{ "schema_version": 1, "artifact_index_sha256": "f066a0154a9101b359c3a4d4fa6a83fb12b7126a9cf0f19cb772611d08cc07ee", "base_awq_checkpoint_index_sha256": "da62ba7bfef57f8117f35fc78bf5daf14c0ab5eb1b8173b4d712c1f4d294819b", "host_side_checkpoint_audit": { "status": "PASS", "indexed_tensor_count": 222747, "carrier_checkpoint_completion_sha256": "70fb852c73620edbf75858ab36c519005a5b4cec6597d2493c1355b4b901f077" }, "awq_calibration_expert_coverage": { "status": "PASS_WITH_SELECTIVE_AUGMENTATION", "report_filename": "EXPERT_COVERAGE.json", "natural_unique_expert_ids": "512/512", "natural_layer_expert_pairs": "24568/24576", "natural_layer_expert_coverage_fraction": 0.9996744791666666, "fully_naturally_covered_layers": "45/48", "zero_natural_token_pairs": 8, "below_128_natural_token_pairs": 93, "selectively_augmented_layer_expert_pairs": 93, "runtime_zero_coverage_fallback_pairs": 0, "calibration_records": 684, "calibration_active_tokens": 202750, "native_route_matrix_sha256": "6c987b6d2a1506978f38bc0b4fa167d205250ba8f0629f05adddc880a26828b1", "selective_bypass_target_manifest_sha256": "9d4c2372bb065d2f9aadce457d98b0314c3844e1989e8fc6e36ce2a7088770c9" }, "awq_fp16_kv_quality": { "status": "ACCEPTED", "contract": { "hardware": "4x NVIDIA Tesla V100-PCIE-32GB", "tensor_parallel_size": 4, "mtp": 0, "activation_dtype": "float16", "kv_cache_dtype": "float16", "enforce_eager": true, "prefix_caching": false }, "results": { "basic_generation": "4/4", "needle_retrieval_1k_to_128k": "6/6 exact", "heldout_tool_selection": "10/12", "repeat_exact_stability": "6/6 cases over two repeats", "gsm8k_five_shot_subset": "29/32", "humaneval_mbpp_functional_subset": "9/10", "ifeval_strict": "3/5 prompts; 9/12 instructions", "ifeval_loose": "3/5 prompts; 9/12 instructions" }, "acceptance_marker_sha256": "47ceb8120dede77f057d9a1ab62196e6dd476c8e5a8512257028069e12c48a81" }, "calibrated_qsa_e4m3_kv_quality": { "status": "ACCEPTED", "scale_overlay_bundled": true, "scale_filename": "model-kvscales.safetensors", "merged_index_entry_count": 24, "scale_file_sha256": "c89b08bacde7d869342190b2f95046a83af37080241ae00ccae3a3d6888852fd", "scale_tensor_count": 24, "contract": { "hardware": "4x NVIDIA Tesla V100-PCIE-32GB", "tensor_parallel_size": 4, "mtp": 0, "activation_dtype": "float16", "main_qsa_kv_cache_dtype": "fp8_e4m3", "qsa_indexer_cache_dtype": "float16", "enforce_eager": true, "prefix_caching": false, "gpu_memory_utilization": 0.89 }, "calibration": { "release_distribution_processed_tokens": 4130597, "coverage": [ "Chinese", "English", "code", "multi-turn dialogue", "tool use", "1K context", "16K context", "64K context", "128K context" ], "scale_min": 0.01710728369653225, "scale_max": 0.080636166036129, "maximum_saturation_ratio": 0, "report_sha256": "0612b360e005c8c3cb9332a8cd38d1c13b834f14831584514b9b1775422a9a31" }, "results": { "heldout_tool_selection": "10/12; zero correctness regressions versus route-neutral FP16", "heldout_tool_exact_stability": "4/6 cases over two repeats; route-neutral FP16 was 3/6", "needlebench_128k": "6/6 exact; zero correctness regressions versus route-neutral FP16", "matched_first_tokens": "18/18", "gsm8k_five_shot_subset": "29/32; zero flips versus FP16", "humaneval_mbpp_functional_subset": "9/10; identical to FP16", "ifeval_strict": "3/5 prompts; 9/12 instructions; identical to FP16", "ifeval_loose": "3/5 prompts; 9/12 instructions; identical to FP16", "selected_block_recall_mean": 0.9959058567, "selected_block_recall_min": 0.9918218198, "qsa_output_cosine_min": 0.9985972106, "qsa_output_relative_l2_max": 0.0529616075 }, "acceptance_marker_sha256": "0ed39d23b078d449052f58bf254030cc5c6658e4e9d5f2853926a6ffe4e70d8d" }, "image_video_validation": { "status": "ACCEPTED_WITH_PENDING_QWEN4EXP_ENABLEMENT", "qwen4exp_enablement_in_current_release": false, "qwen4exp_enablement_pr": "https://github.com/1CatAI/1Cat-vLLM/pull/458", "qwen4exp_enablement_commit": "dd68b9e8f44d248ad7082b871243a66c5fc1696a", "production_accepted": false, "contract": { "hardware": "4x NVIDIA Tesla V100-PCIE-32GB", "tensor_parallel_size": 4, "mtp": 0, "activation_dtype": "float16", "enforce_eager": true, "prefix_caching": false, "gpu_memory_utilization": 0.89, "maximum_model_length": 32768, "maximum_sequences": 1, "video_frames": 16, "repeats": 2 }, "scope": { "image_cases": 5, "video_cases": 4, "requests_per_kv_arm": 18, "maximum_observed_prompt_tokens": 3182 }, "visual_payload_integrity": { "tensor_count": 333, "bitwise_equal_to_reference": "333/333", "mismatch_count": 0, "aggregate_sha256": "cb57bb7af556cab0c09f457e72fe0894b42d7db08437478a62beb760b69c984c" }, "results": { "fp16_kv_semantic_score": "18/18", "e4m3_calibrated_kv_semantic_score": "18/18", "successful_http_requests_per_arm": "18/18", "runtime_or_request_errors": 0, "repeat_exact_stability_per_arm": "9/9 cases", "cross_arm_exact_answers": "16/18", "cross_arm_whitespace_normalized_answers": "18/18", "e4m3_calibrated_scale_gate": "24/24" }, "capacity_observation": { "fp16_kv_tokens": 34859, "e4m3_kv_tokens": 59234, "ratio": 1.699247826960039, "formal_performance_claim": false }, "limitations": [ "Validation requires Qwen4Exp model-integration PR #458 until it is included in a 1Cat-vLLM release.", "This was a compact regression set rather than a broad multimodal benchmark.", "Long-context multimodal composition was not tested." ], "internal_report_sha256": "9432f9978a480e90c9d134ab7b52cce57c1dc0f294b1872a424504baf1a56e0d" }, "calibrated_qsa_e4m3_kv_performance": { "status": "ACCEPTED", "contract": { "hardware": "4x NVIDIA Tesla V100-PCIE-32GB", "tensor_parallel_size": 4, "mtp": 0, "activation_dtype": "float16", "gpu_memory_utilization": 0.89, "contexts": [1024, 4096, 16384, 32768, 65536, 131072], "concurrencies": [1, 4, 8], "output_tokens": 256, "scored_repeats": 3, "chunked_prefill_tokens": 8192, "cuda_graphs": "FULL_AND_PIECEWISE", "prefix_caching": false }, "capacity": { "fp16_kv_tokens": 310480, "e4m3_kv_tokens": 569579, "ratio": 1.834511079618655 }, "matrix": { "grid_cells": 18, "matched_ok_cells": 15, "fp16_ok_cells": 15, "e4m3_ok_cells": 17, "fp16_scored_requests": 174, "e4m3_scored_requests": 210, "request_failures": 0 }, "matched_16k_and_longer_geometric_deltas_pct": { "prefill_throughput": -3.42101961847342, "decode_throughput": -2.5445261709874822, "server_e2e_elapsed": 3.1165392932944957 }, "caveats": [ "FP16 P1024 and generic E4M3 G6 P256 XQA use different reduction structures.", "Bitcast decoding and scale hoisting were tested together rather than isolated.", "The result supports a capacity gain with a small long-context latency cost, not a general speedup claim." ], "analysis_sha256": "bcf263b3a623b15eff953148740ec22b98f0f7a779eb648f4dd321e958227bfb", "acceptance_marker_sha256": "c0f3acc6f1e7e9c4b7f5450a95035d86513a9f0aa45f1967951f205197b413b7" }, "not_accepted": [ "TP8", "MTP", "prefix caching", "256K context", "generic Transformers loading", "upstream vLLM", "Qwen4Exp multimodal startup without the pending enablement change", "production deployment" ], "huggingface_upload_authorized": false, "huggingface_upload_performed": false }