{ "status": "prepared_not_published", "base_repository": "https://github.com/vcruz305/llama.cpp", "base_revision": "5210c7c5ed61dddaee6ed476623abf4b63093d16", "base_source_archive_sha256": "3e0ad09b1f2175a170d87310731ac12d1bf52d8ab0b1f4fbdb19298457ae73e6", "patch_sha256": "1a70ed3aacb89fec18ff7fdd263702a767f1e2606b3a57bad425035c1ed29fe6", "model": "pfeifferj/DeepSeek-V4.1-Flash-GSQ-RCO-GGUF", "model_file": "DeepSeek-V4.1-Flash-GSQ-RCO-3.0bit.gguf", "model_sha256": "11f46543370256ef616b6e458b6950e148625b5f8b7545173d124445d3393c53", "official_reference_revision": "dba1be0a40aa45a94ad051997016db3960a90277", "runtime_artifacts_sha256": { "libllama.so.0.4.0": "9abf951448ec289625e27aff805986ea80780afc1c7e9b745162f3fd5cf3d7d8", "libggml-cuda.so.0.23.0": "b33b8b786f7c61c1cc10a6699d1f5d778eedfa094dec34f9013c38a82c8f5648", "mc-score-deepseek-f32": "e5b8a91ba2547b02a14668335b590ddcac185b6b606c3b10f50b0cf29834fb4c", "llama-perplexity": "51197334ee7c016569cbbb893eb20150cc2714ef46a161c34f00ae5175da3632" }, "evaluation_settings": { "context": 2048, "batch": 2048, "microbatch": 2048, "sequences": 1, "threads": 16, "flash_attention": false, "cache_k": "f32", "cache_v": "f32", "CUDA_VISIBLE_DEVICES": "2,3", "NVIDIA_TF32_OVERRIDE": "0", "GGML_CUDA_MMF_F32_DISABLE": "1", "GGML_CUDA_REFERENCE_F32": "1", "engram_placement": "lazy CPU mmap", "method": "single prompt forward, raw answer-letter single-token log-probabilities, variable option counts from archived task index, no chat template", "tokenization": { "scorer_add_special": true, "gguf_add_bos_token": false, "gguf_add_eos_token": false, "effective_raw_prompt": "No forced BOS or EOS; preserve original Spark scorer and model metadata behavior", "chat_prompt": "Pinned official encoder supplies BOS explicitly" } }, "throughput_pilot": { "questions": 20, "decode_seconds": 32.749756, "seconds_per_question": 1.6374878, "projected_2000_decode_minutes": 54.582926666666665 }, "ppl_sidecar_patch_sha256": "2ed02e1d9c7a77303ef50153792b3a40da19cadc394ae0d79aa76d59c8ba2068", "question_state_isolation": "research/question-state-isolation.md", "tokenizer_parity": { "status": "PASS", "vocabulary_size": 129280, "compared": "Every native token ID, exact token string, and decoded byte piece including special tokens", "token_string_mapping_sha256": "ad4e020abc59468b66b83039c5663753efe861c0af3cda9034bfc283d298059c", "token_byte_mapping_sha256": "dced44b89abfc32450c7398912681c477f0da4db310b0caa1bbdcf14ed0e9c7d", "tokenizer_sha256": "c90dfa01249db1be4245780a052ede752e1361c612ac6d08e2bdada7d599476b", "native_dump_sha256": "c53162bbc06168cdb73b3e69f41c13fb1ed4b8603dd33ac0f85e54e78df9bebc" }, "accepted_mmlu": { "correct": 1220, "n": 2000, "accuracy": 0.61, "tsv_sha256": "7ac61bdb80ce8c0c41079c2d2364968442380785bfdb33246d3b0846704254e1", "physical_gpu_pairs": [ [ 2, 3 ], [ 0, 1 ] ], "cross_pair_parity": "All20 predictions and every printed option log probability match exactly", "final_merge": { "status": "PASS", "rows": 2000, "primary": [ 0, 1500 ], "tail": [ 1700, 2000 ], "sha256": "7ac61bdb80ce8c0c41079c2d2364968442380785bfdb33246d3b0846704254e1", "reused_initial_rows": 824, "second_restart": 1379, "middle": [ 1500, 1700 ], "same_binary_and_settings": true } } }