{ "schema": "daecore.serving-preparation-summary.v1", "date": "2026-09-28", "hardware": "Windows x64, NVIDIA RTX 3060 Ti 8 GiB", "unmeasured_targets": [ "AMD", "Intel", "Linux x64" ], "runtimes": { "onnxruntime": "1.24.4", "vulkan_plugin": "0.4.0", "backend": "Vulkan", "storage_buffer_cache": "lazyRelease" }, "model": "ettin", "qualification": { "queries": 970, "unjudged_top20": 0, "quality_evidence": "serving.json", "quality_receipt_sha256": "805a1b0acd7baaf248e182908bc23cd52171e40aa8f1281799a20d15b58c6905", "full_replay_seconds_p50_p95": [ 1.7189999999827705, 2.610000000044238 ], "matched_20_pool_seconds_p50_p95": { "vulkan": [ 1.5565476999909151, 2.1063475799834124 ], "predecessor_directml": [ 1.9197023500164505, 2.701977299965802 ] }, "pressure": "One preflight refusal after 322 queries with lazy release; resumed all remaining rows with zero further retries. An earlier default-cache run stopped after 424. No native OOM observed.", "precision": "FP32 Vulkan versus FP16 CUDA; rankings not bit-exact", "selector": "The provider comparison transfers the preceding CUDA mapping; it does not measure the newly fitted release mapping." }, "worker_recovery": { "receipt_sha256": "46205d39521bde40d1228c837e57f88e1a941d42f6846f4a244c48b513347cc6", "outputs_identical_after_owned_worker_crash": true }, "public_reranking": { "scope": "Retained matched public pools retrieved by upstream Gemma; upstream and production Ettin PyTorch FP16; not a new Vulkan public replay", "source_sha256": "0814db560081af8376e4b1f3e3d2bcd5c4d9b8b208f7fb8b7172f113e8123ef3", "panels": { "fiqa": { "queries": 648, "ndcg@10": { "upstream": 0.48603492061219766, "finetuned": 0.452680922415071 } }, "scifact": { "queries": 300, "ndcg@10": { "upstream": 0.7487436478294831, "finetuned": 0.7543833016238701 } } } }, "maximum_envelope": { "tokens": 1153, "max_abs_logit_delta_vs_cpu": 5.91278076171875e-05 } }