| |
| """Write judgments.json for the Nex-N2.5-mini cards. Sentence structure is authored; every number is |
| computed from summary.json. A judgment whose inputs are missing is NOT emitted (the renderer then |
| shows JUDGMENT PENDING). usage: nex_judge.py <summary.json> <judgments.json>""" |
| import json, math, statistics, sys |
|
|
| S = json.load(open(sys.argv[1])); T = S.get("tiers") or {}; J = {} |
| MiB = 1024 ** 2 |
| UB = S.get("n_ubatch") if S.get("n_ubatch") is not None else 1024 |
| SPEED_EQ = 3.0 |
| TWINS = (("q106i", "q106"), ("q102i", "q102"), ("q103i", "q103")) |
| STD_TAGS = ("q106", "q102", "q103") |
| IMAT_TAGS = ("q106i", "q102i", "q103i") |
| NAMES = {"q106": "STRIX_LEAN", "q102": "COHERENT", "q103": "FAST", |
| "q106i": "STRIX_LEAN", "q102i": "COHERENT", "q103i": "FAST"} |
|
|
|
|
| def pct(a, b): |
| return (b - a) / a * 100 |
|
|
|
|
| def sig(a, ea, b, eb): |
| return abs(b - a) / math.sqrt(ea * ea + eb * eb) |
|
|
|
|
| def have(*v): |
| return all(x is not None for x in v) |
|
|
|
|
| def bench(label): |
| return next((x for x in reversed(S.get("bench") or []) if x["label"] == label), None) |
|
|
|
|
| def gate(label): |
| return next((x for x in reversed(S.get("gates") or []) if x.get("label") == label), None) |
|
|
|
|
| def bn(tag, dev): |
| return bench(f"n-{tag}-{dev}") |
|
|
|
|
| def twin_spread(key): |
| """Largest |gap| between an imatrix file and its standard twin: a measured noise floor.""" |
| if not all(T.get(i, {}).get("same_tensor_types_as_standard") is True for i, _ in TWINS): |
| return None |
| vals = [] |
| for ti, ts in TWINS: |
| for dev in ("rocm", "vk"): |
| x, y = bn(ti, dev), bn(ts, dev) |
| if not (x and y and x.get(key) is not None and y.get(key) is not None): |
| return None |
| vals.append(abs(pct(y[key], x[key]))) |
| return max(vals) if vals else None |
|
|
|
|
| TG_NOISE, PP_NOISE = twin_spread("tg_median"), twin_spread("pp_median") |
| DEC_EQ = max(SPEED_EQ, TG_NOISE) if TG_NOISE is not None else None |
| PRE_EQ = max(SPEED_EQ, PP_NOISE) if PP_NOISE is not None else None |
|
|
|
|
| def rel(g, what, eq): |
| if abs(g) < eq: |
| return f"{what} within {abs(g):.1f} % of" |
| return f"{what} {abs(g):.1f} % {'faster' if g > 0 else 'slower'} than" |
|
|
|
|
| def cmp_line(a_tag, b_tag, a_nm, b_nm): |
| """-> (text, a_advantage, a_disadvantage) or (None, None, None).""" |
| ra, rb, va, vb = bn(a_tag, "rocm"), bn(b_tag, "rocm"), bn(a_tag, "vk"), bn(b_tag, "vk") |
| if not all((ra, rb, va, vb)) or DEC_EQ is None or PRE_EQ is None: |
| return None, None, None |
| if not have(ra.get("tg_median"), rb.get("tg_median"), va.get("tg_median"), vb.get("tg_median"), |
| ra.get("pp_median"), rb.get("pp_median"), |
| ra.get("tg_min"), ra.get("tg_max"), rb.get("tg_min"), rb.get("tg_max")): |
| return None, None, None |
| g_r, g_v = pct(rb["tg_median"], ra["tg_median"]), pct(vb["tg_median"], va["tg_median"]) |
| g_p = pct(rb["pp_median"], ra["pp_median"]) |
| txt = (f"`{a_nm}` {rel(g_r, 'decodes', DEC_EQ)} `{b_nm}` on ROCm0 ({ra['tg_median']:.2f} vs " |
| f"{rb['tg_median']:.2f} tok/s; per-run ranges {ra['tg_min']:.2f}–{ra['tg_max']:.2f} and " |
| f"{rb['tg_min']:.2f}–{rb['tg_max']:.2f}) and {rel(g_v, 'decodes', DEC_EQ)[len('decodes '):]} it on Vulkan0 " |
| f"({va['tg_median']:.2f} vs {vb['tg_median']:.2f}), and {rel(g_p, 'prefills', PRE_EQ)} it on ROCm0 " |
| f"({ra['pp_median']:.0f} vs {rb['pp_median']:.0f} tok/s)") |
| adv = g_p >= PRE_EQ or g_r >= DEC_EQ or g_v >= DEC_EQ |
| dis = g_p <= -PRE_EQ or g_r <= -DEC_EQ or g_v <= -DEC_EQ |
| return txt, adv, dis |
|
|
|
|
| NOISE_NOTE = ("" if DEC_EQ is None or PRE_EQ is None else |
| f"Speed gaps below {DEC_EQ:.1f} % (decode) and {PRE_EQ:.1f} % (prefill) are called a tie: the larger of " |
| f"{SPEED_EQ:.0f} % and the widest gap measured between files that do identical work per token (each imatrix " |
| f"file and its standard twin: decode {TG_NOISE:.1f} %, prefill {PP_NOISE:.1f} %).") |
|
|
|
|
| def kld_cmp(a, b): |
| """KLD of tag a against tag b -> (pct change b->a, sigma, word). 'lower'/'higher' only at >= 2 sigma.""" |
| x, y = T[a], T[b] |
| d = pct(y["kld_mean"], x["kld_mean"]) |
| sg = sig(y["kld_mean"], y["kld_err"], x["kld_mean"], x["kld_err"]) |
| word = "within noise of" if sg < 2 else ("lower than" if d < 0 else "higher than") |
| return d, sg, word |
|
|
|
|
| def kld_ready(tags): |
| return all(have(T.get(k, {}).get("kld_mean"), T.get(k, {}).get("kld_err"), |
| T.get(k, {}).get("size_bytes")) for k in tags) |
|
|
|
|
| def kld_rank_text(tags): |
| items = sorted(tags, key=lambda t: T[t]["kld_mean"]) |
| parts = [] |
| for i, tag in enumerate(items): |
| x = T[tag] |
| bit = f"`{NAMES[tag]}` {x['kld_mean']:.4f}" |
| if i > 0: |
| lo = T[items[0]]["kld_mean"] |
| parts.append(f"{bit} ({pct(lo, x['kld_mean']):+.1f} % vs `{NAMES[items[0]]}`)") |
| else: |
| parts.append(bit) |
| return ", ".join(parts) |
|
|
|
|
| def size_vs(a, b): |
| da, db = T[a]["size_bytes"], T[b]["size_bytes"] |
| if da == db: |
| return f"`{NAMES[a]}` and `{NAMES[b]}` are the same size ({da / MiB:.0f} MiB)" |
| smaller, larger = (a, b) if da < db else (b, a) |
| return (f"`{NAMES[smaller]}` is {(T[larger]['size_bytes'] - T[smaller]['size_bytes']) / MiB:.0f} MiB " |
| f"smaller than `{NAMES[larger]}`") |
|
|
|
|
| |
| rep, rep_v, ref = S.get("repeat") or {}, S.get("repeat_vk") or {}, S.get("reference") or {} |
| if rep.get("result") == "MATCH" and rep.get("rows") is not None and have(ref.get("cpu_chunk1"), ref.get("vulkan0_chunk1")): |
| both = rep_v.get("result") == "MATCH" and rep_v.get("rows") is not None |
| J["quality_provenance"] = ( |
| f"Measured directly on these files, against BF16 logits computed **on the CPU** in the same session " |
| f"(first-window perplexity {ref['cpu_chunk1']:.4f}; Vulkan0 gave {ref['vulkan0_chunk1']:.4f} for the same " |
| f"window). Every file was graded on ROCm0 (the columns above) and again on Vulkan0. The STRIX_LEAN grade was run " |
| f"twice {'on each backend' if both else 'on ROCm0'} and every per-chunk row matched: {rep['rows']} of " |
| f"{rep['rows']} on ROCm0" + (f", {rep_v['rows']} of {rep_v['rows']} on Vulkan0." if both else ".") |
| + " Why not the GPU for the reference: see [Known issues](#known-issues-and-limits).") |
|
|
| |
| vk_rows = [(t, T[t]["vk"]) for t in STD_TAGS + IMAT_TAGS if (T.get(t) or {}).get("vk") and have( |
| T[t].get("kld_mean"), T[t].get("kld_err"), T[t]["vk"].get("kld_mean"), T[t]["vk"].get("kld_err"))] |
| if len(vk_rows) == 6: |
| parts_b, lower_vk, lower_rocm = [], 0, 0 |
| for t, v in vk_rows: |
| a_, b_ = T[t]["kld_mean"], v["kld_mean"] |
| sg_ = sig(a_, T[t]["kld_err"], b_, v["kld_err"]) |
| if sg_ >= 2: |
| lower_vk += b_ < a_; lower_rocm += a_ < b_ |
| pc_ = f"{pct(a_, b_):+.1f}" |
| pc_ = "0.0" if pc_ in ("+0.0", "-0.0") else pc_ |
| parts_b.append(f"{'imatrix ' if t.endswith('i') else ''}{NAMES[t]} {a_:.4f} / {b_:.4f} ({pc_} %, {sg_:.1f}σ)") |
| verdict = ("The two backends agree within noise on every file." if not (lower_vk or lower_rocm) else |
| f"Vulkan0's output is measurably closer to BF16 on {lower_vk} of 6 files and ROCm0's on {lower_rocm}.") |
| J["backend_quality_note"] = ( |
| f"**Same files, same reference, graded on each backend** — KLD ROCm0 / Vulkan0: " + "; ".join(parts_b) + f". {verdict}") |
|
|
|
|
| |
| def twin_kld_ok(st, im): |
| keys = ("kld_mean", "kld_err", "kld_median", "kld_p99", "same_top_p") |
| return have(*(T.get(st, {}).get(k) for k in keys), *(T.get(im, {}).get(k) for k in keys)) |
|
|
|
|
| if all(twin_kld_ok(st, im) for im, st in TWINS): |
| bits = [] |
| improved, worse = [], [] |
| for im, st in TWINS: |
| nm = NAMES[st] |
| a, b = T[st], T[im] |
| dk = pct(a["kld_mean"], b["kld_mean"]) |
| sg = sig(a["kld_mean"], a["kld_err"], b["kld_mean"], b["kld_err"]) |
| dmed = pct(a["kld_median"], b["kld_median"]) |
| dp99 = pct(a["kld_p99"], b["kld_p99"]) |
| dtop = b["same_top_p"] - a["same_top_p"] |
| if dk < 0 and sg >= 2: |
| verb = "improves" |
| improved.append(nm) |
| elif dk > 0 and sg >= 2: |
| verb = "is measurably worse than" |
| worse.append(nm) |
| else: |
| verb = "is within noise of" |
| bits.append( |
| f"**{nm}** {verb} the standard file on mean KLD " |
| f"({dk:+.1f} %, {sg:.1f}σ; {a['kld_mean']:.4f} → {b['kld_mean']:.4f}); " |
| f"median {dmed:+.1f} %, 99th-pct {dp99:+.1f} %, top-1 {dtop:+.2f} pp") |
| if len(improved) == 3: |
| head = "**The imatrix measurably improves all three tiers.** " |
| elif improved: |
| head = f"**The imatrix measurably improves {', '.join(improved)}.** " |
| else: |
| head = "**The imatrix does not measurably improve any of the three tiers on this corpus.** " |
| if worse: |
| head += f"**It is measurably worse on {', '.join(worse)}.** " |
| J["imat_verdict"] = head + "; ".join(bits) + "." |
|
|
|
|
| |
| def backend_line(tag): |
| r, v = bn(tag, "rocm"), bn(tag, "vk") |
| if not have(r, v) or DEC_EQ is None or PRE_EQ is None: |
| return None |
| if not have(r.get("tg_median"), v.get("tg_median"), r.get("pp_median"), v.get("pp_median"), |
| r.get("tg_min"), r.get("tg_max"), v.get("tg_min"), v.get("tg_max")): |
| return None |
| g_t, g_p = pct(v["tg_median"], r["tg_median"]), pct(v["pp_median"], r["pp_median"]) |
| return (f"`{NAMES[tag]}`: ROCm0 {rel(g_t, 'decodes', DEC_EQ)} Vulkan0 " |
| f"({r['tg_median']:.2f} vs {v['tg_median']:.2f} tok/s; ranges " |
| f"{r['tg_min']:.2f}–{r['tg_max']:.2f} / {v['tg_min']:.2f}–{v['tg_max']:.2f}) and " |
| f"{rel(g_p, 'prefills', PRE_EQ)} it ({r['pp_median']:.0f} vs {v['pp_median']:.0f} tok/s)") |
|
|
|
|
| if DEC_EQ is not None and PRE_EQ is not None: |
| lines = [backend_line(t) for t in STD_TAGS] |
| cr, pr = bench("n-q106-rocm"), bench("n-q106-rocm-prose") |
| cv, pv = bench("n-q106-vk"), bench("n-q106-vk-prose") |
| prose = None |
| if have(cr, pr, cr and cr.get("tg_median"), pr and pr.get("tg_median")): |
| g = pct(cr["tg_median"], pr["tg_median"]) |
| prose = (f"STRIX_LEAN workload range on ROCm0: code {cr['tg_median']:.2f} tok/s vs prose " |
| f"{pr['tg_median']:.2f} ({rel(g, 'prose decodes', DEC_EQ)} code)") |
| if have(cv, pv, cv and cv.get("tg_median"), pv and pv.get("tg_median")): |
| gv = pct(cv["tg_median"], pv["tg_median"]) |
| prose += (f"; Vulkan0 code {cv['tg_median']:.2f} vs prose {pv['tg_median']:.2f} " |
| f"({rel(gv, 'prose decodes', DEC_EQ)} code)") |
| prose += "." |
|
|
| def _span(wl): |
| v = [(b_.get("prompt_n_min") if b_.get("prompt_n_min") is not None else b_.get("prompt_n"), |
| b_.get("prompt_n_max") if b_.get("prompt_n_max") is not None else b_.get("prompt_n")) |
| for b_ in S.get("bench") or [] if b_.get("workload") == wl] |
| v = [x for x in v if None not in x] |
| if not v: |
| return None |
| lo, hi = min(a for a, _ in v), max(b for _, b in v) |
| return f"{lo:,}" if lo == hi else f"{lo:,}–{hi:,}" |
| sc, sp = _span("code"), _span("prose") |
| if sc and sp: |
| prose += (f" Prompt lengths: code {sc} tokens (the first 30,000 characters of `convert_hf_to_gguf.py` plus " |
| f"an instruction), prose {sp} tokens (the first 34,000 characters of wikitext-2 *train* plus a " |
| f"writing instruction).") |
| if all(lines) and prose: |
| J["speed_note"] = " ".join(x + "." for x in lines) + " " + prose + (f" {NOISE_NOTE}" if NOISE_NOTE else "") |
|
|
|
|
| |
| d = gate("n-c3-q106") |
| if d and d.get("all_reused") and d.get("all_aligned") and have(d.get("n"), d.get("L"), d.get("expect_cache_n"), |
| d.get("identical")): |
| wm = [r_["warm_prompt_ms"] for r_ in d.get("rows") or [] if r_.get("warm_prompt_ms") is not None] |
| cm = [r_["cold_prompt_ms"] for r_ in d.get("rows") or [] if r_.get("cold_prompt_ms") is not None] |
| t_line = "" |
| if wm and cm: |
| t_line = (f" — median prefill **{statistics.median(wm) / 1000:.1f} s instead of " |
| f"{statistics.median(cm) / 1000:.1f} s** cold " |
| f"({statistics.median(cm) / statistics.median(wm):.1f}× faster)") |
| J["cache_note"] = ( |
| f"In {d['n']} request pairs sharing a long prefix at one fixed prompt length of {d['L']:,} tokens, every " |
| f"second request resumed from the checkpoint the first one left {4 + UB:,} tokens before its end — " |
| f"**{d['expect_cache_n']:,} tokens reused " |
| f"({d['expect_cache_n'] / d['L'] * 100:.0f} %), {d['L'] - d['expect_cache_n']:,} processed**{t_line}. " |
| f"Each warm reply was byte-identical to a cold run of the same prompt in **{d['identical']}/{d['n']}** " |
| f"exchanges.\n\n" |
| f"llama-server processes the last `n_ubatch` + 4 tokens of every prompt as two batches so it can checkpoint " |
| f"there ({UB:,} + 4 = {4 + UB:,} tokens with the `-ub {UB}` used in these measurements, where the server " |
| f"default is `-ub 512`; [upstream PR #20288](https://github.com/ggml-org/llama.cpp/pull/20288)). A turn that " |
| f"resumes from a checkpoint left by a prompt of a *different* length therefore splits its tail differently " |
| f"from a cold run, and float rounding can flip a greedy token.") |
|
|
|
|
| |
| def _tools_ok(x): |
| return bool(x) and have(x.get("passed"), x.get("total"), x.get("detail")) |
|
|
|
|
| tl = gate("n-tools-q106") |
| FX = [gate(l) for l in ("n-tools-q106-roff", "n-tools-q106-roff-r2", "n-tools-q106-roff-r3")] |
| TD = S.get("tools_diag") or {} |
| TF = S.get("template_fix") or {} |
| if _tools_ok(tl) and all(_tools_ok(x) for x in FX) and have( |
| TD.get("stock_on_leaks"), TD.get("stock_on_replies"), TD.get("stock_on_reasoning_extracted"), |
| TD.get("nested_off_http500"), TD.get("nested_off_attempts")): |
| misses, on_n, on_ok, off_n, off_ok = [], 0, 0, 0, 0 |
| for i, x in enumerate(FX, 1): |
| for k, v in x["detail"].items(): |
| name, think = k.split("|think=") |
| if think == "True": |
| on_n += 1; on_ok += bool(v) |
| else: |
| off_n += 1; off_ok += bool(v) |
| if not v: |
| misses.append("`%s` with thinking %s (pass %d)" % (name, "on" if think == "True" else "off", i)) |
| fp = TD.get("flag_probes") or {} |
| flag_txt = "; ".join("%s: %d of %d replies still had reasoning in `content`" % (lab, fp[key]["leaks"], fp[key]["n"]) |
| for key, lab in (("fmt-deepseek", "`--reasoning-format deepseek`"), |
| ("srv-kwargs-high", "`--chat-template-kwargs` with `reasoning_effort`"), |
| ("reasoning-on", "`--reasoning on`")) if key in fp) |
| J["tools_note"] = ( |
| "**Stock chat template: %d/%d.** Every thinking-on check failed. Re-run with the raw replies kept, %d of %d " |
| "thinking-on replies carried the reasoning and a `</think>` in `content`, and %d had any " |
| "`reasoning_content`. llama-server builds its reasoning parser by rendering the template with " |
| "`enable_thinking` on and off; this template ignores `enable_thinking` (it switches on `reasoning_effort`), " |
| "so the parser finds no reasoning markers and extracts nothing. Server switches did not help (%s). With " |
| "thinking off, `nested-object` failed on an HTTP 500 — see [Known issues](#known-issues-and-limits).\n\n" |
| "**With the included `%s` and `--reasoning off` (the quick start): %s over three passes of the same suite " |
| "(%d/%d)** — thinking off %d/%d, thinking on %d/%d; the misses were %s. A pass requires a native " |
| "`tool_calls` entry with the right arguments and no think tags in `content`. Each check is a single sample at " |
| "the recommended temperature 0.7." % ( |
| tl["passed"], tl["total"], TD["stock_on_leaks"], TD["stock_on_replies"], |
| TD["stock_on_reasoning_extracted"], flag_txt or "not measured", TF.get("file") or "—", |
| ", ".join("%d/%d" % (x["passed"], x["total"]) for x in FX), |
| sum(x["passed"] for x in FX), sum(x["total"] for x in FX), off_ok, off_n, on_ok, on_n, |
| ", ".join(misses) if misses else "none")) |
|
|
|
|
| |
| PR = TF.get("probes_roff") or {} |
| if PR and all(_tools_ok(x) for x in FX): |
| def _probe(prefix): |
| rows_ = [v for k, v in PR.items() if k.split("|")[0] == prefix] |
| return len(rows_), sum(1 for v in rows_ if v.get("leaks")), sum(1 for v in rows_ if v.get("reasoning_len")) |
| on_ok = sum(bool(v) for x in FX for k, v in x["detail"].items() if k.endswith("|think=True")) |
| on_n = sum(1 for x in FX for k in x["detail"] if k.endswith("|think=True")) |
| off_ok = sum(bool(v) for x in FX for k, v in x["detail"].items() if k.endswith("|think=False")) |
| off_n = sum(1 for x in FX for k in x["detail"] if k.endswith("|think=False")) |
| lines = [] |
| for label, prefix, think, dest in ( |
| ("no `chat_template_kwargs`", "no-kwargs", "off (the server default with `--reasoning off`)", "—"), |
| ("`\"enable_thinking\": false`", "enable_thinking=false", "off", "—"), |
| ("`\"reasoning_effort\": \"none\"`", "reasoning_effort=none", "off", "—"), |
| ("`\"reasoning_effort\": \"high\"`", "reasoning_effort=high", "on", "**`content`** — do not use"), |
| ("`\"reasoning_effort\": \"medium\"`", "reasoning_effort=medium", "adaptive", "**`content`** — do not use")): |
| n, leaks, _ = _probe(prefix) |
| if n: |
| lines.append("| %s | %s | %s | %d of %d replies with think tags in `content` |" % (label, think, dest, leaks, n)) |
| lines.insert(1 if lines else 0, |
| "| `\"enable_thinking\": true` | on | `reasoning_content` | tool suite with thinking on: %d/%d " |
| "(a pass requires no think tags in `content`) |" % (on_ok, on_n)) |
| J["template_note"] = ( |
| "Measured on the standard STRIX_LEAN file with the included template file and `--reasoning off` (greedy " |
| "probes: a direct question, a " |
| "one-word instruction and a tool request, each with the tool schema attached; tool suite: 3 passes):\n\n" |
| "| request | thinking | reasoning ends up in | measured |\n| --- | --- | --- | --- |\n" + "\n".join(lines) |
| + "\n\nSo: switch thinking with `enable_thinking` only. Thinking-off tool checks: %d/%d." % (off_ok, off_n)) |
|
|
|
|
| |
| def vwhy(x): |
| if x.get("server_died"): |
| import re as _re |
| errs = [_re.sub(r"^[0-9.]+ [IWE] (srv +)?", "", e).strip() for e in x.get("server_log_errors") or []] |
| first = next((e for e in errs if "failed" in e.lower() or "error" in e.lower() or "abort" in e.lower()), |
| None) or (errs[0] if errs else None) |
| return ("the server aborted" + (f" (`{first[:90]}`)" if first else "")) |
| if x.get("error"): |
| return f"the request failed ({x['error'][:80]})" |
| exp = x.get("expected") or "" |
| nexp = len(exp.split(",")) if exp else None |
| hits = x.get("hits") or [] |
| if nexp: |
| return f"the reply named {len(hits)} of {nexp} expected terms" |
| return "the reply did not pass" |
|
|
|
|
| on, off = gate("n-vision-q106-faon"), gate("n-vision-q106-faoff") |
| if on and off: |
| def vok(x): |
| return x.get("result") == "PASS" |
| if vok(on) and vok(off): |
| J["vision_note"] = "✅ **Images work with `-fa on` and `-fa off`.**" |
| elif vok(off) and not vok(on): |
| J["vision_note"] = (f"⛔ **With `-fa on`, image requests fail** — {vwhy(on)}. With `-fa off` they work. " |
| f"**For image input, serve with `-fa off`.**") |
| J["vision_quickstart_warning"] = ("> ⛔ **Images:** with `-fa on`, image requests fail on this build " |
| "([measured](#vision)). If you send images, use `-fa off`.") |
| elif vok(on) and not vok(off): |
| J["vision_note"] = (f"⛔ **With `-fa off`, image requests fail** — {vwhy(off)}. With `-fa on` they work. " |
| f"**For image input, keep `-fa on` (the quick-start default).**") |
| else: |
| J["vision_note"] = (f"⛔ **Images failed in both `-fa` settings.** `-fa on`: {vwhy(on)}; " |
| f"`-fa off`: {vwhy(off)}.") |
| vfx = gate("n-vision-q106-roff-faon") |
| if vfx and "vision_note" in J: |
| J["vision_note"] += (" The `-fa on` image test was repeated with the included template file and " |
| f"`--reasoning off` (the quick start): {'✅ passed' if vok(vfx) else '❌ ' + vwhy(vfx)}.") |
|
|
|
|
| |
| rows_sz = [r for r in (S.get("sizing") or []) if r.get("label") == "strix-lean"] |
| if rows_sz: |
| def mem_cell(r, key): |
| if r.get("result") == "LOAD_FAIL": |
| return "did not load" |
| v = r.get(key) |
| return f"{v:.2f} GiB" if v is not None else "—" |
| body = "\n".join( |
| f"| {r['ctx']:,} | {mem_cell(r, 'footprint_loaded_gib')} | {mem_cell(r, 'footprint_after_8k_gib')} |" |
| for r in rows_sz if r.get("ctx") is not None) |
| J["memory_note"] = ( |
| "Measured footprint (drop in `MemAvailable`) of STRIX_LEAN with the vision projector, q8_0 KV cache, " |
| "`-cram 512`, one slot, no draft head:\n\n" |
| "| context | after load | after one request (30,000-character code prompt) |\n" |
| "| ---: | ---: | ---: |\n" + body + |
| "\n\nNo row was decode-benchmarked beyond that one request. Nothing beyond these rows was measured.") |
|
|
|
|
| |
| def recommend(lean, coh, fast): |
| """STRIX_LEAN is the flagship tier; the data decides whether COHERENT's quality or FAST's speed is worth taking |
| instead. -> (default tag, markdown) or (None, None) when an input is missing.""" |
| lc, lc_adv, lc_dis = cmp_line(lean, coh, "STRIX_LEAN", "COHERENT") |
| fl, fl_adv, fl_dis = cmp_line(fast, lean, "FAST", "STRIX_LEAN") |
| if lc is None or fl is None: |
| return None, None |
| L, C, F = T[lean], T[coh], T[fast] |
| d_c, s_c, w_c = kld_cmp(coh, lean) |
| more = (C["size_bytes"] - L["size_bytes"]) / MiB |
| size_c = f"{abs(more):.0f} MiB {'more' if more > 0 else 'less'}" |
| if w_c == "lower than" and lc_adv and not lc_dis: |
| default = lean |
| head = (f"**Start with `STRIX_LEAN`; take `COHERENT` if quality matters more than speed.** {lc}. `COHERENT`'s " |
| f"KLD is {abs(d_c):.1f} % lower ({s_c:.1f}σ) for {size_c}.") |
| elif w_c == "lower than": |
| default = coh |
| head = (f"**Start with `COHERENT`.** Its KLD is {abs(d_c):.1f} % lower than `STRIX_LEAN`'s ({s_c:.1f}σ) for " |
| f"{size_c}, and the speed comparison does not clearly favour `STRIX_LEAN`: {lc}.") |
| elif w_c == "within noise of" and lc_dis and not lc_adv: |
| default = coh |
| head = (f"**Start with `COHERENT`.** Its KLD is {w_c} `STRIX_LEAN`'s ({d_c:+.1f} %, {s_c:.1f}σ) and it is " |
| f"measurably faster: {lc}. It costs {size_c}.") |
| else: |
| default = lean |
| head = (f"**Start with `STRIX_LEAN`.** `COHERENT`'s KLD is {w_c} it ({d_c:+.1f} %, {s_c:.1f}σ) for " |
| f"{size_c}; {lc}.") |
| d_f, s_f, w_f = kld_cmp(fast, lean) |
| kf = f"its KLD is {w_f} `STRIX_LEAN`'s ({F['kld_mean']:.4f} vs {L['kld_mean']:.4f}, {d_f:+.1f} %, {s_f:.1f}σ)" |
| if fl_adv and not fl_dis: |
| fast_txt = f"**Take `FAST` for speed:** {fl}; {kf}." |
| else: |
| fast_txt = f"`FAST` does not buy a clear speed gain here: {fl}; {kf}." |
| sizes = f"{size_vs(lean, coh)}; {size_vs(lean, fast)}." |
| body = (f"{head}\n\n{fast_txt}\n\nKLD order (lower is closer to BF16): {kld_rank_text((lean, coh, fast))}. " |
| f"{sizes} {NOISE_NOTE}") |
| return default, body |
|
|
|
|
| def imat_pointer(): |
| """Point the standard card at the imatrix repo; name only measurable (>= 2 sigma) twin improvements.""" |
| repo = S.get("model_repo_imat") |
| if not repo: |
| return "" |
| link = f"[imatrix build](https://huggingface.co/{repo})" |
| bits = [] |
| for im, st in TWINS: |
| if not twin_kld_ok(st, im): |
| return "" |
| d, sg, w = kld_cmp(im, st) |
| if w == "lower than": |
| bits.append(f"`{NAMES[st]}` {T[st]['kld_mean']:.4f} → {T[im]['kld_mean']:.4f} ({d:+.1f} %, {sg:.1f}σ)") |
| if bits: |
| return f"**The {link} is measurably closer to BF16 at the same size:** " + "; ".join(bits) + "." |
| return f"Importance-matrix twins of all three files: **{link}** (not measurably closer to BF16 on this corpus)." |
|
|
|
|
| if kld_ready(STD_TAGS) and DEC_EQ is not None and PRE_EQ is not None: |
| default, body = recommend("q106", "q102", "q103") |
| if default: |
| ptr = imat_pointer() |
| J["std_recommendation"] = body + (f"\n\n{ptr}" if ptr else "") |
| J["std_default"] = T[default]["file"] |
|
|
| if kld_ready(IMAT_TAGS) and DEC_EQ is not None and PRE_EQ is not None: |
| default_i, body_i = recommend("q106i", "q102i", "q103i") |
| if default_i: |
| cross = "" |
| if kld_ready(("q102",)): |
| d, sg, w = kld_cmp("q106i", "q102") |
| cross = (f"\n\nFor scale: the imatrix `STRIX_LEAN`'s KLD is {w} the *standard* `COHERENT`'s " |
| f"({T['q106i']['kld_mean']:.4f} vs {T['q102']['kld_mean']:.4f}, {d:+.1f} %, {sg:.1f}σ), " |
| f"{size_vs('q106i', 'q102')}.") |
| J["imat_recommendation"] = body_i + cross |
| J["imat_default"] = T[default_i]["file"] |
|
|
| parts = [] |
| for nm, ti, ts in (("STRIX_LEAN", "q106i", "q106"), ("COHERENT", "q102i", "q102"), ("FAST", "q103i", "q103")): |
| for dev, dn in (("rocm", "ROCm0"), ("vk", "Vulkan0")): |
| a_, b_ = bn(ti, dev), bn(ts, dev) |
| if a_ and b_ and have(a_.get("tg_median"), b_.get("tg_median")): |
| parts.append(f"{nm} on {dn} {a_['tg_median']:.2f} vs {b_['tg_median']:.2f} tok/s") |
| same = [T.get(i, {}).get("same_tensor_types_as_standard") for i, _ in TWINS] |
| _hk = [T.get(i, {}).get("header_keys_only_in_imatrix") for i, _ in TWINS] |
| _hs = [T.get(i, {}).get("header_keys_only_in_standard") for i, _ in TWINS] |
| HDR = "" |
| if all(x is not None for x in _hk + _hs) and not any(_hs) and len({tuple(x) for x in _hk}) == 1 and _hk[0]: |
| HDR = ", all in the GGUF header, which carries %d extra keys in each imatrix file: %s" % ( |
| len(_hk[0]), ", ".join("`%s`" % k for k in _hk[0])) |
| deltas = [T.get(i, {}).get("file_size_delta_bytes") for i, _ in TWINS] |
| if len(parts) == 6 and all(x is True for x in same) and all(d is not None for d in deltas): |
| J["imat_speed_note"] = ( |
| f"The imatrix changes scale values, not tensor types or sizes: tensor by tensor, each imatrix file has the " |
| f"same names, types and byte sizes as its standard twin, so the work per token is the same (the files differ " |
| f"in size by {deltas[0]} / {deltas[1]} / {deltas[2]} bytes{HDR})." + " " |
| f"Measured speed still differs — imatrix vs standard: " + "; ".join(parts) + "." |
| + (f" The largest of these decode gaps is {TG_NOISE:.1f} %; both cards call decode gaps below {DEC_EQ:.1f} % " |
| f"a tie." if TG_NOISE is not None and DEC_EQ is not None else "")) |
|
|
|
|
| |
| ki = [ |
| "- **No MTP head.** The checkpoint's `config.json` declares `mtp_num_hidden_layers: 1`, but the weights contain " |
| "**no** `mtp.*` tensors (1,026 tensors total). There is no multi-token-prediction head and no speculative decoding " |
| "on these files — do not pass `--spec-type draft-mtp`.", |
| "- **The stock chat template ignores `enable_thinking`, and llama-server cannot separate its reasoning** " |
| "(see [Tool calling](#tool-calling)). Serve with the included `chat_template_enable_thinking.jinja` and " |
| "`--reasoning off`, and switch thinking per request with `enable_thinking` — see " |
| "[Reasoning controls](#reasoning-controls).", |
| "- **Earlier assistant turns are re-rendered with their reasoning**, so multi-turn contexts grow faster than with " |
| "templates that drop it.", |
| "- **`llama-server`'s host-RAM prompt cache defaults to 8 GiB** (`-cram 8192`). On a shared box, set `-cram` explicitly.", |
| "- Measured on Linux only (Ryzen AI Max+ 395, ROCm 7.2.4, unpatched `d3ca537`).", |
| ] |
| _pn_vals = [b_.get("prompt_n_max") if b_.get("prompt_n_max") is not None else b_.get("prompt_n") |
| for b_ in (S.get("bench") or [])] |
| _pn = max((v for v in _pn_vals if v is not None), default=None) |
| _sz = {r_.get("ctx"): r_ for r_ in (S.get("sizing") or []) if r_.get("label") == "strix-lean"} |
| _big = max(_sz) if _sz else None |
| _ctx_txt = ("" if _big is None else |
| f" (a {_big:,}-token context was loaded in the memory test, not benchmarked)" |
| if _sz[_big].get("result") != "LOAD_FAIL" else "") |
| for _c, _r in sorted(_sz.items()): |
| if _r.get("result") == "LOAD_FAIL": |
| ki.append(f"- **A {_c:,}-token context did not load** in the memory test (STRIX_LEAN + vision projector, " |
| f"q8_0 KV cache, one slot) on this box.") |
| ki.append(f"- **Not measured:** decode beyond a {_pn:,}-token prompt{_ctx_txt}, " if _pn is not None else |
| "- **Not measured:** decode at long context, ") |
| ki[-1] += ("long-context quality, video input, concurrency above 1, and task-level accuracy. Perplexity/KLD measure " |
| "next-token fidelity to BF16 on prose, not reasoning or code correctness.") |
| if have(ref.get("hip_rocm0_chunk1"), ref.get("cpu_chunk1"), ref.get("hip_rocm0_final")) and have(S.get("bf16", {}).get("ppl")): |
| ki.insert(0, f"- **Do not run the BF16 GGUF of this model on ROCm0 with this build.** `d3ca537` computes it wrong on " |
| f"that path: wikitext-2 perplexity {ref['hip_rocm0_final']:.1f} on ROCm0 vs {S['bf16']['ppl']:.2f} on the CPU " |
| f"(first window {ref['hip_rocm0_chunk1']:.1f} vs {ref['cpu_chunk1']:.2f}; also wrong with `-fa off`). " |
| f"The 4-bit files are not affected — their ROCm0 grades are in the table — and the BF16 file is not " |
| f"published here; the quality reference was computed on the CPU instead.") |
| if have(TD.get("nested_off_http500"), TD.get("nested_off_attempts")): |
| ki.insert(1, "- **llama-server rejects a tool call whose required arguments are not in the order the schema lists " |
| "them** — HTTP 500, *The model produced output that does not match the expected peg-native format*: " |
| "this build's parser for the XML tool-call format expects required arguments in definition order. " |
| "On the standard STRIX_LEAN file with the stock template and thinking off, the `nested-object` " |
| "request (three required arguments) " |
| "hit it in the tool suite%s and in %d of %d repeats; the model had written a well-formed call with " |
| "the arguments reordered. Be ready to retry on this error." % ( |
| "" if TD.get("gate_http500_logged") else " (not confirmed in the suite's server log)", |
| TD["nested_off_http500"], TD["nested_off_attempts"])) |
| _PR = (S.get("template_fix") or {}).get("probes_roff") or {} |
| _lk = [v for k, v in _PR.items() if k.split("|")[0] in ("reasoning_effort=high", "reasoning_effort=medium")] |
| if _lk: |
| ki.insert(2, "- **With the included template, do not set `reasoning_effort` to `high` or `medium`** — the reasoning " |
| "goes back into `content` (%d of %d probe replies). `enable_thinking: true` is the way to turn " |
| "thinking on." % (sum(1 for v in _lk if v.get("leaks")), len(_lk))) |
| _seat = [r_ for r_ in (S.get("seats") or {}).values() if r_.get("thinking_reasoning_len") is not None] |
| _short = [r_ for r_ in _seat if not r_.get("thinking_reply") and r_.get("thinking_reasoning_len")] |
| if _short: |
| ki.insert(3, "- **With thinking on, a very short answer can stay inside the think block.** Served as in the quick " |
| "start, `Reply with the single word: ready` (no tools, `enable_thinking: true`, temperature 0) came " |
| "back with the word in `reasoning_content` and an empty `content` on %d of %d FAST files tested.%s" % ( |
| len(_short), len(_seat), |
| " With thinking off the same request returned `ready` in `content`." |
| if all((r_.get("default_reply") or "").strip().lower().startswith("ready") for r_ in _seat) else "")) |
| J["std_known_issues"] = "\n".join(ki) |
| J["imat_known_issues"] = "\n".join([ |
| "- Calibration text is general-purpose English/code (bartowski `calibration_datav3.txt`); an imatrix built from " |
| "your own domain can do better on that domain.", |
| "- Graded on wikitext-2 *test*, a different corpus from the calibration text. Improvements on reasoning/code " |
| "tasks were not measured.", |
| ] + ki) |
|
|
| json.dump(J, open(sys.argv[2], "w"), indent=2) |
| print("judgments written:", sorted(J)) |
| for k, v in J.items(): |
| print(f"\n[{k}]\n{v}") |
|
|