#!/usr/bin/env python3 # SPDX-License-Identifier: MIT OR Apache-2.0 """Draw the README's speed cards from a release-gate record, so they cannot drift from what was measured. Each card names the record it was drawn from. The proof cards (bit-exact counts) are unchanged by speed and are left as they are. usage: python3 tools/cards.py [testing/results/.txt] (default: the newest record)""" import re, sys from pathlib import Path ROOT = Path(__file__).resolve().parents[1] MONO = "ui-monospace,SFMono-Regular,Menlo,Consolas,monospace" GOLD, BLUE, GREEN, INK, DIM, BG, EDGE = "#D9A23A", "#5AD1FF", "#56D364", "#e6edf3", "#8b949e", "#0d1117", "#30363d" def vkey(p: Path): return tuple(int(x) for x in p.stem.split(".")) def section(text: str, name: str) -> str: m = re.search(rf"^## {re.escape(name)}\n(.*?)(?=^## |\Z)", text, re.S | re.M) return m.group(1) if m else "" def esc(s: str) -> str: return s.replace("&", "&").replace("<", "<").replace(">", ">") def card(title: str, rows: list, foot: list, unit: str) -> str: """rows: (label, bankml value, llama.cpp value, ratio text); bars share one scale per card.""" h = 54 + 46 * len(rows) + 18 * len(foot) + 10 top = max(max(b, g) for _, b, g, _ in rows) scale = 280 / top t = lambda x, y, s, size=12, fill=INK, w="400", anchor="start": ( # noqa: E731 f"{esc(s)}") out = [f"", f"", t(20, 32, title, 15, INK, "700")] y = 70 for label, b, g, ratio in rows: out.append(t(20, y, label, 12, DIM)) for dy, v, col, who in ((-2, b, GOLD, "bankml"), (10, g, BLUE, "llama.cpp")): w = max(2.0, v * scale) out.append(f"") out.append(t(round(208 + w, 1), y + dy + 8, f"{v:g} {unit} · {who}", 10.5)) out.append(t(700, y, ratio, 12, GREEN, "700", "end")) y += 46 for line in foot: out.append(t(20, y, line, 10.5, DIM)) y += 18 return "".join(out) + "\n" def main(): rec = Path(sys.argv[1]) if len(sys.argv) > 1 else max((ROOT / "testing/results").glob("*.txt"), key=vkey) text = rec.read_text() v = rec.stem cpu = re.search(r"· (.+?) · (\d+) threads", text) machine = cpu.group(1).replace(" with Radeon Vega Mobile Gfx", "") if cpu else "this machine" # 1. ternary kernel, per projection (decode) and the prefill tile q2 = section(text, "ab_vs_ggml_q2_0") rows = [] for m in re.finditer(r"blk\.0\.(\w+)\.weight\s+(\d+x\d+)\s+ggml\s+[\d.]+/\s*([\d.]+).*?bankml\s+[\d.]+/\s*([\d.]+).*?\|\s*([\d.]+)x", q2): rows.append((f"{m.group(1)} {m.group(2).replace('x', ' × ')}", float(m.group(4)), float(m.group(3)), f"{float(m.group(5)):.2f}×")) pf = re.search(r"prefill .*?ggml per-pair [\d.]+/([\d.]+).*?1x4 tile [\d.]+/([\d.]+) \(([\d.]+)x\)", q2) if pf: rows.append(("prefill tile 1×4", float(pf.group(2)), float(pf.group(1)), f"{float(pf.group(3)):.2f}×")) if rows: (ROOT / "docs/cards/ternary_speed.svg").write_text(card( "Ternary Q2_0_g64 kernel vs llama.cpp b11192 — same bits, 1 thread", rows, [f"median ns per 64-weight block (prefill: per block·column) · {machine} · 1 thread", f"gate record testing/results/{v}.txt · every oracle bit-exact in the same run"], "ns")) # 2. 1-bit kernel: decode GEMV and the prefill tile q1 = section(text, "ab_vs_ggml") g = re.search(r"ggml b11192 [\d.]+/([\d.]+) \|.*?vec_dot_act [\d.]+/([\d.]+) \(([\d.]+)x\)", q1) p = re.search(r"ggml per-pair [\d.]+/([\d.]+).*?Q8Act tile [\d.]+/([\d.]+) \(([\d.]+)x\)", q1) if g and p: (ROOT / "docs/cards/q1_speed.svg").write_text(card( "1-bit Q1_0 kernel vs llama.cpp b11192 — same bits, 1 thread", [("decode GEMV 12288 × 4096", float(g.group(2)), float(g.group(1)), f"{float(g.group(3)):.2f}×"), ("prefill tile 1×4", float(p.group(2)), float(p.group(1)), f"{float(p.group(3)):.2f}×")], [f"median ns per 128-weight block (prefill: per block·column) · {machine} · 1 thread", f"gate record testing/results/{v}.txt · decode at parity: the 1-bit kernel is at the core's instruction limit"], "ns")) # 3. one ternary token, all matmuls, 3 threads db = section(text, "decode_budget_q2_0") m = re.search(r"3 thread\(s\): ggml b11192 ([\d.]+)/([\d.]+) s/token.*?bankml ([\d.]+)/([\d.]+) s/token \| ([\d.]+)x \| matmul-only ceiling ggml ([\d.]+), bankml ([\d.]+)", db) floor = re.search(r"3 thread\(s\): [\d.]+ GB/s read → floor for one ternary token \(([\d.]+) GB\) ([\d.]+) s", section(text, "bench_memory_floor")) if m: gmin, gmed, bmin, bmed = (float(m.group(k)) for k in (1, 2, 3, 4)) b, gg = bmin, gmin note = ("the model stayed resident in memory" if bmed < 0.6 else "NOT resident: too little free memory to cache the 2.31 GB file, so this run measured the disk") (ROOT / "docs/cards/decode_budget.svg").write_text(card( "One ternary token: all 253 matmuls, 3 threads", [("253 matmuls, 1 token", b, gg, f"{gg / b:.2f}×")], [f"fastest of the runs (min) · median {bmed:g} s vs {gmed:g} s ({gmed / bmed:.2f}×) · {machine}", note, f"matmul ceiling {m.group(7)} tok/s vs {m.group(6)}" + (f" · memory floor {floor.group(2)} s/token ({floor.group(1)} GB read)" if floor else ""), f"gate record testing/results/{v}.txt · matmuls only; since 0.3.0 bankML serves the whole model itself, token-identical"], "s")) print(f"cards drawn from testing/results/{v}.txt") if __name__ == "__main__": main()