"""Generate the comparison only from completed measurements.""" import math from common import RUN, read_json, stamp def english_python_ppl(scores): groups = [scores[k] for k in ["en", "code"] if k in scores] if len(groups) != 2: return None return math.exp(sum(g["tokens"] * math.log(g["ppl"]) for g in groups) / sum(g["tokens"] for g in groups)) def interval(correct, n): z = 1.96 p = correct / n center = (p + z*z/(2*n))/(1+z*z/n) radius = z*math.sqrt(p*(1-p)/n+z*z/(4*n*n))/(1+z*z/n) return f"{100*p:.1f}% ({100*(center-radius):.1f}–{100*(center+radius):.1f})" def main(): lines = ["# Local Swift / HyperQwen comparison", "", "Updated: " + stamp(), "", "RTX 3090 measurements. A task is one independently scored problem; tool tasks include every model call. " "Output tokens include reasoning. Latency is observed serving time, not a GPU-kernel compute measurement. " "Quality runs use bounded budgets and one seed (15027). Thinking tasks use temperature 1.0, top_p 0.95, top_k 20; nonthinking tasks use greedy decoding. These are not publisher leaderboard scores.", "", "The Qwen reference is the existing AutoRound fast checkpoint; Swift 1.0 is the existing AWQ/INT8-head checkpoint. " "The Swift 1.5 INT8-head baseline and INT4-head fast derivative share the same upstream AWQ body. " "A BF16 reference measurement is outside this four-checkpoint comparison.", "", "## Fixed-output speed and perplexity", "", "512 output tokens/request, concurrency 1; natural task lengths are reported separately. Perplexity uses English and Python only; Danish is excluded.", "", "| Model | Decode tok/s (median) | TTFT (s) | EN PPL | Code PPL | EN/Python PPL |", "|---|---:|---:|---:|---:|---:|"] tags = ["qwen-fast", "swift10", "swift15-baseline", "swift15-fast"] for tag in tags: directory = RUN / "results" / tag speed = read_json(directory / "speed-c1.json") if (directory / "speed-c1.json").exists() else {} ppl = read_json(directory / "perplexity.json").get("scores", {}) if (directory / "perplexity.json").exists() else {} def f(v): return "pending" if v is None else f"{v:.3f}" lines.append("| " + " | ".join([tag, f(speed.get("median_decode_tps")), f(speed.get("median_ttft_seconds")), *[f(ppl.get(k, {}).get("ppl")) for k in ["en", "code"]], f(english_python_ppl(ppl))]) + " |") lines += ["", "## Swift 1.5 multi-user serving", "", "Same fast checkpoint, no MTP, concurrency eight. INT8 applies to MLP activations only.", "", "| Runtime | Aggregate output tok/s | Combined PPL |", "|---|---:|---:|"] for tag in ["swift15-fast-batch", "swift15-fast-batch-int8"]: directory = RUN / "results" / tag if not (directory / "speed-c8.json").exists(): continue speed = read_json(directory / "speed-c8.json") ppl = english_python_ppl(read_json(directory / "perplexity.json")["scores"]) if (directory / "perplexity.json").exists() else None lines.append(f"| {tag} | {speed['aggregate_output_tps']:.2f} | {ppl if ppl is not None else 'pending'} |") for mode in ["pilot", "full"]: lines += ["", "## " + ("Sequential task latency pilot" if mode == "pilot" else "Quality suite (concurrency recorded per run)"), "", "Accuracy includes all attempts. Truncated responses count as wrong; their count is reported separately. Parentheses show descriptive 95% Wilson intervals. Small differences are inconclusive.", "", "| Model | Task suite | Correct/total | Accuracy (95% CI) | Output tokens/task | Total tokens/task | Request seconds/task | Truncated | Errors |", "|---|---|---:|---:|---:|---:|---:|---:|---:|"] for tag in tags: path = RUN / "results" / (tag + "-" + mode) / "summary.json" if not path.exists(): continue summary = read_json(path) for suite, values in summary["suites"].items(): v = values accuracy = interval(v['correct'],v['attempted']) lines.append(f"| {tag} | {suite} | {v['correct']}/{v['attempted']} | {accuracy} | " f"{v['mean_output_tokens']:.1f} | {v['mean_total_tokens']:.1f} | {v['mean_model_seconds']:.2f} | {v['truncated']} | {v['errors']} |") lines += ["", "Full-suite request latencies overlap and must not be summed as GPU time. " "Per-run JSON summaries also report wall-clock makespan and wall seconds per correct task, including failed attempts. " "Resumed runs preserve attempt timings and omit a misleading whole-suite makespan.", "", "LiveCodeBench is a fixed 100-problem, stdin-only v6 subset with a custom deterministic judge, " "not an official full LCB score. IFBench uses the official instruction verifier. " "The 20-task latency pilot has five tasks per category and is too small to establish quality equivalence.", ""] (RUN / "REPORT.md").write_text("\n".join(lines)) print(RUN / "REPORT.md") if __name__ == "__main__": main()