"""The README headline must be generated, and must stay in sync. It drifted once: the controller comparison carried numbers from a `runs/arm-*` checkpoint after the shipped one had changed. That is the same failure this repository criticises NanoJev's demo for, so it gets a test rather than a convention. """ from __future__ import annotations import importlib.util import json import tempfile from pathlib import Path import pytest ROOT = Path(__file__).resolve().parents[1] def _load(): spec = importlib.util.spec_from_file_location( "readme_table", ROOT / "scripts" / "readme_table.py") mod = importlib.util.module_from_spec(spec) spec.loader.exec_module(mod) return mod rt = _load() def _run_dir(tmp_path: Path) -> Path: play = tmp_path / "play" play.mkdir() def maze(sr, eff, cells, n=36): return {"summary": {"maze": {"solve_rate": sr, "mean_efficiency": eff, "mean_distinct_cells": cells, "episodes": n}}} def snake(f, mx, surv, n=6): return {"summary": {"snake": {"mean_food": f, "max_food": mx, "survival_rate": surv, "episodes": n}}} for name, body in {"maze_model": maze(0.61, 0.812, 40.2), "maze_model_field": maze(1.0, 0.993, 22.1), "maze_random": maze(0.08, 0.140, 88.0), "maze_reference": maze(1.0, 1.0, 21.0), "snake_model": snake(9.33, 14, 0.5), "snake_random": snake(0.83, 2, 0.0)}.items(): (play / f"{name}_summary.json").write_text(json.dumps(body)) # Keyed exactly as scripts/evaluate.py writes it: Path(split).stem, so a # bare split name. The previous fixture used "data/frozen/test.jsonl", # which nothing produces, and so certified a lookup that matched nothing # in the real artifact and silently dropped the accuracy table. (tmp_path / "eval.json").write_text(json.dumps({"test": { "groups": {"maze/choice": {"n": 97, "boards": 97, "accuracy": 0.9794, "uniform_control": 0.4905}}}})) return tmp_path def test_headline_reports_accuracy_beside_its_control(tmp_path): block = "\n".join(rt.render(_run_dir(tmp_path))) assert "0.9794" in block and "0.4905" in block, \ "an accuracy must never be shown without the control next to it" def test_controls_are_not_bolded(tmp_path): """Bolding the floor's solve rate reads as a boast about the floor.""" block = "\n".join(rt.render(_run_dir(tmp_path))) floor = next(l for l in block.splitlines() if "`random` (floor)" in l and "0.08" in l) assert "**0.08**" not in floor model = next(l for l in block.splitlines() if "(network alone)" in l) assert "**0.61**" in model def test_episode_counts_come_from_the_data(tmp_path): """Hardcoding them is the drift this script exists to remove.""" run = _run_dir(tmp_path) p = run / "play" / "maze_model_summary.json" body = json.loads(p.read_text()) body["summary"]["maze"]["episodes"] = 99 p.write_text(json.dumps(body)) assert "99 episodes" in "\n".join(rt.render(run)) def test_missing_artifacts_say_so_rather_than_render_an_empty_table(tmp_path): block = "\n".join(rt.render(tmp_path)) assert "benchmark.sh" in block assert "| ---" not in block def test_fixture_key_matches_what_evaluate_py_writes(): """The fixture's eval.json key must be the one evaluate.py produces. This is the test that was missing. The headline lookup and the evaluation writer had never been run against each other, so the fixture was free to invent a key format, and it did -- leaving a generator that rendered no accuracy table at all against a real artifact while the suite stayed green. Reading the writer's keying expression keeps the two joined. """ src = (ROOT / "scripts" / "evaluate.py").read_text() assert "name = Path(path).stem" in src, ( "evaluate.py no longer keys results by split stem; update pick_split " "and this fixture together") def test_headline_survives_a_real_evaluate_py_key(tmp_path): block = "\n".join(rt.render(_run_dir(tmp_path))) assert "Decision accuracy" in block and "0.9794" in block assert "`test` split" in block, "the headline must name the split it quotes" def test_unrecognised_eval_shape_complains_instead_of_rendering_nothing(tmp_path): """Silence was the original failure; an empty table looked like no data.""" run = _run_dir(tmp_path) (run / "eval.json").write_text(json.dumps({"data/frozen/weird.jsonl": { "groups": {"maze/choice": {"n": 1, "accuracy": 1.0, "uniform_control": 0.5}}}})) block = "\n".join(rt.render(run)) assert "no recognised split" in block assert "0.9794" not in block def _memory_run(tmp_path: Path, model_11=0.92, random_11=0.92, model_21=0.33, random_21=0.83) -> Path: play = tmp_path / "play" play.mkdir(exist_ok=True) def sizes(s11, e11, s21, e21): return {"summary": {"maze": {"by_size": { "11": {"episodes": 12, "solve_rate": s11, "mean_efficiency": e11}, "21": {"episodes": 12, "solve_rate": s21, "mean_efficiency": e21}}}}} (play / "maze_model_memory_summary.json").write_text(json.dumps( sizes(model_11, 0.926, model_21, 0.869))) (play / "maze_random_memory_summary.json").write_text(json.dumps( sizes(random_11, 0.588, random_21, 0.331))) return tmp_path def test_memory_table_bolds_the_winner_not_the_model(tmp_path): """At size 21 the control wins solve rate, and the table must say so. This is the table's whole reason for existing: `model+memory` cannot be read without its control. A renderer that always bolded the model row would turn a loss into an advertisement, which is exactly the reading this repository faults NanoJev's demo for. """ block = "\n".join(rt.render_memory(_memory_run(tmp_path))) rows = {} for line in block.splitlines(): if line.startswith("| `"): ctrl, size = line.split("|")[1].strip(), line.split("|")[2].strip() rows[(ctrl, size)] = line assert "**0.83**" in rows[("`random+memory`", "21")], \ "the control wins solve rate at 21 and the table must show it" assert "**0.33**" not in rows[("`model+memory`", "21")] # Efficiency is the column that does belong to the network, at both sizes. assert "**0.869**" in rows[("`model+memory`", "21")] assert "**0.926**" in rows[("`model+memory`", "11")] def test_memory_table_ties_are_not_bolded(tmp_path): """Size 11 solve rate is a tie in the real data; a tie is not a win.""" block = "\n".join(rt.render_memory(_memory_run(tmp_path))) for line in block.splitlines(): if line.startswith("| `") and line.split("|")[2].strip() == "11": assert "**0.92**" not in line, "0.92 vs 0.92 is a tie, not a win" def test_memory_table_reports_every_size_it_has(tmp_path): """The finding is the size dependence, so averaging sizes away loses it.""" block = "\n".join(rt.render_memory(_memory_run(tmp_path))) assert len([l for l in block.splitlines() if l.startswith("| `")]) == 4 def test_memory_table_says_so_when_the_control_is_missing(tmp_path): """Rendering the model row alone would be worse than rendering nothing.""" run = _memory_run(tmp_path) (run / "play" / "maze_random_memory_summary.json").unlink() block = "\n".join(rt.render_memory(run)) assert "benchmark.sh" in block and "| ---" not in block def test_readme_prose_still_matches_the_memory_table(): """A generated table under un-generated prose is only half-fixed. The README reads the table for the reader, so a new checkpoint can turn the paragraph into a false statement about its own table. It used to predict the table: solve rate at 11 belongs to the memory, the control *wins* at 21. Both went false when the model improved, which is the wrong way for a claim to be fragile -- the prose should not need editing because the model got better. Those sentences now point at the table's own bolding instead, which `test_memory_table_bolds_the_winner_ not_the_model` covers. What remains is the one directional claim the paragraph still makes on its own authority: efficiency is the column that belongs to the network. If that stops being true the finding has changed, not just the numbers, and the paragraph needs rewriting rather than regenerating. """ run = ROOT / "runs" / "jevon-final" if not (run / "play" / "maze_random_memory_summary.json").is_file(): pytest.skip("no benchmark artifacts yet") play = run / "play" def by_size(name): body = json.loads((play / f"{name}_summary.json").read_text()) return body["summary"]["maze"]["by_size"] model, control = by_size("maze_model_memory"), by_size("maze_random_memory") for size in ("11", "21"): if size not in model or size not in control: continue assert model[size]["mean_efficiency"] > control[size]["mean_efficiency"], ( f"README claims efficiency belongs to the network at size {size}; " "it no longer does") # The paragraph also claims the model arrives "on very nearly the # shortest path". Efficiency is shortest/steps over solved episodes, so # that is a claim about a number, not a mood. for size in ("11", "21"): if size in model and model[size]["solve_rate"] > 0: assert model[size]["mean_efficiency"] >= 0.75, ( f"README says the model arrives on very nearly the shortest " f"path; efficiency at size {size} is " f"{model[size]['mean_efficiency']:.3f}") def test_readme_headline_is_in_sync(): """The shipped README must match what the current artifacts render. Skipped rather than failed when the run directory is absent, so a fresh clone that has not trained yet still passes the suite. """ run = ROOT / "runs" / "jevon-final" if not run.is_dir(): pytest.skip("no shipped run to render from") text = (ROOT / "README.md").read_text() for name, renderer in rt.BLOCKS.items(): b, e = rt.begin(name), rt.end(name) assert b in text and e in text, f"missing markers for {name!r}" head, rest = text.split(b, 1) _, tail = rest.split(e, 1) # rt.block, not a second copy of the wrapping: the copy that used to # live here drifted out of step with main() the moment the markers # moved, and compared marker-less output against a README with them. rebuilt = head + "\n".join(rt.block(name, renderer(run))) + tail assert rebuilt == text, \ f"README {name} block is stale; run scripts/readme_table.py" def test_no_renderer_emits_its_own_markers(tmp_path): """Marker emission belongs to main(), and must stay there. It used to be each renderer's job. A renderer that forgot deleted its own markers on the first write, and every run after that died with "markers not found in README.md" -- an error that points at the README, which was fine until this script ate it. The new block hit exactly that on its first run, which is why the contract is now enforced rather than documented. """ for name, renderer in rt.BLOCKS.items(): body = "\n".join(renderer(tmp_path)) assert "BEGIN:" not in body and "END:" not in body, ( f"{name} renderer emits markers; main() adds them and will " "double them") def test_every_block_renders_something_without_artifacts(tmp_path): """A fresh clone has no run directory. Each block must say so, not crash. Rendering nothing is the failure that matters here: an empty block reads as a section the author forgot to fill in, and the marker pair makes it invisible in the rendered page. """ for name, renderer in rt.BLOCKS.items(): body = [l for l in renderer(tmp_path) if l.strip()] assert body, f"{name} renders nothing when the run directory is empty" def _history(tmp_path, snake, maze=None): run = tmp_path / "run" run.mkdir(exist_ok=True) maze = maze or [1.0] * len(snake) (run / "history.json").write_text(json.dumps( [{"step": (i + 1) * 500, "snake_action": s, "maze_action": m} for i, (s, m) in enumerate(zip(snake, maze))])) return run def test_volatility_infers_the_split_size_from_the_accuracy_grid(): """74 questions, recovered from the values rather than hard-coded. An accuracy over n questions is a multiple of 1/n, so the smallest gap between distinct observed values bounds n. Hard-coding 74 would be a fourth place the split size is written down, and the one nobody would think to update when the generator changes. """ run = _history(Path(tempfile.mkdtemp()), [70 / 74, 73 / 74, 67 / 74, 74 / 74]) body = "\n".join(rt.render_volatility(run)) assert "of 74" in body, body def test_volatility_halves_do_not_drop_the_move_across_the_seam(): """The halves overlap by one eval on purpose. A move is a property of a *pair* of evals, so splitting the list cleanly leaves the move spanning the seam in neither half. Here the only large move is exactly that one: a clean split reports 0.000 for both halves and the table silently loses the finding. """ run = _history(Path(tempfile.mkdtemp()), [0.5, 0.5, 0.5, 0.9, 0.9, 0.9]) body = "\n".join(rt.render_volatility(run)) assert "`0.400`" in body, f"seam move missing:\n{body}" def test_volatility_reports_a_run_that_never_moves(): """A perfectly flat history is a real state, not a reason to divide by zero.""" run = _history(Path(tempfile.mkdtemp()), [0.9] * 6) body = "\n".join(rt.render_volatility(run)) assert "`0.000`" in body, body def _selection(tmp, best, balanced): run = Path(tempfile.mkdtemp()) / "run" run.mkdir(parents=True) (run / "best.json").write_text(json.dumps(best)) (run / "balanced.json").write_text(json.dumps(balanced)) return run def test_shipped_table_bolds_the_lower_loss_and_the_higher_accuracy(): """Loss is the one row where smaller wins. Bolding the larger number there would invert the table's whole argument while looking perfectly well-formed -- it would show `balanced.pt` winning loss, which is the opposite of the trade the section is about. """ run = _selection(None, {"step": 7500, "loss": 0.40, "acc_choice": 0.93}, {"step": 3000, "loss": 0.48, "acc_choice": 0.99}) body = rt.render_shipped(run) loss = next(l for l in body if l.startswith("| eval loss")) choice = next(l for l in body if l.startswith("| `acc_choice`")) assert "**0.4000**" in loss and "**" not in loss.split("|")[3], loss assert "**0.9900**" in choice and "**" not in choice.split("|")[2], choice def test_shipped_table_says_so_when_both_selectors_agree(): """Two identical columns are not a contrast, and printing them as one would be a worse answer than saying there is nothing to show.""" metrics = {"step": 4000, "loss": 0.42, "acc_choice": 0.97} body = "\n".join(rt.render_shipped(_selection(None, metrics, dict(metrics)))) assert "4000" in body and "|" not in body, body def test_shipped_table_does_not_bold_a_tied_row(): run = _selection(None, {"step": 7500, "loss": 0.40, "maze_action": 1.0}, {"step": 3000, "loss": 0.48, "maze_action": 1.0}) maze = next(l for l in rt.render_shipped(run) if l.startswith("| `maze_action`")) assert "**" not in maze, maze def test_shipped_table_skips_metrics_a_run_never_wrote(): run = _selection(None, {"step": 7500, "loss": 0.40}, {"step": 3000, "loss": 0.48}) body = "\n".join(rt.render_shipped(run)) assert "acc_choice" not in body and "eval loss" in body, body def test_readme_prose_still_matches_the_shipped_table(): """The paragraph says `balanced.pt` takes the choice questions. Not true by construction: the balanced selector maximises `min(maze_action, snake_action)`, which is two subsets of the 171 choice questions, so it is possible for `best.pt` to win `acc_choice` overall. If that happens the paragraph is wrong and needs rewriting, so it is asserted rather than assumed. Skipped when the two selectors agreed, which is a real outcome and not a failure. """ run = ROOT / "runs" / "jevon-final" best, balanced = run / "best.json", run / "balanced.json" if not (best.exists() and balanced.exists()): pytest.skip("no selection metadata yet") a, b = json.loads(best.read_text()), json.loads(balanced.read_text()) if a.get("step") == b.get("step"): pytest.skip("both selectors chose the same checkpoint") assert b["acc_choice"] > a["acc_choice"], ( f"README says balanced.pt takes the choice questions; it has " f"{b['acc_choice']:.4f} against best.pt's {a['acc_choice']:.4f}")