jevon / tests /test_results.py
lewislululu's picture
Jevon: a 20M-parameter decision model for Maze and Snake
97f78c9
Raw History Blame Contribute Delete
43.2 kB
"""The report generator must survive the runs you most want reported.
``make_results.py`` is the last step of ``benchmark.sh``, so a crash there
destroys a whole evaluation sweep after it has already been paid for. The two
cases below are the ones that actually bit: a controller that solved nothing
(so ``mean_steps_when_solved`` is ``None``), and a probe file written before
the ``reach`` column existed.
"""
from __future__ import annotations
import importlib.util
import json
import re
import subprocess
import sys
from collections import Counter
from pathlib import Path
import pytest
ROOT = Path(__file__).resolve().parents[1]
_spec = importlib.util.spec_from_file_location(
"make_results", ROOT / "scripts" / "make_results.py")
make_results = importlib.util.module_from_spec(_spec)
sys.modules["make_results"] = make_results
_spec.loader.exec_module(make_results)
def _maze(solve_rate, steps):
return {"summary": {"maze": {
"solve_rate": solve_rate, "mean_steps_when_solved": steps,
"mean_efficiency": 0.0 if steps is None else 1.0,
"mean_collisions": 4.2}}}
def test_unsolved_controller_renders_instead_of_crashing():
# A model that solves nothing is a real result, not a broken input.
text = "\n".join(make_results.play_section({"maze_model": _maze(0.0, None)}))
assert "| 0.00 |" in text
assert "--" in text
def test_solved_controller_still_prints_its_step_count():
text = "\n".join(make_results.play_section({"maze_field": _maze(1.0, 82.0)}))
assert "82.0" in text
def test_probe_row_reports_reach():
row = {"cells": 60, "probe_r2": 0.9, "probe_mae": 1.2, "head_r2": 0.5,
"descent_acc": 0.71, "reach_rate": 1.0, "diameter": 33,
"iterations": 82}
text = "\n".join(make_results.probe_section({"11": row}))
assert "reach" in text
# Bolded, because it is the column that predicts solve rate.
assert "**1.0000**" in text
def test_probe_row_tolerates_a_pre_reach_probe_file():
row = {"cells": 60, "probe_r2": 0.9, "probe_mae": 1.2, "head_r2": 0.5,
"descent_acc": 0.71, "diameter": 33, "iterations": 82}
text = "\n".join(make_results.probe_section({"11": row}))
assert "nan" in text
def test_missing_inputs_are_skipped_not_guessed():
assert make_results.probe_section(None) == []
assert make_results.play_section({}) == []
def test_deadlock_columns_render():
block = _maze(0.0, None)
block["summary"]["maze"].update(deadlock_rate=1.0, mean_distinct_cells=1.0)
text = "\n".join(make_results.play_section({"maze_model": block}))
assert "deadlock" in text
assert "| 1.00 | 1.0 |" in text
def test_summary_without_deadlock_fields_still_renders():
# Summaries recorded before the diagnostic existed must not crash the table.
text = "\n".join(make_results.play_section({"maze_model": _maze(1.0, 82.0)}))
assert "82.0" in text
def test_unlisted_play_configs_still_render_at_the_end(tmp_path):
"""Adding a play config must never silently drop it from the table."""
play = tmp_path / "play"
play.mkdir()
for name in ("maze_model", "maze_brand_new_idea"):
(play / f"{name}_summary.json").write_text(json.dumps({"summary": {"maze": {
"episodes": 1, "solve_rate": 1.0, "mean_steps_when_solved": 10.0,
"mean_efficiency": 1.0, "mean_collisions": 0.0}}}))
summaries = {p.stem.replace("_summary", ""): make_results.read(p)
for p in sorted(play.glob("*_summary.json"))}
rows = [line for line in make_results.play_section(summaries) if line.startswith("| `")]
assert any("maze_brand_new_idea" in r for r in rows)
assert rows[0].startswith("| `maze_model`"), "known rows keep reading order"
def test_the_control_is_printed_directly_below_the_row_it_controls(tmp_path):
play = tmp_path / "play"
play.mkdir()
# Written in an order that sorting would get wrong.
for name in ("maze_random_memory", "maze_model_memory", "maze_model"):
(play / f"{name}_summary.json").write_text(json.dumps({"summary": {"maze": {
"episodes": 1, "solve_rate": 0.5, "mean_steps_when_solved": 10.0,
"mean_efficiency": 0.5, "mean_collisions": 0.0}}}))
summaries = {p.stem.replace("_summary", ""): make_results.read(p)
for p in sorted(play.glob("*_summary.json"))}
rows = [line for line in make_results.play_section(summaries) if line.startswith("| `")]
names = [r.split("`")[1] for r in rows]
assert names.index("maze_random_memory") == names.index("maze_model_memory") + 1
# ------------------------------------------------- data independence
@pytest.mark.parametrize("split", ["val", "test", "ood"])
def test_frozen_splits_draw_a_fresh_board_per_state(split):
"""Every accuracy row's `n` claims that many independent tests. Check it.
Questions that share a board share walls, goal and distance field, so they
succeed or fail together and n overstates the evidence. The generator is
supposed to draw a fresh board per state, which makes n honest -- but that
is a property of the generator, and generators get edited.
This was not hypothetical. An earlier reading of these splits reported the
OOD maze questions as coming from 8 boards, and shipped that claim into
two generated tables. It came from parsing the uid with the wrong field
index, not from the data. So this test reads the board off the tensor the
model is actually fed -- channel 0 and the goal cell -- rather than off a
uid convention, because the convention is what was wrong last time.
"""
import hashlib
path = ROOT / "data" / "frozen" / f"{split}.jsonl"
if not path.is_file():
pytest.skip(f"{path} not generated")
rows = [json.loads(line) for line in path.read_text().splitlines() if line]
for game in ("maze", "snake"):
states = [r for r in rows if r["game"] == game]
if not states:
continue
seen = set()
for r in states:
_, h, w = r["shape"]
seen.add(hashlib.sha1(json.dumps(
[h, w, r["board"][:h * w], r.get("target")]).encode()).digest())
# Not equality: two small mazes can coincide by chance, and three do.
# The guard is against a generator that reuses one board across many
# states, which is a different order of magnitude.
assert len(seen) >= 0.95 * len(states), (
f"{split}/{game}: {len(states)} states share only {len(seen)} "
"distinct boards; `n` no longer measures independent tests and "
"the `boards` column in RESULTS.md is now the denominator to read")
def test_checkpoint_selection_table_matches_the_run_history():
"""The README's `best.pt` vs `balanced.pt` table, checked against history.
That table is the whole argument for selecting on min(maze, snake) rather
than on loss, and it quotes one run at two specific steps. The checkpoints
it describes are long overwritten -- training keeps only the current
best.pt and balanced.pt -- but history.json keeps every eval row forever,
so the claim stays checkable after the evidence for it is gone.
Steps are parsed out of the column headers rather than hardcoded, so
re-pointing the table at different steps is a README edit and this follows
it. A retrained run that no longer shows the trade will fail here, which
is correct: the argument would then need different evidence, not a
quietly stale table.
"""
import re
history = ROOT / "runs" / "jevon-final" / "history.json"
if not history.is_file():
pytest.skip("no training history yet")
rows = {r["step"]: r for r in json.loads(history.read_text())}
text = (ROOT / "README.md").read_text()
header = next((l for l in text.splitlines()
if "`best.pt` (step" in l and "`balanced.pt` (step" in l), None)
assert header, "the checkpoint-selection table header is gone from the README"
steps = [int(s) for s in re.findall(r"step (\d+)", header)]
assert len(steps) == 2, header
for step in steps:
assert step in rows, f"README quotes step {step}, absent from history.json"
# [1:] drops the remainder of the header line itself, which is empty and
# would end the loop on its first iteration.
table = text.split(header, 1)[1].split("\n")[1:]
fields = {"eval loss": "loss", "`acc_boolean`": "acc_boolean",
"`acc_score`": "acc_score", "`acc_choice`": "acc_choice",
"`snake_action`": "snake_action"}
checked = 0
for line in table:
if not line.startswith("|"):
break
cells = [c.strip() for c in line.strip("|").split("|")]
key = fields.get(cells[0])
if key is None or len(cells) != 3:
continue
for step, cell in zip(steps, cells[1:]):
shown = cell.strip("*")
# Compared at the precision the README prints, which is the point:
# acc_boolean is quoted to five decimals precisely to show the two
# checkpoints are identical there.
places = len(shown.split(".")[1]) if "." in shown else 0
actual = f"{rows[step][key]:.{places}f}"
assert shown == actual, (
f"README says {key} = {shown} at step {step}; "
f"history.json says {actual}")
checked += 1
assert checked == 10, f"expected 10 cells checked, got {checked}"
def test_readme_parameter_count_matches_the_shipped_config():
"""The headline calls this a 20M-parameter model. Check it against the
architecture the shipped run actually used.
Built from `config.json` rather than loaded from `balanced.pt`, for two
reasons: an 80MB state-dict load is slow enough to notice in a test suite,
and the config is the thing a reader of the README can inspect. If the two
ever disagreed the checkpoint would not load at all, which
`inference.py` would report far more loudly than this test could.
A bare "20M" tolerates a range, so the assertion is the range a reader
would accept for that phrase, not the exact integer -- the point is to
catch an architecture change that makes the front page wrong, not to
re-pin the number every time a bias term moves. The exact figure in the
header table is pinned separately, by
`test_the_model_card_parameter_count_is_exact`.
Found by content rather than at a fixed line offset. This read line 3
until the Hub front matter was added above it, at which point it was
asserting about `pipeline_tag` -- a test that breaks when an unrelated
paragraph is inserted is a test that will eventually be deleted rather
than fixed.
"""
cfg_path = ROOT / "runs" / "jevon-final" / "config.json"
if not cfg_path.exists():
pytest.skip("no shipped run to check against")
sys.path.insert(0, str(ROOT))
from jevon.modeling import JevonConfig, JevonModel
model = JevonModel(JevonConfig(**json.loads(cfg_path.read_text())))
total = sum(p.numel() for p in model.parameters())
claim = next((line for line in (ROOT / "README.md").read_text().splitlines()
if "20M-parameter" in line), None)
assert claim, "README headline no longer calls this a 20M-parameter model"
assert 19.5e6 <= total <= 20.5e6, (
f"README says 20M, config.json builds {total:,} "
f"({total / 1e6:.2f}M)")
def test_readme_reproduction_command_is_the_one_that_trained_the_checkpoint():
""""Reproduce it with:" has to be true, or the repo is worse than silent.
A reader who runs the quoted command and gets different numbers has no
way to tell whether the model is flaky or the page is stale, and will
reasonably conclude the former. train.py records `sys.argv` verbatim in
`args.json` for exactly this comparison, so the check is against what the
shipped run was actually launched with rather than against a second copy
of the same prose.
Compared as a flag set, not as a string: the README wraps the line with a
backslash and runs under `python` where the run used the venv interpreter
with `-u`, and neither difference changes what gets trained.
"""
args_path = ROOT / "runs" / "jevon-final" / "args.json"
if not args_path.exists():
pytest.skip("no shipped run to check against")
argv = json.loads(args_path.read_text())["argv"]
# Drop the interpreter's own flags and the script path; what remains is
# the training configuration the README is promising.
actual = argv[argv.index("scripts/train.py") + 1:]
text = (ROOT / "README.md").read_text()
quoted = None
for block in re.findall(r"```(?:[a-z]*)\n(.*?)```", text, re.S):
joined = block.replace("\\\n", " ")
for line in joined.splitlines():
if "scripts/train.py" in line and "--out runs/jevon-final" in line:
quoted = line.split("scripts/train.py", 1)[1].split()
assert quoted is not None, (
"README no longer quotes a train.py command for runs/jevon-final; "
"the reproduction instructions have gone missing")
assert quoted == actual, (
f"README says:\n {' '.join(quoted)}\nrun used:\n {' '.join(actual)}")
def test_readme_val_split_composition_matches_the_frozen_data():
"""The selection argument rests on these five numbers.
"73.5% of eval loss is booleans" is why `best.pt` can be wrong, and it is
a property of the frozen split, not of the model -- so it changes if
anyone edits the question profile, and it changes silently, because the
split regenerates without the README noticing. Read out of the data
rather than asserted against constants typed here, which would just be a
sixth copy of the same claim.
"""
path = ROOT / "data" / "frozen" / "val.jsonl"
if not path.exists():
pytest.skip("no frozen val split; run scripts/build_dataset.py")
states = [json.loads(line) for line in path.read_text().splitlines() if line]
kinds = Counter(q["type"] for st in states for q in st["questions"])
total = sum(kinds.values())
text = (ROOT / "README.md").read_text()
claim = next((p for p in text.split("\n\n")
if "frozen validation split is" in p), None)
assert claim, "README no longer describes the val split composition"
for number in (str(total), str(len(states)), str(kinds["boolean"]),
str(kinds["score"]), str(kinds["choice"])):
assert number in claim, (
f"README's val-split paragraph does not mention {number}; "
f"measured {total} questions over {len(states)} states, "
f"{dict(kinds)}")
# The two percentages the argument actually turns on.
for share, label in ((kinds["boolean"] / total, "boolean"),
(kinds["choice"] / total, "choice")):
assert f"{share * 100:.1f}%" in claim, (
f"README's {label} share is stale; measured {share * 100:.1f}%")
def test_readme_selection_example_matches_the_run_it_claims_to_quote():
"""The table making the case for `balanced.pt` is hand-typed, from disk.
It is a historical observation -- two checkpoints that existed at one
instant and were overwritten minutes later -- so it cannot be generated
the way the shipped-checkpoint table is. But the evals it quotes are
still in `history.json`, so it can be checked, and it has to be: the
whole selection argument rests on "boolean and score are identical to
five decimal places, only choice moved". Retrain the run and every cell
is about a different model while the prose reads the same.
"""
hist_path = ROOT / "runs" / "jevon-final" / "history.json"
if not hist_path.exists():
pytest.skip("no shipped run to check against")
history = {r["step"]: r for r in json.loads(hist_path.read_text())}
text = (ROOT / "README.md").read_text()
# Generated blocks contain a table with the same shape; this one is the
# hand-typed one, so the generated ones are removed before searching.
prose = re.sub(r"<!-- BEGIN:.*?<!-- END:\w+ -->", "", text, flags=re.S)
header = re.search(r"\| \| `best\.pt` \(step (\d+)\) \| "
r"`balanced\.pt` \(step (\d+)\) \|\n(.*?)\n\n",
prose, re.S)
assert header, "README no longer shows the mid-training selection table"
best_step, balanced_step = int(header.group(1)), int(header.group(2))
for step in (best_step, balanced_step):
assert step in history, (
f"README quotes step {step}, which this run never evaluated; "
f"evals are at {sorted(history)}")
fields = {"eval loss": "loss", "`acc_boolean`": "acc_boolean",
"`acc_score`": "acc_score", "`acc_choice`": "acc_choice",
"`snake_action`": "snake_action", "`maze_action`": "maze_action"}
checked = 0
for row in header.group(3).splitlines():
cells = [c.strip() for c in row.strip().strip("|").split("|")]
if len(cells) != 3 or cells[0] not in fields:
continue
key = fields[cells[0]]
for cell, step in zip(cells[1:], (best_step, balanced_step)):
quoted = cell.strip("*` ")
actual = history[step][key]
assert quoted == f"{actual:.{len(quoted.split('.')[1])}f}", (
f"README says {key} = {quoted} at step {step}; "
f"history.json says {actual}")
checked += 1
assert checked >= 8, f"only {checked} cells parsed out of the table"
# The two claims the prose makes *about* the table, which are what the
# reader takes away and which no cell states on its own.
gap = history[balanced_step]["acc_choice"] - history[best_step]["acc_choice"]
assert f"**{round(gap * 100)} points" in prose, (
f"README's choice-accuracy gap is stale; measured {gap * 100:.1f}")
ce = [f"{history[s]['ce']:.4f}" for s in (balanced_step, best_step)]
assert f"(`ce` {ce[0]} → {ce[1]})" in prose, (
f"README's cross-entropy pair is stale; measured {ce[0]} → {ce[1]}")
def test_readme_uniform_controls_match_the_frozen_split():
"""`0.4905` and `0.4955` are what a coin scores at this game.
They are the reason a snake score of `0.5135` is described as "a hair
above the control" rather than as half-right, so they carry the weight of
that paragraph. Neither depends on the model -- both are the mean
fraction of tied-optimal candidates per action question -- which is
exactly why they rot unnoticed: the split can be rebuilt without any
checkpoint being retrained.
Computed the way `evaluate.py` computes `uniform_control`, from `probs`
alone, so this does not need a checkpoint either.
"""
path = ROOT / "data" / "frozen" / "val.jsonl"
if not path.exists():
pytest.skip("no frozen val split; run scripts/build_dataset.py")
shares = {}
for line in path.read_text().splitlines():
if not line:
continue
state = json.loads(line)
for q in state["questions"]:
if q["name"] == "action":
top = max(q["probs"])
shares.setdefault(state["game"], []).append(
sum(v >= top - 1e-6 for v in q["probs"]) / len(q["probs"]))
assert set(shares) == {"maze", "snake"}, f"split games: {sorted(shares)}"
text = (ROOT / "README.md").read_text()
claim = next((p for p in text.split("\n\n")
if "uniform controls on this split" in p), None)
assert claim, "README no longer states the uniform controls"
for game, values in sorted(shares.items()):
control = sum(values) / len(values)
assert f"`{control:.4f}` for {game}" in claim, (
f"README's {game} uniform control is stale; measured "
f"{control:.4f} over {len(values)} action questions")
# The snake control is quoted a second time, forty lines down, to call
# `0.5135` "a hair above" it. A second copy of a number is a second thing
# to go stale, and that sentence is the one a reader believes.
snake = sum(shares["snake"]) / len(shares["snake"])
echo = re.search(r"above the `([\d.]+)` control", text)
assert echo, "README no longer compares the failing checkpoint to control"
assert echo.group(1) == f"{snake:.4f}", (
f"README's echoed snake control is stale; it says {echo.group(1)}, "
f"the split says {snake:.4f}")
SHIPPED_WEIGHTS = {"runs/jevon-final/balanced.pt", "runs/jevon-final/best.pt"}
def test_the_shipped_run_metadata_is_committed():
"""Everything the README is generated from has to survive a clone.
`.gitignore` used to read `runs/`, which excluded 72K of JSON along with
241MB of weights. The suite stayed green because every test that reads
those files skips when they are missing -- so on a fresh clone the whole
documentation-drift apparatus passed by not running, which is the exact
failure it was built to prevent.
Checked against `git ls-files` rather than the filesystem: the files are
on disk here either way, and being on disk is not the property that
matters.
"""
run = ROOT / "runs" / "jevon-final"
if not run.is_dir() or not (ROOT / ".git").exists():
pytest.skip("no shipped run, or not a git work tree")
tracked = set(subprocess.run(
["git", "ls-files", "runs/jevon-final"], cwd=ROOT,
capture_output=True, text=True, check=True).stdout.split())
on_disk = {str(p.relative_to(ROOT)) for p in run.rglob("*.json")}
missing = sorted(on_disk - tracked)
assert not missing, (
"the README is generated from files git does not track:\n "
+ "\n ".join(missing)
+ "\nevery test reading them skips, so the suite would be green "
"on a clone that has none of them")
def test_the_shipped_checkpoints_are_committed_and_no_others_are():
"""This repo is a model release, so the weights are the point of it.
The inverse of what this file used to assert. While the weights lived
outside git the rule was "no `.pt` is tracked"; publishing to the Hub
makes that rule the bug -- `from_pretrained` would download a recipe.
Both halves still matter. `balanced.pt` and `best.pt` are the two
checkpoints the model card compares, so both ship; `last.pt` is wherever
training happened to stop, which is not a claim this repo makes, and the
ablation runs under `runs/arm-*` are 80MB each of the same.
"""
if not (ROOT / ".git").exists():
pytest.skip("not a git work tree")
tracked = set(subprocess.run(
["git", "ls-files"], cwd=ROOT,
capture_output=True, text=True, check=True).stdout.split())
shipped = {f for f in tracked if f.endswith(".pt")}
assert shipped == SHIPPED_WEIGHTS, (
f"tracked checkpoints are {sorted(shipped)}; "
f"expected exactly {sorted(SHIPPED_WEIGHTS)}")
def test_the_checkpoints_are_stored_as_lfs_pointers():
"""80MB of raw blob in a git object store is the failure this guards.
The Hub rejects non-LFS files over 10MB, so getting this wrong means the
push fails -- but it fails after the repository has been rewritten to
contain 160MB that `git gc` will not reclaim, and the fix at that point is
to rebuild history. Cheaper to assert before pushing.
Two separate claims, because `.gitattributes` saying `filter=lfs` and the
committed blob *being* a pointer are not the same thing: a file staged
before the attribute existed keeps its raw blob, and check-attr will
cheerfully report `lfs` for it ever after.
"""
if not (ROOT / ".git").exists():
pytest.skip("not a git work tree")
for name in sorted(SHIPPED_WEIGHTS):
attr = subprocess.run(
["git", "check-attr", "filter", "--", name], cwd=ROOT,
capture_output=True, text=True, check=True).stdout.strip()
assert attr.endswith(": lfs"), (
f"{name} is not routed through LFS by .gitattributes ({attr})")
blob = subprocess.run(
["git", "cat-file", "-s", f"HEAD:{name}"], cwd=ROOT,
capture_output=True, text=True)
if blob.returncode != 0:
continue # not committed yet
size = int(blob.stdout)
assert size < 1024, (
f"HEAD:{name} is a {size:,}-byte blob, so it was committed as raw "
f"bytes rather than as an LFS pointer; the Hub will reject the push")
def test_no_script_defaults_to_a_run_that_does_not_ship():
"""`--checkpoint` defaults have to name a directory a clone will have.
Four scripts defaulted to `runs/jevon-s` or `runs/jevon-v3` long after the
shipped run had been renamed `runs/jevon-final`; `runs/jevon-v3` had not
existed on this machine for some time either. Nothing noticed, because
every invocation in the README and in `benchmark.sh` passes the path
explicitly -- the defaults were only ever reached by someone reading
`--help` and trusting it, which is exactly the reader this repo is being
opened for.
Tracked-ness is the test, not existence: a default naming a directory that
happens to sit in this working copy but is gitignored is the same broken
promise to a clone. `train.py --out` is excluded because it names a run to
*create*; defaulting it to the shipped one would overwrite it.
"""
if not (ROOT / ".git").exists():
pytest.skip("not a git work tree")
tracked = subprocess.run(
["git", "ls-files", "runs"], cwd=ROOT,
capture_output=True, text=True, check=True).stdout.split()
ships = {f.split("/")[1] for f in tracked if "/" in f}
cited = []
for path in sorted((ROOT / "scripts").iterdir()):
if path.suffix not in (".py", ".sh"):
continue
for lineno, line in enumerate(path.read_text().splitlines(), 1):
if "--out" in line or line.lstrip().startswith("#"):
continue
# `default="runs/x"` in argparse, `${1:-runs/x}` in shell.
for m in re.finditer(r"(?:default=\"|:-)(runs/([\w.-]+))", line):
cited.append((f"{path.name}:{lineno}", m.group(1), m.group(2)))
assert cited, "no script names a default run directory any more"
stale = [(where, ref) for where, ref, run in cited if run not in ships]
assert not stale, (
"script defaults name run directories that are not in the repo:\n "
+ "\n ".join(f"{where} {ref}" for where, ref in stale)
+ f"\nshipped: {sorted(ships)}")
def test_every_file_the_readme_links_to_is_in_the_repo():
"""A relative link is a promise that the repo contains the thing.
`RESULTS.md` is generated by the last stage of `benchmark.sh` and was
linked from the front page three times while never once being committed --
it is not even gitignored, it was simply never staged, and `git status`
listing it among the run outputs made it easy to skip. So every reader of
a fresh clone got three dead links to the document holding all the detailed
numbers, and the page most likely to be read first was the page most
broken.
Tracked-ness again, not existence: the generated file is always sitting
right here after a benchmark, which is precisely why this went unnoticed
for as long as it did.
"""
if not (ROOT / ".git").exists():
pytest.skip("not a git work tree")
text = (ROOT / "README.md").read_text()
targets = {m.group(1).split("#")[0]
for m in re.finditer(r"\]\(([^)]+)\)", text)}
local = sorted(t for t in targets
if t and not re.match(r"^(https?:|mailto:|#)", t))
assert local, "the README links to nothing in its own repo any more"
tracked = set(subprocess.run(
["git", "ls-files"], cwd=ROOT,
capture_output=True, text=True, check=True).stdout.split())
dead = [t for t in local if t not in tracked
and not (ROOT / t).is_dir()]
assert not dead, (
f"README links to files the repo does not ship: {dead}")
def test_readme_checkpoint_size_follows_from_the_parameter_count():
""""which is 80MB" is a download size, and the reader decides on it.
Checked against what fixes it rather than against the file: fp32 weights
are four bytes a parameter, and the parameter count comes from
`config.json`. That predicts the real file to within 0.1% (the remainder
is zip and pickle overhead), far inside the rounding the prose uses.
Stating it from the config rather than from `os.stat` also keeps the claim
checkable on a clone made without LFS, where the file on disk is a 130-byte
pointer and stat-ing it would make the README's number look wrong.
It was wrong when written -- 77MB, which is the size in MiB, while the
`.gitignore` next to it said 237MB for three files that come to 241MB and
the runtime error said 240MB. Four hand-typed spellings of one number, none
of them agreeing, all of them describing a file the reader cannot stat.
"""
cfg_path = ROOT / "runs" / "jevon-final" / "config.json"
if not cfg_path.exists():
pytest.skip("no shipped run to check against")
sys.path.insert(0, str(ROOT))
from jevon.modeling import JevonConfig, JevonModel
total = sum(p.numel() for p in
JevonModel(JevonConfig(**json.loads(cfg_path.read_text()))).parameters())
megabytes = round(total * 4 / 1e6)
text = (ROOT / "README.md").read_text()
stated = re.search(r"(\d+)MB of fp32 weights", text)
assert stated, "README no longer states the checkpoint size"
assert int(stated.group(1)) == megabytes, (
f"README says {stated.group(1)}MB; {total:,} fp32 parameters are "
f"{megabytes}MB")
def test_the_shell_fallback_prefers_the_same_checkpoint_python_does():
"""Which weights load by default was written down in three places.
`benchmark.sh` and `record.sh` each pick `balanced.pt` and fall back to
`best.pt`; the Python scripts defaulted `--weights` to `best.pt` flatly,
so the drivers ran the checkpoint this repo documents and a reader running
the same script by hand ran the one the README argues against -- and got
numbers that disagree with RESULTS.md for no visible reason. The Python
side now resolves it, and this pins the shell copy to that order.
Parsed from the script rather than restated, so it cannot become a fourth
copy. `record.sh` is chained to `benchmark.sh` by the arcade repo's own
suite, which already compares the two file to file.
"""
sys.path.insert(0, str(ROOT))
from jevon.inference import PREFERRED_WEIGHTS
text = (ROOT / "scripts" / "benchmark.sh").read_text()
line = next((l for l in text.splitlines()
if "balanced.pt" in l and "WEIGHTS=" in l), None)
assert line, "benchmark.sh no longer chooses between checkpoints"
order = re.findall(r"(\w+\.pt)", line)
# `[ -f "$RUN/balanced.pt" ]` names the probe and the assignment, so the
# preferred one appears twice before the fallback appears at all.
seen = list(dict.fromkeys(order))
assert tuple(seen) == tuple(PREFERRED_WEIGHTS), (
f"benchmark.sh prefers {seen}, jevon.inference prefers "
f"{list(PREFERRED_WEIGHTS)}")
def test_the_readme_names_the_entry_points_the_arcade_plays_through():
"""The generator split is a promise to another repo.
`play_maze` used to be the only way in, and the arcade's live viewer would
have needed its own copy of the decision logic to take one step at a time
-- a second implementation of the loop every number in RESULTS.md is
measured from. `maze_steps`/`snake_steps` exist so that copy does not, and
the README says so, which makes them public surface rather than an
internal detail someone may fold back into the wrapper.
The arcade's suite checks the other direction: that its live server calls
these and defines no loop of its own.
"""
text = (ROOT / "README.md").read_text()
source = (ROOT / "scripts" / "play.py").read_text()
promised = [n for n in ("maze_steps", "snake_steps", "play_maze",
"play_snake", "default_max_steps")
if f"`{n}`" in text]
assert len(promised) == 5, f"README stopped naming some of them: {promised}"
for name in promised:
assert f"def {name}(" in source, f"README names {name}, play.py has no such function"
# --------------------------------------------------------------- model card
#
# The front matter and the three lines under "Using it" are what a reader of
# the Hub page actually runs. Everything else on that page is prose about a
# number; these are a promise about an API, and they are the only part of the
# README that can be wrong in a way that greets someone on their first minute.
def _frontmatter() -> dict:
"""The YAML block at the top of the model card, parsed.
Parsed rather than pattern-matched because the failure this guards is a
malformed block, and a regex that tolerates one would pass on exactly the
file that renders on huggingface.co as raw text with `---` at the top.
"""
yaml = pytest.importorskip("yaml")
text = (ROOT / "README.md").read_text()
assert text.startswith("---\n"), (
"the model card has no YAML front matter; the Hub needs it for the "
"licence, the pipeline tag and every tag the model is found by")
closing = text.index("\n---\n", 3)
return yaml.safe_load(text[4:closing])
def test_the_model_card_frontmatter_is_valid_and_agrees_with_the_licence():
meta = _frontmatter()
for key in ("license", "pipeline_tag", "tags"):
assert key in meta, f"model card front matter has no {key!r}: {meta}"
assert isinstance(meta["tags"], list) and meta["tags"], "tags is not a list"
# Three spellings of one licence -- front matter, LICENSE, pyproject -- in
# three vocabularies that do not share a string: the Hub's controlled list
# says `agpl-3.0`, SPDX says `AGPL-3.0-or-later`, and the file itself says
# neither. The Hub shows the first, and the repo is bound by the second and
# third, so a relicence applied to the file everyone edits and not to the
# two nobody does is a page that advertises terms the code does not carry.
licence = (ROOT / "LICENSE").read_text()
assert licence.splitlines()[0].strip() == "GNU AFFERO GENERAL PUBLIC LICENSE"
assert meta["license"] == "agpl-3.0", (
f"front matter says {meta['license']!r}, LICENSE is the AGPL")
assert ('license = { text = "AGPL-3.0-or-later" }'
in (ROOT / "pyproject.toml").read_text())
# Section 13 is the entire reason this is the AGPL and not the GPL, and the
# `## Licence` section says so in prose. The two licences are otherwise
# near-identical documents with near-identical first lines, so dropping the
# wrong one in here would leave every check above green, the page still
# claiming a network clause, and the clause gone.
assert "13. Remote Network Interaction" in licence, (
"LICENSE has no section 13 -- this looks like the GPL, not the AGPL, "
"and the model card's licence section claims the network clause")
def test_the_model_card_names_the_repo_the_code_downloads_from():
"""`from_pretrained("...")` in the prose has to be the id the code defaults to.
Two spellings of one identifier, in two files, and the one on the page is
the one people copy. If they drift, the documented call downloads someone
else's repository or nothing at all -- and nothing else in the suite would
notice, because no test has a network.
"""
sys.path.insert(0, str(ROOT))
from jevon import hub
text = (ROOT / "README.md").read_text()
quoted = set(re.findall(r'from_pretrained\("([^"]+)"\)', text))
assert quoted, "the model card no longer shows a from_pretrained call"
assert hub.REPO_ID in quoted, (
f"the card calls from_pretrained{sorted(quoted)}; "
f"hub.REPO_ID is {hub.REPO_ID!r}")
# The rest of `quoted` should be local run directories, which must ship.
for target in quoted - {hub.REPO_ID}:
assert target in ("runs/jevon-final",), (
f"the card shows from_pretrained({target!r}), which is neither the "
f"published repo nor the shipped run")
assert hub.CHECKPOINT == "runs/jevon-final"
assert f"https://huggingface.co/{hub.REPO_ID}" in text, (
"the card never links to its own Hub page, so a reader who arrived "
"via a mirror has no way back")
def test_the_model_card_parameter_count_is_exact():
""""20M" is checked elsewhere as a range; the header table states the number.
A precise figure is more useful and more fragile: it is the one claim on
the page that a single config change silently falsifies.
"""
cfg_path = ROOT / "runs" / "jevon-final" / "config.json"
if not cfg_path.exists():
pytest.skip("no shipped run to check against")
sys.path.insert(0, str(ROOT))
from jevon.modeling import JevonConfig, JevonModel
total = sum(p.numel() for p in
JevonModel(JevonConfig(**json.loads(cfg_path.read_text()))).parameters())
text = (ROOT / "README.md").read_text()
assert f"| Parameters | {total:,} |" in text, (
f"the header table does not state {total:,} parameters")
def test_the_usage_example_imports_names_that_exist():
"""The first code block on the page, executed as far as import resolution.
Not run -- it loads 80MB and generates a maze -- but every name it imports
is resolved against the real modules. A published `from jevon import
from_pretrained` that raises ImportError is the worst line in the repo to
get wrong, and it is one `__init__.py` edit away at all times.
"""
sys.path.insert(0, str(ROOT))
import importlib
text = (ROOT / "README.md").read_text()
block = re.search(r"## Using it(.*?)\n## ", text, re.S)
assert block, "the model card no longer has a 'Using it' section"
code = re.findall(r"```python\n(.*?)```", block.group(1), re.S)
assert code, "the 'Using it' section shows no Python"
imports = re.findall(r"^from ([\w.]+) import (.+)$", "\n".join(code), re.M)
assert imports, "the example imports nothing"
for module_name, names in imports:
module = importlib.import_module(module_name)
for clause in names.split(","):
name = clause.strip().split(" as ")[0].strip()
if not hasattr(module, name):
# `from envs import maze` imports a submodule, and importing
# the package alone does not bind it -- so `hasattr` answered
# "no" for a line that runs perfectly well, and answered "yes"
# only when some earlier test file happened to have imported
# `envs.maze` first. This whole file passed or failed on the
# order pytest collected it in, which is the one way a guard
# can be both green and meaningless.
try:
importlib.import_module(f"{module_name}.{name}")
continue
except ImportError:
pass
assert hasattr(module, name), (
f"the model card's example does `from {module_name} import "
f"{name}`, which does not resolve")
def test_the_usage_example_asks_for_questions_the_state_offers():
"""The same block, run as far as the board -- everything but the weights.
The example ends in three `answers[...]` lookups, and `question_profile`
omits the action question whenever fewer than two moves are legal. That is
true of this maze's start cell, so the block as first published raised
`KeyError: 'action'` on its own first lookup: the probabilities beside it
were real, measured one step further in, and nothing tied the two
together. Checking the keys costs a maze generation and no checkpoint,
because the profile -- not the network -- decides which questions exist.
"""
sys.path.insert(0, str(ROOT))
from envs import maze as mz
text = (ROOT / "README.md").read_text()
block = re.search(r"## Using it(.*?)\n## ", text, re.S)
assert block, "the model card no longer has a 'Using it' section"
code = "\n".join(re.findall(r"```python\n(.*?)```", block.group(1), re.S))
board = re.search(r"mz\.generate_maze\((\d+), \"(\w+)\", seed=(\d+)\)", code)
assert board, "the example no longer generates a maze the usual way"
state = mz.generate_maze(int(board.group(1)), board.group(2),
int(board.group(3)))
for action in re.findall(r"state\.step\(\"(\w+)\"\)", code):
assert action in state.legal_actions(), (
f"the example steps {action!r} from {state.position}, where the "
f"legal moves are {state.legal_actions()}")
state.step(action)
offered = set(mz.question_profile(state)["questions"])
asked = set(re.findall(r"answers\[\"(\w+)\"\]", code))
assert asked, "the example reads no answers"
assert asked <= offered, (
f"the example reads {sorted(asked - offered)} from a state at "
f"{state.position} that is asked only {sorted(offered)}")
def test_the_commercial_half_of_the_dual_licence_can_be_taken_up():
"""An offer with no document and no address behind it is not an offer.
The AGPL only works as a business arrangement because of the second half:
a company that cannot carry copyleft has somewhere to go. That half lives
in prose, which is where it rots -- an address that drifted between the
model card and COMMERCIAL-LICENSE.md sends the one reader who intends to
pay to a mailbox nobody reads, and nothing about the repository looks
wrong. So the offer is checked the way the rest of this page is: the
document exists, the page links to it, and the address is spelled once.
"""
offer = ROOT / "COMMERCIAL-LICENSE.md"
assert offer.is_file(), (
"the model card offers a commercial licence and there is no "
"COMMERCIAL-LICENSE.md to take it up in")
readme = (ROOT / "README.md").read_text()
assert "COMMERCIAL-LICENSE.md" in readme, (
"COMMERCIAL-LICENSE.md exists but the model card never links to it, "
"so the Hub page shows the copyleft terms and not the way out of them")
# Both files, one address. The model card's copy is the one a reader meets
# first and the offer document's is the one they act on.
# The trailing `\w` matters: `[\w.]+` at the end swallows the full stop of
# any sentence the address ends, and then one address is two.
address = r"[\w.+-]+@[\w-]+(?:\.[\w-]+)+"
emails = set(re.findall(address, offer.read_text()))
emails |= set(re.findall(address, readme))
assert len(emails) == 1, (
f"the commercial licence is offered at more than one address: "
f"{sorted(emails)}")