Download tests/test_results.py from lewislululu/jevon: direct link, hf CLI and curl.
- Browser
- Download file 43.2 kB
-
https://huggingface.co/lewislululu/jevon/resolve/main/tests/test_results.py
- Command line
-
hf download hf://lewislululu/jevon/tests/test_results.py
-
curl -L -o test_results.py https://huggingface.co/lewislululu/jevon/resolve/main/tests/test_results.py
43.2 kB
| """The report generator must survive the runs you most want reported. | |
| ``make_results.py`` is the last step of ``benchmark.sh``, so a crash there | |
| destroys a whole evaluation sweep after it has already been paid for. The two | |
| cases below are the ones that actually bit: a controller that solved nothing | |
| (so ``mean_steps_when_solved`` is ``None``), and a probe file written before | |
| the ``reach`` column existed. | |
| """ | |
| from __future__ import annotations | |
| import importlib.util | |
| import json | |
| import re | |
| import subprocess | |
| import sys | |
| from collections import Counter | |
| from pathlib import Path | |
| import pytest | |
| ROOT = Path(__file__).resolve().parents[1] | |
| _spec = importlib.util.spec_from_file_location( | |
| "make_results", ROOT / "scripts" / "make_results.py") | |
| make_results = importlib.util.module_from_spec(_spec) | |
| sys.modules["make_results"] = make_results | |
| _spec.loader.exec_module(make_results) | |
| def _maze(solve_rate, steps): | |
| return {"summary": {"maze": { | |
| "solve_rate": solve_rate, "mean_steps_when_solved": steps, | |
| "mean_efficiency": 0.0 if steps is None else 1.0, | |
| "mean_collisions": 4.2}}} | |
| def test_unsolved_controller_renders_instead_of_crashing(): | |
| # A model that solves nothing is a real result, not a broken input. | |
| text = "\n".join(make_results.play_section({"maze_model": _maze(0.0, None)})) | |
| assert "| 0.00 |" in text | |
| assert "--" in text | |
| def test_solved_controller_still_prints_its_step_count(): | |
| text = "\n".join(make_results.play_section({"maze_field": _maze(1.0, 82.0)})) | |
| assert "82.0" in text | |
| def test_probe_row_reports_reach(): | |
| row = {"cells": 60, "probe_r2": 0.9, "probe_mae": 1.2, "head_r2": 0.5, | |
| "descent_acc": 0.71, "reach_rate": 1.0, "diameter": 33, | |
| "iterations": 82} | |
| text = "\n".join(make_results.probe_section({"11": row})) | |
| assert "reach" in text | |
| # Bolded, because it is the column that predicts solve rate. | |
| assert "**1.0000**" in text | |
| def test_probe_row_tolerates_a_pre_reach_probe_file(): | |
| row = {"cells": 60, "probe_r2": 0.9, "probe_mae": 1.2, "head_r2": 0.5, | |
| "descent_acc": 0.71, "diameter": 33, "iterations": 82} | |
| text = "\n".join(make_results.probe_section({"11": row})) | |
| assert "nan" in text | |
| def test_missing_inputs_are_skipped_not_guessed(): | |
| assert make_results.probe_section(None) == [] | |
| assert make_results.play_section({}) == [] | |
| def test_deadlock_columns_render(): | |
| block = _maze(0.0, None) | |
| block["summary"]["maze"].update(deadlock_rate=1.0, mean_distinct_cells=1.0) | |
| text = "\n".join(make_results.play_section({"maze_model": block})) | |
| assert "deadlock" in text | |
| assert "| 1.00 | 1.0 |" in text | |
| def test_summary_without_deadlock_fields_still_renders(): | |
| # Summaries recorded before the diagnostic existed must not crash the table. | |
| text = "\n".join(make_results.play_section({"maze_model": _maze(1.0, 82.0)})) | |
| assert "82.0" in text | |
| def test_unlisted_play_configs_still_render_at_the_end(tmp_path): | |
| """Adding a play config must never silently drop it from the table.""" | |
| play = tmp_path / "play" | |
| play.mkdir() | |
| for name in ("maze_model", "maze_brand_new_idea"): | |
| (play / f"{name}_summary.json").write_text(json.dumps({"summary": {"maze": { | |
| "episodes": 1, "solve_rate": 1.0, "mean_steps_when_solved": 10.0, | |
| "mean_efficiency": 1.0, "mean_collisions": 0.0}}})) | |
| summaries = {p.stem.replace("_summary", ""): make_results.read(p) | |
| for p in sorted(play.glob("*_summary.json"))} | |
| rows = [line for line in make_results.play_section(summaries) if line.startswith("| `")] | |
| assert any("maze_brand_new_idea" in r for r in rows) | |
| assert rows[0].startswith("| `maze_model`"), "known rows keep reading order" | |
| def test_the_control_is_printed_directly_below_the_row_it_controls(tmp_path): | |
| play = tmp_path / "play" | |
| play.mkdir() | |
| # Written in an order that sorting would get wrong. | |
| for name in ("maze_random_memory", "maze_model_memory", "maze_model"): | |
| (play / f"{name}_summary.json").write_text(json.dumps({"summary": {"maze": { | |
| "episodes": 1, "solve_rate": 0.5, "mean_steps_when_solved": 10.0, | |
| "mean_efficiency": 0.5, "mean_collisions": 0.0}}})) | |
| summaries = {p.stem.replace("_summary", ""): make_results.read(p) | |
| for p in sorted(play.glob("*_summary.json"))} | |
| rows = [line for line in make_results.play_section(summaries) if line.startswith("| `")] | |
| names = [r.split("`")[1] for r in rows] | |
| assert names.index("maze_random_memory") == names.index("maze_model_memory") + 1 | |
| # ------------------------------------------------- data independence | |
| def test_frozen_splits_draw_a_fresh_board_per_state(split): | |
| """Every accuracy row's `n` claims that many independent tests. Check it. | |
| Questions that share a board share walls, goal and distance field, so they | |
| succeed or fail together and n overstates the evidence. The generator is | |
| supposed to draw a fresh board per state, which makes n honest -- but that | |
| is a property of the generator, and generators get edited. | |
| This was not hypothetical. An earlier reading of these splits reported the | |
| OOD maze questions as coming from 8 boards, and shipped that claim into | |
| two generated tables. It came from parsing the uid with the wrong field | |
| index, not from the data. So this test reads the board off the tensor the | |
| model is actually fed -- channel 0 and the goal cell -- rather than off a | |
| uid convention, because the convention is what was wrong last time. | |
| """ | |
| import hashlib | |
| path = ROOT / "data" / "frozen" / f"{split}.jsonl" | |
| if not path.is_file(): | |
| pytest.skip(f"{path} not generated") | |
| rows = [json.loads(line) for line in path.read_text().splitlines() if line] | |
| for game in ("maze", "snake"): | |
| states = [r for r in rows if r["game"] == game] | |
| if not states: | |
| continue | |
| seen = set() | |
| for r in states: | |
| _, h, w = r["shape"] | |
| seen.add(hashlib.sha1(json.dumps( | |
| [h, w, r["board"][:h * w], r.get("target")]).encode()).digest()) | |
| # Not equality: two small mazes can coincide by chance, and three do. | |
| # The guard is against a generator that reuses one board across many | |
| # states, which is a different order of magnitude. | |
| assert len(seen) >= 0.95 * len(states), ( | |
| f"{split}/{game}: {len(states)} states share only {len(seen)} " | |
| "distinct boards; `n` no longer measures independent tests and " | |
| "the `boards` column in RESULTS.md is now the denominator to read") | |
| def test_checkpoint_selection_table_matches_the_run_history(): | |
| """The README's `best.pt` vs `balanced.pt` table, checked against history. | |
| That table is the whole argument for selecting on min(maze, snake) rather | |
| than on loss, and it quotes one run at two specific steps. The checkpoints | |
| it describes are long overwritten -- training keeps only the current | |
| best.pt and balanced.pt -- but history.json keeps every eval row forever, | |
| so the claim stays checkable after the evidence for it is gone. | |
| Steps are parsed out of the column headers rather than hardcoded, so | |
| re-pointing the table at different steps is a README edit and this follows | |
| it. A retrained run that no longer shows the trade will fail here, which | |
| is correct: the argument would then need different evidence, not a | |
| quietly stale table. | |
| """ | |
| import re | |
| history = ROOT / "runs" / "jevon-final" / "history.json" | |
| if not history.is_file(): | |
| pytest.skip("no training history yet") | |
| rows = {r["step"]: r for r in json.loads(history.read_text())} | |
| text = (ROOT / "README.md").read_text() | |
| header = next((l for l in text.splitlines() | |
| if "`best.pt` (step" in l and "`balanced.pt` (step" in l), None) | |
| assert header, "the checkpoint-selection table header is gone from the README" | |
| steps = [int(s) for s in re.findall(r"step (\d+)", header)] | |
| assert len(steps) == 2, header | |
| for step in steps: | |
| assert step in rows, f"README quotes step {step}, absent from history.json" | |
| # [1:] drops the remainder of the header line itself, which is empty and | |
| # would end the loop on its first iteration. | |
| table = text.split(header, 1)[1].split("\n")[1:] | |
| fields = {"eval loss": "loss", "`acc_boolean`": "acc_boolean", | |
| "`acc_score`": "acc_score", "`acc_choice`": "acc_choice", | |
| "`snake_action`": "snake_action"} | |
| checked = 0 | |
| for line in table: | |
| if not line.startswith("|"): | |
| break | |
| cells = [c.strip() for c in line.strip("|").split("|")] | |
| key = fields.get(cells[0]) | |
| if key is None or len(cells) != 3: | |
| continue | |
| for step, cell in zip(steps, cells[1:]): | |
| shown = cell.strip("*") | |
| # Compared at the precision the README prints, which is the point: | |
| # acc_boolean is quoted to five decimals precisely to show the two | |
| # checkpoints are identical there. | |
| places = len(shown.split(".")[1]) if "." in shown else 0 | |
| actual = f"{rows[step][key]:.{places}f}" | |
| assert shown == actual, ( | |
| f"README says {key} = {shown} at step {step}; " | |
| f"history.json says {actual}") | |
| checked += 1 | |
| assert checked == 10, f"expected 10 cells checked, got {checked}" | |
| def test_readme_parameter_count_matches_the_shipped_config(): | |
| """The headline calls this a 20M-parameter model. Check it against the | |
| architecture the shipped run actually used. | |
| Built from `config.json` rather than loaded from `balanced.pt`, for two | |
| reasons: an 80MB state-dict load is slow enough to notice in a test suite, | |
| and the config is the thing a reader of the README can inspect. If the two | |
| ever disagreed the checkpoint would not load at all, which | |
| `inference.py` would report far more loudly than this test could. | |
| A bare "20M" tolerates a range, so the assertion is the range a reader | |
| would accept for that phrase, not the exact integer -- the point is to | |
| catch an architecture change that makes the front page wrong, not to | |
| re-pin the number every time a bias term moves. The exact figure in the | |
| header table is pinned separately, by | |
| `test_the_model_card_parameter_count_is_exact`. | |
| Found by content rather than at a fixed line offset. This read line 3 | |
| until the Hub front matter was added above it, at which point it was | |
| asserting about `pipeline_tag` -- a test that breaks when an unrelated | |
| paragraph is inserted is a test that will eventually be deleted rather | |
| than fixed. | |
| """ | |
| cfg_path = ROOT / "runs" / "jevon-final" / "config.json" | |
| if not cfg_path.exists(): | |
| pytest.skip("no shipped run to check against") | |
| sys.path.insert(0, str(ROOT)) | |
| from jevon.modeling import JevonConfig, JevonModel | |
| model = JevonModel(JevonConfig(**json.loads(cfg_path.read_text()))) | |
| total = sum(p.numel() for p in model.parameters()) | |
| claim = next((line for line in (ROOT / "README.md").read_text().splitlines() | |
| if "20M-parameter" in line), None) | |
| assert claim, "README headline no longer calls this a 20M-parameter model" | |
| assert 19.5e6 <= total <= 20.5e6, ( | |
| f"README says 20M, config.json builds {total:,} " | |
| f"({total / 1e6:.2f}M)") | |
| def test_readme_reproduction_command_is_the_one_that_trained_the_checkpoint(): | |
| """"Reproduce it with:" has to be true, or the repo is worse than silent. | |
| A reader who runs the quoted command and gets different numbers has no | |
| way to tell whether the model is flaky or the page is stale, and will | |
| reasonably conclude the former. train.py records `sys.argv` verbatim in | |
| `args.json` for exactly this comparison, so the check is against what the | |
| shipped run was actually launched with rather than against a second copy | |
| of the same prose. | |
| Compared as a flag set, not as a string: the README wraps the line with a | |
| backslash and runs under `python` where the run used the venv interpreter | |
| with `-u`, and neither difference changes what gets trained. | |
| """ | |
| args_path = ROOT / "runs" / "jevon-final" / "args.json" | |
| if not args_path.exists(): | |
| pytest.skip("no shipped run to check against") | |
| argv = json.loads(args_path.read_text())["argv"] | |
| # Drop the interpreter's own flags and the script path; what remains is | |
| # the training configuration the README is promising. | |
| actual = argv[argv.index("scripts/train.py") + 1:] | |
| text = (ROOT / "README.md").read_text() | |
| quoted = None | |
| for block in re.findall(r"```(?:[a-z]*)\n(.*?)```", text, re.S): | |
| joined = block.replace("\\\n", " ") | |
| for line in joined.splitlines(): | |
| if "scripts/train.py" in line and "--out runs/jevon-final" in line: | |
| quoted = line.split("scripts/train.py", 1)[1].split() | |
| assert quoted is not None, ( | |
| "README no longer quotes a train.py command for runs/jevon-final; " | |
| "the reproduction instructions have gone missing") | |
| assert quoted == actual, ( | |
| f"README says:\n {' '.join(quoted)}\nrun used:\n {' '.join(actual)}") | |
| def test_readme_val_split_composition_matches_the_frozen_data(): | |
| """The selection argument rests on these five numbers. | |
| "73.5% of eval loss is booleans" is why `best.pt` can be wrong, and it is | |
| a property of the frozen split, not of the model -- so it changes if | |
| anyone edits the question profile, and it changes silently, because the | |
| split regenerates without the README noticing. Read out of the data | |
| rather than asserted against constants typed here, which would just be a | |
| sixth copy of the same claim. | |
| """ | |
| path = ROOT / "data" / "frozen" / "val.jsonl" | |
| if not path.exists(): | |
| pytest.skip("no frozen val split; run scripts/build_dataset.py") | |
| states = [json.loads(line) for line in path.read_text().splitlines() if line] | |
| kinds = Counter(q["type"] for st in states for q in st["questions"]) | |
| total = sum(kinds.values()) | |
| text = (ROOT / "README.md").read_text() | |
| claim = next((p for p in text.split("\n\n") | |
| if "frozen validation split is" in p), None) | |
| assert claim, "README no longer describes the val split composition" | |
| for number in (str(total), str(len(states)), str(kinds["boolean"]), | |
| str(kinds["score"]), str(kinds["choice"])): | |
| assert number in claim, ( | |
| f"README's val-split paragraph does not mention {number}; " | |
| f"measured {total} questions over {len(states)} states, " | |
| f"{dict(kinds)}") | |
| # The two percentages the argument actually turns on. | |
| for share, label in ((kinds["boolean"] / total, "boolean"), | |
| (kinds["choice"] / total, "choice")): | |
| assert f"{share * 100:.1f}%" in claim, ( | |
| f"README's {label} share is stale; measured {share * 100:.1f}%") | |
| def test_readme_selection_example_matches_the_run_it_claims_to_quote(): | |
| """The table making the case for `balanced.pt` is hand-typed, from disk. | |
| It is a historical observation -- two checkpoints that existed at one | |
| instant and were overwritten minutes later -- so it cannot be generated | |
| the way the shipped-checkpoint table is. But the evals it quotes are | |
| still in `history.json`, so it can be checked, and it has to be: the | |
| whole selection argument rests on "boolean and score are identical to | |
| five decimal places, only choice moved". Retrain the run and every cell | |
| is about a different model while the prose reads the same. | |
| """ | |
| hist_path = ROOT / "runs" / "jevon-final" / "history.json" | |
| if not hist_path.exists(): | |
| pytest.skip("no shipped run to check against") | |
| history = {r["step"]: r for r in json.loads(hist_path.read_text())} | |
| text = (ROOT / "README.md").read_text() | |
| # Generated blocks contain a table with the same shape; this one is the | |
| # hand-typed one, so the generated ones are removed before searching. | |
| prose = re.sub(r"<!-- BEGIN:.*?<!-- END:\w+ -->", "", text, flags=re.S) | |
| header = re.search(r"\| \| `best\.pt` \(step (\d+)\) \| " | |
| r"`balanced\.pt` \(step (\d+)\) \|\n(.*?)\n\n", | |
| prose, re.S) | |
| assert header, "README no longer shows the mid-training selection table" | |
| best_step, balanced_step = int(header.group(1)), int(header.group(2)) | |
| for step in (best_step, balanced_step): | |
| assert step in history, ( | |
| f"README quotes step {step}, which this run never evaluated; " | |
| f"evals are at {sorted(history)}") | |
| fields = {"eval loss": "loss", "`acc_boolean`": "acc_boolean", | |
| "`acc_score`": "acc_score", "`acc_choice`": "acc_choice", | |
| "`snake_action`": "snake_action", "`maze_action`": "maze_action"} | |
| checked = 0 | |
| for row in header.group(3).splitlines(): | |
| cells = [c.strip() for c in row.strip().strip("|").split("|")] | |
| if len(cells) != 3 or cells[0] not in fields: | |
| continue | |
| key = fields[cells[0]] | |
| for cell, step in zip(cells[1:], (best_step, balanced_step)): | |
| quoted = cell.strip("*` ") | |
| actual = history[step][key] | |
| assert quoted == f"{actual:.{len(quoted.split('.')[1])}f}", ( | |
| f"README says {key} = {quoted} at step {step}; " | |
| f"history.json says {actual}") | |
| checked += 1 | |
| assert checked >= 8, f"only {checked} cells parsed out of the table" | |
| # The two claims the prose makes *about* the table, which are what the | |
| # reader takes away and which no cell states on its own. | |
| gap = history[balanced_step]["acc_choice"] - history[best_step]["acc_choice"] | |
| assert f"**{round(gap * 100)} points" in prose, ( | |
| f"README's choice-accuracy gap is stale; measured {gap * 100:.1f}") | |
| ce = [f"{history[s]['ce']:.4f}" for s in (balanced_step, best_step)] | |
| assert f"(`ce` {ce[0]} → {ce[1]})" in prose, ( | |
| f"README's cross-entropy pair is stale; measured {ce[0]} → {ce[1]}") | |
| def test_readme_uniform_controls_match_the_frozen_split(): | |
| """`0.4905` and `0.4955` are what a coin scores at this game. | |
| They are the reason a snake score of `0.5135` is described as "a hair | |
| above the control" rather than as half-right, so they carry the weight of | |
| that paragraph. Neither depends on the model -- both are the mean | |
| fraction of tied-optimal candidates per action question -- which is | |
| exactly why they rot unnoticed: the split can be rebuilt without any | |
| checkpoint being retrained. | |
| Computed the way `evaluate.py` computes `uniform_control`, from `probs` | |
| alone, so this does not need a checkpoint either. | |
| """ | |
| path = ROOT / "data" / "frozen" / "val.jsonl" | |
| if not path.exists(): | |
| pytest.skip("no frozen val split; run scripts/build_dataset.py") | |
| shares = {} | |
| for line in path.read_text().splitlines(): | |
| if not line: | |
| continue | |
| state = json.loads(line) | |
| for q in state["questions"]: | |
| if q["name"] == "action": | |
| top = max(q["probs"]) | |
| shares.setdefault(state["game"], []).append( | |
| sum(v >= top - 1e-6 for v in q["probs"]) / len(q["probs"])) | |
| assert set(shares) == {"maze", "snake"}, f"split games: {sorted(shares)}" | |
| text = (ROOT / "README.md").read_text() | |
| claim = next((p for p in text.split("\n\n") | |
| if "uniform controls on this split" in p), None) | |
| assert claim, "README no longer states the uniform controls" | |
| for game, values in sorted(shares.items()): | |
| control = sum(values) / len(values) | |
| assert f"`{control:.4f}` for {game}" in claim, ( | |
| f"README's {game} uniform control is stale; measured " | |
| f"{control:.4f} over {len(values)} action questions") | |
| # The snake control is quoted a second time, forty lines down, to call | |
| # `0.5135` "a hair above" it. A second copy of a number is a second thing | |
| # to go stale, and that sentence is the one a reader believes. | |
| snake = sum(shares["snake"]) / len(shares["snake"]) | |
| echo = re.search(r"above the `([\d.]+)` control", text) | |
| assert echo, "README no longer compares the failing checkpoint to control" | |
| assert echo.group(1) == f"{snake:.4f}", ( | |
| f"README's echoed snake control is stale; it says {echo.group(1)}, " | |
| f"the split says {snake:.4f}") | |
| SHIPPED_WEIGHTS = {"runs/jevon-final/balanced.pt", "runs/jevon-final/best.pt"} | |
| def test_the_shipped_run_metadata_is_committed(): | |
| """Everything the README is generated from has to survive a clone. | |
| `.gitignore` used to read `runs/`, which excluded 72K of JSON along with | |
| 241MB of weights. The suite stayed green because every test that reads | |
| those files skips when they are missing -- so on a fresh clone the whole | |
| documentation-drift apparatus passed by not running, which is the exact | |
| failure it was built to prevent. | |
| Checked against `git ls-files` rather than the filesystem: the files are | |
| on disk here either way, and being on disk is not the property that | |
| matters. | |
| """ | |
| run = ROOT / "runs" / "jevon-final" | |
| if not run.is_dir() or not (ROOT / ".git").exists(): | |
| pytest.skip("no shipped run, or not a git work tree") | |
| tracked = set(subprocess.run( | |
| ["git", "ls-files", "runs/jevon-final"], cwd=ROOT, | |
| capture_output=True, text=True, check=True).stdout.split()) | |
| on_disk = {str(p.relative_to(ROOT)) for p in run.rglob("*.json")} | |
| missing = sorted(on_disk - tracked) | |
| assert not missing, ( | |
| "the README is generated from files git does not track:\n " | |
| + "\n ".join(missing) | |
| + "\nevery test reading them skips, so the suite would be green " | |
| "on a clone that has none of them") | |
| def test_the_shipped_checkpoints_are_committed_and_no_others_are(): | |
| """This repo is a model release, so the weights are the point of it. | |
| The inverse of what this file used to assert. While the weights lived | |
| outside git the rule was "no `.pt` is tracked"; publishing to the Hub | |
| makes that rule the bug -- `from_pretrained` would download a recipe. | |
| Both halves still matter. `balanced.pt` and `best.pt` are the two | |
| checkpoints the model card compares, so both ship; `last.pt` is wherever | |
| training happened to stop, which is not a claim this repo makes, and the | |
| ablation runs under `runs/arm-*` are 80MB each of the same. | |
| """ | |
| if not (ROOT / ".git").exists(): | |
| pytest.skip("not a git work tree") | |
| tracked = set(subprocess.run( | |
| ["git", "ls-files"], cwd=ROOT, | |
| capture_output=True, text=True, check=True).stdout.split()) | |
| shipped = {f for f in tracked if f.endswith(".pt")} | |
| assert shipped == SHIPPED_WEIGHTS, ( | |
| f"tracked checkpoints are {sorted(shipped)}; " | |
| f"expected exactly {sorted(SHIPPED_WEIGHTS)}") | |
| def test_the_checkpoints_are_stored_as_lfs_pointers(): | |
| """80MB of raw blob in a git object store is the failure this guards. | |
| The Hub rejects non-LFS files over 10MB, so getting this wrong means the | |
| push fails -- but it fails after the repository has been rewritten to | |
| contain 160MB that `git gc` will not reclaim, and the fix at that point is | |
| to rebuild history. Cheaper to assert before pushing. | |
| Two separate claims, because `.gitattributes` saying `filter=lfs` and the | |
| committed blob *being* a pointer are not the same thing: a file staged | |
| before the attribute existed keeps its raw blob, and check-attr will | |
| cheerfully report `lfs` for it ever after. | |
| """ | |
| if not (ROOT / ".git").exists(): | |
| pytest.skip("not a git work tree") | |
| for name in sorted(SHIPPED_WEIGHTS): | |
| attr = subprocess.run( | |
| ["git", "check-attr", "filter", "--", name], cwd=ROOT, | |
| capture_output=True, text=True, check=True).stdout.strip() | |
| assert attr.endswith(": lfs"), ( | |
| f"{name} is not routed through LFS by .gitattributes ({attr})") | |
| blob = subprocess.run( | |
| ["git", "cat-file", "-s", f"HEAD:{name}"], cwd=ROOT, | |
| capture_output=True, text=True) | |
| if blob.returncode != 0: | |
| continue # not committed yet | |
| size = int(blob.stdout) | |
| assert size < 1024, ( | |
| f"HEAD:{name} is a {size:,}-byte blob, so it was committed as raw " | |
| f"bytes rather than as an LFS pointer; the Hub will reject the push") | |
| def test_no_script_defaults_to_a_run_that_does_not_ship(): | |
| """`--checkpoint` defaults have to name a directory a clone will have. | |
| Four scripts defaulted to `runs/jevon-s` or `runs/jevon-v3` long after the | |
| shipped run had been renamed `runs/jevon-final`; `runs/jevon-v3` had not | |
| existed on this machine for some time either. Nothing noticed, because | |
| every invocation in the README and in `benchmark.sh` passes the path | |
| explicitly -- the defaults were only ever reached by someone reading | |
| `--help` and trusting it, which is exactly the reader this repo is being | |
| opened for. | |
| Tracked-ness is the test, not existence: a default naming a directory that | |
| happens to sit in this working copy but is gitignored is the same broken | |
| promise to a clone. `train.py --out` is excluded because it names a run to | |
| *create*; defaulting it to the shipped one would overwrite it. | |
| """ | |
| if not (ROOT / ".git").exists(): | |
| pytest.skip("not a git work tree") | |
| tracked = subprocess.run( | |
| ["git", "ls-files", "runs"], cwd=ROOT, | |
| capture_output=True, text=True, check=True).stdout.split() | |
| ships = {f.split("/")[1] for f in tracked if "/" in f} | |
| cited = [] | |
| for path in sorted((ROOT / "scripts").iterdir()): | |
| if path.suffix not in (".py", ".sh"): | |
| continue | |
| for lineno, line in enumerate(path.read_text().splitlines(), 1): | |
| if "--out" in line or line.lstrip().startswith("#"): | |
| continue | |
| # `default="runs/x"` in argparse, `${1:-runs/x}` in shell. | |
| for m in re.finditer(r"(?:default=\"|:-)(runs/([\w.-]+))", line): | |
| cited.append((f"{path.name}:{lineno}", m.group(1), m.group(2))) | |
| assert cited, "no script names a default run directory any more" | |
| stale = [(where, ref) for where, ref, run in cited if run not in ships] | |
| assert not stale, ( | |
| "script defaults name run directories that are not in the repo:\n " | |
| + "\n ".join(f"{where} {ref}" for where, ref in stale) | |
| + f"\nshipped: {sorted(ships)}") | |
| def test_every_file_the_readme_links_to_is_in_the_repo(): | |
| """A relative link is a promise that the repo contains the thing. | |
| `RESULTS.md` is generated by the last stage of `benchmark.sh` and was | |
| linked from the front page three times while never once being committed -- | |
| it is not even gitignored, it was simply never staged, and `git status` | |
| listing it among the run outputs made it easy to skip. So every reader of | |
| a fresh clone got three dead links to the document holding all the detailed | |
| numbers, and the page most likely to be read first was the page most | |
| broken. | |
| Tracked-ness again, not existence: the generated file is always sitting | |
| right here after a benchmark, which is precisely why this went unnoticed | |
| for as long as it did. | |
| """ | |
| if not (ROOT / ".git").exists(): | |
| pytest.skip("not a git work tree") | |
| text = (ROOT / "README.md").read_text() | |
| targets = {m.group(1).split("#")[0] | |
| for m in re.finditer(r"\]\(([^)]+)\)", text)} | |
| local = sorted(t for t in targets | |
| if t and not re.match(r"^(https?:|mailto:|#)", t)) | |
| assert local, "the README links to nothing in its own repo any more" | |
| tracked = set(subprocess.run( | |
| ["git", "ls-files"], cwd=ROOT, | |
| capture_output=True, text=True, check=True).stdout.split()) | |
| dead = [t for t in local if t not in tracked | |
| and not (ROOT / t).is_dir()] | |
| assert not dead, ( | |
| f"README links to files the repo does not ship: {dead}") | |
| def test_readme_checkpoint_size_follows_from_the_parameter_count(): | |
| """"which is 80MB" is a download size, and the reader decides on it. | |
| Checked against what fixes it rather than against the file: fp32 weights | |
| are four bytes a parameter, and the parameter count comes from | |
| `config.json`. That predicts the real file to within 0.1% (the remainder | |
| is zip and pickle overhead), far inside the rounding the prose uses. | |
| Stating it from the config rather than from `os.stat` also keeps the claim | |
| checkable on a clone made without LFS, where the file on disk is a 130-byte | |
| pointer and stat-ing it would make the README's number look wrong. | |
| It was wrong when written -- 77MB, which is the size in MiB, while the | |
| `.gitignore` next to it said 237MB for three files that come to 241MB and | |
| the runtime error said 240MB. Four hand-typed spellings of one number, none | |
| of them agreeing, all of them describing a file the reader cannot stat. | |
| """ | |
| cfg_path = ROOT / "runs" / "jevon-final" / "config.json" | |
| if not cfg_path.exists(): | |
| pytest.skip("no shipped run to check against") | |
| sys.path.insert(0, str(ROOT)) | |
| from jevon.modeling import JevonConfig, JevonModel | |
| total = sum(p.numel() for p in | |
| JevonModel(JevonConfig(**json.loads(cfg_path.read_text()))).parameters()) | |
| megabytes = round(total * 4 / 1e6) | |
| text = (ROOT / "README.md").read_text() | |
| stated = re.search(r"(\d+)MB of fp32 weights", text) | |
| assert stated, "README no longer states the checkpoint size" | |
| assert int(stated.group(1)) == megabytes, ( | |
| f"README says {stated.group(1)}MB; {total:,} fp32 parameters are " | |
| f"{megabytes}MB") | |
| def test_the_shell_fallback_prefers_the_same_checkpoint_python_does(): | |
| """Which weights load by default was written down in three places. | |
| `benchmark.sh` and `record.sh` each pick `balanced.pt` and fall back to | |
| `best.pt`; the Python scripts defaulted `--weights` to `best.pt` flatly, | |
| so the drivers ran the checkpoint this repo documents and a reader running | |
| the same script by hand ran the one the README argues against -- and got | |
| numbers that disagree with RESULTS.md for no visible reason. The Python | |
| side now resolves it, and this pins the shell copy to that order. | |
| Parsed from the script rather than restated, so it cannot become a fourth | |
| copy. `record.sh` is chained to `benchmark.sh` by the arcade repo's own | |
| suite, which already compares the two file to file. | |
| """ | |
| sys.path.insert(0, str(ROOT)) | |
| from jevon.inference import PREFERRED_WEIGHTS | |
| text = (ROOT / "scripts" / "benchmark.sh").read_text() | |
| line = next((l for l in text.splitlines() | |
| if "balanced.pt" in l and "WEIGHTS=" in l), None) | |
| assert line, "benchmark.sh no longer chooses between checkpoints" | |
| order = re.findall(r"(\w+\.pt)", line) | |
| # `[ -f "$RUN/balanced.pt" ]` names the probe and the assignment, so the | |
| # preferred one appears twice before the fallback appears at all. | |
| seen = list(dict.fromkeys(order)) | |
| assert tuple(seen) == tuple(PREFERRED_WEIGHTS), ( | |
| f"benchmark.sh prefers {seen}, jevon.inference prefers " | |
| f"{list(PREFERRED_WEIGHTS)}") | |
| def test_the_readme_names_the_entry_points_the_arcade_plays_through(): | |
| """The generator split is a promise to another repo. | |
| `play_maze` used to be the only way in, and the arcade's live viewer would | |
| have needed its own copy of the decision logic to take one step at a time | |
| -- a second implementation of the loop every number in RESULTS.md is | |
| measured from. `maze_steps`/`snake_steps` exist so that copy does not, and | |
| the README says so, which makes them public surface rather than an | |
| internal detail someone may fold back into the wrapper. | |
| The arcade's suite checks the other direction: that its live server calls | |
| these and defines no loop of its own. | |
| """ | |
| text = (ROOT / "README.md").read_text() | |
| source = (ROOT / "scripts" / "play.py").read_text() | |
| promised = [n for n in ("maze_steps", "snake_steps", "play_maze", | |
| "play_snake", "default_max_steps") | |
| if f"`{n}`" in text] | |
| assert len(promised) == 5, f"README stopped naming some of them: {promised}" | |
| for name in promised: | |
| assert f"def {name}(" in source, f"README names {name}, play.py has no such function" | |
| # --------------------------------------------------------------- model card | |
| # | |
| # The front matter and the three lines under "Using it" are what a reader of | |
| # the Hub page actually runs. Everything else on that page is prose about a | |
| # number; these are a promise about an API, and they are the only part of the | |
| # README that can be wrong in a way that greets someone on their first minute. | |
| def _frontmatter() -> dict: | |
| """The YAML block at the top of the model card, parsed. | |
| Parsed rather than pattern-matched because the failure this guards is a | |
| malformed block, and a regex that tolerates one would pass on exactly the | |
| file that renders on huggingface.co as raw text with `---` at the top. | |
| """ | |
| yaml = pytest.importorskip("yaml") | |
| text = (ROOT / "README.md").read_text() | |
| assert text.startswith("---\n"), ( | |
| "the model card has no YAML front matter; the Hub needs it for the " | |
| "licence, the pipeline tag and every tag the model is found by") | |
| closing = text.index("\n---\n", 3) | |
| return yaml.safe_load(text[4:closing]) | |
| def test_the_model_card_frontmatter_is_valid_and_agrees_with_the_licence(): | |
| meta = _frontmatter() | |
| for key in ("license", "pipeline_tag", "tags"): | |
| assert key in meta, f"model card front matter has no {key!r}: {meta}" | |
| assert isinstance(meta["tags"], list) and meta["tags"], "tags is not a list" | |
| # Three spellings of one licence -- front matter, LICENSE, pyproject -- in | |
| # three vocabularies that do not share a string: the Hub's controlled list | |
| # says `agpl-3.0`, SPDX says `AGPL-3.0-or-later`, and the file itself says | |
| # neither. The Hub shows the first, and the repo is bound by the second and | |
| # third, so a relicence applied to the file everyone edits and not to the | |
| # two nobody does is a page that advertises terms the code does not carry. | |
| licence = (ROOT / "LICENSE").read_text() | |
| assert licence.splitlines()[0].strip() == "GNU AFFERO GENERAL PUBLIC LICENSE" | |
| assert meta["license"] == "agpl-3.0", ( | |
| f"front matter says {meta['license']!r}, LICENSE is the AGPL") | |
| assert ('license = { text = "AGPL-3.0-or-later" }' | |
| in (ROOT / "pyproject.toml").read_text()) | |
| # Section 13 is the entire reason this is the AGPL and not the GPL, and the | |
| # `## Licence` section says so in prose. The two licences are otherwise | |
| # near-identical documents with near-identical first lines, so dropping the | |
| # wrong one in here would leave every check above green, the page still | |
| # claiming a network clause, and the clause gone. | |
| assert "13. Remote Network Interaction" in licence, ( | |
| "LICENSE has no section 13 -- this looks like the GPL, not the AGPL, " | |
| "and the model card's licence section claims the network clause") | |
| def test_the_model_card_names_the_repo_the_code_downloads_from(): | |
| """`from_pretrained("...")` in the prose has to be the id the code defaults to. | |
| Two spellings of one identifier, in two files, and the one on the page is | |
| the one people copy. If they drift, the documented call downloads someone | |
| else's repository or nothing at all -- and nothing else in the suite would | |
| notice, because no test has a network. | |
| """ | |
| sys.path.insert(0, str(ROOT)) | |
| from jevon import hub | |
| text = (ROOT / "README.md").read_text() | |
| quoted = set(re.findall(r'from_pretrained\("([^"]+)"\)', text)) | |
| assert quoted, "the model card no longer shows a from_pretrained call" | |
| assert hub.REPO_ID in quoted, ( | |
| f"the card calls from_pretrained{sorted(quoted)}; " | |
| f"hub.REPO_ID is {hub.REPO_ID!r}") | |
| # The rest of `quoted` should be local run directories, which must ship. | |
| for target in quoted - {hub.REPO_ID}: | |
| assert target in ("runs/jevon-final",), ( | |
| f"the card shows from_pretrained({target!r}), which is neither the " | |
| f"published repo nor the shipped run") | |
| assert hub.CHECKPOINT == "runs/jevon-final" | |
| assert f"https://huggingface.co/{hub.REPO_ID}" in text, ( | |
| "the card never links to its own Hub page, so a reader who arrived " | |
| "via a mirror has no way back") | |
| def test_the_model_card_parameter_count_is_exact(): | |
| """"20M" is checked elsewhere as a range; the header table states the number. | |
| A precise figure is more useful and more fragile: it is the one claim on | |
| the page that a single config change silently falsifies. | |
| """ | |
| cfg_path = ROOT / "runs" / "jevon-final" / "config.json" | |
| if not cfg_path.exists(): | |
| pytest.skip("no shipped run to check against") | |
| sys.path.insert(0, str(ROOT)) | |
| from jevon.modeling import JevonConfig, JevonModel | |
| total = sum(p.numel() for p in | |
| JevonModel(JevonConfig(**json.loads(cfg_path.read_text()))).parameters()) | |
| text = (ROOT / "README.md").read_text() | |
| assert f"| Parameters | {total:,} |" in text, ( | |
| f"the header table does not state {total:,} parameters") | |
| def test_the_usage_example_imports_names_that_exist(): | |
| """The first code block on the page, executed as far as import resolution. | |
| Not run -- it loads 80MB and generates a maze -- but every name it imports | |
| is resolved against the real modules. A published `from jevon import | |
| from_pretrained` that raises ImportError is the worst line in the repo to | |
| get wrong, and it is one `__init__.py` edit away at all times. | |
| """ | |
| sys.path.insert(0, str(ROOT)) | |
| import importlib | |
| text = (ROOT / "README.md").read_text() | |
| block = re.search(r"## Using it(.*?)\n## ", text, re.S) | |
| assert block, "the model card no longer has a 'Using it' section" | |
| code = re.findall(r"```python\n(.*?)```", block.group(1), re.S) | |
| assert code, "the 'Using it' section shows no Python" | |
| imports = re.findall(r"^from ([\w.]+) import (.+)$", "\n".join(code), re.M) | |
| assert imports, "the example imports nothing" | |
| for module_name, names in imports: | |
| module = importlib.import_module(module_name) | |
| for clause in names.split(","): | |
| name = clause.strip().split(" as ")[0].strip() | |
| if not hasattr(module, name): | |
| # `from envs import maze` imports a submodule, and importing | |
| # the package alone does not bind it -- so `hasattr` answered | |
| # "no" for a line that runs perfectly well, and answered "yes" | |
| # only when some earlier test file happened to have imported | |
| # `envs.maze` first. This whole file passed or failed on the | |
| # order pytest collected it in, which is the one way a guard | |
| # can be both green and meaningless. | |
| try: | |
| importlib.import_module(f"{module_name}.{name}") | |
| continue | |
| except ImportError: | |
| pass | |
| assert hasattr(module, name), ( | |
| f"the model card's example does `from {module_name} import " | |
| f"{name}`, which does not resolve") | |
| def test_the_usage_example_asks_for_questions_the_state_offers(): | |
| """The same block, run as far as the board -- everything but the weights. | |
| The example ends in three `answers[...]` lookups, and `question_profile` | |
| omits the action question whenever fewer than two moves are legal. That is | |
| true of this maze's start cell, so the block as first published raised | |
| `KeyError: 'action'` on its own first lookup: the probabilities beside it | |
| were real, measured one step further in, and nothing tied the two | |
| together. Checking the keys costs a maze generation and no checkpoint, | |
| because the profile -- not the network -- decides which questions exist. | |
| """ | |
| sys.path.insert(0, str(ROOT)) | |
| from envs import maze as mz | |
| text = (ROOT / "README.md").read_text() | |
| block = re.search(r"## Using it(.*?)\n## ", text, re.S) | |
| assert block, "the model card no longer has a 'Using it' section" | |
| code = "\n".join(re.findall(r"```python\n(.*?)```", block.group(1), re.S)) | |
| board = re.search(r"mz\.generate_maze\((\d+), \"(\w+)\", seed=(\d+)\)", code) | |
| assert board, "the example no longer generates a maze the usual way" | |
| state = mz.generate_maze(int(board.group(1)), board.group(2), | |
| int(board.group(3))) | |
| for action in re.findall(r"state\.step\(\"(\w+)\"\)", code): | |
| assert action in state.legal_actions(), ( | |
| f"the example steps {action!r} from {state.position}, where the " | |
| f"legal moves are {state.legal_actions()}") | |
| state.step(action) | |
| offered = set(mz.question_profile(state)["questions"]) | |
| asked = set(re.findall(r"answers\[\"(\w+)\"\]", code)) | |
| assert asked, "the example reads no answers" | |
| assert asked <= offered, ( | |
| f"the example reads {sorted(asked - offered)} from a state at " | |
| f"{state.position} that is asked only {sorted(offered)}") | |
| def test_the_commercial_half_of_the_dual_licence_can_be_taken_up(): | |
| """An offer with no document and no address behind it is not an offer. | |
| The AGPL only works as a business arrangement because of the second half: | |
| a company that cannot carry copyleft has somewhere to go. That half lives | |
| in prose, which is where it rots -- an address that drifted between the | |
| model card and COMMERCIAL-LICENSE.md sends the one reader who intends to | |
| pay to a mailbox nobody reads, and nothing about the repository looks | |
| wrong. So the offer is checked the way the rest of this page is: the | |
| document exists, the page links to it, and the address is spelled once. | |
| """ | |
| offer = ROOT / "COMMERCIAL-LICENSE.md" | |
| assert offer.is_file(), ( | |
| "the model card offers a commercial licence and there is no " | |
| "COMMERCIAL-LICENSE.md to take it up in") | |
| readme = (ROOT / "README.md").read_text() | |
| assert "COMMERCIAL-LICENSE.md" in readme, ( | |
| "COMMERCIAL-LICENSE.md exists but the model card never links to it, " | |
| "so the Hub page shows the copyleft terms and not the way out of them") | |
| # Both files, one address. The model card's copy is the one a reader meets | |
| # first and the offer document's is the one they act on. | |
| # The trailing `\w` matters: `[\w.]+` at the end swallows the full stop of | |
| # any sentence the address ends, and then one address is two. | |
| address = r"[\w.+-]+@[\w-]+(?:\.[\w-]+)+" | |
| emails = set(re.findall(address, offer.read_text())) | |
| emails |= set(re.findall(address, readme)) | |
| assert len(emails) == 1, ( | |
| f"the commercial licence is offered at more than one address: " | |
| f"{sorted(emails)}") | |