"""Model invariants, including regressions for three bugs that silently disabled the planner during development.""" from __future__ import annotations import random import sys from pathlib import Path ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) import pytest import torch from torch import nn import envs.maze as mz import envs.snake as sk from jevon.grid import bfs_field, encode_maze, encode_snake, pad_boards, pad_fields from jevon.losses import descent_loss, squash_distance, LossWeights from jevon.losses import decision_loss, field_loss, _grade_score_targets from jevon.modeling import JevonConfig, JevonModel from jevon.planner import SpatialPlanner, default_iterations from jevon.tokenizer import JevonTokenizer from jevon.data import TaskSampler, SamplerConfig, collate, build_corpus def tiny_model(vocab_size: int = 200, **overrides): cfg = JevonConfig(vocab_size=vocab_size, d_model=64, n_heads=4, prefix_layers=2, candidate_layers=1, set_layers=1, planner_dim=32, planner_blocks=2, dropout=0.0, **overrides) return JevonModel(cfg) # ------------------------------------------------------------------ planner def test_every_planner_parameter_receives_gradient(): """Regression: a zero-initialised final mix weight made ``update`` exactly zero, which zeroed the gradient of every parameter feeding it. The planner looked fine and trained nothing.""" planner = SpatialPlanner(8, 32, 2) board = torch.rand(2, 8, 9, 9) passable = (torch.rand(2, 1, 9, 9) > 0.2).float() planner(board, passable, iterations=6, grad_steps=6).sum().backward() missing = [n for n, p in planner.named_parameters() if p.grad is None or not p.grad.any()] assert not missing, f"no gradient reached: {missing}" def test_stem_is_trained_through_the_no_grad_window(): """Regression: detaching after the no-grad iterations severed the stem's only path to the loss, freezing the input projection at initialisation.""" planner = SpatialPlanner(8, 32, 2) board = torch.rand(1, 8, 7, 7) passable = torch.ones(1, 1, 7, 7) planner(board, passable, iterations=40, grad_steps=4).sum().backward() grads = [p.grad for p in planner.stem.parameters() if p.grad is not None] assert grads and sum(g.abs().sum() for g in grads) > 0 def test_long_recurrence_stays_finite(): planner = SpatialPlanner(8, 32, 2) board = torch.rand(1, 8, 11, 11) passable = torch.ones(1, 1, 11, 11) out = planner(board, passable, iterations=512, grad_steps=0) assert torch.isfinite(out).all() def test_iteration_budget_covers_the_true_maze_diameter(): """Regression: the budget was 2.5*size, but a 31x31 corridor maze has a diameter of 312. The distance signal could not physically reach the agent, which pinned maze accuracy at chance.""" for size in (11, 21, 31): worst = 0 for topology in mz.TOPOLOGIES: state = mz.generate_maze(size, topology, seed=size) distances = mz.bfs_distances(size, state.walls, state.goal) worst = max(worst, max(d for d in distances.values() if d < mz.UNREACHABLE)) assert default_iterations(size) >= worst, ( f"size {size}: budget {default_iterations(size)} < diameter {worst}") def test_planner_ignores_the_agent_so_fields_can_be_cached(): """The cache is only sound because the field never depends on the agent.""" model = tiny_model() state = mz.generate_maze(9, "tree", 3) board_a, _, target = encode_maze(state) moved = state.clone() moved.position = next(c for c in mz.bfs_distances(9, state.walls, state.position) if c != state.position and state.free(c)) board_b, _, _ = encode_maze(moved) boards, passable, _ = pad_boards([board_a, board_b]) with torch.no_grad(): field = model.encode_boards(boards, passable, 8, 0) assert torch.allclose(field[0], field[1], atol=1e-5) # --------------------------------------------------------------------- grid @pytest.mark.parametrize("game", ["maze", "snake"]) def test_bfs_field_matches_the_environment_oracle(game): if game == "maze": state = mz.generate_maze(13, "loops", 11) board, _, target = encode_maze(state) reference = mz.bfs_distances(13, state.walls, state.goal) field = bfs_field(board, target) for cell, distance in reference.items(): expected = -1.0 if distance >= mz.UNREACHABLE else float(distance) assert field[cell] == expected else: rng = random.Random(1) state = sk.random_state(12, 20, rng) board, _, target = encode_snake(state) field = bfs_field(board, target) assert field[target] == 0.0 assert (field >= 0).sum() > 0 # Body cells are impassable, so they are never assigned a distance. for cell in list(state.body)[:-1]: assert field[cell] == -1.0 def test_pad_fields_marks_padding_as_unsupervised(): padded = pad_fields([torch.zeros(3, 3), torch.zeros(5, 5)]) assert padded.shape == (2, 5, 5) assert (padded[0, 3:, :] == -1).all() and (padded[0, :, 3:] == -1).all() # -------------------------------------------------------------------- model def build_batch(states=3): cfg = SamplerConfig(maze_sizes=(9,), snake_sizes=(10,)) group = list(TaskSampler(cfg, "train", 5).batch(states)) tokenizer = JevonTokenizer.build( build_corpus(TaskSampler(cfg, "train", 6), 20), max_vocab=500, min_count=1) return group, tokenizer def test_forward_shapes_and_boolean_convention(): group, tokenizer = build_batch() model = tiny_model(len(tokenizer)) batch = collate(group, tokenizer) logits = model(batch, iterations=8, grad_steps=0) assert logits.shape[0] == batch["prefix_ids"].shape[0] probs = logits.detach().softmax(-1) for row, meta in enumerate(batch["meta"]): k = len(meta["probs"]) if meta["type"] == "boolean": assert k == 2, "a Boolean must expose exactly false/true" assert float(probs[row, :k].sum()) == pytest.approx(1.0, abs=1e-3) def test_anchors_actually_reach_the_scoring_head(): """If the anchor were ignored, two candidates naming different cells would score identically whenever their text embeddings matched.""" group, tokenizer = build_batch(1) model = tiny_model(len(tokenizer)).eval() batch = collate(group, tokenizer) with torch.no_grad(): base = model(batch, iterations=8, grad_steps=0) shifted = dict(batch) shifted["anchors"] = torch.zeros_like(batch["anchors"]) moved = model(shifted, iterations=8, grad_steps=0) assert not torch.allclose(base, moved, atol=1e-5) def test_field_loss_ignores_padding_and_unreachable_cells(): pred = torch.zeros(1, 2, 4, 4, requires_grad=True) dist = torch.full((1, 4, 4), -1.0) dist[0, 0, 0] = 6.0 passable = torch.ones(1, 1, 4, 4) loss, stats = field_loss(pred, dist, passable, torch.tensor([4])) loss.backward() # Only the single reachable cell may carry distance gradient. assert pred.grad[0, 0].abs().gt(0).sum() == 1 def test_decision_and_field_losses_compose(): group, tokenizer = build_batch() model = tiny_model(len(tokenizer)) batch = collate(group, tokenizer) cells = model.encode_boards(batch["boards"], batch["passable"], 8, 4) logits = model(batch, iterations=8, grad_steps=4, cells=cells) loss, _ = decision_loss(logits, batch) f_loss, stats = field_loss(model.predict_field(cells), batch["dist_field"], batch["passable"], batch["board_size"]) (loss + f_loss).backward() assert torch.isfinite(torch.tensor(stats["field_r2"])) assert model.field_head.weight.grad.abs().sum() > 0 @pytest.mark.slow def test_planner_can_learn_a_distance_field(): """The central claim: with dense supervision the recurrence learns to propagate distance. Trained on four small boards only.""" torch.manual_seed(0) model = tiny_model() boards, fields = [], [] for seed in range(4): state = mz.generate_maze(9, "tree", 100 + seed) board, _, target = encode_maze(state) boards.append(board) fields.append(bfs_field(board, target)) packed, passable, sizes = pad_boards(boards) dist = pad_fields(fields) opt = torch.optim.Adam( list(model.planner.parameters()) + list(model.field_head.parameters()), lr=3e-3) first = last = None for step in range(400): cells = model.encode_boards(packed, passable, 40, 8) loss, stats = field_loss(model.predict_field(cells), dist, passable, sizes) opt.zero_grad(set_to_none=True) loss.backward() torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0) opt.step() if step == 0: first = stats["field_r2"] last = stats["field_r2"] assert last > 0.75, f"planner failed to learn distance: R^2 {first:.3f} -> {last:.3f}" # ------------------------------------------------- ordinal score questions def test_graded_targets_keep_the_true_level_on_top(): """Smoothing must not move the argmax, or every accuracy shifts with it.""" from jevon.modeling import TYPE_INDEX target = torch.zeros(3, 7) for row, level in enumerate((0, 4, 6)): target[row, level] = 1.0 valid = torch.ones(3, 7, dtype=torch.bool) types = torch.full((3,), TYPE_INDEX["score"]) graded = _grade_score_targets(target, valid, types, 0.7) assert torch.equal(graded.argmax(-1), target.argmax(-1)) assert torch.allclose(graded.sum(-1), torch.ones(3)) # Mass must fall away monotonically from the true level, which is the # whole point: a near-miss should cost less than the far end of the scale. row = graded[1] assert row[4] > row[3] > row[2] > row[1] > row[0] assert row[4] > row[5] > row[6] def test_graded_targets_leave_other_question_types_alone(): from jevon.modeling import TYPE_INDEX target = torch.zeros(2, 4) target[0, 1] = 1.0 target[1, :2] = torch.tensor([0.5, 0.5]) # a genuine tie, not a soft label valid = torch.ones(2, 4, dtype=torch.bool) types = torch.tensor([TYPE_INDEX["choice"], TYPE_INDEX["boolean"]]) graded = _grade_score_targets(target, valid, types, 0.7) assert torch.equal(graded, target) def test_graded_targets_never_put_mass_on_absent_levels(): from jevon.modeling import TYPE_INDEX target = torch.zeros(1, 7) target[0, 2] = 1.0 valid = torch.zeros(1, 7, dtype=torch.bool) valid[0, :4] = True graded = _grade_score_targets(target, valid, types=torch.tensor([TYPE_INDEX["score"]]), tau=0.7) assert float(graded[0, 4:].sum()) == 0.0 assert float(graded.sum()) == pytest.approx(1.0, abs=1e-5) def test_smoothing_reaches_the_loss_only_for_score_rows(): group, tokenizer = build_batch() model = tiny_model(len(tokenizer)).eval() batch = collate(group, tokenizer) logits = model(batch, iterations=8, grad_steps=0) plain, _ = decision_loss(logits, batch, LossWeights()) graded, _ = decision_loss(logits, batch, LossWeights(score_smoothing=0.7)) types = batch["question_type"] from jevon.modeling import TYPE_INDEX if bool(types.eq(TYPE_INDEX["score"]).any()): assert not torch.allclose(plain, graded) else: # nothing to smooth, nothing to change assert torch.allclose(plain, graded) # ------------------------------------------------- exposing the field head def test_exposed_field_changes_what_the_text_side_reads(): """The gathered vector must actually carry the field head's output.""" torch.manual_seed(0) model = tiny_model(expose_field=True).eval() cells = torch.randn(2, model.cfg.planner_dim, 9, 9) readable = model._readable(cells) assert readable.shape[1] == model.cfg.planner_dim + 2 assert torch.allclose(readable[:, :model.cfg.planner_dim], cells) assert torch.allclose(readable[:, model.cfg.planner_dim:], model.predict_field(cells)) def test_exposed_field_is_off_by_default_so_checkpoints_stay_loadable(): plain = tiny_model() assert plain.cfg.expose_field is False cells = torch.randn(1, plain.cfg.planner_dim, 7, 7) assert torch.allclose(plain._readable(cells), cells) # The projection widths are what a checkpoint load actually checks. assert plain.grid_to_text.in_features == plain.cfg.planner_dim assert tiny_model(expose_field=True).grid_to_text.in_features == \ plain.cfg.planner_dim + 2 def test_exposed_field_model_trains_end_to_end(): group, tokenizer = build_batch() model = tiny_model(len(tokenizer), expose_field=True) batch = collate(group, tokenizer) logits = model(batch, iterations=8, grad_steps=4) loss, _ = decision_loss(logits, batch, LossWeights(score_smoothing=0.7)) loss.backward() assert model.field_head.weight.grad is not None assert float(model.field_head.weight.grad.abs().sum()) > 0, \ "the decision loss must now flow back into the field head" # ------------------------------------------------- field target shape def test_linear_target_keeps_neighbour_contrast_uniform(): """The quantity greedy descent depends on is the gap between neighbours. A log target compresses that gap as distance grows, so the hardest comparison on a long path is also the one with the least signal. """ from jevon.losses import squash_distance dist = torch.tensor([[[0.0, 1.0, 40.0, 41.0]]]) size = torch.tensor([11.0]) log = squash_distance(dist, size, "log")[0, 0] near, far = float(log[1] - log[0]), float(log[3] - log[2]) assert near > 10 * far, "log target should collapse far-field contrast" lin = squash_distance(dist, size, "linear")[0, 0] near, far = float(lin[1] - lin[0]), float(lin[3] - lin[2]) # float32 differencing near 0.33 is only good to ~1e-7 absolute; the claim # being tested is uniformity against a 28x collapse, not bit-equality. assert near == pytest.approx(far, rel=1e-4) assert near == pytest.approx(1.0 / 121.0, rel=1e-4) def test_both_target_shapes_are_monotone_and_bounded(): from jevon.losses import squash_distance dist = torch.arange(0.0, 60.0).reshape(1, 1, -1) size = torch.tensor([11.0]) for mode in ("log", "linear"): out = squash_distance(dist, size, mode)[0, 0] assert torch.all(out[1:] > out[:-1]), f"{mode} must be strictly increasing" assert float(out.min()) >= 0.0 and float(out.max()) <= 1.0 def test_field_loss_respects_the_requested_shape(): from jevon.losses import field_loss pred = torch.zeros(1, 2, 5, 5, requires_grad=True) dist = torch.full((1, 5, 5), 20.0) passable = torch.ones(1, 1, 5, 5) size = torch.tensor([5.0]) log_loss, _ = field_loss(pred, dist, passable, size, "log") lin_loss, _ = field_loss(pred, dist, passable, size, "linear") assert not torch.allclose(log_loss, lin_loss) def test_pooled_r2_is_not_the_mean_of_per_batch_r2(): """R² is a ratio of sums; averaging it lets one batch dominate. Two batches, each fit perfectly except for one cell, but with very different target spreads. The mean of the per-batch R² is dragged far below the pooled value that describes the predictions as a whole. """ from jevon.losses import field_loss def one(target_value, spread, err): dist = torch.full((1, 1, 4), float(target_value)) dist[0, 0, 1] += spread pred = torch.zeros(1, 2, 1, 4) pred[0, 0] = torch.nn.functional.pad(dist[0], (0, 0)) pred[0, 0, 0] += err return dist, pred stats = [] for target_value, spread, err in ((100.0, 200.0, 0.02), (40.0, 0.4, 0.02)): dist, pred = one(target_value, spread, err) _, st = field_loss(pred, dist, torch.ones(1, 1, 1, 4), torch.tensor([11.0])) stats.append(st) naive = sum(s["field_r2"] for s in stats) / len(stats) n = sum(s["_f_n"] for s in stats) sse = sum(s["_f_sse"] for s in stats) tsum = sum(s["_f_sum"] for s in stats) tsq = sum(s["_f_sumsq"] for s in stats) pooled = 1.0 - sse / max(tsq - tsum * tsum / n, 1e-9) assert pooled > naive, "pooling must not inherit the low-variance batch's blowup" assert min(s["field_r2"] for s in stats) < naive < pooled # --------------------------------------------------- descent ordering loss def _maze_field(size=9, seed=4242): """A real maze with its true BFS field, shaped as the loss expects.""" from envs import maze as mz st = mz.generate_maze(size, "tree", seed) truth = mz.bfs_distances(size, st.walls, st.goal) dist = torch.full((1, size, size), -1.0) inside = torch.zeros(1, 1, size, size) for cell, d in truth.items(): if st.free(cell): inside[0, 0, cell[0], cell[1]] = 1.0 if d >= 0: dist[0, cell[0], cell[1]] = float(d) return dist, inside, torch.tensor([size]) @pytest.mark.parametrize("mode", ["linear", "log"]) def test_true_field_has_no_ordering_penalty(mode): """The target itself must be a fixed point, or the term fights the regression.""" dist, inside, size = _maze_field() loss, ordered = descent_loss(squash_distance(dist, size, mode), dist, inside, size) assert float(loss) == pytest.approx(0.0, abs=1e-9) assert ordered == pytest.approx(1.0) def test_flat_field_is_penalised_by_exactly_one_margin(): """With no gradient to follow every neighbour misses by the full margin, which pins the loss to a value we can state rather than eyeball.""" dist, inside, size = _maze_field() loss, ordered = descent_loss(torch.zeros_like(dist), dist, inside, size) assert float(loss) == pytest.approx(1.0 / float(size[0]) ** 2, rel=1e-5) assert ordered < 0.2 # no better than picking a corner def test_ordering_degrades_with_noise_not_with_offset(): """Descent reads differences, so a constant shift must not change the score while noise of the same size must -- that is the whole point of the term.""" dist, inside, size = _maze_field() perfect = squash_distance(dist, size, "linear") torch.manual_seed(0) _, shifted = descent_loss(perfect + 0.25, dist, inside, size) _, noised = descent_loss(perfect + torch.randn_like(perfect) * 0.05, dist, inside, size) assert shifted == pytest.approx(1.0) assert noised < 0.8 def test_descent_term_is_off_unless_weighted(): """Existing checkpoints and runs must see byte-identical behaviour at 0.""" dist, inside, size = _maze_field() pred = torch.zeros(1, 2, size[0], size[0]) base, base_stats = field_loss(pred, dist, inside, size, "linear") on, on_stats = field_loss(pred, dist, inside, size, "linear", descent_weight=1.0) assert "field_ordered" not in base_stats assert on_stats["field_ordered"] < 0.2 assert float(on) > float(base) def test_descent_term_reaches_the_planner(): """A term that does not move the planner's weights cannot fix descent.""" torch.manual_seed(0) group, tokenizer = build_batch() model = tiny_model(len(tokenizer)) batch = collate(group, tokenizer) cells = model.encode_boards(batch["boards"], batch["passable"], 8, 4) loss, _ = field_loss(model.predict_field(cells), batch["dist_field"], batch["passable"], batch["board_size"], "linear", descent_weight=1.0, descent_margin=1.0) loss.backward() grads = [p.grad.abs().sum() for p in model.planner.parameters() if p.grad is not None] assert grads and float(sum(grads)) > 0.0 # ------------------------------------------------------- min-plus field head def _maze_board(size=11, seed=900_001, topology="tree"): from envs import maze as mz from jevon.grid import encode_maze, pad_boards st = mz.generate_maze(size, topology, seed) board, _, _ = encode_maze(st) boards, passable, _ = pad_boards([board]) truth = torch.full((size, size), -1.0) for cell, d in mz.bfs_distances(size, st.walls, st.goal).items(): if st.free(cell) and d >= 0: truth[cell] = float(d) return boards, passable, truth def test_min_plus_reproduces_exact_bfs_at_initialisation(): """Costs start at 1 per step, so the recurrence *is* breadth-first search. Starting from the right answer means training only has to adjust it.""" from jevon.planner import MinPlusField, default_iterations boards, passable, truth = _maze_board() head = MinPlusField(32) out = head(torch.randn(1, 32, 11, 11), boards, passable, iterations=default_iterations(11), grad_steps=0) v = out[0, 0].detach() * 11 * 11 # back into step units mask = truth >= 0 assert float((v[mask] - truth[mask]).abs().max()) < 1e-2 @pytest.mark.parametrize("seed", [0, 1, 2]) @pytest.mark.parametrize("size,topology", [(11, "tree"), (21, "loops"), (31, "loops"), (21, "random_obstacle")]) def test_min_plus_field_is_always_traversable(seed, size, topology): """The point of the head: whatever the costs, descent reaches the goal. Randomising the cost weights gives a field with no resemblance to the true distance, and it must *still* have no spurious basin -- that is what makes this an architectural guarantee rather than a well-trained checkpoint. Sized up to 31 and run on `loops` because that is the shape of the claim the README makes, and because a spurious basin needs somewhere to hide: a `tree` maze has exactly one path between any two cells, so descent has almost no opportunity to go wrong there, and an 11x11 board is small enough that a bad field still stumbles into the goal. Cyclic boards at 31x31 are where a guarantee that only held approximately would break. """ from jevon.planner import MinPlusField, default_iterations from scripts.probe import descent_outcomes boards, passable, truth = _maze_board(size=size, topology=topology) torch.manual_seed(seed) head = MinPlusField(32) nn.init.normal_(head.cost.weight, std=1.0) nn.init.normal_(head.cost.bias, std=1.0) out = head(torch.randn(1, 32, size, size), boards, passable, iterations=default_iterations(size), grad_steps=0) assert descent_outcomes(out[0, 0], truth) == pytest.approx(1.0) def test_min_plus_walls_are_never_entered(): """A wall that leaks a finite value would open a shortcut through it.""" from jevon.planner import MinPlusField, default_iterations boards, passable, truth = _maze_board() head = MinPlusField(32) out = head(torch.randn(1, 32, 11, 11), boards, passable, iterations=default_iterations(11), grad_steps=0) walls = (truth < 0) & (passable[0, 0] < 0.5) assert bool((out[0, 0][walls] >= MinPlusField.BIG).all()) def test_min_plus_is_off_by_default(): """Checkpoints trained with the conv head must keep loading.""" from jevon.planner import MinPlusField assert JevonConfig(vocab_size=10).min_plus_field is False assert tiny_model().min_plus is None def test_min_plus_model_trains_end_to_end(): group, tokenizer = build_batch() model = tiny_model(len(tokenizer), min_plus_field=True) batch = collate(group, tokenizer) cells = model.encode_boards(batch["boards"], batch["passable"], 8, 4) pred = model.predict_field(cells, batch["boards"], batch["passable"], iterations=8, grad_steps=4) loss, _ = field_loss(pred, batch["dist_field"], batch["passable"], batch["board_size"], "linear") loss.backward() assert float(model.min_plus.cost.weight.grad.abs().sum()) > 0.0 def test_min_plus_head_refuses_to_guess_the_board(): """Silently falling back to features would produce a field with no walls.""" model = tiny_model(200, min_plus_field=True) with pytest.raises(ValueError, match="needs the board"): model.predict_field(torch.randn(1, 32, 9, 9)) def test_min_plus_leaves_the_planner_supervised(): """The read-out matches BFS at init, so a loss on it is ~0 and its zero-initialised cost conv passes nothing back. If that were the only field head, turning min-plus on would silently switch off the dense supervision the recurrence is trained by.""" group, tokenizer = build_batch() model = tiny_model(len(tokenizer), min_plus_field=True) batch = collate(group, tokenizer) cells = model.encode_boards(batch["boards"], batch["passable"], 8, 4) loss, _ = field_loss(model.supervised_field(cells), batch["dist_field"], batch["passable"], batch["board_size"], "linear") loss.backward() planner_grad = sum(float(p.grad.abs().sum()) for p in model.planner.parameters() if p.grad is not None) assert planner_grad > 0.0 def test_read_out_and_supervised_head_agree_without_min_plus(): """Off by default means literally the same tensor, not merely similar.""" group, tokenizer = build_batch() model = tiny_model(len(tokenizer)) batch = collate(group, tokenizer) cells = model.encode_boards(batch["boards"], batch["passable"], 8, 0) assert torch.equal(model.predict_field(cells), model.supervised_field(cells)) def test_the_planner_never_sees_the_agent(): """Two boards differing only in where the agent stands are one board. This is not a nicety -- `JevonRunner._fields` keys its cache on the board with the focus channel zeroed, but hands the *unzeroed* board to `encode_boards`. Those two agree only because `encode_boards` zeroes it again on its own copy. If the planner ever started reading focus, the key would still collide across an episode and every step after the first would silently get the field computed for the agent's starting cell. Nothing else in the suite would notice: the field would be plausible, self-consistent, and wrong. """ torch.manual_seed(0) model = tiny_model().eval() state = mz.generate_maze(11, "tree", 3) board, _, _ = encode_maze(state) boards, passable, _ = pad_boards([board]) here = boards.clone() there = boards.clone() here[:, JevonModel.FOCUS_CHANNEL] = 0.0 there[:, JevonModel.FOCUS_CHANNEL] = 0.0 # Two different cells, both passable, neither the one the other marks. free = [(r, c) for r in range(state.size) for c in range(state.size) if state.free((r, c))][:2] assert len(free) == 2, "maze too small to put the agent in two places" for cell, grid in zip(free, (here, there)): grid[0, JevonModel.FOCUS_CHANNEL, cell[0], cell[1]] = 1.0 with torch.no_grad(): steps = default_iterations(state.size) a = model.encode_boards(here, passable, steps, 0) b = model.encode_boards(there, passable, steps, 0) assert torch.equal(a, b), ( "moving the agent changed the planner's features, so the field cache " "in JevonRunner._fields is serving stale fields") def test_a_maze_field_is_computed_once_per_episode(): """The README's `planner_calls=1, cache_hits=9` over ten decisions. That pair is the entire evidence for "computed **once**", which is in turn the reason the architecture section claims a per-decision cost independent of episode length. It is also the first thing to break quietly: anything that makes two states of one maze hash differently turns the cache off, and every number in the repo stays exactly the same except a wall-clock figure nobody re-measures. Built on a throwaway checkpoint rather than the shipped one -- the counters are a property of the cache, not of the weights, and a test that needs an 80MB state dict is a test that gets skipped. """ import dataclasses import json as jsonmod import tempfile from jevon.inference import JevonRunner, maze_state_sample cfg = SamplerConfig(maze_sizes=(11,), snake_sizes=(10,)) tokenizer = JevonTokenizer.build( build_corpus(TaskSampler(cfg, "train", 6), 20), max_vocab=500, min_count=1) torch.manual_seed(0) model = tiny_model(len(tokenizer)) with tempfile.TemporaryDirectory() as tmp: path = Path(tmp) tokenizer.save(path / "tokenizer.json") (path / "config.json").write_text( jsonmod.dumps(dataclasses.asdict(model.cfg))) torch.save(model.state_dict(), path / "weights.pt") runner = JevonRunner(path, device="cpu", weights="weights.pt") state = mz.generate_maze(11, "tree", 3) # Walk back and forth over one edge, so the agent is somewhere new on # every decision and the episode cannot end early. Both directions # stay legal for as long as the loop runs. forward = state.legal_actions()[0] back = {"north": "south", "south": "north", "east": "west", "west": "east"}[forward] decisions = 0 for i in range(10): runner.answer([maze_state_sample(state)]) decisions += 1 state.step(forward if i % 2 == 0 else back) assert (runner.planner_calls, runner.planner_cache_hits) == (1, decisions - 1), ( f"{decisions} decisions on one maze cost " f"{runner.planner_calls} planner calls and " f"{runner.planner_cache_hits} cache hits; the field is supposed to " f"be computed once per board") readme = (ROOT / "README.md").read_text() assert (f"`planner_calls={runner.planner_calls}`, " f"`cache_hits={runner.planner_cache_hits}`") in readme, ( f"README's cache figures disagree with the measured " f"planner_calls={runner.planner_calls}, " f"cache_hits={runner.planner_cache_hits}") def test_a_checkpoint_without_weights_says_why(tmp_path): """The shipped run directory has its JSON and not its `.pt` files. Not the shipped one any more -- that carries both checkpoints through Git LFS -- but it is still what a reader gets from `train.py --out runs/mine` before training finishes, and what a clone made without LFS amounts to. Either way the directory exists, parses, builds a model, and only then dies on a missing file. The path alone says neither which of those two situations it is nor what to do about it, and this is the first command a fresh clone runs. """ import json as jsonmod, dataclasses from jevon.inference import JevonRunner tokenizer = JevonTokenizer.build( build_corpus(TaskSampler(SamplerConfig(maze_sizes=(11,), snake_sizes=(10,)), "train", 6), 20), max_vocab=500, min_count=1) tokenizer.save(tmp_path / "tokenizer.json") (tmp_path / "config.json").write_text( jsonmod.dumps(dataclasses.asdict(tiny_model(len(tokenizer)).cfg))) with pytest.raises(FileNotFoundError) as excinfo: JevonRunner(tmp_path, device="cpu", weights="balanced.pt") message = str(excinfo.value) assert "balanced.pt" in message for clue in ("git lfs pull", "scripts/train.py", "--checkpoint"): assert clue in message, f"the error never mentions {clue!r}: {message}"