"""The closed-loop controllers, tested against the bug that faked every maze number. ``question_profile`` only emits an ``action`` question when two or more moves are legal, and its criteria list only those legal moves. ``play_maze`` used to offer all four compass directions anyway and fall back to ``ACTIONS[0]`` when the question was absent, so at a one-exit cell the agent walked north into a wall, did not move, was handed the identical state, and chose the same wall again for the whole step budget. A solve rate of 0.00 reports that failure and a genuinely wandering agent identically, which is why these tests assert on movement rather than on solving. """ from __future__ import annotations import random import sys from pathlib import Path ROOT = Path(__file__).resolve().parents[1] sys.path.insert(0, str(ROOT)) import pytest import envs.maze as mz from jevon.inference import maze_state_sample sys.path.insert(0, str(ROOT / "scripts")) import play as play_mod # noqa: E402 class StubRunner: """Answers every action question with a fixed preference order. Preferring ``ACTIONS[0]`` is the adversarial case: it is exactly what the broken fallback returned, so a controller that still walks into walls under this stub has the old behaviour no matter what the real model would say. """ def __init__(self, order=mz.ACTIONS): self.order = list(order) self.calls = 0 def answer(self, samples): self.calls += 1 out = [] for sample in samples: for question in sample.questions: if question.qtype != "choice": continue keys = list(question.candidate_ids) ranked = sorted(keys, key=lambda k: self.order.index(k) if k in self.order else len(self.order)) probs = {k: 0.01 for k in keys} probs[ranked[0]] = 1.0 - 0.01 * (len(keys) - 1) out.append({"question": question.name, "probabilities": probs}) return out def one_exit_state(): """A maze state whose start cell has exactly one legal move.""" for size in (11, 21): for topology in mz.TOPOLOGIES: for seed in range(40): state = mz.generate_maze(size, topology, seed) if len(state.legal_actions()) == 1: return state raise AssertionError("no dead-end start found; generator changed") def test_a_one_exit_cell_omits_the_action_question(): """The premise of the bug: the model is not consulted at a forced move.""" state = one_exit_state() names = {q.name for q in maze_state_sample(state).questions} assert len(state.legal_actions()) == 1 assert "action" not in names, "profile must not ask about a forced move" # ...and the control: where there is a real choice, the question is there. forked = mz.generate_maze(11, "loops", 5) assert len(forked.legal_actions()) >= 2 assert "action" in {q.name for q in maze_state_sample(forked).questions} def test_forced_move_takes_the_only_legal_action(): state = one_exit_state() only = state.legal_actions()[0] action, _ = play_mod.model_action(StubRunner(), maze_state_sample(state), state.legal_actions(), random.Random(0), 0.0) assert action == only def test_model_action_never_returns_an_action_outside_the_candidate_set(): rng = random.Random(1) for seed in range(12): state = mz.generate_maze(11, "tree", seed) legal = state.legal_actions() action, _ = play_mod.model_action(StubRunner(), maze_state_sample(state), legal, rng, 0.0) assert action in legal def test_north_preferring_model_still_moves_every_step(): """The regression itself: collisions are impossible when only legal moves are offered. Under the old code this run recorded one collision per step for the whole budget and stood on a single square. """ for topology in mz.TOPOLOGIES: state = mz.generate_maze(11, topology, seed=7) out = play_mod.play_maze(StubRunner(), state, "model", 200, random.Random(3)) assert out["collisions"] == 0, f"{topology}: walked into a wall" assert not out["deadlocked"], f"{topology}: deadlocked" assert out["distinct_cells"] > 1, f"{topology}: never left the start cell" def test_the_model_is_consulted_whenever_a_choice_exists(): """A controller that never calls the network would also pass the test above.""" runner = StubRunner() state = mz.generate_maze(11, "loops", seed=5) play_mod.play_maze(runner, state, "model", 60, random.Random(4)) assert runner.calls > 0 def test_snake_offers_fatal_moves_on_purpose(): """Maze filters to legal moves; snake must not, or dying becomes unmeasurable.""" import envs.snake as sk state = sk.new_game(12, seed=3) candidates = state.candidate_actions() assert len(candidates) == 3, "every non-reversing action, safe or not" # ------------------------------------------------------- model selection def _train_mod(): import importlib.util spec = importlib.util.spec_from_file_location( "jevon_train", ROOT / "scripts" / "train.py") mod = importlib.util.module_from_spec(spec) spec.loader.exec_module(mod) return mod def test_balanced_score_refuses_to_average_away_a_traded_game(): """The real case from the final run: great at maze, below control at snake.""" train = _train_mod() traded = {"maze_action": 0.9794, "snake_action": 0.2973} even = {"maze_action": 0.72, "snake_action": 0.70} assert train.balanced_score(traded) == pytest.approx(0.2973) # A mean would rank the traded checkpoint (0.638) above the even one # (0.710); the whole point of the rule is that it does not. assert train.balanced_score(even) > train.balanced_score(traded) def test_balanced_score_scores_a_missing_game_as_zero(): train = _train_mod() assert train.balanced_score({"maze_action": 0.99}) == 0.0 assert train.balanced_score({}) == 0.0 # --------------------------------------------------------------- trace fidelity @pytest.mark.parametrize("size", [11, 21, 31, 51]) def test_recorded_field_survives_its_own_rounding(size): """The trace writer rounds to 4 decimals; the field is normalised by H*W. Those two numbers have to be read together. One BFS step is 1/(H*W) in the recorded units -- 0.0010 at size 31, 0.00038 at 51 -- and the writer keeps four decimals, so the margin is not large. If it ever closes, the viewer shades neighbouring distances identically and the overlay shows a flat basin where the planner had a gradient. That failure is silent everywhere it matters: solve rate, efficiency and every summary number are computed from the unrounded field, so they stay correct while the picture of them becomes false. Size 51 is included although the benchmark stops at 31, because it is the next size someone would reach for and the first where the margin gets thin. """ import torch from envs import maze as mz from jevon.grid import encode_maze, pad_boards from jevon.planner import MinPlusField, default_iterations st = mz.generate_maze(size, "loops", 900_001) board, _, _ = encode_maze(st) boards, passable, _ = pad_boards([board]) field = MinPlusField(32)(torch.randn(1, 32, size, size), boards, passable, iterations=default_iterations(size), grad_steps=0)[0, 0].detach() rounded = torch.round(field * 1e4) / 1e4 # exactly what play.py writes free = [(i, j) for i in range(size) for j in range(size) if board[0, i, j]] levels_before = len({float(field[c]) for c in free}) levels_after = len({float(rounded[c]) for c in free}) assert levels_after == levels_before, ( f"rounding merged {levels_before - levels_after} distance levels at " f"size {size}; the overlay will shade different distances alike") for i, j in free: if (i, j) == st.goal: continue nb = [rounded[i + d, j + e] for d, e in ((1, 0), (-1, 0), (0, 1), (0, -1)) if 0 <= i + d < size and 0 <= j + e < size and board[0, i + d, j + e]] assert not nb or min(nb) < rounded[i, j], ( f"cell {(i, j)} has no downhill neighbour once rounded; the " f"recorded field is not descendable at size {size}")