Jevon: a 20M-parameter decision model for Maze and Snake
Browse filesAnswers typed questions about a grid -- "which way should I move?", "is north
clear?", "how far is the goal?" -- with calibrated probabilities and no token
decoding. A from-scratch reimplementation of the idea behind NanoJev, rebuilt
around the thing that made the original fail at Maze: planning depth. The
backbone is a min-plus recurrence whose fixed point is a shortest-path
distance by construction; the three Jev primitives (Choice, Score, Boolean)
are kept as the interface.
Trained for 9,000 steps on one Apple M4 Max. On the held-out test split it
picks the BFS-optimal maze move 141 times out of 141 and the oracle snake move
89 out of 92, and the planner's distance field matches BFS to within 0.01
cells after a single global rescale.
This repository is the release: the weights (balanced.pt and best.pt, through
Git LFS), the model code, the training and evaluation scripts, the frozen
splits every number is measured on, and a model card whose tables are
generated from the JSON beside them rather than typed. `pytest -q` is 143
checks, most of them joins between a file that writes a number and a file that
quotes it.
AGPL-3.0-or-later, so anything built on it is open source too -- including
a network service, which is what section 13 is for. A commercial licence
without those terms is available: see COMMERCIAL-LICENSE.md.
Watch it play: https://github.com/lewislulu/jevon-arcade
- .gitattributes +21 -0
- .gitignore +35 -0
- COMMERCIAL-LICENSE.md +38 -0
- LICENSE +661 -0
- README.md +810 -0
- RESULTS.md +157 -0
- data/frozen/manifest.json +31 -0
- data/frozen/ood.jsonl +0 -0
- data/frozen/test.jsonl +0 -0
- data/frozen/val.jsonl +0 -0
- envs/__init__.py +1 -0
- envs/maze.py +363 -0
- envs/snake.py +386 -0
- jevon/__init__.py +33 -0
- jevon/data.py +380 -0
- jevon/grid.py +141 -0
- jevon/hub.py +88 -0
- jevon/inference.py +252 -0
- jevon/losses.py +290 -0
- jevon/modeling.py +390 -0
- jevon/planner.py +227 -0
- jevon/tokenizer.py +101 -0
- pyproject.toml +30 -0
- requirements.txt +9 -0
- runs/jevon-final/args.json +49 -0
- runs/jevon-final/balanced.json +18 -0
- runs/jevon-final/balanced.pt +3 -0
- runs/jevon-final/best.json +17 -0
- runs/jevon-final/best.pt +3 -0
- runs/jevon-final/config.json +20 -0
- runs/jevon-final/eval.json +359 -0
- runs/jevon-final/history.json +308 -0
- runs/jevon-final/play/maze_model_field_summary.json +543 -0
- runs/jevon-final/play/maze_model_memory_summary.json +543 -0
- runs/jevon-final/play/maze_model_sampled_summary.json +543 -0
- runs/jevon-final/play/maze_model_summary.json +543 -0
- runs/jevon-final/play/maze_random_memory_summary.json +543 -0
- runs/jevon-final/play/maze_random_summary.json +543 -0
- runs/jevon-final/play/maze_reference_summary.json +543 -0
- runs/jevon-final/play/snake_model_summary.json +74 -0
- runs/jevon-final/play/snake_random_summary.json +74 -0
- runs/jevon-final/play/snake_reference_summary.json +74 -0
- runs/jevon-final/probe.json +42 -0
- runs/jevon-final/tokenizer.json +1 -0
- scripts/benchmark.sh +91 -0
- scripts/build_dataset.py +47 -0
- scripts/evaluate.py +128 -0
- scripts/make_results.py +244 -0
- scripts/play.py +413 -0
- scripts/probe.py +178 -0
|
@@ -33,3 +33,24 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
|
| 37 |
+
# ---------------------------------------------------------------------------
|
| 38 |
+
# Everything above is the list HuggingFace writes into a new repository, kept
|
| 39 |
+
# verbatim: a format added here later is then already routed correctly, and a
|
| 40 |
+
# diff against a fresh Hub repo shows only what this project chose.
|
| 41 |
+
#
|
| 42 |
+
# Checkpoints are why it matters. An 80MB blob in the git object store is a
|
| 43 |
+
# clone people retry rather than finish, and the Hub rejects non-LFS files
|
| 44 |
+
# above 10MB outright -- `*.pt` is covered above, and a clone made without LFS
|
| 45 |
+
# gets a pointer file that `resolve_weights` names and explains rather than
|
| 46 |
+
# dying inside `torch.load`.
|
| 47 |
+
|
| 48 |
+
# The frozen splits stay ordinary git objects. They are 14MB of JSONL, under
|
| 49 |
+
# the Hub's LFS threshold, and keeping them out of LFS means a clone made
|
| 50 |
+
# without LFS still has everything `pytest -q` reads.
|
| 51 |
+
*.jsonl -text
|
| 52 |
+
|
| 53 |
+
# Generated tables live inside README.md and RESULTS.md; a merge driver that
|
| 54 |
+
# tries to be clever about them produces a file that no longer matches the
|
| 55 |
+
# JSON it is generated from, which the suite then fails on.
|
| 56 |
+
*.md -merge
|
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
__pycache__/
|
| 2 |
+
*.py[cod]
|
| 3 |
+
.venv/
|
| 4 |
+
.pytest_cache/
|
| 5 |
+
.DS_Store
|
| 6 |
+
|
| 7 |
+
# Checkpoints are ignored by default, then the shipped run is exempted file by
|
| 8 |
+
# file below. `*.pt` comes first because gitignore is last-match-wins.
|
| 9 |
+
*.pt
|
| 10 |
+
|
| 11 |
+
# Runs are ignored except the one this model card is about, and that one is
|
| 12 |
+
# tracked *completely* -- its JSON metadata and the two checkpoints every
|
| 13 |
+
# number here is scored from. The exception is not housekeeping:
|
| 14 |
+
#
|
| 15 |
+
# - README.md and RESULTS.md are generated from these JSON files, and the
|
| 16 |
+
# tests that check the prose against them skip when they are absent. An
|
| 17 |
+
# earlier `.gitignore` read `runs/`, which shipped a repo whose guards all
|
| 18 |
+
# passed by not running.
|
| 19 |
+
# - The weights are the model. A repo that carries the recipe and not the
|
| 20 |
+
# artefact is not a model release, and `from_pretrained` would have
|
| 21 |
+
# nothing to download.
|
| 22 |
+
#
|
| 23 |
+
# `last.pt` stays out: it is wherever training stopped, which is not a claim
|
| 24 |
+
# this repo makes about anything.
|
| 25 |
+
runs/*
|
| 26 |
+
!runs/jevon-final/
|
| 27 |
+
runs/jevon-final/*
|
| 28 |
+
!runs/jevon-final/*.json
|
| 29 |
+
!runs/jevon-final/balanced.pt
|
| 30 |
+
!runs/jevon-final/best.pt
|
| 31 |
+
!runs/jevon-final/play/
|
| 32 |
+
!runs/jevon-final/play/*.json
|
| 33 |
+
|
| 34 |
+
# Training logs and scratch runs.
|
| 35 |
+
runs/*.log
|
|
@@ -0,0 +1,38 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Commercial licence
|
| 2 |
+
|
| 3 |
+
Jevon is published under the GNU Affero General Public License, version 3 or
|
| 4 |
+
later — the full text is in [LICENSE](LICENSE).
|
| 5 |
+
|
| 6 |
+
The AGPL is a strong copyleft licence. You may use, modify and run this model
|
| 7 |
+
and its code for any purpose, including commercially, on one condition: the
|
| 8 |
+
complete source of whatever you build on it has to be offered under the AGPL
|
| 9 |
+
as well. Section 13 extends that to people who never receive a copy — if your
|
| 10 |
+
users reach Jevon over a network, they are entitled to the source of the
|
| 11 |
+
service they are reaching. Serving it behind an API counts.
|
| 12 |
+
|
| 13 |
+
If your product cannot carry that obligation, a separate commercial licence is
|
| 14 |
+
available. It covers the same artefacts under terms with no copyleft clause,
|
| 15 |
+
so your own source stays yours.
|
| 16 |
+
|
| 17 |
+
**Contact:** Lewis <sudolewis@gmail.com>
|
| 18 |
+
|
| 19 |
+
Useful to include: who you are, what you intend to build, and whether you will
|
| 20 |
+
be distributing the weights or serving them. The second case is the one the
|
| 21 |
+
AGPL treats differently, so it changes the answer.
|
| 22 |
+
|
| 23 |
+
## What both licences cover
|
| 24 |
+
|
| 25 |
+
The model code in `jevon/`, the environments in `envs/`, the training and
|
| 26 |
+
evaluation scripts, and the checkpoints in `runs/jevon-final/`.
|
| 27 |
+
|
| 28 |
+
Copyright © 2026 Lewis. Sole authorship is what makes a second licence
|
| 29 |
+
possible at all: there is no CLA to collect and no third-party contribution
|
| 30 |
+
that would have to be re-licensed first. That stays true only while it stays
|
| 31 |
+
true, so a pull request large enough to be copyrightable will be asked for a
|
| 32 |
+
licence grant before it is merged.
|
| 33 |
+
|
| 34 |
+
## What this is not
|
| 35 |
+
|
| 36 |
+
Not legal advice, and not a warranty. The AGPL's sections 15 and 16 disclaim
|
| 37 |
+
both, and a commercial licence is a negotiated document rather than a file in
|
| 38 |
+
a repository — this page is an offer to start that conversation.
|
|
@@ -0,0 +1,661 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
GNU AFFERO GENERAL PUBLIC LICENSE
|
| 2 |
+
Version 3, 19 November 2007
|
| 3 |
+
|
| 4 |
+
Copyright (C) 2007 Free Software Foundation, Inc. <https://fsf.org/>
|
| 5 |
+
Everyone is permitted to copy and distribute verbatim copies
|
| 6 |
+
of this license document, but changing it is not allowed.
|
| 7 |
+
|
| 8 |
+
Preamble
|
| 9 |
+
|
| 10 |
+
The GNU Affero General Public License is a free, copyleft license for
|
| 11 |
+
software and other kinds of works, specifically designed to ensure
|
| 12 |
+
cooperation with the community in the case of network server software.
|
| 13 |
+
|
| 14 |
+
The licenses for most software and other practical works are designed
|
| 15 |
+
to take away your freedom to share and change the works. By contrast,
|
| 16 |
+
our General Public Licenses are intended to guarantee your freedom to
|
| 17 |
+
share and change all versions of a program--to make sure it remains free
|
| 18 |
+
software for all its users.
|
| 19 |
+
|
| 20 |
+
When we speak of free software, we are referring to freedom, not
|
| 21 |
+
price. Our General Public Licenses are designed to make sure that you
|
| 22 |
+
have the freedom to distribute copies of free software (and charge for
|
| 23 |
+
them if you wish), that you receive source code or can get it if you
|
| 24 |
+
want it, that you can change the software or use pieces of it in new
|
| 25 |
+
free programs, and that you know you can do these things.
|
| 26 |
+
|
| 27 |
+
Developers that use our General Public Licenses protect your rights
|
| 28 |
+
with two steps: (1) assert copyright on the software, and (2) offer
|
| 29 |
+
you this License which gives you legal permission to copy, distribute
|
| 30 |
+
and/or modify the software.
|
| 31 |
+
|
| 32 |
+
A secondary benefit of defending all users' freedom is that
|
| 33 |
+
improvements made in alternate versions of the program, if they
|
| 34 |
+
receive widespread use, become available for other developers to
|
| 35 |
+
incorporate. Many developers of free software are heartened and
|
| 36 |
+
encouraged by the resulting cooperation. However, in the case of
|
| 37 |
+
software used on network servers, this result may fail to come about.
|
| 38 |
+
The GNU General Public License permits making a modified version and
|
| 39 |
+
letting the public access it on a server without ever releasing its
|
| 40 |
+
source code to the public.
|
| 41 |
+
|
| 42 |
+
The GNU Affero General Public License is designed specifically to
|
| 43 |
+
ensure that, in such cases, the modified source code becomes available
|
| 44 |
+
to the community. It requires the operator of a network server to
|
| 45 |
+
provide the source code of the modified version running there to the
|
| 46 |
+
users of that server. Therefore, public use of a modified version, on
|
| 47 |
+
a publicly accessible server, gives the public access to the source
|
| 48 |
+
code of the modified version.
|
| 49 |
+
|
| 50 |
+
An older license, called the Affero General Public License and
|
| 51 |
+
published by Affero, was designed to accomplish similar goals. This is
|
| 52 |
+
a different license, not a version of the Affero GPL, but Affero has
|
| 53 |
+
released a new version of the Affero GPL which permits relicensing under
|
| 54 |
+
this license.
|
| 55 |
+
|
| 56 |
+
The precise terms and conditions for copying, distribution and
|
| 57 |
+
modification follow.
|
| 58 |
+
|
| 59 |
+
TERMS AND CONDITIONS
|
| 60 |
+
|
| 61 |
+
0. Definitions.
|
| 62 |
+
|
| 63 |
+
"This License" refers to version 3 of the GNU Affero General Public License.
|
| 64 |
+
|
| 65 |
+
"Copyright" also means copyright-like laws that apply to other kinds of
|
| 66 |
+
works, such as semiconductor masks.
|
| 67 |
+
|
| 68 |
+
"The Program" refers to any copyrightable work licensed under this
|
| 69 |
+
License. Each licensee is addressed as "you". "Licensees" and
|
| 70 |
+
"recipients" may be individuals or organizations.
|
| 71 |
+
|
| 72 |
+
To "modify" a work means to copy from or adapt all or part of the work
|
| 73 |
+
in a fashion requiring copyright permission, other than the making of an
|
| 74 |
+
exact copy. The resulting work is called a "modified version" of the
|
| 75 |
+
earlier work or a work "based on" the earlier work.
|
| 76 |
+
|
| 77 |
+
A "covered work" means either the unmodified Program or a work based
|
| 78 |
+
on the Program.
|
| 79 |
+
|
| 80 |
+
To "propagate" a work means to do anything with it that, without
|
| 81 |
+
permission, would make you directly or secondarily liable for
|
| 82 |
+
infringement under applicable copyright law, except executing it on a
|
| 83 |
+
computer or modifying a private copy. Propagation includes copying,
|
| 84 |
+
distribution (with or without modification), making available to the
|
| 85 |
+
public, and in some countries other activities as well.
|
| 86 |
+
|
| 87 |
+
To "convey" a work means any kind of propagation that enables other
|
| 88 |
+
parties to make or receive copies. Mere interaction with a user through
|
| 89 |
+
a computer network, with no transfer of a copy, is not conveying.
|
| 90 |
+
|
| 91 |
+
An interactive user interface displays "Appropriate Legal Notices"
|
| 92 |
+
to the extent that it includes a convenient and prominently visible
|
| 93 |
+
feature that (1) displays an appropriate copyright notice, and (2)
|
| 94 |
+
tells the user that there is no warranty for the work (except to the
|
| 95 |
+
extent that warranties are provided), that licensees may convey the
|
| 96 |
+
work under this License, and how to view a copy of this License. If
|
| 97 |
+
the interface presents a list of user commands or options, such as a
|
| 98 |
+
menu, a prominent item in the list meets this criterion.
|
| 99 |
+
|
| 100 |
+
1. Source Code.
|
| 101 |
+
|
| 102 |
+
The "source code" for a work means the preferred form of the work
|
| 103 |
+
for making modifications to it. "Object code" means any non-source
|
| 104 |
+
form of a work.
|
| 105 |
+
|
| 106 |
+
A "Standard Interface" means an interface that either is an official
|
| 107 |
+
standard defined by a recognized standards body, or, in the case of
|
| 108 |
+
interfaces specified for a particular programming language, one that
|
| 109 |
+
is widely used among developers working in that language.
|
| 110 |
+
|
| 111 |
+
The "System Libraries" of an executable work include anything, other
|
| 112 |
+
than the work as a whole, that (a) is included in the normal form of
|
| 113 |
+
packaging a Major Component, but which is not part of that Major
|
| 114 |
+
Component, and (b) serves only to enable use of the work with that
|
| 115 |
+
Major Component, or to implement a Standard Interface for which an
|
| 116 |
+
implementation is available to the public in source code form. A
|
| 117 |
+
"Major Component", in this context, means a major essential component
|
| 118 |
+
(kernel, window system, and so on) of the specific operating system
|
| 119 |
+
(if any) on which the executable work runs, or a compiler used to
|
| 120 |
+
produce the work, or an object code interpreter used to run it.
|
| 121 |
+
|
| 122 |
+
The "Corresponding Source" for a work in object code form means all
|
| 123 |
+
the source code needed to generate, install, and (for an executable
|
| 124 |
+
work) run the object code and to modify the work, including scripts to
|
| 125 |
+
control those activities. However, it does not include the work's
|
| 126 |
+
System Libraries, or general-purpose tools or generally available free
|
| 127 |
+
programs which are used unmodified in performing those activities but
|
| 128 |
+
which are not part of the work. For example, Corresponding Source
|
| 129 |
+
includes interface definition files associated with source files for
|
| 130 |
+
the work, and the source code for shared libraries and dynamically
|
| 131 |
+
linked subprograms that the work is specifically designed to require,
|
| 132 |
+
such as by intimate data communication or control flow between those
|
| 133 |
+
subprograms and other parts of the work.
|
| 134 |
+
|
| 135 |
+
The Corresponding Source need not include anything that users
|
| 136 |
+
can regenerate automatically from other parts of the Corresponding
|
| 137 |
+
Source.
|
| 138 |
+
|
| 139 |
+
The Corresponding Source for a work in source code form is that
|
| 140 |
+
same work.
|
| 141 |
+
|
| 142 |
+
2. Basic Permissions.
|
| 143 |
+
|
| 144 |
+
All rights granted under this License are granted for the term of
|
| 145 |
+
copyright on the Program, and are irrevocable provided the stated
|
| 146 |
+
conditions are met. This License explicitly affirms your unlimited
|
| 147 |
+
permission to run the unmodified Program. The output from running a
|
| 148 |
+
covered work is covered by this License only if the output, given its
|
| 149 |
+
content, constitutes a covered work. This License acknowledges your
|
| 150 |
+
rights of fair use or other equivalent, as provided by copyright law.
|
| 151 |
+
|
| 152 |
+
You may make, run and propagate covered works that you do not
|
| 153 |
+
convey, without conditions so long as your license otherwise remains
|
| 154 |
+
in force. You may convey covered works to others for the sole purpose
|
| 155 |
+
of having them make modifications exclusively for you, or provide you
|
| 156 |
+
with facilities for running those works, provided that you comply with
|
| 157 |
+
the terms of this License in conveying all material for which you do
|
| 158 |
+
not control copyright. Those thus making or running the covered works
|
| 159 |
+
for you must do so exclusively on your behalf, under your direction
|
| 160 |
+
and control, on terms that prohibit them from making any copies of
|
| 161 |
+
your copyrighted material outside their relationship with you.
|
| 162 |
+
|
| 163 |
+
Conveying under any other circumstances is permitted solely under
|
| 164 |
+
the conditions stated below. Sublicensing is not allowed; section 10
|
| 165 |
+
makes it unnecessary.
|
| 166 |
+
|
| 167 |
+
3. Protecting Users' Legal Rights From Anti-Circumvention Law.
|
| 168 |
+
|
| 169 |
+
No covered work shall be deemed part of an effective technological
|
| 170 |
+
measure under any applicable law fulfilling obligations under article
|
| 171 |
+
11 of the WIPO copyright treaty adopted on 20 December 1996, or
|
| 172 |
+
similar laws prohibiting or restricting circumvention of such
|
| 173 |
+
measures.
|
| 174 |
+
|
| 175 |
+
When you convey a covered work, you waive any legal power to forbid
|
| 176 |
+
circumvention of technological measures to the extent such circumvention
|
| 177 |
+
is effected by exercising rights under this License with respect to
|
| 178 |
+
the covered work, and you disclaim any intention to limit operation or
|
| 179 |
+
modification of the work as a means of enforcing, against the work's
|
| 180 |
+
users, your or third parties' legal rights to forbid circumvention of
|
| 181 |
+
technological measures.
|
| 182 |
+
|
| 183 |
+
4. Conveying Verbatim Copies.
|
| 184 |
+
|
| 185 |
+
You may convey verbatim copies of the Program's source code as you
|
| 186 |
+
receive it, in any medium, provided that you conspicuously and
|
| 187 |
+
appropriately publish on each copy an appropriate copyright notice;
|
| 188 |
+
keep intact all notices stating that this License and any
|
| 189 |
+
non-permissive terms added in accord with section 7 apply to the code;
|
| 190 |
+
keep intact all notices of the absence of any warranty; and give all
|
| 191 |
+
recipients a copy of this License along with the Program.
|
| 192 |
+
|
| 193 |
+
You may charge any price or no price for each copy that you convey,
|
| 194 |
+
and you may offer support or warranty protection for a fee.
|
| 195 |
+
|
| 196 |
+
5. Conveying Modified Source Versions.
|
| 197 |
+
|
| 198 |
+
You may convey a work based on the Program, or the modifications to
|
| 199 |
+
produce it from the Program, in the form of source code under the
|
| 200 |
+
terms of section 4, provided that you also meet all of these conditions:
|
| 201 |
+
|
| 202 |
+
a) The work must carry prominent notices stating that you modified
|
| 203 |
+
it, and giving a relevant date.
|
| 204 |
+
|
| 205 |
+
b) The work must carry prominent notices stating that it is
|
| 206 |
+
released under this License and any conditions added under section
|
| 207 |
+
7. This requirement modifies the requirement in section 4 to
|
| 208 |
+
"keep intact all notices".
|
| 209 |
+
|
| 210 |
+
c) You must license the entire work, as a whole, under this
|
| 211 |
+
License to anyone who comes into possession of a copy. This
|
| 212 |
+
License will therefore apply, along with any applicable section 7
|
| 213 |
+
additional terms, to the whole of the work, and all its parts,
|
| 214 |
+
regardless of how they are packaged. This License gives no
|
| 215 |
+
permission to license the work in any other way, but it does not
|
| 216 |
+
invalidate such permission if you have separately received it.
|
| 217 |
+
|
| 218 |
+
d) If the work has interactive user interfaces, each must display
|
| 219 |
+
Appropriate Legal Notices; however, if the Program has interactive
|
| 220 |
+
interfaces that do not display Appropriate Legal Notices, your
|
| 221 |
+
work need not make them do so.
|
| 222 |
+
|
| 223 |
+
A compilation of a covered work with other separate and independent
|
| 224 |
+
works, which are not by their nature extensions of the covered work,
|
| 225 |
+
and which are not combined with it such as to form a larger program,
|
| 226 |
+
in or on a volume of a storage or distribution medium, is called an
|
| 227 |
+
"aggregate" if the compilation and its resulting copyright are not
|
| 228 |
+
used to limit the access or legal rights of the compilation's users
|
| 229 |
+
beyond what the individual works permit. Inclusion of a covered work
|
| 230 |
+
in an aggregate does not cause this License to apply to the other
|
| 231 |
+
parts of the aggregate.
|
| 232 |
+
|
| 233 |
+
6. Conveying Non-Source Forms.
|
| 234 |
+
|
| 235 |
+
You may convey a covered work in object code form under the terms
|
| 236 |
+
of sections 4 and 5, provided that you also convey the
|
| 237 |
+
machine-readable Corresponding Source under the terms of this License,
|
| 238 |
+
in one of these ways:
|
| 239 |
+
|
| 240 |
+
a) Convey the object code in, or embodied in, a physical product
|
| 241 |
+
(including a physical distribution medium), accompanied by the
|
| 242 |
+
Corresponding Source fixed on a durable physical medium
|
| 243 |
+
customarily used for software interchange.
|
| 244 |
+
|
| 245 |
+
b) Convey the object code in, or embodied in, a physical product
|
| 246 |
+
(including a physical distribution medium), accompanied by a
|
| 247 |
+
written offer, valid for at least three years and valid for as
|
| 248 |
+
long as you offer spare parts or customer support for that product
|
| 249 |
+
model, to give anyone who possesses the object code either (1) a
|
| 250 |
+
copy of the Corresponding Source for all the software in the
|
| 251 |
+
product that is covered by this License, on a durable physical
|
| 252 |
+
medium customarily used for software interchange, for a price no
|
| 253 |
+
more than your reasonable cost of physically performing this
|
| 254 |
+
conveying of source, or (2) access to copy the
|
| 255 |
+
Corresponding Source from a network server at no charge.
|
| 256 |
+
|
| 257 |
+
c) Convey individual copies of the object code with a copy of the
|
| 258 |
+
written offer to provide the Corresponding Source. This
|
| 259 |
+
alternative is allowed only occasionally and noncommercially, and
|
| 260 |
+
only if you received the object code with such an offer, in accord
|
| 261 |
+
with subsection 6b.
|
| 262 |
+
|
| 263 |
+
d) Convey the object code by offering access from a designated
|
| 264 |
+
place (gratis or for a charge), and offer equivalent access to the
|
| 265 |
+
Corresponding Source in the same way through the same place at no
|
| 266 |
+
further charge. You need not require recipients to copy the
|
| 267 |
+
Corresponding Source along with the object code. If the place to
|
| 268 |
+
copy the object code is a network server, the Corresponding Source
|
| 269 |
+
may be on a different server (operated by you or a third party)
|
| 270 |
+
that supports equivalent copying facilities, provided you maintain
|
| 271 |
+
clear directions next to the object code saying where to find the
|
| 272 |
+
Corresponding Source. Regardless of what server hosts the
|
| 273 |
+
Corresponding Source, you remain obligated to ensure that it is
|
| 274 |
+
available for as long as needed to satisfy these requirements.
|
| 275 |
+
|
| 276 |
+
e) Convey the object code using peer-to-peer transmission, provided
|
| 277 |
+
you inform other peers where the object code and Corresponding
|
| 278 |
+
Source of the work are being offered to the general public at no
|
| 279 |
+
charge under subsection 6d.
|
| 280 |
+
|
| 281 |
+
A separable portion of the object code, whose source code is excluded
|
| 282 |
+
from the Corresponding Source as a System Library, need not be
|
| 283 |
+
included in conveying the object code work.
|
| 284 |
+
|
| 285 |
+
A "User Product" is either (1) a "consumer product", which means any
|
| 286 |
+
tangible personal property which is normally used for personal, family,
|
| 287 |
+
or household purposes, or (2) anything designed or sold for incorporation
|
| 288 |
+
into a dwelling. In determining whether a product is a consumer product,
|
| 289 |
+
doubtful cases shall be resolved in favor of coverage. For a particular
|
| 290 |
+
product received by a particular user, "normally used" refers to a
|
| 291 |
+
typical or common use of that class of product, regardless of the status
|
| 292 |
+
of the particular user or of the way in which the particular user
|
| 293 |
+
actually uses, or expects or is expected to use, the product. A product
|
| 294 |
+
is a consumer product regardless of whether the product has substantial
|
| 295 |
+
commercial, industrial or non-consumer uses, unless such uses represent
|
| 296 |
+
the only significant mode of use of the product.
|
| 297 |
+
|
| 298 |
+
"Installation Information" for a User Product means any methods,
|
| 299 |
+
procedures, authorization keys, or other information required to install
|
| 300 |
+
and execute modified versions of a covered work in that User Product from
|
| 301 |
+
a modified version of its Corresponding Source. The information must
|
| 302 |
+
suffice to ensure that the continued functioning of the modified object
|
| 303 |
+
code is in no case prevented or interfered with solely because
|
| 304 |
+
modification has been made.
|
| 305 |
+
|
| 306 |
+
If you convey an object code work under this section in, or with, or
|
| 307 |
+
specifically for use in, a User Product, and the conveying occurs as
|
| 308 |
+
part of a transaction in which the right of possession and use of the
|
| 309 |
+
User Product is transferred to the recipient in perpetuity or for a
|
| 310 |
+
fixed term (regardless of how the transaction is characterized), the
|
| 311 |
+
Corresponding Source conveyed under this section must be accompanied
|
| 312 |
+
by the Installation Information. But this requirement does not apply
|
| 313 |
+
if neither you nor any third party retains the ability to install
|
| 314 |
+
modified object code on the User Product (for example, the work has
|
| 315 |
+
been installed in ROM).
|
| 316 |
+
|
| 317 |
+
The requirement to provide Installation Information does not include a
|
| 318 |
+
requirement to continue to provide support service, warranty, or updates
|
| 319 |
+
for a work that has been modified or installed by the recipient, or for
|
| 320 |
+
the User Product in which it has been modified or installed. Access to a
|
| 321 |
+
network may be denied when the modification itself materially and
|
| 322 |
+
adversely affects the operation of the network or violates the rules and
|
| 323 |
+
protocols for communication across the network.
|
| 324 |
+
|
| 325 |
+
Corresponding Source conveyed, and Installation Information provided,
|
| 326 |
+
in accord with this section must be in a format that is publicly
|
| 327 |
+
documented (and with an implementation available to the public in
|
| 328 |
+
source code form), and must require no special password or key for
|
| 329 |
+
unpacking, reading or copying.
|
| 330 |
+
|
| 331 |
+
7. Additional Terms.
|
| 332 |
+
|
| 333 |
+
"Additional permissions" are terms that supplement the terms of this
|
| 334 |
+
License by making exceptions from one or more of its conditions.
|
| 335 |
+
Additional permissions that are applicable to the entire Program shall
|
| 336 |
+
be treated as though they were included in this License, to the extent
|
| 337 |
+
that they are valid under applicable law. If additional permissions
|
| 338 |
+
apply only to part of the Program, that part may be used separately
|
| 339 |
+
under those permissions, but the entire Program remains governed by
|
| 340 |
+
this License without regard to the additional permissions.
|
| 341 |
+
|
| 342 |
+
When you convey a copy of a covered work, you may at your option
|
| 343 |
+
remove any additional permissions from that copy, or from any part of
|
| 344 |
+
it. (Additional permissions may be written to require their own
|
| 345 |
+
removal in certain cases when you modify the work.) You may place
|
| 346 |
+
additional permissions on material, added by you to a covered work,
|
| 347 |
+
for which you have or can give appropriate copyright permission.
|
| 348 |
+
|
| 349 |
+
Notwithstanding any other provision of this License, for material you
|
| 350 |
+
add to a covered work, you may (if authorized by the copyright holders of
|
| 351 |
+
that material) supplement the terms of this License with terms:
|
| 352 |
+
|
| 353 |
+
a) Disclaiming warranty or limiting liability differently from the
|
| 354 |
+
terms of sections 15 and 16 of this License; or
|
| 355 |
+
|
| 356 |
+
b) Requiring preservation of specified reasonable legal notices or
|
| 357 |
+
author attributions in that material or in the Appropriate Legal
|
| 358 |
+
Notices displayed by works containing it; or
|
| 359 |
+
|
| 360 |
+
c) Prohibiting misrepresentation of the origin of that material, or
|
| 361 |
+
requiring that modified versions of such material be marked in
|
| 362 |
+
reasonable ways as different from the original version; or
|
| 363 |
+
|
| 364 |
+
d) Limiting the use for publicity purposes of names of licensors or
|
| 365 |
+
authors of the material; or
|
| 366 |
+
|
| 367 |
+
e) Declining to grant rights under trademark law for use of some
|
| 368 |
+
trade names, trademarks, or service marks; or
|
| 369 |
+
|
| 370 |
+
f) Requiring indemnification of licensors and authors of that
|
| 371 |
+
material by anyone who conveys the material (or modified versions of
|
| 372 |
+
it) with contractual assumptions of liability to the recipient, for
|
| 373 |
+
any liability that these contractual assumptions directly impose on
|
| 374 |
+
those licensors and authors.
|
| 375 |
+
|
| 376 |
+
All other non-permissive additional terms are considered "further
|
| 377 |
+
restrictions" within the meaning of section 10. If the Program as you
|
| 378 |
+
received it, or any part of it, contains a notice stating that it is
|
| 379 |
+
governed by this License along with a term that is a further
|
| 380 |
+
restriction, you may remove that term. If a license document contains
|
| 381 |
+
a further restriction but permits relicensing or conveying under this
|
| 382 |
+
License, you may add to a covered work material governed by the terms
|
| 383 |
+
of that license document, provided that the further restriction does
|
| 384 |
+
not survive such relicensing or conveying.
|
| 385 |
+
|
| 386 |
+
If you add terms to a covered work in accord with this section, you
|
| 387 |
+
must place, in the relevant source files, a statement of the
|
| 388 |
+
additional terms that apply to those files, or a notice indicating
|
| 389 |
+
where to find the applicable terms.
|
| 390 |
+
|
| 391 |
+
Additional terms, permissive or non-permissive, may be stated in the
|
| 392 |
+
form of a separately written license, or stated as exceptions;
|
| 393 |
+
the above requirements apply either way.
|
| 394 |
+
|
| 395 |
+
8. Termination.
|
| 396 |
+
|
| 397 |
+
You may not propagate or modify a covered work except as expressly
|
| 398 |
+
provided under this License. Any attempt otherwise to propagate or
|
| 399 |
+
modify it is void, and will automatically terminate your rights under
|
| 400 |
+
this License (including any patent licenses granted under the third
|
| 401 |
+
paragraph of section 11).
|
| 402 |
+
|
| 403 |
+
However, if you cease all violation of this License, then your
|
| 404 |
+
license from a particular copyright holder is reinstated (a)
|
| 405 |
+
provisionally, unless and until the copyright holder explicitly and
|
| 406 |
+
finally terminates your license, and (b) permanently, if the copyright
|
| 407 |
+
holder fails to notify you of the violation by some reasonable means
|
| 408 |
+
prior to 60 days after the cessation.
|
| 409 |
+
|
| 410 |
+
Moreover, your license from a particular copyright holder is
|
| 411 |
+
reinstated permanently if the copyright holder notifies you of the
|
| 412 |
+
violation by some reasonable means, this is the first time you have
|
| 413 |
+
received notice of violation of this License (for any work) from that
|
| 414 |
+
copyright holder, and you cure the violation prior to 30 days after
|
| 415 |
+
your receipt of the notice.
|
| 416 |
+
|
| 417 |
+
Termination of your rights under this section does not terminate the
|
| 418 |
+
licenses of parties who have received copies or rights from you under
|
| 419 |
+
this License. If your rights have been terminated and not permanently
|
| 420 |
+
reinstated, you do not qualify to receive new licenses for the same
|
| 421 |
+
material under section 10.
|
| 422 |
+
|
| 423 |
+
9. Acceptance Not Required for Having Copies.
|
| 424 |
+
|
| 425 |
+
You are not required to accept this License in order to receive or
|
| 426 |
+
run a copy of the Program. Ancillary propagation of a covered work
|
| 427 |
+
occurring solely as a consequence of using peer-to-peer transmission
|
| 428 |
+
to receive a copy likewise does not require acceptance. However,
|
| 429 |
+
nothing other than this License grants you permission to propagate or
|
| 430 |
+
modify any covered work. These actions infringe copyright if you do
|
| 431 |
+
not accept this License. Therefore, by modifying or propagating a
|
| 432 |
+
covered work, you indicate your acceptance of this License to do so.
|
| 433 |
+
|
| 434 |
+
10. Automatic Licensing of Downstream Recipients.
|
| 435 |
+
|
| 436 |
+
Each time you convey a covered work, the recipient automatically
|
| 437 |
+
receives a license from the original licensors, to run, modify and
|
| 438 |
+
propagate that work, subject to this License. You are not responsible
|
| 439 |
+
for enforcing compliance by third parties with this License.
|
| 440 |
+
|
| 441 |
+
An "entity transaction" is a transaction transferring control of an
|
| 442 |
+
organization, or substantially all assets of one, or subdividing an
|
| 443 |
+
organization, or merging organizations. If propagation of a covered
|
| 444 |
+
work results from an entity transaction, each party to that
|
| 445 |
+
transaction who receives a copy of the work also receives whatever
|
| 446 |
+
licenses to the work the party's predecessor in interest had or could
|
| 447 |
+
give under the previous paragraph, plus a right to possession of the
|
| 448 |
+
Corresponding Source of the work from the predecessor in interest, if
|
| 449 |
+
the predecessor has it or can get it with reasonable efforts.
|
| 450 |
+
|
| 451 |
+
You may not impose any further restrictions on the exercise of the
|
| 452 |
+
rights granted or affirmed under this License. For example, you may
|
| 453 |
+
not impose a license fee, royalty, or other charge for exercise of
|
| 454 |
+
rights granted under this License, and you may not initiate litigation
|
| 455 |
+
(including a cross-claim or counterclaim in a lawsuit) alleging that
|
| 456 |
+
any patent claim is infringed by making, using, selling, offering for
|
| 457 |
+
sale, or importing the Program or any portion of it.
|
| 458 |
+
|
| 459 |
+
11. Patents.
|
| 460 |
+
|
| 461 |
+
A "contributor" is a copyright holder who authorizes use under this
|
| 462 |
+
License of the Program or a work on which the Program is based. The
|
| 463 |
+
work thus licensed is called the contributor's "contributor version".
|
| 464 |
+
|
| 465 |
+
A contributor's "essential patent claims" are all patent claims
|
| 466 |
+
owned or controlled by the contributor, whether already acquired or
|
| 467 |
+
hereafter acquired, that would be infringed by some manner, permitted
|
| 468 |
+
by this License, of making, using, or selling its contributor version,
|
| 469 |
+
but do not include claims that would be infringed only as a
|
| 470 |
+
consequence of further modification of the contributor version. For
|
| 471 |
+
purposes of this definition, "control" includes the right to grant
|
| 472 |
+
patent sublicenses in a manner consistent with the requirements of
|
| 473 |
+
this License.
|
| 474 |
+
|
| 475 |
+
Each contributor grants you a non-exclusive, worldwide, royalty-free
|
| 476 |
+
patent license under the contributor's essential patent claims, to
|
| 477 |
+
make, use, sell, offer for sale, import and otherwise run, modify and
|
| 478 |
+
propagate the contents of its contributor version.
|
| 479 |
+
|
| 480 |
+
In the following three paragraphs, a "patent license" is any express
|
| 481 |
+
agreement or commitment, however denominated, not to enforce a patent
|
| 482 |
+
(such as an express permission to practice a patent or covenant not to
|
| 483 |
+
sue for patent infringement). To "grant" such a patent license to a
|
| 484 |
+
party means to make such an agreement or commitment not to enforce a
|
| 485 |
+
patent against the party.
|
| 486 |
+
|
| 487 |
+
If you convey a covered work, knowingly relying on a patent license,
|
| 488 |
+
and the Corresponding Source of the work is not available for anyone
|
| 489 |
+
to copy, free of charge and under the terms of this License, through a
|
| 490 |
+
publicly available network server or other readily accessible means,
|
| 491 |
+
then you must either (1) cause the Corresponding Source to be so
|
| 492 |
+
available, or (2) arrange to deprive yourself of the benefit of the
|
| 493 |
+
patent license for this particular work, or (3) arrange, in a manner
|
| 494 |
+
consistent with the requirements of this License, to extend the patent
|
| 495 |
+
license to downstream recipients. "Knowingly relying" means you have
|
| 496 |
+
actual knowledge that, but for the patent license, your conveying the
|
| 497 |
+
covered work in a country, or your recipient's use of the covered work
|
| 498 |
+
in a country, would infringe one or more identifiable patents in that
|
| 499 |
+
country that you have reason to believe are valid.
|
| 500 |
+
|
| 501 |
+
If, pursuant to or in connection with a single transaction or
|
| 502 |
+
arrangement, you convey, or propagate by procuring conveyance of, a
|
| 503 |
+
covered work, and grant a patent license to some of the parties
|
| 504 |
+
receiving the covered work authorizing them to use, propagate, modify
|
| 505 |
+
or convey a specific copy of the covered work, then the patent license
|
| 506 |
+
you grant is automatically extended to all recipients of the covered
|
| 507 |
+
work and works based on it.
|
| 508 |
+
|
| 509 |
+
A patent license is "discriminatory" if it does not include within
|
| 510 |
+
the scope of its coverage, prohibits the exercise of, or is
|
| 511 |
+
conditioned on the non-exercise of one or more of the rights that are
|
| 512 |
+
specifically granted under this License. You may not convey a covered
|
| 513 |
+
work if you are a party to an arrangement with a third party that is
|
| 514 |
+
in the business of distributing software, under which you make payment
|
| 515 |
+
to the third party based on the extent of your activity of conveying
|
| 516 |
+
the work, and under which the third party grants, to any of the
|
| 517 |
+
parties who would receive the covered work from you, a discriminatory
|
| 518 |
+
patent license (a) in connection with copies of the covered work
|
| 519 |
+
conveyed by you (or copies made from those copies), or (b) primarily
|
| 520 |
+
for and in connection with specific products or compilations that
|
| 521 |
+
contain the covered work, unless you entered into that arrangement,
|
| 522 |
+
or that patent license was granted, prior to 28 March 2007.
|
| 523 |
+
|
| 524 |
+
Nothing in this License shall be construed as excluding or limiting
|
| 525 |
+
any implied license or other defenses to infringement that may
|
| 526 |
+
otherwise be available to you under applicable patent law.
|
| 527 |
+
|
| 528 |
+
12. No Surrender of Others' Freedom.
|
| 529 |
+
|
| 530 |
+
If conditions are imposed on you (whether by court order, agreement or
|
| 531 |
+
otherwise) that contradict the conditions of this License, they do not
|
| 532 |
+
excuse you from the conditions of this License. If you cannot convey a
|
| 533 |
+
covered work so as to satisfy simultaneously your obligations under this
|
| 534 |
+
License and any other pertinent obligations, then as a consequence you may
|
| 535 |
+
not convey it at all. For example, if you agree to terms that obligate you
|
| 536 |
+
to collect a royalty for further conveying from those to whom you convey
|
| 537 |
+
the Program, the only way you could satisfy both those terms and this
|
| 538 |
+
License would be to refrain entirely from conveying the Program.
|
| 539 |
+
|
| 540 |
+
13. Remote Network Interaction; Use with the GNU General Public License.
|
| 541 |
+
|
| 542 |
+
Notwithstanding any other provision of this License, if you modify the
|
| 543 |
+
Program, your modified version must prominently offer all users
|
| 544 |
+
interacting with it remotely through a computer network (if your version
|
| 545 |
+
supports such interaction) an opportunity to receive the Corresponding
|
| 546 |
+
Source of your version by providing access to the Corresponding Source
|
| 547 |
+
from a network server at no charge, through some standard or customary
|
| 548 |
+
means of facilitating copying of software. This Corresponding Source
|
| 549 |
+
shall include the Corresponding Source for any work covered by version 3
|
| 550 |
+
of the GNU General Public License that is incorporated pursuant to the
|
| 551 |
+
following paragraph.
|
| 552 |
+
|
| 553 |
+
Notwithstanding any other provision of this License, you have
|
| 554 |
+
permission to link or combine any covered work with a work licensed
|
| 555 |
+
under version 3 of the GNU General Public License into a single
|
| 556 |
+
combined work, and to convey the resulting work. The terms of this
|
| 557 |
+
License will continue to apply to the part which is the covered work,
|
| 558 |
+
but the work with which it is combined will remain governed by version
|
| 559 |
+
3 of the GNU General Public License.
|
| 560 |
+
|
| 561 |
+
14. Revised Versions of this License.
|
| 562 |
+
|
| 563 |
+
The Free Software Foundation may publish revised and/or new versions of
|
| 564 |
+
the GNU Affero General Public License from time to time. Such new versions
|
| 565 |
+
will be similar in spirit to the present version, but may differ in detail to
|
| 566 |
+
address new problems or concerns.
|
| 567 |
+
|
| 568 |
+
Each version is given a distinguishing version number. If the
|
| 569 |
+
Program specifies that a certain numbered version of the GNU Affero General
|
| 570 |
+
Public License "or any later version" applies to it, you have the
|
| 571 |
+
option of following the terms and conditions either of that numbered
|
| 572 |
+
version or of any later version published by the Free Software
|
| 573 |
+
Foundation. If the Program does not specify a version number of the
|
| 574 |
+
GNU Affero General Public License, you may choose any version ever published
|
| 575 |
+
by the Free Software Foundation.
|
| 576 |
+
|
| 577 |
+
If the Program specifies that a proxy can decide which future
|
| 578 |
+
versions of the GNU Affero General Public License can be used, that proxy's
|
| 579 |
+
public statement of acceptance of a version permanently authorizes you
|
| 580 |
+
to choose that version for the Program.
|
| 581 |
+
|
| 582 |
+
Later license versions may give you additional or different
|
| 583 |
+
permissions. However, no additional obligations are imposed on any
|
| 584 |
+
author or copyright holder as a result of your choosing to follow a
|
| 585 |
+
later version.
|
| 586 |
+
|
| 587 |
+
15. Disclaimer of Warranty.
|
| 588 |
+
|
| 589 |
+
THERE IS NO WARRANTY FOR THE PROGRAM, TO THE EXTENT PERMITTED BY
|
| 590 |
+
APPLICABLE LAW. EXCEPT WHEN OTHERWISE STATED IN WRITING THE COPYRIGHT
|
| 591 |
+
HOLDERS AND/OR OTHER PARTIES PROVIDE THE PROGRAM "AS IS" WITHOUT WARRANTY
|
| 592 |
+
OF ANY KIND, EITHER EXPRESSED OR IMPLIED, INCLUDING, BUT NOT LIMITED TO,
|
| 593 |
+
THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
| 594 |
+
PURPOSE. THE ENTIRE RISK AS TO THE QUALITY AND PERFORMANCE OF THE PROGRAM
|
| 595 |
+
IS WITH YOU. SHOULD THE PROGRAM PROVE DEFECTIVE, YOU ASSUME THE COST OF
|
| 596 |
+
ALL NECESSARY SERVICING, REPAIR OR CORRECTION.
|
| 597 |
+
|
| 598 |
+
16. Limitation of Liability.
|
| 599 |
+
|
| 600 |
+
IN NO EVENT UNLESS REQUIRED BY APPLICABLE LAW OR AGREED TO IN WRITING
|
| 601 |
+
WILL ANY COPYRIGHT HOLDER, OR ANY OTHER PARTY WHO MODIFIES AND/OR CONVEYS
|
| 602 |
+
THE PROGRAM AS PERMITTED ABOVE, BE LIABLE TO YOU FOR DAMAGES, INCLUDING ANY
|
| 603 |
+
GENERAL, SPECIAL, INCIDENTAL OR CONSEQUENTIAL DAMAGES ARISING OUT OF THE
|
| 604 |
+
USE OR INABILITY TO USE THE PROGRAM (INCLUDING BUT NOT LIMITED TO LOSS OF
|
| 605 |
+
DATA OR DATA BEING RENDERED INACCURATE OR LOSSES SUSTAINED BY YOU OR THIRD
|
| 606 |
+
PARTIES OR A FAILURE OF THE PROGRAM TO OPERATE WITH ANY OTHER PROGRAMS),
|
| 607 |
+
EVEN IF SUCH HOLDER OR OTHER PARTY HAS BEEN ADVISED OF THE POSSIBILITY OF
|
| 608 |
+
SUCH DAMAGES.
|
| 609 |
+
|
| 610 |
+
17. Interpretation of Sections 15 and 16.
|
| 611 |
+
|
| 612 |
+
If the disclaimer of warranty and limitation of liability provided
|
| 613 |
+
above cannot be given local legal effect according to their terms,
|
| 614 |
+
reviewing courts shall apply local law that most closely approximates
|
| 615 |
+
an absolute waiver of all civil liability in connection with the
|
| 616 |
+
Program, unless a warranty or assumption of liability accompanies a
|
| 617 |
+
copy of the Program in return for a fee.
|
| 618 |
+
|
| 619 |
+
END OF TERMS AND CONDITIONS
|
| 620 |
+
|
| 621 |
+
How to Apply These Terms to Your New Programs
|
| 622 |
+
|
| 623 |
+
If you develop a new program, and you want it to be of the greatest
|
| 624 |
+
possible use to the public, the best way to achieve this is to make it
|
| 625 |
+
free software which everyone can redistribute and change under these terms.
|
| 626 |
+
|
| 627 |
+
To do so, attach the following notices to the program. It is safest
|
| 628 |
+
to attach them to the start of each source file to most effectively
|
| 629 |
+
state the exclusion of warranty; and each file should have at least
|
| 630 |
+
the "copyright" line and a pointer to where the full notice is found.
|
| 631 |
+
|
| 632 |
+
<one line to give the program's name and a brief idea of what it does.>
|
| 633 |
+
Copyright (C) <year> <name of author>
|
| 634 |
+
|
| 635 |
+
This program is free software: you can redistribute it and/or modify
|
| 636 |
+
it under the terms of the GNU Affero General Public License as published by
|
| 637 |
+
the Free Software Foundation, either version 3 of the License, or
|
| 638 |
+
(at your option) any later version.
|
| 639 |
+
|
| 640 |
+
This program is distributed in the hope that it will be useful,
|
| 641 |
+
but WITHOUT ANY WARRANTY; without even the implied warranty of
|
| 642 |
+
MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
| 643 |
+
GNU Affero General Public License for more details.
|
| 644 |
+
|
| 645 |
+
You should have received a copy of the GNU Affero General Public License
|
| 646 |
+
along with this program. If not, see <https://www.gnu.org/licenses/>.
|
| 647 |
+
|
| 648 |
+
Also add information on how to contact you by electronic and paper mail.
|
| 649 |
+
|
| 650 |
+
If your software can interact with users remotely through a computer
|
| 651 |
+
network, you should also make sure that it provides a way for users to
|
| 652 |
+
get its source. For example, if your program is a web application, its
|
| 653 |
+
interface could display a "Source" link that leads users to an archive
|
| 654 |
+
of the code. There are many ways you could offer source, and different
|
| 655 |
+
solutions will be better for different programs; see section 13 for the
|
| 656 |
+
specific requirements.
|
| 657 |
+
|
| 658 |
+
You should also get your employer (if you work as a programmer) or school,
|
| 659 |
+
if any, to sign a "copyright disclaimer" for the program, if necessary.
|
| 660 |
+
For more information on this, and how to apply and follow the GNU AGPL, see
|
| 661 |
+
<https://www.gnu.org/licenses/>.
|
|
@@ -1,3 +1,813 @@
|
|
| 1 |
---
|
| 2 |
license: agpl-3.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
---
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
license: agpl-3.0
|
| 3 |
+
pipeline_tag: reinforcement-learning
|
| 4 |
+
tags:
|
| 5 |
+
- maze
|
| 6 |
+
- snake
|
| 7 |
+
- planning
|
| 8 |
+
- grid-world
|
| 9 |
+
- value-iteration-network
|
| 10 |
+
- decision-model
|
| 11 |
+
- pytorch
|
| 12 |
+
metrics:
|
| 13 |
+
- accuracy
|
| 14 |
---
|
| 15 |
+
|
| 16 |
+
# Jevon
|
| 17 |
+
|
| 18 |
+
A 20M-parameter decision model that answers **typed questions about a grid** —
|
| 19 |
+
"which way should I move?", "is north clear?", "how far is the goal?" — with
|
| 20 |
+
calibrated probabilities and no token decoding. It is a from-scratch
|
| 21 |
+
reimplementation of the idea behind [NanoJev][nanojev] (itself a small
|
| 22 |
+
open replication of TypeSafe AI's [Jev][jev]), rebuilt around the thing that
|
| 23 |
+
made the original fail at Maze: **planning depth**.
|
| 24 |
+
|
| 25 |
+
Jevon keeps NanoJev's interface — the three Jev primitives, `Choice`, `Score`
|
| 26 |
+
and `Boolean` — and replaces the backbone.
|
| 27 |
+
|
| 28 |
+
[nanojev]: https://github.com/TianyuCodings/NanoJev
|
| 29 |
+
[jev]: https://typesafe.ai
|
| 30 |
+
|
| 31 |
+
```python
|
| 32 |
+
from jevon import from_pretrained
|
| 33 |
+
|
| 34 |
+
jevon = from_pretrained("lewislululu/jevon") # config + tokenizer + balanced.pt
|
| 35 |
+
```
|
| 36 |
+
|
| 37 |
+
| | |
|
| 38 |
+
| --- | --- |
|
| 39 |
+
| Parameters | 20,105,047 |
|
| 40 |
+
| Checkpoint | `runs/jevon-final/balanced.pt` — 80MB of fp32 weights, via Git LFS |
|
| 41 |
+
| Games | Maze (4 topologies, sizes 11–51) and Snake (8–24) |
|
| 42 |
+
| Question types | `choice`, `boolean`, `score` — the three Jev primitives |
|
| 43 |
+
| Training | 9,000 steps on one Apple M4 Max, from scratch |
|
| 44 |
+
| Watch it play | [jevon-arcade](https://github.com/lewislulu/jevon-arcade) |
|
| 45 |
+
| Licence | [AGPL-3.0-or-later](LICENSE), or a [commercial licence](COMMERCIAL-LICENSE.md) |
|
| 46 |
+
|
| 47 |
+
Weights, code, frozen evaluation splits and every number below are in this one
|
| 48 |
+
repository, and the tables are generated from the JSON beside them rather than
|
| 49 |
+
typed — `pytest -q` fails if this page and that JSON disagree.
|
| 50 |
+
|
| 51 |
+
## Results at a glance
|
| 52 |
+
|
| 53 |
+
<!-- BEGIN:headline (generated by scripts/readme_table.py) -->
|
| 54 |
+
|
| 55 |
+
**Decision accuracy** (held-out `test` split, next to the uniform control). `boards` is the number of distinct boards behind those questions; it tracks `n` because the generator draws a fresh board per state, which is what makes `n` an honest denominator rather than an assumed one:
|
| 56 |
+
|
| 57 |
+
| question | n | boards | accuracy | uniform control |
|
| 58 |
+
| --- | ---: | ---: | ---: | ---: |
|
| 59 |
+
| `maze/choice` | 141 | 140 | **1.0000** | 0.4882 |
|
| 60 |
+
| `snake/choice` | 92 | 92 | **0.9674** | 0.4783 |
|
| 61 |
+
|
| 62 |
+
**Maze**, 36 episodes across topologies and sizes:
|
| 63 |
+
|
| 64 |
+
| controller | solve rate | efficiency | cells visited |
|
| 65 |
+
| --- | ---: | ---: | ---: |
|
| 66 |
+
| `model` (network alone) | **1.00** | 1.000 | 94.2 |
|
| 67 |
+
| `model-field` (architectural) | **1.00** | 1.000 | 94.2 |
|
| 68 |
+
| `random` (floor) | 0.14 | 0.068 | 92.2 |
|
| 69 |
+
| `reference` (BFS) | 1.00 | 1.000 | 94.2 |
|
| 70 |
+
|
| 71 |
+
**Snake**, 6 episodes:
|
| 72 |
+
|
| 73 |
+
| controller | mean food | max food | survival |
|
| 74 |
+
| --- | ---: | ---: | ---: |
|
| 75 |
+
| `model` | **4.50** | 16 | 1.00 |
|
| 76 |
+
| `random` (floor) | 0.67 | 2 | 0.00 |
|
| 77 |
+
| `reference` (oracle) | 23.50 | 31 | 1.00 |
|
| 78 |
+
|
| 79 |
+
<!-- END:headline -->
|
| 80 |
+
|
| 81 |
+
Regenerated from the evaluation JSON by `scripts/readme_table.py`, and
|
| 82 |
+
`--check` fails the test suite if this block drifts from it. Full tables,
|
| 83 |
+
including calibration and the planner probe, are in [RESULTS.md](RESULTS.md).
|
| 84 |
+
|
| 85 |
+
---
|
| 86 |
+
|
| 87 |
+
## Using it
|
| 88 |
+
|
| 89 |
+
```bash
|
| 90 |
+
pip install torch huggingface_hub
|
| 91 |
+
git clone https://github.com/lewislulu/jevon-arcade # or this repo, for the code
|
| 92 |
+
```
|
| 93 |
+
|
| 94 |
+
`from_pretrained` fetches the config, the tokenizer and one checkpoint — about
|
| 95 |
+
80MB, not the whole repository — and caches them under `~/.cache/huggingface`.
|
| 96 |
+
It takes a local run directory just as happily, so the same line works before
|
| 97 |
+
and after publishing:
|
| 98 |
+
|
| 99 |
+
```python
|
| 100 |
+
from jevon import from_pretrained
|
| 101 |
+
from jevon.inference import maze_state_sample
|
| 102 |
+
from envs import maze as mz
|
| 103 |
+
|
| 104 |
+
jevon = from_pretrained("lewislululu/jevon") # or "runs/jevon-final"
|
| 105 |
+
|
| 106 |
+
state = mz.generate_maze(21, "tree", seed=7)
|
| 107 |
+
state.step("south") # the start cell has one exit, so it poses no choice
|
| 108 |
+
|
| 109 |
+
answers = {a["question"]: a for a in jevon.answer([maze_state_sample(state)])}
|
| 110 |
+
answers["action"]["probabilities"]
|
| 111 |
+
# {'north': 0.002, 'south': 0.998} BFS-optimal: south
|
| 112 |
+
answers["clear_west"]["probabilities"]
|
| 113 |
+
# {'false': 0.999, 'true': 0.001} a wall
|
| 114 |
+
answers["solvable"]["probabilities"]
|
| 115 |
+
# {'false': 0.001, 'true': 0.999}
|
| 116 |
+
```
|
| 117 |
+
|
| 118 |
+
One pass answers **every** question the state offers — `action`, the four
|
| 119 |
+
`clear_*` booleans, `distance` and `solvable` — because the state and the
|
| 120 |
+
board are encoded once and the questions cross-attend to them. "Offers" is
|
| 121 |
+
literal: a cell with one exit poses no choice, so no `action` question is
|
| 122 |
+
asked there and none is answered. Nothing is
|
| 123 |
+
decoded as text: each answer is a distribution over that question's own
|
| 124 |
+
candidates, which is what makes the numbers above comparable across states.
|
| 125 |
+
|
| 126 |
+
The planner's distance field is readable directly, and costs nothing extra
|
| 127 |
+
during an episode because it does not depend on the agent:
|
| 128 |
+
|
| 129 |
+
```python
|
| 130 |
+
field = jevon.distance_field(maze_state_sample(state)) # (21, 21), lower is closer
|
| 131 |
+
```
|
| 132 |
+
|
| 133 |
+
On a maze that field is a shortest-path distance up to one global scale — on
|
| 134 |
+
the board above, rescaled by a single constant it matches BFS to within
|
| 135 |
+
**0.0096 cells** — which is why greedy descent on it arrives from anywhere.
|
| 136 |
+
[Building the property in](#building-the-property-in) is the part to read
|
| 137 |
+
before treating that as a result about training; it is a property of the
|
| 138 |
+
recurrence, and the honest apples-to-apples number is the `model` row.
|
| 139 |
+
|
| 140 |
+
Device: `cuda`, then `mps`, then CPU, unless you pass `device=`.
|
| 141 |
+
|
| 142 |
+
To watch it play rather than query it, [jevon-arcade][arcade] runs the maze and
|
| 143 |
+
snake boards in a browser while this model decides, one forward pass per step,
|
| 144 |
+
through the generators in `scripts/play.py` that produced every number above.
|
| 145 |
+
|
| 146 |
+
[arcade]: https://github.com/lewislulu/jevon-arcade
|
| 147 |
+
|
| 148 |
+
## Intended use, and what this is not
|
| 149 |
+
|
| 150 |
+
Jevon is a **research artefact**: a demonstration that a 20M-parameter model
|
| 151 |
+
with the right structural prior solves grid planning that a language-model
|
| 152 |
+
backbone of similar size does not. It is useful for studying amortised value
|
| 153 |
+
iteration, typed decision heads, and for reproducing or disputing the
|
| 154 |
+
comparison with NanoJev below.
|
| 155 |
+
|
| 156 |
+
It is not a general agent and not a language model. It answers a fixed set of
|
| 157 |
+
typed questions about Maze and Snake boards; it has no text output, no
|
| 158 |
+
instruction following, and nothing it learned transfers off a grid.
|
| 159 |
+
|
| 160 |
+
Three limits worth knowing before quoting a number:
|
| 161 |
+
|
| 162 |
+
1. **The maze field is exact by construction, not by training.** A checkpoint
|
| 163 |
+
trained for 60 steps already solves every maze optimally. Any
|
| 164 |
+
`model-field` maze figure describes the architecture. See [Building the
|
| 165 |
+
property in](#building-the-property-in).
|
| 166 |
+
2. **Snake is the weak game.** 4.50 mean food against its own oracle's 23.50.
|
| 167 |
+
There is no planner for Snake the way there is for Maze, because distance
|
| 168 |
+
to food is the wrong field to descend — survival dominates.
|
| 169 |
+
3. **The splits are small.** 141 maze and 92 snake choice questions in `test`.
|
| 170 |
+
Every accuracy here is printed next to its constant-prediction control for
|
| 171 |
+
that reason, and [RESULTS.md](RESULTS.md) gives `n` and the distinct board
|
| 172 |
+
count for each.
|
| 173 |
+
|
| 174 |
+
Trained entirely on synthetic boards from the generators in `envs/`, whose
|
| 175 |
+
oracles are exact. There is no human data anywhere in it.
|
| 176 |
+
|
| 177 |
+
---
|
| 178 |
+
|
| 179 |
+
## Why rebuild it
|
| 180 |
+
|
| 181 |
+
NanoJev's own development log reports that its maze model learned nothing:
|
| 182 |
+
its test-maze atomic accuracy (`0.5625`) is *identical* to its always-true
|
| 183 |
+
control (`0.5625`), and every trained arm solved **0 of 3** mazes within the
|
| 184 |
+
step cap. The impressive "244 attempts" demo comes from the surrounding
|
| 185 |
+
controller — collision memory, untried-edge exploration, repositioning — not
|
| 186 |
+
from the network.
|
| 187 |
+
|
| 188 |
+
Three things caused that, and Jevon fixes each one.
|
| 189 |
+
|
| 190 |
+
### 1. The board was a string
|
| 191 |
+
|
| 192 |
+
NanoJev feeds the maze to a language model as ASCII art. In a flattened token
|
| 193 |
+
sequence, two vertically adjacent cells of a 50×50 maze are ~51 tokens apart,
|
| 194 |
+
so the model has to learn 2-D adjacency through attention, from scratch, for
|
| 195 |
+
every board size. Jevon reads the board as an **8-channel grid**
|
| 196 |
+
(`free, blocked, focus, target, body, tail, decay, game`) shared by both games,
|
| 197 |
+
so one set of spatial weights serves Maze and Snake alike.
|
| 198 |
+
|
| 199 |
+
### 2. There was no iterative computation
|
| 200 |
+
|
| 201 |
+
Shortest-path distance is not a local function. A 21×21 corridor maze has a
|
| 202 |
+
shortest path of 154 cells; a 31×31 has 312. A fixed-depth transformer cannot
|
| 203 |
+
propagate a distance signal that far in one pass, no matter how wide it is.
|
| 204 |
+
|
| 205 |
+
Jevon adds a **`SpatialPlanner`**: a weight-shared gated message-passing block
|
| 206 |
+
applied recurrently over the 4-neighbourhood, masked by passability. `T`
|
| 207 |
+
iterations propagate information `T` cells — amortised value iteration, or
|
| 208 |
+
Bellman–Ford with learned operators. The budget is sized to the board:
|
| 209 |
+
|
| 210 |
+
```python
|
| 211 |
+
default_iterations(size) = min(4096, max(32, 2 * size + size * size // 2))
|
| 212 |
+
# 11 -> 82 21 -> 262 31 -> 542 51 -> 1402
|
| 213 |
+
```
|
| 214 |
+
|
| 215 |
+
Memory stays flat in `T` because only the last `grad_steps` iterations carry
|
| 216 |
+
gradient; earlier ones run under `no_grad` (implicit/phantom gradient).
|
| 217 |
+
|
| 218 |
+
### 3. One question per state cannot teach a 500-step recurrence
|
| 219 |
+
|
| 220 |
+
This was the subtle one. Even with a correct planner and a correct iteration
|
| 221 |
+
budget, a linear probe from the planner's features to true BFS distance scored
|
| 222 |
+
**R² = 0.21** — the field was barely related to distance. The recurrence was
|
| 223 |
+
getting gradient from a *single* `distance` question per state.
|
| 224 |
+
|
| 225 |
+
So the planner gets its own dense objective: a 1×1 convolution reads every cell
|
| 226 |
+
and predicts the BFS distance to the target (log-squashed) plus a reachability
|
| 227 |
+
logit. That is ~`size²` supervised targets per board instead of one, and it is
|
| 228 |
+
defined purely on the board the planner receives, so the target is exactly the
|
| 229 |
+
quantity the planner has the information to compute.
|
| 230 |
+
|
| 231 |
+
```
|
| 232 |
+
tests/test_model.py::test_planner_can_learn_a_distance_field
|
| 233 |
+
R² -31.9 -> 0.79 (4 boards, 400 steps, 9x9)
|
| 234 |
+
```
|
| 235 |
+
|
| 236 |
+
### 4. A graded scale is not seven unrelated categories
|
| 237 |
+
|
| 238 |
+
`distance` and `room` are Jev **Score** questions: their criteria are an
|
| 239 |
+
ordered scale ("within ten moves", "within twenty-five moves"). Encoded as a
|
| 240 |
+
flat softmax with a one-hot target, naming the neighbouring level costs exactly
|
| 241 |
+
as much as naming the opposite end of the scale, and with seven near-synonymous
|
| 242 |
+
levels the marginal mode becomes the easiest minimum.
|
| 243 |
+
|
| 244 |
+
It did, for a long time. `acc_score` sat on **0.3333** at every evaluation
|
| 245 |
+
through the first 1000 steps of a run — exactly the always-level-4 constant —
|
| 246 |
+
while a ridge probe on the very same planner features, binned through the very
|
| 247 |
+
same thresholds, scored:
|
| 248 |
+
|
| 249 |
+
```
|
| 250 |
+
probe -> level 0.4335 best constant 0.3436
|
| 251 |
+
within one level 0.9076
|
| 252 |
+
```
|
| 253 |
+
|
| 254 |
+
The information was there and the head was not using it, and 91% of the probe's
|
| 255 |
+
errors were *one level off* — precisely the structure a one-hot target
|
| 256 |
+
discards.
|
| 257 |
+
|
| 258 |
+
A matched control run says how much of this is the target and how much is just
|
| 259 |
+
impatience: left alone, the one-hot head does eventually escape, reaching
|
| 260 |
+
0.3722 at step 1200. So the constant is an early-training attractor rather than
|
| 261 |
+
a permanent trap, and the question the ablation answers is whether these two
|
| 262 |
+
changes escape it sooner and end higher, not whether escape is possible at all.
|
| 263 |
+
Both are off by default and measured in `runs/arm-*`:
|
| 264 |
+
|
| 265 |
+
* **Ordinal targets** (`--score-smoothing`): score mass decays with distance
|
| 266 |
+
along the scale. The peak stays on the true level, so every reported accuracy
|
| 267 |
+
stays comparable with runs trained without it.
|
| 268 |
+
* **Exposed field** (`--expose-field`): the text side gathers the field head's
|
| 269 |
+
own distance/reach read-out alongside the raw planner features. The field
|
| 270 |
+
head is already trained on every free cell, so its output is the most
|
| 271 |
+
distilled form of exactly what the question asks; without this the
|
| 272 |
+
transformer re-derives the same map from a far sparser signal.
|
| 273 |
+
|
| 274 |
+
### 5. Two ways of being wrong about the same number
|
| 275 |
+
|
| 276 |
+
`field_r2` appeared to fall while the decision metrics rose, which looks like
|
| 277 |
+
the decision loss dragging the planner off its objective. That produced one
|
| 278 |
+
falsified hypothesis and one real bug, and the order matters: measuring first
|
| 279 |
+
is what separated them.
|
| 280 |
+
|
| 281 |
+
**The hypothesis.** If the decision loss were overwhelming the field loss, the
|
| 282 |
+
planner's gradient would show it. It does not:
|
| 283 |
+
|
| 284 |
+
```
|
| 285 |
+
decision loss 0.5050 grad->planner 0.03195
|
| 286 |
+
field loss 0.0653 grad->planner 0.04808 ratio 0.66 : 1
|
| 287 |
+
```
|
| 288 |
+
|
| 289 |
+
The field objective already dominates, so `--field-weight 1.0` is calibrated
|
| 290 |
+
and the tempting fix — raising it — would have been a change made for a reason
|
| 291 |
+
measurement does not support.
|
| 292 |
+
|
| 293 |
+
**The bug.** `field_r2` was computed per evaluation batch and then averaged.
|
| 294 |
+
R² is a ratio of sums and is not averageable: a batch of boards with little
|
| 295 |
+
distance variance drives its own R² arbitrarily negative and drags the mean
|
| 296 |
+
with it. Pooling the sums and dividing once gives a stable number, and the
|
| 297 |
+
"collapse" disappears. The probe, which pools over every cell, had been
|
| 298 |
+
reporting R² ≈ 0.50 on the same checkpoints the whole time.
|
| 299 |
+
|
| 300 |
+
### 6. R² is not the metric the controller uses
|
| 301 |
+
|
| 302 |
+
Greedy descent never reads an absolute distance. It compares a cell's four
|
| 303 |
+
neighbours and steps to the lowest, so what matters is whether the *ordering*
|
| 304 |
+
of neighbours is right — and a path only succeeds if every comparison along it
|
| 305 |
+
succeeds. `scripts/probe.py` reports that directly:
|
| 306 |
+
|
| 307 |
+
```
|
| 308 |
+
size probe R² field head R² descent
|
| 309 |
+
11 0.5013 0.3425 0.6325
|
| 310 |
+
21 0.2697 0.1801 0.6561
|
| 311 |
+
```
|
| 312 |
+
|
| 313 |
+
Descent accuracy 0.65 means a 21×21 maze needs ~130 consecutive correct
|
| 314 |
+
choices; the measured closed-loop solve rate for `--controller model-field` was
|
| 315 |
+
0, exactly as that predicts.
|
| 316 |
+
|
| 317 |
+
The cause is arithmetic. With a log-squashed target, neighbouring cells differ
|
| 318 |
+
by ~`1/d`, so contrast collapses as distance grows — measured on a trained
|
| 319 |
+
checkpoint:
|
| 320 |
+
|
| 321 |
+
```
|
| 322 |
+
true dist descent acc mean target gap
|
| 323 |
+
0-9 0.7273 0.08279
|
| 324 |
+
10-19 0.6458 0.02730
|
| 325 |
+
20-29 0.6531 0.01680
|
| 326 |
+
30-39 0.7429 0.01191
|
| 327 |
+
40-49 0.5000 0.00980 <- 8.5x less contrast
|
| 328 |
+
```
|
| 329 |
+
|
| 330 |
+
and the same checkpoint's field MAE was **0.107** — larger than the gap it had
|
| 331 |
+
to resolve *anywhere* on the board. Rescaling would not help, since it scales
|
| 332 |
+
the error by the same factor; the shape has to change. `--field-target linear`
|
| 333 |
+
makes every neighbouring pair differ by exactly `1/size²` wherever it sits, so
|
| 334 |
+
precision is spent uniformly instead of being concentrated near the goal.
|
| 335 |
+
|
| 336 |
+
That argument is arithmetic and stands on its own. The *empirical* claim it
|
| 337 |
+
invites — that the linear target measurably improves closed-loop descent — does
|
| 338 |
+
not survive the seed-variance check in section 7, and is not made here.
|
| 339 |
+
|
| 340 |
+
### 7. A field the controller can follow is not the same as an accurate field
|
| 341 |
+
|
| 342 |
+
Section 6 explained a solve rate of 0 with descent accuracy 0.65: a 21×21 maze
|
| 343 |
+
needs ~130 consecutive correct choices, and `0.65¹³⁰` is nothing. That
|
| 344 |
+
reasoning is wrong, and the way it is wrong mattered more than the original
|
| 345 |
+
problem.
|
| 346 |
+
|
| 347 |
+
A wrong step is not a lost episode. It moves the agent to another cell, where
|
| 348 |
+
it descends again. What ends an episode is not error *rate* but error
|
| 349 |
+
*structure*: a spurious local minimum, which descent enters and never leaves.
|
| 350 |
+
So the thing to measure is not how often a step is right but how often a walk
|
| 351 |
+
arrives. From every free cell, follow the field downhill and record where it
|
| 352 |
+
ends up:
|
| 353 |
+
|
| 354 |
+
```
|
| 355 |
+
size T reach cycle descent
|
| 356 |
+
11 41 0.272 0.728 0.729
|
| 357 |
+
11 82 0.293 0.707 0.743
|
| 358 |
+
11 164 0.299 0.701 0.736
|
| 359 |
+
21 131 0.070 0.930 0.623
|
| 360 |
+
21 262 0.072 0.928 0.630
|
| 361 |
+
21 524 0.072 0.928 0.641
|
| 362 |
+
```
|
| 363 |
+
|
| 364 |
+
Two things fall out. 70–93% of cells sit in a basin, so the failure is
|
| 365 |
+
structural rather than statistical — descent accuracy of 0.63 and reach of 0.07
|
| 366 |
+
are not two views of one number. And quadrupling the iteration budget moves
|
| 367 |
+
reach by under three points, so the recurrence has converged: propagation
|
| 368 |
+
depth, the thing this architecture was built to supply, was never what was
|
| 369 |
+
missing.
|
| 370 |
+
|
| 371 |
+
#### The direct fix does not work, and measuring the noise is why we know
|
| 372 |
+
|
| 373 |
+
The obvious response is to supervise the property: for every cell, hinge the
|
| 374 |
+
best true-downhill neighbour below every other neighbour by one true step
|
| 375 |
+
(`descent_loss`). It is exactly 0 on the true field under both target shapes,
|
| 376 |
+
so it cannot fight the regression term, and raising its weight does move reach
|
| 377 |
+
in the right direction — 0.104, 0.122, 0.132, 0.154 at weights 0, 1, 5, 20.
|
| 378 |
+
|
| 379 |
+
That looks like a small win. It is not a win at all. Re-running the *unchanged*
|
| 380 |
+
baseline under three seeds gives:
|
| 381 |
+
|
| 382 |
+
```
|
| 383 |
+
seed reach descent acc
|
| 384 |
+
0 0.1040 0.6700
|
| 385 |
+
1 0.0446 0.6283
|
| 386 |
+
2 0.1478 0.6575
|
| 387 |
+
```
|
| 388 |
+
|
| 389 |
+
A 3.3× spread, with the whole ablation sitting comfortably inside it. The
|
| 390 |
+
honest reading is that the hinge's effect is not resolvable at this sample
|
| 391 |
+
size, and any conclusion drawn from that first table would have been an
|
| 392 |
+
artefact. The term is kept — it is principled and costs nothing at weight 0 —
|
| 393 |
+
but it is not what fixed the problem, and this repository does not claim it
|
| 394 |
+
did.
|
| 395 |
+
|
| 396 |
+
One correction to that table, found later. Each row ran in a fresh process,
|
| 397 |
+
and at the time `generate_maze` seeded itself with `hash(topology)`, which
|
| 398 |
+
Python salts per process — so the rows differ in their held-out boards as
|
| 399 |
+
well as their training seed, and the spread bounds the two together rather
|
| 400 |
+
than the seed alone. That is now fixed (`TOPOLOGIES.index`, pinned by a test
|
| 401 |
+
that runs the generator under three hash salts). It does not rescue the
|
| 402 |
+
ablation: a band measured over *more* sources of variation than intended is
|
| 403 |
+
still a band the ablation sits inside, and the four ablation rows were drawn
|
| 404 |
+
from separate processes too, so they were never a controlled comparison in
|
| 405 |
+
the first place. It does mean the number quoted above is an upper bound on
|
| 406 |
+
seed variance specifically, and the honest summary is narrower than it looks:
|
| 407 |
+
this experiment does not resolve the hinge's effect, and it never could
|
| 408 |
+
have.
|
| 409 |
+
|
| 410 |
+
The same caveat retires a claim section 6 would otherwise support: matched
|
| 411 |
+
conv-head baselines differing only in target shape came out at 0.104 and 0.200,
|
| 412 |
+
which is also within the noise band above.
|
| 413 |
+
|
| 414 |
+
#### Building the property in
|
| 415 |
+
|
| 416 |
+
Global monotonicity is all the local constraints holding at once, so a penalty
|
| 417 |
+
that gets ~70% of them right buys far less than 70% of the benefit. Stop asking
|
| 418 |
+
for the property and construct it. `MinPlusField` predicts a per-cell **cost**
|
| 419 |
+
and reads distance off a Bellman–Ford recurrence:
|
| 420 |
+
|
| 421 |
+
```
|
| 422 |
+
v(c) ← cost(c) + min over passable neighbours n of v(n), v(goal) = 0
|
| 423 |
+
```
|
| 424 |
+
|
| 425 |
+
With strictly positive costs the fixed point *is* a shortest-path distance, so
|
| 426 |
+
every non-goal cell has a strictly lower neighbour and greedy descent
|
| 427 |
+
terminates at the goal from anywhere. Reach is 1.0 by construction rather than
|
| 428 |
+
by training — the test suite asserts it with the cost weights randomised, where
|
| 429 |
+
the field bears no resemblance to the true distance and is still traversable.
|
| 430 |
+
Costs are emitted in units of `1/area`, exactly the step size of the `linear`
|
| 431 |
+
target, so the two agree by design rather than by tuning.
|
| 432 |
+
|
| 433 |
+
```
|
| 434 |
+
descent acc reach field MAE time
|
| 435 |
+
conv 0.7115 0.2001 0.05636 65s
|
| 436 |
+
min-plus 1.0000 1.0000 0.00000 67s
|
| 437 |
+
```
|
| 438 |
+
|
| 439 |
+
The `conv` row carries the noise band established above; the `min-plus` row
|
| 440 |
+
does not, because 1.0000 there is a proof obligation the tests discharge
|
| 441 |
+
rather than a measurement that could have come out otherwise. That asymmetry
|
| 442 |
+
is the whole argument for building the property in instead of training for it.
|
| 443 |
+
|
| 444 |
+
**The zero MAE is the caveat, and it belongs next to the headline.** For a
|
| 445 |
+
maze, uniform cost is exactly right; the head initialises at
|
| 446 |
+
`softplus(0.5413) ≈ 1`; so it computes exact BFS before a single gradient step
|
| 447 |
+
(measured error against true BFS: under `1e-2` cells). The maze field is
|
| 448 |
+
therefore a property of the architecture, not something the run learned, and
|
| 449 |
+
any maze number produced by `--controller model-field` has to be read that way.
|
| 450 |
+
|
| 451 |
+
What stays genuinely learned is the cost map — which is where Snake lives,
|
| 452 |
+
since distance to food is not the whole objective there — and whether handing
|
| 453 |
+
the transformer an exact field improves the action head. That second question
|
| 454 |
+
is the apples-to-apples comparison with NanoJev, and it is the one the
|
| 455 |
+
headline `model` controller reports.
|
| 456 |
+
|
| 457 |
+
Closed-loop, the guarantee survives the whole inference path. A checkpoint
|
| 458 |
+
trained for **60 steps** — long enough to confirm the plumbing works and not
|
| 459 |
+
much else — played with `--controller model-field`:
|
| 460 |
+
|
| 461 |
+
```
|
| 462 |
+
size solve rate mean steps mean shortest collisions
|
| 463 |
+
11 1.000 33.0 33.0 0
|
| 464 |
+
21 1.000 82.0 82.0 0
|
| 465 |
+
31 1.000 172.5 172.5 0
|
| 466 |
+
```
|
| 467 |
+
|
| 468 |
+
Every maze solved, by an exactly optimal path. Read that as a statement about
|
| 469 |
+
the architecture: a 60-step checkpoint has learned nothing, and the number
|
| 470 |
+
comes from the recurrence being breadth-first search. It is reported because
|
| 471 |
+
the comparison it replaces — Jev and NanoJev both solving 0 of 128 — is a
|
| 472 |
+
comparison between controllers, and this is what a controller with the right
|
| 473 |
+
structural prior does on the same boards.
|
| 474 |
+
|
| 475 |
+
This is the Value-Iteration-Network idea (Tamar et al., 2016) with the
|
| 476 |
+
max-plus reward recurrence swapped for the min-plus distance one the field head
|
| 477 |
+
was already supervised on.
|
| 478 |
+
|
| 479 |
+
---
|
| 480 |
+
|
| 481 |
+
## Architecture
|
| 482 |
+
|
| 483 |
+
```
|
| 484 |
+
header + question text ──► shared prefix encoder (6 layers) ─┐
|
| 485 |
+
├─► cross-attention ──► score ──► softmax
|
| 486 |
+
candidate k text ────────► candidate encoder (2 layers) ──────┘ │
|
| 487 |
+
│
|
| 488 |
+
8-channel board ──► SpatialPlanner (recurrent, T steps) ──► field ─────┤
|
| 489 |
+
│ │
|
| 490 |
+
└──► 1x1 conv ──► distance + reach (training only)
|
| 491 |
+
```
|
| 492 |
+
|
| 493 |
+
Three structural differences from NanoJev, beyond the planner:
|
| 494 |
+
|
| 495 |
+
**Shared prefix.** NanoJev runs one full forward pass *per candidate*,
|
| 496 |
+
re-encoding the whole state prefix each time. A 50×50 ASCII maze is ~2,600
|
| 497 |
+
tokens, so a 4-candidate decision costs ~10.4K tokens. Jevon encodes the state
|
| 498 |
+
and question **once** and lets candidates cross-attend to it: cost is
|
| 499 |
+
`|prefix| + K·|candidate|` rather than `K·(|prefix| + |candidate|)`. This is
|
| 500 |
+
the "tree sharing" NanoJev lists as unimplemented.
|
| 501 |
+
|
| 502 |
+
**Anchored candidates.** A candidate that names a cell ("Move north to (7,11)")
|
| 503 |
+
carries that coordinate, and the scoring head gathers the planner's feature at
|
| 504 |
+
exactly that cell. The model does not have to re-derive from text which cell a
|
| 505 |
+
candidate refers to.
|
| 506 |
+
|
| 507 |
+
**Field caching.** The planner never sees the agent — the focus channel is
|
| 508 |
+
zeroed before the recurrence — so for a fixed maze the field is constant for an
|
| 509 |
+
entire episode and is computed **once**. Measured: 10 maze decisions in 0.71 s
|
| 510 |
+
with `planner_calls=1`, `cache_hits=9` (71 ms/decision).
|
| 511 |
+
|
| 512 |
+
---
|
| 513 |
+
|
| 514 |
+
## Environments
|
| 515 |
+
|
| 516 |
+
Both games ship with exact oracles, because the oracle *is* the training signal.
|
| 517 |
+
|
| 518 |
+
**Maze** reproduces NanoJev's four topologies (`corridor`, `tree`, `loops`,
|
| 519 |
+
`random_obstacle`) and its ASCII rendering so numbers are comparable. Targets
|
| 520 |
+
are the full BFS-optimal action distribution — every shortest-path move shares
|
| 521 |
+
the mass, so the model is never punished for picking a different optimal move.
|
| 522 |
+
|
| 523 |
+
**Snake** deliberately departs from NanoJev. NanoJev's target is a local greedy
|
| 524 |
+
rule: step toward the food. That self-traps. Jevon's oracle prefers, in order:
|
| 525 |
+
survive → keep the tail reachable → keep room for the body → then close on the
|
| 526 |
+
food. Measured over 20 games on 12×12:
|
| 527 |
+
|
| 528 |
+
| Snake target policy | mean food | max food | mean steps |
|
| 529 |
+
| --- | --- | --- | --- |
|
| 530 |
+
| NanoJev-style greedy | 22.70 | 41 | 229.0 |
|
| 531 |
+
| Jevon survival oracle | **43.00** | **105** | **2000.0** (full horizon, every game) |
|
| 532 |
+
|
| 533 |
+
Half of all sampled Snake states are **constructed directly** as self-avoiding
|
| 534 |
+
walks rather than reached by play. A good policy almost never produces a
|
| 535 |
+
cramped board, so rolling out to one is both slow and rare — yet cramped boards
|
| 536 |
+
are exactly where the survival questions carry signal.
|
| 537 |
+
|
| 538 |
+
---
|
| 539 |
+
|
| 540 |
+
## Question types
|
| 541 |
+
|
| 542 |
+
| Type | Example | Candidates |
|
| 543 |
+
| --- | --- | --- |
|
| 544 |
+
| `choice` | which move to make | 2–4 dynamic |
|
| 545 |
+
| `boolean` | "north is clear", "the goal is reachable" | false / true |
|
| 546 |
+
| `score` | how far the goal is, how much room remains | 5–7 ordered levels |
|
| 547 |
+
|
| 548 |
+
Every accuracy in this repo is reported next to the **constant-prediction
|
| 549 |
+
control** for the same questions. That is not decoration: the original
|
| 550 |
+
`room` question was scored against board area rather than snake length, which
|
| 551 |
+
made *every* sampled state the top level, and the model scored exactly the
|
| 552 |
+
constant baseline (`0.405556`) to six decimal places while appearing to learn.
|
| 553 |
+
|
| 554 |
+
---
|
| 555 |
+
|
| 556 |
+
## Training and reproduction
|
| 557 |
+
|
| 558 |
+
```bash
|
| 559 |
+
uv venv && uv pip install -r requirements.txt
|
| 560 |
+
|
| 561 |
+
python scripts/build_dataset.py # freeze val / test / OOD splits
|
| 562 |
+
|
| 563 |
+
# The shipped checkpoint, exactly. Every flag matters: --min-plus-field is
|
| 564 |
+
# what makes descent reach the goal by construction, --expose-field is what
|
| 565 |
+
# lets the text side read the planner, and --planner-lr-mult compensates for
|
| 566 |
+
# the planner block being applied several hundred times per step but carrying
|
| 567 |
+
# gradient through only the last eight.
|
| 568 |
+
python scripts/train.py --out runs/jevon-final --steps 9000 --eval-every 500 \
|
| 569 |
+
--planner-lr-mult 10 --field-weight 1.0 --seed 11 --min-plus-field --expose-field
|
| 570 |
+
|
| 571 |
+
bash scripts/benchmark.sh runs/jevon-final # eval + probe + play + RESULTS.md
|
| 572 |
+
pytest -q
|
| 573 |
+
```
|
| 574 |
+
|
| 575 |
+
Each run directory records both halves of its own provenance: `config.json`
|
| 576 |
+
is the architecture and `args.json` is the recipe (`argv` verbatim, plus the
|
| 577 |
+
parsed values the defaults filled in). `benchmark.sh` defaults to
|
| 578 |
+
`balanced.pt` — see [Which checkpoint ships](#which-checkpoint-ships).
|
| 579 |
+
|
| 580 |
+
`runs/jevon-final/` ships whole: the JSON this page is generated from, and
|
| 581 |
+
`balanced.pt` and `best.pt`, the two checkpoints [Which checkpoint
|
| 582 |
+
ships](#which-checkpoint-ships) compares. Each is 80MB of fp32 weights and
|
| 583 |
+
goes through **Git LFS**, so a clone needs it:
|
| 584 |
+
|
| 585 |
+
```bash
|
| 586 |
+
git lfs install
|
| 587 |
+
git clone https://huggingface.co/lewislululu/jevon
|
| 588 |
+
```
|
| 589 |
+
|
| 590 |
+
Cloned without LFS, those two paths hold 130-byte pointer files. Loading one
|
| 591 |
+
says exactly that and names the command, rather than failing inside
|
| 592 |
+
`torch.load`.
|
| 593 |
+
|
| 594 |
+
The JSON matters separately from the weights, and for a reason worth stating:
|
| 595 |
+
every test that reads those files degrades to a **skip** when they are
|
| 596 |
+
missing. An earlier `.gitignore` excluded the whole `runs/` directory, which
|
| 597 |
+
shipped a repo whose entire documentation-drift apparatus passed by not
|
| 598 |
+
running. `pytest -q` on a fresh clone now really does check this page against
|
| 599 |
+
the run it describes. `last.pt` stays out — it is wherever training stopped,
|
| 600 |
+
which is not a claim this repo makes.
|
| 601 |
+
|
| 602 |
+
`scripts/play.py` exposes each controller loop twice: as `maze_steps` /
|
| 603 |
+
`snake_steps`, generators that take one decision per `next()`, and as
|
| 604 |
+
`play_maze` / `play_snake`, which drain them. The split exists so that
|
| 605 |
+
[jevon-arcade](https://github.com/lewislulu/jevon-arcade)'s live viewer can take a single forward pass
|
| 606 |
+
per HTTP request without owning a second copy of the decision logic. A
|
| 607 |
+
divergent copy would be a viewer demonstrating a model nobody benchmarked,
|
| 608 |
+
which is the failure these two repos have hit more often than any other. The
|
| 609 |
+
step budget lives there too, in `default_max_steps`, so a live episode and a
|
| 610 |
+
recorded one agree about what running out means.
|
| 611 |
+
|
| 612 |
+
It exposes deliberately separable controllers so that model skill is never
|
| 613 |
+
confused with controller scaffolding:
|
| 614 |
+
|
| 615 |
+
- `model` — the network alone. No search, no memory, no visited set.
|
| 616 |
+
- `model+memory` — plus one bit per cell: prefer a move onto ground not yet
|
| 617 |
+
stood on. This is the same kind of scaffolding this README faults NanoJev's
|
| 618 |
+
demo for, so it is reported separately and never as the model's score.
|
| 619 |
+
- `random+memory` — **the control for the row above.** Identical bookkeeping,
|
| 620 |
+
no network. It exists because `model+memory` cannot be read without it, and
|
| 621 |
+
reading the two together is less flattering than reading one:
|
| 622 |
+
|
| 623 |
+
<!-- BEGIN:memory (generated by scripts/readme_table.py) -->
|
| 624 |
+
|
| 625 |
+
| controller | size | solve | efficiency |
|
| 626 |
+
| --- | ---: | ---: | ---: |
|
| 627 |
+
| `model+memory` | 11 | **1.00** | **1.000** |
|
| 628 |
+
| `random+memory` | 11 | 0.92 | 0.588 |
|
| 629 |
+
| `model+memory` | 21 | **1.00** | **1.000** |
|
| 630 |
+
| `random+memory` | 21 | 0.83 | 0.310 |
|
| 631 |
+
| `model+memory` | 31 | **1.00** | **1.000** |
|
| 632 |
+
| `random+memory` | 31 | 0.75 | 0.147 |
|
| 633 |
+
|
| 634 |
+
12 episodes per size per topology, from the shipped checkpoint.
|
| 635 |
+
|
| 636 |
+
<!-- END:memory -->
|
| 637 |
+
|
| 638 |
+
Read the two rows together, per size. The memory alone is a capable maze
|
| 639 |
+
solver on a small board — near-exhaustive exploration finds the goal — so
|
| 640 |
+
wherever `random+memory` matches `model+memory` on solve rate, arrival is
|
| 641 |
+
the scaffolding's doing and not the network's. The table bolds whichever of
|
| 642 |
+
the pair wins each column rather than always bolding the model, so a bold
|
| 643 |
+
control is a column the model does not own.
|
| 644 |
+
|
| 645 |
+
Efficiency is the column that does belong to the network. When the model
|
| 646 |
+
arrives it arrives on very nearly the shortest path; the control reaches the
|
| 647 |
+
same place several times slower. That gap is what the planner contributes,
|
| 648 |
+
and it is worth being exact about what it is not. The model knows the
|
| 649 |
+
direction and has no state with which to notice it has been somewhere
|
| 650 |
+
before, so cycling is what ends its unsolved episodes — and the memory masks
|
| 651 |
+
that rather than fixing it: when every neighbour has been visited the
|
| 652 |
+
controller falls back to the full legal set and the greedy policy walks back
|
| 653 |
+
into the cycle it just left.
|
| 654 |
+
- `reference` — BFS-optimal (maze) / survival oracle (snake).
|
| 655 |
+
- `random`.
|
| 656 |
+
|
| 657 |
+
**Why the memory is in the controller and not in the board.** The grid has a
|
| 658 |
+
`visited` channel (channel 6, used by snake for tail age) and maze leaves it
|
| 659 |
+
empty on purpose. Filling it would not teach the model to stop cycling: the
|
| 660 |
+
targets are BFS-optimal actions, and the optimal action from a cell is a
|
| 661 |
+
function of `(walls, goal, cell)` alone — where the agent has already been is
|
| 662 |
+
conditionally independent of it. Under this supervision the channel is noise
|
| 663 |
+
by construction, and a better-fitted model would learn to ignore it faster.
|
| 664 |
+
Cycling is approximation error in the action head, not absent memory, so the
|
| 665 |
+
lever that moves it is action accuracy; the architectural answer is
|
| 666 |
+
`model-field`, whose fixed point is a shortest-path distance and which
|
| 667 |
+
therefore cannot cycle at all.
|
| 668 |
+
|
| 669 |
+
The candidate set is `state.legal_actions()`, which is what the model was
|
| 670 |
+
trained to score. Offering it all four compass directions instead let a wall
|
| 671 |
+
win the argmax, and since a blocked move does not change the state, the next
|
| 672 |
+
decision was identical — an agent that stood still for the entire budget.
|
| 673 |
+
Every maze number for `model` predating commit `6afaae8` measured that.
|
| 674 |
+
|
| 675 |
+
Held-out splits are by seed (`seed % 10`: `<8` train, `8` val, `9` test), and
|
| 676 |
+
the OOD split uses board sizes never sampled during training (maze 41, 51;
|
| 677 |
+
snake 20, 24).
|
| 678 |
+
|
| 679 |
+
## Which checkpoint ships
|
| 680 |
+
|
| 681 |
+
Training writes three: `last.pt`, `best.pt` (lowest eval loss) and
|
| 682 |
+
`balanced.pt`. `balanced.pt` is what loads when you do not name one —
|
| 683 |
+
`benchmark.sh`, `record.sh` and every script's `--weights` agree on that, and
|
| 684 |
+
a test pins them to each other so they cannot drift. The reason is a selection
|
| 685 |
+
bug that `best.pt` walks straight into.
|
| 686 |
+
|
| 687 |
+
The frozen validation split is 1325 questions over 180 states: **974 boolean,
|
| 688 |
+
180 score, and 171 choice**. Eval loss averages over all of them, so 73.5% of
|
| 689 |
+
it is booleans and 12.9% is choice — while every headline number in this
|
| 690 |
+
README comes from those 171 choice questions, because those are the gameplay
|
| 691 |
+
decision. The two can move in opposite directions, and they do.
|
| 692 |
+
|
| 693 |
+
One run hit its lowest loss at a step scoring maze `0.9794` and snake
|
| 694 |
+
`0.2973`. The uniform controls on this split are `0.4905` for maze and
|
| 695 |
+
`0.4955` for snake — not `1/k`, because many states have several
|
| 696 |
+
tied-optimal moves — so that checkpoint was a full 20 points *below chance*
|
| 697 |
+
on snake while holding the best loss in the run. It had stopped playing one
|
| 698 |
+
of the two games, and eval loss could not see it.
|
| 699 |
+
|
| 700 |
+
So `balanced.pt` selects on `min(maze_action, snake_action)` instead —
|
| 701 |
+
|
| 702 |
+
```python
|
| 703 |
+
def balanced_score(metrics: dict) -> float:
|
| 704 |
+
return min(metrics.get("maze_action", 0.0), metrics.get("snake_action", 0.0))
|
| 705 |
+
```
|
| 706 |
+
|
| 707 |
+
`min` rather than a mean, because the failure being guarded against is
|
| 708 |
+
exactly the trade of one game for the other, and a mean averages it away. A
|
| 709 |
+
missing metric scores zero rather than being skipped, so a run that never
|
| 710 |
+
reports `snake_action` cannot earn a balanced checkpoint by default.
|
| 711 |
+
|
| 712 |
+
This is not hypothetical, and it is not rare — the shipped run did it too.
|
| 713 |
+
Caught mid-training at step 1500, the two checkpoints on disk at that instant
|
| 714 |
+
made the point better than any argument (both were later superseded; the
|
| 715 |
+
final ones are in `runs/jevon-final/`):
|
| 716 |
+
|
| 717 |
+
| | `best.pt` (step 1500) | `balanced.pt` (step 1000) |
|
| 718 |
+
| --- | ---: | ---: |
|
| 719 |
+
| eval loss | **0.6161** | 0.6269 |
|
| 720 |
+
| `acc_boolean` | 0.94688 | 0.94688 |
|
| 721 |
+
| `acc_score` | 0.42222 | 0.42222 |
|
| 722 |
+
| `acc_choice` | 0.6665 | **0.8469** |
|
| 723 |
+
| `snake_action` | 0.5135 | **0.9459** |
|
| 724 |
+
|
| 725 |
+
`best.pt` won on loss. Boolean and score accuracy are *identical to five
|
| 726 |
+
decimal places* between the two, so neither moved; the only accuracy that
|
| 727 |
+
moved is choice, and it moved **18 points the wrong way**. The loss improved
|
| 728 |
+
because the model sharpened its confidence on booleans it was already getting
|
| 729 |
+
right (`ce` 0.4377 → 0.4286) while its snake play fell to `0.5135`, a hair
|
| 730 |
+
above the `0.4955` control. Selecting on loss would have shipped that.
|
| 731 |
+
|
| 732 |
+
Training later reached maze `1.0000` and snake `0.9595` together, and
|
| 733 |
+
`balanced.pt` followed it up as readily as it had refused to follow it down.
|
| 734 |
+
|
| 735 |
+
The shipped run's own two checkpoints are that same argument, and unlike the
|
| 736 |
+
snapshot above both are on disk for a reader to check:
|
| 737 |
+
|
| 738 |
+
<!-- BEGIN:shipped (generated by scripts/readme_table.py) -->
|
| 739 |
+
|
| 740 |
+
| | `best.pt` (step 8500) | `balanced.pt` (step 3000) |
|
| 741 |
+
| --- | ---: | ---: |
|
| 742 |
+
| eval loss | **0.3999** | 0.4821 |
|
| 743 |
+
| `acc_boolean` | **0.96732** | 0.94688 |
|
| 744 |
+
| `acc_score` | 0.34444 | **0.35000** |
|
| 745 |
+
| `acc_choice` | 0.9889 | **0.9944** |
|
| 746 |
+
| `maze_action` | 1.0000 | 1.0000 |
|
| 747 |
+
| `snake_action` | 0.9730 | **0.9865** |
|
| 748 |
+
|
| 749 |
+
<!-- END:shipped -->
|
| 750 |
+
|
| 751 |
+
Bold marks the better cell in each row — loss being the one row where smaller
|
| 752 |
+
is better. `best.pt` takes loss and the two question types that dominate it;
|
| 753 |
+
`balanced.pt` takes the choice questions, which are the ones every headline
|
| 754 |
+
number in this README is scored on. `benchmark.sh` defaults to `balanced.pt`,
|
| 755 |
+
which is why.
|
| 756 |
+
|
| 757 |
+
### Selection reads `val`; every headline number comes from `test`
|
| 758 |
+
|
| 759 |
+
That separation matters more here than it usually does, because the quantity
|
| 760 |
+
being selected on keeps moving. The eval is deterministic — running it twice
|
| 761 |
+
on a fixed checkpoint returns identical figures to six decimals — so what the
|
| 762 |
+
table below measures is real weight movement between checkpoints, not
|
| 763 |
+
measurement noise.
|
| 764 |
+
|
| 765 |
+
<!-- BEGIN:volatility (generated by scripts/readme_table.py) -->
|
| 766 |
+
|
| 767 |
+
| half of the run | evals | largest eval-to-eval move in `snake_action` | in questions |
|
| 768 |
+
| --- | ---: | ---: | ---: |
|
| 769 |
+
| first (500-5000) | 10 | `0.446` | 33 of 74 |
|
| 770 |
+
| second (5000-9000) | 9 | `0.122` | 9 of 74 |
|
| 771 |
+
|
| 772 |
+
<!-- END:volatility -->
|
| 773 |
+
|
| 774 |
+
The movement shrinks by the factor shown there, and it does not reach zero.
|
| 775 |
+
An earlier draft of this section claimed it stopped, on the evidence of four
|
| 776 |
+
consecutive evals returning a byte-identical `0.9865`; two evals later one
|
| 777 |
+
gave back six questions and the sentence was false. Any claim phrased over
|
| 778 |
+
the last N evals goes stale as soon as training continues past them, which is
|
| 779 |
+
why this one is phrased over halves of the run and generated rather than
|
| 780 |
+
typed.
|
| 781 |
+
|
| 782 |
+
Selecting the maximum of ~18 such evals is a maximum of a noisy estimate, so
|
| 783 |
+
a `val` figure is biased upward by construction — and it is the second-half
|
| 784 |
+
column that sizes that bias, not the first. `test` is scored
|
| 785 |
+
once, by a checkpoint that never competed on it, which is why the headline
|
| 786 |
+
quotes it and names the split it is quoting. `val` is reported too, in
|
| 787 |
+
[RESULTS.md](RESULTS.md), where it can be read against the split that
|
| 788 |
+
selection never touched.
|
| 789 |
+
|
| 790 |
+
## Results
|
| 791 |
+
|
| 792 |
+
See [RESULTS.md](RESULTS.md), which is generated from the evaluation JSON
|
| 793 |
+
rather than typed by hand.
|
| 794 |
+
|
| 795 |
+
## Licence
|
| 796 |
+
|
| 797 |
+
Copyright © 2026 Lewis.
|
| 798 |
+
|
| 799 |
+
[AGPL-3.0-or-later](LICENSE). Use it, modify it, run it for any purpose
|
| 800 |
+
including a commercial one — provided the complete source of whatever you
|
| 801 |
+
build on it is offered under the same licence. Section 13 is why this is the
|
| 802 |
+
AGPL and not the GPL: it extends that to users who only ever reach the model
|
| 803 |
+
over a network, so serving Jevon behind an API is covered where plain GPL
|
| 804 |
+
would not have been.
|
| 805 |
+
|
| 806 |
+
If your product cannot carry that obligation, a
|
| 807 |
+
[commercial licence](COMMERCIAL-LICENSE.md) without the copyleft terms is
|
| 808 |
+
available — sudolewis@gmail.com.
|
| 809 |
+
|
| 810 |
+
The viewer, [jevon-arcade][arcade], is MIT instead. It holds no model code, it
|
| 811 |
+
imports this repository at runtime, and a demo is worth more the more people
|
| 812 |
+
run it — but a distribution that combines the two carries this licence, not
|
| 813 |
+
that one.
|
|
@@ -0,0 +1,157 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Results
|
| 2 |
+
|
| 3 |
+
Generated by `scripts/make_results.py` from the JSON written by `train.py`, `evaluate.py`, `probe.py` and `play.py`. Do not edit by hand.
|
| 4 |
+
|
| 5 |
+
## Baseline: what NanoJev reports for itself
|
| 6 |
+
|
| 7 |
+
From `docs/DEVELOPMENT_RESULTS.md` in the NanoJev repository.
|
| 8 |
+
|
| 9 |
+
| NanoJev task | reported acc | always-true control | above control |
|
| 10 |
+
| --- | ---: | ---: | :---: |
|
| 11 |
+
| `scaled_maze` (test) | 0.5625 | 0.5625 | no |
|
| 12 |
+
| `scaled_maze` (ood) | 0.5625 | 0.5625 | no |
|
| 13 |
+
| `snake_one_step_safety_v4` | 0.9220 | 0.9220 | no |
|
| 14 |
+
|
| 15 |
+
Closed-loop play, same source, 128-step limit:
|
| 16 |
+
|
| 17 |
+
| maze | Jev | NanoJev | reference |
|
| 18 |
+
| --- | ---: | ---: | ---: |
|
| 19 |
+
| all three mazes | 0/128 | 0/128 | solved 1/16, 1/24, 1/96 |
|
| 20 |
+
|
| 21 |
+
> "The learned direct-action models and Jev do not solve any of these
|
| 22 |
+
> three mazes within the shared 128-step limit."
|
| 23 |
+
|
| 24 |
+
Every accuracy in this file is therefore printed next to its
|
| 25 |
+
constant-prediction control, and gameplay is reported as solve rate
|
| 26 |
+
rather than per-question accuracy.
|
| 27 |
+
|
| 28 |
+
## Does the planner actually compute distances?
|
| 29 |
+
|
| 30 |
+
Ridge probe from the planner's per-cell features to true BFS distance, on held-out mazes. `diameter` is the longest true shortest-path in those boards; `T` is the iteration budget.
|
| 31 |
+
|
| 32 |
+
`descent` is the fraction of cells whose lowest-predicted neighbour lies on a true shortest path. `reach` is the fraction from which greedy descent actually arrives at the goal, and it is the one that predicts solve rate: a field can be right 70% of the time per step and still trap most walks in a spurious basin, so `descent` is an upper bound on nothing in particular. R² scores absolute values; only `reach` scores the behaviour.
|
| 33 |
+
|
| 34 |
+
`read-out R²` scores `predict_field()`, the field this controller actually descends -- the min-plus recurrence when the run enabled it, the conv head otherwise. The training curve's `conv field R²` scores a different head; see the note there.
|
| 35 |
+
|
| 36 |
+
| size | cells | probe R² | probe MAE (cells) | read-out R² | descent | reach | diameter | T |
|
| 37 |
+
| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
|
| 38 |
+
| 11 | 215 | **0.0937** | 7.97 | 1.0000 | 1.0000 | **1.0000** | 48 | 82 |
|
| 39 |
+
| 21 | 901 | **0.0420** | 29.71 | 1.0000 | 1.0000 | **1.0000** | 152 | 262 |
|
| 40 |
+
| 31 | 2011 | **0.0316** | 63.06 | 1.0000 | 1.0000 | **1.0000** | 310 | 542 |
|
| 41 |
+
| 51 | 5679 | **0.0068** | 145.23 | 1.0000 | 1.0000 | **1.0000** | 672 | 1402 |
|
| 42 |
+
|
| 43 |
+
## Question-level accuracy
|
| 44 |
+
|
| 45 |
+
`uniform` is the accuracy of guessing uniformly over the same candidates. A model that matches its control has learned nothing, which is the failure NanoJev's own log records for its maze model.
|
| 46 |
+
|
| 47 |
+
`boards` is how many distinct boards the row's questions came from, and it is there so that `n` can be trusted. Questions that share a board are not independent tests of whether the model can read a board -- they share walls, goal and distance field, so they succeed or fail together -- and a row reporting `n` alone cannot tell you whether that is happening. Here it is not: the generator draws a fresh board per state, so `boards` tracks `n` on every split and `n` is the honest denominator. A row where the two diverge is one whose precision should be read off `boards`.
|
| 48 |
+
|
| 49 |
+
### val — 1325 questions, ECE 0.0865
|
| 50 |
+
|
| 51 |
+
| group | n | boards | accuracy | uniform control | TVD | Brier |
|
| 52 |
+
| --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
| 53 |
+
| `maze/action` | 97 | 97 | **1.0000** | 0.4905 | 0.0188 | 0.0023 |
|
| 54 |
+
| `maze/boolean` | 530 | 106 | **1.0000** | 0.5000 | 0.0007 | 0.0000 |
|
| 55 |
+
| `maze/choice` | 97 | 97 | **1.0000** | 0.4905 | 0.0188 | 0.0023 |
|
| 56 |
+
| `maze/clear` | 424 | 106 | **1.0000** | 0.5000 | 0.0004 | 0.0000 |
|
| 57 |
+
| `maze/distance` | 106 | 106 | **0.3962** | 0.1429 | 0.7764 | 0.7640 |
|
| 58 |
+
| `maze/score` | 106 | 106 | **0.3962** | 0.1429 | 0.7764 | 0.7640 |
|
| 59 |
+
| `maze/solvable` | 106 | 106 | **1.0000** | 0.5000 | 0.0018 | 0.0000 |
|
| 60 |
+
| `snake/action` | 74 | 74 | **0.9865** | 0.4955 | 0.3500 | 0.3148 |
|
| 61 |
+
| `snake/boolean` | 444 | 74 | **0.8829** | 0.5000 | 0.2338 | 0.2022 |
|
| 62 |
+
| `snake/choice` | 74 | 74 | **0.9865** | 0.4955 | 0.3500 | 0.3148 |
|
| 63 |
+
| `snake/escape` | 222 | 74 | **0.8108** | 0.5000 | 0.2773 | 0.2891 |
|
| 64 |
+
| `snake/room` | 74 | 74 | **0.2838** | 0.2000 | 0.7990 | 0.7981 |
|
| 65 |
+
| `snake/safe` | 222 | 74 | **0.9550** | 0.5000 | 0.1903 | 0.1152 |
|
| 66 |
+
| `snake/score` | 74 | 74 | **0.2838** | 0.2000 | 0.7990 | 0.7981 |
|
| 67 |
+
|
| 68 |
+
### test — 1765 questions, ECE 0.0590
|
| 69 |
+
|
| 70 |
+
| group | n | boards | accuracy | uniform control | TVD | Brier |
|
| 71 |
+
| --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
| 72 |
+
| `maze/action` | 141 | 140 | **1.0000** | 0.4882 | 0.0396 | 0.0079 |
|
| 73 |
+
| `maze/boolean` | 740 | 147 | **1.0000** | 0.5000 | 0.0007 | 0.0000 |
|
| 74 |
+
| `maze/choice` | 141 | 140 | **1.0000** | 0.4882 | 0.0396 | 0.0079 |
|
| 75 |
+
| `maze/clear` | 592 | 147 | **1.0000** | 0.5000 | 0.0004 | 0.0000 |
|
| 76 |
+
| `maze/distance` | 148 | 147 | **0.2973** | 0.1429 | 0.7847 | 0.7796 |
|
| 77 |
+
| `maze/score` | 148 | 147 | **0.2973** | 0.1429 | 0.7847 | 0.7796 |
|
| 78 |
+
| `maze/solvable` | 148 | 147 | **1.0000** | 0.5000 | 0.0019 | 0.0000 |
|
| 79 |
+
| `snake/action` | 92 | 92 | **0.9674** | 0.4783 | 0.3107 | 0.2770 |
|
| 80 |
+
| `snake/boolean` | 552 | 92 | **0.8895** | 0.5000 | 0.2223 | 0.1938 |
|
| 81 |
+
| `snake/choice` | 92 | 92 | **0.9674** | 0.4783 | 0.3107 | 0.2770 |
|
| 82 |
+
| `snake/escape` | 276 | 92 | **0.8188** | 0.5000 | 0.2628 | 0.2747 |
|
| 83 |
+
| `snake/room` | 92 | 92 | **0.2500** | 0.2000 | 0.7993 | 0.7986 |
|
| 84 |
+
| `snake/safe` | 276 | 92 | **0.9601** | 0.5000 | 0.1818 | 0.1128 |
|
| 85 |
+
| `snake/score` | 92 | 92 | **0.2500** | 0.2000 | 0.7993 | 0.7986 |
|
| 86 |
+
|
| 87 |
+
### ood — 1167 questions, ECE 0.0540
|
| 88 |
+
|
| 89 |
+
| group | n | boards | accuracy | uniform control | TVD | Brier |
|
| 90 |
+
| --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
| 91 |
+
| `maze/action` | 107 | 107 | **1.0000** | 0.5132 | 0.2699 | 0.1504 |
|
| 92 |
+
| `maze/boolean` | 550 | 110 | **1.0000** | 0.5000 | 0.0007 | 0.0000 |
|
| 93 |
+
| `maze/choice` | 107 | 107 | **1.0000** | 0.5132 | 0.2699 | 0.1504 |
|
| 94 |
+
| `maze/clear` | 440 | 110 | **1.0000** | 0.5000 | 0.0004 | 0.0000 |
|
| 95 |
+
| `maze/distance` | 110 | 110 | **0.0545** | 0.1429 | 0.8542 | 0.9160 |
|
| 96 |
+
| `maze/score` | 110 | 110 | **0.0545** | 0.1429 | 0.8542 | 0.9160 |
|
| 97 |
+
| `maze/solvable` | 110 | 110 | **1.0000** | 0.5000 | 0.0020 | 0.0000 |
|
| 98 |
+
| `snake/action` | 50 | 50 | **0.9800** | 0.4867 | 0.3574 | 0.3199 |
|
| 99 |
+
| `snake/boolean` | 300 | 50 | **0.8333** | 0.5000 | 0.2636 | 0.2621 |
|
| 100 |
+
| `snake/choice` | 50 | 50 | **0.9800** | 0.4867 | 0.3574 | 0.3199 |
|
| 101 |
+
| `snake/escape` | 150 | 50 | **0.6800** | 0.5000 | 0.3561 | 0.4471 |
|
| 102 |
+
| `snake/room` | 50 | 50 | **0.2000** | 0.2000 | 0.7989 | 0.7978 |
|
| 103 |
+
| `snake/safe` | 150 | 50 | **0.9867** | 0.5000 | 0.1710 | 0.0771 |
|
| 104 |
+
| `snake/score` | 50 | 50 | **0.2000** | 0.2000 | 0.7989 | 0.7978 |
|
| 105 |
+
|
| 106 |
+
## Closed-loop play
|
| 107 |
+
|
| 108 |
+
Read `model` and `model-field` as answering different questions. `model` is the network's action head alone -- no search, no memory, no visited set -- and is the apples-to-apples comparison with NanoJev. `model-field` descends a min-plus recurrence whose fixed point is a shortest-path distance by construction, so it arrives from anywhere; for a maze it does so at initialisation, before any training, because uniform cost is already the right answer. Its solve rate is a property of the architecture, not a measurement of what this run learned.
|
| 109 |
+
|
| 110 |
+
### Maze
|
| 111 |
+
|
| 112 |
+
`deadlock` is the fraction of episodes in which the controller spent a tenth of its step budget walking into the same wall, and `cells` is how many distinct squares it stood on. They separate two failures a solve rate of 0.00 reports identically: standing still and touring the board. Deadlock should now read 0.00 for every controller, because only legal moves are offered as candidates -- it is kept as a regression guard, not a finding. The failure that remains is cycling: the planner never sees the agent, so a greedy policy that steps A->B->A has no state with which to notice, and `cells` is what exposes it.
|
| 113 |
+
|
| 114 |
+
| controller | solve rate | mean steps | efficiency | collisions | deadlock | cells |
|
| 115 |
+
| --- | ---: | ---: | ---: | ---: | ---: | ---: |
|
| 116 |
+
| `maze_model` | 1.00 | 93.2 | 1.000 | 0.0 | 0.00 | 94.2 |
|
| 117 |
+
| `maze_model_sampled` | 1.00 | 106.1 | 0.926 | 0.0 | 0.00 | 95.4 |
|
| 118 |
+
| `maze_model_field` | 1.00 | 93.2 | 1.000 | 0.0 | 0.00 | 94.2 |
|
| 119 |
+
| `maze_model_memory` | 1.00 | 93.2 | 1.000 | 0.0 | 0.00 | 94.2 |
|
| 120 |
+
| `maze_random_memory` | 0.83 | 689.7 | 0.363 | 0.0 | 0.00 | 188.5 |
|
| 121 |
+
| `maze_reference` | 1.00 | 93.2 | 1.000 | 0.0 | 0.00 | 94.2 |
|
| 122 |
+
| `maze_random` | 0.14 | 560.4 | 0.068 | 0.0 | 0.00 | 92.2 |
|
| 123 |
+
|
| 124 |
+
### Snake
|
| 125 |
+
|
| 126 |
+
| controller | mean food | max food | mean steps | survival |
|
| 127 |
+
| --- | ---: | ---: | ---: | ---: |
|
| 128 |
+
| `snake_model` | 4.50 | 16 | 288.0 | 1.00 |
|
| 129 |
+
| `snake_reference` | 23.50 | 31 | 288.0 | 1.00 |
|
| 130 |
+
| `snake_random` | 0.67 | 2 | 33.8 | 0.00 |
|
| 131 |
+
|
| 132 |
+
## Training curve
|
| 133 |
+
|
| 134 |
+
Frozen validation split, never sampled during training.
|
| 135 |
+
|
| 136 |
+
`conv field R²` scores `supervised_field()` -- the conv head the field loss trains. It is **not** the field the controller descends. Under `--min-plus-field` those are different objects: the read-out is the min-plus recurrence, scored as `read-out R²` in the probe table above, and nothing at inference reads the conv head at all. It stays in the loss as an auxiliary task on the shared trunk, so a negative value here means that auxiliary head has stopped tracking the trunk the action head and the recurrence are shaping -- it does not mean the planner is wrong, and `reach` is where that would show up.
|
| 137 |
+
|
| 138 |
+
| step | loss | acc | choice | boolean | score | conv field R² | maze action | snake action |
|
| 139 |
+
| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |
|
| 140 |
+
| 500 | 0.6430 | 0.8310 | 0.6806 | 0.9469 | 0.3500 | 0.5048 | 0.5773 | 0.8108 |
|
| 141 |
+
| 1000 | 0.6269 | 0.8626 | 0.8469 | 0.9469 | 0.4222 | -2.7876 | 0.7732 | 0.9459 |
|
| 142 |
+
| 1500 | 0.6161 | 0.8393 | 0.6665 | 0.9469 | 0.4222 | 0.0461 | 0.7835 | 0.5135 |
|
| 143 |
+
| 2000 | 0.5041 | 0.8703 | 0.9833 | 0.9469 | 0.3500 | -0.1305 | 1.0000 | 0.9595 |
|
| 144 |
+
| 2500 | 0.4842 | 0.8531 | 0.8499 | 0.9469 | 0.3500 | 0.1714 | 1.0000 | 0.6486 |
|
| 145 |
+
| 3000 | 0.4821 | 0.8718 | 0.9944 | 0.9469 | 0.3500 | -0.2088 | 1.0000 | 0.9865 |
|
| 146 |
+
| 3500 | 0.4725 | 0.8718 | 0.9944 | 0.9469 | 0.3500 | 0.0103 | 1.0000 | 0.9865 |
|
| 147 |
+
| 4000 | 0.4797 | 0.8702 | 0.9828 | 0.9469 | 0.3500 | 0.2466 | 0.9794 | 0.9865 |
|
| 148 |
+
| 4500 | 0.4765 | 0.8710 | 0.9889 | 0.9469 | 0.3500 | 0.2097 | 0.9897 | 0.9865 |
|
| 149 |
+
| 5000 | 0.4733 | 0.8718 | 0.9944 | 0.9469 | 0.3500 | 0.1148 | 1.0000 | 0.9865 |
|
| 150 |
+
| 5500 | 0.4667 | 0.8725 | 0.9944 | 0.9469 | 0.3556 | -0.2881 | 1.0000 | 0.9865 |
|
| 151 |
+
| 6000 | 0.4579 | 0.8673 | 0.9595 | 0.9469 | 0.3500 | -0.1574 | 1.0000 | 0.9054 |
|
| 152 |
+
| 6500 | 0.4543 | 0.8695 | 0.9767 | 0.9469 | 0.3500 | 0.2626 | 1.0000 | 0.9459 |
|
| 153 |
+
| 7000 | 0.4258 | 0.8860 | 0.9889 | 0.9673 | 0.3500 | 0.0325 | 1.0000 | 0.9730 |
|
| 154 |
+
| 7500 | 0.4088 | 0.8831 | 0.9368 | 0.9673 | 0.3778 | -0.4171 | 1.0000 | 0.8514 |
|
| 155 |
+
| 8000 | 0.4030 | 0.8860 | 0.9889 | 0.9673 | 0.3500 | 0.0676 | 1.0000 | 0.9730 |
|
| 156 |
+
| 8500 | 0.3999 | 0.8853 | 0.9889 | 0.9673 | 0.3444 | 0.2261 | 1.0000 | 0.9730 |
|
| 157 |
+
| 9000 | 0.4021 | 0.8958 | 0.9889 | 0.9673 | 0.4222 | 0.1083 | 1.0000 | 0.9730 |
|
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"val": 180,
|
| 3 |
+
"test": 240,
|
| 4 |
+
"ood": 160,
|
| 5 |
+
"config": {
|
| 6 |
+
"train_maze_sizes": [
|
| 7 |
+
7,
|
| 8 |
+
9,
|
| 9 |
+
11,
|
| 10 |
+
15,
|
| 11 |
+
21,
|
| 12 |
+
25,
|
| 13 |
+
31
|
| 14 |
+
],
|
| 15 |
+
"train_snake_sizes": [
|
| 16 |
+
8,
|
| 17 |
+
10,
|
| 18 |
+
12,
|
| 19 |
+
14,
|
| 20 |
+
16
|
| 21 |
+
],
|
| 22 |
+
"ood_maze_sizes": [
|
| 23 |
+
41,
|
| 24 |
+
51
|
| 25 |
+
],
|
| 26 |
+
"ood_snake_sizes": [
|
| 27 |
+
20,
|
| 28 |
+
24
|
| 29 |
+
]
|
| 30 |
+
}
|
| 31 |
+
}
|
|
The diff for this file is too large to render.
See raw diff
|
|
|
|
The diff for this file is too large to render.
See raw diff
|
|
|
|
The diff for this file is too large to render.
See raw diff
|
|
|
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
"""Maze and Snake environments with exact oracles."""
|
|
@@ -0,0 +1,363 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Maze environment: generators, BFS oracle, renderers, and question profiles.
|
| 2 |
+
|
| 3 |
+
Topologies and the ASCII rendering follow NanoJev's `scaled_maze.py` so that
|
| 4 |
+
numbers from the two projects can be placed side by side. Everything else --
|
| 5 |
+
the oracle targets and the structured grid view -- is new.
|
| 6 |
+
|
| 7 |
+
Coordinate convention is zero-based ``(row, col)`` with row 0 at the top.
|
| 8 |
+
"""
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import random
|
| 12 |
+
from collections import deque
|
| 13 |
+
from dataclasses import dataclass, field
|
| 14 |
+
from typing import Iterable, Sequence
|
| 15 |
+
|
| 16 |
+
DIRECTIONS: dict[str, tuple[int, int]] = {
|
| 17 |
+
"north": (-1, 0),
|
| 18 |
+
"east": (0, 1),
|
| 19 |
+
"south": (1, 0),
|
| 20 |
+
"west": (0, -1),
|
| 21 |
+
}
|
| 22 |
+
ACTIONS: tuple[str, ...] = ("north", "east", "south", "west")
|
| 23 |
+
TOPOLOGIES: tuple[str, ...] = ("corridor", "tree", "loops", "random_obstacle")
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
@dataclass
|
| 27 |
+
class MazeState:
|
| 28 |
+
size: int
|
| 29 |
+
walls: frozenset[tuple[int, int]]
|
| 30 |
+
position: tuple[int, int]
|
| 31 |
+
goal: tuple[int, int]
|
| 32 |
+
topology: str = "tree"
|
| 33 |
+
seed: int = 0
|
| 34 |
+
# Set by the simulator, not by the model's view of the world.
|
| 35 |
+
steps: int = 0
|
| 36 |
+
collisions: int = 0
|
| 37 |
+
visited: set[tuple[int, int]] = field(default_factory=set)
|
| 38 |
+
|
| 39 |
+
def free(self, cell: tuple[int, int]) -> bool:
|
| 40 |
+
row, col = cell
|
| 41 |
+
return 0 <= row < self.size and 0 <= col < self.size and cell not in self.walls
|
| 42 |
+
|
| 43 |
+
def legal_actions(self) -> list[str]:
|
| 44 |
+
out = []
|
| 45 |
+
for action in ACTIONS:
|
| 46 |
+
dr, dc = DIRECTIONS[action]
|
| 47 |
+
if self.free((self.position[0] + dr, self.position[1] + dc)):
|
| 48 |
+
out.append(action)
|
| 49 |
+
return out
|
| 50 |
+
|
| 51 |
+
def step(self, action: str) -> bool:
|
| 52 |
+
"""Apply ``action``. Returns True when the agent actually moved."""
|
| 53 |
+
dr, dc = DIRECTIONS[action]
|
| 54 |
+
target = (self.position[0] + dr, self.position[1] + dc)
|
| 55 |
+
self.steps += 1
|
| 56 |
+
if self.free(target):
|
| 57 |
+
self.position = target
|
| 58 |
+
self.visited.add(target)
|
| 59 |
+
return True
|
| 60 |
+
self.collisions += 1
|
| 61 |
+
return False
|
| 62 |
+
|
| 63 |
+
def solved(self) -> bool:
|
| 64 |
+
return self.position == self.goal
|
| 65 |
+
|
| 66 |
+
def clone(self) -> "MazeState":
|
| 67 |
+
return MazeState(self.size, self.walls, self.position, self.goal, self.topology,
|
| 68 |
+
self.seed, self.steps, self.collisions, set(self.visited))
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
# --------------------------------------------------------------------------
|
| 72 |
+
# Generation
|
| 73 |
+
# --------------------------------------------------------------------------
|
| 74 |
+
|
| 75 |
+
def _carve_dfs(size: int, rng: random.Random, straight_bias: float) -> set[tuple[int, int]]:
|
| 76 |
+
"""Randomised DFS on the odd sublattice; returns the set of carved cells."""
|
| 77 |
+
start = (1, 1)
|
| 78 |
+
opened = {start}
|
| 79 |
+
stack = [(start, None)]
|
| 80 |
+
while stack:
|
| 81 |
+
cell, incoming = stack[-1]
|
| 82 |
+
options = []
|
| 83 |
+
for action in ACTIONS:
|
| 84 |
+
dr, dc = DIRECTIONS[action]
|
| 85 |
+
nxt = (cell[0] + 2 * dr, cell[1] + 2 * dc)
|
| 86 |
+
if 1 <= nxt[0] < size - 1 and 1 <= nxt[1] < size - 1 and nxt not in opened:
|
| 87 |
+
options.append((action, nxt))
|
| 88 |
+
if not options:
|
| 89 |
+
stack.pop()
|
| 90 |
+
continue
|
| 91 |
+
# Straight bias keeps corridors long, which makes the planning horizon deep.
|
| 92 |
+
weights = [straight_bias if action == incoming else 1.0 for action, _ in options]
|
| 93 |
+
action, nxt = rng.choices(options, weights=weights, k=1)[0]
|
| 94 |
+
dr, dc = DIRECTIONS[action]
|
| 95 |
+
opened.add((cell[0] + dr, cell[1] + dc))
|
| 96 |
+
opened.add(nxt)
|
| 97 |
+
stack.append((nxt, action))
|
| 98 |
+
return opened
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
def _largest_component(size: int, free: set[tuple[int, int]]) -> set[tuple[int, int]]:
|
| 102 |
+
seen: set[tuple[int, int]] = set()
|
| 103 |
+
best: set[tuple[int, int]] = set()
|
| 104 |
+
for cell in free:
|
| 105 |
+
if cell in seen:
|
| 106 |
+
continue
|
| 107 |
+
queue, comp = deque([cell]), set()
|
| 108 |
+
seen.add(cell)
|
| 109 |
+
while queue:
|
| 110 |
+
cur = queue.popleft()
|
| 111 |
+
comp.add(cur)
|
| 112 |
+
for dr, dc in DIRECTIONS.values():
|
| 113 |
+
nxt = (cur[0] + dr, cur[1] + dc)
|
| 114 |
+
if nxt in free and nxt not in seen:
|
| 115 |
+
seen.add(nxt)
|
| 116 |
+
queue.append(nxt)
|
| 117 |
+
if len(comp) > len(best):
|
| 118 |
+
best = comp
|
| 119 |
+
return best
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
def generate_maze(size: int, topology: str, seed: int) -> MazeState:
|
| 123 |
+
"""Build a solvable maze of the given size and topology."""
|
| 124 |
+
if topology not in TOPOLOGIES:
|
| 125 |
+
raise ValueError(f"unknown topology {topology!r}")
|
| 126 |
+
if size < 5:
|
| 127 |
+
raise ValueError("size must be at least 5")
|
| 128 |
+
# TOPOLOGIES.index, not hash(topology): str.__hash__ is salted per process
|
| 129 |
+
# unless PYTHONHASHSEED is pinned, so hashing the name here handed every
|
| 130 |
+
# process a different maze for the same (size, topology, seed). Training
|
| 131 |
+
# merely resampled, but probe.py and play.py both regenerate their boards,
|
| 132 |
+
# which made their numbers incomparable between runs and unreproducible for
|
| 133 |
+
# anyone cloning the repository.
|
| 134 |
+
base = (seed * 1_000_003) ^ (size * 7919) ^ (TOPOLOGIES.index(topology) * 65_537)
|
| 135 |
+
|
| 136 |
+
# Retry rather than raise. random_obstacle draws its density before it
|
| 137 |
+
# knows whether the result percolates, so a small fraction of seeds leave
|
| 138 |
+
# a largest component too small to use. That used to raise, and since
|
| 139 |
+
# nothing up the stack caught it, one unlucky draw killed a training run
|
| 140 |
+
# hours in -- which is how this was found. Thinning the density on each
|
| 141 |
+
# attempt makes the retry converge instead of resampling the same tail
|
| 142 |
+
# forever, and attempt 0 is byte-identical to the previous behaviour, so
|
| 143 |
+
# only the seeds that already failed move.
|
| 144 |
+
free: set[tuple[int, int]] = set()
|
| 145 |
+
for attempt in range(8):
|
| 146 |
+
rng = random.Random(base ^ (attempt * 2_654_435_761))
|
| 147 |
+
if topology == "random_obstacle":
|
| 148 |
+
density = rng.uniform(0.22, 0.32) * (0.7 ** attempt)
|
| 149 |
+
free = {
|
| 150 |
+
(r, c)
|
| 151 |
+
for r in range(1, size - 1)
|
| 152 |
+
for c in range(1, size - 1)
|
| 153 |
+
if rng.random() > density
|
| 154 |
+
}
|
| 155 |
+
free = _largest_component(size, free)
|
| 156 |
+
else:
|
| 157 |
+
straight = {"corridor": 6.0, "tree": 1.0, "loops": 1.5}[topology]
|
| 158 |
+
free = _carve_dfs(size, rng, straight)
|
| 159 |
+
if topology == "loops":
|
| 160 |
+
# Knock out extra walls to introduce cycles (multiple shortest paths).
|
| 161 |
+
candidates = [
|
| 162 |
+
(r, c)
|
| 163 |
+
for r in range(1, size - 1)
|
| 164 |
+
for c in range(1, size - 1)
|
| 165 |
+
if (r, c) not in free
|
| 166 |
+
and sum((r + dr, c + dc) in free for dr, dc in DIRECTIONS.values()) >= 2
|
| 167 |
+
]
|
| 168 |
+
rng.shuffle(candidates)
|
| 169 |
+
for cell in candidates[: max(1, len(candidates) // 6)]:
|
| 170 |
+
free.add(cell)
|
| 171 |
+
free = _largest_component(size, free)
|
| 172 |
+
if len(free) >= 4:
|
| 173 |
+
break
|
| 174 |
+
else:
|
| 175 |
+
# Unreachable for size >= 5: by attempt 7 the density is under 2%.
|
| 176 |
+
raise RuntimeError(f"degenerate maze: {topology} size={size} seed={seed}")
|
| 177 |
+
walls = frozenset(
|
| 178 |
+
(r, c) for r in range(size) for c in range(size) if (r, c) not in free
|
| 179 |
+
)
|
| 180 |
+
|
| 181 |
+
# Pick a start/goal pair that is far apart so the task needs real planning.
|
| 182 |
+
cells = sorted(free)
|
| 183 |
+
anchor = rng.choice(cells)
|
| 184 |
+
far = bfs_distances(size, walls, anchor)
|
| 185 |
+
start = max((c for c in cells if far[c] < 10**8), key=lambda c: far[c])
|
| 186 |
+
dist = bfs_distances(size, walls, start)
|
| 187 |
+
reachable = [c for c in cells if dist[c] < 10**8 and c != start]
|
| 188 |
+
goal = max(reachable, key=lambda c: dist[c])
|
| 189 |
+
return MazeState(size, walls, start, goal, topology, seed, visited={start})
|
| 190 |
+
|
| 191 |
+
|
| 192 |
+
# --------------------------------------------------------------------------
|
| 193 |
+
# Oracle
|
| 194 |
+
# --------------------------------------------------------------------------
|
| 195 |
+
|
| 196 |
+
UNREACHABLE = 10**9
|
| 197 |
+
|
| 198 |
+
|
| 199 |
+
def bfs_distances(size: int, walls: Iterable[tuple[int, int]],
|
| 200 |
+
source: tuple[int, int]) -> dict[tuple[int, int], int]:
|
| 201 |
+
"""Shortest-path distance in cells from ``source`` to every free cell."""
|
| 202 |
+
wallset = walls if isinstance(walls, (set, frozenset)) else set(walls)
|
| 203 |
+
dist = {source: 0}
|
| 204 |
+
queue = deque([source])
|
| 205 |
+
while queue:
|
| 206 |
+
cur = queue.popleft()
|
| 207 |
+
for dr, dc in DIRECTIONS.values():
|
| 208 |
+
nxt = (cur[0] + dr, cur[1] + dc)
|
| 209 |
+
if not (0 <= nxt[0] < size and 0 <= nxt[1] < size):
|
| 210 |
+
continue
|
| 211 |
+
if nxt in wallset or nxt in dist:
|
| 212 |
+
continue
|
| 213 |
+
dist[nxt] = dist[cur] + 1
|
| 214 |
+
queue.append(nxt)
|
| 215 |
+
out: dict[tuple[int, int], int] = {}
|
| 216 |
+
for r in range(size):
|
| 217 |
+
for c in range(size):
|
| 218 |
+
out[(r, c)] = dist.get((r, c), UNREACHABLE)
|
| 219 |
+
return out
|
| 220 |
+
|
| 221 |
+
|
| 222 |
+
def optimal_action_distribution(state: MazeState) -> dict[str, float]:
|
| 223 |
+
"""Uniform mass over every legal move that lies on *a* shortest path.
|
| 224 |
+
|
| 225 |
+
When the goal is unreachable all legal moves tie, matching NanoJev's stated
|
| 226 |
+
contract, so the two label sets describe the same target concept.
|
| 227 |
+
"""
|
| 228 |
+
actions = state.legal_actions()
|
| 229 |
+
if not actions:
|
| 230 |
+
return {}
|
| 231 |
+
dist = bfs_distances(state.size, state.walls, state.goal)
|
| 232 |
+
scored = []
|
| 233 |
+
for action in actions:
|
| 234 |
+
dr, dc = DIRECTIONS[action]
|
| 235 |
+
nxt = (state.position[0] + dr, state.position[1] + dc)
|
| 236 |
+
scored.append((action, dist[nxt]))
|
| 237 |
+
best = min(value for _, value in scored)
|
| 238 |
+
if best >= UNREACHABLE:
|
| 239 |
+
return {action: 1.0 / len(actions) for action in actions}
|
| 240 |
+
winners = [action for action, value in scored if value == best]
|
| 241 |
+
return {action: (1.0 / len(winners) if action in winners else 0.0) for action in actions}
|
| 242 |
+
|
| 243 |
+
|
| 244 |
+
DISTANCE_LEVELS: tuple[str, ...] = (
|
| 245 |
+
"The goal is on this cell already.",
|
| 246 |
+
"The goal is within two moves.",
|
| 247 |
+
"The goal is within five moves.",
|
| 248 |
+
"The goal is within ten moves.",
|
| 249 |
+
"The goal is within twenty-five moves.",
|
| 250 |
+
"The goal is within sixty moves.",
|
| 251 |
+
"The goal is more than sixty moves away or cannot be reached.",
|
| 252 |
+
)
|
| 253 |
+
_LEVEL_BOUNDS = (0, 2, 5, 10, 25, 60)
|
| 254 |
+
|
| 255 |
+
|
| 256 |
+
def distance_level(steps: int) -> int:
|
| 257 |
+
for index, bound in enumerate(_LEVEL_BOUNDS):
|
| 258 |
+
if steps <= bound:
|
| 259 |
+
return index
|
| 260 |
+
return len(DISTANCE_LEVELS) - 1
|
| 261 |
+
|
| 262 |
+
|
| 263 |
+
# --------------------------------------------------------------------------
|
| 264 |
+
# Views
|
| 265 |
+
# --------------------------------------------------------------------------
|
| 266 |
+
|
| 267 |
+
def render_ascii(state: MazeState) -> str:
|
| 268 |
+
"""NanoJev-compatible full-board ASCII rendering (no cropping)."""
|
| 269 |
+
rows = []
|
| 270 |
+
for row in range(state.size):
|
| 271 |
+
line = []
|
| 272 |
+
for col in range(state.size):
|
| 273 |
+
cell = (row, col)
|
| 274 |
+
char = "#" if cell in state.walls else "."
|
| 275 |
+
if cell == state.goal:
|
| 276 |
+
char = "G"
|
| 277 |
+
if cell == state.position:
|
| 278 |
+
char = "@" if state.position == state.goal else "A"
|
| 279 |
+
line.append(char)
|
| 280 |
+
rows.append("".join(line))
|
| 281 |
+
return (
|
| 282 |
+
f"Maze {state.size}x{state.size}. Coordinates are zero-based (row,column). "
|
| 283 |
+
f"Agent=({state.position[0]},{state.position[1]}), "
|
| 284 |
+
f"goal=({state.goal[0]},{state.goal[1]}).\n"
|
| 285 |
+
"#=wall, .=free, A=agent, G=goal, @=agent at goal. "
|
| 286 |
+
"Move one cell north/east/south/west; no diagonals or wraparound. "
|
| 287 |
+
"Stop on the goal.\n" + "\n".join(rows)
|
| 288 |
+
)
|
| 289 |
+
|
| 290 |
+
|
| 291 |
+
def render_header(state: MazeState) -> str:
|
| 292 |
+
"""The text a grid-aware model sees: framing only, no flattened board."""
|
| 293 |
+
return (
|
| 294 |
+
f"Maze {state.size}x{state.size}. Coordinates are zero-based (row,column). "
|
| 295 |
+
f"Agent=({state.position[0]},{state.position[1]}), "
|
| 296 |
+
f"goal=({state.goal[0]},{state.goal[1]}). Topology={state.topology}. "
|
| 297 |
+
"Move one cell north/east/south/west; no diagonals or wraparound. "
|
| 298 |
+
"Stop on the goal."
|
| 299 |
+
)
|
| 300 |
+
|
| 301 |
+
|
| 302 |
+
# --------------------------------------------------------------------------
|
| 303 |
+
# Question profiles (the Jev decision contract)
|
| 304 |
+
# --------------------------------------------------------------------------
|
| 305 |
+
|
| 306 |
+
def question_profile(state: MazeState, *, include_planning: bool = True) -> dict:
|
| 307 |
+
"""Build the question map plus oracle targets for one maze state."""
|
| 308 |
+
actions = state.legal_actions()
|
| 309 |
+
dist_from_goal = bfs_distances(state.size, state.walls, state.goal)
|
| 310 |
+
questions: dict[str, dict] = {}
|
| 311 |
+
targets: dict[str, dict] = {}
|
| 312 |
+
|
| 313 |
+
if len(actions) >= 2:
|
| 314 |
+
criteria, anchors = {}, {}
|
| 315 |
+
for action in actions:
|
| 316 |
+
dr, dc = DIRECTIONS[action]
|
| 317 |
+
cell = (state.position[0] + dr, state.position[1] + dc)
|
| 318 |
+
criteria[action] = f"Move {action} to ({cell[0]},{cell[1]})."
|
| 319 |
+
anchors[action] = cell
|
| 320 |
+
questions["action"] = {
|
| 321 |
+
"type": "choice",
|
| 322 |
+
"instructions": (
|
| 323 |
+
"Choose a legal next move on a shortest path to the goal. "
|
| 324 |
+
"All equally short next moves tie; if the goal is unreachable, "
|
| 325 |
+
"all legal moves tie."
|
| 326 |
+
),
|
| 327 |
+
"criteria": criteria,
|
| 328 |
+
"anchors": anchors,
|
| 329 |
+
}
|
| 330 |
+
targets["action"] = {"probs": optimal_action_distribution(state)}
|
| 331 |
+
|
| 332 |
+
for action in ACTIONS:
|
| 333 |
+
dr, dc = DIRECTIONS[action]
|
| 334 |
+
cell = (state.position[0] + dr, state.position[1] + dc)
|
| 335 |
+
questions[f"clear_{action}"] = {
|
| 336 |
+
"type": "boolean",
|
| 337 |
+
"instructions": (
|
| 338 |
+
f"Moving {action} from ({state.position[0]},{state.position[1]}) "
|
| 339 |
+
f"reaches ({cell[0]},{cell[1]}), which is inside the maze and not a wall."
|
| 340 |
+
),
|
| 341 |
+
"anchors": {"true": cell},
|
| 342 |
+
}
|
| 343 |
+
targets[f"clear_{action}"] = {"probs": {"true": float(state.free(cell)),
|
| 344 |
+
"false": float(not state.free(cell))}}
|
| 345 |
+
|
| 346 |
+
if include_planning:
|
| 347 |
+
steps = dist_from_goal[state.position]
|
| 348 |
+
questions["distance"] = {
|
| 349 |
+
"type": "score",
|
| 350 |
+
"instructions": "How far is the goal from the agent's current cell?",
|
| 351 |
+
"criteria": list(DISTANCE_LEVELS),
|
| 352 |
+
}
|
| 353 |
+
level = distance_level(steps if steps < UNREACHABLE else UNREACHABLE)
|
| 354 |
+
targets["distance"] = {"level": level, "steps": min(steps, 9999)}
|
| 355 |
+
|
| 356 |
+
questions["solvable"] = {
|
| 357 |
+
"type": "boolean",
|
| 358 |
+
"instructions": "A path of open cells connects the agent to the goal.",
|
| 359 |
+
}
|
| 360 |
+
reachable = dist_from_goal[state.position] < UNREACHABLE
|
| 361 |
+
targets["solvable"] = {"probs": {"true": float(reachable),
|
| 362 |
+
"false": float(not reachable)}}
|
| 363 |
+
return {"questions": questions, "targets": targets}
|
|
@@ -0,0 +1,386 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Snake environment with a survival-aware oracle.
|
| 2 |
+
|
| 3 |
+
NanoJev labels Snake with a purely local rule: avoid an immediate collision,
|
| 4 |
+
then minimise Manhattan distance to the current food. That rule walks the
|
| 5 |
+
snake into pockets it cannot leave, which is why its recorded episodes stall in
|
| 6 |
+
the twenties. The oracle here keeps the same one-step action interface but
|
| 7 |
+
scores each move by whether the snake can still reach its own tail afterwards
|
| 8 |
+
and how much free space the move leaves behind -- the standard strong heuristic
|
| 9 |
+
for Snake -- so the imitation target is worth imitating.
|
| 10 |
+
"""
|
| 11 |
+
from __future__ import annotations
|
| 12 |
+
|
| 13 |
+
import random
|
| 14 |
+
from collections import deque
|
| 15 |
+
from dataclasses import dataclass, field
|
| 16 |
+
from typing import Optional
|
| 17 |
+
|
| 18 |
+
DIRECTIONS: dict[str, tuple[int, int]] = {
|
| 19 |
+
"north": (-1, 0),
|
| 20 |
+
"east": (0, 1),
|
| 21 |
+
"south": (1, 0),
|
| 22 |
+
"west": (0, -1),
|
| 23 |
+
}
|
| 24 |
+
ACTIONS: tuple[str, ...] = ("north", "east", "south", "west")
|
| 25 |
+
OPPOSITE = {"north": "south", "south": "north", "east": "west", "west": "east"}
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
@dataclass
|
| 29 |
+
class SnakeState:
|
| 30 |
+
size: int
|
| 31 |
+
body: list[tuple[int, int]] # head first, tail last
|
| 32 |
+
direction: str
|
| 33 |
+
food: tuple[int, int]
|
| 34 |
+
rng_state: int
|
| 35 |
+
steps: int = 0
|
| 36 |
+
eaten: int = 0
|
| 37 |
+
alive: bool = True
|
| 38 |
+
death: Optional[str] = None
|
| 39 |
+
history: list[str] = field(default_factory=list)
|
| 40 |
+
|
| 41 |
+
@property
|
| 42 |
+
def head(self) -> tuple[int, int]:
|
| 43 |
+
return self.body[0]
|
| 44 |
+
|
| 45 |
+
def clone(self) -> "SnakeState":
|
| 46 |
+
return SnakeState(self.size, list(self.body), self.direction, self.food,
|
| 47 |
+
self.rng_state, self.steps, self.eaten, self.alive,
|
| 48 |
+
self.death, list(self.history))
|
| 49 |
+
|
| 50 |
+
def in_bounds(self, cell: tuple[int, int]) -> bool:
|
| 51 |
+
return 0 <= cell[0] < self.size and 0 <= cell[1] < self.size
|
| 52 |
+
|
| 53 |
+
def candidate_actions(self) -> list[str]:
|
| 54 |
+
"""Every non-reversing action. Fatal moves stay on the list on purpose."""
|
| 55 |
+
if len(self.body) == 1:
|
| 56 |
+
return list(ACTIONS)
|
| 57 |
+
return [a for a in ACTIONS if a != OPPOSITE[self.direction]]
|
| 58 |
+
|
| 59 |
+
def step(self, action: str) -> None:
|
| 60 |
+
if not self.alive:
|
| 61 |
+
return
|
| 62 |
+
if len(self.body) > 1 and action == OPPOSITE[self.direction]:
|
| 63 |
+
action = self.direction # A reversal is not offered; treat it as "keep going".
|
| 64 |
+
dr, dc = DIRECTIONS[action]
|
| 65 |
+
head = (self.head[0] + dr, self.head[1] + dc)
|
| 66 |
+
self.steps += 1
|
| 67 |
+
self.direction = action
|
| 68 |
+
self.history.append(action)
|
| 69 |
+
if not self.in_bounds(head):
|
| 70 |
+
self.alive, self.death = False, "wall"
|
| 71 |
+
return
|
| 72 |
+
eats = head == self.food
|
| 73 |
+
# The tail vacates unless the snake eats, so entering it is legal.
|
| 74 |
+
occupied = set(self.body if eats else self.body[:-1])
|
| 75 |
+
if head in occupied:
|
| 76 |
+
self.alive, self.death = False, "body"
|
| 77 |
+
return
|
| 78 |
+
self.body.insert(0, head)
|
| 79 |
+
if eats:
|
| 80 |
+
self.eaten += 1
|
| 81 |
+
self._respawn_food()
|
| 82 |
+
else:
|
| 83 |
+
self.body.pop()
|
| 84 |
+
|
| 85 |
+
def _next_random(self) -> int:
|
| 86 |
+
# SplitMix64: reproducible, and the stream never enters the model's view.
|
| 87 |
+
self.rng_state = (self.rng_state + 0x9E3779B97F4A7C15) & 0xFFFFFFFFFFFFFFFF
|
| 88 |
+
z = self.rng_state
|
| 89 |
+
z = ((z ^ (z >> 30)) * 0xBF58476D1CE4E5B9) & 0xFFFFFFFFFFFFFFFF
|
| 90 |
+
z = ((z ^ (z >> 27)) * 0x94D049BB133111EB) & 0xFFFFFFFFFFFFFFFF
|
| 91 |
+
return z ^ (z >> 31)
|
| 92 |
+
|
| 93 |
+
def _respawn_food(self) -> None:
|
| 94 |
+
occupied = set(self.body)
|
| 95 |
+
empty = [(r, c) for r in range(self.size) for c in range(self.size)
|
| 96 |
+
if (r, c) not in occupied]
|
| 97 |
+
if not empty:
|
| 98 |
+
self.food = self.head # Board is full; the episode is a win.
|
| 99 |
+
return
|
| 100 |
+
self.food = empty[self._next_random() % len(empty)]
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
def new_game(size: int = 12, seed: int = 0, length: int = 3) -> SnakeState:
|
| 104 |
+
rng = random.Random(seed)
|
| 105 |
+
mid = size // 2
|
| 106 |
+
body = [(mid, mid - i) for i in range(length)]
|
| 107 |
+
state = SnakeState(size=size, body=body, direction="east", food=(0, 0),
|
| 108 |
+
rng_state=(seed * 0x2545F4914F6CDD1D) & 0xFFFFFFFFFFFFFFFF)
|
| 109 |
+
occupied = set(body)
|
| 110 |
+
empty = [(r, c) for r in range(size) for c in range(size) if (r, c) not in occupied]
|
| 111 |
+
state.food = empty[rng.randrange(len(empty))]
|
| 112 |
+
return state
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
def random_state(size: int, length: int, rng: random.Random,
|
| 116 |
+
attempts: int = 8) -> SnakeState | None:
|
| 117 |
+
"""Build a valid snake of roughly ``length`` as a self-avoiding walk.
|
| 118 |
+
|
| 119 |
+
Reaching cramped positions by playing the oracle takes thousands of steps
|
| 120 |
+
per sample and still almost never produces them -- a good policy avoids
|
| 121 |
+
them by construction. Sampling body shapes directly is cheap and covers the
|
| 122 |
+
tight configurations where the survival questions actually carry signal.
|
| 123 |
+
The walk prefers neighbours with fewer free neighbours (Warnsdorff's rule),
|
| 124 |
+
which produces bodies that hug themselves rather than sprawl.
|
| 125 |
+
"""
|
| 126 |
+
length = max(2, min(length, size * size - 2))
|
| 127 |
+
for _ in range(attempts):
|
| 128 |
+
start = (rng.randrange(size), rng.randrange(size))
|
| 129 |
+
body = [start]
|
| 130 |
+
occupied = {start}
|
| 131 |
+
while len(body) < length:
|
| 132 |
+
head = body[-1]
|
| 133 |
+
options = []
|
| 134 |
+
for dr, dc in DIRECTIONS.values():
|
| 135 |
+
cell = (head[0] + dr, head[1] + dc)
|
| 136 |
+
if not (0 <= cell[0] < size and 0 <= cell[1] < size):
|
| 137 |
+
continue
|
| 138 |
+
if cell in occupied:
|
| 139 |
+
continue
|
| 140 |
+
free = sum(
|
| 141 |
+
1 for dr2, dc2 in DIRECTIONS.values()
|
| 142 |
+
if 0 <= cell[0] + dr2 < size and 0 <= cell[1] + dc2 < size
|
| 143 |
+
and (cell[0] + dr2, cell[1] + dc2) not in occupied)
|
| 144 |
+
options.append((free, cell))
|
| 145 |
+
if not options:
|
| 146 |
+
break
|
| 147 |
+
fewest = min(o[0] for o in options)
|
| 148 |
+
cell = rng.choice([c for f, c in options if f == fewest])
|
| 149 |
+
body.append(cell)
|
| 150 |
+
occupied.add(cell)
|
| 151 |
+
if len(body) < max(2, length // 2):
|
| 152 |
+
continue
|
| 153 |
+
# The walk was grown tail-first, so reverse it to put the head at index 0.
|
| 154 |
+
body.reverse()
|
| 155 |
+
empty = [(r, c) for r in range(size) for c in range(size)
|
| 156 |
+
if (r, c) not in occupied]
|
| 157 |
+
if not empty:
|
| 158 |
+
continue
|
| 159 |
+
head, neck = body[0], body[1]
|
| 160 |
+
direction = next((name for name, (dr, dc) in DIRECTIONS.items()
|
| 161 |
+
if (neck[0] + dr, neck[1] + dc) == head), "east")
|
| 162 |
+
return SnakeState(size=size, body=body, direction=direction,
|
| 163 |
+
food=rng.choice(empty),
|
| 164 |
+
rng_state=rng.randrange(1, 2 ** 63))
|
| 165 |
+
return None
|
| 166 |
+
|
| 167 |
+
|
| 168 |
+
# --------------------------------------------------------------------------
|
| 169 |
+
# Oracle
|
| 170 |
+
# --------------------------------------------------------------------------
|
| 171 |
+
|
| 172 |
+
def _flood_fill(size: int, blocked: set[tuple[int, int]],
|
| 173 |
+
source: tuple[int, int]) -> set[tuple[int, int]]:
|
| 174 |
+
if source in blocked or not (0 <= source[0] < size and 0 <= source[1] < size):
|
| 175 |
+
return set()
|
| 176 |
+
seen = {source}
|
| 177 |
+
queue = deque([source])
|
| 178 |
+
while queue:
|
| 179 |
+
cur = queue.popleft()
|
| 180 |
+
for dr, dc in DIRECTIONS.values():
|
| 181 |
+
nxt = (cur[0] + dr, cur[1] + dc)
|
| 182 |
+
if (0 <= nxt[0] < size and 0 <= nxt[1] < size
|
| 183 |
+
and nxt not in blocked and nxt not in seen):
|
| 184 |
+
seen.add(nxt)
|
| 185 |
+
queue.append(nxt)
|
| 186 |
+
return seen
|
| 187 |
+
|
| 188 |
+
|
| 189 |
+
def _bfs_len(size: int, blocked: set[tuple[int, int]], source: tuple[int, int],
|
| 190 |
+
target: tuple[int, int]) -> int:
|
| 191 |
+
if source == target:
|
| 192 |
+
return 0
|
| 193 |
+
dist = {source: 0}
|
| 194 |
+
queue = deque([source])
|
| 195 |
+
while queue:
|
| 196 |
+
cur = queue.popleft()
|
| 197 |
+
for dr, dc in DIRECTIONS.values():
|
| 198 |
+
nxt = (cur[0] + dr, cur[1] + dc)
|
| 199 |
+
if not (0 <= nxt[0] < size and 0 <= nxt[1] < size) or nxt in dist:
|
| 200 |
+
continue
|
| 201 |
+
if nxt in blocked and nxt != target:
|
| 202 |
+
continue
|
| 203 |
+
dist[nxt] = dist[cur] + 1
|
| 204 |
+
if nxt == target:
|
| 205 |
+
return dist[nxt]
|
| 206 |
+
queue.append(nxt)
|
| 207 |
+
return 10**9
|
| 208 |
+
|
| 209 |
+
|
| 210 |
+
def evaluate_action(state: SnakeState, action: str) -> dict:
|
| 211 |
+
"""Score one action without mutating ``state``.
|
| 212 |
+
|
| 213 |
+
Returns immediate safety, the free area the new head can still reach, whether
|
| 214 |
+
the tail stays reachable (the snake can always follow it out of a pocket),
|
| 215 |
+
and the post-move path length to the food.
|
| 216 |
+
"""
|
| 217 |
+
dr, dc = DIRECTIONS[action]
|
| 218 |
+
head = (state.head[0] + dr, state.head[1] + dc)
|
| 219 |
+
size = state.size
|
| 220 |
+
if not state.in_bounds(head):
|
| 221 |
+
return {"safe": False, "reason": "wall", "space": 0,
|
| 222 |
+
"tail_reachable": False, "food_distance": 10**9, "eats": False}
|
| 223 |
+
eats = head == state.food
|
| 224 |
+
occupied = set(state.body if eats else state.body[:-1])
|
| 225 |
+
if head in occupied:
|
| 226 |
+
return {"safe": False, "reason": "body", "space": 0,
|
| 227 |
+
"tail_reachable": False, "food_distance": 10**9, "eats": False}
|
| 228 |
+
|
| 229 |
+
body_after = [head] + (state.body if eats else state.body[:-1])
|
| 230 |
+
blocked = set(body_after)
|
| 231 |
+
reachable = _flood_fill(size, blocked - {head}, head)
|
| 232 |
+
tail = body_after[-1]
|
| 233 |
+
# The tail vacates next tick, so treat it as passable when testing escape.
|
| 234 |
+
tail_reachable = _bfs_len(size, blocked - {head, tail}, head, tail) < 10**9
|
| 235 |
+
food_distance = (0 if eats else
|
| 236 |
+
_bfs_len(size, blocked - {head}, head, state.food))
|
| 237 |
+
return {"safe": True, "reason": None, "space": len(reachable),
|
| 238 |
+
"tail_reachable": tail_reachable, "food_distance": food_distance,
|
| 239 |
+
"eats": eats, "length_after": len(body_after)}
|
| 240 |
+
|
| 241 |
+
|
| 242 |
+
def oracle_action_distribution(state: SnakeState) -> tuple[dict[str, float], dict]:
|
| 243 |
+
"""Uniform mass over the moves a strong player would consider.
|
| 244 |
+
|
| 245 |
+
Preference order: survive this tick, keep the tail reachable, keep enough
|
| 246 |
+
room for the whole body, then close on the food. Ties share mass evenly.
|
| 247 |
+
"""
|
| 248 |
+
actions = state.candidate_actions()
|
| 249 |
+
if not actions:
|
| 250 |
+
return {}, {}
|
| 251 |
+
scored = {a: evaluate_action(state, a) for a in actions}
|
| 252 |
+
survivors = [a for a in actions if scored[a]["safe"]]
|
| 253 |
+
if not survivors:
|
| 254 |
+
return {a: 1.0 / len(actions) for a in actions}, scored
|
| 255 |
+
|
| 256 |
+
pool = [a for a in survivors if scored[a]["tail_reachable"]] or survivors
|
| 257 |
+
need = len(state.body) + 1
|
| 258 |
+
roomy = [a for a in pool if scored[a]["space"] >= need]
|
| 259 |
+
pool = roomy or pool
|
| 260 |
+
|
| 261 |
+
reachable_food = [a for a in pool if scored[a]["food_distance"] < 10**9]
|
| 262 |
+
if reachable_food:
|
| 263 |
+
best = min(scored[a]["food_distance"] for a in reachable_food)
|
| 264 |
+
winners = [a for a in reachable_food if scored[a]["food_distance"] == best]
|
| 265 |
+
else:
|
| 266 |
+
best = max(scored[a]["space"] for a in pool)
|
| 267 |
+
winners = [a for a in pool if scored[a]["space"] == best]
|
| 268 |
+
|
| 269 |
+
probs = {a: (1.0 / len(winners) if a in winners else 0.0) for a in actions}
|
| 270 |
+
return probs, scored
|
| 271 |
+
|
| 272 |
+
|
| 273 |
+
# --------------------------------------------------------------------------
|
| 274 |
+
# Views
|
| 275 |
+
# --------------------------------------------------------------------------
|
| 276 |
+
|
| 277 |
+
def render_ascii(state: SnakeState) -> str:
|
| 278 |
+
grid = [["." for _ in range(state.size)] for _ in range(state.size)]
|
| 279 |
+
for idx, (r, c) in enumerate(state.body):
|
| 280 |
+
grid[r][c] = "H" if idx == 0 else "o"
|
| 281 |
+
fr, fc = state.food
|
| 282 |
+
if grid[fr][fc] == ".":
|
| 283 |
+
grid[fr][fc] = "*"
|
| 284 |
+
rows = ["".join(row) for row in grid]
|
| 285 |
+
return (
|
| 286 |
+
f"Snake {state.size}x{state.size}. Coordinates are zero-based (row,column). "
|
| 287 |
+
f"Head=({state.head[0]},{state.head[1]}), facing {state.direction}, "
|
| 288 |
+
f"length={len(state.body)}, food=({fr},{fc}).\n"
|
| 289 |
+
"H=head, o=body, *=food, .=empty. Move one cell north/east/south/west; "
|
| 290 |
+
"reversing onto your own neck is not offered. Hitting a wall or your body ends the game.\n"
|
| 291 |
+
+ "\n".join(rows)
|
| 292 |
+
)
|
| 293 |
+
|
| 294 |
+
|
| 295 |
+
def render_header(state: SnakeState) -> str:
|
| 296 |
+
return (
|
| 297 |
+
f"Snake {state.size}x{state.size}. Coordinates are zero-based (row,column). "
|
| 298 |
+
f"Head=({state.head[0]},{state.head[1]}), facing {state.direction}, "
|
| 299 |
+
f"length={len(state.body)}, food=({state.food[0]},{state.food[1]}). "
|
| 300 |
+
"Move one cell north/east/south/west; reversing onto your own neck is not "
|
| 301 |
+
"offered. Hitting a wall or your body ends the game."
|
| 302 |
+
)
|
| 303 |
+
|
| 304 |
+
|
| 305 |
+
def question_profile(state: SnakeState) -> dict:
|
| 306 |
+
actions = state.candidate_actions()
|
| 307 |
+
probs, scored = oracle_action_distribution(state)
|
| 308 |
+
questions: dict[str, dict] = {}
|
| 309 |
+
targets: dict[str, dict] = {}
|
| 310 |
+
|
| 311 |
+
if len(actions) >= 2:
|
| 312 |
+
criteria, anchors = {}, {}
|
| 313 |
+
for action in actions:
|
| 314 |
+
dr, dc = DIRECTIONS[action]
|
| 315 |
+
cell = (state.head[0] + dr, state.head[1] + dc)
|
| 316 |
+
criteria[action] = f"Move {action} to ({cell[0]},{cell[1]})."
|
| 317 |
+
anchors[action] = cell
|
| 318 |
+
questions["action"] = {
|
| 319 |
+
"type": "choice",
|
| 320 |
+
"instructions": (
|
| 321 |
+
"Choose the next move that keeps the snake alive and makes the most "
|
| 322 |
+
"progress towards eating the food. Prefer moves that leave an escape "
|
| 323 |
+
"route; equally good moves tie."
|
| 324 |
+
),
|
| 325 |
+
"criteria": criteria,
|
| 326 |
+
"anchors": anchors,
|
| 327 |
+
}
|
| 328 |
+
targets["action"] = {"probs": probs}
|
| 329 |
+
|
| 330 |
+
for action in actions:
|
| 331 |
+
dr, dc = DIRECTIONS[action]
|
| 332 |
+
cell = (state.head[0] + dr, state.head[1] + dc)
|
| 333 |
+
info = scored[action]
|
| 334 |
+
questions[f"safe_{action}"] = {
|
| 335 |
+
"type": "boolean",
|
| 336 |
+
"instructions": (
|
| 337 |
+
f"Moving {action} to ({cell[0]},{cell[1]}) does not immediately end the game."
|
| 338 |
+
),
|
| 339 |
+
"anchors": {"true": cell},
|
| 340 |
+
}
|
| 341 |
+
targets[f"safe_{action}"] = {"probs": {"true": float(info["safe"]),
|
| 342 |
+
"false": float(not info["safe"])}}
|
| 343 |
+
questions[f"escape_{action}"] = {
|
| 344 |
+
"type": "boolean",
|
| 345 |
+
"instructions": (
|
| 346 |
+
f"After moving {action} to ({cell[0]},{cell[1]}) the snake can still "
|
| 347 |
+
"reach its own tail, so it is not sealed into a dead pocket."
|
| 348 |
+
),
|
| 349 |
+
"anchors": {"true": cell},
|
| 350 |
+
}
|
| 351 |
+
ok = info["safe"] and info["tail_reachable"]
|
| 352 |
+
targets[f"escape_{action}"] = {"probs": {"true": float(ok), "false": float(not ok)}}
|
| 353 |
+
|
| 354 |
+
best_space = max((scored[a]["space"] for a in actions), default=0)
|
| 355 |
+
questions["room"] = {
|
| 356 |
+
"type": "score",
|
| 357 |
+
"instructions": "How much open room does the snake still have to manoeuvre in?",
|
| 358 |
+
"criteria": list(ROOM_LEVELS),
|
| 359 |
+
}
|
| 360 |
+
targets["room"] = {"level": room_level(best_space, len(state.body))}
|
| 361 |
+
return {"questions": questions, "targets": targets}
|
| 362 |
+
|
| 363 |
+
|
| 364 |
+
ROOM_LEVELS: tuple[str, ...] = (
|
| 365 |
+
"Trapped: less open room than the snake is long.",
|
| 366 |
+
"Barely enough room to fit the snake; a dead end is likely.",
|
| 367 |
+
"Limited room, under three times the snake's length.",
|
| 368 |
+
"Comfortable room, several times the snake's length.",
|
| 369 |
+
"Wide open, more than eight times the snake's length.",
|
| 370 |
+
)
|
| 371 |
+
|
| 372 |
+
|
| 373 |
+
def room_level(space: int, length: int) -> int:
|
| 374 |
+
"""Rate open room *relative to the snake's own length*.
|
| 375 |
+
|
| 376 |
+
An absolute fraction of the board is the wrong scale: a short snake on a
|
| 377 |
+
large board always scores the top level regardless of how it is placed,
|
| 378 |
+
which collapses the label to a constant and teaches nothing. What actually
|
| 379 |
+
decides whether the snake survives is how much room it has compared with
|
| 380 |
+
the space its own body needs.
|
| 381 |
+
"""
|
| 382 |
+
ratio = space / max(1, length)
|
| 383 |
+
for index, bound in enumerate((1.0, 2.0, 3.0, 8.0)):
|
| 384 |
+
if ratio <= bound:
|
| 385 |
+
return index
|
| 386 |
+
return len(ROOM_LEVELS) - 1
|
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Jevon: a small typed-decision model with a recurrent spatial planner.
|
| 2 |
+
|
| 3 |
+
Copyright (C) 2026 Lewis
|
| 4 |
+
|
| 5 |
+
This program is free software: you can redistribute it and/or modify it under
|
| 6 |
+
the terms of the GNU Affero General Public License as published by the Free
|
| 7 |
+
Software Foundation, either version 3 of the License, or (at your option) any
|
| 8 |
+
later version.
|
| 9 |
+
|
| 10 |
+
It is distributed in the hope that it will be useful, but WITHOUT ANY
|
| 11 |
+
WARRANTY; without even the implied warranty of MERCHANTABILITY or FITNESS FOR
|
| 12 |
+
A PARTICULAR PURPOSE. See the GNU Affero General Public License for more
|
| 13 |
+
details, in LICENSE or at <https://www.gnu.org/licenses/>.
|
| 14 |
+
|
| 15 |
+
A commercial licence without the copyleft terms is available; see
|
| 16 |
+
COMMERCIAL-LICENSE.md.
|
| 17 |
+
"""
|
| 18 |
+
|
| 19 |
+
__version__ = "0.1.0"
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
def __getattr__(name): # PEP 562
|
| 23 |
+
"""Expose `from_pretrained` without making `import jevon` import torch.
|
| 24 |
+
|
| 25 |
+
The Hub helper is the first thing a reader of the model card reaches for,
|
| 26 |
+
so it should be `from jevon import from_pretrained`. Importing `.hub`
|
| 27 |
+
eagerly here would pull torch into every `import jevon`, including the
|
| 28 |
+
ones that only want `__version__`.
|
| 29 |
+
"""
|
| 30 |
+
if name in ("from_pretrained", "REPO_ID", "CHECKPOINT", "DEFAULT_WEIGHTS"):
|
| 31 |
+
from . import hub
|
| 32 |
+
return getattr(hub, name)
|
| 33 |
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
@@ -0,0 +1,380 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Sampling, paraphrasing, and collation of decision records.
|
| 2 |
+
|
| 3 |
+
Both games are procedural, so training draws from an endless seeded stream
|
| 4 |
+
rather than a fixed file. Splits are decided by the map seed: a maze or Snake
|
| 5 |
+
episode belongs to exactly one split for the whole run, and evaluation seeds are
|
| 6 |
+
never sampled during training. Board sizes held out for OOD are never sampled
|
| 7 |
+
during training either, which is what makes the size-generalisation claim mean
|
| 8 |
+
something.
|
| 9 |
+
"""
|
| 10 |
+
from __future__ import annotations
|
| 11 |
+
|
| 12 |
+
import json
|
| 13 |
+
import random
|
| 14 |
+
from dataclasses import dataclass, field
|
| 15 |
+
from pathlib import Path
|
| 16 |
+
from typing import Iterator, Optional, Sequence
|
| 17 |
+
|
| 18 |
+
import torch
|
| 19 |
+
|
| 20 |
+
from . import grid as gridmod
|
| 21 |
+
from .modeling import TYPE_INDEX
|
| 22 |
+
|
| 23 |
+
SPLITS = ("train", "val", "test")
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
def split_for_seed(seed: int) -> str:
|
| 27 |
+
"""Deterministic, collision-free assignment of a map seed to a split."""
|
| 28 |
+
bucket = seed % 10
|
| 29 |
+
if bucket < 8:
|
| 30 |
+
return "train"
|
| 31 |
+
return "val" if bucket == 8 else "test"
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
@dataclass
|
| 35 |
+
class QuestionSample:
|
| 36 |
+
name: str
|
| 37 |
+
qtype: str
|
| 38 |
+
instructions: str
|
| 39 |
+
candidate_ids: list[str]
|
| 40 |
+
candidate_texts: list[str]
|
| 41 |
+
anchors: list[Optional[tuple[int, int]]]
|
| 42 |
+
probs: list[float]
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
@dataclass
|
| 46 |
+
class StateSample:
|
| 47 |
+
uid: str
|
| 48 |
+
game: str
|
| 49 |
+
split: str
|
| 50 |
+
size: int
|
| 51 |
+
topology: str
|
| 52 |
+
board: torch.Tensor
|
| 53 |
+
focus: tuple[int, int]
|
| 54 |
+
target: tuple[int, int]
|
| 55 |
+
header: str
|
| 56 |
+
ascii_board: str
|
| 57 |
+
questions: list[QuestionSample] = field(default_factory=list)
|
| 58 |
+
# Dense supervision for the planner: BFS distance from ``target`` to every
|
| 59 |
+
# cell, -1 where blocked or unreachable. One question per state is far too
|
| 60 |
+
# little signal to teach a several-hundred-step recurrence; this is ~size^2.
|
| 61 |
+
dist_field: Optional[torch.Tensor] = None
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
# --------------------------------------------------------------------------
|
| 65 |
+
# Instruction paraphrasing
|
| 66 |
+
# --------------------------------------------------------------------------
|
| 67 |
+
# Jevon-S learns its vocabulary from this corpus, so it would happily overfit a
|
| 68 |
+
# single phrasing of every question. Paraphrasing at sample time forces the
|
| 69 |
+
# decision to depend on the board rather than on a memorised string.
|
| 70 |
+
|
| 71 |
+
_PREFIXES = ("", "Task: ", "Decide: ", "Question: ", "Please decide. ", "Read the board. ")
|
| 72 |
+
_SUFFIXES = ("", " Answer with a probability for each option.",
|
| 73 |
+
" Weigh the options carefully.", " Ties should share probability evenly.",
|
| 74 |
+
" Report your confidence.")
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
def paraphrase(text: str, rng: random.Random) -> str:
|
| 78 |
+
return f"{rng.choice(_PREFIXES)}{text}{rng.choice(_SUFFIXES)}"
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
# --------------------------------------------------------------------------
|
| 82 |
+
# Samplers
|
| 83 |
+
# --------------------------------------------------------------------------
|
| 84 |
+
|
| 85 |
+
def _questions_from_profile(profile: dict, rng: random.Random,
|
| 86 |
+
augment: bool) -> list[QuestionSample]:
|
| 87 |
+
out: list[QuestionSample] = []
|
| 88 |
+
for name, question in profile["questions"].items():
|
| 89 |
+
target = profile["targets"].get(name)
|
| 90 |
+
if target is None:
|
| 91 |
+
continue
|
| 92 |
+
qtype = question["type"]
|
| 93 |
+
anchors_map = question.get("anchors") or {}
|
| 94 |
+
if qtype == "choice":
|
| 95 |
+
ids = list(question["criteria"])
|
| 96 |
+
texts = [question["criteria"][key] for key in ids]
|
| 97 |
+
probs = [float(target["probs"].get(key, 0.0)) for key in ids]
|
| 98 |
+
anchors = [anchors_map.get(key) for key in ids]
|
| 99 |
+
elif qtype == "boolean":
|
| 100 |
+
# One semantic path, scored against an implicit zero logit.
|
| 101 |
+
ids, texts = ["true"], ["The proposition is true."]
|
| 102 |
+
probs = [float(target["probs"]["false"]), float(target["probs"]["true"])]
|
| 103 |
+
anchors = [anchors_map.get("true")]
|
| 104 |
+
else:
|
| 105 |
+
levels = question["criteria"]
|
| 106 |
+
ids = [str(i) for i in range(len(levels))]
|
| 107 |
+
texts = list(levels)
|
| 108 |
+
probs = [0.0] * len(levels)
|
| 109 |
+
probs[int(target["level"])] = 1.0
|
| 110 |
+
anchors = [None] * len(levels)
|
| 111 |
+
total = sum(probs)
|
| 112 |
+
if total <= 0:
|
| 113 |
+
continue
|
| 114 |
+
probs = [p / total for p in probs]
|
| 115 |
+
instructions = question["instructions"]
|
| 116 |
+
if augment:
|
| 117 |
+
instructions = paraphrase(instructions, rng)
|
| 118 |
+
out.append(QuestionSample(name, qtype, instructions, ids, texts, anchors, probs))
|
| 119 |
+
return out
|
| 120 |
+
|
| 121 |
+
|
| 122 |
+
def sample_maze_state(seed: int, size: int, topology: str, rng: random.Random,
|
| 123 |
+
augment: bool = True) -> Optional[StateSample]:
|
| 124 |
+
from envs import maze as mz
|
| 125 |
+
|
| 126 |
+
state = mz.generate_maze(size, topology, seed)
|
| 127 |
+
# Walk the agent to a random reachable cell so training sees mid-episode
|
| 128 |
+
# states, not only the start. Positions come from BFS, never from rollouts.
|
| 129 |
+
distances = mz.bfs_distances(size, state.walls, state.position)
|
| 130 |
+
reachable = [cell for cell, d in distances.items() if d < mz.UNREACHABLE]
|
| 131 |
+
if len(reachable) > 1:
|
| 132 |
+
state.position = rng.choice(reachable)
|
| 133 |
+
state.visited = {state.position}
|
| 134 |
+
if state.position == state.goal:
|
| 135 |
+
return None
|
| 136 |
+
profile = mz.question_profile(state)
|
| 137 |
+
questions = _questions_from_profile(profile, rng, augment)
|
| 138 |
+
if not questions:
|
| 139 |
+
return None
|
| 140 |
+
board, focus, target = gridmod.encode_maze(state)
|
| 141 |
+
return StateSample(
|
| 142 |
+
dist_field=gridmod.bfs_field(board, target),
|
| 143 |
+
uid=f"maze:{topology}:{size}:{seed}:{state.position[0]}_{state.position[1]}",
|
| 144 |
+
game="maze", split=split_for_seed(seed), size=size, topology=topology,
|
| 145 |
+
board=board, focus=focus, target=target,
|
| 146 |
+
header=mz.render_header(state), ascii_board=mz.render_ascii(state),
|
| 147 |
+
questions=questions)
|
| 148 |
+
|
| 149 |
+
|
| 150 |
+
def sample_snake_state(seed: int, size: int, rng: random.Random,
|
| 151 |
+
augment: bool = True, max_advance: int = 220
|
| 152 |
+
) -> Optional[StateSample]:
|
| 153 |
+
from envs import snake as sk
|
| 154 |
+
|
| 155 |
+
# Half the states come from play, half from directly constructed bodies.
|
| 156 |
+
# Oracle play gives realistic positions but, being a good policy, it almost
|
| 157 |
+
# never produces a cramped board -- which is exactly where the survival and
|
| 158 |
+
# room questions carry their signal. Constructed walks cover that half of
|
| 159 |
+
# the state space at a fraction of the cost of rolling out to it.
|
| 160 |
+
if rng.random() < 0.5:
|
| 161 |
+
target = max(3, int(size * size * rng.uniform(0.04, 0.45)))
|
| 162 |
+
state = sk.random_state(size, target, rng)
|
| 163 |
+
if state is None:
|
| 164 |
+
return None
|
| 165 |
+
else:
|
| 166 |
+
state = sk.new_game(size, seed)
|
| 167 |
+
steps = rng.randrange(0, max_advance)
|
| 168 |
+
for _ in range(steps):
|
| 169 |
+
if not state.alive:
|
| 170 |
+
break
|
| 171 |
+
probs, _ = sk.oracle_action_distribution(state)
|
| 172 |
+
if not probs:
|
| 173 |
+
break
|
| 174 |
+
actions = list(probs)
|
| 175 |
+
if rng.random() < 0.15:
|
| 176 |
+
action = rng.choice(state.candidate_actions())
|
| 177 |
+
else:
|
| 178 |
+
action = max(actions, key=lambda a: probs[a])
|
| 179 |
+
state.step(action)
|
| 180 |
+
if not state.alive or len(state.candidate_actions()) < 1:
|
| 181 |
+
return None
|
| 182 |
+
profile = sk.question_profile(state)
|
| 183 |
+
questions = _questions_from_profile(profile, rng, augment)
|
| 184 |
+
if not questions:
|
| 185 |
+
return None
|
| 186 |
+
board, focus, target = gridmod.encode_snake(state)
|
| 187 |
+
return StateSample(
|
| 188 |
+
dist_field=gridmod.bfs_field(board, target),
|
| 189 |
+
uid=f"snake:{size}:{seed}:{state.steps}", game="snake",
|
| 190 |
+
split=split_for_seed(seed), size=size, topology="snake",
|
| 191 |
+
board=board, focus=focus, target=target,
|
| 192 |
+
header=sk.render_header(state), ascii_board=sk.render_ascii(state),
|
| 193 |
+
questions=questions)
|
| 194 |
+
|
| 195 |
+
|
| 196 |
+
@dataclass
|
| 197 |
+
class SamplerConfig:
|
| 198 |
+
maze_sizes: Sequence[int] = (7, 9, 11, 15, 21, 25, 31)
|
| 199 |
+
maze_topologies: Sequence[str] = ("corridor", "tree", "loops", "random_obstacle")
|
| 200 |
+
snake_sizes: Sequence[int] = (8, 10, 12, 14, 16)
|
| 201 |
+
snake_fraction: float = 0.45
|
| 202 |
+
augment: bool = True
|
| 203 |
+
seed_space: int = 1_000_000
|
| 204 |
+
|
| 205 |
+
|
| 206 |
+
class TaskSampler:
|
| 207 |
+
"""Endless stream of :class:`StateSample` restricted to one split."""
|
| 208 |
+
|
| 209 |
+
def __init__(self, cfg: SamplerConfig, split: str, seed: int = 0):
|
| 210 |
+
self.cfg = cfg
|
| 211 |
+
self.split = split
|
| 212 |
+
self.rng = random.Random(seed)
|
| 213 |
+
|
| 214 |
+
def _draw_seed(self) -> int:
|
| 215 |
+
while True:
|
| 216 |
+
seed = self.rng.randrange(self.cfg.seed_space)
|
| 217 |
+
if split_for_seed(seed) == self.split:
|
| 218 |
+
return seed
|
| 219 |
+
|
| 220 |
+
def __iter__(self) -> Iterator[StateSample]:
|
| 221 |
+
while True:
|
| 222 |
+
sample = self.sample()
|
| 223 |
+
if sample is not None:
|
| 224 |
+
yield sample
|
| 225 |
+
|
| 226 |
+
def sample(self) -> Optional[StateSample]:
|
| 227 |
+
seed = self._draw_seed()
|
| 228 |
+
if self.rng.random() < self.cfg.snake_fraction:
|
| 229 |
+
size = self.rng.choice(list(self.cfg.snake_sizes))
|
| 230 |
+
return sample_snake_state(seed, size, self.rng, self.cfg.augment)
|
| 231 |
+
size = self.rng.choice(list(self.cfg.maze_sizes))
|
| 232 |
+
topology = self.rng.choice(list(self.cfg.maze_topologies))
|
| 233 |
+
return sample_maze_state(seed, size, topology, self.rng, self.cfg.augment)
|
| 234 |
+
|
| 235 |
+
def batch(self, states: int) -> list[StateSample]:
|
| 236 |
+
out: list[StateSample] = []
|
| 237 |
+
while len(out) < states:
|
| 238 |
+
sample = self.sample()
|
| 239 |
+
if sample is not None:
|
| 240 |
+
out.append(sample)
|
| 241 |
+
return out
|
| 242 |
+
|
| 243 |
+
|
| 244 |
+
# --------------------------------------------------------------------------
|
| 245 |
+
# Collation
|
| 246 |
+
# --------------------------------------------------------------------------
|
| 247 |
+
|
| 248 |
+
def collate(states: Sequence[StateSample], tokenizer, *, max_prefix: int = 320,
|
| 249 |
+
max_candidate: int = 48, text_only: bool = False,
|
| 250 |
+
device: str | torch.device = "cpu") -> dict:
|
| 251 |
+
"""Turn states into the tensor batch :meth:`JevonModel.forward` expects."""
|
| 252 |
+
prefixes: list[list[int]] = []
|
| 253 |
+
candidates: list[list[list[int]]] = []
|
| 254 |
+
anchors: list[list[tuple[int, int]]] = []
|
| 255 |
+
targets: list[list[float]] = []
|
| 256 |
+
types: list[int] = []
|
| 257 |
+
state_index: list[int] = []
|
| 258 |
+
focus: list[tuple[int, int]] = []
|
| 259 |
+
target_cell: list[tuple[int, int]] = []
|
| 260 |
+
meta: list[dict] = []
|
| 261 |
+
|
| 262 |
+
for s_idx, state in enumerate(states):
|
| 263 |
+
context = state.ascii_board if text_only else state.header
|
| 264 |
+
for question in state.questions:
|
| 265 |
+
prefix = tokenizer.encode(
|
| 266 |
+
f"{context}\n{question.qtype}\n{question.instructions}",
|
| 267 |
+
add_bos=True, add_eos=True)[:max_prefix]
|
| 268 |
+
prefixes.append(prefix)
|
| 269 |
+
candidates.append([
|
| 270 |
+
tokenizer.encode(text, add_bos=True, add_eos=True)[:max_candidate]
|
| 271 |
+
for text in question.candidate_texts])
|
| 272 |
+
anchors.append([a if a is not None else (-1, -1) for a in question.anchors])
|
| 273 |
+
targets.append(list(question.probs))
|
| 274 |
+
types.append(TYPE_INDEX[question.qtype])
|
| 275 |
+
state_index.append(s_idx)
|
| 276 |
+
focus.append(state.focus)
|
| 277 |
+
target_cell.append(state.target)
|
| 278 |
+
meta.append({"uid": state.uid, "game": state.game, "size": state.size,
|
| 279 |
+
"topology": state.topology, "question": question.name,
|
| 280 |
+
"type": question.qtype, "candidate_ids": question.candidate_ids,
|
| 281 |
+
"probs": list(question.probs)})
|
| 282 |
+
|
| 283 |
+
q = len(prefixes)
|
| 284 |
+
lp = max(len(p) for p in prefixes)
|
| 285 |
+
kmax = max(max(len(c) for c in candidates), max(len(t) for t in targets))
|
| 286 |
+
lc = max(max(len(ids) for ids in group) for group in candidates)
|
| 287 |
+
|
| 288 |
+
prefix_ids = torch.zeros(q, lp, dtype=torch.long)
|
| 289 |
+
cand_ids = torch.zeros(q, kmax, lc, dtype=torch.long)
|
| 290 |
+
cand_valid = torch.zeros(q, kmax, dtype=torch.bool)
|
| 291 |
+
target_probs = torch.zeros(q, kmax)
|
| 292 |
+
anchor_t = torch.full((q, kmax, 2), -1, dtype=torch.long)
|
| 293 |
+
|
| 294 |
+
for i in range(q):
|
| 295 |
+
prefix_ids[i, : len(prefixes[i])] = torch.tensor(prefixes[i])
|
| 296 |
+
for k, ids in enumerate(candidates[i]):
|
| 297 |
+
cand_ids[i, k, : len(ids)] = torch.tensor(ids)
|
| 298 |
+
anchor_t[i, k] = torch.tensor(anchors[i][k])
|
| 299 |
+
# A Boolean has one candidate path but a two-way distribution.
|
| 300 |
+
cand_valid[i, : len(targets[i])] = True
|
| 301 |
+
target_probs[i, : len(targets[i])] = torch.tensor(targets[i])
|
| 302 |
+
|
| 303 |
+
batch = {
|
| 304 |
+
"prefix_ids": prefix_ids,
|
| 305 |
+
"candidate_ids": cand_ids,
|
| 306 |
+
"candidate_valid": cand_valid,
|
| 307 |
+
"target_probs": target_probs,
|
| 308 |
+
"anchors": anchor_t,
|
| 309 |
+
"question_type": torch.tensor(types, dtype=torch.long),
|
| 310 |
+
"question_state": torch.tensor(state_index, dtype=torch.long),
|
| 311 |
+
"focus": torch.tensor(focus, dtype=torch.long),
|
| 312 |
+
"target": torch.tensor(target_cell, dtype=torch.long),
|
| 313 |
+
}
|
| 314 |
+
if not text_only:
|
| 315 |
+
boards, passable, sizes = gridmod.pad_boards([s.board for s in states])
|
| 316 |
+
batch.update(boards=boards, passable=passable, board_size=sizes)
|
| 317 |
+
if all(s.dist_field is not None for s in states):
|
| 318 |
+
batch["dist_field"] = gridmod.pad_fields([s.dist_field for s in states])
|
| 319 |
+
batch = {k: v.to(device) for k, v in batch.items()}
|
| 320 |
+
batch["meta"] = meta
|
| 321 |
+
return batch
|
| 322 |
+
|
| 323 |
+
|
| 324 |
+
def build_corpus(sampler: TaskSampler, states: int = 400) -> list[str]:
|
| 325 |
+
"""Text used to fit the tokenizer vocabulary."""
|
| 326 |
+
corpus: list[str] = []
|
| 327 |
+
for state in sampler.batch(states):
|
| 328 |
+
corpus.append(state.header)
|
| 329 |
+
for question in state.questions:
|
| 330 |
+
corpus.append(question.instructions)
|
| 331 |
+
corpus.extend(question.candidate_texts)
|
| 332 |
+
corpus.extend(_PREFIXES)
|
| 333 |
+
corpus.extend(_SUFFIXES)
|
| 334 |
+
return corpus
|
| 335 |
+
|
| 336 |
+
|
| 337 |
+
def freeze_split(path: str | Path, cfg: SamplerConfig, split: str, states: int,
|
| 338 |
+
seed: int) -> int:
|
| 339 |
+
"""Write a fixed evaluation set so numbers stay comparable across runs."""
|
| 340 |
+
sampler = TaskSampler(cfg, split, seed)
|
| 341 |
+
rows = []
|
| 342 |
+
for state in sampler.batch(states):
|
| 343 |
+
rows.append({
|
| 344 |
+
"uid": state.uid, "game": state.game, "split": state.split,
|
| 345 |
+
"size": state.size, "topology": state.topology,
|
| 346 |
+
"header": state.header, "focus": list(state.focus),
|
| 347 |
+
"target": list(state.target),
|
| 348 |
+
"board": state.board.to(torch.int16).flatten().tolist()
|
| 349 |
+
if state.game == "maze" else state.board.flatten().tolist(),
|
| 350 |
+
"shape": list(state.board.shape),
|
| 351 |
+
"dist_field": state.dist_field.flatten().tolist(),
|
| 352 |
+
"questions": [{
|
| 353 |
+
"name": q.name, "type": q.qtype, "instructions": q.instructions,
|
| 354 |
+
"candidate_ids": q.candidate_ids, "candidate_texts": q.candidate_texts,
|
| 355 |
+
"anchors": [list(a) if a else None for a in q.anchors],
|
| 356 |
+
"probs": q.probs} for q in state.questions],
|
| 357 |
+
})
|
| 358 |
+
Path(path).parent.mkdir(parents=True, exist_ok=True)
|
| 359 |
+
Path(path).write_text("".join(json.dumps(r) + "\n" for r in rows))
|
| 360 |
+
return len(rows)
|
| 361 |
+
|
| 362 |
+
|
| 363 |
+
def load_frozen(path: str | Path) -> list[StateSample]:
|
| 364 |
+
out = []
|
| 365 |
+
for line in Path(path).read_text().splitlines():
|
| 366 |
+
row = json.loads(line)
|
| 367 |
+
board = torch.tensor(row["board"], dtype=torch.float32).view(*row["shape"])
|
| 368 |
+
questions = [QuestionSample(
|
| 369 |
+
q["name"], q["type"], q["instructions"], q["candidate_ids"],
|
| 370 |
+
q["candidate_texts"],
|
| 371 |
+
[tuple(a) if a else None for a in q["anchors"]], q["probs"])
|
| 372 |
+
for q in row["questions"]]
|
| 373 |
+
shape = row["shape"][1:]
|
| 374 |
+
field_t = (torch.tensor(row["dist_field"], dtype=torch.float32).view(*shape)
|
| 375 |
+
if "dist_field" in row else gridmod.bfs_field(board, tuple(row["target"])))
|
| 376 |
+
out.append(StateSample(
|
| 377 |
+
row["uid"], row["game"], row["split"], row["size"], row["topology"],
|
| 378 |
+
board, tuple(row["focus"]), tuple(row["target"]), row["header"], "",
|
| 379 |
+
questions, dist_field=field_t))
|
| 380 |
+
return out
|
|
@@ -0,0 +1,141 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""One board representation shared by every game.
|
| 2 |
+
|
| 3 |
+
Maze and Snake look different on screen but pose the planner the same question:
|
| 4 |
+
propagate reachability out from a target cell, around obstacles, and report what
|
| 5 |
+
each neighbouring cell is worth. Giving both games the same channel layout lets
|
| 6 |
+
a single set of planner weights serve both, so the propagation circuit learned
|
| 7 |
+
on mazes transfers to Snake's flood-fill and back.
|
| 8 |
+
|
| 9 |
+
Channel meanings are fixed:
|
| 10 |
+
|
| 11 |
+
=== ==========================================================
|
| 12 |
+
0 ``free`` cell is enterable on the next tick
|
| 13 |
+
1 ``blocked`` static obstacle (maze wall)
|
| 14 |
+
2 ``focus`` the agent / snake head
|
| 15 |
+
3 ``target`` the goal / the food
|
| 16 |
+
4 ``body`` a dynamic obstacle that will move (snake body)
|
| 17 |
+
5 ``tail`` the body cell that vacates first
|
| 18 |
+
6 ``decay`` how soon a dynamic obstacle clears; maze: visited cells
|
| 19 |
+
7 ``game`` constant plane, 0 for maze and 1 for snake
|
| 20 |
+
=== ==========================================================
|
| 21 |
+
"""
|
| 22 |
+
from __future__ import annotations
|
| 23 |
+
|
| 24 |
+
from collections import deque
|
| 25 |
+
from typing import Sequence
|
| 26 |
+
|
| 27 |
+
import torch
|
| 28 |
+
|
| 29 |
+
FREE_CHANNEL = 0
|
| 30 |
+
|
| 31 |
+
UNIFIED_CHANNELS: tuple[str, ...] = (
|
| 32 |
+
"free", "blocked", "focus", "target", "body", "tail", "decay", "game",
|
| 33 |
+
)
|
| 34 |
+
NUM_CHANNELS = len(UNIFIED_CHANNELS)
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def _empty(size: int) -> torch.Tensor:
|
| 38 |
+
return torch.zeros(NUM_CHANNELS, size, size, dtype=torch.float32)
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
def encode_maze(state) -> tuple[torch.Tensor, tuple[int, int], tuple[int, int]]:
|
| 42 |
+
"""Returns ``(board, focus_cell, target_cell)`` for a :class:`MazeState`."""
|
| 43 |
+
size = state.size
|
| 44 |
+
board = _empty(size)
|
| 45 |
+
wall_rows = torch.zeros(size, size, dtype=torch.bool)
|
| 46 |
+
for r, c in state.walls:
|
| 47 |
+
wall_rows[r, c] = True
|
| 48 |
+
board[0] = (~wall_rows).float()
|
| 49 |
+
board[1] = wall_rows.float()
|
| 50 |
+
board[2, state.position[0], state.position[1]] = 1.0
|
| 51 |
+
board[3, state.goal[0], state.goal[1]] = 1.0
|
| 52 |
+
# `visited` is deliberately left empty, and filling it would not help. The
|
| 53 |
+
# targets are BFS-optimal actions, and the optimal action from a cell is a
|
| 54 |
+
# function of (walls, goal, cell) alone -- where the agent has already been
|
| 55 |
+
# is conditionally independent of it. So under this supervision the channel
|
| 56 |
+
# is noise by construction, and a perfectly fitted model would learn to
|
| 57 |
+
# ignore it. Cycling is approximation error in the action head, not missing
|
| 58 |
+
# memory, which is why the fix lives in the controller and the measured
|
| 59 |
+
# lever is action accuracy. Leaving it empty also keeps the board constant
|
| 60 |
+
# for the whole episode, which is what lets the planner field be cached.
|
| 61 |
+
return board, state.position, state.goal
|
| 62 |
+
|
| 63 |
+
|
| 64 |
+
def encode_snake(state) -> tuple[torch.Tensor, tuple[int, int], tuple[int, int]]:
|
| 65 |
+
"""Returns ``(board, head_cell, food_cell)`` for a :class:`SnakeState`."""
|
| 66 |
+
size = state.size
|
| 67 |
+
board = _empty(size)
|
| 68 |
+
board[0] = 1.0
|
| 69 |
+
length = len(state.body)
|
| 70 |
+
for index, (r, c) in enumerate(state.body):
|
| 71 |
+
board[0, r, c] = 0.0
|
| 72 |
+
board[4, r, c] = 1.0
|
| 73 |
+
# Cells near the tail free up soonest; the planner needs that ordering
|
| 74 |
+
# to tell a real dead end from one that unblocks in a few ticks.
|
| 75 |
+
board[6, r, c] = 1.0 - index / max(length - 1, 1)
|
| 76 |
+
head_r, head_c = state.head
|
| 77 |
+
board[2, head_r, head_c] = 1.0
|
| 78 |
+
tail_r, tail_c = state.body[-1]
|
| 79 |
+
board[5, tail_r, tail_c] = 1.0
|
| 80 |
+
board[0, tail_r, tail_c] = 1.0 # the tail vacates, so it is enterable
|
| 81 |
+
food_r, food_c = state.food
|
| 82 |
+
board[3, food_r, food_c] = 1.0
|
| 83 |
+
board[7] = 1.0
|
| 84 |
+
return board, state.head, state.food
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
def pad_boards(boards: Sequence[torch.Tensor]) -> tuple[torch.Tensor, torch.Tensor,
|
| 88 |
+
torch.Tensor]:
|
| 89 |
+
"""Pad a ragged list of boards to a common size.
|
| 90 |
+
|
| 91 |
+
Returns ``(boards, passable, sizes)``. Padding is marked unpassable so a
|
| 92 |
+
small board in a mixed batch cannot leak signal into the padded margin.
|
| 93 |
+
"""
|
| 94 |
+
height = max(int(b.shape[1]) for b in boards)
|
| 95 |
+
width = max(int(b.shape[2]) for b in boards)
|
| 96 |
+
stacked = torch.zeros(len(boards), NUM_CHANNELS, height, width)
|
| 97 |
+
passable = torch.zeros(len(boards), 1, height, width)
|
| 98 |
+
sizes = torch.zeros(len(boards), dtype=torch.long)
|
| 99 |
+
for i, board in enumerate(boards):
|
| 100 |
+
h, w = board.shape[1], board.shape[2]
|
| 101 |
+
stacked[i, :, :h, :w] = board
|
| 102 |
+
passable[i, 0, :h, :w] = board[0]
|
| 103 |
+
sizes[i] = max(h, w)
|
| 104 |
+
return stacked, passable, sizes
|
| 105 |
+
|
| 106 |
+
|
| 107 |
+
def bfs_field(board: torch.Tensor, target: tuple[int, int]) -> torch.Tensor:
|
| 108 |
+
"""Shortest-path distance from ``target`` to every cell, in cells.
|
| 109 |
+
|
| 110 |
+
Defined purely on the unified board the planner receives -- channel 0 is
|
| 111 |
+
passability for both games -- so the supervision target is exactly the
|
| 112 |
+
quantity the planner has the information to compute. Unreachable and
|
| 113 |
+
blocked cells are ``-1``.
|
| 114 |
+
"""
|
| 115 |
+
free = board[FREE_CHANNEL] > 0.5
|
| 116 |
+
height, width = free.shape
|
| 117 |
+
dist = torch.full((height, width), -1.0)
|
| 118 |
+
if not (0 <= target[0] < height and 0 <= target[1] < width):
|
| 119 |
+
return dist
|
| 120 |
+
queue = deque([target])
|
| 121 |
+
dist[target[0], target[1]] = 0.0
|
| 122 |
+
while queue:
|
| 123 |
+
row, col = queue.popleft()
|
| 124 |
+
step = dist[row, col] + 1
|
| 125 |
+
for dr, dc in ((-1, 0), (1, 0), (0, -1), (0, 1)):
|
| 126 |
+
nr, nc = row + dr, col + dc
|
| 127 |
+
if 0 <= nr < height and 0 <= nc < width and free[nr, nc] \
|
| 128 |
+
and dist[nr, nc] < 0:
|
| 129 |
+
dist[nr, nc] = step
|
| 130 |
+
queue.append((nr, nc))
|
| 131 |
+
return dist
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
def pad_fields(fields: Sequence[torch.Tensor]) -> torch.Tensor:
|
| 135 |
+
"""Pad distance fields to a common shape with ``-1`` (= no supervision)."""
|
| 136 |
+
height = max(f.shape[0] for f in fields)
|
| 137 |
+
width = max(f.shape[1] for f in fields)
|
| 138 |
+
out = torch.full((len(fields), height, width), -1.0)
|
| 139 |
+
for index, field in enumerate(fields):
|
| 140 |
+
out[index, : field.shape[0], : field.shape[1]] = field
|
| 141 |
+
return out
|
|
@@ -0,0 +1,88 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Loading the published checkpoint from the HuggingFace Hub.
|
| 2 |
+
|
| 3 |
+
``JevonRunner`` takes a directory. On the Hub that directory arrives as a
|
| 4 |
+
snapshot download, so this module is the few lines in between -- kept in code
|
| 5 |
+
rather than in the README so the repo id has exactly one spelling, which a
|
| 6 |
+
test then pins the model card to.
|
| 7 |
+
"""
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
from pathlib import Path
|
| 11 |
+
|
| 12 |
+
import torch
|
| 13 |
+
|
| 14 |
+
from .inference import REPO_ID as _REPO_ID, JevonRunner
|
| 15 |
+
|
| 16 |
+
#: The published repository. One spelling, imported by anything that needs it.
|
| 17 |
+
REPO_ID = _REPO_ID
|
| 18 |
+
|
| 19 |
+
#: Where the checkpoint sits inside that repository.
|
| 20 |
+
#:
|
| 21 |
+
#: The Hub convention is weights at the root. This repo keeps the run
|
| 22 |
+
#: directory intact instead, because ``config.json``, ``tokenizer.json`` and
|
| 23 |
+
#: the ``.pt`` files are one artefact: a root-level copy of the config would
|
| 24 |
+
#: be a second spelling of the architecture, free to drift from the weights
|
| 25 |
+
#: sitting beside it, and this project has lost more time to writer/reader
|
| 26 |
+
#: pairs like that than to anything else.
|
| 27 |
+
CHECKPOINT = "runs/jevon-final"
|
| 28 |
+
|
| 29 |
+
#: What ``from_pretrained`` fetches when no checkpoint is named. Selected on
|
| 30 |
+
#: ``min(maze_action, snake_action)`` rather than on aggregate eval loss --
|
| 31 |
+
#: see "Which checkpoint ships" in the model card. Every published number
|
| 32 |
+
#: comes from this file.
|
| 33 |
+
DEFAULT_WEIGHTS = "balanced.pt"
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def default_device() -> str:
|
| 37 |
+
"""The best device actually present, rather than the one this was built on.
|
| 38 |
+
|
| 39 |
+
``JevonRunner`` keeps ``mps`` as its default because every script in this
|
| 40 |
+
repo runs on the machine the model was trained on. A download from the Hub
|
| 41 |
+
is the opposite case -- most of them are CUDA or CPU -- and silently
|
| 42 |
+
falling back from an unavailable ``mps`` would leave a CUDA user on CPU
|
| 43 |
+
wondering why a 20M-parameter model is slow.
|
| 44 |
+
"""
|
| 45 |
+
if torch.cuda.is_available():
|
| 46 |
+
return "cuda"
|
| 47 |
+
if torch.backends.mps.is_available():
|
| 48 |
+
return "mps"
|
| 49 |
+
return "cpu"
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def from_pretrained(repo_id: str | Path = REPO_ID, *, device: str | None = None,
|
| 53 |
+
weights: str | None = None,
|
| 54 |
+
revision: str | None = None) -> JevonRunner:
|
| 55 |
+
"""A loaded ``JevonRunner``, from the Hub or from a local run directory.
|
| 56 |
+
|
| 57 |
+
``repo_id`` may be either. A path that already contains ``config.json`` is
|
| 58 |
+
loaded straight from disk, so the same call works before and after
|
| 59 |
+
publishing and in the tests, which have no network.
|
| 60 |
+
|
| 61 |
+
Only the files needed to answer questions are fetched -- the config, the
|
| 62 |
+
tokenizer and one checkpoint -- so this costs ~80MB rather than the whole
|
| 63 |
+
repository. Name ``weights`` to pull a different one.
|
| 64 |
+
"""
|
| 65 |
+
local = Path(repo_id)
|
| 66 |
+
if (local / "config.json").is_file():
|
| 67 |
+
return JevonRunner(local, device or default_device(), weights)
|
| 68 |
+
|
| 69 |
+
try:
|
| 70 |
+
from huggingface_hub import snapshot_download
|
| 71 |
+
except ImportError as err: # optional: local runs need no Hub
|
| 72 |
+
raise ImportError(
|
| 73 |
+
"from_pretrained() needs huggingface_hub "
|
| 74 |
+
"(`pip install huggingface_hub`). To load a run you already have, "
|
| 75 |
+
"pass its directory instead: from_pretrained('runs/jevon-final')."
|
| 76 |
+
) from err
|
| 77 |
+
|
| 78 |
+
# Named explicitly rather than left to resolve_weights' preference order:
|
| 79 |
+
# the fallback scans a directory, and this one will only ever contain the
|
| 80 |
+
# single file that was just downloaded.
|
| 81 |
+
wanted = weights or DEFAULT_WEIGHTS
|
| 82 |
+
snapshot = snapshot_download(
|
| 83 |
+
str(repo_id), revision=revision,
|
| 84 |
+
allow_patterns=[f"{CHECKPOINT}/config.json",
|
| 85 |
+
f"{CHECKPOINT}/tokenizer.json",
|
| 86 |
+
f"{CHECKPOINT}/{wanted}"])
|
| 87 |
+
return JevonRunner(Path(snapshot) / CHECKPOINT,
|
| 88 |
+
device or default_device(), wanted)
|
|
@@ -0,0 +1,252 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Loading a checkpoint and answering decision requests."""
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import hashlib
|
| 5 |
+
import json
|
| 6 |
+
from pathlib import Path
|
| 7 |
+
from typing import Optional, Sequence
|
| 8 |
+
|
| 9 |
+
import torch
|
| 10 |
+
|
| 11 |
+
from . import grid as gridmod
|
| 12 |
+
from .data import QuestionSample, StateSample, collate
|
| 13 |
+
from .modeling import JevonConfig, JevonModel
|
| 14 |
+
from .planner import default_iterations
|
| 15 |
+
from .tokenizer import JevonTokenizer
|
| 16 |
+
|
| 17 |
+
|
| 18 |
+
# The order `benchmark.sh` and `record.sh` also use. `balanced.pt` is selected
|
| 19 |
+
# on min(maze_action, snake_action) and `best.pt` on aggregate eval loss, which
|
| 20 |
+
# the boolean questions outnumber the action questions in badly enough that a
|
| 21 |
+
# checkpoint which had stopped playing snake once held the lowest loss in the
|
| 22 |
+
# run -- so every number this repo ships comes from `balanced.pt`, and the
|
| 23 |
+
# README argues the case at length. The scripts nonetheless defaulted
|
| 24 |
+
# `--weights` to `best.pt`, handing anyone who read `--help` the checkpoint the
|
| 25 |
+
# documentation exists to argue against, and quietly disagreeing with RESULTS.md.
|
| 26 |
+
PREFERRED_WEIGHTS = ("balanced.pt", "best.pt")
|
| 27 |
+
|
| 28 |
+
# Defined here rather than in `hub.py` because the error below names it and
|
| 29 |
+
# `hub` imports this module; one spelling, no cycle.
|
| 30 |
+
REPO_ID = "lewislululu/jevon"
|
| 31 |
+
|
| 32 |
+
# The first line of a Git LFS pointer file.
|
| 33 |
+
LFS_POINTER = b"version https://git-lfs.github.com/spec/v1"
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def _is_lfs_pointer(path: Path) -> bool:
|
| 37 |
+
"""True for a checkpoint that was cloned without Git LFS.
|
| 38 |
+
|
| 39 |
+
This is the one failure a published repo produces that a local one never
|
| 40 |
+
does, and it does not look like a missing file: the path exists, and holds
|
| 41 |
+
~130 bytes of text naming the object that should have been fetched.
|
| 42 |
+
`torch.load` on that dies inside the zip reader with a message about a
|
| 43 |
+
central directory, which says nothing about the cause -- and the cause is
|
| 44 |
+
one command.
|
| 45 |
+
"""
|
| 46 |
+
if path.stat().st_size > 1024: # a real checkpoint is ~80MB
|
| 47 |
+
return False
|
| 48 |
+
with path.open("rb") as handle:
|
| 49 |
+
return handle.read(len(LFS_POINTER)) == LFS_POINTER
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def resolve_weights(checkpoint_dir: str | Path,
|
| 53 |
+
name: str | None = None) -> Path:
|
| 54 |
+
"""The checkpoint file to load, preferring the one the docs describe.
|
| 55 |
+
|
| 56 |
+
`name` names one explicitly; without it the first of PREFERRED_WEIGHTS
|
| 57 |
+
that exists wins, so a run trained before `balanced.pt` existed still
|
| 58 |
+
loads.
|
| 59 |
+
"""
|
| 60 |
+
path = Path(checkpoint_dir)
|
| 61 |
+
wanted = (name,) if name else PREFERRED_WEIGHTS
|
| 62 |
+
for candidate in wanted:
|
| 63 |
+
target = path / candidate
|
| 64 |
+
if target.is_file():
|
| 65 |
+
if _is_lfs_pointer(target):
|
| 66 |
+
raise FileNotFoundError(
|
| 67 |
+
f"{target} is a Git LFS pointer, not a checkpoint -- this "
|
| 68 |
+
f"clone was made without LFS. Run `git lfs install && git "
|
| 69 |
+
f"lfs pull`, or fetch the file from "
|
| 70 |
+
f"https://huggingface.co/{REPO_ID}.")
|
| 71 |
+
return target
|
| 72 |
+
|
| 73 |
+
# A bare FileNotFoundError names the path but not the reason, and the
|
| 74 |
+
# reason is the non-obvious half. Which reason it is depends on where the
|
| 75 |
+
# reader got this directory: the published run carries both checkpoints,
|
| 76 |
+
# so reaching this line means either LFS was never pulled or this is a run
|
| 77 |
+
# they trained themselves.
|
| 78 |
+
have = sorted(q.name for q in path.glob("*.pt"))
|
| 79 |
+
tried = " or ".join(str(path / c) for c in wanted)
|
| 80 |
+
raise FileNotFoundError(
|
| 81 |
+
f"no weights at {tried}. The published run ships `balanced.pt` and "
|
| 82 |
+
f"`best.pt` through Git LFS: on a clone, `git lfs pull` fetches them. "
|
| 83 |
+
f"Otherwise train one with `scripts/train.py --out runs/my-run`, or "
|
| 84 |
+
f"point --checkpoint at a run that has them."
|
| 85 |
+
+ (f" This directory has: {', '.join(have)}." if have else ""))
|
| 86 |
+
|
| 87 |
+
|
| 88 |
+
class JevonRunner:
|
| 89 |
+
"""A loaded model plus the glue that turns states into distributions."""
|
| 90 |
+
|
| 91 |
+
def __init__(self, checkpoint_dir: str | Path, device: str = "mps",
|
| 92 |
+
weights: str | None = None):
|
| 93 |
+
path = Path(checkpoint_dir)
|
| 94 |
+
chosen = resolve_weights(path, weights)
|
| 95 |
+
cfg_dict = json.loads((path / "config.json").read_text())
|
| 96 |
+
self.cfg = JevonConfig(**cfg_dict)
|
| 97 |
+
self.tokenizer = JevonTokenizer.load(path / "tokenizer.json")
|
| 98 |
+
usable = device != "mps" or torch.backends.mps.is_available()
|
| 99 |
+
self.device = torch.device(device if usable else "cpu")
|
| 100 |
+
self.model = JevonModel(self.cfg).to(self.device)
|
| 101 |
+
self.weights = chosen.name
|
| 102 |
+
state = torch.load(chosen, map_location=self.device)
|
| 103 |
+
self.model.load_state_dict(state)
|
| 104 |
+
self.model.eval()
|
| 105 |
+
self._field_cache: dict[bytes, torch.Tensor] = {}
|
| 106 |
+
self.planner_calls = 0
|
| 107 |
+
self.planner_cache_hits = 0
|
| 108 |
+
|
| 109 |
+
def reset_cache(self) -> None:
|
| 110 |
+
self._field_cache.clear()
|
| 111 |
+
|
| 112 |
+
def _fields(self, batch: dict, steps: int) -> Optional[torch.Tensor]:
|
| 113 |
+
"""Planner fields for the batch, reusing any already computed.
|
| 114 |
+
|
| 115 |
+
The planner never sees the agent, so a maze board hashes to the same key
|
| 116 |
+
for every step of an episode and the field is computed once. Snake
|
| 117 |
+
boards change each tick and simply miss.
|
| 118 |
+
"""
|
| 119 |
+
if self.model.planner is None or batch.get("boards") is None:
|
| 120 |
+
return None
|
| 121 |
+
boards, passable = batch["boards"], batch["passable"]
|
| 122 |
+
probe = boards.clone()
|
| 123 |
+
probe[:, self.model.FOCUS_CHANNEL] = 0.0
|
| 124 |
+
out, missing = [None] * boards.shape[0], []
|
| 125 |
+
keys = []
|
| 126 |
+
for i in range(boards.shape[0]):
|
| 127 |
+
key = hashlib.blake2b(
|
| 128 |
+
probe[i].to(torch.float16).cpu().numpy().tobytes()
|
| 129 |
+
+ steps.to_bytes(4, "big"), digest_size=16).digest()
|
| 130 |
+
keys.append(key)
|
| 131 |
+
cached = self._field_cache.get(key)
|
| 132 |
+
if cached is None:
|
| 133 |
+
missing.append(i)
|
| 134 |
+
else:
|
| 135 |
+
out[i] = cached
|
| 136 |
+
self.planner_cache_hits += 1
|
| 137 |
+
if missing:
|
| 138 |
+
self.planner_calls += len(missing)
|
| 139 |
+
index = torch.tensor(missing, device=boards.device)
|
| 140 |
+
computed = self.model.encode_boards(
|
| 141 |
+
boards[index], passable[index], steps, 0)
|
| 142 |
+
for slot, i in enumerate(missing):
|
| 143 |
+
field = computed[slot]
|
| 144 |
+
out[i] = field
|
| 145 |
+
if len(self._field_cache) < 64:
|
| 146 |
+
self._field_cache[keys[i]] = field
|
| 147 |
+
return torch.stack(out, dim=0)
|
| 148 |
+
|
| 149 |
+
@torch.no_grad()
|
| 150 |
+
def distance_field(self, state: StateSample, *,
|
| 151 |
+
iterations: Optional[int] = None) -> torch.Tensor:
|
| 152 |
+
"""The model's own predicted distance-to-target for every cell.
|
| 153 |
+
|
| 154 |
+
This is the planner's training read-out, reused at play time. It costs
|
| 155 |
+
nothing extra during an episode: the field does not depend on the agent,
|
| 156 |
+
so the cache that serves ``answer`` serves this too.
|
| 157 |
+
"""
|
| 158 |
+
batch = collate([state], self.tokenizer, device=self.device)
|
| 159 |
+
steps = iterations or default_iterations(int(batch["board_size"].max()))
|
| 160 |
+
cells = self._fields(batch, steps)
|
| 161 |
+
if cells is None:
|
| 162 |
+
raise RuntimeError("checkpoint has no planner")
|
| 163 |
+
with torch.no_grad():
|
| 164 |
+
return self.model.predict_field(
|
| 165 |
+
cells, batch["boards"], batch["passable"],
|
| 166 |
+
iterations=steps)[0, 0].float().cpu()
|
| 167 |
+
|
| 168 |
+
def answer(self, states: Sequence[StateSample], *,
|
| 169 |
+
iterations: Optional[int] = None) -> list[dict]:
|
| 170 |
+
"""Score every question on every state in a single batched pass."""
|
| 171 |
+
batch = collate(states, self.tokenizer, device=self.device)
|
| 172 |
+
steps = iterations or default_iterations(int(batch["board_size"].max()))
|
| 173 |
+
cells = self._fields(batch, steps)
|
| 174 |
+
logits = self.model(batch, iterations=steps, grad_steps=0, cells=cells)
|
| 175 |
+
probs = logits.float().softmax(-1).cpu()
|
| 176 |
+
out = []
|
| 177 |
+
for row, meta in enumerate(batch["meta"]):
|
| 178 |
+
ids = meta["candidate_ids"]
|
| 179 |
+
if meta["type"] == "boolean":
|
| 180 |
+
names = ["false", "true"]
|
| 181 |
+
values = probs[row, :2].tolist()
|
| 182 |
+
else:
|
| 183 |
+
names = list(ids)
|
| 184 |
+
values = probs[row, : len(ids)].tolist()
|
| 185 |
+
total = sum(values) or 1.0
|
| 186 |
+
distribution = {n: v / total for n, v in zip(names, values)}
|
| 187 |
+
record = {
|
| 188 |
+
"uid": meta["uid"], "question": meta["question"],
|
| 189 |
+
"type": meta["type"], "probabilities": distribution,
|
| 190 |
+
"choice": max(distribution, key=distribution.get),
|
| 191 |
+
"target": meta["probs"],
|
| 192 |
+
}
|
| 193 |
+
if meta["type"] == "score":
|
| 194 |
+
k = len(names)
|
| 195 |
+
record["expected_score"] = sum(
|
| 196 |
+
i / max(k - 1, 1) * distribution[n] for i, n in enumerate(names))
|
| 197 |
+
out.append(record)
|
| 198 |
+
return out
|
| 199 |
+
|
| 200 |
+
|
| 201 |
+
def maze_state_sample(state, *, uid: str = "live") -> StateSample:
|
| 202 |
+
from envs import maze as mz
|
| 203 |
+
|
| 204 |
+
profile = mz.question_profile(state)
|
| 205 |
+
board, focus, target = gridmod.encode_maze(state)
|
| 206 |
+
return _assemble(profile, board, focus, target, mz.render_header(state),
|
| 207 |
+
mz.render_ascii(state), uid, "maze", state.size, state.topology)
|
| 208 |
+
|
| 209 |
+
|
| 210 |
+
def snake_state_sample(state, *, uid: str = "live") -> StateSample:
|
| 211 |
+
from envs import snake as sk
|
| 212 |
+
|
| 213 |
+
profile = sk.question_profile(state)
|
| 214 |
+
board, focus, target = gridmod.encode_snake(state)
|
| 215 |
+
return _assemble(profile, board, focus, target, sk.render_header(state),
|
| 216 |
+
sk.render_ascii(state), uid, "snake", state.size, "snake")
|
| 217 |
+
|
| 218 |
+
|
| 219 |
+
def _assemble(profile, board, focus, target, header, ascii_board, uid, game, size,
|
| 220 |
+
topology) -> StateSample:
|
| 221 |
+
questions = []
|
| 222 |
+
for name, question in profile["questions"].items():
|
| 223 |
+
qtype = question["type"]
|
| 224 |
+
anchors_map = question.get("anchors") or {}
|
| 225 |
+
if qtype == "choice":
|
| 226 |
+
ids = list(question["criteria"])
|
| 227 |
+
texts = [question["criteria"][k] for k in ids]
|
| 228 |
+
anchors = [anchors_map.get(k) for k in ids]
|
| 229 |
+
probs = [0.0] * len(ids)
|
| 230 |
+
elif qtype == "boolean":
|
| 231 |
+
ids, texts = ["true"], ["The proposition is true."]
|
| 232 |
+
anchors = [anchors_map.get("true")]
|
| 233 |
+
probs = [0.0, 0.0]
|
| 234 |
+
else:
|
| 235 |
+
levels = question["criteria"]
|
| 236 |
+
ids = [str(i) for i in range(len(levels))]
|
| 237 |
+
texts, anchors = list(levels), [None] * len(levels)
|
| 238 |
+
probs = [0.0] * len(levels)
|
| 239 |
+
target_row = profile["targets"].get(name)
|
| 240 |
+
if target_row is not None:
|
| 241 |
+
if qtype == "choice":
|
| 242 |
+
probs = [float(target_row["probs"].get(k, 0.0)) for k in ids]
|
| 243 |
+
elif qtype == "boolean":
|
| 244 |
+
probs = [float(target_row["probs"]["false"]),
|
| 245 |
+
float(target_row["probs"]["true"])]
|
| 246 |
+
else:
|
| 247 |
+
probs = [0.0] * len(ids)
|
| 248 |
+
probs[int(target_row["level"])] = 1.0
|
| 249 |
+
questions.append(QuestionSample(name, qtype, question["instructions"], ids,
|
| 250 |
+
texts, anchors, probs))
|
| 251 |
+
return StateSample(uid, game, "live", size, topology, board, focus, target,
|
| 252 |
+
header, ascii_board, questions)
|
|
@@ -0,0 +1,290 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Training objectives for calibrated decisions.
|
| 2 |
+
|
| 3 |
+
Cross-entropy against the oracle's full distribution is the main term: the
|
| 4 |
+
targets are genuine distributions (all shortest-path moves tie, all equally good
|
| 5 |
+
Snake moves tie), so matching them is matching the right answer *and* the right
|
| 6 |
+
uncertainty. A Brier term is added because cross-entropy alone rewards
|
| 7 |
+
confident near-misses more than it should, and Brier is the proper score that
|
| 8 |
+
penalises overconfidence directly.
|
| 9 |
+
|
| 10 |
+
Question types arrive in very different quantities -- a maze state emits four
|
| 11 |
+
Boolean questions but only one choice -- so each type is reweighted to keep the
|
| 12 |
+
gameplay-relevant choice questions from being drowned out.
|
| 13 |
+
"""
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
from dataclasses import dataclass
|
| 17 |
+
|
| 18 |
+
import torch
|
| 19 |
+
import torch.nn.functional as F
|
| 20 |
+
|
| 21 |
+
from .modeling import TYPE_INDEX
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
@dataclass
|
| 25 |
+
class LossWeights:
|
| 26 |
+
choice: float = 2.0
|
| 27 |
+
boolean: float = 0.6
|
| 28 |
+
score: float = 0.6
|
| 29 |
+
brier: float = 0.5
|
| 30 |
+
# Score levels are a graded scale, not unrelated categories: "within ten
|
| 31 |
+
# moves" and "within twenty-five moves" are neighbours on one axis. A
|
| 32 |
+
# one-hot target says otherwise, so a near-miss earns exactly as much loss
|
| 33 |
+
# as naming the opposite end of the scale, and with seven near-synonymous
|
| 34 |
+
# levels the marginal mode becomes the easiest minimum -- measured, the
|
| 35 |
+
# score head sat on the always-level-4 constant for a whole run while a
|
| 36 |
+
# linear probe on the same features binned to 0.43 against its 0.34. That
|
| 37 |
+
# probe lands within one level 91% of the time, which is the shape the
|
| 38 |
+
# target should have: mass decaying with distance along the scale.
|
| 39 |
+
score_smoothing: float = 0.0
|
| 40 |
+
entropy: float = 0.0 # optional confidence penalty
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def _grade_score_targets(target, valid, types, tau):
|
| 44 |
+
"""Spread score-question mass over neighbouring levels.
|
| 45 |
+
|
| 46 |
+
Candidate ``j`` of a score question *is* level ``j`` (the criteria arrive in
|
| 47 |
+
order), so distance along the candidate axis is distance along the scale.
|
| 48 |
+
The peak stays on the true level, which keeps every reported accuracy
|
| 49 |
+
comparable with runs trained without this.
|
| 50 |
+
"""
|
| 51 |
+
is_score = types.eq(TYPE_INDEX["score"])
|
| 52 |
+
if not bool(is_score.any()):
|
| 53 |
+
return target
|
| 54 |
+
levels = torch.arange(target.shape[1], device=target.device).float()
|
| 55 |
+
true = target.argmax(-1, keepdim=True).float()
|
| 56 |
+
graded = torch.exp(-(levels.unsqueeze(0) - true).abs() / tau)
|
| 57 |
+
graded = graded * valid.float()
|
| 58 |
+
graded = graded / graded.sum(-1, keepdim=True).clamp(min=1e-9)
|
| 59 |
+
return torch.where(is_score.unsqueeze(-1), graded, target)
|
| 60 |
+
|
| 61 |
+
|
| 62 |
+
def decision_loss(logits: torch.Tensor, batch: dict,
|
| 63 |
+
weights: LossWeights = LossWeights()) -> tuple[torch.Tensor, dict]:
|
| 64 |
+
target = batch["target_probs"]
|
| 65 |
+
valid = batch["candidate_valid"]
|
| 66 |
+
types = batch["question_type"]
|
| 67 |
+
if weights.score_smoothing > 0:
|
| 68 |
+
target = _grade_score_targets(target, valid, types, weights.score_smoothing)
|
| 69 |
+
|
| 70 |
+
log_probs = logits.float().log_softmax(-1)
|
| 71 |
+
probs = log_probs.exp()
|
| 72 |
+
ce = -(target * log_probs).sum(-1)
|
| 73 |
+
brier = ((probs - target) ** 2 * valid).sum(-1)
|
| 74 |
+
|
| 75 |
+
per_type = torch.tensor(
|
| 76 |
+
[weights.choice, weights.boolean, weights.score], device=logits.device)
|
| 77 |
+
w = per_type[types]
|
| 78 |
+
# Normalise within the batch so a lopsided type mix cannot change the step size.
|
| 79 |
+
w = w / w.mean().clamp(min=1e-6)
|
| 80 |
+
|
| 81 |
+
loss = (w * (ce + weights.brier * brier)).mean()
|
| 82 |
+
if weights.entropy:
|
| 83 |
+
entropy = -(probs.clamp_min(1e-9).log() * probs * valid).sum(-1)
|
| 84 |
+
loss = loss - weights.entropy * (w * entropy).mean()
|
| 85 |
+
|
| 86 |
+
with torch.no_grad():
|
| 87 |
+
picked = logits.argmax(-1)
|
| 88 |
+
best = target.argmax(-1)
|
| 89 |
+
# A tie counts as correct whenever the pick carries maximal target mass.
|
| 90 |
+
hit = target.gather(1, picked.unsqueeze(1)).squeeze(1)
|
| 91 |
+
top = target.max(-1).values
|
| 92 |
+
correct = (hit >= top - 1e-6).float()
|
| 93 |
+
stats = {
|
| 94 |
+
"loss": loss.detach(),
|
| 95 |
+
"ce": ce.mean().detach(),
|
| 96 |
+
"brier": brier.mean().detach(),
|
| 97 |
+
"acc": correct.mean().detach(),
|
| 98 |
+
"acc_choice": _masked_mean(correct, types == TYPE_INDEX["choice"]),
|
| 99 |
+
"acc_boolean": _masked_mean(correct, types == TYPE_INDEX["boolean"]),
|
| 100 |
+
"acc_score": _masked_mean(correct, types == TYPE_INDEX["score"]),
|
| 101 |
+
"tvd": _masked_mean(0.5 * (probs - target).abs().sum(-1),
|
| 102 |
+
torch.ones_like(types, dtype=torch.bool)),
|
| 103 |
+
}
|
| 104 |
+
del best
|
| 105 |
+
return loss, stats
|
| 106 |
+
|
| 107 |
+
|
| 108 |
+
def _masked_mean(values: torch.Tensor, mask: torch.Tensor) -> torch.Tensor:
|
| 109 |
+
if mask.sum() == 0:
|
| 110 |
+
return torch.tensor(float("nan"), device=values.device)
|
| 111 |
+
return values[mask].mean().detach()
|
| 112 |
+
|
| 113 |
+
|
| 114 |
+
@torch.no_grad()
|
| 115 |
+
def calibration_bins(probs: torch.Tensor, correct: torch.Tensor,
|
| 116 |
+
bins: int = 10) -> dict:
|
| 117 |
+
"""Expected calibration error over equal-width confidence bins."""
|
| 118 |
+
edges = torch.linspace(0, 1, bins + 1, device=probs.device)
|
| 119 |
+
total, ece = probs.numel(), torch.zeros((), device=probs.device)
|
| 120 |
+
rows = []
|
| 121 |
+
for i in range(bins):
|
| 122 |
+
lo, hi = edges[i], edges[i + 1]
|
| 123 |
+
mask = (probs > lo) & (probs <= hi) if i else (probs >= lo) & (probs <= hi)
|
| 124 |
+
count = int(mask.sum())
|
| 125 |
+
if not count:
|
| 126 |
+
rows.append({"bin": [float(lo), float(hi)], "count": 0})
|
| 127 |
+
continue
|
| 128 |
+
conf, acc = probs[mask].mean(), correct[mask].float().mean()
|
| 129 |
+
ece = ece + (count / total) * (conf - acc).abs()
|
| 130 |
+
rows.append({"bin": [float(lo), float(hi)], "count": count,
|
| 131 |
+
"confidence": float(conf), "accuracy": float(acc)})
|
| 132 |
+
return {"ece": float(ece), "bins": rows}
|
| 133 |
+
|
| 134 |
+
|
| 135 |
+
def squash_distance(dist: torch.Tensor, size: torch.Tensor,
|
| 136 |
+
mode: str = "log") -> torch.Tensor:
|
| 137 |
+
"""Map raw cell counts onto [0, 1].
|
| 138 |
+
|
| 139 |
+
``log`` gives near-uniform resolution per doubling, which is the right
|
| 140 |
+
choice if you care about reading a distance off the field. It is the wrong
|
| 141 |
+
choice if you care about *descending* it: neighbouring cells then differ by
|
| 142 |
+
~1/d in target units, so contrast falls from 0.083 near the goal to 0.0098
|
| 143 |
+
forty cells out -- an 8.5x collapse, measured on a trained checkpoint whose
|
| 144 |
+
field MAE was 0.107, i.e. larger than the gap it had to resolve anywhere.
|
| 145 |
+
|
| 146 |
+
``linear`` spends resolution uniformly: every neighbouring pair differs by
|
| 147 |
+
exactly 1/size^2 wherever it sits on the board. Greedy descent has to win
|
| 148 |
+
*every* comparison along a path, so the weakest link sets the solve rate,
|
| 149 |
+
and a uniform target is the one that lifts the weakest link.
|
| 150 |
+
|
| 151 |
+
Rescaling alone would change nothing -- it scales the head's error by the
|
| 152 |
+
same factor -- so this is a change of shape, not of units.
|
| 153 |
+
"""
|
| 154 |
+
area = size.to(dist.dtype).clamp(min=2.0) ** 2
|
| 155 |
+
if mode == "linear":
|
| 156 |
+
return dist.clamp(min=0.0) / area.view(-1, 1, 1)
|
| 157 |
+
return torch.log1p(dist.clamp(min=0.0)) / torch.log1p(area).view(-1, 1, 1)
|
| 158 |
+
|
| 159 |
+
|
| 160 |
+
def _neighbours(x: torch.Tensor, fill: float) -> torch.Tensor:
|
| 161 |
+
"""The four 4-neighbourhood shifts of an ``(S, H, W)`` map, stacked on dim 0.
|
| 162 |
+
|
| 163 |
+
Off-board cells are filled with ``fill`` so a neighbour past the border is
|
| 164 |
+
rejected by the same comparison that rejects a wall, with no edge case.
|
| 165 |
+
"""
|
| 166 |
+
pads = ((0, 0, 1, 0), (0, 0, 0, 1), (1, 0, 0, 0), (0, 1, 0, 0))
|
| 167 |
+
crops = (lambda t: t[:, :-1, :], lambda t: t[:, 1:, :],
|
| 168 |
+
lambda t: t[:, :, :-1], lambda t: t[:, :, 1:])
|
| 169 |
+
return torch.stack([crop(F.pad(x, pad, value=fill))
|
| 170 |
+
for pad, crop in zip(pads, crops)], dim=0)
|
| 171 |
+
|
| 172 |
+
|
| 173 |
+
def descent_loss(dist_pred: torch.Tensor, dist_field: torch.Tensor,
|
| 174 |
+
passable: torch.Tensor, size: torch.Tensor,
|
| 175 |
+
margin_mult: float = 1.0) -> tuple:
|
| 176 |
+
"""Supervise the ordering a greedy controller reads, not just the values.
|
| 177 |
+
|
| 178 |
+
Regressing each cell's distance is supervision on absolute values, but
|
| 179 |
+
descent never uses an absolute value: it compares a cell's four neighbours
|
| 180 |
+
and steps to the smallest. Those two objectives come apart badly here.
|
| 181 |
+
Measured on a trained checkpoint, the field's MAE was 0.107 while the gap
|
| 182 |
+
between neighbouring target values was 0.0098 forty cells out -- the error
|
| 183 |
+
was an order of magnitude larger than the quantity being compared, so the
|
| 184 |
+
ordering was noise even where the values looked good. The consequence is
|
| 185 |
+
not a slightly worse route: 70-93% of cells descended into a spurious basin
|
| 186 |
+
and cycled forever, which is what a solve rate of zero looks like up close.
|
| 187 |
+
|
| 188 |
+
So state the requirement directly. For every cell with a true downhill
|
| 189 |
+
neighbour, the best predicted downhill neighbour must sit below every other
|
| 190 |
+
passable neighbour by a margin, and anything less is a hinge penalty. The
|
| 191 |
+
margin is one true step (``margin_mult / area``), which asks the field to
|
| 192 |
+
separate real neighbours by at least as much as the truth separates them.
|
| 193 |
+
|
| 194 |
+
Returns ``(loss, ordered_fraction)`` where the second value is the share of
|
| 195 |
+
cells whose argmin already points along a shortest path -- the training-time
|
| 196 |
+
reading of the metric the controller's solve rate actually depends on.
|
| 197 |
+
"""
|
| 198 |
+
inside = passable.squeeze(1) > 0.5 # (S, H, W)
|
| 199 |
+
# The goal itself has no step to get right, and unreachable cells have no
|
| 200 |
+
# truth to compare against.
|
| 201 |
+
here_ok = inside & (dist_field > 0)
|
| 202 |
+
|
| 203 |
+
nb_true = _neighbours(dist_field, -1.0)
|
| 204 |
+
nb_inside = _neighbours(inside.to(dist_pred.dtype), 0.0) > 0.5
|
| 205 |
+
nb_pred = _neighbours(dist_pred, 0.0)
|
| 206 |
+
|
| 207 |
+
valid = nb_inside & (nb_true >= 0)
|
| 208 |
+
down = valid & nb_true.eq(dist_field.unsqueeze(0) - 1)
|
| 209 |
+
other = valid & ~down
|
| 210 |
+
|
| 211 |
+
big = torch.finfo(dist_pred.dtype).max
|
| 212 |
+
down_min = torch.where(down, nb_pred, nb_pred.new_full((), big)).amin(dim=0)
|
| 213 |
+
|
| 214 |
+
area = size.to(dist_pred.dtype).clamp(min=2.0) ** 2
|
| 215 |
+
margin = (margin_mult / area).view(-1, 1, 1)
|
| 216 |
+
|
| 217 |
+
rows = (here_ok & down.any(dim=0)).unsqueeze(0)
|
| 218 |
+
penal = torch.relu(down_min.unsqueeze(0) + margin - nb_pred) * (other & rows)
|
| 219 |
+
count = (other & rows).sum().clamp(min=1)
|
| 220 |
+
|
| 221 |
+
with torch.no_grad():
|
| 222 |
+
other_min = torch.where(other, nb_pred, nb_pred.new_full((), big)).amin(dim=0)
|
| 223 |
+
# A cell is ordered when its best downhill neighbour beats every other
|
| 224 |
+
# one. Cells whose neighbours are all downhill are ordered by
|
| 225 |
+
# construction -- any move is correct -- and the ``big`` fill makes that
|
| 226 |
+
# fall out of the same comparison.
|
| 227 |
+
ordered = ((down_min < other_min) & rows.squeeze(0)).sum()
|
| 228 |
+
total = rows.sum().clamp(min=1)
|
| 229 |
+
|
| 230 |
+
return penal.sum() / count, float(ordered / total)
|
| 231 |
+
|
| 232 |
+
|
| 233 |
+
def field_loss(pred: torch.Tensor, dist_field: torch.Tensor,
|
| 234 |
+
passable: torch.Tensor, size: torch.Tensor,
|
| 235 |
+
mode: str = "log", descent_weight: float = 0.0,
|
| 236 |
+
descent_margin: float = 1.0) -> tuple:
|
| 237 |
+
"""Dense planner supervision.
|
| 238 |
+
|
| 239 |
+
``pred`` is (S, 2, H, W): channel 0 regresses the squashed distance to the
|
| 240 |
+
target over reachable cells, channel 1 classifies reachability over every
|
| 241 |
+
cell inside the board. Padding outside a board is excluded from both.
|
| 242 |
+
"""
|
| 243 |
+
inside = passable.squeeze(1) > 0.5 # (S, H, W)
|
| 244 |
+
reachable = dist_field >= 0.0
|
| 245 |
+
target = squash_distance(dist_field, size, mode)
|
| 246 |
+
|
| 247 |
+
dist_pred = pred[:, 0]
|
| 248 |
+
mask = reachable & inside
|
| 249 |
+
count = mask.sum().clamp(min=1)
|
| 250 |
+
dist_term = (torch.nn.functional.smooth_l1_loss(
|
| 251 |
+
dist_pred, target, beta=0.05, reduction="none") * mask).sum() / count
|
| 252 |
+
|
| 253 |
+
reach_term = (torch.nn.functional.binary_cross_entropy_with_logits(
|
| 254 |
+
pred[:, 1], reachable.to(pred.dtype), reduction="none")
|
| 255 |
+
* inside).sum() / inside.sum().clamp(min=1)
|
| 256 |
+
|
| 257 |
+
with torch.no_grad():
|
| 258 |
+
abs_err = ((dist_pred - target).abs() * mask).sum()
|
| 259 |
+
sse = (((dist_pred - target) ** 2) * mask).sum()
|
| 260 |
+
tsum = (target * mask).sum()
|
| 261 |
+
tsq = ((target ** 2) * mask).sum()
|
| 262 |
+
centred = (tsq - tsum ** 2 / count).clamp(min=1e-9)
|
| 263 |
+
r2 = 1.0 - sse / centred
|
| 264 |
+
total = dist_term + 0.3 * reach_term
|
| 265 |
+
extra = {}
|
| 266 |
+
if descent_weight > 0:
|
| 267 |
+
order_term, ordered = descent_loss(dist_pred, dist_field, passable,
|
| 268 |
+
size, descent_margin)
|
| 269 |
+
total = total + descent_weight * order_term
|
| 270 |
+
# Only reported when the term is on, so a caller that averages the
|
| 271 |
+
# stats never has to average a placeholder.
|
| 272 |
+
extra = {"field_order": float(order_term.detach()),
|
| 273 |
+
"field_ordered": ordered}
|
| 274 |
+
|
| 275 |
+
return total, {
|
| 276 |
+
**extra,
|
| 277 |
+
"field": float(dist_term.detach()),
|
| 278 |
+
# A penalty, so lower is better and 0 is perfect. Not to be read as
|
| 279 |
+
# probe.py's `reach_rate`, which is a fraction of cells and where
|
| 280 |
+
# higher is better -- these sit next to each other in a run directory
|
| 281 |
+
# and mean opposite things by the same word.
|
| 282 |
+
"field_reach": float(reach_term.detach()),
|
| 283 |
+
"field_mae": float(abs_err / count), "field_r2": float(r2),
|
| 284 |
+
# Raw sums so a caller spanning several batches can pool the statistic.
|
| 285 |
+
# R^2 is a ratio of sums and is NOT averageable: one batch of boards
|
| 286 |
+
# with little distance variance sends its own R^2 arbitrarily negative
|
| 287 |
+
# and drags a mean with it, which is what made field_r2 look like it
|
| 288 |
+
# was collapsing during training when the pooled value was fine.
|
| 289 |
+
"_f_sse": float(sse), "_f_abs": float(abs_err), "_f_sum": float(tsum),
|
| 290 |
+
"_f_sumsq": float(tsq), "_f_n": float(count)}
|
|
@@ -0,0 +1,390 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Jevon: a System-One decision model with a spatial planner.
|
| 2 |
+
|
| 3 |
+
Contract (identical to Jev/NanoJev): a request carries states, each with a map
|
| 4 |
+
of typed questions -- ``choice`` over 2..255 dynamic candidates, ``boolean``
|
| 5 |
+
over a proposition, ``score`` over 2..10 ordered levels. Every question returns
|
| 6 |
+
a complete probability distribution and nothing is decoded token by token.
|
| 7 |
+
|
| 8 |
+
Three things differ from NanoJev's implementation:
|
| 9 |
+
|
| 10 |
+
1. **Shared prefix.** NanoJev re-encodes the whole state once per candidate.
|
| 11 |
+
Here the state and question are encoded once and candidates cross-attend to
|
| 12 |
+
that encoding, so cost is ``|prefix| + K*|candidate|`` rather than
|
| 13 |
+
``K*(|prefix| + |candidate|)``.
|
| 14 |
+
2. **Structured board input.** A 50x50 maze flattened into ASCII puts
|
| 15 |
+
vertically adjacent cells 51 tokens apart. Jevon reads the board as a
|
| 16 |
+
``(C, H, W)`` tensor through :class:`~jevon.planner.SpatialPlanner`, which
|
| 17 |
+
has the 2D adjacency built in.
|
| 18 |
+
3. **Anchored candidates.** A candidate that names a cell carries that cell's
|
| 19 |
+
coordinate, so the scoring head reads the planner's feature for exactly the
|
| 20 |
+
cell the candidate would move to.
|
| 21 |
+
"""
|
| 22 |
+
from __future__ import annotations
|
| 23 |
+
|
| 24 |
+
from dataclasses import dataclass, field
|
| 25 |
+
from typing import Optional
|
| 26 |
+
|
| 27 |
+
import torch
|
| 28 |
+
from torch import nn
|
| 29 |
+
import torch.nn.functional as F
|
| 30 |
+
|
| 31 |
+
from .planner import MinPlusField, SpatialPlanner, default_iterations
|
| 32 |
+
|
| 33 |
+
QUESTION_TYPES: tuple[str, ...] = ("choice", "boolean", "score")
|
| 34 |
+
TYPE_INDEX = {name: i for i, name in enumerate(QUESTION_TYPES)}
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
@dataclass
|
| 38 |
+
class JevonConfig:
|
| 39 |
+
vocab_size: int = 4096
|
| 40 |
+
d_model: int = 384
|
| 41 |
+
n_heads: int = 6
|
| 42 |
+
prefix_layers: int = 6
|
| 43 |
+
candidate_layers: int = 2
|
| 44 |
+
set_layers: int = 2
|
| 45 |
+
ffn_mult: int = 4
|
| 46 |
+
dropout: float = 0.1
|
| 47 |
+
max_prefix_len: int = 320
|
| 48 |
+
max_candidate_len: int = 48
|
| 49 |
+
# Spatial planner
|
| 50 |
+
grid_channels: int = 8
|
| 51 |
+
planner_dim: int = 96
|
| 52 |
+
planner_blocks: int = 2
|
| 53 |
+
planner_grad_steps: int = 8
|
| 54 |
+
local_patch: int = 5
|
| 55 |
+
use_planner: bool = True
|
| 56 |
+
# Let the text side read the planner's own distance/reach read-out at
|
| 57 |
+
# the cells it gathers, instead of re-deriving it from raw features.
|
| 58 |
+
expose_field: bool = False
|
| 59 |
+
# Read the distance off a min-plus recurrence instead of a 1x1 conv, which
|
| 60 |
+
# makes the field a genuine shortest-path distance and so traversable by
|
| 61 |
+
# construction. Pairs with --field-target linear; see MinPlusField.
|
| 62 |
+
min_plus_field: bool = False
|
| 63 |
+
|
| 64 |
+
def to_dict(self) -> dict:
|
| 65 |
+
return dict(self.__dict__)
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
def _mha(d_model: int, n_heads: int, dropout: float) -> nn.MultiheadAttention:
|
| 69 |
+
return nn.MultiheadAttention(d_model, n_heads, dropout=dropout, batch_first=True)
|
| 70 |
+
|
| 71 |
+
|
| 72 |
+
class EncoderLayer(nn.Module):
|
| 73 |
+
"""Pre-norm self-attention block, optionally with cross-attention."""
|
| 74 |
+
|
| 75 |
+
def __init__(self, cfg: JevonConfig, cross: bool = False):
|
| 76 |
+
super().__init__()
|
| 77 |
+
d = cfg.d_model
|
| 78 |
+
self.norm_self = nn.LayerNorm(d)
|
| 79 |
+
self.self_attn = _mha(d, cfg.n_heads, cfg.dropout)
|
| 80 |
+
self.cross = cross
|
| 81 |
+
if cross:
|
| 82 |
+
self.norm_q = nn.LayerNorm(d)
|
| 83 |
+
self.norm_kv = nn.LayerNorm(d)
|
| 84 |
+
self.cross_attn = _mha(d, cfg.n_heads, cfg.dropout)
|
| 85 |
+
self.norm_ffn = nn.LayerNorm(d)
|
| 86 |
+
self.ffn = nn.Sequential(
|
| 87 |
+
nn.Linear(d, d * cfg.ffn_mult), nn.GELU(),
|
| 88 |
+
nn.Dropout(cfg.dropout), nn.Linear(d * cfg.ffn_mult, d),
|
| 89 |
+
)
|
| 90 |
+
self.drop = nn.Dropout(cfg.dropout)
|
| 91 |
+
|
| 92 |
+
def forward(self, x, pad_mask=None, memory=None, memory_pad_mask=None):
|
| 93 |
+
h = self.norm_self(x)
|
| 94 |
+
attn, _ = self.self_attn(h, h, h, key_padding_mask=pad_mask, need_weights=False)
|
| 95 |
+
x = x + self.drop(attn)
|
| 96 |
+
if self.cross and memory is not None:
|
| 97 |
+
q = self.norm_q(x)
|
| 98 |
+
kv = self.norm_kv(memory)
|
| 99 |
+
attn, _ = self.cross_attn(q, kv, kv, key_padding_mask=memory_pad_mask,
|
| 100 |
+
need_weights=False)
|
| 101 |
+
x = x + self.drop(attn)
|
| 102 |
+
return x + self.drop(self.ffn(self.norm_ffn(x)))
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
class JevonModel(nn.Module):
|
| 106 |
+
def __init__(self, cfg: JevonConfig):
|
| 107 |
+
super().__init__()
|
| 108 |
+
self.cfg = cfg
|
| 109 |
+
d = cfg.d_model
|
| 110 |
+
self.token_embed = nn.Embedding(cfg.vocab_size, d, padding_idx=0)
|
| 111 |
+
self.prefix_pos = nn.Embedding(cfg.max_prefix_len, d)
|
| 112 |
+
self.candidate_pos = nn.Embedding(cfg.max_candidate_len, d)
|
| 113 |
+
self.type_embed = nn.Embedding(len(QUESTION_TYPES), d)
|
| 114 |
+
self.drop = nn.Dropout(cfg.dropout)
|
| 115 |
+
|
| 116 |
+
self.prefix_layers = nn.ModuleList(
|
| 117 |
+
EncoderLayer(cfg) for _ in range(cfg.prefix_layers))
|
| 118 |
+
self.prefix_norm = nn.LayerNorm(d)
|
| 119 |
+
self.candidate_layers = nn.ModuleList(
|
| 120 |
+
EncoderLayer(cfg, cross=True) for _ in range(cfg.candidate_layers))
|
| 121 |
+
self.candidate_norm = nn.LayerNorm(d)
|
| 122 |
+
self.candidate_pool = nn.Linear(d, 1)
|
| 123 |
+
|
| 124 |
+
if cfg.use_planner:
|
| 125 |
+
self.planner = SpatialPlanner(cfg.grid_channels, cfg.planner_dim,
|
| 126 |
+
cfg.planner_blocks)
|
| 127 |
+
gathered = cfg.planner_dim + (2 if cfg.expose_field else 0)
|
| 128 |
+
self.gathered_dim = gathered
|
| 129 |
+
self.grid_to_text = nn.Linear(gathered, d)
|
| 130 |
+
self.grid_token_type = nn.Embedding(4, d) # pooled / focus / target / patch
|
| 131 |
+
self.anchor_proj = nn.Sequential(
|
| 132 |
+
nn.Linear(gathered * 3, d), nn.GELU(), nn.Linear(d, d))
|
| 133 |
+
# Dense read-out used only as a training signal: channel 0 is the
|
| 134 |
+
# squashed distance to the target, channel 1 a reachability logit.
|
| 135 |
+
# Every free cell supervises the recurrence, so the planner learns
|
| 136 |
+
# a real value function instead of whatever the sparse decision
|
| 137 |
+
# questions happen to reward.
|
| 138 |
+
self.field_head = nn.Conv2d(cfg.planner_dim, 2, 1)
|
| 139 |
+
# The min-plus head is a read-out, not a replacement. Its output
|
| 140 |
+
# matches BFS at initialisation, so a field loss on it is ~0 from
|
| 141 |
+
# step one and its zero-initialised cost conv passes no gradient
|
| 142 |
+
# back -- which would quietly retire the dense planner supervision
|
| 143 |
+
# section 3 exists to provide. Keeping the conv head supervised
|
| 144 |
+
# preserves that signal; the min-plus head is what gets read.
|
| 145 |
+
self.min_plus = (MinPlusField(cfg.planner_dim)
|
| 146 |
+
if cfg.min_plus_field else None)
|
| 147 |
+
else:
|
| 148 |
+
self.planner = None
|
| 149 |
+
self.field_head = None
|
| 150 |
+
self.min_plus = None
|
| 151 |
+
|
| 152 |
+
feature_dim = d * 2
|
| 153 |
+
self.score_head = nn.Sequential(
|
| 154 |
+
nn.LayerNorm(feature_dim), nn.Linear(feature_dim, d), nn.GELU(),
|
| 155 |
+
nn.Linear(d, 1))
|
| 156 |
+
self.set_layers = nn.ModuleList(
|
| 157 |
+
EncoderLayer(cfg) for _ in range(cfg.set_layers))
|
| 158 |
+
self.set_norm = nn.LayerNorm(d)
|
| 159 |
+
self.set_out = nn.Linear(d, 1)
|
| 160 |
+
nn.init.zeros_(self.set_out.weight)
|
| 161 |
+
nn.init.zeros_(self.set_out.bias)
|
| 162 |
+
self.count_embed = nn.Linear(1, d)
|
| 163 |
+
|
| 164 |
+
self.apply(self._init)
|
| 165 |
+
|
| 166 |
+
@staticmethod
|
| 167 |
+
def _init(module: nn.Module) -> None:
|
| 168 |
+
if isinstance(module, nn.Linear):
|
| 169 |
+
nn.init.normal_(module.weight, std=0.02)
|
| 170 |
+
if module.bias is not None:
|
| 171 |
+
nn.init.zeros_(module.bias)
|
| 172 |
+
elif isinstance(module, nn.Embedding):
|
| 173 |
+
nn.init.normal_(module.weight, std=0.02)
|
| 174 |
+
|
| 175 |
+
# ------------------------------------------------------------------
|
| 176 |
+
# Channel 2 is the agent/head cell. It is withheld from the planner so the
|
| 177 |
+
# field it computes describes the *board* -- distance to the goal, reachable
|
| 178 |
+
# space around the food -- and not the agent. Two things follow: a maze field
|
| 179 |
+
# stays valid for a whole episode and can be computed once, and the readout
|
| 180 |
+
# below is forced to locate the agent by gathering cells rather than by
|
| 181 |
+
# memorising a position-specific shortcut.
|
| 182 |
+
FOCUS_CHANNEL = 2
|
| 183 |
+
|
| 184 |
+
def encode_boards(self, boards, passable, iterations, grad_steps=None):
|
| 185 |
+
if self.planner is None or boards is None:
|
| 186 |
+
return None
|
| 187 |
+
boards = boards.clone()
|
| 188 |
+
boards[:, self.FOCUS_CHANNEL] = 0.0
|
| 189 |
+
return self.planner(
|
| 190 |
+
boards, passable, iterations=iterations,
|
| 191 |
+
grad_steps=self.cfg.planner_grad_steps if grad_steps is None else grad_steps)
|
| 192 |
+
|
| 193 |
+
def predict_field(self, cells: torch.Tensor, board: Optional[torch.Tensor] = None,
|
| 194 |
+
passable: Optional[torch.Tensor] = None, *,
|
| 195 |
+
iterations: Optional[int] = None,
|
| 196 |
+
grad_steps: Optional[int] = None) -> torch.Tensor:
|
| 197 |
+
"""(S, P, H, W) planner features -> (S, 2, H, W) distance / reach.
|
| 198 |
+
|
| 199 |
+
This is the *read-out*: what the controller descends and what
|
| 200 |
+
``expose_field`` shows the text side. :meth:`supervised_field` is the
|
| 201 |
+
one the field loss trains. They are the same conv head unless
|
| 202 |
+
``min_plus_field`` is set.
|
| 203 |
+
|
| 204 |
+
``board`` and ``passable`` are required only by the min-plus head, which
|
| 205 |
+
needs the walls and the goal to run its recurrence over. The conv head
|
| 206 |
+
ignores them, so existing callers keep working unchanged.
|
| 207 |
+
"""
|
| 208 |
+
if self.min_plus is None:
|
| 209 |
+
return self.field_head(cells)
|
| 210 |
+
if board is None or passable is None:
|
| 211 |
+
raise ValueError(
|
| 212 |
+
"min_plus_field needs the board and passable mask; pass them to "
|
| 213 |
+
"predict_field(cells, board, passable)")
|
| 214 |
+
steps = iterations or default_iterations(min(board.shape[-2:]))
|
| 215 |
+
return self.min_plus(
|
| 216 |
+
cells, board, passable, iterations=steps,
|
| 217 |
+
grad_steps=(self.cfg.planner_grad_steps if grad_steps is None
|
| 218 |
+
else grad_steps))
|
| 219 |
+
|
| 220 |
+
def supervised_field(self, cells: torch.Tensor) -> torch.Tensor:
|
| 221 |
+
"""The field head the loss trains, which is always the conv one.
|
| 222 |
+
|
| 223 |
+
Its job is to keep dense gradient flowing into the recurrence, so it
|
| 224 |
+
stays on even when the min-plus read-out makes its output redundant for
|
| 225 |
+
the controller.
|
| 226 |
+
"""
|
| 227 |
+
return self.field_head(cells)
|
| 228 |
+
|
| 229 |
+
def _readable(self, cells: torch.Tensor, board: Optional[torch.Tensor] = None,
|
| 230 |
+
passable: Optional[torch.Tensor] = None) -> torch.Tensor:
|
| 231 |
+
"""Planner features as the text side sees them.
|
| 232 |
+
|
| 233 |
+
The field head is trained on every free cell, so its output is the most
|
| 234 |
+
distilled form of exactly what the distance question asks. Without this
|
| 235 |
+
the transformer has to re-learn that same linear map from the raw
|
| 236 |
+
planner features, through a far sparser signal.
|
| 237 |
+
|
| 238 |
+
Computed once per forward pass and threaded to both gather sites. The
|
| 239 |
+
min-plus head runs a T-step recurrence, and T is in the hundreds, so
|
| 240 |
+
recomputing it per site was pure duplicated work.
|
| 241 |
+
"""
|
| 242 |
+
if not self.cfg.expose_field:
|
| 243 |
+
return cells
|
| 244 |
+
return torch.cat([cells, self.predict_field(cells, board, passable)], dim=1)
|
| 245 |
+
|
| 246 |
+
def _grid_tokens(self, cells, question_state, focus, target):
|
| 247 |
+
"""Board summary tokens prepended to each question's prefix.
|
| 248 |
+
|
| 249 |
+
``cells`` (S, P, H, W) planner features
|
| 250 |
+
``focus`` (Q, 2) the agent/head cell; ``target`` (Q, 2) goal/food cell.
|
| 251 |
+
"""
|
| 252 |
+
q_cells = cells[question_state] # (Q, P, H, W)
|
| 253 |
+
flat = q_cells.flatten(2) # (Q, P, HW)
|
| 254 |
+
pooled = torch.stack([flat.mean(-1), flat.amax(-1)], dim=1) # (Q, 2, P)
|
| 255 |
+
focus_vec = self._gather_cells(q_cells, focus).unsqueeze(1)
|
| 256 |
+
target_vec = self._gather_cells(q_cells, target).unsqueeze(1)
|
| 257 |
+
patch = self._gather_patch(q_cells, focus, self.cfg.local_patch)
|
| 258 |
+
|
| 259 |
+
tokens = torch.cat([pooled, focus_vec, target_vec, patch], dim=1)
|
| 260 |
+
tokens = self.grid_to_text(tokens)
|
| 261 |
+
kinds = torch.tensor(
|
| 262 |
+
[0, 0, 1, 2] + [3] * patch.shape[1], device=tokens.device)
|
| 263 |
+
return tokens + self.grid_token_type(kinds).unsqueeze(0)
|
| 264 |
+
|
| 265 |
+
@staticmethod
|
| 266 |
+
def _gather_cells(cells, coords):
|
| 267 |
+
"""``cells`` (Q, P, H, W), ``coords`` (Q, 2) -> (Q, P). Negative rows give zeros."""
|
| 268 |
+
q, p, h, w = cells.shape
|
| 269 |
+
rows = coords[:, 0].clamp(0, h - 1)
|
| 270 |
+
cols = coords[:, 1].clamp(0, w - 1)
|
| 271 |
+
index = (rows * w + cols).view(q, 1, 1).expand(q, p, 1)
|
| 272 |
+
out = cells.flatten(2).gather(2, index).squeeze(2)
|
| 273 |
+
return out * (coords[:, :1] >= 0).to(out.dtype)
|
| 274 |
+
|
| 275 |
+
@staticmethod
|
| 276 |
+
def _gather_patch(cells, centre, patch):
|
| 277 |
+
q, p, h, w = cells.shape
|
| 278 |
+
radius = patch // 2
|
| 279 |
+
offsets = torch.arange(-radius, radius + 1, device=cells.device)
|
| 280 |
+
dr = offsets.view(-1, 1).expand(patch, patch).reshape(-1)
|
| 281 |
+
dc = offsets.view(1, -1).expand(patch, patch).reshape(-1)
|
| 282 |
+
rows = (centre[:, :1] + dr.view(1, -1))
|
| 283 |
+
cols = (centre[:, 1:2] + dc.view(1, -1))
|
| 284 |
+
inside = ((rows >= 0) & (rows < h) & (cols >= 0) & (cols < w)).unsqueeze(-1)
|
| 285 |
+
index = (rows.clamp(0, h - 1) * w + cols.clamp(0, w - 1))
|
| 286 |
+
index = index.unsqueeze(1).expand(q, p, patch * patch)
|
| 287 |
+
out = cells.flatten(2).gather(2, index).transpose(1, 2) # (Q, patch^2, P)
|
| 288 |
+
return out * inside.to(out.dtype)
|
| 289 |
+
|
| 290 |
+
# ------------------------------------------------------------------
|
| 291 |
+
def forward(self, batch: dict, *, iterations: Optional[int] = None,
|
| 292 |
+
grad_steps: Optional[int] = None,
|
| 293 |
+
cells: Optional[torch.Tensor] = None) -> torch.Tensor:
|
| 294 |
+
"""Returns candidate logits of shape ``(Q, Kmax)``, padded with -1e9.
|
| 295 |
+
|
| 296 |
+
``cells`` lets a caller supply a planner field computed earlier. A maze
|
| 297 |
+
field depends only on walls and goal, so an episode computes it once and
|
| 298 |
+
reuses it for every step.
|
| 299 |
+
"""
|
| 300 |
+
cfg = self.cfg
|
| 301 |
+
prefix_ids = batch["prefix_ids"] # (Q, Lp)
|
| 302 |
+
cand_ids = batch["candidate_ids"] # (Q, K, Lc)
|
| 303 |
+
cand_valid = batch["candidate_valid"] # (Q, K) bool
|
| 304 |
+
q, kmax, lc = cand_ids.shape
|
| 305 |
+
device = prefix_ids.device
|
| 306 |
+
|
| 307 |
+
if cells is None and cfg.use_planner and batch.get("boards") is not None:
|
| 308 |
+
steps = iterations or default_iterations(int(batch["board_size"].max()))
|
| 309 |
+
cells = self.encode_boards(batch["boards"], batch["passable"], steps, grad_steps)
|
| 310 |
+
|
| 311 |
+
# --- prefix ---------------------------------------------------
|
| 312 |
+
positions = torch.arange(prefix_ids.shape[1], device=device)
|
| 313 |
+
x = self.token_embed(prefix_ids) + self.prefix_pos(positions).unsqueeze(0)
|
| 314 |
+
x = x + self.type_embed(batch["question_type"]).unsqueeze(1)
|
| 315 |
+
prefix_pad = prefix_ids.eq(0)
|
| 316 |
+
readable = (self._readable(cells, batch.get("boards"), batch.get("passable"))
|
| 317 |
+
if cells is not None else None)
|
| 318 |
+
if cells is not None:
|
| 319 |
+
grid_tokens = self._grid_tokens(
|
| 320 |
+
readable, batch["question_state"], batch["focus"], batch["target"])
|
| 321 |
+
x = torch.cat([grid_tokens, x], dim=1)
|
| 322 |
+
prefix_pad = torch.cat(
|
| 323 |
+
[torch.zeros(q, grid_tokens.shape[1], dtype=torch.bool, device=device),
|
| 324 |
+
prefix_pad], dim=1)
|
| 325 |
+
x = self.drop(x)
|
| 326 |
+
for layer in self.prefix_layers:
|
| 327 |
+
x = layer(x, pad_mask=prefix_pad)
|
| 328 |
+
memory = self.prefix_norm(x)
|
| 329 |
+
|
| 330 |
+
# --- candidates cross-attend to the shared prefix --------------
|
| 331 |
+
positions = torch.arange(lc, device=device)
|
| 332 |
+
c = self.token_embed(cand_ids) + self.candidate_pos(positions).view(1, 1, lc, -1)
|
| 333 |
+
c = self.drop(c).reshape(q * kmax, lc, cfg.d_model)
|
| 334 |
+
cand_pad = cand_ids.eq(0).reshape(q * kmax, lc)
|
| 335 |
+
# A fully padded row would make softmax produce NaNs; let it see slot 0.
|
| 336 |
+
empty = cand_pad.all(dim=1)
|
| 337 |
+
cand_pad = cand_pad.clone()
|
| 338 |
+
cand_pad[empty, 0] = False
|
| 339 |
+
mem = memory.unsqueeze(1).expand(q, kmax, memory.shape[1], cfg.d_model)
|
| 340 |
+
mem = mem.reshape(q * kmax, memory.shape[1], cfg.d_model)
|
| 341 |
+
mem_pad = prefix_pad.unsqueeze(1).expand(q, kmax, prefix_pad.shape[1])
|
| 342 |
+
mem_pad = mem_pad.reshape(q * kmax, prefix_pad.shape[1])
|
| 343 |
+
for layer in self.candidate_layers:
|
| 344 |
+
c = layer(c, pad_mask=cand_pad, memory=mem, memory_pad_mask=mem_pad)
|
| 345 |
+
c = self.candidate_norm(c)
|
| 346 |
+
|
| 347 |
+
weights = self.candidate_pool(c).masked_fill(cand_pad.unsqueeze(-1), -1e9)
|
| 348 |
+
pooled = (c * weights.softmax(dim=1)).sum(dim=1).reshape(q, kmax, cfg.d_model)
|
| 349 |
+
|
| 350 |
+
# --- anchor the candidate on the cell it would move to ---------
|
| 351 |
+
if cells is not None:
|
| 352 |
+
anchors = batch["anchors"] # (Q, K, 2)
|
| 353 |
+
q_cells = readable[batch["question_state"]]
|
| 354 |
+
flat_anchor = anchors.reshape(q * kmax, 2)
|
| 355 |
+
rep = q_cells.repeat_interleave(kmax, dim=0)
|
| 356 |
+
anchor_vec = self._gather_cells(rep, flat_anchor).reshape(q, kmax, -1)
|
| 357 |
+
focus_vec = self._gather_cells(q_cells, batch["focus"]).unsqueeze(1)
|
| 358 |
+
target_vec = self._gather_cells(q_cells, batch["target"]).unsqueeze(1)
|
| 359 |
+
pooled = pooled + self.anchor_proj(torch.cat([
|
| 360 |
+
anchor_vec,
|
| 361 |
+
anchor_vec - focus_vec.expand_as(anchor_vec),
|
| 362 |
+
target_vec.expand_as(anchor_vec),
|
| 363 |
+
], dim=-1))
|
| 364 |
+
|
| 365 |
+
# --- set reasoning over the live candidate set -----------------
|
| 366 |
+
count = cand_valid.sum(-1, keepdim=True).clamp(min=1).float().log()
|
| 367 |
+
h = pooled + self.count_embed(count).unsqueeze(1)
|
| 368 |
+
set_pad = ~cand_valid
|
| 369 |
+
set_pad = set_pad.clone()
|
| 370 |
+
set_pad[set_pad.all(dim=1), 0] = False
|
| 371 |
+
for layer in self.set_layers:
|
| 372 |
+
h = layer(h, pad_mask=set_pad)
|
| 373 |
+
h = self.set_norm(h)
|
| 374 |
+
delta = self.set_out(h).squeeze(-1)
|
| 375 |
+
|
| 376 |
+
base = self.score_head(torch.cat([pooled, h], dim=-1)).squeeze(-1)
|
| 377 |
+
logits = base + delta
|
| 378 |
+
|
| 379 |
+
# Boolean questions carry one semantic path; its logit pairs with a
|
| 380 |
+
# fixed zero so the distribution is [P(false), P(true)].
|
| 381 |
+
is_boolean = batch["question_type"].eq(TYPE_INDEX["boolean"])
|
| 382 |
+
if is_boolean.any():
|
| 383 |
+
paired = torch.stack([torch.zeros_like(logits[:, 0]), logits[:, 0]], dim=-1)
|
| 384 |
+
padded = F.pad(paired, (0, max(kmax - 2, 0)), value=-1e9)[:, :kmax]
|
| 385 |
+
logits = torch.where(is_boolean.unsqueeze(-1), padded, logits)
|
| 386 |
+
|
| 387 |
+
return logits.masked_fill(~cand_valid, -1e9)
|
| 388 |
+
|
| 389 |
+
def num_parameters(self) -> int:
|
| 390 |
+
return sum(p.numel() for p in self.parameters())
|
|
@@ -0,0 +1,227 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Recurrent spatial planner: amortised value iteration over a grid.
|
| 2 |
+
|
| 3 |
+
A fixed-depth transformer reading a flattened ASCII board cannot compute a
|
| 4 |
+
shortest path: the answer needs a number of propagation steps proportional to
|
| 5 |
+
the path length, and on a 50x50 corridor maze that is a couple of hundred.
|
| 6 |
+
This module supplies that depth with a weight-shared message-passing block run
|
| 7 |
+
for ``T`` iterations over the 4-neighbourhood, which is the same recurrence
|
| 8 |
+
Bellman-Ford uses. Walls block messages, so the signal follows corridors.
|
| 9 |
+
|
| 10 |
+
Two properties matter for generalisation:
|
| 11 |
+
|
| 12 |
+
* weights are shared across iterations, so ``T`` can be raised at inference to
|
| 13 |
+
suit a larger board than anything seen in training;
|
| 14 |
+
* only the final ``grad_steps`` iterations carry gradient (the earlier ones run
|
| 15 |
+
under ``no_grad``), which keeps memory flat in ``T`` and is the standard
|
| 16 |
+
phantom-gradient trick for implicit/equilibrium models.
|
| 17 |
+
"""
|
| 18 |
+
from __future__ import annotations
|
| 19 |
+
|
| 20 |
+
import torch
|
| 21 |
+
from torch import nn
|
| 22 |
+
import torch.nn.functional as F
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
def _shift(x: torch.Tensor, dr: int, dc: int, fill: float = 0.0) -> torch.Tensor:
|
| 26 |
+
"""Translate a ``(B, D, H, W)`` map, filling the exposed border with ``fill``."""
|
| 27 |
+
if dr:
|
| 28 |
+
x = F.pad(x, (0, 0, max(dr, 0), max(-dr, 0)), value=fill)
|
| 29 |
+
x = x[:, :, : x.shape[2] - abs(dr), :] if dr > 0 else x[:, :, abs(dr):, :]
|
| 30 |
+
if dc:
|
| 31 |
+
x = F.pad(x, (max(dc, 0), max(-dc, 0), 0, 0), value=fill)
|
| 32 |
+
x = x[:, :, :, : x.shape[3] - abs(dc)] if dc > 0 else x[:, :, :, abs(dc):]
|
| 33 |
+
return x
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
class RMSNorm2d(nn.Module):
|
| 37 |
+
def __init__(self, dim: int, eps: float = 1e-6):
|
| 38 |
+
super().__init__()
|
| 39 |
+
self.weight = nn.Parameter(torch.ones(dim))
|
| 40 |
+
self.eps = eps
|
| 41 |
+
|
| 42 |
+
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
| 43 |
+
norm = x.float().pow(2).mean(dim=1, keepdim=True).add(self.eps).rsqrt()
|
| 44 |
+
return (x.float() * norm).to(x.dtype) * self.weight.view(1, -1, 1, 1)
|
| 45 |
+
|
| 46 |
+
|
| 47 |
+
class PlannerBlock(nn.Module):
|
| 48 |
+
"""One propagation step: gather from passable neighbours, then gate an update."""
|
| 49 |
+
|
| 50 |
+
def __init__(self, dim: int, static_dim: int, hidden_mult: int = 2):
|
| 51 |
+
super().__init__()
|
| 52 |
+
self.norm = RMSNorm2d(dim)
|
| 53 |
+
self.message = nn.Conv2d(dim, dim, 1, bias=False)
|
| 54 |
+
# max-pool and mean-pool over neighbours, the current cell, and the
|
| 55 |
+
# static board channels (walls, goal, food) which never change with t.
|
| 56 |
+
fan_in = dim * 3 + static_dim
|
| 57 |
+
hidden = dim * hidden_mult
|
| 58 |
+
self.mix = nn.Sequential(
|
| 59 |
+
nn.Conv2d(fan_in, hidden, 1),
|
| 60 |
+
nn.GELU(),
|
| 61 |
+
nn.Conv2d(hidden, dim * 2, 1),
|
| 62 |
+
)
|
| 63 |
+
# A zero final weight would start the recurrence at exactly the identity
|
| 64 |
+
# but also zero the gradient of everything feeding it, so the first step
|
| 65 |
+
# trains nothing upstream. Small random weights plus a negative gate bias
|
| 66 |
+
# give a near-identity start that is still differentiable everywhere.
|
| 67 |
+
nn.init.normal_(self.mix[-1].weight, std=0.02)
|
| 68 |
+
nn.init.zeros_(self.mix[-1].bias)
|
| 69 |
+
self.gate_bias = nn.Parameter(torch.full((dim,), -2.0))
|
| 70 |
+
|
| 71 |
+
def forward(self, h: torch.Tensor, passable: torch.Tensor,
|
| 72 |
+
static: torch.Tensor) -> torch.Tensor:
|
| 73 |
+
normed = self.norm(h)
|
| 74 |
+
msg = self.message(normed) * passable
|
| 75 |
+
neighbours = torch.stack(
|
| 76 |
+
[_shift(msg, -1, 0), _shift(msg, 1, 0), _shift(msg, 0, -1), _shift(msg, 0, 1)],
|
| 77 |
+
dim=0,
|
| 78 |
+
)
|
| 79 |
+
# Masked max-pool: a wall neighbour must not win the max.
|
| 80 |
+
agg_max = neighbours.amax(dim=0)
|
| 81 |
+
agg_mean = neighbours.mean(dim=0)
|
| 82 |
+
update, gate = self.mix(
|
| 83 |
+
torch.cat([normed, agg_max, agg_mean, static], dim=1)
|
| 84 |
+
).chunk(2, dim=1)
|
| 85 |
+
gate = torch.sigmoid(gate + self.gate_bias.view(1, -1, 1, 1))
|
| 86 |
+
return (h + gate * update) * passable
|
| 87 |
+
|
| 88 |
+
|
| 89 |
+
class SpatialPlanner(nn.Module):
|
| 90 |
+
"""Stack of shared-weight propagation steps over a board."""
|
| 91 |
+
|
| 92 |
+
def __init__(self, in_channels: int, dim: int = 64, blocks: int = 2,
|
| 93 |
+
hidden_mult: int = 2):
|
| 94 |
+
super().__init__()
|
| 95 |
+
self.dim = dim
|
| 96 |
+
self.stem = nn.Sequential(
|
| 97 |
+
nn.Conv2d(in_channels, dim, 3, padding=1),
|
| 98 |
+
nn.GELU(),
|
| 99 |
+
nn.Conv2d(dim, dim, 1),
|
| 100 |
+
)
|
| 101 |
+
self.static_proj = nn.Conv2d(in_channels, dim // 2, 1)
|
| 102 |
+
self.blocks = nn.ModuleList(
|
| 103 |
+
PlannerBlock(dim, dim // 2, hidden_mult) for _ in range(blocks)
|
| 104 |
+
)
|
| 105 |
+
self.out_norm = RMSNorm2d(dim)
|
| 106 |
+
|
| 107 |
+
def forward(self, board: torch.Tensor, passable: torch.Tensor, *,
|
| 108 |
+
iterations: int, grad_steps: int = 8) -> torch.Tensor:
|
| 109 |
+
"""``board``: (B, C, H, W). ``passable``: (B, 1, H, W) with 1 on open cells.
|
| 110 |
+
|
| 111 |
+
Returns per-cell features of shape (B, dim, H, W).
|
| 112 |
+
"""
|
| 113 |
+
static = self.static_proj(board)
|
| 114 |
+
stem = self.stem(board) * passable
|
| 115 |
+
h = stem
|
| 116 |
+
|
| 117 |
+
no_grad_steps = max(0, iterations - max(grad_steps, 1))
|
| 118 |
+
if no_grad_steps:
|
| 119 |
+
with torch.no_grad():
|
| 120 |
+
for step in range(no_grad_steps):
|
| 121 |
+
h = self.blocks[step % len(self.blocks)](h, passable, static)
|
| 122 |
+
# Re-attach the stem so it still receives a gradient. Detaching the
|
| 123 |
+
# loop alone would sever the stem's only path to the loss and leave
|
| 124 |
+
# the input projection frozen at initialisation.
|
| 125 |
+
h = h.detach() + (stem - stem.detach())
|
| 126 |
+
for step in range(no_grad_steps, iterations):
|
| 127 |
+
h = self.blocks[step % len(self.blocks)](h, passable, static)
|
| 128 |
+
return self.out_norm(h)
|
| 129 |
+
|
| 130 |
+
|
| 131 |
+
def default_iterations(size: int, cap: int = 4096) -> int:
|
| 132 |
+
"""Propagation budget for a board of side ``size``.
|
| 133 |
+
|
| 134 |
+
One iteration moves information one cell, so the budget has to cover the
|
| 135 |
+
board's *diameter*, not its side length. A spanning-tree maze winds through
|
| 136 |
+
nearly every free cell: measured diameters are 48 at 11x11, 312 at 31x31 and
|
| 137 |
+
750 at 51x51, all far beyond any multiple of the side length. The quadratic
|
| 138 |
+
term tracks that growth with roughly 2x headroom, which is affordable because
|
| 139 |
+
one iteration is a pair of 1x1 convolutions.
|
| 140 |
+
"""
|
| 141 |
+
return int(min(cap, max(32, 2 * size + (size * size) // 2)))
|
| 142 |
+
|
| 143 |
+
|
| 144 |
+
class MinPlusField(nn.Module):
|
| 145 |
+
"""Shortest-path distance under a learned per-cell cost.
|
| 146 |
+
|
| 147 |
+
The planner's features are read by a 1x1 convolution to get a distance, and
|
| 148 |
+
nothing about that arrangement makes the result traversable. Measured on a
|
| 149 |
+
trained checkpoint, 72% of cells at 11x11 and 93% at 21x21 descended into a
|
| 150 |
+
spurious basin and cycled; penalising the local ordering directly moved the
|
| 151 |
+
reachable fraction only from 0.104 to 0.132 even at five times the weight.
|
| 152 |
+
Global monotonicity is not something a sum of local penalties delivers.
|
| 153 |
+
|
| 154 |
+
A min-plus recurrence delivers it by construction. Iterating
|
| 155 |
+
|
| 156 |
+
v(c) <- cost(c) + min over passable neighbours n of v(n), v(goal) = 0
|
| 157 |
+
|
| 158 |
+
from ``v = BIG`` is Bellman-Ford, so with strictly positive costs the fixed
|
| 159 |
+
point *is* a shortest-path distance: every non-goal cell has a neighbour
|
| 160 |
+
strictly below it, and greedy descent therefore terminates at the goal from
|
| 161 |
+
anywhere. The trap mode is not reduced, it is removed.
|
| 162 |
+
|
| 163 |
+
What is left to learn is a cost per cell -- a local quantity, which is the
|
| 164 |
+
kind of thing a convolution is good at -- while the global structure comes
|
| 165 |
+
from the recurrence. This is the Value-Iteration-Network idea (Tamar et
|
| 166 |
+
al., 2016) with the max-plus reward recurrence replaced by the min-plus
|
| 167 |
+
distance one the field head is actually supervised on.
|
| 168 |
+
|
| 169 |
+
Costs are emitted in units of ``1 / area``, so the output is on the same
|
| 170 |
+
scale as a ``linear`` distance target, whose steps are a uniform
|
| 171 |
+
``1 / area`` apart. That mattered while this head was the one the field
|
| 172 |
+
loss trained. It no longer is: the conv head stayed supervised precisely so
|
| 173 |
+
the dense signal survives, and this head is only ever read. Nothing here is
|
| 174 |
+
compared against a target, so ``--field-target`` is free to be whatever
|
| 175 |
+
trains the planner best, and the two flags are independent.
|
| 176 |
+
"""
|
| 177 |
+
|
| 178 |
+
BIG = 1.0e4
|
| 179 |
+
|
| 180 |
+
def __init__(self, dim: int, target_channel: int = 3):
|
| 181 |
+
super().__init__()
|
| 182 |
+
self.cost = nn.Conv2d(dim, 1, 1)
|
| 183 |
+
self.reach = nn.Conv2d(dim, 1, 1)
|
| 184 |
+
self.target_channel = target_channel
|
| 185 |
+
# softplus(0.54) ~= 1.0, so the recurrence starts out measuring plain
|
| 186 |
+
# step count and the model adjusts from a sensible place rather than
|
| 187 |
+
# from a cost of zero, which would make every distance zero.
|
| 188 |
+
nn.init.zeros_(self.cost.weight)
|
| 189 |
+
nn.init.constant_(self.cost.bias, 0.5413)
|
| 190 |
+
|
| 191 |
+
def forward(self, cells: torch.Tensor, board: torch.Tensor,
|
| 192 |
+
passable: torch.Tensor, *, iterations: int,
|
| 193 |
+
grad_steps: int = 8) -> torch.Tensor:
|
| 194 |
+
"""``(S, dim, H, W)`` features -> ``(S, 2, H, W)`` distance / reach."""
|
| 195 |
+
inside = board.abs().sum(dim=1, keepdim=True) > 0 # (S,1,H,W)
|
| 196 |
+
area = inside.flatten(1).sum(dim=1).clamp(min=4.0)
|
| 197 |
+
area = area.to(cells.dtype).view(-1, 1, 1, 1)
|
| 198 |
+
cost = F.softplus(self.cost(cells)) / area
|
| 199 |
+
|
| 200 |
+
goal = board[:, self.target_channel: self.target_channel + 1] > 0.5
|
| 201 |
+
open_cell = (passable > 0.5) | goal # the goal must be enterable
|
| 202 |
+
zero = torch.zeros((), dtype=cells.dtype, device=cells.device)
|
| 203 |
+
big = torch.full((), self.BIG, dtype=cells.dtype, device=cells.device)
|
| 204 |
+
|
| 205 |
+
v = torch.where(goal, zero, big).expand_as(cost).contiguous()
|
| 206 |
+
|
| 207 |
+
def sweep(v: torch.Tensor) -> torch.Tensor:
|
| 208 |
+
# Shifting in BIG rather than 0 is what makes the border correct
|
| 209 |
+
# without a separate mask: an off-board neighbour looks exactly like
|
| 210 |
+
# an unreached one. Padding cells are already BIG because they are
|
| 211 |
+
# outside ``open_cell``, so the same argument covers them.
|
| 212 |
+
best = _shift(v, -1, 0, self.BIG)
|
| 213 |
+
for dr, dc in ((1, 0), (0, -1), (0, 1)):
|
| 214 |
+
best = torch.minimum(best, _shift(v, dr, dc, self.BIG))
|
| 215 |
+
out = (cost + best).clamp(max=self.BIG)
|
| 216 |
+
out = torch.where(goal, zero, out)
|
| 217 |
+
return torch.where(open_cell, out, big)
|
| 218 |
+
|
| 219 |
+
free_steps = max(0, iterations - max(grad_steps, 1))
|
| 220 |
+
if free_steps:
|
| 221 |
+
with torch.no_grad():
|
| 222 |
+
for _ in range(free_steps):
|
| 223 |
+
v = sweep(v)
|
| 224 |
+
v = v.detach()
|
| 225 |
+
for _ in range(free_steps, iterations):
|
| 226 |
+
v = sweep(v)
|
| 227 |
+
return torch.cat([v, self.reach(cells)], dim=1)
|
|
@@ -0,0 +1,101 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""A compact word-level tokenizer with byte fallback.
|
| 2 |
+
|
| 3 |
+
Jevon-S is trained from scratch, so a 150k-entry subword vocabulary would put
|
| 4 |
+
most of the parameter budget in an embedding table that the task never
|
| 5 |
+
exercises. This tokenizer keeps the vocabulary in the low thousands: words seen
|
| 6 |
+
during corpus construction become single ids, and anything unseen decomposes
|
| 7 |
+
into bytes, so no input is ever rejected.
|
| 8 |
+
"""
|
| 9 |
+
from __future__ import annotations
|
| 10 |
+
|
| 11 |
+
import json
|
| 12 |
+
import re
|
| 13 |
+
from collections import Counter
|
| 14 |
+
from pathlib import Path
|
| 15 |
+
from typing import Iterable, Sequence
|
| 16 |
+
|
| 17 |
+
PAD, UNK, BOS, EOS, SEP = "<pad>", "<unk>", "<bos>", "<eos>", "<sep>"
|
| 18 |
+
SPECIALS: tuple[str, ...] = (PAD, UNK, BOS, EOS, SEP)
|
| 19 |
+
|
| 20 |
+
# Words, signed integers, and single non-space symbols. Coordinates such as
|
| 21 |
+
# "(12,7)" split into "(", "12", ",", "7", ")" so numbers stay atomic.
|
| 22 |
+
_PATTERN = re.compile(r"[A-Za-z]+|\d+|[^\sA-Za-z\d]")
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
class JevonTokenizer:
|
| 26 |
+
def __init__(self, vocab: Sequence[str]):
|
| 27 |
+
self.itos = list(vocab)
|
| 28 |
+
self.stoi = {tok: i for i, tok in enumerate(self.itos)}
|
| 29 |
+
for special in SPECIALS:
|
| 30 |
+
if special not in self.stoi:
|
| 31 |
+
raise ValueError(f"vocabulary is missing {special!r}")
|
| 32 |
+
self.pad_id = self.stoi[PAD]
|
| 33 |
+
self.unk_id = self.stoi[UNK]
|
| 34 |
+
self.bos_id = self.stoi[BOS]
|
| 35 |
+
self.eos_id = self.stoi[EOS]
|
| 36 |
+
self.sep_id = self.stoi[SEP]
|
| 37 |
+
self._byte_base = self.stoi.get("<byte_0>")
|
| 38 |
+
|
| 39 |
+
def __len__(self) -> int:
|
| 40 |
+
return len(self.itos)
|
| 41 |
+
|
| 42 |
+
@classmethod
|
| 43 |
+
def build(cls, corpus: Iterable[str], max_vocab: int = 4096,
|
| 44 |
+
min_count: int = 2) -> "JevonTokenizer":
|
| 45 |
+
counts: Counter[str] = Counter()
|
| 46 |
+
for text in corpus:
|
| 47 |
+
counts.update(_PATTERN.findall(text.lower()))
|
| 48 |
+
vocab = list(SPECIALS)
|
| 49 |
+
vocab += [f"<byte_{i}>" for i in range(256)] # always-available fallback
|
| 50 |
+
room = max_vocab - len(vocab)
|
| 51 |
+
seen = set(vocab)
|
| 52 |
+
for token, count in counts.most_common():
|
| 53 |
+
if room <= 0:
|
| 54 |
+
break
|
| 55 |
+
if count < min_count or token in seen:
|
| 56 |
+
continue
|
| 57 |
+
vocab.append(token)
|
| 58 |
+
seen.add(token)
|
| 59 |
+
room -= 1
|
| 60 |
+
return cls(vocab)
|
| 61 |
+
|
| 62 |
+
def encode(self, text: str, *, add_bos: bool = False,
|
| 63 |
+
add_eos: bool = False) -> list[int]:
|
| 64 |
+
ids: list[int] = [self.bos_id] if add_bos else []
|
| 65 |
+
for token in _PATTERN.findall(text.lower()):
|
| 66 |
+
index = self.stoi.get(token)
|
| 67 |
+
if index is not None:
|
| 68 |
+
ids.append(index)
|
| 69 |
+
elif self._byte_base is not None:
|
| 70 |
+
ids.extend(self._byte_base + b for b in token.encode("utf-8"))
|
| 71 |
+
else:
|
| 72 |
+
ids.append(self.unk_id)
|
| 73 |
+
if add_eos:
|
| 74 |
+
ids.append(self.eos_id)
|
| 75 |
+
return ids
|
| 76 |
+
|
| 77 |
+
def decode(self, ids: Sequence[int]) -> str:
|
| 78 |
+
out, pending = [], bytearray()
|
| 79 |
+
|
| 80 |
+
def flush() -> None:
|
| 81 |
+
if pending:
|
| 82 |
+
out.append(pending.decode("utf-8", errors="replace"))
|
| 83 |
+
pending.clear()
|
| 84 |
+
|
| 85 |
+
for index in ids:
|
| 86 |
+
token = self.itos[index] if 0 <= index < len(self.itos) else UNK
|
| 87 |
+
if token.startswith("<byte_") and token.endswith(">"):
|
| 88 |
+
pending.append(int(token[6:-1]))
|
| 89 |
+
continue
|
| 90 |
+
flush()
|
| 91 |
+
if token not in SPECIALS:
|
| 92 |
+
out.append(token)
|
| 93 |
+
flush()
|
| 94 |
+
return " ".join(out)
|
| 95 |
+
|
| 96 |
+
def save(self, path: str | Path) -> None:
|
| 97 |
+
Path(path).write_text(json.dumps({"vocab": self.itos}, ensure_ascii=False))
|
| 98 |
+
|
| 99 |
+
@classmethod
|
| 100 |
+
def load(cls, path: str | Path) -> "JevonTokenizer":
|
| 101 |
+
return cls(json.loads(Path(path).read_text())["vocab"])
|
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[project]
|
| 2 |
+
name = "jevon"
|
| 3 |
+
version = "0.1.0"
|
| 4 |
+
description = "A small typed-decision model with a recurrent spatial planner"
|
| 5 |
+
readme = "README.md"
|
| 6 |
+
requires-python = ">=3.10"
|
| 7 |
+
# SPDX. The Hub's own vocabulary spells this `agpl-3.0`, which is what
|
| 8 |
+
# README.md's front matter carries; the test suite checks the two agree.
|
| 9 |
+
license = { text = "AGPL-3.0-or-later" }
|
| 10 |
+
dependencies = ["torch>=2.4"]
|
| 11 |
+
|
| 12 |
+
[project.optional-dependencies]
|
| 13 |
+
# `from_pretrained` only. A local run directory loads without it.
|
| 14 |
+
hub = ["huggingface_hub>=0.23"]
|
| 15 |
+
dev = ["pytest>=8.0", "pyyaml>=6.0", "huggingface_hub>=0.23"]
|
| 16 |
+
|
| 17 |
+
[project.urls]
|
| 18 |
+
Homepage = "https://huggingface.co/lewislululu/jevon"
|
| 19 |
+
Arcade = "https://github.com/lewislulu/jevon-arcade"
|
| 20 |
+
|
| 21 |
+
[build-system]
|
| 22 |
+
requires = ["setuptools>=68"]
|
| 23 |
+
build-backend = "setuptools.build_meta"
|
| 24 |
+
|
| 25 |
+
[tool.setuptools]
|
| 26 |
+
packages = ["jevon", "envs"]
|
| 27 |
+
|
| 28 |
+
[tool.pytest.ini_options]
|
| 29 |
+
testpaths = ["tests"]
|
| 30 |
+
markers = ["slow: exercises real optimisation and takes tens of seconds"]
|
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
torch>=2.4
|
| 2 |
+
|
| 3 |
+
# Only `jevon.hub.from_pretrained` needs this; loading a local run directory
|
| 4 |
+
# with `JevonRunner` does not, and the import is deferred so the absence is a
|
| 5 |
+
# clear message rather than an ImportError at startup.
|
| 6 |
+
huggingface_hub>=0.23
|
| 7 |
+
|
| 8 |
+
pytest>=8.0
|
| 9 |
+
pyyaml>=6.0
|
|
@@ -0,0 +1,49 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"argv": [
|
| 3 |
+
"scripts/train.py",
|
| 4 |
+
"--out",
|
| 5 |
+
"runs/jevon-final",
|
| 6 |
+
"--steps",
|
| 7 |
+
"9000",
|
| 8 |
+
"--eval-every",
|
| 9 |
+
"500",
|
| 10 |
+
"--planner-lr-mult",
|
| 11 |
+
"10",
|
| 12 |
+
"--field-weight",
|
| 13 |
+
"1.0",
|
| 14 |
+
"--seed",
|
| 15 |
+
"11",
|
| 16 |
+
"--min-plus-field",
|
| 17 |
+
"--expose-field"
|
| 18 |
+
],
|
| 19 |
+
"args": {
|
| 20 |
+
"out": "runs/jevon-final",
|
| 21 |
+
"steps": 9000,
|
| 22 |
+
"states_per_batch": 12,
|
| 23 |
+
"lr": 0.0003,
|
| 24 |
+
"warmup": 300,
|
| 25 |
+
"weight_decay": 0.01,
|
| 26 |
+
"grad_clip": 1.0,
|
| 27 |
+
"seed": 11,
|
| 28 |
+
"device": "mps",
|
| 29 |
+
"d_model": 384,
|
| 30 |
+
"planner_dim": 96,
|
| 31 |
+
"planner_grad_steps": 8,
|
| 32 |
+
"planner_lr_mult": 10.0,
|
| 33 |
+
"field_target": "log",
|
| 34 |
+
"descent_weight": 0.0,
|
| 35 |
+
"descent_margin": 1.0,
|
| 36 |
+
"score_smoothing": 0.0,
|
| 37 |
+
"min_plus_field": true,
|
| 38 |
+
"expose_field": true,
|
| 39 |
+
"field_weight": 1.0,
|
| 40 |
+
"iteration_jitter": 0.5,
|
| 41 |
+
"snake_fraction": 0.45,
|
| 42 |
+
"eval_every": 500,
|
| 43 |
+
"eval_file": "data/frozen/val.jsonl",
|
| 44 |
+
"log_every": 25,
|
| 45 |
+
"resume": null,
|
| 46 |
+
"no_planner": false
|
| 47 |
+
},
|
| 48 |
+
"note": "recorded from the live process; this run predates train.py writing args.json itself"
|
| 49 |
+
}
|
|
@@ -0,0 +1,18 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"loss": 0.4821357289950053,
|
| 3 |
+
"ce": 0.3864518404006958,
|
| 4 |
+
"brier": 0.1908059984445572,
|
| 5 |
+
"acc": 0.8717548251152039,
|
| 6 |
+
"acc_choice": 0.9944444457689922,
|
| 7 |
+
"acc_boolean": 0.9468833446502686,
|
| 8 |
+
"acc_score": 0.3500000014901161,
|
| 9 |
+
"tvd": 0.20573404928048453,
|
| 10 |
+
"field": 0.12298520555098852,
|
| 11 |
+
"field_reach": 0.007836632352943221,
|
| 12 |
+
"field_mae": 0.14491180615647178,
|
| 13 |
+
"field_r2": -0.20879205446084947,
|
| 14 |
+
"snake_action": 0.9864864864864865,
|
| 15 |
+
"maze_action": 1.0,
|
| 16 |
+
"step": 3000,
|
| 17 |
+
"balanced": 0.9864864864864865
|
| 18 |
+
}
|
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:42940a69544849f3e846f3e2119e3b01e769ba10ee1e21e68cd69203cfa4d64f
|
| 3 |
+
size 80487109
|
|
@@ -0,0 +1,17 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"loss": 0.39986308813095095,
|
| 3 |
+
"ce": 0.3141382733980815,
|
| 4 |
+
"brier": 0.15380982756614686,
|
| 5 |
+
"acc": 0.8852876385052999,
|
| 6 |
+
"acc_choice": 0.9888888915379842,
|
| 7 |
+
"acc_boolean": 0.9673183560371399,
|
| 8 |
+
"acc_score": 0.3444444472591082,
|
| 9 |
+
"tvd": 0.1690053701400757,
|
| 10 |
+
"field": 0.08480164979894957,
|
| 11 |
+
"field_reach": 0.0087276277015917,
|
| 12 |
+
"field_mae": 0.10644824003896998,
|
| 13 |
+
"field_r2": 0.2261388042345791,
|
| 14 |
+
"snake_action": 0.972972972972973,
|
| 15 |
+
"maze_action": 1.0,
|
| 16 |
+
"step": 8500
|
| 17 |
+
}
|
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2e43a60500265548e121acb902abc38953f4efeb959ce87e06f9ab72c8ec88be
|
| 3 |
+
size 80474413
|
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"vocab_size": 497,
|
| 3 |
+
"d_model": 384,
|
| 4 |
+
"n_heads": 6,
|
| 5 |
+
"prefix_layers": 6,
|
| 6 |
+
"candidate_layers": 2,
|
| 7 |
+
"set_layers": 2,
|
| 8 |
+
"ffn_mult": 4,
|
| 9 |
+
"dropout": 0.1,
|
| 10 |
+
"max_prefix_len": 320,
|
| 11 |
+
"max_candidate_len": 48,
|
| 12 |
+
"grid_channels": 8,
|
| 13 |
+
"planner_dim": 96,
|
| 14 |
+
"planner_blocks": 2,
|
| 15 |
+
"planner_grad_steps": 8,
|
| 16 |
+
"local_patch": 5,
|
| 17 |
+
"use_planner": true,
|
| 18 |
+
"expose_field": true,
|
| 19 |
+
"min_plus_field": true
|
| 20 |
+
}
|
|
@@ -0,0 +1,359 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"val": {
|
| 3 |
+
"states": 180,
|
| 4 |
+
"questions": 1325,
|
| 5 |
+
"ece": 0.08648470789194107,
|
| 6 |
+
"groups": {
|
| 7 |
+
"maze/action": {
|
| 8 |
+
"n": 97,
|
| 9 |
+
"boards": 97,
|
| 10 |
+
"accuracy": 1.0,
|
| 11 |
+
"uniform_control": 0.49054983073903113,
|
| 12 |
+
"tvd": 0.018844484970534103,
|
| 13 |
+
"brier": 0.002279498944011423
|
| 14 |
+
},
|
| 15 |
+
"maze/boolean": {
|
| 16 |
+
"n": 530,
|
| 17 |
+
"boards": 106,
|
| 18 |
+
"accuracy": 1.0,
|
| 19 |
+
"uniform_control": 0.5,
|
| 20 |
+
"tvd": 0.0006993013012381095,
|
| 21 |
+
"brier": 1.9829470994958475e-06
|
| 22 |
+
},
|
| 23 |
+
"maze/choice": {
|
| 24 |
+
"n": 97,
|
| 25 |
+
"boards": 97,
|
| 26 |
+
"accuracy": 1.0,
|
| 27 |
+
"uniform_control": 0.49054983073903113,
|
| 28 |
+
"tvd": 0.018844484970534103,
|
| 29 |
+
"brier": 0.002279498944011423
|
| 30 |
+
},
|
| 31 |
+
"maze/clear": {
|
| 32 |
+
"n": 424,
|
| 33 |
+
"boards": 106,
|
| 34 |
+
"accuracy": 1.0,
|
| 35 |
+
"uniform_control": 0.5,
|
| 36 |
+
"tvd": 0.0004252428954954804,
|
| 37 |
+
"brier": 8.113499066392354e-07
|
| 38 |
+
},
|
| 39 |
+
"maze/distance": {
|
| 40 |
+
"n": 106,
|
| 41 |
+
"boards": 106,
|
| 42 |
+
"accuracy": 0.39622641509433965,
|
| 43 |
+
"uniform_control": 0.1428571492433548,
|
| 44 |
+
"tvd": 0.7763906526115706,
|
| 45 |
+
"brier": 0.7639503164111443
|
| 46 |
+
},
|
| 47 |
+
"maze/score": {
|
| 48 |
+
"n": 106,
|
| 49 |
+
"boards": 106,
|
| 50 |
+
"accuracy": 0.39622641509433965,
|
| 51 |
+
"uniform_control": 0.1428571492433548,
|
| 52 |
+
"tvd": 0.7763906526115706,
|
| 53 |
+
"brier": 0.7639503164111443
|
| 54 |
+
},
|
| 55 |
+
"maze/solvable": {
|
| 56 |
+
"n": 106,
|
| 57 |
+
"boards": 106,
|
| 58 |
+
"accuracy": 1.0,
|
| 59 |
+
"uniform_control": 0.5,
|
| 60 |
+
"tvd": 0.001795534924208626,
|
| 61 |
+
"brier": 6.6693358709222975e-06
|
| 62 |
+
},
|
| 63 |
+
"snake/action": {
|
| 64 |
+
"n": 74,
|
| 65 |
+
"boards": 74,
|
| 66 |
+
"accuracy": 0.9864864864864865,
|
| 67 |
+
"uniform_control": 0.4954955098596779,
|
| 68 |
+
"tvd": 0.35000925611176,
|
| 69 |
+
"brier": 0.3147798128456769
|
| 70 |
+
},
|
| 71 |
+
"snake/boolean": {
|
| 72 |
+
"n": 444,
|
| 73 |
+
"boards": 74,
|
| 74 |
+
"accuracy": 0.8828828828828829,
|
| 75 |
+
"uniform_control": 0.5,
|
| 76 |
+
"tvd": 0.23378823729450102,
|
| 77 |
+
"brier": 0.20216913486784743
|
| 78 |
+
},
|
| 79 |
+
"snake/choice": {
|
| 80 |
+
"n": 74,
|
| 81 |
+
"boards": 74,
|
| 82 |
+
"accuracy": 0.9864864864864865,
|
| 83 |
+
"uniform_control": 0.4954955098596779,
|
| 84 |
+
"tvd": 0.35000925611176,
|
| 85 |
+
"brier": 0.3147798128456769
|
| 86 |
+
},
|
| 87 |
+
"snake/escape": {
|
| 88 |
+
"n": 222,
|
| 89 |
+
"boards": 74,
|
| 90 |
+
"accuracy": 0.8108108108108109,
|
| 91 |
+
"uniform_control": 0.5,
|
| 92 |
+
"tvd": 0.2772626264507438,
|
| 93 |
+
"brier": 0.28911788216246675
|
| 94 |
+
},
|
| 95 |
+
"snake/room": {
|
| 96 |
+
"n": 74,
|
| 97 |
+
"boards": 74,
|
| 98 |
+
"accuracy": 0.28378378378378377,
|
| 99 |
+
"uniform_control": 0.20000000298023224,
|
| 100 |
+
"tvd": 0.7990304521612219,
|
| 101 |
+
"brier": 0.7980756687151419
|
| 102 |
+
},
|
| 103 |
+
"snake/safe": {
|
| 104 |
+
"n": 222,
|
| 105 |
+
"boards": 74,
|
| 106 |
+
"accuracy": 0.954954954954955,
|
| 107 |
+
"uniform_control": 0.5,
|
| 108 |
+
"tvd": 0.19031384813825827,
|
| 109 |
+
"brier": 0.1152203875732281
|
| 110 |
+
},
|
| 111 |
+
"snake/score": {
|
| 112 |
+
"n": 74,
|
| 113 |
+
"boards": 74,
|
| 114 |
+
"accuracy": 0.28378378378378377,
|
| 115 |
+
"uniform_control": 0.20000000298023224,
|
| 116 |
+
"tvd": 0.7990304521612219,
|
| 117 |
+
"brier": 0.7980756687151419
|
| 118 |
+
}
|
| 119 |
+
}
|
| 120 |
+
},
|
| 121 |
+
"test": {
|
| 122 |
+
"states": 240,
|
| 123 |
+
"questions": 1765,
|
| 124 |
+
"ece": 0.059004783630371094,
|
| 125 |
+
"groups": {
|
| 126 |
+
"maze/action": {
|
| 127 |
+
"n": 141,
|
| 128 |
+
"boards": 140,
|
| 129 |
+
"accuracy": 1.0,
|
| 130 |
+
"uniform_control": 0.48817967269437534,
|
| 131 |
+
"tvd": 0.03961129047573943,
|
| 132 |
+
"brier": 0.007873877364980677
|
| 133 |
+
},
|
| 134 |
+
"maze/boolean": {
|
| 135 |
+
"n": 740,
|
| 136 |
+
"boards": 147,
|
| 137 |
+
"accuracy": 1.0,
|
| 138 |
+
"uniform_control": 0.5,
|
| 139 |
+
"tvd": 0.0006845150103578596,
|
| 140 |
+
"brier": 2.021968730086637e-06
|
| 141 |
+
},
|
| 142 |
+
"maze/choice": {
|
| 143 |
+
"n": 141,
|
| 144 |
+
"boards": 140,
|
| 145 |
+
"accuracy": 1.0,
|
| 146 |
+
"uniform_control": 0.48817967269437534,
|
| 147 |
+
"tvd": 0.03961129047573943,
|
| 148 |
+
"brier": 0.007873877364980677
|
| 149 |
+
},
|
| 150 |
+
"maze/clear": {
|
| 151 |
+
"n": 592,
|
| 152 |
+
"boards": 147,
|
| 153 |
+
"accuracy": 1.0,
|
| 154 |
+
"uniform_control": 0.5,
|
| 155 |
+
"tvd": 0.00038996367161539763,
|
| 156 |
+
"brier": 7.440200639129634e-07
|
| 157 |
+
},
|
| 158 |
+
"maze/distance": {
|
| 159 |
+
"n": 148,
|
| 160 |
+
"boards": 147,
|
| 161 |
+
"accuracy": 0.2972972972972973,
|
| 162 |
+
"uniform_control": 0.1428571492433548,
|
| 163 |
+
"tvd": 0.7846677939634066,
|
| 164 |
+
"brier": 0.7795776144878285
|
| 165 |
+
},
|
| 166 |
+
"maze/score": {
|
| 167 |
+
"n": 148,
|
| 168 |
+
"boards": 147,
|
| 169 |
+
"accuracy": 0.2972972972972973,
|
| 170 |
+
"uniform_control": 0.1428571492433548,
|
| 171 |
+
"tvd": 0.7846677939634066,
|
| 172 |
+
"brier": 0.7795776144878285
|
| 173 |
+
},
|
| 174 |
+
"maze/solvable": {
|
| 175 |
+
"n": 148,
|
| 176 |
+
"boards": 147,
|
| 177 |
+
"accuracy": 1.0,
|
| 178 |
+
"uniform_control": 0.5,
|
| 179 |
+
"tvd": 0.0018627203653277073,
|
| 180 |
+
"brier": 7.133763394781331e-06
|
| 181 |
+
},
|
| 182 |
+
"snake/action": {
|
| 183 |
+
"n": 92,
|
| 184 |
+
"boards": 92,
|
| 185 |
+
"accuracy": 0.967391304347826,
|
| 186 |
+
"uniform_control": 0.47826088349456375,
|
| 187 |
+
"tvd": 0.3106858467774304,
|
| 188 |
+
"brier": 0.2770434002752654
|
| 189 |
+
},
|
| 190 |
+
"snake/boolean": {
|
| 191 |
+
"n": 552,
|
| 192 |
+
"boards": 92,
|
| 193 |
+
"accuracy": 0.8894927536231884,
|
| 194 |
+
"uniform_control": 0.5,
|
| 195 |
+
"tvd": 0.22231864719314204,
|
| 196 |
+
"brier": 0.19377556414448022
|
| 197 |
+
},
|
| 198 |
+
"snake/choice": {
|
| 199 |
+
"n": 92,
|
| 200 |
+
"boards": 92,
|
| 201 |
+
"accuracy": 0.967391304347826,
|
| 202 |
+
"uniform_control": 0.47826088349456375,
|
| 203 |
+
"tvd": 0.3106858467774304,
|
| 204 |
+
"brier": 0.2770434002752654
|
| 205 |
+
},
|
| 206 |
+
"snake/escape": {
|
| 207 |
+
"n": 276,
|
| 208 |
+
"boards": 92,
|
| 209 |
+
"accuracy": 0.8188405797101449,
|
| 210 |
+
"uniform_control": 0.5,
|
| 211 |
+
"tvd": 0.26280311700226605,
|
| 212 |
+
"brier": 0.2747444752047
|
| 213 |
+
},
|
| 214 |
+
"snake/room": {
|
| 215 |
+
"n": 92,
|
| 216 |
+
"boards": 92,
|
| 217 |
+
"accuracy": 0.25,
|
| 218 |
+
"uniform_control": 0.20000000298023224,
|
| 219 |
+
"tvd": 0.7993024775515432,
|
| 220 |
+
"brier": 0.7986197102329006
|
| 221 |
+
},
|
| 222 |
+
"snake/safe": {
|
| 223 |
+
"n": 276,
|
| 224 |
+
"boards": 92,
|
| 225 |
+
"accuracy": 0.9601449275362319,
|
| 226 |
+
"uniform_control": 0.5,
|
| 227 |
+
"tvd": 0.18183417738401803,
|
| 228 |
+
"brier": 0.1128066530842604
|
| 229 |
+
},
|
| 230 |
+
"snake/score": {
|
| 231 |
+
"n": 92,
|
| 232 |
+
"boards": 92,
|
| 233 |
+
"accuracy": 0.25,
|
| 234 |
+
"uniform_control": 0.20000000298023224,
|
| 235 |
+
"tvd": 0.7993024775515432,
|
| 236 |
+
"brier": 0.7986197102329006
|
| 237 |
+
}
|
| 238 |
+
}
|
| 239 |
+
},
|
| 240 |
+
"ood": {
|
| 241 |
+
"states": 160,
|
| 242 |
+
"questions": 1167,
|
| 243 |
+
"ece": 0.05399147421121597,
|
| 244 |
+
"groups": {
|
| 245 |
+
"maze/action": {
|
| 246 |
+
"n": 107,
|
| 247 |
+
"boards": 107,
|
| 248 |
+
"accuracy": 1.0,
|
| 249 |
+
"uniform_control": 0.5132398789174089,
|
| 250 |
+
"tvd": 0.26991402065364,
|
| 251 |
+
"brier": 0.1504033250384964
|
| 252 |
+
},
|
| 253 |
+
"maze/boolean": {
|
| 254 |
+
"n": 550,
|
| 255 |
+
"boards": 110,
|
| 256 |
+
"accuracy": 1.0,
|
| 257 |
+
"uniform_control": 0.5,
|
| 258 |
+
"tvd": 0.0007242930423341353,
|
| 259 |
+
"brier": 2.258754991337562e-06
|
| 260 |
+
},
|
| 261 |
+
"maze/choice": {
|
| 262 |
+
"n": 107,
|
| 263 |
+
"boards": 107,
|
| 264 |
+
"accuracy": 1.0,
|
| 265 |
+
"uniform_control": 0.5132398789174089,
|
| 266 |
+
"tvd": 0.26991402065364,
|
| 267 |
+
"brier": 0.1504033250384964
|
| 268 |
+
},
|
| 269 |
+
"maze/clear": {
|
| 270 |
+
"n": 440,
|
| 271 |
+
"boards": 110,
|
| 272 |
+
"accuracy": 1.0,
|
| 273 |
+
"uniform_control": 0.5,
|
| 274 |
+
"tvd": 0.0004054153733622198,
|
| 275 |
+
"brier": 7.734664848157991e-07
|
| 276 |
+
},
|
| 277 |
+
"maze/distance": {
|
| 278 |
+
"n": 110,
|
| 279 |
+
"boards": 110,
|
| 280 |
+
"accuracy": 0.05454545454545454,
|
| 281 |
+
"uniform_control": 0.1428571492433548,
|
| 282 |
+
"tvd": 0.8541775340383703,
|
| 283 |
+
"brier": 0.9160033637827093
|
| 284 |
+
},
|
| 285 |
+
"maze/score": {
|
| 286 |
+
"n": 110,
|
| 287 |
+
"boards": 110,
|
| 288 |
+
"accuracy": 0.05454545454545454,
|
| 289 |
+
"uniform_control": 0.1428571492433548,
|
| 290 |
+
"tvd": 0.8541775340383703,
|
| 291 |
+
"brier": 0.9160033637827093
|
| 292 |
+
},
|
| 293 |
+
"maze/solvable": {
|
| 294 |
+
"n": 110,
|
| 295 |
+
"boards": 110,
|
| 296 |
+
"accuracy": 1.0,
|
| 297 |
+
"uniform_control": 0.5,
|
| 298 |
+
"tvd": 0.0019998037182217972,
|
| 299 |
+
"brier": 8.199909017424612e-06
|
| 300 |
+
},
|
| 301 |
+
"snake/action": {
|
| 302 |
+
"n": 50,
|
| 303 |
+
"boards": 50,
|
| 304 |
+
"accuracy": 0.98,
|
| 305 |
+
"uniform_control": 0.4866666805744171,
|
| 306 |
+
"tvd": 0.35739153509028254,
|
| 307 |
+
"brier": 0.3199479156875387
|
| 308 |
+
},
|
| 309 |
+
"snake/boolean": {
|
| 310 |
+
"n": 300,
|
| 311 |
+
"boards": 50,
|
| 312 |
+
"accuracy": 0.8333333333333334,
|
| 313 |
+
"uniform_control": 0.5,
|
| 314 |
+
"tvd": 0.26355026371777057,
|
| 315 |
+
"brier": 0.262066046569186
|
| 316 |
+
},
|
| 317 |
+
"snake/choice": {
|
| 318 |
+
"n": 50,
|
| 319 |
+
"boards": 50,
|
| 320 |
+
"accuracy": 0.98,
|
| 321 |
+
"uniform_control": 0.4866666805744171,
|
| 322 |
+
"tvd": 0.35739153509028254,
|
| 323 |
+
"brier": 0.3199479156875387
|
| 324 |
+
},
|
| 325 |
+
"snake/escape": {
|
| 326 |
+
"n": 150,
|
| 327 |
+
"boards": 50,
|
| 328 |
+
"accuracy": 0.68,
|
| 329 |
+
"uniform_control": 0.5,
|
| 330 |
+
"tvd": 0.35605070618291695,
|
| 331 |
+
"brier": 0.4470668915938586
|
| 332 |
+
},
|
| 333 |
+
"snake/room": {
|
| 334 |
+
"n": 50,
|
| 335 |
+
"boards": 50,
|
| 336 |
+
"accuracy": 0.2,
|
| 337 |
+
"uniform_control": 0.20000000298023224,
|
| 338 |
+
"tvd": 0.798902131319046,
|
| 339 |
+
"brier": 0.7978189885616302
|
| 340 |
+
},
|
| 341 |
+
"snake/safe": {
|
| 342 |
+
"n": 150,
|
| 343 |
+
"boards": 50,
|
| 344 |
+
"accuracy": 0.9866666666666667,
|
| 345 |
+
"uniform_control": 0.5,
|
| 346 |
+
"tvd": 0.1710498212526242,
|
| 347 |
+
"brier": 0.07706520154451331
|
| 348 |
+
},
|
| 349 |
+
"snake/score": {
|
| 350 |
+
"n": 50,
|
| 351 |
+
"boards": 50,
|
| 352 |
+
"accuracy": 0.2,
|
| 353 |
+
"uniform_control": 0.20000000298023224,
|
| 354 |
+
"tvd": 0.798902131319046,
|
| 355 |
+
"brier": 0.7978189885616302
|
| 356 |
+
}
|
| 357 |
+
}
|
| 358 |
+
}
|
| 359 |
+
}
|
|
@@ -0,0 +1,308 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"loss": 0.6430396795272827,
|
| 4 |
+
"ce": 0.4291669289271037,
|
| 5 |
+
"brier": 0.22238269944985709,
|
| 6 |
+
"acc": 0.8309783538182577,
|
| 7 |
+
"acc_choice": 0.680606069167455,
|
| 8 |
+
"acc_boolean": 0.9468833446502686,
|
| 9 |
+
"acc_score": 0.3500000014901161,
|
| 10 |
+
"tvd": 0.2240913321574529,
|
| 11 |
+
"field": 0.0648224376142025,
|
| 12 |
+
"field_reach": 0.008027789788320661,
|
| 13 |
+
"field_mae": 0.0873508279607331,
|
| 14 |
+
"field_r2": 0.5047728275363464,
|
| 15 |
+
"snake_action": 0.8108108108108109,
|
| 16 |
+
"maze_action": 0.5773195876288659,
|
| 17 |
+
"step": 500
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"loss": 0.6268994410832723,
|
| 21 |
+
"ce": 0.43772715926170347,
|
| 22 |
+
"brier": 0.21867869893709818,
|
| 23 |
+
"acc": 0.8626200517018636,
|
| 24 |
+
"acc_choice": 0.8468686898549398,
|
| 25 |
+
"acc_boolean": 0.9468833446502686,
|
| 26 |
+
"acc_score": 0.4222222218910853,
|
| 27 |
+
"tvd": 0.23100693821907042,
|
| 28 |
+
"field": 0.24834836920102438,
|
| 29 |
+
"field_reach": 0.00785924393373231,
|
| 30 |
+
"field_mae": 0.27680033871509113,
|
| 31 |
+
"field_r2": -2.787584755271071,
|
| 32 |
+
"snake_action": 0.9459459459459459,
|
| 33 |
+
"maze_action": 0.7731958762886598,
|
| 34 |
+
"step": 1000
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"loss": 0.6161262154579162,
|
| 38 |
+
"ce": 0.42860496044158936,
|
| 39 |
+
"brier": 0.2166995237270991,
|
| 40 |
+
"acc": 0.8393356005350748,
|
| 41 |
+
"acc_choice": 0.666464650630951,
|
| 42 |
+
"acc_boolean": 0.9468833446502686,
|
| 43 |
+
"acc_score": 0.4222222218910853,
|
| 44 |
+
"tvd": 0.2255744288365046,
|
| 45 |
+
"field": 0.09811665092905363,
|
| 46 |
+
"field_reach": 0.007819027624403436,
|
| 47 |
+
"field_mae": 0.12170105225125379,
|
| 48 |
+
"field_r2": 0.046098211747254925,
|
| 49 |
+
"snake_action": 0.5135135135135135,
|
| 50 |
+
"maze_action": 0.7835051546391752,
|
| 51 |
+
"step": 1500
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"loss": 0.5040864706039428,
|
| 55 |
+
"ce": 0.39263187646865844,
|
| 56 |
+
"brier": 0.19500995179017386,
|
| 57 |
+
"acc": 0.8703148086865743,
|
| 58 |
+
"acc_choice": 0.9833333373069764,
|
| 59 |
+
"acc_boolean": 0.9468833446502686,
|
| 60 |
+
"acc_score": 0.3500000014901161,
|
| 61 |
+
"tvd": 0.21229490141073862,
|
| 62 |
+
"field": 0.10865951577822368,
|
| 63 |
+
"field_reach": 0.007855451099264126,
|
| 64 |
+
"field_mae": 0.13416509528904144,
|
| 65 |
+
"field_r2": -0.13046405368830816,
|
| 66 |
+
"snake_action": 0.9594594594594594,
|
| 67 |
+
"maze_action": 1.0,
|
| 68 |
+
"step": 2000
|
| 69 |
+
},
|
| 70 |
+
{
|
| 71 |
+
"loss": 0.48423184355099996,
|
| 72 |
+
"ce": 0.3865963856379191,
|
| 73 |
+
"brier": 0.19062800109386444,
|
| 74 |
+
"acc": 0.853052802880605,
|
| 75 |
+
"acc_choice": 0.8498989899953207,
|
| 76 |
+
"acc_boolean": 0.9468833446502686,
|
| 77 |
+
"acc_score": 0.3500000014901161,
|
| 78 |
+
"tvd": 0.20264210999011995,
|
| 79 |
+
"field": 0.09101995428403219,
|
| 80 |
+
"field_reach": 0.00942669656748573,
|
| 81 |
+
"field_mae": 0.11243784826426176,
|
| 82 |
+
"field_r2": 0.17141795311611863,
|
| 83 |
+
"snake_action": 0.6486486486486487,
|
| 84 |
+
"maze_action": 1.0,
|
| 85 |
+
"step": 2500
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"loss": 0.4821357289950053,
|
| 89 |
+
"ce": 0.3864518404006958,
|
| 90 |
+
"brier": 0.1908059984445572,
|
| 91 |
+
"acc": 0.8717548251152039,
|
| 92 |
+
"acc_choice": 0.9944444457689922,
|
| 93 |
+
"acc_boolean": 0.9468833446502686,
|
| 94 |
+
"acc_score": 0.3500000014901161,
|
| 95 |
+
"tvd": 0.20573404928048453,
|
| 96 |
+
"field": 0.12298520555098852,
|
| 97 |
+
"field_reach": 0.007836632352943221,
|
| 98 |
+
"field_mae": 0.14491180615647178,
|
| 99 |
+
"field_r2": -0.20879205446084947,
|
| 100 |
+
"snake_action": 0.9864864864864865,
|
| 101 |
+
"maze_action": 1.0,
|
| 102 |
+
"step": 3000
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"loss": 0.4724984308083852,
|
| 106 |
+
"ce": 0.3778689761956533,
|
| 107 |
+
"brier": 0.18896289964516957,
|
| 108 |
+
"acc": 0.8717548251152039,
|
| 109 |
+
"acc_choice": 0.9944444457689922,
|
| 110 |
+
"acc_boolean": 0.9468833446502686,
|
| 111 |
+
"acc_score": 0.3500000014901161,
|
| 112 |
+
"tvd": 0.19179949859778087,
|
| 113 |
+
"field": 0.10719771832227706,
|
| 114 |
+
"field_reach": 0.007806899095885456,
|
| 115 |
+
"field_mae": 0.12956693910768763,
|
| 116 |
+
"field_r2": 0.010277769420561245,
|
| 117 |
+
"snake_action": 0.9864864864864865,
|
| 118 |
+
"maze_action": 1.0,
|
| 119 |
+
"step": 3500
|
| 120 |
+
},
|
| 121 |
+
{
|
| 122 |
+
"loss": 0.47965741753578184,
|
| 123 |
+
"ce": 0.378563779592514,
|
| 124 |
+
"brier": 0.19039177348216374,
|
| 125 |
+
"acc": 0.8702458739280701,
|
| 126 |
+
"acc_choice": 0.9828282872835795,
|
| 127 |
+
"acc_boolean": 0.9468833446502686,
|
| 128 |
+
"acc_score": 0.3500000014901161,
|
| 129 |
+
"tvd": 0.18964516123135886,
|
| 130 |
+
"field": 0.08818436115980148,
|
| 131 |
+
"field_reach": 0.00800159627882143,
|
| 132 |
+
"field_mae": 0.1098015543827379,
|
| 133 |
+
"field_r2": 0.2466056585560258,
|
| 134 |
+
"snake_action": 0.9864864864864865,
|
| 135 |
+
"maze_action": 0.979381443298969,
|
| 136 |
+
"step": 4000
|
| 137 |
+
},
|
| 138 |
+
{
|
| 139 |
+
"loss": 0.47649228970209756,
|
| 140 |
+
"ce": 0.3793632904688517,
|
| 141 |
+
"brier": 0.19014093180497488,
|
| 142 |
+
"acc": 0.8710301876068115,
|
| 143 |
+
"acc_choice": 0.9888888915379842,
|
| 144 |
+
"acc_boolean": 0.9468833446502686,
|
| 145 |
+
"acc_score": 0.3500000014901161,
|
| 146 |
+
"tvd": 0.19629547695318858,
|
| 147 |
+
"field": 0.09085945089658101,
|
| 148 |
+
"field_reach": 0.007979249480801325,
|
| 149 |
+
"field_mae": 0.11260708531644582,
|
| 150 |
+
"field_r2": 0.20965784971410284,
|
| 151 |
+
"snake_action": 0.9864864864864865,
|
| 152 |
+
"maze_action": 0.9896907216494846,
|
| 153 |
+
"step": 4500
|
| 154 |
+
},
|
| 155 |
+
{
|
| 156 |
+
"loss": 0.47328211069107057,
|
| 157 |
+
"ce": 0.37875324686368306,
|
| 158 |
+
"brier": 0.18869826843341192,
|
| 159 |
+
"acc": 0.8717548251152039,
|
| 160 |
+
"acc_choice": 0.9944444457689922,
|
| 161 |
+
"acc_boolean": 0.9468833446502686,
|
| 162 |
+
"acc_score": 0.3500000014901161,
|
| 163 |
+
"tvd": 0.1954862008492152,
|
| 164 |
+
"field": 0.09279383917649588,
|
| 165 |
+
"field_reach": 0.008081573558350404,
|
| 166 |
+
"field_mae": 0.11689527545097957,
|
| 167 |
+
"field_r2": 0.11475372426767272,
|
| 168 |
+
"snake_action": 0.9864864864864865,
|
| 169 |
+
"maze_action": 1.0,
|
| 170 |
+
"step": 5000
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"loss": 0.4666572948296865,
|
| 174 |
+
"ce": 0.3724243978659312,
|
| 175 |
+
"brier": 0.18684656272331873,
|
| 176 |
+
"acc": 0.8724794665972392,
|
| 177 |
+
"acc_choice": 0.9944444457689922,
|
| 178 |
+
"acc_boolean": 0.9468833446502686,
|
| 179 |
+
"acc_score": 0.35555555671453476,
|
| 180 |
+
"tvd": 0.1899937520424525,
|
| 181 |
+
"field": 0.11715173323949178,
|
| 182 |
+
"field_reach": 0.008021636578875283,
|
| 183 |
+
"field_mae": 0.14379776442470568,
|
| 184 |
+
"field_r2": -0.2881043439838695,
|
| 185 |
+
"snake_action": 0.9864864864864865,
|
| 186 |
+
"maze_action": 1.0,
|
| 187 |
+
"step": 5500
|
| 188 |
+
},
|
| 189 |
+
{
|
| 190 |
+
"loss": 0.45790409445762636,
|
| 191 |
+
"ce": 0.3650842885176341,
|
| 192 |
+
"brier": 0.1832037756840388,
|
| 193 |
+
"acc": 0.8672664841016133,
|
| 194 |
+
"acc_choice": 0.959494952360789,
|
| 195 |
+
"acc_boolean": 0.9468833446502686,
|
| 196 |
+
"acc_score": 0.3500000014901161,
|
| 197 |
+
"tvd": 0.19846215347448984,
|
| 198 |
+
"field": 0.10857781569163004,
|
| 199 |
+
"field_reach": 0.008120695035904646,
|
| 200 |
+
"field_mae": 0.13411058857249636,
|
| 201 |
+
"field_r2": -0.1573779587982147,
|
| 202 |
+
"snake_action": 0.9054054054054054,
|
| 203 |
+
"maze_action": 1.0,
|
| 204 |
+
"step": 6000
|
| 205 |
+
},
|
| 206 |
+
{
|
| 207 |
+
"loss": 0.45433027148246763,
|
| 208 |
+
"ce": 0.3633299271265666,
|
| 209 |
+
"brier": 0.18349071244398754,
|
| 210 |
+
"acc": 0.8695062915484111,
|
| 211 |
+
"acc_choice": 0.9766666690508524,
|
| 212 |
+
"acc_boolean": 0.9468833446502686,
|
| 213 |
+
"acc_score": 0.3500000014901161,
|
| 214 |
+
"tvd": 0.20805268486340842,
|
| 215 |
+
"field": 0.08403789947430293,
|
| 216 |
+
"field_reach": 0.007841536616130422,
|
| 217 |
+
"field_mae": 0.10577063097931902,
|
| 218 |
+
"field_r2": 0.26262213095402853,
|
| 219 |
+
"snake_action": 0.9459459459459459,
|
| 220 |
+
"maze_action": 1.0,
|
| 221 |
+
"step": 6500
|
| 222 |
+
},
|
| 223 |
+
{
|
| 224 |
+
"loss": 0.4257788856824239,
|
| 225 |
+
"ce": 0.338992706934611,
|
| 226 |
+
"brier": 0.16655596445004145,
|
| 227 |
+
"acc": 0.8860452135403951,
|
| 228 |
+
"acc_choice": 0.9888888915379842,
|
| 229 |
+
"acc_boolean": 0.9673183560371399,
|
| 230 |
+
"acc_score": 0.3500000014901161,
|
| 231 |
+
"tvd": 0.1928422600030899,
|
| 232 |
+
"field": 0.10394345372915267,
|
| 233 |
+
"field_reach": 0.00819415650330484,
|
| 234 |
+
"field_mae": 0.12675953488266162,
|
| 235 |
+
"field_r2": 0.03251710439413491,
|
| 236 |
+
"snake_action": 0.972972972972973,
|
| 237 |
+
"maze_action": 1.0,
|
| 238 |
+
"step": 7000
|
| 239 |
+
},
|
| 240 |
+
{
|
| 241 |
+
"loss": 0.40878687500953675,
|
| 242 |
+
"ce": 0.32246087193489076,
|
| 243 |
+
"brier": 0.1595507149895032,
|
| 244 |
+
"acc": 0.8830990632375081,
|
| 245 |
+
"acc_choice": 0.9367676854133606,
|
| 246 |
+
"acc_boolean": 0.9673183560371399,
|
| 247 |
+
"acc_score": 0.3777777726451556,
|
| 248 |
+
"tvd": 0.16677065094312032,
|
| 249 |
+
"field": 0.13400054325660068,
|
| 250 |
+
"field_reach": 0.008717572259289834,
|
| 251 |
+
"field_mae": 0.15709024260702828,
|
| 252 |
+
"field_r2": -0.41707112606313124,
|
| 253 |
+
"snake_action": 0.8513513513513513,
|
| 254 |
+
"maze_action": 1.0,
|
| 255 |
+
"step": 7500
|
| 256 |
+
},
|
| 257 |
+
{
|
| 258 |
+
"loss": 0.4029513160387675,
|
| 259 |
+
"ce": 0.31686999201774596,
|
| 260 |
+
"brier": 0.15476668129364649,
|
| 261 |
+
"acc": 0.8860452135403951,
|
| 262 |
+
"acc_choice": 0.9888888915379842,
|
| 263 |
+
"acc_boolean": 0.9673183560371399,
|
| 264 |
+
"acc_score": 0.3500000014901161,
|
| 265 |
+
"tvd": 0.17156118750572205,
|
| 266 |
+
"field": 0.10043507615725199,
|
| 267 |
+
"field_reach": 0.00858972800488118,
|
| 268 |
+
"field_mae": 0.12331134686881152,
|
| 269 |
+
"field_r2": 0.06755961142045463,
|
| 270 |
+
"snake_action": 0.972972972972973,
|
| 271 |
+
"maze_action": 1.0,
|
| 272 |
+
"step": 8000
|
| 273 |
+
},
|
| 274 |
+
{
|
| 275 |
+
"loss": 0.39986308813095095,
|
| 276 |
+
"ce": 0.3141382733980815,
|
| 277 |
+
"brier": 0.15380982756614686,
|
| 278 |
+
"acc": 0.8852876385052999,
|
| 279 |
+
"acc_choice": 0.9888888915379842,
|
| 280 |
+
"acc_boolean": 0.9673183560371399,
|
| 281 |
+
"acc_score": 0.3444444472591082,
|
| 282 |
+
"tvd": 0.1690053701400757,
|
| 283 |
+
"field": 0.08480164979894957,
|
| 284 |
+
"field_reach": 0.0087276277015917,
|
| 285 |
+
"field_mae": 0.10644824003896998,
|
| 286 |
+
"field_r2": 0.2261388042345791,
|
| 287 |
+
"snake_action": 0.972972972972973,
|
| 288 |
+
"maze_action": 1.0,
|
| 289 |
+
"step": 8500
|
| 290 |
+
},
|
| 291 |
+
{
|
| 292 |
+
"loss": 0.4020693858464559,
|
| 293 |
+
"ce": 0.31669941544532776,
|
| 294 |
+
"brier": 0.1551360418399175,
|
| 295 |
+
"acc": 0.8958186864852905,
|
| 296 |
+
"acc_choice": 0.9888888915379842,
|
| 297 |
+
"acc_boolean": 0.9673183560371399,
|
| 298 |
+
"acc_score": 0.4222222218910853,
|
| 299 |
+
"tvd": 0.155249418814977,
|
| 300 |
+
"field": 0.09742864519357682,
|
| 301 |
+
"field_reach": 0.008850584997950743,
|
| 302 |
+
"field_mae": 0.11998297499025644,
|
| 303 |
+
"field_r2": 0.10834436235203748,
|
| 304 |
+
"snake_action": 0.972972972972973,
|
| 305 |
+
"maze_action": 1.0,
|
| 306 |
+
"step": 9000
|
| 307 |
+
}
|
| 308 |
+
]
|
|
@@ -0,0 +1,543 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "model-field",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"solved": true,
|
| 6 |
+
"steps": 32,
|
| 7 |
+
"collisions": 0,
|
| 8 |
+
"distinct_cells": 33,
|
| 9 |
+
"longest_stuck": 0,
|
| 10 |
+
"deadlocked": false,
|
| 11 |
+
"game": "maze",
|
| 12 |
+
"size": 11,
|
| 13 |
+
"topology": "corridor",
|
| 14 |
+
"seed": 901111,
|
| 15 |
+
"shortest": 32,
|
| 16 |
+
"efficiency": 1.0
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"solved": true,
|
| 20 |
+
"steps": 42,
|
| 21 |
+
"collisions": 0,
|
| 22 |
+
"distinct_cells": 43,
|
| 23 |
+
"longest_stuck": 0,
|
| 24 |
+
"deadlocked": false,
|
| 25 |
+
"game": "maze",
|
| 26 |
+
"size": 11,
|
| 27 |
+
"topology": "corridor",
|
| 28 |
+
"seed": 901148,
|
| 29 |
+
"shortest": 42,
|
| 30 |
+
"efficiency": 1.0
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"solved": true,
|
| 34 |
+
"steps": 40,
|
| 35 |
+
"collisions": 0,
|
| 36 |
+
"distinct_cells": 41,
|
| 37 |
+
"longest_stuck": 0,
|
| 38 |
+
"deadlocked": false,
|
| 39 |
+
"game": "maze",
|
| 40 |
+
"size": 11,
|
| 41 |
+
"topology": "corridor",
|
| 42 |
+
"seed": 901185,
|
| 43 |
+
"shortest": 40,
|
| 44 |
+
"efficiency": 1.0
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"solved": true,
|
| 48 |
+
"steps": 44,
|
| 49 |
+
"collisions": 0,
|
| 50 |
+
"distinct_cells": 45,
|
| 51 |
+
"longest_stuck": 0,
|
| 52 |
+
"deadlocked": false,
|
| 53 |
+
"game": "maze",
|
| 54 |
+
"size": 11,
|
| 55 |
+
"topology": "tree",
|
| 56 |
+
"seed": 901111,
|
| 57 |
+
"shortest": 44,
|
| 58 |
+
"efficiency": 1.0
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"solved": true,
|
| 62 |
+
"steps": 34,
|
| 63 |
+
"collisions": 0,
|
| 64 |
+
"distinct_cells": 35,
|
| 65 |
+
"longest_stuck": 0,
|
| 66 |
+
"deadlocked": false,
|
| 67 |
+
"game": "maze",
|
| 68 |
+
"size": 11,
|
| 69 |
+
"topology": "tree",
|
| 70 |
+
"seed": 901148,
|
| 71 |
+
"shortest": 34,
|
| 72 |
+
"efficiency": 1.0
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"solved": true,
|
| 76 |
+
"steps": 36,
|
| 77 |
+
"collisions": 0,
|
| 78 |
+
"distinct_cells": 37,
|
| 79 |
+
"longest_stuck": 0,
|
| 80 |
+
"deadlocked": false,
|
| 81 |
+
"game": "maze",
|
| 82 |
+
"size": 11,
|
| 83 |
+
"topology": "tree",
|
| 84 |
+
"seed": 901185,
|
| 85 |
+
"shortest": 36,
|
| 86 |
+
"efficiency": 1.0
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"solved": true,
|
| 90 |
+
"steps": 15,
|
| 91 |
+
"collisions": 0,
|
| 92 |
+
"distinct_cells": 16,
|
| 93 |
+
"longest_stuck": 0,
|
| 94 |
+
"deadlocked": false,
|
| 95 |
+
"game": "maze",
|
| 96 |
+
"size": 11,
|
| 97 |
+
"topology": "loops",
|
| 98 |
+
"seed": 901111,
|
| 99 |
+
"shortest": 15,
|
| 100 |
+
"efficiency": 1.0
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"solved": true,
|
| 104 |
+
"steps": 22,
|
| 105 |
+
"collisions": 0,
|
| 106 |
+
"distinct_cells": 23,
|
| 107 |
+
"longest_stuck": 0,
|
| 108 |
+
"deadlocked": false,
|
| 109 |
+
"game": "maze",
|
| 110 |
+
"size": 11,
|
| 111 |
+
"topology": "loops",
|
| 112 |
+
"seed": 901148,
|
| 113 |
+
"shortest": 22,
|
| 114 |
+
"efficiency": 1.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"solved": true,
|
| 118 |
+
"steps": 18,
|
| 119 |
+
"collisions": 0,
|
| 120 |
+
"distinct_cells": 19,
|
| 121 |
+
"longest_stuck": 0,
|
| 122 |
+
"deadlocked": false,
|
| 123 |
+
"game": "maze",
|
| 124 |
+
"size": 11,
|
| 125 |
+
"topology": "loops",
|
| 126 |
+
"seed": 901185,
|
| 127 |
+
"shortest": 18,
|
| 128 |
+
"efficiency": 1.0
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"solved": true,
|
| 132 |
+
"steps": 15,
|
| 133 |
+
"collisions": 0,
|
| 134 |
+
"distinct_cells": 16,
|
| 135 |
+
"longest_stuck": 0,
|
| 136 |
+
"deadlocked": false,
|
| 137 |
+
"game": "maze",
|
| 138 |
+
"size": 11,
|
| 139 |
+
"topology": "random_obstacle",
|
| 140 |
+
"seed": 901111,
|
| 141 |
+
"shortest": 15,
|
| 142 |
+
"efficiency": 1.0
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"solved": true,
|
| 146 |
+
"steps": 16,
|
| 147 |
+
"collisions": 0,
|
| 148 |
+
"distinct_cells": 17,
|
| 149 |
+
"longest_stuck": 0,
|
| 150 |
+
"deadlocked": false,
|
| 151 |
+
"game": "maze",
|
| 152 |
+
"size": 11,
|
| 153 |
+
"topology": "random_obstacle",
|
| 154 |
+
"seed": 901148,
|
| 155 |
+
"shortest": 16,
|
| 156 |
+
"efficiency": 1.0
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"solved": true,
|
| 160 |
+
"steps": 13,
|
| 161 |
+
"collisions": 0,
|
| 162 |
+
"distinct_cells": 14,
|
| 163 |
+
"longest_stuck": 0,
|
| 164 |
+
"deadlocked": false,
|
| 165 |
+
"game": "maze",
|
| 166 |
+
"size": 11,
|
| 167 |
+
"topology": "random_obstacle",
|
| 168 |
+
"seed": 901185,
|
| 169 |
+
"shortest": 13,
|
| 170 |
+
"efficiency": 1.0
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"solved": true,
|
| 174 |
+
"steps": 142,
|
| 175 |
+
"collisions": 0,
|
| 176 |
+
"distinct_cells": 143,
|
| 177 |
+
"longest_stuck": 0,
|
| 178 |
+
"deadlocked": false,
|
| 179 |
+
"game": "maze",
|
| 180 |
+
"size": 21,
|
| 181 |
+
"topology": "corridor",
|
| 182 |
+
"seed": 902121,
|
| 183 |
+
"shortest": 142,
|
| 184 |
+
"efficiency": 1.0
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"solved": true,
|
| 188 |
+
"steps": 116,
|
| 189 |
+
"collisions": 0,
|
| 190 |
+
"distinct_cells": 117,
|
| 191 |
+
"longest_stuck": 0,
|
| 192 |
+
"deadlocked": false,
|
| 193 |
+
"game": "maze",
|
| 194 |
+
"size": 21,
|
| 195 |
+
"topology": "corridor",
|
| 196 |
+
"seed": 902158,
|
| 197 |
+
"shortest": 116,
|
| 198 |
+
"efficiency": 1.0
|
| 199 |
+
},
|
| 200 |
+
{
|
| 201 |
+
"solved": true,
|
| 202 |
+
"steps": 156,
|
| 203 |
+
"collisions": 0,
|
| 204 |
+
"distinct_cells": 157,
|
| 205 |
+
"longest_stuck": 0,
|
| 206 |
+
"deadlocked": false,
|
| 207 |
+
"game": "maze",
|
| 208 |
+
"size": 21,
|
| 209 |
+
"topology": "corridor",
|
| 210 |
+
"seed": 902195,
|
| 211 |
+
"shortest": 156,
|
| 212 |
+
"efficiency": 1.0
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"solved": true,
|
| 216 |
+
"steps": 118,
|
| 217 |
+
"collisions": 0,
|
| 218 |
+
"distinct_cells": 119,
|
| 219 |
+
"longest_stuck": 0,
|
| 220 |
+
"deadlocked": false,
|
| 221 |
+
"game": "maze",
|
| 222 |
+
"size": 21,
|
| 223 |
+
"topology": "tree",
|
| 224 |
+
"seed": 902121,
|
| 225 |
+
"shortest": 118,
|
| 226 |
+
"efficiency": 1.0
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"solved": true,
|
| 230 |
+
"steps": 172,
|
| 231 |
+
"collisions": 0,
|
| 232 |
+
"distinct_cells": 173,
|
| 233 |
+
"longest_stuck": 0,
|
| 234 |
+
"deadlocked": false,
|
| 235 |
+
"game": "maze",
|
| 236 |
+
"size": 21,
|
| 237 |
+
"topology": "tree",
|
| 238 |
+
"seed": 902158,
|
| 239 |
+
"shortest": 172,
|
| 240 |
+
"efficiency": 1.0
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"solved": true,
|
| 244 |
+
"steps": 140,
|
| 245 |
+
"collisions": 0,
|
| 246 |
+
"distinct_cells": 141,
|
| 247 |
+
"longest_stuck": 0,
|
| 248 |
+
"deadlocked": false,
|
| 249 |
+
"game": "maze",
|
| 250 |
+
"size": 21,
|
| 251 |
+
"topology": "tree",
|
| 252 |
+
"seed": 902195,
|
| 253 |
+
"shortest": 140,
|
| 254 |
+
"efficiency": 1.0
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"solved": true,
|
| 258 |
+
"steps": 46,
|
| 259 |
+
"collisions": 0,
|
| 260 |
+
"distinct_cells": 47,
|
| 261 |
+
"longest_stuck": 0,
|
| 262 |
+
"deadlocked": false,
|
| 263 |
+
"game": "maze",
|
| 264 |
+
"size": 21,
|
| 265 |
+
"topology": "loops",
|
| 266 |
+
"seed": 902121,
|
| 267 |
+
"shortest": 46,
|
| 268 |
+
"efficiency": 1.0
|
| 269 |
+
},
|
| 270 |
+
{
|
| 271 |
+
"solved": true,
|
| 272 |
+
"steps": 40,
|
| 273 |
+
"collisions": 0,
|
| 274 |
+
"distinct_cells": 41,
|
| 275 |
+
"longest_stuck": 0,
|
| 276 |
+
"deadlocked": false,
|
| 277 |
+
"game": "maze",
|
| 278 |
+
"size": 21,
|
| 279 |
+
"topology": "loops",
|
| 280 |
+
"seed": 902158,
|
| 281 |
+
"shortest": 40,
|
| 282 |
+
"efficiency": 1.0
|
| 283 |
+
},
|
| 284 |
+
{
|
| 285 |
+
"solved": true,
|
| 286 |
+
"steps": 42,
|
| 287 |
+
"collisions": 0,
|
| 288 |
+
"distinct_cells": 43,
|
| 289 |
+
"longest_stuck": 0,
|
| 290 |
+
"deadlocked": false,
|
| 291 |
+
"game": "maze",
|
| 292 |
+
"size": 21,
|
| 293 |
+
"topology": "loops",
|
| 294 |
+
"seed": 902195,
|
| 295 |
+
"shortest": 42,
|
| 296 |
+
"efficiency": 1.0
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"solved": true,
|
| 300 |
+
"steps": 41,
|
| 301 |
+
"collisions": 0,
|
| 302 |
+
"distinct_cells": 42,
|
| 303 |
+
"longest_stuck": 0,
|
| 304 |
+
"deadlocked": false,
|
| 305 |
+
"game": "maze",
|
| 306 |
+
"size": 21,
|
| 307 |
+
"topology": "random_obstacle",
|
| 308 |
+
"seed": 902121,
|
| 309 |
+
"shortest": 41,
|
| 310 |
+
"efficiency": 1.0
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"solved": true,
|
| 314 |
+
"steps": 36,
|
| 315 |
+
"collisions": 0,
|
| 316 |
+
"distinct_cells": 37,
|
| 317 |
+
"longest_stuck": 0,
|
| 318 |
+
"deadlocked": false,
|
| 319 |
+
"game": "maze",
|
| 320 |
+
"size": 21,
|
| 321 |
+
"topology": "random_obstacle",
|
| 322 |
+
"seed": 902158,
|
| 323 |
+
"shortest": 36,
|
| 324 |
+
"efficiency": 1.0
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"solved": true,
|
| 328 |
+
"steps": 36,
|
| 329 |
+
"collisions": 0,
|
| 330 |
+
"distinct_cells": 37,
|
| 331 |
+
"longest_stuck": 0,
|
| 332 |
+
"deadlocked": false,
|
| 333 |
+
"game": "maze",
|
| 334 |
+
"size": 21,
|
| 335 |
+
"topology": "random_obstacle",
|
| 336 |
+
"seed": 902195,
|
| 337 |
+
"shortest": 36,
|
| 338 |
+
"efficiency": 1.0
|
| 339 |
+
},
|
| 340 |
+
{
|
| 341 |
+
"solved": true,
|
| 342 |
+
"steps": 264,
|
| 343 |
+
"collisions": 0,
|
| 344 |
+
"distinct_cells": 265,
|
| 345 |
+
"longest_stuck": 0,
|
| 346 |
+
"deadlocked": false,
|
| 347 |
+
"game": "maze",
|
| 348 |
+
"size": 31,
|
| 349 |
+
"topology": "corridor",
|
| 350 |
+
"seed": 903131,
|
| 351 |
+
"shortest": 264,
|
| 352 |
+
"efficiency": 1.0
|
| 353 |
+
},
|
| 354 |
+
{
|
| 355 |
+
"solved": true,
|
| 356 |
+
"steps": 194,
|
| 357 |
+
"collisions": 0,
|
| 358 |
+
"distinct_cells": 195,
|
| 359 |
+
"longest_stuck": 0,
|
| 360 |
+
"deadlocked": false,
|
| 361 |
+
"game": "maze",
|
| 362 |
+
"size": 31,
|
| 363 |
+
"topology": "corridor",
|
| 364 |
+
"seed": 903168,
|
| 365 |
+
"shortest": 194,
|
| 366 |
+
"efficiency": 1.0
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"solved": true,
|
| 370 |
+
"steps": 286,
|
| 371 |
+
"collisions": 0,
|
| 372 |
+
"distinct_cells": 287,
|
| 373 |
+
"longest_stuck": 0,
|
| 374 |
+
"deadlocked": false,
|
| 375 |
+
"game": "maze",
|
| 376 |
+
"size": 31,
|
| 377 |
+
"topology": "corridor",
|
| 378 |
+
"seed": 903205,
|
| 379 |
+
"shortest": 286,
|
| 380 |
+
"efficiency": 1.0
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"solved": true,
|
| 384 |
+
"steps": 268,
|
| 385 |
+
"collisions": 0,
|
| 386 |
+
"distinct_cells": 269,
|
| 387 |
+
"longest_stuck": 0,
|
| 388 |
+
"deadlocked": false,
|
| 389 |
+
"game": "maze",
|
| 390 |
+
"size": 31,
|
| 391 |
+
"topology": "tree",
|
| 392 |
+
"seed": 903131,
|
| 393 |
+
"shortest": 268,
|
| 394 |
+
"efficiency": 1.0
|
| 395 |
+
},
|
| 396 |
+
{
|
| 397 |
+
"solved": true,
|
| 398 |
+
"steps": 282,
|
| 399 |
+
"collisions": 0,
|
| 400 |
+
"distinct_cells": 283,
|
| 401 |
+
"longest_stuck": 0,
|
| 402 |
+
"deadlocked": false,
|
| 403 |
+
"game": "maze",
|
| 404 |
+
"size": 31,
|
| 405 |
+
"topology": "tree",
|
| 406 |
+
"seed": 903168,
|
| 407 |
+
"shortest": 282,
|
| 408 |
+
"efficiency": 1.0
|
| 409 |
+
},
|
| 410 |
+
{
|
| 411 |
+
"solved": true,
|
| 412 |
+
"steps": 252,
|
| 413 |
+
"collisions": 0,
|
| 414 |
+
"distinct_cells": 253,
|
| 415 |
+
"longest_stuck": 0,
|
| 416 |
+
"deadlocked": false,
|
| 417 |
+
"game": "maze",
|
| 418 |
+
"size": 31,
|
| 419 |
+
"topology": "tree",
|
| 420 |
+
"seed": 903205,
|
| 421 |
+
"shortest": 252,
|
| 422 |
+
"efficiency": 1.0
|
| 423 |
+
},
|
| 424 |
+
{
|
| 425 |
+
"solved": true,
|
| 426 |
+
"steps": 66,
|
| 427 |
+
"collisions": 0,
|
| 428 |
+
"distinct_cells": 67,
|
| 429 |
+
"longest_stuck": 0,
|
| 430 |
+
"deadlocked": false,
|
| 431 |
+
"game": "maze",
|
| 432 |
+
"size": 31,
|
| 433 |
+
"topology": "loops",
|
| 434 |
+
"seed": 903131,
|
| 435 |
+
"shortest": 66,
|
| 436 |
+
"efficiency": 1.0
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"solved": true,
|
| 440 |
+
"steps": 72,
|
| 441 |
+
"collisions": 0,
|
| 442 |
+
"distinct_cells": 73,
|
| 443 |
+
"longest_stuck": 0,
|
| 444 |
+
"deadlocked": false,
|
| 445 |
+
"game": "maze",
|
| 446 |
+
"size": 31,
|
| 447 |
+
"topology": "loops",
|
| 448 |
+
"seed": 903168,
|
| 449 |
+
"shortest": 72,
|
| 450 |
+
"efficiency": 1.0
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"solved": true,
|
| 454 |
+
"steps": 88,
|
| 455 |
+
"collisions": 0,
|
| 456 |
+
"distinct_cells": 89,
|
| 457 |
+
"longest_stuck": 0,
|
| 458 |
+
"deadlocked": false,
|
| 459 |
+
"game": "maze",
|
| 460 |
+
"size": 31,
|
| 461 |
+
"topology": "loops",
|
| 462 |
+
"seed": 903205,
|
| 463 |
+
"shortest": 88,
|
| 464 |
+
"efficiency": 1.0
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"solved": true,
|
| 468 |
+
"steps": 57,
|
| 469 |
+
"collisions": 0,
|
| 470 |
+
"distinct_cells": 58,
|
| 471 |
+
"longest_stuck": 0,
|
| 472 |
+
"deadlocked": false,
|
| 473 |
+
"game": "maze",
|
| 474 |
+
"size": 31,
|
| 475 |
+
"topology": "random_obstacle",
|
| 476 |
+
"seed": 903131,
|
| 477 |
+
"shortest": 57,
|
| 478 |
+
"efficiency": 1.0
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"solved": true,
|
| 482 |
+
"steps": 56,
|
| 483 |
+
"collisions": 0,
|
| 484 |
+
"distinct_cells": 57,
|
| 485 |
+
"longest_stuck": 0,
|
| 486 |
+
"deadlocked": false,
|
| 487 |
+
"game": "maze",
|
| 488 |
+
"size": 31,
|
| 489 |
+
"topology": "random_obstacle",
|
| 490 |
+
"seed": 903168,
|
| 491 |
+
"shortest": 56,
|
| 492 |
+
"efficiency": 1.0
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"solved": true,
|
| 496 |
+
"steps": 58,
|
| 497 |
+
"collisions": 0,
|
| 498 |
+
"distinct_cells": 59,
|
| 499 |
+
"longest_stuck": 0,
|
| 500 |
+
"deadlocked": false,
|
| 501 |
+
"game": "maze",
|
| 502 |
+
"size": 31,
|
| 503 |
+
"topology": "random_obstacle",
|
| 504 |
+
"seed": 903205,
|
| 505 |
+
"shortest": 58,
|
| 506 |
+
"efficiency": 1.0
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"summary": {
|
| 510 |
+
"maze": {
|
| 511 |
+
"episodes": 36,
|
| 512 |
+
"solve_rate": 1.0,
|
| 513 |
+
"mean_steps_when_solved": 93.19444444444444,
|
| 514 |
+
"mean_efficiency": 1.0,
|
| 515 |
+
"mean_collisions": 0.0,
|
| 516 |
+
"deadlock_rate": 0.0,
|
| 517 |
+
"mean_distinct_cells": 94.19444444444444,
|
| 518 |
+
"by_size": {
|
| 519 |
+
"11": {
|
| 520 |
+
"episodes": 12,
|
| 521 |
+
"solve_rate": 1.0,
|
| 522 |
+
"mean_steps": 27.25,
|
| 523 |
+
"mean_efficiency": 1.0,
|
| 524 |
+
"mean_shortest": 27.25
|
| 525 |
+
},
|
| 526 |
+
"21": {
|
| 527 |
+
"episodes": 12,
|
| 528 |
+
"solve_rate": 1.0,
|
| 529 |
+
"mean_steps": 90.41666666666667,
|
| 530 |
+
"mean_efficiency": 1.0,
|
| 531 |
+
"mean_shortest": 90.41666666666667
|
| 532 |
+
},
|
| 533 |
+
"31": {
|
| 534 |
+
"episodes": 12,
|
| 535 |
+
"solve_rate": 1.0,
|
| 536 |
+
"mean_steps": 161.91666666666666,
|
| 537 |
+
"mean_efficiency": 1.0,
|
| 538 |
+
"mean_shortest": 161.91666666666666
|
| 539 |
+
}
|
| 540 |
+
}
|
| 541 |
+
}
|
| 542 |
+
}
|
| 543 |
+
}
|
|
@@ -0,0 +1,543 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "model+memory",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"solved": true,
|
| 6 |
+
"steps": 32,
|
| 7 |
+
"collisions": 0,
|
| 8 |
+
"distinct_cells": 33,
|
| 9 |
+
"longest_stuck": 0,
|
| 10 |
+
"deadlocked": false,
|
| 11 |
+
"game": "maze",
|
| 12 |
+
"size": 11,
|
| 13 |
+
"topology": "corridor",
|
| 14 |
+
"seed": 901111,
|
| 15 |
+
"shortest": 32,
|
| 16 |
+
"efficiency": 1.0
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"solved": true,
|
| 20 |
+
"steps": 42,
|
| 21 |
+
"collisions": 0,
|
| 22 |
+
"distinct_cells": 43,
|
| 23 |
+
"longest_stuck": 0,
|
| 24 |
+
"deadlocked": false,
|
| 25 |
+
"game": "maze",
|
| 26 |
+
"size": 11,
|
| 27 |
+
"topology": "corridor",
|
| 28 |
+
"seed": 901148,
|
| 29 |
+
"shortest": 42,
|
| 30 |
+
"efficiency": 1.0
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"solved": true,
|
| 34 |
+
"steps": 40,
|
| 35 |
+
"collisions": 0,
|
| 36 |
+
"distinct_cells": 41,
|
| 37 |
+
"longest_stuck": 0,
|
| 38 |
+
"deadlocked": false,
|
| 39 |
+
"game": "maze",
|
| 40 |
+
"size": 11,
|
| 41 |
+
"topology": "corridor",
|
| 42 |
+
"seed": 901185,
|
| 43 |
+
"shortest": 40,
|
| 44 |
+
"efficiency": 1.0
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"solved": true,
|
| 48 |
+
"steps": 44,
|
| 49 |
+
"collisions": 0,
|
| 50 |
+
"distinct_cells": 45,
|
| 51 |
+
"longest_stuck": 0,
|
| 52 |
+
"deadlocked": false,
|
| 53 |
+
"game": "maze",
|
| 54 |
+
"size": 11,
|
| 55 |
+
"topology": "tree",
|
| 56 |
+
"seed": 901111,
|
| 57 |
+
"shortest": 44,
|
| 58 |
+
"efficiency": 1.0
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"solved": true,
|
| 62 |
+
"steps": 34,
|
| 63 |
+
"collisions": 0,
|
| 64 |
+
"distinct_cells": 35,
|
| 65 |
+
"longest_stuck": 0,
|
| 66 |
+
"deadlocked": false,
|
| 67 |
+
"game": "maze",
|
| 68 |
+
"size": 11,
|
| 69 |
+
"topology": "tree",
|
| 70 |
+
"seed": 901148,
|
| 71 |
+
"shortest": 34,
|
| 72 |
+
"efficiency": 1.0
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"solved": true,
|
| 76 |
+
"steps": 36,
|
| 77 |
+
"collisions": 0,
|
| 78 |
+
"distinct_cells": 37,
|
| 79 |
+
"longest_stuck": 0,
|
| 80 |
+
"deadlocked": false,
|
| 81 |
+
"game": "maze",
|
| 82 |
+
"size": 11,
|
| 83 |
+
"topology": "tree",
|
| 84 |
+
"seed": 901185,
|
| 85 |
+
"shortest": 36,
|
| 86 |
+
"efficiency": 1.0
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"solved": true,
|
| 90 |
+
"steps": 15,
|
| 91 |
+
"collisions": 0,
|
| 92 |
+
"distinct_cells": 16,
|
| 93 |
+
"longest_stuck": 0,
|
| 94 |
+
"deadlocked": false,
|
| 95 |
+
"game": "maze",
|
| 96 |
+
"size": 11,
|
| 97 |
+
"topology": "loops",
|
| 98 |
+
"seed": 901111,
|
| 99 |
+
"shortest": 15,
|
| 100 |
+
"efficiency": 1.0
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"solved": true,
|
| 104 |
+
"steps": 22,
|
| 105 |
+
"collisions": 0,
|
| 106 |
+
"distinct_cells": 23,
|
| 107 |
+
"longest_stuck": 0,
|
| 108 |
+
"deadlocked": false,
|
| 109 |
+
"game": "maze",
|
| 110 |
+
"size": 11,
|
| 111 |
+
"topology": "loops",
|
| 112 |
+
"seed": 901148,
|
| 113 |
+
"shortest": 22,
|
| 114 |
+
"efficiency": 1.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"solved": true,
|
| 118 |
+
"steps": 18,
|
| 119 |
+
"collisions": 0,
|
| 120 |
+
"distinct_cells": 19,
|
| 121 |
+
"longest_stuck": 0,
|
| 122 |
+
"deadlocked": false,
|
| 123 |
+
"game": "maze",
|
| 124 |
+
"size": 11,
|
| 125 |
+
"topology": "loops",
|
| 126 |
+
"seed": 901185,
|
| 127 |
+
"shortest": 18,
|
| 128 |
+
"efficiency": 1.0
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"solved": true,
|
| 132 |
+
"steps": 15,
|
| 133 |
+
"collisions": 0,
|
| 134 |
+
"distinct_cells": 16,
|
| 135 |
+
"longest_stuck": 0,
|
| 136 |
+
"deadlocked": false,
|
| 137 |
+
"game": "maze",
|
| 138 |
+
"size": 11,
|
| 139 |
+
"topology": "random_obstacle",
|
| 140 |
+
"seed": 901111,
|
| 141 |
+
"shortest": 15,
|
| 142 |
+
"efficiency": 1.0
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"solved": true,
|
| 146 |
+
"steps": 16,
|
| 147 |
+
"collisions": 0,
|
| 148 |
+
"distinct_cells": 17,
|
| 149 |
+
"longest_stuck": 0,
|
| 150 |
+
"deadlocked": false,
|
| 151 |
+
"game": "maze",
|
| 152 |
+
"size": 11,
|
| 153 |
+
"topology": "random_obstacle",
|
| 154 |
+
"seed": 901148,
|
| 155 |
+
"shortest": 16,
|
| 156 |
+
"efficiency": 1.0
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"solved": true,
|
| 160 |
+
"steps": 13,
|
| 161 |
+
"collisions": 0,
|
| 162 |
+
"distinct_cells": 14,
|
| 163 |
+
"longest_stuck": 0,
|
| 164 |
+
"deadlocked": false,
|
| 165 |
+
"game": "maze",
|
| 166 |
+
"size": 11,
|
| 167 |
+
"topology": "random_obstacle",
|
| 168 |
+
"seed": 901185,
|
| 169 |
+
"shortest": 13,
|
| 170 |
+
"efficiency": 1.0
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"solved": true,
|
| 174 |
+
"steps": 142,
|
| 175 |
+
"collisions": 0,
|
| 176 |
+
"distinct_cells": 143,
|
| 177 |
+
"longest_stuck": 0,
|
| 178 |
+
"deadlocked": false,
|
| 179 |
+
"game": "maze",
|
| 180 |
+
"size": 21,
|
| 181 |
+
"topology": "corridor",
|
| 182 |
+
"seed": 902121,
|
| 183 |
+
"shortest": 142,
|
| 184 |
+
"efficiency": 1.0
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"solved": true,
|
| 188 |
+
"steps": 116,
|
| 189 |
+
"collisions": 0,
|
| 190 |
+
"distinct_cells": 117,
|
| 191 |
+
"longest_stuck": 0,
|
| 192 |
+
"deadlocked": false,
|
| 193 |
+
"game": "maze",
|
| 194 |
+
"size": 21,
|
| 195 |
+
"topology": "corridor",
|
| 196 |
+
"seed": 902158,
|
| 197 |
+
"shortest": 116,
|
| 198 |
+
"efficiency": 1.0
|
| 199 |
+
},
|
| 200 |
+
{
|
| 201 |
+
"solved": true,
|
| 202 |
+
"steps": 156,
|
| 203 |
+
"collisions": 0,
|
| 204 |
+
"distinct_cells": 157,
|
| 205 |
+
"longest_stuck": 0,
|
| 206 |
+
"deadlocked": false,
|
| 207 |
+
"game": "maze",
|
| 208 |
+
"size": 21,
|
| 209 |
+
"topology": "corridor",
|
| 210 |
+
"seed": 902195,
|
| 211 |
+
"shortest": 156,
|
| 212 |
+
"efficiency": 1.0
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"solved": true,
|
| 216 |
+
"steps": 118,
|
| 217 |
+
"collisions": 0,
|
| 218 |
+
"distinct_cells": 119,
|
| 219 |
+
"longest_stuck": 0,
|
| 220 |
+
"deadlocked": false,
|
| 221 |
+
"game": "maze",
|
| 222 |
+
"size": 21,
|
| 223 |
+
"topology": "tree",
|
| 224 |
+
"seed": 902121,
|
| 225 |
+
"shortest": 118,
|
| 226 |
+
"efficiency": 1.0
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"solved": true,
|
| 230 |
+
"steps": 172,
|
| 231 |
+
"collisions": 0,
|
| 232 |
+
"distinct_cells": 173,
|
| 233 |
+
"longest_stuck": 0,
|
| 234 |
+
"deadlocked": false,
|
| 235 |
+
"game": "maze",
|
| 236 |
+
"size": 21,
|
| 237 |
+
"topology": "tree",
|
| 238 |
+
"seed": 902158,
|
| 239 |
+
"shortest": 172,
|
| 240 |
+
"efficiency": 1.0
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"solved": true,
|
| 244 |
+
"steps": 140,
|
| 245 |
+
"collisions": 0,
|
| 246 |
+
"distinct_cells": 141,
|
| 247 |
+
"longest_stuck": 0,
|
| 248 |
+
"deadlocked": false,
|
| 249 |
+
"game": "maze",
|
| 250 |
+
"size": 21,
|
| 251 |
+
"topology": "tree",
|
| 252 |
+
"seed": 902195,
|
| 253 |
+
"shortest": 140,
|
| 254 |
+
"efficiency": 1.0
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"solved": true,
|
| 258 |
+
"steps": 46,
|
| 259 |
+
"collisions": 0,
|
| 260 |
+
"distinct_cells": 47,
|
| 261 |
+
"longest_stuck": 0,
|
| 262 |
+
"deadlocked": false,
|
| 263 |
+
"game": "maze",
|
| 264 |
+
"size": 21,
|
| 265 |
+
"topology": "loops",
|
| 266 |
+
"seed": 902121,
|
| 267 |
+
"shortest": 46,
|
| 268 |
+
"efficiency": 1.0
|
| 269 |
+
},
|
| 270 |
+
{
|
| 271 |
+
"solved": true,
|
| 272 |
+
"steps": 40,
|
| 273 |
+
"collisions": 0,
|
| 274 |
+
"distinct_cells": 41,
|
| 275 |
+
"longest_stuck": 0,
|
| 276 |
+
"deadlocked": false,
|
| 277 |
+
"game": "maze",
|
| 278 |
+
"size": 21,
|
| 279 |
+
"topology": "loops",
|
| 280 |
+
"seed": 902158,
|
| 281 |
+
"shortest": 40,
|
| 282 |
+
"efficiency": 1.0
|
| 283 |
+
},
|
| 284 |
+
{
|
| 285 |
+
"solved": true,
|
| 286 |
+
"steps": 42,
|
| 287 |
+
"collisions": 0,
|
| 288 |
+
"distinct_cells": 43,
|
| 289 |
+
"longest_stuck": 0,
|
| 290 |
+
"deadlocked": false,
|
| 291 |
+
"game": "maze",
|
| 292 |
+
"size": 21,
|
| 293 |
+
"topology": "loops",
|
| 294 |
+
"seed": 902195,
|
| 295 |
+
"shortest": 42,
|
| 296 |
+
"efficiency": 1.0
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"solved": true,
|
| 300 |
+
"steps": 41,
|
| 301 |
+
"collisions": 0,
|
| 302 |
+
"distinct_cells": 42,
|
| 303 |
+
"longest_stuck": 0,
|
| 304 |
+
"deadlocked": false,
|
| 305 |
+
"game": "maze",
|
| 306 |
+
"size": 21,
|
| 307 |
+
"topology": "random_obstacle",
|
| 308 |
+
"seed": 902121,
|
| 309 |
+
"shortest": 41,
|
| 310 |
+
"efficiency": 1.0
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"solved": true,
|
| 314 |
+
"steps": 36,
|
| 315 |
+
"collisions": 0,
|
| 316 |
+
"distinct_cells": 37,
|
| 317 |
+
"longest_stuck": 0,
|
| 318 |
+
"deadlocked": false,
|
| 319 |
+
"game": "maze",
|
| 320 |
+
"size": 21,
|
| 321 |
+
"topology": "random_obstacle",
|
| 322 |
+
"seed": 902158,
|
| 323 |
+
"shortest": 36,
|
| 324 |
+
"efficiency": 1.0
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"solved": true,
|
| 328 |
+
"steps": 36,
|
| 329 |
+
"collisions": 0,
|
| 330 |
+
"distinct_cells": 37,
|
| 331 |
+
"longest_stuck": 0,
|
| 332 |
+
"deadlocked": false,
|
| 333 |
+
"game": "maze",
|
| 334 |
+
"size": 21,
|
| 335 |
+
"topology": "random_obstacle",
|
| 336 |
+
"seed": 902195,
|
| 337 |
+
"shortest": 36,
|
| 338 |
+
"efficiency": 1.0
|
| 339 |
+
},
|
| 340 |
+
{
|
| 341 |
+
"solved": true,
|
| 342 |
+
"steps": 264,
|
| 343 |
+
"collisions": 0,
|
| 344 |
+
"distinct_cells": 265,
|
| 345 |
+
"longest_stuck": 0,
|
| 346 |
+
"deadlocked": false,
|
| 347 |
+
"game": "maze",
|
| 348 |
+
"size": 31,
|
| 349 |
+
"topology": "corridor",
|
| 350 |
+
"seed": 903131,
|
| 351 |
+
"shortest": 264,
|
| 352 |
+
"efficiency": 1.0
|
| 353 |
+
},
|
| 354 |
+
{
|
| 355 |
+
"solved": true,
|
| 356 |
+
"steps": 194,
|
| 357 |
+
"collisions": 0,
|
| 358 |
+
"distinct_cells": 195,
|
| 359 |
+
"longest_stuck": 0,
|
| 360 |
+
"deadlocked": false,
|
| 361 |
+
"game": "maze",
|
| 362 |
+
"size": 31,
|
| 363 |
+
"topology": "corridor",
|
| 364 |
+
"seed": 903168,
|
| 365 |
+
"shortest": 194,
|
| 366 |
+
"efficiency": 1.0
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"solved": true,
|
| 370 |
+
"steps": 286,
|
| 371 |
+
"collisions": 0,
|
| 372 |
+
"distinct_cells": 287,
|
| 373 |
+
"longest_stuck": 0,
|
| 374 |
+
"deadlocked": false,
|
| 375 |
+
"game": "maze",
|
| 376 |
+
"size": 31,
|
| 377 |
+
"topology": "corridor",
|
| 378 |
+
"seed": 903205,
|
| 379 |
+
"shortest": 286,
|
| 380 |
+
"efficiency": 1.0
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"solved": true,
|
| 384 |
+
"steps": 268,
|
| 385 |
+
"collisions": 0,
|
| 386 |
+
"distinct_cells": 269,
|
| 387 |
+
"longest_stuck": 0,
|
| 388 |
+
"deadlocked": false,
|
| 389 |
+
"game": "maze",
|
| 390 |
+
"size": 31,
|
| 391 |
+
"topology": "tree",
|
| 392 |
+
"seed": 903131,
|
| 393 |
+
"shortest": 268,
|
| 394 |
+
"efficiency": 1.0
|
| 395 |
+
},
|
| 396 |
+
{
|
| 397 |
+
"solved": true,
|
| 398 |
+
"steps": 282,
|
| 399 |
+
"collisions": 0,
|
| 400 |
+
"distinct_cells": 283,
|
| 401 |
+
"longest_stuck": 0,
|
| 402 |
+
"deadlocked": false,
|
| 403 |
+
"game": "maze",
|
| 404 |
+
"size": 31,
|
| 405 |
+
"topology": "tree",
|
| 406 |
+
"seed": 903168,
|
| 407 |
+
"shortest": 282,
|
| 408 |
+
"efficiency": 1.0
|
| 409 |
+
},
|
| 410 |
+
{
|
| 411 |
+
"solved": true,
|
| 412 |
+
"steps": 252,
|
| 413 |
+
"collisions": 0,
|
| 414 |
+
"distinct_cells": 253,
|
| 415 |
+
"longest_stuck": 0,
|
| 416 |
+
"deadlocked": false,
|
| 417 |
+
"game": "maze",
|
| 418 |
+
"size": 31,
|
| 419 |
+
"topology": "tree",
|
| 420 |
+
"seed": 903205,
|
| 421 |
+
"shortest": 252,
|
| 422 |
+
"efficiency": 1.0
|
| 423 |
+
},
|
| 424 |
+
{
|
| 425 |
+
"solved": true,
|
| 426 |
+
"steps": 66,
|
| 427 |
+
"collisions": 0,
|
| 428 |
+
"distinct_cells": 67,
|
| 429 |
+
"longest_stuck": 0,
|
| 430 |
+
"deadlocked": false,
|
| 431 |
+
"game": "maze",
|
| 432 |
+
"size": 31,
|
| 433 |
+
"topology": "loops",
|
| 434 |
+
"seed": 903131,
|
| 435 |
+
"shortest": 66,
|
| 436 |
+
"efficiency": 1.0
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"solved": true,
|
| 440 |
+
"steps": 72,
|
| 441 |
+
"collisions": 0,
|
| 442 |
+
"distinct_cells": 73,
|
| 443 |
+
"longest_stuck": 0,
|
| 444 |
+
"deadlocked": false,
|
| 445 |
+
"game": "maze",
|
| 446 |
+
"size": 31,
|
| 447 |
+
"topology": "loops",
|
| 448 |
+
"seed": 903168,
|
| 449 |
+
"shortest": 72,
|
| 450 |
+
"efficiency": 1.0
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"solved": true,
|
| 454 |
+
"steps": 88,
|
| 455 |
+
"collisions": 0,
|
| 456 |
+
"distinct_cells": 89,
|
| 457 |
+
"longest_stuck": 0,
|
| 458 |
+
"deadlocked": false,
|
| 459 |
+
"game": "maze",
|
| 460 |
+
"size": 31,
|
| 461 |
+
"topology": "loops",
|
| 462 |
+
"seed": 903205,
|
| 463 |
+
"shortest": 88,
|
| 464 |
+
"efficiency": 1.0
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"solved": true,
|
| 468 |
+
"steps": 57,
|
| 469 |
+
"collisions": 0,
|
| 470 |
+
"distinct_cells": 58,
|
| 471 |
+
"longest_stuck": 0,
|
| 472 |
+
"deadlocked": false,
|
| 473 |
+
"game": "maze",
|
| 474 |
+
"size": 31,
|
| 475 |
+
"topology": "random_obstacle",
|
| 476 |
+
"seed": 903131,
|
| 477 |
+
"shortest": 57,
|
| 478 |
+
"efficiency": 1.0
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"solved": true,
|
| 482 |
+
"steps": 56,
|
| 483 |
+
"collisions": 0,
|
| 484 |
+
"distinct_cells": 57,
|
| 485 |
+
"longest_stuck": 0,
|
| 486 |
+
"deadlocked": false,
|
| 487 |
+
"game": "maze",
|
| 488 |
+
"size": 31,
|
| 489 |
+
"topology": "random_obstacle",
|
| 490 |
+
"seed": 903168,
|
| 491 |
+
"shortest": 56,
|
| 492 |
+
"efficiency": 1.0
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"solved": true,
|
| 496 |
+
"steps": 58,
|
| 497 |
+
"collisions": 0,
|
| 498 |
+
"distinct_cells": 59,
|
| 499 |
+
"longest_stuck": 0,
|
| 500 |
+
"deadlocked": false,
|
| 501 |
+
"game": "maze",
|
| 502 |
+
"size": 31,
|
| 503 |
+
"topology": "random_obstacle",
|
| 504 |
+
"seed": 903205,
|
| 505 |
+
"shortest": 58,
|
| 506 |
+
"efficiency": 1.0
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"summary": {
|
| 510 |
+
"maze": {
|
| 511 |
+
"episodes": 36,
|
| 512 |
+
"solve_rate": 1.0,
|
| 513 |
+
"mean_steps_when_solved": 93.19444444444444,
|
| 514 |
+
"mean_efficiency": 1.0,
|
| 515 |
+
"mean_collisions": 0.0,
|
| 516 |
+
"deadlock_rate": 0.0,
|
| 517 |
+
"mean_distinct_cells": 94.19444444444444,
|
| 518 |
+
"by_size": {
|
| 519 |
+
"11": {
|
| 520 |
+
"episodes": 12,
|
| 521 |
+
"solve_rate": 1.0,
|
| 522 |
+
"mean_steps": 27.25,
|
| 523 |
+
"mean_efficiency": 1.0,
|
| 524 |
+
"mean_shortest": 27.25
|
| 525 |
+
},
|
| 526 |
+
"21": {
|
| 527 |
+
"episodes": 12,
|
| 528 |
+
"solve_rate": 1.0,
|
| 529 |
+
"mean_steps": 90.41666666666667,
|
| 530 |
+
"mean_efficiency": 1.0,
|
| 531 |
+
"mean_shortest": 90.41666666666667
|
| 532 |
+
},
|
| 533 |
+
"31": {
|
| 534 |
+
"episodes": 12,
|
| 535 |
+
"solve_rate": 1.0,
|
| 536 |
+
"mean_steps": 161.91666666666666,
|
| 537 |
+
"mean_efficiency": 1.0,
|
| 538 |
+
"mean_shortest": 161.91666666666666
|
| 539 |
+
}
|
| 540 |
+
}
|
| 541 |
+
}
|
| 542 |
+
}
|
| 543 |
+
}
|
|
@@ -0,0 +1,543 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "model",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"solved": true,
|
| 6 |
+
"steps": 32,
|
| 7 |
+
"collisions": 0,
|
| 8 |
+
"distinct_cells": 33,
|
| 9 |
+
"longest_stuck": 0,
|
| 10 |
+
"deadlocked": false,
|
| 11 |
+
"game": "maze",
|
| 12 |
+
"size": 11,
|
| 13 |
+
"topology": "corridor",
|
| 14 |
+
"seed": 901111,
|
| 15 |
+
"shortest": 32,
|
| 16 |
+
"efficiency": 1.0
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"solved": true,
|
| 20 |
+
"steps": 42,
|
| 21 |
+
"collisions": 0,
|
| 22 |
+
"distinct_cells": 43,
|
| 23 |
+
"longest_stuck": 0,
|
| 24 |
+
"deadlocked": false,
|
| 25 |
+
"game": "maze",
|
| 26 |
+
"size": 11,
|
| 27 |
+
"topology": "corridor",
|
| 28 |
+
"seed": 901148,
|
| 29 |
+
"shortest": 42,
|
| 30 |
+
"efficiency": 1.0
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"solved": true,
|
| 34 |
+
"steps": 40,
|
| 35 |
+
"collisions": 0,
|
| 36 |
+
"distinct_cells": 41,
|
| 37 |
+
"longest_stuck": 0,
|
| 38 |
+
"deadlocked": false,
|
| 39 |
+
"game": "maze",
|
| 40 |
+
"size": 11,
|
| 41 |
+
"topology": "corridor",
|
| 42 |
+
"seed": 901185,
|
| 43 |
+
"shortest": 40,
|
| 44 |
+
"efficiency": 1.0
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"solved": true,
|
| 48 |
+
"steps": 44,
|
| 49 |
+
"collisions": 0,
|
| 50 |
+
"distinct_cells": 45,
|
| 51 |
+
"longest_stuck": 0,
|
| 52 |
+
"deadlocked": false,
|
| 53 |
+
"game": "maze",
|
| 54 |
+
"size": 11,
|
| 55 |
+
"topology": "tree",
|
| 56 |
+
"seed": 901111,
|
| 57 |
+
"shortest": 44,
|
| 58 |
+
"efficiency": 1.0
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"solved": true,
|
| 62 |
+
"steps": 34,
|
| 63 |
+
"collisions": 0,
|
| 64 |
+
"distinct_cells": 35,
|
| 65 |
+
"longest_stuck": 0,
|
| 66 |
+
"deadlocked": false,
|
| 67 |
+
"game": "maze",
|
| 68 |
+
"size": 11,
|
| 69 |
+
"topology": "tree",
|
| 70 |
+
"seed": 901148,
|
| 71 |
+
"shortest": 34,
|
| 72 |
+
"efficiency": 1.0
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"solved": true,
|
| 76 |
+
"steps": 36,
|
| 77 |
+
"collisions": 0,
|
| 78 |
+
"distinct_cells": 37,
|
| 79 |
+
"longest_stuck": 0,
|
| 80 |
+
"deadlocked": false,
|
| 81 |
+
"game": "maze",
|
| 82 |
+
"size": 11,
|
| 83 |
+
"topology": "tree",
|
| 84 |
+
"seed": 901185,
|
| 85 |
+
"shortest": 36,
|
| 86 |
+
"efficiency": 1.0
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"solved": true,
|
| 90 |
+
"steps": 15,
|
| 91 |
+
"collisions": 0,
|
| 92 |
+
"distinct_cells": 16,
|
| 93 |
+
"longest_stuck": 0,
|
| 94 |
+
"deadlocked": false,
|
| 95 |
+
"game": "maze",
|
| 96 |
+
"size": 11,
|
| 97 |
+
"topology": "loops",
|
| 98 |
+
"seed": 901111,
|
| 99 |
+
"shortest": 15,
|
| 100 |
+
"efficiency": 1.0
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"solved": true,
|
| 104 |
+
"steps": 22,
|
| 105 |
+
"collisions": 0,
|
| 106 |
+
"distinct_cells": 23,
|
| 107 |
+
"longest_stuck": 0,
|
| 108 |
+
"deadlocked": false,
|
| 109 |
+
"game": "maze",
|
| 110 |
+
"size": 11,
|
| 111 |
+
"topology": "loops",
|
| 112 |
+
"seed": 901148,
|
| 113 |
+
"shortest": 22,
|
| 114 |
+
"efficiency": 1.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"solved": true,
|
| 118 |
+
"steps": 18,
|
| 119 |
+
"collisions": 0,
|
| 120 |
+
"distinct_cells": 19,
|
| 121 |
+
"longest_stuck": 0,
|
| 122 |
+
"deadlocked": false,
|
| 123 |
+
"game": "maze",
|
| 124 |
+
"size": 11,
|
| 125 |
+
"topology": "loops",
|
| 126 |
+
"seed": 901185,
|
| 127 |
+
"shortest": 18,
|
| 128 |
+
"efficiency": 1.0
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"solved": true,
|
| 132 |
+
"steps": 15,
|
| 133 |
+
"collisions": 0,
|
| 134 |
+
"distinct_cells": 16,
|
| 135 |
+
"longest_stuck": 0,
|
| 136 |
+
"deadlocked": false,
|
| 137 |
+
"game": "maze",
|
| 138 |
+
"size": 11,
|
| 139 |
+
"topology": "random_obstacle",
|
| 140 |
+
"seed": 901111,
|
| 141 |
+
"shortest": 15,
|
| 142 |
+
"efficiency": 1.0
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"solved": true,
|
| 146 |
+
"steps": 16,
|
| 147 |
+
"collisions": 0,
|
| 148 |
+
"distinct_cells": 17,
|
| 149 |
+
"longest_stuck": 0,
|
| 150 |
+
"deadlocked": false,
|
| 151 |
+
"game": "maze",
|
| 152 |
+
"size": 11,
|
| 153 |
+
"topology": "random_obstacle",
|
| 154 |
+
"seed": 901148,
|
| 155 |
+
"shortest": 16,
|
| 156 |
+
"efficiency": 1.0
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"solved": true,
|
| 160 |
+
"steps": 13,
|
| 161 |
+
"collisions": 0,
|
| 162 |
+
"distinct_cells": 14,
|
| 163 |
+
"longest_stuck": 0,
|
| 164 |
+
"deadlocked": false,
|
| 165 |
+
"game": "maze",
|
| 166 |
+
"size": 11,
|
| 167 |
+
"topology": "random_obstacle",
|
| 168 |
+
"seed": 901185,
|
| 169 |
+
"shortest": 13,
|
| 170 |
+
"efficiency": 1.0
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"solved": true,
|
| 174 |
+
"steps": 144,
|
| 175 |
+
"collisions": 0,
|
| 176 |
+
"distinct_cells": 143,
|
| 177 |
+
"longest_stuck": 0,
|
| 178 |
+
"deadlocked": false,
|
| 179 |
+
"game": "maze",
|
| 180 |
+
"size": 21,
|
| 181 |
+
"topology": "corridor",
|
| 182 |
+
"seed": 902121,
|
| 183 |
+
"shortest": 142,
|
| 184 |
+
"efficiency": 0.9861111111111112
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"solved": true,
|
| 188 |
+
"steps": 116,
|
| 189 |
+
"collisions": 0,
|
| 190 |
+
"distinct_cells": 117,
|
| 191 |
+
"longest_stuck": 0,
|
| 192 |
+
"deadlocked": false,
|
| 193 |
+
"game": "maze",
|
| 194 |
+
"size": 21,
|
| 195 |
+
"topology": "corridor",
|
| 196 |
+
"seed": 902158,
|
| 197 |
+
"shortest": 116,
|
| 198 |
+
"efficiency": 1.0
|
| 199 |
+
},
|
| 200 |
+
{
|
| 201 |
+
"solved": true,
|
| 202 |
+
"steps": 156,
|
| 203 |
+
"collisions": 0,
|
| 204 |
+
"distinct_cells": 157,
|
| 205 |
+
"longest_stuck": 0,
|
| 206 |
+
"deadlocked": false,
|
| 207 |
+
"game": "maze",
|
| 208 |
+
"size": 21,
|
| 209 |
+
"topology": "corridor",
|
| 210 |
+
"seed": 902195,
|
| 211 |
+
"shortest": 156,
|
| 212 |
+
"efficiency": 1.0
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"solved": true,
|
| 216 |
+
"steps": 118,
|
| 217 |
+
"collisions": 0,
|
| 218 |
+
"distinct_cells": 119,
|
| 219 |
+
"longest_stuck": 0,
|
| 220 |
+
"deadlocked": false,
|
| 221 |
+
"game": "maze",
|
| 222 |
+
"size": 21,
|
| 223 |
+
"topology": "tree",
|
| 224 |
+
"seed": 902121,
|
| 225 |
+
"shortest": 118,
|
| 226 |
+
"efficiency": 1.0
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"solved": true,
|
| 230 |
+
"steps": 174,
|
| 231 |
+
"collisions": 0,
|
| 232 |
+
"distinct_cells": 173,
|
| 233 |
+
"longest_stuck": 0,
|
| 234 |
+
"deadlocked": false,
|
| 235 |
+
"game": "maze",
|
| 236 |
+
"size": 21,
|
| 237 |
+
"topology": "tree",
|
| 238 |
+
"seed": 902158,
|
| 239 |
+
"shortest": 172,
|
| 240 |
+
"efficiency": 0.9885057471264368
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"solved": true,
|
| 244 |
+
"steps": 140,
|
| 245 |
+
"collisions": 0,
|
| 246 |
+
"distinct_cells": 141,
|
| 247 |
+
"longest_stuck": 0,
|
| 248 |
+
"deadlocked": false,
|
| 249 |
+
"game": "maze",
|
| 250 |
+
"size": 21,
|
| 251 |
+
"topology": "tree",
|
| 252 |
+
"seed": 902195,
|
| 253 |
+
"shortest": 140,
|
| 254 |
+
"efficiency": 1.0
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"solved": true,
|
| 258 |
+
"steps": 46,
|
| 259 |
+
"collisions": 0,
|
| 260 |
+
"distinct_cells": 47,
|
| 261 |
+
"longest_stuck": 0,
|
| 262 |
+
"deadlocked": false,
|
| 263 |
+
"game": "maze",
|
| 264 |
+
"size": 21,
|
| 265 |
+
"topology": "loops",
|
| 266 |
+
"seed": 902121,
|
| 267 |
+
"shortest": 46,
|
| 268 |
+
"efficiency": 1.0
|
| 269 |
+
},
|
| 270 |
+
{
|
| 271 |
+
"solved": true,
|
| 272 |
+
"steps": 40,
|
| 273 |
+
"collisions": 0,
|
| 274 |
+
"distinct_cells": 41,
|
| 275 |
+
"longest_stuck": 0,
|
| 276 |
+
"deadlocked": false,
|
| 277 |
+
"game": "maze",
|
| 278 |
+
"size": 21,
|
| 279 |
+
"topology": "loops",
|
| 280 |
+
"seed": 902158,
|
| 281 |
+
"shortest": 40,
|
| 282 |
+
"efficiency": 1.0
|
| 283 |
+
},
|
| 284 |
+
{
|
| 285 |
+
"solved": true,
|
| 286 |
+
"steps": 42,
|
| 287 |
+
"collisions": 0,
|
| 288 |
+
"distinct_cells": 43,
|
| 289 |
+
"longest_stuck": 0,
|
| 290 |
+
"deadlocked": false,
|
| 291 |
+
"game": "maze",
|
| 292 |
+
"size": 21,
|
| 293 |
+
"topology": "loops",
|
| 294 |
+
"seed": 902195,
|
| 295 |
+
"shortest": 42,
|
| 296 |
+
"efficiency": 1.0
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"solved": true,
|
| 300 |
+
"steps": 41,
|
| 301 |
+
"collisions": 0,
|
| 302 |
+
"distinct_cells": 42,
|
| 303 |
+
"longest_stuck": 0,
|
| 304 |
+
"deadlocked": false,
|
| 305 |
+
"game": "maze",
|
| 306 |
+
"size": 21,
|
| 307 |
+
"topology": "random_obstacle",
|
| 308 |
+
"seed": 902121,
|
| 309 |
+
"shortest": 41,
|
| 310 |
+
"efficiency": 1.0
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"solved": true,
|
| 314 |
+
"steps": 36,
|
| 315 |
+
"collisions": 0,
|
| 316 |
+
"distinct_cells": 37,
|
| 317 |
+
"longest_stuck": 0,
|
| 318 |
+
"deadlocked": false,
|
| 319 |
+
"game": "maze",
|
| 320 |
+
"size": 21,
|
| 321 |
+
"topology": "random_obstacle",
|
| 322 |
+
"seed": 902158,
|
| 323 |
+
"shortest": 36,
|
| 324 |
+
"efficiency": 1.0
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"solved": true,
|
| 328 |
+
"steps": 36,
|
| 329 |
+
"collisions": 0,
|
| 330 |
+
"distinct_cells": 37,
|
| 331 |
+
"longest_stuck": 0,
|
| 332 |
+
"deadlocked": false,
|
| 333 |
+
"game": "maze",
|
| 334 |
+
"size": 21,
|
| 335 |
+
"topology": "random_obstacle",
|
| 336 |
+
"seed": 902195,
|
| 337 |
+
"shortest": 36,
|
| 338 |
+
"efficiency": 1.0
|
| 339 |
+
},
|
| 340 |
+
{
|
| 341 |
+
"solved": true,
|
| 342 |
+
"steps": 328,
|
| 343 |
+
"collisions": 0,
|
| 344 |
+
"distinct_cells": 266,
|
| 345 |
+
"longest_stuck": 0,
|
| 346 |
+
"deadlocked": false,
|
| 347 |
+
"game": "maze",
|
| 348 |
+
"size": 31,
|
| 349 |
+
"topology": "corridor",
|
| 350 |
+
"seed": 903131,
|
| 351 |
+
"shortest": 264,
|
| 352 |
+
"efficiency": 0.8048780487804879
|
| 353 |
+
},
|
| 354 |
+
{
|
| 355 |
+
"solved": true,
|
| 356 |
+
"steps": 238,
|
| 357 |
+
"collisions": 0,
|
| 358 |
+
"distinct_cells": 195,
|
| 359 |
+
"longest_stuck": 0,
|
| 360 |
+
"deadlocked": false,
|
| 361 |
+
"game": "maze",
|
| 362 |
+
"size": 31,
|
| 363 |
+
"topology": "corridor",
|
| 364 |
+
"seed": 903168,
|
| 365 |
+
"shortest": 194,
|
| 366 |
+
"efficiency": 0.8151260504201681
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"solved": true,
|
| 370 |
+
"steps": 332,
|
| 371 |
+
"collisions": 0,
|
| 372 |
+
"distinct_cells": 287,
|
| 373 |
+
"longest_stuck": 0,
|
| 374 |
+
"deadlocked": false,
|
| 375 |
+
"game": "maze",
|
| 376 |
+
"size": 31,
|
| 377 |
+
"topology": "corridor",
|
| 378 |
+
"seed": 903205,
|
| 379 |
+
"shortest": 286,
|
| 380 |
+
"efficiency": 0.8614457831325302
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"solved": true,
|
| 384 |
+
"steps": 314,
|
| 385 |
+
"collisions": 0,
|
| 386 |
+
"distinct_cells": 270,
|
| 387 |
+
"longest_stuck": 0,
|
| 388 |
+
"deadlocked": false,
|
| 389 |
+
"game": "maze",
|
| 390 |
+
"size": 31,
|
| 391 |
+
"topology": "tree",
|
| 392 |
+
"seed": 903131,
|
| 393 |
+
"shortest": 268,
|
| 394 |
+
"efficiency": 0.8535031847133758
|
| 395 |
+
},
|
| 396 |
+
{
|
| 397 |
+
"solved": true,
|
| 398 |
+
"steps": 330,
|
| 399 |
+
"collisions": 0,
|
| 400 |
+
"distinct_cells": 286,
|
| 401 |
+
"longest_stuck": 0,
|
| 402 |
+
"deadlocked": false,
|
| 403 |
+
"game": "maze",
|
| 404 |
+
"size": 31,
|
| 405 |
+
"topology": "tree",
|
| 406 |
+
"seed": 903168,
|
| 407 |
+
"shortest": 282,
|
| 408 |
+
"efficiency": 0.8545454545454545
|
| 409 |
+
},
|
| 410 |
+
{
|
| 411 |
+
"solved": true,
|
| 412 |
+
"steps": 314,
|
| 413 |
+
"collisions": 0,
|
| 414 |
+
"distinct_cells": 254,
|
| 415 |
+
"longest_stuck": 0,
|
| 416 |
+
"deadlocked": false,
|
| 417 |
+
"game": "maze",
|
| 418 |
+
"size": 31,
|
| 419 |
+
"topology": "tree",
|
| 420 |
+
"seed": 903205,
|
| 421 |
+
"shortest": 252,
|
| 422 |
+
"efficiency": 0.802547770700637
|
| 423 |
+
},
|
| 424 |
+
{
|
| 425 |
+
"solved": true,
|
| 426 |
+
"steps": 82,
|
| 427 |
+
"collisions": 0,
|
| 428 |
+
"distinct_cells": 69,
|
| 429 |
+
"longest_stuck": 0,
|
| 430 |
+
"deadlocked": false,
|
| 431 |
+
"game": "maze",
|
| 432 |
+
"size": 31,
|
| 433 |
+
"topology": "loops",
|
| 434 |
+
"seed": 903131,
|
| 435 |
+
"shortest": 66,
|
| 436 |
+
"efficiency": 0.8048780487804879
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"solved": true,
|
| 440 |
+
"steps": 82,
|
| 441 |
+
"collisions": 0,
|
| 442 |
+
"distinct_cells": 74,
|
| 443 |
+
"longest_stuck": 0,
|
| 444 |
+
"deadlocked": false,
|
| 445 |
+
"game": "maze",
|
| 446 |
+
"size": 31,
|
| 447 |
+
"topology": "loops",
|
| 448 |
+
"seed": 903168,
|
| 449 |
+
"shortest": 72,
|
| 450 |
+
"efficiency": 0.8780487804878049
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"solved": true,
|
| 454 |
+
"steps": 120,
|
| 455 |
+
"collisions": 0,
|
| 456 |
+
"distinct_cells": 90,
|
| 457 |
+
"longest_stuck": 0,
|
| 458 |
+
"deadlocked": false,
|
| 459 |
+
"game": "maze",
|
| 460 |
+
"size": 31,
|
| 461 |
+
"topology": "loops",
|
| 462 |
+
"seed": 903205,
|
| 463 |
+
"shortest": 88,
|
| 464 |
+
"efficiency": 0.7333333333333333
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"solved": true,
|
| 468 |
+
"steps": 75,
|
| 469 |
+
"collisions": 0,
|
| 470 |
+
"distinct_cells": 65,
|
| 471 |
+
"longest_stuck": 0,
|
| 472 |
+
"deadlocked": false,
|
| 473 |
+
"game": "maze",
|
| 474 |
+
"size": 31,
|
| 475 |
+
"topology": "random_obstacle",
|
| 476 |
+
"seed": 903131,
|
| 477 |
+
"shortest": 57,
|
| 478 |
+
"efficiency": 0.76
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"solved": true,
|
| 482 |
+
"steps": 102,
|
| 483 |
+
"collisions": 0,
|
| 484 |
+
"distinct_cells": 69,
|
| 485 |
+
"longest_stuck": 0,
|
| 486 |
+
"deadlocked": false,
|
| 487 |
+
"game": "maze",
|
| 488 |
+
"size": 31,
|
| 489 |
+
"topology": "random_obstacle",
|
| 490 |
+
"seed": 903168,
|
| 491 |
+
"shortest": 56,
|
| 492 |
+
"efficiency": 0.5490196078431373
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"solved": true,
|
| 496 |
+
"steps": 88,
|
| 497 |
+
"collisions": 0,
|
| 498 |
+
"distinct_cells": 74,
|
| 499 |
+
"longest_stuck": 0,
|
| 500 |
+
"deadlocked": false,
|
| 501 |
+
"game": "maze",
|
| 502 |
+
"size": 31,
|
| 503 |
+
"topology": "random_obstacle",
|
| 504 |
+
"seed": 903205,
|
| 505 |
+
"shortest": 58,
|
| 506 |
+
"efficiency": 0.6590909090909091
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"summary": {
|
| 510 |
+
"maze": {
|
| 511 |
+
"episodes": 36,
|
| 512 |
+
"solve_rate": 1.0,
|
| 513 |
+
"mean_steps_when_solved": 106.13888888888889,
|
| 514 |
+
"mean_efficiency": 0.9264176063907187,
|
| 515 |
+
"mean_collisions": 0.0,
|
| 516 |
+
"deadlock_rate": 0.0,
|
| 517 |
+
"mean_distinct_cells": 95.41666666666667,
|
| 518 |
+
"by_size": {
|
| 519 |
+
"11": {
|
| 520 |
+
"episodes": 12,
|
| 521 |
+
"solve_rate": 1.0,
|
| 522 |
+
"mean_steps": 27.25,
|
| 523 |
+
"mean_efficiency": 1.0,
|
| 524 |
+
"mean_shortest": 27.25
|
| 525 |
+
},
|
| 526 |
+
"21": {
|
| 527 |
+
"episodes": 12,
|
| 528 |
+
"solve_rate": 1.0,
|
| 529 |
+
"mean_steps": 90.75,
|
| 530 |
+
"mean_efficiency": 0.9978847381864623,
|
| 531 |
+
"mean_shortest": 90.41666666666667
|
| 532 |
+
},
|
| 533 |
+
"31": {
|
| 534 |
+
"episodes": 12,
|
| 535 |
+
"solve_rate": 1.0,
|
| 536 |
+
"mean_steps": 200.41666666666666,
|
| 537 |
+
"mean_efficiency": 0.7813680809856938,
|
| 538 |
+
"mean_shortest": 161.91666666666666
|
| 539 |
+
}
|
| 540 |
+
}
|
| 541 |
+
}
|
| 542 |
+
}
|
| 543 |
+
}
|
|
@@ -0,0 +1,543 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "model",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"solved": true,
|
| 6 |
+
"steps": 32,
|
| 7 |
+
"collisions": 0,
|
| 8 |
+
"distinct_cells": 33,
|
| 9 |
+
"longest_stuck": 0,
|
| 10 |
+
"deadlocked": false,
|
| 11 |
+
"game": "maze",
|
| 12 |
+
"size": 11,
|
| 13 |
+
"topology": "corridor",
|
| 14 |
+
"seed": 901111,
|
| 15 |
+
"shortest": 32,
|
| 16 |
+
"efficiency": 1.0
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"solved": true,
|
| 20 |
+
"steps": 42,
|
| 21 |
+
"collisions": 0,
|
| 22 |
+
"distinct_cells": 43,
|
| 23 |
+
"longest_stuck": 0,
|
| 24 |
+
"deadlocked": false,
|
| 25 |
+
"game": "maze",
|
| 26 |
+
"size": 11,
|
| 27 |
+
"topology": "corridor",
|
| 28 |
+
"seed": 901148,
|
| 29 |
+
"shortest": 42,
|
| 30 |
+
"efficiency": 1.0
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"solved": true,
|
| 34 |
+
"steps": 40,
|
| 35 |
+
"collisions": 0,
|
| 36 |
+
"distinct_cells": 41,
|
| 37 |
+
"longest_stuck": 0,
|
| 38 |
+
"deadlocked": false,
|
| 39 |
+
"game": "maze",
|
| 40 |
+
"size": 11,
|
| 41 |
+
"topology": "corridor",
|
| 42 |
+
"seed": 901185,
|
| 43 |
+
"shortest": 40,
|
| 44 |
+
"efficiency": 1.0
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"solved": true,
|
| 48 |
+
"steps": 44,
|
| 49 |
+
"collisions": 0,
|
| 50 |
+
"distinct_cells": 45,
|
| 51 |
+
"longest_stuck": 0,
|
| 52 |
+
"deadlocked": false,
|
| 53 |
+
"game": "maze",
|
| 54 |
+
"size": 11,
|
| 55 |
+
"topology": "tree",
|
| 56 |
+
"seed": 901111,
|
| 57 |
+
"shortest": 44,
|
| 58 |
+
"efficiency": 1.0
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"solved": true,
|
| 62 |
+
"steps": 34,
|
| 63 |
+
"collisions": 0,
|
| 64 |
+
"distinct_cells": 35,
|
| 65 |
+
"longest_stuck": 0,
|
| 66 |
+
"deadlocked": false,
|
| 67 |
+
"game": "maze",
|
| 68 |
+
"size": 11,
|
| 69 |
+
"topology": "tree",
|
| 70 |
+
"seed": 901148,
|
| 71 |
+
"shortest": 34,
|
| 72 |
+
"efficiency": 1.0
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"solved": true,
|
| 76 |
+
"steps": 36,
|
| 77 |
+
"collisions": 0,
|
| 78 |
+
"distinct_cells": 37,
|
| 79 |
+
"longest_stuck": 0,
|
| 80 |
+
"deadlocked": false,
|
| 81 |
+
"game": "maze",
|
| 82 |
+
"size": 11,
|
| 83 |
+
"topology": "tree",
|
| 84 |
+
"seed": 901185,
|
| 85 |
+
"shortest": 36,
|
| 86 |
+
"efficiency": 1.0
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"solved": true,
|
| 90 |
+
"steps": 15,
|
| 91 |
+
"collisions": 0,
|
| 92 |
+
"distinct_cells": 16,
|
| 93 |
+
"longest_stuck": 0,
|
| 94 |
+
"deadlocked": false,
|
| 95 |
+
"game": "maze",
|
| 96 |
+
"size": 11,
|
| 97 |
+
"topology": "loops",
|
| 98 |
+
"seed": 901111,
|
| 99 |
+
"shortest": 15,
|
| 100 |
+
"efficiency": 1.0
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"solved": true,
|
| 104 |
+
"steps": 22,
|
| 105 |
+
"collisions": 0,
|
| 106 |
+
"distinct_cells": 23,
|
| 107 |
+
"longest_stuck": 0,
|
| 108 |
+
"deadlocked": false,
|
| 109 |
+
"game": "maze",
|
| 110 |
+
"size": 11,
|
| 111 |
+
"topology": "loops",
|
| 112 |
+
"seed": 901148,
|
| 113 |
+
"shortest": 22,
|
| 114 |
+
"efficiency": 1.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"solved": true,
|
| 118 |
+
"steps": 18,
|
| 119 |
+
"collisions": 0,
|
| 120 |
+
"distinct_cells": 19,
|
| 121 |
+
"longest_stuck": 0,
|
| 122 |
+
"deadlocked": false,
|
| 123 |
+
"game": "maze",
|
| 124 |
+
"size": 11,
|
| 125 |
+
"topology": "loops",
|
| 126 |
+
"seed": 901185,
|
| 127 |
+
"shortest": 18,
|
| 128 |
+
"efficiency": 1.0
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"solved": true,
|
| 132 |
+
"steps": 15,
|
| 133 |
+
"collisions": 0,
|
| 134 |
+
"distinct_cells": 16,
|
| 135 |
+
"longest_stuck": 0,
|
| 136 |
+
"deadlocked": false,
|
| 137 |
+
"game": "maze",
|
| 138 |
+
"size": 11,
|
| 139 |
+
"topology": "random_obstacle",
|
| 140 |
+
"seed": 901111,
|
| 141 |
+
"shortest": 15,
|
| 142 |
+
"efficiency": 1.0
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"solved": true,
|
| 146 |
+
"steps": 16,
|
| 147 |
+
"collisions": 0,
|
| 148 |
+
"distinct_cells": 17,
|
| 149 |
+
"longest_stuck": 0,
|
| 150 |
+
"deadlocked": false,
|
| 151 |
+
"game": "maze",
|
| 152 |
+
"size": 11,
|
| 153 |
+
"topology": "random_obstacle",
|
| 154 |
+
"seed": 901148,
|
| 155 |
+
"shortest": 16,
|
| 156 |
+
"efficiency": 1.0
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"solved": true,
|
| 160 |
+
"steps": 13,
|
| 161 |
+
"collisions": 0,
|
| 162 |
+
"distinct_cells": 14,
|
| 163 |
+
"longest_stuck": 0,
|
| 164 |
+
"deadlocked": false,
|
| 165 |
+
"game": "maze",
|
| 166 |
+
"size": 11,
|
| 167 |
+
"topology": "random_obstacle",
|
| 168 |
+
"seed": 901185,
|
| 169 |
+
"shortest": 13,
|
| 170 |
+
"efficiency": 1.0
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"solved": true,
|
| 174 |
+
"steps": 142,
|
| 175 |
+
"collisions": 0,
|
| 176 |
+
"distinct_cells": 143,
|
| 177 |
+
"longest_stuck": 0,
|
| 178 |
+
"deadlocked": false,
|
| 179 |
+
"game": "maze",
|
| 180 |
+
"size": 21,
|
| 181 |
+
"topology": "corridor",
|
| 182 |
+
"seed": 902121,
|
| 183 |
+
"shortest": 142,
|
| 184 |
+
"efficiency": 1.0
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"solved": true,
|
| 188 |
+
"steps": 116,
|
| 189 |
+
"collisions": 0,
|
| 190 |
+
"distinct_cells": 117,
|
| 191 |
+
"longest_stuck": 0,
|
| 192 |
+
"deadlocked": false,
|
| 193 |
+
"game": "maze",
|
| 194 |
+
"size": 21,
|
| 195 |
+
"topology": "corridor",
|
| 196 |
+
"seed": 902158,
|
| 197 |
+
"shortest": 116,
|
| 198 |
+
"efficiency": 1.0
|
| 199 |
+
},
|
| 200 |
+
{
|
| 201 |
+
"solved": true,
|
| 202 |
+
"steps": 156,
|
| 203 |
+
"collisions": 0,
|
| 204 |
+
"distinct_cells": 157,
|
| 205 |
+
"longest_stuck": 0,
|
| 206 |
+
"deadlocked": false,
|
| 207 |
+
"game": "maze",
|
| 208 |
+
"size": 21,
|
| 209 |
+
"topology": "corridor",
|
| 210 |
+
"seed": 902195,
|
| 211 |
+
"shortest": 156,
|
| 212 |
+
"efficiency": 1.0
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"solved": true,
|
| 216 |
+
"steps": 118,
|
| 217 |
+
"collisions": 0,
|
| 218 |
+
"distinct_cells": 119,
|
| 219 |
+
"longest_stuck": 0,
|
| 220 |
+
"deadlocked": false,
|
| 221 |
+
"game": "maze",
|
| 222 |
+
"size": 21,
|
| 223 |
+
"topology": "tree",
|
| 224 |
+
"seed": 902121,
|
| 225 |
+
"shortest": 118,
|
| 226 |
+
"efficiency": 1.0
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"solved": true,
|
| 230 |
+
"steps": 172,
|
| 231 |
+
"collisions": 0,
|
| 232 |
+
"distinct_cells": 173,
|
| 233 |
+
"longest_stuck": 0,
|
| 234 |
+
"deadlocked": false,
|
| 235 |
+
"game": "maze",
|
| 236 |
+
"size": 21,
|
| 237 |
+
"topology": "tree",
|
| 238 |
+
"seed": 902158,
|
| 239 |
+
"shortest": 172,
|
| 240 |
+
"efficiency": 1.0
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"solved": true,
|
| 244 |
+
"steps": 140,
|
| 245 |
+
"collisions": 0,
|
| 246 |
+
"distinct_cells": 141,
|
| 247 |
+
"longest_stuck": 0,
|
| 248 |
+
"deadlocked": false,
|
| 249 |
+
"game": "maze",
|
| 250 |
+
"size": 21,
|
| 251 |
+
"topology": "tree",
|
| 252 |
+
"seed": 902195,
|
| 253 |
+
"shortest": 140,
|
| 254 |
+
"efficiency": 1.0
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"solved": true,
|
| 258 |
+
"steps": 46,
|
| 259 |
+
"collisions": 0,
|
| 260 |
+
"distinct_cells": 47,
|
| 261 |
+
"longest_stuck": 0,
|
| 262 |
+
"deadlocked": false,
|
| 263 |
+
"game": "maze",
|
| 264 |
+
"size": 21,
|
| 265 |
+
"topology": "loops",
|
| 266 |
+
"seed": 902121,
|
| 267 |
+
"shortest": 46,
|
| 268 |
+
"efficiency": 1.0
|
| 269 |
+
},
|
| 270 |
+
{
|
| 271 |
+
"solved": true,
|
| 272 |
+
"steps": 40,
|
| 273 |
+
"collisions": 0,
|
| 274 |
+
"distinct_cells": 41,
|
| 275 |
+
"longest_stuck": 0,
|
| 276 |
+
"deadlocked": false,
|
| 277 |
+
"game": "maze",
|
| 278 |
+
"size": 21,
|
| 279 |
+
"topology": "loops",
|
| 280 |
+
"seed": 902158,
|
| 281 |
+
"shortest": 40,
|
| 282 |
+
"efficiency": 1.0
|
| 283 |
+
},
|
| 284 |
+
{
|
| 285 |
+
"solved": true,
|
| 286 |
+
"steps": 42,
|
| 287 |
+
"collisions": 0,
|
| 288 |
+
"distinct_cells": 43,
|
| 289 |
+
"longest_stuck": 0,
|
| 290 |
+
"deadlocked": false,
|
| 291 |
+
"game": "maze",
|
| 292 |
+
"size": 21,
|
| 293 |
+
"topology": "loops",
|
| 294 |
+
"seed": 902195,
|
| 295 |
+
"shortest": 42,
|
| 296 |
+
"efficiency": 1.0
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"solved": true,
|
| 300 |
+
"steps": 41,
|
| 301 |
+
"collisions": 0,
|
| 302 |
+
"distinct_cells": 42,
|
| 303 |
+
"longest_stuck": 0,
|
| 304 |
+
"deadlocked": false,
|
| 305 |
+
"game": "maze",
|
| 306 |
+
"size": 21,
|
| 307 |
+
"topology": "random_obstacle",
|
| 308 |
+
"seed": 902121,
|
| 309 |
+
"shortest": 41,
|
| 310 |
+
"efficiency": 1.0
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"solved": true,
|
| 314 |
+
"steps": 36,
|
| 315 |
+
"collisions": 0,
|
| 316 |
+
"distinct_cells": 37,
|
| 317 |
+
"longest_stuck": 0,
|
| 318 |
+
"deadlocked": false,
|
| 319 |
+
"game": "maze",
|
| 320 |
+
"size": 21,
|
| 321 |
+
"topology": "random_obstacle",
|
| 322 |
+
"seed": 902158,
|
| 323 |
+
"shortest": 36,
|
| 324 |
+
"efficiency": 1.0
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"solved": true,
|
| 328 |
+
"steps": 36,
|
| 329 |
+
"collisions": 0,
|
| 330 |
+
"distinct_cells": 37,
|
| 331 |
+
"longest_stuck": 0,
|
| 332 |
+
"deadlocked": false,
|
| 333 |
+
"game": "maze",
|
| 334 |
+
"size": 21,
|
| 335 |
+
"topology": "random_obstacle",
|
| 336 |
+
"seed": 902195,
|
| 337 |
+
"shortest": 36,
|
| 338 |
+
"efficiency": 1.0
|
| 339 |
+
},
|
| 340 |
+
{
|
| 341 |
+
"solved": true,
|
| 342 |
+
"steps": 264,
|
| 343 |
+
"collisions": 0,
|
| 344 |
+
"distinct_cells": 265,
|
| 345 |
+
"longest_stuck": 0,
|
| 346 |
+
"deadlocked": false,
|
| 347 |
+
"game": "maze",
|
| 348 |
+
"size": 31,
|
| 349 |
+
"topology": "corridor",
|
| 350 |
+
"seed": 903131,
|
| 351 |
+
"shortest": 264,
|
| 352 |
+
"efficiency": 1.0
|
| 353 |
+
},
|
| 354 |
+
{
|
| 355 |
+
"solved": true,
|
| 356 |
+
"steps": 194,
|
| 357 |
+
"collisions": 0,
|
| 358 |
+
"distinct_cells": 195,
|
| 359 |
+
"longest_stuck": 0,
|
| 360 |
+
"deadlocked": false,
|
| 361 |
+
"game": "maze",
|
| 362 |
+
"size": 31,
|
| 363 |
+
"topology": "corridor",
|
| 364 |
+
"seed": 903168,
|
| 365 |
+
"shortest": 194,
|
| 366 |
+
"efficiency": 1.0
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"solved": true,
|
| 370 |
+
"steps": 286,
|
| 371 |
+
"collisions": 0,
|
| 372 |
+
"distinct_cells": 287,
|
| 373 |
+
"longest_stuck": 0,
|
| 374 |
+
"deadlocked": false,
|
| 375 |
+
"game": "maze",
|
| 376 |
+
"size": 31,
|
| 377 |
+
"topology": "corridor",
|
| 378 |
+
"seed": 903205,
|
| 379 |
+
"shortest": 286,
|
| 380 |
+
"efficiency": 1.0
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"solved": true,
|
| 384 |
+
"steps": 268,
|
| 385 |
+
"collisions": 0,
|
| 386 |
+
"distinct_cells": 269,
|
| 387 |
+
"longest_stuck": 0,
|
| 388 |
+
"deadlocked": false,
|
| 389 |
+
"game": "maze",
|
| 390 |
+
"size": 31,
|
| 391 |
+
"topology": "tree",
|
| 392 |
+
"seed": 903131,
|
| 393 |
+
"shortest": 268,
|
| 394 |
+
"efficiency": 1.0
|
| 395 |
+
},
|
| 396 |
+
{
|
| 397 |
+
"solved": true,
|
| 398 |
+
"steps": 282,
|
| 399 |
+
"collisions": 0,
|
| 400 |
+
"distinct_cells": 283,
|
| 401 |
+
"longest_stuck": 0,
|
| 402 |
+
"deadlocked": false,
|
| 403 |
+
"game": "maze",
|
| 404 |
+
"size": 31,
|
| 405 |
+
"topology": "tree",
|
| 406 |
+
"seed": 903168,
|
| 407 |
+
"shortest": 282,
|
| 408 |
+
"efficiency": 1.0
|
| 409 |
+
},
|
| 410 |
+
{
|
| 411 |
+
"solved": true,
|
| 412 |
+
"steps": 252,
|
| 413 |
+
"collisions": 0,
|
| 414 |
+
"distinct_cells": 253,
|
| 415 |
+
"longest_stuck": 0,
|
| 416 |
+
"deadlocked": false,
|
| 417 |
+
"game": "maze",
|
| 418 |
+
"size": 31,
|
| 419 |
+
"topology": "tree",
|
| 420 |
+
"seed": 903205,
|
| 421 |
+
"shortest": 252,
|
| 422 |
+
"efficiency": 1.0
|
| 423 |
+
},
|
| 424 |
+
{
|
| 425 |
+
"solved": true,
|
| 426 |
+
"steps": 66,
|
| 427 |
+
"collisions": 0,
|
| 428 |
+
"distinct_cells": 67,
|
| 429 |
+
"longest_stuck": 0,
|
| 430 |
+
"deadlocked": false,
|
| 431 |
+
"game": "maze",
|
| 432 |
+
"size": 31,
|
| 433 |
+
"topology": "loops",
|
| 434 |
+
"seed": 903131,
|
| 435 |
+
"shortest": 66,
|
| 436 |
+
"efficiency": 1.0
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"solved": true,
|
| 440 |
+
"steps": 72,
|
| 441 |
+
"collisions": 0,
|
| 442 |
+
"distinct_cells": 73,
|
| 443 |
+
"longest_stuck": 0,
|
| 444 |
+
"deadlocked": false,
|
| 445 |
+
"game": "maze",
|
| 446 |
+
"size": 31,
|
| 447 |
+
"topology": "loops",
|
| 448 |
+
"seed": 903168,
|
| 449 |
+
"shortest": 72,
|
| 450 |
+
"efficiency": 1.0
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"solved": true,
|
| 454 |
+
"steps": 88,
|
| 455 |
+
"collisions": 0,
|
| 456 |
+
"distinct_cells": 89,
|
| 457 |
+
"longest_stuck": 0,
|
| 458 |
+
"deadlocked": false,
|
| 459 |
+
"game": "maze",
|
| 460 |
+
"size": 31,
|
| 461 |
+
"topology": "loops",
|
| 462 |
+
"seed": 903205,
|
| 463 |
+
"shortest": 88,
|
| 464 |
+
"efficiency": 1.0
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"solved": true,
|
| 468 |
+
"steps": 57,
|
| 469 |
+
"collisions": 0,
|
| 470 |
+
"distinct_cells": 58,
|
| 471 |
+
"longest_stuck": 0,
|
| 472 |
+
"deadlocked": false,
|
| 473 |
+
"game": "maze",
|
| 474 |
+
"size": 31,
|
| 475 |
+
"topology": "random_obstacle",
|
| 476 |
+
"seed": 903131,
|
| 477 |
+
"shortest": 57,
|
| 478 |
+
"efficiency": 1.0
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"solved": true,
|
| 482 |
+
"steps": 56,
|
| 483 |
+
"collisions": 0,
|
| 484 |
+
"distinct_cells": 57,
|
| 485 |
+
"longest_stuck": 0,
|
| 486 |
+
"deadlocked": false,
|
| 487 |
+
"game": "maze",
|
| 488 |
+
"size": 31,
|
| 489 |
+
"topology": "random_obstacle",
|
| 490 |
+
"seed": 903168,
|
| 491 |
+
"shortest": 56,
|
| 492 |
+
"efficiency": 1.0
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"solved": true,
|
| 496 |
+
"steps": 58,
|
| 497 |
+
"collisions": 0,
|
| 498 |
+
"distinct_cells": 59,
|
| 499 |
+
"longest_stuck": 0,
|
| 500 |
+
"deadlocked": false,
|
| 501 |
+
"game": "maze",
|
| 502 |
+
"size": 31,
|
| 503 |
+
"topology": "random_obstacle",
|
| 504 |
+
"seed": 903205,
|
| 505 |
+
"shortest": 58,
|
| 506 |
+
"efficiency": 1.0
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"summary": {
|
| 510 |
+
"maze": {
|
| 511 |
+
"episodes": 36,
|
| 512 |
+
"solve_rate": 1.0,
|
| 513 |
+
"mean_steps_when_solved": 93.19444444444444,
|
| 514 |
+
"mean_efficiency": 1.0,
|
| 515 |
+
"mean_collisions": 0.0,
|
| 516 |
+
"deadlock_rate": 0.0,
|
| 517 |
+
"mean_distinct_cells": 94.19444444444444,
|
| 518 |
+
"by_size": {
|
| 519 |
+
"11": {
|
| 520 |
+
"episodes": 12,
|
| 521 |
+
"solve_rate": 1.0,
|
| 522 |
+
"mean_steps": 27.25,
|
| 523 |
+
"mean_efficiency": 1.0,
|
| 524 |
+
"mean_shortest": 27.25
|
| 525 |
+
},
|
| 526 |
+
"21": {
|
| 527 |
+
"episodes": 12,
|
| 528 |
+
"solve_rate": 1.0,
|
| 529 |
+
"mean_steps": 90.41666666666667,
|
| 530 |
+
"mean_efficiency": 1.0,
|
| 531 |
+
"mean_shortest": 90.41666666666667
|
| 532 |
+
},
|
| 533 |
+
"31": {
|
| 534 |
+
"episodes": 12,
|
| 535 |
+
"solve_rate": 1.0,
|
| 536 |
+
"mean_steps": 161.91666666666666,
|
| 537 |
+
"mean_efficiency": 1.0,
|
| 538 |
+
"mean_shortest": 161.91666666666666
|
| 539 |
+
}
|
| 540 |
+
}
|
| 541 |
+
}
|
| 542 |
+
}
|
| 543 |
+
}
|
|
@@ -0,0 +1,543 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "random+memory",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"solved": true,
|
| 6 |
+
"steps": 32,
|
| 7 |
+
"collisions": 0,
|
| 8 |
+
"distinct_cells": 33,
|
| 9 |
+
"longest_stuck": 0,
|
| 10 |
+
"deadlocked": false,
|
| 11 |
+
"game": "maze",
|
| 12 |
+
"size": 11,
|
| 13 |
+
"topology": "corridor",
|
| 14 |
+
"seed": 901111,
|
| 15 |
+
"shortest": 32,
|
| 16 |
+
"efficiency": 1.0
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"solved": true,
|
| 20 |
+
"steps": 42,
|
| 21 |
+
"collisions": 0,
|
| 22 |
+
"distinct_cells": 43,
|
| 23 |
+
"longest_stuck": 0,
|
| 24 |
+
"deadlocked": false,
|
| 25 |
+
"game": "maze",
|
| 26 |
+
"size": 11,
|
| 27 |
+
"topology": "corridor",
|
| 28 |
+
"seed": 901148,
|
| 29 |
+
"shortest": 42,
|
| 30 |
+
"efficiency": 1.0
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"solved": true,
|
| 34 |
+
"steps": 120,
|
| 35 |
+
"collisions": 0,
|
| 36 |
+
"distinct_cells": 49,
|
| 37 |
+
"longest_stuck": 0,
|
| 38 |
+
"deadlocked": false,
|
| 39 |
+
"game": "maze",
|
| 40 |
+
"size": 11,
|
| 41 |
+
"topology": "corridor",
|
| 42 |
+
"seed": 901185,
|
| 43 |
+
"shortest": 40,
|
| 44 |
+
"efficiency": 0.3333333333333333
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"solved": true,
|
| 48 |
+
"steps": 44,
|
| 49 |
+
"collisions": 0,
|
| 50 |
+
"distinct_cells": 45,
|
| 51 |
+
"longest_stuck": 0,
|
| 52 |
+
"deadlocked": false,
|
| 53 |
+
"game": "maze",
|
| 54 |
+
"size": 11,
|
| 55 |
+
"topology": "tree",
|
| 56 |
+
"seed": 901111,
|
| 57 |
+
"shortest": 44,
|
| 58 |
+
"efficiency": 1.0
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"solved": true,
|
| 62 |
+
"steps": 34,
|
| 63 |
+
"collisions": 0,
|
| 64 |
+
"distinct_cells": 35,
|
| 65 |
+
"longest_stuck": 0,
|
| 66 |
+
"deadlocked": false,
|
| 67 |
+
"game": "maze",
|
| 68 |
+
"size": 11,
|
| 69 |
+
"topology": "tree",
|
| 70 |
+
"seed": 901148,
|
| 71 |
+
"shortest": 34,
|
| 72 |
+
"efficiency": 1.0
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"solved": true,
|
| 76 |
+
"steps": 74,
|
| 77 |
+
"collisions": 0,
|
| 78 |
+
"distinct_cells": 47,
|
| 79 |
+
"longest_stuck": 0,
|
| 80 |
+
"deadlocked": false,
|
| 81 |
+
"game": "maze",
|
| 82 |
+
"size": 11,
|
| 83 |
+
"topology": "tree",
|
| 84 |
+
"seed": 901185,
|
| 85 |
+
"shortest": 36,
|
| 86 |
+
"efficiency": 0.4864864864864865
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"solved": true,
|
| 90 |
+
"steps": 217,
|
| 91 |
+
"collisions": 0,
|
| 92 |
+
"distinct_cells": 48,
|
| 93 |
+
"longest_stuck": 0,
|
| 94 |
+
"deadlocked": false,
|
| 95 |
+
"game": "maze",
|
| 96 |
+
"size": 11,
|
| 97 |
+
"topology": "loops",
|
| 98 |
+
"seed": 901111,
|
| 99 |
+
"shortest": 15,
|
| 100 |
+
"efficiency": 0.06912442396313365
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"solved": false,
|
| 104 |
+
"steps": 484,
|
| 105 |
+
"collisions": 0,
|
| 106 |
+
"distinct_cells": 47,
|
| 107 |
+
"longest_stuck": 0,
|
| 108 |
+
"deadlocked": false,
|
| 109 |
+
"game": "maze",
|
| 110 |
+
"size": 11,
|
| 111 |
+
"topology": "loops",
|
| 112 |
+
"seed": 901148,
|
| 113 |
+
"shortest": 22,
|
| 114 |
+
"efficiency": 0.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"solved": true,
|
| 118 |
+
"steps": 26,
|
| 119 |
+
"collisions": 0,
|
| 120 |
+
"distinct_cells": 27,
|
| 121 |
+
"longest_stuck": 0,
|
| 122 |
+
"deadlocked": false,
|
| 123 |
+
"game": "maze",
|
| 124 |
+
"size": 11,
|
| 125 |
+
"topology": "loops",
|
| 126 |
+
"seed": 901185,
|
| 127 |
+
"shortest": 18,
|
| 128 |
+
"efficiency": 0.6923076923076923
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"solved": true,
|
| 132 |
+
"steps": 155,
|
| 133 |
+
"collisions": 0,
|
| 134 |
+
"distinct_cells": 58,
|
| 135 |
+
"longest_stuck": 0,
|
| 136 |
+
"deadlocked": false,
|
| 137 |
+
"game": "maze",
|
| 138 |
+
"size": 11,
|
| 139 |
+
"topology": "random_obstacle",
|
| 140 |
+
"seed": 901111,
|
| 141 |
+
"shortest": 15,
|
| 142 |
+
"efficiency": 0.0967741935483871
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"solved": true,
|
| 146 |
+
"steps": 28,
|
| 147 |
+
"collisions": 0,
|
| 148 |
+
"distinct_cells": 27,
|
| 149 |
+
"longest_stuck": 0,
|
| 150 |
+
"deadlocked": false,
|
| 151 |
+
"game": "maze",
|
| 152 |
+
"size": 11,
|
| 153 |
+
"topology": "random_obstacle",
|
| 154 |
+
"seed": 901148,
|
| 155 |
+
"shortest": 16,
|
| 156 |
+
"efficiency": 0.5714285714285714
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"solved": true,
|
| 160 |
+
"steps": 61,
|
| 161 |
+
"collisions": 0,
|
| 162 |
+
"distinct_cells": 30,
|
| 163 |
+
"longest_stuck": 0,
|
| 164 |
+
"deadlocked": false,
|
| 165 |
+
"game": "maze",
|
| 166 |
+
"size": 11,
|
| 167 |
+
"topology": "random_obstacle",
|
| 168 |
+
"seed": 901185,
|
| 169 |
+
"shortest": 13,
|
| 170 |
+
"efficiency": 0.21311475409836064
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"solved": true,
|
| 174 |
+
"steps": 666,
|
| 175 |
+
"collisions": 0,
|
| 176 |
+
"distinct_cells": 175,
|
| 177 |
+
"longest_stuck": 0,
|
| 178 |
+
"deadlocked": false,
|
| 179 |
+
"game": "maze",
|
| 180 |
+
"size": 21,
|
| 181 |
+
"topology": "corridor",
|
| 182 |
+
"seed": 902121,
|
| 183 |
+
"shortest": 142,
|
| 184 |
+
"efficiency": 0.2132132132132132
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"solved": true,
|
| 188 |
+
"steps": 164,
|
| 189 |
+
"collisions": 0,
|
| 190 |
+
"distinct_cells": 127,
|
| 191 |
+
"longest_stuck": 0,
|
| 192 |
+
"deadlocked": false,
|
| 193 |
+
"game": "maze",
|
| 194 |
+
"size": 21,
|
| 195 |
+
"topology": "corridor",
|
| 196 |
+
"seed": 902158,
|
| 197 |
+
"shortest": 116,
|
| 198 |
+
"efficiency": 0.7073170731707317
|
| 199 |
+
},
|
| 200 |
+
{
|
| 201 |
+
"solved": true,
|
| 202 |
+
"steps": 156,
|
| 203 |
+
"collisions": 0,
|
| 204 |
+
"distinct_cells": 157,
|
| 205 |
+
"longest_stuck": 0,
|
| 206 |
+
"deadlocked": false,
|
| 207 |
+
"game": "maze",
|
| 208 |
+
"size": 21,
|
| 209 |
+
"topology": "corridor",
|
| 210 |
+
"seed": 902195,
|
| 211 |
+
"shortest": 156,
|
| 212 |
+
"efficiency": 1.0
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"solved": false,
|
| 216 |
+
"steps": 1764,
|
| 217 |
+
"collisions": 0,
|
| 218 |
+
"distinct_cells": 133,
|
| 219 |
+
"longest_stuck": 0,
|
| 220 |
+
"deadlocked": false,
|
| 221 |
+
"game": "maze",
|
| 222 |
+
"size": 21,
|
| 223 |
+
"topology": "tree",
|
| 224 |
+
"seed": 902121,
|
| 225 |
+
"shortest": 118,
|
| 226 |
+
"efficiency": 0.0
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"solved": true,
|
| 230 |
+
"steps": 312,
|
| 231 |
+
"collisions": 0,
|
| 232 |
+
"distinct_cells": 189,
|
| 233 |
+
"longest_stuck": 0,
|
| 234 |
+
"deadlocked": false,
|
| 235 |
+
"game": "maze",
|
| 236 |
+
"size": 21,
|
| 237 |
+
"topology": "tree",
|
| 238 |
+
"seed": 902158,
|
| 239 |
+
"shortest": 172,
|
| 240 |
+
"efficiency": 0.5512820512820513
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"solved": true,
|
| 244 |
+
"steps": 1536,
|
| 245 |
+
"collisions": 0,
|
| 246 |
+
"distinct_cells": 193,
|
| 247 |
+
"longest_stuck": 0,
|
| 248 |
+
"deadlocked": false,
|
| 249 |
+
"game": "maze",
|
| 250 |
+
"size": 21,
|
| 251 |
+
"topology": "tree",
|
| 252 |
+
"seed": 902195,
|
| 253 |
+
"shortest": 140,
|
| 254 |
+
"efficiency": 0.09114583333333333
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"solved": true,
|
| 258 |
+
"steps": 1296,
|
| 259 |
+
"collisions": 0,
|
| 260 |
+
"distinct_cells": 211,
|
| 261 |
+
"longest_stuck": 0,
|
| 262 |
+
"deadlocked": false,
|
| 263 |
+
"game": "maze",
|
| 264 |
+
"size": 21,
|
| 265 |
+
"topology": "loops",
|
| 266 |
+
"seed": 902121,
|
| 267 |
+
"shortest": 46,
|
| 268 |
+
"efficiency": 0.035493827160493825
|
| 269 |
+
},
|
| 270 |
+
{
|
| 271 |
+
"solved": true,
|
| 272 |
+
"steps": 98,
|
| 273 |
+
"collisions": 0,
|
| 274 |
+
"distinct_cells": 79,
|
| 275 |
+
"longest_stuck": 0,
|
| 276 |
+
"deadlocked": false,
|
| 277 |
+
"game": "maze",
|
| 278 |
+
"size": 21,
|
| 279 |
+
"topology": "loops",
|
| 280 |
+
"seed": 902158,
|
| 281 |
+
"shortest": 40,
|
| 282 |
+
"efficiency": 0.40816326530612246
|
| 283 |
+
},
|
| 284 |
+
{
|
| 285 |
+
"solved": false,
|
| 286 |
+
"steps": 1764,
|
| 287 |
+
"collisions": 0,
|
| 288 |
+
"distinct_cells": 171,
|
| 289 |
+
"longest_stuck": 0,
|
| 290 |
+
"deadlocked": false,
|
| 291 |
+
"game": "maze",
|
| 292 |
+
"size": 21,
|
| 293 |
+
"topology": "loops",
|
| 294 |
+
"seed": 902195,
|
| 295 |
+
"shortest": 42,
|
| 296 |
+
"efficiency": 0.0
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"solved": true,
|
| 300 |
+
"steps": 1705,
|
| 301 |
+
"collisions": 0,
|
| 302 |
+
"distinct_cells": 200,
|
| 303 |
+
"longest_stuck": 0,
|
| 304 |
+
"deadlocked": false,
|
| 305 |
+
"game": "maze",
|
| 306 |
+
"size": 21,
|
| 307 |
+
"topology": "random_obstacle",
|
| 308 |
+
"seed": 902121,
|
| 309 |
+
"shortest": 41,
|
| 310 |
+
"efficiency": 0.02404692082111437
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"solved": true,
|
| 314 |
+
"steps": 928,
|
| 315 |
+
"collisions": 0,
|
| 316 |
+
"distinct_cells": 168,
|
| 317 |
+
"longest_stuck": 0,
|
| 318 |
+
"deadlocked": false,
|
| 319 |
+
"game": "maze",
|
| 320 |
+
"size": 21,
|
| 321 |
+
"topology": "random_obstacle",
|
| 322 |
+
"seed": 902158,
|
| 323 |
+
"shortest": 36,
|
| 324 |
+
"efficiency": 0.03879310344827586
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"solved": true,
|
| 328 |
+
"steps": 1218,
|
| 329 |
+
"collisions": 0,
|
| 330 |
+
"distinct_cells": 264,
|
| 331 |
+
"longest_stuck": 0,
|
| 332 |
+
"deadlocked": false,
|
| 333 |
+
"game": "maze",
|
| 334 |
+
"size": 21,
|
| 335 |
+
"topology": "random_obstacle",
|
| 336 |
+
"seed": 902195,
|
| 337 |
+
"shortest": 36,
|
| 338 |
+
"efficiency": 0.029556650246305417
|
| 339 |
+
},
|
| 340 |
+
{
|
| 341 |
+
"solved": true,
|
| 342 |
+
"steps": 744,
|
| 343 |
+
"collisions": 0,
|
| 344 |
+
"distinct_cells": 313,
|
| 345 |
+
"longest_stuck": 0,
|
| 346 |
+
"deadlocked": false,
|
| 347 |
+
"game": "maze",
|
| 348 |
+
"size": 31,
|
| 349 |
+
"topology": "corridor",
|
| 350 |
+
"seed": 903131,
|
| 351 |
+
"shortest": 264,
|
| 352 |
+
"efficiency": 0.3548387096774194
|
| 353 |
+
},
|
| 354 |
+
{
|
| 355 |
+
"solved": false,
|
| 356 |
+
"steps": 3844,
|
| 357 |
+
"collisions": 0,
|
| 358 |
+
"distinct_cells": 253,
|
| 359 |
+
"longest_stuck": 0,
|
| 360 |
+
"deadlocked": false,
|
| 361 |
+
"game": "maze",
|
| 362 |
+
"size": 31,
|
| 363 |
+
"topology": "corridor",
|
| 364 |
+
"seed": 903168,
|
| 365 |
+
"shortest": 194,
|
| 366 |
+
"efficiency": 0.0
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"solved": true,
|
| 370 |
+
"steps": 1488,
|
| 371 |
+
"collisions": 0,
|
| 372 |
+
"distinct_cells": 369,
|
| 373 |
+
"longest_stuck": 0,
|
| 374 |
+
"deadlocked": false,
|
| 375 |
+
"game": "maze",
|
| 376 |
+
"size": 31,
|
| 377 |
+
"topology": "corridor",
|
| 378 |
+
"seed": 903205,
|
| 379 |
+
"shortest": 286,
|
| 380 |
+
"efficiency": 0.1922043010752688
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"solved": true,
|
| 384 |
+
"steps": 2036,
|
| 385 |
+
"collisions": 0,
|
| 386 |
+
"distinct_cells": 375,
|
| 387 |
+
"longest_stuck": 0,
|
| 388 |
+
"deadlocked": false,
|
| 389 |
+
"game": "maze",
|
| 390 |
+
"size": 31,
|
| 391 |
+
"topology": "tree",
|
| 392 |
+
"seed": 903131,
|
| 393 |
+
"shortest": 268,
|
| 394 |
+
"efficiency": 0.13163064833005894
|
| 395 |
+
},
|
| 396 |
+
{
|
| 397 |
+
"solved": true,
|
| 398 |
+
"steps": 1320,
|
| 399 |
+
"collisions": 0,
|
| 400 |
+
"distinct_cells": 345,
|
| 401 |
+
"longest_stuck": 0,
|
| 402 |
+
"deadlocked": false,
|
| 403 |
+
"game": "maze",
|
| 404 |
+
"size": 31,
|
| 405 |
+
"topology": "tree",
|
| 406 |
+
"seed": 903168,
|
| 407 |
+
"shortest": 282,
|
| 408 |
+
"efficiency": 0.21363636363636362
|
| 409 |
+
},
|
| 410 |
+
{
|
| 411 |
+
"solved": false,
|
| 412 |
+
"steps": 3844,
|
| 413 |
+
"collisions": 0,
|
| 414 |
+
"distinct_cells": 335,
|
| 415 |
+
"longest_stuck": 0,
|
| 416 |
+
"deadlocked": false,
|
| 417 |
+
"game": "maze",
|
| 418 |
+
"size": 31,
|
| 419 |
+
"topology": "tree",
|
| 420 |
+
"seed": 903205,
|
| 421 |
+
"shortest": 252,
|
| 422 |
+
"efficiency": 0.0
|
| 423 |
+
},
|
| 424 |
+
{
|
| 425 |
+
"solved": true,
|
| 426 |
+
"steps": 1218,
|
| 427 |
+
"collisions": 0,
|
| 428 |
+
"distinct_cells": 344,
|
| 429 |
+
"longest_stuck": 0,
|
| 430 |
+
"deadlocked": false,
|
| 431 |
+
"game": "maze",
|
| 432 |
+
"size": 31,
|
| 433 |
+
"topology": "loops",
|
| 434 |
+
"seed": 903131,
|
| 435 |
+
"shortest": 66,
|
| 436 |
+
"efficiency": 0.054187192118226604
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"solved": true,
|
| 440 |
+
"steps": 644,
|
| 441 |
+
"collisions": 0,
|
| 442 |
+
"distinct_cells": 264,
|
| 443 |
+
"longest_stuck": 0,
|
| 444 |
+
"deadlocked": false,
|
| 445 |
+
"game": "maze",
|
| 446 |
+
"size": 31,
|
| 447 |
+
"topology": "loops",
|
| 448 |
+
"seed": 903168,
|
| 449 |
+
"shortest": 72,
|
| 450 |
+
"efficiency": 0.11180124223602485
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"solved": true,
|
| 454 |
+
"steps": 430,
|
| 455 |
+
"collisions": 0,
|
| 456 |
+
"distinct_cells": 125,
|
| 457 |
+
"longest_stuck": 0,
|
| 458 |
+
"deadlocked": false,
|
| 459 |
+
"game": "maze",
|
| 460 |
+
"size": 31,
|
| 461 |
+
"topology": "loops",
|
| 462 |
+
"seed": 903205,
|
| 463 |
+
"shortest": 88,
|
| 464 |
+
"efficiency": 0.20465116279069767
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"solved": false,
|
| 468 |
+
"steps": 3844,
|
| 469 |
+
"collisions": 0,
|
| 470 |
+
"distinct_cells": 485,
|
| 471 |
+
"longest_stuck": 0,
|
| 472 |
+
"deadlocked": false,
|
| 473 |
+
"game": "maze",
|
| 474 |
+
"size": 31,
|
| 475 |
+
"topology": "random_obstacle",
|
| 476 |
+
"seed": 903131,
|
| 477 |
+
"shortest": 57,
|
| 478 |
+
"efficiency": 0.0
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"solved": true,
|
| 482 |
+
"steps": 2154,
|
| 483 |
+
"collisions": 0,
|
| 484 |
+
"distinct_cells": 446,
|
| 485 |
+
"longest_stuck": 0,
|
| 486 |
+
"deadlocked": false,
|
| 487 |
+
"game": "maze",
|
| 488 |
+
"size": 31,
|
| 489 |
+
"topology": "random_obstacle",
|
| 490 |
+
"seed": 903168,
|
| 491 |
+
"shortest": 56,
|
| 492 |
+
"efficiency": 0.025998142989786442
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"solved": true,
|
| 496 |
+
"steps": 1746,
|
| 497 |
+
"collisions": 0,
|
| 498 |
+
"distinct_cells": 577,
|
| 499 |
+
"longest_stuck": 0,
|
| 500 |
+
"deadlocked": false,
|
| 501 |
+
"game": "maze",
|
| 502 |
+
"size": 31,
|
| 503 |
+
"topology": "random_obstacle",
|
| 504 |
+
"seed": 903205,
|
| 505 |
+
"shortest": 58,
|
| 506 |
+
"efficiency": 0.033218785796105384
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"summary": {
|
| 510 |
+
"maze": {
|
| 511 |
+
"episodes": 36,
|
| 512 |
+
"solve_rate": 0.8333333333333334,
|
| 513 |
+
"mean_steps_when_solved": 689.7333333333333,
|
| 514 |
+
"mean_efficiency": 0.3627915980599186,
|
| 515 |
+
"mean_collisions": 0.0,
|
| 516 |
+
"deadlock_rate": 0.0,
|
| 517 |
+
"mean_distinct_cells": 188.52777777777777,
|
| 518 |
+
"by_size": {
|
| 519 |
+
"11": {
|
| 520 |
+
"episodes": 12,
|
| 521 |
+
"solve_rate": 0.9166666666666666,
|
| 522 |
+
"mean_steps": 75.72727272727273,
|
| 523 |
+
"mean_efficiency": 0.5875063141059967,
|
| 524 |
+
"mean_shortest": 27.25
|
| 525 |
+
},
|
| 526 |
+
"21": {
|
| 527 |
+
"episodes": 12,
|
| 528 |
+
"solve_rate": 0.8333333333333334,
|
| 529 |
+
"mean_steps": 807.9,
|
| 530 |
+
"mean_efficiency": 0.30990119379816417,
|
| 531 |
+
"mean_shortest": 90.41666666666667
|
| 532 |
+
},
|
| 533 |
+
"31": {
|
| 534 |
+
"episodes": 12,
|
| 535 |
+
"solve_rate": 0.75,
|
| 536 |
+
"mean_steps": 1308.888888888889,
|
| 537 |
+
"mean_efficiency": 0.14690739429443908,
|
| 538 |
+
"mean_shortest": 161.91666666666666
|
| 539 |
+
}
|
| 540 |
+
}
|
| 541 |
+
}
|
| 542 |
+
}
|
| 543 |
+
}
|
|
@@ -0,0 +1,543 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "random",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"solved": false,
|
| 6 |
+
"steps": 484,
|
| 7 |
+
"collisions": 0,
|
| 8 |
+
"distinct_cells": 14,
|
| 9 |
+
"longest_stuck": 0,
|
| 10 |
+
"deadlocked": false,
|
| 11 |
+
"game": "maze",
|
| 12 |
+
"size": 11,
|
| 13 |
+
"topology": "corridor",
|
| 14 |
+
"seed": 901111,
|
| 15 |
+
"shortest": 32,
|
| 16 |
+
"efficiency": 0.0
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"solved": false,
|
| 20 |
+
"steps": 484,
|
| 21 |
+
"collisions": 0,
|
| 22 |
+
"distinct_cells": 25,
|
| 23 |
+
"longest_stuck": 0,
|
| 24 |
+
"deadlocked": false,
|
| 25 |
+
"game": "maze",
|
| 26 |
+
"size": 11,
|
| 27 |
+
"topology": "corridor",
|
| 28 |
+
"seed": 901148,
|
| 29 |
+
"shortest": 42,
|
| 30 |
+
"efficiency": 0.0
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"solved": false,
|
| 34 |
+
"steps": 484,
|
| 35 |
+
"collisions": 0,
|
| 36 |
+
"distinct_cells": 42,
|
| 37 |
+
"longest_stuck": 0,
|
| 38 |
+
"deadlocked": false,
|
| 39 |
+
"game": "maze",
|
| 40 |
+
"size": 11,
|
| 41 |
+
"topology": "corridor",
|
| 42 |
+
"seed": 901185,
|
| 43 |
+
"shortest": 40,
|
| 44 |
+
"efficiency": 0.0
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"solved": false,
|
| 48 |
+
"steps": 484,
|
| 49 |
+
"collisions": 0,
|
| 50 |
+
"distinct_cells": 26,
|
| 51 |
+
"longest_stuck": 0,
|
| 52 |
+
"deadlocked": false,
|
| 53 |
+
"game": "maze",
|
| 54 |
+
"size": 11,
|
| 55 |
+
"topology": "tree",
|
| 56 |
+
"seed": 901111,
|
| 57 |
+
"shortest": 44,
|
| 58 |
+
"efficiency": 0.0
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"solved": true,
|
| 62 |
+
"steps": 444,
|
| 63 |
+
"collisions": 0,
|
| 64 |
+
"distinct_cells": 35,
|
| 65 |
+
"longest_stuck": 0,
|
| 66 |
+
"deadlocked": false,
|
| 67 |
+
"game": "maze",
|
| 68 |
+
"size": 11,
|
| 69 |
+
"topology": "tree",
|
| 70 |
+
"seed": 901148,
|
| 71 |
+
"shortest": 34,
|
| 72 |
+
"efficiency": 0.07657657657657657
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"solved": false,
|
| 76 |
+
"steps": 484,
|
| 77 |
+
"collisions": 0,
|
| 78 |
+
"distinct_cells": 23,
|
| 79 |
+
"longest_stuck": 0,
|
| 80 |
+
"deadlocked": false,
|
| 81 |
+
"game": "maze",
|
| 82 |
+
"size": 11,
|
| 83 |
+
"topology": "tree",
|
| 84 |
+
"seed": 901185,
|
| 85 |
+
"shortest": 36,
|
| 86 |
+
"efficiency": 0.0
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"solved": false,
|
| 90 |
+
"steps": 484,
|
| 91 |
+
"collisions": 0,
|
| 92 |
+
"distinct_cells": 47,
|
| 93 |
+
"longest_stuck": 0,
|
| 94 |
+
"deadlocked": false,
|
| 95 |
+
"game": "maze",
|
| 96 |
+
"size": 11,
|
| 97 |
+
"topology": "loops",
|
| 98 |
+
"seed": 901111,
|
| 99 |
+
"shortest": 15,
|
| 100 |
+
"efficiency": 0.0
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"solved": false,
|
| 104 |
+
"steps": 484,
|
| 105 |
+
"collisions": 0,
|
| 106 |
+
"distinct_cells": 36,
|
| 107 |
+
"longest_stuck": 0,
|
| 108 |
+
"deadlocked": false,
|
| 109 |
+
"game": "maze",
|
| 110 |
+
"size": 11,
|
| 111 |
+
"topology": "loops",
|
| 112 |
+
"seed": 901148,
|
| 113 |
+
"shortest": 22,
|
| 114 |
+
"efficiency": 0.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"solved": true,
|
| 118 |
+
"steps": 426,
|
| 119 |
+
"collisions": 0,
|
| 120 |
+
"distinct_cells": 48,
|
| 121 |
+
"longest_stuck": 0,
|
| 122 |
+
"deadlocked": false,
|
| 123 |
+
"game": "maze",
|
| 124 |
+
"size": 11,
|
| 125 |
+
"topology": "loops",
|
| 126 |
+
"seed": 901185,
|
| 127 |
+
"shortest": 18,
|
| 128 |
+
"efficiency": 0.04225352112676056
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"solved": false,
|
| 132 |
+
"steps": 484,
|
| 133 |
+
"collisions": 0,
|
| 134 |
+
"distinct_cells": 56,
|
| 135 |
+
"longest_stuck": 0,
|
| 136 |
+
"deadlocked": false,
|
| 137 |
+
"game": "maze",
|
| 138 |
+
"size": 11,
|
| 139 |
+
"topology": "random_obstacle",
|
| 140 |
+
"seed": 901111,
|
| 141 |
+
"shortest": 15,
|
| 142 |
+
"efficiency": 0.0
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"solved": false,
|
| 146 |
+
"steps": 484,
|
| 147 |
+
"collisions": 0,
|
| 148 |
+
"distinct_cells": 53,
|
| 149 |
+
"longest_stuck": 0,
|
| 150 |
+
"deadlocked": false,
|
| 151 |
+
"game": "maze",
|
| 152 |
+
"size": 11,
|
| 153 |
+
"topology": "random_obstacle",
|
| 154 |
+
"seed": 901148,
|
| 155 |
+
"shortest": 16,
|
| 156 |
+
"efficiency": 0.0
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"solved": false,
|
| 160 |
+
"steps": 484,
|
| 161 |
+
"collisions": 0,
|
| 162 |
+
"distinct_cells": 30,
|
| 163 |
+
"longest_stuck": 0,
|
| 164 |
+
"deadlocked": false,
|
| 165 |
+
"game": "maze",
|
| 166 |
+
"size": 11,
|
| 167 |
+
"topology": "random_obstacle",
|
| 168 |
+
"seed": 901185,
|
| 169 |
+
"shortest": 13,
|
| 170 |
+
"efficiency": 0.0
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"solved": false,
|
| 174 |
+
"steps": 1764,
|
| 175 |
+
"collisions": 0,
|
| 176 |
+
"distinct_cells": 23,
|
| 177 |
+
"longest_stuck": 0,
|
| 178 |
+
"deadlocked": false,
|
| 179 |
+
"game": "maze",
|
| 180 |
+
"size": 21,
|
| 181 |
+
"topology": "corridor",
|
| 182 |
+
"seed": 902121,
|
| 183 |
+
"shortest": 142,
|
| 184 |
+
"efficiency": 0.0
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"solved": false,
|
| 188 |
+
"steps": 1764,
|
| 189 |
+
"collisions": 0,
|
| 190 |
+
"distinct_cells": 57,
|
| 191 |
+
"longest_stuck": 0,
|
| 192 |
+
"deadlocked": false,
|
| 193 |
+
"game": "maze",
|
| 194 |
+
"size": 21,
|
| 195 |
+
"topology": "corridor",
|
| 196 |
+
"seed": 902158,
|
| 197 |
+
"shortest": 116,
|
| 198 |
+
"efficiency": 0.0
|
| 199 |
+
},
|
| 200 |
+
{
|
| 201 |
+
"solved": false,
|
| 202 |
+
"steps": 1764,
|
| 203 |
+
"collisions": 0,
|
| 204 |
+
"distinct_cells": 45,
|
| 205 |
+
"longest_stuck": 0,
|
| 206 |
+
"deadlocked": false,
|
| 207 |
+
"game": "maze",
|
| 208 |
+
"size": 21,
|
| 209 |
+
"topology": "corridor",
|
| 210 |
+
"seed": 902195,
|
| 211 |
+
"shortest": 156,
|
| 212 |
+
"efficiency": 0.0
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"solved": false,
|
| 216 |
+
"steps": 1764,
|
| 217 |
+
"collisions": 0,
|
| 218 |
+
"distinct_cells": 116,
|
| 219 |
+
"longest_stuck": 0,
|
| 220 |
+
"deadlocked": false,
|
| 221 |
+
"game": "maze",
|
| 222 |
+
"size": 21,
|
| 223 |
+
"topology": "tree",
|
| 224 |
+
"seed": 902121,
|
| 225 |
+
"shortest": 118,
|
| 226 |
+
"efficiency": 0.0
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"solved": false,
|
| 230 |
+
"steps": 1764,
|
| 231 |
+
"collisions": 0,
|
| 232 |
+
"distinct_cells": 28,
|
| 233 |
+
"longest_stuck": 0,
|
| 234 |
+
"deadlocked": false,
|
| 235 |
+
"game": "maze",
|
| 236 |
+
"size": 21,
|
| 237 |
+
"topology": "tree",
|
| 238 |
+
"seed": 902158,
|
| 239 |
+
"shortest": 172,
|
| 240 |
+
"efficiency": 0.0
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"solved": false,
|
| 244 |
+
"steps": 1764,
|
| 245 |
+
"collisions": 0,
|
| 246 |
+
"distinct_cells": 84,
|
| 247 |
+
"longest_stuck": 0,
|
| 248 |
+
"deadlocked": false,
|
| 249 |
+
"game": "maze",
|
| 250 |
+
"size": 21,
|
| 251 |
+
"topology": "tree",
|
| 252 |
+
"seed": 902195,
|
| 253 |
+
"shortest": 140,
|
| 254 |
+
"efficiency": 0.0
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"solved": false,
|
| 258 |
+
"steps": 1764,
|
| 259 |
+
"collisions": 0,
|
| 260 |
+
"distinct_cells": 67,
|
| 261 |
+
"longest_stuck": 0,
|
| 262 |
+
"deadlocked": false,
|
| 263 |
+
"game": "maze",
|
| 264 |
+
"size": 21,
|
| 265 |
+
"topology": "loops",
|
| 266 |
+
"seed": 902121,
|
| 267 |
+
"shortest": 46,
|
| 268 |
+
"efficiency": 0.0
|
| 269 |
+
},
|
| 270 |
+
{
|
| 271 |
+
"solved": false,
|
| 272 |
+
"steps": 1764,
|
| 273 |
+
"collisions": 0,
|
| 274 |
+
"distinct_cells": 105,
|
| 275 |
+
"longest_stuck": 0,
|
| 276 |
+
"deadlocked": false,
|
| 277 |
+
"game": "maze",
|
| 278 |
+
"size": 21,
|
| 279 |
+
"topology": "loops",
|
| 280 |
+
"seed": 902158,
|
| 281 |
+
"shortest": 40,
|
| 282 |
+
"efficiency": 0.0
|
| 283 |
+
},
|
| 284 |
+
{
|
| 285 |
+
"solved": false,
|
| 286 |
+
"steps": 1764,
|
| 287 |
+
"collisions": 0,
|
| 288 |
+
"distinct_cells": 121,
|
| 289 |
+
"longest_stuck": 0,
|
| 290 |
+
"deadlocked": false,
|
| 291 |
+
"game": "maze",
|
| 292 |
+
"size": 21,
|
| 293 |
+
"topology": "loops",
|
| 294 |
+
"seed": 902195,
|
| 295 |
+
"shortest": 42,
|
| 296 |
+
"efficiency": 0.0
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"solved": false,
|
| 300 |
+
"steps": 1764,
|
| 301 |
+
"collisions": 0,
|
| 302 |
+
"distinct_cells": 121,
|
| 303 |
+
"longest_stuck": 0,
|
| 304 |
+
"deadlocked": false,
|
| 305 |
+
"game": "maze",
|
| 306 |
+
"size": 21,
|
| 307 |
+
"topology": "random_obstacle",
|
| 308 |
+
"seed": 902121,
|
| 309 |
+
"shortest": 41,
|
| 310 |
+
"efficiency": 0.0
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"solved": true,
|
| 314 |
+
"steps": 494,
|
| 315 |
+
"collisions": 0,
|
| 316 |
+
"distinct_cells": 130,
|
| 317 |
+
"longest_stuck": 0,
|
| 318 |
+
"deadlocked": false,
|
| 319 |
+
"game": "maze",
|
| 320 |
+
"size": 21,
|
| 321 |
+
"topology": "random_obstacle",
|
| 322 |
+
"seed": 902158,
|
| 323 |
+
"shortest": 36,
|
| 324 |
+
"efficiency": 0.0728744939271255
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"solved": true,
|
| 328 |
+
"steps": 894,
|
| 329 |
+
"collisions": 0,
|
| 330 |
+
"distinct_cells": 152,
|
| 331 |
+
"longest_stuck": 0,
|
| 332 |
+
"deadlocked": false,
|
| 333 |
+
"game": "maze",
|
| 334 |
+
"size": 21,
|
| 335 |
+
"topology": "random_obstacle",
|
| 336 |
+
"seed": 902195,
|
| 337 |
+
"shortest": 36,
|
| 338 |
+
"efficiency": 0.040268456375838924
|
| 339 |
+
},
|
| 340 |
+
{
|
| 341 |
+
"solved": false,
|
| 342 |
+
"steps": 3844,
|
| 343 |
+
"collisions": 0,
|
| 344 |
+
"distinct_cells": 48,
|
| 345 |
+
"longest_stuck": 0,
|
| 346 |
+
"deadlocked": false,
|
| 347 |
+
"game": "maze",
|
| 348 |
+
"size": 31,
|
| 349 |
+
"topology": "corridor",
|
| 350 |
+
"seed": 903131,
|
| 351 |
+
"shortest": 264,
|
| 352 |
+
"efficiency": 0.0
|
| 353 |
+
},
|
| 354 |
+
{
|
| 355 |
+
"solved": false,
|
| 356 |
+
"steps": 3844,
|
| 357 |
+
"collisions": 0,
|
| 358 |
+
"distinct_cells": 73,
|
| 359 |
+
"longest_stuck": 0,
|
| 360 |
+
"deadlocked": false,
|
| 361 |
+
"game": "maze",
|
| 362 |
+
"size": 31,
|
| 363 |
+
"topology": "corridor",
|
| 364 |
+
"seed": 903168,
|
| 365 |
+
"shortest": 194,
|
| 366 |
+
"efficiency": 0.0
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"solved": false,
|
| 370 |
+
"steps": 3844,
|
| 371 |
+
"collisions": 0,
|
| 372 |
+
"distinct_cells": 85,
|
| 373 |
+
"longest_stuck": 0,
|
| 374 |
+
"deadlocked": false,
|
| 375 |
+
"game": "maze",
|
| 376 |
+
"size": 31,
|
| 377 |
+
"topology": "corridor",
|
| 378 |
+
"seed": 903205,
|
| 379 |
+
"shortest": 286,
|
| 380 |
+
"efficiency": 0.0
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"solved": false,
|
| 384 |
+
"steps": 3844,
|
| 385 |
+
"collisions": 0,
|
| 386 |
+
"distinct_cells": 62,
|
| 387 |
+
"longest_stuck": 0,
|
| 388 |
+
"deadlocked": false,
|
| 389 |
+
"game": "maze",
|
| 390 |
+
"size": 31,
|
| 391 |
+
"topology": "tree",
|
| 392 |
+
"seed": 903131,
|
| 393 |
+
"shortest": 268,
|
| 394 |
+
"efficiency": 0.0
|
| 395 |
+
},
|
| 396 |
+
{
|
| 397 |
+
"solved": false,
|
| 398 |
+
"steps": 3844,
|
| 399 |
+
"collisions": 0,
|
| 400 |
+
"distinct_cells": 58,
|
| 401 |
+
"longest_stuck": 0,
|
| 402 |
+
"deadlocked": false,
|
| 403 |
+
"game": "maze",
|
| 404 |
+
"size": 31,
|
| 405 |
+
"topology": "tree",
|
| 406 |
+
"seed": 903168,
|
| 407 |
+
"shortest": 282,
|
| 408 |
+
"efficiency": 0.0
|
| 409 |
+
},
|
| 410 |
+
{
|
| 411 |
+
"solved": false,
|
| 412 |
+
"steps": 3844,
|
| 413 |
+
"collisions": 0,
|
| 414 |
+
"distinct_cells": 75,
|
| 415 |
+
"longest_stuck": 0,
|
| 416 |
+
"deadlocked": false,
|
| 417 |
+
"game": "maze",
|
| 418 |
+
"size": 31,
|
| 419 |
+
"topology": "tree",
|
| 420 |
+
"seed": 903205,
|
| 421 |
+
"shortest": 252,
|
| 422 |
+
"efficiency": 0.0
|
| 423 |
+
},
|
| 424 |
+
{
|
| 425 |
+
"solved": false,
|
| 426 |
+
"steps": 3844,
|
| 427 |
+
"collisions": 0,
|
| 428 |
+
"distinct_cells": 376,
|
| 429 |
+
"longest_stuck": 0,
|
| 430 |
+
"deadlocked": false,
|
| 431 |
+
"game": "maze",
|
| 432 |
+
"size": 31,
|
| 433 |
+
"topology": "loops",
|
| 434 |
+
"seed": 903131,
|
| 435 |
+
"shortest": 66,
|
| 436 |
+
"efficiency": 0.0
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"solved": false,
|
| 440 |
+
"steps": 3844,
|
| 441 |
+
"collisions": 0,
|
| 442 |
+
"distinct_cells": 166,
|
| 443 |
+
"longest_stuck": 0,
|
| 444 |
+
"deadlocked": false,
|
| 445 |
+
"game": "maze",
|
| 446 |
+
"size": 31,
|
| 447 |
+
"topology": "loops",
|
| 448 |
+
"seed": 903168,
|
| 449 |
+
"shortest": 72,
|
| 450 |
+
"efficiency": 0.0
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"solved": false,
|
| 454 |
+
"steps": 3844,
|
| 455 |
+
"collisions": 0,
|
| 456 |
+
"distinct_cells": 178,
|
| 457 |
+
"longest_stuck": 0,
|
| 458 |
+
"deadlocked": false,
|
| 459 |
+
"game": "maze",
|
| 460 |
+
"size": 31,
|
| 461 |
+
"topology": "loops",
|
| 462 |
+
"seed": 903205,
|
| 463 |
+
"shortest": 88,
|
| 464 |
+
"efficiency": 0.0
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"solved": false,
|
| 468 |
+
"steps": 3844,
|
| 469 |
+
"collisions": 0,
|
| 470 |
+
"distinct_cells": 246,
|
| 471 |
+
"longest_stuck": 0,
|
| 472 |
+
"deadlocked": false,
|
| 473 |
+
"game": "maze",
|
| 474 |
+
"size": 31,
|
| 475 |
+
"topology": "random_obstacle",
|
| 476 |
+
"seed": 903131,
|
| 477 |
+
"shortest": 57,
|
| 478 |
+
"efficiency": 0.0
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"solved": false,
|
| 482 |
+
"steps": 3844,
|
| 483 |
+
"collisions": 0,
|
| 484 |
+
"distinct_cells": 315,
|
| 485 |
+
"longest_stuck": 0,
|
| 486 |
+
"deadlocked": false,
|
| 487 |
+
"game": "maze",
|
| 488 |
+
"size": 31,
|
| 489 |
+
"topology": "random_obstacle",
|
| 490 |
+
"seed": 903168,
|
| 491 |
+
"shortest": 56,
|
| 492 |
+
"efficiency": 0.0
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"solved": true,
|
| 496 |
+
"steps": 544,
|
| 497 |
+
"collisions": 0,
|
| 498 |
+
"distinct_cells": 155,
|
| 499 |
+
"longest_stuck": 0,
|
| 500 |
+
"deadlocked": false,
|
| 501 |
+
"game": "maze",
|
| 502 |
+
"size": 31,
|
| 503 |
+
"topology": "random_obstacle",
|
| 504 |
+
"seed": 903205,
|
| 505 |
+
"shortest": 58,
|
| 506 |
+
"efficiency": 0.10661764705882353
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"summary": {
|
| 510 |
+
"maze": {
|
| 511 |
+
"episodes": 36,
|
| 512 |
+
"solve_rate": 0.1388888888888889,
|
| 513 |
+
"mean_steps_when_solved": 560.4,
|
| 514 |
+
"mean_efficiency": 0.06771813901302501,
|
| 515 |
+
"mean_collisions": 0.0,
|
| 516 |
+
"deadlock_rate": 0.0,
|
| 517 |
+
"mean_distinct_cells": 92.25,
|
| 518 |
+
"by_size": {
|
| 519 |
+
"11": {
|
| 520 |
+
"episodes": 12,
|
| 521 |
+
"solve_rate": 0.16666666666666666,
|
| 522 |
+
"mean_steps": 435.0,
|
| 523 |
+
"mean_efficiency": 0.05941504885166857,
|
| 524 |
+
"mean_shortest": 27.25
|
| 525 |
+
},
|
| 526 |
+
"21": {
|
| 527 |
+
"episodes": 12,
|
| 528 |
+
"solve_rate": 0.16666666666666666,
|
| 529 |
+
"mean_steps": 694.0,
|
| 530 |
+
"mean_efficiency": 0.05657147515148221,
|
| 531 |
+
"mean_shortest": 90.41666666666667
|
| 532 |
+
},
|
| 533 |
+
"31": {
|
| 534 |
+
"episodes": 12,
|
| 535 |
+
"solve_rate": 0.08333333333333333,
|
| 536 |
+
"mean_steps": 544.0,
|
| 537 |
+
"mean_efficiency": 0.10661764705882353,
|
| 538 |
+
"mean_shortest": 161.91666666666666
|
| 539 |
+
}
|
| 540 |
+
}
|
| 541 |
+
}
|
| 542 |
+
}
|
| 543 |
+
}
|
|
@@ -0,0 +1,543 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "reference",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"solved": true,
|
| 6 |
+
"steps": 32,
|
| 7 |
+
"collisions": 0,
|
| 8 |
+
"distinct_cells": 33,
|
| 9 |
+
"longest_stuck": 0,
|
| 10 |
+
"deadlocked": false,
|
| 11 |
+
"game": "maze",
|
| 12 |
+
"size": 11,
|
| 13 |
+
"topology": "corridor",
|
| 14 |
+
"seed": 901111,
|
| 15 |
+
"shortest": 32,
|
| 16 |
+
"efficiency": 1.0
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"solved": true,
|
| 20 |
+
"steps": 42,
|
| 21 |
+
"collisions": 0,
|
| 22 |
+
"distinct_cells": 43,
|
| 23 |
+
"longest_stuck": 0,
|
| 24 |
+
"deadlocked": false,
|
| 25 |
+
"game": "maze",
|
| 26 |
+
"size": 11,
|
| 27 |
+
"topology": "corridor",
|
| 28 |
+
"seed": 901148,
|
| 29 |
+
"shortest": 42,
|
| 30 |
+
"efficiency": 1.0
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"solved": true,
|
| 34 |
+
"steps": 40,
|
| 35 |
+
"collisions": 0,
|
| 36 |
+
"distinct_cells": 41,
|
| 37 |
+
"longest_stuck": 0,
|
| 38 |
+
"deadlocked": false,
|
| 39 |
+
"game": "maze",
|
| 40 |
+
"size": 11,
|
| 41 |
+
"topology": "corridor",
|
| 42 |
+
"seed": 901185,
|
| 43 |
+
"shortest": 40,
|
| 44 |
+
"efficiency": 1.0
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"solved": true,
|
| 48 |
+
"steps": 44,
|
| 49 |
+
"collisions": 0,
|
| 50 |
+
"distinct_cells": 45,
|
| 51 |
+
"longest_stuck": 0,
|
| 52 |
+
"deadlocked": false,
|
| 53 |
+
"game": "maze",
|
| 54 |
+
"size": 11,
|
| 55 |
+
"topology": "tree",
|
| 56 |
+
"seed": 901111,
|
| 57 |
+
"shortest": 44,
|
| 58 |
+
"efficiency": 1.0
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"solved": true,
|
| 62 |
+
"steps": 34,
|
| 63 |
+
"collisions": 0,
|
| 64 |
+
"distinct_cells": 35,
|
| 65 |
+
"longest_stuck": 0,
|
| 66 |
+
"deadlocked": false,
|
| 67 |
+
"game": "maze",
|
| 68 |
+
"size": 11,
|
| 69 |
+
"topology": "tree",
|
| 70 |
+
"seed": 901148,
|
| 71 |
+
"shortest": 34,
|
| 72 |
+
"efficiency": 1.0
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"solved": true,
|
| 76 |
+
"steps": 36,
|
| 77 |
+
"collisions": 0,
|
| 78 |
+
"distinct_cells": 37,
|
| 79 |
+
"longest_stuck": 0,
|
| 80 |
+
"deadlocked": false,
|
| 81 |
+
"game": "maze",
|
| 82 |
+
"size": 11,
|
| 83 |
+
"topology": "tree",
|
| 84 |
+
"seed": 901185,
|
| 85 |
+
"shortest": 36,
|
| 86 |
+
"efficiency": 1.0
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"solved": true,
|
| 90 |
+
"steps": 15,
|
| 91 |
+
"collisions": 0,
|
| 92 |
+
"distinct_cells": 16,
|
| 93 |
+
"longest_stuck": 0,
|
| 94 |
+
"deadlocked": false,
|
| 95 |
+
"game": "maze",
|
| 96 |
+
"size": 11,
|
| 97 |
+
"topology": "loops",
|
| 98 |
+
"seed": 901111,
|
| 99 |
+
"shortest": 15,
|
| 100 |
+
"efficiency": 1.0
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"solved": true,
|
| 104 |
+
"steps": 22,
|
| 105 |
+
"collisions": 0,
|
| 106 |
+
"distinct_cells": 23,
|
| 107 |
+
"longest_stuck": 0,
|
| 108 |
+
"deadlocked": false,
|
| 109 |
+
"game": "maze",
|
| 110 |
+
"size": 11,
|
| 111 |
+
"topology": "loops",
|
| 112 |
+
"seed": 901148,
|
| 113 |
+
"shortest": 22,
|
| 114 |
+
"efficiency": 1.0
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"solved": true,
|
| 118 |
+
"steps": 18,
|
| 119 |
+
"collisions": 0,
|
| 120 |
+
"distinct_cells": 19,
|
| 121 |
+
"longest_stuck": 0,
|
| 122 |
+
"deadlocked": false,
|
| 123 |
+
"game": "maze",
|
| 124 |
+
"size": 11,
|
| 125 |
+
"topology": "loops",
|
| 126 |
+
"seed": 901185,
|
| 127 |
+
"shortest": 18,
|
| 128 |
+
"efficiency": 1.0
|
| 129 |
+
},
|
| 130 |
+
{
|
| 131 |
+
"solved": true,
|
| 132 |
+
"steps": 15,
|
| 133 |
+
"collisions": 0,
|
| 134 |
+
"distinct_cells": 16,
|
| 135 |
+
"longest_stuck": 0,
|
| 136 |
+
"deadlocked": false,
|
| 137 |
+
"game": "maze",
|
| 138 |
+
"size": 11,
|
| 139 |
+
"topology": "random_obstacle",
|
| 140 |
+
"seed": 901111,
|
| 141 |
+
"shortest": 15,
|
| 142 |
+
"efficiency": 1.0
|
| 143 |
+
},
|
| 144 |
+
{
|
| 145 |
+
"solved": true,
|
| 146 |
+
"steps": 16,
|
| 147 |
+
"collisions": 0,
|
| 148 |
+
"distinct_cells": 17,
|
| 149 |
+
"longest_stuck": 0,
|
| 150 |
+
"deadlocked": false,
|
| 151 |
+
"game": "maze",
|
| 152 |
+
"size": 11,
|
| 153 |
+
"topology": "random_obstacle",
|
| 154 |
+
"seed": 901148,
|
| 155 |
+
"shortest": 16,
|
| 156 |
+
"efficiency": 1.0
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"solved": true,
|
| 160 |
+
"steps": 13,
|
| 161 |
+
"collisions": 0,
|
| 162 |
+
"distinct_cells": 14,
|
| 163 |
+
"longest_stuck": 0,
|
| 164 |
+
"deadlocked": false,
|
| 165 |
+
"game": "maze",
|
| 166 |
+
"size": 11,
|
| 167 |
+
"topology": "random_obstacle",
|
| 168 |
+
"seed": 901185,
|
| 169 |
+
"shortest": 13,
|
| 170 |
+
"efficiency": 1.0
|
| 171 |
+
},
|
| 172 |
+
{
|
| 173 |
+
"solved": true,
|
| 174 |
+
"steps": 142,
|
| 175 |
+
"collisions": 0,
|
| 176 |
+
"distinct_cells": 143,
|
| 177 |
+
"longest_stuck": 0,
|
| 178 |
+
"deadlocked": false,
|
| 179 |
+
"game": "maze",
|
| 180 |
+
"size": 21,
|
| 181 |
+
"topology": "corridor",
|
| 182 |
+
"seed": 902121,
|
| 183 |
+
"shortest": 142,
|
| 184 |
+
"efficiency": 1.0
|
| 185 |
+
},
|
| 186 |
+
{
|
| 187 |
+
"solved": true,
|
| 188 |
+
"steps": 116,
|
| 189 |
+
"collisions": 0,
|
| 190 |
+
"distinct_cells": 117,
|
| 191 |
+
"longest_stuck": 0,
|
| 192 |
+
"deadlocked": false,
|
| 193 |
+
"game": "maze",
|
| 194 |
+
"size": 21,
|
| 195 |
+
"topology": "corridor",
|
| 196 |
+
"seed": 902158,
|
| 197 |
+
"shortest": 116,
|
| 198 |
+
"efficiency": 1.0
|
| 199 |
+
},
|
| 200 |
+
{
|
| 201 |
+
"solved": true,
|
| 202 |
+
"steps": 156,
|
| 203 |
+
"collisions": 0,
|
| 204 |
+
"distinct_cells": 157,
|
| 205 |
+
"longest_stuck": 0,
|
| 206 |
+
"deadlocked": false,
|
| 207 |
+
"game": "maze",
|
| 208 |
+
"size": 21,
|
| 209 |
+
"topology": "corridor",
|
| 210 |
+
"seed": 902195,
|
| 211 |
+
"shortest": 156,
|
| 212 |
+
"efficiency": 1.0
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"solved": true,
|
| 216 |
+
"steps": 118,
|
| 217 |
+
"collisions": 0,
|
| 218 |
+
"distinct_cells": 119,
|
| 219 |
+
"longest_stuck": 0,
|
| 220 |
+
"deadlocked": false,
|
| 221 |
+
"game": "maze",
|
| 222 |
+
"size": 21,
|
| 223 |
+
"topology": "tree",
|
| 224 |
+
"seed": 902121,
|
| 225 |
+
"shortest": 118,
|
| 226 |
+
"efficiency": 1.0
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"solved": true,
|
| 230 |
+
"steps": 172,
|
| 231 |
+
"collisions": 0,
|
| 232 |
+
"distinct_cells": 173,
|
| 233 |
+
"longest_stuck": 0,
|
| 234 |
+
"deadlocked": false,
|
| 235 |
+
"game": "maze",
|
| 236 |
+
"size": 21,
|
| 237 |
+
"topology": "tree",
|
| 238 |
+
"seed": 902158,
|
| 239 |
+
"shortest": 172,
|
| 240 |
+
"efficiency": 1.0
|
| 241 |
+
},
|
| 242 |
+
{
|
| 243 |
+
"solved": true,
|
| 244 |
+
"steps": 140,
|
| 245 |
+
"collisions": 0,
|
| 246 |
+
"distinct_cells": 141,
|
| 247 |
+
"longest_stuck": 0,
|
| 248 |
+
"deadlocked": false,
|
| 249 |
+
"game": "maze",
|
| 250 |
+
"size": 21,
|
| 251 |
+
"topology": "tree",
|
| 252 |
+
"seed": 902195,
|
| 253 |
+
"shortest": 140,
|
| 254 |
+
"efficiency": 1.0
|
| 255 |
+
},
|
| 256 |
+
{
|
| 257 |
+
"solved": true,
|
| 258 |
+
"steps": 46,
|
| 259 |
+
"collisions": 0,
|
| 260 |
+
"distinct_cells": 47,
|
| 261 |
+
"longest_stuck": 0,
|
| 262 |
+
"deadlocked": false,
|
| 263 |
+
"game": "maze",
|
| 264 |
+
"size": 21,
|
| 265 |
+
"topology": "loops",
|
| 266 |
+
"seed": 902121,
|
| 267 |
+
"shortest": 46,
|
| 268 |
+
"efficiency": 1.0
|
| 269 |
+
},
|
| 270 |
+
{
|
| 271 |
+
"solved": true,
|
| 272 |
+
"steps": 40,
|
| 273 |
+
"collisions": 0,
|
| 274 |
+
"distinct_cells": 41,
|
| 275 |
+
"longest_stuck": 0,
|
| 276 |
+
"deadlocked": false,
|
| 277 |
+
"game": "maze",
|
| 278 |
+
"size": 21,
|
| 279 |
+
"topology": "loops",
|
| 280 |
+
"seed": 902158,
|
| 281 |
+
"shortest": 40,
|
| 282 |
+
"efficiency": 1.0
|
| 283 |
+
},
|
| 284 |
+
{
|
| 285 |
+
"solved": true,
|
| 286 |
+
"steps": 42,
|
| 287 |
+
"collisions": 0,
|
| 288 |
+
"distinct_cells": 43,
|
| 289 |
+
"longest_stuck": 0,
|
| 290 |
+
"deadlocked": false,
|
| 291 |
+
"game": "maze",
|
| 292 |
+
"size": 21,
|
| 293 |
+
"topology": "loops",
|
| 294 |
+
"seed": 902195,
|
| 295 |
+
"shortest": 42,
|
| 296 |
+
"efficiency": 1.0
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"solved": true,
|
| 300 |
+
"steps": 41,
|
| 301 |
+
"collisions": 0,
|
| 302 |
+
"distinct_cells": 42,
|
| 303 |
+
"longest_stuck": 0,
|
| 304 |
+
"deadlocked": false,
|
| 305 |
+
"game": "maze",
|
| 306 |
+
"size": 21,
|
| 307 |
+
"topology": "random_obstacle",
|
| 308 |
+
"seed": 902121,
|
| 309 |
+
"shortest": 41,
|
| 310 |
+
"efficiency": 1.0
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"solved": true,
|
| 314 |
+
"steps": 36,
|
| 315 |
+
"collisions": 0,
|
| 316 |
+
"distinct_cells": 37,
|
| 317 |
+
"longest_stuck": 0,
|
| 318 |
+
"deadlocked": false,
|
| 319 |
+
"game": "maze",
|
| 320 |
+
"size": 21,
|
| 321 |
+
"topology": "random_obstacle",
|
| 322 |
+
"seed": 902158,
|
| 323 |
+
"shortest": 36,
|
| 324 |
+
"efficiency": 1.0
|
| 325 |
+
},
|
| 326 |
+
{
|
| 327 |
+
"solved": true,
|
| 328 |
+
"steps": 36,
|
| 329 |
+
"collisions": 0,
|
| 330 |
+
"distinct_cells": 37,
|
| 331 |
+
"longest_stuck": 0,
|
| 332 |
+
"deadlocked": false,
|
| 333 |
+
"game": "maze",
|
| 334 |
+
"size": 21,
|
| 335 |
+
"topology": "random_obstacle",
|
| 336 |
+
"seed": 902195,
|
| 337 |
+
"shortest": 36,
|
| 338 |
+
"efficiency": 1.0
|
| 339 |
+
},
|
| 340 |
+
{
|
| 341 |
+
"solved": true,
|
| 342 |
+
"steps": 264,
|
| 343 |
+
"collisions": 0,
|
| 344 |
+
"distinct_cells": 265,
|
| 345 |
+
"longest_stuck": 0,
|
| 346 |
+
"deadlocked": false,
|
| 347 |
+
"game": "maze",
|
| 348 |
+
"size": 31,
|
| 349 |
+
"topology": "corridor",
|
| 350 |
+
"seed": 903131,
|
| 351 |
+
"shortest": 264,
|
| 352 |
+
"efficiency": 1.0
|
| 353 |
+
},
|
| 354 |
+
{
|
| 355 |
+
"solved": true,
|
| 356 |
+
"steps": 194,
|
| 357 |
+
"collisions": 0,
|
| 358 |
+
"distinct_cells": 195,
|
| 359 |
+
"longest_stuck": 0,
|
| 360 |
+
"deadlocked": false,
|
| 361 |
+
"game": "maze",
|
| 362 |
+
"size": 31,
|
| 363 |
+
"topology": "corridor",
|
| 364 |
+
"seed": 903168,
|
| 365 |
+
"shortest": 194,
|
| 366 |
+
"efficiency": 1.0
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"solved": true,
|
| 370 |
+
"steps": 286,
|
| 371 |
+
"collisions": 0,
|
| 372 |
+
"distinct_cells": 287,
|
| 373 |
+
"longest_stuck": 0,
|
| 374 |
+
"deadlocked": false,
|
| 375 |
+
"game": "maze",
|
| 376 |
+
"size": 31,
|
| 377 |
+
"topology": "corridor",
|
| 378 |
+
"seed": 903205,
|
| 379 |
+
"shortest": 286,
|
| 380 |
+
"efficiency": 1.0
|
| 381 |
+
},
|
| 382 |
+
{
|
| 383 |
+
"solved": true,
|
| 384 |
+
"steps": 268,
|
| 385 |
+
"collisions": 0,
|
| 386 |
+
"distinct_cells": 269,
|
| 387 |
+
"longest_stuck": 0,
|
| 388 |
+
"deadlocked": false,
|
| 389 |
+
"game": "maze",
|
| 390 |
+
"size": 31,
|
| 391 |
+
"topology": "tree",
|
| 392 |
+
"seed": 903131,
|
| 393 |
+
"shortest": 268,
|
| 394 |
+
"efficiency": 1.0
|
| 395 |
+
},
|
| 396 |
+
{
|
| 397 |
+
"solved": true,
|
| 398 |
+
"steps": 282,
|
| 399 |
+
"collisions": 0,
|
| 400 |
+
"distinct_cells": 283,
|
| 401 |
+
"longest_stuck": 0,
|
| 402 |
+
"deadlocked": false,
|
| 403 |
+
"game": "maze",
|
| 404 |
+
"size": 31,
|
| 405 |
+
"topology": "tree",
|
| 406 |
+
"seed": 903168,
|
| 407 |
+
"shortest": 282,
|
| 408 |
+
"efficiency": 1.0
|
| 409 |
+
},
|
| 410 |
+
{
|
| 411 |
+
"solved": true,
|
| 412 |
+
"steps": 252,
|
| 413 |
+
"collisions": 0,
|
| 414 |
+
"distinct_cells": 253,
|
| 415 |
+
"longest_stuck": 0,
|
| 416 |
+
"deadlocked": false,
|
| 417 |
+
"game": "maze",
|
| 418 |
+
"size": 31,
|
| 419 |
+
"topology": "tree",
|
| 420 |
+
"seed": 903205,
|
| 421 |
+
"shortest": 252,
|
| 422 |
+
"efficiency": 1.0
|
| 423 |
+
},
|
| 424 |
+
{
|
| 425 |
+
"solved": true,
|
| 426 |
+
"steps": 66,
|
| 427 |
+
"collisions": 0,
|
| 428 |
+
"distinct_cells": 67,
|
| 429 |
+
"longest_stuck": 0,
|
| 430 |
+
"deadlocked": false,
|
| 431 |
+
"game": "maze",
|
| 432 |
+
"size": 31,
|
| 433 |
+
"topology": "loops",
|
| 434 |
+
"seed": 903131,
|
| 435 |
+
"shortest": 66,
|
| 436 |
+
"efficiency": 1.0
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"solved": true,
|
| 440 |
+
"steps": 72,
|
| 441 |
+
"collisions": 0,
|
| 442 |
+
"distinct_cells": 73,
|
| 443 |
+
"longest_stuck": 0,
|
| 444 |
+
"deadlocked": false,
|
| 445 |
+
"game": "maze",
|
| 446 |
+
"size": 31,
|
| 447 |
+
"topology": "loops",
|
| 448 |
+
"seed": 903168,
|
| 449 |
+
"shortest": 72,
|
| 450 |
+
"efficiency": 1.0
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"solved": true,
|
| 454 |
+
"steps": 88,
|
| 455 |
+
"collisions": 0,
|
| 456 |
+
"distinct_cells": 89,
|
| 457 |
+
"longest_stuck": 0,
|
| 458 |
+
"deadlocked": false,
|
| 459 |
+
"game": "maze",
|
| 460 |
+
"size": 31,
|
| 461 |
+
"topology": "loops",
|
| 462 |
+
"seed": 903205,
|
| 463 |
+
"shortest": 88,
|
| 464 |
+
"efficiency": 1.0
|
| 465 |
+
},
|
| 466 |
+
{
|
| 467 |
+
"solved": true,
|
| 468 |
+
"steps": 57,
|
| 469 |
+
"collisions": 0,
|
| 470 |
+
"distinct_cells": 58,
|
| 471 |
+
"longest_stuck": 0,
|
| 472 |
+
"deadlocked": false,
|
| 473 |
+
"game": "maze",
|
| 474 |
+
"size": 31,
|
| 475 |
+
"topology": "random_obstacle",
|
| 476 |
+
"seed": 903131,
|
| 477 |
+
"shortest": 57,
|
| 478 |
+
"efficiency": 1.0
|
| 479 |
+
},
|
| 480 |
+
{
|
| 481 |
+
"solved": true,
|
| 482 |
+
"steps": 56,
|
| 483 |
+
"collisions": 0,
|
| 484 |
+
"distinct_cells": 57,
|
| 485 |
+
"longest_stuck": 0,
|
| 486 |
+
"deadlocked": false,
|
| 487 |
+
"game": "maze",
|
| 488 |
+
"size": 31,
|
| 489 |
+
"topology": "random_obstacle",
|
| 490 |
+
"seed": 903168,
|
| 491 |
+
"shortest": 56,
|
| 492 |
+
"efficiency": 1.0
|
| 493 |
+
},
|
| 494 |
+
{
|
| 495 |
+
"solved": true,
|
| 496 |
+
"steps": 58,
|
| 497 |
+
"collisions": 0,
|
| 498 |
+
"distinct_cells": 59,
|
| 499 |
+
"longest_stuck": 0,
|
| 500 |
+
"deadlocked": false,
|
| 501 |
+
"game": "maze",
|
| 502 |
+
"size": 31,
|
| 503 |
+
"topology": "random_obstacle",
|
| 504 |
+
"seed": 903205,
|
| 505 |
+
"shortest": 58,
|
| 506 |
+
"efficiency": 1.0
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"summary": {
|
| 510 |
+
"maze": {
|
| 511 |
+
"episodes": 36,
|
| 512 |
+
"solve_rate": 1.0,
|
| 513 |
+
"mean_steps_when_solved": 93.19444444444444,
|
| 514 |
+
"mean_efficiency": 1.0,
|
| 515 |
+
"mean_collisions": 0.0,
|
| 516 |
+
"deadlock_rate": 0.0,
|
| 517 |
+
"mean_distinct_cells": 94.19444444444444,
|
| 518 |
+
"by_size": {
|
| 519 |
+
"11": {
|
| 520 |
+
"episodes": 12,
|
| 521 |
+
"solve_rate": 1.0,
|
| 522 |
+
"mean_steps": 27.25,
|
| 523 |
+
"mean_efficiency": 1.0,
|
| 524 |
+
"mean_shortest": 27.25
|
| 525 |
+
},
|
| 526 |
+
"21": {
|
| 527 |
+
"episodes": 12,
|
| 528 |
+
"solve_rate": 1.0,
|
| 529 |
+
"mean_steps": 90.41666666666667,
|
| 530 |
+
"mean_efficiency": 1.0,
|
| 531 |
+
"mean_shortest": 90.41666666666667
|
| 532 |
+
},
|
| 533 |
+
"31": {
|
| 534 |
+
"episodes": 12,
|
| 535 |
+
"solve_rate": 1.0,
|
| 536 |
+
"mean_steps": 161.91666666666666,
|
| 537 |
+
"mean_efficiency": 1.0,
|
| 538 |
+
"mean_shortest": 161.91666666666666
|
| 539 |
+
}
|
| 540 |
+
}
|
| 541 |
+
}
|
| 542 |
+
}
|
| 543 |
+
}
|
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "model",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"food": 0,
|
| 6 |
+
"steps": 288,
|
| 7 |
+
"alive": true,
|
| 8 |
+
"death": null,
|
| 9 |
+
"length": 3,
|
| 10 |
+
"game": "snake",
|
| 11 |
+
"size": 12,
|
| 12 |
+
"seed": 610084
|
| 13 |
+
},
|
| 14 |
+
{
|
| 15 |
+
"food": 4,
|
| 16 |
+
"steps": 288,
|
| 17 |
+
"alive": true,
|
| 18 |
+
"death": null,
|
| 19 |
+
"length": 7,
|
| 20 |
+
"game": "snake",
|
| 21 |
+
"size": 12,
|
| 22 |
+
"seed": 610137
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"food": 4,
|
| 26 |
+
"steps": 288,
|
| 27 |
+
"alive": true,
|
| 28 |
+
"death": null,
|
| 29 |
+
"length": 7,
|
| 30 |
+
"game": "snake",
|
| 31 |
+
"size": 12,
|
| 32 |
+
"seed": 610190
|
| 33 |
+
},
|
| 34 |
+
{
|
| 35 |
+
"food": 0,
|
| 36 |
+
"steps": 288,
|
| 37 |
+
"alive": true,
|
| 38 |
+
"death": null,
|
| 39 |
+
"length": 3,
|
| 40 |
+
"game": "snake",
|
| 41 |
+
"size": 12,
|
| 42 |
+
"seed": 610243
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"food": 16,
|
| 46 |
+
"steps": 288,
|
| 47 |
+
"alive": true,
|
| 48 |
+
"death": null,
|
| 49 |
+
"length": 19,
|
| 50 |
+
"game": "snake",
|
| 51 |
+
"size": 12,
|
| 52 |
+
"seed": 610296
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"food": 3,
|
| 56 |
+
"steps": 288,
|
| 57 |
+
"alive": true,
|
| 58 |
+
"death": null,
|
| 59 |
+
"length": 6,
|
| 60 |
+
"game": "snake",
|
| 61 |
+
"size": 12,
|
| 62 |
+
"seed": 610349
|
| 63 |
+
}
|
| 64 |
+
],
|
| 65 |
+
"summary": {
|
| 66 |
+
"snake": {
|
| 67 |
+
"episodes": 6,
|
| 68 |
+
"mean_food": 4.5,
|
| 69 |
+
"max_food": 16,
|
| 70 |
+
"mean_steps": 288.0,
|
| 71 |
+
"survival_rate": 1.0
|
| 72 |
+
}
|
| 73 |
+
}
|
| 74 |
+
}
|
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "random",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"food": 0,
|
| 6 |
+
"steps": 7,
|
| 7 |
+
"alive": false,
|
| 8 |
+
"death": "wall",
|
| 9 |
+
"length": 3,
|
| 10 |
+
"game": "snake",
|
| 11 |
+
"size": 12,
|
| 12 |
+
"seed": 610084
|
| 13 |
+
},
|
| 14 |
+
{
|
| 15 |
+
"food": 0,
|
| 16 |
+
"steps": 19,
|
| 17 |
+
"alive": false,
|
| 18 |
+
"death": "wall",
|
| 19 |
+
"length": 3,
|
| 20 |
+
"game": "snake",
|
| 21 |
+
"size": 12,
|
| 22 |
+
"seed": 610137
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"food": 2,
|
| 26 |
+
"steps": 60,
|
| 27 |
+
"alive": false,
|
| 28 |
+
"death": "body",
|
| 29 |
+
"length": 5,
|
| 30 |
+
"game": "snake",
|
| 31 |
+
"size": 12,
|
| 32 |
+
"seed": 610190
|
| 33 |
+
},
|
| 34 |
+
{
|
| 35 |
+
"food": 0,
|
| 36 |
+
"steps": 85,
|
| 37 |
+
"alive": false,
|
| 38 |
+
"death": "wall",
|
| 39 |
+
"length": 3,
|
| 40 |
+
"game": "snake",
|
| 41 |
+
"size": 12,
|
| 42 |
+
"seed": 610243
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"food": 1,
|
| 46 |
+
"steps": 21,
|
| 47 |
+
"alive": false,
|
| 48 |
+
"death": "wall",
|
| 49 |
+
"length": 4,
|
| 50 |
+
"game": "snake",
|
| 51 |
+
"size": 12,
|
| 52 |
+
"seed": 610296
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"food": 1,
|
| 56 |
+
"steps": 11,
|
| 57 |
+
"alive": false,
|
| 58 |
+
"death": "wall",
|
| 59 |
+
"length": 4,
|
| 60 |
+
"game": "snake",
|
| 61 |
+
"size": 12,
|
| 62 |
+
"seed": 610349
|
| 63 |
+
}
|
| 64 |
+
],
|
| 65 |
+
"summary": {
|
| 66 |
+
"snake": {
|
| 67 |
+
"episodes": 6,
|
| 68 |
+
"mean_food": 0.6666666666666666,
|
| 69 |
+
"max_food": 2,
|
| 70 |
+
"mean_steps": 33.833333333333336,
|
| 71 |
+
"survival_rate": 0.0
|
| 72 |
+
}
|
| 73 |
+
}
|
| 74 |
+
}
|
|
@@ -0,0 +1,74 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"controller": "reference",
|
| 3 |
+
"episodes": [
|
| 4 |
+
{
|
| 5 |
+
"food": 16,
|
| 6 |
+
"steps": 288,
|
| 7 |
+
"alive": true,
|
| 8 |
+
"death": null,
|
| 9 |
+
"length": 19,
|
| 10 |
+
"game": "snake",
|
| 11 |
+
"size": 12,
|
| 12 |
+
"seed": 610084
|
| 13 |
+
},
|
| 14 |
+
{
|
| 15 |
+
"food": 30,
|
| 16 |
+
"steps": 288,
|
| 17 |
+
"alive": true,
|
| 18 |
+
"death": null,
|
| 19 |
+
"length": 33,
|
| 20 |
+
"game": "snake",
|
| 21 |
+
"size": 12,
|
| 22 |
+
"seed": 610137
|
| 23 |
+
},
|
| 24 |
+
{
|
| 25 |
+
"food": 31,
|
| 26 |
+
"steps": 288,
|
| 27 |
+
"alive": true,
|
| 28 |
+
"death": null,
|
| 29 |
+
"length": 34,
|
| 30 |
+
"game": "snake",
|
| 31 |
+
"size": 12,
|
| 32 |
+
"seed": 610190
|
| 33 |
+
},
|
| 34 |
+
{
|
| 35 |
+
"food": 22,
|
| 36 |
+
"steps": 288,
|
| 37 |
+
"alive": true,
|
| 38 |
+
"death": null,
|
| 39 |
+
"length": 25,
|
| 40 |
+
"game": "snake",
|
| 41 |
+
"size": 12,
|
| 42 |
+
"seed": 610243
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"food": 13,
|
| 46 |
+
"steps": 288,
|
| 47 |
+
"alive": true,
|
| 48 |
+
"death": null,
|
| 49 |
+
"length": 16,
|
| 50 |
+
"game": "snake",
|
| 51 |
+
"size": 12,
|
| 52 |
+
"seed": 610296
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"food": 29,
|
| 56 |
+
"steps": 288,
|
| 57 |
+
"alive": true,
|
| 58 |
+
"death": null,
|
| 59 |
+
"length": 32,
|
| 60 |
+
"game": "snake",
|
| 61 |
+
"size": 12,
|
| 62 |
+
"seed": 610349
|
| 63 |
+
}
|
| 64 |
+
],
|
| 65 |
+
"summary": {
|
| 66 |
+
"snake": {
|
| 67 |
+
"episodes": 6,
|
| 68 |
+
"mean_food": 23.5,
|
| 69 |
+
"max_food": 31,
|
| 70 |
+
"mean_steps": 288.0,
|
| 71 |
+
"survival_rate": 1.0
|
| 72 |
+
}
|
| 73 |
+
}
|
| 74 |
+
}
|
|
@@ -0,0 +1,42 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"11": {
|
| 3 |
+
"cells": 215,
|
| 4 |
+
"probe_r2": 0.09365391731262207,
|
| 5 |
+
"probe_mae": 7.971122741699219,
|
| 6 |
+
"head_r2": 0.9999997019767761,
|
| 7 |
+
"descent_acc": 1.0,
|
| 8 |
+
"reach_rate": 1.0,
|
| 9 |
+
"diameter": 48,
|
| 10 |
+
"iterations": 82
|
| 11 |
+
},
|
| 12 |
+
"21": {
|
| 13 |
+
"cells": 901,
|
| 14 |
+
"probe_r2": 0.04196900129318237,
|
| 15 |
+
"probe_mae": 29.71183204650879,
|
| 16 |
+
"head_r2": 0.9999999403953552,
|
| 17 |
+
"descent_acc": 1.0,
|
| 18 |
+
"reach_rate": 1.0,
|
| 19 |
+
"diameter": 152,
|
| 20 |
+
"iterations": 262
|
| 21 |
+
},
|
| 22 |
+
"31": {
|
| 23 |
+
"cells": 2011,
|
| 24 |
+
"probe_r2": 0.031648457050323486,
|
| 25 |
+
"probe_mae": 63.062259674072266,
|
| 26 |
+
"head_r2": 1.0,
|
| 27 |
+
"descent_acc": 1.0,
|
| 28 |
+
"reach_rate": 1.0,
|
| 29 |
+
"diameter": 310,
|
| 30 |
+
"iterations": 542
|
| 31 |
+
},
|
| 32 |
+
"51": {
|
| 33 |
+
"cells": 5679,
|
| 34 |
+
"probe_r2": 0.006791293621063232,
|
| 35 |
+
"probe_mae": 145.2346954345703,
|
| 36 |
+
"head_r2": 1.0,
|
| 37 |
+
"descent_acc": 1.0,
|
| 38 |
+
"reach_rate": 1.0,
|
| 39 |
+
"diameter": 672,
|
| 40 |
+
"iterations": 1402
|
| 41 |
+
}
|
| 42 |
+
}
|
|
@@ -0,0 +1 @@
|
|
|
|
|
|
|
| 1 |
+
{"vocab": ["<pad>", "<unk>", "<bos>", "<eos>", "<sep>", "<byte_0>", "<byte_1>", "<byte_2>", "<byte_3>", "<byte_4>", "<byte_5>", "<byte_6>", "<byte_7>", "<byte_8>", "<byte_9>", "<byte_10>", "<byte_11>", "<byte_12>", "<byte_13>", "<byte_14>", "<byte_15>", "<byte_16>", "<byte_17>", "<byte_18>", "<byte_19>", "<byte_20>", "<byte_21>", "<byte_22>", "<byte_23>", "<byte_24>", "<byte_25>", "<byte_26>", "<byte_27>", "<byte_28>", "<byte_29>", "<byte_30>", "<byte_31>", "<byte_32>", "<byte_33>", "<byte_34>", "<byte_35>", "<byte_36>", "<byte_37>", "<byte_38>", "<byte_39>", "<byte_40>", "<byte_41>", "<byte_42>", "<byte_43>", "<byte_44>", "<byte_45>", "<byte_46>", "<byte_47>", "<byte_48>", "<byte_49>", "<byte_50>", "<byte_51>", "<byte_52>", "<byte_53>", "<byte_54>", "<byte_55>", "<byte_56>", "<byte_57>", "<byte_58>", "<byte_59>", "<byte_60>", "<byte_61>", "<byte_62>", "<byte_63>", "<byte_64>", "<byte_65>", "<byte_66>", "<byte_67>", "<byte_68>", "<byte_69>", "<byte_70>", "<byte_71>", "<byte_72>", "<byte_73>", "<byte_74>", "<byte_75>", "<byte_76>", "<byte_77>", "<byte_78>", "<byte_79>", "<byte_80>", "<byte_81>", "<byte_82>", "<byte_83>", "<byte_84>", "<byte_85>", "<byte_86>", "<byte_87>", "<byte_88>", "<byte_89>", "<byte_90>", "<byte_91>", "<byte_92>", "<byte_93>", "<byte_94>", "<byte_95>", "<byte_96>", "<byte_97>", "<byte_98>", "<byte_99>", "<byte_100>", "<byte_101>", "<byte_102>", "<byte_103>", "<byte_104>", "<byte_105>", "<byte_106>", "<byte_107>", "<byte_108>", "<byte_109>", "<byte_110>", "<byte_111>", "<byte_112>", "<byte_113>", "<byte_114>", "<byte_115>", "<byte_116>", "<byte_117>", "<byte_118>", "<byte_119>", "<byte_120>", "<byte_121>", "<byte_122>", "<byte_123>", "<byte_124>", "<byte_125>", "<byte_126>", "<byte_127>", "<byte_128>", "<byte_129>", "<byte_130>", "<byte_131>", "<byte_132>", "<byte_133>", "<byte_134>", "<byte_135>", "<byte_136>", "<byte_137>", "<byte_138>", "<byte_139>", "<byte_140>", "<byte_141>", "<byte_142>", "<byte_143>", "<byte_144>", "<byte_145>", "<byte_146>", "<byte_147>", "<byte_148>", "<byte_149>", "<byte_150>", "<byte_151>", "<byte_152>", "<byte_153>", "<byte_154>", "<byte_155>", "<byte_156>", "<byte_157>", "<byte_158>", "<byte_159>", "<byte_160>", "<byte_161>", "<byte_162>", "<byte_163>", "<byte_164>", "<byte_165>", "<byte_166>", "<byte_167>", "<byte_168>", "<byte_169>", "<byte_170>", "<byte_171>", "<byte_172>", "<byte_173>", "<byte_174>", "<byte_175>", "<byte_176>", "<byte_177>", "<byte_178>", "<byte_179>", "<byte_180>", "<byte_181>", "<byte_182>", "<byte_183>", "<byte_184>", "<byte_185>", "<byte_186>", "<byte_187>", "<byte_188>", "<byte_189>", "<byte_190>", "<byte_191>", "<byte_192>", "<byte_193>", "<byte_194>", "<byte_195>", "<byte_196>", "<byte_197>", "<byte_198>", "<byte_199>", "<byte_200>", "<byte_201>", "<byte_202>", "<byte_203>", "<byte_204>", "<byte_205>", "<byte_206>", "<byte_207>", "<byte_208>", "<byte_209>", "<byte_210>", "<byte_211>", "<byte_212>", "<byte_213>", "<byte_214>", "<byte_215>", "<byte_216>", "<byte_217>", "<byte_218>", "<byte_219>", "<byte_220>", "<byte_221>", "<byte_222>", "<byte_223>", "<byte_224>", "<byte_225>", "<byte_226>", "<byte_227>", "<byte_228>", "<byte_229>", "<byte_230>", "<byte_231>", "<byte_232>", "<byte_233>", "<byte_234>", "<byte_235>", "<byte_236>", "<byte_237>", "<byte_238>", "<byte_239>", "<byte_240>", "<byte_241>", "<byte_242>", "<byte_243>", "<byte_244>", "<byte_245>", "<byte_246>", "<byte_247>", "<byte_248>", "<byte_249>", "<byte_250>", "<byte_251>", "<byte_252>", "<byte_253>", "<byte_254>", "<byte_255>", ".", "the", ",", "is", "(", ")", "goal", "a", "to", "moves", "proposition", "true", "not", "moving", "move", ":", "snake", "maze", "from", "within", "=", "/", "and", "wall", "east", "west", "probability", "south", "north", "1", "reaches", "which", "inside", "decide", ";", "cell", "5", "3", "7", "your", "agent", "on", "-", "room", "2", "'", "s", "open", "4", "or", "6", "next", "tie", "ties", "should", "share", "evenly", "9", "report", "confidence", "length", "own", "game", "does", "end", "still", "dead", "weigh", "options", "carefully", "8", "answer", "with", "for", "each", "option", "than", "five", "sixty", "task", "read", "board", "path", "legal", "all", "question", "please", "11", "immediately", "after", "can", "reach", "its", "tail", "so", "it", "sealed", "into", "pocket", "times", "x", "coordinates", "are", "zero", "based", "row", "column", "one", "how", "?", "more", "choose", "equally", "10", "13", "12", "15", "food", "that", "topology", "no", "diagonals", "wraparound", "stop", "far", "current", "this", "already", "two", "ten", "twenty", "away", "cannot", "be", "reached", "of", "cells", "connects", "shortest", "short", "if", "unreachable", "0", "14", "16", "21", "head", "facing", "reversing", "onto", "neck", "offered", "hitting", "body", "ends", "keeps", "alive", "makes", "most", "progress", "towards", "eating", "prefer", "leave", "an", "escape", "route", "good", "much", "have", "manoeuvre", "in", "trapped", "less", "long", "barely", "enough", "fit", "likely", "limited", "under", "three", "comfortable", "several", "wide", "eight", "17", "19", "25", "22", "29", "31", "corridor", "tree", "loops", "23", "random", "_", "obstacle", "20", "18", "27", "28", "24", "26", "30", "36", "85", "79", "33", "35", "59", "61", "56", "64", "50", "39", "71", "40", "32", "69", "62", "54", "37", "41", "111", "43", "70", "106", "101", "98", "47", "58", "49", "38", "60"]}
|
|
@@ -0,0 +1,91 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env bash
|
| 2 |
+
# Full evaluation suite for a trained checkpoint.
|
| 3 |
+
#
|
| 4 |
+
# scripts/benchmark.sh runs/jevon-final [arcade-data-dir] [weights]
|
| 5 |
+
#
|
| 6 |
+
# `weights` selects which checkpoint the whole suite is run against, and
|
| 7 |
+
# defaults to balanced.pt -- the one chosen on min(maze_action, snake_action)
|
| 8 |
+
# rather than on aggregate eval loss, which the boolean questions dominate.
|
| 9 |
+
# Falls back to best.pt for runs trained before that existed. Every number in
|
| 10 |
+
# RESULTS.md comes from the single checkpoint named here, so the evaluation,
|
| 11 |
+
# the probe and all the play configs cannot silently disagree about which
|
| 12 |
+
# model they measured.
|
| 13 |
+
#
|
| 14 |
+
# Writes eval.json, probe.json and play/*.json under the run directory, then
|
| 15 |
+
# regenerates RESULTS.md from them.
|
| 16 |
+
set -euo pipefail
|
| 17 |
+
|
| 18 |
+
RUN="${1:-runs/jevon-final}"
|
| 19 |
+
ARCADE="${2:-}"
|
| 20 |
+
WEIGHTS="${3:-}"
|
| 21 |
+
PY="${PYTHON:-$(dirname "$0")/../../.venv/bin/python}"
|
| 22 |
+
[ -x "$PY" ] || PY=python3
|
| 23 |
+
PLAY="$RUN/play"
|
| 24 |
+
mkdir -p "$PLAY"
|
| 25 |
+
|
| 26 |
+
if [ -z "$WEIGHTS" ]; then
|
| 27 |
+
if [ -f "$RUN/balanced.pt" ]; then WEIGHTS=balanced.pt; else WEIGHTS=best.pt; fi
|
| 28 |
+
fi
|
| 29 |
+
[ -f "$RUN/$WEIGHTS" ] || { echo "no such checkpoint: $RUN/$WEIGHTS" >&2; exit 1; }
|
| 30 |
+
echo "==> benchmarking $RUN/$WEIGHTS"
|
| 31 |
+
|
| 32 |
+
echo "==> question-level metrics and calibration"
|
| 33 |
+
"$PY" scripts/evaluate.py --checkpoint "$RUN" --weights "$WEIGHTS" \
|
| 34 |
+
--splits data/frozen/val.jsonl,data/frozen/test.jsonl,data/frozen/ood.jsonl \
|
| 35 |
+
--out "$RUN/eval.json"
|
| 36 |
+
|
| 37 |
+
echo "==> planner field probe"
|
| 38 |
+
"$PY" scripts/probe.py --checkpoint "$RUN" --weights "$WEIGHTS" \
|
| 39 |
+
--out "$RUN/probe.json"
|
| 40 |
+
|
| 41 |
+
play () { # play <name> <game> <controller> <extra...>
|
| 42 |
+
local name="$1" game="$2" ctrl="$3"; shift 3
|
| 43 |
+
echo "==> play: $name"
|
| 44 |
+
local trace=()
|
| 45 |
+
# --trace-max-steps shortens the replay, never the episode, so the summary is
|
| 46 |
+
# still measured over the full run. Without it a random walk on a 31x31 board
|
| 47 |
+
# contributes its whole 3844-step cap and the arcade data directory grows to
|
| 48 |
+
# megabytes of JSON nobody will scrub through. record.sh in the arcade repo
|
| 49 |
+
# uses the same cap; they must agree or the two repos ship different replays.
|
| 50 |
+
[ -n "$ARCADE" ] && trace=(--trace-out "$ARCADE/${name}.json" --trace-max-steps 400)
|
| 51 |
+
"$PY" scripts/play.py --checkpoint "$RUN" --weights "$WEIGHTS" \
|
| 52 |
+
--game "$game" --controller "$ctrl" \
|
| 53 |
+
--out "$PLAY/${name}_summary.json" "${trace[@]}" "$@"
|
| 54 |
+
[ -n "$ARCADE" ] && cp "$PLAY/${name}_summary.json" "$ARCADE/" || true
|
| 55 |
+
}
|
| 56 |
+
|
| 57 |
+
MAZE_ARGS=(--episodes 3 --maze-sizes 11,21,31)
|
| 58 |
+
SNAKE_ARGS=(--episodes 6 --snake-sizes 12)
|
| 59 |
+
|
| 60 |
+
play maze_model maze model "${MAZE_ARGS[@]}"
|
| 61 |
+
# Same network, same absence of search or memory, but sampling from the
|
| 62 |
+
# distribution instead of taking its argmax. Now that only legal moves are
|
| 63 |
+
# offered, a greedy controller cannot bang into a wall, but it can still cycle:
|
| 64 |
+
# the planner is stateless, so a two-cell oscillation repeats forever. Sampling
|
| 65 |
+
# breaks that with no memory at all, if the policy is calibrated enough for its
|
| 66 |
+
# second choice to be worth taking. Whether it is, is the point of measuring.
|
| 67 |
+
play maze_model_sampled maze model "${MAZE_ARGS[@]}" --temperature 1.0
|
| 68 |
+
play maze_model_field maze model-field "${MAZE_ARGS[@]}"
|
| 69 |
+
play maze_model_memory maze model+memory "${MAZE_ARGS[@]}"
|
| 70 |
+
play maze_reference maze reference "${MAZE_ARGS[@]}"
|
| 71 |
+
play maze_random maze random "${MAZE_ARGS[@]}"
|
| 72 |
+
# The control for maze_model_memory, and the reason that row cannot be read on
|
| 73 |
+
# its own. Identical bookkeeping, no network: whatever this solves, the memory
|
| 74 |
+
# solved. On a small board near-exhaustive exploration is enough to match the
|
| 75 |
+
# model on solve rate, so the two separate on path length, not on arrival.
|
| 76 |
+
play maze_random_memory maze random+memory "${MAZE_ARGS[@]}"
|
| 77 |
+
play snake_model snake model "${SNAKE_ARGS[@]}"
|
| 78 |
+
play snake_reference snake reference "${SNAKE_ARGS[@]}"
|
| 79 |
+
play snake_random snake random "${SNAKE_ARGS[@]}"
|
| 80 |
+
|
| 81 |
+
echo "==> RESULTS.md"
|
| 82 |
+
"$PY" scripts/make_results.py --run "$RUN" --eval "$RUN/eval.json" \
|
| 83 |
+
--probe "$RUN/probe.json" --play-dir "$PLAY"
|
| 84 |
+
|
| 85 |
+
# The README headline is generated from the same JSON, so it cannot disagree
|
| 86 |
+
# with RESULTS.md. Only refreshed when benchmarking the run the README is
|
| 87 |
+
# about; benchmarking an experimental arm must not rewrite the front page.
|
| 88 |
+
if [ "$RUN" = "runs/jevon-final" ]; then
|
| 89 |
+
echo "==> README headline"
|
| 90 |
+
"$PY" scripts/readme_table.py --run "$RUN"
|
| 91 |
+
fi
|
|
@@ -0,0 +1,47 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Freeze evaluation splits so every run is scored on identical states."""
|
| 3 |
+
from __future__ import annotations
|
| 4 |
+
|
| 5 |
+
import argparse
|
| 6 |
+
import json
|
| 7 |
+
import sys
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
|
| 10 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 11 |
+
sys.path.insert(0, str(ROOT))
|
| 12 |
+
|
| 13 |
+
from jevon.data import SamplerConfig, freeze_split
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def main() -> None:
|
| 17 |
+
p = argparse.ArgumentParser(description=__doc__)
|
| 18 |
+
p.add_argument("--out-dir", default="data/frozen")
|
| 19 |
+
p.add_argument("--val-states", type=int, default=180)
|
| 20 |
+
p.add_argument("--test-states", type=int, default=240)
|
| 21 |
+
p.add_argument("--ood-states", type=int, default=160)
|
| 22 |
+
p.add_argument("--seed", type=int, default=20260919)
|
| 23 |
+
args = p.parse_args()
|
| 24 |
+
|
| 25 |
+
out = Path(args.out_dir)
|
| 26 |
+
out.mkdir(parents=True, exist_ok=True)
|
| 27 |
+
base = SamplerConfig(augment=False)
|
| 28 |
+
# Sizes never sampled in training; this is the size-generalisation probe.
|
| 29 |
+
ood = SamplerConfig(maze_sizes=(41, 51), snake_sizes=(20, 24), augment=False)
|
| 30 |
+
|
| 31 |
+
manifest = {}
|
| 32 |
+
manifest["val"] = freeze_split(out / "val.jsonl", base, "val", args.val_states,
|
| 33 |
+
args.seed)
|
| 34 |
+
manifest["test"] = freeze_split(out / "test.jsonl", base, "test",
|
| 35 |
+
args.test_states, args.seed + 1)
|
| 36 |
+
manifest["ood"] = freeze_split(out / "ood.jsonl", ood, "test", args.ood_states,
|
| 37 |
+
args.seed + 2)
|
| 38 |
+
manifest["config"] = {"train_maze_sizes": list(base.maze_sizes),
|
| 39 |
+
"train_snake_sizes": list(base.snake_sizes),
|
| 40 |
+
"ood_maze_sizes": list(ood.maze_sizes),
|
| 41 |
+
"ood_snake_sizes": list(ood.snake_sizes)}
|
| 42 |
+
(out / "manifest.json").write_text(json.dumps(manifest, indent=2))
|
| 43 |
+
print(json.dumps(manifest, indent=2))
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
if __name__ == "__main__":
|
| 47 |
+
main()
|
|
@@ -0,0 +1,128 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Question-level evaluation: accuracy, distribution error, and calibration.
|
| 3 |
+
|
| 4 |
+
Reported per split, per game, and per question type. Two controls are printed
|
| 5 |
+
alongside every accuracy so the numbers cannot be read as better than they are:
|
| 6 |
+
``majority`` is the accuracy of always answering the most common label, and
|
| 7 |
+
``uniform`` is the accuracy of guessing uniformly over the offered candidates.
|
| 8 |
+
NanoJev's maze atomic accuracy matched its own always-true control exactly,
|
| 9 |
+
which is the failure mode these columns are here to expose.
|
| 10 |
+
"""
|
| 11 |
+
from __future__ import annotations
|
| 12 |
+
|
| 13 |
+
import argparse
|
| 14 |
+
import json
|
| 15 |
+
import sys
|
| 16 |
+
from collections import defaultdict
|
| 17 |
+
from pathlib import Path
|
| 18 |
+
|
| 19 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 20 |
+
sys.path.insert(0, str(ROOT))
|
| 21 |
+
|
| 22 |
+
import torch
|
| 23 |
+
|
| 24 |
+
from jevon.data import collate, load_frozen
|
| 25 |
+
from jevon.inference import JevonRunner
|
| 26 |
+
from jevon.losses import calibration_bins
|
| 27 |
+
from jevon.planner import default_iterations
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
def main() -> None:
|
| 31 |
+
p = argparse.ArgumentParser(description=__doc__)
|
| 32 |
+
p.add_argument("--checkpoint", default="runs/jevon-final")
|
| 33 |
+
# None means "prefer balanced.pt", the checkpoint every shipped number
|
| 34 |
+
# comes from. See jevon.inference.resolve_weights.
|
| 35 |
+
p.add_argument("--weights", default=None)
|
| 36 |
+
p.add_argument("--splits", default="data/frozen/test.jsonl,data/frozen/ood.jsonl")
|
| 37 |
+
p.add_argument("--batch-states", type=int, default=8)
|
| 38 |
+
p.add_argument("--device", default="mps")
|
| 39 |
+
p.add_argument("--out", default=None)
|
| 40 |
+
args = p.parse_args()
|
| 41 |
+
|
| 42 |
+
runner = JevonRunner(args.checkpoint, args.device, args.weights)
|
| 43 |
+
report = {}
|
| 44 |
+
|
| 45 |
+
for path in args.splits.split(","):
|
| 46 |
+
path = path.strip()
|
| 47 |
+
if not path or not Path(path).exists():
|
| 48 |
+
continue
|
| 49 |
+
name = Path(path).stem
|
| 50 |
+
states = load_frozen(path)
|
| 51 |
+
buckets: dict[tuple[str, str], list] = defaultdict(list)
|
| 52 |
+
# How many distinct boards a bucket's questions came from. Several
|
| 53 |
+
# questions about one maze would not be several independent tests of
|
| 54 |
+
# whether the model can read a maze -- they share walls, goal and
|
| 55 |
+
# distance field, so they succeed or fail together -- and n alone
|
| 56 |
+
# cannot tell you whether that is happening. Measured here it is not:
|
| 57 |
+
# the generator draws a fresh board per state, so boards tracks n on
|
| 58 |
+
# every split. The column exists to keep that checkable rather than
|
| 59 |
+
# assumed, because it is a property of the generator and generators
|
| 60 |
+
# get edited.
|
| 61 |
+
board_ids: dict[tuple[str, str], set] = defaultdict(set)
|
| 62 |
+
conf, hit = [], []
|
| 63 |
+
with torch.no_grad():
|
| 64 |
+
for start in range(0, len(states), args.batch_states):
|
| 65 |
+
group = states[start:start + args.batch_states]
|
| 66 |
+
batch = collate(group, runner.tokenizer, device=runner.device)
|
| 67 |
+
steps = default_iterations(int(batch["board_size"].max()))
|
| 68 |
+
logits = runner.model(batch, iterations=steps, grad_steps=0)
|
| 69 |
+
probs = logits.float().softmax(-1).cpu()
|
| 70 |
+
target = batch["target_probs"].cpu()
|
| 71 |
+
for row, meta in enumerate(batch["meta"]):
|
| 72 |
+
k = len(meta["probs"])
|
| 73 |
+
pr, tg = probs[row, :k], target[row, :k]
|
| 74 |
+
picked = int(pr.argmax())
|
| 75 |
+
correct = float(tg[picked] >= tg.max() - 1e-6)
|
| 76 |
+
tvd = float(0.5 * (pr - tg).abs().sum())
|
| 77 |
+
brier = float(((pr - tg) ** 2).sum())
|
| 78 |
+
# Controls computed on the very same questions.
|
| 79 |
+
majority = float(tg[int(tg.argmax())] >= tg.max() - 1e-6)
|
| 80 |
+
uniform = float(tg.max() * 0 + (tg >= tg.max() - 1e-6).float().mean())
|
| 81 |
+
key = (meta["game"], meta["type"])
|
| 82 |
+
# The uid schema differs per game -- maze is
|
| 83 |
+
# "maze:<topology>:<size>:<seed>:<row>_<col>" and snake is
|
| 84 |
+
# "snake:<size>:<seed>:<step>" -- but the last segment is
|
| 85 |
+
# the within-board position in both, so dropping it names
|
| 86 |
+
# the board either way. Cross-checked against a hash of
|
| 87 |
+
# channel 0 and the goal cell: the two agree to within
|
| 88 |
+
# three coincidental duplicates across all three splits.
|
| 89 |
+
board = (meta["size"], meta["topology"],
|
| 90 |
+
meta["uid"].rsplit(":", 1)[0])
|
| 91 |
+
kind_key = (meta["game"], meta["question"].split("_")[0])
|
| 92 |
+
buckets[key].append((correct, tvd, brier, uniform))
|
| 93 |
+
buckets[kind_key].append((correct, tvd, brier, uniform))
|
| 94 |
+
board_ids[key].add(board)
|
| 95 |
+
board_ids[kind_key].add(board)
|
| 96 |
+
del majority
|
| 97 |
+
conf.append(float(pr.max()))
|
| 98 |
+
hit.append(correct)
|
| 99 |
+
rows = {}
|
| 100 |
+
for (game, kind), values in sorted(buckets.items()):
|
| 101 |
+
n = len(values)
|
| 102 |
+
rows[f"{game}/{kind}"] = {
|
| 103 |
+
"n": n,
|
| 104 |
+
"boards": len(board_ids[(game, kind)]),
|
| 105 |
+
"accuracy": sum(v[0] for v in values) / n,
|
| 106 |
+
"uniform_control": sum(v[3] for v in values) / n,
|
| 107 |
+
"tvd": sum(v[1] for v in values) / n,
|
| 108 |
+
"brier": sum(v[2] for v in values) / n,
|
| 109 |
+
}
|
| 110 |
+
calib = calibration_bins(torch.tensor(conf), torch.tensor(hit))
|
| 111 |
+
report[name] = {"states": len(states), "questions": len(conf),
|
| 112 |
+
"ece": calib["ece"], "groups": rows}
|
| 113 |
+
print(f"\n=== {name}: {len(states)} states / {len(conf)} questions "
|
| 114 |
+
f"(ECE {calib['ece']:.4f}) ===")
|
| 115 |
+
print(f"{'group':28s} {'n':>5s} {'brds':>5s} {'acc':>7s} {'unif':>7s} "
|
| 116 |
+
f"{'tvd':>7s} {'brier':>7s}")
|
| 117 |
+
for key, row in rows.items():
|
| 118 |
+
print(f"{key:28s} {row['n']:5d} {row['boards']:5d} "
|
| 119 |
+
f"{row['accuracy']:7.4f} {row['uniform_control']:7.4f} "
|
| 120 |
+
f"{row['tvd']:7.4f} {row['brier']:7.4f}")
|
| 121 |
+
|
| 122 |
+
if args.out:
|
| 123 |
+
Path(args.out).parent.mkdir(parents=True, exist_ok=True)
|
| 124 |
+
Path(args.out).write_text(json.dumps(report, indent=2))
|
| 125 |
+
|
| 126 |
+
|
| 127 |
+
if __name__ == "__main__":
|
| 128 |
+
main()
|
|
@@ -0,0 +1,244 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Generate RESULTS.md from the evaluation artifacts.
|
| 3 |
+
|
| 4 |
+
Every number in the document comes from a JSON file written by another
|
| 5 |
+
script, so the documented results and the measured results cannot drift.
|
| 6 |
+
Missing inputs are skipped with a note rather than guessed at.
|
| 7 |
+
"""
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
import argparse
|
| 11 |
+
import json
|
| 12 |
+
from pathlib import Path
|
| 13 |
+
|
| 14 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
def read(path: Path):
|
| 18 |
+
try:
|
| 19 |
+
return json.loads(path.read_text())
|
| 20 |
+
except (OSError, json.JSONDecodeError):
|
| 21 |
+
return None
|
| 22 |
+
|
| 23 |
+
|
| 24 |
+
def training_section(run: Path, has_probe: bool = True) -> list[str]:
|
| 25 |
+
history = read(run / "history.json")
|
| 26 |
+
if not history:
|
| 27 |
+
return []
|
| 28 |
+
out = ["## Training curve", "",
|
| 29 |
+
"Frozen validation split, never sampled during training.", "",
|
| 30 |
+
"`conv field R²` scores `supervised_field()` -- the conv head the "
|
| 31 |
+
"field loss trains. It is **not** the field the controller "
|
| 32 |
+
"descends. Under `--min-plus-field` those are different objects: "
|
| 33 |
+
"the read-out is the min-plus recurrence" +
|
| 34 |
+
(", scored as `read-out R²` in the probe table above" if has_probe
|
| 35 |
+
else "") +
|
| 36 |
+
", and nothing at inference reads the conv "
|
| 37 |
+
"head at all. It stays in the loss as an auxiliary task on the "
|
| 38 |
+
"shared trunk, so a negative value here means that auxiliary head "
|
| 39 |
+
"has stopped tracking the trunk the action head and the recurrence "
|
| 40 |
+
"are shaping -- it does not mean the planner is wrong, and `reach` "
|
| 41 |
+
"is where that would show up.", "",
|
| 42 |
+
"| step | loss | acc | choice | boolean | score | conv field R² | maze action | snake action |",
|
| 43 |
+
"| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |"]
|
| 44 |
+
for row in history:
|
| 45 |
+
out.append(
|
| 46 |
+
f"| {row['step']} | {row['loss']:.4f} | {row['acc']:.4f} | "
|
| 47 |
+
f"{row.get('acc_choice', float('nan')):.4f} | "
|
| 48 |
+
f"{row.get('acc_boolean', float('nan')):.4f} | "
|
| 49 |
+
f"{row.get('acc_score', float('nan')):.4f} | "
|
| 50 |
+
f"{row.get('field_r2', float('nan')):.4f} | "
|
| 51 |
+
f"{row.get('maze_action', float('nan')):.4f} | "
|
| 52 |
+
f"{row.get('snake_action', float('nan')):.4f} |")
|
| 53 |
+
return out + [""]
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
def eval_section(data) -> list[str]:
|
| 57 |
+
if not data:
|
| 58 |
+
return []
|
| 59 |
+
out = ["## Question-level accuracy", "",
|
| 60 |
+
"`uniform` is the accuracy of guessing uniformly over the same "
|
| 61 |
+
"candidates. A model that matches its control has learned nothing, "
|
| 62 |
+
"which is the failure NanoJev's own log records for its maze model.", "",
|
| 63 |
+
"`boards` is how many distinct boards the row's questions came "
|
| 64 |
+
"from, and it is there so that `n` can be trusted. Questions that "
|
| 65 |
+
"share a board are not independent tests of whether the model can "
|
| 66 |
+
"read a board -- they share walls, goal and distance field, so "
|
| 67 |
+
"they succeed or fail together -- and a row reporting `n` alone "
|
| 68 |
+
"cannot tell you whether that is happening. Here it is not: the "
|
| 69 |
+
"generator draws a fresh board per state, so `boards` tracks `n` "
|
| 70 |
+
"on every split and `n` is the honest denominator. A row where "
|
| 71 |
+
"the two diverge is one whose precision should be read off "
|
| 72 |
+
"`boards`.", ""]
|
| 73 |
+
for split, block in data.items():
|
| 74 |
+
out += [f"### {split} — {block['questions']} questions, ECE {block['ece']:.4f}", "",
|
| 75 |
+
"| group | n | boards | accuracy | uniform control | TVD | Brier |",
|
| 76 |
+
"| --- | ---: | ---: | ---: | ---: | ---: | ---: |"]
|
| 77 |
+
for name, row in block["groups"].items():
|
| 78 |
+
boards = row.get("boards")
|
| 79 |
+
out.append(f"| `{name}` | {row['n']} | "
|
| 80 |
+
f"{'--' if boards is None else boards} | "
|
| 81 |
+
f"**{row['accuracy']:.4f}** | "
|
| 82 |
+
f"{row['uniform_control']:.4f} | {row['tvd']:.4f} | "
|
| 83 |
+
f"{row['brier']:.4f} |")
|
| 84 |
+
out.append("")
|
| 85 |
+
return out
|
| 86 |
+
|
| 87 |
+
|
| 88 |
+
def probe_section(data) -> list[str]:
|
| 89 |
+
if not data:
|
| 90 |
+
return []
|
| 91 |
+
out = ["## Does the planner actually compute distances?", "",
|
| 92 |
+
"Ridge probe from the planner's per-cell features to true BFS "
|
| 93 |
+
"distance, on held-out mazes. `diameter` is the longest true "
|
| 94 |
+
"shortest-path in those boards; `T` is the iteration budget.", "",
|
| 95 |
+
"`descent` is the fraction of cells whose lowest-predicted "
|
| 96 |
+
"neighbour lies on a true shortest path. `reach` is the fraction "
|
| 97 |
+
"from which greedy descent actually arrives at the goal, and it "
|
| 98 |
+
"is the one that predicts solve rate: a field can be right 70% of "
|
| 99 |
+
"the time per step and still trap most walks in a spurious basin, "
|
| 100 |
+
"so `descent` is an upper bound on nothing in particular. R² "
|
| 101 |
+
"scores absolute values; only `reach` scores the behaviour.", "",
|
| 102 |
+
"`read-out R²` scores `predict_field()`, the field this "
|
| 103 |
+
"controller actually descends -- the min-plus recurrence when "
|
| 104 |
+
"the run enabled it, the conv head otherwise. The training "
|
| 105 |
+
"curve's `conv field R²` scores a different head; see the "
|
| 106 |
+
"note there.", "",
|
| 107 |
+
"| size | cells | probe R² | probe MAE (cells) | read-out R² | descent | reach | diameter | T |",
|
| 108 |
+
"| ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |"]
|
| 109 |
+
for size, row in data.items():
|
| 110 |
+
out.append(f"| {size} | {row['cells']} | **{row['probe_r2']:.4f}** | "
|
| 111 |
+
f"{row['probe_mae']:.2f} | {row['head_r2']:.4f} | "
|
| 112 |
+
f"{row.get('descent_acc', float('nan')):.4f} | "
|
| 113 |
+
f"**{row.get('reach_rate', float('nan')):.4f}** | "
|
| 114 |
+
f"{row['diameter']} | {row['iterations']} |")
|
| 115 |
+
return out + [""]
|
| 116 |
+
|
| 117 |
+
|
| 118 |
+
# Reading order, not alphabetical order. Sorting by filename puts
|
| 119 |
+
# `maze_random_memory` three rows below the `maze_model_memory` it is the
|
| 120 |
+
# control for, and `model+memory` above the plain `model` it is scaffolding
|
| 121 |
+
# over. Anything not listed here still renders, at the end, so adding a play
|
| 122 |
+
# config can never silently drop it from the table.
|
| 123 |
+
PLAY_ORDER = ["maze_model", "maze_model_sampled", "maze_model_field",
|
| 124 |
+
"maze_model_memory", "maze_random_memory", "maze_reference",
|
| 125 |
+
"maze_random", "snake_model", "snake_reference", "snake_random"]
|
| 126 |
+
|
| 127 |
+
|
| 128 |
+
def play_section(summaries: dict) -> list[str]:
|
| 129 |
+
maze, snake = [], []
|
| 130 |
+
ordered = sorted(summaries.items(),
|
| 131 |
+
key=lambda kv: (PLAY_ORDER.index(kv[0])
|
| 132 |
+
if kv[0] in PLAY_ORDER else len(PLAY_ORDER),
|
| 133 |
+
kv[0]))
|
| 134 |
+
for label, data in ordered:
|
| 135 |
+
block = data.get("summary", data) if data else {}
|
| 136 |
+
if "maze" in block:
|
| 137 |
+
m = block["maze"]
|
| 138 |
+
steps = m.get("mean_steps_when_solved")
|
| 139 |
+
dead = m.get("deadlock_rate")
|
| 140 |
+
maze.append(f"| `{label}` | {m['solve_rate']:.2f} | "
|
| 141 |
+
f"{'--' if steps is None else format(steps, '.1f')} | "
|
| 142 |
+
f"{m['mean_efficiency']:.3f} | {m['mean_collisions']:.1f} | "
|
| 143 |
+
f"{'--' if dead is None else format(dead, '.2f')} | "
|
| 144 |
+
f"{m.get('mean_distinct_cells', float('nan')):.1f} |")
|
| 145 |
+
if "snake" in block:
|
| 146 |
+
s = block["snake"]
|
| 147 |
+
snake.append(f"| `{label}` | {s['mean_food']:.2f} | {s['max_food']} | "
|
| 148 |
+
f"{s['mean_steps']:.1f} | {s['survival_rate']:.2f} |")
|
| 149 |
+
out = []
|
| 150 |
+
if maze:
|
| 151 |
+
out += ["## Closed-loop play", "",
|
| 152 |
+
"Read `model` and `model-field` as answering different "
|
| 153 |
+
"questions. `model` is the network's action head alone -- no "
|
| 154 |
+
"search, no memory, no visited set -- and is the "
|
| 155 |
+
"apples-to-apples comparison with NanoJev. `model-field` "
|
| 156 |
+
"descends a min-plus recurrence whose fixed point is a "
|
| 157 |
+
"shortest-path distance by construction, so it arrives from "
|
| 158 |
+
"anywhere; for a maze it does so at initialisation, before "
|
| 159 |
+
"any training, because uniform cost is already the right "
|
| 160 |
+
"answer. Its solve rate is a property of the architecture, "
|
| 161 |
+
"not a measurement of what this run learned.", "",
|
| 162 |
+
"### Maze", "",
|
| 163 |
+
"`deadlock` is the fraction of episodes in which the "
|
| 164 |
+
"controller spent a tenth of its step budget walking into the "
|
| 165 |
+
"same wall, and `cells` is how many distinct squares it stood "
|
| 166 |
+
"on. They separate two failures a solve rate of 0.00 reports "
|
| 167 |
+
"identically: standing still and touring the board. Deadlock "
|
| 168 |
+
"should now read 0.00 for every controller, because only legal "
|
| 169 |
+
"moves are offered as candidates -- it is kept as a regression "
|
| 170 |
+
"guard, not a finding. The failure that remains is cycling: "
|
| 171 |
+
"the planner never sees the agent, so a greedy policy that "
|
| 172 |
+
"steps A->B->A has no state with which to notice, and `cells` "
|
| 173 |
+
"is what exposes it.", "",
|
| 174 |
+
"| controller | solve rate | mean steps | efficiency | collisions | deadlock | cells |",
|
| 175 |
+
"| --- | ---: | ---: | ---: | ---: | ---: | ---: |"] + maze + [""]
|
| 176 |
+
if snake:
|
| 177 |
+
out += ["### Snake", "",
|
| 178 |
+
"| controller | mean food | max food | mean steps | survival |",
|
| 179 |
+
"| --- | ---: | ---: | ---: | ---: |"] + snake + [""]
|
| 180 |
+
return out
|
| 181 |
+
|
| 182 |
+
|
| 183 |
+
def nanojev_section() -> list[str]:
|
| 184 |
+
"""The numbers NanoJev reports for itself, quoted from its own repo.
|
| 185 |
+
|
| 186 |
+
Transcribed from docs/DEVELOPMENT_RESULTS.md in TianyuCodings/NanoJev so
|
| 187 |
+
the comparison is against what its authors measured, not a rerun of ours.
|
| 188 |
+
The point of the table is the control column: where a reported accuracy
|
| 189 |
+
equals the always-true rate, the number is reporting the class balance.
|
| 190 |
+
"""
|
| 191 |
+
return [
|
| 192 |
+
"## Baseline: what NanoJev reports for itself", "",
|
| 193 |
+
"From `docs/DEVELOPMENT_RESULTS.md` in the NanoJev repository.", "",
|
| 194 |
+
"| NanoJev task | reported acc | always-true control | above control |",
|
| 195 |
+
"| --- | ---: | ---: | :---: |",
|
| 196 |
+
"| `scaled_maze` (test) | 0.5625 | 0.5625 | no |",
|
| 197 |
+
"| `scaled_maze` (ood) | 0.5625 | 0.5625 | no |",
|
| 198 |
+
"| `snake_one_step_safety_v4` | 0.9220 | 0.9220 | no |",
|
| 199 |
+
"",
|
| 200 |
+
"Closed-loop play, same source, 128-step limit:", "",
|
| 201 |
+
"| maze | Jev | NanoJev | reference |",
|
| 202 |
+
"| --- | ---: | ---: | ---: |",
|
| 203 |
+
"| all three mazes | 0/128 | 0/128 | solved 1/16, 1/24, 1/96 |",
|
| 204 |
+
"",
|
| 205 |
+
"> \"The learned direct-action models and Jev do not solve any of these",
|
| 206 |
+
"> three mazes within the shared 128-step limit.\"", "",
|
| 207 |
+
"Every accuracy in this file is therefore printed next to its",
|
| 208 |
+
"constant-prediction control, and gameplay is reported as solve rate",
|
| 209 |
+
"rather than per-question accuracy.", "",
|
| 210 |
+
]
|
| 211 |
+
|
| 212 |
+
|
| 213 |
+
def main() -> None:
|
| 214 |
+
p = argparse.ArgumentParser(description=__doc__)
|
| 215 |
+
p.add_argument("--run", default="runs/jevon-final")
|
| 216 |
+
p.add_argument("--eval", default="runs/jevon-final/eval.json")
|
| 217 |
+
p.add_argument("--probe", default="runs/jevon-final/probe.json")
|
| 218 |
+
p.add_argument("--play-dir", default="runs/jevon-final/play")
|
| 219 |
+
p.add_argument("--out", default="RESULTS.md")
|
| 220 |
+
args = p.parse_args()
|
| 221 |
+
|
| 222 |
+
summaries = {}
|
| 223 |
+
play_dir = Path(args.play_dir)
|
| 224 |
+
if play_dir.is_dir():
|
| 225 |
+
for path in sorted(play_dir.glob("*_summary.json")):
|
| 226 |
+
summaries[path.stem.replace("_summary", "")] = read(path)
|
| 227 |
+
|
| 228 |
+
lines = ["# Results", "",
|
| 229 |
+
"Generated by `scripts/make_results.py` from the JSON written by "
|
| 230 |
+
"`train.py`, `evaluate.py`, `probe.py` and `play.py`. Do not edit "
|
| 231 |
+
"by hand.", ""]
|
| 232 |
+
lines += nanojev_section()
|
| 233 |
+
probe = read(Path(args.probe))
|
| 234 |
+
lines += probe_section(probe)
|
| 235 |
+
lines += eval_section(read(Path(args.eval)))
|
| 236 |
+
lines += play_section(summaries)
|
| 237 |
+
lines += training_section(Path(args.run), has_probe=bool(probe))
|
| 238 |
+
|
| 239 |
+
Path(args.out).write_text("\n".join(lines))
|
| 240 |
+
print(f"wrote {args.out} ({len(lines)} lines)")
|
| 241 |
+
|
| 242 |
+
|
| 243 |
+
if __name__ == "__main__":
|
| 244 |
+
main()
|
|
@@ -0,0 +1,413 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Closed-loop gameplay evaluation.
|
| 3 |
+
|
| 4 |
+
Controllers are kept deliberately separable so it is clear what is doing the
|
| 5 |
+
work. ``model`` asks the network for the action distribution and does nothing
|
| 6 |
+
else -- no search, no visited set, no memory. ``model+memory`` adds the thin
|
| 7 |
+
bookkeeping a deployed agent would have, which is roughly what NanoJev's
|
| 8 |
+
showcase controller provides, and is reported separately rather than folded
|
| 9 |
+
into the headline number.
|
| 10 |
+
|
| 11 |
+
``random+memory`` is that second row's control, and the reason it exists: the
|
| 12 |
+
memory is doing enough of the work that ``model+memory`` says very little on
|
| 13 |
+
its own. Running the identical bookkeeping with no network behind it says how
|
| 14 |
+
much.
|
| 15 |
+
"""
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
import argparse
|
| 19 |
+
import json
|
| 20 |
+
import math
|
| 21 |
+
import random
|
| 22 |
+
import sys
|
| 23 |
+
from pathlib import Path
|
| 24 |
+
|
| 25 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 26 |
+
sys.path.insert(0, str(ROOT))
|
| 27 |
+
|
| 28 |
+
import torch
|
| 29 |
+
|
| 30 |
+
from envs import maze as mz
|
| 31 |
+
from envs import snake as sk
|
| 32 |
+
from jevon.inference import JevonRunner, maze_state_sample, snake_state_sample
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
# ---------------------------------------------------------------- policies
|
| 36 |
+
|
| 37 |
+
def model_action(runner, sample, legal, rng, temperature):
|
| 38 |
+
answers = {a["question"]: a for a in runner.answer([sample])}
|
| 39 |
+
row = answers.get("action")
|
| 40 |
+
if row is None:
|
| 41 |
+
# question_profile omits the action question when fewer than two moves
|
| 42 |
+
# are legal, because there is nothing to choose between. The answer is
|
| 43 |
+
# then forced, and taking it is not a guess -- but taking legal[0] is.
|
| 44 |
+
# That fallback used to fire at every one-exit cell and return `north`,
|
| 45 |
+
# a wall, so the agent never moved and was handed the identical state
|
| 46 |
+
# forever. Every maze figure for this controller measured that, not the
|
| 47 |
+
# network.
|
| 48 |
+
return (legal[0] if len(legal) == 1 else rng.choice(legal)), {}
|
| 49 |
+
distribution = {k: v for k, v in row["probabilities"].items() if k in legal}
|
| 50 |
+
if not distribution:
|
| 51 |
+
return rng.choice(legal), row["probabilities"]
|
| 52 |
+
if temperature <= 0:
|
| 53 |
+
return max(distribution, key=distribution.get), distribution
|
| 54 |
+
keys = list(distribution)
|
| 55 |
+
weights = [max(distribution[k], 1e-9) ** (1.0 / temperature) for k in keys]
|
| 56 |
+
return rng.choices(keys, weights=weights, k=1)[0], distribution
|
| 57 |
+
|
| 58 |
+
|
| 59 |
+
def field_action(runner, sample, state, legal, rng):
|
| 60 |
+
"""Move to the neighbouring cell the planner thinks is closest to the goal.
|
| 61 |
+
|
| 62 |
+
Pure greedy descent on the model's own learned field -- no search, no
|
| 63 |
+
memory, no visited set. It reads the planner's training read-out instead of
|
| 64 |
+
the text-conditioned action head, so it measures whether the recurrence
|
| 65 |
+
learned a usable value function rather than whether the language side
|
| 66 |
+
learned to describe one.
|
| 67 |
+
"""
|
| 68 |
+
field = runner.distance_field(sample)
|
| 69 |
+
scores = {}
|
| 70 |
+
for action in legal:
|
| 71 |
+
dr, dc = mz.DIRECTIONS[action]
|
| 72 |
+
cell = (state.position[0] + dr, state.position[1] + dc)
|
| 73 |
+
if not state.free(cell):
|
| 74 |
+
continue
|
| 75 |
+
scores[action] = float(field[cell])
|
| 76 |
+
if not scores:
|
| 77 |
+
return rng.choice(legal), {}
|
| 78 |
+
best = min(scores.values())
|
| 79 |
+
# Report it as a distribution so traces render the same way as the others.
|
| 80 |
+
weights = {a: math.exp(-(v - best) * 8.0) for a, v in scores.items()}
|
| 81 |
+
total = sum(weights.values())
|
| 82 |
+
return min(scores, key=scores.get), {a: w / total for a, w in weights.items()}
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def episode_field(runner, state):
|
| 86 |
+
"""The model's predicted distance-to-goal for every cell, once per episode.
|
| 87 |
+
|
| 88 |
+
The planner never sees the agent, so this is constant for the whole
|
| 89 |
+
episode -- the same property that lets the controller cache it. Recorded
|
| 90 |
+
so the viewer can show what the model believes about the board, not just
|
| 91 |
+
which move it picked.
|
| 92 |
+
"""
|
| 93 |
+
if runner is None:
|
| 94 |
+
return None
|
| 95 |
+
try:
|
| 96 |
+
field = runner.distance_field(maze_state_sample(state))
|
| 97 |
+
except RuntimeError: # checkpoint trained without a planner
|
| 98 |
+
return None
|
| 99 |
+
return [[round(float(v), 4) for v in row] for row in field]
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
class Recorder(list):
|
| 103 |
+
"""A step log with a ceiling, so a trace stays viewable without lying.
|
| 104 |
+
|
| 105 |
+
A random walk on a 31x31 maze runs to its 3844-step cap, and thirty-six of
|
| 106 |
+
those come to roughly six megabytes of JSON that nobody will ever scrub
|
| 107 |
+
through. Truncating the *recording* keeps the episode itself untouched, so
|
| 108 |
+
the summary -- solve rate, step count, collisions -- is still the real one;
|
| 109 |
+
only the replay is shortened, and ``truncated`` says by how much.
|
| 110 |
+
"""
|
| 111 |
+
|
| 112 |
+
def __init__(self, cap: int = 0):
|
| 113 |
+
super().__init__()
|
| 114 |
+
self.cap = cap
|
| 115 |
+
self.seen = 0
|
| 116 |
+
|
| 117 |
+
def append(self, item) -> None:
|
| 118 |
+
self.seen += 1
|
| 119 |
+
if not self.cap or len(self) < self.cap:
|
| 120 |
+
super().append(item)
|
| 121 |
+
|
| 122 |
+
@property
|
| 123 |
+
def truncated(self) -> int:
|
| 124 |
+
return self.seen - len(self)
|
| 125 |
+
|
| 126 |
+
|
| 127 |
+
def maze_steps(runner, state, controller, max_steps, rng, temperature=0.0):
|
| 128 |
+
"""One decision per ``next()``, yielding the step it just recorded.
|
| 129 |
+
|
| 130 |
+
Written as a generator so the live server in the arcade repo can take a
|
| 131 |
+
single forward pass per request while ``play_maze`` below drains it in a
|
| 132 |
+
loop. The alternative was a second copy of this decision logic behind an
|
| 133 |
+
HTTP handler, which is the failure mode these two repos have hit more than
|
| 134 |
+
any other -- and this is the loop every shipped number is measured from,
|
| 135 |
+
so a divergent copy would be a viewer demonstrating a model nobody
|
| 136 |
+
benchmarked. Returns the episode result, so callers need ``StopIteration``
|
| 137 |
+
or ``yield from``.
|
| 138 |
+
"""
|
| 139 |
+
# The set of cells stood on, read twice. It is the ``model+memory``
|
| 140 |
+
# controller's whole memory -- prefer a move that leaves ground already
|
| 141 |
+
# covered -- and it is also the wandering diagnostic, because an agent that
|
| 142 |
+
# visited one cell and an agent that toured forty both score 0.00.
|
| 143 |
+
#
|
| 144 |
+
# Memory used to mean remembering which (cell, action) pairs had bounced
|
| 145 |
+
# off a wall. Restricting the candidate set to legal moves made that set
|
| 146 |
+
# permanently empty, which would have left ``model+memory`` a silent
|
| 147 |
+
# duplicate of ``model`` and its row in the results table a lie. Cycles are
|
| 148 |
+
# what end these episodes now: the planner is stateless, so a greedy policy
|
| 149 |
+
# that steps A->B->A repeats it for the whole budget.
|
| 150 |
+
seen: set[tuple[int, int]] = {state.position}
|
| 151 |
+
stuck = longest_stuck = 0
|
| 152 |
+
while state.steps < max_steps and not state.solved():
|
| 153 |
+
# The action question offers only legal moves as criteria, so ranking
|
| 154 |
+
# all four here asks the model to score candidates it was never trained
|
| 155 |
+
# on and lets a wall win.
|
| 156 |
+
legal = state.legal_actions() or list(mz.ACTIONS)
|
| 157 |
+
# The memory, applied before any policy sees the list, so `model+memory`
|
| 158 |
+
# and `random+memory` differ in exactly one thing: who picks among the
|
| 159 |
+
# survivors. Anything `random+memory` already achieves is the memory's
|
| 160 |
+
# doing, not the network's.
|
| 161 |
+
options = legal
|
| 162 |
+
if controller.endswith("+memory"):
|
| 163 |
+
r, c = state.position
|
| 164 |
+
fresh = []
|
| 165 |
+
for a in legal:
|
| 166 |
+
dr, dc = mz.DIRECTIONS[a]
|
| 167 |
+
if (r + dr, c + dc) not in seen:
|
| 168 |
+
fresh.append(a)
|
| 169 |
+
options = fresh or legal
|
| 170 |
+
|
| 171 |
+
if controller == "reference":
|
| 172 |
+
probs = mz.optimal_action_distribution(state)
|
| 173 |
+
action = max(probs, key=probs.get) if probs else rng.choice(legal)
|
| 174 |
+
distribution = probs
|
| 175 |
+
elif controller.startswith("random"):
|
| 176 |
+
action, distribution = rng.choice(options), {}
|
| 177 |
+
elif controller == "model-field":
|
| 178 |
+
sample = maze_state_sample(state)
|
| 179 |
+
action, distribution = field_action(runner, sample, state, legal, rng)
|
| 180 |
+
else:
|
| 181 |
+
sample = maze_state_sample(state)
|
| 182 |
+
action, distribution = model_action(runner, sample, options, rng, temperature)
|
| 183 |
+
before = state.position
|
| 184 |
+
moved = state.step(action)
|
| 185 |
+
if not moved:
|
| 186 |
+
stuck += 1
|
| 187 |
+
longest_stuck = max(longest_stuck, stuck)
|
| 188 |
+
else:
|
| 189 |
+
stuck = 0
|
| 190 |
+
seen.add(state.position)
|
| 191 |
+
yield {"position": list(before), "action": action, "moved": moved,
|
| 192 |
+
"probabilities": {k: round(float(v), 4)
|
| 193 |
+
for k, v in (distribution or {}).items()}}
|
| 194 |
+
return {"solved": state.solved(), "steps": state.steps,
|
| 195 |
+
"collisions": state.collisions, "distinct_cells": len(seen),
|
| 196 |
+
"longest_stuck": longest_stuck,
|
| 197 |
+
# Kept as a guard rather than a finding. It fired on every episode
|
| 198 |
+
# while blocked moves were being offered as candidates; with only
|
| 199 |
+
# legal moves on the list it should stay at zero, so a non-zero
|
| 200 |
+
# value means something has regressed.
|
| 201 |
+
"deadlocked": longest_stuck >= max(20, max_steps // 10)}
|
| 202 |
+
|
| 203 |
+
|
| 204 |
+
def snake_steps(runner, state, controller, max_steps, rng, temperature=0.0):
|
| 205 |
+
"""One decision per ``next()``. See ``maze_steps``."""
|
| 206 |
+
while state.alive and state.steps < max_steps:
|
| 207 |
+
legal = state.candidate_actions()
|
| 208 |
+
if controller == "reference":
|
| 209 |
+
probs, _ = sk.oracle_action_distribution(state)
|
| 210 |
+
action = max(probs, key=probs.get) if probs else rng.choice(legal)
|
| 211 |
+
distribution = probs
|
| 212 |
+
elif controller == "random":
|
| 213 |
+
action, distribution = rng.choice(legal), {}
|
| 214 |
+
else:
|
| 215 |
+
sample = snake_state_sample(state)
|
| 216 |
+
action, distribution = model_action(runner, sample, legal, rng, temperature)
|
| 217 |
+
head, food, body = state.head, state.food, list(state.body)
|
| 218 |
+
state.step(action)
|
| 219 |
+
yield {"head": list(head), "food": list(food),
|
| 220 |
+
"body": [list(c) for c in body], "action": action,
|
| 221 |
+
"probabilities": {k: round(float(v), 4)
|
| 222 |
+
for k, v in (distribution or {}).items()}}
|
| 223 |
+
return {"food": state.eaten, "steps": state.steps, "alive": state.alive,
|
| 224 |
+
"death": state.death, "length": len(state.body)}
|
| 225 |
+
|
| 226 |
+
|
| 227 |
+
def default_max_steps(game: str, size: int) -> int:
|
| 228 |
+
"""The step budget an episode gets when none is given.
|
| 229 |
+
|
| 230 |
+
Lives here rather than in each caller because the arcade's live server
|
| 231 |
+
plays the same games through the same generators, and a budget that
|
| 232 |
+
differed between the two would make a live episode and a recorded one
|
| 233 |
+
disagree about what "failed" means -- the model looks worse or better
|
| 234 |
+
purely by where it was run.
|
| 235 |
+
"""
|
| 236 |
+
return (4 if game == "maze" else 2) * size * size
|
| 237 |
+
|
| 238 |
+
|
| 239 |
+
def _drain(steps, record):
|
| 240 |
+
"""Run a step generator to the end, recording as it goes.
|
| 241 |
+
|
| 242 |
+
The wrappers below keep ``play.py``'s own callers unchanged: the generator
|
| 243 |
+
is the shared thing, and running it to completion is what batch recording
|
| 244 |
+
happens to want.
|
| 245 |
+
"""
|
| 246 |
+
while True:
|
| 247 |
+
try:
|
| 248 |
+
step = next(steps)
|
| 249 |
+
except StopIteration as stop:
|
| 250 |
+
return stop.value
|
| 251 |
+
if record is not None:
|
| 252 |
+
record.append(step)
|
| 253 |
+
|
| 254 |
+
|
| 255 |
+
def play_maze(runner, state, controller, max_steps, rng, temperature=0.0,
|
| 256 |
+
record=None):
|
| 257 |
+
return _drain(maze_steps(runner, state, controller, max_steps, rng,
|
| 258 |
+
temperature), record)
|
| 259 |
+
|
| 260 |
+
|
| 261 |
+
def play_snake(runner, state, controller, max_steps, rng, temperature=0.0,
|
| 262 |
+
record=None):
|
| 263 |
+
return _drain(snake_steps(runner, state, controller, max_steps, rng,
|
| 264 |
+
temperature), record)
|
| 265 |
+
|
| 266 |
+
|
| 267 |
+
# ---------------------------------------------------------------- driver
|
| 268 |
+
|
| 269 |
+
def main() -> None:
|
| 270 |
+
p = argparse.ArgumentParser(description=__doc__)
|
| 271 |
+
p.add_argument("--checkpoint", default="runs/jevon-final")
|
| 272 |
+
# None means "prefer balanced.pt", the checkpoint every shipped number
|
| 273 |
+
# comes from. See jevon.inference.resolve_weights.
|
| 274 |
+
p.add_argument("--weights", default=None)
|
| 275 |
+
p.add_argument("--game", choices=("maze", "snake", "both"), default="both")
|
| 276 |
+
p.add_argument("--controller", default="model",
|
| 277 |
+
choices=("model", "model+memory", "model-field",
|
| 278 |
+
"reference", "random", "random+memory"))
|
| 279 |
+
p.add_argument("--episodes", type=int, default=12)
|
| 280 |
+
p.add_argument("--maze-sizes", default="11,21,31,51")
|
| 281 |
+
p.add_argument("--maze-topologies", default="corridor,tree,loops,random_obstacle")
|
| 282 |
+
p.add_argument("--snake-sizes", default="12")
|
| 283 |
+
p.add_argument("--max-steps", type=int, default=0, help="0 => 4*size^2 for maze, 2*size^2 for snake")
|
| 284 |
+
p.add_argument("--temperature", type=float, default=0.0)
|
| 285 |
+
p.add_argument("--seed", type=int, default=17)
|
| 286 |
+
p.add_argument("--device", default="mps")
|
| 287 |
+
p.add_argument("--out", default=None)
|
| 288 |
+
p.add_argument("--trace-out", default=None, help="write per-step replay JSON")
|
| 289 |
+
p.add_argument("--trace-max-steps", type=int, default=0,
|
| 290 |
+
help="record at most this many steps per episode; the "
|
| 291 |
+
"episode still runs to completion, so summaries are "
|
| 292 |
+
"unaffected (0 => record everything)")
|
| 293 |
+
args = p.parse_args()
|
| 294 |
+
|
| 295 |
+
if args.controller == "model-field" and args.game != "maze":
|
| 296 |
+
# Snake's field is distance to food, which is the wrong thing to
|
| 297 |
+
# descend greedily -- survival dominates. Silently falling back to
|
| 298 |
+
# the action head would mislabel a row in the results table.
|
| 299 |
+
p.error("--controller model-field only applies to --game maze")
|
| 300 |
+
|
| 301 |
+
rng = random.Random(args.seed)
|
| 302 |
+
runner = None
|
| 303 |
+
if args.controller.startswith("model"):
|
| 304 |
+
runner = JevonRunner(args.checkpoint, args.device, args.weights)
|
| 305 |
+
|
| 306 |
+
results = {"controller": args.controller, "episodes": []}
|
| 307 |
+
traces = []
|
| 308 |
+
|
| 309 |
+
if args.game in ("maze", "both"):
|
| 310 |
+
for size in [int(s) for s in args.maze_sizes.split(",")]:
|
| 311 |
+
for topology in args.maze_topologies.split(","):
|
| 312 |
+
for i in range(args.episodes):
|
| 313 |
+
seed = 900_000 + i * 37 + size * 101
|
| 314 |
+
state = mz.generate_maze(size, topology, seed)
|
| 315 |
+
shortest = mz.bfs_distances(size, state.walls,
|
| 316 |
+
state.goal)[state.position]
|
| 317 |
+
cap = args.max_steps or default_max_steps("maze", size)
|
| 318 |
+
record = Recorder(args.trace_max_steps) if args.trace_out else None
|
| 319 |
+
# Before play, because play mutates the agent's position --
|
| 320 |
+
# the field itself does not depend on it, but recording it
|
| 321 |
+
# first keeps that independence obvious rather than assumed.
|
| 322 |
+
field = episode_field(runner, state) if record is not None else None
|
| 323 |
+
out = play_maze(runner, state, args.controller, cap, rng,
|
| 324 |
+
args.temperature, record)
|
| 325 |
+
out.update(game="maze", size=size, topology=topology, seed=seed,
|
| 326 |
+
shortest=shortest,
|
| 327 |
+
efficiency=shortest / out["steps"] if out["solved"] else 0.0)
|
| 328 |
+
results["episodes"].append(out)
|
| 329 |
+
if record is not None:
|
| 330 |
+
traces.append({"game": "maze", "size": size,
|
| 331 |
+
"topology": topology, "seed": seed,
|
| 332 |
+
"walls": sorted(map(list, state.walls)),
|
| 333 |
+
"goal": list(state.goal), "result": out,
|
| 334 |
+
"field": field, "steps": list(record),
|
| 335 |
+
"truncated": record.truncated})
|
| 336 |
+
print(f"maze {size:3d} {topology:16s} seed={seed} "
|
| 337 |
+
f"solved={out['solved']} steps={out['steps']:5d} "
|
| 338 |
+
f"shortest={shortest} collisions={out['collisions']}",
|
| 339 |
+
flush=True)
|
| 340 |
+
|
| 341 |
+
if args.game in ("snake", "both"):
|
| 342 |
+
for size in [int(s) for s in args.snake_sizes.split(",")]:
|
| 343 |
+
for i in range(args.episodes):
|
| 344 |
+
seed = 610_000 + i * 53 + size * 7
|
| 345 |
+
state = sk.new_game(size, seed)
|
| 346 |
+
cap = args.max_steps or default_max_steps("snake", size)
|
| 347 |
+
record = Recorder(args.trace_max_steps) if args.trace_out else None
|
| 348 |
+
out = play_snake(runner, state, args.controller, cap, rng,
|
| 349 |
+
args.temperature, record)
|
| 350 |
+
out.update(game="snake", size=size, seed=seed)
|
| 351 |
+
results["episodes"].append(out)
|
| 352 |
+
if record is not None:
|
| 353 |
+
traces.append({"game": "snake", "size": size, "seed": seed,
|
| 354 |
+
"result": out, "steps": list(record),
|
| 355 |
+
"truncated": record.truncated})
|
| 356 |
+
print(f"snake {size:3d} seed={seed} food={out['food']:3d} "
|
| 357 |
+
f"steps={out['steps']:4d} alive={out['alive']} "
|
| 358 |
+
f"death={out['death']}", flush=True)
|
| 359 |
+
|
| 360 |
+
maze_eps = [e for e in results["episodes"] if e["game"] == "maze"]
|
| 361 |
+
snake_eps = [e for e in results["episodes"] if e["game"] == "snake"]
|
| 362 |
+
summary = {}
|
| 363 |
+
if maze_eps:
|
| 364 |
+
solved = [e for e in maze_eps if e["solved"]]
|
| 365 |
+
summary["maze"] = {
|
| 366 |
+
"episodes": len(maze_eps),
|
| 367 |
+
"solve_rate": len(solved) / len(maze_eps),
|
| 368 |
+
"mean_steps_when_solved": (sum(e["steps"] for e in solved) / len(solved)
|
| 369 |
+
if solved else None),
|
| 370 |
+
"mean_efficiency": (sum(e["efficiency"] for e in solved) / len(solved)
|
| 371 |
+
if solved else 0.0),
|
| 372 |
+
"mean_collisions": sum(e["collisions"] for e in maze_eps) / len(maze_eps),
|
| 373 |
+
"deadlock_rate": sum(e.get("deadlocked", False)
|
| 374 |
+
for e in maze_eps) / len(maze_eps),
|
| 375 |
+
"mean_distinct_cells": sum(e.get("distinct_cells", 0)
|
| 376 |
+
for e in maze_eps) / len(maze_eps),
|
| 377 |
+
"by_size": {},
|
| 378 |
+
}
|
| 379 |
+
for size in sorted({e["size"] for e in maze_eps}):
|
| 380 |
+
rows = [e for e in maze_eps if e["size"] == size]
|
| 381 |
+
ok = [e for e in rows if e["solved"]]
|
| 382 |
+
summary["maze"]["by_size"][str(size)] = {
|
| 383 |
+
"episodes": len(rows),
|
| 384 |
+
"solve_rate": len(ok) / len(rows),
|
| 385 |
+
"mean_steps": sum(e["steps"] for e in ok) / len(ok) if ok else None,
|
| 386 |
+
# Over solved episodes only, matching the aggregate above: an
|
| 387 |
+
# unsolved episode has efficiency 0 by definition, so including
|
| 388 |
+
# them would make this a second, noisier solve rate rather than
|
| 389 |
+
# a statement about path length. It is here because the
|
| 390 |
+
# model-vs-control comparison separates on efficiency and not
|
| 391 |
+
# on arrival, and that comparison is per size.
|
| 392 |
+
"mean_efficiency": (sum(e["efficiency"] for e in ok) / len(ok)
|
| 393 |
+
if ok else 0.0),
|
| 394 |
+
"mean_shortest": sum(e["shortest"] for e in rows) / len(rows)}
|
| 395 |
+
if snake_eps:
|
| 396 |
+
summary["snake"] = {
|
| 397 |
+
"episodes": len(snake_eps),
|
| 398 |
+
"mean_food": sum(e["food"] for e in snake_eps) / len(snake_eps),
|
| 399 |
+
"max_food": max(e["food"] for e in snake_eps),
|
| 400 |
+
"mean_steps": sum(e["steps"] for e in snake_eps) / len(snake_eps),
|
| 401 |
+
"survival_rate": sum(e["alive"] for e in snake_eps) / len(snake_eps)}
|
| 402 |
+
results["summary"] = summary
|
| 403 |
+
print(json.dumps(summary, indent=2))
|
| 404 |
+
if args.out:
|
| 405 |
+
Path(args.out).parent.mkdir(parents=True, exist_ok=True)
|
| 406 |
+
Path(args.out).write_text(json.dumps(results, indent=2))
|
| 407 |
+
if args.trace_out:
|
| 408 |
+
Path(args.trace_out).parent.mkdir(parents=True, exist_ok=True)
|
| 409 |
+
Path(args.trace_out).write_text(json.dumps(traces))
|
| 410 |
+
|
| 411 |
+
|
| 412 |
+
if __name__ == "__main__":
|
| 413 |
+
main()
|
|
@@ -0,0 +1,178 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Measure what the planner's field actually encodes.
|
| 3 |
+
|
| 4 |
+
Fits a ridge probe from the planner's per-cell features to the true BFS
|
| 5 |
+
distance and reports R^2. This is the diagnostic that caught the central
|
| 6 |
+
training bug: with only the sparse decision questions supervising it, the
|
| 7 |
+
recurrence scored R^2 = 0.21 -- it was running, it was differentiable, its
|
| 8 |
+
accuracy numbers looked reasonable, and it was not computing distances.
|
| 9 |
+
|
| 10 |
+
Also reports the field head's own predictions, which is what the decision
|
| 11 |
+
heads actually read.
|
| 12 |
+
"""
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
import argparse
|
| 16 |
+
import json
|
| 17 |
+
import sys
|
| 18 |
+
from pathlib import Path
|
| 19 |
+
|
| 20 |
+
ROOT = Path(__file__).resolve().parents[1]
|
| 21 |
+
sys.path.insert(0, str(ROOT))
|
| 22 |
+
|
| 23 |
+
import torch
|
| 24 |
+
|
| 25 |
+
import envs.maze as mz
|
| 26 |
+
from jevon.grid import bfs_field, encode_maze, pad_boards
|
| 27 |
+
from jevon.inference import resolve_weights
|
| 28 |
+
from jevon.modeling import JevonConfig, JevonModel
|
| 29 |
+
from jevon.planner import default_iterations
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
def ridge_r2(features: torch.Tensor, target: torch.Tensor) -> tuple[float, float]:
|
| 33 |
+
design = torch.cat([features, torch.ones(len(features), 1)], 1)
|
| 34 |
+
gram = design.T @ design + 1e-3 * torch.eye(design.shape[1])
|
| 35 |
+
weights = torch.linalg.lstsq(gram, design.T @ target.unsqueeze(1)).solution
|
| 36 |
+
pred = (design @ weights).squeeze(1)
|
| 37 |
+
ss_res = ((pred - target) ** 2).sum()
|
| 38 |
+
ss_tot = ((target - target.mean()) ** 2).sum()
|
| 39 |
+
return float(1 - ss_res / ss_tot), float((pred - target).abs().mean())
|
| 40 |
+
|
| 41 |
+
|
| 42 |
+
def descent_accuracy(head: torch.Tensor, truth: torch.Tensor) -> float:
|
| 43 |
+
"""Fraction of cells whose lowest-predicted neighbour is a correct step.
|
| 44 |
+
|
| 45 |
+
R^2 scores the field's absolute values, but greedy descent only ever
|
| 46 |
+
compares a cell's four neighbours, so a field can be numerically mediocre
|
| 47 |
+
and still route perfectly -- or fit well on average and still invert the
|
| 48 |
+
one comparison that matters. This is the quantity that predicts solve
|
| 49 |
+
rate, so it is measured rather than inferred from R^2.
|
| 50 |
+
"""
|
| 51 |
+
head = head.detach()
|
| 52 |
+
hits = total = 0
|
| 53 |
+
h, w = truth.shape
|
| 54 |
+
for r in range(h):
|
| 55 |
+
for c in range(w):
|
| 56 |
+
here = float(truth[r, c])
|
| 57 |
+
if here <= 0: # blocked, unreachable, or the goal
|
| 58 |
+
continue
|
| 59 |
+
options = [(rr, cc) for rr, cc in
|
| 60 |
+
((r - 1, c), (r + 1, c), (r, c - 1), (r, c + 1))
|
| 61 |
+
if 0 <= rr < h and 0 <= cc < w and truth[rr, cc] >= 0]
|
| 62 |
+
if not options:
|
| 63 |
+
continue
|
| 64 |
+
pick = min(options, key=lambda x: float(head[x[0], x[1]]))
|
| 65 |
+
# Correct iff that neighbour lies on *a* shortest path.
|
| 66 |
+
hits += float(truth[pick]) == here - 1
|
| 67 |
+
total += 1
|
| 68 |
+
return hits / max(total, 1)
|
| 69 |
+
|
| 70 |
+
|
| 71 |
+
def descent_outcomes(head: torch.Tensor, truth: torch.Tensor,
|
| 72 |
+
cap_mult: int = 8) -> float:
|
| 73 |
+
"""Fraction of cells from which greedy descent actually reaches the goal.
|
| 74 |
+
|
| 75 |
+
Per-step accuracy is not the quantity a controller lives or dies by. A
|
| 76 |
+
field that is right 99% of the time but contains one spurious basin traps
|
| 77 |
+
forever; a field that is right 70% of the time with no basin still arrives,
|
| 78 |
+
just by a longer route. Solve rate equals *this* number, so report it next
|
| 79 |
+
to the per-step figure and let the gap between them speak.
|
| 80 |
+
"""
|
| 81 |
+
head = head.detach()
|
| 82 |
+
h, w = truth.shape
|
| 83 |
+
goal = min(((r, c) for r in range(h) for c in range(w)
|
| 84 |
+
if float(truth[r, c]) == 0.0), default=None)
|
| 85 |
+
if goal is None:
|
| 86 |
+
return 0.0
|
| 87 |
+
reached = total = 0
|
| 88 |
+
cap = cap_mult * h * w
|
| 89 |
+
for r in range(h):
|
| 90 |
+
for c in range(w):
|
| 91 |
+
if float(truth[r, c]) < 0:
|
| 92 |
+
continue
|
| 93 |
+
total += 1
|
| 94 |
+
cur, seen = (r, c), set()
|
| 95 |
+
for _ in range(cap):
|
| 96 |
+
if cur == goal:
|
| 97 |
+
reached += 1
|
| 98 |
+
break
|
| 99 |
+
if cur in seen:
|
| 100 |
+
break # a cycle: descent never escapes
|
| 101 |
+
seen.add(cur)
|
| 102 |
+
options = [(rr, cc) for rr, cc in
|
| 103 |
+
((cur[0] - 1, cur[1]), (cur[0] + 1, cur[1]),
|
| 104 |
+
(cur[0], cur[1] - 1), (cur[0], cur[1] + 1))
|
| 105 |
+
if 0 <= rr < h and 0 <= cc < w and truth[rr, cc] >= 0]
|
| 106 |
+
if not options:
|
| 107 |
+
break
|
| 108 |
+
cur = min(options, key=lambda x: float(head[x[0], x[1]]))
|
| 109 |
+
return reached / max(total, 1)
|
| 110 |
+
|
| 111 |
+
|
| 112 |
+
def main() -> None:
|
| 113 |
+
p = argparse.ArgumentParser(description=__doc__)
|
| 114 |
+
p.add_argument("--checkpoint", default="runs/jevon-final")
|
| 115 |
+
# None means "prefer balanced.pt", the checkpoint every shipped number
|
| 116 |
+
# comes from. See jevon.inference.resolve_weights.
|
| 117 |
+
p.add_argument("--weights", default=None)
|
| 118 |
+
p.add_argument("--sizes", default="11,21,31,51")
|
| 119 |
+
p.add_argument("--episodes", type=int, default=4)
|
| 120 |
+
p.add_argument("--device", default="cpu")
|
| 121 |
+
p.add_argument("--out", default=None)
|
| 122 |
+
args = p.parse_args()
|
| 123 |
+
|
| 124 |
+
root = Path(args.checkpoint)
|
| 125 |
+
cfg = JevonConfig(**json.loads((root / "config.json").read_text()))
|
| 126 |
+
model = JevonModel(cfg)
|
| 127 |
+
model.load_state_dict(torch.load(resolve_weights(root, args.weights),
|
| 128 |
+
map_location="cpu"))
|
| 129 |
+
model.eval().to(args.device)
|
| 130 |
+
|
| 131 |
+
report = {}
|
| 132 |
+
print(f"{'size':>5s} {'cells':>7s} {'probe R2':>9s} {'probe MAE':>10s} "
|
| 133 |
+
f"{'head R2':>8s} {'descent':>8s} {'reach':>7s} "
|
| 134 |
+
f"{'diameter':>9s} {'T':>6s}")
|
| 135 |
+
for size in [int(s) for s in args.sizes.split(",")]:
|
| 136 |
+
feats, heads, dists, descents, reaches = [], [], [], [], []
|
| 137 |
+
diameter = 0
|
| 138 |
+
steps = default_iterations(size)
|
| 139 |
+
with torch.no_grad():
|
| 140 |
+
for i in range(args.episodes):
|
| 141 |
+
topology = mz.TOPOLOGIES[i % len(mz.TOPOLOGIES)]
|
| 142 |
+
# Seeds ending in 9 are the held-out test split.
|
| 143 |
+
state = mz.generate_maze(size, topology, 70_009 + i * 10)
|
| 144 |
+
board, _, target = encode_maze(state)
|
| 145 |
+
boards, passable, _ = pad_boards([board])
|
| 146 |
+
cells = model.encode_boards(boards.to(args.device),
|
| 147 |
+
passable.to(args.device), steps, 0)
|
| 148 |
+
head = model.predict_field(
|
| 149 |
+
cells, boards.to(args.device),
|
| 150 |
+
passable.to(args.device), iterations=steps)[0, 0]
|
| 151 |
+
truth = bfs_field(board, target)
|
| 152 |
+
mask = truth >= 0
|
| 153 |
+
diameter = max(diameter, int(truth.max()))
|
| 154 |
+
descents.append(descent_accuracy(head.cpu(), truth))
|
| 155 |
+
reaches.append(descent_outcomes(head.cpu(), truth))
|
| 156 |
+
feats.append(cells[0, :, mask].T.cpu())
|
| 157 |
+
heads.append(head[mask].cpu())
|
| 158 |
+
dists.append(truth[mask])
|
| 159 |
+
features = torch.cat(feats)
|
| 160 |
+
truth = torch.cat(dists)
|
| 161 |
+
r2, mae = ridge_r2(features, truth)
|
| 162 |
+
head_r2, _ = ridge_r2(torch.cat(heads).unsqueeze(1), truth)
|
| 163 |
+
descent = sum(descents) / max(len(descents), 1)
|
| 164 |
+
reach = sum(reaches) / max(len(reaches), 1)
|
| 165 |
+
report[size] = {"cells": len(truth), "probe_r2": r2, "probe_mae": mae,
|
| 166 |
+
"head_r2": head_r2, "descent_acc": descent,
|
| 167 |
+
"reach_rate": reach,
|
| 168 |
+
"diameter": diameter, "iterations": steps}
|
| 169 |
+
print(f"{size:5d} {len(truth):7d} {r2:9.4f} {mae:10.2f} {head_r2:8.4f} "
|
| 170 |
+
f"{descent:8.4f} {reach:7.3f} {diameter:9d} {steps:6d}")
|
| 171 |
+
|
| 172 |
+
if args.out:
|
| 173 |
+
Path(args.out).parent.mkdir(parents=True, exist_ok=True)
|
| 174 |
+
Path(args.out).write_text(json.dumps(report, indent=2))
|
| 175 |
+
|
| 176 |
+
|
| 177 |
+
if __name__ == "__main__":
|
| 178 |
+
main()
|