{ "base_revision": "1a9eb8af2754ad2329a24cfe50d388cb559441d0", "cache_revision": "19f554feadb744961d7f478ea099a11ea7f58771", "parameters_including_encoder": 3994676, "seed": 92872, "steps_target": 2000, "batch_size": 24, "arms": [ "one_pass_control", "recurrent_four_pass" ], "architecture": "Reuse pretrained MobileNetV3 and spatial decoder once. Repeat SAME two QueryBlocks; after first pass apply damped residual update .25. Pool spatial features using current color-region weights and append normalized pooled tokens to memory. No extra parameters.", "loss": "Same fixed selected whole-image targets and losses from region pilot. Control single output; recurrent mean loss across all four passes. No region auxiliary or critic. Matched input batches, initial weights, optimizer and update count.", "learning_rates": { "encoder": 5e-06, "decoder": 0.0001 }, "evaluation": "Fresh 96-image final holdout excluding earlier evaluation IDs and byte-identical training images. Development reused, explicitly nonindependent. 1/2/4 trained-depth and8 extrapolation. Equal-guided-chroma processing. Raw metrics also retained.", "limitations": [ "Single seed short architecture pilot", "8 passes beyond trained depth may worsen", "Fixed targets inherit earlier model selection bias", "Does not test diffusion or demonstrate architectural ceiling", "No guarantees extra passes help", "Near duplicates and upstream pretraining overlap unknown" ], "research": [ "https://arxiv.org/abs/1806.02919", "https://arxiv.org/abs/1904.05290", "https://arxiv.org/abs/2212.11613", "https://arxiv.org/abs/2111.05826" ], "budget": { "hardware": "l4x1", "hard_timeout_minutes": 30, "maximum_hardware_usd_at_published_rate": 0.4, "training_graceful_stop_minutes": 21 }, "release": "Experimental checkpoints to main subfolder, visual review required before production" }