{ "date": "2026-09-11", "requested_re10k_budget": "8localincludingtarget+4memory=12totalframes", "live": { "purpose": "LIVE12 matches total frame-slot budget of8local(includingtarget)+4memory; nativecontiguouscontext preserved", "checkpoint": "/share_1/users/bonan_ding/.tmp/worldmem_baselines_20260909/checkpoints/live/live-re10k.ckpt", "checkpoint_global_step": 20500, "weight_policy": "raw state_dict; no EMA tensors in released checkpoint", "parameter_counts": { "backbone_parameters": 772346944, "backbone_parameter_tensors": 174, "vae_parameters": 66456019, "vae_parameter_tensors": 312, "trainable_parameters": 0 }, "optimizer_groups": [], "lr_stages": [], "lora": false, "activation_checkpointing": false, "batch_per_gpu": 1, "full_workers": 8, "generation": "frozen inference only", "unchanged_sampler": { "steps": 18, "flow_shift": 3.0, "chunk_frames": 1, "context_noise": true, "kv_cache": false, "network_precision": "native fp16 autocast", "vae_precision": "fp32 posterior sample scale0.2325" }, "comparison": { "previous_live": { "window_slots": 32, "past_rgb": 31, "target_slots": 1, "memory_slots": 0 }, "new_live": { "window_slots": 12, "past_rgb": 11, "target_slots": 1, "memory_slots": 0 }, "ours_re10k": { "window_slots": 12, "past_rgb": 7, "target_slots": 1, "memory_slots": 4 } }, "selection": "native most-recent contiguous window; no new retrieval", "architecture_input_t": 32, "frames": { "observed": 100, "full_generated": 300, "smoke_generated": 104, "trajectory": "G0..G199,G199..G0", "reference": "raw DFoT test256 RGB", "seed": 0, "full_scenes": 100 }, "boundaries": { "first_prediction": { "target": 100, "context": [ 89, 100 ] }, "twelfth_prediction": { "target": 111, "context": [ 100, 111 ] }, "turnaround": { "target": 200, "source": 199, "context": [ 189, 200 ] }, "last_smoke": { "target": 203, "source": 196 }, "last_full": { "target": 399, "source": 0 } }, "berzelius_full_job": "deployment/evaluate_live_8h200.sh runs104-framepreflight andaggregation beforefull100case evaluation", "prior_results": "preserved; new live_re10k_w12 prefix", "status": "prepared; CPUchecks/GPUsmoke/publication tracked separately" }, "geometry_forcing": { "user_selected_protocol": "Native GF workflow on our100observed/300generated reverse loop; accepted16slots, not12slot comparison", "dataset": "RE10K final100", "scene_manifest": "deployment/manifests/re10k_real_trajectory_h100_f399_scopes_v1.csv", "seed": 0, "case_seed": "1000003*metadata_index+7", "observations_available": 100, "observed_rgb_consumed": 1, "anchor_timeline": 99, "scored_timeline": [ 100, 400 ], "source_trajectory": "G0..G199,G199..G0", "generated_frames": 300, "native_input_frames": 301, "model_slots": 16, "keyframe_count": 19, "keyframe_timeline_indices": [ 99, 116, 132, 149, 166, 182, 199, 216, 232, 249, 266, 282, 299, 316, 332, 349, 366, 382, 399 ], "keyframe_sliding_context": 8, "prediction_guidance": { "name": "stabilized_vanilla", "scale": 4.0, "stabilization": 0.02 }, "interpolation_guidance": { "name": "vanilla", "scale": 1.5 }, "interpolation_max_batch_size": 4, "camera_padding": "Nativehelperrepeatlastcamera forshortbatch;11activecameraentriesbecome16 onlastkeyframebatch", "parameter_counts": { "parameters": 458818051, "parameter_tensors": 494, "trainable_parameters": 0, "trainable_tensors": 0 }, "weights": { "source": "https://huggingface.co/Haoyuwu/GeometryForcing/blob/main/geometry_forcing_state_dict.ckpt", "policy": "released state_dict; no EMA substitution", "backbone_tensors": 496 }, "optimizer_groups": [], "lr_stages": [], "lora": [], "activation_checkpointing": false, "sampler": "DDIM50 eta0; continuouspred_v; fp16autocast", "scoring": "unchangedcommon rawRGB/AlexLPIPS/meanframePSNR/SSIM/pooledFID; notexactpaper-metricreproduction", "outputs": "geometry-forcing_native_re10k_*", "validation": "strictcheckpointCPU+actualnativeplannerrecording-sampler integrationpassed; GPUimagequalitypendingonescenesmoke", "initial_submission": "deployment/smoke_geometry_forcing_1h200.sh", "full_submission_after_review": "deployment/evaluate_geometry_forcing_8h200.sh" }, "minecraft": { "audit_date": "2026-09-11", "scope": "Read-only source audit; only this requested evidence JSON created. No evaluator edits, installations or GPU jobs.", "source_root": "/share_1/users/bonan_ding/.tmp/worldmem_baseline_evals_hf_20260910", "requested_reference_budget": { "local_total_slots": 8, "retrieved_memory_slots": 4, "total_slots": 12, "local_includes_target": true }, "minecraft_reference_caveat": { "finding": "The inspected current Minecraft DeMemWM source-free full-evaluation launcher actually requests 8 local plus up to 8 memory slots, not RE10K 8+4.", "local_total_slots": 8, "past_local_slots_at_chunk1": 7, "current_target_slots": 1, "memory_capacity": { "anchor": 2, "dynamic": 4, "revisit": 2, "total": 8 }, "memory_count_caveat": "These are capacities; valid selections and noise routing can reduce active memory at a denoising step.", "temporal_unit": "One RGB frame per latent position, frame_stack=1; VAE encode flattens time and batch into independent images.", "sources": [ "WorldMem/scripts/evaluate_dememwm_sourcefree_ref_long_horizon_8gpu.sh:104-120,141", "WorldMem/algorithms/dememwm/df_video.py:2620-2640", "WorldMem/configurations/algorithm/df_base.yaml:6", "WorldMem/algorithms/worldmem/df_video.py:572-581" ] }, "decmem": { "native_config": { "stm_sliding_window": 8, "num_frame_per_block": 4, "ltm_topk": 80, "ltm_tile_size": [ 1, 8, 8 ], "local_attn_size": -1, "sink_size": 0 }, "temporal_vae": { "rgb_to_latent": "1 + 4 + 4 + ... RGB frames -> 1 + 1 + 1 + ... latent positions; causal feature caches mean this is a sampling/compression ratio, not a strict receptive-field boundary.", "observed_rgb": 600, "left_padding_rgb": 1, "observed_latents": 151, "full_timeline_latents": 276, "generated_rgb": 500 }, "stm_during_steady_state_denoising": { "cached_previous_latents": 4, "current_block_latents": 4, "total_latent_positions": 8, "nominal_previous_rgb": 16, "nominal_current_rgb": 16, "attention_tokens_per_latent": 880, "attention_tokens_total": 7040, "why": "Each clean commit retains (stm_sliding_window - num_frame_per_block) latent positions. Denoising concatenates those cached K/V with the current block; it does not have 8 prior latent frames." }, "first_generated_block": { "latent_indices_in_stm": [ 144, 145, 146, 147, 148, 149, 150, 151 ], "previous_cached_latents": [ 144, 145, 146, 147 ], "current_clean_latents": [ 148, 149, 150 ], "current_generated_latents": [ 151 ], "nominal_observed_source_rgb_in_stm": [ 672, 699 ], "generated_source_rgb": [ 700, 703 ], "explanation": "151 observed latent positions leave three clean positions in the first 4-latent denoising block." }, "ltm": { "selection_unit": "Spatial key tiles per query tile and attention head, not frames.", "spatial_dit_grid": [ 22, 40 ], "tiles_per_latent": 15, "fine_attention_topk_tiles": 80, "padded_tokens_per_tile": 64, "fine_attention_max_padded_key_tokens_per_query_tile": 5120, "valid_tile_sizes": [ 48, 64 ], "cached_history_policy": "All committed prior blocks remain in tiled and compressed K/V; the current block is concatenated for each forward. No fixed temporal-memory-slot cap or eviction.", "includes_current_block": true, "selection_varies_per_query_head": true, "memory_frame_count": null, "not_equivalent_to_four_memory_frames": true, "why_topk_is_not_frame_count": "15 spatial tiles belong to one latent position; top-80 can combine regions from many times and is selected independently per query/head. Even changing topk to 4 or 60 would not mean four recalled RGB images." }, "persistent_image_anchor": null, "text_or_clip_image_conditioning": false, "vae_decoder_state": "Full causal observed+generated timeline decoded after denoising; this temporal decoder state is separate from transformer attention capacity.", "matches_12_rgb_frame_capacity": false, "sources": [ "DecMem/configs/decmem.yaml:23-40,56", "DecMem/evaluation/worldmem_adapter.py:21-26,82-112,124-137", "DecMem/pipeline/causal_diffusion_inference.py:78-111,247-340", "DecMem/wan/modules/causal_model.py:271-309", "DecMem/wan/modules/ltm_processor.py:284-329,371-418", "DecMem/wan/modules/vae.py:517-574" ] }, "matrix_game_2": { "native_config": { "local_attn_size": 6, "num_frame_per_block": 3, "sink_size": 0, "action_window_size": 3, "action_vae_time_compression_ratio": 4 }, "temporal_vae": { "rgb_to_latent": "1 + 4 + 4 + ... RGB frames -> 1 + 1 + 1 + ... latent positions; causal VAE features can depend on preceding RGB.", "observed_rgb": 600, "left_padding_rgb": 9, "observed_latents": 153, "generated_latents": 126, "generated_rgb_before_scoring_trim": 504, "scored_rgb": 500 }, "self_attention_during_denoising": { "cached_previous_latents": 3, "current_block_latents": 3, "total_latent_positions": 6, "nominal_previous_rgb": 12, "nominal_current_rgb": 12, "attention_tokens_per_latent": 880, "attention_tokens_total": 5280, "first_target_attention_latent_indices": [ 150, 151, 152, 153, 154, 155 ], "first_target_previous_source_rgb": [ 688, 699 ], "first_target_current_source_rgb": [ 700, 711 ], "why": "A 6-latent cache evicts 3 positions before appending each new 3-latent block; the current block consumes half of the configured local attention size." }, "retrieved_memory_slots": 0, "persistent_image_anchor": { "source_frame": 100, "selection": "First observed frame, including identical left-padding copies; not the last observed frame.", "clip_tokens": 257, "cross_attention_layers": 30, "included_in_local_attn_size_6": false, "policy": "The same image CLIP context is passed for every block, cached as cross-attention K/V. This is a persistent image condition, not four retrieved frames." }, "other_conditioning": { "cond_concat": "Native VAE encoding of a first-image-plus-zero-video sequence and an initial-frame mask; sliced for each block. No future ground-truth images.", "actions": "Separate keyboard/mouse K/V caches use the same 6-latent capacity; actions are controls and not additional observed RGB slots." }, "state_caveat": "All observed latent blocks prefill generator K/V and temporal decoder state. Cached hidden representations and autoregressive generated images can propagate historical information, so direct 6-position attention is not a hard bound on all historical influence.", "matches_12_rgb_frame_capacity": false, "sources": [ "Matrix-Game/Matrix-Game-2/worldmem_adapter.py:13-18,70-101,111-118", "Matrix-Game/Matrix-Game-2/configs/distilled_model/universal/config.json:27-34,43-49", "Matrix-Game/Matrix-Game-2/configs/inference_yaml/inference_universal.yaml:14", "Matrix-Game/Matrix-Game-2/pipeline/causal_inference.py:110-116,254-279,371-432", "Matrix-Game/Matrix-Game-2/wan/modules/causal_model.py:173-210", "Matrix-Game/Matrix-Game-2/wan/modules/model.py:228-258" ] }, "conclusion": "Neither Minecraft baseline implements an 8-image-local plus 4-image-memory layout. Report their native compression, current-block size, memory selection unit and persistent conditioning separately. Changing their settings to the integer 12 would not establish the same visual-information budget.", "validation": { "performed": [ "Read effective adapters and their called native cache/attention code.", "Executed only the actual pure-Python timeline functions extracted with AST: DecMem [151,1,276,500], Matrix [153,9,126].", "Checked first generated-block source-index arithmetic against each causal VAE packing rule." ], "not_performed": [ "No model execution, new GPU smoke or numerical comparison under changed capacities." ] } }, "validation": { "cpu": { "scope": "CPU native-sampler/control-boundary test with recording denoiser; reduced-size native Transformer mask/RoPE test underCPUfp16autocast, notcheckpointGPUevaluation", "passed": true, "window_frames": 12, "condition_frames": 11, "first_window": [ 89, 100 ], "all_generated_context_first_window": [ 100, 111 ], "turnaround_window": [ 189, 200 ], "last_window": [ 192, 203 ], "native_denoiser_calls": 1872, "generated_frames": 104, "observed_latents_unchanged": true, "real_scene": "516faacb7687d0c7", "architecture_input_t_retained": 32, "native_transformer_runtime_t": 12, "test_setup_correction": "Initial CPUTransformer test omitted autocast and hit Half/Float mismatch; passed withnativefp16autocast, no modelsourcechange." }, "launcher": { "python_syntax": "passed", "shell_syntax": "passed", "berzelius_cache_values": { "TMPDIR": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache", "TRITON_CACHE_DIR": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache" }, "target_gpu_preflight_before_full": true, "separate_w12_output": true, "full_scope": { "cases": 100, "observed": 100, "generated": 300, "seed": 0, "workers": 8 }, "slurm_resources_unchanged": true }, "local_gpu_job": 4826, "local_gpu_status": "submitted; completion not verified", "berzelius_gpu_preflight": "mandatory104generatedframes beforefull100cases; anyfailurestopsjob", "geometry_native": { "cpu_integration": true, "strict_checkpoint_load": true, "cuda_execution": false, "gpu_image_quality_pending": true } }, "scope": "LIVE12 retained; GF nativeworkflow classifiedoffline; causal12controls prepared; Minecraftmethodsunchanged", "geometry_forcing_previous_dense": { "audit_date": "2026-09-11", "method": "Geometry Forcing", "dataset": "RE10K", "scope": "Read-only source/provenance audit; only this audit artifact was written. No baseline source changes, model inference, installs, or jobs.", "source_root": "/share_1/users/bonan_ding/.tmp/worldmem_baseline_evals_hf_20260910", "official_code_commit": "f5dfc6c5c3cbd2c0dad0121c0f7b161d2886e683", "checkpoint": { "source": "https://huggingface.co/Haoyuwu/GeometryForcing/blob/main/geometry_forcing_state_dict.ckpt", "loaded_backbone_tensors": 496, "parameters": 458818051, "weight_policy": "released state_dict; no EMA substitution", "existing_runtime_evidence": "/share_1/users/bonan_ding/.tmp/worldmem_baselines_20260909/artifacts/smoke-suite-4780/geometry-forcing/method_manifest_rank0.json" }, "current_context": { "observed_prefix_available_to_adapter": 100, "observed_rgb_actually_used_initially": [ 93, 100 ], "past_rgb_frames_per_prediction": 7, "generated_target_slots_per_prediction": 1, "retrieved_memory_reference_frames": 0, "active_slots_total": 8, "physical_backbone_slots": 16, "padding_slots": 8, "padding_semantics": "Pure-noise slots with context_mask=-1 and repeated current-target camera; not extra observations or memory. These slots remain in the transformer computation; this is not hard attention exclusion.", "rolling_rule": "At timeline t, use RGB t-7:t and target camera t; append generated t and discard oldest history frame.", "future_ground_truth_rgb_access": false, "reversal_mapping": "Timeline199=G199, timeline200=G199; camera indices follow the fixed duplicated-turnaround trajectory." }, "encoding_and_decoding": { "diffusion_space": "RGB pixel space", "input_output_shape_per_slot": [ 3, 256, 256 ], "temporal_compression_factor": 1, "vae": null, "extra_encoder_reference_frames": 0, "decoder_history_or_warmup_frames": 0, "spatial_processing": "U-ViT patch embedding and spatial down/up-sampling occur within the denoiser and preserve all time positions.", "output_processing": "Dataset channel normalization is inverted, then pixels are clamped to [0,1].", "vggt_alignment_encoder_at_inference": false }, "comparison_to_confirmed_ours": { "ours_local_slots_including_target": 8, "ours_past_local_frames": 7, "ours_memory_reference_slots": 4, "ours_total_active_slots": 12, "ours_total_reference_rgb_slots": 11, "deMemWM_memory_roles": { "anchor": 1, "dynamic": 2, "revisit": 1 }, "worldmem_memory_roles": { "anchor": 0, "dynamic": 4, "revisit": 0 }, "verdict": "Current GF matches the seven-frame local history and stays below the 12-active-slot budget, but does not match total reference capacity: 7 references versus ours up to 11.", "memory_vs_contiguous": "Ours selects four additional historical memory slots and routes them through its memory mechanism. GF has no such retrieval path; a 12-active-slot GF would use 11 contiguous past frames, not seven local plus four retrieved memories.", "archived_dememwm_manifest": "/tmp/hf_full_final100_80749/results/re10k_faithful_dememwm_full_final100_train731edd42758e_evaldc3cff220ba0_step70000_job80749/run_manifest.json", "archived_code_evidence": { "repository": "/share_1/users/bonan_ding/.tmp/WorldMem-vmem-fair-eval-0b88eda", "dememwm_eval_commit": "dc3cff220ba0af892e354c206ef58d33ef76577c", "worldmem_eval_commit": "73c1cc4f6787", "file": "algorithms/worldmem/dfot_realestate10k.py", "local_target_inclusion_lines": [ 4265, 4266, 4322, 4324, 4325, 4329, 4354, 4365 ], "worldmem_four_reference_requirement_lines": [ 1792, 1793, 1802, 1803 ] } }, "twelve_active_slot_feasibility": { "static_api_compatible": true, "proposed_reference_frames": 11, "proposed_target_slots": 1, "proposed_padding_slots": 4, "physical_backbone_slots_retained": 16, "requires_new_weights_or_training": false, "requires_adapter_edit": true, "needed_changes": "Use last11 history RGB, target-11:target+1 camera indices, 4 repeated target-camera pad slots, sampling length12, and updated provenance. Changing dataset.context_length alone has no effect on these hardcoded rollout dimensions.", "checkpoint_compatibility_basis": "Only active context/length/mask change; native maximum16 and all backbone tensors remain unchanged. Native sampler explicitly supports lengths<=16 with padding.", "do_not_set_physical_backbone_length_to_12": "Unnecessary model-configuration change; keep the released 16-slot backbone and pad4.", "runtime_validation_at_12": "Not performed; existing verified smoke used active8." }, "sampler_unchanged": { "steps": 50, "ddim_eta": 0.0, "guidance": "stabilized_vanilla", "guidance_scale": 4.0, "stabilization_level": 0.02, "precision": "fp16 autocast with fp32 camera processing" }, "sources": [ { "file": "GeometryForcing/evaluation/worldmem_adapter.py", "lines": [ 23, 32, 48, 50, 73, 91, 98, 100, 101, 107, 110, 111, 112 ], "supports": "Actual 7-history + 1-target input; camera padding to 16; strict released backbone load; normalization only; rolling history." }, { "file": "shared/run_and_score.py", "lines": [ 35, 36, 96, 97 ], "supports": "Production entrypoint selects the released checkpoint and exposes only first 100 RGB observations to Geometry Forcing." }, { "file": "GeometryForcing/configurations/dataset/realestate10k.yaml", "lines": [ 9, 12 ], "supports": "Native maximum is 16 frames at 256-square resolution." }, { "file": "GeometryForcing/configurations/dataset/base_video.yaml", "lines": [ 13, 14, 18, 23 ], "supports": "Latent diffusion disabled, temporal factor 1, direct 3-channel RGB inputs." }, { "file": "GeometryForcing/algorithms/dfot/dfot_video.py", "lines": [ 125, 128, 209, 215, 1155, 1187, 1192, 1209, 1214, 1219, 1347, 1348 ], "supports": "No VAE is loaded; frame-to-time-token conversion is 1:1; active lengths <=16 accepted and padded to native horizon, then trimmed." }, { "file": "GeometryForcing/algorithms/dfot/dfot_video.py", "lines": [ 940, 961, 962, 963, 1275, 1316, 1338, 1340 ], "supports": "Padding receives maximal noise level, is excluded from returned generated positions, and is held rather than generated." }, { "file": "GeometryForcing/algorithms/dfot/history_guidance.py", "lines": [ 343, 345, 369, 370 ], "supports": "History is mask>=1, targets mask==0, padding mask==-1; padding is not an additional RGB reference." }, { "file": "GeometryForcing/algorithms/dfot/backbones/u_vit/u_vit3d_pose.py", "lines": [ 63, 68, 81, 83, 89, 90, 123, 128, 131, 139, 140, 154, 155 ], "supports": "U-ViT requires its configured 16 physical time slots; per-frame spatial encoding/decoding preserves the time dimension. No per-frame padding attention mask is passed to this forward API." }, { "file": "GeometryForcing/algorithms/dfot/dfot_video_pose.py", "lines": [ 136, 142, 143, 147, 148 ], "supports": "VGGT alignment encoder is only constructed with alignment_coeff>0; adapter sets it to zero." } ], "recommendation": "Retain current GF results explicitly labeled local8/no-memory. A GF active12 rerun is feasible for equal total frame budget, but is a separate experimental variant requiring authorized adapter changes and a smoke run; no change is made by this audit." }, "geometry_forcing_budget_exception": "Previous native16-slot run retained only as an offline diagnostic; new past-only tests target12active slots.", "geometry_forcing_native_classification": "Offline diagnostic only: interpolation uses later generated keyframes; not an eligible strict autoregressive baseline.", "geometry_forcing_causal_controls": { "scope": "GF early-collapse diagnostics, not full benchmark", "modes": [ "ar8_cfg4", "ar12_cfg4", "ar12_cfg1", "teacher12_cfg4" ], "dataset": "RE10K monitor8 first two fixed scenes", "frames": 32, "scored_timeline": [ 100, 132 ], "primary_active_budget": 12, "physical_backbone_slots": 16, "causal_conditioning": "RGB strictly beforetarget; cameras only through currenttarget; currentcamera repeated fornoise-padding", "teacher_control": "Only pastRGB replacedwithGT; same masks andstabilization, diagnostic-only", "interpolation": "nevercalled", "frozen": true, "parameters": 458818051, "optimizer": null, "lr": null, "lora": null, "sampling": "DDIM50eta0; stabilization.02; originalstate_dict", "rng": "private/globalstatesrecorded;12slotCFGpair hasthe same targetnoise;8vs12changes targetslot", "cpu_validation_files": [ "cli_validation.json", "causality_validation.json", "checkpoint_interface_validation.json" ], "gpu_validation": "pendingBerzeliussubmission" } }