worldmem-baseline-evals / deployment /context_budget_audit.json
BonanDing's picture
Add strictly past-only Geometry Forcing context and guidance diagnostics
03a4c98 verified
Raw History Blame Contribute Delete
26.5 kB
{
"date": "2026-09-11",
"requested_re10k_budget": "8localincludingtarget+4memory=12totalframes",
"live": {
"purpose": "LIVE12 matches total frame-slot budget of8local(includingtarget)+4memory; nativecontiguouscontext preserved",
"checkpoint": "/share_1/users/bonan_ding/.tmp/worldmem_baselines_20260909/checkpoints/live/live-re10k.ckpt",
"checkpoint_global_step": 20500,
"weight_policy": "raw state_dict; no EMA tensors in released checkpoint",
"parameter_counts": {
"backbone_parameters": 772346944,
"backbone_parameter_tensors": 174,
"vae_parameters": 66456019,
"vae_parameter_tensors": 312,
"trainable_parameters": 0
},
"optimizer_groups": [],
"lr_stages": [],
"lora": false,
"activation_checkpointing": false,
"batch_per_gpu": 1,
"full_workers": 8,
"generation": "frozen inference only",
"unchanged_sampler": {
"steps": 18,
"flow_shift": 3.0,
"chunk_frames": 1,
"context_noise": true,
"kv_cache": false,
"network_precision": "native fp16 autocast",
"vae_precision": "fp32 posterior sample scale0.2325"
},
"comparison": {
"previous_live": {
"window_slots": 32,
"past_rgb": 31,
"target_slots": 1,
"memory_slots": 0
},
"new_live": {
"window_slots": 12,
"past_rgb": 11,
"target_slots": 1,
"memory_slots": 0
},
"ours_re10k": {
"window_slots": 12,
"past_rgb": 7,
"target_slots": 1,
"memory_slots": 4
}
},
"selection": "native most-recent contiguous window; no new retrieval",
"architecture_input_t": 32,
"frames": {
"observed": 100,
"full_generated": 300,
"smoke_generated": 104,
"trajectory": "G0..G199,G199..G0",
"reference": "raw DFoT test256 RGB",
"seed": 0,
"full_scenes": 100
},
"boundaries": {
"first_prediction": {
"target": 100,
"context": [
89,
100
]
},
"twelfth_prediction": {
"target": 111,
"context": [
100,
111
]
},
"turnaround": {
"target": 200,
"source": 199,
"context": [
189,
200
]
},
"last_smoke": {
"target": 203,
"source": 196
},
"last_full": {
"target": 399,
"source": 0
}
},
"berzelius_full_job": "deployment/evaluate_live_8h200.sh runs104-framepreflight andaggregation beforefull100case evaluation",
"prior_results": "preserved; new live_re10k_w12 prefix",
"status": "prepared; CPUchecks/GPUsmoke/publication tracked separately"
},
"geometry_forcing": {
"user_selected_protocol": "Native GF workflow on our100observed/300generated reverse loop; accepted16slots, not12slot comparison",
"dataset": "RE10K final100",
"scene_manifest": "deployment/manifests/re10k_real_trajectory_h100_f399_scopes_v1.csv",
"seed": 0,
"case_seed": "1000003*metadata_index+7",
"observations_available": 100,
"observed_rgb_consumed": 1,
"anchor_timeline": 99,
"scored_timeline": [
100,
400
],
"source_trajectory": "G0..G199,G199..G0",
"generated_frames": 300,
"native_input_frames": 301,
"model_slots": 16,
"keyframe_count": 19,
"keyframe_timeline_indices": [
99,
116,
132,
149,
166,
182,
199,
216,
232,
249,
266,
282,
299,
316,
332,
349,
366,
382,
399
],
"keyframe_sliding_context": 8,
"prediction_guidance": {
"name": "stabilized_vanilla",
"scale": 4.0,
"stabilization": 0.02
},
"interpolation_guidance": {
"name": "vanilla",
"scale": 1.5
},
"interpolation_max_batch_size": 4,
"camera_padding": "Nativehelperrepeatlastcamera forshortbatch;11activecameraentriesbecome16 onlastkeyframebatch",
"parameter_counts": {
"parameters": 458818051,
"parameter_tensors": 494,
"trainable_parameters": 0,
"trainable_tensors": 0
},
"weights": {
"source": "https://huggingface.co/Haoyuwu/GeometryForcing/blob/main/geometry_forcing_state_dict.ckpt",
"policy": "released state_dict; no EMA substitution",
"backbone_tensors": 496
},
"optimizer_groups": [],
"lr_stages": [],
"lora": [],
"activation_checkpointing": false,
"sampler": "DDIM50 eta0; continuouspred_v; fp16autocast",
"scoring": "unchangedcommon rawRGB/AlexLPIPS/meanframePSNR/SSIM/pooledFID; notexactpaper-metricreproduction",
"outputs": "geometry-forcing_native_re10k_*",
"validation": "strictcheckpointCPU+actualnativeplannerrecording-sampler integrationpassed; GPUimagequalitypendingonescenesmoke",
"initial_submission": "deployment/smoke_geometry_forcing_1h200.sh",
"full_submission_after_review": "deployment/evaluate_geometry_forcing_8h200.sh"
},
"minecraft": {
"audit_date": "2026-09-11",
"scope": "Read-only source audit; only this requested evidence JSON created. No evaluator edits, installations or GPU jobs.",
"source_root": "/share_1/users/bonan_ding/.tmp/worldmem_baseline_evals_hf_20260910",
"requested_reference_budget": {
"local_total_slots": 8,
"retrieved_memory_slots": 4,
"total_slots": 12,
"local_includes_target": true
},
"minecraft_reference_caveat": {
"finding": "The inspected current Minecraft DeMemWM source-free full-evaluation launcher actually requests 8 local plus up to 8 memory slots, not RE10K 8+4.",
"local_total_slots": 8,
"past_local_slots_at_chunk1": 7,
"current_target_slots": 1,
"memory_capacity": {
"anchor": 2,
"dynamic": 4,
"revisit": 2,
"total": 8
},
"memory_count_caveat": "These are capacities; valid selections and noise routing can reduce active memory at a denoising step.",
"temporal_unit": "One RGB frame per latent position, frame_stack=1; VAE encode flattens time and batch into independent images.",
"sources": [
"WorldMem/scripts/evaluate_dememwm_sourcefree_ref_long_horizon_8gpu.sh:104-120,141",
"WorldMem/algorithms/dememwm/df_video.py:2620-2640",
"WorldMem/configurations/algorithm/df_base.yaml:6",
"WorldMem/algorithms/worldmem/df_video.py:572-581"
]
},
"decmem": {
"native_config": {
"stm_sliding_window": 8,
"num_frame_per_block": 4,
"ltm_topk": 80,
"ltm_tile_size": [
1,
8,
8
],
"local_attn_size": -1,
"sink_size": 0
},
"temporal_vae": {
"rgb_to_latent": "1 + 4 + 4 + ... RGB frames -> 1 + 1 + 1 + ... latent positions; causal feature caches mean this is a sampling/compression ratio, not a strict receptive-field boundary.",
"observed_rgb": 600,
"left_padding_rgb": 1,
"observed_latents": 151,
"full_timeline_latents": 276,
"generated_rgb": 500
},
"stm_during_steady_state_denoising": {
"cached_previous_latents": 4,
"current_block_latents": 4,
"total_latent_positions": 8,
"nominal_previous_rgb": 16,
"nominal_current_rgb": 16,
"attention_tokens_per_latent": 880,
"attention_tokens_total": 7040,
"why": "Each clean commit retains (stm_sliding_window - num_frame_per_block) latent positions. Denoising concatenates those cached K/V with the current block; it does not have 8 prior latent frames."
},
"first_generated_block": {
"latent_indices_in_stm": [
144,
145,
146,
147,
148,
149,
150,
151
],
"previous_cached_latents": [
144,
145,
146,
147
],
"current_clean_latents": [
148,
149,
150
],
"current_generated_latents": [
151
],
"nominal_observed_source_rgb_in_stm": [
672,
699
],
"generated_source_rgb": [
700,
703
],
"explanation": "151 observed latent positions leave three clean positions in the first 4-latent denoising block."
},
"ltm": {
"selection_unit": "Spatial key tiles per query tile and attention head, not frames.",
"spatial_dit_grid": [
22,
40
],
"tiles_per_latent": 15,
"fine_attention_topk_tiles": 80,
"padded_tokens_per_tile": 64,
"fine_attention_max_padded_key_tokens_per_query_tile": 5120,
"valid_tile_sizes": [
48,
64
],
"cached_history_policy": "All committed prior blocks remain in tiled and compressed K/V; the current block is concatenated for each forward. No fixed temporal-memory-slot cap or eviction.",
"includes_current_block": true,
"selection_varies_per_query_head": true,
"memory_frame_count": null,
"not_equivalent_to_four_memory_frames": true,
"why_topk_is_not_frame_count": "15 spatial tiles belong to one latent position; top-80 can combine regions from many times and is selected independently per query/head. Even changing topk to 4 or 60 would not mean four recalled RGB images."
},
"persistent_image_anchor": null,
"text_or_clip_image_conditioning": false,
"vae_decoder_state": "Full causal observed+generated timeline decoded after denoising; this temporal decoder state is separate from transformer attention capacity.",
"matches_12_rgb_frame_capacity": false,
"sources": [
"DecMem/configs/decmem.yaml:23-40,56",
"DecMem/evaluation/worldmem_adapter.py:21-26,82-112,124-137",
"DecMem/pipeline/causal_diffusion_inference.py:78-111,247-340",
"DecMem/wan/modules/causal_model.py:271-309",
"DecMem/wan/modules/ltm_processor.py:284-329,371-418",
"DecMem/wan/modules/vae.py:517-574"
]
},
"matrix_game_2": {
"native_config": {
"local_attn_size": 6,
"num_frame_per_block": 3,
"sink_size": 0,
"action_window_size": 3,
"action_vae_time_compression_ratio": 4
},
"temporal_vae": {
"rgb_to_latent": "1 + 4 + 4 + ... RGB frames -> 1 + 1 + 1 + ... latent positions; causal VAE features can depend on preceding RGB.",
"observed_rgb": 600,
"left_padding_rgb": 9,
"observed_latents": 153,
"generated_latents": 126,
"generated_rgb_before_scoring_trim": 504,
"scored_rgb": 500
},
"self_attention_during_denoising": {
"cached_previous_latents": 3,
"current_block_latents": 3,
"total_latent_positions": 6,
"nominal_previous_rgb": 12,
"nominal_current_rgb": 12,
"attention_tokens_per_latent": 880,
"attention_tokens_total": 5280,
"first_target_attention_latent_indices": [
150,
151,
152,
153,
154,
155
],
"first_target_previous_source_rgb": [
688,
699
],
"first_target_current_source_rgb": [
700,
711
],
"why": "A 6-latent cache evicts 3 positions before appending each new 3-latent block; the current block consumes half of the configured local attention size."
},
"retrieved_memory_slots": 0,
"persistent_image_anchor": {
"source_frame": 100,
"selection": "First observed frame, including identical left-padding copies; not the last observed frame.",
"clip_tokens": 257,
"cross_attention_layers": 30,
"included_in_local_attn_size_6": false,
"policy": "The same image CLIP context is passed for every block, cached as cross-attention K/V. This is a persistent image condition, not four retrieved frames."
},
"other_conditioning": {
"cond_concat": "Native VAE encoding of a first-image-plus-zero-video sequence and an initial-frame mask; sliced for each block. No future ground-truth images.",
"actions": "Separate keyboard/mouse K/V caches use the same 6-latent capacity; actions are controls and not additional observed RGB slots."
},
"state_caveat": "All observed latent blocks prefill generator K/V and temporal decoder state. Cached hidden representations and autoregressive generated images can propagate historical information, so direct 6-position attention is not a hard bound on all historical influence.",
"matches_12_rgb_frame_capacity": false,
"sources": [
"Matrix-Game/Matrix-Game-2/worldmem_adapter.py:13-18,70-101,111-118",
"Matrix-Game/Matrix-Game-2/configs/distilled_model/universal/config.json:27-34,43-49",
"Matrix-Game/Matrix-Game-2/configs/inference_yaml/inference_universal.yaml:14",
"Matrix-Game/Matrix-Game-2/pipeline/causal_inference.py:110-116,254-279,371-432",
"Matrix-Game/Matrix-Game-2/wan/modules/causal_model.py:173-210",
"Matrix-Game/Matrix-Game-2/wan/modules/model.py:228-258"
]
},
"conclusion": "Neither Minecraft baseline implements an 8-image-local plus 4-image-memory layout. Report their native compression, current-block size, memory selection unit and persistent conditioning separately. Changing their settings to the integer 12 would not establish the same visual-information budget.",
"validation": {
"performed": [
"Read effective adapters and their called native cache/attention code.",
"Executed only the actual pure-Python timeline functions extracted with AST: DecMem [151,1,276,500], Matrix [153,9,126].",
"Checked first generated-block source-index arithmetic against each causal VAE packing rule."
],
"not_performed": [
"No model execution, new GPU smoke or numerical comparison under changed capacities."
]
}
},
"validation": {
"cpu": {
"scope": "CPU native-sampler/control-boundary test with recording denoiser; reduced-size native Transformer mask/RoPE test underCPUfp16autocast, notcheckpointGPUevaluation",
"passed": true,
"window_frames": 12,
"condition_frames": 11,
"first_window": [
89,
100
],
"all_generated_context_first_window": [
100,
111
],
"turnaround_window": [
189,
200
],
"last_window": [
192,
203
],
"native_denoiser_calls": 1872,
"generated_frames": 104,
"observed_latents_unchanged": true,
"real_scene": "516faacb7687d0c7",
"architecture_input_t_retained": 32,
"native_transformer_runtime_t": 12,
"test_setup_correction": "Initial CPUTransformer test omitted autocast and hit Half/Float mismatch; passed withnativefp16autocast, no modelsourcechange."
},
"launcher": {
"python_syntax": "passed",
"shell_syntax": "passed",
"berzelius_cache_values": {
"TMPDIR": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache",
"TRITON_CACHE_DIR": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache"
},
"target_gpu_preflight_before_full": true,
"separate_w12_output": true,
"full_scope": {
"cases": 100,
"observed": 100,
"generated": 300,
"seed": 0,
"workers": 8
},
"slurm_resources_unchanged": true
},
"local_gpu_job": 4826,
"local_gpu_status": "submitted; completion not verified",
"berzelius_gpu_preflight": "mandatory104generatedframes beforefull100cases; anyfailurestopsjob",
"geometry_native": {
"cpu_integration": true,
"strict_checkpoint_load": true,
"cuda_execution": false,
"gpu_image_quality_pending": true
}
},
"scope": "LIVE12 retained; GF nativeworkflow classifiedoffline; causal12controls prepared; Minecraftmethodsunchanged",
"geometry_forcing_previous_dense": {
"audit_date": "2026-09-11",
"method": "Geometry Forcing",
"dataset": "RE10K",
"scope": "Read-only source/provenance audit; only this audit artifact was written. No baseline source changes, model inference, installs, or jobs.",
"source_root": "/share_1/users/bonan_ding/.tmp/worldmem_baseline_evals_hf_20260910",
"official_code_commit": "f5dfc6c5c3cbd2c0dad0121c0f7b161d2886e683",
"checkpoint": {
"source": "https://huggingface.co/Haoyuwu/GeometryForcing/blob/main/geometry_forcing_state_dict.ckpt",
"loaded_backbone_tensors": 496,
"parameters": 458818051,
"weight_policy": "released state_dict; no EMA substitution",
"existing_runtime_evidence": "/share_1/users/bonan_ding/.tmp/worldmem_baselines_20260909/artifacts/smoke-suite-4780/geometry-forcing/method_manifest_rank0.json"
},
"current_context": {
"observed_prefix_available_to_adapter": 100,
"observed_rgb_actually_used_initially": [
93,
100
],
"past_rgb_frames_per_prediction": 7,
"generated_target_slots_per_prediction": 1,
"retrieved_memory_reference_frames": 0,
"active_slots_total": 8,
"physical_backbone_slots": 16,
"padding_slots": 8,
"padding_semantics": "Pure-noise slots with context_mask=-1 and repeated current-target camera; not extra observations or memory. These slots remain in the transformer computation; this is not hard attention exclusion.",
"rolling_rule": "At timeline t, use RGB t-7:t and target camera t; append generated t and discard oldest history frame.",
"future_ground_truth_rgb_access": false,
"reversal_mapping": "Timeline199=G199, timeline200=G199; camera indices follow the fixed duplicated-turnaround trajectory."
},
"encoding_and_decoding": {
"diffusion_space": "RGB pixel space",
"input_output_shape_per_slot": [
3,
256,
256
],
"temporal_compression_factor": 1,
"vae": null,
"extra_encoder_reference_frames": 0,
"decoder_history_or_warmup_frames": 0,
"spatial_processing": "U-ViT patch embedding and spatial down/up-sampling occur within the denoiser and preserve all time positions.",
"output_processing": "Dataset channel normalization is inverted, then pixels are clamped to [0,1].",
"vggt_alignment_encoder_at_inference": false
},
"comparison_to_confirmed_ours": {
"ours_local_slots_including_target": 8,
"ours_past_local_frames": 7,
"ours_memory_reference_slots": 4,
"ours_total_active_slots": 12,
"ours_total_reference_rgb_slots": 11,
"deMemWM_memory_roles": {
"anchor": 1,
"dynamic": 2,
"revisit": 1
},
"worldmem_memory_roles": {
"anchor": 0,
"dynamic": 4,
"revisit": 0
},
"verdict": "Current GF matches the seven-frame local history and stays below the 12-active-slot budget, but does not match total reference capacity: 7 references versus ours up to 11.",
"memory_vs_contiguous": "Ours selects four additional historical memory slots and routes them through its memory mechanism. GF has no such retrieval path; a 12-active-slot GF would use 11 contiguous past frames, not seven local plus four retrieved memories.",
"archived_dememwm_manifest": "/tmp/hf_full_final100_80749/results/re10k_faithful_dememwm_full_final100_train731edd42758e_evaldc3cff220ba0_step70000_job80749/run_manifest.json",
"archived_code_evidence": {
"repository": "/share_1/users/bonan_ding/.tmp/WorldMem-vmem-fair-eval-0b88eda",
"dememwm_eval_commit": "dc3cff220ba0af892e354c206ef58d33ef76577c",
"worldmem_eval_commit": "73c1cc4f6787",
"file": "algorithms/worldmem/dfot_realestate10k.py",
"local_target_inclusion_lines": [
4265,
4266,
4322,
4324,
4325,
4329,
4354,
4365
],
"worldmem_four_reference_requirement_lines": [
1792,
1793,
1802,
1803
]
}
},
"twelve_active_slot_feasibility": {
"static_api_compatible": true,
"proposed_reference_frames": 11,
"proposed_target_slots": 1,
"proposed_padding_slots": 4,
"physical_backbone_slots_retained": 16,
"requires_new_weights_or_training": false,
"requires_adapter_edit": true,
"needed_changes": "Use last11 history RGB, target-11:target+1 camera indices, 4 repeated target-camera pad slots, sampling length12, and updated provenance. Changing dataset.context_length alone has no effect on these hardcoded rollout dimensions.",
"checkpoint_compatibility_basis": "Only active context/length/mask change; native maximum16 and all backbone tensors remain unchanged. Native sampler explicitly supports lengths<=16 with padding.",
"do_not_set_physical_backbone_length_to_12": "Unnecessary model-configuration change; keep the released 16-slot backbone and pad4.",
"runtime_validation_at_12": "Not performed; existing verified smoke used active8."
},
"sampler_unchanged": {
"steps": 50,
"ddim_eta": 0.0,
"guidance": "stabilized_vanilla",
"guidance_scale": 4.0,
"stabilization_level": 0.02,
"precision": "fp16 autocast with fp32 camera processing"
},
"sources": [
{
"file": "GeometryForcing/evaluation/worldmem_adapter.py",
"lines": [
23,
32,
48,
50,
73,
91,
98,
100,
101,
107,
110,
111,
112
],
"supports": "Actual 7-history + 1-target input; camera padding to 16; strict released backbone load; normalization only; rolling history."
},
{
"file": "shared/run_and_score.py",
"lines": [
35,
36,
96,
97
],
"supports": "Production entrypoint selects the released checkpoint and exposes only first 100 RGB observations to Geometry Forcing."
},
{
"file": "GeometryForcing/configurations/dataset/realestate10k.yaml",
"lines": [
9,
12
],
"supports": "Native maximum is 16 frames at 256-square resolution."
},
{
"file": "GeometryForcing/configurations/dataset/base_video.yaml",
"lines": [
13,
14,
18,
23
],
"supports": "Latent diffusion disabled, temporal factor 1, direct 3-channel RGB inputs."
},
{
"file": "GeometryForcing/algorithms/dfot/dfot_video.py",
"lines": [
125,
128,
209,
215,
1155,
1187,
1192,
1209,
1214,
1219,
1347,
1348
],
"supports": "No VAE is loaded; frame-to-time-token conversion is 1:1; active lengths <=16 accepted and padded to native horizon, then trimmed."
},
{
"file": "GeometryForcing/algorithms/dfot/dfot_video.py",
"lines": [
940,
961,
962,
963,
1275,
1316,
1338,
1340
],
"supports": "Padding receives maximal noise level, is excluded from returned generated positions, and is held rather than generated."
},
{
"file": "GeometryForcing/algorithms/dfot/history_guidance.py",
"lines": [
343,
345,
369,
370
],
"supports": "History is mask>=1, targets mask==0, padding mask==-1; padding is not an additional RGB reference."
},
{
"file": "GeometryForcing/algorithms/dfot/backbones/u_vit/u_vit3d_pose.py",
"lines": [
63,
68,
81,
83,
89,
90,
123,
128,
131,
139,
140,
154,
155
],
"supports": "U-ViT requires its configured 16 physical time slots; per-frame spatial encoding/decoding preserves the time dimension. No per-frame padding attention mask is passed to this forward API."
},
{
"file": "GeometryForcing/algorithms/dfot/dfot_video_pose.py",
"lines": [
136,
142,
143,
147,
148
],
"supports": "VGGT alignment encoder is only constructed with alignment_coeff>0; adapter sets it to zero."
}
],
"recommendation": "Retain current GF results explicitly labeled local8/no-memory. A GF active12 rerun is feasible for equal total frame budget, but is a separate experimental variant requiring authorized adapter changes and a smoke run; no change is made by this audit."
},
"geometry_forcing_budget_exception": "Previous native16-slot run retained only as an offline diagnostic; new past-only tests target12active slots.",
"geometry_forcing_native_classification": "Offline diagnostic only: interpolation uses later generated keyframes; not an eligible strict autoregressive baseline.",
"geometry_forcing_causal_controls": {
"scope": "GF early-collapse diagnostics, not full benchmark",
"modes": [
"ar8_cfg4",
"ar12_cfg4",
"ar12_cfg1",
"teacher12_cfg4"
],
"dataset": "RE10K monitor8 first two fixed scenes",
"frames": 32,
"scored_timeline": [
100,
132
],
"primary_active_budget": 12,
"physical_backbone_slots": 16,
"causal_conditioning": "RGB strictly beforetarget; cameras only through currenttarget; currentcamera repeated fornoise-padding",
"teacher_control": "Only pastRGB replacedwithGT; same masks andstabilization, diagnostic-only",
"interpolation": "nevercalled",
"frozen": true,
"parameters": 458818051,
"optimizer": null,
"lr": null,
"lora": null,
"sampling": "DDIM50eta0; stabilization.02; originalstate_dict",
"rng": "private/globalstatesrecorded;12slotCFGpair hasthe same targetnoise;8vs12changes targetslot",
"cpu_validation_files": [
"cli_validation.json",
"causality_validation.json",
"checkpoint_interface_validation.json"
],
"gpu_validation": "pendingBerzeliussubmission"
}
}