Download deployment/context_budget_audit.json from BonanDing/worldmem-baseline-evals: direct link, hf CLI and curl.
- Browser
- Download file 26.5 kB
-
https://huggingface.co/BonanDing/worldmem-baseline-evals/resolve/main/deployment/context_budget_audit.json
- Command line
-
hf download hf://BonanDing/worldmem-baseline-evals/deployment/context_budget_audit.json
-
curl -L -o context_budget_audit.json https://huggingface.co/BonanDing/worldmem-baseline-evals/resolve/main/deployment/context_budget_audit.json
26.5 kB
| { | |
| "date": "2026-09-11", | |
| "requested_re10k_budget": "8localincludingtarget+4memory=12totalframes", | |
| "live": { | |
| "purpose": "LIVE12 matches total frame-slot budget of8local(includingtarget)+4memory; nativecontiguouscontext preserved", | |
| "checkpoint": "/share_1/users/bonan_ding/.tmp/worldmem_baselines_20260909/checkpoints/live/live-re10k.ckpt", | |
| "checkpoint_global_step": 20500, | |
| "weight_policy": "raw state_dict; no EMA tensors in released checkpoint", | |
| "parameter_counts": { | |
| "backbone_parameters": 772346944, | |
| "backbone_parameter_tensors": 174, | |
| "vae_parameters": 66456019, | |
| "vae_parameter_tensors": 312, | |
| "trainable_parameters": 0 | |
| }, | |
| "optimizer_groups": [], | |
| "lr_stages": [], | |
| "lora": false, | |
| "activation_checkpointing": false, | |
| "batch_per_gpu": 1, | |
| "full_workers": 8, | |
| "generation": "frozen inference only", | |
| "unchanged_sampler": { | |
| "steps": 18, | |
| "flow_shift": 3.0, | |
| "chunk_frames": 1, | |
| "context_noise": true, | |
| "kv_cache": false, | |
| "network_precision": "native fp16 autocast", | |
| "vae_precision": "fp32 posterior sample scale0.2325" | |
| }, | |
| "comparison": { | |
| "previous_live": { | |
| "window_slots": 32, | |
| "past_rgb": 31, | |
| "target_slots": 1, | |
| "memory_slots": 0 | |
| }, | |
| "new_live": { | |
| "window_slots": 12, | |
| "past_rgb": 11, | |
| "target_slots": 1, | |
| "memory_slots": 0 | |
| }, | |
| "ours_re10k": { | |
| "window_slots": 12, | |
| "past_rgb": 7, | |
| "target_slots": 1, | |
| "memory_slots": 4 | |
| } | |
| }, | |
| "selection": "native most-recent contiguous window; no new retrieval", | |
| "architecture_input_t": 32, | |
| "frames": { | |
| "observed": 100, | |
| "full_generated": 300, | |
| "smoke_generated": 104, | |
| "trajectory": "G0..G199,G199..G0", | |
| "reference": "raw DFoT test256 RGB", | |
| "seed": 0, | |
| "full_scenes": 100 | |
| }, | |
| "boundaries": { | |
| "first_prediction": { | |
| "target": 100, | |
| "context": [ | |
| 89, | |
| 100 | |
| ] | |
| }, | |
| "twelfth_prediction": { | |
| "target": 111, | |
| "context": [ | |
| 100, | |
| 111 | |
| ] | |
| }, | |
| "turnaround": { | |
| "target": 200, | |
| "source": 199, | |
| "context": [ | |
| 189, | |
| 200 | |
| ] | |
| }, | |
| "last_smoke": { | |
| "target": 203, | |
| "source": 196 | |
| }, | |
| "last_full": { | |
| "target": 399, | |
| "source": 0 | |
| } | |
| }, | |
| "berzelius_full_job": "deployment/evaluate_live_8h200.sh runs104-framepreflight andaggregation beforefull100case evaluation", | |
| "prior_results": "preserved; new live_re10k_w12 prefix", | |
| "status": "prepared; CPUchecks/GPUsmoke/publication tracked separately" | |
| }, | |
| "geometry_forcing": { | |
| "user_selected_protocol": "Native GF workflow on our100observed/300generated reverse loop; accepted16slots, not12slot comparison", | |
| "dataset": "RE10K final100", | |
| "scene_manifest": "deployment/manifests/re10k_real_trajectory_h100_f399_scopes_v1.csv", | |
| "seed": 0, | |
| "case_seed": "1000003*metadata_index+7", | |
| "observations_available": 100, | |
| "observed_rgb_consumed": 1, | |
| "anchor_timeline": 99, | |
| "scored_timeline": [ | |
| 100, | |
| 400 | |
| ], | |
| "source_trajectory": "G0..G199,G199..G0", | |
| "generated_frames": 300, | |
| "native_input_frames": 301, | |
| "model_slots": 16, | |
| "keyframe_count": 19, | |
| "keyframe_timeline_indices": [ | |
| 99, | |
| 116, | |
| 132, | |
| 149, | |
| 166, | |
| 182, | |
| 199, | |
| 216, | |
| 232, | |
| 249, | |
| 266, | |
| 282, | |
| 299, | |
| 316, | |
| 332, | |
| 349, | |
| 366, | |
| 382, | |
| 399 | |
| ], | |
| "keyframe_sliding_context": 8, | |
| "prediction_guidance": { | |
| "name": "stabilized_vanilla", | |
| "scale": 4.0, | |
| "stabilization": 0.02 | |
| }, | |
| "interpolation_guidance": { | |
| "name": "vanilla", | |
| "scale": 1.5 | |
| }, | |
| "interpolation_max_batch_size": 4, | |
| "camera_padding": "Nativehelperrepeatlastcamera forshortbatch;11activecameraentriesbecome16 onlastkeyframebatch", | |
| "parameter_counts": { | |
| "parameters": 458818051, | |
| "parameter_tensors": 494, | |
| "trainable_parameters": 0, | |
| "trainable_tensors": 0 | |
| }, | |
| "weights": { | |
| "source": "https://huggingface.co/Haoyuwu/GeometryForcing/blob/main/geometry_forcing_state_dict.ckpt", | |
| "policy": "released state_dict; no EMA substitution", | |
| "backbone_tensors": 496 | |
| }, | |
| "optimizer_groups": [], | |
| "lr_stages": [], | |
| "lora": [], | |
| "activation_checkpointing": false, | |
| "sampler": "DDIM50 eta0; continuouspred_v; fp16autocast", | |
| "scoring": "unchangedcommon rawRGB/AlexLPIPS/meanframePSNR/SSIM/pooledFID; notexactpaper-metricreproduction", | |
| "outputs": "geometry-forcing_native_re10k_*", | |
| "validation": "strictcheckpointCPU+actualnativeplannerrecording-sampler integrationpassed; GPUimagequalitypendingonescenesmoke", | |
| "initial_submission": "deployment/smoke_geometry_forcing_1h200.sh", | |
| "full_submission_after_review": "deployment/evaluate_geometry_forcing_8h200.sh" | |
| }, | |
| "minecraft": { | |
| "audit_date": "2026-09-11", | |
| "scope": "Read-only source audit; only this requested evidence JSON created. No evaluator edits, installations or GPU jobs.", | |
| "source_root": "/share_1/users/bonan_ding/.tmp/worldmem_baseline_evals_hf_20260910", | |
| "requested_reference_budget": { | |
| "local_total_slots": 8, | |
| "retrieved_memory_slots": 4, | |
| "total_slots": 12, | |
| "local_includes_target": true | |
| }, | |
| "minecraft_reference_caveat": { | |
| "finding": "The inspected current Minecraft DeMemWM source-free full-evaluation launcher actually requests 8 local plus up to 8 memory slots, not RE10K 8+4.", | |
| "local_total_slots": 8, | |
| "past_local_slots_at_chunk1": 7, | |
| "current_target_slots": 1, | |
| "memory_capacity": { | |
| "anchor": 2, | |
| "dynamic": 4, | |
| "revisit": 2, | |
| "total": 8 | |
| }, | |
| "memory_count_caveat": "These are capacities; valid selections and noise routing can reduce active memory at a denoising step.", | |
| "temporal_unit": "One RGB frame per latent position, frame_stack=1; VAE encode flattens time and batch into independent images.", | |
| "sources": [ | |
| "WorldMem/scripts/evaluate_dememwm_sourcefree_ref_long_horizon_8gpu.sh:104-120,141", | |
| "WorldMem/algorithms/dememwm/df_video.py:2620-2640", | |
| "WorldMem/configurations/algorithm/df_base.yaml:6", | |
| "WorldMem/algorithms/worldmem/df_video.py:572-581" | |
| ] | |
| }, | |
| "decmem": { | |
| "native_config": { | |
| "stm_sliding_window": 8, | |
| "num_frame_per_block": 4, | |
| "ltm_topk": 80, | |
| "ltm_tile_size": [ | |
| 1, | |
| 8, | |
| 8 | |
| ], | |
| "local_attn_size": -1, | |
| "sink_size": 0 | |
| }, | |
| "temporal_vae": { | |
| "rgb_to_latent": "1 + 4 + 4 + ... RGB frames -> 1 + 1 + 1 + ... latent positions; causal feature caches mean this is a sampling/compression ratio, not a strict receptive-field boundary.", | |
| "observed_rgb": 600, | |
| "left_padding_rgb": 1, | |
| "observed_latents": 151, | |
| "full_timeline_latents": 276, | |
| "generated_rgb": 500 | |
| }, | |
| "stm_during_steady_state_denoising": { | |
| "cached_previous_latents": 4, | |
| "current_block_latents": 4, | |
| "total_latent_positions": 8, | |
| "nominal_previous_rgb": 16, | |
| "nominal_current_rgb": 16, | |
| "attention_tokens_per_latent": 880, | |
| "attention_tokens_total": 7040, | |
| "why": "Each clean commit retains (stm_sliding_window - num_frame_per_block) latent positions. Denoising concatenates those cached K/V with the current block; it does not have 8 prior latent frames." | |
| }, | |
| "first_generated_block": { | |
| "latent_indices_in_stm": [ | |
| 144, | |
| 145, | |
| 146, | |
| 147, | |
| 148, | |
| 149, | |
| 150, | |
| 151 | |
| ], | |
| "previous_cached_latents": [ | |
| 144, | |
| 145, | |
| 146, | |
| 147 | |
| ], | |
| "current_clean_latents": [ | |
| 148, | |
| 149, | |
| 150 | |
| ], | |
| "current_generated_latents": [ | |
| 151 | |
| ], | |
| "nominal_observed_source_rgb_in_stm": [ | |
| 672, | |
| 699 | |
| ], | |
| "generated_source_rgb": [ | |
| 700, | |
| 703 | |
| ], | |
| "explanation": "151 observed latent positions leave three clean positions in the first 4-latent denoising block." | |
| }, | |
| "ltm": { | |
| "selection_unit": "Spatial key tiles per query tile and attention head, not frames.", | |
| "spatial_dit_grid": [ | |
| 22, | |
| 40 | |
| ], | |
| "tiles_per_latent": 15, | |
| "fine_attention_topk_tiles": 80, | |
| "padded_tokens_per_tile": 64, | |
| "fine_attention_max_padded_key_tokens_per_query_tile": 5120, | |
| "valid_tile_sizes": [ | |
| 48, | |
| 64 | |
| ], | |
| "cached_history_policy": "All committed prior blocks remain in tiled and compressed K/V; the current block is concatenated for each forward. No fixed temporal-memory-slot cap or eviction.", | |
| "includes_current_block": true, | |
| "selection_varies_per_query_head": true, | |
| "memory_frame_count": null, | |
| "not_equivalent_to_four_memory_frames": true, | |
| "why_topk_is_not_frame_count": "15 spatial tiles belong to one latent position; top-80 can combine regions from many times and is selected independently per query/head. Even changing topk to 4 or 60 would not mean four recalled RGB images." | |
| }, | |
| "persistent_image_anchor": null, | |
| "text_or_clip_image_conditioning": false, | |
| "vae_decoder_state": "Full causal observed+generated timeline decoded after denoising; this temporal decoder state is separate from transformer attention capacity.", | |
| "matches_12_rgb_frame_capacity": false, | |
| "sources": [ | |
| "DecMem/configs/decmem.yaml:23-40,56", | |
| "DecMem/evaluation/worldmem_adapter.py:21-26,82-112,124-137", | |
| "DecMem/pipeline/causal_diffusion_inference.py:78-111,247-340", | |
| "DecMem/wan/modules/causal_model.py:271-309", | |
| "DecMem/wan/modules/ltm_processor.py:284-329,371-418", | |
| "DecMem/wan/modules/vae.py:517-574" | |
| ] | |
| }, | |
| "matrix_game_2": { | |
| "native_config": { | |
| "local_attn_size": 6, | |
| "num_frame_per_block": 3, | |
| "sink_size": 0, | |
| "action_window_size": 3, | |
| "action_vae_time_compression_ratio": 4 | |
| }, | |
| "temporal_vae": { | |
| "rgb_to_latent": "1 + 4 + 4 + ... RGB frames -> 1 + 1 + 1 + ... latent positions; causal VAE features can depend on preceding RGB.", | |
| "observed_rgb": 600, | |
| "left_padding_rgb": 9, | |
| "observed_latents": 153, | |
| "generated_latents": 126, | |
| "generated_rgb_before_scoring_trim": 504, | |
| "scored_rgb": 500 | |
| }, | |
| "self_attention_during_denoising": { | |
| "cached_previous_latents": 3, | |
| "current_block_latents": 3, | |
| "total_latent_positions": 6, | |
| "nominal_previous_rgb": 12, | |
| "nominal_current_rgb": 12, | |
| "attention_tokens_per_latent": 880, | |
| "attention_tokens_total": 5280, | |
| "first_target_attention_latent_indices": [ | |
| 150, | |
| 151, | |
| 152, | |
| 153, | |
| 154, | |
| 155 | |
| ], | |
| "first_target_previous_source_rgb": [ | |
| 688, | |
| 699 | |
| ], | |
| "first_target_current_source_rgb": [ | |
| 700, | |
| 711 | |
| ], | |
| "why": "A 6-latent cache evicts 3 positions before appending each new 3-latent block; the current block consumes half of the configured local attention size." | |
| }, | |
| "retrieved_memory_slots": 0, | |
| "persistent_image_anchor": { | |
| "source_frame": 100, | |
| "selection": "First observed frame, including identical left-padding copies; not the last observed frame.", | |
| "clip_tokens": 257, | |
| "cross_attention_layers": 30, | |
| "included_in_local_attn_size_6": false, | |
| "policy": "The same image CLIP context is passed for every block, cached as cross-attention K/V. This is a persistent image condition, not four retrieved frames." | |
| }, | |
| "other_conditioning": { | |
| "cond_concat": "Native VAE encoding of a first-image-plus-zero-video sequence and an initial-frame mask; sliced for each block. No future ground-truth images.", | |
| "actions": "Separate keyboard/mouse K/V caches use the same 6-latent capacity; actions are controls and not additional observed RGB slots." | |
| }, | |
| "state_caveat": "All observed latent blocks prefill generator K/V and temporal decoder state. Cached hidden representations and autoregressive generated images can propagate historical information, so direct 6-position attention is not a hard bound on all historical influence.", | |
| "matches_12_rgb_frame_capacity": false, | |
| "sources": [ | |
| "Matrix-Game/Matrix-Game-2/worldmem_adapter.py:13-18,70-101,111-118", | |
| "Matrix-Game/Matrix-Game-2/configs/distilled_model/universal/config.json:27-34,43-49", | |
| "Matrix-Game/Matrix-Game-2/configs/inference_yaml/inference_universal.yaml:14", | |
| "Matrix-Game/Matrix-Game-2/pipeline/causal_inference.py:110-116,254-279,371-432", | |
| "Matrix-Game/Matrix-Game-2/wan/modules/causal_model.py:173-210", | |
| "Matrix-Game/Matrix-Game-2/wan/modules/model.py:228-258" | |
| ] | |
| }, | |
| "conclusion": "Neither Minecraft baseline implements an 8-image-local plus 4-image-memory layout. Report their native compression, current-block size, memory selection unit and persistent conditioning separately. Changing their settings to the integer 12 would not establish the same visual-information budget.", | |
| "validation": { | |
| "performed": [ | |
| "Read effective adapters and their called native cache/attention code.", | |
| "Executed only the actual pure-Python timeline functions extracted with AST: DecMem [151,1,276,500], Matrix [153,9,126].", | |
| "Checked first generated-block source-index arithmetic against each causal VAE packing rule." | |
| ], | |
| "not_performed": [ | |
| "No model execution, new GPU smoke or numerical comparison under changed capacities." | |
| ] | |
| } | |
| }, | |
| "validation": { | |
| "cpu": { | |
| "scope": "CPU native-sampler/control-boundary test with recording denoiser; reduced-size native Transformer mask/RoPE test underCPUfp16autocast, notcheckpointGPUevaluation", | |
| "passed": true, | |
| "window_frames": 12, | |
| "condition_frames": 11, | |
| "first_window": [ | |
| 89, | |
| 100 | |
| ], | |
| "all_generated_context_first_window": [ | |
| 100, | |
| 111 | |
| ], | |
| "turnaround_window": [ | |
| 189, | |
| 200 | |
| ], | |
| "last_window": [ | |
| 192, | |
| 203 | |
| ], | |
| "native_denoiser_calls": 1872, | |
| "generated_frames": 104, | |
| "observed_latents_unchanged": true, | |
| "real_scene": "516faacb7687d0c7", | |
| "architecture_input_t_retained": 32, | |
| "native_transformer_runtime_t": 12, | |
| "test_setup_correction": "Initial CPUTransformer test omitted autocast and hit Half/Float mismatch; passed withnativefp16autocast, no modelsourcechange." | |
| }, | |
| "launcher": { | |
| "python_syntax": "passed", | |
| "shell_syntax": "passed", | |
| "berzelius_cache_values": { | |
| "TMPDIR": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache", | |
| "TRITON_CACHE_DIR": "/proj/cvl/users/x_fahkh2/worldmem-baseline-evals/cache" | |
| }, | |
| "target_gpu_preflight_before_full": true, | |
| "separate_w12_output": true, | |
| "full_scope": { | |
| "cases": 100, | |
| "observed": 100, | |
| "generated": 300, | |
| "seed": 0, | |
| "workers": 8 | |
| }, | |
| "slurm_resources_unchanged": true | |
| }, | |
| "local_gpu_job": 4826, | |
| "local_gpu_status": "submitted; completion not verified", | |
| "berzelius_gpu_preflight": "mandatory104generatedframes beforefull100cases; anyfailurestopsjob", | |
| "geometry_native": { | |
| "cpu_integration": true, | |
| "strict_checkpoint_load": true, | |
| "cuda_execution": false, | |
| "gpu_image_quality_pending": true | |
| } | |
| }, | |
| "scope": "LIVE12 retained; GF nativeworkflow classifiedoffline; causal12controls prepared; Minecraftmethodsunchanged", | |
| "geometry_forcing_previous_dense": { | |
| "audit_date": "2026-09-11", | |
| "method": "Geometry Forcing", | |
| "dataset": "RE10K", | |
| "scope": "Read-only source/provenance audit; only this audit artifact was written. No baseline source changes, model inference, installs, or jobs.", | |
| "source_root": "/share_1/users/bonan_ding/.tmp/worldmem_baseline_evals_hf_20260910", | |
| "official_code_commit": "f5dfc6c5c3cbd2c0dad0121c0f7b161d2886e683", | |
| "checkpoint": { | |
| "source": "https://huggingface.co/Haoyuwu/GeometryForcing/blob/main/geometry_forcing_state_dict.ckpt", | |
| "loaded_backbone_tensors": 496, | |
| "parameters": 458818051, | |
| "weight_policy": "released state_dict; no EMA substitution", | |
| "existing_runtime_evidence": "/share_1/users/bonan_ding/.tmp/worldmem_baselines_20260909/artifacts/smoke-suite-4780/geometry-forcing/method_manifest_rank0.json" | |
| }, | |
| "current_context": { | |
| "observed_prefix_available_to_adapter": 100, | |
| "observed_rgb_actually_used_initially": [ | |
| 93, | |
| 100 | |
| ], | |
| "past_rgb_frames_per_prediction": 7, | |
| "generated_target_slots_per_prediction": 1, | |
| "retrieved_memory_reference_frames": 0, | |
| "active_slots_total": 8, | |
| "physical_backbone_slots": 16, | |
| "padding_slots": 8, | |
| "padding_semantics": "Pure-noise slots with context_mask=-1 and repeated current-target camera; not extra observations or memory. These slots remain in the transformer computation; this is not hard attention exclusion.", | |
| "rolling_rule": "At timeline t, use RGB t-7:t and target camera t; append generated t and discard oldest history frame.", | |
| "future_ground_truth_rgb_access": false, | |
| "reversal_mapping": "Timeline199=G199, timeline200=G199; camera indices follow the fixed duplicated-turnaround trajectory." | |
| }, | |
| "encoding_and_decoding": { | |
| "diffusion_space": "RGB pixel space", | |
| "input_output_shape_per_slot": [ | |
| 3, | |
| 256, | |
| 256 | |
| ], | |
| "temporal_compression_factor": 1, | |
| "vae": null, | |
| "extra_encoder_reference_frames": 0, | |
| "decoder_history_or_warmup_frames": 0, | |
| "spatial_processing": "U-ViT patch embedding and spatial down/up-sampling occur within the denoiser and preserve all time positions.", | |
| "output_processing": "Dataset channel normalization is inverted, then pixels are clamped to [0,1].", | |
| "vggt_alignment_encoder_at_inference": false | |
| }, | |
| "comparison_to_confirmed_ours": { | |
| "ours_local_slots_including_target": 8, | |
| "ours_past_local_frames": 7, | |
| "ours_memory_reference_slots": 4, | |
| "ours_total_active_slots": 12, | |
| "ours_total_reference_rgb_slots": 11, | |
| "deMemWM_memory_roles": { | |
| "anchor": 1, | |
| "dynamic": 2, | |
| "revisit": 1 | |
| }, | |
| "worldmem_memory_roles": { | |
| "anchor": 0, | |
| "dynamic": 4, | |
| "revisit": 0 | |
| }, | |
| "verdict": "Current GF matches the seven-frame local history and stays below the 12-active-slot budget, but does not match total reference capacity: 7 references versus ours up to 11.", | |
| "memory_vs_contiguous": "Ours selects four additional historical memory slots and routes them through its memory mechanism. GF has no such retrieval path; a 12-active-slot GF would use 11 contiguous past frames, not seven local plus four retrieved memories.", | |
| "archived_dememwm_manifest": "/tmp/hf_full_final100_80749/results/re10k_faithful_dememwm_full_final100_train731edd42758e_evaldc3cff220ba0_step70000_job80749/run_manifest.json", | |
| "archived_code_evidence": { | |
| "repository": "/share_1/users/bonan_ding/.tmp/WorldMem-vmem-fair-eval-0b88eda", | |
| "dememwm_eval_commit": "dc3cff220ba0af892e354c206ef58d33ef76577c", | |
| "worldmem_eval_commit": "73c1cc4f6787", | |
| "file": "algorithms/worldmem/dfot_realestate10k.py", | |
| "local_target_inclusion_lines": [ | |
| 4265, | |
| 4266, | |
| 4322, | |
| 4324, | |
| 4325, | |
| 4329, | |
| 4354, | |
| 4365 | |
| ], | |
| "worldmem_four_reference_requirement_lines": [ | |
| 1792, | |
| 1793, | |
| 1802, | |
| 1803 | |
| ] | |
| } | |
| }, | |
| "twelve_active_slot_feasibility": { | |
| "static_api_compatible": true, | |
| "proposed_reference_frames": 11, | |
| "proposed_target_slots": 1, | |
| "proposed_padding_slots": 4, | |
| "physical_backbone_slots_retained": 16, | |
| "requires_new_weights_or_training": false, | |
| "requires_adapter_edit": true, | |
| "needed_changes": "Use last11 history RGB, target-11:target+1 camera indices, 4 repeated target-camera pad slots, sampling length12, and updated provenance. Changing dataset.context_length alone has no effect on these hardcoded rollout dimensions.", | |
| "checkpoint_compatibility_basis": "Only active context/length/mask change; native maximum16 and all backbone tensors remain unchanged. Native sampler explicitly supports lengths<=16 with padding.", | |
| "do_not_set_physical_backbone_length_to_12": "Unnecessary model-configuration change; keep the released 16-slot backbone and pad4.", | |
| "runtime_validation_at_12": "Not performed; existing verified smoke used active8." | |
| }, | |
| "sampler_unchanged": { | |
| "steps": 50, | |
| "ddim_eta": 0.0, | |
| "guidance": "stabilized_vanilla", | |
| "guidance_scale": 4.0, | |
| "stabilization_level": 0.02, | |
| "precision": "fp16 autocast with fp32 camera processing" | |
| }, | |
| "sources": [ | |
| { | |
| "file": "GeometryForcing/evaluation/worldmem_adapter.py", | |
| "lines": [ | |
| 23, | |
| 32, | |
| 48, | |
| 50, | |
| 73, | |
| 91, | |
| 98, | |
| 100, | |
| 101, | |
| 107, | |
| 110, | |
| 111, | |
| 112 | |
| ], | |
| "supports": "Actual 7-history + 1-target input; camera padding to 16; strict released backbone load; normalization only; rolling history." | |
| }, | |
| { | |
| "file": "shared/run_and_score.py", | |
| "lines": [ | |
| 35, | |
| 36, | |
| 96, | |
| 97 | |
| ], | |
| "supports": "Production entrypoint selects the released checkpoint and exposes only first 100 RGB observations to Geometry Forcing." | |
| }, | |
| { | |
| "file": "GeometryForcing/configurations/dataset/realestate10k.yaml", | |
| "lines": [ | |
| 9, | |
| 12 | |
| ], | |
| "supports": "Native maximum is 16 frames at 256-square resolution." | |
| }, | |
| { | |
| "file": "GeometryForcing/configurations/dataset/base_video.yaml", | |
| "lines": [ | |
| 13, | |
| 14, | |
| 18, | |
| 23 | |
| ], | |
| "supports": "Latent diffusion disabled, temporal factor 1, direct 3-channel RGB inputs." | |
| }, | |
| { | |
| "file": "GeometryForcing/algorithms/dfot/dfot_video.py", | |
| "lines": [ | |
| 125, | |
| 128, | |
| 209, | |
| 215, | |
| 1155, | |
| 1187, | |
| 1192, | |
| 1209, | |
| 1214, | |
| 1219, | |
| 1347, | |
| 1348 | |
| ], | |
| "supports": "No VAE is loaded; frame-to-time-token conversion is 1:1; active lengths <=16 accepted and padded to native horizon, then trimmed." | |
| }, | |
| { | |
| "file": "GeometryForcing/algorithms/dfot/dfot_video.py", | |
| "lines": [ | |
| 940, | |
| 961, | |
| 962, | |
| 963, | |
| 1275, | |
| 1316, | |
| 1338, | |
| 1340 | |
| ], | |
| "supports": "Padding receives maximal noise level, is excluded from returned generated positions, and is held rather than generated." | |
| }, | |
| { | |
| "file": "GeometryForcing/algorithms/dfot/history_guidance.py", | |
| "lines": [ | |
| 343, | |
| 345, | |
| 369, | |
| 370 | |
| ], | |
| "supports": "History is mask>=1, targets mask==0, padding mask==-1; padding is not an additional RGB reference." | |
| }, | |
| { | |
| "file": "GeometryForcing/algorithms/dfot/backbones/u_vit/u_vit3d_pose.py", | |
| "lines": [ | |
| 63, | |
| 68, | |
| 81, | |
| 83, | |
| 89, | |
| 90, | |
| 123, | |
| 128, | |
| 131, | |
| 139, | |
| 140, | |
| 154, | |
| 155 | |
| ], | |
| "supports": "U-ViT requires its configured 16 physical time slots; per-frame spatial encoding/decoding preserves the time dimension. No per-frame padding attention mask is passed to this forward API." | |
| }, | |
| { | |
| "file": "GeometryForcing/algorithms/dfot/dfot_video_pose.py", | |
| "lines": [ | |
| 136, | |
| 142, | |
| 143, | |
| 147, | |
| 148 | |
| ], | |
| "supports": "VGGT alignment encoder is only constructed with alignment_coeff>0; adapter sets it to zero." | |
| } | |
| ], | |
| "recommendation": "Retain current GF results explicitly labeled local8/no-memory. A GF active12 rerun is feasible for equal total frame budget, but is a separate experimental variant requiring authorized adapter changes and a smoke run; no change is made by this audit." | |
| }, | |
| "geometry_forcing_budget_exception": "Previous native16-slot run retained only as an offline diagnostic; new past-only tests target12active slots.", | |
| "geometry_forcing_native_classification": "Offline diagnostic only: interpolation uses later generated keyframes; not an eligible strict autoregressive baseline.", | |
| "geometry_forcing_causal_controls": { | |
| "scope": "GF early-collapse diagnostics, not full benchmark", | |
| "modes": [ | |
| "ar8_cfg4", | |
| "ar12_cfg4", | |
| "ar12_cfg1", | |
| "teacher12_cfg4" | |
| ], | |
| "dataset": "RE10K monitor8 first two fixed scenes", | |
| "frames": 32, | |
| "scored_timeline": [ | |
| 100, | |
| 132 | |
| ], | |
| "primary_active_budget": 12, | |
| "physical_backbone_slots": 16, | |
| "causal_conditioning": "RGB strictly beforetarget; cameras only through currenttarget; currentcamera repeated fornoise-padding", | |
| "teacher_control": "Only pastRGB replacedwithGT; same masks andstabilization, diagnostic-only", | |
| "interpolation": "nevercalled", | |
| "frozen": true, | |
| "parameters": 458818051, | |
| "optimizer": null, | |
| "lr": null, | |
| "lora": null, | |
| "sampling": "DDIM50eta0; stabilization.02; originalstate_dict", | |
| "rng": "private/globalstatesrecorded;12slotCFGpair hasthe same targetnoise;8vs12changes targetslot", | |
| "cpu_validation_files": [ | |
| "cli_validation.json", | |
| "causality_validation.json", | |
| "checkpoint_interface_validation.json" | |
| ], | |
| "gpu_validation": "pendingBerzeliussubmission" | |
| } | |
| } | |