{ "action_contract": { "action_dimensions": 10, "action_horizon": 15, "action_vector_order": [ "relative_ee_xyz_3", "relative_ee_rotation_6d_6", "absolute_gripper_1" ], "coordinate_frame": "widowx_reference_base/teleop_aligned_tool", "gripper_source": "action.gripper_position", "gripper_target_representation": "absolute_normalized_0_1", "observation_dimensions": 10, "observation_horizon": 1, "pose_rotation_representation": "rotation_6d", "pose_source": "observation.ee_pose", "pose_source_semantics": "future_achieved_pose", "pose_target_representation": "relative_to_current_achieved_pose", "source_fps": 30, "target_fps": 5, "tool_frame_definition": { "jaw_symmetry_correction": [ [ 1.0, 0.0, 0.0 ], [ 0.0, 1.0, 0.0 ], [ 0.0, 0.0, 1.0 ] ], "paired_teleop_alignment": "derived_identity_from_panda_robotiq_mapping_and_widowx_jaw_axes", "recording_frame_id": "widowx_teleop_recording_frame_v1", "recording_raw_from_canonical": [ [ 0.0, 0.0, -1.0 ], [ 0.0, 1.0, 0.0 ], [ 1.0, 0.0, 0.0 ] ], "reference_from_robot_base": [ [ 1.0, 0.0, 0.0, -0.1 ], [ 0.0, 1.0, 0.0, 0.0 ], [ 0.0, 0.0, 1.0, 0.0 ], [ 0.0, 0.0, 0.0, 1.0 ] ], "reference_robot": "widowx250", "reference_tool_frame_definition_sha256": "f3aef5e783001a5718e4a0c42020b06d137fb429c04e544a67934dac7decf4ba", "robot_name": "panda-widowx", "schema_version": 1, "tool_frame_definition": { "axis_semantics": [ "forward", "left_finger", "palm_up" ], "canonical_reference_robot": "widowx250", "canonical_tool_frame_id": "widowx_tool_canonical_v1", "evidence": [ "compiled wx250s/gripper_link +X finger-forward equals grip_site raw +Z", "compiled left_finger positive slide opens along grip_site raw +Y", "grip_site is Ry(+90deg) relative to the transplanted WidowX gripper body" ], "raw_forward_axis": [ 0.0, 0.0, 1.0 ], "raw_from_canonical": [ [ 0.0, 0.0, -1.0 ], [ 0.0, 1.0, 0.0 ], [ 1.0, 0.0, 0.0 ] ], "raw_left_axis": [ 0.0, 1.0, 0.0 ], "raw_palm_up_axis": [ -1.0, 0.0, 0.0 ], "robot_name": "panda-widowx", "robot_xml_sha256": "5f8b242ead3600740065e135c609c34ed40bfdc29201d50a5b366abf0b13da3c", "schema_version": 1 }, "tool_frame_definition_sha256": "360ba397b682c1e042beebb3dd4f5cf47561601cef9c3d14c6558184ad37ecdd", "transform_semantics": "T_widowxbase_recordingtool=inv(T_world_widowxbase)@T_world_robotbase@T_robotbase_raw@T_raw_recordingtool", "world_from_reference_base": [ [ 1.0, 0.0, 0.0, 0.1 ], [ 0.0, 1.0, 0.0, 0.0 ], [ 0.0, 0.0, 1.0, 0.42 ], [ 0.0, 0.0, 0.0, 1.0 ] ], "world_from_robot_base": [ [ 1.0, 0.0, 0.0, 0.0 ], [ 0.0, 1.0, 0.0, 0.0 ], [ 0.0, 0.0, 1.0, 0.42 ], [ 0.0, 0.0, 0.0, 1.0 ] ] }, "tool_frame_definition_sha256": "0cf1ec052e9add9d54e479106a5617f30717a9a82da9c437505a1e6480ff6956", "tool_frame_id": "widowx_teleop_recording_frame_v1", "unused_command_pose": "action.ee_pose" }, "artifact_type": "mimicvideo_world2action_decoder", "base_artifacts": { "action_frame_initialization": { "numeric_transform": "identity", "reason": "widowx250 raw grip_site defines the canonical reference", "reference_robot": "widowx250", "schema_version": 1, "source_dataset_repo_id": "dreamdifferent/vam-cross-target-widowx250-native", "source_tool_frame_id": "native_grip_site_v1", "target_tool_frame_id": "widowx_teleop_recording_frame_v1" }, "initial_action_decoder": { "bytes": 998172132, "delivery": "standalone_hub", "iteration": 2374, "kind": "pretrained_world2action_decoder", "local_path": "initial_action_decoders/dreamdifferent--vam-cross-target-widowx250-native-2cam-action-decoder-93750cc/checkpoints/model/iter_000002374.pt", "manifest_path": "vam_cross_action_decoder_manifest.json", "path": "checkpoints/model/iter_000002374.pt", "path_in_repo": "checkpoints/model/iter_000002374.pt", "receipt_path": "initial_action_decoders/dreamdifferent--vam-cross-target-widowx250-native-2cam-action-decoder-93750cc/VAM_CROSS_INITIAL_ACTION_DECODER.json", "repo_id": "dreamdifferent/vam-cross-target-widowx250-native-2cam-action-decoder", "revision": "93750cccda01620e3c028477e4c49bc5c996a68d", "sha256": "3a9a75686164351468ef9b99098ae9152a1f07082b869e1ad03c9e1e1cf6885c", "source_iteration": 2374, "training_dataset_repo_id": "dreamdifferent/vam-cross-target-widowx250-native", "training_robot_name": "widowx250" }, "initial_video_backbone": { "bytes": 3913057284, "delivery": "standalone_hub", "kind": "fused_video2world_dit", "local_path": "initial_backbones/dreamdifferent--widowx250-video-fused-f0cea76/checkpoints/video_backbone/iter_000001060_fused.pt", "manifest_path": "vam_cross_fused_video_lora_manifest.json", "path_in_repo": "checkpoints/video_backbone/iter_000001060_fused.pt", "receipt_path": "initial_backbones/dreamdifferent--widowx250-video-fused-f0cea76/VAM_CROSS_INITIAL_VIDEO_BACKBONE.json", "repo_id": "dreamdifferent/widowx250-video-fused", "revision": "f0cea76b62c5dd66b06b9f965932ddea32a7b546", "sha256": "d0f24c049bee63b03d3b62747b240a2d1822ddd5f83a52fcd866a882e80122b1", "source_iteration": 1060 }, "mimicvideo_commit": "e3355dbc93132b576c02f920a59b4fc18a4f5906", "video_lora": { "bytes": 741588124, "delivery": "standalone_hub", "iteration": 200, "kind": "video2world_lora", "local_path": "selected_video_loras/dreamdifferent--vam-cross-level2-panda-widowx-widowx-texture-video-lora-iter200-5a22b23/checkpoints/model/iter_000000200.pt", "manifest_path": "vam_cross_video_lora_manifest.json", "path": "checkpoints/model/iter_000000200.pt", "path_in_repo": "checkpoints/model/iter_000000200.pt", "receipt_path": "selected_video_loras/dreamdifferent--vam-cross-level2-panda-widowx-widowx-texture-video-lora-iter200-5a22b23/VAM_CROSS_SELECTED_VIDEO_LORA.json", "repo_id": "dreamdifferent/vam-cross-level2-panda-widowx-widowx-texture-video-lora-iter200", "revision": "5a22b23f6f4bf80b249d6215d15260b7b59acbcf", "sha256": "45ff819e0044fc5a77d894e8f5d36caef9c28d68fb470ada0fe89724922e9b13", "source_iteration": 200 } }, "checkpoint": { "complete_components": [ "model", "optim", "scheduler", "trainer" ], "filename": "iter_000001800.pt", "iteration": 1800, "reached_requested_max_iterations": false, "requested_max_iterations": 3000, "selection": "explicit", "source_job_id": null, "termination_reason": "unknown", "uploaded_training_state": false }, "code": { "vam_cross_commit": "673f60cfc089a23dc920afa21526c10450a4dbbf" }, "created_at_utc": "2026-09-27T15:20:49.227697+00:00", "dataset": { "camera_keys": [ "observation.images.corner_cam", "observation.images.front_cam" ], "commit": "994be7f8952807008c316db7943ec732bf70b978", "episodes": 256, "frames": 54349, "instruction": null, "repo_id": "dreamdifferent/vam-cross-level2-panda-widowx-widowx-texture", "revision": "994be7f8952807008c316db7943ec732bf70b978", "robot_type": "panda-widowx", "source_fps": 30, "target_fps": 5, "task_instructions": [ "reach toward the bowl and stop immediately before contact", "reach toward the pot and stop immediately before contact", "reach toward the tomato and stop immediately before contact", "touch the pan without attempting to move it", "touch the dish sponge without grasping it", "touch the egg without grasping it", "grasp the apple without lifting it", "grasp the candle without lifting it", "release the grasped pepper shaker while it rests on the table", "release the held cup while keeping it at its current position", "lift the apple vertically and hold it without horizontal movement", "lift the bar of soap vertically and hold it without horizontal movement", "lower the held egg while maintaining the grasp", "lower the held potato while maintaining the grasp", "move the held cup horizontally a short distance", "move the held mug horizontally a short distance", "push the dish sponge forward without grasping it", "push the bar of soap forward without grasping it", "pull the potato backward toward the robot", "pull the bar of soap backward toward the robot", "rotate the potato in place around the vertical axis without translation", "rotate the dish sponge in place around the vertical axis without translation", "release the held tomato from a low height onto the table", "release the held candle from a low height onto the table" ] }, "files": { "checkpoints/model/iter_000001800.pt": { "bytes": 998172132, "sha256": "e9cf5ae373416faa966d04c9707080f815a3710fd29b53c4d837380b9b3bfa47" }, "config.yaml": { "bytes": 72384, "sha256": "0044315d397aac5cd66a86f1d8d5a48a5db9e1da22545373785493ea513cd219" }, "latest_checkpoint.txt": { "bytes": 18, "sha256": "9d7680d3739b134ff8ac9e2eee33528b7d4381b4c646db6384bc14d2e65c4469" }, "vam_cross_video2world_config.json": { "bytes": 17838, "sha256": "5ff1fe0e2d9b095a0db73757275980464fbcd25616630798f5b2eecc152210b3" }, "vam_cross_world2action_config.json": { "bytes": 10218, "sha256": "b84772ea506143d0ac5c747941203217daa5b5a014599d65b699f8b9a515725b" } }, "hub_repo_id": "dreamdifferent/vam-cross-level2-panda-widowx-widowx-texture-teleopaligned-videolora200-action-decoder-iter1800", "model": { "data_config": "bridge", "experiment": "w2a_panda_widowx_level2_widowx_texture_2cam_hstack_action_iter2374_videolora_iter200_widowx_teleop_recording_frame_v1", "initialization": "pretrained_action_decoder", "learning_rate": 0.0001, "stage": "world2action", "xattn_layer": 20 }, "private": false, "schema_version": 1 }