figurek1m's picture
Publish World2Action decoder iter_000001800.pt
0b7015e verified
Raw History Blame Contribute Delete
11.9 kB
{
"action_contract": {
"action_dimensions": 10,
"action_horizon": 15,
"action_vector_order": [
"relative_ee_xyz_3",
"relative_ee_rotation_6d_6",
"absolute_gripper_1"
],
"coordinate_frame": "widowx_reference_base/teleop_aligned_tool",
"gripper_source": "action.gripper_position",
"gripper_target_representation": "absolute_normalized_0_1",
"observation_dimensions": 10,
"observation_horizon": 1,
"pose_rotation_representation": "rotation_6d",
"pose_source": "observation.ee_pose",
"pose_source_semantics": "future_achieved_pose",
"pose_target_representation": "relative_to_current_achieved_pose",
"source_fps": 30,
"target_fps": 5,
"tool_frame_definition": {
"jaw_symmetry_correction": [
[
0.9999801549651472,
0.000586648295754,
-0.0062726007092316
],
[
-0.000306217139993,
-0.9899520639783521,
-0.1414030312831526
],
[
-0.0062925278659326,
0.141402145912918,
-0.9899322387033764
]
],
"paired_teleop_alignment": "same_so101_target_orientation_fitted_constant_gauge_removed",
"recording_frame_id": "widowx_teleop_recording_frame_v1",
"recording_raw_from_canonical": [
[
0.9999801549651472,
0.000586648295754,
-0.0062726007092316
],
[
0.0062925278659326,
-0.141402145912918,
0.9899322387033764
],
[
-0.000306217139993,
-0.9899520639783521,
-0.1414030312831526
]
],
"reference_from_robot_base": [
[
1.0,
0.0,
0.0,
0.04999999999999999
],
[
0.0,
1.0,
0.0,
0.0
],
[
0.0,
0.0,
1.0,
0.0
],
[
0.0,
0.0,
0.0,
1.0
]
],
"reference_robot": "widowx250",
"reference_tool_frame_definition_sha256": "f3aef5e783001a5718e4a0c42020b06d137fb429c04e544a67934dac7decf4ba",
"robot_name": "so101",
"schema_version": 1,
"tool_frame_definition": {
"axis_semantics": [
"forward",
"left_finger",
"palm_up"
],
"canonical_reference_robot": "widowx250",
"canonical_tool_frame_id": "widowx_tool_canonical_v1",
"evidence": [
"gripperframe raw +X follows the gripper body -Z fingertip direction",
"moving jaw lies on gripperframe raw +Z relative to the fixed jaw",
"single-jaw geometry uses the moving-jaw side as the positive lateral axis"
],
"raw_forward_axis": [
1.0,
0.0,
0.0
],
"raw_from_canonical": [
[
1.0,
0.0,
0.0
],
[
0.0,
0.0,
-1.0
],
[
0.0,
1.0,
0.0
]
],
"raw_left_axis": [
0.0,
0.0,
1.0
],
"raw_palm_up_axis": [
0.0,
-1.0,
0.0
],
"robot_name": "so101",
"robot_xml_sha256": "37e01af73a28743167321f584af79f88f2d725eb8cc6ce35001360ef1fe93268",
"schema_version": 1
},
"tool_frame_definition_sha256": "728ae1e2c2f0603cfedf3794b32f3707f05a8297e0739ed92e1264babc0f0783",
"transform_semantics": "T_widowxbase_recordingtool=inv(T_world_widowxbase)@T_world_robotbase@T_robotbase_raw@T_raw_recordingtool",
"world_from_reference_base": [
[
1.0,
0.0,
0.0,
0.1
],
[
0.0,
1.0,
0.0,
0.0
],
[
0.0,
0.0,
1.0,
0.42
],
[
0.0,
0.0,
0.0,
1.0
]
],
"world_from_robot_base": [
[
1.0,
0.0,
0.0,
0.15
],
[
0.0,
1.0,
0.0,
0.0
],
[
0.0,
0.0,
1.0,
0.42
],
[
0.0,
0.0,
0.0,
1.0
]
]
},
"tool_frame_definition_sha256": "37cdea9413fef920e6bd14f79596514761c6fdb5c7ee0fee3450d7f91f27c813",
"tool_frame_id": "widowx_teleop_recording_frame_v1",
"unused_command_pose": "action.ee_pose"
},
"artifact_type": "mimicvideo_world2action_decoder",
"base_artifacts": {
"action_frame_initialization": {
"numeric_transform": "identity",
"reason": "widowx250 raw grip_site defines the canonical reference",
"reference_robot": "widowx250",
"schema_version": 1,
"source_dataset_repo_id": "dreamdifferent/vam-cross-target-widowx250-native",
"source_tool_frame_id": "native_grip_site_v1",
"target_tool_frame_id": "widowx_teleop_recording_frame_v1"
},
"initial_action_decoder": {
"bytes": 998172132,
"delivery": "standalone_hub",
"iteration": 2374,
"kind": "pretrained_world2action_decoder",
"local_path": "initial_action_decoders/dreamdifferent--vam-cross-target-widowx250-native-2cam-action-decoder-93750cc/checkpoints/model/iter_000002374.pt",
"manifest_path": "vam_cross_action_decoder_manifest.json",
"path": "checkpoints/model/iter_000002374.pt",
"path_in_repo": "checkpoints/model/iter_000002374.pt",
"receipt_path": "initial_action_decoders/dreamdifferent--vam-cross-target-widowx250-native-2cam-action-decoder-93750cc/VAM_CROSS_INITIAL_ACTION_DECODER.json",
"repo_id": "dreamdifferent/vam-cross-target-widowx250-native-2cam-action-decoder",
"revision": "93750cccda01620e3c028477e4c49bc5c996a68d",
"sha256": "3a9a75686164351468ef9b99098ae9152a1f07082b869e1ad03c9e1e1cf6885c",
"source_iteration": 2374,
"training_dataset_repo_id": "dreamdifferent/vam-cross-target-widowx250-native",
"training_robot_name": "widowx250"
},
"initial_video_backbone": {
"bytes": 3913057284,
"delivery": "standalone_hub",
"kind": "fused_video2world_dit",
"local_path": "initial_backbones/dreamdifferent--widowx250-video-fused-f0cea76/checkpoints/video_backbone/iter_000001060_fused.pt",
"manifest_path": "vam_cross_fused_video_lora_manifest.json",
"path_in_repo": "checkpoints/video_backbone/iter_000001060_fused.pt",
"receipt_path": "initial_backbones/dreamdifferent--widowx250-video-fused-f0cea76/VAM_CROSS_INITIAL_VIDEO_BACKBONE.json",
"repo_id": "dreamdifferent/widowx250-video-fused",
"revision": "f0cea76b62c5dd66b06b9f965932ddea32a7b546",
"sha256": "d0f24c049bee63b03d3b62747b240a2d1822ddd5f83a52fcd866a882e80122b1",
"source_iteration": 1060
},
"mimicvideo_commit": "e3355dbc93132b576c02f920a59b4fc18a4f5906",
"video_lora": {
"bytes": 741588124,
"delivery": "standalone_hub",
"iteration": 400,
"kind": "video2world_lora",
"local_path": "selected_video_loras/dreamdifferent--vam-cross-level4-so101-widowx-texture-video-lora-iter-400-d660324/checkpoints/model/iter_000000400.pt",
"manifest_path": "vam_cross_video_lora_manifest.json",
"path": "checkpoints/model/iter_000000400.pt",
"path_in_repo": "checkpoints/model/iter_000000400.pt",
"receipt_path": "selected_video_loras/dreamdifferent--vam-cross-level4-so101-widowx-texture-video-lora-iter-400-d660324/VAM_CROSS_SELECTED_VIDEO_LORA.json",
"repo_id": "dreamdifferent/vam-cross-level4-so101-widowx-texture-video-lora-iter-400",
"revision": "d66032413191d12b18eb5c97be9f649db3ffcc26",
"sha256": "ab7741672f83baefc5a4bd56a4ba867eba87c9ad5eb118ba60037b1dac3b4f82",
"source_iteration": 400
}
},
"checkpoint": {
"complete_components": [
"model",
"optim",
"scheduler",
"trainer"
],
"filename": "iter_000001800.pt",
"iteration": 1800,
"reached_requested_max_iterations": true,
"requested_max_iterations": 1800,
"selection": "explicit",
"source_job_id": null,
"termination_reason": "completed",
"uploaded_training_state": false
},
"code": {
"vam_cross_commit": "3de3a6a94784b9524ac11a4d3a403ca9921c8397"
},
"created_at_utc": "2026-09-24T13:56:56.940404+00:00",
"dataset": {
"camera_keys": [
"observation.images.corner_cam",
"observation.images.front_cam"
],
"commit": "4d2d4b0418eccc9f9398a2745ad2e4ed766a4ef6",
"episodes": 151,
"frames": 54340,
"instruction": null,
"repo_id": "dreamdifferent/vam-cross-level4-so101-widowx-texture",
"revision": "4d2d4b0418eccc9f9398a2745ad2e4ed766a4ef6",
"robot_type": "so101",
"source_fps": 30,
"target_fps": 5,
"task_instructions": [
"pick up the apple and place it into the pan",
"pick up the apple and place it into the pot",
"pick up the apple and place it into the bowl",
"pick up the candle and place it into the pan",
"pick up the candle and place it into the pot",
"pick up the cup and place it into the bowl",
"pick up the cup and place it into the pan",
"pick up the cup and place it into the pot",
"pick up the dish sponge and place it into the bowl",
"pick up the dish sponge and place it into the pan",
"pick up the dish sponge and place it into the pot",
"pick up the egg and place it into the bowl",
"pick up the egg and place it into the pan",
"pick up the egg and place it into the pot",
"pick up the mug and place it into the bowl",
"pick up the mug and place it into the pan",
"pick up the mug and place it into the pot",
"pick up the pepper shaker and place it into the bowl",
"pick up the pepper shaker and place it into the pan",
"pick up the pepper shaker and place it into the pot",
"pick up the potato and place it into the bowl",
"pick up the potato and place it into the pan",
"pick up the potato and place it into the pot",
"pick up the tomato and place it into the bowl",
"pick up the tomato and place it into the pan",
"pick up the tomato and place it into the pot",
"pick up the bar of soap and place it into the bowl",
"pick up the bar of soap and place it into the pan",
"pick up the bar of soap and place it into the pot"
]
},
"files": {
"checkpoints/model/iter_000001800.pt": {
"bytes": 998172132,
"sha256": "c3bffbfb4b7f41e63981e17ac53a7880e017d790e6957dd76c85a2877ac1fcfe"
},
"config.yaml": {
"bytes": 72298,
"sha256": "711f24342a692db2bca4e3b48487dc127e760be096eebc8af9af9136d32f961f"
},
"latest_checkpoint.txt": {
"bytes": 18,
"sha256": "9d7680d3739b134ff8ac9e2eee33528b7d4381b4c646db6384bc14d2e65c4469"
},
"vam_cross_video2world_config.json": {
"bytes": 17585,
"sha256": "684f9b3cbf2f9077d557922c22650b26d00290e6c2db4aa1743d33d8e6f39290"
},
"vam_cross_world2action_config.json": {
"bytes": 10407,
"sha256": "67e0bd7227ab91a1173a2edd7f5a0db725dad63a306fa0527462fe791e01a22b"
}
},
"hub_repo_id": "dreamdifferent/vam-cross-level4-so101-widowx-texture-teleopaligned-videolora400-action-decoder-iter1800",
"model": {
"data_config": "bridge",
"experiment": "w2a_so101_level4_widowx_texture_2cam_hstack_action_iter2374_videolora_iter400_widowx_teleop_recording_frame_v1",
"initialization": "pretrained_action_decoder",
"learning_rate": 0.0001,
"stage": "world2action",
"xattn_layer": 20
},
"private": false,
"schema_version": 1
}