arunos728 commited on
Commit
1ef1594
·
verified ·
1 Parent(s): f05ac09

fastwam precision8 vert576, step 80k

Browse files
Files changed (1) hide show
  1. config.yaml +202 -0
config.yaml ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ output_dir: ./runs/robodojo_precision8_uncond_3cam576_1e-4/2026-09-24_13-48-32
2
+ batch_size: 8
3
+ num_workers: 8
4
+ prefetch_factor: 4
5
+ lr_scheduler_type: cosine
6
+ learning_rate: 0.0001
7
+ num_epochs: 18
8
+ max_steps: 100000
9
+ log_every: 10
10
+ save_every: 5000
11
+ eval_every: 200
12
+ eval_num_inference_steps: 10
13
+ gradient_accumulation_steps: 1
14
+ mixed_precision: bf16
15
+ seed: 42
16
+ max_grad_norm: 1.0
17
+ weight_decay: 0.01
18
+ resume: runs/robodojo_precision8_uncond_3cam576_1e-4/2026-09-24_13-48-32/checkpoints/state/step_055000
19
+ wandb:
20
+ enabled: false
21
+ workspace: null
22
+ project: fast-wam
23
+ name: robodojo_precision8_uncond_3cam576_1e-4
24
+ group: null
25
+ mode: online
26
+ data:
27
+ train:
28
+ _target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset
29
+ dataset_dirs:
30
+ - /fsx/rlwrld/heeseung/datasets/lerobot_v21_precision8_320x240
31
+ shape_meta:
32
+ images:
33
+ - key: cam_high
34
+ raw_shape:
35
+ - 3
36
+ - 240
37
+ - 320
38
+ shape:
39
+ - 3
40
+ - 192
41
+ - 256
42
+ - key: cam_left_wrist
43
+ raw_shape:
44
+ - 3
45
+ - 240
46
+ - 320
47
+ shape:
48
+ - 3
49
+ - 192
50
+ - 256
51
+ - key: cam_right_wrist
52
+ raw_shape:
53
+ - 3
54
+ - 240
55
+ - 320
56
+ shape:
57
+ - 3
58
+ - 192
59
+ - 256
60
+ action:
61
+ - key: default
62
+ lerobot_key: action
63
+ raw_shape: 14
64
+ shape: 14
65
+ state:
66
+ - key: default
67
+ lerobot_key: observation.state
68
+ raw_shape: 14
69
+ shape: 14
70
+ num_frames: 33
71
+ global_sample_stride: 1
72
+ action_video_freq_ratio: 4
73
+ video_size:
74
+ - 576
75
+ - 256
76
+ camera_key: null
77
+ val_set_proportion: 0.0
78
+ is_training_set: true
79
+ skip_padding_as_possible: false
80
+ concat_multi_camera: vertical
81
+ processor:
82
+ _target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor
83
+ shape_meta:
84
+ images:
85
+ - key: cam_high
86
+ raw_shape:
87
+ - 3
88
+ - 240
89
+ - 320
90
+ shape:
91
+ - 3
92
+ - 192
93
+ - 256
94
+ - key: cam_left_wrist
95
+ raw_shape:
96
+ - 3
97
+ - 240
98
+ - 320
99
+ shape:
100
+ - 3
101
+ - 192
102
+ - 256
103
+ - key: cam_right_wrist
104
+ raw_shape:
105
+ - 3
106
+ - 240
107
+ - 320
108
+ shape:
109
+ - 3
110
+ - 192
111
+ - 256
112
+ action:
113
+ - key: default
114
+ lerobot_key: action
115
+ raw_shape: 14
116
+ shape: 14
117
+ state:
118
+ - key: default
119
+ lerobot_key: observation.state
120
+ raw_shape: 14
121
+ shape: 14
122
+ num_obs_steps: 33
123
+ num_output_cameras: 3
124
+ action_output_dim: 14
125
+ proprio_output_dim: 14
126
+ action_state_transforms: null
127
+ use_stepwise_action_norm: false
128
+ norm_default_mode: z-score
129
+ norm_exception_mode: null
130
+ action_state_merger:
131
+ _target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign
132
+ train_transforms:
133
+ - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
134
+ - _target_: torchvision.transforms.Resize
135
+ size:
136
+ - 192
137
+ - 256
138
+ val_transforms:
139
+ - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor
140
+ - _target_: torchvision.transforms.Resize
141
+ size:
142
+ - 192
143
+ - 256
144
+ text_embedding_cache_dir: ./data/text_embeds_cache/lerobot_v21_precision8_320x240
145
+ context_len: 128
146
+ model:
147
+ _target_: fastwam.runtime.create_fastwam
148
+ model_id: Wan-AI/Wan2.2-TI2V-5B
149
+ tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B
150
+ tokenizer_max_len: 128
151
+ load_text_encoder: false
152
+ proprio_dim: 14
153
+ redirect_common_files: false
154
+ mot_checkpoint_mixed_attn: false
155
+ action_dit_pretrained_path: checkpoints/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt
156
+ skip_dit_load_from_pretrain: false
157
+ video_dit_config:
158
+ has_image_input: false
159
+ patch_size:
160
+ - 1
161
+ - 2
162
+ - 2
163
+ in_dim: 48
164
+ hidden_dim: 3072
165
+ ffn_dim: 14336
166
+ freq_dim: 256
167
+ text_dim: 4096
168
+ out_dim: 48
169
+ num_heads: 24
170
+ attn_head_dim: 128
171
+ num_layers: 30
172
+ eps: 1.0e-06
173
+ seperated_timestep: true
174
+ require_clip_embedding: false
175
+ require_vae_embedding: false
176
+ fuse_vae_embedding_in_latents: true
177
+ use_gradient_checkpointing: false
178
+ video_attention_mask_mode: first_frame_causal
179
+ action_conditioned: false
180
+ action_dim: 14
181
+ action_group_causal_mask_mode: group_diagonal
182
+ action_dit_config:
183
+ action_dim: 14
184
+ hidden_dim: 1024
185
+ ffn_dim: 4096
186
+ num_heads: 24
187
+ attn_head_dim: 128
188
+ num_layers: 30
189
+ text_dim: 4096
190
+ freq_dim: 256
191
+ eps: 1.0e-06
192
+ use_gradient_checkpointing: false
193
+ video_scheduler:
194
+ train_shift: 5.0
195
+ infer_shift: 5.0
196
+ num_train_timesteps: 1000
197
+ action_scheduler:
198
+ train_shift: 5.0
199
+ infer_shift: 5.0
200
+ num_train_timesteps: 1000
201
+ loss:
202
+ lambda_action: 1.0