output_dir: ./runs/robotwin_joint_3cam_384_1e-4_ep10_clean_task_drop-10/2026-08-13_22-26-04 batch_size: null global_batch_size: 128 num_workers: 8 lr_scheduler_type: cosine learning_rate: 0.0001 num_epochs: 10 max_steps: null log_every: 10 save_every: 10000 save_state_every: 500000 eval_every: 500 eval_num_inference_steps: 10 gradient_accumulation_steps: 1 split_video_action_loss_backward: false split_video_action_loss_mode: alternating_batches mixed_precision: bf16 seed: 42 max_grad_norm: 1.0 weight_decay: 0.01 resume: null sampling: weighted: false default_weight: 1.0 combine: max replacement: true num_samples: null rules: [] overwrite_video_latents: false video_latent_save_dtype: fp16 video_latent_batch_size: 8 video_latent_num_workers: 8 wandb: enabled: true workspace: tehaji007 project: fast-wam name: robotwin_joint_3cam_384_1e-4_ep10_clean_task_drop-10 group: null mode: offline data: train: _target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset dataset_dirs: - ./data/robotwin2.0/robotwin2.0 shape_meta: images: - key: cam_high raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_left_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_right_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 action: - key: default raw_shape: 14 shape: 14 state: - key: default raw_shape: 14 shape: 14 num_frames: 33 global_sample_stride: 1 action_video_freq_ratio: 4 video_size: - 384 - 320 camera_key: null val_set_proportion: 0.01 is_training_set: true pretrained_norm_stats: ./data/robotwin2.0/dataset_stats.json skip_padding_as_possible: false concat_multi_camera: robotwin use_precomputed_video_latents: true video_latent_cache_dir: ./data/video_latents/robotwin_uncond_3cam_384_1e-4_all_tasks strict_video_latent_loading: true video_latent_storage_format: npz_compressed video_latent_shard_size: 10000 use_precomputed_metadata_cache: true metadata_cache_dir: ./data/video_latents/robotwin_uncond_3cam_384_1e-4_all_tasks_metadata_cache metadata_cache_shard_size: 10000 subset_sample_indices_file: null prompt_subset_file: null robotwin_clean_only: true robotwin_clean_group_size: 550 robotwin_clean_episodes_per_group: 50 robotwin_clean_random_select: false robotwin_clean_random_seed: 0 robotwin_clean_pool_episodes_per_group: 50 processor: _target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor shape_meta: images: - key: cam_high raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_left_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_right_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 action: - key: default raw_shape: 14 shape: 14 state: - key: default raw_shape: 14 shape: 14 num_obs_steps: 33 num_output_cameras: 3 action_output_dim: 14 proprio_output_dim: 14 action_state_transforms: null use_stepwise_action_norm: false norm_default_mode: z-score norm_exception_mode: null action_state_merger: _target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign train_transforms: - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: - 240 - 320 val_transforms: - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: - 240 - 320 text_embedding_cache_dir: ./data/text_embeds_cache/robotwin context_len: 128 robotwin_drop_tasks: - blocks_ranking_rgb - click_bell - handover_mic - move_stapler_pad - pick_diverse_bottles - place_a2b_right - place_can_basket - place_object_stand - shake_bottle_horizontally - stack_bowls_three robotwin_action_loss_drop_tasks: [] val: _target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset dataset_dirs: - ./data/robotwin2.0/robotwin2.0/ shape_meta: images: - key: cam_high raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_left_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_right_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 action: - key: default raw_shape: 14 shape: 14 state: - key: default raw_shape: 14 shape: 14 num_frames: 33 global_sample_stride: 1 action_video_freq_ratio: 4 video_size: - 384 - 320 camera_key: null val_set_proportion: 0.01 is_training_set: false pretrained_norm_stats: ./data/robotwin2.0/dataset_stats.json skip_padding_as_possible: false concat_multi_camera: robotwin use_precomputed_video_latents: false video_latent_cache_dir: null strict_video_latent_loading: true video_latent_storage_format: npz_compressed video_latent_shard_size: 10000 use_precomputed_metadata_cache: false metadata_cache_dir: null metadata_cache_shard_size: 10000 subset_sample_indices_file: null prompt_subset_file: null robotwin_clean_only: true robotwin_clean_group_size: 550 robotwin_clean_episodes_per_group: 50 robotwin_clean_random_select: false robotwin_clean_random_seed: 0 robotwin_clean_pool_episodes_per_group: 50 processor: _target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor shape_meta: images: - key: cam_high raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_left_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 - key: cam_right_wrist raw_shape: - 3 - 480 - 640 shape: - 3 - 240 - 320 action: - key: default raw_shape: 14 shape: 14 state: - key: default raw_shape: 14 shape: 14 num_obs_steps: 33 num_output_cameras: 3 action_output_dim: 14 proprio_output_dim: 14 action_state_transforms: null use_stepwise_action_norm: false norm_default_mode: z-score norm_exception_mode: null action_state_merger: _target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign train_transforms: - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: - 240 - 320 val_transforms: - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor - _target_: torchvision.transforms.Resize size: - 240 - 320 text_embedding_cache_dir: ./data/text_embeds_cache/robotwin context_len: 128 model: _target_: fastwam.runtime.create_fastwam_joint model_id: Wan-AI/Wan2.2-TI2V-5B tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B tokenizer_max_len: 128 load_text_encoder: false proprio_dim: 14 redirect_common_files: false mot_checkpoint_mixed_attn: false action_dit_pretrained_path: checkpoints/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt skip_dit_load_from_pretrain: false keep_dit_layers: null keep_dit_layer_indices: null video_dit_config: has_image_input: false patch_size: - 1 - 2 - 2 in_dim: 48 hidden_dim: 3072 ffn_dim: 14336 freq_dim: 256 text_dim: 4096 out_dim: 48 num_heads: 24 attn_head_dim: 128 num_layers: 30 eps: 1.0e-06 seperated_timestep: true require_clip_embedding: false require_vae_embedding: false fuse_vae_embedding_in_latents: true use_gradient_checkpointing: false video_attention_mask_mode: first_frame_causal action_conditioned: false action_dim: 14 action_group_causal_mask_mode: group_diagonal action_dit_config: action_dim: 14 hidden_dim: 1024 ffn_dim: 4096 num_heads: 24 attn_head_dim: 128 num_layers: 30 text_dim: 4096 freq_dim: 256 eps: 1.0e-06 use_gradient_checkpointing: false video_scheduler: train_shift: 5.0 infer_shift: 5.0 num_train_timesteps: 1000 action_scheduler: train_shift: 5.0 infer_shift: 5.0 num_train_timesteps: 1000 loss: lambda_action: 1.0