Download config.yaml from deformable-bench/fastwam-flatten-silk: direct link, hf CLI and curl.
- Browser
- Download file 5.06 kB
-
https://huggingface.co/deformable-bench/fastwam-flatten-silk/resolve/main/config.yaml
- Command line
-
hf download hf://deformable-bench/fastwam-flatten-silk/config.yaml
-
curl -L -o config.yaml https://huggingface.co/deformable-bench/fastwam-flatten-silk/resolve/main/config.yaml
5.06 kB
| output_dir: /horizon-bucket/robot_lab/users/wenkang.hu-labs/projects/deformable_bench/runs/flatten-silk/fastwam/production-20260718T232050Z | |
| batch_size: 16 | |
| num_workers: 0 | |
| lr_scheduler_type: cosine | |
| learning_rate: 0.0001 | |
| num_epochs: 5 | |
| max_steps: 3145 | |
| log_every: 10 | |
| save_every: 0 | |
| eval_every: 0 | |
| eval_num_inference_steps: 10 | |
| gradient_accumulation_steps: 1 | |
| mixed_precision: bf16 | |
| seed: 42 | |
| max_grad_norm: 1.0 | |
| weight_decay: 0.01 | |
| resume: null | |
| wandb: | |
| enabled: false | |
| workspace: null | |
| project: fast-wam | |
| name: silk_grasp_uncond_3cam_384_1e-4 | |
| group: null | |
| mode: online | |
| data: | |
| train: | |
| _target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset | |
| dataset_dirs: | |
| - /horizon-bucket/robot_lab/users/wenkang.hu-labs/projects/deformable_bench/datasets/huggingface/deformable-bench/flatten-silk/e465c61e9c03674cd46f215c3c1fa86e4071a6e2 | |
| shape_meta: | |
| images: | |
| - key: static_cam | |
| raw_shape: | |
| - 3 | |
| - 720 | |
| - 1280 | |
| shape: | |
| - 3 | |
| - 240 | |
| - 320 | |
| - key: left_hand_cam | |
| raw_shape: | |
| - 3 | |
| - 720 | |
| - 1280 | |
| shape: | |
| - 3 | |
| - 240 | |
| - 320 | |
| - key: right_hand_cam | |
| raw_shape: | |
| - 3 | |
| - 720 | |
| - 1280 | |
| shape: | |
| - 3 | |
| - 240 | |
| - 320 | |
| action: | |
| - key: default | |
| raw_shape: 14 | |
| shape: 14 | |
| state: | |
| - key: default | |
| raw_shape: 14 | |
| shape: 14 | |
| num_frames: 33 | |
| global_sample_stride: 1 | |
| action_video_freq_ratio: 4 | |
| video_size: | |
| - 384 | |
| - 320 | |
| camera_key: null | |
| val_set_proportion: 0.0 | |
| is_training_set: true | |
| pretrained_norm_stats: null | |
| skip_padding_as_possible: false | |
| concat_multi_camera: robotwin | |
| processor: | |
| _target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor | |
| shape_meta: | |
| images: | |
| - key: static_cam | |
| raw_shape: | |
| - 3 | |
| - 720 | |
| - 1280 | |
| shape: | |
| - 3 | |
| - 240 | |
| - 320 | |
| - key: left_hand_cam | |
| raw_shape: | |
| - 3 | |
| - 720 | |
| - 1280 | |
| shape: | |
| - 3 | |
| - 240 | |
| - 320 | |
| - key: right_hand_cam | |
| raw_shape: | |
| - 3 | |
| - 720 | |
| - 1280 | |
| shape: | |
| - 3 | |
| - 240 | |
| - 320 | |
| action: | |
| - key: default | |
| raw_shape: 14 | |
| shape: 14 | |
| state: | |
| - key: default | |
| raw_shape: 14 | |
| shape: 14 | |
| num_obs_steps: 33 | |
| num_output_cameras: 3 | |
| action_output_dim: 14 | |
| proprio_output_dim: 14 | |
| action_state_transforms: null | |
| use_stepwise_action_norm: false | |
| norm_default_mode: z-score | |
| norm_exception_mode: null | |
| action_state_merger: | |
| _target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign | |
| train_transforms: | |
| - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor | |
| - _target_: torchvision.transforms.Resize | |
| size: | |
| - 240 | |
| - 320 | |
| val_transforms: | |
| - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor | |
| - _target_: torchvision.transforms.Resize | |
| size: | |
| - 240 | |
| - 320 | |
| text_embedding_cache_dir: /horizon-bucket/robot_lab/users/wenkang.hu-labs/projects/deformable_bench/derived/fastwam/text-embeddings/deformable-bench-four-datasets/v3-wan22ti2v5b-t5len128 | |
| context_len: 128 | |
| model: | |
| _target_: fastwam.runtime.create_fastwam | |
| model_id: Wan-AI/Wan2.2-TI2V-5B | |
| tokenizer_model_id: Wan-AI/Wan2.1-T2V-1.3B | |
| tokenizer_max_len: 128 | |
| load_text_encoder: false | |
| proprio_dim: 14 | |
| redirect_common_files: false | |
| mot_checkpoint_mixed_attn: false | |
| action_dit_pretrained_path: /horizon-bucket/robot_lab/users/wenkang.hu-labs/projects/deformable_bench/derived/fastwam/actiondit/sha256-7bd65e1986accaaf2e8dd5e59b5e98ef348d7c2f1d88ab2df010eb954a441445/ActionDiT_linear_interp_Wan22_alphascale_1024hdim.pt | |
| skip_dit_load_from_pretrain: false | |
| video_dit_config: | |
| has_image_input: false | |
| patch_size: | |
| - 1 | |
| - 2 | |
| - 2 | |
| in_dim: 48 | |
| hidden_dim: 3072 | |
| ffn_dim: 14336 | |
| freq_dim: 256 | |
| text_dim: 4096 | |
| out_dim: 48 | |
| num_heads: 24 | |
| attn_head_dim: 128 | |
| num_layers: 30 | |
| eps: 1.0e-06 | |
| seperated_timestep: true | |
| require_clip_embedding: false | |
| require_vae_embedding: false | |
| fuse_vae_embedding_in_latents: true | |
| use_gradient_checkpointing: false | |
| video_attention_mask_mode: first_frame_causal | |
| action_conditioned: false | |
| action_dim: 14 | |
| action_group_causal_mask_mode: group_diagonal | |
| action_dit_config: | |
| action_dim: 14 | |
| hidden_dim: 1024 | |
| ffn_dim: 4096 | |
| num_heads: 24 | |
| attn_head_dim: 128 | |
| num_layers: 30 | |
| text_dim: 4096 | |
| freq_dim: 256 | |
| eps: 1.0e-06 | |
| use_gradient_checkpointing: false | |
| video_scheduler: | |
| train_shift: 5.0 | |
| infer_shift: 5.0 | |
| num_train_timesteps: 1000 | |
| action_scheduler: | |
| train_shift: 5.0 | |
| infer_shift: 5.0 | |
| num_train_timesteps: 1000 | |
| loss: | |
| lambda_action: 1.0 | |