Download config.yaml from yuyangalin/ImageWAM-FLUX.2-4B-LIBERO: direct link, hf CLI and curl.
- Browser
- Download file 5.97 kB
-
https://huggingface.co/yuyangalin/ImageWAM-FLUX.2-4B-LIBERO/resolve/main/config.yaml
- Command line
-
hf download hf://yuyangalin/ImageWAM-FLUX.2-4B-LIBERO/config.yaml
-
curl -L -o config.yaml https://huggingface.co/yuyangalin/ImageWAM-FLUX.2-4B-LIBERO/resolve/main/config.yaml
5.97 kB
| output_dir: ./runs/libero_flux2_klein_4b_base_fastwam/2026-05-27_04-13-04 | |
| batch_size: 10 | |
| num_workers: 24 | |
| prefetch_factor: 2 | |
| persistent_workers: true | |
| lr_scheduler_type: cosine | |
| learning_rate: 0.0001 | |
| warmup_steps: null | |
| num_epochs: 10 | |
| max_steps: null | |
| log_every: 10 | |
| save_every: 1000 | |
| keep_latest_state_only: true | |
| eval_every: 100 | |
| eval_num_inference_steps: 10 | |
| eval_num_samples: 8 | |
| rank_timer_every: 10 | |
| rank_timer_sync_cuda: true | |
| qwen_cache_batch_size: 32 | |
| qwen_cache_overwrite: false | |
| qwen_cache_save_workers: 8 | |
| flux2_qwen3_model_spec: null | |
| precache_arrow: true | |
| precache_num_workers: 4 | |
| build_lerobot_meta_cache: true | |
| build_norm_stats: true | |
| precache_fuse_norm_stats: true | |
| force_rebuild_norm_stats: false | |
| norm_stats_use_arrow_cache: true | |
| norm_stats_bulk_read_arrow: true | |
| norm_stats_cache_enabled: true | |
| norm_stats_cache_dir: null | |
| precache_profile: false | |
| precache_profile_min_sec: 0.0 | |
| gradient_accumulation_steps: 1 | |
| mixed_precision: bf16 | |
| seed: 42 | |
| max_grad_norm: 1.0 | |
| weight_decay: 0.01 | |
| resume: null | |
| wandb: | |
| enabled: true | |
| workspace: arisilin | |
| project: fast-wam | |
| name: libero_flux2_klein_4b_base_fastwam | |
| group: null | |
| mode: online | |
| data: | |
| train: | |
| _target_: fastwam.datasets.lerobot.robot_video_dataset.RobotVideoDataset | |
| dataset_dirs: | |
| - /dockerdata/data/libero/libero_spatial_no_noops_lerobot | |
| - /dockerdata/data/libero/libero_object_no_noops_lerobot | |
| - /dockerdata/data/libero/libero_goal_no_noops_lerobot | |
| - /dockerdata/data/libero/libero_10_no_noops_lerobot | |
| shape_meta: | |
| images: | |
| - key: image | |
| raw_shape: | |
| - 3 | |
| - 512 | |
| - 512 | |
| shape: | |
| - 3 | |
| - 224 | |
| - 224 | |
| - key: wrist_image | |
| raw_shape: | |
| - 3 | |
| - 512 | |
| - 512 | |
| shape: | |
| - 3 | |
| - 224 | |
| - 224 | |
| action: | |
| - key: default | |
| raw_shape: 7 | |
| shape: 7 | |
| state: | |
| - key: default | |
| raw_shape: 8 | |
| shape: 8 | |
| num_frames: 17 | |
| global_sample_stride: 1 | |
| action_video_freq_ratio: 1 | |
| video_size: | |
| - 224 | |
| - 448 | |
| camera_key: null | |
| val_set_proportion: 0.0 | |
| is_training_set: true | |
| skip_padding_as_possible: false | |
| concat_multi_camera: horizontal | |
| video_augmentation: | |
| _target_: fastwam.datasets.lerobot.transforms.image.VideoAugmentation | |
| p: 0.65 | |
| augment_types: | |
| - corrupt_only | |
| - color_only | |
| - both | |
| - both | |
| color_jitter: | |
| brightness: 0.3 | |
| contrast: 0.3 | |
| saturation: 0.25 | |
| hue: 0.04 | |
| gamma: | |
| range: | |
| - 0.8 | |
| - 1.25 | |
| gaussian_noise: | |
| std: 0.015 | |
| random_resized_crop: | |
| scale: | |
| - 0.92 | |
| - 1.0 | |
| ratio: preserve | |
| rotate: | |
| degrees: 8.0 | |
| fill: mean | |
| exposure: | |
| ev_range: | |
| - -0.25 | |
| - 0.25 | |
| require_text_cache: false | |
| endpoint_frames_only: true | |
| qwen_text_cache_dir: /dockerdata/data/libero/flux2_qwen3_cache_4b | |
| qwen_context_len: 512 | |
| pretrained_norm_stats: null | |
| processor: | |
| _target_: fastwam.datasets.lerobot.processors.fastwam_processor.FastWAMProcessor | |
| shape_meta: | |
| images: | |
| - key: image | |
| raw_shape: | |
| - 3 | |
| - 512 | |
| - 512 | |
| shape: | |
| - 3 | |
| - 224 | |
| - 224 | |
| - key: wrist_image | |
| raw_shape: | |
| - 3 | |
| - 512 | |
| - 512 | |
| shape: | |
| - 3 | |
| - 224 | |
| - 224 | |
| action: | |
| - key: default | |
| raw_shape: 7 | |
| shape: 7 | |
| state: | |
| - key: default | |
| raw_shape: 8 | |
| shape: 8 | |
| num_obs_steps: 17 | |
| image_obs_steps: 2 | |
| num_output_cameras: 2 | |
| action_output_dim: 7 | |
| proprio_output_dim: 8 | |
| delta_action_dim_mask: | |
| default: | |
| - true | |
| - true | |
| - true | |
| - true | |
| - true | |
| - true | |
| - false | |
| action_state_transforms: null | |
| use_stepwise_action_norm: false | |
| norm_default_mode: min/max | |
| norm_exception_mode: null | |
| action_state_merger: | |
| _target_: fastwam.datasets.lerobot.transforms.action_state_merger.ConcatLeftAlign | |
| train_transforms: | |
| - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor | |
| - _target_: torchvision.transforms.Resize | |
| size: | |
| - 224 | |
| - 224 | |
| val_transforms: | |
| - _target_: fastwam.datasets.lerobot.transforms.image.ToTensor | |
| - _target_: torchvision.transforms.Resize | |
| size: | |
| - 224 | |
| - 224 | |
| text_embedding_cache_dir: null | |
| context_len: 128 | |
| qwen_text_cache_format: qwen3_flux2 | |
| model: | |
| _target_: fastwam.runtime.create_fastwam_flux2_klein | |
| flux2_src_path: /apdcephfs_nj7/share_305204761/alixzhang/flux2 | |
| flux2_model_path: /dockerdata/models/FLUX.2-klein-base-4B/FLUX.2-klein-base-4B/flux-2-klein-base-4b.safetensors | |
| ae_model_path: /dockerdata/models/FLUX.2-dev/FLUX.2-dev/ae.safetensors | |
| variant: klein-base-4b | |
| qwen3_model_spec: Qwen/Qwen3-4B | |
| load_text_encoder: false | |
| proprio_dim: 8 | |
| mot_checkpoint_mixed_attn: true | |
| mot_gqa_implementation: repeat | |
| mot_force_flash_attention: false | |
| pack_proprio_after_text: true | |
| action_dit_pretrained_path: checkpoints/action_dit_flux2_4b_libero_init.pt | |
| flux2_lora_config: | |
| enabled: false | |
| rank: 16 | |
| alpha: 16.0 | |
| dropout: 0.0 | |
| save_lora_merged: false | |
| save_trainable_only: false | |
| target_suffixes: | |
| - qkv | |
| - proj | |
| - linear1 | |
| - linear2 | |
| - img_mlp.0 | |
| - img_mlp.2 | |
| - txt_mlp.0 | |
| - txt_mlp.2 | |
| action_dit_config: | |
| action_dim: 7 | |
| hidden_dim: 1024 | |
| num_heads: 24 | |
| attn_head_dim: 128 | |
| num_layers_double: 5 | |
| num_layers_single: 20 | |
| mlp_ratio: 4.0 | |
| max_action_horizon: 64 | |
| use_gradient_checkpointing: true | |
| video_scheduler: | |
| train_shift: 5.0 | |
| infer_shift: 5.0 | |
| num_train_timesteps: 1000 | |
| action_scheduler: | |
| train_shift: 5.0 | |
| infer_shift: 5.0 | |
| num_train_timesteps: 1000 | |
| loss: | |
| lambda_video: 0.5 | |
| lambda_action: 1.0 | |