LeRobot
Safetensors
Wan2.2
English
Robotics
LeRobot
Safetensors
vision-language-action
imitation-learning
fastwam
robotwin
Instructions to use ZibinDong/fastwam_robotwin_uncond_3cam_384 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- LeRobot
How to use ZibinDong/fastwam_robotwin_uncond_3cam_384 with LeRobot:
- Wan2.2
How to use ZibinDong/fastwam_robotwin_uncond_3cam_384 with Wan2.2:
# No code snippets available yet for this library. # To use this model, check the repository files and the library's documentation. # Want to help? PRs adding snippets are welcome at: # https://github.com/huggingface/huggingface.js
- Notebooks
- Google Colab
- Kaggle
| { | |
| "n_obs_steps": 1, | |
| "input_features": { | |
| "observation.images.image": { | |
| "type": "VISUAL", | |
| "shape": [ | |
| 3, | |
| 384, | |
| 320 | |
| ] | |
| }, | |
| "observation.state": { | |
| "type": "STATE", | |
| "shape": [ | |
| 14 | |
| ] | |
| } | |
| }, | |
| "output_features": { | |
| "action": { | |
| "type": "ACTION", | |
| "shape": [ | |
| 14 | |
| ] | |
| } | |
| }, | |
| "device": null, | |
| "use_amp": false, | |
| "push_to_hub": false, | |
| "repo_id": null, | |
| "private": null, | |
| "tags": null, | |
| "license": null, | |
| "action_dim": 14, | |
| "proprio_dim": 14, | |
| "action_horizon": 32, | |
| "n_action_steps": 32, | |
| "num_video_frames": 33, | |
| "image_size": [ | |
| 384, | |
| 320 | |
| ], | |
| "context_len": 128, | |
| "model_id": ".", | |
| "tokenizer_model_id": ".", | |
| "tokenizer_max_len": 128, | |
| "load_text_encoder": true, | |
| "mot_checkpoint_mixed_attn": false, | |
| "torch_dtype": "bfloat16", | |
| "video_scheduler": { | |
| "train_shift": 5, | |
| "infer_shift": 5, | |
| "num_train_timesteps": 1000 | |
| }, | |
| "action_scheduler": { | |
| "train_shift": 5, | |
| "infer_shift": 5, | |
| "num_train_timesteps": 1000 | |
| }, | |
| "loss": { | |
| "lambda_video": 1, | |
| "lambda_action": 1 | |
| }, | |
| "video_dit_config": { | |
| "has_image_input": false, | |
| "patch_size": [ | |
| 1, | |
| 2, | |
| 2 | |
| ], | |
| "in_dim": 48, | |
| "hidden_dim": 3072, | |
| "ffn_dim": 14336, | |
| "freq_dim": 256, | |
| "text_dim": 4096, | |
| "out_dim": 48, | |
| "num_heads": 24, | |
| "attn_head_dim": 128, | |
| "num_layers": 30, | |
| "eps": 1e-06, | |
| "seperated_timestep": true, | |
| "require_clip_embedding": false, | |
| "require_vae_embedding": false, | |
| "fuse_vae_embedding_in_latents": true, | |
| "use_gradient_checkpointing": false, | |
| "video_attention_mask_mode": "first_frame_causal", | |
| "action_conditioned": false, | |
| "action_dim": 14, | |
| "action_group_causal_mask_mode": "group_diagonal" | |
| }, | |
| "action_dit_config": { | |
| "action_dim": 14, | |
| "hidden_dim": 1024, | |
| "ffn_dim": 4096, | |
| "num_heads": 24, | |
| "attn_head_dim": 128, | |
| "num_layers": 30, | |
| "text_dim": 4096, | |
| "freq_dim": 256, | |
| "eps": 1e-06, | |
| "use_gradient_checkpointing": false | |
| }, | |
| "normalization_mapping": { | |
| "VISUAL": "MEAN_STD", | |
| "STATE": "MIN_MAX", | |
| "ACTION": "MIN_MAX" | |
| }, | |
| "optimizer_lr": 0.0001, | |
| "optimizer_weight_decay": 0.01, | |
| "type": "fastwam", | |
| "toggle_action_dimensions": [] | |
| } | |