mpark commited on
Commit
a1179c1
·
verified ·
1 Parent(s): a38b99a

Config and normalizer in the format of the current DiscoDemo code

Browse files
Files changed (3) hide show
  1. README.md +1 -1
  2. config.yaml +50 -369
  3. normalizer.pt +2 -2
README.md CHANGED
@@ -24,7 +24,7 @@ Rolling it out generates demonstrations such as those in the Stage2 dataset belo
24
  | File | Content |
25
  |---|---|
26
  | `actor.pt` | Actor network weights (`network_state_dict`). Optimizer state is not included. |
27
- | `normalizer.pt` | Running statistics used to normalize the policy input. |
28
  | `config.yaml` | Full training configuration (environment, curriculum, agent). |
29
 
30
  Only what is needed to roll the policy out is released; the critic and other training-only state are omitted.
 
24
  | File | Content |
25
  |---|---|
26
  | `actor.pt` | Actor network weights (`network_state_dict`). Optimizer state is not included. |
27
+ | `normalizer.pt` | Observation bounds used to normalize the policy input. |
28
  | `config.yaml` | Full training configuration (environment, curriculum, agent). |
29
 
30
  Only what is needed to roll the policy out is released; the critic and other training-only state are omitted.
config.yaml CHANGED
@@ -1,3 +1,22 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  agent:
2
  agent_type: flashSAC
3
  seed: 0
@@ -8,9 +27,7 @@ agent:
8
  sample_batch_size: 2048
9
  normalize_reward: true
10
  normalized_G_max: 5.0
11
- reward_grmax_clamp: 0.0
12
  asymmetric_observation: false
13
- store_ep_uid: false
14
  learning_rate_init: 0.0003
15
  learning_rate_peak: 0.0003
16
  learning_rate_end: 0.00015
@@ -21,14 +38,12 @@ agent:
21
  actor_num_blocks: 2
22
  actor_hidden_dim: 128
23
  actor_bc_alpha: 0.0
24
- actor_action_magnitude_coef: 0.0
25
  actor_noise_zeta_mu: 2.0
26
  actor_noise_zeta_max: 16
27
  actor_update_period: 2
28
  critic_num_blocks: 2
29
  critic_hidden_dim: 256
30
  critic_num_bins: 101
31
- critic_value_range: 5
32
  critic_min_v: -5
33
  critic_max_v: 5
34
  critic_target_update_tau: 0.01
@@ -42,211 +57,32 @@ agent:
42
  use_amp: true
43
  load_optimizer: true
44
  load_reward_normalizer: true
45
- normalize_inputs: true
46
- normalize_clamp: true
47
- normalize_inputs_mode: bounds
48
- normalize_inputs_skip_keys: []
49
- env:
50
- env_type: robolab
51
- env_name: Fr3PegInsertRound
52
- num_env_steps: 50000000
53
- seed: 0
54
- num_train_envs: 2048
55
- num_eval_envs: 0
56
- num_record_envs: 1
57
- rescale_action: true
58
- max_episode_steps: null
59
- robot_variant: fr3
60
- sysid_workcell: libs/real2sim/assets/workcell.json
61
- canonical_env_version: fr3-peg-insert-round-v3
62
- gripper_close_width_m: null
63
- gripper_stiffness: null
64
- gripper_damping: null
65
- gripper_delay: true
66
- gripper_delay_obs: true
67
- gripper_delay_obs_mode: elapsed
68
- gripper_lock: closed
69
- action_repeat: 1
70
- task: libs/real2sim/real2sim/tasks/fr3_peg_insert_round.py
71
- reward_mode: sparse
72
- headless: true
73
- seed: 0
74
- load_replay_buffer_on_resume: true
75
- agent_load_skip_input_layer: false
76
- num_env_steps: 50000000
77
- num_train_envs: 2048
78
- num_eval_envs: 0
79
- num_record_envs: 1
80
- num_eval_episodes: 1
81
- num_record_episodes: 1
82
- gamma: 0.99
83
- n_step: 8
84
- num_interaction_steps: 24414.0625
85
- updates_per_interaction_step: 128
86
- num_update_steps: 3125000.0
87
- evaluation_per_interaction_step: 244
88
- metrics_per_interaction_step: 244
89
- recording_per_interaction_step: 813
90
- logging_per_interaction_step: 24
91
- save_checkpoint_per_interaction_step: 244
92
- save_buffer_per_interaction_step: 244
93
- keep_only_latest_periodic_buffer: true
94
- eval_random_seed: 12345
95
- reconfigure_per_interaction_step: null
96
  rfcl:
97
- num_demos: 20
98
- skip_failed: true
99
- shuffle_demos: false
100
- reward_mode: sparse
101
  demo_ratio: 0.5
102
  truncate_after_success: 50
103
- skip_reverse_curriculum: false
104
  reverse:
105
- curriculum_method: per_demo
106
  reverse_step_size: 16
107
- per_env_buffer_size: 3
108
  advance_threshold: 0.9
109
- start_step_sampler: multienv-zvf
110
- demo_horizon_to_max_steps_ratio: 1
111
- verbose: 1
112
- num_train_envs: 20
113
- updates_per_interaction_step: null
114
- curriculum_keyframes:
115
- enabled: false
116
- path: null
117
- remap_dense_resume: false
118
- multi_env: null
119
  minimum_episode_steps: 30
120
- zvf_hi: 2.0
121
- zvf_lo: 0.0
122
- zvf_window_K: 10
123
- zvf_sample_weight: geometric
124
- zvf_mask: false
125
- forward:
126
- enabled: true
127
- num_seeds: 1000
128
- rho: 1.0
129
- nu: 0.2
130
- score_fn: success_once_score
131
- score_transform: rankmin
132
- score_temperature: 0.1
133
- staleness_transform: rankmin
134
- staleness_temperature: 0.1
135
- staleness_coef: 0.1
136
- transition:
137
- enabled: true
138
- solved_frac_threshold: 0.5
139
- save_checkpoint_and_buffer: true
140
- force_forward_at_is: 0
141
  init_pose: pregrasp:0.03,15
142
- init_pose_source: onthefly
143
- init_pose_root: null
144
- init_pose_start_row: null
145
- demo_path: null
146
- demo_bank: capx20-round-v3
147
- resume_forward: false
148
  eval_worker:
149
- enabled: true
150
- measure_safety: false
151
  num_envs: 50
152
- video_max_steps: 0
153
- video_speed: 2.0
154
- video_scale: 0.5
155
- video_quality: 5
156
- frontier_eval:
157
- enabled: true
158
- num_demos: 8
159
- envs_per_demo: 6
160
- frame_offset: 0
161
- video: true
162
- exclude_terminal_frontiers: false
163
- terminal_slack_steps: -1
164
- eval_success_window_size: 0
165
- max_episode_steps: 0
166
- success_hold_steps: 20
167
- success_reward_mode: oneshot
168
- success_as_truncation: false
169
- timeout_as_termination: true
170
- distractor_obs: true
171
- settle_from_reverse: true
172
- adaptive_horizon:
173
- enabled: false
174
- multiplier: 2.0
175
- min_steps: 50
176
- min_episodes: 10
177
- max_relative_action_rate_k: null
178
- max_relative_action_rate_abs:
179
- - 0.0111075
180
- - 0.0258575
181
- - 0.009622
182
- - 0.033109
183
- - 0.015942
184
- - 0.0208835
185
- - 0.020363
186
  skill:
187
- z_dim: 0
188
- z_unit: false
189
- z_norm_match: false
190
- alpha: 0.0
191
- phi_space: object
192
  phi_hidden:
193
  - 256
194
  - 256
195
  phi_lr: 0.0001
196
  dual_slack: 0.001
197
- alpha_mode: raw
198
- budget_warmup_updates: 500
199
- budget_freeze_after: 0
200
  lam_init: 1.0
201
- gate_capacity: 1048576
202
- skill_type: metra
203
- z_discrete: false
204
- n_skills: 5
205
- disc_hidden:
206
- - 256
207
- - 256
208
- disc_lr: 0.0001
209
- partial_credit:
210
- coef: 0.0
211
- living_cost:
212
- coef: 0.0
213
  safety_penalty:
214
- enabled: true
215
- apply_in_reverse: true
216
- cap: 0.1
217
- qvel:
218
- coef: 0.0
219
- deadband: 0.12
220
- sat: 4.44
221
- qacc:
222
- coef: 0.0
223
- deadband: 14.0
224
- sat: 27183
225
- action_smoothness:
226
- coef: 0.0
227
- deadband: 0.1
228
- sat: 0.265
229
- rel_action:
230
- coef: 0.0
231
- deadband: 0.5
232
- norm: 2
233
- sat: 1.0
234
- gripper_flip:
235
- coef: 0.0
236
- deadband: 0.0
237
- sat: 1.0
238
- eef_lin_vel:
239
- weight: 0.0
240
- deadband: 0.18
241
- sat: 1.5
242
  illegal_contact:
243
  weight: 0.0001
244
  sat: 100
245
- allowed_pairs:
246
- - - panda_leftfinger
247
- - peg
248
- - - panda_rightfinger
249
- - peg
250
  robot_bodies:
251
  - panda_link5
252
  - panda_link6
@@ -258,34 +94,11 @@ safety_penalty:
258
  - peg
259
  - board
260
  - table
261
- obj_lin_vel:
262
- weight: 0.0
263
- deadband: 0.05
264
- sat: 2.04
265
- obj_ang_vel:
266
- weight: 0.0
267
- deadband: 0.5
268
- sat: 28.8
269
- obj_acc:
270
- weight: 0.0
271
- deadband: 0.0
272
- sat: 29.6
273
- obj_impact:
274
- weight: 0.0
275
- deadband: 1.0
276
- sat: 553
277
- other_vel:
278
- weight: 0.0
279
- deadband: 0.05
280
- sat: 2.18
281
- other_disp:
282
- weight: 0.0
283
- deadband: 0.02
284
- sat: 1.98
285
- other_impact:
286
- weight: 0.0
287
- deadband: 1.0
288
- sat: 414
289
  obj_press:
290
  weight: 1.0e-06
291
  deadband_axial: 5.0
@@ -293,12 +106,27 @@ safety_penalty:
293
  sat: 100.0
294
  surfaces:
295
  - board
296
- save_video: true
297
- video_camera: exo_wrist
298
- disable_train_cameras: true
299
- control_decimation: 6
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
300
  jerk_limited_action:
301
- enabled: true
302
  max_joint_velocity:
303
  - 0.22215
304
  - 0.51715
@@ -323,151 +151,4 @@ jerk_limited_action:
323
  - 9.90014
324
  - 6.67391
325
  - 10.55754
326
- reset_reverse_curriculum_on_resume: false
327
- obs_bounds:
328
- object_xy:
329
- - - 0.25
330
- - 0.65
331
- - - -0.25
332
- - 0.15
333
- object_xy_margin_m: 0.1
334
- object_z_m:
335
- - 0.0
336
- - 0.6
337
- eef_box_m:
338
- x:
339
- - 0.2
340
- - 0.85
341
- y:
342
- - -0.45
343
- - 0.45
344
- z:
345
- - 0.0
346
- - 0.7
347
- joints: physical
348
- resolved:
349
- terms:
350
- - - target0_pos
351
- - 3
352
- - - target0_quat
353
- - 4
354
- - - target0_tcp
355
- - 3
356
- - - container_pos
357
- - 3
358
- - - container_quat
359
- - 4
360
- - - tcp_to_container
361
- - 3
362
- - - arm_joint_pos
363
- - 7
364
- - - gripper_pos
365
- - 1
366
- - - eef_position
367
- - 3
368
- - - eef_orientation
369
- - 4
370
- - - gripper_delay
371
- - 2
372
- - - prev_action
373
- - 8
374
- low:
375
- - 0.25
376
- - -0.25
377
- - 0.0
378
- - -1.0
379
- - -1.0
380
- - -1.0
381
- - -1.0
382
- - -0.6000000238418579
383
- - -0.699999988079071
384
- - -0.699999988079071
385
- - 0.25
386
- - -0.25
387
- - 0.0
388
- - -1.0
389
- - -1.0
390
- - -1.0
391
- - -1.0
392
- - -0.6000000238418579
393
- - -0.699999988079071
394
- - -0.699999988079071
395
- - -2.8973002433776855
396
- - -1.7627999782562256
397
- - -2.8973000049591064
398
- - -3.0717999935150146
399
- - -2.8973000049591064
400
- - -0.017499923706054688
401
- - -3.6826982498168945
402
- - 0.0
403
- - 0.20000000298023224
404
- - -0.44999998807907104
405
- - 0.0
406
- - -1.0
407
- - -1.0
408
- - -1.0
409
- - -1.0
410
- - 0.0
411
- - 0.0
412
- - -1.0
413
- - -1.0
414
- - -1.0
415
- - -1.0
416
- - -1.0
417
- - -1.0
418
- - -1.0
419
- - -1.0
420
- high:
421
- - 0.6499999761581421
422
- - 0.15000000596046448
423
- - 0.6000000238418579
424
- - 1.0
425
- - 1.0
426
- - 1.0
427
- - 1.0
428
- - 0.44999998807907104
429
- - 0.6000000238418579
430
- - 0.6000000238418579
431
- - 0.6499999761581421
432
- - 0.15000000596046448
433
- - 0.6000000238418579
434
- - 1.0
435
- - 1.0
436
- - 1.0
437
- - 1.0
438
- - 0.44999998807907104
439
- - 0.6000000238418579
440
- - 0.6000000238418579
441
- - 2.8977999687194824
442
- - 1.7627999782562256
443
- - 2.8973000049591064
444
- - -0.06979990005493164
445
- - 2.8973000049591064
446
- - 3.752500057220459
447
- - 2.1119017601013184
448
- - 1.0
449
- - 0.8500000238418579
450
- - 0.44999998807907104
451
- - 0.699999988079071
452
- - 1.0
453
- - 1.0
454
- - 1.0
455
- - 1.0
456
- - 1.0
457
- - 1.0
458
- - 1.0
459
- - 1.0
460
- - 1.0
461
- - 1.0
462
- - 1.0
463
- - 1.0
464
- - 1.0
465
- - 1.0
466
- max_relative_action_rate:
467
- - 0.0111075
468
- - 0.0258575
469
- - 0.009622
470
- - 0.033109
471
- - 0.015942
472
- - 0.0208835
473
- - 0.020363
 
1
+ project_name: DiscoDemo
2
+ entity_name: null
3
+ group_name: fmb_round
4
+ exp_name: prfcl-fmb_round
5
+ seed: 0
6
+ logger_type: wandb
7
+ save_path: libs/FlashSAC/models/fmb_round/prfcl-fmb_round/seed0-TIMESTAMP
8
+ agent_load_path: null
9
+ num_env_steps: 50000000
10
+ num_train_envs: 2048
11
+ updates_per_interaction_step: 128
12
+ gamma: 0.99
13
+ n_step: 8
14
+ env:
15
+ workcell: libs/real2sim/assets/workcell.json
16
+ task: libs/real2sim/real2sim/tasks/fmb_round.py
17
+ env_version: fr3-fmb_round
18
+ gripper_lock: closed
19
+ success_hold_steps: 20
20
  agent:
21
  agent_type: flashSAC
22
  seed: 0
 
27
  sample_batch_size: 2048
28
  normalize_reward: true
29
  normalized_G_max: 5.0
 
30
  asymmetric_observation: false
 
31
  learning_rate_init: 0.0003
32
  learning_rate_peak: 0.0003
33
  learning_rate_end: 0.00015
 
38
  actor_num_blocks: 2
39
  actor_hidden_dim: 128
40
  actor_bc_alpha: 0.0
 
41
  actor_noise_zeta_mu: 2.0
42
  actor_noise_zeta_max: 16
43
  actor_update_period: 2
44
  critic_num_blocks: 2
45
  critic_hidden_dim: 256
46
  critic_num_bins: 101
 
47
  critic_min_v: -5
48
  critic_max_v: 5
49
  critic_target_update_tau: 0.01
 
57
  use_amp: true
58
  load_optimizer: true
59
  load_reward_normalizer: true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
60
  rfcl:
 
 
 
 
61
  demo_ratio: 0.5
62
  truncate_after_success: 50
 
63
  reverse:
 
64
  reverse_step_size: 16
 
65
  advance_threshold: 0.9
66
+ frontier_window: 10
 
 
 
 
 
 
 
 
 
67
  minimum_episode_steps: 30
68
+ solved_frac_threshold: 0.5
69
+ demo_path: data/demo_banks/fmb_round/human_demos.h5
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
70
  init_pose: pregrasp:0.03,15
 
 
 
 
 
 
71
  eval_worker:
 
 
72
  num_envs: 50
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
73
  skill:
 
 
 
 
 
74
  phi_hidden:
75
  - 256
76
  - 256
77
  phi_lr: 0.0001
78
  dual_slack: 0.001
 
 
 
79
  lam_init: 1.0
80
+ alpha: 0.0
81
+ z_dim: 0
 
 
 
 
 
 
 
 
 
 
82
  safety_penalty:
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
83
  illegal_contact:
84
  weight: 0.0001
85
  sat: 100
 
 
 
 
 
86
  robot_bodies:
87
  - panda_link5
88
  - panda_link6
 
94
  - peg
95
  - board
96
  - table
97
+ allowed_pairs:
98
+ - - panda_leftfinger
99
+ - peg
100
+ - - panda_rightfinger
101
+ - peg
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
102
  obj_press:
103
  weight: 1.0e-06
104
  deadband_axial: 5.0
 
106
  sat: 100.0
107
  surfaces:
108
  - board
109
+ obs_bounds:
110
+ object_z_m:
111
+ - 0.0
112
+ - 0.6
113
+ eef_box_m:
114
+ x:
115
+ - 0.2
116
+ - 0.85
117
+ 'y':
118
+ - -0.45
119
+ - 0.45
120
+ z:
121
+ - 0.0
122
+ - 0.7
123
+ object_xy:
124
+ - - 0.25
125
+ - 0.65
126
+ - - -0.25
127
+ - 0.15
128
+ task_name: fmb_round
129
  jerk_limited_action:
 
130
  max_joint_velocity:
131
  - 0.22215
132
  - 0.51715
 
151
  - 9.90014
152
  - 6.67391
153
  - 10.55754
154
+ method_name: prfcl
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
normalizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:95db1c12b4e976b513cf21aaabb8b35f4aa324391498d562a5271b97ca3a058d
3
- size 3002
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b237a781e41219873d94bb327cb71697cf9f4b462c65ff98b96a4a92b9c2253
3
+ size 2364