hqr-robotic commited on
Commit
5a48a00
·
1 Parent(s): 794a7cd

upload fastwam-idm and fastwam-joint LIBERO fine-tuned checkpoints

Browse files
fastwam_idm_libero_full_finetune_bs16/checkpoints/step-021700-epoch-10-loss=0.0906.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9b5dd21865788790f756f4279052a36fd4cb110643c767ca90c509d8faefd8ac
3
+ size 49625513412
fastwam_idm_libero_full_finetune_bs16/config.json ADDED
@@ -0,0 +1,475 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_action_window_size": 32,
3
+ "_ckpt_root": "./checkpoints",
4
+ "_data_root_paths": [
5
+ "datasets/libero_spatial_no_noops_lerobotv2.1",
6
+ "datasets/libero_object_no_noops_lerobotv2.1",
7
+ "datasets/libero_goal_no_noops_lerobotv2.1",
8
+ "datasets/libero_10_no_noops_lerobotv2.1"
9
+ ],
10
+ "_frame_sample_stride": 4,
11
+ "_frame_window_size": 9,
12
+ "_statistic_name": "libero_all_no_noops",
13
+ "_text_prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}",
14
+ "_tokenizer": "./checkpoints/fastwam_base_full/tokenizer",
15
+ "eval": {
16
+ "manager": {
17
+ "launch_delay": 0.5,
18
+ "master_port_base": 29690,
19
+ "max_tasks_per_gpu": 2,
20
+ "monitor_interval": 5,
21
+ "num_gpus": 8,
22
+ "status_interval": 30
23
+ },
24
+ "runner": {
25
+ "dataset": {
26
+ "img_buffer_len": 1,
27
+ "transforms": [
28
+ {
29
+ "img_keys": [
30
+ "agentview_image",
31
+ "robot0_eye_in_hand_image"
32
+ ],
33
+ "type": "ProcessLiberoEvalInputs"
34
+ },
35
+ {
36
+ "image_resize_strategy": "resize-naive",
37
+ "input_sizes": [
38
+ [
39
+ 3,
40
+ 224,
41
+ 224
42
+ ],
43
+ [
44
+ 3,
45
+ 224,
46
+ 224
47
+ ]
48
+ ],
49
+ "means": [
50
+ [
51
+ 127.5,
52
+ 127.5,
53
+ 127.5
54
+ ],
55
+ [
56
+ 127.5,
57
+ 127.5,
58
+ 127.5
59
+ ]
60
+ ],
61
+ "stds": [
62
+ [
63
+ 127.5,
64
+ 127.5,
65
+ 127.5
66
+ ],
67
+ [
68
+ 127.5,
69
+ 127.5,
70
+ 127.5
71
+ ]
72
+ ],
73
+ "type": "TransformImage"
74
+ },
75
+ {
76
+ "norm_type": "min_max",
77
+ "out_key": "states",
78
+ "stat_key": "proprio",
79
+ "state_dim": 8,
80
+ "type": "LiberoProprioFromInputs"
81
+ },
82
+ {
83
+ "max_len": 128,
84
+ "prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}",
85
+ "tokenizer": {
86
+ "model_path": "./checkpoints/fastwam_base_full/tokenizer",
87
+ "type": "PretrainedTokenizer"
88
+ },
89
+ "type": "LiberoPromptFromInputs",
90
+ "use_conversation": false
91
+ },
92
+ {
93
+ "frame_window_size": 1,
94
+ "num_views": 2,
95
+ "tile_direction": "horizontal",
96
+ "type": "PrepareVideo"
97
+ }
98
+ ],
99
+ "type": "LiberoParquetEvalDataset"
100
+ },
101
+ "denormalize_action": {
102
+ "action_dim": 7,
103
+ "norm_type": "min_max",
104
+ "type": "DenormalizeLiberoAction"
105
+ },
106
+ "enable_mixed_precision_training": true,
107
+ "eval_chunk_size": 10,
108
+ "eval_shard_strategy": "task",
109
+ "inference_seed": 42,
110
+ "max_steps": {
111
+ "libero_10": 700,
112
+ "libero_goal": 400,
113
+ "libero_object": 400,
114
+ "libero_spatial": 400
115
+ },
116
+ "mixed_precision_dtype": "bf16",
117
+ "model_build_device": "cuda",
118
+ "model_build_dtype": "bf16",
119
+ "model_family": "fastwam",
120
+ "norm_stats_key": "libero_all_no_noops",
121
+ "num_inference_steps": 10,
122
+ "num_steps_wait": 30,
123
+ "num_trials_per_task": 50,
124
+ "preprocess_every_step": false,
125
+ "resize_size": 224,
126
+ "save_multi_view_rollout_videos": true,
127
+ "save_rollout_videos": true,
128
+ "seed": 42,
129
+ "task_ids": null,
130
+ "task_suite_name": [
131
+ "libero_spatial",
132
+ "libero_object",
133
+ "libero_goal",
134
+ "libero_10"
135
+ ],
136
+ "type": "LiberoEvalRunner"
137
+ }
138
+ },
139
+ "eval_dataset": null,
140
+ "inference_model": {
141
+ "action_horizon": 32,
142
+ "frame_window_size": 9,
143
+ "mot_checkpoint_mixed_attn": true,
144
+ "num_views": 2,
145
+ "pretrained_name_or_path": "./checkpoints/fastwam_base_full/fastwam_base_full.safetensors",
146
+ "proprio_dim": 8,
147
+ "torch_dtype": "bf16",
148
+ "type": "FastWAMVLA",
149
+ "vla_head": {
150
+ "action_dit_config": {
151
+ "action_dim": 7,
152
+ "attn_head_dim": 128,
153
+ "eps": 1e-06,
154
+ "ffn_dim": 4096,
155
+ "freq_dim": 256,
156
+ "hidden_dim": 1024,
157
+ "num_heads": 24,
158
+ "num_layers": 30,
159
+ "text_dim": 4096,
160
+ "use_gradient_checkpointing": true
161
+ },
162
+ "action_scheduler": {
163
+ "infer_shift": 5.0,
164
+ "num_train_timesteps": 1000,
165
+ "train_shift": 5.0
166
+ },
167
+ "loss": {
168
+ "lambda_action": 1.0,
169
+ "lambda_video": 1.0
170
+ },
171
+ "type": "FastWAMIDMHead",
172
+ "video_dit_config": {
173
+ "action_conditioned": false,
174
+ "action_dim": 7,
175
+ "action_group_causal_mask_mode": "group_diagonal",
176
+ "attn_head_dim": 128,
177
+ "eps": 1e-06,
178
+ "ffn_dim": 14336,
179
+ "freq_dim": 256,
180
+ "fuse_vae_embedding_in_latents": true,
181
+ "has_image_input": false,
182
+ "hidden_dim": 3072,
183
+ "in_dim": 48,
184
+ "num_heads": 24,
185
+ "num_layers": 30,
186
+ "out_dim": 48,
187
+ "patch_size": [
188
+ 1,
189
+ 2,
190
+ 2
191
+ ],
192
+ "require_clip_embedding": false,
193
+ "require_vae_embedding": false,
194
+ "seperated_timestep": true,
195
+ "text_dim": 4096,
196
+ "use_gradient_checkpointing": true,
197
+ "video_attention_mask_mode": "first_frame_causal"
198
+ },
199
+ "video_scheduler": {
200
+ "infer_shift": 5.0,
201
+ "num_train_timesteps": 1000,
202
+ "train_shift": 5.0
203
+ }
204
+ },
205
+ "vlm_backbone": {
206
+ "text_embed_cache_context_len": 128,
207
+ "text_embed_cache_device": "cpu",
208
+ "text_embed_cache_size": 256,
209
+ "type": "Wan22Backbone"
210
+ }
211
+ },
212
+ "model": {
213
+ "action_horizon": 32,
214
+ "frame_window_size": 9,
215
+ "mot_checkpoint_mixed_attn": true,
216
+ "num_views": 2,
217
+ "pretrained_name_or_path": "./checkpoints/fastwam_base_full/fastwam_base_full.safetensors",
218
+ "proprio_dim": 8,
219
+ "torch_dtype": "bf16",
220
+ "type": "FastWAMVLA",
221
+ "vla_head": {
222
+ "action_dit_config": {
223
+ "action_dim": 7,
224
+ "attn_head_dim": 128,
225
+ "eps": 1e-06,
226
+ "ffn_dim": 4096,
227
+ "freq_dim": 256,
228
+ "hidden_dim": 1024,
229
+ "num_heads": 24,
230
+ "num_layers": 30,
231
+ "text_dim": 4096,
232
+ "use_gradient_checkpointing": true
233
+ },
234
+ "action_scheduler": {
235
+ "infer_shift": 5.0,
236
+ "num_train_timesteps": 1000,
237
+ "train_shift": 5.0
238
+ },
239
+ "loss": {
240
+ "lambda_action": 1.0,
241
+ "lambda_video": 1.0
242
+ },
243
+ "type": "FastWAMIDMHead",
244
+ "video_dit_config": {
245
+ "action_conditioned": false,
246
+ "action_dim": 7,
247
+ "action_group_causal_mask_mode": "group_diagonal",
248
+ "attn_head_dim": 128,
249
+ "eps": 1e-06,
250
+ "ffn_dim": 14336,
251
+ "freq_dim": 256,
252
+ "fuse_vae_embedding_in_latents": true,
253
+ "has_image_input": false,
254
+ "hidden_dim": 3072,
255
+ "in_dim": 48,
256
+ "num_heads": 24,
257
+ "num_layers": 30,
258
+ "out_dim": 48,
259
+ "patch_size": [
260
+ 1,
261
+ 2,
262
+ 2
263
+ ],
264
+ "require_clip_embedding": false,
265
+ "require_vae_embedding": false,
266
+ "seperated_timestep": true,
267
+ "text_dim": 4096,
268
+ "use_gradient_checkpointing": true,
269
+ "video_attention_mask_mode": "first_frame_causal"
270
+ },
271
+ "video_scheduler": {
272
+ "infer_shift": 5.0,
273
+ "num_train_timesteps": 1000,
274
+ "train_shift": 5.0
275
+ }
276
+ },
277
+ "vlm_backbone": {
278
+ "text_embed_cache_context_len": 128,
279
+ "text_embed_cache_device": "cpu",
280
+ "text_embed_cache_size": 256,
281
+ "type": "Wan22Backbone"
282
+ }
283
+ },
284
+ "runner": {
285
+ "collator": {
286
+ "keys": [
287
+ "states",
288
+ "images",
289
+ "img_masks",
290
+ "actions",
291
+ "action_masks",
292
+ "embodiment_ids",
293
+ "frame_masks",
294
+ "lang_tokens",
295
+ "lang_masks"
296
+ ],
297
+ "meta_keys": [
298
+ "task_description",
299
+ "info",
300
+ "stats",
301
+ "timestamp"
302
+ ],
303
+ "type": "DictCollator"
304
+ },
305
+ "enable_gradient_checkpointing": false,
306
+ "enable_mixed_precision_training": true,
307
+ "evaluator": {
308
+ "eval_every": 1000,
309
+ "num_inference_steps": 10,
310
+ "save_video": true,
311
+ "seed": 42,
312
+ "type": "training-eval",
313
+ "video_fps": 8
314
+ },
315
+ "grad_accumulation_steps": 1,
316
+ "lr_scheduler": {
317
+ "betas": [
318
+ 0.9,
319
+ 0.95
320
+ ],
321
+ "min_lr_ratio": 0.01,
322
+ "type": "linear-warmup+cosine-decay-min-lr",
323
+ "warmup_ratio": 0.05,
324
+ "weight_decay_style": "uniform"
325
+ },
326
+ "max_epochs": 10,
327
+ "max_grad_norm": 1.0,
328
+ "max_keep_ckpts": 10,
329
+ "max_steps": null,
330
+ "metric": {
331
+ "active_trackers": [
332
+ "jsonl",
333
+ "wandb"
334
+ ],
335
+ "run_dir": "work_dirs",
336
+ "type": "VLAMetric",
337
+ "window_size": 1
338
+ },
339
+ "mixed_precision_dtype": "bf16",
340
+ "optimizer": {
341
+ "lr": 0.0001,
342
+ "type": "AdamW",
343
+ "weight_decay": 0.01
344
+ },
345
+ "reduce_in_full_precision": true,
346
+ "sampler": null,
347
+ "save_epoch_interval": 1,
348
+ "save_iter_interval": 10000,
349
+ "sharding_strategy": "shard-grad-op",
350
+ "type": "FSDPTrainRunner"
351
+ },
352
+ "seed": 42,
353
+ "train_dataloader": {
354
+ "dataset": {
355
+ "batch_shard_size": 8,
356
+ "datasets": {
357
+ "action_key": "action",
358
+ "action_window_size": 32,
359
+ "data_root_path": [
360
+ "datasets/libero_spatial_no_noops_lerobotv2.1",
361
+ "datasets/libero_object_no_noops_lerobotv2.1",
362
+ "datasets/libero_goal_no_noops_lerobotv2.1",
363
+ "datasets/libero_10_no_noops_lerobotv2.1"
364
+ ],
365
+ "frame_sample_stride": 4,
366
+ "frame_window_size": 9,
367
+ "statistic_name": "libero_all_no_noops",
368
+ "transforms": [
369
+ {
370
+ "embodiment_id": 0,
371
+ "name_mappings": {
372
+ "actions": [
373
+ "actions"
374
+ ],
375
+ "observation.state": [
376
+ "states"
377
+ ]
378
+ },
379
+ "parquet_keys": [
380
+ "observation.state",
381
+ "timestamp",
382
+ "actions",
383
+ "info",
384
+ "stats",
385
+ "action_masks"
386
+ ],
387
+ "type": "ProcessParquetInputs",
388
+ "video_backend": "torchcodec",
389
+ "video_keys": [
390
+ "observation.images.image",
391
+ "observation.images.wrist_image"
392
+ ]
393
+ },
394
+ {
395
+ "backend": "torchvision",
396
+ "height": 224,
397
+ "scale_to_unit_interval": true,
398
+ "type": "ResizeImages",
399
+ "width": 224
400
+ },
401
+ {
402
+ "means": [
403
+ 0.5,
404
+ 0.5,
405
+ 0.5
406
+ ],
407
+ "stds": [
408
+ 0.5,
409
+ 0.5,
410
+ 0.5
411
+ ],
412
+ "type": "NormalizeImages"
413
+ },
414
+ {
415
+ "action_dim": 7,
416
+ "action_key": "action",
417
+ "delta_action_dim_mask": [
418
+ true,
419
+ true,
420
+ true,
421
+ true,
422
+ true,
423
+ true,
424
+ false
425
+ ],
426
+ "norm_type": "min_max",
427
+ "pad_invalid_action_delta_dims": true,
428
+ "state_dim": 8,
429
+ "state_key": "proprio",
430
+ "type": "NormalizeStatesAndActions"
431
+ },
432
+ {
433
+ "frame_window_size": 9,
434
+ "num_views": 2,
435
+ "tile_direction": "horizontal",
436
+ "type": "PrepareVideo"
437
+ },
438
+ {
439
+ "max_len": 128,
440
+ "prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}",
441
+ "tokenizer": {
442
+ "model_path": "./checkpoints/fastwam_base_full/tokenizer",
443
+ "type": "PretrainedTokenizer"
444
+ },
445
+ "type": "LiberoPromptFromInputs",
446
+ "use_conversation": false
447
+ }
448
+ ],
449
+ "type": "ParquetDataset",
450
+ "use_delta": false,
451
+ "window_start_idx": 0
452
+ },
453
+ "name_mappings": {
454
+ "action": [
455
+ "action"
456
+ ],
457
+ "observation.state": [
458
+ "proprio"
459
+ ]
460
+ },
461
+ "reshuffle_each_epoch": true,
462
+ "seed": 42,
463
+ "statistic_keys": [
464
+ "observation.state",
465
+ "timestamp",
466
+ "action"
467
+ ],
468
+ "statistic_name": "libero_all_no_noops",
469
+ "type": "DistributedRepeatingDataset"
470
+ },
471
+ "per_device_batch_size": 8,
472
+ "per_device_num_workers": 8
473
+ },
474
+ "val_dataloader": null
475
+ }
fastwam_idm_libero_full_finetune_bs16/config.yaml ADDED
@@ -0,0 +1,382 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _action_window_size: 32
2
+ _ckpt_root: ./checkpoints
3
+ _data_root_paths:
4
+ - datasets/libero_spatial_no_noops_lerobotv2.1
5
+ - datasets/libero_object_no_noops_lerobotv2.1
6
+ - datasets/libero_goal_no_noops_lerobotv2.1
7
+ - datasets/libero_10_no_noops_lerobotv2.1
8
+ _frame_sample_stride: 4
9
+ _frame_window_size: 9
10
+ _statistic_name: libero_all_no_noops
11
+ _text_prompt_template: 'A video recorded from a robot''s point of view executing the
12
+ following instruction: {task}'
13
+ _tokenizer: ./checkpoints/fastwam_base_full/tokenizer
14
+ eval:
15
+ manager:
16
+ launch_delay: 0.5
17
+ master_port_base: 29690
18
+ max_tasks_per_gpu: 2
19
+ monitor_interval: 5
20
+ num_gpus: 8
21
+ status_interval: 30
22
+ runner:
23
+ dataset:
24
+ img_buffer_len: 1
25
+ transforms:
26
+ - img_keys:
27
+ - agentview_image
28
+ - robot0_eye_in_hand_image
29
+ type: ProcessLiberoEvalInputs
30
+ - image_resize_strategy: resize-naive
31
+ input_sizes:
32
+ - - 3
33
+ - 224
34
+ - 224
35
+ - - 3
36
+ - 224
37
+ - 224
38
+ means:
39
+ - - 127.5
40
+ - 127.5
41
+ - 127.5
42
+ - - 127.5
43
+ - 127.5
44
+ - 127.5
45
+ stds:
46
+ - - 127.5
47
+ - 127.5
48
+ - 127.5
49
+ - - 127.5
50
+ - 127.5
51
+ - 127.5
52
+ type: TransformImage
53
+ - norm_type: min_max
54
+ out_key: states
55
+ stat_key: proprio
56
+ state_dim: 8
57
+ type: LiberoProprioFromInputs
58
+ - max_len: 128
59
+ prompt_template: 'A video recorded from a robot''s point of view executing
60
+ the following instruction: {task}'
61
+ tokenizer:
62
+ model_path: ./checkpoints/fastwam_base_full/tokenizer
63
+ type: PretrainedTokenizer
64
+ type: LiberoPromptFromInputs
65
+ use_conversation: false
66
+ - frame_window_size: 1
67
+ num_views: 2
68
+ tile_direction: horizontal
69
+ type: PrepareVideo
70
+ type: LiberoParquetEvalDataset
71
+ denormalize_action:
72
+ action_dim: 7
73
+ norm_type: min_max
74
+ type: DenormalizeLiberoAction
75
+ enable_mixed_precision_training: true
76
+ eval_chunk_size: 10
77
+ eval_shard_strategy: task
78
+ inference_seed: 42
79
+ max_steps:
80
+ libero_10: 700
81
+ libero_goal: 400
82
+ libero_object: 400
83
+ libero_spatial: 400
84
+ mixed_precision_dtype: bf16
85
+ model_build_device: cuda
86
+ model_build_dtype: bf16
87
+ model_family: fastwam
88
+ norm_stats_key: libero_all_no_noops
89
+ num_inference_steps: 10
90
+ num_steps_wait: 30
91
+ num_trials_per_task: 50
92
+ preprocess_every_step: false
93
+ resize_size: 224
94
+ save_multi_view_rollout_videos: true
95
+ save_rollout_videos: true
96
+ seed: 42
97
+ task_ids: null
98
+ task_suite_name:
99
+ - libero_spatial
100
+ - libero_object
101
+ - libero_goal
102
+ - libero_10
103
+ type: LiberoEvalRunner
104
+ eval_dataset: null
105
+ inference_model:
106
+ action_horizon: 32
107
+ frame_window_size: 9
108
+ mot_checkpoint_mixed_attn: true
109
+ num_views: 2
110
+ pretrained_name_or_path: ./checkpoints/fastwam_base_full/fastwam_base_full.safetensors
111
+ proprio_dim: 8
112
+ torch_dtype: bf16
113
+ type: FastWAMVLA
114
+ vla_head:
115
+ action_dit_config:
116
+ action_dim: 7
117
+ attn_head_dim: 128
118
+ eps: 1.0e-06
119
+ ffn_dim: 4096
120
+ freq_dim: 256
121
+ hidden_dim: 1024
122
+ num_heads: 24
123
+ num_layers: 30
124
+ text_dim: 4096
125
+ use_gradient_checkpointing: true
126
+ action_scheduler:
127
+ infer_shift: 5.0
128
+ num_train_timesteps: 1000
129
+ train_shift: 5.0
130
+ loss:
131
+ lambda_action: 1.0
132
+ lambda_video: 1.0
133
+ type: FastWAMIDMHead
134
+ video_dit_config:
135
+ action_conditioned: false
136
+ action_dim: 7
137
+ action_group_causal_mask_mode: group_diagonal
138
+ attn_head_dim: 128
139
+ eps: 1.0e-06
140
+ ffn_dim: 14336
141
+ freq_dim: 256
142
+ fuse_vae_embedding_in_latents: true
143
+ has_image_input: false
144
+ hidden_dim: 3072
145
+ in_dim: 48
146
+ num_heads: 24
147
+ num_layers: 30
148
+ out_dim: 48
149
+ patch_size:
150
+ - 1
151
+ - 2
152
+ - 2
153
+ require_clip_embedding: false
154
+ require_vae_embedding: false
155
+ seperated_timestep: true
156
+ text_dim: 4096
157
+ use_gradient_checkpointing: true
158
+ video_attention_mask_mode: first_frame_causal
159
+ video_scheduler:
160
+ infer_shift: 5.0
161
+ num_train_timesteps: 1000
162
+ train_shift: 5.0
163
+ vlm_backbone:
164
+ text_embed_cache_context_len: 128
165
+ text_embed_cache_device: cpu
166
+ text_embed_cache_size: 256
167
+ type: Wan22Backbone
168
+ model:
169
+ action_horizon: 32
170
+ frame_window_size: 9
171
+ mot_checkpoint_mixed_attn: true
172
+ num_views: 2
173
+ pretrained_name_or_path: ./checkpoints/fastwam_base_full/fastwam_base_full.safetensors
174
+ proprio_dim: 8
175
+ torch_dtype: bf16
176
+ type: FastWAMVLA
177
+ vla_head:
178
+ action_dit_config:
179
+ action_dim: 7
180
+ attn_head_dim: 128
181
+ eps: 1.0e-06
182
+ ffn_dim: 4096
183
+ freq_dim: 256
184
+ hidden_dim: 1024
185
+ num_heads: 24
186
+ num_layers: 30
187
+ text_dim: 4096
188
+ use_gradient_checkpointing: true
189
+ action_scheduler:
190
+ infer_shift: 5.0
191
+ num_train_timesteps: 1000
192
+ train_shift: 5.0
193
+ loss:
194
+ lambda_action: 1.0
195
+ lambda_video: 1.0
196
+ type: FastWAMIDMHead
197
+ video_dit_config:
198
+ action_conditioned: false
199
+ action_dim: 7
200
+ action_group_causal_mask_mode: group_diagonal
201
+ attn_head_dim: 128
202
+ eps: 1.0e-06
203
+ ffn_dim: 14336
204
+ freq_dim: 256
205
+ fuse_vae_embedding_in_latents: true
206
+ has_image_input: false
207
+ hidden_dim: 3072
208
+ in_dim: 48
209
+ num_heads: 24
210
+ num_layers: 30
211
+ out_dim: 48
212
+ patch_size:
213
+ - 1
214
+ - 2
215
+ - 2
216
+ require_clip_embedding: false
217
+ require_vae_embedding: false
218
+ seperated_timestep: true
219
+ text_dim: 4096
220
+ use_gradient_checkpointing: true
221
+ video_attention_mask_mode: first_frame_causal
222
+ video_scheduler:
223
+ infer_shift: 5.0
224
+ num_train_timesteps: 1000
225
+ train_shift: 5.0
226
+ vlm_backbone:
227
+ text_embed_cache_context_len: 128
228
+ text_embed_cache_device: cpu
229
+ text_embed_cache_size: 256
230
+ type: Wan22Backbone
231
+ runner:
232
+ collator:
233
+ keys:
234
+ - states
235
+ - images
236
+ - img_masks
237
+ - actions
238
+ - action_masks
239
+ - embodiment_ids
240
+ - frame_masks
241
+ - lang_tokens
242
+ - lang_masks
243
+ meta_keys:
244
+ - task_description
245
+ - info
246
+ - stats
247
+ - timestamp
248
+ type: DictCollator
249
+ enable_gradient_checkpointing: false
250
+ enable_mixed_precision_training: true
251
+ evaluator:
252
+ eval_every: 1000
253
+ num_inference_steps: 10
254
+ save_video: true
255
+ seed: 42
256
+ type: training-eval
257
+ video_fps: 8
258
+ grad_accumulation_steps: 1
259
+ lr_scheduler:
260
+ betas:
261
+ - 0.9
262
+ - 0.95
263
+ min_lr_ratio: 0.01
264
+ type: linear-warmup+cosine-decay-min-lr
265
+ warmup_ratio: 0.05
266
+ weight_decay_style: uniform
267
+ max_epochs: 10
268
+ max_grad_norm: 1.0
269
+ max_keep_ckpts: 10
270
+ max_steps: null
271
+ metric:
272
+ active_trackers:
273
+ - jsonl
274
+ - wandb
275
+ run_dir: work_dirs
276
+ type: VLAMetric
277
+ window_size: 1
278
+ mixed_precision_dtype: bf16
279
+ optimizer:
280
+ lr: 0.0001
281
+ type: AdamW
282
+ weight_decay: 0.01
283
+ reduce_in_full_precision: true
284
+ sampler: null
285
+ save_epoch_interval: 1
286
+ save_iter_interval: 10000
287
+ sharding_strategy: shard-grad-op
288
+ type: FSDPTrainRunner
289
+ seed: 42
290
+ train_dataloader:
291
+ dataset:
292
+ batch_shard_size: 8
293
+ datasets:
294
+ action_key: action
295
+ action_window_size: 32
296
+ data_root_path:
297
+ - datasets/libero_spatial_no_noops_lerobotv2.1
298
+ - datasets/libero_object_no_noops_lerobotv2.1
299
+ - datasets/libero_goal_no_noops_lerobotv2.1
300
+ - datasets/libero_10_no_noops_lerobotv2.1
301
+ frame_sample_stride: 4
302
+ frame_window_size: 9
303
+ statistic_name: libero_all_no_noops
304
+ transforms:
305
+ - embodiment_id: 0
306
+ name_mappings:
307
+ actions:
308
+ - actions
309
+ observation.state:
310
+ - states
311
+ parquet_keys:
312
+ - observation.state
313
+ - timestamp
314
+ - actions
315
+ - info
316
+ - stats
317
+ - action_masks
318
+ type: ProcessParquetInputs
319
+ video_backend: torchcodec
320
+ video_keys:
321
+ - observation.images.image
322
+ - observation.images.wrist_image
323
+ - backend: torchvision
324
+ height: 224
325
+ scale_to_unit_interval: true
326
+ type: ResizeImages
327
+ width: 224
328
+ - means:
329
+ - 0.5
330
+ - 0.5
331
+ - 0.5
332
+ stds:
333
+ - 0.5
334
+ - 0.5
335
+ - 0.5
336
+ type: NormalizeImages
337
+ - action_dim: 7
338
+ action_key: action
339
+ delta_action_dim_mask:
340
+ - true
341
+ - true
342
+ - true
343
+ - true
344
+ - true
345
+ - true
346
+ - false
347
+ norm_type: min_max
348
+ pad_invalid_action_delta_dims: true
349
+ state_dim: 8
350
+ state_key: proprio
351
+ type: NormalizeStatesAndActions
352
+ - frame_window_size: 9
353
+ num_views: 2
354
+ tile_direction: horizontal
355
+ type: PrepareVideo
356
+ - max_len: 128
357
+ prompt_template: 'A video recorded from a robot''s point of view executing
358
+ the following instruction: {task}'
359
+ tokenizer:
360
+ model_path: ./checkpoints/fastwam_base_full/tokenizer
361
+ type: PretrainedTokenizer
362
+ type: LiberoPromptFromInputs
363
+ use_conversation: false
364
+ type: ParquetDataset
365
+ use_delta: false
366
+ window_start_idx: 0
367
+ name_mappings:
368
+ action:
369
+ - action
370
+ observation.state:
371
+ - proprio
372
+ reshuffle_each_epoch: true
373
+ seed: 42
374
+ statistic_keys:
375
+ - observation.state
376
+ - timestamp
377
+ - action
378
+ statistic_name: libero_all_no_noops
379
+ type: DistributedRepeatingDataset
380
+ per_device_batch_size: 8
381
+ per_device_num_workers: 8
382
+ val_dataloader: null
fastwam_idm_libero_full_finetune_bs16/dataset_statistics.json ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "libero_all_no_noops": {
3
+ "proprio": {
4
+ "mean": [
5
+ -0.04644819089290944,
6
+ 0.034403383486704416,
7
+ 0.7655374256288966,
8
+ 2.9686952302705496,
9
+ -0.21987845450605217,
10
+ -0.12717501774779605,
11
+ 0.027024820750829286,
12
+ -0.02728454226240833
13
+ ],
14
+ "std": [
15
+ 0.1047811675710572,
16
+ 0.15155194421998813,
17
+ 0.37755264388745513,
18
+ 0.34584562447365863,
19
+ 0.9210392242484864,
20
+ 0.3221193452952537,
21
+ 0.014130245844896644,
22
+ 0.014014258634582654
23
+ ],
24
+ "min": [
25
+ -0.48278069496154785,
26
+ -0.3309336006641388,
27
+ 0.008128181099891663,
28
+ 1.002794623374939,
29
+ -3.6312508583068848,
30
+ -1.842738389968872,
31
+ -0.005453015677630901,
32
+ -0.042015016078948975
33
+ ],
34
+ "max": [
35
+ 0.2103137969970703,
36
+ 0.3904264271259308,
37
+ 1.472778081893921,
38
+ 3.7248642444610596,
39
+ 3.5618896484375,
40
+ 1.3863215446472168,
41
+ 0.04232141748070717,
42
+ 0.0013126095291227102
43
+ ],
44
+ "q01": null,
45
+ "q99": null
46
+ },
47
+ "timestamp": {
48
+ "mean": [
49
+ 4.769871954139705
50
+ ],
51
+ "std": [
52
+ 3.675230472000918
53
+ ],
54
+ "min": [
55
+ 0.0
56
+ ],
57
+ "max": [
58
+ 25.2
59
+ ],
60
+ "q01": null,
61
+ "q99": null
62
+ },
63
+ "action": {
64
+ "mean": [
65
+ 0.0618831284549259,
66
+ 0.08701013409377131,
67
+ -0.09098753582797833,
68
+ 0.0005711871427809853,
69
+ 0.005523421982727656,
70
+ -0.005017155657792082,
71
+ 0.5262555238677551
72
+ ],
73
+ "std": [
74
+ 0.33503345581181854,
75
+ 0.3785447232455365,
76
+ 0.44376765641279264,
77
+ 0.03920756640316542,
78
+ 0.06304128440900067,
79
+ 0.07838785125536755,
80
+ 0.4993101721429526
81
+ ],
82
+ "min": [
83
+ -0.9375,
84
+ -0.9375,
85
+ -0.9375,
86
+ -0.24214285612106323,
87
+ -0.375,
88
+ -0.3642857074737549,
89
+ 0.0
90
+ ],
91
+ "max": [
92
+ 0.9375,
93
+ 0.9375,
94
+ 0.9375,
95
+ 0.3557142913341522,
96
+ 0.375,
97
+ 0.375,
98
+ 1.0
99
+ ],
100
+ "q01": null,
101
+ "q99": null
102
+ }
103
+ }
104
+ }
fastwam_idm_libero_full_finetune_bs16/fastwam_idm_libero_full_finetune_2026_08_17_21_48_53.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
fastwam_joint_libero_full_finetune_bs16/checkpoints/step-021700-epoch-10-loss=0.0980.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4898467d9d35440764760f4646acf16a546159b02208a310aab8e22862e97f78
3
+ size 49625513412
fastwam_joint_libero_full_finetune_bs16/config.json ADDED
@@ -0,0 +1,475 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_action_window_size": 32,
3
+ "_ckpt_root": "./checkpoints",
4
+ "_data_root_paths": [
5
+ "datasets/libero_spatial_no_noops_lerobotv2.1",
6
+ "datasets/libero_object_no_noops_lerobotv2.1",
7
+ "datasets/libero_goal_no_noops_lerobotv2.1",
8
+ "datasets/libero_10_no_noops_lerobotv2.1"
9
+ ],
10
+ "_frame_sample_stride": 4,
11
+ "_frame_window_size": 9,
12
+ "_statistic_name": "libero_all_no_noops",
13
+ "_text_prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}",
14
+ "_tokenizer": "./checkpoints/fastwam_base_full/tokenizer",
15
+ "eval": {
16
+ "manager": {
17
+ "launch_delay": 0.5,
18
+ "master_port_base": 29690,
19
+ "max_tasks_per_gpu": 2,
20
+ "monitor_interval": 5,
21
+ "num_gpus": 8,
22
+ "status_interval": 30
23
+ },
24
+ "runner": {
25
+ "dataset": {
26
+ "img_buffer_len": 1,
27
+ "transforms": [
28
+ {
29
+ "img_keys": [
30
+ "agentview_image",
31
+ "robot0_eye_in_hand_image"
32
+ ],
33
+ "type": "ProcessLiberoEvalInputs"
34
+ },
35
+ {
36
+ "image_resize_strategy": "resize-naive",
37
+ "input_sizes": [
38
+ [
39
+ 3,
40
+ 224,
41
+ 224
42
+ ],
43
+ [
44
+ 3,
45
+ 224,
46
+ 224
47
+ ]
48
+ ],
49
+ "means": [
50
+ [
51
+ 127.5,
52
+ 127.5,
53
+ 127.5
54
+ ],
55
+ [
56
+ 127.5,
57
+ 127.5,
58
+ 127.5
59
+ ]
60
+ ],
61
+ "stds": [
62
+ [
63
+ 127.5,
64
+ 127.5,
65
+ 127.5
66
+ ],
67
+ [
68
+ 127.5,
69
+ 127.5,
70
+ 127.5
71
+ ]
72
+ ],
73
+ "type": "TransformImage"
74
+ },
75
+ {
76
+ "norm_type": "min_max",
77
+ "out_key": "states",
78
+ "stat_key": "proprio",
79
+ "state_dim": 8,
80
+ "type": "LiberoProprioFromInputs"
81
+ },
82
+ {
83
+ "max_len": 128,
84
+ "prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}",
85
+ "tokenizer": {
86
+ "model_path": "./checkpoints/fastwam_base_full/tokenizer",
87
+ "type": "PretrainedTokenizer"
88
+ },
89
+ "type": "LiberoPromptFromInputs",
90
+ "use_conversation": false
91
+ },
92
+ {
93
+ "frame_window_size": 1,
94
+ "num_views": 2,
95
+ "tile_direction": "horizontal",
96
+ "type": "PrepareVideo"
97
+ }
98
+ ],
99
+ "type": "LiberoParquetEvalDataset"
100
+ },
101
+ "denormalize_action": {
102
+ "action_dim": 7,
103
+ "norm_type": "min_max",
104
+ "type": "DenormalizeLiberoAction"
105
+ },
106
+ "enable_mixed_precision_training": true,
107
+ "eval_chunk_size": 10,
108
+ "eval_shard_strategy": "task",
109
+ "inference_seed": 42,
110
+ "max_steps": {
111
+ "libero_10": 700,
112
+ "libero_goal": 400,
113
+ "libero_object": 400,
114
+ "libero_spatial": 400
115
+ },
116
+ "mixed_precision_dtype": "bf16",
117
+ "model_build_device": "cuda",
118
+ "model_build_dtype": "bf16",
119
+ "model_family": "fastwam",
120
+ "norm_stats_key": "libero_all_no_noops",
121
+ "num_inference_steps": 10,
122
+ "num_steps_wait": 30,
123
+ "num_trials_per_task": 50,
124
+ "preprocess_every_step": false,
125
+ "resize_size": 224,
126
+ "save_multi_view_rollout_videos": true,
127
+ "save_rollout_videos": true,
128
+ "seed": 42,
129
+ "task_ids": null,
130
+ "task_suite_name": [
131
+ "libero_spatial",
132
+ "libero_object",
133
+ "libero_goal",
134
+ "libero_10"
135
+ ],
136
+ "type": "LiberoEvalRunner"
137
+ }
138
+ },
139
+ "eval_dataset": null,
140
+ "inference_model": {
141
+ "action_horizon": 32,
142
+ "frame_window_size": 9,
143
+ "mot_checkpoint_mixed_attn": true,
144
+ "num_views": 2,
145
+ "pretrained_name_or_path": "./checkpoints/fastwam_base_full/fastwam_base_full.safetensors",
146
+ "proprio_dim": 8,
147
+ "torch_dtype": "bf16",
148
+ "type": "FastWAMVLA",
149
+ "vla_head": {
150
+ "action_dit_config": {
151
+ "action_dim": 7,
152
+ "attn_head_dim": 128,
153
+ "eps": 1e-06,
154
+ "ffn_dim": 4096,
155
+ "freq_dim": 256,
156
+ "hidden_dim": 1024,
157
+ "num_heads": 24,
158
+ "num_layers": 30,
159
+ "text_dim": 4096,
160
+ "use_gradient_checkpointing": true
161
+ },
162
+ "action_scheduler": {
163
+ "infer_shift": 5.0,
164
+ "num_train_timesteps": 1000,
165
+ "train_shift": 5.0
166
+ },
167
+ "loss": {
168
+ "lambda_action": 1.0,
169
+ "lambda_video": 1.0
170
+ },
171
+ "type": "FastWAMJointHead",
172
+ "video_dit_config": {
173
+ "action_conditioned": false,
174
+ "action_dim": 7,
175
+ "action_group_causal_mask_mode": "group_diagonal",
176
+ "attn_head_dim": 128,
177
+ "eps": 1e-06,
178
+ "ffn_dim": 14336,
179
+ "freq_dim": 256,
180
+ "fuse_vae_embedding_in_latents": true,
181
+ "has_image_input": false,
182
+ "hidden_dim": 3072,
183
+ "in_dim": 48,
184
+ "num_heads": 24,
185
+ "num_layers": 30,
186
+ "out_dim": 48,
187
+ "patch_size": [
188
+ 1,
189
+ 2,
190
+ 2
191
+ ],
192
+ "require_clip_embedding": false,
193
+ "require_vae_embedding": false,
194
+ "seperated_timestep": true,
195
+ "text_dim": 4096,
196
+ "use_gradient_checkpointing": true,
197
+ "video_attention_mask_mode": "first_frame_causal"
198
+ },
199
+ "video_scheduler": {
200
+ "infer_shift": 5.0,
201
+ "num_train_timesteps": 1000,
202
+ "train_shift": 5.0
203
+ }
204
+ },
205
+ "vlm_backbone": {
206
+ "text_embed_cache_context_len": 128,
207
+ "text_embed_cache_device": "cpu",
208
+ "text_embed_cache_size": 256,
209
+ "type": "Wan22Backbone"
210
+ }
211
+ },
212
+ "model": {
213
+ "action_horizon": 32,
214
+ "frame_window_size": 9,
215
+ "mot_checkpoint_mixed_attn": true,
216
+ "num_views": 2,
217
+ "pretrained_name_or_path": "./checkpoints/fastwam_base_full/fastwam_base_full.safetensors",
218
+ "proprio_dim": 8,
219
+ "torch_dtype": "bf16",
220
+ "type": "FastWAMVLA",
221
+ "vla_head": {
222
+ "action_dit_config": {
223
+ "action_dim": 7,
224
+ "attn_head_dim": 128,
225
+ "eps": 1e-06,
226
+ "ffn_dim": 4096,
227
+ "freq_dim": 256,
228
+ "hidden_dim": 1024,
229
+ "num_heads": 24,
230
+ "num_layers": 30,
231
+ "text_dim": 4096,
232
+ "use_gradient_checkpointing": true
233
+ },
234
+ "action_scheduler": {
235
+ "infer_shift": 5.0,
236
+ "num_train_timesteps": 1000,
237
+ "train_shift": 5.0
238
+ },
239
+ "loss": {
240
+ "lambda_action": 1.0,
241
+ "lambda_video": 1.0
242
+ },
243
+ "type": "FastWAMJointHead",
244
+ "video_dit_config": {
245
+ "action_conditioned": false,
246
+ "action_dim": 7,
247
+ "action_group_causal_mask_mode": "group_diagonal",
248
+ "attn_head_dim": 128,
249
+ "eps": 1e-06,
250
+ "ffn_dim": 14336,
251
+ "freq_dim": 256,
252
+ "fuse_vae_embedding_in_latents": true,
253
+ "has_image_input": false,
254
+ "hidden_dim": 3072,
255
+ "in_dim": 48,
256
+ "num_heads": 24,
257
+ "num_layers": 30,
258
+ "out_dim": 48,
259
+ "patch_size": [
260
+ 1,
261
+ 2,
262
+ 2
263
+ ],
264
+ "require_clip_embedding": false,
265
+ "require_vae_embedding": false,
266
+ "seperated_timestep": true,
267
+ "text_dim": 4096,
268
+ "use_gradient_checkpointing": true,
269
+ "video_attention_mask_mode": "first_frame_causal"
270
+ },
271
+ "video_scheduler": {
272
+ "infer_shift": 5.0,
273
+ "num_train_timesteps": 1000,
274
+ "train_shift": 5.0
275
+ }
276
+ },
277
+ "vlm_backbone": {
278
+ "text_embed_cache_context_len": 128,
279
+ "text_embed_cache_device": "cpu",
280
+ "text_embed_cache_size": 256,
281
+ "type": "Wan22Backbone"
282
+ }
283
+ },
284
+ "runner": {
285
+ "collator": {
286
+ "keys": [
287
+ "states",
288
+ "images",
289
+ "img_masks",
290
+ "actions",
291
+ "action_masks",
292
+ "embodiment_ids",
293
+ "frame_masks",
294
+ "lang_tokens",
295
+ "lang_masks"
296
+ ],
297
+ "meta_keys": [
298
+ "task_description",
299
+ "info",
300
+ "stats",
301
+ "timestamp"
302
+ ],
303
+ "type": "DictCollator"
304
+ },
305
+ "enable_gradient_checkpointing": false,
306
+ "enable_mixed_precision_training": true,
307
+ "evaluator": {
308
+ "eval_every": 1000,
309
+ "num_inference_steps": 10,
310
+ "save_video": true,
311
+ "seed": 42,
312
+ "type": "training-eval",
313
+ "video_fps": 8
314
+ },
315
+ "grad_accumulation_steps": 1,
316
+ "lr_scheduler": {
317
+ "betas": [
318
+ 0.9,
319
+ 0.95
320
+ ],
321
+ "min_lr_ratio": 0.01,
322
+ "type": "linear-warmup+cosine-decay-min-lr",
323
+ "warmup_ratio": 0.05,
324
+ "weight_decay_style": "uniform"
325
+ },
326
+ "max_epochs": 10,
327
+ "max_grad_norm": 1.0,
328
+ "max_keep_ckpts": 10,
329
+ "max_steps": null,
330
+ "metric": {
331
+ "active_trackers": [
332
+ "jsonl",
333
+ "wandb"
334
+ ],
335
+ "run_dir": "work_dirs",
336
+ "type": "VLAMetric",
337
+ "window_size": 1
338
+ },
339
+ "mixed_precision_dtype": "bf16",
340
+ "optimizer": {
341
+ "lr": 0.0001,
342
+ "type": "AdamW",
343
+ "weight_decay": 0.01
344
+ },
345
+ "reduce_in_full_precision": true,
346
+ "sampler": null,
347
+ "save_epoch_interval": 1,
348
+ "save_iter_interval": 10000,
349
+ "sharding_strategy": "shard-grad-op",
350
+ "type": "FSDPTrainRunner"
351
+ },
352
+ "seed": 42,
353
+ "train_dataloader": {
354
+ "dataset": {
355
+ "batch_shard_size": 8,
356
+ "datasets": {
357
+ "action_key": "action",
358
+ "action_window_size": 32,
359
+ "data_root_path": [
360
+ "datasets/libero_spatial_no_noops_lerobotv2.1",
361
+ "datasets/libero_object_no_noops_lerobotv2.1",
362
+ "datasets/libero_goal_no_noops_lerobotv2.1",
363
+ "datasets/libero_10_no_noops_lerobotv2.1"
364
+ ],
365
+ "frame_sample_stride": 4,
366
+ "frame_window_size": 9,
367
+ "statistic_name": "libero_all_no_noops",
368
+ "transforms": [
369
+ {
370
+ "embodiment_id": 0,
371
+ "name_mappings": {
372
+ "actions": [
373
+ "actions"
374
+ ],
375
+ "observation.state": [
376
+ "states"
377
+ ]
378
+ },
379
+ "parquet_keys": [
380
+ "observation.state",
381
+ "timestamp",
382
+ "actions",
383
+ "info",
384
+ "stats",
385
+ "action_masks"
386
+ ],
387
+ "type": "ProcessParquetInputs",
388
+ "video_backend": "torchcodec",
389
+ "video_keys": [
390
+ "observation.images.image",
391
+ "observation.images.wrist_image"
392
+ ]
393
+ },
394
+ {
395
+ "backend": "torchvision",
396
+ "height": 224,
397
+ "scale_to_unit_interval": true,
398
+ "type": "ResizeImages",
399
+ "width": 224
400
+ },
401
+ {
402
+ "means": [
403
+ 0.5,
404
+ 0.5,
405
+ 0.5
406
+ ],
407
+ "stds": [
408
+ 0.5,
409
+ 0.5,
410
+ 0.5
411
+ ],
412
+ "type": "NormalizeImages"
413
+ },
414
+ {
415
+ "action_dim": 7,
416
+ "action_key": "action",
417
+ "delta_action_dim_mask": [
418
+ true,
419
+ true,
420
+ true,
421
+ true,
422
+ true,
423
+ true,
424
+ false
425
+ ],
426
+ "norm_type": "min_max",
427
+ "pad_invalid_action_delta_dims": true,
428
+ "state_dim": 8,
429
+ "state_key": "proprio",
430
+ "type": "NormalizeStatesAndActions"
431
+ },
432
+ {
433
+ "frame_window_size": 9,
434
+ "num_views": 2,
435
+ "tile_direction": "horizontal",
436
+ "type": "PrepareVideo"
437
+ },
438
+ {
439
+ "max_len": 128,
440
+ "prompt_template": "A video recorded from a robot's point of view executing the following instruction: {task}",
441
+ "tokenizer": {
442
+ "model_path": "./checkpoints/fastwam_base_full/tokenizer",
443
+ "type": "PretrainedTokenizer"
444
+ },
445
+ "type": "LiberoPromptFromInputs",
446
+ "use_conversation": false
447
+ }
448
+ ],
449
+ "type": "ParquetDataset",
450
+ "use_delta": false,
451
+ "window_start_idx": 0
452
+ },
453
+ "name_mappings": {
454
+ "action": [
455
+ "action"
456
+ ],
457
+ "observation.state": [
458
+ "proprio"
459
+ ]
460
+ },
461
+ "reshuffle_each_epoch": true,
462
+ "seed": 42,
463
+ "statistic_keys": [
464
+ "observation.state",
465
+ "timestamp",
466
+ "action"
467
+ ],
468
+ "statistic_name": "libero_all_no_noops",
469
+ "type": "DistributedRepeatingDataset"
470
+ },
471
+ "per_device_batch_size": 8,
472
+ "per_device_num_workers": 8
473
+ },
474
+ "val_dataloader": null
475
+ }
fastwam_joint_libero_full_finetune_bs16/config.yaml ADDED
@@ -0,0 +1,382 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ _action_window_size: 32
2
+ _ckpt_root: ./checkpoints
3
+ _data_root_paths:
4
+ - datasets/libero_spatial_no_noops_lerobotv2.1
5
+ - datasets/libero_object_no_noops_lerobotv2.1
6
+ - datasets/libero_goal_no_noops_lerobotv2.1
7
+ - datasets/libero_10_no_noops_lerobotv2.1
8
+ _frame_sample_stride: 4
9
+ _frame_window_size: 9
10
+ _statistic_name: libero_all_no_noops
11
+ _text_prompt_template: 'A video recorded from a robot''s point of view executing the
12
+ following instruction: {task}'
13
+ _tokenizer: ./checkpoints/fastwam_base_full/tokenizer
14
+ eval:
15
+ manager:
16
+ launch_delay: 0.5
17
+ master_port_base: 29690
18
+ max_tasks_per_gpu: 2
19
+ monitor_interval: 5
20
+ num_gpus: 8
21
+ status_interval: 30
22
+ runner:
23
+ dataset:
24
+ img_buffer_len: 1
25
+ transforms:
26
+ - img_keys:
27
+ - agentview_image
28
+ - robot0_eye_in_hand_image
29
+ type: ProcessLiberoEvalInputs
30
+ - image_resize_strategy: resize-naive
31
+ input_sizes:
32
+ - - 3
33
+ - 224
34
+ - 224
35
+ - - 3
36
+ - 224
37
+ - 224
38
+ means:
39
+ - - 127.5
40
+ - 127.5
41
+ - 127.5
42
+ - - 127.5
43
+ - 127.5
44
+ - 127.5
45
+ stds:
46
+ - - 127.5
47
+ - 127.5
48
+ - 127.5
49
+ - - 127.5
50
+ - 127.5
51
+ - 127.5
52
+ type: TransformImage
53
+ - norm_type: min_max
54
+ out_key: states
55
+ stat_key: proprio
56
+ state_dim: 8
57
+ type: LiberoProprioFromInputs
58
+ - max_len: 128
59
+ prompt_template: 'A video recorded from a robot''s point of view executing
60
+ the following instruction: {task}'
61
+ tokenizer:
62
+ model_path: ./checkpoints/fastwam_base_full/tokenizer
63
+ type: PretrainedTokenizer
64
+ type: LiberoPromptFromInputs
65
+ use_conversation: false
66
+ - frame_window_size: 1
67
+ num_views: 2
68
+ tile_direction: horizontal
69
+ type: PrepareVideo
70
+ type: LiberoParquetEvalDataset
71
+ denormalize_action:
72
+ action_dim: 7
73
+ norm_type: min_max
74
+ type: DenormalizeLiberoAction
75
+ enable_mixed_precision_training: true
76
+ eval_chunk_size: 10
77
+ eval_shard_strategy: task
78
+ inference_seed: 42
79
+ max_steps:
80
+ libero_10: 700
81
+ libero_goal: 400
82
+ libero_object: 400
83
+ libero_spatial: 400
84
+ mixed_precision_dtype: bf16
85
+ model_build_device: cuda
86
+ model_build_dtype: bf16
87
+ model_family: fastwam
88
+ norm_stats_key: libero_all_no_noops
89
+ num_inference_steps: 10
90
+ num_steps_wait: 30
91
+ num_trials_per_task: 50
92
+ preprocess_every_step: false
93
+ resize_size: 224
94
+ save_multi_view_rollout_videos: true
95
+ save_rollout_videos: true
96
+ seed: 42
97
+ task_ids: null
98
+ task_suite_name:
99
+ - libero_spatial
100
+ - libero_object
101
+ - libero_goal
102
+ - libero_10
103
+ type: LiberoEvalRunner
104
+ eval_dataset: null
105
+ inference_model:
106
+ action_horizon: 32
107
+ frame_window_size: 9
108
+ mot_checkpoint_mixed_attn: true
109
+ num_views: 2
110
+ pretrained_name_or_path: ./checkpoints/fastwam_base_full/fastwam_base_full.safetensors
111
+ proprio_dim: 8
112
+ torch_dtype: bf16
113
+ type: FastWAMVLA
114
+ vla_head:
115
+ action_dit_config:
116
+ action_dim: 7
117
+ attn_head_dim: 128
118
+ eps: 1.0e-06
119
+ ffn_dim: 4096
120
+ freq_dim: 256
121
+ hidden_dim: 1024
122
+ num_heads: 24
123
+ num_layers: 30
124
+ text_dim: 4096
125
+ use_gradient_checkpointing: true
126
+ action_scheduler:
127
+ infer_shift: 5.0
128
+ num_train_timesteps: 1000
129
+ train_shift: 5.0
130
+ loss:
131
+ lambda_action: 1.0
132
+ lambda_video: 1.0
133
+ type: FastWAMJointHead
134
+ video_dit_config:
135
+ action_conditioned: false
136
+ action_dim: 7
137
+ action_group_causal_mask_mode: group_diagonal
138
+ attn_head_dim: 128
139
+ eps: 1.0e-06
140
+ ffn_dim: 14336
141
+ freq_dim: 256
142
+ fuse_vae_embedding_in_latents: true
143
+ has_image_input: false
144
+ hidden_dim: 3072
145
+ in_dim: 48
146
+ num_heads: 24
147
+ num_layers: 30
148
+ out_dim: 48
149
+ patch_size:
150
+ - 1
151
+ - 2
152
+ - 2
153
+ require_clip_embedding: false
154
+ require_vae_embedding: false
155
+ seperated_timestep: true
156
+ text_dim: 4096
157
+ use_gradient_checkpointing: true
158
+ video_attention_mask_mode: first_frame_causal
159
+ video_scheduler:
160
+ infer_shift: 5.0
161
+ num_train_timesteps: 1000
162
+ train_shift: 5.0
163
+ vlm_backbone:
164
+ text_embed_cache_context_len: 128
165
+ text_embed_cache_device: cpu
166
+ text_embed_cache_size: 256
167
+ type: Wan22Backbone
168
+ model:
169
+ action_horizon: 32
170
+ frame_window_size: 9
171
+ mot_checkpoint_mixed_attn: true
172
+ num_views: 2
173
+ pretrained_name_or_path: ./checkpoints/fastwam_base_full/fastwam_base_full.safetensors
174
+ proprio_dim: 8
175
+ torch_dtype: bf16
176
+ type: FastWAMVLA
177
+ vla_head:
178
+ action_dit_config:
179
+ action_dim: 7
180
+ attn_head_dim: 128
181
+ eps: 1.0e-06
182
+ ffn_dim: 4096
183
+ freq_dim: 256
184
+ hidden_dim: 1024
185
+ num_heads: 24
186
+ num_layers: 30
187
+ text_dim: 4096
188
+ use_gradient_checkpointing: true
189
+ action_scheduler:
190
+ infer_shift: 5.0
191
+ num_train_timesteps: 1000
192
+ train_shift: 5.0
193
+ loss:
194
+ lambda_action: 1.0
195
+ lambda_video: 1.0
196
+ type: FastWAMJointHead
197
+ video_dit_config:
198
+ action_conditioned: false
199
+ action_dim: 7
200
+ action_group_causal_mask_mode: group_diagonal
201
+ attn_head_dim: 128
202
+ eps: 1.0e-06
203
+ ffn_dim: 14336
204
+ freq_dim: 256
205
+ fuse_vae_embedding_in_latents: true
206
+ has_image_input: false
207
+ hidden_dim: 3072
208
+ in_dim: 48
209
+ num_heads: 24
210
+ num_layers: 30
211
+ out_dim: 48
212
+ patch_size:
213
+ - 1
214
+ - 2
215
+ - 2
216
+ require_clip_embedding: false
217
+ require_vae_embedding: false
218
+ seperated_timestep: true
219
+ text_dim: 4096
220
+ use_gradient_checkpointing: true
221
+ video_attention_mask_mode: first_frame_causal
222
+ video_scheduler:
223
+ infer_shift: 5.0
224
+ num_train_timesteps: 1000
225
+ train_shift: 5.0
226
+ vlm_backbone:
227
+ text_embed_cache_context_len: 128
228
+ text_embed_cache_device: cpu
229
+ text_embed_cache_size: 256
230
+ type: Wan22Backbone
231
+ runner:
232
+ collator:
233
+ keys:
234
+ - states
235
+ - images
236
+ - img_masks
237
+ - actions
238
+ - action_masks
239
+ - embodiment_ids
240
+ - frame_masks
241
+ - lang_tokens
242
+ - lang_masks
243
+ meta_keys:
244
+ - task_description
245
+ - info
246
+ - stats
247
+ - timestamp
248
+ type: DictCollator
249
+ enable_gradient_checkpointing: false
250
+ enable_mixed_precision_training: true
251
+ evaluator:
252
+ eval_every: 1000
253
+ num_inference_steps: 10
254
+ save_video: true
255
+ seed: 42
256
+ type: training-eval
257
+ video_fps: 8
258
+ grad_accumulation_steps: 1
259
+ lr_scheduler:
260
+ betas:
261
+ - 0.9
262
+ - 0.95
263
+ min_lr_ratio: 0.01
264
+ type: linear-warmup+cosine-decay-min-lr
265
+ warmup_ratio: 0.05
266
+ weight_decay_style: uniform
267
+ max_epochs: 10
268
+ max_grad_norm: 1.0
269
+ max_keep_ckpts: 10
270
+ max_steps: null
271
+ metric:
272
+ active_trackers:
273
+ - jsonl
274
+ - wandb
275
+ run_dir: work_dirs
276
+ type: VLAMetric
277
+ window_size: 1
278
+ mixed_precision_dtype: bf16
279
+ optimizer:
280
+ lr: 0.0001
281
+ type: AdamW
282
+ weight_decay: 0.01
283
+ reduce_in_full_precision: true
284
+ sampler: null
285
+ save_epoch_interval: 1
286
+ save_iter_interval: 10000
287
+ sharding_strategy: shard-grad-op
288
+ type: FSDPTrainRunner
289
+ seed: 42
290
+ train_dataloader:
291
+ dataset:
292
+ batch_shard_size: 8
293
+ datasets:
294
+ action_key: action
295
+ action_window_size: 32
296
+ data_root_path:
297
+ - datasets/libero_spatial_no_noops_lerobotv2.1
298
+ - datasets/libero_object_no_noops_lerobotv2.1
299
+ - datasets/libero_goal_no_noops_lerobotv2.1
300
+ - datasets/libero_10_no_noops_lerobotv2.1
301
+ frame_sample_stride: 4
302
+ frame_window_size: 9
303
+ statistic_name: libero_all_no_noops
304
+ transforms:
305
+ - embodiment_id: 0
306
+ name_mappings:
307
+ actions:
308
+ - actions
309
+ observation.state:
310
+ - states
311
+ parquet_keys:
312
+ - observation.state
313
+ - timestamp
314
+ - actions
315
+ - info
316
+ - stats
317
+ - action_masks
318
+ type: ProcessParquetInputs
319
+ video_backend: torchcodec
320
+ video_keys:
321
+ - observation.images.image
322
+ - observation.images.wrist_image
323
+ - backend: torchvision
324
+ height: 224
325
+ scale_to_unit_interval: true
326
+ type: ResizeImages
327
+ width: 224
328
+ - means:
329
+ - 0.5
330
+ - 0.5
331
+ - 0.5
332
+ stds:
333
+ - 0.5
334
+ - 0.5
335
+ - 0.5
336
+ type: NormalizeImages
337
+ - action_dim: 7
338
+ action_key: action
339
+ delta_action_dim_mask:
340
+ - true
341
+ - true
342
+ - true
343
+ - true
344
+ - true
345
+ - true
346
+ - false
347
+ norm_type: min_max
348
+ pad_invalid_action_delta_dims: true
349
+ state_dim: 8
350
+ state_key: proprio
351
+ type: NormalizeStatesAndActions
352
+ - frame_window_size: 9
353
+ num_views: 2
354
+ tile_direction: horizontal
355
+ type: PrepareVideo
356
+ - max_len: 128
357
+ prompt_template: 'A video recorded from a robot''s point of view executing
358
+ the following instruction: {task}'
359
+ tokenizer:
360
+ model_path: ./checkpoints/fastwam_base_full/tokenizer
361
+ type: PretrainedTokenizer
362
+ type: LiberoPromptFromInputs
363
+ use_conversation: false
364
+ type: ParquetDataset
365
+ use_delta: false
366
+ window_start_idx: 0
367
+ name_mappings:
368
+ action:
369
+ - action
370
+ observation.state:
371
+ - proprio
372
+ reshuffle_each_epoch: true
373
+ seed: 42
374
+ statistic_keys:
375
+ - observation.state
376
+ - timestamp
377
+ - action
378
+ statistic_name: libero_all_no_noops
379
+ type: DistributedRepeatingDataset
380
+ per_device_batch_size: 8
381
+ per_device_num_workers: 8
382
+ val_dataloader: null
fastwam_joint_libero_full_finetune_bs16/dataset_statistics.json ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "libero_all_no_noops": {
3
+ "proprio": {
4
+ "mean": [
5
+ -0.04644819089290944,
6
+ 0.034403383486704416,
7
+ 0.7655374256288966,
8
+ 2.9686952302705496,
9
+ -0.21987845450605217,
10
+ -0.12717501774779605,
11
+ 0.027024820750829286,
12
+ -0.02728454226240833
13
+ ],
14
+ "std": [
15
+ 0.1047811675710572,
16
+ 0.15155194421998813,
17
+ 0.37755264388745513,
18
+ 0.34584562447365863,
19
+ 0.9210392242484864,
20
+ 0.3221193452952537,
21
+ 0.014130245844896644,
22
+ 0.014014258634582654
23
+ ],
24
+ "min": [
25
+ -0.48278069496154785,
26
+ -0.3309336006641388,
27
+ 0.008128181099891663,
28
+ 1.002794623374939,
29
+ -3.6312508583068848,
30
+ -1.842738389968872,
31
+ -0.005453015677630901,
32
+ -0.042015016078948975
33
+ ],
34
+ "max": [
35
+ 0.2103137969970703,
36
+ 0.3904264271259308,
37
+ 1.472778081893921,
38
+ 3.7248642444610596,
39
+ 3.5618896484375,
40
+ 1.3863215446472168,
41
+ 0.04232141748070717,
42
+ 0.0013126095291227102
43
+ ],
44
+ "q01": null,
45
+ "q99": null
46
+ },
47
+ "timestamp": {
48
+ "mean": [
49
+ 4.769871954139705
50
+ ],
51
+ "std": [
52
+ 3.675230472000918
53
+ ],
54
+ "min": [
55
+ 0.0
56
+ ],
57
+ "max": [
58
+ 25.2
59
+ ],
60
+ "q01": null,
61
+ "q99": null
62
+ },
63
+ "action": {
64
+ "mean": [
65
+ 0.0618831284549259,
66
+ 0.08701013409377131,
67
+ -0.09098753582797833,
68
+ 0.0005711871427809853,
69
+ 0.005523421982727656,
70
+ -0.005017155657792082,
71
+ 0.5262555238677551
72
+ ],
73
+ "std": [
74
+ 0.33503345581181854,
75
+ 0.3785447232455365,
76
+ 0.44376765641279264,
77
+ 0.03920756640316542,
78
+ 0.06304128440900067,
79
+ 0.07838785125536755,
80
+ 0.4993101721429526
81
+ ],
82
+ "min": [
83
+ -0.9375,
84
+ -0.9375,
85
+ -0.9375,
86
+ -0.24214285612106323,
87
+ -0.375,
88
+ -0.3642857074737549,
89
+ 0.0
90
+ ],
91
+ "max": [
92
+ 0.9375,
93
+ 0.9375,
94
+ 0.9375,
95
+ 0.3557142913341522,
96
+ 0.375,
97
+ 0.375,
98
+ 1.0
99
+ ],
100
+ "q01": null,
101
+ "q99": null
102
+ }
103
+ }
104
+ }
fastwam_joint_libero_full_finetune_bs16/fastwam_joint_libero_full_finetune_2026_08_17_22_14_43.jsonl ADDED
The diff for this file is too large to render. See raw diff