File size: 1,165 Bytes
2622c40
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
# @package _global_
# FlashWAM (M1_FusedKV_RopeFixed): decoupled MoT, 1-layer action expert,
# fused_kv KV source, action-aligned video RoPE fix.
# FROM SCRATCH on the pusht dataset (see data/pusht_2cam.yaml).
# Recipe is identical to the dish_utensil runs so the two datasets stay
# comparable: 4 GPUs x batch 8 x accum 1 = global 32, 30 epochs, resume: null.
#
# 32,131 frames / 32 = 1,005 steps per epoch -> 30,150 steps total,
# 6 checkpoints at every 5th epoch. Submitted INDEPENDENTLY of its twin (no
# --dependency), per the user's standing no-chaining preference.
#
# save_every is a PLACEHOLDER — the sbatch recomputes it from meta/info.json
# at launch and passes the real value as a CLI override.

defaults:
  - override /data: pusht_2cam
  - override /model: lift_flashwam_m1_fusedkv_ropefixed
  - _self_

# dataloading
batch_size: 8   # 4 GPUs x 8 x accum 1 = global 32
num_workers: 12

# scheduler
lr_scheduler_type: "cosine"
learning_rate: 1e-4
num_epochs: 30
max_steps: null
log_every: 10
save_every: 5025   # PLACEHOLDER -- overridden at launch, see header
eval_every: 0

# training
gradient_accumulation_steps: 1
weight_decay: 1e-2
resume: null