File size: 6,036 Bytes
880dff9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
#!/usr/bin/env bash
# 8 卡端到端 smoke:单测 → 数据体检 → 8 卡 arope 臂 30 步(含 save_state)→ 8 卡 plain 臂 10 步 → 用 smoke ckpt 推理
# → check_ckpt 键集核对 → 从 state 续训 2 步。
# 任一步失败即停(set -e),每步打印耗时;全程日志 tee 到 outputs/smoke_all.log。
# 用法:bash scripts/smoke_all.sh            # 全部
#      FROM=c bash scripts/smoke_all.sh     # 从第 c 步开始(修完某步后不必从头重跑)
# 需要 8 张空闲 GPU(0–7);单测与推理只用 GPU 0。
# 评测 clip 是 val_eybx 的普通 clip;它是 11 条 train 转场的 src_b,dataset.py 已按 src 归属把这些转场剔出训练集。
set -euo pipefail
ROOT=/opt/dlami/nvme/zhiyangdeng/ActionRoPE
PY=$ROOT/.venv/bin/python
ACC=$ROOT/.venv/bin/accelerate
export DIFFSYNTH_SKIP_DOWNLOAD=True
export PYTHONPATH="$ROOT"
FROM="${FROM:-a}"
CLIP="${CLIP:-clip_Eybx_200000958_000273}"    # val_eybx,21 个 cell 动作全为 7,tidal_flats,sidecar valid
INFER_STEPS="${INFER_STEPS:-50}"
LOG=$ROOT/outputs/smoke_all.log
mkdir -p "$ROOT/outputs" "$ROOT/outputs/samples"
exec > >(tee -a "$LOG") 2>&1

T_ALL=$(date +%s)
# 步骤按字母排序,FROM 之前的跳过;结束时打印耗时
step_begin() { STEP="$1"; shift; echo; echo "===== [$STEP] $* ($(date '+%F %T')) ====="; T_STEP=$(date +%s); }
step_end()   { echo "----- [$STEP] 通过,耗时 $(( $(date +%s) - T_STEP )) s -----"; }
want()       { [[ "$1" > "$FROM" || "$1" == "$FROM" ]]; }

# (a) 单测:几何(CPU)+ 模型等价性 + 推理 VAE/位移,全部在 GPU 0
if want a; then
  step_begin a "pytest tests/ -x -q(GPU 0)"
  CUDA_VISIBLE_DEVICES=0 "$PY" -m pytest "$ROOT/tests/" -x -q -p no:cacheprovider
  step_end
fi

# (b) 数据体检:目录/软链 + 文本表
if want b; then
  step_begin b "inspect_dataset --check layout text_table"
  "$PY" -m tools.inspect_dataset --check layout text_table
  step_end
fi

# (c) 8 卡 ZeRO-2 arope 臂:256 个 clip,30 步,第 15/30 步存权重(+ save_state 两槽轮转)并验证 8 个 val clip
if want c; then
  step_begin c "8 卡 arope 臂 30 步 → outputs/smoke_arope"
  mkdir -p "$ROOT/outputs/smoke_arope"
  CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 "$ACC" launch --config_file "$ROOT/configs/accelerate_zero2.yaml" \
    "$ROOT/actionrope/train.py" \
    --arm arope --limit 256 --max_steps 30 --save_every 15 --save_state --val_every 15 --val_n 8 \
    --warmup_steps 5 --num_workers 2 --seed 0 \
    --output "$ROOT/outputs/smoke_arope" 2>&1 | tee "$ROOT/outputs/smoke_arope/train.log"
  test -s "$ROOT/outputs/smoke_arope/step-30.safetensors"
  test -f "$ROOT/outputs/smoke_arope/state/trainer_state.json"
  step_end
fi

# (d) 8 卡 ZeRO-2 plain 臂:10 步,第 5/10 步验证,第 10 步存权重
if want d; then
  step_begin d "8 卡 plain 臂 10 步 → outputs/smoke_plain"
  mkdir -p "$ROOT/outputs/smoke_plain"
  CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 "$ACC" launch --config_file "$ROOT/configs/accelerate_zero2.yaml" \
    "$ROOT/actionrope/train.py" \
    --arm plain --limit 256 --max_steps 10 --save_every 10 --val_every 5 --val_n 8 \
    --warmup_steps 5 --num_workers 2 --seed 0 \
    --output "$ROOT/outputs/smoke_plain" 2>&1 | tee "$ROOT/outputs/smoke_plain/train.log"
  test -s "$ROOT/outputs/smoke_plain/step-10.safetensors"
  step_end
fi

# (e) arope smoke ckpt 推理:replay 一条 val clip(与 GT 比 PSNR)+ 自造 right 1.0× 指令
if want e; then
  step_begin e "arope step-30 推理:replay $CLIP + right:1.0:21(GPU 0,$INFER_STEPS 步)"
  CKPT=$ROOT/outputs/smoke_arope/step-30.safetensors
  CUDA_VISIBLE_DEVICES=0 "$PY" -m actionrope.infer --ckpt "$CKPT" --replay "$CLIP" --steps "$INFER_STEPS" \
    --out "$ROOT/outputs/samples/smoke_e2e_arope_replay.mp4"
  CUDA_VISIBLE_DEVICES=0 "$PY" -m actionrope.infer --ckpt "$CKPT" --first_frame "$CLIP" --scene tidal_flats \
    --actions "right:1.0:21" --steps "$INFER_STEPS" --seed 0 \
    --out "$ROOT/outputs/samples/smoke_e2e_arope_right_x1.mp4"
  step_end
fi

# (f) plain smoke ckpt 推理:同一首帧、同一物理指令(动作走文本,动作词 = 真实屏幕方向)
if want f; then
  step_begin f "plain step-10 推理:right:1.0:21(GPU 0,$INFER_STEPS 步)"
  CUDA_VISIBLE_DEVICES=0 "$PY" -m actionrope.infer --arm plain --ckpt "$ROOT/outputs/smoke_plain/step-10.safetensors" \
    --first_frame "$CLIP" --scene tidal_flats --actions "right:1.0:21" --steps "$INFER_STEPS" --seed 0 \
    --out "$ROOT/outputs/samples/smoke_e2e_plain_right_x1.mp4"
  step_end
fi

# (g) 导出的 safetensors 能 strict 装回 WanModel(CPU),并确认权重确实变了
if want g; then
  step_begin g "check_ckpt outputs/smoke_arope/step-30.safetensors(CPU)"
  "$PY" "$ROOT/tests/check_ckpt.py" "$ROOT/outputs/smoke_arope/step-30.safetensors" --arm arope \
    --json_out "$ROOT/outputs/smoke_arope/check_ckpt.json"
  "$PY" -c "import json; r=json.load(open('$ROOT/outputs/smoke_arope/check_ckpt.json')); assert r['strict_load_ok'] and r['n_changed_tensors']>0, r"
  step_end
fi

# (h) 从 state 软链(第 30 步的优化器状态)续训 2 步:trainer_state 的 step=30 ⇒ 到 32 停
if want h; then
  step_begin h "8 卡 arope 臂从 outputs/smoke_arope/state 续训 2 步 → outputs/smoke_arope_resume"
  mkdir -p "$ROOT/outputs/smoke_arope_resume"
  CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 "$ACC" launch --config_file "$ROOT/configs/accelerate_zero2.yaml" \
    "$ROOT/actionrope/train.py" \
    --arm arope --limit 256 --max_steps 32 --save_every 32 --val_every 0 \
    --warmup_steps 5 --num_workers 2 --seed 0 --resume "$ROOT/outputs/smoke_arope/state" \
    --output "$ROOT/outputs/smoke_arope_resume" 2>&1 | tee "$ROOT/outputs/smoke_arope_resume/train.log"
  test -s "$ROOT/outputs/smoke_arope_resume/step-32.safetensors"
  grep -q "从 step 30 续训" "$ROOT/outputs/smoke_arope_resume/train.log"
  step_end
fi

echo
echo "===== smoke_all 全部通过,总耗时 $(( $(date +%s) - T_ALL )) s ($(date '+%F %T')) ====="