Download code/scripts/smoke_all.sh from teawhite/ActionRoPE: direct link, hf CLI and curl.
- Browser
- Download file 6.04 kB
-
https://huggingface.co/teawhite/ActionRoPE/resolve/main/code/scripts/smoke_all.sh
- Command line
-
hf download hf://teawhite/ActionRoPE/code/scripts/smoke_all.sh
-
curl -L -o smoke_all.sh https://huggingface.co/teawhite/ActionRoPE/resolve/main/code/scripts/smoke_all.sh
6.04 kB
| # 8 卡端到端 smoke:单测 → 数据体检 → 8 卡 arope 臂 30 步(含 save_state)→ 8 卡 plain 臂 10 步 → 用 smoke ckpt 推理 | |
| # → check_ckpt 键集核对 → 从 state 续训 2 步。 | |
| # 任一步失败即停(set -e),每步打印耗时;全程日志 tee 到 outputs/smoke_all.log。 | |
| # 用法:bash scripts/smoke_all.sh # 全部 | |
| # FROM=c bash scripts/smoke_all.sh # 从第 c 步开始(修完某步后不必从头重跑) | |
| # 需要 8 张空闲 GPU(0–7);单测与推理只用 GPU 0。 | |
| # 评测 clip 是 val_eybx 的普通 clip;它是 11 条 train 转场的 src_b,dataset.py 已按 src 归属把这些转场剔出训练集。 | |
| set -euo pipefail | |
| ROOT=/opt/dlami/nvme/zhiyangdeng/ActionRoPE | |
| PY=$ROOT/.venv/bin/python | |
| ACC=$ROOT/.venv/bin/accelerate | |
| export DIFFSYNTH_SKIP_DOWNLOAD=True | |
| export PYTHONPATH="$ROOT" | |
| FROM="${FROM:-a}" | |
| CLIP="${CLIP:-clip_Eybx_200000958_000273}" # val_eybx,21 个 cell 动作全为 7,tidal_flats,sidecar valid | |
| INFER_STEPS="${INFER_STEPS:-50}" | |
| LOG=$ROOT/outputs/smoke_all.log | |
| mkdir -p "$ROOT/outputs" "$ROOT/outputs/samples" | |
| exec > >(tee -a "$LOG") 2>&1 | |
| T_ALL=$(date +%s) | |
| # 步骤按字母排序,FROM 之前的跳过;结束时打印耗时 | |
| step_begin() { STEP="$1"; shift; echo; echo "===== [$STEP] $* ($(date '+%F %T')) ====="; T_STEP=$(date +%s); } | |
| step_end() { echo "----- [$STEP] 通过,耗时 $(( $(date +%s) - T_STEP )) s -----"; } | |
| want() { [[ "$1" > "$FROM" || "$1" == "$FROM" ]]; } | |
| # (a) 单测:几何(CPU)+ 模型等价性 + 推理 VAE/位移,全部在 GPU 0 | |
| if want a; then | |
| step_begin a "pytest tests/ -x -q(GPU 0)" | |
| CUDA_VISIBLE_DEVICES=0 "$PY" -m pytest "$ROOT/tests/" -x -q -p no:cacheprovider | |
| step_end | |
| fi | |
| # (b) 数据体检:目录/软链 + 文本表 | |
| if want b; then | |
| step_begin b "inspect_dataset --check layout text_table" | |
| "$PY" -m tools.inspect_dataset --check layout text_table | |
| step_end | |
| fi | |
| # (c) 8 卡 ZeRO-2 arope 臂:256 个 clip,30 步,第 15/30 步存权重(+ save_state 两槽轮转)并验证 8 个 val clip | |
| if want c; then | |
| step_begin c "8 卡 arope 臂 30 步 → outputs/smoke_arope" | |
| mkdir -p "$ROOT/outputs/smoke_arope" | |
| CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 "$ACC" launch --config_file "$ROOT/configs/accelerate_zero2.yaml" \ | |
| "$ROOT/actionrope/train.py" \ | |
| --arm arope --limit 256 --max_steps 30 --save_every 15 --save_state --val_every 15 --val_n 8 \ | |
| --warmup_steps 5 --num_workers 2 --seed 0 \ | |
| --output "$ROOT/outputs/smoke_arope" 2>&1 | tee "$ROOT/outputs/smoke_arope/train.log" | |
| test -s "$ROOT/outputs/smoke_arope/step-30.safetensors" | |
| test -f "$ROOT/outputs/smoke_arope/state/trainer_state.json" | |
| step_end | |
| fi | |
| # (d) 8 卡 ZeRO-2 plain 臂:10 步,第 5/10 步验证,第 10 步存权重 | |
| if want d; then | |
| step_begin d "8 卡 plain 臂 10 步 → outputs/smoke_plain" | |
| mkdir -p "$ROOT/outputs/smoke_plain" | |
| CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 "$ACC" launch --config_file "$ROOT/configs/accelerate_zero2.yaml" \ | |
| "$ROOT/actionrope/train.py" \ | |
| --arm plain --limit 256 --max_steps 10 --save_every 10 --val_every 5 --val_n 8 \ | |
| --warmup_steps 5 --num_workers 2 --seed 0 \ | |
| --output "$ROOT/outputs/smoke_plain" 2>&1 | tee "$ROOT/outputs/smoke_plain/train.log" | |
| test -s "$ROOT/outputs/smoke_plain/step-10.safetensors" | |
| step_end | |
| fi | |
| # (e) arope smoke ckpt 推理:replay 一条 val clip(与 GT 比 PSNR)+ 自造 right 1.0× 指令 | |
| if want e; then | |
| step_begin e "arope step-30 推理:replay $CLIP + right:1.0:21(GPU 0,$INFER_STEPS 步)" | |
| CKPT=$ROOT/outputs/smoke_arope/step-30.safetensors | |
| CUDA_VISIBLE_DEVICES=0 "$PY" -m actionrope.infer --ckpt "$CKPT" --replay "$CLIP" --steps "$INFER_STEPS" \ | |
| --out "$ROOT/outputs/samples/smoke_e2e_arope_replay.mp4" | |
| CUDA_VISIBLE_DEVICES=0 "$PY" -m actionrope.infer --ckpt "$CKPT" --first_frame "$CLIP" --scene tidal_flats \ | |
| --actions "right:1.0:21" --steps "$INFER_STEPS" --seed 0 \ | |
| --out "$ROOT/outputs/samples/smoke_e2e_arope_right_x1.mp4" | |
| step_end | |
| fi | |
| # (f) plain smoke ckpt 推理:同一首帧、同一物理指令(动作走文本,动作词 = 真实屏幕方向) | |
| if want f; then | |
| step_begin f "plain step-10 推理:right:1.0:21(GPU 0,$INFER_STEPS 步)" | |
| CUDA_VISIBLE_DEVICES=0 "$PY" -m actionrope.infer --arm plain --ckpt "$ROOT/outputs/smoke_plain/step-10.safetensors" \ | |
| --first_frame "$CLIP" --scene tidal_flats --actions "right:1.0:21" --steps "$INFER_STEPS" --seed 0 \ | |
| --out "$ROOT/outputs/samples/smoke_e2e_plain_right_x1.mp4" | |
| step_end | |
| fi | |
| # (g) 导出的 safetensors 能 strict 装回 WanModel(CPU),并确认权重确实变了 | |
| if want g; then | |
| step_begin g "check_ckpt outputs/smoke_arope/step-30.safetensors(CPU)" | |
| "$PY" "$ROOT/tests/check_ckpt.py" "$ROOT/outputs/smoke_arope/step-30.safetensors" --arm arope \ | |
| --json_out "$ROOT/outputs/smoke_arope/check_ckpt.json" | |
| "$PY" -c "import json; r=json.load(open('$ROOT/outputs/smoke_arope/check_ckpt.json')); assert r['strict_load_ok'] and r['n_changed_tensors']>0, r" | |
| step_end | |
| fi | |
| # (h) 从 state 软链(第 30 步的优化器状态)续训 2 步:trainer_state 的 step=30 ⇒ 到 32 停 | |
| if want h; then | |
| step_begin h "8 卡 arope 臂从 outputs/smoke_arope/state 续训 2 步 → outputs/smoke_arope_resume" | |
| mkdir -p "$ROOT/outputs/smoke_arope_resume" | |
| CUDA_VISIBLE_DEVICES=0,1,2,3,4,5,6,7 "$ACC" launch --config_file "$ROOT/configs/accelerate_zero2.yaml" \ | |
| "$ROOT/actionrope/train.py" \ | |
| --arm arope --limit 256 --max_steps 32 --save_every 32 --val_every 0 \ | |
| --warmup_steps 5 --num_workers 2 --seed 0 --resume "$ROOT/outputs/smoke_arope/state" \ | |
| --output "$ROOT/outputs/smoke_arope_resume" 2>&1 | tee "$ROOT/outputs/smoke_arope_resume/train.log" | |
| test -s "$ROOT/outputs/smoke_arope_resume/step-32.safetensors" | |
| grep -q "从 step 30 续训" "$ROOT/outputs/smoke_arope_resume/train.log" | |
| step_end | |
| fi | |
| echo | |
| echo "===== smoke_all 全部通过,总耗时 $(( $(date +%s) - T_ALL )) s ($(date '+%F %T')) =====" | |