ProCreations's picture
download
raw
3.35 kB
#!/bin/bash
# Final render job: runs on an HF Jobs multi-GPU node. Code + outputs live in the bucket mounted at /out.
set -euo pipefail
export PYTHONUNBUFFERED=1
WORK=/work; mkdir -p $WORK; cd $WORK
RUN=${RUN:-final}
FRAMES=${FRAMES:-7200}
WIDTH=${WIDTH:-3840}; HEIGHT=${HEIGHT:-2160}; SS=${SS:-2}
CRF=${CRF:-20}; PRESET=${PRESET:-medium}; CODEC=${CODEC:-hevc}
START=${START:-0}
OUTDIR=/out/runs/$RUN; mkdir -p $OUTDIR
log(){ echo "[$(date +%H:%M:%S)] $*"; }
log "run=$RUN frames=$START..$((START+FRAMES)) ${WIDTH}x${HEIGHT} ss=$SS crf=$CRF preset=$PRESET"
nvidia-smi --query-gpu=name,memory.total --format=csv
NGPU=$(nvidia-smi -L | wc -l); NCPU=$(nproc)
log "gpus=$NGPU cpus=$NCPU"
cp /out/code/llmviz.py /out/code/capture.py $WORK/
pip install -q "transformers>=5.5" safetensors numpy pillow imageio-ffmpeg "huggingface_hub>=1.0" 2>&1 | tail -1 || true
FFMPEG=$(python -c "import imageio_ffmpeg;print(imageio_ffmpeg.get_ffmpeg_exe())")
log "ffmpeg=$FFMPEG"
if [ ! -f $WORK/model/config.json ]; then
log "downloading model"; hf download openbmb/MiniCPM5-2B --local-dir $WORK/model >/dev/null 2>&1 || python -c "from huggingface_hub import snapshot_download; snapshot_download('openbmb/MiniCPM5-2B', local_dir='$WORK/model')"
fi
if [ ! -f $WORK/acts.npz ]; then
log "capturing activations (real generation on GPU 0)"
CUDA_VISIBLE_DEVICES=0 MODEL=$WORK/model OUT=$WORK/acts.npz N_NEW=${N_NEW:-96} python capture.py 2>&1 | grep -v Warning | tail -8
cp $WORK/acts.json $OUTDIR/ 2>/dev/null || true
fi
# ---- parallel render: one worker per GPU, contiguous frame ranges
PER=$(( (FRAMES + NGPU - 1) / NGPU ))
THREADS=$(( NCPU / NGPU )); [ $THREADS -lt 2 ] && THREADS=2
log "rendering: $NGPU workers x $PER frames, $THREADS encoder threads each"
PIDS=()
for ((g=0; g<NGPU; g++)); do
S=$((START + g*PER)); E=$((S + PER)); [ $E -gt $((START+FRAMES)) ] && E=$((START+FRAMES))
[ $S -ge $E ] && continue
CUDA_VISIBLE_DEVICES=$g python llmviz.py --acts $WORK/acts.npz --model $WORK/model --width $WIDTH --height $HEIGHT --ss $SS \
--start $S --end $E --out $WORK/seg_$g.mp4 --ffmpeg $FFMPEG --codec $CODEC --crf $CRF --preset $PRESET --threads $THREADS > $WORK/render_$g.log 2>&1 &
PIDS+=($!)
done
# progress heartbeat
while :; do
alive=0; for p in "${PIDS[@]}"; do kill -0 $p 2>/dev/null && alive=$((alive+1)); done
log "workers alive: $alive/${#PIDS[@]} $(for ((g=0; g<NGPU; g++)); do grep -h '\[render\]' $WORK/render_$g.log 2>/dev/null | tail -1 | sed 's/.*frame/g'$g' f/'; done | tr '\n' ' ')"
[ $alive -eq 0 ] && break
sleep 60
done
FAIL=0; for p in "${PIDS[@]}"; do wait $p || FAIL=1; done
for ((g=0; g<NGPU; g++)); do tail -3 $WORK/render_$g.log; done
[ $FAIL -ne 0 ] && { log "A WORKER FAILED"; cp $WORK/render_*.log $OUTDIR/; exit 1; }
# ---- concat
rm -f list.txt; for ((g=0; g<NGPU; g++)); do [ -f seg_$g.mp4 ] && echo "file 'seg_$g.mp4'" >> list.txt; done
$FFMPEG -y -hide_banner -loglevel error -f concat -safe 0 -i list.txt -c copy -movflags +faststart $WORK/$RUN.mp4
python - <<PY
import subprocess, json
exe="$FFMPEG".replace("ffmpeg","ffprobe") if False else None
PY
$FFMPEG -hide_banner -i $WORK/$RUN.mp4 2>&1 | grep -E "Duration|Stream" || true
ls -la $WORK/$RUN.mp4
cp $WORK/$RUN.mp4 $OUTDIR/$RUN.mp4
cp $WORK/render_*.log $OUTDIR/ || true
cp $WORK/acts.npz $OUTDIR/acts.npz || true
log "DONE -> $OUTDIR/$RUN.mp4"

Xet Storage Details

Size:
3.35 kB
·
Xet hash:
8f64df2af954a480916dd118e0fb6d9b30002117bcfdb96177965e2160d19cc9

Xet efficiently stores files, intelligently splitting them into unique chunks and accelerating uploads and downloads. More info.