Download scripts/launch_shards.sh from fzzhang/svd-code: direct link, hf CLI and curl.
- Browser
- Download file 1.4 kB
-
https://huggingface.co/fzzhang/svd-code/resolve/main/scripts/launch_shards.sh
- Command line
-
hf download hf://fzzhang/svd-code/scripts/launch_shards.sh
-
curl -L -o launch_shards.sh https://huggingface.co/fzzhang/svd-code/resolve/main/scripts/launch_shards.sh
1.4 kB
| # Launch 8 shards on this worker, one per GPU. | |
| # Worker index is derived from $MY_POD_NAME (Arnold pod name ends with worker-N). | |
| # | |
| # Env vars: | |
| # CONFIG — path to the SDG YAML config (default: math53K config) | |
| # LIMIT — optional --limit value for smoke tests (e.g. LIMIT=128) | |
| # NUM_SHARDS — total shards across all workers (default: 32 = 4 nodes × 8 GPU) | |
| # | |
| # Outputs: | |
| # logs/worker_${WORKER_IDX}/shard_NNN.log | |
| set -e | |
| WORKER_IDX=$(echo "$MY_POD_NAME" | grep -oE 'worker-[0-9]+$' | grep -oE '[0-9]+') | |
| if [ -z "$WORKER_IDX" ]; then | |
| echo "ERROR: could not infer worker index from MY_POD_NAME=$MY_POD_NAME" >&2 | |
| exit 1 | |
| fi | |
| CONFIG="${CONFIG:-sdg/configs/qwen3_4b_openthoughts3_math53K_instill_n8_valredundancy5_round1.yaml}" | |
| NUM_SHARDS="${NUM_SHARDS:-32}" | |
| GPUS_PER_WORKER=8 | |
| START=$((WORKER_IDX * GPUS_PER_WORKER)) | |
| mkdir -p "logs/worker_${WORKER_IDX}" | |
| LIMIT_FLAG="" | |
| if [ -n "$LIMIT" ]; then LIMIT_FLAG="--limit $LIMIT"; fi | |
| echo "worker $WORKER_IDX launching shards $START..$((START+GPUS_PER_WORKER-1)) (config=$CONFIG, limit=${LIMIT:-none})" | |
| for i in $(seq 0 $((GPUS_PER_WORKER - 1))); do | |
| SHARD_ID=$((START + i)) | |
| CUDA_VISIBLE_DEVICES=$i nohup python -m sdg.generate \ | |
| --config "$CONFIG" --shard-id "$SHARD_ID" --num-shards "$NUM_SHARDS" $LIMIT_FLAG \ | |
| > "logs/worker_${WORKER_IDX}/shard_${SHARD_ID}.log" 2>&1 & | |
| done | |
| echo "launched 8 shards; PIDs: $(jobs -p)" | |