File size: 1,869 Bytes
58258b8 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 | #!/bin/bash
# Per-worker monitor: GPU util, current tqdm state for each shard, completion counts,
# and the cluster-wide MongoDB cache size.
#
# Works on any worker (uses $ARNOLD_WORKER_0_HOST for cross-worker mongo reachability,
# so it doesn't matter whether you're on worker_0 or not).
#
# Usage:
# ./scripts/monitor.sh # one shot
# watch -n 30 ./scripts/monitor.sh # auto-refresh
# while true; do clear; ./scripts/monitor.sh; sleep 30; done # if `watch` unavailable
RUN_NAME="${RUN_NAME:-qwen3_4b_openthoughts3_math53K_instill_n8_valredundancy5_round1}"
echo "=== $(hostname) $(date +%H:%M:%S) ==="
nvidia-smi --query-gpu=index,utilization.gpu --format=csv,noheader
echo
echo "=== Shards on this worker ==="
# Show the most recent tqdm update on each shard's log.
# tqdm uses \r to overwrite a single line — naive `tail -1` returns the entire
# line and `cut -cN` truncates from the wrong end. Convert \r → \n and take
# the actual most-recent state.
for f in logs/worker_*/shard_*.log; do
[ -e "$f" ] || continue
recent=$(tail -c 500 "$f" 2>/dev/null | tr '\r' '\n' | tail -1)
echo "$(basename "$f"): $recent"
done
echo
STATS=$(ls "output/${RUN_NAME}/shards"/*/stats.json 2>/dev/null | wc -l)
COMPLETE=$(grep -l "PIPELINE COMPLETE" logs/worker_*/shard_*.log 2>/dev/null | wc -l)
# Mongo cache count — use ARNOLD_WORKER_0_HOST (IPv6) so this works from any worker,
# not just worker_0. 127.0.0.1 would only work on worker_0.
MONGO=$(python3 -c "
from pymongo import MongoClient
import os
host = os.environ['ARNOLD_WORKER_0_HOST']
port = os.environ['ARNOLD_WORKER_0_PORT']
print(MongoClient(f'mongodb://[{host}]:{port}', serverSelectionTimeoutMS=3000).sdg_cache.inference_cache.count_documents({}))
" 2>/dev/null)
echo "stats.json: $STATS/8 | PIPELINE COMPLETE: $COMPLETE/8 | mongo cache: ${MONGO:-<unreachable>}"
|