#!/bin/bash # Per-worker monitor: GPU util, current tqdm state for each shard, completion counts, # and the cluster-wide MongoDB cache size. # # Works on any worker (uses $ARNOLD_WORKER_0_HOST for cross-worker mongo reachability, # so it doesn't matter whether you're on worker_0 or not). # # Usage: # ./scripts/monitor.sh # one shot # watch -n 30 ./scripts/monitor.sh # auto-refresh # while true; do clear; ./scripts/monitor.sh; sleep 30; done # if `watch` unavailable RUN_NAME="${RUN_NAME:-qwen3_4b_openthoughts3_math53K_instill_n8_valredundancy5_round1}" echo "=== $(hostname) $(date +%H:%M:%S) ===" nvidia-smi --query-gpu=index,utilization.gpu --format=csv,noheader echo echo "=== Shards on this worker ===" # Show the most recent tqdm update on each shard's log. # tqdm uses \r to overwrite a single line — naive `tail -1` returns the entire # line and `cut -cN` truncates from the wrong end. Convert \r → \n and take # the actual most-recent state. for f in logs/worker_*/shard_*.log; do [ -e "$f" ] || continue recent=$(tail -c 500 "$f" 2>/dev/null | tr '\r' '\n' | tail -1) echo "$(basename "$f"): $recent" done echo STATS=$(ls "output/${RUN_NAME}/shards"/*/stats.json 2>/dev/null | wc -l) COMPLETE=$(grep -l "PIPELINE COMPLETE" logs/worker_*/shard_*.log 2>/dev/null | wc -l) # Mongo cache count — use ARNOLD_WORKER_0_HOST (IPv6) so this works from any worker, # not just worker_0. 127.0.0.1 would only work on worker_0. MONGO=$(python3 -c " from pymongo import MongoClient import os host = os.environ['ARNOLD_WORKER_0_HOST'] port = os.environ['ARNOLD_WORKER_0_PORT'] print(MongoClient(f'mongodb://[{host}]:{port}', serverSelectionTimeoutMS=3000).sdg_cache.inference_cache.count_documents({})) " 2>/dev/null) echo "stats.json: $STATS/8 | PIPELINE COMPLETE: $COMPLETE/8 | mongo cache: ${MONGO:-}"