diffusion-planner-p150 / code /scripts /container_smoke.sh
changh95's picture
tt-model push diffusion-planner-p150 (container)
4d9b003 verified
Raw History Blame Contribute Delete
6.79 kB
#!/usr/bin/env bash
# SPDX-License-Identifier: Apache-2.0
# Serve ONE serve profile of the BUILT container package, run the smoke test against it, keep the evidence, and always
# stop it -- in one device-lock window. Run it once per serve profile (`tt-model profiles
# <staged dir>/tt_kernel_manifest.json` lists them; without --profile the package's default profile is served):
#
# ROOT=/home/ubuntu/experiments/tt-models
# $ROOT/bin/devrun -t 3600 -k 150 -- env -u HF_TOKEN -u HUGGING_FACE_HUB_TOKEN sg docker -c \
# "bash code/scripts/container_smoke.sh $ROOT/build/diffusion-planner-p150 [port] [--profile NAME] [--log-dir DIR]"
#
# The smoke test FAILS unless /info reports ETH dispatch and the 12x10 grid and runs what the package pins for the
# profile (dispatch, CQs, variant, weights revision); it also compares the output with the stored CPU reference of the
# sample when one exists (code/tt_diffusion_planner/server/smoke_test.py). Exit code: the smoke test's (0 = PASS); 1 when
# serve fails, 2 on a usage error.
#
# Evidence, kept whatever the outcome (--log-dir, default: logs/smoke/ of this repo, next to code/), named
# <name>[-<profile>]-<UTC time>.*:
# .container.log the container's whole log (boot, requests, shutdown), followed while the container stops
# .info.json GET /info as soon as the server is READY (device: dispatch, grid, cores; pins; versions)
# .smoke.json the smoke test's /predict output (the SMOKE_OUT environment variable overrides this path)
# .result.json profile, port, exit code, start / end times
#
# `tt-model serve` returns once the server is READY and leaves the container running, so the stop must happen before
# the lock is released, also when devrun's timeout TERMs this script: the EXIT trap saves the log, then stops the
# container cleanly with SIGTERM (120 s grace, hence devrun -k 150), never `docker kill`, which would leave the chip
# dirty. Needs docker access (sg docker) and python3.
set -u
usage() { echo "usage: container_smoke.sh <staged package dir> [port] [--profile NAME] [--log-dir DIR]" >&2; }
PROFILE=""
LOG_DIR=""
POSITIONAL=()
while [ $# -gt 0 ]; do
case "$1" in
--profile) [ $# -ge 2 ] || { usage; exit 2; }; PROFILE="$2"; shift 2 ;;
--profile=*) PROFILE="${1#--profile=}"; shift ;;
--log-dir) [ $# -ge 2 ] || { usage; exit 2; }; LOG_DIR="$2"; shift 2 ;;
--log-dir=*) LOG_DIR="${1#--log-dir=}"; shift ;;
-h|--help) usage; exit 0 ;;
-*) echo "container_smoke.sh: unknown option $1" >&2; usage; exit 2 ;;
*) POSITIONAL+=("$1"); shift ;;
esac
done
[ ${#POSITIONAL[@]} -ge 1 ] && [ ${#POSITIONAL[@]} -le 2 ] || { usage; exit 2; }
STAGED="${POSITIONAL[0]}"
PORT="${POSITIONAL[1]:-20000}"
MANIFEST="$STAGED/tt_kernel_manifest.json"
[ -f "$MANIFEST" ] || { echo "container_smoke.sh: $MANIFEST not found (run tt-model package first)" >&2; exit 2; }
HERE="$(cd "$(dirname "$0")" && pwd)"
PROFILE_ARGS=()
[ -n "$PROFILE" ] && PROFILE_ARGS=(--profile "$PROFILE")
# The package name and the served profile (tt-model's rule: --profile, else default_profile, else the first serve
# profile), hence the container name tt-model gives it: tt-model-<name>-<profile> (tt_kernel/container.py).
read -r NAME PROFILE_NAME < <(python3 - "$MANIFEST" "$PROFILE" <<'EOF'
import json, sys
m = json.load(open(sys.argv[1]))
c = m.get("container") or {}
profiles = c.get("serve_profiles") or [{}]
print(m.get("name") or "model", sys.argv[2] or c.get("default_profile") or profiles[0].get("name") or "default")
EOF
)
[ -n "${NAME:-}" ] || { echo "container_smoke.sh: cannot read the package name from $MANIFEST" >&2; exit 2; }
CONTAINER="tt-model-$NAME-$PROFILE_NAME"
[ -n "$LOG_DIR" ] || LOG_DIR="$(cd "$HERE/../.." && pwd)/logs/smoke"
if ! mkdir -p "$LOG_DIR" 2>/dev/null || [ ! -w "$LOG_DIR" ]; then
echo "container_smoke.sh: cannot write $LOG_DIR; keeping the evidence in ${TMPDIR:-/tmp}" >&2
LOG_DIR="${TMPDIR:-/tmp}"
fi
LOG_DIR="$(cd "$LOG_DIR" && pwd)"
STARTED="$(date -u +%Y-%m-%dT%H:%M:%SZ)"
STEM="$LOG_DIR/$NAME${PROFILE:+-$PROFILE}-$(date -u +%Y%m%dT%H%M%SZ)"
SMOKE_JSON="${SMOKE_OUT:-$STEM.smoke.json}"
fetch() { # fetch URL FILE: one GET with a 30 s timeout; the body goes to FILE
python3 - "$1" "$2" <<'EOF'
import sys, urllib.request
try:
with urllib.request.urlopen(sys.argv[1], timeout=30) as r:
body = r.read()
except Exception as e: # noqa: BLE001 -- best effort: the smoke test reports the server's state
sys.exit(f"container_smoke.sh: GET {sys.argv[1]} failed: {e}")
open(sys.argv[2], "wb").write(body)
EOF
}
CHILD=""
STOPPED=0
cleanup() {
local rc=$?
[ "$STOPPED" = 1 ] && return
STOPPED=1
if [ -n "$CHILD" ]; then kill -TERM "$CHILD" 2>/dev/null; wait "$CHILD" 2>/dev/null; fi
# The log is lost with the container: follow it (whole history, then the shutdown lines) while it stops.
local logger=""
if docker inspect "$CONTAINER" >/dev/null 2>&1; then
docker logs --follow "$CONTAINER" > "$STEM.container.log" 2>&1 &
logger=$!
else
tt-model logs "${PROFILE_ARGS[@]}" "$MANIFEST" > "$STEM.container.log" 2>&1 || true
fi
tt-model stop "${PROFILE_ARGS[@]}" "$MANIFEST" || true
if [ -n "$logger" ]; then
for _ in $(seq 1 30); do kill -0 "$logger" 2>/dev/null || break; sleep 1; done
kill "$logger" 2>/dev/null
wait "$logger" 2>/dev/null
fi
python3 - "$STEM.result.json" "$NAME" "$PROFILE_NAME" "$PORT" "$rc" "$STARTED" "$CONTAINER" "$MANIFEST" <<'EOF'
import datetime, json, sys
out, name, profile, port, rc, started, container, manifest = sys.argv[1:]
ended = datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
json.dump({"bundle": name, "profile": profile, "port": int(port) if port.isdigit() else port, "rc": int(rc),
"result": "PASS" if rc == "0" else "FAIL",
"started": started, "ended": ended, "container": container, "manifest": manifest}, open(out, "w"), indent=1)
EOF
echo "container_smoke.sh: rc=$rc; evidence in $STEM.*"
}
trap cleanup EXIT
trap 'exit 143' TERM
trap 'exit 130' INT
# Each step runs in the background and is waited for: bash defers a trap while a FOREGROUND command runs, so a TERM
# sent to this script alone (devrun's timeout signals the whole process group) would otherwise wait for the step.
step() {
"$@" &
CHILD=$!
wait "$CHILD"
local rc=$?
CHILD=""
return "$rc"
}
# --port and --profile BEFORE the target: options after it are passed through to the container (tt-model cli rule)
step tt-model serve --port "$PORT" "${PROFILE_ARGS[@]}" "$MANIFEST" || exit 1
step fetch "http://127.0.0.1:$PORT/info" "$STEM.info.json"
step python3 "$HERE/../tt_diffusion_planner/server/smoke_test.py" --url "http://127.0.0.1:$PORT" --wait 600 \
--manifest "$MANIFEST" "${PROFILE_ARGS[@]}" --out "$SMOKE_JSON"