#!/usr/bin/env bash # SPDX-License-Identifier: Apache-2.0 # Serve ONE serve profile of the BUILT container package, run the smoke test against it, keep the evidence, and always # stop it -- in one device-lock window. Run it once per serve profile (`tt-model profiles # /tt_kernel_manifest.json` lists them; without --profile the package's default profile is served): # # ROOT=/home/ubuntu/experiments/tt-models # $ROOT/bin/devrun -t 3600 -k 150 -- env -u HF_TOKEN -u HUGGING_FACE_HUB_TOKEN sg docker -c \ # "bash code/scripts/container_smoke.sh $ROOT/build/meteor-p150 [port] [--profile NAME] [--log-dir DIR]" # # The smoke test FAILS unless /info reports ETH dispatch and the 12x10 grid and runs what the package pins for the # profile (dispatch, CQs, variant, weights revision); it also compares the output with the stored CPU reference of the # sample when one exists (code/tt_meteor/server/smoke_test.py). Exit code: the smoke test's (0 = PASS); 1 when # serve fails, 2 on a usage error. # # Evidence, kept whatever the outcome (--log-dir, default: logs/smoke/ of this repo, next to code/), named # [-]-.*: # .container.log the container's whole log (boot, requests, shutdown), followed while the container stops # .info.json GET /info as soon as the server is READY (device: dispatch, grid, cores; pins; versions) # .smoke.json the smoke test's /predict output (the SMOKE_OUT environment variable overrides this path) # .result.json profile, port, exit code, start / end times # # `tt-model serve` returns once the server is READY and leaves the container running, so the stop must happen before # the lock is released, also when devrun's timeout TERMs this script: the EXIT trap saves the log, then stops the # container cleanly with SIGTERM (120 s grace, hence devrun -k 150), never `docker kill`, which would leave the chip # dirty. Needs docker access (sg docker) and python3. set -u usage() { echo "usage: container_smoke.sh [port] [--profile NAME] [--log-dir DIR]" >&2; } PROFILE="" LOG_DIR="" POSITIONAL=() while [ $# -gt 0 ]; do case "$1" in --profile) [ $# -ge 2 ] || { usage; exit 2; }; PROFILE="$2"; shift 2 ;; --profile=*) PROFILE="${1#--profile=}"; shift ;; --log-dir) [ $# -ge 2 ] || { usage; exit 2; }; LOG_DIR="$2"; shift 2 ;; --log-dir=*) LOG_DIR="${1#--log-dir=}"; shift ;; -h|--help) usage; exit 0 ;; -*) echo "container_smoke.sh: unknown option $1" >&2; usage; exit 2 ;; *) POSITIONAL+=("$1"); shift ;; esac done [ ${#POSITIONAL[@]} -ge 1 ] && [ ${#POSITIONAL[@]} -le 2 ] || { usage; exit 2; } STAGED="${POSITIONAL[0]}" PORT="${POSITIONAL[1]:-20000}" MANIFEST="$STAGED/tt_kernel_manifest.json" [ -f "$MANIFEST" ] || { echo "container_smoke.sh: $MANIFEST not found (run tt-model package first)" >&2; exit 2; } HERE="$(cd "$(dirname "$0")" && pwd)" PROFILE_ARGS=() [ -n "$PROFILE" ] && PROFILE_ARGS=(--profile "$PROFILE") # The package name and the served profile (tt-model's rule: --profile, else default_profile, else the first serve # profile), hence the container name tt-model gives it: tt-model-- (tt_kernel/container.py). read -r NAME PROFILE_NAME < <(python3 - "$MANIFEST" "$PROFILE" <<'EOF' import json, sys m = json.load(open(sys.argv[1])) c = m.get("container") or {} profiles = c.get("serve_profiles") or [{}] print(m.get("name") or "model", sys.argv[2] or c.get("default_profile") or profiles[0].get("name") or "default") EOF ) [ -n "${NAME:-}" ] || { echo "container_smoke.sh: cannot read the package name from $MANIFEST" >&2; exit 2; } CONTAINER="tt-model-$NAME-$PROFILE_NAME" [ -n "$LOG_DIR" ] || LOG_DIR="$(cd "$HERE/../.." && pwd)/logs/smoke" if ! mkdir -p "$LOG_DIR" 2>/dev/null || [ ! -w "$LOG_DIR" ]; then echo "container_smoke.sh: cannot write $LOG_DIR; keeping the evidence in ${TMPDIR:-/tmp}" >&2 LOG_DIR="${TMPDIR:-/tmp}" fi LOG_DIR="$(cd "$LOG_DIR" && pwd)" STARTED="$(date -u +%Y-%m-%dT%H:%M:%SZ)" STEM="$LOG_DIR/$NAME${PROFILE:+-$PROFILE}-$(date -u +%Y%m%dT%H%M%SZ)" SMOKE_JSON="${SMOKE_OUT:-$STEM.smoke.json}" fetch() { # fetch URL FILE: one GET with a 30 s timeout; the body goes to FILE python3 - "$1" "$2" <<'EOF' import sys, urllib.request try: with urllib.request.urlopen(sys.argv[1], timeout=30) as r: body = r.read() except Exception as e: # noqa: BLE001 -- best effort: the smoke test reports the server's state sys.exit(f"container_smoke.sh: GET {sys.argv[1]} failed: {e}") open(sys.argv[2], "wb").write(body) EOF } CHILD="" STOPPED=0 cleanup() { local rc=$? [ "$STOPPED" = 1 ] && return STOPPED=1 if [ -n "$CHILD" ]; then kill -TERM "$CHILD" 2>/dev/null; wait "$CHILD" 2>/dev/null; fi # The log is lost with the container: follow it (whole history, then the shutdown lines) while it stops. local logger="" if docker inspect "$CONTAINER" >/dev/null 2>&1; then docker logs --follow "$CONTAINER" > "$STEM.container.log" 2>&1 & logger=$! else tt-model logs "${PROFILE_ARGS[@]}" "$MANIFEST" > "$STEM.container.log" 2>&1 || true fi tt-model stop "${PROFILE_ARGS[@]}" "$MANIFEST" || true if [ -n "$logger" ]; then for _ in $(seq 1 30); do kill -0 "$logger" 2>/dev/null || break; sleep 1; done kill "$logger" 2>/dev/null wait "$logger" 2>/dev/null fi python3 - "$STEM.result.json" "$NAME" "$PROFILE_NAME" "$PORT" "$rc" "$STARTED" "$CONTAINER" "$MANIFEST" <<'EOF' import datetime, json, sys out, name, profile, port, rc, started, container, manifest = sys.argv[1:] ended = datetime.datetime.now(datetime.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ") json.dump({"bundle": name, "profile": profile, "port": int(port) if port.isdigit() else port, "rc": int(rc), "result": "PASS" if rc == "0" else "FAIL", "started": started, "ended": ended, "container": container, "manifest": manifest}, open(out, "w"), indent=1) EOF echo "container_smoke.sh: rc=$rc; evidence in $STEM.*" } trap cleanup EXIT trap 'exit 143' TERM trap 'exit 130' INT # Each step runs in the background and is waited for: bash defers a trap while a FOREGROUND command runs, so a TERM # sent to this script alone (devrun's timeout signals the whole process group) would otherwise wait for the step. step() { "$@" & CHILD=$! wait "$CHILD" local rc=$? CHILD="" return "$rc" } # --port and --profile BEFORE the target: options after it are passed through to the container (tt-model cli rule) step tt-model serve --port "$PORT" "${PROFILE_ARGS[@]}" "$MANIFEST" || exit 1 step fetch "http://127.0.0.1:$PORT/info" "$STEM.info.json" step python3 "$HERE/../tt_meteor/server/smoke_test.py" --url "http://127.0.0.1:$PORT" --wait 600 \ --manifest "$MANIFEST" "${PROFILE_ARGS[@]}" --out "$SMOKE_JSON"