Files
AI-Profile-Router/experiments/bonsai2-ab/run-go.sh
T

86 lines
3.3 KiB
Bash

#!/usr/bin/env bash
# Run after the user has explicitly approved GPU inference. Restore the initial
# production profile even if a benchmark or model load fails.
set -Eeuo pipefail
[[ ${1:-} == --go && $(hostname) == athena ]] || { echo 'Usage on Athena: run-go.sh --go' >&2; exit 2; }
cd /opt/mike-ai/experiments/bonsai2-ab
OUT=/data/benchmarks/bonsai2-ab
mkdir -p "$OUT"
ORIGINAL=$(docker ps --format '{{.Names}}' | sed -n 's/^mike-ai-llama-\(fast\|medium\|large\|ultra\|uncensored\)$/\1/p')
[[ $(wc -w <<<"$ORIGINAL") == 1 ]] || { echo 'Expected exactly one active production profile' >&2; exit 1; }
echo "$ORIGINAL" > "$OUT/original-profile.txt"
controller() {
docker exec mike-ai-profile-controller python3 -c '
import os, sys, urllib.request
token = os.environ.get("CONTROLLER_TOKEN", "").strip()
if not token:
token = open(os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token"), encoding="utf-8").read().strip()
req = urllib.request.Request("http://127.0.0.1:8090" + sys.argv[1], data=b"{}", headers={"Authorization": "Bearer " + token, "Content-Type": "application/json"}, method="POST")
print(urllib.request.urlopen(req, timeout=180).read().decode())
' "$1"
}
monitor_pid=''
finish() {
local rc=$?
trap - EXIT INT TERM
if [[ -n "$monitor_pid" ]]; then kill "$monitor_pid" 2>/dev/null || true; wait "$monitor_pid" 2>/dev/null || true; fi
./case.sh fast stop || true
controller "/profiles/$ORIGINAL/activate" || true
docker ps --format '{{.Names}} {{.Status}}' | grep -E 'mike-ai-(llama-|router|profile-controller|bonsai2-ab)' || true
echo "AB_RUN_EXIT=$rc ORIGINAL=$ORIGINAL" | tee -a "$OUT/run-go.log"
exit "$rc"
}
trap finish EXIT INT TERM
wait_health() {
local url=$1
for ((i=0;i<360;i++)); do
if curl -fsS --max-time 2 "$url/health" >/dev/null 2>&1; then return 0; fi
sleep 2
done
echo "Health timeout: $url" >&2
return 1
}
record() {
local label=$1 base=$2 model=$3
echo "START $label $(date -Is)" | tee -a "$OUT/run-go.log"
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits > "$OUT/$label-before-gpu.csv"
python3 gpu_monitor.py --output "$OUT/$label-gpu.jsonl" --go >/dev/null 2>&1 &
monitor_pid=$!
python3 measure.py --label "$label" --base "$base" --model "$model" --output "$OUT/$label.json" --go 2>&1 | tee "$OUT/$label.log"
kill "$monitor_pid" 2>/dev/null || true
wait "$monitor_pid" 2>/dev/null || true
monitor_pid=''
echo "END $label $(date -Is)" | tee -a "$OUT/run-go.log"
}
qwen() {
local case=$1 ip
controller "/profiles/$case/activate"
ip=$(docker inspect "mike-ai-llama-$case" --format '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}')
wait_health "http://$ip:8080"
record "qwen-$case" "http://$ip:8080" "qwen-$case"
}
bonsai() {
local case=$1
controller /inference/stop
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits > "$OUT/bonsai-$case-idle-gpu.csv"
./case.sh "$case" create
./case.sh "$case" start --go
wait_health http://127.0.0.1:5006
record "bonsai-$case" http://127.0.0.1:5006 bonsai2-test
./case.sh "$case" stop
}
# Qwen Medium was measured first while already active, before this script.
[[ -s "$OUT/qwen-medium.json" ]] || { echo 'Qwen Medium baseline incomplete' >&2; exit 1; }
qwen fast
bonsai fast
bonsai medium
qwen ultra
bonsai ultra