Files

85 lines
3.7 KiB
Bash

#!/usr/bin/env bash
set -Eeuo pipefail
[[ ${1:-} == --go && $(hostname) == athena ]] || { echo 'Usage on Athena: validate-splits.sh --go [fast|medium|ultra|all]' >&2; exit 2; }
TARGET=${2:-all}
[[ $TARGET =~ ^(fast|medium|ultra|all)$ ]] || { echo 'Invalid target profile' >&2; exit 2; }
cd /opt/mike-ai/experiments/ornith15-ab
OUT=/data/benchmarks/ornith15-ab/split-validation
mkdir -p "$OUT"
ORIGINAL=$(docker ps --format '{{.Names}}' | sed -n 's/^mike-ai-llama-\(fast\|medium\|large\|ultra\|uncensored\)$/\1/p')
[[ $(wc -w <<<"$ORIGINAL") == 1 ]] || { echo 'Expected exactly one active production profile' >&2; exit 1; }
WHISPER_WAS_RUNNING=false
docker inspect -f '{{.State.Running}}' mike-ai-whisper 2>/dev/null | grep -qx true && WHISPER_WAS_RUNNING=true
TTS_WAS_RUNNING=false
docker inspect -f '{{.State.Running}}' mike-ai-qwen3-tts 2>/dev/null | grep -qx true && TTS_WAS_RUNNING=true
controller() {
docker exec mike-ai-profile-controller python3 -c '
import os, sys, urllib.request
token = os.environ.get("CONTROLLER_TOKEN", "").strip()
if not token:
token = open(os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token"), encoding="utf-8").read().strip()
req = urllib.request.Request("http://127.0.0.1:8090" + sys.argv[1], data=b"{}", headers={"Authorization": "Bearer " + token, "Content-Type": "application/json"}, method="POST")
print(urllib.request.urlopen(req, timeout=180).read().decode())
' "$1"
}
finish() {
local rc=$?
trap - EXIT HUP INT TERM
docker stop mike-ai-ornith15-ab >/dev/null 2>&1 || true
if $TTS_WAS_RUNNING; then docker start mike-ai-qwen3-tts >/dev/null 2>&1 || true; fi
if $WHISPER_WAS_RUNNING; then docker start mike-ai-whisper >/dev/null 2>&1 || true; fi
controller "/profiles/$ORIGINAL/activate" || true
echo "SPLIT_VALIDATION_EXIT=$rc ORIGINAL=$ORIGINAL" | tee -a "$OUT/run.log"
exit "$rc"
}
trap finish EXIT HUP INT TERM
wait_health() {
local profile=$1
for ((i=0;i<300;i++)); do
if curl -fsS --max-time 2 http://127.0.0.1:5007/health >/dev/null 2>&1; then return 0; fi
if ! docker inspect -f '{{.State.Running}}' mike-ai-ornith15-ab 2>/dev/null | grep -qx true; then
echo "LOAD_FAILED $profile" | tee -a "$OUT/run.log"
docker logs --tail 80 mike-ai-ornith15-ab > "$OUT/$profile-container.log" 2>&1 || true
return 1
fi
sleep 2
done
echo "HEALTH_TIMEOUT $profile" | tee -a "$OUT/run.log"
docker logs --tail 80 mike-ai-ornith15-ab > "$OUT/$profile-container.log" 2>&1 || true
return 1
}
probe() {
local profile=$1
echo "START $profile $(date -Is)" | tee -a "$OUT/run.log"
./case.sh "$profile" create
./case.sh "$profile" start --go
wait_health "$profile"
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/$profile-loaded-gpu.csv"
curl -fsS --max-time 180 http://127.0.0.1:5007/v1/chat/completions \
-H 'Content-Type: application/json' \
-d '{"model":"ornith15-test","messages":[{"role":"user","content":"Antworte exakt mit: SPLIT OK"}],"max_tokens":64,"temperature":0}' \
> "$OUT/$profile-response.json"
grep -q 'SPLIT OK' "$OUT/$profile-response.json"
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/$profile-generated-gpu.csv"
echo "PASS $profile $(date -Is)" | tee -a "$OUT/run.log"
./case.sh "$profile" stop
}
controller /inference/stop
if $WHISPER_WAS_RUNNING; then docker stop mike-ai-whisper >/dev/null; fi
if $TTS_WAS_RUNNING; then docker stop mike-ai-qwen3-tts >/dev/null; fi
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/idle-gpu.csv"
if [[ $TARGET == all ]]; then
probe fast
probe medium
probe ultra
else
probe "$TARGET"
fi