#!/usr/bin/env bash set -Eeuo pipefail [[ ${1:-} == --go && $(hostname) == athena ]] || { echo 'Usage on Athena: validate-splits.sh --go [fast|medium|ultra|all]' >&2; exit 2; } TARGET=${2:-all} [[ $TARGET =~ ^(fast|medium|ultra|all)$ ]] || { echo 'Invalid target profile' >&2; exit 2; } cd /opt/mike-ai/experiments/ornith15-ab OUT=/data/benchmarks/ornith15-ab/split-validation mkdir -p "$OUT" ORIGINAL=$(docker ps --format '{{.Names}}' | sed -n 's/^mike-ai-llama-\(fast\|medium\|large\|ultra\|uncensored\)$/\1/p') [[ $(wc -w <<<"$ORIGINAL") == 1 ]] || { echo 'Expected exactly one active production profile' >&2; exit 1; } WHISPER_WAS_RUNNING=false docker inspect -f '{{.State.Running}}' mike-ai-whisper 2>/dev/null | grep -qx true && WHISPER_WAS_RUNNING=true TTS_WAS_RUNNING=false docker inspect -f '{{.State.Running}}' mike-ai-qwen3-tts 2>/dev/null | grep -qx true && TTS_WAS_RUNNING=true controller() { docker exec mike-ai-profile-controller python3 -c ' import os, sys, urllib.request token = os.environ.get("CONTROLLER_TOKEN", "").strip() if not token: token = open(os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token"), encoding="utf-8").read().strip() req = urllib.request.Request("http://127.0.0.1:8090" + sys.argv[1], data=b"{}", headers={"Authorization": "Bearer " + token, "Content-Type": "application/json"}, method="POST") print(urllib.request.urlopen(req, timeout=180).read().decode()) ' "$1" } finish() { local rc=$? trap - EXIT HUP INT TERM docker stop mike-ai-ornith15-ab >/dev/null 2>&1 || true if $TTS_WAS_RUNNING; then docker start mike-ai-qwen3-tts >/dev/null 2>&1 || true; fi if $WHISPER_WAS_RUNNING; then docker start mike-ai-whisper >/dev/null 2>&1 || true; fi controller "/profiles/$ORIGINAL/activate" || true echo "SPLIT_VALIDATION_EXIT=$rc ORIGINAL=$ORIGINAL" | tee -a "$OUT/run.log" exit "$rc" } trap finish EXIT HUP INT TERM wait_health() { local profile=$1 for ((i=0;i<300;i++)); do if curl -fsS --max-time 2 http://127.0.0.1:5007/health >/dev/null 2>&1; then return 0; fi if ! docker inspect -f '{{.State.Running}}' mike-ai-ornith15-ab 2>/dev/null | grep -qx true; then echo "LOAD_FAILED $profile" | tee -a "$OUT/run.log" docker logs --tail 80 mike-ai-ornith15-ab > "$OUT/$profile-container.log" 2>&1 || true return 1 fi sleep 2 done echo "HEALTH_TIMEOUT $profile" | tee -a "$OUT/run.log" docker logs --tail 80 mike-ai-ornith15-ab > "$OUT/$profile-container.log" 2>&1 || true return 1 } probe() { local profile=$1 echo "START $profile $(date -Is)" | tee -a "$OUT/run.log" ./case.sh "$profile" create ./case.sh "$profile" start --go wait_health "$profile" nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/$profile-loaded-gpu.csv" curl -fsS --max-time 180 http://127.0.0.1:5007/v1/chat/completions \ -H 'Content-Type: application/json' \ -d '{"model":"ornith15-test","messages":[{"role":"user","content":"Antworte exakt mit: SPLIT OK"}],"max_tokens":64,"temperature":0}' \ > "$OUT/$profile-response.json" grep -q 'SPLIT OK' "$OUT/$profile-response.json" nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/$profile-generated-gpu.csv" echo "PASS $profile $(date -Is)" | tee -a "$OUT/run.log" ./case.sh "$profile" stop } controller /inference/stop if $WHISPER_WAS_RUNNING; then docker stop mike-ai-whisper >/dev/null; fi if $TTS_WAS_RUNNING; then docker stop mike-ai-qwen3-tts >/dev/null; fi nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/idle-gpu.csv" if [[ $TARGET == all ]]; then probe fast probe medium probe ultra else probe "$TARGET" fi