85 lines
3.7 KiB
Bash
85 lines
3.7 KiB
Bash
#!/usr/bin/env bash
|
|
set -Eeuo pipefail
|
|
[[ ${1:-} == --go && $(hostname) == athena ]] || { echo 'Usage on Athena: validate-splits.sh --go [fast|medium|ultra|all]' >&2; exit 2; }
|
|
TARGET=${2:-all}
|
|
[[ $TARGET =~ ^(fast|medium|ultra|all)$ ]] || { echo 'Invalid target profile' >&2; exit 2; }
|
|
cd /opt/mike-ai/experiments/ornith15-ab
|
|
OUT=/data/benchmarks/ornith15-ab/split-validation
|
|
mkdir -p "$OUT"
|
|
|
|
ORIGINAL=$(docker ps --format '{{.Names}}' | sed -n 's/^mike-ai-llama-\(fast\|medium\|large\|ultra\|uncensored\)$/\1/p')
|
|
[[ $(wc -w <<<"$ORIGINAL") == 1 ]] || { echo 'Expected exactly one active production profile' >&2; exit 1; }
|
|
WHISPER_WAS_RUNNING=false
|
|
docker inspect -f '{{.State.Running}}' mike-ai-whisper 2>/dev/null | grep -qx true && WHISPER_WAS_RUNNING=true
|
|
TTS_WAS_RUNNING=false
|
|
docker inspect -f '{{.State.Running}}' mike-ai-qwen3-tts 2>/dev/null | grep -qx true && TTS_WAS_RUNNING=true
|
|
|
|
controller() {
|
|
docker exec mike-ai-profile-controller python3 -c '
|
|
import os, sys, urllib.request
|
|
token = os.environ.get("CONTROLLER_TOKEN", "").strip()
|
|
if not token:
|
|
token = open(os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token"), encoding="utf-8").read().strip()
|
|
req = urllib.request.Request("http://127.0.0.1:8090" + sys.argv[1], data=b"{}", headers={"Authorization": "Bearer " + token, "Content-Type": "application/json"}, method="POST")
|
|
print(urllib.request.urlopen(req, timeout=180).read().decode())
|
|
' "$1"
|
|
}
|
|
|
|
finish() {
|
|
local rc=$?
|
|
trap - EXIT HUP INT TERM
|
|
docker stop mike-ai-ornith15-ab >/dev/null 2>&1 || true
|
|
if $TTS_WAS_RUNNING; then docker start mike-ai-qwen3-tts >/dev/null 2>&1 || true; fi
|
|
if $WHISPER_WAS_RUNNING; then docker start mike-ai-whisper >/dev/null 2>&1 || true; fi
|
|
controller "/profiles/$ORIGINAL/activate" || true
|
|
echo "SPLIT_VALIDATION_EXIT=$rc ORIGINAL=$ORIGINAL" | tee -a "$OUT/run.log"
|
|
exit "$rc"
|
|
}
|
|
trap finish EXIT HUP INT TERM
|
|
|
|
wait_health() {
|
|
local profile=$1
|
|
for ((i=0;i<300;i++)); do
|
|
if curl -fsS --max-time 2 http://127.0.0.1:5007/health >/dev/null 2>&1; then return 0; fi
|
|
if ! docker inspect -f '{{.State.Running}}' mike-ai-ornith15-ab 2>/dev/null | grep -qx true; then
|
|
echo "LOAD_FAILED $profile" | tee -a "$OUT/run.log"
|
|
docker logs --tail 80 mike-ai-ornith15-ab > "$OUT/$profile-container.log" 2>&1 || true
|
|
return 1
|
|
fi
|
|
sleep 2
|
|
done
|
|
echo "HEALTH_TIMEOUT $profile" | tee -a "$OUT/run.log"
|
|
docker logs --tail 80 mike-ai-ornith15-ab > "$OUT/$profile-container.log" 2>&1 || true
|
|
return 1
|
|
}
|
|
|
|
probe() {
|
|
local profile=$1
|
|
echo "START $profile $(date -Is)" | tee -a "$OUT/run.log"
|
|
./case.sh "$profile" create
|
|
./case.sh "$profile" start --go
|
|
wait_health "$profile"
|
|
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/$profile-loaded-gpu.csv"
|
|
curl -fsS --max-time 180 http://127.0.0.1:5007/v1/chat/completions \
|
|
-H 'Content-Type: application/json' \
|
|
-d '{"model":"ornith15-test","messages":[{"role":"user","content":"Antworte exakt mit: SPLIT OK"}],"max_tokens":64,"temperature":0}' \
|
|
> "$OUT/$profile-response.json"
|
|
grep -q 'SPLIT OK' "$OUT/$profile-response.json"
|
|
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/$profile-generated-gpu.csv"
|
|
echo "PASS $profile $(date -Is)" | tee -a "$OUT/run.log"
|
|
./case.sh "$profile" stop
|
|
}
|
|
|
|
controller /inference/stop
|
|
if $WHISPER_WAS_RUNNING; then docker stop mike-ai-whisper >/dev/null; fi
|
|
if $TTS_WAS_RUNNING; then docker stop mike-ai-qwen3-tts >/dev/null; fi
|
|
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits | tee "$OUT/idle-gpu.csv"
|
|
|
|
if [[ $TARGET == all ]]; then
|
|
probe fast
|
|
probe medium
|
|
probe ultra
|
|
else
|
|
probe "$TARGET"
|
|
fi
|