Document Ornith 1.5 A/B benchmark
This commit is contained in:
Executable
+97
@@ -0,0 +1,97 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
[[ ${1:-} == --go && $(hostname) == athena ]] || { echo 'Usage on Athena: run-go.sh --go' >&2; exit 2; }
|
||||
cd /opt/mike-ai/experiments/ornith15-ab
|
||||
OUT=/data/benchmarks/ornith15-ab
|
||||
mkdir -p "$OUT"
|
||||
ORIGINAL=$(docker ps --format '{{.Names}}' | sed -n 's/^mike-ai-llama-\(fast\|medium\|large\|ultra\|uncensored\)$/\1/p')
|
||||
[[ $(wc -w <<<"$ORIGINAL") == 1 ]] || { echo 'Expected exactly one active production profile' >&2; exit 1; }
|
||||
WHISPER_WAS_RUNNING=false
|
||||
docker inspect -f '{{.State.Running}}' mike-ai-whisper 2>/dev/null | grep -qx true && WHISPER_WAS_RUNNING=true
|
||||
TTS_WAS_RUNNING=false
|
||||
docker inspect -f '{{.State.Running}}' mike-ai-qwen3-tts 2>/dev/null | grep -qx true && TTS_WAS_RUNNING=true
|
||||
printf '%s\n' "$ORIGINAL" > "$OUT/original-profile.txt"
|
||||
|
||||
controller() {
|
||||
docker exec mike-ai-profile-controller python3 -c '
|
||||
import os, sys, urllib.request
|
||||
token = os.environ.get("CONTROLLER_TOKEN", "").strip()
|
||||
if not token:
|
||||
token = open(os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token"), encoding="utf-8").read().strip()
|
||||
req = urllib.request.Request("http://127.0.0.1:8090" + sys.argv[1], data=b"{}", headers={"Authorization": "Bearer " + token, "Content-Type": "application/json"}, method="POST")
|
||||
print(urllib.request.urlopen(req, timeout=180).read().decode())
|
||||
' "$1"
|
||||
}
|
||||
|
||||
monitor_pid=''
|
||||
finish() {
|
||||
local rc=$?
|
||||
trap - EXIT HUP INT TERM
|
||||
if [[ -n "$monitor_pid" ]]; then kill "$monitor_pid" 2>/dev/null || true; wait "$monitor_pid" 2>/dev/null || true; fi
|
||||
./case.sh fast stop || true
|
||||
if $TTS_WAS_RUNNING; then docker start mike-ai-qwen3-tts >/dev/null 2>&1 || true; fi
|
||||
if $WHISPER_WAS_RUNNING; then docker start mike-ai-whisper >/dev/null 2>&1 || true; fi
|
||||
controller "/profiles/$ORIGINAL/activate" || true
|
||||
docker ps --format '{{.Names}} {{.Status}}' | grep -E 'mike-ai-(llama-|router|profile-controller|whisper|ornith15-ab)' || true
|
||||
echo "AB_RUN_EXIT=$rc ORIGINAL=$ORIGINAL WHISPER_RESTORED=$WHISPER_WAS_RUNNING TTS_RESTORED=$TTS_WAS_RUNNING" | tee -a "$OUT/run-go.log"
|
||||
exit "$rc"
|
||||
}
|
||||
trap finish EXIT HUP INT TERM
|
||||
|
||||
wait_health() {
|
||||
local url=$1
|
||||
for ((i=0;i<450;i++)); do
|
||||
curl -fsS --max-time 2 "$url/health" >/dev/null 2>&1 && return 0
|
||||
sleep 2
|
||||
done
|
||||
echo "Health timeout: $url" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
record() {
|
||||
local label=$1 base=$2 model=$3
|
||||
echo "START $label $(date -Is)" | tee -a "$OUT/run-go.log"
|
||||
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits > "$OUT/$label-before-gpu.csv"
|
||||
python3 gpu_monitor.py --output "$OUT/$label-gpu.jsonl" --go >/dev/null 2>&1 &
|
||||
monitor_pid=$!
|
||||
python3 measure.py --label "$label" --base "$base" --model "$model" --output "$OUT/$label.json" --go 2>&1 | tee "$OUT/$label.log"
|
||||
kill "$monitor_pid" 2>/dev/null || true
|
||||
wait "$monitor_pid" 2>/dev/null || true
|
||||
monitor_pid=''
|
||||
echo "END $label $(date -Is)" | tee -a "$OUT/run-go.log"
|
||||
}
|
||||
|
||||
qwen() {
|
||||
local profile=$1 ip
|
||||
if [[ -s "$OUT/qwen-$profile.json" ]]; then
|
||||
echo "SKIP qwen-$profile existing complete result" | tee -a "$OUT/run-go.log"
|
||||
return 0
|
||||
fi
|
||||
controller "/profiles/$profile/activate"
|
||||
ip=$(docker inspect "mike-ai-llama-$profile" --format '{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}')
|
||||
wait_health "http://$ip:8080"
|
||||
record "qwen-$profile" "http://$ip:8080" "qwen-$profile"
|
||||
}
|
||||
|
||||
ornith() {
|
||||
local profile=$1
|
||||
./case.sh "$profile" create
|
||||
./case.sh "$profile" start --go
|
||||
wait_health http://127.0.0.1:5007
|
||||
record "ornith-$profile" http://127.0.0.1:5007 ornith15-test
|
||||
./case.sh "$profile" stop
|
||||
}
|
||||
|
||||
# Fresh production baselines under their real service conditions.
|
||||
qwen medium
|
||||
qwen fast
|
||||
qwen ultra
|
||||
|
||||
# Ornith needs the 3060 memory normally occupied by Qwen-TTS.
|
||||
controller /inference/stop
|
||||
if $WHISPER_WAS_RUNNING; then docker stop mike-ai-whisper >/dev/null; fi
|
||||
if $TTS_WAS_RUNNING; then docker stop mike-ai-qwen3-tts >/dev/null; fi
|
||||
nvidia-smi --query-gpu=uuid,name,memory.used,memory.free --format=csv,noheader,nounits > "$OUT/ornith-idle-gpu.csv"
|
||||
ornith fast
|
||||
ornith medium
|
||||
ornith ultra
|
||||
Reference in New Issue
Block a user