#!/bin/bash # FLUX GPU-Resident-Benchmark mit GARANTIERTER Qwen-Recovery. # # Ablauf: # 1. Aktives Qwen-Profil aus override.conf merken # 2. llama.cpp stoppen, VRAM-Abgabe verifizieren # 3. Benchmark ausführen (Python, GPU-resident, ohne CPU-Offload) # 4. IMMER (trap EXIT): llama.cpp wieder starten, Readiness prüfen # # Usage: flux_gpu_benchmark.sh set -u SERVICE=mike-ai-llama-ui.service PROFILE_DIR=/etc/systemd/system/mike-ai-llama-ui.service.d PY=/opt/mike-ai/ai-profile-router/venv/bin/python SCRIPT="$(cd "$(dirname "$0")" && pwd)/flux_gpu_benchmark.py" ROUTER_STATUS=http://127.0.0.1:8081/status # Aktives Profil bestimmen: override.conf bytegenau mit den Profil-Dateien # vergleichen (robuster als String-Matching). PROFILE="" for p in fast medium long; do if cmp -s "$PROFILE_DIR/override.conf" "$PROFILE_DIR/profile-$p.conf.disabled" 2>/dev/null; then PROFILE=$p break fi done if [ -z "$PROFILE" ]; then echo "FEHLER: aktives Profil nicht erkannt (override.conf passt zu keinem Profil-File)" exit 1 fi echo "=== Aktives Qwen-Profil: $PROFILE (wird nach dem Test wiederhergestellt) ===" restore() { echo echo "=== RECOVERY: stelle Profil $PROFILE wieder her ===" systemctl start "$SERVICE" 2>&1 || echo "systemctl start fehlgeschlagen" for i in $(seq 1 150); do if curl -sf "$ROUTER_STATUS" 2>/dev/null | python3 -c ' import json, sys d = json.load(sys.stdin) u = d["upstream"] assert u["reachable"] and u["model"], d print("ready:", u["model"], "ctx", u["ctx"]) ' 2>/dev/null; then echo "=== RECOVERY OK: llama.cpp ist wieder inference-ready ===" return 0 fi sleep 2 done echo "=== RECOVERY FEHLGESCHLAGEN: bitte manuell prüfen (systemctl status $SERVICE) ===" return 1 } trap restore EXIT echo "=== Stoppe $SERVICE ===" systemctl stop "$SERVICE" sleep 3 echo "=== VRAM nach Stop (MiB) ===" nvidia-smi --query-gpu=memory.used --format=csv,noheader,nounits echo "=== Starte Benchmark ===" "$PY" "$SCRIPT" RC=$? echo "=== Benchmark-Exit-Code: $RC ===" exit $RC