dev: lokale Tests mit Mock-llama.cpp
This commit is contained in:
Executable
+13
@@ -0,0 +1,13 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Simuliert /usr/local/bin/llama-profile für lokale Tests.
|
||||||
|
# Setzt nur die Fake-override.conf (kein systemctl, kein whiptail).
|
||||||
|
set -e
|
||||||
|
D="$(cd "$(dirname "$0")" && pwd)/fake-profile-dir"
|
||||||
|
|
||||||
|
case "${1:-}" in
|
||||||
|
fast|medium|long) ;;
|
||||||
|
*) echo "Usage: fake-llama-profile {fast|medium|long}"; exit 1 ;;
|
||||||
|
esac
|
||||||
|
|
||||||
|
cp "$D/profile-$1.conf.disabled" "$D/override.conf"
|
||||||
|
echo "fake: Profil '$1' gesetzt"
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
ExecStart=/usr/bin/mock-llama --ctx-size 73728
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
ExecStart=/usr/bin/mock-llama --ctx-size 73728
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
ExecStart=/usr/bin/mock-llama --ctx-size 131072
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
ExecStart=/usr/bin/mock-llama --ctx-size 94208
|
||||||
Executable
+121
@@ -0,0 +1,121 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Mock-llama.cpp für lokale Tests des AI Profile Router.
|
||||||
|
|
||||||
|
Simuliert die relevanten Endpunkte von llama.cpp:
|
||||||
|
GET /health
|
||||||
|
GET /v1/models (n_ctx wird aus der Fake-override.conf gelesen)
|
||||||
|
POST /v1/chat/completions (non-streaming, streaming, Tool Calls)
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import time
|
||||||
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||||
|
|
||||||
|
PROFILE_DIR = os.environ.get("MOCK_PROFILE_DIR", "dev/fake-profile-dir")
|
||||||
|
PORT = int(os.environ.get("MOCK_PORT", "18080"))
|
||||||
|
|
||||||
|
|
||||||
|
def current_ctx() -> int:
|
||||||
|
"""Liest --ctx-size aus der Fake-override.conf (simuliert Profilwechsel)."""
|
||||||
|
try:
|
||||||
|
with open(os.path.join(PROFILE_DIR, "override.conf"), encoding="utf-8") as f:
|
||||||
|
for line in f:
|
||||||
|
if "--ctx-size" in line:
|
||||||
|
return int(line.split("--ctx-size")[1].split()[0])
|
||||||
|
except (OSError, ValueError, IndexError):
|
||||||
|
pass
|
||||||
|
return 73728
|
||||||
|
|
||||||
|
|
||||||
|
class Handler(BaseHTTPRequestHandler):
|
||||||
|
def do_GET(self):
|
||||||
|
if self.path == "/health":
|
||||||
|
self._json(200, {"status": "ok"})
|
||||||
|
elif self.path == "/v1/models":
|
||||||
|
ctx = current_ctx()
|
||||||
|
self._json(200, {
|
||||||
|
"object": "list",
|
||||||
|
"data": [{
|
||||||
|
"id": f"mock-model-{ctx}",
|
||||||
|
"object": "model",
|
||||||
|
"owned_by": "mock",
|
||||||
|
"meta": {"n_ctx": ctx},
|
||||||
|
}],
|
||||||
|
})
|
||||||
|
else:
|
||||||
|
self._json(404, {"error": {"message": "not found"}})
|
||||||
|
|
||||||
|
def do_POST(self):
|
||||||
|
length = int(self.headers.get("Content-Length") or 0)
|
||||||
|
body = json.loads(self.rfile.read(length) or b"{}")
|
||||||
|
if self.path == "/v1/chat/completions":
|
||||||
|
if body.get("stream"):
|
||||||
|
self._stream(body)
|
||||||
|
else:
|
||||||
|
self._json(200, self._completion(body))
|
||||||
|
else:
|
||||||
|
self._json(404, {"error": {"message": "not found"}})
|
||||||
|
|
||||||
|
def _completion(self, body: dict) -> dict:
|
||||||
|
model = body.get("model")
|
||||||
|
if body.get("tools"):
|
||||||
|
name = body["tools"][0]["function"]["name"]
|
||||||
|
message = {
|
||||||
|
"role": "assistant",
|
||||||
|
"content": None,
|
||||||
|
"tool_calls": [{
|
||||||
|
"id": "call_mock_1",
|
||||||
|
"type": "function",
|
||||||
|
"function": {"name": name, "arguments": '{"city": "Berlin"}'},
|
||||||
|
}],
|
||||||
|
}
|
||||||
|
finish = "tool_calls"
|
||||||
|
else:
|
||||||
|
message = {"role": "assistant",
|
||||||
|
"content": f"Mock-Antwort (Modell: {model})"}
|
||||||
|
finish = "stop"
|
||||||
|
return {
|
||||||
|
"id": "chatcmpl-mock",
|
||||||
|
"object": "chat.completion",
|
||||||
|
"created": int(time.time()),
|
||||||
|
"model": model,
|
||||||
|
"choices": [{"index": 0, "message": message,
|
||||||
|
"finish_reason": finish}],
|
||||||
|
"usage": {"prompt_tokens": 10, "completion_tokens": 5,
|
||||||
|
"total_tokens": 15},
|
||||||
|
}
|
||||||
|
|
||||||
|
def _stream(self, body: dict) -> None:
|
||||||
|
self.send_response(200)
|
||||||
|
self.send_header("Content-Type", "text/event-stream")
|
||||||
|
self.send_header("Connection", "close")
|
||||||
|
self.end_headers()
|
||||||
|
model = body.get("model")
|
||||||
|
for tok in ["Mock-", "Streaming", "-Antwort", f" ({model})"]:
|
||||||
|
chunk = {
|
||||||
|
"id": "chatcmpl-mock",
|
||||||
|
"object": "chat.completion.chunk",
|
||||||
|
"model": model,
|
||||||
|
"choices": [{"index": 0, "delta": {"content": tok},
|
||||||
|
"finish_reason": None}],
|
||||||
|
}
|
||||||
|
self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode())
|
||||||
|
self.wfile.flush()
|
||||||
|
time.sleep(0.05)
|
||||||
|
self.wfile.write(b"data: [DONE]\n\n")
|
||||||
|
self.wfile.flush()
|
||||||
|
|
||||||
|
def _json(self, code: int, payload: dict) -> None:
|
||||||
|
body = json.dumps(payload).encode()
|
||||||
|
self.send_response(code)
|
||||||
|
self.send_header("Content-Type", "application/json")
|
||||||
|
self.send_header("Content-Length", str(len(body)))
|
||||||
|
self.send_header("Connection", "close")
|
||||||
|
self.end_headers()
|
||||||
|
self.wfile.write(body)
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
print(f"Mock-llama.cpp auf 127.0.0.1:{PORT} (Profile-Dir: {PROFILE_DIR})")
|
||||||
|
ThreadingHTTPServer(("127.0.0.1", PORT), Handler).serve_forever()
|
||||||
Executable
+180
@@ -0,0 +1,180 @@
|
|||||||
|
#!/bin/bash
|
||||||
|
# Lokaler Test des AI Profile Router (Mock-llama.cpp + Fake-Profil-Skript).
|
||||||
|
# Startet Mock + Router, führt alle Testfälle aus, beendet beide.
|
||||||
|
set -uo pipefail
|
||||||
|
cd "$(dirname "$0")/.."
|
||||||
|
|
||||||
|
UP_PORT=18080
|
||||||
|
RT_PORT=18081
|
||||||
|
BASE="http://127.0.0.1:$RT_PORT"
|
||||||
|
FAKE_DIR="$PWD/dev/fake-profile-dir"
|
||||||
|
PASS=0
|
||||||
|
FAIL=0
|
||||||
|
|
||||||
|
cleanup() {
|
||||||
|
kill "${MOCK_PID:-}" "${ROUTER_PID:-}" 2>/dev/null || true
|
||||||
|
rm -f /tmp/mock_pid2
|
||||||
|
wait 2>/dev/null || true
|
||||||
|
}
|
||||||
|
trap cleanup EXIT
|
||||||
|
|
||||||
|
ok() { echo " PASS: $1"; PASS=$((PASS+1)); }
|
||||||
|
bad() { echo " FAIL: $1"; FAIL=$((FAIL+1)); }
|
||||||
|
|
||||||
|
# --- Mock-llama.cpp starten --------------------------------------------------
|
||||||
|
echo "== Starte Mock-llama.cpp (Port $UP_PORT)"
|
||||||
|
MOCK_PROFILE_DIR="$FAKE_DIR" MOCK_PORT="$UP_PORT" python3 dev/mock_upstream.py >/tmp/mock_upstream.log 2>&1 &
|
||||||
|
MOCK_PID=$!
|
||||||
|
sleep 0.5
|
||||||
|
|
||||||
|
# --- Router starten -----------------------------------------------------------
|
||||||
|
echo "== Starte Router (Port $RT_PORT)"
|
||||||
|
ROUTER_HOST=127.0.0.1 ROUTER_PORT="$RT_PORT" \
|
||||||
|
UPSTREAM_URL="http://127.0.0.1:$UP_PORT" \
|
||||||
|
PROFILE_SCRIPT="$PWD/dev/fake-llama-profile.sh" \
|
||||||
|
PROFILE_DIR="$FAKE_DIR" \
|
||||||
|
SWITCH_TIMEOUT=30 \
|
||||||
|
python3 router/ai_profile_router.py >/tmp/router_test.log 2>&1 &
|
||||||
|
ROUTER_PID=$!
|
||||||
|
sleep 0.5
|
||||||
|
|
||||||
|
# --- 1. /v1/models -------------------------------------------------------------
|
||||||
|
echo "== Test 1: /v1/models"
|
||||||
|
RESP=$(curl -sf "$BASE/v1/models")
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
ids={m["id"]:m for m in d["data"]}
|
||||||
|
assert set(ids)=={"qwen-fast","qwen-medium","qwen-long"}, ids
|
||||||
|
assert ids["qwen-fast"]["context_length"]==73728
|
||||||
|
assert ids["qwen-medium"]["context_length"]==94208
|
||||||
|
assert ids["qwen-long"]["context_length"]==131072
|
||||||
|
' && ok "drei virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
||||||
|
|
||||||
|
# --- 2. /status -----------------------------------------------------------------
|
||||||
|
echo "== Test 2: /status"
|
||||||
|
RESP=$(curl -sf "$BASE/status")
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
assert d["current_profile"]=="fast", d
|
||||||
|
assert d["upstream"]["reachable"] is True, d
|
||||||
|
assert d["upstream"]["ctx"]==73728, d
|
||||||
|
' && ok "Status zeigt Profil fast + erreichbares Upstream" || bad "/status"
|
||||||
|
|
||||||
|
# --- 3. Normales Forwarding (non-streaming) -------------------------------------
|
||||||
|
echo "== Test 3: Forwarding non-streaming (qwen-fast)"
|
||||||
|
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||||
|
-d '{"model":"qwen-fast","messages":[{"role":"user","content":"Hallo"}]}')
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
assert d["model"].startswith("mock-model-"), d
|
||||||
|
assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
|
||||||
|
' && ok "Request wurde weitergeleitet, Modell ersetzt" || bad "Forwarding"
|
||||||
|
|
||||||
|
# --- 4. Streaming ----------------------------------------------------------------
|
||||||
|
echo "== Test 4: Streaming (SSE)"
|
||||||
|
RESP=$(curl -sfN "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||||
|
-d '{"model":"qwen-fast","stream":true,"messages":[{"role":"user","content":"Hallo"}]}')
|
||||||
|
echo "$RESP"
|
||||||
|
echo "$RESP" | grep -q "data: " && echo "$RESP" | grep -q "\[DONE\]" \
|
||||||
|
&& ok "SSE-Stream mit [DONE] erhalten" || bad "Streaming"
|
||||||
|
|
||||||
|
# --- 5. Tool Calls -----------------------------------------------------------------
|
||||||
|
echo "== Test 5: Tool Calls"
|
||||||
|
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||||
|
-d '{"model":"qwen-fast","messages":[{"role":"user","content":"Wetter?"}],
|
||||||
|
"tools":[{"type":"function","function":{"name":"get_weather","parameters":{}}}]}' )
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
tc=d["choices"][0]["message"]["tool_calls"]
|
||||||
|
assert tc[0]["function"]["name"]=="get_weather", d
|
||||||
|
assert d["choices"][0]["finish_reason"]=="tool_calls", d
|
||||||
|
' && ok "Tool Calls transparent weitergereicht" || bad "Tool Calls"
|
||||||
|
|
||||||
|
# --- 6. Profilwechsel fast -> medium ------------------------------------------------
|
||||||
|
echo "== Test 6: Profilwechsel fast -> medium"
|
||||||
|
RESP=$(curl -sf -X POST "$BASE/medium")
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
assert d["profile"]=="medium" and d["context_length"]==94208, d
|
||||||
|
' && ok "Profil medium aktiv" || bad "Profilwechsel medium"
|
||||||
|
curl -sf "$BASE/status" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
assert d["current_profile"]=="medium", d
|
||||||
|
assert d["upstream"]["ctx"]==94208, d
|
||||||
|
' && ok "Status bestätigt medium (ctx 94208)" || bad "Status nach Wechsel"
|
||||||
|
|
||||||
|
# --- 7. Profilwechsel medium -> fast --------------------------------------------------
|
||||||
|
echo "== Test 7: Profilwechsel medium -> fast"
|
||||||
|
RESP=$(curl -sf -X POST "$BASE/fast")
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
assert d["profile"]=="fast" and d["context_length"]==73728, d
|
||||||
|
' && ok "Profil fast wieder aktiv" || bad "Profilwechsel fast"
|
||||||
|
|
||||||
|
# --- 8. Virtuelles Modell triggert Profilwechsel ----------------------------------------
|
||||||
|
echo "== Test 8: Chat mit qwen-long triggert Wechsel auf long"
|
||||||
|
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||||
|
-d '{"model":"qwen-long","messages":[{"role":"user","content":"Hallo"}]}')
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
assert d["model"]=="mock-model-131072", d
|
||||||
|
' && ok "qwen-long hat Profil long aktiviert und weitergeleitet" || bad "virtuelles Modell"
|
||||||
|
|
||||||
|
# --- 9. Ungültiges Profil ------------------------------------------------------------------
|
||||||
|
echo "== Test 9: Ungültiges Profil"
|
||||||
|
CODE=$(curl -s -o /tmp/err9.json -w "%{http_code}" -X POST "$BASE/huge")
|
||||||
|
cat /tmp/err9.json; echo
|
||||||
|
[ "$CODE" = "400" ] && ok "400 bei unbekanntem Profil (POST /huge)" || bad "erwartet 400, bekam $CODE"
|
||||||
|
|
||||||
|
CODE=$(curl -s -o /tmp/err9b.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
||||||
|
-H "Content-Type: application/json" -d '{"model":"qwen-huge","messages":[]}')
|
||||||
|
cat /tmp/err9b.json; echo
|
||||||
|
[ "$CODE" = "400" ] && ok "400 bei unbekanntem virtuellen Modell (qwen-huge)" || bad "erwartet 400, bekam $CODE"
|
||||||
|
|
||||||
|
# --- 10. llama.cpp down -> 502, danach Recovery ---------------------------------------------------
|
||||||
|
echo "== Test 10: Upstream down -> 502, danach Recovery"
|
||||||
|
# Profil auf fast setzen (aus Test 8 ist long aktiv)
|
||||||
|
curl -sf -X POST "$BASE/fast" >/dev/null
|
||||||
|
kill "$MOCK_PID" 2>/dev/null; wait "$MOCK_PID" 2>/dev/null || true
|
||||||
|
sleep 0.5
|
||||||
|
CODE=$(curl -s -o /tmp/err10.json -w "%{http_code}" -X POST "$BASE/v1/chat/completions" \
|
||||||
|
-H "Content-Type: application/json" -d '{"model":"qwen-fast","messages":[]}')
|
||||||
|
cat /tmp/err10.json; echo
|
||||||
|
[ "$CODE" = "502" ] && ok "502 bei downem Upstream (Profil bereits aktiv)" || bad "erwartet 502, bekam $CODE"
|
||||||
|
|
||||||
|
# Mock nach ~3 s neu starten (simuliert systemctl restart durch das Profil-Skript)
|
||||||
|
rm -f /tmp/mock_pid2
|
||||||
|
(
|
||||||
|
sleep 3
|
||||||
|
MOCK_PROFILE_DIR="$FAKE_DIR" MOCK_PORT="$UP_PORT" python3 dev/mock_upstream.py >/tmp/mock_upstream2.log 2>&1 &
|
||||||
|
echo $! > /tmp/mock_pid2
|
||||||
|
) &
|
||||||
|
RESP=$(curl -sf -X POST "$BASE/fast")
|
||||||
|
echo "$RESP" | python3 -m json.tool
|
||||||
|
# neuen Mock als MOCK_PID übernehmen, damit Cleanup ihn beendet
|
||||||
|
[ -f /tmp/mock_pid2 ] && MOCK_PID=$(cat /tmp/mock_pid2)
|
||||||
|
echo "$RESP" | python3 -c '
|
||||||
|
import json,sys
|
||||||
|
d=json.load(sys.stdin)
|
||||||
|
assert d["profile"]=="fast" and d["model"]=="mock-model-73728", d
|
||||||
|
' && ok "Recovery: /fast wartet auf Upstream, dann Erfolg" || bad "Recovery"
|
||||||
|
|
||||||
|
# --- Ergebnis --------------------------------------------------------------------------------------------
|
||||||
|
echo
|
||||||
|
echo "== Ergebnis: $PASS bestanden, $FAIL fehlgeschlagen =="
|
||||||
|
[ "$FAIL" -eq 0 ]
|
||||||
Reference in New Issue
Block a user