Expose all profiles through llama.cpp discovery

This commit is contained in:
Mikei386
2026-09-15 11:28:34 +02:00
parent 23f4970072
commit 80917e72b4
3 changed files with 52 additions and 0 deletions
+13
View File
@@ -131,6 +131,19 @@ assert ids["qwen-ultra"]["context_length"]==262144
assert ids["qwen-uncensored"]["context_length"]==80000
' && ok "fünf virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
# llama.cpp-Clients fragen zuerst den nativen Endpunkt ab. Er muss ebenfalls
# den gesamten virtuellen Katalog liefern und darf nicht auf das aktive Profil
# des Upstreams zusammenschrumpfen.
RESP=$(curl -sf "$BASE/models?autoload=false")
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
ids={m["id"]:m for m in d["data"]}
assert set(ids)=={"qwen-fast","qwen-medium","qwen-large","qwen-ultra","qwen-uncensored"}, ids
assert ids["qwen-fast"]["status"]["value"]=="loaded", ids
assert all(ids[name]["status"]["value"]=="unloaded" for name in ids if name!="qwen-fast"), ids
' && ok "nativer /models-Katalog enthält alle fünf Profile" || bad "/models"
# --- 2. /status -----------------------------------------------------------------
echo "== Test 2: /status"
RESP=$(curl -sf "$BASE/status")