Expose all profiles through llama.cpp discovery
This commit is contained in:
@@ -131,6 +131,19 @@ assert ids["qwen-ultra"]["context_length"]==262144
|
||||
assert ids["qwen-uncensored"]["context_length"]==80000
|
||||
' && ok "fünf virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
|
||||
|
||||
# llama.cpp-Clients fragen zuerst den nativen Endpunkt ab. Er muss ebenfalls
|
||||
# den gesamten virtuellen Katalog liefern und darf nicht auf das aktive Profil
|
||||
# des Upstreams zusammenschrumpfen.
|
||||
RESP=$(curl -sf "$BASE/models?autoload=false")
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
ids={m["id"]:m for m in d["data"]}
|
||||
assert set(ids)=={"qwen-fast","qwen-medium","qwen-large","qwen-ultra","qwen-uncensored"}, ids
|
||||
assert ids["qwen-fast"]["status"]["value"]=="loaded", ids
|
||||
assert all(ids[name]["status"]["value"]=="unloaded" for name in ids if name!="qwen-fast"), ids
|
||||
' && ok "nativer /models-Katalog enthält alle fünf Profile" || bad "/models"
|
||||
|
||||
# --- 2. /status -----------------------------------------------------------------
|
||||
echo "== Test 2: /status"
|
||||
RESP=$(curl -sf "$BASE/status")
|
||||
|
||||
Reference in New Issue
Block a user