Expose all profiles through llama.cpp discovery

This commit is contained in:
Mikei386
2026-09-15 11:28:34 +02:00
parent 23f4970072
commit 80917e72b4
3 changed files with 52 additions and 0 deletions
+7
View File
@@ -133,6 +133,13 @@ verbindet Mikrofon → Athena Whisper → normalen OpenClaw-Agenten → aktives
Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten
bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis
und bleibt deshalb bei normalen Container-Updates bestehen. und bleibt deshalb bei normalen Container-Updates bestehen.
OpenClaw wird über den Provider **llama.cpp → Existing llama-server** mit
`http://192.168.1.212:8081/v1` verbunden. Der Router beantwortet sowohl
`/models` als auch `/v1/models` mit allen fünf virtuellen Profilen. Dadurch
erkennt OpenClaw die vollständige Auswahl automatisch, während Laden,
Entladen und Umschalten weiterhin ausschließlich der Athena Profile Router
übernimmt.
- Portainer: `https://192.168.1.212:9443` - Portainer: `https://192.168.1.212:9443`
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119` - Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
+13
View File
@@ -131,6 +131,19 @@ assert ids["qwen-ultra"]["context_length"]==262144
assert ids["qwen-uncensored"]["context_length"]==80000 assert ids["qwen-uncensored"]["context_length"]==80000
' && ok "fünf virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models" ' && ok "fünf virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models"
# llama.cpp-Clients fragen zuerst den nativen Endpunkt ab. Er muss ebenfalls
# den gesamten virtuellen Katalog liefern und darf nicht auf das aktive Profil
# des Upstreams zusammenschrumpfen.
RESP=$(curl -sf "$BASE/models?autoload=false")
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
ids={m["id"]:m for m in d["data"]}
assert set(ids)=={"qwen-fast","qwen-medium","qwen-large","qwen-ultra","qwen-uncensored"}, ids
assert ids["qwen-fast"]["status"]["value"]=="loaded", ids
assert all(ids[name]["status"]["value"]=="unloaded" for name in ids if name!="qwen-fast"), ids
' && ok "nativer /models-Katalog enthält alle fünf Profile" || bad "/models"
# --- 2. /status ----------------------------------------------------------------- # --- 2. /status -----------------------------------------------------------------
echo "== Test 2: /status" echo "== Test 2: /status"
RESP=$(curl -sf "$BASE/status") RESP=$(curl -sf "$BASE/status")
+32
View File
@@ -1859,6 +1859,13 @@ class Handler(BaseHTTPRequestHandler):
self._send_auth_required() self._send_auth_required()
elif path == "/v1/models" and self.command == "GET": elif path == "/v1/models" and self.command == "GET":
self._send_json(200, self._models_payload()) self._send_json(200, self._models_payload())
elif path == "/models" and self.command == "GET":
# llama.cpp clients (notably OpenClaw's existing-server
# provider) probe the native catalog before /v1/models. Do
# not proxy this request to the one currently active profile,
# otherwise the remaining switchable profiles disappear from
# discovery.
self._send_json(200, self._llamacpp_models_payload())
elif path == "/status" and self.command == "GET": elif path == "/status" and self.command == "GET":
self._send_json(200, self._status_payload()) self._send_json(200, self._status_payload())
elif path == "/mode" and self.command == "GET": elif path == "/mode" and self.command == "GET":
@@ -2076,6 +2083,31 @@ class Handler(BaseHTTPRequestHandler):
"data": models, "data": models,
} }
@staticmethod
def _llamacpp_models_payload() -> dict:
"""Return every switchable profile in llama.cpp's native catalog."""
active = current_profile()
models = []
for name, ctx in PROFILES.items():
model_id = EXPECTED_MODELS.get(name) or f"qwen-{name}"
models.append({
"id": model_id,
"object": "model",
"created": 0,
"owned_by": "ai-profile-router",
"context_length": ctx,
"context_window": ctx,
"meta": {"n_ctx_train": ctx},
"status": {
"value": "loaded" if name == active else "unloaded",
},
"architecture": {
"input_modalities": ["text", "image"],
"output_modalities": ["text"],
},
})
return {"data": models}
def _status_payload(self) -> dict: def _status_payload(self) -> dict:
up = upstream_status() up = upstream_status()
img = STATE.image img = STATE.image