diff --git a/README.md b/README.md index 93b373b..6ec2002 100644 --- a/README.md +++ b/README.md @@ -133,6 +133,13 @@ verbindet Mikrofon → Athena Whisper → normalen OpenClaw-Agenten → aktives Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis und bleibt deshalb bei normalen Container-Updates bestehen. + +OpenClaw wird über den Provider **llama.cpp → Existing llama-server** mit +`http://192.168.1.212:8081/v1` verbunden. Der Router beantwortet sowohl +`/models` als auch `/v1/models` mit allen fünf virtuellen Profilen. Dadurch +erkennt OpenClaw die vollständige Auswahl automatisch, während Laden, +Entladen und Umschalten weiterhin ausschließlich der Athena Profile Router +übernimmt. - Portainer: `https://192.168.1.212:9443` - Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119` diff --git a/dev/test_local.sh b/dev/test_local.sh index cf2b38b..f188f49 100755 --- a/dev/test_local.sh +++ b/dev/test_local.sh @@ -131,6 +131,19 @@ assert ids["qwen-ultra"]["context_length"]==262144 assert ids["qwen-uncensored"]["context_length"]==80000 ' && ok "fünf virtuelle Modelle mit korrekten Context Windows" || bad "/v1/models" +# llama.cpp-Clients fragen zuerst den nativen Endpunkt ab. Er muss ebenfalls +# den gesamten virtuellen Katalog liefern und darf nicht auf das aktive Profil +# des Upstreams zusammenschrumpfen. +RESP=$(curl -sf "$BASE/models?autoload=false") +echo "$RESP" | python3 -c ' +import json,sys +d=json.load(sys.stdin) +ids={m["id"]:m for m in d["data"]} +assert set(ids)=={"qwen-fast","qwen-medium","qwen-large","qwen-ultra","qwen-uncensored"}, ids +assert ids["qwen-fast"]["status"]["value"]=="loaded", ids +assert all(ids[name]["status"]["value"]=="unloaded" for name in ids if name!="qwen-fast"), ids +' && ok "nativer /models-Katalog enthält alle fünf Profile" || bad "/models" + # --- 2. /status ----------------------------------------------------------------- echo "== Test 2: /status" RESP=$(curl -sf "$BASE/status") diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index e6a3e55..60a7ee8 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -1859,6 +1859,13 @@ class Handler(BaseHTTPRequestHandler): self._send_auth_required() elif path == "/v1/models" and self.command == "GET": self._send_json(200, self._models_payload()) + elif path == "/models" and self.command == "GET": + # llama.cpp clients (notably OpenClaw's existing-server + # provider) probe the native catalog before /v1/models. Do + # not proxy this request to the one currently active profile, + # otherwise the remaining switchable profiles disappear from + # discovery. + self._send_json(200, self._llamacpp_models_payload()) elif path == "/status" and self.command == "GET": self._send_json(200, self._status_payload()) elif path == "/mode" and self.command == "GET": @@ -2076,6 +2083,31 @@ class Handler(BaseHTTPRequestHandler): "data": models, } + @staticmethod + def _llamacpp_models_payload() -> dict: + """Return every switchable profile in llama.cpp's native catalog.""" + active = current_profile() + models = [] + for name, ctx in PROFILES.items(): + model_id = EXPECTED_MODELS.get(name) or f"qwen-{name}" + models.append({ + "id": model_id, + "object": "model", + "created": 0, + "owned_by": "ai-profile-router", + "context_length": ctx, + "context_window": ctx, + "meta": {"n_ctx_train": ctx}, + "status": { + "value": "loaded" if name == active else "unloaded", + }, + "architecture": { + "input_modalities": ["text", "image"], + "output_modalities": ["text"], + }, + }) + return {"data": models} + def _status_payload(self) -> dict: up = upstream_status() img = STATE.image