diff --git a/README.md b/README.md index 0316b12..6e2b3b0 100644 --- a/README.md +++ b/README.md @@ -40,14 +40,16 @@ Neustart an; danach wird derselbe Befehl erneut ausgeführt. |---|---|---| | Open WebUI | `:8080` | Chat und Administration | | Profile Router | `:8081` | OpenAI-kompatible API, Profilwahl | -| llama.cpp | nur Docker-intern | Inferenz, Vision, MCP | +| llama.cpp | nur Docker-intern | Inferenz und integrierte Vision | | Profile Controller | nur Docker-intern | eng begrenzter Containerwechsel | -| SearXNG | nur Docker-intern | Websuche | +| MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid | -Bildgenerierung, TTS/STT sowie Home-Assistant-, ARR- und Unraid-MCPs werden -bewusst nicht automatisch aktiviert. Sie erhalten später eigene Container und -kleinstmögliche Rechte. Die Bildanalyse ist bereits Bestandteil des -multimodalen Qwen-Modells. +Bildgenerierung und TTS/STT bleiben optionale Dienste. Web-, Home-Assistant-, +ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter +`platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne +`mike-ai-tools`-Netz; llama.cpp erhält keine MCP-Konfiguration und keine +Infrastruktur-Secrets. Die Bildanalyse ist Bestandteil des multimodalen +Qwen-Modells. ## Dokumentation diff --git a/compose.yaml b/compose.yaml index 88d6e90..6863b78 100644 --- a/compose.yaml +++ b/compose.yaml @@ -11,15 +11,12 @@ x-llama-common: &llama-common - /tmp:size=1g,mode=1777 volumes: - "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro" - - ./platform/docker/mcp-standard.json:/etc/mike-ai/mcp-standard.json:ro environment: - SEARXNG_URL: http://searxng:8080 NVIDIA_DRIVER_CAPABILITIES: compute,utility dns: ["${AI_DNS:-1.1.1.1}"] networks: inference: aliases: [llama-upstream] - search: {} security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: @@ -36,7 +33,6 @@ services: labels: com.mike-ai.llama-profile: fast environment: - SEARXNG_URL: http://searxng:8080 NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: @@ -85,8 +81,6 @@ services: - "0.8" - --top-k - "20" - - --mcp-servers-config - - /etc/mike-ai/mcp-standard.json - --device - CUDA0 - --split-mode @@ -108,7 +102,6 @@ services: labels: com.mike-ai.llama-profile: medium environment: - SEARXNG_URL: http://searxng:8080 NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: @@ -157,8 +150,6 @@ services: - "0.8" - --top-k - "20" - - --mcp-servers-config - - /etc/mike-ai/mcp-standard.json - --device - CUDA0 - --split-mode @@ -170,7 +161,6 @@ services: labels: com.mike-ai.llama-profile: long environment: - SEARXNG_URL: http://searxng:8080 NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: @@ -221,8 +211,6 @@ services: - "0.8" - --top-k - "20" - - --mcp-servers-config - - /etc/mike-ai/mcp-standard.json - --device - CUDA0 - --split-mode @@ -276,8 +264,6 @@ services: - all - --no-mmap - --no-ui - - --mcp-servers-config - - /etc/mike-ai/mcp-standard.json - --device - CUDA0 - --split-mode @@ -355,26 +341,19 @@ services: OPENAI_API_BASE_URLS: http://router:8081/v1 OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false} + # Seed native MCP connections on a fresh Open WebUI database. Secrets + # stay inside the tool containers, so these internal URLs need no keys. + TOOL_SERVER_CONNECTIONS: >- + [{"url":"http://mike-ai-mcp-web:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"web-local","name":"Web (lokal)","description":"Kompakte Websuche und Quellenvergleich"}},{"url":"http://mike-ai-mcp-homeassistant:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"homeassistant-local","name":"Home Assistant (lokal)","description":"Home-Assistant-Werkzeuge mit serverseitigem Token"}},{"url":"http://mike-ai-mcp-arr:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"arr-local","name":"ARR (lokal)","description":"Sonarr- und Radarr-Werkzeuge"}},{"url":"http://mike-ai-mcp-unraid-official:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"unraid-readonly-local","name":"Unraid (lokal, read-only)","description":"Begrenzte Unraid-Diagnose"}}] DO_NOT_TRACK: "true" SCARF_NO_ANALYTICS: "true" ports: - "${AI_BIND_ADDRESS:-127.0.0.1}:8080:8080" dns: ["${AI_DNS:-1.1.1.1}"] - networks: [frontend] + networks: [frontend, tools] depends_on: [router] security_opt: ["no-new-privileges:true"] - searxng: - image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60 - container_name: mike-ai-searxng - restart: unless-stopped - volumes: - - ./platform/web-search/searxng-settings.yml:/etc/searxng/settings.yml:ro - networks: [search] - dns: ["${AI_DNS:-1.1.1.1}"] - security_opt: ["no-new-privileges:true"] - cap_drop: [ALL] - networks: frontend: internal: false @@ -388,10 +367,9 @@ networks: internal: true ipam: config: [{subnet: 172.30.30.0/24}] - search: - internal: false - ipam: - config: [{subnet: 172.30.40.0/24}] + tools: + external: true + name: mike-ai-tools volumes: open-webui-data: diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md index 45de60d..47744b8 100644 --- a/docs/ARCHITECTURE.md +++ b/docs/ARCHITECTURE.md @@ -22,7 +22,11 @@ Heimnetz / VPN-Clients +-- llama-medium > exakt einer aktiv +-- llama-long --/ +-- llama-experimental - +-- SearXNG + Web-MCP + +-- internes MCP-Netz + +-- Web-MCP + TinySearch + SearXNG + +-- Home-Assistant-MCP-Relay + +-- ARR-MCP + +-- Unraid-MCP ``` ## Container und Vertrauensgrenzen @@ -33,7 +37,8 @@ Heimnetz / VPN-Clients | Profile Router | nur WireGuard, Port 8081 | OpenAI-API und Profilwahl | | Profile Controller | nein | startet ausschließlich vier bekannte Profile | | llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision | -| SearXNG | nein | Websuche für den lokalen Web-MCP | +| MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche | +| SearXNG/TinySearch | nein | Suchbackend des Web-MCP | Nur der Profile Controller erhält den Docker-Socket. Der Router erhält weder Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `long` @@ -73,35 +78,36 @@ nicht automatisch in die Produktionsprofile aufgenommen. Internetzugang der KI über zuhause laufen, braucht der Heim-Peer zusätzlich IP-Forwarding und NAT ins Heim-WAN. -## Nicht automatisch installiert +## Optionale Erweiterungen -Bildgenerierung, Whisper, TTS sowie Home-Assistant-, ARR- und Unraid-MCPs sind -Erweiterungen. Sie benötigen eigene Modelle, Rechte oder Secrets und bleiben -im sauberen Basissystem deaktiviert. Multimodale Bildanalyse erfolgt direkt -über Qwen plus Projektor. Nicht installierte Worker-Endpunkte antworten klar - mit `feature_disabled`, statt alte systemd-Pfade aufzurufen. +Bildgenerierung, Whisper und TTS benötigen eigene Modelle und bleiben im +Basissystem deaktiviert. Home Assistant, ARR und Unraid sind vorbereitete +MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only +Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen +plus Projektor. Nicht installierte Worker-Endpunkte antworten klar mit +`feature_disabled`, statt alte systemd-Pfade aufzurufen. ## Zentrale MCP-Werkzeugebene -Werkzeuge werden nicht fest in Open WebUI, Hermes oder einen anderen Client -eingebaut. Sie laufen als zentrale, über WireGuard erreichbare MCP-Server. Alle -MCP-fähigen Oberflächen verwenden dadurch dieselben geprüften Werkzeuge, ohne -Secrets oder Installationen zu duplizieren. +Werkzeuge werden nicht in llama.cpp eingebaut. Sie laufen als eigene, +zentrale MCP-Container. Open WebUI greift intern darauf zu. Für externe Clients +wie Hermes wird später ein authentifizierter MCP-Gateway über WireGuard +vorgeschaltet; die unauthentifizierten internen Ports werden niemals direkt +veröffentlicht. So können alle Oberflächen dieselben geprüften Werkzeuge +verwenden, ohne Secrets zu duplizieren. Die Trenneinheit ist **ein Container pro Fachbereich und Vertrauensstufe** – nicht ein Container pro einzelner Funktion und nicht ein gemeinsamer Allzweck-MCP mit sämtlichen Zugangsdaten. ```text -Open WebUI ──┐ -Hermes Agent ├── mcp-gateway ──┬── web-mcp -weitere MCP- ┘ ├── home-assistant-mcp-read -Clients ├── home-assistant-mcp-write - ├── arr-mcp-read - ├── arr-mcp-write - ├── unraid-mcp-read - ├── unraid-mcp-admin - └── sandbox-mcp +Open WebUI ── internes Netz ───────────┬── web-mcp + ├── home-assistant-mcp + ├── arr-mcp + └── unraid-mcp-read + +Hermes Agent ─ WireGuard ─┐ +weitere MCP-Clients ──────┴── mcp-gateway (später) ── dasselbe interne Netz ``` | Container | Werkzeugbereich | Standardrecht | diff --git a/docs/COMPONENTS.md b/docs/COMPONENTS.md index 0fbbdd0..b3e5ee5 100644 --- a/docs/COMPONENTS.md +++ b/docs/COMPONENTS.md @@ -5,17 +5,19 @@ | AI Profile Router | `router/` | vollständig | Kern | | llama.cpp | ggml-org/llama.cpp, festgeschriebener Commit | Buildskript und Commit | Kern | | Qwen-Profile | `platform/profiles/` | vollständig, Modelle ausgenommen | Kern | -| Websuche | TinySearch + SearXNG | Compose und sichere Grundkonfiguration | Kern | -| Web-MCP-Fassade | `platform/web-search/web_search_mcp.py` | vollständig | Kern | -| Home-Assistant-MCP | separates privates Repository | nur Integration dokumentiert | optional | -| ARR-MCP | separates privates Repository | nur Integration dokumentiert | optional | -| Unraid-MCP | separates Repository/Installation | read-only Integration dokumentiert | optional | +| MCP-Tool-Stack | `platform/mcp/compose.yaml` | vollständig | Kern | +| Websuche | TinySearch + SearXNG | intern, ohne veröffentlichten Port | Kern | +| Web-MCP-Fassade | `platform/web-search/web_search_mcp.py` | eigener Container | Kern | +| Home-Assistant-MCP | HA-Endpunkt plus lokaler Relay | eigener optionaler Container | optional | +| ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional | +| Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional | | Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional | | XTTS-v2 | Coqui | Worker, Service und Lockdatei | optional | | FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional | | LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional | | Glances | Distribution | nur Betriebsrolle dokumentiert | optional | -Separate MCP-Repositories werden nicht in dieses Repository kopiert. Ihre -Versionen sollen künftig in einem Release-Manifest referenziert werden. So -bleiben Zuständigkeiten klar und Updates können unabhängig getestet werden. +Upstream-Komponenten werden nicht ungeprüft einkopiert. Images, Python-Pakete +und lokale Patches sind in Dockerfiles, Compose-Mounts und Dokumentation +explizit benannt. So bleiben Zuständigkeiten klar und Updates können +unabhängig getestet werden. diff --git a/docs/CURRENT_REFERENCE.md b/docs/CURRENT_REFERENCE.md index 70f3245..7569c06 100644 --- a/docs/CURRENT_REFERENCE.md +++ b/docs/CURRENT_REFERENCE.md @@ -31,7 +31,7 @@ Zielplattform. | Hauptdienst | `mike-ai-llama-ui.service` | | llama.cpp-Port | 8080, auf dem alten Host noch im LAN gebunden | | Client-Port | 8081 über den Router | -| MCP-Konfiguration | `/etc/mike-ai/mcp-servers.json` | +| MCP-Konfiguration | getrennte Container unter `/opt/mike-ai/mcp-containers` | ### Aktives Fast-Profil diff --git a/docs/INSTALLATION.md b/docs/INSTALLATION.md index 1fbafa6..9e188b1 100644 --- a/docs/INSTALLATION.md +++ b/docs/INSTALLATION.md @@ -74,6 +74,33 @@ Zusätzlich prüfen: Uni-LAN sieht keine KI-Ports; Heimnetz erreicht beide; gestopptes WireGuard lässt KI-Container nicht ins Internet; jeder Profilwechsel startet exakt einen llama-Container; Text, Tool Call und Bild funktionieren. +## Werkzeug-Container + +Der Installer startet Websuche automatisch in einem privaten Docker-Netz. +Weitere Bereiche werden nur aktiviert, wenn ihre root-only Konfiguration schon +vorhanden ist: + +```text +/etc/mike-ai/homeassistant-admin-mcp.env +/etc/mike-ai/arr-mcp.env +/etc/mike-ai/runraid/.env +/usr/local/bin/runraid Version 0.4.2 +``` + +Nach dem Nachreichen einer Datei genügt: + +```bash +sudo /opt/mike-ai/stack/platform/mcp/install-tools.sh +``` + +Auf einer frischen Open-WebUI-Datenbank werden die internen MCP-Adressen über +`TOOL_SERVER_CONNECTIONS` vorbelegt. Bei einer übernommenen Datenbank müssen +die Einträge einmal unter **Admin-Einstellungen → Externe Werkzeuge** geprüft +oder importiert werden. Die Endpunkte stehen in `platform/mcp/README.md`. +Kein MCP-Port wird auf dem Host veröffentlicht. Externe Clients wie Hermes +benötigen später den authentifizierten WireGuard-Gateway und dürfen nicht +direkt auf das interne Werkzeugnetz zugreifen. + Die öffentliche Standardkonfiguration nutzt `UD-IQ4_XS`. Das bislang schnellste Referenzprofil nutzt dagegen die lokal vorhandene `IQ4-MIX`-Datei. Für eine bitgenaue Migration diese Datei anhand der in `CURRENT_REFERENCE.md` diff --git a/docs/RECOVERY_REQUIREMENTS.md b/docs/RECOVERY_REQUIREMENTS.md index df4a1b7..02ec6c5 100644 --- a/docs/RECOVERY_REQUIREMENTS.md +++ b/docs/RECOVERY_REQUIREMENTS.md @@ -36,17 +36,24 @@ Pflichtrollen: - FLUX.2 klein - XTTS-v2 und verwendete Stimme -## 2. Externe Komponenten und Commits – offen +## 2. Externe Komponenten und Commits – teilweise gesichert -Für jedes separate Projekt benötigen wir Repository und Commit: +Im Repository gesichert sind inzwischen: + +- getrennte MCP-Container und internes Netz +- Web-MCP-Fassade sowie gepinnte TinySearch-/SearXNG-Images +- ARR-MCP 1.0.1 und der aktuell eingesetzte kompakte Sonarr-Patch +- Home-Assistant-Relay ohne eingebettetes Token +- Startlogik und Health-Checks + +Noch extern zu beschaffen und exakt festzuhalten sind: - Home-Assistant-MCP -- ARR-MCP -- Unraid read-only MCP +- `runraid` 0.4.2 für den read-only Unraid-MCP - gegebenenfalls eigener Unraid-Administrations-MCP - LLama-GUI, falls sie erhalten bleibt -Jede Komponente bekommt zusätzlich: +Jede noch externe Komponente bekommt zusätzlich: - Installationsbefehl - Systembenutzer @@ -156,7 +163,7 @@ Festlegen, welche Daten persistent sein sollen: - Benchmarkresultate: eigenes Repository - Logs: ohne Prompt- und Tool-Antwortinhalte -## 9. Ende-zu-Ende-Installer – implementiert, Hardware-Abnahme offen +## 9. Ende-zu-Ende-Installer – weitgehend implementiert, Praxistest offen Der Ablauf ist jetzt in `install.sh` zusammengeführt: @@ -170,6 +177,11 @@ enable-selected-mcp-profiles run-acceptance-tests ``` +`platform/mcp/install-tools.sh` installiert den Webbereich automatisch und +aktiviert HA, ARR und Unraid nur bei vorhandenen Secret-/Programmdateien. Offen +bleiben ein kompletter Leerhost-Probelauf und der automatisierte Import einer +bereits bestehenden Open-WebUI-Datenbank. + Jeder Schritt muss wiederholbar, einzeln prüfbar und bei Fehlern abbrechbar sein. Ein fehlgeschlagener Schritt darf keinen halb aktivierten Dienst hinterlassen. diff --git a/docs/SECURITY.md b/docs/SECURITY.md index 80090c7..493456c 100644 --- a/docs/SECURITY.md +++ b/docs/SECURITY.md @@ -21,7 +21,13 @@ verlässt sich nicht allein auf UFW. - Profile Controller: einzige Socket-Ausnahme; feste Profile und nur List/Start/Stop, keine frei wählbaren Images, Befehle oder Mounts. - Open WebUI: einziges persistentes Chat-Volume. -- SearXNG: intern, Suchanfragen ohne Chatverlauf. +- MCP-Fachcontainer: intern, getrennte Secrets und keine Host-Ports. +- TinySearch/SearXNG: intern, Suchanfragen ohne Chatverlauf. + +llama.cpp bekommt weder MCP-Konfiguration noch HA-, ARR- oder Unraid-Secrets. +Open WebUI kennt nur interne MCP-URLs; Authentisierung zu den Zielsystemen +findet im jeweiligen Fachcontainer statt. Externe MCP-Clients werden erst über +einen authentifizierten WireGuard-Gateway zugelassen. Ein Docker-Socket bleibt grundsätzlich privilegiert. Der Controller reduziert die erreichbare Funktion stark, ersetzt aber keine zusätzliche Socket-Proxy- diff --git a/install.sh b/install.sh index 15ceb8e..14495fa 100755 --- a/install.sh +++ b/install.sh @@ -232,7 +232,7 @@ set -euo pipefail WG=$WG_INTERFACE HOME_NET=$WG_HOME_SUBNET TABLE=51820 -for NET in 172.30.10.0/24 172.30.30.0/24 172.30.40.0/24; do +for NET in 172.30.10.0/24 172.30.30.0/24 172.30.40.0/24 172.30.50.0/24; do ip rule add from \"\$NET\" table \"\$TABLE\" priority 12000 2>/dev/null || true done ip route replace \"\$HOME_NET\" dev \"\$WG\" @@ -280,6 +280,10 @@ build_and_start() { cd "$STACK_DIR" docker build --build-arg LLAMA_CPP_COMMIT="$commit" \ -f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local . + # Creates the shared internal tools network before Open WebUI is created. + # Web search always starts; HA/ARR/Unraid only start when their root-only + # secret files and required local artifacts are present. + "$STACK_DIR/platform/mcp/install-tools.sh" docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \ llama-fast llama-medium llama-long llama-experimental docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \ diff --git a/platform/docker/mcp-standard.json b/platform/docker/mcp-standard.json deleted file mode 100644 index 8c018cb..0000000 --- a/platform/docker/mcp-standard.json +++ /dev/null @@ -1,12 +0,0 @@ -{ - "mcpServers": { - "web": { - "command": "/usr/bin/python3", - "args": ["/opt/mike-ai/mcp/web_search_mcp.py"], - "env": { - "SEARXNG_URL": "http://searxng:8080" - }, - "timeout_ms": 120000 - } - } -} diff --git a/platform/install-core.sh b/platform/install-core.sh index 722f563..1037398 100755 --- a/platform/install-core.sh +++ b/platform/install-core.sh @@ -10,30 +10,26 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" PLATFORM="$REPO_ROOT/platform" PROFILE_TARGET=/opt/mike-ai/platform/profiles DROPIN=/etc/systemd/system/mike-ai-llama-ui.service.d -WEB_TARGET=/opt/mike-ai/web-search +MCP_ROOT=/opt/mike-ai/mcp-containers +MCP_TARGET=$MCP_ROOT/platform/mcp +MCP_SEARCH_TARGET=$MCP_ROOT/platform/web-search -install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" "$WEB_TARGET" /etc/mike-ai +install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" \ + "$MCP_TARGET" "$MCP_ROOT/platform/web-search" /etc/mike-ai install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \ /etc/systemd/system/mike-ai-llama-ui.service -install -m 0644 "$PLATFORM/systemd/mike-ai-web-search.service" \ - /etc/systemd/system/mike-ai-web-search.service install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf" install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf" install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf" -install -m 0644 "$PLATFORM/web-search/compose.yaml" "$WEB_TARGET/compose.yaml" -install -m 0644 "$PLATFORM/web-search/tinysearch_config.json" \ - "$WEB_TARGET/tinysearch_config.json" -install -m 0755 "$PLATFORM/web-search/web_search_mcp.py" \ - "$WEB_TARGET/web_search_mcp.py" -if [[ ! -e "$WEB_TARGET/searxng-settings.yml" ]]; then - install -m 0600 "$PLATFORM/web-search/searxng-settings.example.yml" \ - "$WEB_TARGET/searxng-settings.yml.example" -fi - -if [[ ! -e /etc/mike-ai/mcp-servers.json ]]; then - install -m 0600 "$PLATFORM/mcp/mcp-servers.example.json" \ - /etc/mike-ai/mcp-servers.json.example +rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/" +rsync -a --delete --exclude searxng-settings.yml \ + "$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/" +if [[ ! -s $MCP_SEARCH_TARGET/searxng-settings.yml ]]; then + install -m 0640 "$PLATFORM/web-search/searxng-settings.example.yml" \ + "$MCP_SEARCH_TARGET/searxng-settings.yml" + sed -i "s/CHANGE_ME_GENERATE_RANDOM_SECRET/$(openssl rand -hex 32)/" \ + "$MCP_SEARCH_TARGET/searxng-settings.yml" fi systemctl daemon-reload @@ -43,7 +39,7 @@ Kernkonfiguration installiert, aber noch nicht gestartet. Vor dem Start: 1. Modellpfade und Hashes gegen manifest.local.yaml prüfen. -2. /etc/mike-ai/mcp-servers.json mit minimalen Servern erstellen. -3. llama.cpp bauen. -4. Danach: llama-profile fast +2. llama.cpp bauen und mit `llama-profile fast` starten. +3. Fach-Secrets unter /etc/mike-ai ablegen. +4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh EOF diff --git a/platform/mcp/Dockerfile.arr b/platform/mcp/Dockerfile.arr new file mode 100644 index 0000000..746913e --- /dev/null +++ b/platform/mcp/Dockerfile.arr @@ -0,0 +1,12 @@ +FROM python:3.13-slim AS builder +COPY --from=ghcr.io/astral-sh/uv:0.11.7 /uv /uvx /bin/ +RUN uv pip install --system --break-system-packages "arr-mcp[mcp]==1.0.1" + +FROM python:3.13-slim +COPY --from=builder /usr/local /usr/local +RUN groupadd --system --gid 10001 mcp \ + && useradd --system --uid 10001 --gid 10001 --no-create-home mcp +USER 10001:10001 +EXPOSE 8000 +ENTRYPOINT ["arr-mcp"] +CMD ["--transport", "streamable-http", "--host", "0.0.0.0", "--port", "8000", "--auth-type", "none"] diff --git a/platform/mcp/Dockerfile.homeassistant-relay b/platform/mcp/Dockerfile.homeassistant-relay new file mode 100644 index 0000000..799d715 --- /dev/null +++ b/platform/mcp/Dockerfile.homeassistant-relay @@ -0,0 +1,8 @@ +FROM nginx:1.29-alpine +COPY platform/mcp/homeassistant.conf.template /etc/nginx/templates/homeassistant.conf.template +COPY platform/mcp/ha-relay-entrypoint.sh /usr/local/bin/ha-relay-entrypoint +RUN chmod 0755 /usr/local/bin/ha-relay-entrypoint \ + && mkdir -p /tmp/client_temp /tmp/proxy_temp \ + && chown -R nginx:nginx /tmp/client_temp /tmp/proxy_temp +EXPOSE 8000 +ENTRYPOINT ["/usr/local/bin/ha-relay-entrypoint"] diff --git a/platform/mcp/Dockerfile.unraid-ssh b/platform/mcp/Dockerfile.unraid-ssh new file mode 100644 index 0000000..5701134 --- /dev/null +++ b/platform/mcp/Dockerfile.unraid-ssh @@ -0,0 +1,15 @@ +FROM python:3.13-slim + +ARG MCP_PROXY_VERSION=0.12.0 +RUN apt-get update \ + && apt-get install -y --no-install-recommends openssh-client \ + && rm -rf /var/lib/apt/lists/* \ + && pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp>=1.17,<2" \ + && useradd --system --uid 10001 --create-home --home-dir /app mcp + +RUN touch /app/unraid_mcp.py && chown 10001:10001 /app/unraid_mcp.py +USER 10001:10001 +WORKDIR /app +EXPOSE 8000 +ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"] +CMD ["python", "/app/unraid_mcp.py"] diff --git a/platform/mcp/Dockerfile.web b/platform/mcp/Dockerfile.web new file mode 100644 index 0000000..179cde2 --- /dev/null +++ b/platform/mcp/Dockerfile.web @@ -0,0 +1,14 @@ +FROM python:3.13-slim + +ARG MCP_PROXY_VERSION=0.12.0 +RUN pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp>=1.17,<2" + +RUN useradd --system --uid 10001 --create-home --home-dir /app mcp +COPY web-search/web_search_mcp.py /app/web_search_mcp.py +RUN chown -R 10001:10001 /app + +USER 10001:10001 +WORKDIR /app +EXPOSE 8000 +ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"] +CMD ["python", "/app/web_search_mcp.py"] diff --git a/platform/mcp/README.md b/platform/mcp/README.md index c8c1ed8..7a9aeb2 100644 --- a/platform/mcp/README.md +++ b/platform/mcp/README.md @@ -1,43 +1,74 @@ -# MCP-Architektur +# Zentrale MCP-Werkzeugebene -Die produktive MCP-Konfiguration ist absichtlich nicht Bestandteil des Git- -Repositories, weil sie lokale Pfade und Zugangsdaten referenziert. Das Beispiel -zeigt nur die Struktur. +MCP-Werkzeuge sind **keine llama.cpp-Startparameter**. Sie laufen als kleine, +voneinander getrennte Container und werden von OpenWebUI, Hermes oder einem +anderen MCP-Client gezielt ausgewählt. Das hält Tool-Schemas aus normalen +Prompts heraus, verhindert den früher beobachteten Kontextverbrauch von über +200.000 Tokens und macht Werkzeuge unabhängig vom geladenen Modellprofil. -## Empfohlene Server +## Container -- `web`: Websuche über lokales TinySearch/SearXNG -- `homeassistant`: Administration mit eigenem, minimal berechtigtem Token -- `arr`: Sonarr/Radarr über spezialisierte Aktionen -- `unraid-readonly`: Diagnose ohne Schreiboperationen +| Container | Endpunkt im Netz `mike-ai-tools` | Zweck | Standard | +|---|---|---|---| +| `mcp-web` | `http://mike-ai-mcp-web:8000/mcp` | kompakte Websuche und Quellenvergleich | an | +| `mcp-homeassistant` | `http://mike-ai-mcp-homeassistant:8000/mcp` | Relay zum nativen HA-MCP; Token bleibt serverseitig | Profil `homeassistant` | +| `mcp-arr` | `http://mike-ai-mcp-arr:8000/mcp` | Sonarr/Radarr/Prowlarr mit serverseitiger Policy | Profil `arr` | +| `mcp-unraid-official` | `http://mike-ai-mcp-unraid-official:8000/mcp` | offizieller, read-only begrenzter Unraid-Zugang | Profil `unraid` | +| `mcp-unraid-ssh` | `http://mike-ai-mcp-unraid-ssh:8000/mcp` | erweiterte Diagnose über einen erzwungenen SSH-Befehl | optional (`extended`) | -## Getrennte Konfigurationen +TinySearch und SearXNG sind interne Abhängigkeiten des Web-MCPs und werden +nicht direkt als allgemeine Werkzeuge angeboten. -Statt alle Werkzeuge ständig zu laden, werden mehrere Dateien empfohlen: +## Sicherheitsmodell -```text -/etc/mike-ai/mcp-standard.json -/etc/mike-ai/mcp-homeassistant.json -/etc/mike-ai/mcp-arr.json -/etc/mike-ai/mcp-unraid-readonly.json -/etc/mike-ai/mcp-unraid-write.json +- Kein MCP-Port wird auf eine Host-Adresse veröffentlicht. +- Nur Clients im privaten Docker-Netz `mike-ai-tools` erreichen die Endpunkte. +- Secrets bleiben in Dateien unter `/etc/mike-ai` und werden read-only + eingehängt. Sie gehören weder in Git noch in OpenWebUI-Tooldefinitionen. +- Jeder Container ist read-only, verliert Linux-Capabilities und hat + `no-new-privileges`. +- Der SSH-basierte Unraid-Container ist nicht Teil des Standardstarts. +- Ein allgemeiner Host-Shell-MCP wird bewusst nicht angeboten. + +## Start + +```bash +sudo platform/mcp/install-tools.sh ``` -Das jeweilige Profil verweist nur auf die benötigte Datei. Dadurch werden die -Tool-Schemas kleiner, das Kontextfenster bleibt frei und kleine Modelle müssen -weniger Werkzeuge unterscheiden. +Der Grundstart enthält nur Websuche. Bereits konfigurierte Fachbereiche werden +explizit ergänzt: -Credentials werden von schmalen Wrapper-Programmen wie `run-arr-mcp` oder -`runraid` aus geschützten Environment-Dateien geladen. Das JSON selbst enthält -weder Werte noch Pfade zu einzelnen Tokens. +Das Skript erkennt vorhandene Secret-Dateien und aktiviert dadurch automatisch +`homeassistant`, `arr` und `unraid`. Ohne Fach-Secrets startet nur der sichere +Webbereich. -## Schreibzugriff +Für den derzeit migrierten Container kann der Name `Open-WebUI` lauten. Der +Netzwerkbefehl ist idempotent zu behandeln. -Schreibende Server gehören nicht in `mcp-standard.json`. Sie benötigen eine -Vorschau und ein an die exakte Änderung gebundenes Approval Ticket. +Die lokale Installation benötigt die vorhandenen Secret-Dateien: -## Shell +```text +/etc/mike-ai/homeassistant-admin-mcp.env +/etc/mike-ai/arr-mcp.env +/etc/mike-ai/runraid/.env +``` -Ein allgemeiner Shell-MCP ist nicht Teil der Zielplattform. Insbesondere -`python3`, `ssh`, `scp`, `curl` und `systemctl` dürfen nicht gemeinsam als -scheinbar harmlose Allowlist angeboten werden. +Die erweiterte Unraid-Diagnose benötigt zusätzlich die Konfigurationsdatei, +den eingeschränkten Schlüssel und die bekannte Hostsignatur. Sie wird nur mit +`--profile extended` gestartet. + +TinySearch speichert sein lokales Embedding-Modell in einem Docker-Volume. +Nach einer Erstinstallation wird das Modell einmalig im Container mit +`tinysearch setup` geladen. Das Volume bleibt bei Containerupdates erhalten. + +## Client-Auswahl + +Werkzeuge werden nicht pauschal an jedes Modell gehängt. Für Home-Assistant- +Fragen wird HA ausgewählt, für Medien ARR, für Recherche Web und für die NAS +Unraid. Mehrere Werkzeuge werden nur aktiviert, wenn die Aufgabe tatsächlich +mehrere Bereiche verbindet. + +Schreibende Aktionen bleiben hinter der jeweiligen serverseitigen Policy und +einem Vorschau-/Bestätigungsablauf. Ein Client-Schalter allein darf niemals +eine read-only Policy aufheben. diff --git a/platform/mcp/compose.yaml b/platform/mcp/compose.yaml new file mode 100644 index 0000000..c8e49b0 --- /dev/null +++ b/platform/mcp/compose.yaml @@ -0,0 +1,147 @@ +name: mike-ai-tools + +x-tool-common: &tool-common + restart: unless-stopped + read_only: true + tmpfs: + - /tmp:rw,noexec,nosuid,nodev,size=64m + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + networks: [tools] + logging: + options: + max-size: 10m + max-file: "3" + +services: + mcp-web: + <<: *tool-common + build: + context: .. + dockerfile: mcp/Dockerfile.web + image: mike-ai/mcp-web:local + container_name: mike-ai-mcp-web + environment: + TINYSEARCH_MCP_URL: http://tinysearch:8000/mcp + SEARXNG_URL: http://searxng:8080 + WEB_SEARCH_BUDGET_MAX_RELATED: "6" + depends_on: + tinysearch: + condition: service_started + + searxng: + <<: *tool-common + image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60 + container_name: mike-ai-tools-searxng + volumes: + - ${SEARXNG_SETTINGS_FILE:-../web-search/searxng-settings.example.yml}:/etc/searxng/settings.yml:ro + networks: [tools, egress] + + tinysearch: + <<: *tool-common + image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c + container_name: mike-ai-tools-tinysearch + shm_size: 1gb + volumes: + - tinysearch-models:/data/models + - ../web-search/tinysearch_config.json:/config/tinysearch_config.json:ro + environment: + MCP_TRANSPORT: streamable-http + MCP_HOST: 0.0.0.0 + MCP_PORT: "8000" + TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json + TINYSEARCH_SEARCH_BACKEND: searxng + SEARXNG_URL: http://searxng:8080/search + depends_on: [searxng] + cap_add: [SETUID, SETGID, CHOWN] + networks: [tools, egress] + # The image's built-in `tinysearch doctor` also requires a writable + # configuration directory, although normal server operation does not. + # Check the service socket instead so read-only hardening remains intact. + healthcheck: + test: ["CMD", "python", "-c", "import socket; s=socket.create_connection(('127.0.0.1', 8000), 2); s.close()"] + interval: 30s + timeout: 5s + retries: 5 + start_period: 20s + + mcp-homeassistant: + <<: *tool-common + build: + context: ../.. + dockerfile: platform/mcp/Dockerfile.homeassistant-relay + image: mike-ai/mcp-homeassistant-relay:local + container_name: mike-ai-mcp-homeassistant + profiles: [homeassistant] + volumes: + - ${HA_ENV_FILE:-/etc/mike-ai/homeassistant-admin-mcp.env}:/run/secrets/homeassistant.env:ro + cap_add: [CHOWN, SETUID, SETGID] + networks: [tools, egress] + + mcp-arr: + <<: *tool-common + build: + context: . + dockerfile: Dockerfile.arr + image: mike-ai/mcp-arr:1.0.1-patched + container_name: mike-ai-mcp-arr + profiles: [arr] + env_file: + - ${ARR_ENV_FILE:-/etc/mike-ai/arr-mcp.env} + volumes: + # The local fork adds bounded read-only Sonarr pseudo-actions. Keep the + # patch explicit until upstream publishes a self-contained 2.x image. + - ${ARR_SONARR_PATCH:-./patches/mcp_sonarr.py}:/usr/local/lib/python3.13/site-packages/arr_mcp/mcp/mcp_sonarr.py:ro + networks: [tools, egress] + + mcp-unraid-official: + <<: *tool-common + image: debian:13-slim + container_name: mike-ai-mcp-unraid-official + profiles: [unraid] + env_file: + - ${RUNRAID_ENV_FILE:-/etc/mike-ai/runraid/.env} + environment: + UNRAID_RMCP_HOST: 0.0.0.0 + UNRAID_RMCP_PORT: "8000" + UNRAID_RMCP_DISABLE_HTTP_AUTH: "true" + UNRAID_NOAUTH: "true" + UNRAID_RMCP_ALLOWED_HOSTS: "mike-ai-mcp-unraid-official:8000,mike-ai-mcp-unraid-official,localhost:8000,127.0.0.1:8000" + volumes: + - ${RUNRAID_BINARY:-/usr/local/bin/runraid}:/usr/local/bin/unraid:ro + entrypoint: ["/usr/local/bin/unraid"] + command: ["serve"] + networks: [tools, egress] + + mcp-unraid-ssh: + <<: *tool-common + profiles: [extended] + build: + context: . + dockerfile: Dockerfile.unraid-ssh + image: mike-ai/mcp-unraid-ssh:local + container_name: mike-ai-mcp-unraid-ssh + environment: + UNRAID_MCP_CONFIG: /run/config/unraid-mcp.json + volumes: + - ${UNRAID_MCP_SOURCE:-/opt/mike-ai/unraid-agent/unraid_mcp.py}:/app/unraid_mcp.py:ro + - ${UNRAID_MCP_CONFIG:-/etc/mike-ai/unraid-mcp.json}:/run/config/unraid-mcp.json:ro + - ${UNRAID_SSH_KEY:-/etc/mike-ai/keys/unraid_root}:/etc/mike-ai/keys/unraid_root:ro + - ${UNRAID_KNOWN_HOSTS:-/etc/mike-ai/ssh/known_hosts_unraid_ai}:/etc/mike-ai/ssh/known_hosts_unraid_ai:ro + - unraid-audit:/var/log/mike-ai + networks: [tools, egress] + +networks: + tools: + name: mike-ai-tools + internal: true + ipam: + config: [{subnet: 172.30.40.0/24}] + egress: + name: mike-ai-tools-egress + ipam: + config: [{subnet: 172.30.50.0/24}] + +volumes: + tinysearch-models: + unraid-audit: diff --git a/platform/mcp/ha-relay-entrypoint.sh b/platform/mcp/ha-relay-entrypoint.sh new file mode 100644 index 0000000..6eb5bc4 --- /dev/null +++ b/platform/mcp/ha-relay-entrypoint.sh @@ -0,0 +1,23 @@ +#!/bin/sh +set -eu + +config=/run/secrets/homeassistant.env +if [ ! -r "$config" ]; then + echo "Home Assistant secret file is missing" >&2 + exit 1 +fi +set -a +. "$config" +set +a +: "${HASS_URL:?HASS_URL is required}" +: "${HASS_TOKEN:?HASS_TOKEN is required}" + +upstream=${HASS_URL%/} +escaped_token=$(printf '%s' "$HASS_TOKEN" | sed 's/[&/]/\\&/g') +escaped_upstream=$(printf '%s' "$upstream" | sed 's/[&/]/\\&/g') +sed -e "s/__HASS_TOKEN__/$escaped_token/g" \ + -e "s/__HASS_UPSTREAM__/$escaped_upstream/g" \ + /etc/nginx/templates/homeassistant.conf.template \ + > /tmp/nginx.conf +unset HASS_TOKEN +exec nginx -c /tmp/nginx.conf -g 'daemon off;' diff --git a/platform/mcp/homeassistant.conf.template b/platform/mcp/homeassistant.conf.template new file mode 100644 index 0000000..8d2047a --- /dev/null +++ b/platform/mcp/homeassistant.conf.template @@ -0,0 +1,30 @@ +worker_processes 1; +pid /tmp/nginx.pid; +error_log /dev/stderr warn; + +events { worker_connections 128; } + +http { + access_log /dev/stdout; + client_body_temp_path /tmp/client_temp; + proxy_temp_path /tmp/proxy_temp; + fastcgi_temp_path /tmp/fastcgi_temp; + uwsgi_temp_path /tmp/uwsgi_temp; + scgi_temp_path /tmp/scgi_temp; + proxy_buffering off; + proxy_read_timeout 600s; + proxy_send_timeout 600s; + + server { + listen 8000; + location /mcp { + proxy_pass __HASS_UPSTREAM__/api/hass_mcp; + proxy_http_version 1.1; + proxy_ssl_server_name on; + proxy_ssl_name $proxy_host; + proxy_set_header Authorization "Bearer __HASS_TOKEN__"; + proxy_set_header Host $proxy_host; + proxy_set_header Connection ""; + } + } +} diff --git a/platform/mcp/install-tools.sh b/platform/mcp/install-tools.sh new file mode 100755 index 0000000..d4f3443 --- /dev/null +++ b/platform/mcp/install-tools.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +set -Eeuo pipefail + +[[ $EUID -eq 0 ]] || { echo "Bitte als root ausführen." >&2; exit 1; } + +MCP_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +COMPOSE=(docker compose -f "$MCP_DIR/compose.yaml") +export SEARXNG_SETTINGS_FILE="${SEARXNG_SETTINGS_FILE:-$MCP_DIR/../web-search/searxng-settings.yml}" + +[[ -s $SEARXNG_SETTINGS_FILE ]] || { + echo "SearXNG-Konfiguration fehlt: $SEARXNG_SETTINGS_FILE" >&2 + exit 1 +} + +profiles=() +if [[ -s /etc/mike-ai/homeassistant-admin-mcp.env ]]; then + profiles+=(--profile homeassistant) +else + echo "Home Assistant bleibt aus: Secret-Datei fehlt." +fi +if [[ -s /etc/mike-ai/arr-mcp.env ]]; then + profiles+=(--profile arr) +else + echo "ARR bleibt aus: Secret-Datei fehlt." +fi +if [[ -s /etc/mike-ai/runraid/.env && -x /usr/local/bin/runraid ]]; then + profiles+=(--profile unraid) +else + echo "Unraid bleibt aus: runraid 0.4.2 oder Secret-Datei fehlt." +fi + +# TinySearch keeps the embedding bundle outside the container. Download it +# once on a fresh host; subsequent rebuilds reuse the named volume. +docker volume create mike-ai-tools_tinysearch-models >/dev/null +tiny_image="marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c" +if ! docker run --rm --entrypoint test \ + -v mike-ai-tools_tinysearch-models:/data/models "$tiny_image" \ + -f /data/models/all-minilm-l6-v2-onnx/model.onnx; then + echo "TinySearch-Modell wird einmalig geladen." + docker run --rm -v mike-ai-tools_tinysearch-models:/data/models \ + "$tiny_image" tinysearch setup +fi + +"${COMPOSE[@]}" "${profiles[@]}" up -d --build + +for webui in mike-ai-open-webui Open-WebUI; do + if docker container inspect "$webui" >/dev/null 2>&1; then + docker network connect mike-ai-tools "$webui" 2>/dev/null || true + fi +done + +"${COMPOSE[@]}" "${profiles[@]}" ps diff --git a/platform/mcp/mcp-servers.example.json b/platform/mcp/mcp-servers.example.json deleted file mode 100644 index 0545fa3..0000000 --- a/platform/mcp/mcp-servers.example.json +++ /dev/null @@ -1,23 +0,0 @@ -{ - "mcpServers": { - "web": { - "command": "/usr/bin/python3", - "args": ["/opt/mike-ai/web-search/web_search_mcp.py"] - }, - "homeassistant": { - "command": "/usr/local/bin/homeassistant-native-mcp", - "args": [], - "timeout_ms": 60000 - }, - "arr": { - "command": "/usr/local/bin/run-arr-mcp", - "args": [], - "timeout_ms": 30000 - }, - "unraid-readonly": { - "command": "/usr/local/bin/runraid", - "args": ["mcp"], - "timeout_ms": 60000 - } - } -} diff --git a/platform/mcp/patches/mcp_sonarr.py b/platform/mcp/patches/mcp_sonarr.py new file mode 100644 index 0000000..3ce7967 --- /dev/null +++ b/platform/mcp/patches/mcp_sonarr.py @@ -0,0 +1,329 @@ +"""Sonarr condensed action-routed MCP tool. + +CONCEPT:ECO-4.82 — gitlab-style organized per-service tool surface. +""" + +import os +import json +import re +from typing import Any + +from agent_utilities.mcp_utilities import dispatch, run_blocking +from fastmcp import FastMCP +from pydantic import Field + +from arr_mcp.auth import get_sonarr_client + + +READ_ONLY_ACTIONS = frozenset( + { + "get_system_status", "get_health", "get_diskspace", "get_ping", + "get_series", "get_series_id", "get_series_lookup", "lookup_series", + "get_episode", "get_episode_id", "get_episodefile", "get_episodefile_id", + "get_calendar", "get_calendar_id", "get_history", "get_history_series", + "get_history_since", "get_queue", "get_queue_details", "get_queue_status", + "get_wanted_missing", "get_wanted_missing_id", "get_wanted_cutoff", + "get_wanted_cutoff_id", "get_qualityprofile", "get_qualityprofile_id", + "get_languageprofile", "get_languageprofile_id", "get_tag", "get_tag_id", + "get_tag_detail", "get_tag_detail_id", "get_command", "get_command_id", + "get_release", + } +) + +PSEUDO_ACTIONS = frozenset({"find_series", "get_season_summary", "search_releases"}) + +WRITE_ACTIONS = frozenset({ + "post_command", + "post_release", + "put_episode_id", + "put_episode_monitor", + "put_series_id", + "put_series", + "put_wanted", + "post_wanted", +}) +MAX_COLLECTION_ITEMS = 50 + + +def _plain(value: Any) -> Any: + if hasattr(value, "model_dump") and callable(value.model_dump): + return value.model_dump() + if hasattr(value, "dict") and callable(value.dict): + return value.dict() + if isinstance(value, list): + return [_plain(item) for item in value] + if isinstance(value, dict): + return {str(key): _plain(item) for key, item in value.items()} + return value + + +def _unwrap(value: Any) -> Any: + value = _plain(value) + if isinstance(value, dict) and set(value) == {"result"}: + return value["result"] + return value + + +def _pick(item: dict[str, Any], fields: tuple[str, ...]) -> dict[str, Any]: + return {field: item[field] for field in fields if item.get(field) is not None} + + +def _compact_series(item: dict[str, Any], include_seasons: bool = False) -> dict[str, Any]: + result = _pick( + item, + ("id", "title", "sortTitle", "year", "status", "monitored", "path", "tvdbId"), + ) + statistics = item.get("statistics") or {} + if isinstance(statistics, dict): + result["statistics"] = _pick( + statistics, + ("seasonCount", "episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"), + ) + if include_seasons: + result["seasons"] = [ + { + **_pick(season, ("seasonNumber", "monitored")), + "statistics": _pick( + season.get("statistics") or {}, + ("episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"), + ), + } + for season in item.get("seasons", []) + if isinstance(season, dict) + ] + return result + + +def _compact_episode(item: dict[str, Any]) -> dict[str, Any]: + return _pick( + item, + ("id", "seriesId", "seasonNumber", "episodeNumber", "title", "airDate", "airDateUtc", "monitored", "hasFile", "episodeFileId"), + ) + + +def _compact_file(item: dict[str, Any]) -> dict[str, Any]: + quality = item.get("quality") or {} + quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None + result = _pick( + item, + ("id", "seriesId", "seasonNumber", "relativePath", "path", "size", "dateAdded", "releaseGroup"), + ) + if quality_name: + result["quality"] = quality_name + return result + + +def _compact_release(item: dict[str, Any]) -> dict[str, Any]: + quality = item.get("quality") or {} + quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None + result = _pick( + item, + ( + "guid", "title", "indexer", "indexerId", "size", "age", "ageHours", + "seeders", "leechers", "protocol", "downloadAllowed", "releaseWeight", + ), + ) + if quality_name: + result["quality"] = quality_name + rejections = item.get("rejections") + if isinstance(rejections, list) and rejections: + result["rejections"] = [str(reason)[:180] for reason in rejections[:5]] + return result + + +def _bounded(items: list[Any], compact) -> dict[str, Any]: + total = len(items) + return { + "total": total, + "returned": min(total, MAX_COLLECTION_ITEMS), + "truncated": total > MAX_COLLECTION_ITEMS, + "items": [compact(item) for item in items[:MAX_COLLECTION_ITEMS] if isinstance(item, dict)], + "next_step": ( + "Use find_series or narrower Sonarr parameters; do not repeat the same broad request." + if total > MAX_COLLECTION_ITEMS else None + ), + } + + +def _compact_result(action: str, value: Any) -> Any: + value = _unwrap(value) + if isinstance(value, list): + if action in {"get_series", "get_series_lookup", "lookup_series"}: + return _bounded(value, _compact_series) + if action in {"get_episode", "get_calendar", "get_wanted_missing", "get_wanted_cutoff"}: + return _bounded(value, _compact_episode) + if action == "get_episodefile": + return _bounded(value, _compact_file) + if action == "get_release": + return _bounded(value, _compact_release) + return _bounded(value, lambda item: item) + if isinstance(value, dict) and action in {"get_series_id"}: + return _compact_series(value, include_seasons=True) + if isinstance(value, dict) and action in {"get_episode_id"}: + return _compact_episode(value) + if isinstance(value, dict) and action in {"get_episodefile_id"}: + return _compact_file(value) + return value + + +async def _find_series(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: + query = str(kwargs.get("query", "")).strip() + if len(query) < 2: + raise ValueError("find_series requires params_json with a query of at least 2 characters") + limit = max(1, min(int(kwargs.get("limit", 8)), 15)) + raw = _unwrap(await run_blocking(dispatch, client, "get_series", {}, service="arr-sonarr")) + words = [word for word in re.findall(r"[a-z0-9]+", query.casefold()) if len(word) > 1] + matches = [] + for item in raw if isinstance(raw, list) else []: + haystack = " ".join( + str(item.get(field, "")) for field in ("title", "sortTitle", "originalTitle", "alternateTitles") + ).casefold() + if all(word in haystack for word in words): + matches.append(_compact_series(item, include_seasons=True)) + return { + "query": query, + "matches": matches[:limit], + "match_count": len(matches), + "truncated": len(matches) > limit, + "task_complete": True, + "instruction": "Use the returned series id for details. Do not call get_series for discovery.", + } + + +async def _season_summary(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: + series_id = int(kwargs["series_id"]) + season_number = int(kwargs["season_number"]) + series = _unwrap( + await run_blocking(dispatch, client, "get_series_id", {"id": series_id}, service="arr-sonarr") + ) + episodes = _unwrap( + await run_blocking( + dispatch, + client, + "get_episode", + {"seriesId": series_id, "seasonNumber": season_number}, + service="arr-sonarr", + ) + ) + files = _unwrap( + await run_blocking( + dispatch, + client, + "get_episodefile", + {"seriesId": series_id}, + service="arr-sonarr", + ) + ) + selected_episodes = [ + _compact_episode(item) for item in episodes + if isinstance(item, dict) and item.get("seasonNumber") == season_number + ] if isinstance(episodes, list) else [] + selected_files = [ + _compact_file(item) for item in files + if isinstance(item, dict) and item.get("seasonNumber") == season_number + ] if isinstance(files, list) else [] + groups = sorted({str(item.get("releaseGroup")) for item in selected_files if item.get("releaseGroup")}) + return { + "series": _compact_series(series) if isinstance(series, dict) else {"id": series_id}, + "season_number": season_number, + "episode_count": len(selected_episodes), + "file_count": len(selected_files), + "release_groups": groups, + "episodes": selected_episodes[:30], + "files": selected_files[:30], + "task_complete": True, + "instruction": "This is the complete compact season answer. Do not repeat broad series or episode queries.", + } + + +async def _search_releases(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: + series_id = kwargs.get("series_id") + episode_id = kwargs.get("episode_id") + season_number = kwargs.get("season_number") + release_group = str(kwargs.get("release_group", "")).strip() + if series_id is None and episode_id is None: + raise ValueError("search_releases requires series_id or episode_id") + query: dict[str, Any] = {} + if series_id is not None: + query["seriesId"] = int(series_id) + if episode_id is not None: + query["episodeId"] = int(episode_id) + if season_number is not None: + query["seasonNumber"] = int(season_number) + raw = await run_blocking(dispatch, client, "get_release", query, service="arr-sonarr") + raw = _unwrap(raw) + if release_group and isinstance(raw, list): + needle = release_group.casefold() + raw = [ + item for item in raw + if isinstance(item, dict) + and needle in ( + str(item.get("releaseGroup", "")) + " " + str(item.get("title", "")) + ).casefold() + ] + compact = _compact_result("get_release", raw) + return { + "task_complete": True, + "search_scope": { + "series_id": series_id, + "episode_id": episode_id, + "season_number": season_number, + "release_group_filter": release_group or None, + }, + "monitoring_changed": False, + "download_started": False, + "results": compact, + "instruction": "These are Sonarr indexer results. Do not use web search to replace them. Never download unless the user separately approves a write action.", + } + + +def register_sonarr_tools(mcp: FastMCP) -> None: + @mcp.tool(tags={"sonarr"}) + async def sonarr_action( + action: str = Field( + description="Read-only Sonarr action. Use find_series {query} for titles, get_season_summary {series_id, season_number} for holdings, and search_releases {series_id, season_number, optional release_group} to query configured Sonarr indexers without downloading or changing monitoring. Avoid broad get_series/get_episode calls." + ), + params_json: str = Field( + default="{}", + description="JSON string of parameters to pass to the action.", + ), + ) -> Any: + """Query Sonarr through a server-side allowlist (read-only by default; write actions when ARR_MCP_WRITE=1).""" + if action in {"list_actions", "help", "actions"}: + return { + "service": "sonarr", + "access_mode": "write" if os.environ.get("ARR_MCP_WRITE", "").strip().lower() in ("1", "true", "yes", "on") else "read-only", + "actions": sorted(READ_ONLY_ACTIONS), + "write_actions": sorted(WRITE_ACTIONS) if os.environ.get("ARR_MCP_WRITE", "").strip().lower() in ("1", "true", "yes", "on") else [], + "preferred_compact_actions": sorted(PSEUDO_ACTIONS), + } + allow_write = os.environ.get("ARR_MCP_WRITE", "").strip().lower() in ( + "1", "true", "yes", "on" + ) + if action in READ_ONLY_ACTIONS | PSEUDO_ACTIONS: + pass + elif allow_write and action in WRITE_ACTIONS: + pass + else: + if allow_write: + raise PermissionError( + f"Sonarr MCP write mode is enabled, but action '{action}' " + "is not in the allowed write set. Allowed: " + f"{sorted(WRITE_ACTIONS)}" + ) + raise PermissionError( + f"Sonarr action '{action}' is blocked by the server-side " + "read-only policy. Set ARR_MCP_WRITE=1 to enable write mode." + ) + client = get_sonarr_client() + kwargs = {k: v for k, v in json.loads(params_json).items() if v is not None} + if action == "find_series": + return await _find_series(client, kwargs) + if action == "get_season_summary": + return await _season_summary(client, kwargs) + if action == "search_releases": + return await _search_releases(client, kwargs) + result = await run_blocking( + dispatch, client, action, kwargs, service="arr-sonarr" + ) + return _compact_result(action, result) diff --git a/platform/profiles/profile-fast.conf b/platform/profiles/profile-fast.conf index 48b29ac..d0d21ef 100644 --- a/platform/profiles/profile-fast.conf +++ b/platform/profiles/profile-fast.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Fast 76.8K MTP2 with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/platform/profiles/profile-long.conf b/platform/profiles/profile-long.conf index 20b3ee2..46d4fac 100644 --- a/platform/profiles/profile-long.conf +++ b/platform/profiles/profile-long.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0 diff --git a/platform/profiles/profile-medium.conf b/platform/profiles/profile-medium.conf index b43d02b..37cac76 100644 --- a/platform/profiles/profile-medium.conf +++ b/platform/profiles/profile-medium.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none diff --git a/platform/systemd/mike-ai-web-search.service b/platform/systemd/mike-ai-web-search.service deleted file mode 100644 index 81ed583..0000000 --- a/platform/systemd/mike-ai-web-search.service +++ /dev/null @@ -1,16 +0,0 @@ -[Unit] -Description=Local AI Web Search (TinySearch + SearXNG) -Requires=docker.service -After=docker.service network-online.target -Before=mike-ai-llama-ui.service - -[Service] -Type=oneshot -RemainAfterExit=yes -WorkingDirectory=/opt/mike-ai/web-search -ExecStart=/usr/bin/docker compose up -d -ExecStop=/usr/bin/docker compose down -TimeoutStartSec=180 - -[Install] -WantedBy=multi-user.target diff --git a/platform/web-search/compose.yaml b/platform/web-search/compose.yaml deleted file mode 100644 index 2af64c7..0000000 --- a/platform/web-search/compose.yaml +++ /dev/null @@ -1,45 +0,0 @@ -name: local-ai-web-search - -services: - searxng: - image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60 - restart: unless-stopped - volumes: - - ./searxng-settings.yml:/etc/searxng/settings.yml:ro - networks: [search] - healthcheck: - test: ["CMD", "wget", "-q", "--spider", "http://127.0.0.1:8080/healthz"] - interval: 30s - timeout: 10s - retries: 5 - start_period: 30s - security_opt: ["no-new-privileges:true"] - - tinysearch: - image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c - restart: unless-stopped - ports: - - "127.0.0.1:8000:8000" - shm_size: "1gb" - volumes: - - tinysearch-models:/data/models - - ./tinysearch_config.json:/config/tinysearch_config.json:ro - environment: - MCP_TRANSPORT: streamable-http - MCP_HOST: 0.0.0.0 - MCP_PORT: 8000 - TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json - TINYSEARCH_SEARCH_BACKEND: searxng - SEARXNG_URL: http://searxng:8080/search - depends_on: - searxng: - condition: service_healthy - networks: [search] - security_opt: ["no-new-privileges:true"] - -networks: - search: - driver: bridge - -volumes: - tinysearch-models: diff --git a/platform/web-search/web_search_mcp.py b/platform/web-search/web_search_mcp.py index 716ea35..1aedd82 100644 --- a/platform/web-search/web_search_mcp.py +++ b/platform/web-search/web_search_mcp.py @@ -10,12 +10,11 @@ from facts verified by crawled pages or primary APIs. from __future__ import annotations import ipaddress +import asyncio import html import json import os import re -import select -import subprocess import sys import threading import time @@ -28,9 +27,9 @@ from xml.etree import ElementTree SERVER_VERSION = "2.1.0" -TINYSEARCH_CONTAINER = os.environ.get( - "TINYSEARCH_CONTAINER", "mike-ai-web-search-tinysearch-1" -) +TINYSEARCH_MCP_URL = os.environ.get( + "TINYSEARCH_MCP_URL", "http://tinysearch:8000/mcp" +).rstrip("/") SEARXNG_URL = os.environ.get("SEARXNG_URL", "").rstrip("/") CHILD_TIMEOUT_SECONDS = float(os.environ.get("TINYSEARCH_CHILD_TIMEOUT", "110")) HTTP_TIMEOUT_SECONDS = float(os.environ.get("WEB_API_TIMEOUT", "18")) @@ -801,90 +800,46 @@ def general_discovery(query: str, limit: int) -> tuple[list[dict[str, Any]], lis class TinySearchClient: - """Minimal synchronous MCP client for TinySearch's stdio server.""" + """Synchronous facade for TinySearch's Streamable HTTP MCP endpoint. + + TinySearch used to be reached by executing ``docker exec`` on the host. + That required access to the Docker socket and coupled this service to a + particular container name. The containerized tool layer instead talks to + TinySearch over the private Docker network. + """ def __init__(self) -> None: - self._next_id = 1 self._lock = threading.Lock() - self._process = subprocess.Popen( - [ - "/usr/bin/docker", - "exec", - "-e", - "MCP_TRANSPORT=stdio", - "-i", - TINYSEARCH_CONTAINER, - "tinysearch", - "mcp", - ], - stdin=subprocess.PIPE, - stdout=subprocess.PIPE, - stderr=subprocess.DEVNULL, - text=True, - encoding="utf-8", - errors="replace", - bufsize=1, - ) - self._request( - "initialize", - { - "protocolVersion": "2024-11-05", - "capabilities": {}, - "clientInfo": {"name": "mike-ai-web-facade", "version": SERVER_VERSION}, - }, - ) - self._notify("notifications/initialized", {}) - def _notify(self, method: str, params: dict[str, Any]) -> None: - assert self._process.stdin is not None - message = {"jsonrpc": "2.0", "method": method, "params": params} - self._process.stdin.write(json.dumps(message, separators=(",", ":")) + "\n") - self._process.stdin.flush() + async def _call_async(self, name: str, arguments: dict[str, Any]) -> str: + # Imported lazily so the dependency-free stdio implementation still + # gives a useful startup error outside its production container. + from mcp import ClientSession + from mcp.client.streamable_http import streamablehttp_client - def _request(self, method: str, params: dict[str, Any]) -> dict[str, Any]: - with self._lock: - if self._process.poll() is not None: - raise RuntimeError("TinySearch child process is not running") - request_id = self._next_id - self._next_id += 1 - assert self._process.stdin is not None - assert self._process.stdout is not None - message = { - "jsonrpc": "2.0", - "id": request_id, - "method": method, - "params": params, - } - self._process.stdin.write(json.dumps(message, separators=(",", ":")) + "\n") - self._process.stdin.flush() - deadline = datetime.now().timestamp() + CHILD_TIMEOUT_SECONDS - while True: - remaining = deadline - datetime.now().timestamp() - if remaining <= 0: - raise TimeoutError(f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s") - ready, _, _ = select.select([self._process.stdout], [], [], remaining) - if not ready: - raise TimeoutError(f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s") - line = self._process.stdout.readline() - if not line: - raise RuntimeError("TinySearch closed its output stream") - payload = json.loads(line) - if payload.get("id") != request_id: - continue - if "error" in payload: - raise RuntimeError(f"TinySearch error: {payload['error']}") - return payload.get("result") or {} + async with streamablehttp_client( + TINYSEARCH_MCP_URL, + timeout=CHILD_TIMEOUT_SECONDS, + sse_read_timeout=CHILD_TIMEOUT_SECONDS, + ) as (read_stream, write_stream, _): + async with ClientSession(read_stream, write_stream) as session: + await session.initialize() + result = await session.call_tool(name, arguments) + if result.isError: + raise RuntimeError(clean_text(str(result.content))) + return "\n".join( + item.text for item in result.content + if getattr(item, "type", None) == "text" + ) def call(self, name: str, arguments: dict[str, Any]) -> str: - result = self._request("tools/call", {"name": name, "arguments": arguments}) - if result.get("isError"): - raise RuntimeError(clean_text(str(result.get("content")))) - texts = [ - item.get("text", "") - for item in result.get("content", []) - if isinstance(item, dict) and item.get("type") == "text" - ] - return "\n".join(texts) + with self._lock: + try: + return asyncio.run(asyncio.wait_for( + self._call_async(name, arguments), CHILD_TIMEOUT_SECONDS)) + except TimeoutError as exc: + raise TimeoutError( + f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s") from exc _client: TinySearchClient | None = None