Containerize MCP tool services

This commit is contained in:
Mikei386
2026-08-20 23:54:02 +02:00
parent 3ab9628088
commit f0d552ef58
28 changed files with 858 additions and 305 deletions
+8 -6
View File
@@ -40,14 +40,16 @@ Neustart an; danach wird derselbe Befehl erneut ausgeführt.
|---|---|---| |---|---|---|
| Open WebUI | `<WG-IP>:8080` | Chat und Administration | | Open WebUI | `<WG-IP>:8080` | Chat und Administration |
| Profile Router | `<WG-IP>:8081` | OpenAI-kompatible API, Profilwahl | | Profile Router | `<WG-IP>:8081` | OpenAI-kompatible API, Profilwahl |
| llama.cpp | nur Docker-intern | Inferenz, Vision, MCP | | llama.cpp | nur Docker-intern | Inferenz und integrierte Vision |
| Profile Controller | nur Docker-intern | eng begrenzter Containerwechsel | | Profile Controller | nur Docker-intern | eng begrenzter Containerwechsel |
| SearXNG | nur Docker-intern | Websuche | | MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid |
Bildgenerierung, TTS/STT sowie Home-Assistant-, ARR- und Unraid-MCPs werden Bildgenerierung und TTS/STT bleiben optionale Dienste. Web-, Home-Assistant-,
bewusst nicht automatisch aktiviert. Sie erhalten später eigene Container und ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter
kleinstmögliche Rechte. Die Bildanalyse ist bereits Bestandteil des `platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne
multimodalen Qwen-Modells. `mike-ai-tools`-Netz; llama.cpp erhält keine MCP-Konfiguration und keine
Infrastruktur-Secrets. Die Bildanalyse ist Bestandteil des multimodalen
Qwen-Modells.
## Dokumentation ## Dokumentation
+8 -30
View File
@@ -11,15 +11,12 @@ x-llama-common: &llama-common
- /tmp:size=1g,mode=1777 - /tmp:size=1g,mode=1777
volumes: volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro" - "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
- ./platform/docker/mcp-standard.json:/etc/mike-ai/mcp-standard.json:ro
environment: environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_DRIVER_CAPABILITIES: compute,utility NVIDIA_DRIVER_CAPABILITIES: compute,utility
dns: ["${AI_DNS:-1.1.1.1}"] dns: ["${AI_DNS:-1.1.1.1}"]
networks: networks:
inference: inference:
aliases: [llama-upstream] aliases: [llama-upstream]
search: {}
security_opt: ["no-new-privileges:true"] security_opt: ["no-new-privileges:true"]
cap_drop: [ALL] cap_drop: [ALL]
healthcheck: healthcheck:
@@ -36,7 +33,6 @@ services:
labels: labels:
com.mike-ai.llama-profile: fast com.mike-ai.llama-profile: fast
environment: environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0} NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility NVIDIA_DRIVER_CAPABILITIES: compute,utility
command: command:
@@ -85,8 +81,6 @@ services:
- "0.8" - "0.8"
- --top-k - --top-k
- "20" - "20"
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device - --device
- CUDA0 - CUDA0
- --split-mode - --split-mode
@@ -108,7 +102,6 @@ services:
labels: labels:
com.mike-ai.llama-profile: medium com.mike-ai.llama-profile: medium
environment: environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0} NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility NVIDIA_DRIVER_CAPABILITIES: compute,utility
command: command:
@@ -157,8 +150,6 @@ services:
- "0.8" - "0.8"
- --top-k - --top-k
- "20" - "20"
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device - --device
- CUDA0 - CUDA0
- --split-mode - --split-mode
@@ -170,7 +161,6 @@ services:
labels: labels:
com.mike-ai.llama-profile: long com.mike-ai.llama-profile: long
environment: environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0} NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility NVIDIA_DRIVER_CAPABILITIES: compute,utility
command: command:
@@ -221,8 +211,6 @@ services:
- "0.8" - "0.8"
- --top-k - --top-k
- "20" - "20"
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device - --device
- CUDA0 - CUDA0
- --split-mode - --split-mode
@@ -276,8 +264,6 @@ services:
- all - all
- --no-mmap - --no-mmap
- --no-ui - --no-ui
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device - --device
- CUDA0 - CUDA0
- --split-mode - --split-mode
@@ -355,26 +341,19 @@ services:
OPENAI_API_BASE_URLS: http://router:8081/v1 OPENAI_API_BASE_URLS: http://router:8081/v1
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false} ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
# Seed native MCP connections on a fresh Open WebUI database. Secrets
# stay inside the tool containers, so these internal URLs need no keys.
TOOL_SERVER_CONNECTIONS: >-
[{"url":"http://mike-ai-mcp-web:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"web-local","name":"Web (lokal)","description":"Kompakte Websuche und Quellenvergleich"}},{"url":"http://mike-ai-mcp-homeassistant:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"homeassistant-local","name":"Home Assistant (lokal)","description":"Home-Assistant-Werkzeuge mit serverseitigem Token"}},{"url":"http://mike-ai-mcp-arr:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"arr-local","name":"ARR (lokal)","description":"Sonarr- und Radarr-Werkzeuge"}},{"url":"http://mike-ai-mcp-unraid-official:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"unraid-readonly-local","name":"Unraid (lokal, read-only)","description":"Begrenzte Unraid-Diagnose"}}]
DO_NOT_TRACK: "true" DO_NOT_TRACK: "true"
SCARF_NO_ANALYTICS: "true" SCARF_NO_ANALYTICS: "true"
ports: ports:
- "${AI_BIND_ADDRESS:-127.0.0.1}:8080:8080" - "${AI_BIND_ADDRESS:-127.0.0.1}:8080:8080"
dns: ["${AI_DNS:-1.1.1.1}"] dns: ["${AI_DNS:-1.1.1.1}"]
networks: [frontend] networks: [frontend, tools]
depends_on: [router] depends_on: [router]
security_opt: ["no-new-privileges:true"] security_opt: ["no-new-privileges:true"]
searxng:
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
container_name: mike-ai-searxng
restart: unless-stopped
volumes:
- ./platform/web-search/searxng-settings.yml:/etc/searxng/settings.yml:ro
networks: [search]
dns: ["${AI_DNS:-1.1.1.1}"]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
networks: networks:
frontend: frontend:
internal: false internal: false
@@ -388,10 +367,9 @@ networks:
internal: true internal: true
ipam: ipam:
config: [{subnet: 172.30.30.0/24}] config: [{subnet: 172.30.30.0/24}]
search: tools:
internal: false external: true
ipam: name: mike-ai-tools
config: [{subnet: 172.30.40.0/24}]
volumes: volumes:
open-webui-data: open-webui-data:
+27 -21
View File
@@ -22,7 +22,11 @@ Heimnetz / VPN-Clients
+-- llama-medium > exakt einer aktiv +-- llama-medium > exakt einer aktiv
+-- llama-long --/ +-- llama-long --/
+-- llama-experimental +-- llama-experimental
+-- SearXNG + Web-MCP +-- internes MCP-Netz
+-- Web-MCP + TinySearch + SearXNG
+-- Home-Assistant-MCP-Relay
+-- ARR-MCP
+-- Unraid-MCP
``` ```
## Container und Vertrauensgrenzen ## Container und Vertrauensgrenzen
@@ -33,7 +37,8 @@ Heimnetz / VPN-Clients
| Profile Router | nur WireGuard, Port 8081 | OpenAI-API und Profilwahl | | Profile Router | nur WireGuard, Port 8081 | OpenAI-API und Profilwahl |
| Profile Controller | nein | startet ausschließlich vier bekannte Profile | | Profile Controller | nein | startet ausschließlich vier bekannte Profile |
| llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision | | llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision |
| SearXNG | nein | Websuche für den lokalen Web-MCP | | MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche |
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
Nur der Profile Controller erhält den Docker-Socket. Der Router erhält weder Nur der Profile Controller erhält den Docker-Socket. Der Router erhält weder
Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `long` Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `long`
@@ -73,35 +78,36 @@ nicht automatisch in die Produktionsprofile aufgenommen.
Internetzugang der KI über zuhause laufen, braucht der Heim-Peer zusätzlich Internetzugang der KI über zuhause laufen, braucht der Heim-Peer zusätzlich
IP-Forwarding und NAT ins Heim-WAN. IP-Forwarding und NAT ins Heim-WAN.
## Nicht automatisch installiert ## Optionale Erweiterungen
Bildgenerierung, Whisper, TTS sowie Home-Assistant-, ARR- und Unraid-MCPs sind Bildgenerierung, Whisper und TTS benötigen eigene Modelle und bleiben im
Erweiterungen. Sie benötigen eigene Modelle, Rechte oder Secrets und bleiben Basissystem deaktiviert. Home Assistant, ARR und Unraid sind vorbereitete
im sauberen Basissystem deaktiviert. Multimodale Bildanalyse erfolgt direkt MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only
über Qwen plus Projektor. Nicht installierte Worker-Endpunkte antworten klar Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen
mit `feature_disabled`, statt alte systemd-Pfade aufzurufen. plus Projektor. Nicht installierte Worker-Endpunkte antworten klar mit
`feature_disabled`, statt alte systemd-Pfade aufzurufen.
## Zentrale MCP-Werkzeugebene ## Zentrale MCP-Werkzeugebene
Werkzeuge werden nicht fest in Open WebUI, Hermes oder einen anderen Client Werkzeuge werden nicht in llama.cpp eingebaut. Sie laufen als eigene,
eingebaut. Sie laufen als zentrale, über WireGuard erreichbare MCP-Server. Alle zentrale MCP-Container. Open WebUI greift intern darauf zu. Für externe Clients
MCP-fähigen Oberflächen verwenden dadurch dieselben geprüften Werkzeuge, ohne wie Hermes wird später ein authentifizierter MCP-Gateway über WireGuard
Secrets oder Installationen zu duplizieren. vorgeschaltet; die unauthentifizierten internen Ports werden niemals direkt
veröffentlicht. So können alle Oberflächen dieselben geprüften Werkzeuge
verwenden, ohne Secrets zu duplizieren.
Die Trenneinheit ist **ein Container pro Fachbereich und Vertrauensstufe** – Die Trenneinheit ist **ein Container pro Fachbereich und Vertrauensstufe** –
nicht ein Container pro einzelner Funktion und nicht ein gemeinsamer nicht ein Container pro einzelner Funktion und nicht ein gemeinsamer
Allzweck-MCP mit sämtlichen Zugangsdaten. Allzweck-MCP mit sämtlichen Zugangsdaten.
```text ```text
Open WebUI ──┐ Open WebUI ── internes Netz ───────────┬── web-mcp
Hermes Agent ├── mcp-gateway ──┬── web-mcp ├── home-assistant-mcp
weitere MCP- ┘ ├── home-assistant-mcp-read ├── arr-mcp
Clients ├── home-assistant-mcp-write └── unraid-mcp-read
├── arr-mcp-read
├── arr-mcp-write Hermes Agent ─ WireGuard ─┐
├── unraid-mcp-read weitere MCP-Clients ──────┴── mcp-gateway (später) ── dasselbe interne Netz
├── unraid-mcp-admin
└── sandbox-mcp
``` ```
| Container | Werkzeugbereich | Standardrecht | | Container | Werkzeugbereich | Standardrecht |
+10 -8
View File
@@ -5,17 +5,19 @@
| AI Profile Router | `router/` | vollständig | Kern | | AI Profile Router | `router/` | vollständig | Kern |
| llama.cpp | ggml-org/llama.cpp, festgeschriebener Commit | Buildskript und Commit | Kern | | llama.cpp | ggml-org/llama.cpp, festgeschriebener Commit | Buildskript und Commit | Kern |
| Qwen-Profile | `platform/profiles/` | vollständig, Modelle ausgenommen | Kern | | Qwen-Profile | `platform/profiles/` | vollständig, Modelle ausgenommen | Kern |
| Websuche | TinySearch + SearXNG | Compose und sichere Grundkonfiguration | Kern | | MCP-Tool-Stack | `platform/mcp/compose.yaml` | vollständig | Kern |
| Web-MCP-Fassade | `platform/web-search/web_search_mcp.py` | vollständig | Kern | | Websuche | TinySearch + SearXNG | intern, ohne veröffentlichten Port | Kern |
| Home-Assistant-MCP | separates privates Repository | nur Integration dokumentiert | optional | | Web-MCP-Fassade | `platform/web-search/web_search_mcp.py` | eigener Container | Kern |
| ARR-MCP | separates privates Repository | nur Integration dokumentiert | optional | | Home-Assistant-MCP | HA-Endpunkt plus lokaler Relay | eigener optionaler Container | optional |
| Unraid-MCP | separates Repository/Installation | read-only Integration dokumentiert | optional | | ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional |
| Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional |
| Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional | | Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional |
| XTTS-v2 | Coqui | Worker, Service und Lockdatei | optional | | XTTS-v2 | Coqui | Worker, Service und Lockdatei | optional |
| FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional | | FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional |
| LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional | | LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional |
| Glances | Distribution | nur Betriebsrolle dokumentiert | optional | | Glances | Distribution | nur Betriebsrolle dokumentiert | optional |
Separate MCP-Repositories werden nicht in dieses Repository kopiert. Ihre Upstream-Komponenten werden nicht ungeprüft einkopiert. Images, Python-Pakete
Versionen sollen künftig in einem Release-Manifest referenziert werden. So und lokale Patches sind in Dockerfiles, Compose-Mounts und Dokumentation
bleiben Zuständigkeiten klar und Updates können unabhängig getestet werden. explizit benannt. So bleiben Zuständigkeiten klar und Updates können
unabhängig getestet werden.
+1 -1
View File
@@ -31,7 +31,7 @@ Zielplattform.
| Hauptdienst | `mike-ai-llama-ui.service` | | Hauptdienst | `mike-ai-llama-ui.service` |
| llama.cpp-Port | 8080, auf dem alten Host noch im LAN gebunden | | llama.cpp-Port | 8080, auf dem alten Host noch im LAN gebunden |
| Client-Port | 8081 über den Router | | Client-Port | 8081 über den Router |
| MCP-Konfiguration | `/etc/mike-ai/mcp-servers.json` | | MCP-Konfiguration | getrennte Container unter `/opt/mike-ai/mcp-containers` |
### Aktives Fast-Profil ### Aktives Fast-Profil
+27
View File
@@ -74,6 +74,33 @@ Zusätzlich prüfen: Uni-LAN sieht keine KI-Ports; Heimnetz erreicht beide;
gestopptes WireGuard lässt KI-Container nicht ins Internet; jeder Profilwechsel gestopptes WireGuard lässt KI-Container nicht ins Internet; jeder Profilwechsel
startet exakt einen llama-Container; Text, Tool Call und Bild funktionieren. startet exakt einen llama-Container; Text, Tool Call und Bild funktionieren.
## Werkzeug-Container
Der Installer startet Websuche automatisch in einem privaten Docker-Netz.
Weitere Bereiche werden nur aktiviert, wenn ihre root-only Konfiguration schon
vorhanden ist:
```text
/etc/mike-ai/homeassistant-admin-mcp.env
/etc/mike-ai/arr-mcp.env
/etc/mike-ai/runraid/.env
/usr/local/bin/runraid Version 0.4.2
```
Nach dem Nachreichen einer Datei genügt:
```bash
sudo /opt/mike-ai/stack/platform/mcp/install-tools.sh
```
Auf einer frischen Open-WebUI-Datenbank werden die internen MCP-Adressen über
`TOOL_SERVER_CONNECTIONS` vorbelegt. Bei einer übernommenen Datenbank müssen
die Einträge einmal unter **Admin-Einstellungen → Externe Werkzeuge** geprüft
oder importiert werden. Die Endpunkte stehen in `platform/mcp/README.md`.
Kein MCP-Port wird auf dem Host veröffentlicht. Externe Clients wie Hermes
benötigen später den authentifizierten WireGuard-Gateway und dürfen nicht
direkt auf das interne Werkzeugnetz zugreifen.
Die öffentliche Standardkonfiguration nutzt `UD-IQ4_XS`. Das bislang schnellste Die öffentliche Standardkonfiguration nutzt `UD-IQ4_XS`. Das bislang schnellste
Referenzprofil nutzt dagegen die lokal vorhandene `IQ4-MIX`-Datei. Für eine Referenzprofil nutzt dagegen die lokal vorhandene `IQ4-MIX`-Datei. Für eine
bitgenaue Migration diese Datei anhand der in `CURRENT_REFERENCE.md` bitgenaue Migration diese Datei anhand der in `CURRENT_REFERENCE.md`
+18 -6
View File
@@ -36,17 +36,24 @@ Pflichtrollen:
- FLUX.2 klein - FLUX.2 klein
- XTTS-v2 und verwendete Stimme - XTTS-v2 und verwendete Stimme
## 2. Externe Komponenten und Commits – offen ## 2. Externe Komponenten und Commits – teilweise gesichert
Für jedes separate Projekt benötigen wir Repository und Commit: Im Repository gesichert sind inzwischen:
- getrennte MCP-Container und internes Netz
- Web-MCP-Fassade sowie gepinnte TinySearch-/SearXNG-Images
- ARR-MCP 1.0.1 und der aktuell eingesetzte kompakte Sonarr-Patch
- Home-Assistant-Relay ohne eingebettetes Token
- Startlogik und Health-Checks
Noch extern zu beschaffen und exakt festzuhalten sind:
- Home-Assistant-MCP - Home-Assistant-MCP
- ARR-MCP - `runraid` 0.4.2 für den read-only Unraid-MCP
- Unraid read-only MCP
- gegebenenfalls eigener Unraid-Administrations-MCP - gegebenenfalls eigener Unraid-Administrations-MCP
- LLama-GUI, falls sie erhalten bleibt - LLama-GUI, falls sie erhalten bleibt
Jede Komponente bekommt zusätzlich: Jede noch externe Komponente bekommt zusätzlich:
- Installationsbefehl - Installationsbefehl
- Systembenutzer - Systembenutzer
@@ -156,7 +163,7 @@ Festlegen, welche Daten persistent sein sollen:
- Benchmarkresultate: eigenes Repository - Benchmarkresultate: eigenes Repository
- Logs: ohne Prompt- und Tool-Antwortinhalte - Logs: ohne Prompt- und Tool-Antwortinhalte
## 9. Ende-zu-Ende-Installer – implementiert, Hardware-Abnahme offen ## 9. Ende-zu-Ende-Installer – weitgehend implementiert, Praxistest offen
Der Ablauf ist jetzt in `install.sh` zusammengeführt: Der Ablauf ist jetzt in `install.sh` zusammengeführt:
@@ -170,6 +177,11 @@ enable-selected-mcp-profiles
run-acceptance-tests run-acceptance-tests
``` ```
`platform/mcp/install-tools.sh` installiert den Webbereich automatisch und
aktiviert HA, ARR und Unraid nur bei vorhandenen Secret-/Programmdateien. Offen
bleiben ein kompletter Leerhost-Probelauf und der automatisierte Import einer
bereits bestehenden Open-WebUI-Datenbank.
Jeder Schritt muss wiederholbar, einzeln prüfbar und bei Fehlern abbrechbar Jeder Schritt muss wiederholbar, einzeln prüfbar und bei Fehlern abbrechbar
sein. Ein fehlgeschlagener Schritt darf keinen halb aktivierten Dienst sein. Ein fehlgeschlagener Schritt darf keinen halb aktivierten Dienst
hinterlassen. hinterlassen.
+7 -1
View File
@@ -21,7 +21,13 @@ verlässt sich nicht allein auf UFW.
- Profile Controller: einzige Socket-Ausnahme; feste Profile und nur - Profile Controller: einzige Socket-Ausnahme; feste Profile und nur
List/Start/Stop, keine frei wählbaren Images, Befehle oder Mounts. List/Start/Stop, keine frei wählbaren Images, Befehle oder Mounts.
- Open WebUI: einziges persistentes Chat-Volume. - Open WebUI: einziges persistentes Chat-Volume.
- SearXNG: intern, Suchanfragen ohne Chatverlauf. - MCP-Fachcontainer: intern, getrennte Secrets und keine Host-Ports.
- TinySearch/SearXNG: intern, Suchanfragen ohne Chatverlauf.
llama.cpp bekommt weder MCP-Konfiguration noch HA-, ARR- oder Unraid-Secrets.
Open WebUI kennt nur interne MCP-URLs; Authentisierung zu den Zielsystemen
findet im jeweiligen Fachcontainer statt. Externe MCP-Clients werden erst über
einen authentifizierten WireGuard-Gateway zugelassen.
Ein Docker-Socket bleibt grundsätzlich privilegiert. Der Controller reduziert Ein Docker-Socket bleibt grundsätzlich privilegiert. Der Controller reduziert
die erreichbare Funktion stark, ersetzt aber keine zusätzliche Socket-Proxy- die erreichbare Funktion stark, ersetzt aber keine zusätzliche Socket-Proxy-
+5 -1
View File
@@ -232,7 +232,7 @@ set -euo pipefail
WG=$WG_INTERFACE WG=$WG_INTERFACE
HOME_NET=$WG_HOME_SUBNET HOME_NET=$WG_HOME_SUBNET
TABLE=51820 TABLE=51820
for NET in 172.30.10.0/24 172.30.30.0/24 172.30.40.0/24; do for NET in 172.30.10.0/24 172.30.30.0/24 172.30.40.0/24 172.30.50.0/24; do
ip rule add from \"\$NET\" table \"\$TABLE\" priority 12000 2>/dev/null || true ip rule add from \"\$NET\" table \"\$TABLE\" priority 12000 2>/dev/null || true
done done
ip route replace \"\$HOME_NET\" dev \"\$WG\" ip route replace \"\$HOME_NET\" dev \"\$WG\"
@@ -280,6 +280,10 @@ build_and_start() {
cd "$STACK_DIR" cd "$STACK_DIR"
docker build --build-arg LLAMA_CPP_COMMIT="$commit" \ docker build --build-arg LLAMA_CPP_COMMIT="$commit" \
-f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local . -f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local .
# Creates the shared internal tools network before Open WebUI is created.
# Web search always starts; HA/ARR/Unraid only start when their root-only
# secret files and required local artifacts are present.
"$STACK_DIR/platform/mcp/install-tools.sh"
docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \ docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \
llama-fast llama-medium llama-long llama-experimental llama-fast llama-medium llama-long llama-experimental
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \ docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
-12
View File
@@ -1,12 +0,0 @@
{
"mcpServers": {
"web": {
"command": "/usr/bin/python3",
"args": ["/opt/mike-ai/mcp/web_search_mcp.py"],
"env": {
"SEARXNG_URL": "http://searxng:8080"
},
"timeout_ms": 120000
}
}
}
+16 -20
View File
@@ -10,30 +10,26 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
PLATFORM="$REPO_ROOT/platform" PLATFORM="$REPO_ROOT/platform"
PROFILE_TARGET=/opt/mike-ai/platform/profiles PROFILE_TARGET=/opt/mike-ai/platform/profiles
DROPIN=/etc/systemd/system/mike-ai-llama-ui.service.d DROPIN=/etc/systemd/system/mike-ai-llama-ui.service.d
WEB_TARGET=/opt/mike-ai/web-search MCP_ROOT=/opt/mike-ai/mcp-containers
MCP_TARGET=$MCP_ROOT/platform/mcp
MCP_SEARCH_TARGET=$MCP_ROOT/platform/web-search
install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" "$WEB_TARGET" /etc/mike-ai install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" \
"$MCP_TARGET" "$MCP_ROOT/platform/web-search" /etc/mike-ai
install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \ install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
/etc/systemd/system/mike-ai-llama-ui.service /etc/systemd/system/mike-ai-llama-ui.service
install -m 0644 "$PLATFORM/systemd/mike-ai-web-search.service" \
/etc/systemd/system/mike-ai-web-search.service
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf" install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf" install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf" install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
install -m 0644 "$PLATFORM/web-search/compose.yaml" "$WEB_TARGET/compose.yaml" rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/"
install -m 0644 "$PLATFORM/web-search/tinysearch_config.json" \ rsync -a --delete --exclude searxng-settings.yml \
"$WEB_TARGET/tinysearch_config.json" "$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/"
install -m 0755 "$PLATFORM/web-search/web_search_mcp.py" \ if [[ ! -s $MCP_SEARCH_TARGET/searxng-settings.yml ]]; then
"$WEB_TARGET/web_search_mcp.py" install -m 0640 "$PLATFORM/web-search/searxng-settings.example.yml" \
if [[ ! -e "$WEB_TARGET/searxng-settings.yml" ]]; then "$MCP_SEARCH_TARGET/searxng-settings.yml"
install -m 0600 "$PLATFORM/web-search/searxng-settings.example.yml" \ sed -i "s/CHANGE_ME_GENERATE_RANDOM_SECRET/$(openssl rand -hex 32)/" \
"$WEB_TARGET/searxng-settings.yml.example" "$MCP_SEARCH_TARGET/searxng-settings.yml"
fi
if [[ ! -e /etc/mike-ai/mcp-servers.json ]]; then
install -m 0600 "$PLATFORM/mcp/mcp-servers.example.json" \
/etc/mike-ai/mcp-servers.json.example
fi fi
systemctl daemon-reload systemctl daemon-reload
@@ -43,7 +39,7 @@ Kernkonfiguration installiert, aber noch nicht gestartet.
Vor dem Start: Vor dem Start:
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen. 1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
2. /etc/mike-ai/mcp-servers.json mit minimalen Servern erstellen. 2. llama.cpp bauen und mit `llama-profile fast` starten.
3. llama.cpp bauen. 3. Fach-Secrets unter /etc/mike-ai ablegen.
4. Danach: llama-profile fast 4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh
EOF EOF
+12
View File
@@ -0,0 +1,12 @@
FROM python:3.13-slim AS builder
COPY --from=ghcr.io/astral-sh/uv:0.11.7 /uv /uvx /bin/
RUN uv pip install --system --break-system-packages "arr-mcp[mcp]==1.0.1"
FROM python:3.13-slim
COPY --from=builder /usr/local /usr/local
RUN groupadd --system --gid 10001 mcp \
&& useradd --system --uid 10001 --gid 10001 --no-create-home mcp
USER 10001:10001
EXPOSE 8000
ENTRYPOINT ["arr-mcp"]
CMD ["--transport", "streamable-http", "--host", "0.0.0.0", "--port", "8000", "--auth-type", "none"]
@@ -0,0 +1,8 @@
FROM nginx:1.29-alpine
COPY platform/mcp/homeassistant.conf.template /etc/nginx/templates/homeassistant.conf.template
COPY platform/mcp/ha-relay-entrypoint.sh /usr/local/bin/ha-relay-entrypoint
RUN chmod 0755 /usr/local/bin/ha-relay-entrypoint \
&& mkdir -p /tmp/client_temp /tmp/proxy_temp \
&& chown -R nginx:nginx /tmp/client_temp /tmp/proxy_temp
EXPOSE 8000
ENTRYPOINT ["/usr/local/bin/ha-relay-entrypoint"]
+15
View File
@@ -0,0 +1,15 @@
FROM python:3.13-slim
ARG MCP_PROXY_VERSION=0.12.0
RUN apt-get update \
&& apt-get install -y --no-install-recommends openssh-client \
&& rm -rf /var/lib/apt/lists/* \
&& pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp>=1.17,<2" \
&& useradd --system --uid 10001 --create-home --home-dir /app mcp
RUN touch /app/unraid_mcp.py && chown 10001:10001 /app/unraid_mcp.py
USER 10001:10001
WORKDIR /app
EXPOSE 8000
ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"]
CMD ["python", "/app/unraid_mcp.py"]
+14
View File
@@ -0,0 +1,14 @@
FROM python:3.13-slim
ARG MCP_PROXY_VERSION=0.12.0
RUN pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp>=1.17,<2"
RUN useradd --system --uid 10001 --create-home --home-dir /app mcp
COPY web-search/web_search_mcp.py /app/web_search_mcp.py
RUN chown -R 10001:10001 /app
USER 10001:10001
WORKDIR /app
EXPOSE 8000
ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"]
CMD ["python", "/app/web_search_mcp.py"]
+61 -30
View File
@@ -1,43 +1,74 @@
# MCP-Architektur # Zentrale MCP-Werkzeugebene
Die produktive MCP-Konfiguration ist absichtlich nicht Bestandteil des Git- MCP-Werkzeuge sind **keine llama.cpp-Startparameter**. Sie laufen als kleine,
Repositories, weil sie lokale Pfade und Zugangsdaten referenziert. Das Beispiel voneinander getrennte Container und werden von OpenWebUI, Hermes oder einem
zeigt nur die Struktur. anderen MCP-Client gezielt ausgewählt. Das hält Tool-Schemas aus normalen
Prompts heraus, verhindert den früher beobachteten Kontextverbrauch von über
200.000 Tokens und macht Werkzeuge unabhängig vom geladenen Modellprofil.
## Empfohlene Server ## Container
- `web`: Websuche über lokales TinySearch/SearXNG | Container | Endpunkt im Netz `mike-ai-tools` | Zweck | Standard |
- `homeassistant`: Administration mit eigenem, minimal berechtigtem Token |---|---|---|---|
- `arr`: Sonarr/Radarr über spezialisierte Aktionen | `mcp-web` | `http://mike-ai-mcp-web:8000/mcp` | kompakte Websuche und Quellenvergleich | an |
- `unraid-readonly`: Diagnose ohne Schreiboperationen | `mcp-homeassistant` | `http://mike-ai-mcp-homeassistant:8000/mcp` | Relay zum nativen HA-MCP; Token bleibt serverseitig | Profil `homeassistant` |
| `mcp-arr` | `http://mike-ai-mcp-arr:8000/mcp` | Sonarr/Radarr/Prowlarr mit serverseitiger Policy | Profil `arr` |
| `mcp-unraid-official` | `http://mike-ai-mcp-unraid-official:8000/mcp` | offizieller, read-only begrenzter Unraid-Zugang | Profil `unraid` |
| `mcp-unraid-ssh` | `http://mike-ai-mcp-unraid-ssh:8000/mcp` | erweiterte Diagnose über einen erzwungenen SSH-Befehl | optional (`extended`) |
## Getrennte Konfigurationen TinySearch und SearXNG sind interne Abhängigkeiten des Web-MCPs und werden
nicht direkt als allgemeine Werkzeuge angeboten.
Statt alle Werkzeuge ständig zu laden, werden mehrere Dateien empfohlen: ## Sicherheitsmodell
```text - Kein MCP-Port wird auf eine Host-Adresse veröffentlicht.
/etc/mike-ai/mcp-standard.json - Nur Clients im privaten Docker-Netz `mike-ai-tools` erreichen die Endpunkte.
/etc/mike-ai/mcp-homeassistant.json - Secrets bleiben in Dateien unter `/etc/mike-ai` und werden read-only
/etc/mike-ai/mcp-arr.json eingehängt. Sie gehören weder in Git noch in OpenWebUI-Tooldefinitionen.
/etc/mike-ai/mcp-unraid-readonly.json - Jeder Container ist read-only, verliert Linux-Capabilities und hat
/etc/mike-ai/mcp-unraid-write.json `no-new-privileges`.
- Der SSH-basierte Unraid-Container ist nicht Teil des Standardstarts.
- Ein allgemeiner Host-Shell-MCP wird bewusst nicht angeboten.
## Start
```bash
sudo platform/mcp/install-tools.sh
``` ```
Das jeweilige Profil verweist nur auf die benötigte Datei. Dadurch werden die Der Grundstart enthält nur Websuche. Bereits konfigurierte Fachbereiche werden
Tool-Schemas kleiner, das Kontextfenster bleibt frei und kleine Modelle müssen explizit ergänzt:
weniger Werkzeuge unterscheiden.
Credentials werden von schmalen Wrapper-Programmen wie `run-arr-mcp` oder Das Skript erkennt vorhandene Secret-Dateien und aktiviert dadurch automatisch
`runraid` aus geschützten Environment-Dateien geladen. Das JSON selbst enthält `homeassistant`, `arr` und `unraid`. Ohne Fach-Secrets startet nur der sichere
weder Werte noch Pfade zu einzelnen Tokens. Webbereich.
## Schreibzugriff Für den derzeit migrierten Container kann der Name `Open-WebUI` lauten. Der
Netzwerkbefehl ist idempotent zu behandeln.
Schreibende Server gehören nicht in `mcp-standard.json`. Sie benötigen eine Die lokale Installation benötigt die vorhandenen Secret-Dateien:
Vorschau und ein an die exakte Änderung gebundenes Approval Ticket.
## Shell ```text
/etc/mike-ai/homeassistant-admin-mcp.env
/etc/mike-ai/arr-mcp.env
/etc/mike-ai/runraid/.env
```
Ein allgemeiner Shell-MCP ist nicht Teil der Zielplattform. Insbesondere Die erweiterte Unraid-Diagnose benötigt zusätzlich die Konfigurationsdatei,
`python3`, `ssh`, `scp`, `curl` und `systemctl` dürfen nicht gemeinsam als den eingeschränkten Schlüssel und die bekannte Hostsignatur. Sie wird nur mit
scheinbar harmlose Allowlist angeboten werden. `--profile extended` gestartet.
TinySearch speichert sein lokales Embedding-Modell in einem Docker-Volume.
Nach einer Erstinstallation wird das Modell einmalig im Container mit
`tinysearch setup` geladen. Das Volume bleibt bei Containerupdates erhalten.
## Client-Auswahl
Werkzeuge werden nicht pauschal an jedes Modell gehängt. Für Home-Assistant-
Fragen wird HA ausgewählt, für Medien ARR, für Recherche Web und für die NAS
Unraid. Mehrere Werkzeuge werden nur aktiviert, wenn die Aufgabe tatsächlich
mehrere Bereiche verbindet.
Schreibende Aktionen bleiben hinter der jeweiligen serverseitigen Policy und
einem Vorschau-/Bestätigungsablauf. Ein Client-Schalter allein darf niemals
eine read-only Policy aufheben.
+147
View File
@@ -0,0 +1,147 @@
name: mike-ai-tools
x-tool-common: &tool-common
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:rw,noexec,nosuid,nodev,size=64m
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
networks: [tools]
logging:
options:
max-size: 10m
max-file: "3"
services:
mcp-web:
<<: *tool-common
build:
context: ..
dockerfile: mcp/Dockerfile.web
image: mike-ai/mcp-web:local
container_name: mike-ai-mcp-web
environment:
TINYSEARCH_MCP_URL: http://tinysearch:8000/mcp
SEARXNG_URL: http://searxng:8080
WEB_SEARCH_BUDGET_MAX_RELATED: "6"
depends_on:
tinysearch:
condition: service_started
searxng:
<<: *tool-common
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
container_name: mike-ai-tools-searxng
volumes:
- ${SEARXNG_SETTINGS_FILE:-../web-search/searxng-settings.example.yml}:/etc/searxng/settings.yml:ro
networks: [tools, egress]
tinysearch:
<<: *tool-common
image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c
container_name: mike-ai-tools-tinysearch
shm_size: 1gb
volumes:
- tinysearch-models:/data/models
- ../web-search/tinysearch_config.json:/config/tinysearch_config.json:ro
environment:
MCP_TRANSPORT: streamable-http
MCP_HOST: 0.0.0.0
MCP_PORT: "8000"
TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json
TINYSEARCH_SEARCH_BACKEND: searxng
SEARXNG_URL: http://searxng:8080/search
depends_on: [searxng]
cap_add: [SETUID, SETGID, CHOWN]
networks: [tools, egress]
# The image's built-in `tinysearch doctor` also requires a writable
# configuration directory, although normal server operation does not.
# Check the service socket instead so read-only hardening remains intact.
healthcheck:
test: ["CMD", "python", "-c", "import socket; s=socket.create_connection(('127.0.0.1', 8000), 2); s.close()"]
interval: 30s
timeout: 5s
retries: 5
start_period: 20s
mcp-homeassistant:
<<: *tool-common
build:
context: ../..
dockerfile: platform/mcp/Dockerfile.homeassistant-relay
image: mike-ai/mcp-homeassistant-relay:local
container_name: mike-ai-mcp-homeassistant
profiles: [homeassistant]
volumes:
- ${HA_ENV_FILE:-/etc/mike-ai/homeassistant-admin-mcp.env}:/run/secrets/homeassistant.env:ro
cap_add: [CHOWN, SETUID, SETGID]
networks: [tools, egress]
mcp-arr:
<<: *tool-common
build:
context: .
dockerfile: Dockerfile.arr
image: mike-ai/mcp-arr:1.0.1-patched
container_name: mike-ai-mcp-arr
profiles: [arr]
env_file:
- ${ARR_ENV_FILE:-/etc/mike-ai/arr-mcp.env}
volumes:
# The local fork adds bounded read-only Sonarr pseudo-actions. Keep the
# patch explicit until upstream publishes a self-contained 2.x image.
- ${ARR_SONARR_PATCH:-./patches/mcp_sonarr.py}:/usr/local/lib/python3.13/site-packages/arr_mcp/mcp/mcp_sonarr.py:ro
networks: [tools, egress]
mcp-unraid-official:
<<: *tool-common
image: debian:13-slim
container_name: mike-ai-mcp-unraid-official
profiles: [unraid]
env_file:
- ${RUNRAID_ENV_FILE:-/etc/mike-ai/runraid/.env}
environment:
UNRAID_RMCP_HOST: 0.0.0.0
UNRAID_RMCP_PORT: "8000"
UNRAID_RMCP_DISABLE_HTTP_AUTH: "true"
UNRAID_NOAUTH: "true"
UNRAID_RMCP_ALLOWED_HOSTS: "mike-ai-mcp-unraid-official:8000,mike-ai-mcp-unraid-official,localhost:8000,127.0.0.1:8000"
volumes:
- ${RUNRAID_BINARY:-/usr/local/bin/runraid}:/usr/local/bin/unraid:ro
entrypoint: ["/usr/local/bin/unraid"]
command: ["serve"]
networks: [tools, egress]
mcp-unraid-ssh:
<<: *tool-common
profiles: [extended]
build:
context: .
dockerfile: Dockerfile.unraid-ssh
image: mike-ai/mcp-unraid-ssh:local
container_name: mike-ai-mcp-unraid-ssh
environment:
UNRAID_MCP_CONFIG: /run/config/unraid-mcp.json
volumes:
- ${UNRAID_MCP_SOURCE:-/opt/mike-ai/unraid-agent/unraid_mcp.py}:/app/unraid_mcp.py:ro
- ${UNRAID_MCP_CONFIG:-/etc/mike-ai/unraid-mcp.json}:/run/config/unraid-mcp.json:ro
- ${UNRAID_SSH_KEY:-/etc/mike-ai/keys/unraid_root}:/etc/mike-ai/keys/unraid_root:ro
- ${UNRAID_KNOWN_HOSTS:-/etc/mike-ai/ssh/known_hosts_unraid_ai}:/etc/mike-ai/ssh/known_hosts_unraid_ai:ro
- unraid-audit:/var/log/mike-ai
networks: [tools, egress]
networks:
tools:
name: mike-ai-tools
internal: true
ipam:
config: [{subnet: 172.30.40.0/24}]
egress:
name: mike-ai-tools-egress
ipam:
config: [{subnet: 172.30.50.0/24}]
volumes:
tinysearch-models:
unraid-audit:
+23
View File
@@ -0,0 +1,23 @@
#!/bin/sh
set -eu
config=/run/secrets/homeassistant.env
if [ ! -r "$config" ]; then
echo "Home Assistant secret file is missing" >&2
exit 1
fi
set -a
. "$config"
set +a
: "${HASS_URL:?HASS_URL is required}"
: "${HASS_TOKEN:?HASS_TOKEN is required}"
upstream=${HASS_URL%/}
escaped_token=$(printf '%s' "$HASS_TOKEN" | sed 's/[&/]/\\&/g')
escaped_upstream=$(printf '%s' "$upstream" | sed 's/[&/]/\\&/g')
sed -e "s/__HASS_TOKEN__/$escaped_token/g" \
-e "s/__HASS_UPSTREAM__/$escaped_upstream/g" \
/etc/nginx/templates/homeassistant.conf.template \
> /tmp/nginx.conf
unset HASS_TOKEN
exec nginx -c /tmp/nginx.conf -g 'daemon off;'
+30
View File
@@ -0,0 +1,30 @@
worker_processes 1;
pid /tmp/nginx.pid;
error_log /dev/stderr warn;
events { worker_connections 128; }
http {
access_log /dev/stdout;
client_body_temp_path /tmp/client_temp;
proxy_temp_path /tmp/proxy_temp;
fastcgi_temp_path /tmp/fastcgi_temp;
uwsgi_temp_path /tmp/uwsgi_temp;
scgi_temp_path /tmp/scgi_temp;
proxy_buffering off;
proxy_read_timeout 600s;
proxy_send_timeout 600s;
server {
listen 8000;
location /mcp {
proxy_pass __HASS_UPSTREAM__/api/hass_mcp;
proxy_http_version 1.1;
proxy_ssl_server_name on;
proxy_ssl_name $proxy_host;
proxy_set_header Authorization "Bearer __HASS_TOKEN__";
proxy_set_header Host $proxy_host;
proxy_set_header Connection "";
}
}
}
+52
View File
@@ -0,0 +1,52 @@
#!/usr/bin/env bash
set -Eeuo pipefail
[[ $EUID -eq 0 ]] || { echo "Bitte als root ausführen." >&2; exit 1; }
MCP_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
COMPOSE=(docker compose -f "$MCP_DIR/compose.yaml")
export SEARXNG_SETTINGS_FILE="${SEARXNG_SETTINGS_FILE:-$MCP_DIR/../web-search/searxng-settings.yml}"
[[ -s $SEARXNG_SETTINGS_FILE ]] || {
echo "SearXNG-Konfiguration fehlt: $SEARXNG_SETTINGS_FILE" >&2
exit 1
}
profiles=()
if [[ -s /etc/mike-ai/homeassistant-admin-mcp.env ]]; then
profiles+=(--profile homeassistant)
else
echo "Home Assistant bleibt aus: Secret-Datei fehlt."
fi
if [[ -s /etc/mike-ai/arr-mcp.env ]]; then
profiles+=(--profile arr)
else
echo "ARR bleibt aus: Secret-Datei fehlt."
fi
if [[ -s /etc/mike-ai/runraid/.env && -x /usr/local/bin/runraid ]]; then
profiles+=(--profile unraid)
else
echo "Unraid bleibt aus: runraid 0.4.2 oder Secret-Datei fehlt."
fi
# TinySearch keeps the embedding bundle outside the container. Download it
# once on a fresh host; subsequent rebuilds reuse the named volume.
docker volume create mike-ai-tools_tinysearch-models >/dev/null
tiny_image="marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c"
if ! docker run --rm --entrypoint test \
-v mike-ai-tools_tinysearch-models:/data/models "$tiny_image" \
-f /data/models/all-minilm-l6-v2-onnx/model.onnx; then
echo "TinySearch-Modell wird einmalig geladen."
docker run --rm -v mike-ai-tools_tinysearch-models:/data/models \
"$tiny_image" tinysearch setup
fi
"${COMPOSE[@]}" "${profiles[@]}" up -d --build
for webui in mike-ai-open-webui Open-WebUI; do
if docker container inspect "$webui" >/dev/null 2>&1; then
docker network connect mike-ai-tools "$webui" 2>/dev/null || true
fi
done
"${COMPOSE[@]}" "${profiles[@]}" ps
-23
View File
@@ -1,23 +0,0 @@
{
"mcpServers": {
"web": {
"command": "/usr/bin/python3",
"args": ["/opt/mike-ai/web-search/web_search_mcp.py"]
},
"homeassistant": {
"command": "/usr/local/bin/homeassistant-native-mcp",
"args": [],
"timeout_ms": 60000
},
"arr": {
"command": "/usr/local/bin/run-arr-mcp",
"args": [],
"timeout_ms": 30000
},
"unraid-readonly": {
"command": "/usr/local/bin/runraid",
"args": ["mcp"],
"timeout_ms": 60000
}
}
}
+329
View File
@@ -0,0 +1,329 @@
"""Sonarr condensed action-routed MCP tool.
CONCEPT:ECO-4.82 — gitlab-style organized per-service tool surface.
"""
import os
import json
import re
from typing import Any
from agent_utilities.mcp_utilities import dispatch, run_blocking
from fastmcp import FastMCP
from pydantic import Field
from arr_mcp.auth import get_sonarr_client
READ_ONLY_ACTIONS = frozenset(
{
"get_system_status", "get_health", "get_diskspace", "get_ping",
"get_series", "get_series_id", "get_series_lookup", "lookup_series",
"get_episode", "get_episode_id", "get_episodefile", "get_episodefile_id",
"get_calendar", "get_calendar_id", "get_history", "get_history_series",
"get_history_since", "get_queue", "get_queue_details", "get_queue_status",
"get_wanted_missing", "get_wanted_missing_id", "get_wanted_cutoff",
"get_wanted_cutoff_id", "get_qualityprofile", "get_qualityprofile_id",
"get_languageprofile", "get_languageprofile_id", "get_tag", "get_tag_id",
"get_tag_detail", "get_tag_detail_id", "get_command", "get_command_id",
"get_release",
}
)
PSEUDO_ACTIONS = frozenset({"find_series", "get_season_summary", "search_releases"})
WRITE_ACTIONS = frozenset({
"post_command",
"post_release",
"put_episode_id",
"put_episode_monitor",
"put_series_id",
"put_series",
"put_wanted",
"post_wanted",
})
MAX_COLLECTION_ITEMS = 50
def _plain(value: Any) -> Any:
if hasattr(value, "model_dump") and callable(value.model_dump):
return value.model_dump()
if hasattr(value, "dict") and callable(value.dict):
return value.dict()
if isinstance(value, list):
return [_plain(item) for item in value]
if isinstance(value, dict):
return {str(key): _plain(item) for key, item in value.items()}
return value
def _unwrap(value: Any) -> Any:
value = _plain(value)
if isinstance(value, dict) and set(value) == {"result"}:
return value["result"]
return value
def _pick(item: dict[str, Any], fields: tuple[str, ...]) -> dict[str, Any]:
return {field: item[field] for field in fields if item.get(field) is not None}
def _compact_series(item: dict[str, Any], include_seasons: bool = False) -> dict[str, Any]:
result = _pick(
item,
("id", "title", "sortTitle", "year", "status", "monitored", "path", "tvdbId"),
)
statistics = item.get("statistics") or {}
if isinstance(statistics, dict):
result["statistics"] = _pick(
statistics,
("seasonCount", "episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"),
)
if include_seasons:
result["seasons"] = [
{
**_pick(season, ("seasonNumber", "monitored")),
"statistics": _pick(
season.get("statistics") or {},
("episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"),
),
}
for season in item.get("seasons", [])
if isinstance(season, dict)
]
return result
def _compact_episode(item: dict[str, Any]) -> dict[str, Any]:
return _pick(
item,
("id", "seriesId", "seasonNumber", "episodeNumber", "title", "airDate", "airDateUtc", "monitored", "hasFile", "episodeFileId"),
)
def _compact_file(item: dict[str, Any]) -> dict[str, Any]:
quality = item.get("quality") or {}
quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None
result = _pick(
item,
("id", "seriesId", "seasonNumber", "relativePath", "path", "size", "dateAdded", "releaseGroup"),
)
if quality_name:
result["quality"] = quality_name
return result
def _compact_release(item: dict[str, Any]) -> dict[str, Any]:
quality = item.get("quality") or {}
quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None
result = _pick(
item,
(
"guid", "title", "indexer", "indexerId", "size", "age", "ageHours",
"seeders", "leechers", "protocol", "downloadAllowed", "releaseWeight",
),
)
if quality_name:
result["quality"] = quality_name
rejections = item.get("rejections")
if isinstance(rejections, list) and rejections:
result["rejections"] = [str(reason)[:180] for reason in rejections[:5]]
return result
def _bounded(items: list[Any], compact) -> dict[str, Any]:
total = len(items)
return {
"total": total,
"returned": min(total, MAX_COLLECTION_ITEMS),
"truncated": total > MAX_COLLECTION_ITEMS,
"items": [compact(item) for item in items[:MAX_COLLECTION_ITEMS] if isinstance(item, dict)],
"next_step": (
"Use find_series or narrower Sonarr parameters; do not repeat the same broad request."
if total > MAX_COLLECTION_ITEMS else None
),
}
def _compact_result(action: str, value: Any) -> Any:
value = _unwrap(value)
if isinstance(value, list):
if action in {"get_series", "get_series_lookup", "lookup_series"}:
return _bounded(value, _compact_series)
if action in {"get_episode", "get_calendar", "get_wanted_missing", "get_wanted_cutoff"}:
return _bounded(value, _compact_episode)
if action == "get_episodefile":
return _bounded(value, _compact_file)
if action == "get_release":
return _bounded(value, _compact_release)
return _bounded(value, lambda item: item)
if isinstance(value, dict) and action in {"get_series_id"}:
return _compact_series(value, include_seasons=True)
if isinstance(value, dict) and action in {"get_episode_id"}:
return _compact_episode(value)
if isinstance(value, dict) and action in {"get_episodefile_id"}:
return _compact_file(value)
return value
async def _find_series(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]:
query = str(kwargs.get("query", "")).strip()
if len(query) < 2:
raise ValueError("find_series requires params_json with a query of at least 2 characters")
limit = max(1, min(int(kwargs.get("limit", 8)), 15))
raw = _unwrap(await run_blocking(dispatch, client, "get_series", {}, service="arr-sonarr"))
words = [word for word in re.findall(r"[a-z0-9]+", query.casefold()) if len(word) > 1]
matches = []
for item in raw if isinstance(raw, list) else []:
haystack = " ".join(
str(item.get(field, "")) for field in ("title", "sortTitle", "originalTitle", "alternateTitles")
).casefold()
if all(word in haystack for word in words):
matches.append(_compact_series(item, include_seasons=True))
return {
"query": query,
"matches": matches[:limit],
"match_count": len(matches),
"truncated": len(matches) > limit,
"task_complete": True,
"instruction": "Use the returned series id for details. Do not call get_series for discovery.",
}
async def _season_summary(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]:
series_id = int(kwargs["series_id"])
season_number = int(kwargs["season_number"])
series = _unwrap(
await run_blocking(dispatch, client, "get_series_id", {"id": series_id}, service="arr-sonarr")
)
episodes = _unwrap(
await run_blocking(
dispatch,
client,
"get_episode",
{"seriesId": series_id, "seasonNumber": season_number},
service="arr-sonarr",
)
)
files = _unwrap(
await run_blocking(
dispatch,
client,
"get_episodefile",
{"seriesId": series_id},
service="arr-sonarr",
)
)
selected_episodes = [
_compact_episode(item) for item in episodes
if isinstance(item, dict) and item.get("seasonNumber") == season_number
] if isinstance(episodes, list) else []
selected_files = [
_compact_file(item) for item in files
if isinstance(item, dict) and item.get("seasonNumber") == season_number
] if isinstance(files, list) else []
groups = sorted({str(item.get("releaseGroup")) for item in selected_files if item.get("releaseGroup")})
return {
"series": _compact_series(series) if isinstance(series, dict) else {"id": series_id},
"season_number": season_number,
"episode_count": len(selected_episodes),
"file_count": len(selected_files),
"release_groups": groups,
"episodes": selected_episodes[:30],
"files": selected_files[:30],
"task_complete": True,
"instruction": "This is the complete compact season answer. Do not repeat broad series or episode queries.",
}
async def _search_releases(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]:
series_id = kwargs.get("series_id")
episode_id = kwargs.get("episode_id")
season_number = kwargs.get("season_number")
release_group = str(kwargs.get("release_group", "")).strip()
if series_id is None and episode_id is None:
raise ValueError("search_releases requires series_id or episode_id")
query: dict[str, Any] = {}
if series_id is not None:
query["seriesId"] = int(series_id)
if episode_id is not None:
query["episodeId"] = int(episode_id)
if season_number is not None:
query["seasonNumber"] = int(season_number)
raw = await run_blocking(dispatch, client, "get_release", query, service="arr-sonarr")
raw = _unwrap(raw)
if release_group and isinstance(raw, list):
needle = release_group.casefold()
raw = [
item for item in raw
if isinstance(item, dict)
and needle in (
str(item.get("releaseGroup", "")) + " " + str(item.get("title", ""))
).casefold()
]
compact = _compact_result("get_release", raw)
return {
"task_complete": True,
"search_scope": {
"series_id": series_id,
"episode_id": episode_id,
"season_number": season_number,
"release_group_filter": release_group or None,
},
"monitoring_changed": False,
"download_started": False,
"results": compact,
"instruction": "These are Sonarr indexer results. Do not use web search to replace them. Never download unless the user separately approves a write action.",
}
def register_sonarr_tools(mcp: FastMCP) -> None:
@mcp.tool(tags={"sonarr"})
async def sonarr_action(
action: str = Field(
description="Read-only Sonarr action. Use find_series {query} for titles, get_season_summary {series_id, season_number} for holdings, and search_releases {series_id, season_number, optional release_group} to query configured Sonarr indexers without downloading or changing monitoring. Avoid broad get_series/get_episode calls."
),
params_json: str = Field(
default="{}",
description="JSON string of parameters to pass to the action.",
),
) -> Any:
"""Query Sonarr through a server-side allowlist (read-only by default; write actions when ARR_MCP_WRITE=1)."""
if action in {"list_actions", "help", "actions"}:
return {
"service": "sonarr",
"access_mode": "write" if os.environ.get("ARR_MCP_WRITE", "").strip().lower() in ("1", "true", "yes", "on") else "read-only",
"actions": sorted(READ_ONLY_ACTIONS),
"write_actions": sorted(WRITE_ACTIONS) if os.environ.get("ARR_MCP_WRITE", "").strip().lower() in ("1", "true", "yes", "on") else [],
"preferred_compact_actions": sorted(PSEUDO_ACTIONS),
}
allow_write = os.environ.get("ARR_MCP_WRITE", "").strip().lower() in (
"1", "true", "yes", "on"
)
if action in READ_ONLY_ACTIONS | PSEUDO_ACTIONS:
pass
elif allow_write and action in WRITE_ACTIONS:
pass
else:
if allow_write:
raise PermissionError(
f"Sonarr MCP write mode is enabled, but action '{action}' "
"is not in the allowed write set. Allowed: "
f"{sorted(WRITE_ACTIONS)}"
)
raise PermissionError(
f"Sonarr action '{action}' is blocked by the server-side "
"read-only policy. Set ARR_MCP_WRITE=1 to enable write mode."
)
client = get_sonarr_client()
kwargs = {k: v for k, v in json.loads(params_json).items() if v is not None}
if action == "find_series":
return await _find_series(client, kwargs)
if action == "get_season_summary":
return await _season_summary(client, kwargs)
if action == "search_releases":
return await _search_releases(client, kwargs)
result = await run_blocking(
dispatch, client, action, kwargs, service="arr-sonarr"
)
return _compact_result(action, result)
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Fast 76.8K MTP2 with CPU Vision
[Service] [Service]
ExecStart= ExecStart=
ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16 ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU with CPU Vision
[Service] [Service]
ExecStart= ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0 ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure with CPU Vision
[Service] [Service]
ExecStart= ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none
@@ -1,16 +0,0 @@
[Unit]
Description=Local AI Web Search (TinySearch + SearXNG)
Requires=docker.service
After=docker.service network-online.target
Before=mike-ai-llama-ui.service
[Service]
Type=oneshot
RemainAfterExit=yes
WorkingDirectory=/opt/mike-ai/web-search
ExecStart=/usr/bin/docker compose up -d
ExecStop=/usr/bin/docker compose down
TimeoutStartSec=180
[Install]
WantedBy=multi-user.target
-45
View File
@@ -1,45 +0,0 @@
name: local-ai-web-search
services:
searxng:
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
restart: unless-stopped
volumes:
- ./searxng-settings.yml:/etc/searxng/settings.yml:ro
networks: [search]
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://127.0.0.1:8080/healthz"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30s
security_opt: ["no-new-privileges:true"]
tinysearch:
image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c
restart: unless-stopped
ports:
- "127.0.0.1:8000:8000"
shm_size: "1gb"
volumes:
- tinysearch-models:/data/models
- ./tinysearch_config.json:/config/tinysearch_config.json:ro
environment:
MCP_TRANSPORT: streamable-http
MCP_HOST: 0.0.0.0
MCP_PORT: 8000
TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json
TINYSEARCH_SEARCH_BACKEND: searxng
SEARXNG_URL: http://searxng:8080/search
depends_on:
searxng:
condition: service_healthy
networks: [search]
security_opt: ["no-new-privileges:true"]
networks:
search:
driver: bridge
volumes:
tinysearch-models:
+37 -82
View File
@@ -10,12 +10,11 @@ from facts verified by crawled pages or primary APIs.
from __future__ import annotations from __future__ import annotations
import ipaddress import ipaddress
import asyncio
import html import html
import json import json
import os import os
import re import re
import select
import subprocess
import sys import sys
import threading import threading
import time import time
@@ -28,9 +27,9 @@ from xml.etree import ElementTree
SERVER_VERSION = "2.1.0" SERVER_VERSION = "2.1.0"
TINYSEARCH_CONTAINER = os.environ.get( TINYSEARCH_MCP_URL = os.environ.get(
"TINYSEARCH_CONTAINER", "mike-ai-web-search-tinysearch-1" "TINYSEARCH_MCP_URL", "http://tinysearch:8000/mcp"
) ).rstrip("/")
SEARXNG_URL = os.environ.get("SEARXNG_URL", "").rstrip("/") SEARXNG_URL = os.environ.get("SEARXNG_URL", "").rstrip("/")
CHILD_TIMEOUT_SECONDS = float(os.environ.get("TINYSEARCH_CHILD_TIMEOUT", "110")) CHILD_TIMEOUT_SECONDS = float(os.environ.get("TINYSEARCH_CHILD_TIMEOUT", "110"))
HTTP_TIMEOUT_SECONDS = float(os.environ.get("WEB_API_TIMEOUT", "18")) HTTP_TIMEOUT_SECONDS = float(os.environ.get("WEB_API_TIMEOUT", "18"))
@@ -801,90 +800,46 @@ def general_discovery(query: str, limit: int) -> tuple[list[dict[str, Any]], lis
class TinySearchClient: class TinySearchClient:
"""Minimal synchronous MCP client for TinySearch's stdio server.""" """Synchronous facade for TinySearch's Streamable HTTP MCP endpoint.
TinySearch used to be reached by executing ``docker exec`` on the host.
That required access to the Docker socket and coupled this service to a
particular container name. The containerized tool layer instead talks to
TinySearch over the private Docker network.
"""
def __init__(self) -> None: def __init__(self) -> None:
self._next_id = 1
self._lock = threading.Lock() self._lock = threading.Lock()
self._process = subprocess.Popen(
[
"/usr/bin/docker",
"exec",
"-e",
"MCP_TRANSPORT=stdio",
"-i",
TINYSEARCH_CONTAINER,
"tinysearch",
"mcp",
],
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.DEVNULL,
text=True,
encoding="utf-8",
errors="replace",
bufsize=1,
)
self._request(
"initialize",
{
"protocolVersion": "2024-11-05",
"capabilities": {},
"clientInfo": {"name": "mike-ai-web-facade", "version": SERVER_VERSION},
},
)
self._notify("notifications/initialized", {})
def _notify(self, method: str, params: dict[str, Any]) -> None: async def _call_async(self, name: str, arguments: dict[str, Any]) -> str:
assert self._process.stdin is not None # Imported lazily so the dependency-free stdio implementation still
message = {"jsonrpc": "2.0", "method": method, "params": params} # gives a useful startup error outside its production container.
self._process.stdin.write(json.dumps(message, separators=(",", ":")) + "\n") from mcp import ClientSession
self._process.stdin.flush() from mcp.client.streamable_http import streamablehttp_client
def _request(self, method: str, params: dict[str, Any]) -> dict[str, Any]: async with streamablehttp_client(
with self._lock: TINYSEARCH_MCP_URL,
if self._process.poll() is not None: timeout=CHILD_TIMEOUT_SECONDS,
raise RuntimeError("TinySearch child process is not running") sse_read_timeout=CHILD_TIMEOUT_SECONDS,
request_id = self._next_id ) as (read_stream, write_stream, _):
self._next_id += 1 async with ClientSession(read_stream, write_stream) as session:
assert self._process.stdin is not None await session.initialize()
assert self._process.stdout is not None result = await session.call_tool(name, arguments)
message = { if result.isError:
"jsonrpc": "2.0", raise RuntimeError(clean_text(str(result.content)))
"id": request_id, return "\n".join(
"method": method, item.text for item in result.content
"params": params, if getattr(item, "type", None) == "text"
} )
self._process.stdin.write(json.dumps(message, separators=(",", ":")) + "\n")
self._process.stdin.flush()
deadline = datetime.now().timestamp() + CHILD_TIMEOUT_SECONDS
while True:
remaining = deadline - datetime.now().timestamp()
if remaining <= 0:
raise TimeoutError(f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s")
ready, _, _ = select.select([self._process.stdout], [], [], remaining)
if not ready:
raise TimeoutError(f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s")
line = self._process.stdout.readline()
if not line:
raise RuntimeError("TinySearch closed its output stream")
payload = json.loads(line)
if payload.get("id") != request_id:
continue
if "error" in payload:
raise RuntimeError(f"TinySearch error: {payload['error']}")
return payload.get("result") or {}
def call(self, name: str, arguments: dict[str, Any]) -> str: def call(self, name: str, arguments: dict[str, Any]) -> str:
result = self._request("tools/call", {"name": name, "arguments": arguments}) with self._lock:
if result.get("isError"): try:
raise RuntimeError(clean_text(str(result.get("content")))) return asyncio.run(asyncio.wait_for(
texts = [ self._call_async(name, arguments), CHILD_TIMEOUT_SECONDS))
item.get("text", "") except TimeoutError as exc:
for item in result.get("content", []) raise TimeoutError(
if isinstance(item, dict) and item.get("type") == "text" f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s") from exc
]
return "\n".join(texts)
_client: TinySearchClient | None = None _client: TinySearchClient | None = None