Containerize MCP tool services

This commit is contained in:
Mikei386
2026-08-20 23:54:02 +02:00
parent 3ab9628088
commit f0d552ef58
28 changed files with 858 additions and 305 deletions
+8 -6
View File
@@ -40,14 +40,16 @@ Neustart an; danach wird derselbe Befehl erneut ausgeführt.
|---|---|---|
| Open WebUI | `<WG-IP>:8080` | Chat und Administration |
| Profile Router | `<WG-IP>:8081` | OpenAI-kompatible API, Profilwahl |
| llama.cpp | nur Docker-intern | Inferenz, Vision, MCP |
| llama.cpp | nur Docker-intern | Inferenz und integrierte Vision |
| Profile Controller | nur Docker-intern | eng begrenzter Containerwechsel |
| SearXNG | nur Docker-intern | Websuche |
| MCP-Tool-Stack | nur Docker-intern | Web, Home Assistant, ARR und Unraid |
Bildgenerierung, TTS/STT sowie Home-Assistant-, ARR- und Unraid-MCPs werden
bewusst nicht automatisch aktiviert. Sie erhalten später eigene Container und
kleinstmögliche Rechte. Die Bildanalyse ist bereits Bestandteil des
multimodalen Qwen-Modells.
Bildgenerierung und TTS/STT bleiben optionale Dienste. Web-, Home-Assistant-,
ARR- und Unraid-Werkzeuge besitzen dagegen bereits getrennte Container unter
`platform/mcp/`. Open WebUI erreicht sie ausschließlich über das interne
`mike-ai-tools`-Netz; llama.cpp erhält keine MCP-Konfiguration und keine
Infrastruktur-Secrets. Die Bildanalyse ist Bestandteil des multimodalen
Qwen-Modells.
## Dokumentation
+8 -30
View File
@@ -11,15 +11,12 @@ x-llama-common: &llama-common
- /tmp:size=1g,mode=1777
volumes:
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
- ./platform/docker/mcp-standard.json:/etc/mike-ai/mcp-standard.json:ro
environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_DRIVER_CAPABILITIES: compute,utility
dns: ["${AI_DNS:-1.1.1.1}"]
networks:
inference:
aliases: [llama-upstream]
search: {}
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
@@ -36,7 +33,6 @@ services:
labels:
com.mike-ai.llama-profile: fast
environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${FAST_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
@@ -85,8 +81,6 @@ services:
- "0.8"
- --top-k
- "20"
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device
- CUDA0
- --split-mode
@@ -108,7 +102,6 @@ services:
labels:
com.mike-ai.llama-profile: medium
environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${MEDIUM_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
@@ -157,8 +150,6 @@ services:
- "0.8"
- --top-k
- "20"
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device
- CUDA0
- --split-mode
@@ -170,7 +161,6 @@ services:
labels:
com.mike-ai.llama-profile: long
environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${LONG_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
@@ -221,8 +211,6 @@ services:
- "0.8"
- --top-k
- "20"
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device
- CUDA0
- --split-mode
@@ -276,8 +264,6 @@ services:
- all
- --no-mmap
- --no-ui
- --mcp-servers-config
- /etc/mike-ai/mcp-standard.json
- --device
- CUDA0
- --split-mode
@@ -355,26 +341,19 @@ services:
OPENAI_API_BASE_URLS: http://router:8081/v1
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
# Seed native MCP connections on a fresh Open WebUI database. Secrets
# stay inside the tool containers, so these internal URLs need no keys.
TOOL_SERVER_CONNECTIONS: >-
[{"url":"http://mike-ai-mcp-web:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"web-local","name":"Web (lokal)","description":"Kompakte Websuche und Quellenvergleich"}},{"url":"http://mike-ai-mcp-homeassistant:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"homeassistant-local","name":"Home Assistant (lokal)","description":"Home-Assistant-Werkzeuge mit serverseitigem Token"}},{"url":"http://mike-ai-mcp-arr:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"arr-local","name":"ARR (lokal)","description":"Sonarr- und Radarr-Werkzeuge"}},{"url":"http://mike-ai-mcp-unraid-official:8000/mcp","path":"","type":"mcp","auth_type":"none","headers":null,"key":"","config":{"enable":true,"access_grants":[]},"info":{"id":"unraid-readonly-local","name":"Unraid (lokal, read-only)","description":"Begrenzte Unraid-Diagnose"}}]
DO_NOT_TRACK: "true"
SCARF_NO_ANALYTICS: "true"
ports:
- "${AI_BIND_ADDRESS:-127.0.0.1}:8080:8080"
dns: ["${AI_DNS:-1.1.1.1}"]
networks: [frontend]
networks: [frontend, tools]
depends_on: [router]
security_opt: ["no-new-privileges:true"]
searxng:
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
container_name: mike-ai-searxng
restart: unless-stopped
volumes:
- ./platform/web-search/searxng-settings.yml:/etc/searxng/settings.yml:ro
networks: [search]
dns: ["${AI_DNS:-1.1.1.1}"]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
networks:
frontend:
internal: false
@@ -388,10 +367,9 @@ networks:
internal: true
ipam:
config: [{subnet: 172.30.30.0/24}]
search:
internal: false
ipam:
config: [{subnet: 172.30.40.0/24}]
tools:
external: true
name: mike-ai-tools
volumes:
open-webui-data:
+27 -21
View File
@@ -22,7 +22,11 @@ Heimnetz / VPN-Clients
+-- llama-medium > exakt einer aktiv
+-- llama-long --/
+-- llama-experimental
+-- SearXNG + Web-MCP
+-- internes MCP-Netz
+-- Web-MCP + TinySearch + SearXNG
+-- Home-Assistant-MCP-Relay
+-- ARR-MCP
+-- Unraid-MCP
```
## Container und Vertrauensgrenzen
@@ -33,7 +37,8 @@ Heimnetz / VPN-Clients
| Profile Router | nur WireGuard, Port 8081 | OpenAI-API und Profilwahl |
| Profile Controller | nein | startet ausschließlich vier bekannte Profile |
| llama.cpp Profile | nein | Inferenz, Tool Calling, integrierte Vision |
| SearXNG | nein | Websuche für den lokalen Web-MCP |
| MCP-Tool-Stack | nein | voneinander getrennte Werkzeugbereiche |
| SearXNG/TinySearch | nein | Suchbackend des Web-MCP |
Nur der Profile Controller erhält den Docker-Socket. Der Router erhält weder
Socket noch Shell-Zugriff und kann dem Controller nur `fast`, `medium`, `long`
@@ -73,35 +78,36 @@ nicht automatisch in die Produktionsprofile aufgenommen.
Internetzugang der KI über zuhause laufen, braucht der Heim-Peer zusätzlich
IP-Forwarding und NAT ins Heim-WAN.
## Nicht automatisch installiert
## Optionale Erweiterungen
Bildgenerierung, Whisper, TTS sowie Home-Assistant-, ARR- und Unraid-MCPs sind
Erweiterungen. Sie benötigen eigene Modelle, Rechte oder Secrets und bleiben
im sauberen Basissystem deaktiviert. Multimodale Bildanalyse erfolgt direkt
über Qwen plus Projektor. Nicht installierte Worker-Endpunkte antworten klar
mit `feature_disabled`, statt alte systemd-Pfade aufzurufen.
Bildgenerierung, Whisper und TTS benötigen eigene Modelle und bleiben im
Basissystem deaktiviert. Home Assistant, ARR und Unraid sind vorbereitete
MCP-Profile: Sie werden erst gestartet, wenn die jeweilige root-only
Secret-Datei vorhanden ist. Multimodale Bildanalyse erfolgt direkt über Qwen
plus Projektor. Nicht installierte Worker-Endpunkte antworten klar mit
`feature_disabled`, statt alte systemd-Pfade aufzurufen.
## Zentrale MCP-Werkzeugebene
Werkzeuge werden nicht fest in Open WebUI, Hermes oder einen anderen Client
eingebaut. Sie laufen als zentrale, über WireGuard erreichbare MCP-Server. Alle
MCP-fähigen Oberflächen verwenden dadurch dieselben geprüften Werkzeuge, ohne
Secrets oder Installationen zu duplizieren.
Werkzeuge werden nicht in llama.cpp eingebaut. Sie laufen als eigene,
zentrale MCP-Container. Open WebUI greift intern darauf zu. Für externe Clients
wie Hermes wird später ein authentifizierter MCP-Gateway über WireGuard
vorgeschaltet; die unauthentifizierten internen Ports werden niemals direkt
veröffentlicht. So können alle Oberflächen dieselben geprüften Werkzeuge
verwenden, ohne Secrets zu duplizieren.
Die Trenneinheit ist **ein Container pro Fachbereich und Vertrauensstufe** –
nicht ein Container pro einzelner Funktion und nicht ein gemeinsamer
Allzweck-MCP mit sämtlichen Zugangsdaten.
```text
Open WebUI ──┐
Hermes Agent ├── mcp-gateway ──┬── web-mcp
weitere MCP- ┘ ├── home-assistant-mcp-read
Clients ├── home-assistant-mcp-write
├── arr-mcp-read
├── arr-mcp-write
├── unraid-mcp-read
├── unraid-mcp-admin
└── sandbox-mcp
Open WebUI ── internes Netz ───────────┬── web-mcp
├── home-assistant-mcp
├── arr-mcp
└── unraid-mcp-read
Hermes Agent ─ WireGuard ─┐
weitere MCP-Clients ──────┴── mcp-gateway (später) ── dasselbe interne Netz
```
| Container | Werkzeugbereich | Standardrecht |
+10 -8
View File
@@ -5,17 +5,19 @@
| AI Profile Router | `router/` | vollständig | Kern |
| llama.cpp | ggml-org/llama.cpp, festgeschriebener Commit | Buildskript und Commit | Kern |
| Qwen-Profile | `platform/profiles/` | vollständig, Modelle ausgenommen | Kern |
| Websuche | TinySearch + SearXNG | Compose und sichere Grundkonfiguration | Kern |
| Web-MCP-Fassade | `platform/web-search/web_search_mcp.py` | vollständig | Kern |
| Home-Assistant-MCP | separates privates Repository | nur Integration dokumentiert | optional |
| ARR-MCP | separates privates Repository | nur Integration dokumentiert | optional |
| Unraid-MCP | separates Repository/Installation | read-only Integration dokumentiert | optional |
| MCP-Tool-Stack | `platform/mcp/compose.yaml` | vollständig | Kern |
| Websuche | TinySearch + SearXNG | intern, ohne veröffentlichten Port | Kern |
| Web-MCP-Fassade | `platform/web-search/web_search_mcp.py` | eigener Container | Kern |
| Home-Assistant-MCP | HA-Endpunkt plus lokaler Relay | eigener optionaler Container | optional |
| ARR-MCP | `arr-mcp` 1.0.1 plus dokumentierter Sonarr-Patch | eigener optionaler Container | optional |
| Unraid-MCP | lokales `runraid`-Binary | eigener optionaler Container | optional |
| Whisper | ggml-org/whisper.cpp | Service im Router-Deploy | optional |
| XTTS-v2 | Coqui | Worker, Service und Lockdatei | optional |
| FLUX.2 klein | Black Forest Labs | Worker und Modellmanifest | optional |
| LLama-GUI | separates Upstream-Projekt | nur Betriebsrolle dokumentiert | optional |
| Glances | Distribution | nur Betriebsrolle dokumentiert | optional |
Separate MCP-Repositories werden nicht in dieses Repository kopiert. Ihre
Versionen sollen künftig in einem Release-Manifest referenziert werden. So
bleiben Zuständigkeiten klar und Updates können unabhängig getestet werden.
Upstream-Komponenten werden nicht ungeprüft einkopiert. Images, Python-Pakete
und lokale Patches sind in Dockerfiles, Compose-Mounts und Dokumentation
explizit benannt. So bleiben Zuständigkeiten klar und Updates können
unabhängig getestet werden.
+1 -1
View File
@@ -31,7 +31,7 @@ Zielplattform.
| Hauptdienst | `mike-ai-llama-ui.service` |
| llama.cpp-Port | 8080, auf dem alten Host noch im LAN gebunden |
| Client-Port | 8081 über den Router |
| MCP-Konfiguration | `/etc/mike-ai/mcp-servers.json` |
| MCP-Konfiguration | getrennte Container unter `/opt/mike-ai/mcp-containers` |
### Aktives Fast-Profil
+27
View File
@@ -74,6 +74,33 @@ Zusätzlich prüfen: Uni-LAN sieht keine KI-Ports; Heimnetz erreicht beide;
gestopptes WireGuard lässt KI-Container nicht ins Internet; jeder Profilwechsel
startet exakt einen llama-Container; Text, Tool Call und Bild funktionieren.
## Werkzeug-Container
Der Installer startet Websuche automatisch in einem privaten Docker-Netz.
Weitere Bereiche werden nur aktiviert, wenn ihre root-only Konfiguration schon
vorhanden ist:
```text
/etc/mike-ai/homeassistant-admin-mcp.env
/etc/mike-ai/arr-mcp.env
/etc/mike-ai/runraid/.env
/usr/local/bin/runraid Version 0.4.2
```
Nach dem Nachreichen einer Datei genügt:
```bash
sudo /opt/mike-ai/stack/platform/mcp/install-tools.sh
```
Auf einer frischen Open-WebUI-Datenbank werden die internen MCP-Adressen über
`TOOL_SERVER_CONNECTIONS` vorbelegt. Bei einer übernommenen Datenbank müssen
die Einträge einmal unter **Admin-Einstellungen → Externe Werkzeuge** geprüft
oder importiert werden. Die Endpunkte stehen in `platform/mcp/README.md`.
Kein MCP-Port wird auf dem Host veröffentlicht. Externe Clients wie Hermes
benötigen später den authentifizierten WireGuard-Gateway und dürfen nicht
direkt auf das interne Werkzeugnetz zugreifen.
Die öffentliche Standardkonfiguration nutzt `UD-IQ4_XS`. Das bislang schnellste
Referenzprofil nutzt dagegen die lokal vorhandene `IQ4-MIX`-Datei. Für eine
bitgenaue Migration diese Datei anhand der in `CURRENT_REFERENCE.md`
+18 -6
View File
@@ -36,17 +36,24 @@ Pflichtrollen:
- FLUX.2 klein
- XTTS-v2 und verwendete Stimme
## 2. Externe Komponenten und Commits – offen
## 2. Externe Komponenten und Commits – teilweise gesichert
Für jedes separate Projekt benötigen wir Repository und Commit:
Im Repository gesichert sind inzwischen:
- getrennte MCP-Container und internes Netz
- Web-MCP-Fassade sowie gepinnte TinySearch-/SearXNG-Images
- ARR-MCP 1.0.1 und der aktuell eingesetzte kompakte Sonarr-Patch
- Home-Assistant-Relay ohne eingebettetes Token
- Startlogik und Health-Checks
Noch extern zu beschaffen und exakt festzuhalten sind:
- Home-Assistant-MCP
- ARR-MCP
- Unraid read-only MCP
- `runraid` 0.4.2 für den read-only Unraid-MCP
- gegebenenfalls eigener Unraid-Administrations-MCP
- LLama-GUI, falls sie erhalten bleibt
Jede Komponente bekommt zusätzlich:
Jede noch externe Komponente bekommt zusätzlich:
- Installationsbefehl
- Systembenutzer
@@ -156,7 +163,7 @@ Festlegen, welche Daten persistent sein sollen:
- Benchmarkresultate: eigenes Repository
- Logs: ohne Prompt- und Tool-Antwortinhalte
## 9. Ende-zu-Ende-Installer – implementiert, Hardware-Abnahme offen
## 9. Ende-zu-Ende-Installer – weitgehend implementiert, Praxistest offen
Der Ablauf ist jetzt in `install.sh` zusammengeführt:
@@ -170,6 +177,11 @@ enable-selected-mcp-profiles
run-acceptance-tests
```
`platform/mcp/install-tools.sh` installiert den Webbereich automatisch und
aktiviert HA, ARR und Unraid nur bei vorhandenen Secret-/Programmdateien. Offen
bleiben ein kompletter Leerhost-Probelauf und der automatisierte Import einer
bereits bestehenden Open-WebUI-Datenbank.
Jeder Schritt muss wiederholbar, einzeln prüfbar und bei Fehlern abbrechbar
sein. Ein fehlgeschlagener Schritt darf keinen halb aktivierten Dienst
hinterlassen.
+7 -1
View File
@@ -21,7 +21,13 @@ verlässt sich nicht allein auf UFW.
- Profile Controller: einzige Socket-Ausnahme; feste Profile und nur
List/Start/Stop, keine frei wählbaren Images, Befehle oder Mounts.
- Open WebUI: einziges persistentes Chat-Volume.
- SearXNG: intern, Suchanfragen ohne Chatverlauf.
- MCP-Fachcontainer: intern, getrennte Secrets und keine Host-Ports.
- TinySearch/SearXNG: intern, Suchanfragen ohne Chatverlauf.
llama.cpp bekommt weder MCP-Konfiguration noch HA-, ARR- oder Unraid-Secrets.
Open WebUI kennt nur interne MCP-URLs; Authentisierung zu den Zielsystemen
findet im jeweiligen Fachcontainer statt. Externe MCP-Clients werden erst über
einen authentifizierten WireGuard-Gateway zugelassen.
Ein Docker-Socket bleibt grundsätzlich privilegiert. Der Controller reduziert
die erreichbare Funktion stark, ersetzt aber keine zusätzliche Socket-Proxy-
+5 -1
View File
@@ -232,7 +232,7 @@ set -euo pipefail
WG=$WG_INTERFACE
HOME_NET=$WG_HOME_SUBNET
TABLE=51820
for NET in 172.30.10.0/24 172.30.30.0/24 172.30.40.0/24; do
for NET in 172.30.10.0/24 172.30.30.0/24 172.30.40.0/24 172.30.50.0/24; do
ip rule add from \"\$NET\" table \"\$TABLE\" priority 12000 2>/dev/null || true
done
ip route replace \"\$HOME_NET\" dev \"\$WG\"
@@ -280,6 +280,10 @@ build_and_start() {
cd "$STACK_DIR"
docker build --build-arg LLAMA_CPP_COMMIT="$commit" \
-f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local .
# Creates the shared internal tools network before Open WebUI is created.
# Web search always starts; HA/ARR/Unraid only start when their root-only
# secret files and required local artifacts are present.
"$STACK_DIR/platform/mcp/install-tools.sh"
docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \
llama-fast llama-medium llama-long llama-experimental
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
-12
View File
@@ -1,12 +0,0 @@
{
"mcpServers": {
"web": {
"command": "/usr/bin/python3",
"args": ["/opt/mike-ai/mcp/web_search_mcp.py"],
"env": {
"SEARXNG_URL": "http://searxng:8080"
},
"timeout_ms": 120000
}
}
}
+16 -20
View File
@@ -10,30 +10,26 @@ REPO_ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
PLATFORM="$REPO_ROOT/platform"
PROFILE_TARGET=/opt/mike-ai/platform/profiles
DROPIN=/etc/systemd/system/mike-ai-llama-ui.service.d
WEB_TARGET=/opt/mike-ai/web-search
MCP_ROOT=/opt/mike-ai/mcp-containers
MCP_TARGET=$MCP_ROOT/platform/mcp
MCP_SEARCH_TARGET=$MCP_ROOT/platform/web-search
install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" "$WEB_TARGET" /etc/mike-ai
install -d -m 0755 "$PROFILE_TARGET" "$DROPIN" \
"$MCP_TARGET" "$MCP_ROOT/platform/web-search" /etc/mike-ai
install -m 0644 "$PLATFORM/systemd/mike-ai-llama-ui.service" \
/etc/systemd/system/mike-ai-llama-ui.service
install -m 0644 "$PLATFORM/systemd/mike-ai-web-search.service" \
/etc/systemd/system/mike-ai-web-search.service
install -m 0755 "$PLATFORM/scripts/llama-profile" /usr/local/bin/llama-profile
install -m 0644 "$PLATFORM/profiles/profile-fast.conf" "$PROFILE_TARGET/profile-fast.conf"
install -m 0644 "$PLATFORM/profiles/profile-medium.conf" "$PROFILE_TARGET/profile-medium.conf"
install -m 0644 "$PLATFORM/profiles/profile-long.conf" "$PROFILE_TARGET/profile-long.conf"
install -m 0644 "$PLATFORM/web-search/compose.yaml" "$WEB_TARGET/compose.yaml"
install -m 0644 "$PLATFORM/web-search/tinysearch_config.json" \
"$WEB_TARGET/tinysearch_config.json"
install -m 0755 "$PLATFORM/web-search/web_search_mcp.py" \
"$WEB_TARGET/web_search_mcp.py"
if [[ ! -e "$WEB_TARGET/searxng-settings.yml" ]]; then
install -m 0600 "$PLATFORM/web-search/searxng-settings.example.yml" \
"$WEB_TARGET/searxng-settings.yml.example"
fi
if [[ ! -e /etc/mike-ai/mcp-servers.json ]]; then
install -m 0600 "$PLATFORM/mcp/mcp-servers.example.json" \
/etc/mike-ai/mcp-servers.json.example
rsync -a --delete "$PLATFORM/mcp/" "$MCP_TARGET/"
rsync -a --delete --exclude searxng-settings.yml \
"$PLATFORM/web-search/" "$MCP_SEARCH_TARGET/"
if [[ ! -s $MCP_SEARCH_TARGET/searxng-settings.yml ]]; then
install -m 0640 "$PLATFORM/web-search/searxng-settings.example.yml" \
"$MCP_SEARCH_TARGET/searxng-settings.yml"
sed -i "s/CHANGE_ME_GENERATE_RANDOM_SECRET/$(openssl rand -hex 32)/" \
"$MCP_SEARCH_TARGET/searxng-settings.yml"
fi
systemctl daemon-reload
@@ -43,7 +39,7 @@ Kernkonfiguration installiert, aber noch nicht gestartet.
Vor dem Start:
1. Modellpfade und Hashes gegen manifest.local.yaml prüfen.
2. /etc/mike-ai/mcp-servers.json mit minimalen Servern erstellen.
3. llama.cpp bauen.
4. Danach: llama-profile fast
2. llama.cpp bauen und mit `llama-profile fast` starten.
3. Fach-Secrets unter /etc/mike-ai ablegen.
4. Danach: /opt/mike-ai/mcp-containers/platform/mcp/install-tools.sh
EOF
+12
View File
@@ -0,0 +1,12 @@
FROM python:3.13-slim AS builder
COPY --from=ghcr.io/astral-sh/uv:0.11.7 /uv /uvx /bin/
RUN uv pip install --system --break-system-packages "arr-mcp[mcp]==1.0.1"
FROM python:3.13-slim
COPY --from=builder /usr/local /usr/local
RUN groupadd --system --gid 10001 mcp \
&& useradd --system --uid 10001 --gid 10001 --no-create-home mcp
USER 10001:10001
EXPOSE 8000
ENTRYPOINT ["arr-mcp"]
CMD ["--transport", "streamable-http", "--host", "0.0.0.0", "--port", "8000", "--auth-type", "none"]
@@ -0,0 +1,8 @@
FROM nginx:1.29-alpine
COPY platform/mcp/homeassistant.conf.template /etc/nginx/templates/homeassistant.conf.template
COPY platform/mcp/ha-relay-entrypoint.sh /usr/local/bin/ha-relay-entrypoint
RUN chmod 0755 /usr/local/bin/ha-relay-entrypoint \
&& mkdir -p /tmp/client_temp /tmp/proxy_temp \
&& chown -R nginx:nginx /tmp/client_temp /tmp/proxy_temp
EXPOSE 8000
ENTRYPOINT ["/usr/local/bin/ha-relay-entrypoint"]
+15
View File
@@ -0,0 +1,15 @@
FROM python:3.13-slim
ARG MCP_PROXY_VERSION=0.12.0
RUN apt-get update \
&& apt-get install -y --no-install-recommends openssh-client \
&& rm -rf /var/lib/apt/lists/* \
&& pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp>=1.17,<2" \
&& useradd --system --uid 10001 --create-home --home-dir /app mcp
RUN touch /app/unraid_mcp.py && chown 10001:10001 /app/unraid_mcp.py
USER 10001:10001
WORKDIR /app
EXPOSE 8000
ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"]
CMD ["python", "/app/unraid_mcp.py"]
+14
View File
@@ -0,0 +1,14 @@
FROM python:3.13-slim
ARG MCP_PROXY_VERSION=0.12.0
RUN pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp>=1.17,<2"
RUN useradd --system --uid 10001 --create-home --home-dir /app mcp
COPY web-search/web_search_mcp.py /app/web_search_mcp.py
RUN chown -R 10001:10001 /app
USER 10001:10001
WORKDIR /app
EXPOSE 8000
ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"]
CMD ["python", "/app/web_search_mcp.py"]
+61 -30
View File
@@ -1,43 +1,74 @@
# MCP-Architektur
# Zentrale MCP-Werkzeugebene
Die produktive MCP-Konfiguration ist absichtlich nicht Bestandteil des Git-
Repositories, weil sie lokale Pfade und Zugangsdaten referenziert. Das Beispiel
zeigt nur die Struktur.
MCP-Werkzeuge sind **keine llama.cpp-Startparameter**. Sie laufen als kleine,
voneinander getrennte Container und werden von OpenWebUI, Hermes oder einem
anderen MCP-Client gezielt ausgewählt. Das hält Tool-Schemas aus normalen
Prompts heraus, verhindert den früher beobachteten Kontextverbrauch von über
200.000 Tokens und macht Werkzeuge unabhängig vom geladenen Modellprofil.
## Empfohlene Server
## Container
- `web`: Websuche über lokales TinySearch/SearXNG
- `homeassistant`: Administration mit eigenem, minimal berechtigtem Token
- `arr`: Sonarr/Radarr über spezialisierte Aktionen
- `unraid-readonly`: Diagnose ohne Schreiboperationen
| Container | Endpunkt im Netz `mike-ai-tools` | Zweck | Standard |
|---|---|---|---|
| `mcp-web` | `http://mike-ai-mcp-web:8000/mcp` | kompakte Websuche und Quellenvergleich | an |
| `mcp-homeassistant` | `http://mike-ai-mcp-homeassistant:8000/mcp` | Relay zum nativen HA-MCP; Token bleibt serverseitig | Profil `homeassistant` |
| `mcp-arr` | `http://mike-ai-mcp-arr:8000/mcp` | Sonarr/Radarr/Prowlarr mit serverseitiger Policy | Profil `arr` |
| `mcp-unraid-official` | `http://mike-ai-mcp-unraid-official:8000/mcp` | offizieller, read-only begrenzter Unraid-Zugang | Profil `unraid` |
| `mcp-unraid-ssh` | `http://mike-ai-mcp-unraid-ssh:8000/mcp` | erweiterte Diagnose über einen erzwungenen SSH-Befehl | optional (`extended`) |
## Getrennte Konfigurationen
TinySearch und SearXNG sind interne Abhängigkeiten des Web-MCPs und werden
nicht direkt als allgemeine Werkzeuge angeboten.
Statt alle Werkzeuge ständig zu laden, werden mehrere Dateien empfohlen:
## Sicherheitsmodell
```text
/etc/mike-ai/mcp-standard.json
/etc/mike-ai/mcp-homeassistant.json
/etc/mike-ai/mcp-arr.json
/etc/mike-ai/mcp-unraid-readonly.json
/etc/mike-ai/mcp-unraid-write.json
- Kein MCP-Port wird auf eine Host-Adresse veröffentlicht.
- Nur Clients im privaten Docker-Netz `mike-ai-tools` erreichen die Endpunkte.
- Secrets bleiben in Dateien unter `/etc/mike-ai` und werden read-only
eingehängt. Sie gehören weder in Git noch in OpenWebUI-Tooldefinitionen.
- Jeder Container ist read-only, verliert Linux-Capabilities und hat
`no-new-privileges`.
- Der SSH-basierte Unraid-Container ist nicht Teil des Standardstarts.
- Ein allgemeiner Host-Shell-MCP wird bewusst nicht angeboten.
## Start
```bash
sudo platform/mcp/install-tools.sh
```
Das jeweilige Profil verweist nur auf die benötigte Datei. Dadurch werden die
Tool-Schemas kleiner, das Kontextfenster bleibt frei und kleine Modelle müssen
weniger Werkzeuge unterscheiden.
Der Grundstart enthält nur Websuche. Bereits konfigurierte Fachbereiche werden
explizit ergänzt:
Credentials werden von schmalen Wrapper-Programmen wie `run-arr-mcp` oder
`runraid` aus geschützten Environment-Dateien geladen. Das JSON selbst enthält
weder Werte noch Pfade zu einzelnen Tokens.
Das Skript erkennt vorhandene Secret-Dateien und aktiviert dadurch automatisch
`homeassistant`, `arr` und `unraid`. Ohne Fach-Secrets startet nur der sichere
Webbereich.
## Schreibzugriff
Für den derzeit migrierten Container kann der Name `Open-WebUI` lauten. Der
Netzwerkbefehl ist idempotent zu behandeln.
Schreibende Server gehören nicht in `mcp-standard.json`. Sie benötigen eine
Vorschau und ein an die exakte Änderung gebundenes Approval Ticket.
Die lokale Installation benötigt die vorhandenen Secret-Dateien:
## Shell
```text
/etc/mike-ai/homeassistant-admin-mcp.env
/etc/mike-ai/arr-mcp.env
/etc/mike-ai/runraid/.env
```
Ein allgemeiner Shell-MCP ist nicht Teil der Zielplattform. Insbesondere
`python3`, `ssh`, `scp`, `curl` und `systemctl` dürfen nicht gemeinsam als
scheinbar harmlose Allowlist angeboten werden.
Die erweiterte Unraid-Diagnose benötigt zusätzlich die Konfigurationsdatei,
den eingeschränkten Schlüssel und die bekannte Hostsignatur. Sie wird nur mit
`--profile extended` gestartet.
TinySearch speichert sein lokales Embedding-Modell in einem Docker-Volume.
Nach einer Erstinstallation wird das Modell einmalig im Container mit
`tinysearch setup` geladen. Das Volume bleibt bei Containerupdates erhalten.
## Client-Auswahl
Werkzeuge werden nicht pauschal an jedes Modell gehängt. Für Home-Assistant-
Fragen wird HA ausgewählt, für Medien ARR, für Recherche Web und für die NAS
Unraid. Mehrere Werkzeuge werden nur aktiviert, wenn die Aufgabe tatsächlich
mehrere Bereiche verbindet.
Schreibende Aktionen bleiben hinter der jeweiligen serverseitigen Policy und
einem Vorschau-/Bestätigungsablauf. Ein Client-Schalter allein darf niemals
eine read-only Policy aufheben.
+147
View File
@@ -0,0 +1,147 @@
name: mike-ai-tools
x-tool-common: &tool-common
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:rw,noexec,nosuid,nodev,size=64m
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
networks: [tools]
logging:
options:
max-size: 10m
max-file: "3"
services:
mcp-web:
<<: *tool-common
build:
context: ..
dockerfile: mcp/Dockerfile.web
image: mike-ai/mcp-web:local
container_name: mike-ai-mcp-web
environment:
TINYSEARCH_MCP_URL: http://tinysearch:8000/mcp
SEARXNG_URL: http://searxng:8080
WEB_SEARCH_BUDGET_MAX_RELATED: "6"
depends_on:
tinysearch:
condition: service_started
searxng:
<<: *tool-common
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
container_name: mike-ai-tools-searxng
volumes:
- ${SEARXNG_SETTINGS_FILE:-../web-search/searxng-settings.example.yml}:/etc/searxng/settings.yml:ro
networks: [tools, egress]
tinysearch:
<<: *tool-common
image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c
container_name: mike-ai-tools-tinysearch
shm_size: 1gb
volumes:
- tinysearch-models:/data/models
- ../web-search/tinysearch_config.json:/config/tinysearch_config.json:ro
environment:
MCP_TRANSPORT: streamable-http
MCP_HOST: 0.0.0.0
MCP_PORT: "8000"
TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json
TINYSEARCH_SEARCH_BACKEND: searxng
SEARXNG_URL: http://searxng:8080/search
depends_on: [searxng]
cap_add: [SETUID, SETGID, CHOWN]
networks: [tools, egress]
# The image's built-in `tinysearch doctor` also requires a writable
# configuration directory, although normal server operation does not.
# Check the service socket instead so read-only hardening remains intact.
healthcheck:
test: ["CMD", "python", "-c", "import socket; s=socket.create_connection(('127.0.0.1', 8000), 2); s.close()"]
interval: 30s
timeout: 5s
retries: 5
start_period: 20s
mcp-homeassistant:
<<: *tool-common
build:
context: ../..
dockerfile: platform/mcp/Dockerfile.homeassistant-relay
image: mike-ai/mcp-homeassistant-relay:local
container_name: mike-ai-mcp-homeassistant
profiles: [homeassistant]
volumes:
- ${HA_ENV_FILE:-/etc/mike-ai/homeassistant-admin-mcp.env}:/run/secrets/homeassistant.env:ro
cap_add: [CHOWN, SETUID, SETGID]
networks: [tools, egress]
mcp-arr:
<<: *tool-common
build:
context: .
dockerfile: Dockerfile.arr
image: mike-ai/mcp-arr:1.0.1-patched
container_name: mike-ai-mcp-arr
profiles: [arr]
env_file:
- ${ARR_ENV_FILE:-/etc/mike-ai/arr-mcp.env}
volumes:
# The local fork adds bounded read-only Sonarr pseudo-actions. Keep the
# patch explicit until upstream publishes a self-contained 2.x image.
- ${ARR_SONARR_PATCH:-./patches/mcp_sonarr.py}:/usr/local/lib/python3.13/site-packages/arr_mcp/mcp/mcp_sonarr.py:ro
networks: [tools, egress]
mcp-unraid-official:
<<: *tool-common
image: debian:13-slim
container_name: mike-ai-mcp-unraid-official
profiles: [unraid]
env_file:
- ${RUNRAID_ENV_FILE:-/etc/mike-ai/runraid/.env}
environment:
UNRAID_RMCP_HOST: 0.0.0.0
UNRAID_RMCP_PORT: "8000"
UNRAID_RMCP_DISABLE_HTTP_AUTH: "true"
UNRAID_NOAUTH: "true"
UNRAID_RMCP_ALLOWED_HOSTS: "mike-ai-mcp-unraid-official:8000,mike-ai-mcp-unraid-official,localhost:8000,127.0.0.1:8000"
volumes:
- ${RUNRAID_BINARY:-/usr/local/bin/runraid}:/usr/local/bin/unraid:ro
entrypoint: ["/usr/local/bin/unraid"]
command: ["serve"]
networks: [tools, egress]
mcp-unraid-ssh:
<<: *tool-common
profiles: [extended]
build:
context: .
dockerfile: Dockerfile.unraid-ssh
image: mike-ai/mcp-unraid-ssh:local
container_name: mike-ai-mcp-unraid-ssh
environment:
UNRAID_MCP_CONFIG: /run/config/unraid-mcp.json
volumes:
- ${UNRAID_MCP_SOURCE:-/opt/mike-ai/unraid-agent/unraid_mcp.py}:/app/unraid_mcp.py:ro
- ${UNRAID_MCP_CONFIG:-/etc/mike-ai/unraid-mcp.json}:/run/config/unraid-mcp.json:ro
- ${UNRAID_SSH_KEY:-/etc/mike-ai/keys/unraid_root}:/etc/mike-ai/keys/unraid_root:ro
- ${UNRAID_KNOWN_HOSTS:-/etc/mike-ai/ssh/known_hosts_unraid_ai}:/etc/mike-ai/ssh/known_hosts_unraid_ai:ro
- unraid-audit:/var/log/mike-ai
networks: [tools, egress]
networks:
tools:
name: mike-ai-tools
internal: true
ipam:
config: [{subnet: 172.30.40.0/24}]
egress:
name: mike-ai-tools-egress
ipam:
config: [{subnet: 172.30.50.0/24}]
volumes:
tinysearch-models:
unraid-audit:
+23
View File
@@ -0,0 +1,23 @@
#!/bin/sh
set -eu
config=/run/secrets/homeassistant.env
if [ ! -r "$config" ]; then
echo "Home Assistant secret file is missing" >&2
exit 1
fi
set -a
. "$config"
set +a
: "${HASS_URL:?HASS_URL is required}"
: "${HASS_TOKEN:?HASS_TOKEN is required}"
upstream=${HASS_URL%/}
escaped_token=$(printf '%s' "$HASS_TOKEN" | sed 's/[&/]/\\&/g')
escaped_upstream=$(printf '%s' "$upstream" | sed 's/[&/]/\\&/g')
sed -e "s/__HASS_TOKEN__/$escaped_token/g" \
-e "s/__HASS_UPSTREAM__/$escaped_upstream/g" \
/etc/nginx/templates/homeassistant.conf.template \
> /tmp/nginx.conf
unset HASS_TOKEN
exec nginx -c /tmp/nginx.conf -g 'daemon off;'
+30
View File
@@ -0,0 +1,30 @@
worker_processes 1;
pid /tmp/nginx.pid;
error_log /dev/stderr warn;
events { worker_connections 128; }
http {
access_log /dev/stdout;
client_body_temp_path /tmp/client_temp;
proxy_temp_path /tmp/proxy_temp;
fastcgi_temp_path /tmp/fastcgi_temp;
uwsgi_temp_path /tmp/uwsgi_temp;
scgi_temp_path /tmp/scgi_temp;
proxy_buffering off;
proxy_read_timeout 600s;
proxy_send_timeout 600s;
server {
listen 8000;
location /mcp {
proxy_pass __HASS_UPSTREAM__/api/hass_mcp;
proxy_http_version 1.1;
proxy_ssl_server_name on;
proxy_ssl_name $proxy_host;
proxy_set_header Authorization "Bearer __HASS_TOKEN__";
proxy_set_header Host $proxy_host;
proxy_set_header Connection "";
}
}
}
+52
View File
@@ -0,0 +1,52 @@
#!/usr/bin/env bash
set -Eeuo pipefail
[[ $EUID -eq 0 ]] || { echo "Bitte als root ausführen." >&2; exit 1; }
MCP_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
COMPOSE=(docker compose -f "$MCP_DIR/compose.yaml")
export SEARXNG_SETTINGS_FILE="${SEARXNG_SETTINGS_FILE:-$MCP_DIR/../web-search/searxng-settings.yml}"
[[ -s $SEARXNG_SETTINGS_FILE ]] || {
echo "SearXNG-Konfiguration fehlt: $SEARXNG_SETTINGS_FILE" >&2
exit 1
}
profiles=()
if [[ -s /etc/mike-ai/homeassistant-admin-mcp.env ]]; then
profiles+=(--profile homeassistant)
else
echo "Home Assistant bleibt aus: Secret-Datei fehlt."
fi
if [[ -s /etc/mike-ai/arr-mcp.env ]]; then
profiles+=(--profile arr)
else
echo "ARR bleibt aus: Secret-Datei fehlt."
fi
if [[ -s /etc/mike-ai/runraid/.env && -x /usr/local/bin/runraid ]]; then
profiles+=(--profile unraid)
else
echo "Unraid bleibt aus: runraid 0.4.2 oder Secret-Datei fehlt."
fi
# TinySearch keeps the embedding bundle outside the container. Download it
# once on a fresh host; subsequent rebuilds reuse the named volume.
docker volume create mike-ai-tools_tinysearch-models >/dev/null
tiny_image="marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c"
if ! docker run --rm --entrypoint test \
-v mike-ai-tools_tinysearch-models:/data/models "$tiny_image" \
-f /data/models/all-minilm-l6-v2-onnx/model.onnx; then
echo "TinySearch-Modell wird einmalig geladen."
docker run --rm -v mike-ai-tools_tinysearch-models:/data/models \
"$tiny_image" tinysearch setup
fi
"${COMPOSE[@]}" "${profiles[@]}" up -d --build
for webui in mike-ai-open-webui Open-WebUI; do
if docker container inspect "$webui" >/dev/null 2>&1; then
docker network connect mike-ai-tools "$webui" 2>/dev/null || true
fi
done
"${COMPOSE[@]}" "${profiles[@]}" ps
-23
View File
@@ -1,23 +0,0 @@
{
"mcpServers": {
"web": {
"command": "/usr/bin/python3",
"args": ["/opt/mike-ai/web-search/web_search_mcp.py"]
},
"homeassistant": {
"command": "/usr/local/bin/homeassistant-native-mcp",
"args": [],
"timeout_ms": 60000
},
"arr": {
"command": "/usr/local/bin/run-arr-mcp",
"args": [],
"timeout_ms": 30000
},
"unraid-readonly": {
"command": "/usr/local/bin/runraid",
"args": ["mcp"],
"timeout_ms": 60000
}
}
}
+329
View File
@@ -0,0 +1,329 @@
"""Sonarr condensed action-routed MCP tool.
CONCEPT:ECO-4.82 — gitlab-style organized per-service tool surface.
"""
import os
import json
import re
from typing import Any
from agent_utilities.mcp_utilities import dispatch, run_blocking
from fastmcp import FastMCP
from pydantic import Field
from arr_mcp.auth import get_sonarr_client
READ_ONLY_ACTIONS = frozenset(
{
"get_system_status", "get_health", "get_diskspace", "get_ping",
"get_series", "get_series_id", "get_series_lookup", "lookup_series",
"get_episode", "get_episode_id", "get_episodefile", "get_episodefile_id",
"get_calendar", "get_calendar_id", "get_history", "get_history_series",
"get_history_since", "get_queue", "get_queue_details", "get_queue_status",
"get_wanted_missing", "get_wanted_missing_id", "get_wanted_cutoff",
"get_wanted_cutoff_id", "get_qualityprofile", "get_qualityprofile_id",
"get_languageprofile", "get_languageprofile_id", "get_tag", "get_tag_id",
"get_tag_detail", "get_tag_detail_id", "get_command", "get_command_id",
"get_release",
}
)
PSEUDO_ACTIONS = frozenset({"find_series", "get_season_summary", "search_releases"})
WRITE_ACTIONS = frozenset({
"post_command",
"post_release",
"put_episode_id",
"put_episode_monitor",
"put_series_id",
"put_series",
"put_wanted",
"post_wanted",
})
MAX_COLLECTION_ITEMS = 50
def _plain(value: Any) -> Any:
if hasattr(value, "model_dump") and callable(value.model_dump):
return value.model_dump()
if hasattr(value, "dict") and callable(value.dict):
return value.dict()
if isinstance(value, list):
return [_plain(item) for item in value]
if isinstance(value, dict):
return {str(key): _plain(item) for key, item in value.items()}
return value
def _unwrap(value: Any) -> Any:
value = _plain(value)
if isinstance(value, dict) and set(value) == {"result"}:
return value["result"]
return value
def _pick(item: dict[str, Any], fields: tuple[str, ...]) -> dict[str, Any]:
return {field: item[field] for field in fields if item.get(field) is not None}
def _compact_series(item: dict[str, Any], include_seasons: bool = False) -> dict[str, Any]:
result = _pick(
item,
("id", "title", "sortTitle", "year", "status", "monitored", "path", "tvdbId"),
)
statistics = item.get("statistics") or {}
if isinstance(statistics, dict):
result["statistics"] = _pick(
statistics,
("seasonCount", "episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"),
)
if include_seasons:
result["seasons"] = [
{
**_pick(season, ("seasonNumber", "monitored")),
"statistics": _pick(
season.get("statistics") or {},
("episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"),
),
}
for season in item.get("seasons", [])
if isinstance(season, dict)
]
return result
def _compact_episode(item: dict[str, Any]) -> dict[str, Any]:
return _pick(
item,
("id", "seriesId", "seasonNumber", "episodeNumber", "title", "airDate", "airDateUtc", "monitored", "hasFile", "episodeFileId"),
)
def _compact_file(item: dict[str, Any]) -> dict[str, Any]:
quality = item.get("quality") or {}
quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None
result = _pick(
item,
("id", "seriesId", "seasonNumber", "relativePath", "path", "size", "dateAdded", "releaseGroup"),
)
if quality_name:
result["quality"] = quality_name
return result
def _compact_release(item: dict[str, Any]) -> dict[str, Any]:
quality = item.get("quality") or {}
quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None
result = _pick(
item,
(
"guid", "title", "indexer", "indexerId", "size", "age", "ageHours",
"seeders", "leechers", "protocol", "downloadAllowed", "releaseWeight",
),
)
if quality_name:
result["quality"] = quality_name
rejections = item.get("rejections")
if isinstance(rejections, list) and rejections:
result["rejections"] = [str(reason)[:180] for reason in rejections[:5]]
return result
def _bounded(items: list[Any], compact) -> dict[str, Any]:
total = len(items)
return {
"total": total,
"returned": min(total, MAX_COLLECTION_ITEMS),
"truncated": total > MAX_COLLECTION_ITEMS,
"items": [compact(item) for item in items[:MAX_COLLECTION_ITEMS] if isinstance(item, dict)],
"next_step": (
"Use find_series or narrower Sonarr parameters; do not repeat the same broad request."
if total > MAX_COLLECTION_ITEMS else None
),
}
def _compact_result(action: str, value: Any) -> Any:
value = _unwrap(value)
if isinstance(value, list):
if action in {"get_series", "get_series_lookup", "lookup_series"}:
return _bounded(value, _compact_series)
if action in {"get_episode", "get_calendar", "get_wanted_missing", "get_wanted_cutoff"}:
return _bounded(value, _compact_episode)
if action == "get_episodefile":
return _bounded(value, _compact_file)
if action == "get_release":
return _bounded(value, _compact_release)
return _bounded(value, lambda item: item)
if isinstance(value, dict) and action in {"get_series_id"}:
return _compact_series(value, include_seasons=True)
if isinstance(value, dict) and action in {"get_episode_id"}:
return _compact_episode(value)
if isinstance(value, dict) and action in {"get_episodefile_id"}:
return _compact_file(value)
return value
async def _find_series(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]:
query = str(kwargs.get("query", "")).strip()
if len(query) < 2:
raise ValueError("find_series requires params_json with a query of at least 2 characters")
limit = max(1, min(int(kwargs.get("limit", 8)), 15))
raw = _unwrap(await run_blocking(dispatch, client, "get_series", {}, service="arr-sonarr"))
words = [word for word in re.findall(r"[a-z0-9]+", query.casefold()) if len(word) > 1]
matches = []
for item in raw if isinstance(raw, list) else []:
haystack = " ".join(
str(item.get(field, "")) for field in ("title", "sortTitle", "originalTitle", "alternateTitles")
).casefold()
if all(word in haystack for word in words):
matches.append(_compact_series(item, include_seasons=True))
return {
"query": query,
"matches": matches[:limit],
"match_count": len(matches),
"truncated": len(matches) > limit,
"task_complete": True,
"instruction": "Use the returned series id for details. Do not call get_series for discovery.",
}
async def _season_summary(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]:
series_id = int(kwargs["series_id"])
season_number = int(kwargs["season_number"])
series = _unwrap(
await run_blocking(dispatch, client, "get_series_id", {"id": series_id}, service="arr-sonarr")
)
episodes = _unwrap(
await run_blocking(
dispatch,
client,
"get_episode",
{"seriesId": series_id, "seasonNumber": season_number},
service="arr-sonarr",
)
)
files = _unwrap(
await run_blocking(
dispatch,
client,
"get_episodefile",
{"seriesId": series_id},
service="arr-sonarr",
)
)
selected_episodes = [
_compact_episode(item) for item in episodes
if isinstance(item, dict) and item.get("seasonNumber") == season_number
] if isinstance(episodes, list) else []
selected_files = [
_compact_file(item) for item in files
if isinstance(item, dict) and item.get("seasonNumber") == season_number
] if isinstance(files, list) else []
groups = sorted({str(item.get("releaseGroup")) for item in selected_files if item.get("releaseGroup")})
return {
"series": _compact_series(series) if isinstance(series, dict) else {"id": series_id},
"season_number": season_number,
"episode_count": len(selected_episodes),
"file_count": len(selected_files),
"release_groups": groups,
"episodes": selected_episodes[:30],
"files": selected_files[:30],
"task_complete": True,
"instruction": "This is the complete compact season answer. Do not repeat broad series or episode queries.",
}
async def _search_releases(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]:
series_id = kwargs.get("series_id")
episode_id = kwargs.get("episode_id")
season_number = kwargs.get("season_number")
release_group = str(kwargs.get("release_group", "")).strip()
if series_id is None and episode_id is None:
raise ValueError("search_releases requires series_id or episode_id")
query: dict[str, Any] = {}
if series_id is not None:
query["seriesId"] = int(series_id)
if episode_id is not None:
query["episodeId"] = int(episode_id)
if season_number is not None:
query["seasonNumber"] = int(season_number)
raw = await run_blocking(dispatch, client, "get_release", query, service="arr-sonarr")
raw = _unwrap(raw)
if release_group and isinstance(raw, list):
needle = release_group.casefold()
raw = [
item for item in raw
if isinstance(item, dict)
and needle in (
str(item.get("releaseGroup", "")) + " " + str(item.get("title", ""))
).casefold()
]
compact = _compact_result("get_release", raw)
return {
"task_complete": True,
"search_scope": {
"series_id": series_id,
"episode_id": episode_id,
"season_number": season_number,
"release_group_filter": release_group or None,
},
"monitoring_changed": False,
"download_started": False,
"results": compact,
"instruction": "These are Sonarr indexer results. Do not use web search to replace them. Never download unless the user separately approves a write action.",
}
def register_sonarr_tools(mcp: FastMCP) -> None:
@mcp.tool(tags={"sonarr"})
async def sonarr_action(
action: str = Field(
description="Read-only Sonarr action. Use find_series {query} for titles, get_season_summary {series_id, season_number} for holdings, and search_releases {series_id, season_number, optional release_group} to query configured Sonarr indexers without downloading or changing monitoring. Avoid broad get_series/get_episode calls."
),
params_json: str = Field(
default="{}",
description="JSON string of parameters to pass to the action.",
),
) -> Any:
"""Query Sonarr through a server-side allowlist (read-only by default; write actions when ARR_MCP_WRITE=1)."""
if action in {"list_actions", "help", "actions"}:
return {
"service": "sonarr",
"access_mode": "write" if os.environ.get("ARR_MCP_WRITE", "").strip().lower() in ("1", "true", "yes", "on") else "read-only",
"actions": sorted(READ_ONLY_ACTIONS),
"write_actions": sorted(WRITE_ACTIONS) if os.environ.get("ARR_MCP_WRITE", "").strip().lower() in ("1", "true", "yes", "on") else [],
"preferred_compact_actions": sorted(PSEUDO_ACTIONS),
}
allow_write = os.environ.get("ARR_MCP_WRITE", "").strip().lower() in (
"1", "true", "yes", "on"
)
if action in READ_ONLY_ACTIONS | PSEUDO_ACTIONS:
pass
elif allow_write and action in WRITE_ACTIONS:
pass
else:
if allow_write:
raise PermissionError(
f"Sonarr MCP write mode is enabled, but action '{action}' "
"is not in the allowed write set. Allowed: "
f"{sorted(WRITE_ACTIONS)}"
)
raise PermissionError(
f"Sonarr action '{action}' is blocked by the server-side "
"read-only policy. Set ARR_MCP_WRITE=1 to enable write mode."
)
client = get_sonarr_client()
kwargs = {k: v for k, v in json.loads(params_json).items() if v is not None}
if action == "find_series":
return await _find_series(client, kwargs)
if action == "get_season_summary":
return await _season_summary(client, kwargs)
if action == "search_releases":
return await _search_releases(client, kwargs)
result = await run_blocking(
dispatch, client, action, kwargs, service="arr-sonarr"
)
return _compact_result(action, result)
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Fast 76.8K MTP2 with CPU Vision
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
ExecStart=/opt/mike-ai/llama.cpp-nvfp4/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-76k-mtp2-vision --ctx-size 76800 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k f16 --spec-draft-type-v f16
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Long 128K MTP2 FFN12 CPU with CPU Vision
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4mix-128k-mtp2-ffn12 --ctx-size 131072 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --override-tensor blk.([0-9]|1[0-1]).ffn_.*=CPU --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none --spec-type draft-mtp --spec-draft-n-max 2 --spec-draft-type-k q4_0 --spec-draft-type-v q4_0
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 92K IQ4_XS Pure with CPU Vision
[Service]
ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --mcp-servers-config /etc/mike-ai/mcp-servers.json --device CUDA0 --split-mode none
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen38-27b-iq4xs-pure-92k --ctx-size 94208 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0 --split-mode none
@@ -1,16 +0,0 @@
[Unit]
Description=Local AI Web Search (TinySearch + SearXNG)
Requires=docker.service
After=docker.service network-online.target
Before=mike-ai-llama-ui.service
[Service]
Type=oneshot
RemainAfterExit=yes
WorkingDirectory=/opt/mike-ai/web-search
ExecStart=/usr/bin/docker compose up -d
ExecStop=/usr/bin/docker compose down
TimeoutStartSec=180
[Install]
WantedBy=multi-user.target
-45
View File
@@ -1,45 +0,0 @@
name: local-ai-web-search
services:
searxng:
image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60
restart: unless-stopped
volumes:
- ./searxng-settings.yml:/etc/searxng/settings.yml:ro
networks: [search]
healthcheck:
test: ["CMD", "wget", "-q", "--spider", "http://127.0.0.1:8080/healthz"]
interval: 30s
timeout: 10s
retries: 5
start_period: 30s
security_opt: ["no-new-privileges:true"]
tinysearch:
image: marcellm01/tinysearch@sha256:5a03d5a1f1b0fabe48f2a26e05db4a84bcb611106a57ec51e42db4549976aa9c
restart: unless-stopped
ports:
- "127.0.0.1:8000:8000"
shm_size: "1gb"
volumes:
- tinysearch-models:/data/models
- ./tinysearch_config.json:/config/tinysearch_config.json:ro
environment:
MCP_TRANSPORT: streamable-http
MCP_HOST: 0.0.0.0
MCP_PORT: 8000
TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json
TINYSEARCH_SEARCH_BACKEND: searxng
SEARXNG_URL: http://searxng:8080/search
depends_on:
searxng:
condition: service_healthy
networks: [search]
security_opt: ["no-new-privileges:true"]
networks:
search:
driver: bridge
volumes:
tinysearch-models:
+37 -82
View File
@@ -10,12 +10,11 @@ from facts verified by crawled pages or primary APIs.
from __future__ import annotations
import ipaddress
import asyncio
import html
import json
import os
import re
import select
import subprocess
import sys
import threading
import time
@@ -28,9 +27,9 @@ from xml.etree import ElementTree
SERVER_VERSION = "2.1.0"
TINYSEARCH_CONTAINER = os.environ.get(
"TINYSEARCH_CONTAINER", "mike-ai-web-search-tinysearch-1"
)
TINYSEARCH_MCP_URL = os.environ.get(
"TINYSEARCH_MCP_URL", "http://tinysearch:8000/mcp"
).rstrip("/")
SEARXNG_URL = os.environ.get("SEARXNG_URL", "").rstrip("/")
CHILD_TIMEOUT_SECONDS = float(os.environ.get("TINYSEARCH_CHILD_TIMEOUT", "110"))
HTTP_TIMEOUT_SECONDS = float(os.environ.get("WEB_API_TIMEOUT", "18"))
@@ -801,90 +800,46 @@ def general_discovery(query: str, limit: int) -> tuple[list[dict[str, Any]], lis
class TinySearchClient:
"""Minimal synchronous MCP client for TinySearch's stdio server."""
"""Synchronous facade for TinySearch's Streamable HTTP MCP endpoint.
TinySearch used to be reached by executing ``docker exec`` on the host.
That required access to the Docker socket and coupled this service to a
particular container name. The containerized tool layer instead talks to
TinySearch over the private Docker network.
"""
def __init__(self) -> None:
self._next_id = 1
self._lock = threading.Lock()
self._process = subprocess.Popen(
[
"/usr/bin/docker",
"exec",
"-e",
"MCP_TRANSPORT=stdio",
"-i",
TINYSEARCH_CONTAINER,
"tinysearch",
"mcp",
],
stdin=subprocess.PIPE,
stdout=subprocess.PIPE,
stderr=subprocess.DEVNULL,
text=True,
encoding="utf-8",
errors="replace",
bufsize=1,
)
self._request(
"initialize",
{
"protocolVersion": "2024-11-05",
"capabilities": {},
"clientInfo": {"name": "mike-ai-web-facade", "version": SERVER_VERSION},
},
)
self._notify("notifications/initialized", {})
def _notify(self, method: str, params: dict[str, Any]) -> None:
assert self._process.stdin is not None
message = {"jsonrpc": "2.0", "method": method, "params": params}
self._process.stdin.write(json.dumps(message, separators=(",", ":")) + "\n")
self._process.stdin.flush()
async def _call_async(self, name: str, arguments: dict[str, Any]) -> str:
# Imported lazily so the dependency-free stdio implementation still
# gives a useful startup error outside its production container.
from mcp import ClientSession
from mcp.client.streamable_http import streamablehttp_client
def _request(self, method: str, params: dict[str, Any]) -> dict[str, Any]:
with self._lock:
if self._process.poll() is not None:
raise RuntimeError("TinySearch child process is not running")
request_id = self._next_id
self._next_id += 1
assert self._process.stdin is not None
assert self._process.stdout is not None
message = {
"jsonrpc": "2.0",
"id": request_id,
"method": method,
"params": params,
}
self._process.stdin.write(json.dumps(message, separators=(",", ":")) + "\n")
self._process.stdin.flush()
deadline = datetime.now().timestamp() + CHILD_TIMEOUT_SECONDS
while True:
remaining = deadline - datetime.now().timestamp()
if remaining <= 0:
raise TimeoutError(f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s")
ready, _, _ = select.select([self._process.stdout], [], [], remaining)
if not ready:
raise TimeoutError(f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s")
line = self._process.stdout.readline()
if not line:
raise RuntimeError("TinySearch closed its output stream")
payload = json.loads(line)
if payload.get("id") != request_id:
continue
if "error" in payload:
raise RuntimeError(f"TinySearch error: {payload['error']}")
return payload.get("result") or {}
async with streamablehttp_client(
TINYSEARCH_MCP_URL,
timeout=CHILD_TIMEOUT_SECONDS,
sse_read_timeout=CHILD_TIMEOUT_SECONDS,
) as (read_stream, write_stream, _):
async with ClientSession(read_stream, write_stream) as session:
await session.initialize()
result = await session.call_tool(name, arguments)
if result.isError:
raise RuntimeError(clean_text(str(result.content)))
return "\n".join(
item.text for item in result.content
if getattr(item, "type", None) == "text"
)
def call(self, name: str, arguments: dict[str, Any]) -> str:
result = self._request("tools/call", {"name": name, "arguments": arguments})
if result.get("isError"):
raise RuntimeError(clean_text(str(result.get("content"))))
texts = [
item.get("text", "")
for item in result.get("content", [])
if isinstance(item, dict) and item.get("type") == "text"
]
return "\n".join(texts)
with self._lock:
try:
return asyncio.run(asyncio.wait_for(
self._call_async(name, arguments), CHILD_TIMEOUT_SECONDS))
except TimeoutError as exc:
raise TimeoutError(
f"TinySearch timed out after {CHILD_TIMEOUT_SECONDS:.0f}s") from exc
_client: TinySearchClient | None = None