diff --git a/.env.example b/.env.example index d9ffd75..12993a9 100644 --- a/.env.example +++ b/.env.example @@ -1,10 +1,10 @@ # Generated as /etc/mike-ai/stack.env by install.sh. Never commit real values. AI_BIND_ADDRESS=10.77.0.2 -MODEL_DIR=/srv/mike-ai/models +MODEL_DIR=/data/models ROUTER_API_KEY=GENERATED_BY_INSTALLER CONTROLLER_TOKEN=GENERATED_BY_INSTALLER -WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER -OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v3 +Z_IMAGE_MODEL_DIR=/data/models/Z-Image-Turbo +IMAGE_GPU_DEVICES=1 PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 diff --git a/ATHENA.md b/ATHENA.md index 3d980be..9a17f6e 100644 --- a/ATHENA.md +++ b/ATHENA.md @@ -1,110 +1,79 @@ # Athena – Betriebsanleitung Diese Datei ist der kurze, verbindliche Einstieg für Menschen und Agenten. -Für normale Arbeiten reicht sie aus. Detaildokumente unter `docs/` werden nur -gelesen, wenn diese Datei ausdrücklich darauf verweist oder eine konkrete -Fehlersuche sie benötigt. -## Aufbau +## Rolle -- Host: Debian, ohne lokalen Notfallzugriff oder KVM. -- Arbeitsbaum und laufender Stack: `/opt/mike-ai/stack`. -- Persistente Daten, Modelle und Backups: `/data`. -- Lokale Konfiguration und Secrets: `/etc/mike-ai` (niemals in Git). -- Benutzerzugriff auf KI-Dienste: über WireGuard, nicht über das Uni-LAN. -- OpenAI-kompatible Modell-API: Profile Router auf Port 8081. -- Oberfläche und Agent: Hermes auf Unraid unter - `/mnt/nvme-storage/appdata/Hermes-Agent`. Athena stellt dafür nur die - OpenAI-kompatible Router-API bereit. OpenWebUI ist abgeschaltetes Rückfallnetz. -- Inferenz: genau ein aktives llama.cpp-Textprofil; der Router wechselt bei - Bedarf zwischen Fast, Medium, Large, Ultra und Uncensored. +Athena ist eine Inferenzmaschine, kein allgemeiner Anwendungsserver. -## Verzeichnisse +Sie betreibt: + +- llama.cpp mit genau einem aktiven Qwen-Profil, +- den OpenAI-kompatiblen Profile Router, +- Z-Image-Turbo für Bilder, +- XTTS und Piper für Sprache, +- das Athena-Dashboard, +- WireGuard-Gateway und Datenbackup, +- den hostgebundenen Athena-Operator. + +Hermes, Benutzeroberfläche und portable Fach-MCPs laufen auf Unraid. Auf Athena +werden keine zweiten Instanzen dieser Dienste angelegt. + +## Pfade | Pfad | Zweck | |---|---| -| `/opt/mike-ai/stack` | Einziger Git-Checkout und einzige Quelle für Deployments | -| `/data/models` | GGUF-Modelle, Projektoren und weitere große Modelldateien | -| `/data` | Persistente Anwendungsdaten und Docker-Backups | -| `/etc/mike-ai` | Lokale Env-Dateien, API-Schlüssel und SSH-Schlüssel | -| `/tmp` | Einmalige Hilfsprogramme und temporäre Arbeitsdateien | +| `/opt/mike-ai/stack` | kanonischer Checkout und Compose-Stack | +| `/data/models` | Modellgewichte | +| `/data/llama-dashboard` | historische Dashboard-Messwerte | +| `/data/docker-backups` | automatische Athena-Backups | +| `/etc/mike-ai` | lokale Konfiguration und Secrets, niemals Git | -Die früheren Checkouts `/data/mike-ai-operator/repository` und -`/root/AI-Profile-Router` sind keine Arbeitsquellen. Sie dürfen nach der -Migration höchstens als gekennzeichnetes Archiv existieren. +## Standardbefehle -## Container-Prinzip +```bash +cd /opt/mike-ai/stack +./manage.sh validate +./manage.sh deploy SERVICE +./manage.sh deploy core +sudo ./smoke-test.sh +``` -Athena betreibt nur Inferenz, Router, Sprache/Bild, Backup und den -hostgebundenen Athena-Operator. Portable Fach-MCPs laufen als getrennte -Unterprozesse im **einen MCPHub-Container auf Unraid**. Sie bleiben unter -`/mcp/NAME` einzeln sichtbar und abschaltbar, brauchen aber nicht je einen -Docker-Container. `config/mcp-registry.json` ist die einzige Serverliste. +`deploy SERVICE` verwendet `--no-deps` und fasst keine anderen Container an. +`deploy core` aktualisiert den vollständigen Athena-Kern. Das aktuell aktive +Qwen-Profil wird vom Profile Controller verwaltet. -## Standardablauf für Änderungen +## Modelle -1. `athena_operator_inspect` einmal für den betroffenen Bereich aufrufen. -2. Mit `athena_operator_search_source` die konkrete Datei finden. -3. Mit `athena_operator_read_source` nur den benötigten Ausschnitt lesen. -4. Änderung über `athena_operator_change` ausführen. -5. Syntax, Compose, Dienstzustand und eine kleine Funktionsprobe prüfen. -6. Geänderte Dateien committen und pushen. -7. Ein manuelles Datenbackup nur nach speicherrelevanten Änderungen auslösen. +- Fast: kurze, interaktive Aufgaben +- Medium/Large/Ultra: steigende Kontextgrößen desselben lokalen Qwen-Modells +- Uncensored: separates lokales Profil +- Z-Image-Turbo: Bildgenerierung; Qwen wird dafür kurz entladen und danach + automatisch wiederhergestellt +- XTTS: RTX 3060; Piper bleibt CPU-Fallback -Nicht bei jedem Zwischenschritt die gesamte Plattform neu untersuchen. Keine -vollständigen Compose-, Installations- oder Dokumentationsdateien in den Chat -laden, wenn ein kleiner Ausschnitt genügt. Derselbe fehlgeschlagene Pfad oder -Werkzeugaufruf wird höchstens einmal wiederholt. +Die verbindlichen Werte stehen in `config/profile-matrix.json` und +`docs/STANDARD_PROFILE_MATRIX.md`. -## Neuer MCP +## Werkzeuge -Portable MCPs werden nach dem Skill `mcphub-deployer` in das reproduzierbare -MCPHub-Image eingebaut. Dazu gehören gepinnte Quelle, genau ein Registry-Eintrag, -Secret-Datei nur im Appdata, Build, Handshake und eine read-only-Probe. Danach -wird ausschließlich MCPHub über Unraid DockerMan neu erstellt. Router, Qwen, -WireGuard und andere Container werden nicht neu gestartet. - -Nur ein Werkzeug, das Athenas Host selbst verwalten muss, gehört in den -Athena-Operator. Es wird kein zweiter allgemeiner Terminal- oder Doku-MCP gebaut. - -## Temporär oder dauerhaft - -- „Nutze Programm X“: wenn es fehlt, nur temporär unter `/tmp` oder in einem - kurzlebigen Container verwenden und anschließend entfernen. -- „Installiere Programm X dauerhaft“: versioniert in den Stack aufnehmen. -- Bestehende Dienste auf Unraid oder im Heimnetz werden weiterverwendet; auf - Athena wird nicht ohne Grund eine zweite Instanz aufgebaut. +Portable Werkzeuge gehören auf Unraid in eigene, per DockerMan verwaltete +Container. Der einzige MCP auf Athena ist der Athena-Operator, weil nur er den +Athena-Host verwalten muss. Neue Fach-MCPs werden nicht in diesen Stack +eingebaut. ## Sicherheitsgrenze -Athena darf ohne ausdrücklichen, aktuellen Auftrag niemals heruntergefahren -oder neu gestartet werden. Ebenfalls tabu sind Änderungen an SSH, LAN, -WireGuard, Firewall, Bootloader, Kernel, Partitionen und Mounts. Diese Grenze -schützt die Erreichbarkeit des entfernten Hosts. - -Innerhalb des vertrauenswürdigen WireGuard-Netzes dürfen die vorgesehenen -Container normal miteinander, mit dem Heimnetz und mit dem Internet -kommunizieren. Keine zusätzlichen Netzwerkbarrieren ohne konkreten Bedarf. - -Secrets dürfen lokal von Athena und dem lokalen Modell verwendet werden. Sie -werden aber weder in Git noch in normalen Werkzeugausgaben oder Chatantworten -veröffentlicht. +Ohne ausdrücklichen aktuellen Auftrag niemals Shutdown, Reboot, Kernel, +Bootloader, Partitionen, Mounts, SSH, LAN, WireGuard oder Firewall ändern. +Secrets dürfen lokal verwendet, aber nie in Git, Logs oder Chatantworten +veröffentlicht werden. ## Fertig bedeutet -Eine Änderung ist erst fertig, wenn: - -- der versionierte Arbeitsbaum die Änderung enthält, -- der betroffene Dienst den neuen Stand verwendet, -- ein fokussierter Test erfolgreich war, -- Git-Status und Commit bekannt sind, -- das automatische Backup läuft und bei Datenänderungen ein Archiv geprüft wurde. - -Bei Unsicherheit wird der konkrete offene Punkt genannt. Es werden keine -Ergebnisse, Werkzeugaufrufe oder erfolgreichen Deployments erfunden. - -## Referenzen - -- Installation und Überblick: `README.md` -- Wiederherstellung: `docs/RECOVERY.md` -- Modellprofile: `docs/STANDARD_PROFILE_MATRIX.md` +- Änderung ist im kanonischen Git-Checkout, +- Compose und Syntax sind gültig, +- betroffener Dienst ist gesund, +- eine kleine Funktionsprobe war erfolgreich, +- Commit und Push sind erfolgt, +- das automatische Backup bleibt gesund. diff --git a/README.md b/README.md index 0542536..ede56f1 100644 --- a/README.md +++ b/README.md @@ -1,68 +1,60 @@ # Athena AI -Ein reproduzierbarer Docker-Stack für Athenas lokale Inferenz. Athena stellt -Router, llama.cpp-Profile, Sprache und Bildgenerierung bereit. Der offizielle -Hermes Agent und MCPHub laufen auf Unraid und werden dort mit Appdata gesichert. +Athena ist die lokale Inferenzmaschine. Der reproduzierbare Docker-Stack stellt +Qwen über eine kleine OpenAI-kompatible Router-API bereit und übernimmt lokale +Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.** -## Aufbau +## Aktueller Aufbau -- `compose.yaml` ist der einzige Einstieg für die KI-Dienste auf Athena. -- Genau ein llama.cpp-Profil ist aktiv. Der Router schaltet zwischen Fast, - Medium, Large, Ultra und Uncensored. -- Hermes verdichtet ältere Assistenten- und Werkzeug-Turns fortlaufend per - Micro-Compaction (alle fünf abgeschlossenen Turns). Das jeweils gewählte - 27B-Hauptprofil erstellt die Zusammenfassung; ein separates, weniger - zuverlässiges Kompressionsmodell wird nicht betrieben. -- Der **Athena Operator** bleibt als einziger hostgebundener administrativer - MCP direkt auf Athena. MCPHub veröffentlicht seinen vorhandenen - WireGuard-HTTP-Endpunkt zentral unter `/mcp/athena-operator`; es gibt keinen - zweiten Operator und keine administrative SSH-Implementierung im Hub. -- Home Assistant, ARR, Unraid, Navidrome und GitHub laufen gemeinsam im - MCPHub-Container auf Unraid, bleiben aber als getrennte MCP-Server unter - `/mcp/NAME` sichtbar, abschaltbar und unabhängig für Clients freigebbar. -- Hermes verwendet für allgemeine Recherche den eingebauten schlüssellosen - Keenable-Provider für Suche und Seitenabruf; der frühere Athena-Webadapter - wird nicht mehr gestartet. -- MCPHubs eigene persistente Einstellungen unter `MCPHub/mcp_settings.json` - sind der produktive Zustand. Oberfläche und offizielle API ändern genau - diese Datei; Container-Updates überschreiben sie nicht. - `config/mcp-registry.json` ist nur der Neuinstallations-Seed. -- Hermes verbindet sich einmal mit MCPHubs gefiltertem `/mcp/hermes`-Endpunkt. Neue - aktivierte Server erscheinen dadurch nach **MCP neu laden**, ohne dass pro - MCP eine weitere Hermes-Konfiguration geschrieben werden muss. -- Modelle und Athena-Backups liegen auf `/data`. Hermes liegt vollständig unter - `/mnt/nvme-storage/appdata/Hermes-Agent`; MCPHub-Zustand, Client-Schlüssel und - MCP-Zugänge liegen unter `/mnt/nvme-storage/appdata/MCPHub`. -- Hermes verwendet unverändert `nousresearch/hermes-agent:latest`. Seine - Profile erreichen Athenas Router über `http://192.168.1.212:8081/v1`. - Die frühere Athena-Instanz bleibt vorerst gestoppt als Rückfall erhalten. -- KI-Oberflächen und APIs sind nur über WireGuard erreichbar. +### Athena + +- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored +- Profile Router auf Port 8081 +- Z-Image-Turbo als exklusiver Bild-Worker auf der RTX 5080 +- XTTS auf der RTX 3060 mit Piper als CPU-Fallback +- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 +- WireGuard-Gateway, Datenbackup und Athena-Operator +- keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz + +### Unraid + +- offizieller Hermes-Agent mit persistentem Appdata +- je ein eigener Container für ARR, Deemix, Navidrome, STRATO und + Nginx Proxy Manager +- MUA/Unraid-MCP als Unraid-Plugin +- Media-Tools als nachrüstbare Werkzeugkiste +- Sicherung durch das vorhandene Unraid-Appdata-Backup + +Hermes nutzt Athenas Router unter `http://192.168.1.212:8081/v1`. Ein MCPHub +ist nicht mehr Bestandteil der produktiven Architektur. ## Installation – ein Befehl -Nach dem Ausfüllen von `config/install.env`: - ```bash -sudo ./install.sh --config config/install.env +cp config/install.env.example /root/mike-ai-install.env +# Werte in /root/mike-ai-install.env eintragen und chmod 600 setzen +sudo ./install.sh --config /root/mike-ai-install.env ``` -Das Skript installiert Docker und NVIDIA-Unterstützung, lädt die konfigurierten -Modelle und startet ausschließlich Athenas Inferenz-Kern plus Operator. +Das Installationsskript baut llama.cpp und die lokalen Images, lädt die +versionierten Modellartefakte und startet ausschließlich den Athena-Kern. -## Bedienung +## Betrieb ```bash -# Gesamten Stack anzeigen -docker compose --env-file /etc/mike-ai/stack.env ps +# Konfiguration prüfen +./manage.sh validate -# Erst anzeigen, dann eine gezielte Komponente ohne Nebenwirkungen ausrollen -./manage.sh --dry-run deploy router +# Gesamten Athena-Kern gezielt aktualisieren +./manage.sh deploy core + +# Nur einen Dienst ausrollen ./manage.sh deploy router -# Sofortiges Datenbackup zusätzlich zum Fünf-Stunden-Zeitplan -docker exec mike-ai-backup backup +# Eindeutige Altcontainer entfernen +./manage.sh purge-legacy -# Kurzer read-only Ende-zu-Ende-Test nach jedem Release +# Read-only Ende-zu-Ende-Test sudo ./smoke-test.sh ``` @@ -93,42 +85,41 @@ bis Hermes' Sitzungsfehler behoben ist. Router-API: `http://:8081/v1` -Hermes-Dashboard: `http://:9119` +## Endpunkte -Hermes-API: `http://:8642` +- Router: `http://192.168.1.212:8081/v1` +- Athena-Dashboard: `http://192.168.1.212:8099` +- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119` -## Ausgegliederte Fach-MCPs +Die Adressen sind nur über die vorgesehenen privaten Netze erreichbar. +## Ausgegliederte MCPs + +- [ARR-MCP](https://git.casaderoll.de/michael/arr-mcp) - [Deemix-MCP](https://git.casaderoll.de/michael/Deemix-MCP) - [Strato-MCP](https://git.casaderoll.de/michael/Strato-MCP) -## Wiederherstellung – ein Befehl +Weitere produktive Container verwenden ihre jeweiligen Upstream-Images und +Unraid-DockerMan-Templates. Details stehen in +[docs/MCP_SERVERS.md](docs/MCP_SERVERS.md). -Nach einer frischen Installation und eingehängtem `/data`: +## Wiederherstellung + +Nach einer frischen Debian-Installation und erneut eingehängtem `/data`: ```bash +sudo ./install.sh --config /root/mike-ai-install.env sudo ./restore.sh /data/docker-backups/athena-latest.tar.gz +sudo ./smoke-test.sh ``` -Details, Prüfschritte und der exakte Sicherungsumfang stehen in -[`docs/RECOVERY.md`](docs/RECOVERY.md). +Der genaue Sicherungsumfang steht in [docs/RECOVERY.md](docs/RECOVERY.md). -## Dokumentation +## Verbindliche Dokumentation -- [`ATHENA.md`](ATHENA.md) – kurze Maschinen- und Operatoranleitung -- [`docs/STANDARD_PROFILE_MATRIX.md`](docs/STANDARD_PROFILE_MATRIX.md) – Profile und Messwerte -- [`docs/MCP_SERVERS.md`](docs/MCP_SERVERS.md) – automatisch erzeugte MCP-Liste -- [`docs/RECOVERY.md`](docs/RECOVERY.md) – Backup und Neuaufbau - -Die MCPHub-Installation, Endpunkte und der schrittweise Rückbau der alten -Athena-MCPs stehen in [`platform/mcphub/README.md`](platform/mcphub/README.md). -Der eigenständig betreibbare Sonarr-/Radarr-Container einschließlich -Debian-Slim-Installer und Unraid-Template liegt unter -[`services/arr-mcp/`](services/arr-mcp/README.md). Er ist für die schrittweise -Ablösung des bisherigen MCPHub-Prozesses vorbereitet, wird durch Athenas -Standardinstallation aber nicht automatisch gestartet. -Für Installation oder Wiederherstellung des Hermes-Gateways auf Unraid liegt unter -[`config/unraid-templates/my-Hermes-Agent-Official.xml`](config/unraid-templates/my-Hermes-Agent-Official.xml) -ein DockerMan-Template, das unverändert das offizielle Nous-Image verwendet. +- [ATHENA.md](ATHENA.md) – kurze Betriebsanleitung +- [docs/STANDARD_PROFILE_MATRIX.md](docs/STANDARD_PROFILE_MATRIX.md) – Profile +- [docs/MCP_SERVERS.md](docs/MCP_SERVERS.md) – produktive Werkzeuge +- [docs/RECOVERY.md](docs/RECOVERY.md) – Backup und Neuaufbau Git enthält keine Secrets, Chatdaten oder Modellgewichte. diff --git a/compose.yaml b/compose.yaml index e49d1ea..ab49ff6 100644 --- a/compose.yaml +++ b/compose.yaml @@ -477,7 +477,6 @@ services: labels: com.mike-ai.llama-profile: experimental environment: - SEARXNG_URL: http://searxng:8080 NVIDIA_VISIBLE_DEVICES: ${EXPERIMENTAL_GPU_DEVICES:-0} NVIDIA_DRIVER_CAPABILITIES: compute,utility command: @@ -532,7 +531,7 @@ services: environment: CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" ALLOWED_PROFILES: fast,medium,large,ultra,uncensored,experimental - IMAGE_WORKER: flux + IMAGE_WORKER: image networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: @@ -572,8 +571,9 @@ services: # consume the complete context before yielding visible output. MAX_GENERATION_TOKENS: "8192" IMAGE_DIR: /data/images - IMAGE_WORKER_URL: http://flux-worker:8086 + IMAGE_WORKER_URL: http://image-worker:8086 IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" + IMAGE_MODEL_NAME: Z-Image-Turbo CHAT_IMAGE_ALLOW_REMOTE_URLS: "false" ENABLE_IMAGE_GENERATION: "true" ENABLE_TTS: "true" @@ -610,31 +610,31 @@ services: tts-gateway: condition: service_healthy - flux-worker: + image-worker: build: - context: platform/docker/flux-worker + context: platform/docker/image-worker args: DIFFUSERS_VERSION: ${DIFFUSERS_VERSION:-0.40.0} TRANSFORMERS_VERSION: ${TRANSFORMERS_VERSION:-5.15.1} ACCELERATE_VERSION: ${ACCELERATE_VERSION:-1.14.0} HF_HUB_VERSION: ${HF_HUB_VERSION:-1.28.0} - image: mike-ai/flux-worker:local - container_name: mike-ai-flux-worker + image: mike-ai/image-worker:local + container_name: mike-ai-image-worker restart: "no" profiles: [image] labels: - com.mike-ai.image-worker: flux + com.mike-ai.image-worker: image gpus: all read_only: true tmpfs: ["/tmp:size=1g,mode=1777"] volumes: - - "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro" + - "${Z_IMAGE_MODEL_DIR:-/data/models/Z-Image-Turbo}:/models/Z-Image-Turbo:ro" - router-images:/data/images environment: NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1} NVIDIA_DRIVER_CAPABILITIES: compute,utility WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" - FLUX_MODEL_DIR: /models/FLUX.2-klein-4B + Z_IMAGE_MODEL_DIR: /models/Z-Image-Turbo IMAGE_DIR: /data/images networks: [inference] security_opt: ["no-new-privileges:true"] @@ -750,168 +750,47 @@ services: retries: 12 start_period: 10s - open-webui: - build: - context: . - dockerfile: platform/openwebui/Dockerfile - image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-agent-loop-v9} - container_name: mike-ai-open-webui + llama-dashboard: + build: ./platform/llama-dashboard + image: mike-ai/llama-dashboard:local + container_name: mike-ai-llama-dashboard restart: unless-stopped - labels: - # SQLite is quiesced briefly while the scheduled data backup is created. - docker-volume-backup.stop-during-backup: "true" - volumes: - - open-webui-data:/app/backend/data - # Upstream-supported static customization hooks. Keeping these files in - # the repository makes the global dark theme reproducible and update-safe. - - ./platform/openwebui/theme/custom.css:/app/build/static/custom.css:ro - - ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro - - ./platform/openwebui/theme/midnight-aurora.svg:/app/build/static/midnight-aurora.svg:ro - - ./platform/openwebui/theme/tool-status:/app/build/static/tool-status:ro - environment: - WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}" - DEFAULT_MODELS: mikeai-medium - OLLAMA_BASE_URL: "" - OPENAI_API_BASE_URLS: http://router:8081/v1 - OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" - # Open WebUI uses the router's OpenAI-compatible image endpoint. The - # router performs the exclusive RTX-5080 hot swap and restores the - # previously active Qwen profile after every image. - ENABLE_IMAGE_GENERATION: "true" - IMAGE_GENERATION_ENGINE: openai - # OpenWebUI v0.9.x exposes a fixed OpenAI image-model dropdown. The - # router accepts this compatibility alias and still executes local - # FLUX.2 Klein; no request is sent to OpenAI. - IMAGE_GENERATION_MODEL: gpt-image-1 - # The local router can return embedded image data. Force that mode so - # OpenWebUI does not reject the router's private Docker/LAN URL through - # its correct SSRF protection. - IMAGE_URL_RESPONSE_MODELS_REGEX_PATTERN: "^$" - IMAGES_OPENAI_API_BASE_URL: http://router:8081/v1 - IMAGES_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" - IMAGE_SIZE: 1024x1024 - IMAGE_STEPS: "4" - AUDIO_TTS_ENGINE: openai - AUDIO_TTS_OPENAI_API_BASE_URL: http://router:8081/v1 - AUDIO_TTS_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" - AUDIO_TTS_MODEL: piper - AUDIO_TTS_VOICE: alloy - ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false} - ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false} - # The derived image reserves the last round for a tool-free synthesis. - # Forty executions permit real multi-domain agent work. Exact-repeat, - # per-tool and total-execution limits in the derived image stop loops. - # Leave continuation headroom after the execution middleware budget: - # one additional model turn is required to synthesize the visible answer. - CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "48" - USER_AGENT: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36" - DO_NOT_TRACK: "true" - SCARF_NO_ANALYTICS: "true" - dns: ["${AI_DNS:-1.1.1.1}"] - networks: [frontend, tools] - depends_on: - wireguard-gateway: - condition: service_healthy - router: - condition: service_healthy - security_opt: ["no-new-privileges:true"] - - hermes: - build: - context: . - dockerfile: platform/hermes/Dockerfile - image: ${HERMES_IMAGE:-mike-ai/hermes-agent:0.20.5-mcpfix1} - container_name: mike-ai-hermes - restart: unless-stopped - command: [/usr/local/bin/start-hermes-managed] - env_file: - - /data/hermes/.env - volumes: - - /data/hermes:/opt/data - - /data/hermes/workspace:/workspace - - ./platform/hermes/start-hermes-managed.sh:/usr/local/bin/start-hermes-managed:ro - - ./platform/hermes/patch-api-mcp-refresh.py:/usr/local/lib/mike-ai/patch-api-mcp-refresh.py:ro - environment: - HERMES_HOME: /opt/data - dns: ["${AI_DNS:-1.1.1.1}"] - networks: [frontend, tools, tools-egress] - depends_on: - wireguard-gateway: - condition: service_healthy - router: - condition: service_healthy - security_opt: ["no-new-privileges:true"] - healthcheck: - test: [CMD, curl, -fsS, "http://127.0.0.1:8642/health"] - interval: 15s - timeout: 5s - retries: 20 - start_period: 45s - - # Optional, fully removable community chat surface. Chat execution goes - # through the existing Hermes gateway. Upstream's container entrypoint - # requires a writable Hermes home for its ownership/init checks; UI-only - # state still remains on a separate bind mount for easy removal. - hermes-webui: - image: ${HERMES_WEBUI_IMAGE:-mike-ai/hermes-webui:0.52.113-hermes-source-v1} - container_name: mike-ai-hermes-webui - restart: unless-stopped - profiles: [hermes-webui] - env_file: - - /data/hermes-webui/.env - volumes: - - /data/hermes:/home/hermeswebui/.hermes - - /data/hermes-webui/state:/state - - /data/hermes-webui/hermes-agent:/home/hermeswebui/.hermes/hermes-agent:ro - - /data/hermes/workspace:/workspace - environment: - HERMES_HOME: /home/hermeswebui/.hermes - HERMES_WEBUI_STATE_DIR: /state - HERMES_WEBUI_HOST: 0.0.0.0 - HERMES_WEBUI_PORT: "8787" - HERMES_WEBUI_CHAT_BACKEND: gateway - HERMES_WEBUI_GATEWAY_BASE_URL: http://hermes:8642 - HERMES_API_URL: http://hermes:8642 - HERMES_WEBUI_AGENT_DIR: /home/hermeswebui/.hermes/hermes-agent - HERMES_WEBUI_GATEWAY_USE_RUNS_API: "true" - HERMES_SKIP_CHMOD: "1" - WANTED_UID: "10000" - WANTED_GID: "10000" - networks: [frontend] - depends_on: - hermes: - condition: service_healthy - security_opt: ["no-new-privileges:true"] - healthcheck: - test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8787/health', timeout=3)"] - interval: 15s - timeout: 5s - retries: 20 - start_period: 45s - - # Persistent VPN listener for the optional WebUI. Sharing the existing - # WireGuard network namespace avoids recreating the remote-access gateway - # merely to add one listener. - hermes-webui-vpn-proxy: - image: mike-ai/wireguard-gateway:local - container_name: mike-ai-hermes-webui-vpn-proxy - restart: unless-stopped - profiles: [hermes-webui] network_mode: "service:wireguard-gateway" - entrypoint: [socat] - command: - - TCP-LISTEN:8787,bind=192.168.1.212,reuseaddr,fork - - TCP:hermes-webui:8787 + gpus: all read_only: true tmpfs: - - /tmp:size=4m,mode=1777 - cap_drop: [ALL] - security_opt: ["no-new-privileges:true"] + - /tmp:size=16m,mode=1777 + volumes: + - /proc:/host/proc:ro + - /data:/host/data:ro + - /data/models:/host/models:ro + - /data/llama-dashboard:/var/lib/llama-dashboard + environment: + DASHBOARD_HOST: 0.0.0.0 + DASHBOARD_PORT: "8099" + ROUTER_URL: http://router:8081 + ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" + HOST_PROC: /host/proc + HOST_DATA: /host/data + HOST_MODELS: /host/models + DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3 + DASHBOARD_HISTORY_INTERVAL: "15" + DASHBOARD_DETAIL_RETENTION_DAYS: "21" + NVIDIA_VISIBLE_DEVICES: all + NVIDIA_DRIVER_CAPABILITIES: compute,utility depends_on: wireguard-gateway: condition: service_healthy - hermes-webui: + router: condition: service_healthy + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + healthcheck: + test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8099/health', timeout=2)"] + interval: 10s + timeout: 3s + retries: 12 + start_period: 10s backup: image: ${BACKUP_IMAGE:-offen/docker-volume-backup@sha256:19102d8e59eb1d598cf8c647c2b21100abaadc5a1c808ac643fa612e323c3013} @@ -954,7 +833,6 @@ networks: name: mike-ai-tools-egress volumes: - open-webui-data: piper-data: router-state: router-images: diff --git a/config/github-mcp.env.example b/config/github-mcp.env.example deleted file mode 100644 index fb1fe5b..0000000 --- a/config/github-mcp.env.example +++ /dev/null @@ -1,9 +0,0 @@ -# Create a dedicated fine-grained GitHub token. Grant repository contents and -# metadata read-only; do not grant write permissions. Keep the real file only -# at /etc/mike-ai/github-mcp.env with mode 0600. -GITHUB_PERSONAL_ACCESS_TOKEN= - -# Four deliberately bounded repository-reading tools. Do not replace this -# with the broad default toolsets unless the resulting schemas were reviewed. -GITHUB_TOOLS=search_repositories,get_repository_tree,get_file_contents,search_code -GITHUB_READ_ONLY=1 diff --git a/config/install.env.example b/config/install.env.example index 71ae05b..6ad7d4d 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -3,7 +3,7 @@ AI_HOSTNAME=ki-host ADMIN_USER=mike -MODEL_DIR=/srv/mike-ai/models +MODEL_DIR=/data/models # Installing a new NVIDIA driver can require one reboot. In that case this # installer exits with code 20 (NVIDIA) or 21 (stable NIC rename); rerun the @@ -15,8 +15,8 @@ NVIDIA_DRIVER_BRANCH= NVIDIA_MIN_DRIVER_MAJOR=570 TEXT_GPU_DEVICES=0 SECONDARY_GPU_DEVICES=1 -IMAGE_GPU_DEVICES=0 -FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B +IMAGE_GPU_DEVICES=1 +Z_IMAGE_MODEL_DIR=/data/models/Z-Image-Turbo # Headless remote reachability. Firmware power-loss recovery is configured # separately once at the physical machine. @@ -88,13 +88,6 @@ UNCENSORED_MTP_MAX=2 EXPERIMENTAL_CONTEXT=76800 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 -OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v3 -OPENWEBUI_ENABLE_SIGNUP=false -OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false -# Removable community Hermes chat surface. Set false to keep only the official -# Hermes Dashboard/API and native clients. -INSTALL_HERMES_WEBUI=true -HERMES_WEBUI_IMAGE=mike-ai/hermes-webui:0.52.113-hermes-source-v1 PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 diff --git a/config/mcp-extensions/himalaya.json b/config/mcp-extensions/himalaya.json deleted file mode 100644 index e0ffb24..0000000 --- a/config/mcp-extensions/himalaya.json +++ /dev/null @@ -1,47 +0,0 @@ -{ - "server": { - "id": "himalaya", - "hermes_id": "himalaya", - "name": "Himalaya Mail", - "description": "Apple-unabhängiger Mailzugriff über die Himalaya CLI. Lesen, suchen und Anhänge verwalten; schreibende Mailaktionen nur auf ausdrücklichen Auftrag.", - "url": "http://192.168.1.2:8787/mcp/himalaya", - "clients": ["hermes"], - "timeout": 300, - "deployment": { - "version": "himalaya-mcp 2.1.2 / himalaya-cli 2.1.0", - "source": "https://github.com/Data-Wise/himalaya-mcp", - "required_env": ["HIMALAYA_CONFIG"], - "required_files": ["himalaya-config.toml"] - }, - "hub": { - "type": "stdio", - "secret_file": "himalaya.env", - "command": "/usr/local/bin/run-with-env", - "args": [ - "/run/secrets/mcphub/himalaya.env", - "--", - "node", - "/app/data/extensions/himalaya/index.js" - ], - "env": { - "HIMALAYA_BINARY": "/app/data/extensions/himalaya/himalaya", - "MCP_TRANSPORT": "stdio" - }, - "enabled": false - } - }, - "artifacts": [ - { - "source": "/app/data/work/himalaya/himalaya", - "path": "himalaya", - "sha256": "7bc31ca0ea596218d97f1b2637e14c6653b1ebf9741711ac0f8a675384d67472", - "mode": "0755" - }, - { - "source": "/app/data/work/himalaya/index.js", - "path": "index.js", - "sha256": "c8a94a46b33e3e683d38bdfbd5d84f4441e4668780facb5f3e90fc22e68ab683", - "mode": "0644" - } - ] -} diff --git a/config/mcp-registry.json b/config/mcp-registry.json deleted file mode 100644 index 4fb7b0f..0000000 --- a/config/mcp-registry.json +++ /dev/null @@ -1,186 +0,0 @@ -{ - "version": 1, - "servers": [ - { - "id": "mcphub-all", - "hermes_id": "mcphub", - "name": "MCPHub", - "description": "Zentraler Zugang zu allen auf Unraid aktivierten MCP-Servern. Neue Server erscheinen nach einem MCP-Reload ohne Änderung der Hermes-Konfiguration.", - "url": "http://192.168.1.2:8787/mcp/hermes", - "clients": ["hermes"], - "env_file": "/etc/mike-ai/mcphub-client.env", - "key_env": "MCPHUB_BEARER_TOKEN", - "auth_type": "bearer", - "timeout": 900 - }, - { - "id": "mcphub-admin-local", - "name": "MCPHub Administration", - "description": "Installiert und verwaltet HTTP-, npm- und Python-MCPs direkt über die offizielle MCPHub-API. Kein Unraid-Terminal erforderlich.", - "url": "http://192.168.1.2:8787/mcp/mcphub-admin", - "clients": [], - "timeout": 300, - "hub": { - "type": "stdio", - "command": "python3", - "args": ["/opt/casaderoll/mcps/mcphub_admin_mcp.py"], - "env": { - "MCPHUB_API_URL": "http://127.0.0.1:3000/api", - "MCPHUB_API_TOKEN_FILE": "/app/data/client-token", - "MCPHUB_CLIENT_GROUP": "hermes" - }, - "enabled": true - } - }, - { - "id": "athena-operator-local", - "hermes_id": "athena-operator", - "name": "Athena Operator", - "description": "Zentrale administrative Schnittstelle für Athena. Beginne mit athena_operator_inspect(subject=guide). Verwaltet Docker, Modelle, MCPs, Git und Backups; Stromversorgung und Remote-Erreichbarkeit bleiben blockiert.", - "url": "http://192.168.1.2:8787/mcp/athena-operator", - "clients": ["openwebui"], - "env_file": "/etc/mike-ai/mcphub-client.env", - "key_env": "MCPHUB_BEARER_TOKEN", - "auth_type": "bearer", - "timeout": 900, - "hub": { - "type": "streamable-http", - "url": "http://192.168.1.212:8202/mcp", - "owner": "admin", - "enabled": true - } - }, - { - "id": "github-local", - "hermes_id": "github", - "name": "GitHub (offiziell, read-only)", - "description": "Repository-Suche, echte Datei-Inhalte und gezielte Code-Suche. Keine rekursiven Komplettbäume oder Schreibzugriffe.", - "url": "http://192.168.1.2:8787/mcp/github", - "clients": ["openwebui"], - "env_file": "/etc/mike-ai/mcphub-client.env", - "key_env": "MCPHUB_BEARER_TOKEN", - "auth_type": "bearer", - "timeout": 300, - "functions": "github-search_repositories,github-get_file_contents,github-search_code", - "hub": { - "type": "stdio", - "command": "/usr/local/bin/run-with-env", - "args": ["/run/secrets/mcphub/github.env", "--", "/usr/local/bin/github-mcp-server", "stdio", "--read-only", "--tools", "search_repositories,get_file_contents,search_code"], - "enabled": true - } - }, - { - "id": "homeassistant-local", - "hermes_id": "homeassistant-admin", - "name": "Home Assistant", - "description": "Entitäten, Zustände, Historie, Automationen, Dashboards, Diagnose und freigegebene YAML-Dateien. Änderungen nur auf ausdrücklichen Auftrag.", - "url": "http://192.168.1.2:8787/mcp/homeassistant", - "clients": ["openwebui"], - "env_file": "/etc/mike-ai/mcphub-client.env", - "key_env": "MCPHUB_BEARER_TOKEN", - "auth_type": "bearer", - "timeout": 300, - "hub": { - "type": "streamable-http", - "secret_file": "homeassistant.env", - "url": "${HASS_URL}/api/hass_mcp", - "headers": {"Authorization": "Bearer ${HASS_TOKEN}"}, - "owner": "admin", - "enabled": true - } - }, - { - "id": "arr-local", - "hermes_id": "arr", - "name": "Sonarr und Radarr", - "description": "Serien, Filme, Queue, Indexer-Suche und kompakte Medieninventare. Für Codec-Fragen radarr_movie_codec_inventory verwenden; keine rohen API-Requests oder Dateisystem-Scans.", - "url": "http://192.168.1.2:8787/mcp/arr", - "clients": ["openwebui"], - "env_file": "/etc/mike-ai/mcphub-client.env", - "key_env": "MCPHUB_BEARER_TOKEN", - "auth_type": "bearer", - "timeout": 600, - "hub": { - "type": "stdio", - "command": "/usr/local/bin/run-with-env", - "args": ["/run/secrets/mcphub/arr.env", "--", "arr-mcp", "--transport", "stdio", "--auth-type", "none"], - "enabled": true - } - }, - { - "id": "navidrome-local", - "hermes_id": "navidrome", - "name": "Navidrome", - "description": "Persönliche Musikbibliothek: Titel, Alben, Künstler, Playlists, Favoriten und Hörverlauf.", - "url": "http://192.168.1.2:8787/mcp/navidrome", - "clients": ["openwebui"], - "env_file": "/etc/mike-ai/mcphub-client.env", - "key_env": "MCPHUB_BEARER_TOKEN", - "auth_type": "bearer", - "timeout": 300, - "hub": { - "type": "stdio", - "command": "/usr/local/bin/run-with-env", - "args": ["/run/secrets/mcphub/navidrome.env", "--", "node", "/opt/casaderoll/navidrome/dist/index.js"], - "env": {"MCP_TRANSPORT": "stdio", "MCP_HTTP_EXPOSE": "false", "WEBUI_ENABLED": "false"}, - "enabled": true - } - }, - { - "id": "mua", - "hermes_id": "unraid", - "name": "MUA (Unraid-Verwaltung)", - "description": "Unraid-Verwaltung über das vorhandene MUA-Plugin. Zustand zuerst lesen, engste Änderung ausführen, danach verifizieren.", - "url": "http://192.168.1.2:8787/mcp/unraid", - "key_env": "MCPHUB_BEARER_TOKEN", - "env_file": "/etc/mike-ai/mcphub-client.env", - "auth_type": "bearer", - "clients": ["openwebui"], - "timeout": 900, - "hub": { - "type": "streamable-http", - "secret_file": "mua.env", - "url": "${MUA_MCP_URL}", - "headers": {"Authorization": "Bearer ${MUA_MCP_BEARER_TOKEN}"}, - "owner": "admin", - "enabled": true - } - }, - { - "id": "fritzbox-local", - "hermes_id": "fritzbox", - "name": "FRITZ!Box", - "description": "FRITZ!Box-Status, WAN/Glasfaser, Netzwerkgeräte, WLAN und Telefonie. Zuerst die aktive WAN-Verbindung ermitteln; Änderungen nur auf ausdrücklichen Auftrag.", - "url": "http://192.168.1.2:8787/mcp/fritzbox", - "key_env": "MCPHUB_BEARER_TOKEN", - "env_file": "/etc/mike-ai/mcphub-client.env", - "auth_type": "bearer", - "clients": ["openwebui"], - "timeout": 300, - "tool_include": [ - "fritzbox-list_services", - "fritzbox-list_actions", - "fritzbox-describe_action", - "fritzbox-call_action" - ], - "hub": { - "type": "stdio", - "command": "/usr/local/bin/run-with-env", - "args": ["/run/secrets/mcphub/fritzbox.env", "--", "/opt/casaderoll/fritz-mcp"], - "enabled": true - } - }, - { - "id": "mua-readonly-local", - "name": "MUA (Unraid read-only)", - "description": "Automatisch nutzbare Unraid-Diagnose für Container, Logs, System, Storage, Shares und Medieninventare. Keine Änderungen oder freie Shell.", - "url": "http://192.168.1.2:8787/mcp/unraid", - "key_env": "MCPHUB_BEARER_TOKEN", - "env_file": "/etc/mike-ai/mcphub-client.env", - "auth_type": "bearer", - "clients": ["openwebui"], - "timeout": 900, - "functions": "unraid-unraid_docker_list,unraid-unraid_docker_inspect,unraid-unraid_docker_logs,unraid-unraid_docker_analyze_logs,unraid-unraid_docker_processes,unraid-unraid_docker_stats,unraid-unraid_docker_info,unraid-unraid_docker_update_status,unraid-unraid_ca_search,unraid-unraid_network_inventory,unraid-unraid_network_list,unraid-unraid_network_inspect,unraid-unraid_network_host_state,unraid-unraid_network_audit_tcp,unraid-unraid_network_lan_probe,unraid-unraid_system_health,unraid-unraid_storage_status,unraid-unraid_disk_health,unraid-unraid_notifications_list,unraid-unraid_shares_list,unraid-unraid_share_inspect,unraid-unraid_files_inventory,unraid-unraid_system_connection_test,unraid-unraid_system_shell_readonly" - } - ] -} diff --git a/config/mcphub-client.env.example b/config/mcphub-client.env.example deleted file mode 100644 index 85fa705..0000000 --- a/config/mcphub-client.env.example +++ /dev/null @@ -1,4 +0,0 @@ -# Shared bearer key generated by platform/mcphub/configure-settings.py. -# The live value is stored in MCPHub appdata/client-token and copied only to -# /etc/mike-ai/mcphub-client.env on clients. -MCPHUB_BEARER_TOKEN=REPLACE_WITH_LOCAL_MCPHUB_CLIENT_TOKEN diff --git a/config/mua-mcp.env.example b/config/mua-mcp.env.example deleted file mode 100644 index 5a6853e..0000000 --- a/config/mua-mcp.env.example +++ /dev/null @@ -1,4 +0,0 @@ -# Root-only auf Athena unter /etc/mike-ai/mua-mcp.env ablegen (Modus 0600). -# Der echte Bearer-Token gehört niemals ins Git-Repository. -MUA_MCP_URL=http://192.168.1.2:3002/mcp -MUA_MCP_BEARER_TOKEN=REPLACE_WITH_MUA_BEARER_TOKEN diff --git a/config/navidrome-mcp.env.example b/config/navidrome-mcp.env.example deleted file mode 100644 index 169168d..0000000 --- a/config/navidrome-mcp.env.example +++ /dev/null @@ -1,8 +0,0 @@ -# Root-only deployment secret. Copy to /etc/mike-ai/navidrome-mcp.env, -# replace the two placeholders and chmod 600. Never commit the real file. -NAVIDROME_URL=http://192.168.1.2:4533 -NAVIDROME_USERNAME=REPLACE_WITH_DEDICATED_USER -NAVIDROME_PASSWORD=REPLACE_WITH_DEDICATED_PASSWORD -# Optional: enables seven public Last.fm discovery/recommendation tools. -# The Last.fm shared secret is not required and must not be stored here. -LASTFM_API_KEY= diff --git a/config/unraid-templates/my-ARR-MCP.xml b/config/unraid-templates/my-ARR-MCP.xml deleted file mode 100644 index 23b9aa3..0000000 --- a/config/unraid-templates/my-ARR-MCP.xml +++ /dev/null @@ -1,23 +0,0 @@ - - - ARR-MCP - mike-ai/arr-mcp:1.1.0 - bridge - sh - false - https://git.casaderoll.de/michael/AI-Profile-Router - https://git.casaderoll.de/michael/AI-Profile-Router/src/branch/main/services/arr-mcp/README.md - Eigenständiger Sonarr- und Radarr-MCP für Hermes und andere MCP-Clients. Das lokale Image wird reproduzierbar aus dem privaten Git gebaut. Der Container nutzt kompakte, modellfreundliche Werkzeuge, ein begrenztes Codec-Inventar und ticketgebundene Sonarr-Schreibaktionen. Konfiguration und Schlüssel liegen ausschließlich in Unraid-Appdata. - AI: - - false - https://raw.githubusercontent.com/Sonarr/Sonarr/develop/Logo/256.png - --read-only --cap-drop=ALL --security-opt=no-new-privileges --pids-limit=256 --tmpfs /tmp:rw,noexec,nosuid,nodev,size=64m --env-file=/mnt/nvme-storage/appdata/ARR-MCP/arr-mcp.env - - - - - - Das lokale Image mike-ai/arr-mcp:1.1.0 muss vorher mit services/arr-mcp/install-on-unraid.sh gebaut worden sein. - 8207 - diff --git a/config/unraid-templates/my-Hermes-Agent-Official.xml b/config/unraid-templates/my-Hermes-Agent-Official.xml deleted file mode 100644 index f96e309..0000000 --- a/config/unraid-templates/my-Hermes-Agent-Official.xml +++ /dev/null @@ -1,50 +0,0 @@ - - - Hermes-Agent - nousresearch/hermes-agent:latest - https://hub.docker.com/r/nousresearch/hermes-agent - bridge - - bash - false - https://github.com/NousResearch/hermes-agent/issues - https://hermes-agent.nousresearch.com/ - https://github.com/NousResearch/hermes-agent/blob/main/website/docs/user-guide/docker.md - Offizieller Hermes Agent von Nous Research als persistenter Gateway- und Dashboard-Dienst. Dieses Unraid-Template verwendet unverändert das offizielle Image nousresearch/hermes-agent:latest; es enthält keinen Fork und keine eigene Build-Schicht. Sämtliche Konfigurationen, Sitzungen, Skills, Erinnerungen und Zugangsdaten liegen dauerhaft unter /opt/data im Unraid-Appdata. - -Vor dem ersten Start müssen sichere Werte für API-Key, Dashboard-Passwort und Dashboard-Secret eingetragen werden. Der Container startet den offiziellen Befehl gateway run. - -Dokumentation: https://hermes-agent.nousresearch.com/docs/user-guide/docker - AI: - http://[IP]:[PORT:9119]/ - false - https://raw.githubusercontent.com/NousResearch/hermes-agent/main/web/public/favicon.ico - --shm-size=1g - gateway run - - - - - - - /mnt/nvme-storage/appdata/Hermes-Agent - - 8642 - 9119 - - 10000 - 10000 - - true - 0.0.0.0 - 8642 - - * - - 1 - 0.0.0.0 - 9119 - admin - - - diff --git a/config/unraid-templates/my-MCPHub.xml b/config/unraid-templates/my-MCPHub.xml deleted file mode 100644 index 58336e9..0000000 --- a/config/unraid-templates/my-MCPHub.xml +++ /dev/null @@ -1,38 +0,0 @@ - - - MCPHub - casaderoll/mcphub:1.2.5 - https://hub.docker.com/r/samanhappy/mcphub - bridge - - - sh - false - https://github.com/samanhappy/mcphub/issues - https://github.com/samanhappy/mcphub - https://github.com/samanhappy/mcphub#readme - Zentrale MCP-Verwaltung mit Weboberfläche. Das lokale CasaDeRoll-Image basiert reproduzierbar auf MCPHub 1.0.32 und enthält die versionierten ARR-, Navidrome-, GitHub- und FritzBox-Laufzeiten. Zusätzliche portable MCPs liegen updatefest im gemounteten Appdata-Verzeichnis und benötigen keinen Image-Neubau. Home Assistant, MUA und der Athena Operator werden als vorhandene HTTP-MCPs eingebunden. Einzelne Server bleiben unter /mcp/NAME getrennt sichtbar und schaltbar. Hermes nutzt für allgemeine Webrecherche seine eingebauten Werkzeuge. - -Weboberfläche: http://[IP]:[PORT:3000]/ -Benutzer beim ersten Start: admin -Wenn kein Admin-Passwort eingetragen wird, erzeugt MCPHub eines und schreibt es ins Containerprotokoll. - Tools:Utilities AI: - http://[IP]:[PORT:3000]/ - - https://github.com/samanhappy.png - --restart=unless-stopped - - - 0 - - - - 8787 - /mnt/nvme-storage/appdata/MCPHub - /mnt/nvme-storage/appdata/MCPHub/secrets - - production - 120000 - Europe/Berlin - - diff --git a/dev/test_hass_mcp_yaml_guard.py b/dev/test_hass_mcp_yaml_guard.py deleted file mode 100644 index 48ff617..0000000 --- a/dev/test_hass_mcp_yaml_guard.py +++ /dev/null @@ -1,91 +0,0 @@ -#!/usr/bin/env python3 -"""Dependency-free safety checks for the hass_mcp YAML overlay.""" - -from __future__ import annotations - -import importlib.util -import pathlib -import sys -import types - - -def module(name: str, **attributes): - value = types.ModuleType(name) - for key, item in attributes.items(): - setattr(value, key, item) - sys.modules[name] = value - return value - - -class ToolError(Exception): - pass - - -def decorator(**_kwargs): - return lambda function: function - - -module("homeassistant") -module("homeassistant.core", HomeAssistant=object) -module("homeassistant.util", slugify=lambda value: str(value).lower().replace(" ", "_")) -module("guarded") -module("guarded.tools") -module("guarded.identity", user_context=lambda: None) -module("guarded.protocol", ToolError=ToolError, internal_error=lambda message, error: RuntimeError(f"{message}: {error}")) -module( - "guarded.registry", - LIMIT_FIELD={"type": "integer"}, - OFFSET_FIELD={"type": "integer"}, - paginate=lambda items, limit, offset: {"items": items[offset : offset + limit]}, - schema=lambda **kwargs: kwargs, - tool=decorator, -) - -source = pathlib.Path(__file__).parents[1] / "platform/mcp/patches/hass_mcp/yaml_config.py" -spec = importlib.util.spec_from_file_location("guarded.tools.yaml_config", source) -assert spec and spec.loader -guard = importlib.util.module_from_spec(spec) -sys.modules[spec.name] = guard -spec.loader.exec_module(guard) - -assert set(guard._KINDS) == {"automation", "script", "scene", "configuration"} -assert all("secret" not in filename for filename, _, _ in guard._KINDS.values()) -assert guard._redact_line("api_key: abc") == "api_key: " -assert guard._redact_line("token: abc") == "token: " -assert guard._redact_line("value: !secret private_name") == "value: !secret " -assert guard._redact_line("alias: Safe automation") == "alias: Safe automation" - -for sensitive in ("token: abc", "password: abc", "value: !secret private_name"): - try: - guard._reject_sensitive_replacement(sensitive) - except ToolError: - pass - else: - raise AssertionError(f"sensitive replacement was accepted: {sensitive}") - -change = guard._change_record("automation", "update", "before", {"id": "demo"}) -preview = guard._preview(change) -assert preview["changed"] is False -assert preview["confirmation_required"] is True -guard._consume_ticket(change, preview["approval_ticket"]) -try: - guard._consume_ticket(change, preview["approval_ticket"]) -except ToolError: - pass -else: - raise AssertionError("one-time approval ticket was reusable") - -other = guard._change_record("automation", "update", "different", {"id": "demo"}) -ticket = guard._preview(change)["approval_ticket"] -try: - guard._consume_ticket(other, ticket) -except ToolError: - pass -else: - raise AssertionError("ticket accepted a different current file fingerprint") - -diff = guard._source_diff("automations.yaml", "a\nb\n", "a\nc\n") -assert any("-b" in line for line in diff) -assert any("+c" in line for line in diff) - -print("hass_mcp_yaml_guard_tests=ok") diff --git a/dev/test_mcp_registry.py b/dev/test_mcp_registry.py deleted file mode 100644 index 2158753..0000000 --- a/dev/test_mcp_registry.py +++ /dev/null @@ -1,106 +0,0 @@ -#!/usr/bin/env python3 -"""Tests for the single declarative MCP client registry.""" - -from __future__ import annotations - -import importlib.util -import json -import sqlite3 -import tempfile -import unittest -from pathlib import Path - - -ROOT = Path(__file__).parents[1] -SOURCE = ROOT / "platform/mcp/sync-clients.py" - - -def load_module(): - spec = importlib.util.spec_from_file_location("sync_clients", SOURCE) - module = importlib.util.module_from_spec(spec) - assert spec.loader - spec.loader.exec_module(module) - return module - - -class RegistryTests(unittest.TestCase): - def setUp(self): - self.module = load_module() - self.temp = tempfile.TemporaryDirectory() - self.root = Path(self.temp.name) - self.registry = self.root / "registry.json" - self.registry.write_text(json.dumps({"version": 1, "servers": [{ - "id": "one", "name": "One", "description": "Test", "url": "http://one/mcp", - "clients": ["hermes", "openwebui"], "timeout": 123, - }]})) - - def tearDown(self): - self.temp.cleanup() - - def test_same_registry_generates_both_clients(self): - items = self.module.active(self.registry, "hermes") - block = self.module.hermes_block(items) - self.assertIn("one:", block) - db = self.root / "webui.db" - con = sqlite3.connect(db) - con.execute("create table config (key text primary key, value text, updated_at integer)") - con.commit(); con.close() - self.module.update_openwebui(db, self.module.active(self.registry, "openwebui")) - con = sqlite3.connect(db) - value = json.loads(con.execute("select value from config where key='tool_server.connections'").fetchone()[0]) - con.close() - self.assertEqual(value[0]["info"]["id"], "one") - - def test_old_platform_context_registration_is_removed(self): - db = self.root / "webui.db" - con = sqlite3.connect(db) - con.execute("create table config (key text primary key, value text, updated_at integer)") - con.execute("insert into config values (?,?,?)", ("tool_server.connections", json.dumps([ - {"info": {"id": "athena-platform"}, "url": "http://old/mcp"}, - {"info": {"id": "unmanaged"}, "url": "http://keep/mcp"}, - ]), 0)) - con.commit(); con.close() - self.module.update_openwebui(db, self.module.active(self.registry, "openwebui")) - con = sqlite3.connect(db) - ids = [item["info"]["id"] for item in json.loads(con.execute("select value from config where key='tool_server.connections'").fetchone()[0])] - con.close() - self.assertEqual(ids, ["unmanaged", "one"]) - - def test_production_registry_has_unique_ids_and_fritzbox(self): - document = json.loads((ROOT / "config/mcp-registry.json").read_text()) - ids = [item["id"] for item in document["servers"]] - hermes_ids = [ - item.get("hermes_id", item["id"]) - for item in document["servers"] if "hermes" in item.get("clients", []) - ] - self.assertEqual(len(ids), len(set(ids))) - self.assertEqual(len(hermes_ids), len(set(hermes_ids))) - fritz = next(item for item in document["servers"] if item["id"] == "fritzbox-local") - self.assertEqual(fritz["hub"]["type"], "stdio") - self.assertIn("fritz-mcp", fritz["hub"]["args"][-1]) - self.assertEqual(len(fritz["tool_include"]), 4) - - def test_hermes_tool_filter_is_generated(self): - item = { - "id": "wide", "name": "Wide", "description": "Test", - "url": "http://wide/mcp", "clients": ["hermes"], - "tool_include": ["list", "describe", "call"], - } - block = self.module.hermes_block([item]) - self.assertIn(" tools:\n include:", block) - self.assertIn(' - "describe"', block) - - def test_raw_mcphub_token_can_replace_host_specific_env_file(self): - item = { - "id": "hub", "name": "Hub", "description": "Test", - "url": "http://hub/mcp/test", "clients": ["hermes"], - "env_file": "/missing/client.env", - "key_env": "MCPHUB_BEARER_TOKEN", - } - self.module.CLIENT_TOKEN = "local-token" - self.assertTrue(self.module.enabled(item)) - self.assertEqual(self.module.resolved(item), ("http://hub/mcp/test", "local-token")) - - -if __name__ == "__main__": - unittest.main() diff --git a/dev/test_mcphub_deploy_extension.py b/dev/test_mcphub_deploy_extension.py deleted file mode 100644 index 4bb718e..0000000 --- a/dev/test_mcphub_deploy_extension.py +++ /dev/null @@ -1,91 +0,0 @@ -from __future__ import annotations - -import argparse -import hashlib -import importlib.util -import json -import pathlib -import tempfile -import unittest - - -SOURCE = pathlib.Path(__file__).parents[1] / "platform/mcphub/deploy-extension.py" -SPEC = importlib.util.spec_from_file_location("deploy_extension", SOURCE) -assert SPEC and SPEC.loader -deploy = importlib.util.module_from_spec(SPEC) -SPEC.loader.exec_module(deploy) - - -class DeployExtensionTest(unittest.TestCase): - def setUp(self) -> None: - self.temp = tempfile.TemporaryDirectory() - self.root = pathlib.Path(self.temp.name) - self.appdata = self.root / "appdata" - self.work = self.appdata / "work/example" - self.secrets = self.root / "secrets" - self.registry = self.appdata / "config/mcp-registry.json" - self.work.mkdir(parents=True) - self.secrets.mkdir() - self.registry.parent.mkdir(parents=True) - self.registry.write_text('{"version":1,"servers":[]}\n') - artifact = self.work / "index.js" - artifact.write_text("console.log('ok')\n") - digest = hashlib.sha256(artifact.read_bytes()).hexdigest() - self.manifest = self.work / "manifest.json" - self.manifest.write_text(json.dumps({ - "server": { - "id": "example", "hermes_id": "example", "name": "Example", - "description": "Example MCP", "url": "http://host/mcp/example", - "clients": ["hermes"], - "deployment": { - "required_env": ["EXAMPLE_TOKEN"], - "required_files": ["example-config.toml"], - }, - "hub": { - "type": "stdio", "secret_file": "example.env", - "command": "node", "args": ["/app/data/extensions/example/index.js"], - "enabled": True, - }, - }, - "artifacts": [{ - "source": str(artifact), "path": "index.js", - "sha256": digest, "mode": "0644", - }], - })) - - def tearDown(self) -> None: - self.temp.cleanup() - - def args(self, **extra: object) -> argparse.Namespace: - values = { - "appdata": self.appdata, "registry": self.registry, - "secrets": self.secrets, "manifest": self.manifest, "id": "example", - "skip_api": True, - } - values.update(extra) - return argparse.Namespace(**values) - - def registered(self) -> dict: - return json.loads(self.registry.read_text())["servers"][0] - - def test_missing_secret_forces_disabled_and_unpublished(self) -> None: - deploy.stage(self.args()) - server = self.registered() - self.assertFalse(server["hub"]["enabled"]) - self.assertEqual(server["clients"], []) - self.assertTrue((self.appdata / "extensions/example/index.js").is_file()) - with self.assertRaises(SystemExit): - deploy.set_enabled(self.args(), True) - - def test_complete_secret_allows_activation(self) -> None: - (self.secrets / "example.env").write_text("EXAMPLE_TOKEN=value\n") - (self.secrets / "example-config.toml").write_text("account = 'example'\n") - deploy.stage(self.args()) - deploy.set_enabled(self.args(), True) - server = self.registered() - self.assertTrue(server["hub"]["enabled"]) - self.assertEqual(server["clients"], ["hermes"]) - - -if __name__ == "__main__": - unittest.main() diff --git a/dev/test_mcphub_git_installer.py b/dev/test_mcphub_git_installer.py deleted file mode 100644 index a3896cb..0000000 --- a/dev/test_mcphub_git_installer.py +++ /dev/null @@ -1,131 +0,0 @@ -from __future__ import annotations - -import importlib.util -import json -import pathlib -import sys -import tempfile -import unittest -from unittest import mock - - -SOURCE = pathlib.Path(__file__).parents[1] / "platform/mcphub/mcphub_git_installer.py" -SPEC = importlib.util.spec_from_file_location("mcphub_git_installer", SOURCE) -assert SPEC and SPEC.loader -installer = importlib.util.module_from_spec(SPEC) -sys.modules[SPEC.name] = installer -SPEC.loader.exec_module(installer) - - -class GitInstallerTest(unittest.TestCase): - def setUp(self) -> None: - self.temp = tempfile.TemporaryDirectory() - self.root = pathlib.Path(self.temp.name) - self.appdata = self.root / "appdata" - self.secrets = self.root / "secrets" - self.secrets.mkdir() - self.commits = iter(["a" * 40, "b" * 40, "c" * 40]) - - def tearDown(self) -> None: - self.temp.cleanup() - - def spec(self, **values: object): - data = { - "name": "example", - "repository": "https://github.com/example/mcp", - "ref": "main", - "runtime": "python", - "entrypoint": "example-mcp", - "arguments": ("--stdio",), - "required_env": ("EXAMPLE_TOKEN",), - } - data.update(values) - return installer.GitInstallSpec(**data) - - def clone(self, _spec, destination: pathlib.Path) -> str: - destination.mkdir(parents=True, exist_ok=True) - (destination / "pyproject.toml").write_text("[project]\nname='example'\n") - return next(self.commits) - - @staticmethod - def clone_same(_spec, destination: pathlib.Path) -> str: - destination.mkdir(parents=True, exist_ok=True) - (destination / "pyproject.toml").write_text("[project]\nname='example'\n") - return "a" * 40 - - @staticmethod - def build(_source: pathlib.Path, release: pathlib.Path, _entrypoint: str) -> list[str]: - release.mkdir(parents=True) - executable = release / ".venv/bin/example-mcp" - executable.parent.mkdir(parents=True) - executable.write_text("ok") - return [str(executable)] - - def test_install_is_disabled_and_reports_only_missing_key_names(self) -> None: - with mock.patch.object(installer, "_clone", self.clone), mock.patch.object( - installer, "_python_release", self.build - ): - result = installer.prepare_release(self.spec(), self.appdata, self.secrets) - self.assertFalse(result["config"]["enabled"]) - self.assertFalse(result["credentials_ready"]) - self.assertEqual(result["missing_env"], ["EXAMPLE_TOKEN"]) - self.assertEqual(result["config"]["command"], "/usr/local/bin/run-with-env") - self.assertEqual(result["config"]["args"][-1], "--stdio") - - def test_same_release_reuses_build_without_duplicating_arguments(self) -> None: - with mock.patch.object(installer, "_clone", self.clone_same), mock.patch.object( - installer, "_python_release", side_effect=self.build - ) as build: - first = installer.prepare_release(self.spec(), self.appdata, self.secrets) - second = installer.prepare_release(self.spec(), self.appdata, self.secrets) - self.assertEqual(build.call_count, 1) - self.assertEqual(first["config"]["args"], second["config"]["args"]) - self.assertEqual(second["config"]["args"].count("--stdio"), 1) - - def test_update_and_rollback_preserve_both_releases(self) -> None: - with mock.patch.object(installer, "_clone", self.clone), mock.patch.object( - installer, "_python_release", self.build - ): - first = installer.prepare_release(self.spec(), self.appdata, self.secrets) - second = installer.prepare_release(self.spec(ref="v2"), self.appdata, self.secrets) - self.assertEqual(second["previous_release"], first["release"]) - rolled = installer.rollback_release("example", self.appdata) - self.assertEqual(rolled["release"], first["release"]) - self.assertEqual(installer.current_release("example", self.appdata, self.secrets)["release"], first["release"]) - - def test_failed_update_leaves_previous_state_current(self) -> None: - with mock.patch.object(installer, "_clone", self.clone), mock.patch.object( - installer, "_python_release", self.build - ): - first = installer.prepare_release(self.spec(), self.appdata, self.secrets) - with mock.patch.object(installer, "_clone", self.clone), mock.patch.object( - installer, "_python_release", side_effect=installer.GitInstallError("build failed") - ): - with self.assertRaises(installer.GitInstallError): - installer.prepare_release(self.spec(ref="broken"), self.appdata, self.secrets) - current = installer.current_release("example", self.appdata, self.secrets) - self.assertEqual(current["release"], first["release"]) - - def test_registry_updates_only_matching_server(self) -> None: - registry = self.appdata / "config/mcp-registry.json" - registry.parent.mkdir(parents=True) - registry.write_text(json.dumps({"version": 1, "servers": [{"id": "keep", "hermes_id": "keep"}]})) - result = { - "name": "example", "repository": "https://github.com/example/mcp.git", - "requested_ref": "main", "commit": "a" * 40, "release": "a" * 12, - "required_env": [], "secret_file": None, - "config": {"type": "stdio", "command": "example", "args": [], "enabled": False}, - } - installer.update_registry(registry, installer.registry_entry(result, "Example")) - servers = json.loads(registry.read_text())["servers"] - self.assertEqual({item["id"] for item in servers}, {"keep", "example-local"}) - - def test_rejects_non_github_and_escaping_subdirectory(self) -> None: - with self.assertRaises(installer.GitInstallError): - installer.normalize_spec(self.spec(repository="https://evil.example/repo")) - with self.assertRaises(installer.GitInstallError): - installer.normalize_spec(self.spec(subdirectory="../escape")) - - -if __name__ == "__main__": - unittest.main() diff --git a/dev/test_mcphub_settings.py b/dev/test_mcphub_settings.py deleted file mode 100644 index c4629e1..0000000 --- a/dev/test_mcphub_settings.py +++ /dev/null @@ -1,94 +0,0 @@ -#!/usr/bin/env python3 -"""Regression tests for declarative, update-safe MCPHub settings.""" - -from __future__ import annotations - -import importlib.util -import json -import os -import tempfile -import unittest -from pathlib import Path - - -ROOT = Path(__file__).parents[1] -SOURCE = ROOT / "platform/mcphub/configure-settings.py" - - -def load_module(): - spec = importlib.util.spec_from_file_location("configure_settings", SOURCE) - module = importlib.util.module_from_spec(spec) - assert spec.loader - spec.loader.exec_module(module) - return module - - -class MCPHubSettingsTests(unittest.TestCase): - def setUp(self): - self.module = load_module() - self.temp = tempfile.TemporaryDirectory() - self.root = Path(self.temp.name) - self.secrets = self.root / "secrets" - self.secrets.mkdir() - (self.secrets / "remote.env").write_text( - "REMOTE_URL=http://example.test/mcp\nTOKEN=secret-value\n", - encoding="utf-8", - ) - self.registry = self.root / "registry.json" - self.registry.write_text(json.dumps({ - "version": 1, - "servers": [ - { - "id": "remote-local", - "hermes_id": "remote", - "hub": { - "type": "streamable-http", - "secret_file": "remote.env", - "url": "${REMOTE_URL}", - "headers": {"Authorization": "Bearer ${TOKEN}"}, - "enabled": True, - }, - }, - {"id": "client-only", "url": "http://unused/mcp"}, - ], - }), encoding="utf-8") - - def tearDown(self): - self.temp.cleanup() - - def test_registry_renders_only_hub_servers_and_expands_secrets(self): - servers = self.module.registry_servers(self.registry, self.secrets, {}) - self.assertEqual(list(servers), ["remote"]) - self.assertEqual(servers["remote"]["url"], "http://example.test/mcp") - self.assertEqual( - servers["remote"]["headers"]["Authorization"], - "Bearer secret-value", - ) - - def test_existing_enabled_toggle_survives_reconciliation(self): - servers = self.module.registry_servers( - self.registry, self.secrets, {"remote": {"enabled": False}} - ) - self.assertFalse(servers["remote"]["enabled"]) - - def test_missing_secret_fails_closed(self): - os.unlink(self.secrets / "remote.env") - with self.assertRaises(SystemExit): - self.module.registry_servers(self.registry, self.secrets, {}) - - def test_hermes_group_is_recoverable_and_bounded(self): - settings = {"groups": [{"id": "keep", "name": "other", "servers": []}]} - self.module.ensure_hermes_group(settings) - self.module.ensure_hermes_group(settings) - groups = settings["groups"] - self.assertEqual(len([group for group in groups if group["name"] == "hermes"]), 1) - hermes = next(group for group in groups if group["name"] == "hermes") - fritzbox = next(item for item in hermes["servers"] if item["name"] == "fritzbox") - self.assertEqual( - fritzbox["tools"], - ["list_services", "list_actions", "describe_action", "call_action"], - ) - - -if __name__ == "__main__": - unittest.main() diff --git a/dev/test_openwebui_filters.py b/dev/test_openwebui_filters.py deleted file mode 100644 index 9f5710f..0000000 --- a/dev/test_openwebui_filters.py +++ /dev/null @@ -1,616 +0,0 @@ -#!/usr/bin/env python3 -"""Offline tests for the versioned OpenWebUI filters. - -The production container provides pydantic. A tiny local stand-in keeps these -logic tests dependency-free and prevents test setup from reaching the network. -""" - -from __future__ import annotations - -import importlib.util -import json -import sys -import tempfile -import types -import unittest -from pathlib import Path - - -class _BaseModel: - def __init__(self, **values): - annotations = {} - for base in reversed(type(self).__mro__): - annotations.update(getattr(base, "__annotations__", {})) - for name in annotations: - setattr(self, name, values.get(name, getattr(type(self), name, None))) - - -fake_pydantic = types.ModuleType("pydantic") -fake_pydantic.BaseModel = _BaseModel -fake_pydantic.Field = lambda default=None, **kwargs: default -sys.modules.setdefault("pydantic", fake_pydantic) - -FILTER_DIR = Path(__file__).parents[1] / "platform" / "openwebui" / "filters" -ACTION_DIR = Path(__file__).parents[1] / "platform" / "openwebui" / "actions" - - -def _load(name: str): - spec = importlib.util.spec_from_file_location(name, FILTER_DIR / f"{name}.py") - module = importlib.util.module_from_spec(spec) - assert spec.loader is not None - spec.loader.exec_module(module) - return module - - -def _load_action(name: str): - spec = importlib.util.spec_from_file_location(name, ACTION_DIR / f"{name}.py") - module = importlib.util.module_from_spec(spec) - assert spec.loader is not None - spec.loader.exec_module(module) - return module - - -class StabilityGuardTests(unittest.IsolatedAsyncioTestCase): - async def asyncSetUp(self): - self.module = _load("stability_guard") - self.guard = self.module.Filter() - - async def test_large_tool_output_is_bounded(self): - body = { - "model": "qwen-fast", - "messages": [ - {"role": "user", "content": "Prüfe das Log."}, - {"role": "tool", "tool_call_id": "x", "content": "A" * 50000}, - ], - } - result = await self.guard.inlet(body) - content = result["messages"][1]["content"] - self.assertLessEqual(len(content), self.guard.valves.max_single_tool_chars) - self.assertIn("Werkzeugausgabe gekürzt", content) - - async def test_uncensored_uses_its_80k_context_limit(self): - self.assertEqual( - self.guard._context_limit("mikeai-uncensored"), - self.guard.valves.uncensored_context_tokens, - ) - - async def test_duplicate_calls_disable_tools(self): - call = { - "id": "call", - "type": "function", - "function": {"name": "search", "arguments": '{"q":"same"}'}, - } - body = { - "model": "qwen-fast", - "tools": [{"type": "function", "function": {"name": "search"}}], - "tool_ids": ["server:mcp:web"], - "messages": [ - {"role": "user", "content": "Suche genau einmal."}, - {"role": "assistant", "tool_calls": [call]}, - {"role": "tool", "tool_call_id": "1", "content": "nichts"}, - {"role": "assistant", "tool_calls": [call]}, - {"role": "tool", "tool_call_id": "2", "content": "nichts"}, - {"role": "assistant", "tool_calls": [call]}, - ], - } - result = await self.guard.inlet(body) - self.assertEqual(result["tools"], []) - self.assertEqual(result["tool_ids"], []) - self.assertIn("Weitere Werkzeugaufrufe", result["messages"][0]["content"]) - - async def test_old_context_is_compacted_before_current_turn(self): - self.guard.valves.default_context_tokens = 10000 - self.guard.valves.hard_context_ratio = 0.8 - self.guard.valves.reserved_output_tokens = 1000 - body = { - "model": "unknown", - "messages": [ - {"role": "system", "content": "Sicher arbeiten."}, - {"role": "user", "content": "alt " * 15000}, - {"role": "assistant", "content": "altantwort " * 8000}, - {"role": "tool", "tool_call_id": "old", "content": "log " * 20000}, - {"role": "user", "content": "Aktuelle wichtige Frage"}, - ], - } - result = await self.guard.inlet(body) - self.assertEqual(result["messages"][-1]["content"], "Aktuelle wichtige Frage") - self.assertLess(len(result["messages"][1]["content"]), 3000) - - async def test_private_csv_disables_web_and_requires_local_table_analysis(self): - body = { - "model": "qwen-fast", - "features": {"web_search": True, "code_interpreter": True}, - "tool_ids": ["server:mcp:web-local", "server:mcp:arr-local"], - "tools": [ - {"type": "function", "function": {"name": "search_web"}}, - {"type": "function", "function": {"name": "execute_code"}}, - ], - "metadata": {"files": [{"name": "private-bank.csv"}]}, - "messages": [ - {"role": "user", "content": "Sortiere Ein- und Ausgänge."}, - ], - } - result = await self.guard.inlet(body) - self.assertEqual(result["tool_ids"], []) - self.assertFalse(result["features"]["web_search"]) - self.assertTrue(result["features"]["code_interpreter"]) - self.assertEqual( - [tool["function"]["name"] for tool in result["tools"]], - ["execute_code"], - ) - self.assertIn("private table rule", result["messages"][0]["content"]) - self.assertIn("sep=None", result["messages"][0]["content"]) - self.assertIn("delimiter", result["messages"][0]["content"]) - - async def test_private_csv_is_detected_from_user_text_without_metadata(self): - body = { - "model": "qwen-fast", - "tools": [ - {"type": "function", "function": {"name": "search_web"}}, - {"type": "function", "function": {"name": "execute_code"}}, - ], - "messages": [ - {"role": "user", "content": "Werte bitte diese CSV meines Bankkontos aus."}, - ], - } - result = await self.guard.inlet(body) - self.assertEqual( - [tool["function"]["name"] for tool in result["tools"]], - ["execute_code"], - ) - - async def test_private_csv_stops_before_openwebui_hard_tool_limit(self): - calls = [] - for index in range(8): - calls.extend( - [ - { - "role": "assistant", - "tool_calls": [ - { - "function": { - "name": "execute_code", - "arguments": '{"code":"step %d"}' % index, - } - } - ], - }, - {"role": "tool", "content": "ok", "tool_call_id": str(index)}, - ] - ) - body = { - "model": "qwen-fast", - "tools": [ - {"type": "function", "function": {"name": "execute_code"}}, - ], - "messages": [ - {"role": "user", "content": "Werte diese CSV aus."}, - *calls, - ], - } - result = await self.guard.inlet(body) - self.assertEqual(result["tools"], []) - self.assertIn("vorhandenen Ergebnisse", result["messages"][0]["content"]) - - -class AutoToolSelectorTests(unittest.IsolatedAsyncioTestCase): - async def asyncSetUp(self): - self.module = _load("auto_tool_selector") - self.selector = self.module.Filter() - - async def _select(self, prompt: str, existing=None): - body = { - "model": "mikeai-medium", - "messages": [{"role": "user", "content": prompt}], - } - if existing is not None: - body["tool_ids"] = existing - return await self.selector.inlet(body) - - async def test_homeassistant_is_selected_for_room_temperature(self): - result = await self._select("Wie warm ist es gerade in der Küche?") - self.assertEqual( - result["tool_ids"], ["server:mcp:homeassistant-local"] - ) - self.assertIn("Never call ha_list_states merely", result["messages"][0]["content"]) - self.assertIn("Never infer an automation entity_id", result["messages"][0]["content"]) - - async def test_hyphenated_homeassistant_and_tool_name_are_selected(self): - result = await self._select( - "Führe einen Home-Assistant-Test aus und nutze ha_list_states genau einmal." - ) - self.assertEqual(result["tool_ids"], ["server:mcp:homeassistant-local"]) - - async def test_general_web_is_available_without_site_specific_rules(self): - result = await self._select("Erkläre mir kurz, wie ein Fahrrad funktioniert.") - self.assertTrue(result["features"]["web_search"]) - self.assertTrue(result["metadata"]["features"]["web_search"]) - self.assertNotIn("tool_ids", result) - - async def test_weather_gets_native_web_and_compact_fallback(self): - result = await self._select("Soll es heute in Rastatt regnen?") - self.assertEqual(result["tool_ids"], ["server:mcp:web-general-local"]) - - async def test_youtube_channel_question_gets_general_web_fallback(self): - result = await self._select( - "Welches Video steht aktuell oben auf dem YouTube-Kanal The Proper People?" - ) - self.assertEqual(result["tool_ids"], ["server:mcp:web-general-local"]) - - async def test_unraid_uses_readonly_not_mua(self): - result = await self._select( - "Welche Docker-Container laufen aktuell auf Unraid?" - ) - self.assertEqual( - result["tool_ids"], ["server:mcp:mua-readonly-local"] - ) - self.assertNotIn("server:mcp:mua", result["tool_ids"]) - self.assertIn("bounded evidence ladder", result["messages"][0]["content"]) - self.assertIn("Do not dump complete configuration files", result["messages"][0]["content"]) - - async def test_explicit_unraid_update_gets_read_and_management_tools(self): - result = await self._select( - "Prüfe auf Unraid alle Docker-Updates, führe die Updates durch und kontrolliere danach den Zustand." - ) - self.assertEqual( - result["tool_ids"], - ["server:mcp:mua-readonly-local", "server:mcp:mua"], - ) - self.assertIn("single batched update workflow", result["messages"][0]["content"]) - - async def test_unraid_update_question_stays_readonly(self): - result = await self._select( - "Gibt es auf Unraid Updates für Docker-Container? Bitte nur prüfen." - ) - self.assertEqual( - result["tool_ids"], ["server:mcp:mua-readonly-local"] - ) - - async def test_readonly_unraid_media_audit_does_not_attach_admin_or_arr(self): - result = await self._select( - "Prüfe auf Unraid, welche Folgen meiner Hörspielserie Die drei Fragezeichen fehlen. " - "Nur lesen, ohne Downloads, Umbenennungen oder sonstige Änderungen." - ) - self.assertEqual( - result["tool_ids"], ["server:mcp:mua-readonly-local"] - ) - self.assertIn("unraid_files_inventory", result["messages"][0]["content"]) - - async def test_unraid_media_audit_with_deezer_enables_native_web(self): - result = await self._select( - "Prüfe auf Unraid, welche Folgen meiner Hörspielserie fehlen, ermittle die " - "aktuelle offizielle Liste online und prüfe jede fehlende Folge bei Deezer. " - "Nur lesen, ohne Downloads oder Änderungen." - ) - self.assertEqual( - result["tool_ids"], - ["server:mcp:mua-readonly-local", "server:mcp:web-general-local"], - ) - self.assertTrue(result["features"]["web_search"]) - self.assertTrue(result["metadata"]["features"]["web_search"]) - self.assertIn("search_web/fetch_url", result["messages"][0]["content"]) - self.assertEqual(result["reasoning_effort"], "medium") - - async def test_unraid_media_write_gets_portable_operator_and_web(self): - result = await self._select( - "Ermittle online das neueste Video, lade es mit yt-dlp auf dem Unraid-Host " - "in einen temporären Ordner, konvertiere es mit ffmpeg und lege die fertige " - "Datei im Share Transfer ab." - ) - self.assertEqual( - result["tool_ids"], - [ - "server:mcp:athena-operator-local", - "server:mcp:mua-readonly-local", - "server:mcp:mua", - "server:mcp:web-general-local", - ], - ) - self.assertIn("direct MUA management or shell tool", result["messages"][0]["content"]) - self.assertIn("Do not ask the user to enable another tool", result["messages"][0]["content"]) - self.assertIn("start-status-result pattern", result["messages"][0]["content"]) - self.assertIn("transient-by-default dependency handling", result["messages"][0]["content"]) - self.assertIn("task-local copy under /tmp", result["messages"][0]["content"]) - self.assertTrue(result["metadata"]["mikeai_long_operator_task"]) - - async def test_generic_remote_host_file_operation_gets_operator(self): - result = await self._select( - "Führe auf dem Server ein vorhandenes Skript aus und speichere die neue Datei unter /data/export." - ) - self.assertEqual(result["tool_ids"], ["server:mcp:athena-operator-local"]) - self.assertTrue(result["metadata"]["mikeai_long_operator_task"]) - - async def test_readonly_diagnostics_do_not_get_long_operator_budget(self): - result = await self._select("Prüfe nur lesend den Zustand von Unraid.") - self.assertNotIn("mikeai_long_operator_task", result["metadata"]) - - async def test_narrow_safety_clause_does_not_cancel_authorized_unraid_write(self): - result = await self._select( - "Lade das Video auf dem Unraid-Host herunter, konvertiere es und lege es " - "im Transfer-Share ab. Installiere dabei kein Paket dauerhaft und ändere " - "keine Unraid-Systemkonfiguration." - ) - self.assertEqual( - result["tool_ids"], - [ - "server:mcp:athena-operator-local", - "server:mcp:mua-readonly-local", - "server:mcp:mua", - ], - ) - - async def test_plain_download_advice_does_not_attach_operator(self): - result = await self._select("Erkläre mir, wie ein Browser einen Download technisch durchführt.") - self.assertNotIn("tool_ids", result) - - async def test_voice_transcription_variants_select_unraid(self): - result = await self._select( - "Welche Dacher Contäner laufen aktuell auf dem Anrate Server?" - ) - self.assertEqual( - result["tool_ids"], ["server:mcp:mua-readonly-local"] - ) - - async def test_navidrome_is_selected(self): - result = await self._select( - "Schau in Navidrome nach ähnlichen Titeln und meiner Playlist." - ) - self.assertEqual(result["tool_ids"], ["server:mcp:navidrome-local"]) - - async def test_platform_context_is_selected(self): - result = await self._select( - "Wie ist der KI-Host Athena aufgebaut und wo liegt der Recovery-Koffer?" - ) - self.assertEqual(result["tool_ids"], ["server:mcp:athena-operator-local"]) - - async def test_athena_operator_is_selected_for_platform_work(self): - result = await self._select( - "Baue und deploye auf Athena einen neuen MCP-Container." - ) - self.assertEqual( - result["tool_ids"], ["server:mcp:athena-operator-local"] - ) - - async def test_mcp_build_from_github_gets_source_and_operator(self): - result = await self._select( - "Ich möchte hierfür einen MCP bauen: https://github.com/foo/bar" - ) - self.assertEqual( - result["tool_ids"], - ["server:mcp:github-local", "server:mcp:athena-operator-local"], - ) - - async def test_existing_unraid_backend_gets_operator_and_runtime_evidence(self): - result = await self._select( - "Baue aus https://github.com/foo/deemix einen MCP. Deemix läuft bereits als Container auf Unraid; prüfe ihn zuerst." - ) - self.assertEqual( - result["tool_ids"], - [ - "server:mcp:github-local", - "server:mcp:athena-operator-local", - "server:mcp:mua-readonly-local", - ], - ) - self.assertIn("integrate or relay", result["messages"][0]["content"]) - - async def test_readonly_integration_plan_does_not_attach_unraid_admin(self): - result = await self._select( - "Analysiere https://github.com/foo/deemix und prüfe den auf Unraid " - "laufenden Container und entwirf einen Athena-MCP. Erstelle keinen zweiten Container, ändere nichts " - "und lies keine Secrets." - ) - self.assertEqual( - result["tool_ids"], - [ - "server:mcp:github-local", - "server:mcp:athena-operator-local", - "server:mcp:mua-readonly-local", - ], - ) - - async def test_github_and_explicit_web_adds_compact_general_web(self): - result = await self._select( - "Prüfe dieses GitHub Repository und suche zusätzlich im Netz nach Nutzerstimmen." - ) - self.assertEqual( - result["tool_ids"], - ["server:mcp:github-local", "server:mcp:web-general-local"], - ) - - async def test_plain_public_web_request_attaches_general_not_legacy_web(self): - result = await self._select( - "Suche im Netz auf MakerWorld einen Schlümpfe-Schlüsselanhänger." - ) - self.assertNotIn("server:mcp:web-local", result.get("tool_ids", [])) - self.assertIn("server:mcp:web-general-local", result.get("tool_ids", [])) - - async def test_ebay_research_gets_general_marketplace_protocol(self): - result = await self._select( - "Suche auf eBay nach einem vollständigen Highscreen 386 PC mit Preis und Versand." - ) - self.assertEqual(result["tool_ids"], ["server:mcp:web-general-local"]) - self.assertIn("Marketplace research protocol", result["messages"][0]["content"]) - self.assertIn("roughly three searches and five page fetches", result["messages"][0]["content"]) - self.assertTrue(result["metadata"]["mikeai_marketplace_research"]) - - async def test_generic_marketplace_research_does_not_need_site_rule(self): - result = await self._select( - "Finde auf einem Marktplatz aktuelle Angebote für einen gebrauchten Synthesizer." - ) - self.assertEqual(result["tool_ids"], ["server:mcp:web-general-local"]) - self.assertIn("deduplicate by listing URL or item number", result["messages"][0]["content"]) - self.assertTrue(result["metadata"]["mikeai_marketplace_research"]) - - async def test_manual_tool_is_preserved(self): - result = await self._select( - "Prüfe Sonarr.", ["server:mcp:manually-selected"] - ) - self.assertEqual( - result["tool_ids"], - ["server:mcp:manually-selected", "server:mcp:arr-local"], - ) - - async def test_plain_chat_gets_no_tools(self): - result = await self._select("Erkläre mir den Unterschied zwischen RAM und SSD.") - self.assertNotIn("tool_ids", result) - - async def test_selection_adds_write_safety_rule(self): - result = await self._select("Zeige mir die Home Assistant Automatisierungen.") - self.assertEqual(result["messages"][0]["role"], "system") - self.assertIn("Availability is not authorization", result["messages"][0]["content"]) - - -class MetricsTests(unittest.IsolatedAsyncioTestCase): - async def test_metrics_file_contains_no_chat_content_or_ids(self): - module = _load("local_performance_metrics") - metrics = module.Filter() - with tempfile.TemporaryDirectory() as directory: - path = Path(directory) / "metrics.jsonl" - metrics.valves.metrics_path = str(path) - metadata = {"message_id": "secret-message-id", "chat_id": "secret-chat-id"} - await metrics.inlet( - { - "model": "qwen-fast", - "messages": [{"role": "user", "content": "private prompt"}], - "tools": [{"name": "tool"}], - }, - __metadata__=metadata, - ) - await metrics.outlet( - { - "model": "qwen-fast", - "messages": [ - { - "role": "assistant", - "content": "private answer", - "usage": { - "prompt_tokens": 10, - "completion_tokens": 4, - "total_tokens": 14, - }, - } - ], - }, - __metadata__=metadata, - ) - raw = path.read_text() - record = json.loads(raw) - self.assertEqual(record["prompt_tokens"], 10) - self.assertNotIn("private", raw) - self.assertNotIn("secret", raw) - - -class SecretRedactionTests(unittest.IsolatedAsyncioTestCase): - async def test_only_tool_and_assistant_content_is_redacted(self): - module = _load("secret_redaction") - guard = module.Filter() - token = "eyJ" + "A" * 24 + "." + "B" * 24 + "." + "C" * 16 - body = { - "messages": [ - {"role": "user", "content": f"Absichtlich lokal nutzen: {token}"}, - {"role": "tool", "content": f'{{"api_key":"1234567890abcdef"}} {token}'}, - ] - } - result = await guard.inlet(body) - self.assertIn(token, result["messages"][0]["content"]) - self.assertNotIn(token, result["messages"][1]["content"]) - self.assertNotIn("1234567890abcdef", result["messages"][1]["content"]) - - outlet = { - "messages": [ - {"role": "assistant", "content": "Bearer " + "abcdefghijklmnopqrstuvwxyz"} - ] - } - result = await guard.outlet(outlet) - self.assertNotIn("abcdefghijklmnopqrstuvwxyz", result["messages"][0]["content"]) - - async def test_documentation_placeholders_and_paths_are_not_redacted(self): - module = _load("secret_redaction") - guard = module.Filter() - content = "\n".join( - [ - 'ROUTER_API_KEY=${ROUTER_API_KEY:?required}', - 'API_KEY=/etc/mike-ai/router-api-key', - 'access_token=', - 'password=[REDACTED]', - 'API-Key: Inhalt von /etc/mike-ai/router-api-key', - ] - ) - redacted, counts = guard._redact(content) - self.assertEqual(redacted, content) - self.assertEqual(counts, {}) - - async def test_notification_reports_category_and_origin_but_not_value(self): - module = _load("secret_redaction") - guard = module.Filter() - events = [] - - async def emit(event): - events.append(event) - - secret = "realistic-secret-value-123456" - body = {"messages": [{"role": "tool", "content": f"api_key={secret}"}]} - result = await guard.inlet(body, __event_emitter__=emit) - self.assertNotIn(secret, result["messages"][0]["content"]) - description = events[0]["data"]["description"] - self.assertIn("API-Key: 1", description) - self.assertIn("Werkzeugausgaben", description) - self.assertIn("nicht protokolliert", description) - self.assertNotIn(secret, description) - - -class QuickActionTests(unittest.IsolatedAsyncioTestCase): - async def asyncSetUp(self): - self.module = _load_action("quick_actions") - self.action = self.module.Action() - self.body = { - "id": "assistant-1", - "model": "qwen-fast", - "messages": [ - {"id": "user-1", "role": "user", "content": "Frage"}, - {"id": "assistant-1", "role": "assistant", "content": "Lange Antwort"}, - ], - } - - async def test_summary_appends_to_selected_message(self): - async def fake_completion(*args, **kwargs): - return "- Kurze Antwort" - - self.action._completion = fake_completion - result = await self.action.action(self.body, "summary") - self.assertEqual(result["messages"][0]["id"], "assistant-1") - self.assertIn("### Kurzfassung", result["messages"][0]["content"]) - self.assertIn("Kurze Antwort", result["messages"][0]["content"]) - - async def test_source_check_uses_web_evidence(self): - completions = iter(["Qwen Fakten", "Bestätigt: Aussage [https://example.invalid]"]) - - async def fake_completion(*args, **kwargs): - return next(completions) - - async def fake_web(query): - self.assertEqual(query, "Qwen Fakten") - return '{"sources":[{"url":"https://example.invalid"}]}' - - self.action._completion = fake_completion - self.action._web_research = fake_web - result = await self.action.action(self.body, "sources") - self.assertIn("### Quellenprüfung", result["messages"][0]["content"]) - self.assertIn("example.invalid", result["messages"][0]["content"]) - - async def test_markdown_copy_uses_browser_clipboard(self): - calls = [] - - async def event_call(event): - calls.append(event) - return True - - await self.action.action(self.body, "copy_markdown", __event_call__=event_call) - self.assertEqual(calls[0]["type"], "execute") - self.assertIn("navigator.clipboard.writeText", calls[0]["data"]["code"]) - self.assertIn("Lange Antwort", calls[0]["data"]["code"]) - - -if __name__ == "__main__": - unittest.main() diff --git a/dev/test_radarr_patch.py b/dev/test_radarr_patch.py deleted file mode 100644 index 7de0ef8..0000000 --- a/dev/test_radarr_patch.py +++ /dev/null @@ -1,84 +0,0 @@ -#!/usr/bin/env python3 -"""Focused offline tests for the model-oriented Radarr overlay.""" - -from __future__ import annotations - -import importlib.util -import sys -import types -import unittest -from pathlib import Path - - -SOURCE = Path(__file__).parents[1] / "platform/mcp/patches/mcp_radarr.py" - - -def load_module(): - fastmcp = types.ModuleType("fastmcp") - fastmcp.FastMCP = object - sys.modules["fastmcp"] = fastmcp - pydantic = types.ModuleType("pydantic") - pydantic.Field = lambda *args, **kwargs: kwargs.get("default") - sys.modules["pydantic"] = pydantic - auth = types.ModuleType("arr_mcp.auth") - auth.get_radarr_client = lambda: None - sys.modules["arr_mcp"] = types.ModuleType("arr_mcp") - sys.modules["arr_mcp.auth"] = auth - spec = importlib.util.spec_from_file_location("radarr_patch", SOURCE) - module = importlib.util.module_from_spec(spec) - assert spec.loader - spec.loader.exec_module(module) - return module - - -class RadarrPatchTests(unittest.TestCase): - def setUp(self): - self.module = load_module() - self.movies = [{ - "id": 12, "title": "Example", "year": 2024, "hasFile": True, - "alternateTitles": [{"title": "large unwanted block"}], - "movieFile": { - "id": 44, "relativePath": "Example.mkv", "size": 2147483648, - "quality": {"quality": {"name": "Bluray-1080p"}}, - "mediaInfo": { - "videoCodec": "x264", "resolution": "1920x1080", - "videoBitDepth": 8, "audioCodec": "EAC3", - "audioLanguages": "ger/eng", "subtitles": "ger", - }, - }, - }] - - def test_inventory_is_compact_and_alias_aware(self): - result = self.module._compact_inventory(self.movies, codecs="h264") - self.assertEqual(result["totalMatched"], 1) - self.assertEqual(result["movies"][0]["videoCodec"], "x264") - self.assertEqual(result["movies"][0]["sizeGiB"], 2.0) - self.assertNotIn("alternateTitles", result["movies"][0]) - - def test_filter_and_pagination_return_valid_bounded_data(self): - result = self.module._compact_inventory(self.movies * 5, query="example", offset=1, limit=2) - self.assertEqual(result["returned"], 2) - self.assertTrue(result["hasMore"]) - - def test_surface_has_only_explicit_read_tools(self): - class FakeMcp: - def __init__(self): - self.names = [] - - def tool(self, **_kwargs): - def decorate(function): - self.names.append(function.__name__) - return function - return decorate - - mcp = FakeMcp() - self.module.register_radarr_tools(mcp) - self.assertEqual( - mcp.names, - ["radarr_find_movie", "radarr_movie_codec_inventory", "radarr_search_releases"], - ) - self.assertNotIn("radarr_action", mcp.names) - - -if __name__ == "__main__": - unittest.main() diff --git a/dev/test_sonarr_release_grab.py b/dev/test_sonarr_release_grab.py deleted file mode 100755 index cc82eb6..0000000 --- a/dev/test_sonarr_release_grab.py +++ /dev/null @@ -1,195 +0,0 @@ -#!/usr/bin/env python3 -"""Focused offline tests for the model-oriented Sonarr overlay.""" - -from __future__ import annotations - -import importlib.util -import sys -import types -import unittest -from pathlib import Path -from unittest.mock import patch - - -SOURCE = Path(__file__).parents[1] / "platform/mcp/patches/mcp_sonarr.py" - - -def load_module(): - async def run_blocking(function, *args, **kwargs): - kwargs.pop("service", None) - return function(*args, **kwargs) - - def dispatch(client, action, kwargs, **_options): - return getattr(client, action)(**kwargs) - - utilities = types.ModuleType("agent_utilities.mcp_utilities") - utilities.dispatch = dispatch - utilities.run_blocking = run_blocking - sys.modules["agent_utilities"] = types.ModuleType("agent_utilities") - sys.modules["agent_utilities.mcp_utilities"] = utilities - fastmcp = types.ModuleType("fastmcp") - fastmcp.FastMCP = object - sys.modules["fastmcp"] = fastmcp - pydantic = types.ModuleType("pydantic") - pydantic.Field = lambda *args, **kwargs: kwargs.get("default") - sys.modules["pydantic"] = pydantic - auth = types.ModuleType("arr_mcp.auth") - auth.get_sonarr_client = lambda: None - sys.modules["arr_mcp"] = types.ModuleType("arr_mcp") - sys.modules["arr_mcp.auth"] = auth - spec = importlib.util.spec_from_file_location("sonarr_patch", SOURCE) - module = importlib.util.module_from_spec(spec) - assert spec.loader - spec.loader.exec_module(module) - return module - - -class FakeSonarrClient: - def __init__(self, *, rejected: bool = False) -> None: - self.posted = [] - self.release = { - "guid": "exact-guid", - "title": "Murder.She.Wrote.S07.German.AC3D.DL.1080p.WebHD.x265-FuN", - "indexer": "Test Indexer", - "indexerId": 7, - "size": 27_600_000_000, - "protocol": "usenet", - "downloadAllowed": not rejected, - "releaseGroup": "FuN", - "seasonNumber": 7, - "fullSeason": True, - "rejections": ["Existing file has equal or better quality"] if rejected else [], - } - - def get_release(self, **_kwargs): - return [dict(self.release)] - - def get_episode(self, **_kwargs): - return [ - {"id": 1, "seasonNumber": 7, "episodeNumber": 1, "hasFile": True}, - {"id": 2, "seasonNumber": 7, "episodeNumber": 2, "hasFile": False}, - ] - - def post_release(self, data=None, **kwargs): - payload = data if data is not None else kwargs - self.posted.append(payload) - return payload - - -class ReleaseGrabTests(unittest.IsolatedAsyncioTestCase): - def setUp(self) -> None: - self.module = load_module() - self.module._APPROVALS.clear() - - async def test_exact_release_requires_preview_and_ticket(self) -> None: - client = FakeSonarrClient() - scope = {"series_id": 42, "season_number": 7, "guid": "exact-guid"} - preview = await self.module._preview_release_grab(client, scope) - self.assertFalse(preview["download_started"]) - self.assertEqual(preview["existing_episode_files_in_season"], 1) - self.assertTrue(preview["approval_ticket"]) - - result = await self.module._grab_release( - client, - {**scope, "confirm": True, "approval_ticket": preview["approval_ticket"]}, - ) - self.assertTrue(result["download_started"]) - self.assertFalse(result["replacement_guaranteed"]) - self.assertEqual(len(client.posted), 1) - self.assertEqual(client.posted[0]["guid"], "exact-guid") - - async def test_rejected_release_needs_force_in_preview(self) -> None: - client = FakeSonarrClient(rejected=True) - scope = {"series_id": 42, "season_number": 7, "guid": "exact-guid"} - blocked = await self.module._preview_release_grab(client, scope) - self.assertTrue(blocked["force_required"]) - self.assertIsNone(blocked["approval_ticket"]) - - approved = await self.module._preview_release_grab(client, {**scope, "force": True}) - self.assertFalse(approved["force_required"]) - self.assertTrue(approved["approval_ticket"]) - - async def test_guid_must_still_match_current_sonarr_results(self) -> None: - client = FakeSonarrClient() - with self.assertRaisesRegex(ValueError, "no longer present"): - await self.module._preview_release_grab( - client, - {"series_id": 42, "season_number": 7, "guid": "different-guid"}, - ) - - async def test_search_can_limit_results_to_group_and_season_pack(self) -> None: - client = FakeSonarrClient() - client.get_release = lambda **_kwargs: [ - dict(client.release), - { - **client.release, - "guid": "episode-guid", - "title": "Mord.ist.ihr.Hobby.S07E02.German.1080p-FuN", - "fullSeason": False, - }, - { - **client.release, - "guid": "other-group", - "title": "Mord.ist.ihr.Hobby.S07.German.1080p-HQC", - "releaseGroup": "HQC", - }, - ] - result = await self.module._search_releases( - client, - { - "series_id": 42, - "season_number": 7, - "release_group": "FuN", - "season_pack_only": True, - }, - ) - self.assertEqual(result["results"]["total"], 1) - self.assertEqual(result["results"]["items"][0]["guid"], "exact-guid") - - def test_read_only_surface_is_explicit_and_has_no_generic_action(self) -> None: - class FakeMcp: - def __init__(self): - self.names = [] - - def tool(self, **_kwargs): - def decorate(function): - self.names.append(function.__name__) - return function - return decorate - - with patch.dict("os.environ", {"ARR_MCP_WRITE": "0"}): - mcp = FakeMcp() - self.module.register_sonarr_tools(mcp) - self.assertEqual( - mcp.names, - [ - "sonarr_find_series", - "sonarr_get_season_summary", - "sonarr_search_releases", - "sonarr_system_status", - ], - ) - self.assertNotIn("sonarr_action", mcp.names) - - def test_write_tools_are_registered_only_when_enabled(self) -> None: - class FakeMcp: - def __init__(self): - self.names = [] - - def tool(self, **_kwargs): - def decorate(function): - self.names.append(function.__name__) - return function - return decorate - - with patch.dict("os.environ", {"ARR_MCP_WRITE": "1"}): - mcp = FakeMcp() - self.module.register_sonarr_tools(mcp) - self.assertIn("sonarr_preview_release_grab", mcp.names) - self.assertIn("sonarr_grab_release", mcp.names) - self.assertIn("sonarr_preview_episode_search", mcp.names) - self.assertIn("sonarr_start_episode_search", mcp.names) - - -if __name__ == "__main__": - unittest.main() diff --git a/dev/test_web_search_mcp.py b/dev/test_web_search_mcp.py deleted file mode 100644 index 37ef82a..0000000 --- a/dev/test_web_search_mcp.py +++ /dev/null @@ -1,148 +0,0 @@ -#!/usr/bin/env python3 -"""Regression tests for the compact web MCP facade.""" - -from __future__ import annotations - -import importlib.util -import json -import pathlib -import unittest -from unittest import mock - - -ROOT = pathlib.Path(__file__).resolve().parents[1] -SPEC = importlib.util.spec_from_file_location( - "web_search_mcp", ROOT / "platform/web-search/web_search_mcp.py" -) -WEB = importlib.util.module_from_spec(SPEC) -assert SPEC.loader -SPEC.loader.exec_module(WEB) - - -class WebSearchMcpTests(unittest.TestCase): - def setUp(self) -> None: - WEB._search_attempts.clear() - - def test_tool_surface_stays_small_and_explicit(self) -> None: - self.assertEqual( - [tool["name"] for tool in WEB.TOOLS], - ["web_search", "web_read", "web_youtube", "web_compare", "web_shop", "web_research"], - ) - - def test_current_queries_do_not_get_wikipedia_noise(self) -> None: - with ( - mock.patch.object(WEB, "SEARXNG_URL", "http://searxng:8080"), - mock.patch.object(WEB, "searxng_json", return_value={"results": []}), - mock.patch.object(WEB, "wikipedia_search") as wikipedia, - ): - results, _ = WEB.general_discovery("latest video The Proper People", 4) - self.assertEqual(results, []) - wikipedia.assert_not_called() - - def test_empty_fresh_search_relaxes_once_and_marks_result(self) -> None: - hit = {"title": "Current page", "url": "https://example.com/current"} - with mock.patch.object( - WEB, - "general_discovery", - side_effect=[([], []), ([hit], [])], - ) as discovery: - result = WEB.web_search({ - "query": "current test release", - "freshness": "week", - "max_results": 3, - }) - self.assertTrue(result["task_complete"]) - self.assertFalse(result["freshness_applied"]) - self.assertIn("unfiltered", result["backend_warning"]) - self.assertEqual(discovery.call_count, 2) - - def test_related_search_budget_is_enforced(self) -> None: - with mock.patch.object(WEB, "SEARCH_BUDGET_MAX_RELATED_CALLS", 2): - self.assertTrue(WEB.consume_search_budget("latest Proper People video")[0]) - self.assertTrue(WEB.consume_search_budget("Proper People newest video")[0]) - self.assertFalse(WEB.consume_search_budget("newest video by Proper People")[0]) - - def test_youtube_feed_provides_order_and_dates(self) -> None: - feed = b''' - - new123Newest - 2026-08-23T12:00:00+00:00 - The Proper People - old456Older - 2026-08-10T12:00:00+00:00 - The Proper People - ''' - with mock.patch.object(WEB, "fetch_public_bytes", return_value=feed): - rows = WEB.youtube_feed_records( - "https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw", 2 - ) - self.assertEqual([row["title"] for row in rows], ["Newest", "Older"]) - self.assertEqual(rows[0]["published_at"], "2026-08-23T12:00:00+00:00") - - def test_latest_youtube_is_one_bounded_specialist_operation(self) -> None: - row = { - "title": "Newest", - "url": "https://www.youtube.com/watch?v=new123", - "source_kind": "youtube_channel_feed", - } - with ( - mock.patch.object(WEB, "resolve_youtube_channel", return_value="https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw"), - mock.patch.object(WEB, "youtube_feed_records", return_value=[row]), - mock.patch.object(WEB, "run_ytdlp") as ytdlp, - ): - result = WEB.web_youtube({"query": "The Proper People", "mode": "latest"}) - self.assertTrue(result["task_complete"]) - self.assertEqual(result["results"][0]["title"], "Newest") - ytdlp.assert_not_called() - - def test_latest_long_youtube_uses_verified_videos_tab(self) -> None: - row = { - "title": "Newest long video", - "url": "https://www.youtube.com/watch?v=long123", - "content_type": "long", - "content_type_verified": True, - "source_kind": "youtube_videos_tab", - } - with ( - mock.patch.object( - WEB, - "resolve_youtube_channel", - return_value="https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw", - ), - mock.patch.object(WEB, "youtube_tab_records", return_value=[row]) as tab, - mock.patch.object(WEB, "youtube_feed_records") as feed, - ): - result = WEB.web_youtube({ - "query": "The Proper People", - "mode": "latest", - "content_type": "long", - }) - tab.assert_called_once_with( - "https://www.youtube.com/channel/UCcem9I78ybZLHLRUlkUO3sw", "long", 5 - ) - feed.assert_not_called() - self.assertEqual(result["content_type_filter"], "long") - self.assertTrue(result["results"][0]["content_type_verified"]) - - def test_youtube_rejects_content_filter_outside_latest_mode(self) -> None: - with self.assertRaisesRegex(ValueError, "only supported with mode=latest"): - WEB.web_youtube({ - "query": "The Proper People", - "mode": "search", - "content_type": "long", - }) - - def test_web_read_does_not_consume_search_loop_budget(self) -> None: - page = {"url": "https://example.com/a", "page_evidence": ["Evidence"]} - with mock.patch.object(WEB, "scrape", return_value=[page]): - result = json.loads(WEB.call_tool("web_read", { - "url": "https://example.com/a", - "question": "What does this page say?", - })) - self.assertTrue(result["task_complete"]) - self.assertEqual(WEB._search_attempts, []) - - -if __name__ == "__main__": - unittest.main() diff --git a/docs/ARCHITECTURE.md b/docs/ARCHITECTURE.md new file mode 100644 index 0000000..00ca4bf --- /dev/null +++ b/docs/ARCHITECTURE.md @@ -0,0 +1,35 @@ +# Aktuelle Architektur + +```mermaid +flowchart LR + C[Hermes Desktop / Web / Mobil] --> H[Hermes Agent
Unraid] + H -->|OpenAI API| R[Profile Router
Athena :8081] + R --> P[Profile Controller] + P --> Q[genau ein llama.cpp-Profil
Qwen Fast / Medium / Large / Ultra / Uncensored] + R --> I[Z-Image-Turbo
RTX 5080, bei Bedarf] + R --> T[XTTS RTX 3060
Piper CPU-Fallback] + + H --> U[MUA / Unraid MCP] + H --> A[ARR-MCP] + H --> D[Deemix-MCP] + H --> N[Navidrome-MCP] + H --> S[STRATO-MCP] + H --> X[Nginx-Proxy-Manager-MCP] + U --> M[Media-Tools
ffmpeg / ffprobe / yt-dlp] + + W[WireGuard-Gateway
Athena] --- R + W --- B[Athena Dashboard :8099] + W --- O[Athena Operator] + K[Backup alle 5 Stunden] --> DATA[/data und /etc/mike-ai] +``` + +## Verantwortung + +- **Unraid** hält Hermes, Chats, Skills, Fach-MCPs und deren Appdata. +- **Athena** rechnet: Text, Bild und Sprache; Router und Dashboard koordinieren + und beobachten die Inferenz. +- **MUA** verwaltet Unraid. **Athena Operator** bleibt auf den Athena-Host + begrenzt. +- Der Router ist die einzige Modelladresse, die Hermes kennen muss. + +Die visuelle Fassung liegt als `athena-architecture-map.png` neben dieser Datei. diff --git a/docs/MCP_SERVERS.md b/docs/MCP_SERVERS.md index 09b7da1..e4b5cc4 100644 --- a/docs/MCP_SERVERS.md +++ b/docs/MCP_SERVERS.md @@ -1,15 +1,30 @@ -# MCP-Server +# Produktive MCP- und Werkzeugdienste -Diese Datei wird aus `config/mcp-registry.json` erzeugt. Änderungen gehören nur in die JSON-Registry. +Die portablen Werkzeuge laufen als getrennte Container auf Unraid. Hermes bindet +sie direkt ein; MCPHub ist nicht mehr im Datenpfad. -| Server | Hermes-ID | Endpunkt | Clients | Werkzeuge | -|---|---|---|---|---:| -| Athena Operator | `athena-operator` | `http://192.168.1.2:8787/mcp/athena-operator` | hermes, openwebui | alle | -| GitHub (offiziell, read-only) | `github` | `http://192.168.1.2:8787/mcp/github` | hermes, openwebui | alle | -| Home Assistant | `homeassistant-admin` | `http://192.168.1.2:8787/mcp/homeassistant` | hermes, openwebui | alle | -| Sonarr und Radarr | `arr` | `http://192.168.1.2:8787/mcp/arr` | hermes, openwebui | alle | -| Navidrome | `navidrome` | `http://192.168.1.2:8787/mcp/navidrome` | hermes, openwebui | alle | -| MUA (Unraid-Verwaltung) | `unraid` | `http://192.168.1.2:8787/mcp/unraid` | hermes, openwebui | alle | -| FRITZ!Box | `fritzbox` | `http://192.168.1.2:8787/mcp/fritzbox` | hermes, openwebui | 4 | +| Dienst | Endpunkt | Betrieb | +|---|---|---| +| MUA / Unraid | `http://192.168.1.2:3002/mcp` | Unraid-Plugin | +| ARR | `http://192.168.1.2:8207/mcp` | `ARR-MCP` | +| Deemix | `http://192.168.1.2:8209/mcp` | `Deemix-MCP` | +| Navidrome | `http://192.168.1.2:8210/mcp` | `Navidrome-MCP` | +| STRATO DNS | `http://192.168.1.2:2030/mcp` | `Strato-MCP` | +| Nginx Proxy Manager | `http://192.168.1.2:8767/mcp` | `Nginx-Proxy-Manager-MCP` | +| Home Assistant | Home-Assistant-MCP-Endpunkt | derzeit in Hermes deaktiviert | +| Athena Operator | interner Athena-Dienst | nur für Athena-Administration | -Allgemeine Webrecherche ist ein eingebautes Hermes-Werkzeug und kein MCPHub-Server. +`Media-Tools` ist kein MCP. Der Container ist eine persistente Werkzeugkiste +für ffmpeg, ffprobe, yt-dlp und ähnliche Hilfsprogramme und wird über das +Unraid-Terminalwerkzeug angesprochen. + +Allgemeine Websuche und Terminal sind Hermes-eigene Werkzeuge. Sie benötigen +keinen zusätzlichen Athena-Container. + +## Zuständigkeit + +- Athena-Host verändern: Athena Operator +- Unraid und Container verwalten: MUA / Unraid +- Mediendienste: jeweiliger Fach-MCP +- fehlende CLI-Medienwerkzeuge: Media-Tools +- allgemeine Recherche: Hermes-Webwerkzeug diff --git a/docs/RECOVERY.md b/docs/RECOVERY.md index 45c09fb..6e2a760 100644 --- a/docs/RECOVERY.md +++ b/docs/RECOVERY.md @@ -1,87 +1,62 @@ # Backup und Wiederherstellung -## Was automatisch gesichert wird +## Athena -Der Container `mike-ai-backup` erstellt alle fünf Stunden ein komprimiertes -Archiv unter `/data/docker-backups` und behält 14 Tage. Netzwerk, WireGuard, -Router und Qwen bleiben dabei erreichbar. Hermes läuft unabhängig auf Unraid. +`mike-ai-backup` erzeugt alle fünf Stunden ein Archiv unter +`/data/docker-backups` und behält 14 Tage. Gesichert werden: -Enthalten sind: +- `/etc/mike-ai` mit lokaler Konfiguration, +- Router-Zustand und erzeugte Bilder, +- Piper-Daten, +- der kanonische Stack als zusätzlicher Snapshot. -- `/etc/mike-ai` mit lokalen Konfigurationen und Secrets -- Router-Zustand und Router-Bildablage -- Piper-Daten -- ein Quellbaum-Snapshot als zusätzliche Bequemlichkeit +Nicht in das Archiv gehören die großen Modellgewichte unter `/data/models`. +Sie bleiben auf der Daten-SSD oder werden anhand der gepinnten Angaben in +`config/install.env.example` erneut geladen. Die Dashboard-Historie liegt +dauerhaft unter `/data/llama-dashboard`. -Nicht kopiert werden `/data/models`, die nur noch als Rückfall vorhandene Kopie -`/data/hermes` und `/data/hermes-webui`: Sie liegen dauerhaft auf der Daten-SSD -und überleben den Austausch der Debian-Systemplatte. Docker-Images werden aus -dem Compose-Stack reproduziert und gehören nicht ins Backup. +### Neuaufbau -Die portablen Fach-MCPs und Hermes gehören nicht mehr zum Athena-Systembackup. -MCPHub liegt einschließlich externer Registry und portabler Erweiterungen unter -`/mnt/nvme-storage/appdata/MCPHub`, Hermes vollständig unter -`/mnt/nvme-storage/appdata/Hermes-Agent`. Beide Verzeichnisse werden vom -bestehenden Unraid-Appdata-Backup gesichert. Für ein vollständiges -Desaster-Recovery müssen daher sowohl Athenas `/data` als auch dieses -Unraid-Appdata-Backup verfügbar sein. - -## Hermes auf Unraid wiederherstellen - -1. `/mnt/nvme-storage/appdata/Hermes-Agent` aus dem Unraid-Appdata-Backup - wiederherstellen. -2. `config/unraid-templates/my-Hermes-Agent-Official.xml` nach - `/boot/config/plugins/dockerMan/templates-user/` kopieren. -3. In Unraid **Docker → Add Container → User Templates → Hermes-Agent** wählen, - die maskierten Schlüssel aus der wiederhergestellten `.env` übernehmen und - den Container starten. -4. `http://127.0.0.1:8642/health` im Container beziehungsweise - `http://:9119` im Browser prüfen. -5. Falls sich Athenas VPN-Adresse geändert hat, in `config.yaml` und in allen - `profiles/*/config.yaml` die Router-Basis-URL anpassen. - -## Manuelles Backup - -```bash -docker exec mike-ai-backup backup -``` - -Die Datei `/data/docker-backups/athena-latest.tar.gz` zeigt danach auf das -neueste erfolgreiche Archiv. - -## Neuaufbau - -1. Debian installieren und `/data` wieder unter demselben Pfad einhängen. -2. Repository klonen. -3. Installation einmal ausführen: +1. Debian installieren und `/data` wieder am bisherigen Pfad einhängen. +2. Dieses Repository klonen. +3. Installationsdatei ausfüllen und Installation starten: ```bash - sudo ./install.sh --config config/install.env + sudo ./install.sh --config /root/mike-ai-install.env ``` -4. Zustand mit einem Befehl wiederherstellen: +4. Letztes Datenarchiv einspielen: ```bash sudo ./restore.sh /data/docker-backups/athena-latest.tar.gz + sudo ./smoke-test.sh ``` -5. Falls `/etc/mike-ai/mcphub-client.env` nicht im Athena-Backup enthalten - war, den Wert aus Unraids - `/mnt/nvme-storage/appdata/MCPHub/client-token` einmalig als - `MCPHUB_BEARER_TOKEN=...` eintragen und anschließend - `platform/mcp/sync-clients.py` ausführen. Der Token gehört nicht ins Git. +Das Restore verändert weder SSH noch LAN, WireGuard, Kernel, Partitionen oder +Mounts. -Das Restore stoppt ausschließlich Container, deren Volumes zurückgeschrieben -werden. SSH, LAN und WireGuard werden nicht verändert. +## Unraid + +Hermes und die Fach-MCPs sind kein Bestandteil des Athena-Backups. Sie werden +durch das vorhandene Unraid-Appdata-Backup gesichert: + +- `/mnt/nvme-storage/appdata/Hermes-Agent` +- die jeweiligen Appdata-Verzeichnisse der MCP-Container +- DockerMan-Templates unter + `/boot/config/plugins/dockerMan/templates-user/` + +Container-Images stammen aus den dokumentierten Registries beziehungsweise den +eigenen Gitea-Repositories. Damit besteht die Wiederherstellung aus +Appdata-Restore plus Neuerstellung über die jeweilige Template-XML. ## Kontrolle ```bash docker compose --env-file /etc/mike-ai/stack.env ps test -s /data/docker-backups/athena-latest.tar.gz +curl -fsS http://192.168.1.212:8099/health +sudo ./smoke-test.sh ``` -Danach einen Router-Request, einen Hermes-Zweiturn-Chat und je einen read-only -MCP-Aufruf über `http://UNRAID-IP:8787/mcp/NAME` testen. Alte Recovery-Koffer -sind für Neuinstallationen nicht mehr erforderlich; Git, Athenas Datenbackup -und das Unraid-Appdata-Backup bilden die Wiederherstellung. +Anschließend einen Hermes-Chat, einen Router-Aufruf und je eine kleine +read-only-Abfrage der benötigten MCPs testen. diff --git a/docs/athena-architecture-map.png b/docs/athena-architecture-map.png new file mode 100644 index 0000000..78accf8 Binary files /dev/null and b/docs/athena-architecture-map.png differ diff --git a/install.sh b/install.sh index 2d23869..9860403 100755 --- a/install.sh +++ b/install.sh @@ -301,30 +301,14 @@ install_stack_files() { install -d -m 0700 "$SECRETS_DIR" [[ -s $SECRETS_DIR/router-api-key ]] || openssl rand -base64 48 >$SECRETS_DIR/router-api-key [[ -s $SECRETS_DIR/controller-token ]] || openssl rand -base64 48 >$SECRETS_DIR/controller-token - [[ -s $SECRETS_DIR/webui-secret ]] || openssl rand -base64 48 >$SECRETS_DIR/webui-secret chmod 0600 "$SECRETS_DIR"/* - local searx="$STACK_DIR/platform/web-search/searxng-settings.yml" - if [[ ! -s $searx ]]; then - cp "$STACK_DIR/platform/web-search/searxng-settings.example.yml" "$searx" - sed -i "s/CHANGE_ME_GENERATE_RANDOM_SECRET/$(openssl rand -hex 32)/" "$searx" - fi - # The official image reads this as its unprivileged uid (977). - chown root:977 "$searx" - chmod 0640 "$searx" - cat >$SECRETS_DIR/stack.env <&2; exit 2; } run python3 "$ROOT_DIR/platform/scripts/sync-profile-matrix.py" --check - run python3 "$ROOT_DIR/platform/scripts/render-mcp-registry.py" --check if [[ $1 == core ]]; then shift [[ $# -eq 0 ]] || { echo "core akzeptiert keine weiteren Services" >&2; exit 2; } run "$ROOT_DIR/platform/mcp/install-tools.sh" run "${compose[@]}" up -d --build \ - wireguard-gateway piper xtts tts-gateway profile-controller router backup + wireguard-gateway piper xtts tts-gateway profile-controller router llama-dashboard backup else run "${compose[@]}" up -d --build --no-deps "$@" fi ;; - stop-legacy) - # Reversible cleanup: stop only obsolete Athena frontends/tool backends. + purge-legacy) + # Entfernt nur ersetzte Athena-Oberflächen und ausgelagerte Fach-MCPs. for name in mike-ai-open-webui mike-ai-hermes mike-ai-hermes-webui \ mike-ai-hermes-webui-vpn-proxy mike-ai-mcp-web mike-ai-tools-searxng \ - mike-ai-tools-tinysearch mike-ai-mcp-platform-context; do + mike-ai-tools-tinysearch mike-ai-mcp-platform-context mike-ai-mcp-arr \ + mike-ai-mcp-deemix mike-ai-mcp-github mike-ai-mcp-homeassistant \ + mike-ai-mcp-navidrome; do if docker inspect "$name" >/dev/null 2>&1; then - run docker stop "$name" + run docker rm -f "$name" fi done ;; diff --git a/platform/docker/flux-worker/Dockerfile b/platform/docker/image-worker/Dockerfile similarity index 70% rename from platform/docker/flux-worker/Dockerfile rename to platform/docker/image-worker/Dockerfile index 3d44d7a..5d36bfd 100644 --- a/platform/docker/flux-worker/Dockerfile +++ b/platform/docker/image-worker/Dockerfile @@ -7,15 +7,15 @@ ARG HF_HUB_VERSION=1.28.0 RUN apt-get update && apt-get install -y --no-install-recommends python3.12-venv && \ rm -rf /var/lib/apt/lists/* && \ - python -m venv --system-site-packages /opt/flux-venv && \ - /opt/flux-venv/bin/pip install --no-cache-dir \ + python -m venv --system-site-packages /opt/image-venv && \ + /opt/image-venv/bin/pip install --no-cache-dir \ "diffusers==${DIFFUSERS_VERSION}" \ "transformers==${TRANSFORMERS_VERSION}" \ "accelerate==${ACCELERATE_VERSION}" \ "huggingface-hub==${HF_HUB_VERSION}" \ sentencepiece protobuf safetensors pillow && \ - useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin flux + useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker -COPY flux_worker.py /app/flux_worker.py +COPY image_worker.py /app/image_worker.py USER 10002:10002 -ENTRYPOINT ["/opt/flux-venv/bin/python", "/app/flux_worker.py"] +ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker.py"] diff --git a/platform/docker/flux-worker/flux_worker.py b/platform/docker/image-worker/image_worker.py similarity index 79% rename from platform/docker/flux-worker/flux_worker.py rename to platform/docker/image-worker/image_worker.py index f19bbce..60ede3a 100644 --- a/platform/docker/flux-worker/flux_worker.py +++ b/platform/docker/image-worker/image_worker.py @@ -1,5 +1,5 @@ #!/usr/bin/env python3 -"""Private FLUX.2 Klein Distilled worker used only during a GPU hot swap.""" +"""Private Z-Image-Turbo worker used only during a GPU hot swap.""" from __future__ import annotations @@ -14,7 +14,7 @@ from pathlib import Path HOST = os.environ.get("WORKER_HOST", "0.0.0.0") PORT = int(os.environ.get("WORKER_PORT", "8086")) TOKEN = os.environ.get("WORKER_TOKEN", "").strip() -MODEL_DIR = os.environ.get("FLUX_MODEL_DIR", "/models/FLUX.2-klein-4B") +MODEL_DIR = os.environ.get("Z_IMAGE_MODEL_DIR", "/models/Z-Image-Turbo") OUTPUT_DIR = Path(os.environ.get("IMAGE_DIR", "/data/images")).resolve() PIPE = None LOAD_SECONDS = 0.0 @@ -33,14 +33,15 @@ def load_pipeline() -> None: if PIPE is not None: return import torch - from diffusers import DiffusionPipeline + from diffusers import ZImagePipeline started = time.monotonic() - PIPE = DiffusionPipeline.from_pretrained( - MODEL_DIR, torch_dtype=torch.bfloat16) - # The full pipeline leaves too little activation headroom on a 16 GiB - # RTX 5080. Model CPU offload keeps each active component on CUDA while - # parking inactive components in system RAM between the four steps. - PIPE.enable_model_cpu_offload() + PIPE = ZImagePipeline.from_pretrained( + MODEL_DIR, torch_dtype=torch.bfloat16, low_cpu_mem_usage=False) + # The Qwen text encoder and the DiT do not fit together in the usable + # 16 GiB of the RTX 5080. Sequential offload keeps only the active + # submodule on CUDA. This is slower than a fully resident pipeline, but + # deterministic and leaves the RTX 3060 available for XTTS. + PIPE.enable_sequential_cpu_offload() if hasattr(PIPE, "enable_vae_slicing"): PIPE.enable_vae_slicing() if hasattr(PIPE, "enable_vae_tiling"): @@ -61,16 +62,16 @@ def generate(data: dict) -> dict: if (width, height) not in {(1024, 1024), (1536, 1024), (1024, 1536), (1920, 1088), (1088, 1920)}: raise ValueError("unsupported image size") - steps = int(data.get("steps", 4)) - guidance = float(data.get("guidance", 1.0)) - if steps != 4 or guidance != 1.0: - raise ValueError("distilled FLUX.2 Klein requires steps=4 and guidance=1.0") + steps = int(data.get("steps", 9)) + guidance = float(data.get("guidance", 0.0)) + if steps != 9 or guidance != 0.0: + raise ValueError("Z-Image-Turbo requires steps=9 and guidance=0.0") seed = data.get("seed") generator = None if seed is None else torch.Generator(device="cuda").manual_seed(int(seed)) load_pipeline() started = time.monotonic() image = PIPE(prompt=prompt, height=height, width=width, - num_inference_steps=4, guidance_scale=1.0, + num_inference_steps=9, guidance_scale=0.0, generator=generator).images[0] OUTPUT_DIR.mkdir(parents=True, exist_ok=True) output = OUTPUT_DIR / filename @@ -83,7 +84,7 @@ def generate(data: dict) -> dict: class Handler(BaseHTTPRequestHandler): def log_message(self, fmt: str, *args: object) -> None: # Never log request bodies/prompts. - print(f"[flux-worker] {self.client_address[0]} {fmt % args}", flush=True) + print(f"[z-image-worker] {self.client_address[0]} {fmt % args}", flush=True) def reply(self, status: int, payload: dict) -> None: body = json.dumps(payload, separators=(",", ":")).encode() @@ -112,7 +113,7 @@ class Handler(BaseHTTPRequestHandler): raise ValueError("invalid request size") self.reply(200, generate(json.loads(self.rfile.read(length)))) except Exception as exc: - print(f"[flux-worker] generation failed: " + print(f"[z-image-worker] generation failed: " f"{type(exc).__name__}: {str(exc)[:1000]}", flush=True) self.reply(400, {"status": "error", "message": str(exc)}) diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index ceb72ce..2a4efab 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -24,7 +24,7 @@ ALLOWED = tuple(x.strip() for x in os.environ.get( "ALLOWED_PROFILES", "fast,medium,large,ultra,uncensored,experimental").split(",") if x.strip()) LABEL_KEY = "com.mike-ai.llama-profile" IMAGE_LABEL_KEY = "com.mike-ai.image-worker" -IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "flux") +IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image") LOCK = threading.Lock() log = logging.getLogger("profile-controller") @@ -98,7 +98,7 @@ def set_image_worker(running: bool) -> dict: with LOCK: item = image_container() if running: - # A FLUX worker may never overlap a llama profile on the 5080. + # The image worker may never overlap a llama profile on the 5080. for profile_item in containers().values(): stop_container(profile_item) if item.get("State") != "running": diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh index 73b4a06..c6f0e72 100644 --- a/platform/docker/wireguard-gateway/entrypoint.sh +++ b/platform/docker/wireguard-gateway/entrypoint.sh @@ -64,7 +64,7 @@ wg_ipv4=$(ip -4 -o address show dev wg0 | awk 'NR == 1 { split($4, address, "/") # The VPN is Athena's normal application network. Nothing below is published # on the physical university interface: every listener is bound inside this # namespace to the Fritzbox-assigned WireGuard address. Clients on the home -# VPN may use OpenWebUI, the router and every useful MCP directly. +# VPN clients may use the router, speech services and Athena operator directly. proxy_pids="" start_proxy() { listen_port=$1 @@ -74,29 +74,10 @@ start_proxy() { } start_proxy 22 172.30.10.1:22 -start_proxy 8080 open-webui:8080 start_proxy 8081 router:8081 start_proxy 8085 tts-gateway:8085 start_proxy 8091 piper:8085 start_proxy 8092 xtts:80 -start_proxy 9119 hermes:9119 -start_proxy 8642 hermes:8642 - -# MCP endpoints. Optional services keep their listener even while stopped and -# begin working automatically as soon as their container is started. start_proxy 8202 mcp-athena-operator:8000 -# Portable general web MCP for Pi, Hermes and other clients. OpenWebUI uses -# its native broad search by default; both paths are site-agnostic. -start_proxy 8203 tinysearch:8000 -start_proxy 8204 mcp-github:8000 -start_proxy 8205 mcp-homeassistant:8000 -start_proxy 8206 mcp-arr:8000 -start_proxy 8207 mcp-navidrome:3000 -start_proxy 8208 mcp-unraid-ssh:8000 - -# Search backends are also directly available for diagnostics and alternative -# clients. Normal chat clients should prefer the MCP endpoint on 8203. -start_proxy 8210 searxng:8080 -start_proxy 8211 tinysearch:8000 wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}') diff --git a/platform/llama-dashboard/Dockerfile b/platform/llama-dashboard/Dockerfile new file mode 100644 index 0000000..0377c1a --- /dev/null +++ b/platform/llama-dashboard/Dockerfile @@ -0,0 +1,14 @@ +FROM python:3.13-slim + +ENV PYTHONDONTWRITEBYTECODE=1 \ + PYTHONUNBUFFERED=1 + +WORKDIR /app +COPY app.py /app/app.py + +RUN useradd --uid 10020 --create-home --shell /usr/sbin/nologin dashboard + +USER 10020:10020 +EXPOSE 8099 + +CMD ["python", "/app/app.py"] diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py new file mode 100644 index 0000000..86e9728 --- /dev/null +++ b/platform/llama-dashboard/app.py @@ -0,0 +1,775 @@ +#!/usr/bin/env python3 +from __future__ import annotations + +import json +import os +import shutil +import sqlite3 +import subprocess +import threading +import time +import urllib.error +import urllib.parse +import urllib.request +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer +from pathlib import Path +from typing import Any + + +HOST = os.getenv("DASHBOARD_HOST", "0.0.0.0") +PORT = int(os.getenv("DASHBOARD_PORT", "8099")) +ROUTER_URL = os.getenv("ROUTER_URL", "http://router:8081").rstrip("/") +ROUTER_API_KEY = os.getenv("ROUTER_API_KEY", "") +HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) +HOST_DATA = os.getenv("HOST_DATA", "/host/data") +HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) +STARTED = time.time() +HISTORY_DB = Path(os.getenv("DASHBOARD_HISTORY_DB", "/var/lib/llama-dashboard/history.sqlite3")) +HISTORY_INTERVAL = max(5, int(os.getenv("DASHBOARD_HISTORY_INTERVAL", "15"))) +DETAIL_RETENTION_DAYS = max(1, int(os.getenv("DASHBOARD_DETAIL_RETENTION_DAYS", "21"))) + + +def _number(value: str) -> int | float | None: + value = value.strip() + if not value or value.lower() in {"n/a", "[n/a]", "not supported"}: + return None + try: + number = float(value) + return int(number) if number.is_integer() else number + except ValueError: + return None + + +def _read_text(path: Path) -> str: + try: + return path.read_text(encoding="utf-8", errors="replace") + except OSError: + return "" + + +class CpuSampler: + def __init__(self) -> None: + self._lock = threading.Lock() + self._previous: tuple[int, int] | None = None + self._previous_net: tuple[float, int, int] | None = None + + def sample(self) -> dict[str, Any]: + stat = _read_text(HOST_PROC / "stat").splitlines() + cpu_line = next((line for line in stat if line.startswith("cpu ")), "") + values = [int(item) for item in cpu_line.split()[1:] if item.isdigit()] + total = sum(values) + idle = sum(values[3:5]) if len(values) >= 5 else 0 + with self._lock: + usage = None + if self._previous and total > self._previous[0]: + delta_total = total - self._previous[0] + delta_idle = idle - self._previous[1] + usage = round(100 * (1 - delta_idle / delta_total), 1) + self._previous = (total, idle) + + mem: dict[str, int] = {} + for line in _read_text(HOST_PROC / "meminfo").splitlines(): + if ":" not in line: + continue + key, raw = line.split(":", 1) + try: + mem[key] = int(raw.strip().split()[0]) * 1024 + except (ValueError, IndexError): + continue + total_mem = mem.get("MemTotal", 0) + available = mem.get("MemAvailable", 0) + used_mem = max(0, total_mem - available) + load = _read_text(HOST_PROC / "loadavg").split() + uptime_raw = _read_text(HOST_PROC / "uptime").split() + uptime = float(uptime_raw[0]) if uptime_raw else None + cpu_count = sum(1 for line in stat if line.startswith("cpu") and len(line) > 3 and line[3].isdigit()) + + disk: dict[str, Any] = {} + try: + usage_disk = shutil.disk_usage(HOST_DATA) + disk = {"total": usage_disk.total, "used": usage_disk.used, "free": usage_disk.free} + except OSError: + pass + rx_bytes = 0 + tx_bytes = 0 + interfaces = 0 + for line in _read_text(HOST_PROC / "net/dev").splitlines()[2:]: + if ":" not in line: + continue + name, values_raw = line.split(":", 1) + if name.strip() == "lo": + continue + values_net = values_raw.split() + if len(values_net) < 9: + continue + try: + rx_bytes += int(values_net[0]) + tx_bytes += int(values_net[8]) + interfaces += 1 + except ValueError: + continue + now = time.monotonic() + rx_rate = None + tx_rate = None + with self._lock: + if self._previous_net and now > self._previous_net[0]: + elapsed = now - self._previous_net[0] + rx_rate = max(0, rx_bytes - self._previous_net[1]) / elapsed + tx_rate = max(0, tx_bytes - self._previous_net[2]) / elapsed + self._previous_net = (now, rx_bytes, tx_bytes) + return { + "usage_percent": usage, + "logical_cpus": cpu_count, + "load": [float(item) for item in load[:3]] if len(load) >= 3 else [], + "memory": {"total": total_mem, "used": used_mem, "available": available}, + "disk_data": disk, + "network": { + "interfaces": interfaces, + "rx_bytes": rx_bytes, + "tx_bytes": tx_bytes, + "rx_bytes_per_second": round(rx_rate, 1) if rx_rate is not None else None, + "tx_bytes_per_second": round(tx_rate, 1) if tx_rate is not None else None, + }, + "host_uptime_seconds": uptime, + } + + +CPU = CpuSampler() + + +GPU_FIELDS = [ + "index", "name", "uuid", "utilization.gpu", "utilization.memory", + "memory.total", "memory.used", "memory.free", "temperature.gpu", + "power.draw", "power.limit", "clocks.current.graphics", + "clocks.current.memory", "fan.speed", "pstate", +] + + +def gpu_status() -> tuple[list[dict[str, Any]], str | None]: + command = [ + "nvidia-smi", + f"--query-gpu={','.join(GPU_FIELDS)}", + "--format=csv,noheader,nounits", + ] + try: + result = subprocess.run(command, capture_output=True, text=True, timeout=4, check=True) + except (OSError, subprocess.SubprocessError) as exc: + return [], str(exc) + cards: list[dict[str, Any]] = [] + for line in result.stdout.splitlines(): + values = [value.strip() for value in line.split(",")] + if len(values) != len(GPU_FIELDS): + continue + raw = dict(zip(GPU_FIELDS, values)) + cards.append({ + "index": _number(raw["index"]), + "name": raw["name"], + "uuid": raw["uuid"], + "gpu_percent": _number(raw["utilization.gpu"]), + "memory_controller_percent": _number(raw["utilization.memory"]), + "memory_total_mib": _number(raw["memory.total"]), + "memory_used_mib": _number(raw["memory.used"]), + "memory_free_mib": _number(raw["memory.free"]), + "temperature_c": _number(raw["temperature.gpu"]), + "power_w": _number(raw["power.draw"]), + "power_limit_w": _number(raw["power.limit"]), + "graphics_clock_mhz": _number(raw["clocks.current.graphics"]), + "memory_clock_mhz": _number(raw["clocks.current.memory"]), + "fan_percent": _number(raw["fan.speed"]), + "pstate": raw["pstate"], + }) + return cards, None + + +def gpu_processes() -> list[dict[str, Any]]: + command = [ + "nvidia-smi", + "--query-compute-apps=gpu_uuid,pid,process_name,used_memory", + "--format=csv,noheader,nounits", + ] + try: + result = subprocess.run(command, capture_output=True, text=True, timeout=4, check=True) + except (OSError, subprocess.SubprocessError): + return [] + processes = [] + for line in result.stdout.splitlines(): + values = [value.strip() for value in line.split(",", 3)] + if len(values) == 4: + processes.append({ + "gpu_uuid": values[0], "pid": _number(values[1]), + "name": values[2], "memory_mib": _number(values[3]), + }) + return processes + + +def llama_runtime() -> dict[str, Any]: + """Read the running llama.cpp command line from the host procfs.""" + options = { + "--model": "model_path", + "--alias": "alias", + "--ctx-size": "context_size", + "--batch-size": "batch_size", + "--ubatch-size": "ubatch_size", + "--parallel": "parallel", + "--threads": "threads", + "--threads-batch": "threads_batch", + "--device": "device", + "--tensor-split": "tensor_split", + "--cache-type-k": "cache_k", + "--cache-type-v": "cache_v", + "--reasoning-budget": "reasoning_budget", + "--spec-draft-n-max": "mtp_draft_tokens", + } + flags = { + "--flash-attn": "flash_attention", + "--cache-prompt": "prompt_cache", + "--mmproj-offload": "vision_offload", + } + try: + entries = list(HOST_PROC.iterdir()) + except OSError: + return {} + for entry in entries: + if not entry.name.isdigit(): + continue + raw = _read_text(entry / "cmdline") + if not raw: + continue + args = [item for item in raw.split("\0") if item] + if "--model" not in args or not any("llama" in item.lower() or item.endswith("/server") for item in args[:2]): + continue + result: dict[str, Any] = {"pid": int(entry.name), "executable": args[0]} + for index, arg in enumerate(args): + if arg in options and index + 1 < len(args): + value: Any = args[index + 1] + if value.isdigit(): + value = int(value) + result[options[arg]] = value + if arg in flags: + value = True + if index + 1 < len(args) and args[index + 1].lower() in {"on", "off", "true", "false"}: + value = args[index + 1].lower() in {"on", "true"} + result[flags[arg]] = value + if result.get("model_path"): + result["model_file"] = Path(str(result["model_path"])).name + return result + return {} + + +def router_status() -> tuple[dict[str, Any], str | None]: + headers = {"Accept": "application/json"} + if ROUTER_API_KEY: + headers["Authorization"] = f"Bearer {ROUTER_API_KEY}" + request = urllib.request.Request(f"{ROUTER_URL}/status", headers=headers) + try: + with urllib.request.urlopen(request, timeout=4) as response: + return json.load(response), None + except (OSError, urllib.error.URLError, json.JSONDecodeError) as exc: + return {}, str(exc) + + +_MODEL_LOCK = threading.Lock() +_MODEL_AT = 0.0 +_MODEL_CACHE: tuple[list[dict[str, Any]], dict[str, Any]] = ([], {"count": 0, "total_size": 0}) + + +def model_inventory() -> tuple[list[dict[str, Any]], dict[str, Any]]: + global _MODEL_AT, _MODEL_CACHE + now = time.monotonic() + with _MODEL_LOCK: + if now - _MODEL_AT < 30: + return _MODEL_CACHE + files: list[dict[str, Any]] = [] + total = 0 + try: + candidates = sorted(HOST_MODELS.rglob("*.gguf")) + except OSError: + candidates = [] + for path in candidates[:100]: + try: + stat = path.stat() + except OSError: + continue + total += stat.st_size + files.append({ + "name": path.name, + "relative_path": str(path.relative_to(HOST_MODELS)), + "size": stat.st_size, + "modified": stat.st_mtime, + }) + result = (files, {"count": len(files), "total_size": total}) + with _MODEL_LOCK: + _MODEL_CACHE = result + _MODEL_AT = now + return result + + +class EventTracker: + def __init__(self) -> None: + self._lock = threading.Lock() + self._last: dict[str, Any] = {} + self._events: list[dict[str, Any]] = [] + + def update(self, router: dict[str, Any], router_error: str | None) -> list[dict[str, Any]]: + state = { + "profile": router.get("current_profile"), + "model": (router.get("upstream") or {}).get("model"), + "available": (router.get("qwen") or {}).get("available"), + "switching": router.get("switching"), + "image_phase": (router.get("image") or {}).get("phase"), + "image_model": (router.get("image") or {}).get("model"), + "image_loaded": (router.get("image") or {}).get("model_loaded"), + "router_error": bool(router_error), + } + labels = { + "profile": "Profil", + "model": "Modell", + "available": "Inferenz bereit", + "switching": "Profilwechsel", + "image_phase": "Bildgenerierung", + "image_model": "Bildmodell", + "image_loaded": "Bildmodell geladen", + "router_error": "Routerfehler", + } + with self._lock: + if self._last: + for key, value in state.items(): + old = self._last.get(key) + if value != old: + self._events.insert(0, { + "timestamp": time.time(), + "name": labels[key], + "from": old, + "to": value, + }) + self._last = state + self._events = self._events[:20] + return list(self._events) + + +EVENTS = EventTracker() + +_COLLECT_LOCK = threading.Lock() +_COLLECT_AT = 0.0 +_COLLECT_CACHE: dict[str, Any] = {} + + +def collect() -> dict[str, Any]: + global _COLLECT_AT, _COLLECT_CACHE + now = time.monotonic() + with _COLLECT_LOCK: + if now - _COLLECT_AT < 0.75 and _COLLECT_CACHE: + return _COLLECT_CACHE + gpus, gpu_error = gpu_status() + router, router_error = router_status() + models, model_summary = model_inventory() + result = { + "timestamp": time.time(), + "dashboard_uptime_seconds": round(time.time() - STARTED, 1), + "cpu": CPU.sample(), + "gpus": gpus, + "gpu_processes": gpu_processes(), + "llama_runtime": llama_runtime(), + "router": router, + "models": models, + "model_summary": model_summary, + "events": EVENTS.update(router, router_error), + "errors": {"gpu": gpu_error, "router": router_error}, + } + with _COLLECT_LOCK: + _COLLECT_CACHE = result + _COLLECT_AT = now + return result + + +class HistoryStore: + """Small persistent telemetry store; never stores prompts or responses.""" + + TOKEN_KEYS = ("prompt_tokens_total", "prompt_tokens_cached_total", "tokens_predicted_total") + + def __init__(self, path: Path) -> None: + path.parent.mkdir(parents=True, exist_ok=True) + self._db = sqlite3.connect(path, check_same_thread=False) + self._db.row_factory = sqlite3.Row + self._lock = threading.Lock() + with self._db: + self._db.executescript(""" + PRAGMA journal_mode=WAL; + PRAGMA synchronous=NORMAL; + CREATE TABLE IF NOT EXISTS meta (key TEXT PRIMARY KEY, value TEXT NOT NULL); + CREATE TABLE IF NOT EXISTS samples_raw ( + ts INTEGER NOT NULL, gpu_index INTEGER NOT NULL, + gpu_util REAL, memory_used_mib REAL, temperature_c REAL, power_w REAL, + profile TEXT, model TEXT + ); + CREATE INDEX IF NOT EXISTS idx_samples_raw_ts ON samples_raw(ts); + CREATE TABLE IF NOT EXISTS samples_hourly ( + hour_ts INTEGER NOT NULL, gpu_index INTEGER NOT NULL, + gpu_util_sum REAL, gpu_util_max REAL, memory_used_sum REAL, memory_used_max REAL, + temperature_sum REAL, temperature_max REAL, power_sum REAL, samples INTEGER NOT NULL, + PRIMARY KEY(hour_ts, gpu_index) + ); + CREATE TABLE IF NOT EXISTS token_totals ( + id INTEGER PRIMARY KEY CHECK(id=1), prompt_tokens INTEGER NOT NULL DEFAULT 0, + cached_tokens INTEGER NOT NULL DEFAULT 0, output_tokens INTEGER NOT NULL DEFAULT 0 + ); + INSERT OR IGNORE INTO token_totals(id) VALUES(1); + CREATE TABLE IF NOT EXISTS token_hourly ( + hour_ts INTEGER NOT NULL, profile TEXT NOT NULL, model TEXT NOT NULL, + prompt_tokens INTEGER NOT NULL DEFAULT 0, cached_tokens INTEGER NOT NULL DEFAULT 0, + output_tokens INTEGER NOT NULL DEFAULT 0, + PRIMARY KEY(hour_ts, profile, model) + ); + CREATE TABLE IF NOT EXISTS model_events ( + ts INTEGER NOT NULL, previous_profile TEXT, profile TEXT, + previous_model TEXT, model TEXT + ); + CREATE INDEX IF NOT EXISTS idx_model_events_ts ON model_events(ts); + """) + + def _meta(self, key: str) -> str | None: + row = self._db.execute("SELECT value FROM meta WHERE key=?", (key,)).fetchone() + return str(row[0]) if row else None + + def _set_meta(self, key: str, value: Any) -> None: + self._db.execute( + "INSERT INTO meta(key,value) VALUES(?,?) ON CONFLICT(key) DO UPDATE SET value=excluded.value", + (key, str(value)), + ) + + def record(self, snapshot: dict[str, Any]) -> None: + ts = int(snapshot.get("timestamp") or time.time()) + router = snapshot.get("router") or {} + image = router.get("image") or {} + image_active = image.get("phase") not in (None, "idle") + profile = str("image" if image_active else + (router.get("current_profile") or "unknown")) + model = str((image.get("model") if image_active else + (router.get("upstream") or {}).get("model")) or "unknown") + metrics = ((router.get("llama_telemetry") or {}).get("metrics") or {}) + runtime = snapshot.get("llama_runtime") or {} + runtime_id = f"{runtime.get('pid', 'none')}:{runtime.get('model_file') or model}" + hour = ts - ts % 3600 + + with self._lock, self._db: + for gpu in snapshot.get("gpus") or []: + if gpu.get("index") is None: + continue + self._db.execute( + "INSERT INTO samples_raw VALUES(?,?,?,?,?,?,?,?)", + (ts, int(gpu["index"]), gpu.get("gpu_percent"), gpu.get("memory_used_mib"), + gpu.get("temperature_c"), gpu.get("power_w"), profile, model), + ) + + previous_runtime = self._meta("counter_runtime") + deltas: list[int] = [] + for key in self.TOKEN_KEYS: + current = max(0, int(float(metrics.get(key) or 0))) + previous = int(self._meta(f"counter_{key}") or 0) + delta = current - previous if previous_runtime == runtime_id and current >= previous else current + deltas.append(max(0, delta)) + self._set_meta(f"counter_{key}", current) + self._set_meta("counter_runtime", runtime_id) + if any(deltas): + self._db.execute( + "UPDATE token_totals SET prompt_tokens=prompt_tokens+?, cached_tokens=cached_tokens+?, output_tokens=output_tokens+? WHERE id=1", + deltas, + ) + self._db.execute( + """INSERT INTO token_hourly VALUES(?,?,?,?,?,?) + ON CONFLICT(hour_ts,profile,model) DO UPDATE SET + prompt_tokens=prompt_tokens+excluded.prompt_tokens, + cached_tokens=cached_tokens+excluded.cached_tokens, + output_tokens=output_tokens+excluded.output_tokens""", + (hour, profile, model, *deltas), + ) + + previous_profile = self._meta("last_profile") + previous_model = self._meta("last_model") + if previous_profile is not None and (profile != previous_profile or model != previous_model): + self._db.execute( + "INSERT INTO model_events VALUES(?,?,?,?,?)", + (ts, previous_profile, profile, previous_model, model), + ) + self._set_meta("last_profile", profile) + self._set_meta("last_model", model) + + def compact(self) -> None: + cutoff = int(time.time()) - DETAIL_RETENTION_DAYS * 86400 + with self._lock, self._db: + self._db.execute(""" + INSERT INTO samples_hourly + SELECT ts-ts%3600, gpu_index, SUM(gpu_util), MAX(gpu_util), + SUM(memory_used_mib), MAX(memory_used_mib), SUM(temperature_c), + MAX(temperature_c), SUM(power_w), COUNT(*) + FROM samples_raw WHERE ts < ? GROUP BY ts-ts%3600, gpu_index + ON CONFLICT(hour_ts,gpu_index) DO UPDATE SET + gpu_util_sum=gpu_util_sum+excluded.gpu_util_sum, + gpu_util_max=MAX(gpu_util_max,excluded.gpu_util_max), + memory_used_sum=memory_used_sum+excluded.memory_used_sum, + memory_used_max=MAX(memory_used_max,excluded.memory_used_max), + temperature_sum=temperature_sum+excluded.temperature_sum, + temperature_max=MAX(temperature_max,excluded.temperature_max), + power_sum=power_sum+excluded.power_sum, + samples=samples+excluded.samples + """, (cutoff,)) + self._db.execute("DELETE FROM samples_raw WHERE ts < ?", (cutoff,)) + + def query(self, range_name: str) -> dict[str, Any]: + ranges = { + "1h": (3600, 60), "24h": (86400, 300), "7d": (7 * 86400, 1800), + "21d": (21 * 86400, 3600), "all": (0, 3600), + } + seconds, bucket = ranges.get(range_name, ranges["24h"]) + now = int(time.time()) + start = 0 if seconds == 0 else now - seconds + detail_cutoff = now - DETAIL_RETENTION_DAYS * 86400 + with self._lock: + points: list[dict[str, Any]] = [] + if start < detail_cutoff: + for row in self._db.execute(""" + SELECT hour_ts ts,gpu_index,gpu_util_sum/samples gpu_util,gpu_util_max, + memory_used_sum/samples memory_used_mib,memory_used_max, + temperature_sum/samples temperature_c,temperature_max, + power_sum/samples power_w + FROM samples_hourly WHERE hour_ts>=? ORDER BY hour_ts,gpu_index + """, (start,)): + points.append(dict(row)) + raw_start = max(start, detail_cutoff) + for row in self._db.execute(f""" + SELECT (ts/{bucket})*{bucket} ts,gpu_index,AVG(gpu_util) gpu_util,MAX(gpu_util) gpu_util_max, + AVG(memory_used_mib) memory_used_mib,MAX(memory_used_mib) memory_used_max, + AVG(temperature_c) temperature_c,MAX(temperature_c) temperature_max, + AVG(power_w) power_w + FROM samples_raw WHERE ts>=? GROUP BY (ts/{bucket}),gpu_index ORDER BY ts,gpu_index + """, (raw_start,)): + points.append(dict(row)) + totals = dict(self._db.execute("SELECT * FROM token_totals WHERE id=1").fetchone()) + range_tokens = dict(self._db.execute( + "SELECT COALESCE(SUM(prompt_tokens),0) prompt_tokens, COALESCE(SUM(cached_tokens),0) cached_tokens, COALESCE(SUM(output_tokens),0) output_tokens FROM token_hourly WHERE hour_ts>=?", + (start,), + ).fetchone()) + profile_usage = [dict(row) for row in self._db.execute(""" + SELECT profile,model,SUM(prompt_tokens) prompt_tokens, + SUM(cached_tokens) cached_tokens,SUM(output_tokens) output_tokens + FROM token_hourly WHERE hour_ts>=? GROUP BY profile,model + ORDER BY SUM(prompt_tokens+cached_tokens+output_tokens) DESC + """, (start,))] + events = [dict(row) for row in self._db.execute( + "SELECT * FROM model_events WHERE ts>=? ORDER BY ts DESC LIMIT 50", (start,) + )] + raw_info = dict(self._db.execute( + "SELECT COUNT(*) rows, MIN(ts) oldest, MAX(ts) newest FROM samples_raw" + ).fetchone()) + points.sort(key=lambda item: (item["ts"], item["gpu_index"])) + return { + "range": range_name if range_name in ranges else "24h", + "detail_retention_days": DETAIL_RETENTION_DAYS, + "sample_interval_seconds": HISTORY_INTERVAL, + "points": points, + "token_totals": totals, + "range_tokens": range_tokens, + "profile_usage": profile_usage, + "model_events": events, + "storage": raw_info, + } + + +HISTORY = HistoryStore(HISTORY_DB) + + +def history_collector() -> None: + next_compaction = 0.0 + while True: + started = time.monotonic() + try: + HISTORY.record(collect()) + if time.time() >= next_compaction: + HISTORY.compact() + next_compaction = time.time() + 3600 + except Exception as exc: + print(f"history collector: {exc}", flush=True) + time.sleep(max(1, HISTORY_INTERVAL - (time.monotonic() - started))) + + +HTML = r''' + +Athena · llama.cpp Dashboard +
+
Mike AI · Live Telemetry

Athena llama.cpp Dashboard

verbinde …
+
+
Aktives Profil
–
Router wird abgefragt
+
Modell
–
–
+
CPU
–
–
+
System-RAM
–
–
+
+
Inferenz · Live
+
Generierung
–
Token pro Sekunde
+
Prompt-Einlesen
–
Token pro Sekunde
+
Prompt-Cache
–
–
+
Anfragen
–
–
+
Slots und Kontextfenster
Telemetrie wird geladen …
+
MTP / Speculative Decoding
–
–Draft-Token
–akzeptiert
–Prüfschritte
–Draft-Tiefe
+
Tokenzähler seit Modellstart
+
Modell-Eigenschaften
+
Langzeitstatistik
+
Eingabe-Tokens gesamt
–
neu + Prompt-Cache
+
Ausgabe-Tokens gesamt
–
seit Beginn der Aufzeichnung
+
Historie
–
21 Tage detailliert, danach Stundenwerte
+
GPU-Verlauf
Auslastung und Temperatur beider Karten
Historie wird geladen …
+
Nutzung nach Profil und Modell
Prozentanteil im oben gewählten Zeitraum
Nutzungsverteilung wird geladen …
+
Dauerhafte Modellwechsel
ZeitProfilModell
Noch keine Wechsel aufgezeichnet
+
Router · Profile · Dienste
+
Inferenz
–
–aktive Anfragen
–Router-Uptime
–Profilwechsel
–/data belegt
+
Verfügbare Profile
+
Zusatzdienste
+
Netzwerk und Host
+
Hardware · Dateien · Laufzeit
+
GPU-Prozesse
GPUProzessPIDVRAM
–
+
Ereignisse seit Dashboard-Start
Noch keine Zustandsänderung
+
llama.cpp Laufzeitkonfiguration
–wird gelesen
+
Verfügbare GGUF-Dateien
–
DateiPfadGrößeGeändert
–
+ +
+
''' + + +FULL_JS = r''' +const deNum=n=>n==null?'–':Number(n).toLocaleString('de-DE',{maximumFractionDigits:1}); +const bytesRate=n=>n==null?'–':n>=1048576?`${(n/1048576).toFixed(1)} MiB/s`:`${(n/1024).toFixed(1)} KiB/s`; +const metric=(value,label)=>`
${value??'–'}${label}
`; +const boolLabel=v=>v===true?'ja':v===false?'nein':'–'; +function slotHtml(s){ + let used=s.context_used||0,total=s.n_ctx||0,p=total?Math.min(100,used/total*100):0; + let state=s.processing?'arbeitet':'frei'; + return `
Slot ${s.id??'–'} · ${state}${s.speculative?'MTP aktiv':'Standard'}
${deNum(used)} / ${deNum(total)} Token${p.toFixed(1)} %
${metric(deNum(s.prompt_tokens),'Prompt')}${metric(deNum(s.prompt_cached),'Cache-Token')}${metric(deNum(s.decoded_tokens),'generiert')}${metric(deNum(s.remaining_generation),'Ausgabe übrig')}${metric(s.task_id??'–','Task-ID')}${metric(s.temperature??'–','Temperatur')}${metric(boolLabel(s.stream),'Streaming')}${metric(deNum(s.max_tokens),'Ausgabelimit')}
`; +} +async function refreshFull(){ + try{ + let response=await fetch('/api/status',{cache:'no-store'}); if(!response.ok) return; + let d=await response.json(),rt=d.router||{},lt=rt.llama_telemetry||{},met=lt.metrics||{},props=lt.props||{},lr=d.llama_runtime||{},cpu=d.cpu||{},net=cpu.network||{}; + let gen=met.predicted_tokens_seconds,prompt=met.prompt_tokens_seconds; + $('generationRate').textContent=gen==null?'–':`${deNum(gen)} tok/s`; + let genAvg=met.tokens_predicted_seconds_total?met.tokens_predicted_total/met.tokens_predicted_seconds_total:null; + $('generationSub').textContent=genAvg==null?'Aktuelle llama.cpp-Messung':`Gesamtdurchschnitt ${deNum(genAvg)} tok/s`; + $('promptRate').textContent=prompt==null?'–':`${deNum(prompt)} tok/s`; + let promptAvg=met.prompt_seconds_total?met.prompt_tokens_total/met.prompt_seconds_total:null; + $('promptSub').textContent=promptAvg==null?'Aktuelle llama.cpp-Messung':`Gesamtdurchschnitt ${deNum(promptAvg)} tok/s`; + let cached=Number(met.prompt_tokens_cached_total||0),processed=Number(met.prompt_tokens_total||0),hit=(cached+processed)?cached/(cached+processed)*100:null; + $('cacheHit').textContent=hit==null?'–':`${hit.toFixed(1)} %`;$('cacheBar').style.width=`${hit||0}%`; + $('cacheSub').textContent=hit==null?'Keine Cache-Metrik':`${deNum(cached)} wiederverwendet · ${deNum(processed)} neu`; + let running=Number(met.requests_processing??(rt.qwen||{}).active_chats??0),waiting=Number(met.requests_deferred||0); + $('requestState').textContent=`${running} aktiv · ${waiting} wartet`; + $('requestSub').textContent=`${(lt.slots||[]).filter(s=>!s.processing).length} freie Slots`; + $('slotCards').innerHTML=(lt.slots||[]).map(slotHtml).join('')||'
Slot-Telemetrie momentan nicht verfügbar
'; + let drafted=Number(met.spec_decode_num_draft_tokens_total||0),accepted=Number(met.spec_decode_num_accepted_tokens_total||0),acceptance=drafted?accepted/drafted*100:null; + $('mtpAcceptance').textContent=acceptance==null?'–':`${acceptance.toFixed(1)} % akzeptiert`;$('mtpBar').style.width=`${acceptance||0}%`; + $('mtpDrafted').textContent=deNum(drafted);$('mtpAccepted').textContent=deNum(accepted);$('mtpSteps').textContent=deNum(met.spec_decode_num_drafts_total);$('mtpDepth').textContent=lr.mtp_draft_tokens??'–'; + $('counters').innerHTML=metric(deNum(met.prompt_tokens_total),'Prompt neu')+metric(deNum(met.prompt_tokens_cached_total),'Prompt aus Cache')+metric(deNum(met.tokens_predicted_total),'generierte Token')+metric(deNum(met.n_decode_total),'Decode-Aufrufe')+metric(deNum(met.n_tokens_max),'größte Sequenz')+metric(deNum(met.n_busy_slots_per_decode),'Slots je Decode'); + let modalities=Object.entries(props.modalities||{}).filter(([,v])=>v).map(([k])=>k).join(', ')||'–'; + $('modelDetails').innerHTML=metric(props.model_ftype||'–','Quantisierung')+metric(props.total_slots??lr.parallel??'–','Slots')+metric(modalities,'Modalitäten')+metric(deNum(props.default_context||lr.context_size),'Kontext')+metric(props.model_alias||lr.alias||'–','Alias')+metric(lr.reasoning_budget??'–','Reasoning-Budget'); + $('profiles').innerHTML=Object.entries(rt.profiles||{}).map(([name,ctx])=>`
${name}${Number(ctx).toLocaleString('de-DE')} Token${name===rt.current_profile?' · aktiv':''}
`).join('')||'
Keine Profile gemeldet
'; + let tts=rt.tts||{},stt=rt.stt||{},img=rt.image||{}; + $('services').innerHTML=metric(tts.ready?'bereit':'nicht bereit',`TTS · ${tts.engine||'–'}`)+metric(tts.speaker||'–','Stimme')+metric(stt.reachable?'bereit':'aus','STT')+metric(img.worker||'–','Bild-Worker')+metric(img.model||'–','Bildmodell')+metric(img.phase==='idle'?'inaktiv':imagePhaseLabel(img.phase),'Bildstatus')+metric(img.model_loaded?'geladen':'entladen','Modellzustand')+metric(img.last_seconds==null?'–':`${deNum(img.last_seconds)} s`,'letztes Bild'); + $('hostMetrics').innerHTML=metric(bytesRate(net.rx_bytes_per_second),'Netzwerk empfangen')+metric(bytesRate(net.tx_bytes_per_second),'Netzwerk gesendet')+metric(dur(cpu.host_uptime_seconds),'Host-Uptime')+metric(dur(d.dashboard_uptime_seconds),'Dashboard-Uptime')+metric((cpu.load||[]).join(' / ')||'–','Load 1/5/15')+metric(net.interfaces??'–','Interfaces'); + let summary=d.model_summary||{};$('modelSummary').textContent=`${summary.count??0} Dateien · ${gib(summary.total_size||0)} gesamt`; + $('modelFiles').innerHTML=(d.models||[]).map(f=>`${f.name}${f.relative_path}${gib(f.size)}${new Date(f.modified*1000).toLocaleString('de-DE')}`).join('')||'Keine GGUF-Dateien im eingebundenen Modellordner'; + $('events').innerHTML=(d.events||[]).map(e=>`
${e.name}: ${String(e.from??'–')} → ${String(e.to??'–')}
`).join('')||'
Noch keine Zustandsänderung
'; + }catch(_){/* Die bestehende Verbindungsanzeige meldet Fehler bereits sichtbar. */} +} +refreshFull();setInterval(refreshFull,1000); +''' + + +HISTORY_JS = r''' +let historyRange='24h',usageMode='total',lastProfileUsage=[]; +const historyColors=['#45d7ff','#ffb454','#66e3a4','#c39bff']; +const hiddenHistorySeries=new Set();let lastHistoryPoints=[]; +function drawHistory(points){ + lastHistoryPoints=points; + const canvas=$('gpuHistoryChart'),rect=canvas.getBoundingClientRect(),ratio=window.devicePixelRatio||1; + canvas.width=Math.max(1,Math.floor(rect.width*ratio));canvas.height=Math.max(1,Math.floor(rect.height*ratio)); + const x=canvas.getContext('2d');x.scale(ratio,ratio);const w=rect.width,h=rect.height,pad={l:42,r:44,t:16,b:28}; + x.clearRect(0,0,w,h);x.strokeStyle='#213044';x.fillStyle='#8fa1b5';x.font='11px system-ui';x.lineWidth=1; + for(let i=0;i<=4;i++){let y=pad.t+(h-pad.t-pad.b)*i/4;x.beginPath();x.moveTo(pad.l,y);x.lineTo(w-pad.r,y);x.stroke();x.fillText(`${100-i*25}%`,4,y+4);x.fillText(`${100-i*25}°`,w-pad.r+7,y+4)} + if(!points.length){x.fillText('Noch keine historischen Messwerte',pad.l+10,h/2);return} + const min=Math.min(...points.map(p=>p.ts)),max=Math.max(...points.map(p=>p.ts)); + const px=t=>pad.l+(t-min)/Math.max(1,max-min)*(w-pad.l-pad.r), py=v=>pad.t+(100-Math.max(0,Math.min(100,v)))/100*(h-pad.t-pad.b); + const span=max-min,ticks=5; + for(let i=0;ip.gpu_index))].sort(); + $('gpuLegend').innerHTML=ids.flatMap((id,idx)=>[['load',`GPU ${id} Auslastung`,historyColors[idx*2%historyColors.length]],['temp',`GPU ${id} Temperatur`,historyColors[(idx*2+1)%historyColors.length]]].map(([kind,label,color])=>{let key=`${id}:${kind}`;return ``})).join(''); + ids.forEach((id,idx)=>{let rows=points.filter(p=>p.gpu_index===id),load=historyColors[idx*2%historyColors.length],temp=historyColors[(idx*2+1)%historyColors.length]; + [[load,'gpu_util','load'],[temp,'temperature_c','temp']].forEach(([color,key,kind])=>{if(hiddenHistorySeries.has(`${id}:${kind}`))return;x.beginPath();x.strokeStyle=color;x.lineWidth=2;let first=true;rows.forEach(p=>{if(p[key]==null)return;let xx=px(p.ts),yy=py(Number(p[key]));first?(x.moveTo(xx,yy),first=false):x.lineTo(xx,yy)});x.stroke()})}); +} +async function refreshHistory(){try{let r=await fetch(`/api/history?range=${historyRange}`,{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),t=d.token_totals||{},input=Number(t.prompt_tokens||0)+Number(t.cached_tokens||0);$('historyInput').textContent=deNum(input);$('historyInputSub').textContent=`${deNum(t.prompt_tokens)} neu · ${deNum(t.cached_tokens)} aus Cache`;$('historyOutput').textContent=deNum(t.output_tokens||0);$('historyStorage').textContent=`${deNum((d.storage||{}).rows||0)} Messpunkte`;drawHistory(d.points||[]);renderProfileUsage(d.profile_usage||[]);$('historyNote').textContent=`Bereich ${d.range} · Messung alle ${d.sample_interval_seconds}s · Detaildaten ${d.detail_retention_days} Tage`;$('historyEvents').innerHTML=(d.model_events||[]).slice(0,15).map(e=>`${new Date(e.ts*1000).toLocaleString('de-DE')}${e.previous_profile||'–'} → ${e.profile||'–'}${e.previous_model||'–'} → ${e.model||'–'}`).join('')||'Noch keine Wechsel aufgezeichnet'}catch(e){$('historyNote').textContent=`Historie nicht verfügbar: ${e.message}`}} +function renderProfileUsage(rows){lastProfileUsage=rows;let value=r=>usageMode==='output'?Number(r.output_tokens||0):Number(r.prompt_tokens||0)+Number(r.cached_tokens||0)+Number(r.output_tokens||0),sum=rows.reduce((n,r)=>n+value(r),0);$('profileUsage').innerHTML=rows.map((r,i)=>{let v=value(r),p=sum?v/sum*100:0,input=Number(r.prompt_tokens||0)+Number(r.cached_tokens||0);return `
${r.profile||'unbekannt'} · ${r.model||'–'}${p.toFixed(1)} %
${deNum(input)} Eingabe · ${deNum(r.output_tokens||0)} Ausgabe${deNum(v)} gewertet
`}).join('')||'
In diesem Zeitraum wurden noch keine Token aufgezeichnet.
'} +$('usageModes').addEventListener('click',e=>{let b=e.target.closest('button[data-mode]');if(!b)return;usageMode=b.dataset.mode;document.querySelectorAll('#usageModes button').forEach(x=>x.classList.toggle('active',x===b));renderProfileUsage(lastProfileUsage)}); +$('historyRanges').addEventListener('click',e=>{let b=e.target.closest('button[data-range]');if(!b)return;historyRange=b.dataset.range;document.querySelectorAll('#historyRanges button').forEach(x=>x.classList.toggle('active',x===b));refreshHistory()}); +$('gpuLegend').addEventListener('click',e=>{let b=e.target.closest('button[data-series]');if(!b)return;let key=b.dataset.series;hiddenHistorySeries.has(key)?hiddenHistorySeries.delete(key):hiddenHistorySeries.add(key);drawHistory(lastHistoryPoints)}); +window.addEventListener('resize',()=>refreshHistory());refreshHistory();setInterval(refreshHistory,15000); +''' + + +class Handler(BaseHTTPRequestHandler): + server_version = "AthenaDashboard/1.0" + + def log_message(self, fmt: str, *args: Any) -> None: + return + + def _send(self, status: int, body: bytes, content_type: str) -> None: + self.send_response(status) + self.send_header("Content-Type", content_type) + self.send_header("Content-Length", str(len(body))) + self.send_header("Cache-Control", "no-store") + self.send_header("X-Content-Type-Options", "nosniff") + self.end_headers() + self.wfile.write(body) + + def do_GET(self) -> None: + path = self.path.split("?", 1)[0] + if path == "/": + self._send(200, HTML.encode(), "text/html; charset=utf-8") + elif path == "/full.js": + self._send(200, FULL_JS.encode(), "text/javascript; charset=utf-8") + elif path == "/history.js": + self._send(200, HISTORY_JS.encode(), "text/javascript; charset=utf-8") + elif path == "/health": + self._send(200, b'{"status":"ok"}', "application/json") + elif path == "/api/status": + body = json.dumps(collect(), ensure_ascii=False, separators=(",", ":")).encode() + self._send(200, body, "application/json; charset=utf-8") + elif path == "/api/history": + query = urllib.parse.parse_qs(urllib.parse.urlsplit(self.path).query) + range_name = query.get("range", ["24h"])[0] + body = json.dumps(HISTORY.query(range_name), ensure_ascii=False, separators=(",", ":")).encode() + self._send(200, body, "application/json; charset=utf-8") + else: + self._send(404, b'{"error":"not found"}', "application/json") + + +if __name__ == "__main__": + threading.Thread(target=history_collector, name="history-collector", daemon=True).start() + ThreadingHTTPServer((HOST, PORT), Handler).serve_forever() diff --git a/platform/mcp/Dockerfile.arr b/platform/mcp/Dockerfile.arr deleted file mode 100644 index 746913e..0000000 --- a/platform/mcp/Dockerfile.arr +++ /dev/null @@ -1,12 +0,0 @@ -FROM python:3.13-slim AS builder -COPY --from=ghcr.io/astral-sh/uv:0.11.7 /uv /uvx /bin/ -RUN uv pip install --system --break-system-packages "arr-mcp[mcp]==1.0.1" - -FROM python:3.13-slim -COPY --from=builder /usr/local /usr/local -RUN groupadd --system --gid 10001 mcp \ - && useradd --system --uid 10001 --gid 10001 --no-create-home mcp -USER 10001:10001 -EXPOSE 8000 -ENTRYPOINT ["arr-mcp"] -CMD ["--transport", "streamable-http", "--host", "0.0.0.0", "--port", "8000", "--auth-type", "none"] diff --git a/platform/mcp/Dockerfile.github b/platform/mcp/Dockerfile.github deleted file mode 100644 index a80cce6..0000000 --- a/platform/mcp/Dockerfile.github +++ /dev/null @@ -1,25 +0,0 @@ -FROM ghcr.io/github/github-mcp-server@sha256:1817b57d43916532dc002bdc5f344d639bd9fb54a9148d42168458f7c3280567 AS github - -FROM python:3.13-slim@sha256:ffb752e139c0a19692a43af8d8523b274222dd68eebad5d583b45c2201c6e30a - -ARG MCP_PROXY_VERSION=0.12.0 -RUN pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp==1.29.0" - -# GitHub publishes a minimal image containing only the official Go binary. -# mcp-proxy contributes transport conversion only; GitHub API behavior and -# every exposed tool remain implemented by GitHub's official MCP server. -COPY --from=github /server/github-mcp-server /usr/local/bin/github-mcp-server - -RUN useradd --system --uid 10001 --create-home --home-dir /app mcp -USER 10001:10001 -WORKDIR /app -EXPOSE 8000 -# This is the same OpenWebUI-compatible stateless transport used by Athena's -# other Python/stdio MCP adapters. The official GitHub binary remains the only -# component implementing GitHub operations. -# mcp-proxy intentionally starts stdio children with a minimal environment. -# Explicit pass-through is required so the GitHub subprocess receives the PAT -# already injected into this container by Docker. The value is never placed on -# the command line, image, logs or Open WebUI connection record. -ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--pass-environment", "--"] -CMD ["/usr/local/bin/github-mcp-server", "stdio", "--read-only", "--tools", "search_repositories,get_file_contents,search_code"] diff --git a/platform/mcp/Dockerfile.homeassistant-relay b/platform/mcp/Dockerfile.homeassistant-relay deleted file mode 100644 index 799d715..0000000 --- a/platform/mcp/Dockerfile.homeassistant-relay +++ /dev/null @@ -1,8 +0,0 @@ -FROM nginx:1.29-alpine -COPY platform/mcp/homeassistant.conf.template /etc/nginx/templates/homeassistant.conf.template -COPY platform/mcp/ha-relay-entrypoint.sh /usr/local/bin/ha-relay-entrypoint -RUN chmod 0755 /usr/local/bin/ha-relay-entrypoint \ - && mkdir -p /tmp/client_temp /tmp/proxy_temp \ - && chown -R nginx:nginx /tmp/client_temp /tmp/proxy_temp -EXPOSE 8000 -ENTRYPOINT ["/usr/local/bin/ha-relay-entrypoint"] diff --git a/platform/mcp/Dockerfile.navidrome b/platform/mcp/Dockerfile.navidrome deleted file mode 100644 index 9eebe45..0000000 --- a/platform/mcp/Dockerfile.navidrome +++ /dev/null @@ -1,8 +0,0 @@ -FROM ghcr.io/blakeem/navidrome-mcp:2.2.0@sha256:047f911a5a8f7cc8f185bb4d6e7ca6c435542edefff4694a00c2f718ab0ee7f5 - -# llama.cpp's tool-schema converter requires every JSON-Schema regex to be -# fully anchored. Upstream 2.2.0 omits the trailing `$` on exactly two radio -# URL fields. Fail the build if upstream changes instead of patching blindly. -USER root -RUN node -e 'const fs=require("node:fs"); const p="/app/dist/tools/handlers/radio-handlers.js"; let s=fs.readFileSync(p,"utf8"); const a="pattern: '\''^https?://.+'\''"; const b="pattern: '\''^https?://.+$'\''"; const n=s.split(a).length-1; if(n!==2) throw new Error(`expected 2 schema patterns, found ${n}`); fs.writeFileSync(p,s.split(a).join(b));' -USER node diff --git a/platform/mcp/Dockerfile.unraid-ssh b/platform/mcp/Dockerfile.unraid-ssh deleted file mode 100644 index 5701134..0000000 --- a/platform/mcp/Dockerfile.unraid-ssh +++ /dev/null @@ -1,15 +0,0 @@ -FROM python:3.13-slim - -ARG MCP_PROXY_VERSION=0.12.0 -RUN apt-get update \ - && apt-get install -y --no-install-recommends openssh-client \ - && rm -rf /var/lib/apt/lists/* \ - && pip install --no-cache-dir "mcp-proxy==${MCP_PROXY_VERSION}" "mcp>=1.17,<2" \ - && useradd --system --uid 10001 --create-home --home-dir /app mcp - -RUN touch /app/unraid_mcp.py && chown 10001:10001 /app/unraid_mcp.py -USER 10001:10001 -WORKDIR /app -EXPOSE 8000 -ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"] -CMD ["python", "/app/unraid_mcp.py"] diff --git a/platform/mcp/Dockerfile.web b/platform/mcp/Dockerfile.web deleted file mode 100644 index 760df02..0000000 --- a/platform/mcp/Dockerfile.web +++ /dev/null @@ -1,18 +0,0 @@ -FROM python:3.13-slim - -ARG MCP_PROXY_VERSION=0.12.0 -ARG YT_DLP_VERSION=2026.7.4 -RUN pip install --no-cache-dir \ - "mcp-proxy==${MCP_PROXY_VERSION}" \ - "mcp>=1.17,<2" \ - "yt-dlp==${YT_DLP_VERSION}" - -RUN useradd --system --uid 10001 --create-home --home-dir /app mcp -COPY web-search/web_search_mcp.py /app/web_search_mcp.py -RUN chown -R 10001:10001 /app - -USER 10001:10001 -WORKDIR /app -EXPOSE 8000 -ENTRYPOINT ["mcp-proxy", "--host", "0.0.0.0", "--port", "8000", "--stateless", "--"] -CMD ["python", "/app/web_search_mcp.py"] diff --git a/platform/mcp/compose.github-maintenance.yaml b/platform/mcp/compose.github-maintenance.yaml deleted file mode 100644 index 6f32790..0000000 --- a/platform/mcp/compose.github-maintenance.yaml +++ /dev/null @@ -1,12 +0,0 @@ -services: - mcp-github: - # Temporary, operator-controlled maintenance mode. It deliberately omits - # delete, merge, repository creation, issue mutation and workflow tools. - command: - - /usr/local/bin/github-mcp-server - - stdio - - --tools - - search_repositories,get_repository_tree,get_file_contents,search_code,list_branches,create_branch,create_or_update_file,push_files,create_pull_request - environment: - GITHUB_READ_ONLY: "0" - GITHUB_TOOLS: search_repositories,get_repository_tree,get_file_contents,search_code,list_branches,create_branch,create_or_update_file,push_files,create_pull_request diff --git a/platform/mcp/compose.yaml b/platform/mcp/compose.yaml index caf158a..6b0e130 100644 --- a/platform/mcp/compose.yaml +++ b/platform/mcp/compose.yaml @@ -12,147 +12,8 @@ x-tool-common: &tool-common max-file: "3" services: - mcp-web: - <<: *tool-common - # Historical site-specific facade. Kept only for rollback while the - # default portable endpoint points directly at TinySearch's broad MCP. - profiles: [legacy-web] - build: - context: .. - dockerfile: mcp/Dockerfile.web - image: mike-ai/mcp-web:local - container_name: mike-ai-mcp-web - # The relay fetches and validates public result pages itself. It therefore - # needs both the private tool network and the explicitly separated egress - # network; keeping it on `tools` only makes search discovery work while - # every page fetch fails. - networks: [tools, tools-egress] - dns: ["${AI_DNS:-1.1.1.1}"] - environment: - TINYSEARCH_MCP_URL: http://tinysearch:8000/mcp - SEARXNG_URL: http://searxng:8080 - WEB_SEARCH_BUDGET_MAX_RELATED: "3" - YOUTUBE_TIMEOUT: "45" - depends_on: - tinysearch: - condition: service_started - healthcheck: - test: ["CMD", "python", "-c", "import socket; s=socket.create_connection(('127.0.0.1', 8000), 2); s.close()"] - interval: 30s - timeout: 5s - retries: 5 - start_period: 15s - - searxng: - <<: *tool-common - image: searxng/searxng@sha256:e45d5894bfaa0bf8773b9f283795ae57f1c15ddb29c8cecb70b3665b0ce9ec60 - container_name: mike-ai-tools-searxng - dns: ["${AI_DNS:-1.1.1.1}"] - volumes: - - ${SEARXNG_SETTINGS_FILE:-../web-search/searxng-settings.example.yml}:/etc/searxng/settings.yml:ro - networks: [tools, tools-egress] - - tinysearch: - <<: *tool-common - # TinySearch v0.6.1. This release fixes the crawler behavior observed with - # v0.5.1 and adds the current search -> scrape_urls workflow. - image: marcellm01/tinysearch@sha256:7a7d0585f5000f462e699e42b97409715826a9e2edcd09a166afa93a4b7cba31 - container_name: mike-ai-tools-tinysearch - dns: ["${AI_DNS:-1.1.1.1}"] - # Crawl4AI keeps transient browser/session state here. The container stays - # read-only; only this disposable runtime directory (and /tmp from the - # common hardening block) is writable. - tmpfs: - - /tmp:rw,noexec,nosuid,nodev,size=64m - - /home/tinysearch/.crawl4ai:rw,nosuid,nodev,size=256m,mode=1777 - shm_size: 1gb - volumes: - - tinysearch-models:/data/models - - ../web-search/tinysearch_config.json:/config/tinysearch_config.json:ro - environment: - MCP_TRANSPORT: streamable-http - MCP_HOST: 0.0.0.0 - MCP_PORT: "8000" - TINYSEARCH_CONFIG_PATH: /config/tinysearch_config.json - TINYSEARCH_SEARCH_BACKEND: searxng - SEARXNG_URL: http://searxng:8080/search - depends_on: [searxng] - cap_add: [SETUID, SETGID, CHOWN] - networks: [tools, tools-egress] - # The image's built-in `tinysearch doctor` also requires a writable - # configuration directory, although normal server operation does not. - # Check the service socket instead so read-only hardening remains intact. - healthcheck: - test: ["CMD", "python", "-c", "import socket; s=socket.create_connection(('127.0.0.1', 8000), 2); s.close()"] - interval: 30s - timeout: 5s - retries: 5 - start_period: 20s - - mcp-homeassistant: - <<: *tool-common - build: - context: ../.. - dockerfile: platform/mcp/Dockerfile.homeassistant-relay - image: mike-ai/mcp-homeassistant-relay:local - container_name: mike-ai-mcp-homeassistant - profiles: [homeassistant] - # Keep the public TLS hostname for SNI/certificate validation, but route it - # to the private reverse proxy through WireGuard. Public DNS may otherwise - # resolve to the Fritzbox WAN address, which is unreachable/hairpinned from - # Athena's remote-site containers. - extra_hosts: - - "ha.casaderoll.de:${HOME_LAN_PROXY_IP:-192.168.1.2}" - volumes: - - ${HA_ENV_FILE:-/etc/mike-ai/homeassistant-admin-mcp.env}:/run/secrets/homeassistant.env:ro - cap_add: [CHOWN, SETUID, SETGID] - networks: [tools, tools-egress] - - mcp-arr: - <<: *tool-common - build: - context: . - dockerfile: Dockerfile.arr - image: mike-ai/mcp-arr:1.0.1-patched - container_name: mike-ai-mcp-arr - profiles: [arr] - env_file: - - ${ARR_ENV_FILE:-/etc/mike-ai/arr-mcp.env} - volumes: - # The local fork adds bounded read-only Sonarr pseudo-actions. Keep the - # patch explicit until upstream publishes a self-contained 2.x image. - - ${ARR_SONARR_PATCH:-./patches/mcp_sonarr.py}:/usr/local/lib/python3.13/site-packages/arr_mcp/mcp/mcp_sonarr.py:ro - # Upstream's generic "Execute any Radarr API action" text gives small - # models no routing boundary. This overlay changes guidance only. - - ${ARR_RADARR_PATCH:-./patches/mcp_radarr.py}:/usr/local/lib/python3.13/site-packages/arr_mcp/mcp/mcp_radarr.py:ro - networks: [tools, tools-egress] - - mcp-navidrome: - <<: *tool-common - # Version and amd64 manifest are pinned. The image contains no mpv, so it - # cannot play audio on the headless AI host and does not expose playback - # controls. It talks to Navidrome only through its authenticated API. - build: - context: . - dockerfile: Dockerfile.navidrome - image: mike-ai/mcp-navidrome:2.2.0-schemafix1 - container_name: mike-ai-mcp-navidrome - profiles: [navidrome] - env_file: - - ${NAVIDROME_MCP_ENV_FILE:-/etc/mike-ai/navidrome-mcp.env} - environment: - MCP_TRANSPORT: http - MCP_HTTP_EXPOSE: "true" - MCP_HTTP_PORT: "3000" - # OpenWebUI uses the Docker name; Pi/Hermes may reach the same endpoint - # directly through Athena's WireGuard address and VPN port 8207. - MCP_HTTP_ALLOWED_HOSTS: "mike-ai-mcp-navidrome:3000,mike-ai-mcp-navidrome,${VPN_SERVICE_IP:-192.168.1.212}:8207,${VPN_SERVICE_IP:-192.168.1.212}" - WEBUI_ENABLED: "false" - tmpfs: - - /tmp:rw,noexec,nosuid,nodev,size=64m - - /config:rw,noexec,nosuid,nodev,size=4m,mode=0700 - networks: [tools, tools-egress] - + # Einziger MCP auf Athena: die schmale Fassade zum root-eigenen Operator. + # Alle portablen Fach-MCPs laufen als eigene Container auf Unraid. mcp-athena-operator: <<: *tool-common build: @@ -163,61 +24,10 @@ services: environment: ATHENA_OPERATOR_SOCKET: /operator/operator.sock volumes: - # The unprivileged MCP facade sees only the root-owned executor socket. - # Docker, source, models, Git credentials and host paths remain on the - # executor side and are reachable only through structured operations. - /run/mike-ai-operator:/operator:ro - networks: [tools] healthcheck: - test: ["CMD", "python", "-c", "import socket; s=socket.create_connection(('127.0.0.1',8000),2); s.close()"] + test: [CMD, python, -c, "import socket; s=socket.create_connection(('127.0.0.1',8000),2); s.close()"] interval: 30s timeout: 5s retries: 5 start_period: 10s - - mcp-github: - <<: *tool-common - build: - context: . - dockerfile: Dockerfile.github - image: mike-ai/mcp-github:github-v1.10.1-mcp-proxy-v0.12.0 - container_name: mike-ai-mcp-github - profiles: [github] - env_file: - - ${GITHUB_MCP_ENV_FILE:-/etc/mike-ai/github-mcp.env} - environment: - # These server-side limits remain authoritative even if a client asks - # for broader toolsets. The token itself must also remain read-only. - GITHUB_TOOLS: search_repositories,get_file_contents,search_code - GITHUB_READ_ONLY: "1" - networks: [tools, tools-egress] - healthcheck: - test: ["CMD", "python", "-c", "import socket; s=socket.create_connection(('127.0.0.1',8000),2); s.close()"] - interval: 30s - timeout: 5s - retries: 5 - start_period: 15s - - mcp-unraid-ssh: - <<: *tool-common - profiles: [extended] - build: - context: . - dockerfile: Dockerfile.unraid-ssh - image: mike-ai/mcp-unraid-ssh:local - container_name: mike-ai-mcp-unraid-ssh - environment: - UNRAID_MCP_CONFIG: /run/config/unraid-mcp.json - volumes: - - ${UNRAID_MCP_SOURCE:-/opt/mike-ai/unraid-agent/unraid_mcp.py}:/app/unraid_mcp.py:ro - - ${UNRAID_MCP_CONFIG:-/etc/mike-ai/unraid-mcp.json}:/run/config/unraid-mcp.json:ro - - ${UNRAID_SSH_KEY:-/etc/mike-ai/keys/unraid_root}:/etc/mike-ai/keys/unraid_root:ro - - ${UNRAID_KNOWN_HOSTS:-/etc/mike-ai/ssh/known_hosts_unraid_ai}:/etc/mike-ai/ssh/known_hosts_unraid_ai:ro - - unraid-audit:/var/log/mike-ai - networks: [tools, tools-egress] - -volumes: - tinysearch-models: - name: mike-ai-tools_tinysearch-models - external: true - unraid-audit: diff --git a/platform/mcp/github-mcp-mode.sh b/platform/mcp/github-mcp-mode.sh deleted file mode 100755 index bb3a874..0000000 --- a/platform/mcp/github-mcp-mode.sh +++ /dev/null @@ -1,49 +0,0 @@ -#!/usr/bin/env bash -set -Eeuo pipefail - -MODE=${1:-} -CONFIRM=${2:-} -ROOT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) -PROJECT=mike-ai-tools -BASE=(docker compose -p "$PROJECT" --profile github -f "$ROOT_DIR/compose.yaml") -MAINTENANCE=(-f "$ROOT_DIR/compose.github-maintenance.yaml") - -die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; } -[[ $EUID -eq 0 ]] || die "Bitte als root ausführen." - -case "$MODE" in - read) - "${BASE[@]}" up -d --no-deps --force-recreate mcp-github - ;; - maintenance) - [[ $CONFIRM == --confirm ]] || die \ - "Wartungsmodus nur mit: $0 maintenance --confirm" - "${BASE[@]}" "${MAINTENANCE[@]}" up -d --no-deps --force-recreate mcp-github - ;; - status) - command_line=$(docker inspect -f '{{json .Config.Cmd}}' mike-ai-mcp-github 2>/dev/null || true) - if [[ $command_line == *create_branch* ]]; then - printf 'GitHub MCP: WARTUNGSMODUS (begrenzter Schreibzugriff)\n' - elif [[ $command_line == *--read-only* ]]; then - printf 'GitHub MCP: NUR LESEN\n' - else - die "GitHub-MCP-Modus ist nicht eindeutig; Konfiguration prüfen." - fi - exit 0 - ;; - *) - die "Aufruf: $0 {status|read|maintenance --confirm}" - ;; -esac - -# Open WebUI caches MCP capabilities. A short backend restart makes the new -# allowlist deterministic for all profiles without touching model services. -docker restart mike-ai-open-webui >/dev/null -for _ in $(seq 1 30); do - [[ $(docker inspect -f '{{.State.Health.Status}}' mike-ai-mcp-github 2>/dev/null || true) == healthy ]] && break - sleep 1 -done -[[ $(docker inspect -f '{{.State.Health.Status}}' mike-ai-mcp-github 2>/dev/null || true) == healthy ]] || \ - die "GitHub MCP wurde nicht gesund. Zurücksetzen mit: $0 read" - -"$0" status diff --git a/platform/mcp/ha-relay-entrypoint.sh b/platform/mcp/ha-relay-entrypoint.sh deleted file mode 100644 index 6eb5bc4..0000000 --- a/platform/mcp/ha-relay-entrypoint.sh +++ /dev/null @@ -1,23 +0,0 @@ -#!/bin/sh -set -eu - -config=/run/secrets/homeassistant.env -if [ ! -r "$config" ]; then - echo "Home Assistant secret file is missing" >&2 - exit 1 -fi -set -a -. "$config" -set +a -: "${HASS_URL:?HASS_URL is required}" -: "${HASS_TOKEN:?HASS_TOKEN is required}" - -upstream=${HASS_URL%/} -escaped_token=$(printf '%s' "$HASS_TOKEN" | sed 's/[&/]/\\&/g') -escaped_upstream=$(printf '%s' "$upstream" | sed 's/[&/]/\\&/g') -sed -e "s/__HASS_TOKEN__/$escaped_token/g" \ - -e "s/__HASS_UPSTREAM__/$escaped_upstream/g" \ - /etc/nginx/templates/homeassistant.conf.template \ - > /tmp/nginx.conf -unset HASS_TOKEN -exec nginx -c /tmp/nginx.conf -g 'daemon off;' diff --git a/platform/mcp/homeassistant.conf.template b/platform/mcp/homeassistant.conf.template deleted file mode 100644 index 8d2047a..0000000 --- a/platform/mcp/homeassistant.conf.template +++ /dev/null @@ -1,30 +0,0 @@ -worker_processes 1; -pid /tmp/nginx.pid; -error_log /dev/stderr warn; - -events { worker_connections 128; } - -http { - access_log /dev/stdout; - client_body_temp_path /tmp/client_temp; - proxy_temp_path /tmp/proxy_temp; - fastcgi_temp_path /tmp/fastcgi_temp; - uwsgi_temp_path /tmp/uwsgi_temp; - scgi_temp_path /tmp/scgi_temp; - proxy_buffering off; - proxy_read_timeout 600s; - proxy_send_timeout 600s; - - server { - listen 8000; - location /mcp { - proxy_pass __HASS_UPSTREAM__/api/hass_mcp; - proxy_http_version 1.1; - proxy_ssl_server_name on; - proxy_ssl_name $proxy_host; - proxy_set_header Authorization "Bearer __HASS_TOKEN__"; - proxy_set_header Host $proxy_host; - proxy_set_header Connection ""; - } - } -} diff --git a/platform/mcp/install-hass-mcp-yaml-guard.sh b/platform/mcp/install-hass-mcp-yaml-guard.sh deleted file mode 100755 index 4f80997..0000000 --- a/platform/mcp/install-hass-mcp-yaml-guard.sh +++ /dev/null @@ -1,27 +0,0 @@ -#!/usr/bin/env bash -set -Eeuo pipefail -umask 077 - -repo_dir=$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd) -source_file=$repo_dir/platform/mcp/patches/hass_mcp/yaml_config.py -ha_config_dir=${1:-} - -die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; } -[[ $EUID -eq 0 ]] || die "Bitte als root auf dem Home-Assistant-Host ausführen." -[[ -n $ha_config_dir ]] || die "Aufruf: $0 /pfad/zum/home-assistant-config" -[[ -s $source_file ]] || die "Patchdatei fehlt: $source_file" - -target=$ha_config_dir/custom_components/hass_mcp/tools/yaml_config.py -[[ -s $target ]] || die "Native hass_mcp-Installation nicht gefunden: $target" - -backup_dir=$ha_config_dir/.hass_mcp_patch_backups -mkdir -p "$backup_dir" -chmod 700 "$backup_dir" -stamp=$(date +%Y%m%d-%H%M%S) -backup=$backup_dir/yaml_config.py.before-guard-$stamp -cp -p "$target" "$backup" -chmod 600 "$backup" -install -m 0644 "$source_file" "$target" - -printf 'YAML Guard installiert. Sicherung: %s\n' "$backup" -printf 'Home Assistant muss jetzt kontrolliert neu gestartet werden.\n' diff --git a/platform/mcp/install-tools.sh b/platform/mcp/install-tools.sh index 95021dd..501d8e8 100755 --- a/platform/mcp/install-tools.sh +++ b/platform/mcp/install-tools.sh @@ -16,14 +16,10 @@ docker network inspect mike-ai-tools >/dev/null 2>&1 || \ docker network inspect mike-ai-tools-egress >/dev/null 2>&1 || \ docker network create --subnet 172.30.50.0/24 mike-ai-tools-egress >/dev/null -# One administrative MCP exposes ATHENA.md plus the bounded host operator. +# Einziger MCP auf Athena: ATHENA.md plus der begrenzte Host-Operator. "$MCP_DIR/../operator/install-operator.sh" -services=(mcp-athena-operator) -# Portable Fach-MCPs laufen zentral im MCPHub auf Unraid. Ihre Compose-Blöcke -# bleiben vorläufig als explizite Rollback-Profile erhalten, werden bei einer -# normalen Athena-Installation aber weder gebaut noch gestartet. -echo "ARR, GitHub, Home Assistant und Navidrome werden über MCPHub auf Unraid bereitgestellt." -"${COMPOSE[@]}" up -d --build "${services[@]}" +echo "Portable Fach-MCPs laufen als eigene Container auf Unraid." +"${COMPOSE[@]}" up -d --build mcp-athena-operator "${COMPOSE[@]}" ps diff --git a/platform/mcp/patches/hass_mcp/yaml_config.py b/platform/mcp/patches/hass_mcp/yaml_config.py deleted file mode 100644 index 9bca6db..0000000 --- a/platform/mcp/patches/hass_mcp/yaml_config.py +++ /dev/null @@ -1,707 +0,0 @@ -"""Guarded YAML access for Home Assistant configuration files. - -The language model is untrusted. Reads are bounded and redact likely inline -credentials. Every mutation is previewed first, bound to the exact current file -hash, backed up, written atomically, checked by Home Assistant and rolled back -when validation or reload fails. ``secrets.yaml`` is never addressable. -""" - -from __future__ import annotations - -import copy -import difflib -import hashlib -import json -import os -import re -import secrets -import time -from pathlib import Path -from typing import Any - -from homeassistant.core import HomeAssistant -from homeassistant.util import slugify - -from ..identity import user_context -from ..protocol import ToolError, internal_error -from ..registry import LIMIT_FIELD, OFFSET_FIELD, paginate, schema, tool - -# kind -> (filename, parsed structure, reload service domain) -_KINDS: dict[str, tuple[str, str, str | None]] = { - "automation": ("automations.yaml", "list", "automation"), - "script": ("scripts.yaml", "dict", "script"), - "scene": ("scenes.yaml", "list", "scene"), - # configuration.yaml is intentionally raw-read/replace only. Treating its - # top-level keys as CRUD records would be dangerously misleading. - "configuration": ("configuration.yaml", "dict", None), -} -_STRUCTURED_KINDS = frozenset({"automation", "script", "scene"}) -_OPS = ( - "list", - "get", - "read_source", - "find_source", - "find_commented_blocks", - "list_backups", - "create", - "update", - "delete", - "replace_source_text", - "restore_backup", - "reload", -) -_MUTATING_OPS = frozenset({"create", "update", "delete", "replace_source_text", "restore_backup"}) -_TICKET_TTL_SECONDS = 600 -_MAX_SOURCE_BYTES = 2 * 1024 * 1024 -_MAX_REPLACEMENT_CHARS = 50_000 -_PENDING: dict[str, dict[str, Any]] = {} -_SENSITIVE_LINE = re.compile( - r"(?i)^(?P\s*[^#\n]*(?:password|passwd|token|secret|api[_-]?key|authorization)[^:]*:\s*).*$" -) -_SECRET_REFERENCE = re.compile(r"!secret\s+[^\s#]+", re.IGNORECASE) - - -@tool( - name="ha_yaml_config", - description=( - "Safely inspect and edit Home Assistant YAML. Structured CRUD is limited to " - "automations.yaml, scripts.yaml and scenes.yaml. Raw source operations also " - "allow configuration.yaml so commented-out blocks can be found and reviewed. " - "secrets.yaml and arbitrary paths are impossible. Use find_commented_blocks once " - "to inventory fully commented YAML entries; use read_source/find_source only for " - "other comments or exact YAML text. Every mutation first returns a diff/preview " - "and one-time approval_ticket; only repeat the exact unchanged call with " - "confirm=true after explicit user approval. Writes create a backup, are atomic, " - "run Home Assistant config validation, roll back on failure, and reload the " - "affected domain when supported. Never claim a preview changed Home Assistant." - ), - input_schema=schema( - properties={ - "kind": {"type": "string", "enum": list(_KINDS)}, - "op": {"type": "string", "enum": list(_OPS)}, - "id": { - "type": "string", - "description": "Entry id for structured get/create/update/delete.", - }, - "config": { - "type": "object", - "additionalProperties": True, - "description": "Complete entry config for structured create/update.", - }, - "query": { - "type": "string", - "maxLength": 500, - "description": "Case-insensitive literal text for find_source.", - }, - "start_line": { - "type": "integer", - "minimum": 1, - "default": 1, - "description": "First 1-based line returned by read_source.", - }, - "max_lines": { - "type": "integer", - "minimum": 1, - "maximum": 400, - "default": 120, - "description": "Bounded source lines returned by read_source/find_source.", - }, - "old_text": { - "type": "string", - "minLength": 1, - "maxLength": _MAX_REPLACEMENT_CHARS, - "description": "Exact unique YAML source text to replace.", - }, - "new_text": { - "type": "string", - "maxLength": _MAX_REPLACEMENT_CHARS, - "description": "Replacement YAML source text; may be empty to remove a block.", - }, - "backup_id": { - "type": "string", - "pattern": r"^[a-z0-9_.-]+$", - "description": "Opaque filename returned by list_backups.", - }, - "confirm": { - "type": "boolean", - "default": False, - "description": "True only after the user approved the exact preview.", - }, - "approval_ticket": { - "type": "string", - "description": "One-time ticket from the unchanged mutation preview.", - }, - "limit": LIMIT_FIELD, - "offset": OFFSET_FIELD, - }, - required=["kind", "op"], - ), - read_only=False, - idempotent=False, - requires_admin=True, - write_ops=["create", "update", "replace_source_text", "reload"], - destructive_ops=["delete", "restore_backup"], - admin_ops=["list", "get", "read_source", "find_source", "find_commented_blocks", "list_backups"], -) -async def ha_yaml_config( - hass: HomeAssistant, - kind: str, - op: str, - id: str | None = None, - config: dict[str, Any] | None = None, - query: str | None = None, - start_line: int = 1, - max_lines: int = 120, - old_text: str | None = None, - new_text: str | None = None, - backup_id: str | None = None, - confirm: bool = False, - approval_ticket: str | None = None, - limit: int = 100, - offset: int = 0, -) -> dict[str, Any]: - if kind not in _KINDS: - raise ToolError(f"unknown kind '{kind}'") - if op not in _OPS: - raise ToolError(f"unknown op '{op}'") - - filename, structure, reload_domain = _KINDS[kind] - path = Path(hass.config.path(filename)) - - if op in {"list", "get", "create", "update", "delete", "reload"} and kind not in _STRUCTURED_KINDS: - raise ToolError( - "configuration.yaml supports only read_source, find_source, list_backups, " - "replace_source_text and restore_backup" - ) - - if op == "read_source": - return await _read_source(hass, path, start_line, max_lines) - if op == "find_source": - if not query: - raise ToolError("op=find_source requires query") - return await _find_source(hass, path, query, max_lines) - if op == "find_commented_blocks": - if kind not in {"automation", "scene"}: - raise ToolError("find_commented_blocks supports automation and scene list files") - return await _find_commented_blocks(hass, path, max_lines) - if op == "list_backups": - return await _list_backups(hass, filename, limit, offset) - if op == "replace_source_text": - if old_text is None or new_text is None: - raise ToolError("op=replace_source_text requires old_text and new_text") - _reject_sensitive_replacement(old_text, new_text) - current = await _read_text(hass, path) - if current.count(old_text) != 1: - raise ToolError( - f"old_text must occur exactly once in {filename}; found {current.count(old_text)} occurrences" - ) - proposed = current.replace(old_text, new_text, 1) - await _validate_yaml_text(hass, proposed, structure, filename) - change = _change_record(kind, op, current, {"old_text": old_text, "new_text": new_text}) - if not confirm: - return _preview(change, _source_diff(filename, current, proposed)) - _consume_ticket(change, approval_ticket) - return await _commit_text(hass, path, filename, structure, reload_domain, current, proposed) - if op == "restore_backup": - if not backup_id: - raise ToolError("op=restore_backup requires backup_id from list_backups") - current = await _read_text(hass, path) - restored = await _read_backup(hass, filename, backup_id) - await _validate_yaml_text(hass, restored, structure, filename) - change = _change_record(kind, op, current, {"backup_id": backup_id}) - if not confirm: - return _preview(change, _source_diff(filename, current, restored), extra={"backup_id": backup_id}) - _consume_ticket(change, approval_ticket) - return await _commit_text(hass, path, filename, structure, reload_domain, current, restored) - - data = await _load(hass, path, structure) - if op == "list": - return paginate(_to_list(data, structure), limit, offset) - if op == "get": - if not id: - raise ToolError("op=get requires id") - item = _find(data, structure, id) - if item is None: - raise ToolError(f"{kind} '{id}' not found in {filename}") - return item - if op == "reload": - await _reload(hass, reload_domain) - return {"reloaded": reload_domain, "changed_file": False} - - current_text = await _read_text(hass, path) - proposed_data = _copy_data(data) - result: dict[str, Any] - if op == "create": - if not config: - raise ToolError("op=create requires config") - # The generated id must be deterministic so the exact preview can be - # confirmed in a second call without silently proposing another entry. - generated_id = f"mcp_{hashlib.sha256(json.dumps(config, sort_keys=True).encode()).hexdigest()[:16]}" - new_id = id or config.get("id") or generated_id - if _find(proposed_data, structure, new_id) is not None: - raise ToolError(f"{kind} '{new_id}' already exists") - if structure == "list": - proposed_data.append({"id": new_id, **{k: v for k, v in config.items() if k != "id"}}) - else: - proposed_data[new_id] = config - result = { - "operation": "create", - "id": new_id, - "proposed_entry": _find(proposed_data, structure, new_id), - } - elif op == "update": - if not id or not config: - raise ToolError("op=update requires id and config") - before = _find(proposed_data, structure, id) - if before is None or not _replace(proposed_data, structure, id, config): - raise ToolError(f"{kind} '{id}' not found") - result = {"operation": "update", "id": id, "current_entry": before, "proposed_entry": _find(proposed_data, structure, id)} - elif op == "delete": - if not id: - raise ToolError("op=delete requires id") - before = _find(proposed_data, structure, id) - if before is None or not _remove(proposed_data, structure, id): - raise ToolError(f"{kind} '{id}' not found") - result = {"operation": "delete", "id": id, "current_entry": before} - else: - raise ToolError(f"unsupported op '{op}'") - - change = _change_record(kind, op, current_text, {"id": id, "config": config, "result": result}) - if not confirm: - return _preview(change, extra=result) - _consume_ticket(change, approval_ticket) - committed = await _commit_data(hass, path, filename, structure, reload_domain, current_text, proposed_data) - return {**result, **committed} - - -def _copy_data(data: Any) -> Any: - return copy.deepcopy(data) - - -def _fingerprint(text: str) -> str: - return hashlib.sha256(text.encode("utf-8")).hexdigest() - - -def _change_record(kind: str, op: str, current: str, arguments: dict[str, Any]) -> dict[str, Any]: - return { - "kind": kind, - "op": op, - "current_sha256": _fingerprint(current), - "arguments": arguments, - } - - -def _new_ticket(change: dict[str, Any]) -> str: - now = time.time() - for key, value in list(_PENDING.items()): - if value["expires_at"] <= now: - _PENDING.pop(key, None) - ticket = secrets.token_urlsafe(18) - _PENDING[ticket] = { - "fingerprint": _fingerprint(json.dumps(change, sort_keys=True, separators=(",", ":"))), - "expires_at": now + _TICKET_TTL_SECONDS, - } - return ticket - - -def _consume_ticket(change: dict[str, Any], ticket: str | None) -> None: - record = _PENDING.pop(ticket, None) if ticket else None - expected = _fingerprint(json.dumps(change, sort_keys=True, separators=(",", ":"))) - if not record or record["expires_at"] <= time.time() or record["fingerprint"] != expected: - raise ToolError( - "approval_ticket is missing, expired, already used, or does not match the exact " - "change/current file. Run the same operation without confirm, show the preview, " - "then repeat unchanged with confirm=true only after explicit user approval." - ) - - -def _preview(change: dict[str, Any], diff: list[str] | None = None, extra: dict[str, Any] | None = None) -> dict[str, Any]: - return { - "changed": False, - "confirmation_required": True, - "approval_ticket": _new_ticket(change), - "ticket_expires_in_seconds": _TICKET_TTL_SECONDS, - "current_sha256": change["current_sha256"], - **(extra or {}), - **({"diff": diff, "diff_truncated": len(diff) >= 120} if diff is not None else {}), - "model_instruction": ( - "This is a preview only. Show it to the user and stop. Do not claim anything was " - "changed. After explicit approval repeat the exact call with confirm=true and approval_ticket." - ), - } - - -def _redact_line(line: str) -> str: - match = _SENSITIVE_LINE.match(line) - if match: - return f"{match.group('prefix')}" - return _SECRET_REFERENCE.sub("!secret ", line) - - -def _reject_sensitive_replacement(*values: str) -> None: - for value in values: - if any(_SENSITIVE_LINE.match(line) for line in value.splitlines()) or _SECRET_REFERENCE.search(value): - raise ToolError( - "Raw replacement containing credential-like keys or !secret references is refused. " - "Edit that material locally outside the LLM context." - ) - - -async def _read_text(hass: HomeAssistant, path: Path) -> str: - def _read() -> str: - if not path.exists(): - return "" - if path.stat().st_size > _MAX_SOURCE_BYTES: - raise ToolError(f"{path.name} exceeds the {_MAX_SOURCE_BYTES} byte safety limit") - return path.read_text(encoding="utf-8") - - return await hass.async_add_executor_job(_read) - - -async def _read_source(hass: HomeAssistant, path: Path, start_line: int, max_lines: int) -> dict[str, Any]: - text = await _read_text(hass, path) - lines = text.splitlines() - start = max(1, start_line) - count = max(1, min(max_lines, 400)) - selected = lines[start - 1 : start - 1 + count] - return { - "file": path.name, - "sha256": _fingerprint(text), - "total_lines": len(lines), - "start_line": start, - "returned_lines": len(selected), - "has_more": start - 1 + len(selected) < len(lines), - "lines": [{"line": start + index, "text": _redact_line(line)} for index, line in enumerate(selected)], - "redaction_note": "Credential-like values and !secret reference names are redacted.", - } - - -async def _find_source(hass: HomeAssistant, path: Path, query: str, max_lines: int) -> dict[str, Any]: - text = await _read_text(hass, path) - lines = text.splitlines() - hits = [index for index, line in enumerate(lines) if query.casefold() in line.casefold()] - cap = max(1, min(max_lines, 400)) - selected = hits[:cap] - return { - "file": path.name, - "sha256": _fingerprint(text), - "authoritative_match_count": len(hits), - "returned_count": len(selected), - "has_more": len(hits) > len(selected), - "matches": [{"line": index + 1, "text": _redact_line(lines[index])} for index in selected], - "redaction_note": "Credential-like values and !secret reference names are redacted.", - } - - -def _extract_commented_blocks(text: str, max_lines: int) -> tuple[list[dict[str, Any]], bool]: - """Return top-level YAML list entries whose every source line is commented. - - This intentionally recognizes only the conservative ``# - id:`` form used - by Home Assistant's automations/scenes editor. Ordinary prose comments, - partially disabled entries and nested comments are not treated as entries. - """ - lines = text.splitlines() - start_pattern = re.compile(r"^\s*#\s*-\s+id\s*:\s*(.*?)\s*$", re.IGNORECASE) - alias_pattern = re.compile(r"^\s*#\s+alias\s*:\s*(.*?)\s*$", re.IGNORECASE) - blocks: list[dict[str, Any]] = [] - consumed = 0 - index = 0 - truncated = False - - def clean_scalar(value: str) -> str: - value = value.strip() - if len(value) >= 2 and value[0] == value[-1] and value[0] in {"'", '"'}: - return value[1:-1] - return value - - while index < len(lines): - match = start_pattern.match(lines[index]) - if not match: - index += 1 - continue - start = index - block_lines = [lines[index]] - index += 1 - while index < len(lines): - if start_pattern.match(lines[index]): - break - if not lines[index].strip() or not re.match(r"^\s*#", lines[index]): - break - block_lines.append(lines[index]) - index += 1 - if consumed + len(block_lines) > max_lines: - truncated = True - break - alias = None - for line in block_lines: - alias_match = alias_pattern.match(line) - if alias_match: - alias = clean_scalar(alias_match.group(1)) - break - blocks.append( - { - "start_line": start + 1, - "end_line": start + len(block_lines), - "id": clean_scalar(match.group(1)), - "alias": alias, - "source": [ - {"line": start + offset + 1, "text": _redact_line(line)} - for offset, line in enumerate(block_lines) - ], - } - ) - consumed += len(block_lines) - return blocks, truncated - - -async def _find_commented_blocks( - hass: HomeAssistant, path: Path, max_lines: int -) -> dict[str, Any]: - text = await _read_text(hass, path) - cap = max(1, min(max_lines, 400)) - blocks, truncated = _extract_commented_blocks(text, cap) - return { - "file": path.name, - "sha256": _fingerprint(text), - "authoritative_block_count": len(blocks) if not truncated else None, - "returned_block_count": len(blocks), - "returned_source_lines": sum(len(block["source"]) for block in blocks), - "has_more": truncated, - "blocks": blocks, - "recognition_rule": "Only fully commented top-level '# - id:' YAML list entries are returned.", - "redaction_note": "Credential-like values and !secret reference names are redacted.", - } - - -def _source_diff(filename: str, before: str, after: str) -> list[str]: - return list( - difflib.unified_diff( - before.splitlines(), - after.splitlines(), - fromfile=f"{filename}:before", - tofile=f"{filename}:after", - lineterm="", - n=3, - ) - )[:120] - - -def _backup_dir(hass: HomeAssistant) -> Path: - return Path(hass.config.path(".hass_mcp_backups", "yaml")) - - -async def _create_backup(hass: HomeAssistant, filename: str, content: str) -> str: - backup_id = f"{filename}.{time.strftime('%Y%m%d-%H%M%S')}.{_fingerprint(content)[:10]}.bak" - directory = _backup_dir(hass) - - def _write() -> None: - directory.mkdir(mode=0o700, parents=True, exist_ok=True) - target = directory / backup_id - target.write_text(content, encoding="utf-8") - target.chmod(0o600) - - await hass.async_add_executor_job(_write) - return backup_id - - -async def _list_backups(hass: HomeAssistant, filename: str, limit: int, offset: int) -> dict[str, Any]: - directory = _backup_dir(hass) - - def _list() -> list[dict[str, Any]]: - if not directory.exists(): - return [] - rows = [] - for path in directory.glob(f"{filename}.*.bak"): - stat = path.stat() - rows.append({"backup_id": path.name, "size": stat.st_size, "created_unix": int(stat.st_mtime)}) - return sorted(rows, key=lambda row: row["created_unix"], reverse=True) - - return paginate(await hass.async_add_executor_job(_list), limit, offset) - - -async def _read_backup(hass: HomeAssistant, filename: str, backup_id: str) -> str: - if Path(backup_id).name != backup_id or not backup_id.startswith(f"{filename}.") or not backup_id.endswith(".bak"): - raise ToolError("backup_id is not valid for this YAML kind") - path = _backup_dir(hass) / backup_id - - def _read() -> str: - if not path.is_file(): - raise ToolError("backup_id not found") - if path.stat().st_size > _MAX_SOURCE_BYTES: - raise ToolError("backup exceeds safety limit") - return path.read_text(encoding="utf-8") - - return await hass.async_add_executor_job(_read) - - -async def _validate_yaml_text(hass: HomeAssistant, content: str, structure: str, filename: str) -> Any: - from homeassistant.util.yaml import parse_yaml - - def _parse() -> Any: - parsed = parse_yaml(content) if content.strip() else ([] if structure == "list" else {}) - if structure == "list" and not isinstance(parsed, list): - raise ToolError(f"{filename} must be a YAML list, got {type(parsed).__name__}") - if structure == "dict" and not isinstance(parsed, dict): - raise ToolError(f"{filename} must be a YAML mapping, got {type(parsed).__name__}") - return parsed - - return await hass.async_add_executor_job(_parse) - - -async def _check_full_config(hass: HomeAssistant) -> dict[str, Any]: - try: - from homeassistant.components.config.core import async_check_ha_config_file - except ImportError: - from homeassistant.config import async_check_ha_config_file - - result = await async_check_ha_config_file(hass) - if result is None: - return {"valid": True} - if isinstance(result, str): - return {"valid": not bool(result), "error": result or None} - errors = getattr(result, "errors", None) - if errors: - return {"valid": False, "error": str(errors)} - return {"valid": True, "result": str(result)} - - -async def _atomic_write(hass: HomeAssistant, path: Path, content: str) -> None: - def _write() -> None: - temporary = path.with_name(f".{path.name}.hass-mcp-{secrets.token_hex(6)}.tmp") - try: - with temporary.open("w", encoding="utf-8") as handle: - handle.write(content) - handle.flush() - os.fsync(handle.fileno()) - os.replace(temporary, path) - finally: - if temporary.exists(): - temporary.unlink() - - await hass.async_add_executor_job(_write) - - -async def _commit_text( - hass: HomeAssistant, - path: Path, - filename: str, - structure: str, - reload_domain: str | None, - before: str, - proposed: str, -) -> dict[str, Any]: - current = await _read_text(hass, path) - if _fingerprint(current) != _fingerprint(before): - raise ToolError("YAML file changed after preview; refusing stale write and requiring a new preview") - await _validate_yaml_text(hass, proposed, structure, filename) - backup_id = await _create_backup(hass, filename, before) - await _atomic_write(hass, path, proposed) - validation = await _check_full_config(hass) - if not validation["valid"]: - await _atomic_write(hass, path, before) - raise ToolError(f"Home Assistant config validation failed; original restored from {backup_id}: {validation.get('error')}") - try: - if reload_domain: - await _reload(hass, reload_domain) - except Exception: - await _atomic_write(hass, path, before) - if reload_domain: - try: - await _reload(hass, reload_domain) - except Exception: - pass - raise - readback = await _read_text(hass, path) - return { - "changed": readback == proposed, - "file": filename, - "backup_id": backup_id, - "full_config_valid": True, - "reloaded": reload_domain, - "restart_required": reload_domain is None, - "new_sha256": _fingerprint(readback), - "exact_readback_match": readback == proposed, - } - - -async def _commit_data( - hass: HomeAssistant, - path: Path, - filename: str, - structure: str, - reload_domain: str | None, - before: str, - data: Any, -) -> dict[str, Any]: - from homeassistant.util.yaml import save_yaml - - def _render() -> str: - temporary = path.with_name(f".{path.name}.hass-mcp-render-{secrets.token_hex(6)}.tmp") - try: - save_yaml(str(temporary), data) - return temporary.read_text(encoding="utf-8") - finally: - if temporary.exists(): - temporary.unlink() - - proposed = await hass.async_add_executor_job(_render) - return await _commit_text(hass, path, filename, structure, reload_domain, before, proposed) - - -async def _load(hass: HomeAssistant, path: Path, structure: str) -> Any: - return await _validate_yaml_text(hass, await _read_text(hass, path), structure, path.name) - - -async def _reload(hass: HomeAssistant, domain: str | None) -> None: - if not domain: - return - try: - await hass.services.async_call(domain, "reload", {}, blocking=True, context=user_context()) - except Exception as error: - raise internal_error(f"{domain}.reload failed", error) from error - - -def _derive_entity_id(domain: str, structure: str, new_id: str, config: dict[str, Any]) -> str: - slug = slugify(new_id) if structure == "dict" else slugify(config.get("alias") or new_id) - return f"{domain}.{slug}" - - -def _to_list(data: Any, structure: str) -> list[dict[str, Any]]: - if structure == "list": - return list(data) - return [{"id": key, **value} for key, value in data.items()] - - -def _find(data: Any, structure: str, id: str) -> dict[str, Any] | None: - if structure == "list": - for entry in data: - if entry.get("id") == id or entry.get("alias") == id: - return entry - return None - return {"id": id, **data[id]} if id in data else None - - -def _replace(data: Any, structure: str, id: str, new: dict[str, Any]) -> bool: - if structure == "list": - for index, entry in enumerate(data): - if entry.get("id") == id: - data[index] = {"id": id, **{key: value for key, value in new.items() if key != "id"}} - return True - return False - if id in data: - data[id] = new - return True - return False - - -def _remove(data: Any, structure: str, id: str) -> bool: - if structure == "list": - for index, entry in enumerate(data): - if entry.get("id") == id: - del data[index] - return True - return False - if id in data: - del data[id] - return True - return False diff --git a/platform/mcp/patches/mcp_radarr.py b/platform/mcp/patches/mcp_radarr.py deleted file mode 100644 index 6daf104..0000000 --- a/platform/mcp/patches/mcp_radarr.py +++ /dev/null @@ -1,187 +0,0 @@ -"""Small, explicit, read-only Radarr tools built on the upstream API client.""" - -import asyncio -from typing import Any - -from fastmcp import FastMCP -from pydantic import Field - -from arr_mcp.auth import get_radarr_client - - -async def _call(client: Any, action: str, kwargs: dict[str, Any] | None = None) -> Any: - """Call one known upstream API method without exposing dynamic dispatch.""" - return await asyncio.to_thread(getattr(client, action), **(kwargs or {})) - - -def _plain(value: Any) -> Any: - if hasattr(value, "model_dump") and callable(value.model_dump): - return value.model_dump() - if hasattr(value, "dict") and callable(value.dict): - return value.dict() - return value - - -def _movies(value: Any) -> list[dict[str, Any]]: - value = _plain(value) - if isinstance(value, dict) and "result" in value: - value = value["result"] - return [item for item in value if isinstance(item, dict)] if isinstance(value, list) else [] - - -def _compact_movie(movie: dict[str, Any]) -> dict[str, Any]: - return { - key: movie[key] - for key in ("id", "title", "originalTitle", "year", "status", "monitored", "hasFile", "path", "tmdbId") - if movie.get(key) is not None - } - - -def _compact_release(item: dict[str, Any]) -> dict[str, Any]: - quality = item.get("quality") or {} - quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None - result = { - key: item[key] - for key in ("guid", "title", "indexer", "indexerId", "size", "age", "seeders", "leechers", "protocol", "downloadAllowed", "releaseGroup") - if item.get(key) is not None - } - if quality_name: - result["quality"] = quality_name - if isinstance(item.get("rejections"), list) and item["rejections"]: - result["rejections"] = [str(reason)[:180] for reason in item["rejections"][:5]] - return result - - -def _codec_aliases(value: str) -> set[str]: - aliases = { - "h264": {"h264", "x264", "avc"}, - "x264": {"h264", "x264", "avc"}, - "avc": {"h264", "x264", "avc"}, - "h265": {"h265", "x265", "hevc"}, - "x265": {"h265", "x265", "hevc"}, - "hevc": {"h265", "x265", "hevc"}, - } - requested: set[str] = set() - for item in value.split(","): - key = item.strip().casefold() - if key: - requested.update(aliases.get(key, {key})) - return requested - - -def _compact_inventory( - movies: list[dict[str, Any]], *, codecs: str = "", query: str = "", - offset: int = 0, limit: int = 200, -) -> dict[str, Any]: - wanted = _codec_aliases(codecs) - needle = query.strip().casefold() - rows: list[dict[str, Any]] = [] - for movie in movies: - movie_file = movie.get("movieFile") or {} - if not movie.get("hasFile") or not movie_file: - continue - media = movie_file.get("mediaInfo") or {} - codec = str(media.get("videoCodec") or "unknown") - if wanted and codec.casefold() not in wanted: - continue - title = str(movie.get("title") or "") - if needle and needle not in title.casefold(): - continue - quality = movie_file.get("quality") or {} - quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None - size = int(movie_file.get("size") or 0) - rows.append({ - "radarrId": movie.get("id"), - "title": title, - "year": movie.get("year"), - "movieFileId": movie_file.get("id"), - "relativePath": movie_file.get("relativePath"), - "sizeBytes": size, - "sizeGiB": round(size / 1073741824, 2), - "quality": quality_name, - "resolution": media.get("resolution"), - "videoCodec": codec, - "videoBitDepth": media.get("videoBitDepth"), - "audioCodec": media.get("audioCodec"), - "audioLanguages": media.get("audioLanguages"), - "subtitles": media.get("subtitles"), - }) - rows.sort(key=lambda row: (str(row["title"]).casefold(), row.get("year") or 0)) - total = len(rows) - start = max(0, int(offset)) - count = min(500, max(1, int(limit))) - selected = rows[start:start + count] - return { - "totalMatched": total, - "offset": start, - "returned": len(selected), - "hasMore": start + len(selected) < total, - "movies": selected, - } - - -def register_radarr_tools(mcp: FastMCP) -> None: - @mcp.tool(tags={"radarr"}) - async def radarr_find_movie( - query: str = Field(description="Movie title or title fragment."), - limit: int = Field(default=10, ge=1, le=25), - ) -> Any: - """Find a movie already managed by Radarr. READ ONLY. Never starts a search or download.""" - needle = query.strip().casefold() - if len(needle) < 2: - raise ValueError("query must contain at least two characters") - movies = _movies(await _call(get_radarr_client(), "get_movie")) - matches = [ - _compact_movie(movie) - for movie in movies - if needle in " ".join(str(movie.get(key, "")) for key in ("title", "originalTitle", "sortTitle")).casefold() - ] - return {"query": query, "match_count": len(matches), "matches": matches[:limit], "truncated": len(matches) > limit} - - @mcp.tool(tags={"radarr"}) - async def radarr_movie_codec_inventory( - video_codecs: str = Field( - default="", - description="Optional comma-separated filter, e.g. h264, x264, h265, x265 or hevc.", - ), - query: str = Field(default="", description="Optional case-insensitive title fragment."), - offset: int = Field(default=0, ge=0), - limit: int = Field(default=200, ge=1, le=500), - ) -> Any: - """Compact authoritative Radarr movie-file inventory. Use for codec, resolution, language and size questions instead of get_movie, raw API requests or filesystem scans. Results are valid bounded JSON without alternate titles, images, overviews or ratings.""" - client = get_radarr_client() - response = _plain(await _call(client, "get_movie")) - movies = _movies(response) - if not movies and response not in ([], {"result": []}): - raise RuntimeError("Radarr get_movie returned an unexpected response") - return _compact_inventory( - movies, codecs=video_codecs, query=query, offset=offset, limit=limit, - ) - - @mcp.tool(tags={"radarr"}) - async def radarr_search_releases( - movie_id: int = Field(ge=1, description="Exact Radarr movie id returned by radarr_find_movie."), - release_group: str = Field(default="", description="Optional release-group filter."), - limit: int = Field(default=50, ge=1, le=100), - ) -> Any: - """Search Radarr's configured indexers for one movie. READ ONLY: never grabs or downloads a release.""" - raw = _plain(await _call(get_radarr_client(), "get_release", {"movieId": movie_id})) - if isinstance(raw, dict) and "result" in raw: - raw = raw["result"] - releases = [item for item in raw if isinstance(item, dict)] if isinstance(raw, list) else [] - needle = release_group.strip().casefold() - if needle: - releases = [ - item for item in releases - if needle in (str(item.get("releaseGroup", "")) + " " + str(item.get("title", ""))).casefold() - ] - compact = [_compact_release(item) for item in releases[:limit]] - return { - "movie_id": movie_id, - "release_group_filter": release_group or None, - "total": len(releases), - "returned": len(compact), - "truncated": len(releases) > limit, - "results": compact, - "download_started": False, - } diff --git a/platform/mcp/patches/mcp_sonarr.py b/platform/mcp/patches/mcp_sonarr.py deleted file mode 100644 index 786e81e..0000000 --- a/platform/mcp/patches/mcp_sonarr.py +++ /dev/null @@ -1,676 +0,0 @@ -"""Sonarr condensed action-routed MCP tool. - -CONCEPT:ECO-4.82 — gitlab-style organized per-service tool surface. -""" - -import asyncio -import os -import json -import re -import secrets -import time -from typing import Any - -from fastmcp import FastMCP -from pydantic import Field - -from arr_mcp.auth import get_sonarr_client - - -MAX_COLLECTION_ITEMS = 50 -APPROVAL_TTL_SECONDS = 600 -_APPROVALS: dict[str, tuple[float, str]] = {} - - -async def _call(client: Any, action: str, kwargs: dict[str, Any] | None = None) -> Any: - """Call one known upstream API method without exposing dynamic dispatch.""" - method = getattr(client, action) - return await asyncio.to_thread(method, **(kwargs or {})) - - -def _plain(value: Any) -> Any: - if hasattr(value, "model_dump") and callable(value.model_dump): - return value.model_dump() - if hasattr(value, "dict") and callable(value.dict): - return value.dict() - if isinstance(value, list): - return [_plain(item) for item in value] - if isinstance(value, dict): - return {str(key): _plain(item) for key, item in value.items()} - return value - - -def _unwrap(value: Any) -> Any: - value = _plain(value) - if isinstance(value, dict) and set(value) == {"result"}: - return value["result"] - return value - - -def _pick(item: dict[str, Any], fields: tuple[str, ...]) -> dict[str, Any]: - return {field: item[field] for field in fields if item.get(field) is not None} - - -def _compact_series(item: dict[str, Any], include_seasons: bool = False) -> dict[str, Any]: - result = _pick( - item, - ("id", "title", "sortTitle", "year", "status", "monitored", "path", "tvdbId"), - ) - statistics = item.get("statistics") or {} - if isinstance(statistics, dict): - result["statistics"] = _pick( - statistics, - ("seasonCount", "episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"), - ) - if include_seasons: - result["seasons"] = [ - { - **_pick(season, ("seasonNumber", "monitored")), - "statistics": _pick( - season.get("statistics") or {}, - ("episodeFileCount", "episodeCount", "totalEpisodeCount", "sizeOnDisk", "percentOfEpisodes"), - ), - } - for season in item.get("seasons", []) - if isinstance(season, dict) - ] - return result - - -def _compact_episode(item: dict[str, Any]) -> dict[str, Any]: - return _pick( - item, - ("id", "seriesId", "seasonNumber", "episodeNumber", "title", "airDate", "airDateUtc", "monitored", "hasFile", "episodeFileId"), - ) - - -def _compact_file(item: dict[str, Any]) -> dict[str, Any]: - quality = item.get("quality") or {} - quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None - result = _pick( - item, - ("id", "seriesId", "seasonNumber", "relativePath", "path", "size", "dateAdded", "releaseGroup"), - ) - if quality_name: - result["quality"] = quality_name - return result - - -def _compact_release(item: dict[str, Any]) -> dict[str, Any]: - quality = item.get("quality") or {} - quality_name = (quality.get("quality") or {}).get("name") if isinstance(quality, dict) else None - result = _pick( - item, - ( - "guid", "title", "indexer", "indexerId", "size", "age", "ageHours", - "seeders", "leechers", "protocol", "downloadAllowed", "releaseWeight", - "releaseGroup", "seasonNumber", "fullSeason", - ), - ) - if quality_name: - result["quality"] = quality_name - rejections = item.get("rejections") - if isinstance(rejections, list) and rejections: - result["rejections"] = [str(reason)[:180] for reason in rejections[:5]] - return result - - -def _bounded(items: list[Any], compact) -> dict[str, Any]: - total = len(items) - return { - "total": total, - "returned": min(total, MAX_COLLECTION_ITEMS), - "truncated": total > MAX_COLLECTION_ITEMS, - "items": [compact(item) for item in items[:MAX_COLLECTION_ITEMS] if isinstance(item, dict)], - "next_step": ( - "Use find_series or narrower Sonarr parameters; do not repeat the same broad request." - if total > MAX_COLLECTION_ITEMS else None - ), - } - - -def _compact_result(action: str, value: Any) -> Any: - value = _unwrap(value) - if isinstance(value, list): - if action in {"get_series", "get_series_lookup", "lookup_series"}: - return _bounded(value, _compact_series) - if action in {"get_episode", "get_calendar", "get_wanted_missing", "get_wanted_cutoff"}: - return _bounded(value, _compact_episode) - if action == "get_episodefile": - return _bounded(value, _compact_file) - if action == "get_release": - return _bounded(value, _compact_release) - return _bounded(value, lambda item: item) - if isinstance(value, dict) and action in {"get_series_id"}: - return _compact_series(value, include_seasons=True) - if isinstance(value, dict) and action in {"get_episode_id"}: - return _compact_episode(value) - if isinstance(value, dict) and action in {"get_episodefile_id"}: - return _compact_file(value) - return value - - -async def _find_series(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - query = str(kwargs.get("query", "")).strip() - if len(query) < 2: - raise ValueError("query must contain at least two characters") - limit = max(1, min(int(kwargs.get("limit", 8)), 15)) - raw = _unwrap(await _call(client, "get_series")) - words = [word for word in re.findall(r"[a-z0-9]+", query.casefold()) if len(word) > 1] - matches = [] - for item in raw if isinstance(raw, list) else []: - haystack = " ".join( - str(item.get(field, "")) for field in ("title", "sortTitle", "originalTitle", "alternateTitles") - ).casefold() - if all(word in haystack for word in words): - matches.append(_compact_series(item, include_seasons=False)) - return { - "query": query, - "matches": matches[:limit], - "match_count": len(matches), - "truncated": len(matches) > limit, - "task_complete": True, - "instruction": "Use the returned series id for details. Do not call get_series for discovery.", - } - - -async def _season_summary(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - series_id = int(kwargs["series_id"]) - season_number = int(kwargs["season_number"]) - series = _unwrap(await _call(client, "get_series_id", {"id": series_id})) - episodes = _unwrap( - await _call(client, "get_episode", {"seriesId": series_id, "seasonNumber": season_number}) - ) - files = _unwrap( - await _call(client, "get_episodefile", {"seriesId": series_id}) - ) - selected_episodes = [ - _compact_episode(item) for item in episodes - if isinstance(item, dict) and item.get("seasonNumber") == season_number - ] if isinstance(episodes, list) else [] - selected_files = [ - _compact_file(item) for item in files - if isinstance(item, dict) and item.get("seasonNumber") == season_number - ] if isinstance(files, list) else [] - groups = sorted({str(item.get("releaseGroup")) for item in selected_files if item.get("releaseGroup")}) - return { - "series": _compact_series(series) if isinstance(series, dict) else {"id": series_id}, - "season_number": season_number, - "episode_count": len(selected_episodes), - "file_count": len(selected_files), - "release_groups": groups, - "episodes": selected_episodes[:30], - "files": selected_files[:30], - "task_complete": True, - "instruction": "This is the complete compact season answer. Do not repeat broad series or episode queries.", - } - - -async def _search_releases(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - series_id = kwargs.get("series_id") - episode_id = kwargs.get("episode_id") - season_number = kwargs.get("season_number") - release_group = str(kwargs.get("release_group", "")).strip() - season_pack_only = kwargs.get("season_pack_only") is True - if series_id is None and episode_id is None: - raise ValueError("search_releases requires series_id or episode_id") - query: dict[str, Any] = {} - if series_id is not None: - query["seriesId"] = int(series_id) - if episode_id is not None: - query["episodeId"] = int(episode_id) - if season_number is not None: - query["seasonNumber"] = int(season_number) - raw = await _call(client, "get_release", query) - raw = _unwrap(raw) - if release_group and isinstance(raw, list): - needle = release_group.casefold() - raw = [ - item for item in raw - if isinstance(item, dict) - and needle in ( - str(item.get("releaseGroup", "")) + " " + str(item.get("title", "")) - ).casefold() - ] - if season_pack_only: - if season_number is None: - raise ValueError("season_pack_only=true requires season_number") - season_token = rf"(?:^|[. _-])S0*{int(season_number)}(?:[. _-]|$)" - episode_token = rf"S0*{int(season_number)}E\d+" - raw = [ - item for item in raw - if isinstance(item, dict) - and ( - item.get("fullSeason") is True - or ( - re.search(season_token, str(item.get("title", "")), re.IGNORECASE) - and not re.search(episode_token, str(item.get("title", "")), re.IGNORECASE) - ) - ) - ] if isinstance(raw, list) else raw - compact = _compact_result("get_release", raw) - return { - "task_complete": True, - "search_scope": { - "series_id": series_id, - "episode_id": episode_id, - "season_number": season_number, - "release_group_filter": release_group or None, - "season_pack_only": season_pack_only, - }, - "monitoring_changed": False, - "download_started": False, - "results": compact, - "instruction": ( - "These are Sonarr indexer results. Do not use web search to replace them. " - "If the user requests one specific release, release group, or complete season pack, " - "NEVER substitute an automatic episode search: preview that exact result with " - "preview_release_grab, then wait for explicit approval before grab_release." - ), - } - - -def _episode_numbers(value: Any) -> list[int]: - if value is None: - return [] - if not isinstance(value, list) or len(value) > 100: - raise ValueError("episode_numbers must be a JSON list with at most 100 entries") - numbers = sorted({int(item) for item in value}) - if any(item < 0 or item > 9999 for item in numbers): - raise ValueError("episode_numbers contains an invalid episode number") - return numbers - - -async def _resolve_episode_search(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - series_id = int(kwargs["series_id"]) - season_number = int(kwargs["season_number"]) - requested_numbers = _episode_numbers(kwargs.get("episode_numbers")) - if series_id < 1 or season_number < 0: - raise ValueError("series_id and season_number must be non-negative identifiers") - - series = _unwrap(await _call(client, "get_series_id", {"id": series_id})) - episodes = _unwrap( - await _call(client, "get_episode", {"seriesId": series_id, "seasonNumber": season_number}) - ) - candidates = [ - item for item in episodes - if isinstance(item, dict) - and int(item.get("seasonNumber", -1)) == season_number - and (not requested_numbers or int(item.get("episodeNumber", -1)) in requested_numbers) - ] if isinstance(episodes, list) else [] - if requested_numbers: - found_numbers = {int(item.get("episodeNumber", -1)) for item in candidates} - missing_metadata = sorted(set(requested_numbers) - found_numbers) - if missing_metadata: - raise ValueError(f"Sonarr has no episode metadata for episode numbers: {missing_metadata}") - missing = [item for item in candidates if not bool(item.get("hasFile"))] - if not candidates: - raise ValueError("No Sonarr episodes match the requested scope") - if len(missing) > 100: - raise ValueError("Refusing to search more than 100 missing episodes at once") - - compact = [_compact_episode(item) for item in missing] - scope = { - "series_id": series_id, - "series_title": str(series.get("title", "")) if isinstance(series, dict) else "", - "season_number": season_number, - "requested_episode_numbers": requested_numbers, - "missing_episode_ids": [int(item["id"]) for item in missing], - "missing_episode_numbers": [int(item["episodeNumber"]) for item in missing], - } - fingerprint = json.dumps(scope, ensure_ascii=False, sort_keys=True, separators=(",", ":")) - return { - "scope": scope, - "fingerprint": fingerprint, - "selected_episode_count": len(candidates), - "already_present_count": len(candidates) - len(missing), - "unmonitored_missing_count": sum(not bool(item.get("monitored")) for item in missing), - "missing_episodes": compact, - } - - -async def _preview_episode_search(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - resolved = await _resolve_episode_search(client, kwargs) - ticket = secrets.token_urlsafe(24) - now = time.monotonic() - for old_ticket, (expires, _) in list(_APPROVALS.items()): - if expires <= now: - _APPROVALS.pop(old_ticket, None) - _APPROVALS[ticket] = (now + APPROVAL_TTL_SECONDS, resolved["fingerprint"]) - return { - "action": "preview-only", - **{key: value for key, value in resolved.items() if key != "fingerprint"}, - "monitoring_changed": False, - "download_started": False, - "approval_ticket": ticket, - "approval_expires_in_seconds": APPROVAL_TTL_SECONDS, - "next_step": ( - "Review series, season and episode list. Only after explicit approval call " - "start_episode_search with exactly the same scope, confirm=true and this ticket." - ), - } - - -async def _start_episode_search(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - if kwargs.get("confirm") is not True: - raise PermissionError("confirm=true is required after reviewing preview_episode_search") - ticket = str(kwargs.get("approval_ticket", "")) - if not ticket: - raise PermissionError("approval_ticket is required") - resolved = await _resolve_episode_search(client, kwargs) - approval = _APPROVALS.pop(ticket, None) - if approval is None or approval[0] <= time.monotonic(): - raise PermissionError("Approval ticket is missing, expired or already used") - if not secrets.compare_digest(approval[1], resolved["fingerprint"]): - raise PermissionError("Approval ticket does not match this exact episode search") - episode_ids = resolved["scope"]["missing_episode_ids"] - if not episode_ids: - return { - "ok": True, - "command_started": False, - "reason": "All selected episodes already have files", - "scope": resolved["scope"], - } - command = _unwrap( - await _call(client, "post_command", {"data": {"name": "EpisodeSearch", "episodeIds": episode_ids}}) - ) - return { - "ok": True, - "command_started": True, - "sonarr_command": _pick(command, ("id", "name", "status", "queued", "startedOn")) if isinstance(command, dict) else command, - "scope": resolved["scope"], - "monitoring_changed": False, - "download_may_start_immediately": True, - "instruction": ( - "Sonarr is now searching its configured indexers and may immediately grab/download " - "the best acceptable release for every approved episode. EpisodeSearch is NOT a " - "read-only manual-search preview and is NOT limited by the monitored flag. Check the " - "queue before claiming that nothing was downloaded." - ), - } - - -def _release_query(kwargs: dict[str, Any]) -> dict[str, Any]: - series_id = int(kwargs["series_id"]) - if series_id < 1: - raise ValueError("series_id must be a positive Sonarr identifier") - query: dict[str, Any] = {"seriesId": series_id} - if kwargs.get("season_number") is not None: - season_number = int(kwargs["season_number"]) - if season_number < 0: - raise ValueError("season_number must be non-negative") - query["seasonNumber"] = season_number - if kwargs.get("episode_id") is not None: - episode_id = int(kwargs["episode_id"]) - if episode_id < 1: - raise ValueError("episode_id must be a positive Sonarr identifier") - query["episodeId"] = episode_id - return query - - -async def _resolve_release_grab(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - guid = str(kwargs.get("guid", "")).strip() - if not guid: - raise ValueError( - "guid is required; copy it from the exact search_releases result the user selected" - ) - query = _release_query(kwargs) - raw = _unwrap( - await _call(client, "get_release", query) - ) - matches = [ - item for item in raw if isinstance(item, dict) and str(item.get("guid", "")) == guid - ] if isinstance(raw, list) else [] - if len(matches) != 1: - raise ValueError( - "The exact release GUID is no longer present in Sonarr's current indexer results; " - "run search_releases again and do not guess or substitute another release" - ) - release = matches[0] - compact = _compact_release(release) - rejections = compact.get("rejections") or [] - blocked_without_force = release.get("downloadAllowed") is False or bool(rejections) - force = kwargs.get("force") is True - - existing_file_count = None - season_number = query.get("seasonNumber") - if season_number is not None: - episodes = _unwrap( - await _call(client, "get_episode", {"seriesId": query["seriesId"], "seasonNumber": season_number}) - ) - if isinstance(episodes, list): - existing_file_count = sum( - bool(item.get("hasFile")) for item in episodes if isinstance(item, dict) - ) - - stable_release = _pick( - release, - ( - "guid", "title", "indexer", "indexerId", "size", "protocol", - "downloadAllowed", "releaseGroup", "seasonNumber", "fullSeason", - ), - ) - raw_rejections = release.get("rejections") - stable_release["rejections"] = ( - [str(reason) for reason in raw_rejections] - if isinstance(raw_rejections, list) - else [] - ) - scope = { - "series_id": query["seriesId"], - "season_number": query.get("seasonNumber"), - "episode_id": query.get("episodeId"), - "guid": guid, - "force": force, - } - fingerprint = json.dumps( - {"scope": scope, "release": stable_release}, - ensure_ascii=False, - sort_keys=True, - separators=(",", ":"), - ) - return { - "scope": scope, - "fingerprint": fingerprint, - "release": compact, - "raw_release": release, - "existing_episode_files_in_season": existing_file_count, - "blocked_without_force": blocked_without_force, - } - - -async def _preview_release_grab(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - resolved = await _resolve_release_grab(client, kwargs) - ticket = secrets.token_urlsafe(24) - now = time.monotonic() - for old_ticket, (expires, _) in list(_APPROVALS.items()): - if expires <= now: - _APPROVALS.pop(old_ticket, None) - force_required = resolved["blocked_without_force"] and not resolved["scope"]["force"] - if not force_required: - _APPROVALS[ticket] = (now + APPROVAL_TTL_SECONDS, resolved["fingerprint"]) - return { - "action": "preview-only", - "scope": resolved["scope"], - "release": resolved["release"], - "existing_episode_files_in_season": resolved["existing_episode_files_in_season"], - "monitoring_changed": False, - "download_started": False, - "existing_files_deleted": False, - "replacement_guaranteed": False, - "warning": ( - "Grabbing a season pack does not itself delete or guarantee replacement of existing " - "episode files. Sonarr applies its import, quality-profile and upgrade rules after download." - ), - "force_required": force_required, - "approval_ticket": None if force_required else ticket, - "approval_expires_in_seconds": None if force_required else APPROVAL_TTL_SECONDS, - "next_step": ( - "This result has Sonarr rejections or downloadAllowed=false. Explain the rejections and " - "only after the user explicitly accepts them call preview_release_grab again with force=true." - if force_required else - "Show the exact title, size, indexer, rejections and overwrite warning. Only after explicit " - "approval call grab_release with exactly the same scope, confirm=true and this ticket." - ), - } - - -async def _grab_release(client: Any, kwargs: dict[str, Any]) -> dict[str, Any]: - if kwargs.get("confirm") is not True: - raise PermissionError("confirm=true is required after reviewing preview_release_grab") - ticket = str(kwargs.get("approval_ticket", "")) - if not ticket: - raise PermissionError("approval_ticket is required") - resolved = await _resolve_release_grab(client, kwargs) - if resolved["blocked_without_force"] and not resolved["scope"]["force"]: - raise PermissionError( - "This release has Sonarr rejections or downloadAllowed=false; an explicitly approved " - "force=true preview is required" - ) - approval = _APPROVALS.pop(ticket, None) - if approval is None or approval[0] <= time.monotonic(): - raise PermissionError("Approval ticket is missing, expired or already used") - if not secrets.compare_digest(approval[1], resolved["fingerprint"]): - raise PermissionError("Approval ticket does not match this exact release grab") - result = _unwrap( - await _call(client, "post_release", {"data": resolved["raw_release"]}) - ) - return { - "ok": True, - "download_started": True, - "selected_release": resolved["release"], - "sonarr_result": _compact_release(result) if isinstance(result, dict) else result, - "monitoring_changed": False, - "existing_files_deleted": False, - "replacement_guaranteed": False, - "instruction": ( - "The exact approved release was sent to Sonarr's configured download client. " - "Do not claim that an existing episode was overwritten; verify queue/import history later." - ), - } - - -def register_sonarr_tools(mcp: FastMCP) -> None: - @mcp.tool(tags={"sonarr"}) - async def sonarr_find_series( - query: str = Field(description="Series title or title fragment, for example Mord ist ihr Hobby."), - limit: int = Field(default=8, ge=1, le=15), - ) -> Any: - """Find a Sonarr series by name. READ ONLY. Never changes monitoring and never starts a search or download.""" - client = get_sonarr_client() - return await _find_series(client, {"query": query, "limit": limit}) - - @mcp.tool(tags={"sonarr"}) - async def sonarr_get_season_summary( - series_id: int = Field(description="Exact Sonarr series id returned by sonarr_find_series."), - season_number: int = Field(ge=0, description="Season number."), - ) -> Any: - """Return episodes and existing files for one Sonarr season. READ ONLY. File names do not prove audio language.""" - return await _season_summary( - get_sonarr_client(), - {"series_id": series_id, "season_number": season_number}, - ) - - @mcp.tool(tags={"sonarr"}) - async def sonarr_search_releases( - series_id: int = Field(description="Exact Sonarr series id."), - season_number: int | None = Field(default=None, ge=0), - episode_id: int | None = Field(default=None, ge=1), - release_group: str = Field(default="", description="Optional release-group filter, for example FuN."), - season_pack_only: bool = Field(default=False, description="Only complete season packs. Requires season_number."), - ) -> Any: - """Search Sonarr's configured indexers and return compact matching releases. READ ONLY: does not alter monitoring, start automatic search, grab, or download anything.""" - return await _search_releases( - get_sonarr_client(), - { - "series_id": series_id, - "season_number": season_number, - "episode_id": episode_id, - "release_group": release_group, - "season_pack_only": season_pack_only, - }, - ) - - @mcp.tool(tags={"sonarr"}) - async def sonarr_system_status() -> Any: - """Return compact Sonarr version and runtime status. READ ONLY.""" - return _compact_result("get_system_status", await _call(get_sonarr_client(), "get_system_status")) - - if os.environ.get("ARR_MCP_WRITE", "").strip().lower() not in ("1", "true", "yes", "on"): - return - - @mcp.tool(tags={"sonarr", "write"}) - async def sonarr_preview_release_grab( - series_id: int = Field(description="Exact Sonarr series id."), - guid: str = Field(description="Exact GUID returned by sonarr_search_releases."), - season_number: int | None = Field(default=None, ge=0), - episode_id: int | None = Field(default=None, ge=1), - force: bool = Field(default=False), - ) -> Any: - """Preview one exact release grab and issue a short-lived approval ticket. Does not download anything.""" - return await _preview_release_grab( - get_sonarr_client(), - { - "series_id": series_id, - "guid": guid, - "season_number": season_number, - "episode_id": episode_id, - "force": force, - }, - ) - - @mcp.tool(tags={"sonarr", "write"}) - async def sonarr_grab_release( - series_id: int = Field(description="Same series id used for the preview."), - guid: str = Field(description="Same exact release GUID used for the preview."), - approval_ticket: str = Field(description="Ticket returned by sonarr_preview_release_grab."), - confirm: bool = Field(description="Must be true after explicit user approval."), - season_number: int | None = Field(default=None, ge=0), - episode_id: int | None = Field(default=None, ge=1), - force: bool = Field(default=False), - ) -> Any: - """Grab exactly one previously previewed release. WRITE: can immediately start a download.""" - return await _grab_release( - get_sonarr_client(), - { - "series_id": series_id, - "guid": guid, - "approval_ticket": approval_ticket, - "confirm": confirm, - "season_number": season_number, - "episode_id": episode_id, - "force": force, - }, - ) - - @mcp.tool(tags={"sonarr", "write"}) - async def sonarr_preview_episode_search( - series_id: int = Field(description="Exact Sonarr series id."), - season_number: int = Field(ge=0), - episode_numbers: list[int] | None = Field(default=None, description="Optional episode numbers; omit for all missing episodes in the season."), - ) -> Any: - """Preview an automatic Sonarr episode search. Does not change monitoring or download anything.""" - return await _preview_episode_search( - get_sonarr_client(), - {"series_id": series_id, "season_number": season_number, "episode_numbers": episode_numbers}, - ) - - @mcp.tool(tags={"sonarr", "write"}) - async def sonarr_start_episode_search( - series_id: int = Field(description="Same series id used for the preview."), - season_number: int = Field(ge=0), - approval_ticket: str = Field(description="Ticket returned by sonarr_preview_episode_search."), - confirm: bool = Field(description="Must be true after explicit user approval."), - episode_numbers: list[int] | None = Field(default=None), - ) -> Any: - """Start a previously previewed automatic episode search. WRITE: may immediately download releases.""" - return await _start_episode_search( - get_sonarr_client(), - { - "series_id": series_id, - "season_number": season_number, - "episode_numbers": episode_numbers, - "approval_ticket": approval_ticket, - "confirm": confirm, - }, - ) diff --git a/platform/mcp/sync-clients.py b/platform/mcp/sync-clients.py deleted file mode 100755 index c94db1f..0000000 --- a/platform/mcp/sync-clients.py +++ /dev/null @@ -1,177 +0,0 @@ -#!/usr/bin/env python3 -"""Generate Hermes and OpenWebUI MCP registrations from one JSON registry.""" - -from __future__ import annotations - -import argparse -import json -import os -import pathlib -import sqlite3 -import time - - -BEGIN = "# BEGIN MANAGED MCP SERVERS" -END = "# END MANAGED MCP SERVERS" -CLIENT_TOKEN = "" - - -def env_file(path: str) -> dict[str, str]: - values: dict[str, str] = {} - source = pathlib.Path(path) - if not source.is_file(): - return values - for raw in source.read_text(encoding="utf-8", errors="replace").splitlines(): - line = raw.strip() - if not line or line.startswith("#") or "=" not in line: - continue - key, value = line.split("=", 1) - values[key.strip()] = value.strip().strip('"').strip("'") - return values - - -def enabled(item: dict) -> bool: - required = item.get("required_file") - if required and not pathlib.Path(required).is_file(): - return False - source = item.get("env_file") - if source: - values = env_file(source) - url_ready = bool(item.get("url")) or bool(values.get(item.get("url_env", ""))) - key_ready = (not item.get("key_env") - or bool(values.get(item["key_env"])) - or (item.get("key_env") == "MCPHUB_BEARER_TOKEN" - and bool(CLIENT_TOKEN))) - return url_ready and key_ready - return True - - -def resolved(item: dict) -> tuple[str, str]: - if item.get("env_file"): - values = env_file(item["env_file"]) - url = item.get("url") or values[item["url_env"]] - key = values.get(item.get("key_env", ""), "") - if not key and item.get("key_env") == "MCPHUB_BEARER_TOKEN": - key = CLIENT_TOKEN - return url, key - return item["url"], "" - - -def active(registry: pathlib.Path, client: str) -> list[dict]: - document = json.loads(registry.read_text(encoding="utf-8")) - if document.get("version") != 1 or not isinstance(document.get("servers"), list): - raise SystemExit("Unsupported MCP registry schema") - return [item for item in document["servers"] if client in item.get("clients", []) and enabled(item)] - - -def yaml_quote(value: str) -> str: - return json.dumps(value, ensure_ascii=False) - - -def hermes_block(items: list[dict]) -> str: - lines = [BEGIN, "mcp_servers:"] - for item in items: - url, key = resolved(item) - lines.extend([ - f" {item.get('hermes_id', item['id'])}:", - f" url: {yaml_quote(url)}", - ]) - if key: - lines.extend([" headers:", f" Authorization: {yaml_quote('Bearer ' + key)}"]) - if "tool_include" in item: - lines.append(" tools:") - lines.append(" include:") - for tool in item["tool_include"]: - lines.append(f" - {yaml_quote(str(tool))}") - elif "tool_exclude" in item: - lines.append(" tools:") - lines.append(" exclude:") - for tool in item["tool_exclude"]: - lines.append(f" - {yaml_quote(str(tool))}") - lines.extend([ - f" timeout: {int(item.get('timeout', 300))}", - " connect_timeout: 30", - " supports_parallel_tool_calls: false", - ]) - lines.append(END) - return "\n".join(lines) + "\n" - - -def update_hermes(path: pathlib.Path, block: str) -> None: - if not path.is_file(): - return - text = path.read_text(encoding="utf-8") - if BEGIN in text and END in text: - prefix, rest = text.split(BEGIN, 1) - _, suffix = rest.split(END, 1) - text = prefix.rstrip() + "\n\n" + block + suffix.lstrip("\n") - else: - marker = "\nmcp_servers:" - if marker in text: - text = text.split(marker, 1)[0].rstrip() + "\n\n" + block - else: - text = text.rstrip() + "\n\n" + block - path.write_text(text, encoding="utf-8") - - -def openwebui_connection(item: dict) -> dict: - url, key = resolved(item) - config = {"enable": True, "access_grants": []} - if item.get("functions"): - config["function_name_filter_list"] = item["functions"] - elif item.get("tool_include"): - config["function_name_filter_list"] = ",".join(item["tool_include"]) - return { - "url": url, "path": "", "type": "mcp", - "auth_type": item.get("auth_type", "none"), "headers": None, - "key": key, "config": config, - "info": {"id": item["id"], "name": item["name"], "description": item["description"]}, - } - - -def update_openwebui(db: pathlib.Path, items: list[dict]) -> None: - con = sqlite3.connect(db) - now = int(time.time()) - row = con.execute("select value from config where key=?", ("tool_server.connections",)).fetchone() - old = json.loads(row[0]) if row else [] - if not isinstance(old, list): - raise SystemExit("Unexpected OpenWebUI tool_server.connections format") - managed_ids = { - "athena-platform", "athena-operator-local", "web-general-local", "github-local", - "homeassistant-local", "arr-local", "navidrome-local", "mua", - "mua-readonly-local", "athena-terminal-local", "unraid-readonly-local", "web-local", - } - keep = [entry for entry in old if str((entry.get("info") or {}).get("id", "")) not in managed_ids] - keep.extend(openwebui_connection(item) for item in items) - with con: - con.execute( - """insert into config (key,value,updated_at) values (?,?,?) - on conflict(key) do update set value=excluded.value,updated_at=excluded.updated_at""", - ("tool_server.connections", json.dumps(keep, ensure_ascii=False), now), - ) - con.close() - - -def main() -> None: - global CLIENT_TOKEN - parser = argparse.ArgumentParser() - parser.add_argument("--registry", type=pathlib.Path, required=True) - parser.add_argument("--hermes", type=pathlib.Path, action="append", default=[]) - parser.add_argument("--openwebui-db", type=pathlib.Path) - parser.add_argument("--mcphub-token-file", type=pathlib.Path) - args = parser.parse_args() - if args.mcphub_token_file: - CLIENT_TOKEN = args.mcphub_token_file.read_text(encoding="utf-8").strip() - if not CLIENT_TOKEN: - raise SystemExit("MCPHub token file is empty") - if args.hermes: - block = hermes_block(active(args.registry, "hermes")) - for path in args.hermes: - update_hermes(path, block) - if args.openwebui_db: - update_openwebui(args.openwebui_db, active(args.registry, "openwebui")) - print("MCP_CLIENT_SYNC_OK") - - -if __name__ == "__main__": - main() diff --git a/platform/mcp/verify-navidrome.sh b/platform/mcp/verify-navidrome.sh deleted file mode 100755 index fc59a1b..0000000 --- a/platform/mcp/verify-navidrome.sh +++ /dev/null @@ -1,96 +0,0 @@ -#!/usr/bin/env bash -set -Eeuo pipefail - -CONTAINER=${OPENWEBUI_CONTAINER:-mike-ai-open-webui} -ENV_FILE=${NAVIDROME_MCP_ENV_FILE:-/etc/mike-ai/navidrome-mcp.env} - -die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; } -[[ $EUID -eq 0 ]] || die "Bitte als root ausführen." -[[ -s $ENV_FILE ]] || die "Navidrome-Secret-Datei fehlt." -[[ $(docker inspect -f '{{.State.Health.Status}}' mike-ai-mcp-navidrome 2>/dev/null || true) == healthy ]] || \ - die "Navidrome-MCP ist nicht gesund." -[[ $(docker inspect -f '{{.State.Health.Status}}' "$CONTAINER" 2>/dev/null || true) == healthy ]] || \ - die "OpenWebUI ist nicht gesund." - -expect_lastfm=false -grep -q '^LASTFM_API_KEY=..' "$ENV_FILE" && expect_lastfm=true - -result=$(docker exec -i -e EXPECT_LASTFM="$expect_lastfm" "$CONTAINER" python - <<'PY' -import asyncio -import json -import os -import sqlite3 -from mcp import ClientSession -from mcp.client.streamable_http import streamablehttp_client - -EXPECTED_LASTFM = { - "get_similar_artists", "get_similar_tracks", "get_artist_info", - "get_top_tracks_by_artist", "get_trending_music", "get_artist_albums", - "get_album_info", -} -PLAYBACK = {"play_songs", "pause", "set_volume"} - -def find_unanchored(value, path=""): - bad = [] - if isinstance(value, dict): - for key, child in value.items(): - here = f"{path}.{key}" if path else key - if key == "pattern" and ( - not isinstance(child, str) - or not child.startswith("^") - or not child.endswith("$") - ): - bad.append(here) - bad.extend(find_unanchored(child, here)) - elif isinstance(value, list): - for index, child in enumerate(value): - bad.extend(find_unanchored(child, f"{path}[{index}]")) - return bad - -async def verify(): - async with streamablehttp_client( - "http://mike-ai-mcp-navidrome:3000/mcp" - ) as (read, write, _): - async with ClientSession(read, write) as session: - await session.initialize() - result = await session.list_tools() - names = {tool.name for tool in result.tools} - bad = [] - for tool in result.tools: - bad.extend(find_unanchored(tool.inputSchema, tool.name)) - if bad: - raise SystemExit("Unverankerte JSON-Schema-Patterns: " + ", ".join(bad)) - if PLAYBACK & names: - raise SystemExit("Playback-Werkzeuge sind auf dem Headless-Host aktiv.") - expect_lastfm = os.environ.get("EXPECT_LASTFM") == "true" - if expect_lastfm and not EXPECTED_LASTFM <= names: - raise SystemExit("Last.fm-Werkzeugkatalog ist unvollständig.") - if not expect_lastfm and EXPECTED_LASTFM & names: - raise SystemExit("Last.fm-Werkzeuge sind ohne konfigurierten Schlüssel aktiv.") - if expect_lastfm: - # Public metadata only. Do not print the returned chart data. - response = await session.call_tool( - "get_trending_music", {"type": "artists", "limit": 1} - ) - if response.isError: - raise SystemExit("Öffentliche Last.fm-Testabfrage ist fehlgeschlagen.") - - con = sqlite3.connect("/app/backend/data/webui.db") - row = con.execute( - "select value from config where key=?", ("tool_server.connections",) - ).fetchone() - connections = json.loads(row[0]) if row else [] - ids = { - str((connection.get("info") or {}).get("id", "")) - for connection in connections if isinstance(connection, dict) - } - if "navidrome-local" not in ids: - raise SystemExit("OpenWebUI-Verbindung navidrome-local fehlt.") - print(f"NAVIDROME_ACCEPTANCE_OK tools={len(names)} lastfm={str(expect_lastfm).lower()}") - -asyncio.run(verify()) -PY -) -[[ $result == NAVIDROME_ACCEPTANCE_OK\ * ]] || \ - die "Navidrome-Abnahme lieferte keinen gültigen Erfolgsmarker." -printf '%s\n' "$result" diff --git a/platform/models/manifest.example.yaml b/platform/models/manifest.example.yaml index 6c6cea4..8bd7368 100644 --- a/platform/models/manifest.example.yaml +++ b/platform/models/manifest.example.yaml @@ -26,11 +26,11 @@ models: file: ggml-large-v3-turbo.bin target: /opt/mike-ai/models/whisper/ggml-large-v3-turbo.bin sha256: "REPLACE_AFTER_VERIFICATION" - flux: + image: role: image-generation - source: black-forest-labs/FLUX.2-klein-4B - target: /data/models/FLUX.2-klein-4B - revision: "303481f0390afb112393f9d77e8f0be72fcefeb7" + source: Tongyi-MAI/Z-Image-Turbo + target: /data/models/Z-Image-Turbo + revision: "f332072aa78be7aecdf3ee76d5c247082da564a6" xtts: role: text-to-speech source: coqui/XTTS-v2 diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index 698f093..5f71941 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -18,7 +18,7 @@ Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra, Kommandos: POST /fast, /medium, /large, /ultra, /uncensored GET /status (Zustand) -Bildgenerierung (FLUX.2 [klein] 4B Base): +Bildgenerierung (Z-Image-Turbo): POST /v1/images/generations (OpenAI-kompatibel) GET /images (Liste) GET /images/ (PNG-Download) @@ -37,7 +37,7 @@ Der Router leitet /v1/audio/speech und /v1/audio/transcriptions per HTTP an die Worker weiter. Der Router agiert als Modell-Orchestrator: vor der Generierung wird -llama.cpp gestoppt, der Bild-Worker lädt FLUX, generiert und entlädt +llama.cpp gestoppt, der Bild-Worker lädt Z-Image, generiert und entlädt das Modell wieder; danach wird das vorherige Qwen-Profil wiederher- gestellt und erst dann geantwortet (try/finally – Qwen wird auch bei Fehlgeschlagener Generierung wiederhergestellt). @@ -116,7 +116,7 @@ CONNECT_TIMEOUT = float(os.environ.get("CONNECT_TIMEOUT", "10")) # s, Connect POLL_INTERVAL = float(os.environ.get("POLL_INTERVAL", "2")) # s, Polling-Intervall MAX_GENERATION_TOKENS = int(os.environ.get("MAX_GENERATION_TOKENS", "8192")) -# --- Bildgenerierung (FLUX.2 [klein] 4B Base) --- +# --- Bildgenerierung (Z-Image-Turbo) --- LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service") SYSTEMCTL_BIN = os.environ.get("SYSTEMCTL_BIN", "systemctl") IMAGE_WORKER = os.environ.get( @@ -125,6 +125,7 @@ IMAGE_PYTHON = os.environ.get( "IMAGE_PYTHON", "/opt/mike-ai/ai-profile-router/venv/bin/python") IMAGE_WORKER_URL = os.environ.get("IMAGE_WORKER_URL", "").rstrip("/") IMAGE_WORKER_TOKEN = os.environ.get("IMAGE_WORKER_TOKEN", "").strip() +IMAGE_MODEL_NAME = os.environ.get("IMAGE_MODEL_NAME", "Z-Image-Turbo") IMAGE_DIR = os.environ.get( "IMAGE_DIR", "/opt/mike-ai/ai-profile-router/images") IMAGE_WORKER_LOG = os.environ.get( @@ -144,8 +145,7 @@ CHAT_IMAGE_MAX_BYTES = int(os.environ.get( CHAT_IMAGE_ALLOW_REMOTE_URLS = os.environ.get( "CHAT_IMAGE_ALLOW_REMOTE_URLS", "false").lower() in {"1", "true", "yes"} -# Erlaubte Auflösungen (Breite x Höhe). FLUX.2 klein ist für 1 MP -# ausgelegt; 1920x1088 (≈2 MP) wird zusätzlich unterstützt. +# Erlaubte Auflösungen (Breite x Höhe). IMAGE_SIZES = { "1024x1024": (1024, 1024), "1536x1024": (1536, 1024), @@ -153,10 +153,8 @@ IMAGE_SIZES = { "1920x1088": (1920, 1088), "1088x1920": (1088, 1920), } -# FLUX.2 Klein Distilled ist fest auf vier Schritte und Guidance 1.0 -# destilliert. Qualitätsstufen bleiben aus OpenAI-Kompatibilitätsgründen -# akzeptiert, ändern aber bewusst nicht die offiziellen Sampling-Werte. -IMAGE_QUALITY = {"standard": 4, "high": 4} +# Z-Image-Turbo nutzt neun Scheduler-Schritte (acht DiT-Forwards) ohne CFG. +IMAGE_QUALITY = {"standard": 9, "high": 9} IMAGE_DEFAULT_QUALITY = "standard" IMAGE_MAX_N = 4 @@ -433,6 +431,136 @@ def upstream_status() -> dict: "ctx": (m.get("meta") or {}).get("n_ctx")} +_TELEMETRY_LOCK = threading.Lock() +_TELEMETRY_AT = 0.0 +_TELEMETRY_CACHE: dict = {} + + +def _upstream_read(path: str, *, timeout: float = 1.5) -> tuple[int, bytes]: + """Read a bounded, read-only llama.cpp telemetry endpoint.""" + conn = http.client.HTTPConnection(UPSTREAM_HOST, UPSTREAM_PORT, + timeout=timeout) + try: + conn.request("GET", path, headers={"Accept": "application/json,text/plain"}) + resp = conn.getresponse() + return resp.status, resp.read(2 * 1024 * 1024) + finally: + conn.close() + + +def _parse_prometheus_metrics(raw: str) -> dict: + wanted = { + "llamacpp:prompt_tokens_total", + "llamacpp:prompt_tokens_cached_total", + "llamacpp:prompt_seconds_total", + "llamacpp:tokens_predicted_total", + "llamacpp:tokens_predicted_seconds_total", + "llamacpp:n_decode_total", + "llamacpp:n_tokens_max", + "llamacpp:spec_decode_num_draft_tokens_total", + "llamacpp:spec_decode_num_accepted_tokens_total", + "llamacpp:spec_decode_num_drafts_total", + "llamacpp:prompt_tokens_seconds", + "llamacpp:predicted_tokens_seconds", + "llamacpp:requests_processing", + "llamacpp:requests_deferred", + "llamacpp:n_busy_slots_per_decode", + } + result: dict[str, int | float] = {} + for line in raw.splitlines(): + if not line or line.startswith("#") or " " not in line: + continue + name, value = line.rsplit(None, 1) + if "{" in name or name not in wanted: + continue + try: + number = float(value) + result[name.removeprefix("llamacpp:")] = ( + int(number) if number.is_integer() else number + ) + except ValueError: + continue + return result + + +def upstream_telemetry() -> dict: + """Compact llama.cpp slots, rates, cache and MTP telemetry. + + The result is cached briefly because the dashboard refreshes every second. + Failures never affect inference or the normal router status response. + """ + global _TELEMETRY_AT, _TELEMETRY_CACHE + now = time.monotonic() + with _TELEMETRY_LOCK: + if now - _TELEMETRY_AT < 0.75 and _TELEMETRY_CACHE: + return _TELEMETRY_CACHE + result: dict = {"available": False, "slots": [], "metrics": {}} + errors: dict[str, str] = {} + try: + status, body = _upstream_read("/slots") + if status == 200: + raw_slots = json.loads(body) + for slot in raw_slots if isinstance(raw_slots, list) else []: + next_token = (slot.get("next_token") or [{}])[0] + params = slot.get("params") or {} + prompt = int(slot.get("n_prompt_tokens") or 0) + decoded = int(next_token.get("n_decoded") or 0) + n_ctx = int(slot.get("n_ctx") or 0) + result["slots"].append({ + "id": slot.get("id"), + "task_id": slot.get("id_task"), + "processing": bool(slot.get("is_processing")), + "speculative": bool(slot.get("speculative")), + "n_ctx": n_ctx, + "prompt_tokens": prompt, + "prompt_processed": int(slot.get("n_prompt_tokens_processed") or 0), + "prompt_cached": int(slot.get("n_prompt_tokens_cache") or 0), + "decoded_tokens": decoded, + "context_used": min(n_ctx, prompt + decoded) if n_ctx else prompt + decoded, + "remaining_generation": next_token.get("n_remain"), + "max_tokens": params.get("max_tokens", params.get("n_predict")), + "temperature": params.get("temperature"), + "stream": params.get("stream"), + }) + else: + errors["slots"] = f"HTTP {status}" + except (OSError, ValueError, KeyError, TypeError, http.client.HTTPException) as exc: + errors["slots"] = str(exc) + try: + status, body = _upstream_read("/metrics") + if status == 200: + result["metrics"] = _parse_prometheus_metrics( + body.decode("utf-8", "replace") + ) + else: + errors["metrics"] = f"HTTP {status}" + except (OSError, ValueError, http.client.HTTPException) as exc: + errors["metrics"] = str(exc) + try: + status, body = _upstream_read("/props") + if status == 200: + props = json.loads(body) + result["props"] = { + "total_slots": props.get("total_slots"), + "model_alias": props.get("model_alias"), + "model_ftype": props.get("model_ftype"), + "model_path": props.get("model_path"), + "modalities": props.get("modalities") or {}, + "default_context": ((props.get("default_generation_settings") or {}) + .get("n_ctx")), + } + else: + errors["props"] = f"HTTP {status}" + except (OSError, ValueError, TypeError, http.client.HTTPException) as exc: + errors["props"] = str(exc) + result["available"] = bool(result["slots"] or result["metrics"]) + if errors: + result["errors"] = errors + _TELEMETRY_CACHE = result + _TELEMETRY_AT = now + return result + + # --------------------------------------------------------------------------- # Profile # --------------------------------------------------------------------------- @@ -614,7 +742,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None: # --------------------------------------------------------------------------- -# Bildgenerierung (FLUX.2 [klein] 4B Base) +# Bildgenerierung (Z-Image-Turbo) # --------------------------------------------------------------------------- class _Worker: @@ -753,8 +881,9 @@ class _RemoteWorker: deadline = time.monotonic() + IMAGE_START_TIMEOUT while time.monotonic() < deadline: try: - self._request("GET", "/health", timeout=3) + health = self._request("GET", "/health", timeout=3) self.running = True + self.model_loaded = bool(health.get("model_loaded")) return except (OSError, urllib.error.URLError, TimeoutError, RuntimeError): time.sleep(1) @@ -804,7 +933,7 @@ def _vram_used_mib() -> int | None: def _wait_vram_free(threshold_mib: int = 1000, timeout: float | None = None) -> None: - """Wartet, bis der VRAM unter threshold_mib fällt (FLUX entladen). + """Wartet, bis der VRAM unter threshold_mib fällt (Bildmodell entladen). Wird nach dem Beenden des Bild-Workers aufgerufen, um sicherzustellen, dass der VRAM (inkl. CUDA-Kontext) frei ist, bevor Qwen neu startet. @@ -937,7 +1066,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int, "guidance": guidance, "quality": quality, "seconds": resp.get("seconds"), - "model": "FLUX.2-klein-4B", + "model": IMAGE_MODEL_NAME, "created": time.strftime("%Y-%m-%dT%H:%M:%S"), } meta_path = os.path.join(IMAGE_DIR, filename[:-4] + ".json") @@ -1475,10 +1604,17 @@ class Handler(BaseHTTPRequestHandler): and bool(up.get("model"))), "active_chats": active_chats, }, + "llama_telemetry": (upstream_telemetry() if up["reachable"] else { + "available": False, + "slots": [], + "metrics": {}, + "errors": {"upstream": up.get("error", "not reachable")}, + }), "image": { "phase": img.phase, "worker": "running" if (img.worker and img.worker.alive()) else "stopped", + "model": IMAGE_MODEL_NAME if img.phase != "idle" else None, "model_loaded": bool(img.worker and img.worker.model_loaded), "last_image": img.last_image, "last_seconds": img.last_seconds, @@ -1543,19 +1679,19 @@ class Handler(BaseHTTPRequestHandler): "invalid_request_error", "invalid_quality") return steps = data.get("steps", IMAGE_QUALITY[quality]) - if not isinstance(steps, int) or isinstance(steps, bool) or steps != 4: - self._send_error(400, "FLUX.2 Klein Distilled erfordert 'steps'=4", + if not isinstance(steps, int) or isinstance(steps, bool) or steps != 9: + self._send_error(400, "Z-Image-Turbo erfordert 'steps'=9", "invalid_request_error", "invalid_steps") return - guidance = data.get("guidance", 1.0) + guidance = data.get("guidance", 0.0) try: guidance = float(guidance) except (TypeError, ValueError): self._send_error(400, "'guidance' muss eine Zahl sein", "invalid_request_error", "invalid_guidance") return - if guidance != 1.0: - self._send_error(400, "FLUX.2 Klein Distilled erfordert 'guidance'=1.0", + if guidance != 0.0: + self._send_error(400, "Z-Image-Turbo erfordert 'guidance'=0.0", "invalid_request_error", "invalid_guidance") return diff --git a/smoke-test.sh b/smoke-test.sh index 6d22e8a..dca5138 100755 --- a/smoke-test.sh +++ b/smoke-test.sh @@ -14,7 +14,7 @@ command -v docker >/dev/null || fail "Docker fehlt." cd "$ROOT_DIR" ./manage.sh validate >/dev/null -pass "Profilmatrix, MCP-Liste und Compose-Konfiguration stimmen" +pass "Profilmatrix und Compose-Konfiguration stimmen" healthy() { local name=$1 state health @@ -30,6 +30,7 @@ for name in \ mike-ai-piper \ mike-ai-xtts \ mike-ai-tts-gateway \ + mike-ai-llama-dashboard \ mike-ai-mcp-athena-operator \ mike-ai-backup; do healthy "$name" || fail "$name fehlt oder ist nicht gesund" @@ -49,11 +50,16 @@ for legacy in \ mike-ai-mcp-web \ mike-ai-tools-searxng \ mike-ai-tools-tinysearch \ - mike-ai-mcp-platform-context; do - [[ $(docker inspect -f '{{.State.Running}}' "$legacy" 2>/dev/null || true) != true ]] || \ - fail "Altlast laeuft noch: $legacy" + mike-ai-mcp-platform-context \ + mike-ai-mcp-arr \ + mike-ai-mcp-deemix \ + mike-ai-mcp-github \ + mike-ai-mcp-homeassistant \ + mike-ai-mcp-navidrome; do + [[ -z $(docker inspect -f '{{.Name}}' "$legacy" 2>/dev/null || true) ]] || \ + fail "Altlast existiert noch: $legacy" done -pass "OpenWebUI, alte Hermes-Dienste und migrierte MCPs sind aus" +pass "OpenWebUI, alte Hermes-Dienste und migrierte MCPs sind entfernt" docker exec mike-ai-router python - <<'PY' >/dev/null import json