diff --git a/README.md b/README.md index 5f4a074..4953871 100644 --- a/README.md +++ b/README.md @@ -16,7 +16,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.** - Whisper.cpp `ggml-small` auf der CPU für lokale deutsche Spracherkennung - Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 - Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio, - BS-RoFormer-Stimmtrennung, OmniVoice und X-VC + BS-RoFormer-Stimmtrennung, OmniVoice, X-VC, Seed-VC und Applio/RVC - Portainer CE als optionale Container-Ansicht auf Port 9443 - WireGuard-Gateway, Datenbackup und Athena-Operator - keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz @@ -127,10 +127,12 @@ Details, Installation, Prüfung und Rollback stehen in - Spuren trennen (BS-RoFormer + Demucs, 2/4/6 Stems): `http://192.168.1.212:8007` - Voice Studio (OmniVoice, Text zu Stimme): `http://192.168.1.212:8008` - Voice Changer (X-VC, Audio zu Audio; native 16 kHz plus optional restaurierte 44,1 kHz): `http://192.168.1.212:8009` +- Seed-VC (Zero-Shot Sprache/Gesang zu Referenzstimme): `http://192.168.1.212:8010` +- Applio (RVC-Inferenz, Modelle und Training): `http://192.168.1.212:8011` Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren außerdem `/athena music`, `/athena stems`, `/athena voice`, -`/athena voicechange`, `/athena llm` und `/athena status`; Details stehen in +`/athena voicechange`, `/athena seedvc`, `/athena applio`, `/athena llm` und `/athena status`; Details stehen in [docs/OPERATING_MODES.md](docs/OPERATING_MODES.md). Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über diff --git a/compose.yaml b/compose.yaml index 9899dff..460dd19 100644 --- a/compose.yaml +++ b/compose.yaml @@ -573,6 +573,8 @@ services: SEPARATOR_WORKER: bs-roformer VOICE_WORKER: vevo2 VOICE_CHANGE_WORKER: xvc + SEED_VC_WORKER: seed-vc + APPLIO_WORKER: applio networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: @@ -632,6 +634,8 @@ services: ENABLE_MUSIC_MODE: "true" MUSIC_START_TIMEOUT: "600" VOICE_CHANGE_START_TIMEOUT: "600" + SEED_VC_START_TIMEOUT: "900" + APPLIO_START_TIMEOUT: "900" STT_WORKER_URL: http://whisper:8084 STT_TIMEOUT: "300" networks: [frontend, control, inference] @@ -865,6 +869,8 @@ services: SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}" VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}" VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}" + SEED_VC_UI_URL: "${SEED_VC_UI_URL:-http://192.168.1.212:8010/}" + APPLIO_UI_URL: "${APPLIO_UI_URL:-http://192.168.1.212:8011/}" HOST_PROC: /host/proc HOST_DATA: /host/data HOST_MODELS: /host/models diff --git a/dev/test_dashboard_modes.py b/dev/test_dashboard_modes.py index 168b1bf..c2b3126 100644 --- a/dev/test_dashboard_modes.py +++ b/dev/test_dashboard_modes.py @@ -70,6 +70,20 @@ class DashboardModeTests(unittest.TestCase): request = urlopen.call_args.args[0] self.assertEqual(json.loads(request.data), {"mode": "voicechange"}) + def test_seed_vc_mode_is_forwarded_to_router(self): + with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen: + status, body = self.dashboard.change_mode("seedvc") + self.assertEqual(status, 202) + self.assertEqual(body, {"status": "accepted"}) + self.assertEqual(json.loads(urlopen.call_args.args[0].data), {"mode": "seedvc"}) + + def test_applio_mode_is_forwarded_to_router(self): + with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen: + status, body = self.dashboard.change_mode("applio") + self.assertEqual(status, 202) + self.assertEqual(body, {"status": "accepted"}) + self.assertEqual(json.loads(urlopen.call_args.args[0].data), {"mode": "applio"}) + def test_unknown_mode_is_rejected_without_router_request(self): with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen: status, body = self.dashboard.change_mode("unknown") diff --git a/docs/CONTAINER_INVENTORY.md b/docs/CONTAINER_INVENTORY.md index 49ef421..7b34fd1 100644 --- a/docs/CONTAINER_INVENTORY.md +++ b/docs/CONTAINER_INVENTORY.md @@ -2,7 +2,7 @@ Stand: 9. September 2026 -Athena besteht derzeit aus 23 Docker-Containern. Nicht jeder Container enthält +Athena besteht derzeit aus 25 Docker-Containern. Nicht jeder Container enthält ein KI-Modell: Router, Oberflächen, Netzwerk, Steuerung und Sicherung sind gewöhnliche Dienste. Die rechenintensiven GPU-Worker werden absichtlich nur bei Bedarf gestartet. Ein Container im Zustand `Created` oder `Exited (0)` ist daher @@ -11,6 +11,7 @@ nicht automatisch ein ungenutzter Rest. | Container | Modell oder wesentliche Komponente | Aufgabe | |---|---|---| | `mike-ai-backup` | kein Modell; Offen Docker Volume Backup | Sichert `/data`, `/etc/mike-ai`, den Stack und die persistenten Docker-Volumes im Fünf-Stunden-Takt. | +| `mike-ai-applio-studio` | Applio/RVC; Stimmenmodelle werden nutzerseitig ergänzt | Vollständige RVC-Oberfläche für Inferenz, Modellverwaltung und Training auf der RTX 5080. | | `mike-ai-image-worker` | FLUX.2 Klein 9B FP8, Qwen3-8B NF4 Textencoder und VAE | Erzeugt und bearbeitet Bilder transaktional; nutzt während eines Auftrags RTX 5080 und RTX 3060. | | `mike-ai-llama-beta1` | Qwen3.8-27B GSQ-RCO `IQ3_S-MTP`, Qwen-MMProj BF16 | Experimentelles Beta-Profil mit 112.000 Token Kontext; wird nur auf Anforderung geladen. | | `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung und Betriebsarten an und bietet die Modusumschaltung. | @@ -27,6 +28,7 @@ nicht automatisch ein ungenutzter Rest. | `mike-ai-profile-controller` | kein Modell | Startet und stoppt ausschließlich freigegebene Modellprofile und Spezialworker in einer sicheren Reihenfolge. | | `mike-ai-qwen3-tts` | `Qwen/Qwen3-TTS-12Hz-1.7B-Base`, Stimme Serena | Hochwertige deutsche Sprachausgabe auf der RTX 3060 im LLM-Betrieb. | | `mike-ai-router` | kein eigenes Modell | Einzige OpenAI-kompatible Modelladresse; koordiniert Profile, Bildaufträge, Sprache und Betriebsarten. | +| `mike-ai-seed-vc-studio` | Seed-VC V1; Gewichte werden beim ersten Auftrag geladen | Zero-Shot-Sprach- und Gesangswandlung anhand einer Referenzaufnahme auf der RTX 5080. | | `mike-ai-stem-separator` | BS-RoFormer Viperx 1297, `htdemucs_ft`, `htdemucs_6s`, `MossFormer2_SE_48K` | Trennt Gesang, Instrumente oder Sprache/Hintergrundgeräusche im exklusiven Separationsmodus. | | `mike-ai-tts-gateway` | kein eigenes Modell | Normalisiert Text, leitet TTS an Qwen3-TTS weiter und fällt bei Bedarf auf Piper zurück. | | `mike-ai-voice-studio` | `k2-fsa/OmniVoice` 0.2.1 mit Whisper-ASR | Erzeugt Text-to-Speech mit einer Referenzstimme; kein Audio-to-Audio-Voice-Changer. | diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md index a4c5541..655adbe 100644 --- a/docs/OPERATING_MODES.md +++ b/docs/OPERATING_MODES.md @@ -1,6 +1,6 @@ # Athena-Betriebsmodi -Athena besitzt fünf gegenseitig exklusive Betriebsmodi: +Athena besitzt sieben gegenseitig exklusive Betriebsmodi: - `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt. - `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt. @@ -11,6 +11,10 @@ Athena besitzt fünf gegenseitig exklusive Betriebsmodi: - `voicechange`: X-VC überträgt eine vorhandene Sprachaufnahme auf eine Referenzstimme und bewahrt dabei Inhalt und Timing. Alle anderen GPU-Dienste sind gestoppt. +- `seedvc`: Seed-VC V1 wandelt Sprache oder Gesang ohne Training anhand einer + Referenzaufnahme um. Alle anderen GPU-Dienste sind gestoppt. +- `applio`: Applio stellt RVC-Inferenz, Modellverwaltung und Training bereit. + Alle anderen GPU-Dienste sind gestoppt. Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus @@ -19,7 +23,7 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen ## Bedienung Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**, -**Voice Studio** und **Voice Changer** bereit. Im Musikmodus werden zwei Oberflächen angeboten: +**Voice Studio**, **X-VC**, **Seed-VC** und **Applio / RVC** bereit. Im Musikmodus werden zwei Oberflächen angeboten: - **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der @@ -71,6 +75,16 @@ des verwendeten GLM-4-Voice-Tokenizers ist Chinesisch und Englisch; Deutsch bleibt deshalb bis zur Hörabnahme ein Qualitätstest und kein zugesagter Produktionspfad. X-VC läuft ausschließlich auf der RTX 5080. +Seed-VC ist unter `http://192.168.1.212:8010` erreichbar. Die gepinnte V1- +Oberfläche unterstützt Zero-Shot-Sprach- und Gesangswandlung. Der Code ist auf +den archivierten Upstream-Commit `51383efd921027683c89e5348211d93ff12ac2a8` +fixiert; Gewichte werden beim ersten Auftrag persistent zwischengespeichert. + +Applio ist unter `http://192.168.1.212:8011` erreichbar. Der RVC-Pfad besitzt +eine eigene Modellbibliothek, Inferenz und Training. Hochwertige Inferenz +benötigt ein passendes RVC-Stimmenmodell. Der Code ist auf Commit +`7fa68ec2166ab1331c539704159fa14901e94e5a` fixiert. + Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom Router lokal beantwortet, auch wenn gerade kein LLM geladen ist: @@ -79,6 +93,8 @@ Router lokal beantwortet, auch wenn gerade kein LLM geladen ist: /athena stems /athena voice /athena voicechange +/athena seedvc +/athena applio /athena llm /athena status ``` @@ -91,6 +107,8 @@ POST /mode {"mode":"music"} POST /mode {"mode":"separation"} POST /mode {"mode":"voice"} POST /mode {"mode":"voicechange"} +POST /mode {"mode":"seedvc"} +POST /mode {"mode":"applio"} POST /mode {"mode":"llm"} ``` @@ -99,7 +117,9 @@ Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in `com.mike-ai.music-worker=acestep` beziehungsweise `com.mike-ai.stem-separator=bs-roformer` oder `com.mike-ai.voice-worker=vevo2` beziehungsweise -`com.mike-ai.voice-change-worker=xvc` markierten Container; freie +`com.mike-ai.voice-change-worker=xvc`, +`com.mike-ai.seed-vc-worker=seed-vc` oder +`com.mike-ai.applio-worker=applio` markierten Container; freie Container- oder Docker-Befehle werden nicht entgegengenommen. ## Wiederanlauf diff --git a/docs/TESTED_MODELS.md b/docs/TESTED_MODELS.md index b7d6a28..bb29494 100644 --- a/docs/TESTED_MODELS.md +++ b/docs/TESTED_MODELS.md @@ -61,6 +61,8 @@ Titelgenerierung und Kontextkompression in Hermes. | 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen | | 09.09.2026 | `chenxie95/X-VC`, Code `49df8c591eafc48b096e466d96f9839f9c0dd739`, UI-Basis `d761cd6421e85376b2656dfefd8471d7f35a42be` | Offizielles Beispiel Ende-zu-Ende gewandelt: 5,20 s Audio in 1,07 s (RTF 0,21), gültiges 16-kHz-Mono-PCM-WAV; Modell belegt rund 2,9 GiB auf der RTX 5080. GLM-4-Voice-Tokenizer dokumentiert Chinesisch und Englisch | **technischer Start- und Konvertierungstest bestanden**; deutsche Hörabnahme offen | | 09.09.2026 | Resemble Enhance 0.0.1, Modellrevision `4e3510ce4a8391159f665903544c5150bee7b2cb` | 14,56 s native X-VC-Ausgabe bei 16 kHz wurden auf der RTX 5080 in 3,55 s zu 44,1-kHz-PCM-WAV restauriert. 3,27 % der gemessenen Signalenergie lagen danach oberhalb 8 kHz; damit ist der Pfad keine bloße Neuabtastung. Wegen der alten Upstream-Pins läuft die reine Inferenz mit NumPy 1.26.4/SciPy 1.11.4 auf dem bestehenden Torch-2.8/CUDA-12.8-Unterbau | **technisch produktiv als optionaler A/B-Pfad**; Hörabnahme entscheidet, ob die rekonstruierten Höhen subjektiv besser oder künstlicher klingen | +| 09.09.2026 | `Plachtaa/seed-vc` V1, Code `51383efd921027683c89e5348211d93ff12ac2a8` | Gepinntes CUDA-12.8-/Torch-2.7.1-Image auf RTX 5080 gestartet; offizielle Gradio-Oberfläche und Healthcheck antworten. Gewichte folgen beim ersten echten Auftrag in den persistenten Hugging-Face-Cache | **technischer Starttest bestanden**; Konvertierungs- und Hörtest offen | +| 09.09.2026 | `IAHispano/Applio`, Code `7fa68ec2166ab1331c539704159fa14901e94e5a` | Gepinntes CUDA-12.8-fähiges Image auf RTX 5080 gestartet; vollständige Applio/RVC-Oberfläche antwortet. Rund 1,8 GiB Basisgewichte und die Konfiguration wurden persistent ausgelagert; kein Zielstimmenmodell vorinstalliert | **technischer Start- und Persistenztest bestanden**; Konvertierungstest mit einem ausgewählten RVC-Modell offen | ## Musikgenerierung diff --git a/experiments/applio-rvc/Dockerfile b/experiments/applio-rvc/Dockerfile new file mode 100644 index 0000000..a9761d7 --- /dev/null +++ b/experiments/applio-rvc/Dockerfile @@ -0,0 +1,25 @@ +# syntax=docker/dockerfile:1 +FROM python:3.12-trixie + +ARG APPLIO_COMMIT=7fa68ec2166ab1331c539704159fa14901e94e5a +ENV PATH=/app/.venv/bin:$PATH \ + HF_HOME=/models/huggingface \ + PIP_DISABLE_PIP_VERSION_CHECK=1 + +RUN apt-get update && apt-get install -y --no-install-recommends \ + ca-certificates curl ffmpeg git libportaudio2 \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app +RUN git clone https://github.com/IAHispano/Applio.git . \ + && git checkout "$APPLIO_COMMIT" \ + && python3 -m venv /app/.venv \ + && pip install --no-cache-dir --upgrade pip \ + && pip install --no-cache-dir python-ffmpeg \ + && pip install --no-cache-dir torch==2.7.1 torchvision==0.22.1 torchaudio==2.7.1 \ + --index-url https://download.pytorch.org/whl/cu128 \ + && sed -i '/^torch==/d;/^torchvision==/d;/^torchaudio==/d' requirements.txt \ + && pip install --no-cache-dir -r requirements.txt + +EXPOSE 6969 +CMD ["python3", "app.py", "--server-name", "0.0.0.0", "--port", "6969"] diff --git a/experiments/applio-rvc/README.md b/experiments/applio-rvc/README.md new file mode 100644 index 0000000..4a3a7ba --- /dev/null +++ b/experiments/applio-rvc/README.md @@ -0,0 +1,14 @@ +# Applio / RVC Studio + +Reproduzierbarer, experimenteller Applio-Worker mit der offiziellen +Weboberfläche. Der Build ist auf Upstream-Commit +`7fa68ec2166ab1331c539704159fa14901e94e5a` festgeschrieben. + +- Dashboard-Modus: `Applio / RVC` +- WireGuard-URL: `http://192.168.1.212:8011/` +- GPU: RTX 5080, exklusiv zu LLM, Musik- und anderen Voice-Modi +- Persistenz: Basisgewichte, importierte/trainierte Modelle, Konfiguration, + Logs und Hugging-Face-Cache unter `/data/voice/applio` + +Applio stellt die RVC-Werkzeuge und deren Oberfläche bereit. Eine konkrete +Zielstimme wird anschließend in der Oberfläche importiert oder trainiert. diff --git a/experiments/applio-rvc/compose.yaml b/experiments/applio-rvc/compose.yaml new file mode 100644 index 0000000..073faeb --- /dev/null +++ b/experiments/applio-rvc/compose.yaml @@ -0,0 +1,41 @@ +services: + applio-studio: + build: . + image: mike-ai/applio-studio:7fa68ec + container_name: mike-ai-applio-studio + restart: "no" + labels: + com.mike-ai.applio-worker: applio + environment: + NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID} + NVIDIA_DRIVER_CAPABILITIES: compute,utility + HF_HOME: /models/huggingface + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + ports: + - "127.0.0.1:8011:6969" + volumes: + - /data/voice/applio/huggingface:/models/huggingface + - /data/voice/applio/logs:/app/logs + - /data/voice/applio/models:/app/rvc/models + - /data/voice/applio/config.json:/app/assets/config.json + healthcheck: + test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:6969/ >/dev/null"] + interval: 5s + timeout: 3s + start_period: 900s + retries: 3 + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"] + capabilities: [gpu] + networks: + frontend: + aliases: [applio-studio] + +networks: + frontend: + name: mike-ai_frontend + external: true diff --git a/experiments/seed-vc/Dockerfile b/experiments/seed-vc/Dockerfile new file mode 100644 index 0000000..f5cd577 --- /dev/null +++ b/experiments/seed-vc/Dockerfile @@ -0,0 +1,23 @@ +FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu22.04 + +ARG SEED_VC_COMMIT=51383efd921027683c89e5348211d93ff12ac2a8 +ENV DEBIAN_FRONTEND=noninteractive \ + PIP_DISABLE_PIP_VERSION_CHECK=1 \ + HF_HOME=/models/huggingface + +RUN apt-get update && apt-get install -y --no-install-recommends \ + build-essential ca-certificates curl ffmpeg git libsndfile1 python3 python3-dev python3-pip \ + && rm -rf /var/lib/apt/lists/* + +WORKDIR /app +RUN git clone https://github.com/Plachtaa/seed-vc.git . \ + && git checkout "$SEED_VC_COMMIT" \ + && sed -i '/^torch\($\|[= <>=]\)/d;/^torchvision\($\|[= <>=]\)/d;/^torchaudio\($\|[= <>=]\)/d' requirements.txt \ + && python3 -m pip install --no-cache-dir \ + torch==2.7.1 torchvision==0.22.1 torchaudio==2.7.1 \ + --index-url https://download.pytorch.org/whl/cu128 \ + && python3 -m pip install --no-cache-dir -r requirements.txt \ + && sed -i 's/demo\.launch()/demo.launch(server_name="0.0.0.0", server_port=7860)/' app.py + +EXPOSE 7860 +CMD ["python3", "app.py", "--enable-v1"] diff --git a/experiments/seed-vc/README.md b/experiments/seed-vc/README.md new file mode 100644 index 0000000..86b0714 --- /dev/null +++ b/experiments/seed-vc/README.md @@ -0,0 +1,14 @@ +# Seed-VC Studio + +Reproduzierbarer, experimenteller Seed-VC-v2-Worker mit der offiziellen +Gradio-Oberfläche. Der Build ist auf Upstream-Commit +`51383efd921027683c89e5348211d93ff12ac2a8` festgeschrieben. + +- Dashboard-Modus: `Seed-VC` +- WireGuard-URL: `http://192.168.1.212:8010/` +- GPU: RTX 5080, exklusiv zu LLM, Musik- und anderen Voice-Modi +- Persistenz: Hugging-Face-Cache unter `/data/huggingface` + +Das Image enthält den Programmcode. Die benötigten Modellgewichte werden beim +ersten tatsächlichen Einsatz in den persistenten Cache geladen. + diff --git a/experiments/seed-vc/compose.yaml b/experiments/seed-vc/compose.yaml new file mode 100644 index 0000000..76a2851 --- /dev/null +++ b/experiments/seed-vc/compose.yaml @@ -0,0 +1,39 @@ +services: + seed-vc-studio: + build: . + image: mike-ai/seed-vc-studio:51383efd + container_name: mike-ai-seed-vc-studio + restart: "no" + labels: + com.mike-ai.seed-vc-worker: seed-vc + environment: + NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID} + NVIDIA_DRIVER_CAPABILITIES: compute,utility + HF_HOME: /models/huggingface + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + ports: + - "127.0.0.1:8010:7860" + volumes: + - /data/voice/seed-vc/huggingface:/models/huggingface + - /data/voice/seed-vc/output:/app/results + healthcheck: + test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:7860/ >/dev/null"] + interval: 5s + timeout: 3s + start_period: 900s + retries: 3 + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"] + capabilities: [gpu] + networks: + frontend: + aliases: [seed-vc-studio] + +networks: + frontend: + name: mike-ai_frontend + external: true diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index fd6e644..7ef6f9d 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -36,6 +36,10 @@ VOICE_LABEL_KEY = "com.mike-ai.voice-worker" VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip() VOICE_CHANGE_LABEL_KEY = "com.mike-ai.voice-change-worker" VOICE_CHANGE_WORKER = os.environ.get("VOICE_CHANGE_WORKER", "").strip() +SEED_VC_LABEL_KEY = "com.mike-ai.seed-vc-worker" +SEED_VC_WORKER = os.environ.get("SEED_VC_WORKER", "").strip() +APPLIO_LABEL_KEY = "com.mike-ai.applio-worker" +APPLIO_WORKER = os.environ.get("APPLIO_WORKER", "").strip() LOCK = threading.Lock() log = logging.getLogger("profile-controller") @@ -148,6 +152,28 @@ def voice_change_container() -> dict: return matches[0] +def seed_vc_container() -> dict: + if not SEED_VC_WORKER: + raise RuntimeError("Seed-VC worker is not configured") + matches = [item for item in labelled_containers(SEED_VC_LABEL_KEY) + if item.get("Labels", {}).get(SEED_VC_LABEL_KEY) == SEED_VC_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one Seed-VC worker {SEED_VC_WORKER!r}, found {len(matches)}") + return matches[0] + + +def applio_container() -> dict: + if not APPLIO_WORKER: + raise RuntimeError("Applio worker is not configured") + matches = [item for item in labelled_containers(APPLIO_LABEL_KEY) + if item.get("Labels", {}).get(APPLIO_LABEL_KEY) == APPLIO_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one Applio worker {APPLIO_WORKER!r}, found {len(matches)}") + return matches[0] + + def stop_music_if_configured() -> None: if MUSIC_WORKER: stop_container(music_container(), timeout=30) @@ -168,6 +194,27 @@ def stop_voice_change_if_configured() -> None: stop_container(voice_change_container(), timeout=30) +def stop_seed_vc_if_configured() -> None: + if SEED_VC_WORKER: + stop_container(seed_vc_container(), timeout=30) + + +def stop_applio_if_configured() -> None: + if APPLIO_WORKER: + stop_container(applio_container(), timeout=30) + + +def stop_voice_tools(except_kind: str | None = None) -> None: + if except_kind != "voice": + stop_voice_if_configured() + if except_kind != "voicechange": + stop_voice_change_if_configured() + if except_kind != "seedvc": + stop_seed_vc_if_configured() + if except_kind != "applio": + stop_applio_if_configured() + + def stop_container(item: dict, timeout: int = 120) -> None: if item.get("State") != "running": return @@ -207,8 +254,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: stop_container(tts_container(), timeout=30) stop_music_if_configured() stop_separator_if_configured() - stop_voice_if_configured() - stop_voice_change_if_configured() + stop_voice_tools() for other in image_containers(): if other["Id"] != item["Id"]: stop_container(other, timeout=20) @@ -235,8 +281,7 @@ def set_music_worker(running: bool) -> dict: stop_container(worker, timeout=20) stop_container(tts_container(), timeout=30) stop_separator_if_configured() - stop_voice_if_configured() - stop_voice_change_if_configured() + stop_voice_tools() start_container(item) else: stop_container(item, timeout=30) @@ -255,8 +300,7 @@ def set_separator_worker(running: bool) -> dict: stop_container(worker, timeout=20) stop_container(tts_container(), timeout=30) stop_music_if_configured() - stop_voice_if_configured() - stop_voice_change_if_configured() + stop_voice_tools() start_container(item) else: stop_container(item, timeout=30) @@ -276,7 +320,7 @@ def set_voice_worker(running: bool) -> dict: stop_container(tts_container(), timeout=30) stop_music_if_configured() stop_separator_if_configured() - stop_voice_change_if_configured() + stop_voice_tools("voice") start_container(item) else: stop_container(item, timeout=30) @@ -296,7 +340,7 @@ def set_voice_change_worker(running: bool) -> dict: stop_container(tts_container(), timeout=30) stop_music_if_configured() stop_separator_if_configured() - stop_voice_if_configured() + stop_voice_tools("voicechange") start_container(item) else: stop_container(item, timeout=30) @@ -304,6 +348,46 @@ def set_voice_change_worker(running: bool) -> dict: "state": "running" if running else "stopped"} +def set_seed_vc_worker(running: bool) -> dict: + """Start Seed-VC exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = seed_vc_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_separator_if_configured() + stop_voice_tools("seedvc") + start_container(item) + else: + stop_container(item, timeout=30) + return {"seed_vc_worker": SEED_VC_WORKER, + "state": "running" if running else "stopped"} + + +def set_applio_worker(running: bool) -> dict: + """Start Applio exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = applio_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_separator_if_configured() + stop_voice_tools("applio") + start_container(item) + else: + stop_container(item, timeout=30) + return {"applio_worker": APPLIO_WORKER, + "state": "running" if running else "stopped"} + + def active_profile(items: dict[str, dict] | None = None) -> str | None: items = items or containers() active = [name for name, item in items.items() if item.get("State") == "running"] @@ -321,8 +405,7 @@ def activate(profile: str) -> dict: stop_container(worker) stop_music_if_configured() stop_separator_if_configured() - stop_voice_if_configured() - stop_voice_change_if_configured() + stop_voice_tools() start_container(tts_container()) items = containers() missing = [name for name in ALLOWED if name not in items] @@ -415,6 +498,20 @@ class Handler(BaseHTTPRequestHandler): "unhealthy" if "(unhealthy)" in voice_change_status else "starting" if voice_change.get("State") == "running" else "stopped") + seed_vc = seed_vc_container() if SEED_VC_WORKER else {} + seed_vc_status = seed_vc.get("Status", "") + seed_vc_health = ("disabled" if not SEED_VC_WORKER else + "healthy" if "(healthy)" in seed_vc_status else + "unhealthy" if "(unhealthy)" in seed_vc_status else + "starting" if seed_vc.get("State") == "running" else + "stopped") + applio = applio_container() if APPLIO_WORKER else {} + applio_status = applio.get("Status", "") + applio_health = ("disabled" if not APPLIO_WORKER else + "healthy" if "(healthy)" in applio_status else + "unhealthy" if "(unhealthy)" in applio_status else + "starting" if applio.get("State") == "running" else + "stopped") self.reply(200, {"active_profile": active_profile(items), "music_worker": music.get("State", "disabled"), "music_health": music_health, @@ -424,6 +521,10 @@ class Handler(BaseHTTPRequestHandler): "voice_health": voice_health, "voice_change_worker": voice_change.get("State", "disabled"), "voice_change_health": voice_change_health, + "seed_vc_worker": seed_vc.get("State", "disabled"), + "seed_vc_health": seed_vc_health, + "applio_worker": applio.get("State", "disabled"), + "applio_health": applio_health, "profiles": {name: items.get(name, {}).get( "State", "missing") for name in ALLOWED}}) except Exception as exc: @@ -469,6 +570,20 @@ class Handler(BaseHTTPRequestHandler): log.exception("voice-change worker transition failed") self.reply(503, {"error": str(exc)}) return + if self.path in {"/workers/seed-vc/start", "/workers/seed-vc/stop"}: + try: + self.reply(200, set_seed_vc_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("Seed-VC worker transition failed") + self.reply(503, {"error": str(exc)}) + return + if self.path in {"/workers/applio/start", "/workers/applio/stop"}: + try: + self.reply(200, set_applio_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("Applio worker transition failed") + self.reply(503, {"error": str(exc)}) + return worker_paths = { "/workers/image/start": (IMAGE_WORKER, True), "/workers/image/stop": (IMAGE_WORKER, False), diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh index 9c6e5b6..ad18bd0 100644 --- a/platform/docker/wireguard-gateway/entrypoint.sh +++ b/platform/docker/wireguard-gateway/entrypoint.sh @@ -83,6 +83,8 @@ start_proxy 7862 music-worker:7860 start_proxy 8007 stem-separator:8080 start_proxy 8008 voice-studio:8008 start_proxy 8009 xvc-studio:8009 +start_proxy 8010 seed-vc-studio:7860 +start_proxy 8011 applio-studio:6969 start_proxy 8202 mcp-athena-operator:8000 start_proxy 9443 portainer:9443 diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py index 0a7606b..87e93ee 100644 --- a/platform/llama-dashboard/app.py +++ b/platform/llama-dashboard/app.py @@ -30,6 +30,8 @@ MUSIC_ORIGINAL_UI_URL = os.getenv( SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/") VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/") VOICE_CHANGE_UI_URL = os.getenv("VOICE_CHANGE_UI_URL", "http://192.168.1.212:8009/") +SEED_VC_UI_URL = os.getenv("SEED_VC_UI_URL", "http://192.168.1.212:8010/") +APPLIO_UI_URL = os.getenv("APPLIO_UI_URL", "http://192.168.1.212:8011/") HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) HOST_DATA = os.getenv("HOST_DATA", "/host/data") HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) @@ -279,7 +281,7 @@ def router_status() -> tuple[dict[str, Any], str | None]: def change_mode(mode: str) -> tuple[int, dict[str, Any]]: - if mode not in {"llm", "music", "separation", "voice", "voicechange"}: + if mode not in {"llm", "music", "separation", "voice", "voicechange", "seedvc", "applio"}: return 400, {"error": "invalid mode"} headers = {"Accept": "application/json", "Content-Type": "application/json"} if ROUTER_API_KEY: @@ -637,7 +639,7 @@ HTML = r'''
Mike AI · Live Telemetry

Athena llama.cpp Dashboard

verbinde …
- +
Aktives Profil
–
Router wird abgefragt
Modell
–
–
CPU
–
–
@@ -676,23 +678,21 @@ const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `
GPU ${g.index}
${g.name}
${g.pstate||'–'}
${pct(g.gpu_percent)}GPU-Kern
${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiBVRAM
${g.temperature_c??'–'} °CTemperatur
${g.power_w??'–'} / ${g.power_limit_w??'–'} WLeistung
${g.graphics_clock_mhz??'–'} MHzGrafiktakt
${g.memory_clock_mhz??'–'} MHzSpeichertakt
${pct(g.memory_controller_percent)}Memory Controller
${pct(g.fan_percent)}Lüfter
GPU-Auslastung${load.toFixed(1)} %
VRAM-Belegung${p.toFixed(1)} %
${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei
`} const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit'); let modeBusy=false; -async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}} +async function setMode(mode){if(modeBusy)return;modeBusy=true;for(const id of ['llmMode','musicMode','separationMode','voiceMode','voiceChangeMode','seedVcMode','applioMode'])$(id).disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}} async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'
Keine GPU-Daten verfügbar
';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}${p.name}${p.pid}${p.memory_mib??'–'} MiB`).join('')||'Keine Compute-Prozesse gemeldet';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`
${v??'–'}${k}
`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);