From 040a2df48b77f6a6b25f3de7cfddb066dc35beb2 Mon Sep 17 00:00:00 2001
From: Mikei386 <44135113+Mikei386@users.noreply.github.com>
Date: Sun, 13 Sep 2026 18:50:34 +0200
Subject: [PATCH] Synchronize Athena operating modes with live deployment
---
compose.yaml | 184 ++---
.../profile-controller/profile_controller.py | 495 ++++++++++++-
.../docker/wireguard-gateway/entrypoint.sh | 12 +-
platform/llama-dashboard/app.py | 171 ++++-
router/ai_profile_router.py | 685 +++++++++++++++++-
5 files changed, 1349 insertions(+), 198 deletions(-)
diff --git a/compose.yaml b/compose.yaml
index 3a99093..538478a 100644
--- a/compose.yaml
+++ b/compose.yaml
@@ -50,11 +50,6 @@ services:
- /tmp:size=16m,mode=1777
volumes:
- "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro"
- # Namespace-sharing services cannot publish ports themselves. The owner
- # must keep these bindings so a gateway recreation cannot hide their UIs.
- ports:
- - "8099:8099"
- - "9443:9443"
networks:
frontend:
ipv4_address: 172.30.10.254
@@ -237,89 +232,6 @@ services:
- --spec-draft-p-min
- "0.05"
- # Beta 1 keeps the complete GSQ-RCO text model and KV cache on the RTX 5080.
- # Only the multimodal projector runs on the RTX 3060. The measured hard
- # boundary is 196608 tokens; 192K deliberately retains runtime headroom.
- llama-beta1:
- <<: *llama-common
- container_name: mike-ai-llama-beta1
- labels:
- com.mike-ai.llama-profile: beta1
- environment:
- NVIDIA_VISIBLE_DEVICES: ${BETA1_GPU_DEVICES:-0,1}
- NVIDIA_DRIVER_CAPABILITIES: compute,utility
- MTMD_BACKEND_DEVICE: CUDA1
- command:
- - --model
- - "/models/${BETA1_MODEL_FILE:?BETA1_MODEL_FILE is required}"
- - --mmproj
- - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- - --mmproj-offload
- - --mmproj-device
- - CUDA1
- - --alias
- - qwen-beta-1
- - --ctx-size
- - "${BETA1_CONTEXT:-192000}"
- - --flash-attn
- - "on"
- - --cache-type-k
- - q4_0
- - --cache-type-v
- - q4_0
- - --cache-prompt
- - --cache-ram
- - "${LLAMA_CACHE_RAM_MIB:-32768}"
- - --threads
- - "${LLAMA_THREADS:-6}"
- - --threads-batch
- - "${LLAMA_THREADS_BATCH:-6}"
- - --batch-size
- - "${BETA1_BATCH_SIZE:-2048}"
- - --ubatch-size
- - "${BETA1_UBATCH_SIZE:-128}"
- - --parallel
- - "${BETA1_PARALLEL_SLOTS:-1}"
- - --kv-unified
- - --jinja
- - --reasoning
- - auto
- - --reasoning-preserve
- - --host
- - 0.0.0.0
- - --port
- - "8080"
- - --metrics
- - --fit
- - "off"
- - --n-gpu-layers
- - all
- - --load-mode
- - none
- - --no-ui
- - --temperature
- - "1.0"
- - --top-p
- - "0.95"
- - --top-k
- - "20"
- - --device
- - CUDA0
- - --main-gpu
- - "0"
- - --split-mode
- - none
- - --spec-type
- - draft-mtp
- - --spec-draft-n-max
- - "3"
- - --spec-draft-type-k
- - f16
- - --spec-draft-type-v
- - f16
- - --spec-draft-p-min
- - "0.05"
-
llama-large:
<<: *llama-common
container_name: mike-ai-llama-large
@@ -575,8 +487,17 @@ services:
- /var/run/docker.sock:/var/run/docker.sock
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
- ALLOWED_PROFILES: fast,medium,beta1,large,ultra,uncensored
+ ALLOWED_PROFILES: fast,medium,large,ultra,uncensored
IMAGE_WORKER: image
+ RESTORE_WORKER: restore
+ TTS_WORKER: qwen3
+ MUSIC_WORKER: acestep
+ YUE2_WORKER: yue2
+ SEPARATOR_WORKER: bs-roformer
+ VOICE_WORKER: vevo2
+ VOICE_CHANGE_WORKER: xvc
+ APPLIO_WORKER: applio
+ TRELLIS_WORKER: trellis2-q8
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
@@ -612,6 +533,8 @@ services:
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
SWITCH_TIMEOUT: "600"
REQUEST_TIMEOUT: "600"
+ YUE2_START_TIMEOUT: "600"
+ TRELLIS_START_TIMEOUT: "900"
# Last-resort guard for every OpenAI-compatible client. Without a
# request limit llama.cpp uses n_predict=-1 and a reasoning loop can
# consume the complete context before yielding visible output.
@@ -621,19 +544,21 @@ services:
IMAGE_DIR: /data/images
IMAGE_WORKER_URL: http://image-worker:8086
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
- IMAGE_MODEL_NAME: FLUX.2-klein-4B
+ IMAGE_MODEL_NAME: FLUX.2-klein-9B-fp8-beta
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "true"
ENABLE_TTS: "true"
- # Stable OpenAI compatibility names remain piper/alloy because an
- # existing Open WebUI database persists those values. The gateway maps
- # alloy to XTTS speaker Annmarie Nele and automatically falls back to
- # Piper if XTTS is unavailable, busy or returns an error.
+ # The gateway keeps text normalization, output conversion and native
+ # PCM streaming in one stable API in front of Qwen3-TTS.
TTS_WORKER_URL: http://tts-gateway:8085
- TTS_MODEL: piper
+ TTS_MODEL: qwen3-tts
TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy
ENABLE_STT: "true"
+ ENABLE_MUSIC_MODE: "true"
+ MUSIC_START_TIMEOUT: "600"
+ VOICE_CHANGE_START_TIMEOUT: "600"
+ APPLIO_START_TIMEOUT: "900"
STT_WORKER_URL: http://whisper:8084
STT_TIMEOUT: "300"
networks: [frontend, control, inference]
@@ -655,8 +580,6 @@ services:
condition: service_healthy
profile-controller:
condition: service_healthy
- piper:
- condition: service_healthy
tts-gateway:
condition: service_healthy
whisper:
@@ -680,13 +603,15 @@ services:
read_only: true
tmpfs: ["/tmp:size=1g,mode=1777"]
volumes:
- - "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro"
+ - "${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}:/models/components:ro"
+ - "${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}:/models/fp8:ro"
- router-images:/data/images
environment:
- NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1}
+ NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
- FLUX_MODEL_DIR: /models/FLUX.2-klein-4B
+ FLUX_COMPONENT_DIR: /models/components
+ FLUX_TRANSFORMER_FILE: /models/fp8/flux-2-klein-9b-fp8.safetensors
IMAGE_DIR: /data/images
networks: [inference]
security_opt: ["no-new-privileges:true"]
@@ -697,41 +622,12 @@ services:
timeout: 3s
retries: 12
- piper:
- build:
- context: platform/docker/piper
- args:
- PIPER_TTS_VERSION: ${PIPER_TTS_VERSION:-1.6.0}
- image: mike-ai/piper:local
- container_name: mike-ai-piper
- restart: unless-stopped
- read_only: true
- tmpfs:
- - /tmp:size=256m,mode=1777
- volumes:
- - piper-data:/data
- environment:
- PIPER_DATA_DIR: /data
- PIPER_VOICE: ${PIPER_VOICE:-de_DE-thorsten-high}
- PIPER_VOICE_ALIAS: alloy
- PIPER_HOST: 0.0.0.0
- PIPER_PORT: "8085"
- PIPER_MAX_TEXT_CHARS: "8000"
- networks: [frontend]
- security_opt: ["no-new-privileges:true"]
- cap_drop: [ALL]
- cap_add: [CHOWN, SETUID, SETGID]
- healthcheck:
- test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
- interval: 10s
- timeout: 5s
- retries: 30
- start_period: 120s
-
qwen3-tts:
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
container_name: mike-ai-qwen3-tts
restart: unless-stopped
+ labels:
+ com.mike-ai.tts-worker: qwen3
deploy:
resources:
reservations:
@@ -783,17 +679,12 @@ services:
QWEN_TTS_VOICE: serena
QWEN_TTS_LANGUAGE: German
QWEN_TTS_TIMEOUT: "120"
- PIPER_URL: http://piper:8085
TTS_VOICE_ALIAS: alloy
TTS_DEFAULT_LANGUAGE: de
# Mixed-language clip stitching caused long pauses and unintelligible
# transitions. Keep full sentences in one stable German voice.
TTS_CODE_SWITCH_ENABLED: "false"
- PIPER_TIMEOUT: "120"
networks: [frontend]
- depends_on:
- piper:
- condition: service_healthy
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
@@ -843,7 +734,7 @@ services:
image: mike-ai/llama-dashboard:local
container_name: mike-ai-llama-dashboard
restart: unless-stopped
- network_mode: "service:wireguard-gateway"
+ networks: [frontend]
gpus: all
read_only: true
tmpfs:
@@ -852,15 +743,26 @@ services:
- /proc:/host/proc:ro
- /data:/host/data:ro
- /data/models:/host/models:ro
+ - /data/emergency-backups:/backups:ro
- /data/llama-dashboard:/var/lib/llama-dashboard
environment:
DASHBOARD_HOST: 0.0.0.0
DASHBOARD_PORT: "8099"
ROUTER_URL: http://router:8081
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
+ MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
+ MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
+ SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
+ VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
+ VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}"
+ APPLIO_UI_URL: "${APPLIO_UI_URL:-http://192.168.1.212:8011/}"
+ MIKES_APPLIO_UI_URL: "${MIKES_APPLIO_UI_URL:-http://192.168.1.212:8012/}"
+ TRELLIS_UI_URL: "${TRELLIS_UI_URL:-http://192.168.1.212:8013/}"
+ YUE2_UI_URL: "${YUE2_UI_URL:-http://192.168.1.212:8014/}"
HOST_PROC: /host/proc
HOST_DATA: /host/data
HOST_MODELS: /host/models
+ DASHBOARD_BACKUP_DIR: /backups
DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3
DASHBOARD_HISTORY_INTERVAL: "15"
DASHBOARD_DETAIL_RETENTION_DAYS: "21"
@@ -884,7 +786,7 @@ services:
image: ${PORTAINER_IMAGE:-portainer/portainer-ce@sha256:511f3f06c96fe3b993ebeaafde311c1959cae73a7ef825dba6397d51b450dffa}
container_name: mike-ai-portainer
restart: unless-stopped
- network_mode: "service:wireguard-gateway"
+ networks: [frontend]
command: [--no-setup-token]
volumes:
- /var/run/docker.sock:/var/run/docker.sock
@@ -908,8 +810,9 @@ services:
- /var/run/docker.sock:/var/run/docker.sock:ro
- /data/docker-backups:/archive
- /etc/mike-ai:/backup/etc-mike-ai:ro
- - /opt/mike-ai/stack:/backup/stack:ro
- - piper-data:/backup/volumes/piper-data:ro
+ # Include every deployed specialist UI/worker source tree, not just the
+ # core checkout. Images themselves remain reproducible and are rebuilt.
+ - /opt/mike-ai:/backup/opt-mike-ai:ro
- router-state:/backup/volumes/router-state:ro
- router-images:/backup/volumes/router-images:ro
- portainer-data:/backup/volumes/portainer-data:ro
@@ -936,7 +839,6 @@ networks:
name: mike-ai-tools-egress
volumes:
- piper-data:
whisper-data:
router-state:
router-images:
diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py
index 2a4efab..4fb3660 100644
--- a/platform/docker/profile-controller/profile_controller.py
+++ b/platform/docker/profile-controller/profile_controller.py
@@ -13,6 +13,7 @@ import logging
import os
import socket
import threading
+import time
import urllib.parse
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
@@ -25,6 +26,22 @@ ALLOWED = tuple(x.strip() for x in os.environ.get(
LABEL_KEY = "com.mike-ai.llama-profile"
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
+RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore")
+TTS_LABEL_KEY = "com.mike-ai.tts-worker"
+TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3")
+MUSIC_LABEL_KEY = "com.mike-ai.music-worker"
+MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip()
+YUE2_WORKER = os.environ.get("YUE2_WORKER", "").strip()
+SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator"
+SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
+VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
+VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
+VOICE_CHANGE_LABEL_KEY = "com.mike-ai.voice-change-worker"
+VOICE_CHANGE_WORKER = os.environ.get("VOICE_CHANGE_WORKER", "").strip()
+APPLIO_LABEL_KEY = "com.mike-ai.applio-worker"
+APPLIO_WORKER = os.environ.get("APPLIO_WORKER", "").strip()
+TRELLIS_LABEL_KEY = "com.mike-ai.trellis-worker"
+TRELLIS_WORKER = os.environ.get("TRELLIS_WORKER", "").strip()
LOCK = threading.Lock()
log = logging.getLogger("profile-controller")
@@ -68,15 +85,152 @@ def labelled_containers(label: str) -> list[dict]:
return json.loads(body)
-def image_container() -> dict:
+def image_container(kind: str = IMAGE_WORKER) -> dict:
matches = [item for item in labelled_containers(IMAGE_LABEL_KEY)
- if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == IMAGE_WORKER]
+ if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == kind]
if len(matches) != 1:
raise RuntimeError(
- f"expected exactly one image worker {IMAGE_WORKER!r}, found {len(matches)}")
+ f"expected exactly one image worker {kind!r}, found {len(matches)}")
return matches[0]
+def image_containers() -> list[dict]:
+ """All allowlisted GPU workers that must never overlap an LLM."""
+ allowed = {IMAGE_WORKER, RESTORE_WORKER}
+ return [item for item in labelled_containers(IMAGE_LABEL_KEY)
+ if item.get("Labels", {}).get(IMAGE_LABEL_KEY) in allowed]
+
+
+def tts_container() -> dict:
+ matches = [item for item in labelled_containers(TTS_LABEL_KEY)
+ if item.get("Labels", {}).get(TTS_LABEL_KEY) == TTS_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one TTS worker {TTS_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def music_container() -> dict:
+ if not MUSIC_WORKER:
+ raise RuntimeError("music worker is not configured")
+ matches = [item for item in labelled_containers(MUSIC_LABEL_KEY)
+ if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == MUSIC_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one music worker {MUSIC_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def yue2_container() -> dict:
+ if not YUE2_WORKER:
+ raise RuntimeError("YuE2 worker is not configured")
+ matches = [item for item in labelled_containers(MUSIC_LABEL_KEY)
+ if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == YUE2_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one YuE2 worker {YUE2_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def separator_container() -> dict:
+ if not SEPARATOR_WORKER:
+ raise RuntimeError("stem separator is not configured")
+ matches = [item for item in labelled_containers(SEPARATOR_LABEL_KEY)
+ if item.get("Labels", {}).get(SEPARATOR_LABEL_KEY) == SEPARATOR_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one stem separator {SEPARATOR_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def voice_container() -> dict:
+ if not VOICE_WORKER:
+ raise RuntimeError("voice worker is not configured")
+ matches = [item for item in labelled_containers(VOICE_LABEL_KEY)
+ if item.get("Labels", {}).get(VOICE_LABEL_KEY) == VOICE_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one voice worker {VOICE_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def voice_change_container() -> dict:
+ if not VOICE_CHANGE_WORKER:
+ raise RuntimeError("voice-change worker is not configured")
+ matches = [item for item in labelled_containers(VOICE_CHANGE_LABEL_KEY)
+ if item.get("Labels", {}).get(VOICE_CHANGE_LABEL_KEY) == VOICE_CHANGE_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one voice-change worker {VOICE_CHANGE_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def applio_container() -> dict:
+ if not APPLIO_WORKER:
+ raise RuntimeError("Applio worker is not configured")
+ matches = [item for item in labelled_containers(APPLIO_LABEL_KEY)
+ if item.get("Labels", {}).get(APPLIO_LABEL_KEY) == APPLIO_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one Applio worker {APPLIO_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def trellis_container() -> dict:
+ if not TRELLIS_WORKER:
+ raise RuntimeError("TRELLIS worker is not configured")
+ matches = [item for item in labelled_containers(TRELLIS_LABEL_KEY)
+ if item.get("Labels", {}).get(TRELLIS_LABEL_KEY) == TRELLIS_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one TRELLIS worker {TRELLIS_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
+def stop_music_if_configured() -> None:
+ if MUSIC_WORKER:
+ stop_container(music_container(), timeout=30)
+
+
+def stop_yue2_if_configured() -> None:
+ if YUE2_WORKER:
+ stop_container(yue2_container(), timeout=30)
+
+
+def stop_separator_if_configured() -> None:
+ if SEPARATOR_WORKER:
+ stop_container(separator_container(), timeout=30)
+
+
+def stop_voice_if_configured() -> None:
+ if VOICE_WORKER:
+ stop_container(voice_container(), timeout=30)
+
+
+def stop_voice_change_if_configured() -> None:
+ if VOICE_CHANGE_WORKER:
+ stop_container(voice_change_container(), timeout=30)
+
+
+def stop_applio_if_configured() -> None:
+ if APPLIO_WORKER:
+ stop_container(applio_container(), timeout=30)
+
+
+def stop_trellis_if_configured() -> None:
+ if TRELLIS_WORKER:
+ stop_container(trellis_container(), timeout=30)
+
+
+def stop_voice_tools(except_kind: str | None = None) -> None:
+ if except_kind != "voice":
+ stop_voice_if_configured()
+ if except_kind != "voicechange":
+ stop_voice_change_if_configured()
+ if except_kind != "applio":
+ stop_applio_if_configured()
+
+
def stop_container(item: dict, timeout: int = 120) -> None:
if item.get("State") != "running":
return
@@ -85,6 +239,31 @@ def stop_container(item: dict, timeout: int = 120) -> None:
raise RuntimeError(f"failed to stop container: HTTP {status}")
+def start_container(item: dict) -> None:
+ status, _ = docker_request("POST", f"/containers/{item['Id']}/start")
+ if status not in (204, 304):
+ raise RuntimeError(f"failed to start container: HTTP {status}")
+
+
+def wait_container_healthy(item: dict, timeout: int) -> None:
+ deadline = time.monotonic() + timeout
+ while time.monotonic() < deadline:
+ status, body = docker_request("GET", f"/containers/{item['Id']}/json")
+ if status != 200:
+ raise RuntimeError(f"failed to inspect container: HTTP {status}")
+ state = json.loads(body).get("State", {})
+ health = state.get("Health", {}).get("Status")
+ if state.get("Status") == "running" and health == "healthy":
+ return
+ if state.get("Status") in {"dead", "exited"} or health == "unhealthy":
+ raise RuntimeError(
+ f"container {item.get('Names', ['unknown'])[0]} failed: "
+ f"state={state.get('Status')} health={health}")
+ time.sleep(1)
+ raise RuntimeError(
+ f"container {item.get('Names', ['unknown'])[0]} did not become healthy")
+
+
def stop_inference() -> dict:
with LOCK:
items = containers()
@@ -94,22 +273,186 @@ def stop_inference() -> dict:
return {"active_profile": None, "previous_profile": previous}
-def set_image_worker(running: bool) -> dict:
+def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
+ if kind not in {IMAGE_WORKER, RESTORE_WORKER}:
+ raise ValueError("worker is not allowlisted")
with LOCK:
- item = image_container()
+ item = image_container(kind)
if running:
# The image worker may never overlap a llama profile on the 5080.
for profile_item in containers().values():
stop_container(profile_item)
- if item.get("State") != "running":
- status, _ = docker_request("POST", f"/containers/{item['Id']}/start")
- if status not in (204, 304):
- raise RuntimeError(f"failed to start image worker: HTTP {status}")
+ # The 9B beta text encoder temporarily borrows the RTX 3060 from
+ # Qwen3-TTS. TTS is unavailable during this exclusive GPU phase.
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_yue2_if_configured()
+ stop_separator_if_configured()
+ stop_voice_tools()
+ stop_trellis_if_configured()
+ for other in image_containers():
+ if other["Id"] != item["Id"]:
+ stop_container(other, timeout=20)
+ start_container(item)
else:
# CUDA/PyTorch may not react promptly to SIGTERM after an OOM.
# Bound recovery time and let Docker issue SIGKILL afterwards.
stop_container(item, timeout=20)
- return {"image_worker": "running" if running else "stopped"}
+ # TTS is restored by the following profile activation. Keeping it
+ # stopped here lets the router verify that both GPUs really
+ # released the image model before Qwen and TTS are reloaded.
+ return {"image_worker": kind,
+ "state": "running" if running else "stopped"}
+
+
+def set_music_worker(running: bool) -> dict:
+ """Start ACE-Step exclusively, or stop it before LLM restoration."""
+ with LOCK:
+ item = music_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_separator_if_configured()
+ stop_voice_tools()
+ stop_yue2_if_configured()
+ stop_trellis_if_configured()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"music_worker": MUSIC_WORKER,
+ "state": "running" if running else "stopped"}
+
+
+def set_yue2_worker(running: bool) -> dict:
+ """Start YuE2 exclusively, or stop it before another mode is loaded."""
+ with LOCK:
+ item = yue2_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_separator_if_configured()
+ stop_voice_tools()
+ stop_trellis_if_configured()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"yue2_worker": YUE2_WORKER,
+ "state": "running" if running else "stopped"}
+
+
+def set_separator_worker(running: bool) -> dict:
+ """Start vocal separation exclusively, or stop it before LLM restoration."""
+ with LOCK:
+ item = separator_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_yue2_if_configured()
+ stop_voice_tools()
+ stop_trellis_if_configured()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"separator_worker": SEPARATOR_WORKER,
+ "state": "running" if running else "stopped"}
+
+
+def set_voice_worker(running: bool) -> dict:
+ """Start OmniVoice exclusively, or stop it before LLM restoration."""
+ with LOCK:
+ item = voice_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_yue2_if_configured()
+ stop_separator_if_configured()
+ stop_voice_tools("voice")
+ stop_trellis_if_configured()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"voice_worker": VOICE_WORKER,
+ "state": "running" if running else "stopped"}
+
+
+def set_voice_change_worker(running: bool) -> dict:
+ """Start X-VC exclusively, or stop it before another mode is loaded."""
+ with LOCK:
+ item = voice_change_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_yue2_if_configured()
+ stop_separator_if_configured()
+ stop_voice_tools("voicechange")
+ stop_trellis_if_configured()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"voice_change_worker": VOICE_CHANGE_WORKER,
+ "state": "running" if running else "stopped"}
+
+
+def set_applio_worker(running: bool) -> dict:
+ """Start Applio exclusively, or stop it before another mode is loaded."""
+ with LOCK:
+ item = applio_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_yue2_if_configured()
+ stop_separator_if_configured()
+ stop_voice_tools("applio")
+ stop_trellis_if_configured()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"applio_worker": APPLIO_WORKER,
+ "state": "running" if running else "stopped"}
+
+
+def set_trellis_worker(running: bool) -> dict:
+ """Start TRELLIS.2 exclusively, or stop it before another mode is loaded."""
+ with LOCK:
+ item = trellis_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_yue2_if_configured()
+ stop_separator_if_configured()
+ stop_voice_tools()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"trellis_worker": TRELLIS_WORKER,
+ "state": "running" if running else "stopped"}
def active_profile(items: dict[str, dict] | None = None) -> str | None:
@@ -125,7 +468,14 @@ def activate(profile: str) -> dict:
raise ValueError("profile is not allowlisted")
with LOCK:
# Defensive mutual exclusion even if a caller bypasses the router.
- stop_container(image_container())
+ for worker in image_containers():
+ stop_container(worker)
+ stop_music_if_configured()
+ stop_yue2_if_configured()
+ stop_separator_if_configured()
+ stop_voice_tools()
+ stop_trellis_if_configured()
+ start_container(tts_container())
items = containers()
missing = [name for name in ALLOWED if name not in items]
if missing:
@@ -189,7 +539,70 @@ class Handler(BaseHTTPRequestHandler):
return
try:
items = containers()
+ music = music_container() if MUSIC_WORKER else {}
+ music_status = music.get("Status", "")
+ music_health = ("disabled" if not MUSIC_WORKER else
+ "healthy" if "(healthy)" in music_status else
+ "unhealthy" if "(unhealthy)" in music_status else
+ "starting" if music.get("State") == "running" else
+ "stopped")
+ yue2 = yue2_container() if YUE2_WORKER else {}
+ yue2_status = yue2.get("Status", "")
+ yue2_health = ("disabled" if not YUE2_WORKER else
+ "healthy" if "(healthy)" in yue2_status else
+ "unhealthy" if "(unhealthy)" in yue2_status else
+ "starting" if yue2.get("State") == "running" else
+ "stopped")
+ separator = separator_container() if SEPARATOR_WORKER else {}
+ separator_status = separator.get("Status", "")
+ separator_health = ("disabled" if not SEPARATOR_WORKER else
+ "healthy" if "(healthy)" in separator_status else
+ "unhealthy" if "(unhealthy)" in separator_status else
+ "starting" if separator.get("State") == "running" else
+ "stopped")
+ voice = voice_container() if VOICE_WORKER else {}
+ voice_status = voice.get("Status", "")
+ voice_health = ("disabled" if not VOICE_WORKER else
+ "healthy" if "(healthy)" in voice_status else
+ "unhealthy" if "(unhealthy)" in voice_status else
+ "starting" if voice.get("State") == "running" else
+ "stopped")
+ voice_change = voice_change_container() if VOICE_CHANGE_WORKER else {}
+ voice_change_status = voice_change.get("Status", "")
+ voice_change_health = ("disabled" if not VOICE_CHANGE_WORKER else
+ "healthy" if "(healthy)" in voice_change_status else
+ "unhealthy" if "(unhealthy)" in voice_change_status else
+ "starting" if voice_change.get("State") == "running" else
+ "stopped")
+ applio = applio_container() if APPLIO_WORKER else {}
+ applio_status = applio.get("Status", "")
+ applio_health = ("disabled" if not APPLIO_WORKER else
+ "healthy" if "(healthy)" in applio_status else
+ "unhealthy" if "(unhealthy)" in applio_status else
+ "starting" if applio.get("State") == "running" else
+ "stopped")
+ trellis = trellis_container() if TRELLIS_WORKER else {}
+ trellis_status = trellis.get("Status", "")
+ trellis_health = ("disabled" if not TRELLIS_WORKER else
+ "healthy" if "(healthy)" in trellis_status else
+ "unhealthy" if "(unhealthy)" in trellis_status else
+ "starting" if trellis.get("State") == "running" else
+ "stopped")
self.reply(200, {"active_profile": active_profile(items),
+ "music_worker": music.get("State", "disabled"),
+ "music_health": music_health,
+ "yue2_worker": yue2.get("State", "disabled"),
+ "yue2_health": yue2_health,
+ "separator_worker": separator.get("State", "disabled"),
+ "separator_health": separator_health,
+ "voice_worker": voice.get("State", "disabled"),
+ "voice_health": voice_health,
+ "voice_change_worker": voice_change.get("State", "disabled"),
+ "voice_change_health": voice_change_health,
+ "applio_worker": applio.get("State", "disabled"),
+ "applio_health": applio_health,
+ "trellis_worker": trellis.get("State", "disabled"),
+ "trellis_health": trellis_health,
"profiles": {name: items.get(name, {}).get(
"State", "missing") for name in ALLOWED}})
except Exception as exc:
@@ -207,9 +620,65 @@ class Handler(BaseHTTPRequestHandler):
log.exception("stopping inference failed")
self.reply(503, {"error": str(exc)})
return
- if self.path in ("/workers/image/start", "/workers/image/stop"):
+ if self.path in {"/workers/music/start", "/workers/music/stop"}:
try:
- self.reply(200, set_image_worker(self.path.endswith("/start")))
+ self.reply(200, set_music_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("music worker transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
+ if self.path in {"/workers/yue2/start", "/workers/yue2/stop"}:
+ try:
+ self.reply(200, set_yue2_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("YuE2 worker transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
+ if self.path in {"/workers/separator/start", "/workers/separator/stop"}:
+ try:
+ self.reply(200, set_separator_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("stem separator transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
+ if self.path in {"/workers/voice/start", "/workers/voice/stop"}:
+ try:
+ self.reply(200, set_voice_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("voice worker transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
+ if self.path in {"/workers/voice-change/start", "/workers/voice-change/stop"}:
+ try:
+ self.reply(200, set_voice_change_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("voice-change worker transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
+ if self.path in {"/workers/applio/start", "/workers/applio/stop"}:
+ try:
+ self.reply(200, set_applio_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("Applio worker transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
+ if self.path in {"/workers/trellis/start", "/workers/trellis/stop"}:
+ try:
+ self.reply(200, set_trellis_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("TRELLIS worker transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
+ worker_paths = {
+ "/workers/image/start": (IMAGE_WORKER, True),
+ "/workers/image/stop": (IMAGE_WORKER, False),
+ "/workers/restore/start": (RESTORE_WORKER, True),
+ "/workers/restore/stop": (RESTORE_WORKER, False),
+ }
+ if self.path in worker_paths:
+ try:
+ kind, running = worker_paths[self.path]
+ self.reply(200, set_image_worker(running, kind))
except Exception as exc:
log.exception("image worker transition failed")
self.reply(503, {"error": str(exc)})
diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh
index 0dabbb5..34c2043 100644
--- a/platform/docker/wireguard-gateway/entrypoint.sh
+++ b/platform/docker/wireguard-gateway/entrypoint.sh
@@ -76,7 +76,17 @@ start_proxy() {
start_proxy 22 172.30.10.1:22
start_proxy 8081 router:8081
start_proxy 8085 tts-gateway:8085
-start_proxy 8091 piper:8085
+start_proxy 8099 llama-dashboard:8099
+start_proxy 7861 music-ui:3000
+start_proxy 7862 music-worker:7860
+start_proxy 8007 stem-separator:8080
+start_proxy 8008 voice-studio:8008
+start_proxy 8009 xvc-studio:8009
+start_proxy 8011 applio-studio:6969
+start_proxy 8012 mikes-applio-ui:8012
+start_proxy 8013 trellis-studio:8080
+start_proxy 8014 yue2-studio:8014
start_proxy 8202 mcp-athena-operator:8000
+start_proxy 9443 portainer:9443
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')
diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py
index 4026ef6..2f5a925 100644
--- a/platform/llama-dashboard/app.py
+++ b/platform/llama-dashboard/app.py
@@ -20,15 +20,59 @@ HOST = os.getenv("DASHBOARD_HOST", "0.0.0.0")
PORT = int(os.getenv("DASHBOARD_PORT", "8099"))
ROUTER_URL = os.getenv("ROUTER_URL", "http://router:8081").rstrip("/")
ROUTER_API_KEY = os.getenv("ROUTER_API_KEY", "")
+MUSIC_COMMUNITY_UI_URL = os.getenv(
+ "MUSIC_COMMUNITY_UI_URL",
+ os.getenv("MUSIC_UI_URL", "http://192.168.1.212:7861/"),
+)
+MUSIC_ORIGINAL_UI_URL = os.getenv(
+ "MUSIC_ORIGINAL_UI_URL", "http://192.168.1.212:7862/"
+)
+SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
+VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
+VOICE_CHANGE_UI_URL = os.getenv("VOICE_CHANGE_UI_URL", "http://192.168.1.212:8009/")
+APPLIO_UI_URL = os.getenv("APPLIO_UI_URL", "http://192.168.1.212:8011/")
+MIKES_APPLIO_UI_URL = os.getenv(
+ "MIKES_APPLIO_UI_URL", "http://192.168.1.212:8012/"
+)
+TRELLIS_UI_URL = os.getenv("TRELLIS_UI_URL", "http://192.168.1.212:8013/")
+YUE2_UI_URL = os.getenv("YUE2_UI_URL", "http://192.168.1.212:8014/")
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
+BACKUP_DIR = Path(os.getenv("DASHBOARD_BACKUP_DIR", "/host/data/emergency-backups"))
STARTED = time.time()
HISTORY_DB = Path(os.getenv("DASHBOARD_HISTORY_DB", "/var/lib/llama-dashboard/history.sqlite3"))
HISTORY_INTERVAL = max(5, int(os.getenv("DASHBOARD_HISTORY_INTERVAL", "15")))
DETAIL_RETENTION_DAYS = max(1, int(os.getenv("DASHBOARD_DETAIL_RETENTION_DAYS", "21")))
+def backup_inventory() -> list[dict[str, Any]]:
+ try:
+ candidates = sorted(
+ BACKUP_DIR.glob("athena-portable-*.tar.zst.age"),
+ key=lambda item: item.stat().st_mtime,
+ reverse=True,
+ )[:5]
+ except OSError:
+ return []
+ result = []
+ for path in candidates:
+ try:
+ stat = path.stat()
+ checksum_file = path.with_name(path.name + ".sha256")
+ checksum = _read_text(checksum_file).split(maxsplit=1)[0]
+ result.append({
+ "name": path.name,
+ "size": stat.st_size,
+ "modified": stat.st_mtime,
+ "sha256": checksum if len(checksum) == 64 else None,
+ "download_url": "/api/backups/download/" + urllib.parse.quote(path.name),
+ })
+ except OSError:
+ continue
+ return result
+
+
def _number(value: str) -> int | float | None:
value = value.strip()
if not value or value.lower() in {"n/a", "[n/a]", "not supported"}:
@@ -268,6 +312,28 @@ def router_status() -> tuple[dict[str, Any], str | None]:
return {}, str(exc)
+def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
+ if mode not in {"llm", "music", "yue2", "separation", "voice",
+ "voicechange", "applio", "trellis"}:
+ return 400, {"error": "invalid mode"}
+ headers = {"Accept": "application/json", "Content-Type": "application/json"}
+ if ROUTER_API_KEY:
+ headers["Authorization"] = f"Bearer {ROUTER_API_KEY}"
+ request = urllib.request.Request(
+ f"{ROUTER_URL}/mode", data=json.dumps({"mode": mode}).encode(),
+ headers=headers, method="POST")
+ try:
+ with urllib.request.urlopen(request, timeout=15) as response:
+ return response.status, json.load(response)
+ except urllib.error.HTTPError as exc:
+ try:
+ return exc.code, json.loads(exc.read())
+ except (ValueError, json.JSONDecodeError):
+ return exc.code, {"error": str(exc)}
+ except (OSError, urllib.error.URLError) as exc:
+ return 503, {"error": str(exc)}
+
+
_MODEL_LOCK = threading.Lock()
_MODEL_AT = 0.0
_MODEL_CACHE: tuple[list[dict[str, Any]], dict[str, Any]] = ([], {"count": 0, "total_size": 0})
@@ -602,9 +668,12 @@ HTML = r'''
.span4 .metrics{grid-template-columns:repeat(2,1fr)}
.history-controls{display:flex;flex-wrap:wrap;gap:7px;margin:10px 0 14px}.history-controls button{border:1px solid var(--line);background:#09111b;color:var(--muted);padding:6px 10px;border-radius:8px;cursor:pointer}.history-controls button.active{color:var(--cyan);border-color:var(--cyan)}.chart-legend{display:flex;flex-wrap:wrap;gap:8px;margin:2px 0 10px}.chart-legend button{border:1px solid var(--series);background:#09111b;color:var(--text);padding:6px 10px;border-radius:8px;cursor:pointer}.chart-legend button::before{content:'';display:inline-block;width:10px;height:3px;background:var(--series);margin:0 7px 3px 0}.chart-legend button.off{opacity:.4;text-decoration:line-through}.chart{width:100%;height:250px;display:block}.token-total{font-size:24px;font-weight:750;margin-top:5px}.history-note{color:var(--muted);font-size:11px;margin-top:8px}
.usage-list{display:grid;gap:13px;margin-top:8px}.usage-head{display:flex;justify-content:space-between;gap:14px;align-items:baseline}.usage-head b{font-size:16px}.usage-head span{color:var(--muted)}.usage-meta{display:flex;justify-content:space-between;gap:12px;color:var(--muted);font-size:11px;margin-top:5px}
+.mode-row{display:flex;align-items:center;justify-content:space-between;gap:18px;flex-wrap:wrap}.mode-buttons,.mode-buttons span{display:flex;gap:9px;flex-wrap:wrap}.mode-buttons button,.mode-buttons a{border:1px solid var(--line);background:#09111b;color:var(--text);padding:9px 14px;border-radius:9px;cursor:pointer;text-decoration:none;font:inherit}.mode-buttons button.active{border-color:var(--cyan);color:var(--cyan);box-shadow:inset 0 0 0 1px #45d7ff33}.mode-buttons button:disabled{opacity:.45;cursor:wait}.mode-buttons span[hidden]{display:none}.mode-buttons a.stable{border-color:#66e3a466;color:var(--green)}.mode-buttons a.experimental{border-color:#ffc65c66;color:var(--amber)}.mode-error{color:var(--red)}
+.download{display:inline-block;border:1px solid #66e3a466;color:var(--green);padding:6px 10px;border-radius:8px;text-decoration:none}.hash{font:11px ui-monospace,SFMono-Regular,monospace;color:var(--muted);word-break:break-all}
Mike AI · Live Telemetry
Athena llama.cpp Dashboard
verbinde …
+ Athena Betriebsmodus
–
Status wird geladen …
Aktives Profil
–
Router wird abgefragt
Modell
–
–
CPU
–
–
@@ -636,14 +705,28 @@ HTML = r'''
Ereignisse seit Dashboard-Start
Noch keine Zustandsänderung
llama.cpp Laufzeitkonfiguration
+ Notfall-Backups · herunterladen
+ Verschlüsselte portable Sicherungen
Unersetzliche Daten ohne erneut ladbare Modellgewichte · maximal fünf Generationen
Zum Wiederherstellen wird der separat verwahrte Age-Schlüssel benötigt.
-
+