diff --git a/compose.yaml b/compose.yaml index 3a99093..538478a 100644 --- a/compose.yaml +++ b/compose.yaml @@ -50,11 +50,6 @@ services: - /tmp:size=16m,mode=1777 volumes: - "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro" - # Namespace-sharing services cannot publish ports themselves. The owner - # must keep these bindings so a gateway recreation cannot hide their UIs. - ports: - - "8099:8099" - - "9443:9443" networks: frontend: ipv4_address: 172.30.10.254 @@ -237,89 +232,6 @@ services: - --spec-draft-p-min - "0.05" - # Beta 1 keeps the complete GSQ-RCO text model and KV cache on the RTX 5080. - # Only the multimodal projector runs on the RTX 3060. The measured hard - # boundary is 196608 tokens; 192K deliberately retains runtime headroom. - llama-beta1: - <<: *llama-common - container_name: mike-ai-llama-beta1 - labels: - com.mike-ai.llama-profile: beta1 - environment: - NVIDIA_VISIBLE_DEVICES: ${BETA1_GPU_DEVICES:-0,1} - NVIDIA_DRIVER_CAPABILITIES: compute,utility - MTMD_BACKEND_DEVICE: CUDA1 - command: - - --model - - "/models/${BETA1_MODEL_FILE:?BETA1_MODEL_FILE is required}" - - --mmproj - - "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}" - - --mmproj-offload - - --mmproj-device - - CUDA1 - - --alias - - qwen-beta-1 - - --ctx-size - - "${BETA1_CONTEXT:-192000}" - - --flash-attn - - "on" - - --cache-type-k - - q4_0 - - --cache-type-v - - q4_0 - - --cache-prompt - - --cache-ram - - "${LLAMA_CACHE_RAM_MIB:-32768}" - - --threads - - "${LLAMA_THREADS:-6}" - - --threads-batch - - "${LLAMA_THREADS_BATCH:-6}" - - --batch-size - - "${BETA1_BATCH_SIZE:-2048}" - - --ubatch-size - - "${BETA1_UBATCH_SIZE:-128}" - - --parallel - - "${BETA1_PARALLEL_SLOTS:-1}" - - --kv-unified - - --jinja - - --reasoning - - auto - - --reasoning-preserve - - --host - - 0.0.0.0 - - --port - - "8080" - - --metrics - - --fit - - "off" - - --n-gpu-layers - - all - - --load-mode - - none - - --no-ui - - --temperature - - "1.0" - - --top-p - - "0.95" - - --top-k - - "20" - - --device - - CUDA0 - - --main-gpu - - "0" - - --split-mode - - none - - --spec-type - - draft-mtp - - --spec-draft-n-max - - "3" - - --spec-draft-type-k - - f16 - - --spec-draft-type-v - - f16 - - --spec-draft-p-min - - "0.05" - llama-large: <<: *llama-common container_name: mike-ai-llama-large @@ -575,8 +487,17 @@ services: - /var/run/docker.sock:/var/run/docker.sock environment: CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" - ALLOWED_PROFILES: fast,medium,beta1,large,ultra,uncensored + ALLOWED_PROFILES: fast,medium,large,ultra,uncensored IMAGE_WORKER: image + RESTORE_WORKER: restore + TTS_WORKER: qwen3 + MUSIC_WORKER: acestep + YUE2_WORKER: yue2 + SEPARATOR_WORKER: bs-roformer + VOICE_WORKER: vevo2 + VOICE_CHANGE_WORKER: xvc + APPLIO_WORKER: applio + TRELLIS_WORKER: trellis2-q8 networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: @@ -612,6 +533,8 @@ services: PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" SWITCH_TIMEOUT: "600" REQUEST_TIMEOUT: "600" + YUE2_START_TIMEOUT: "600" + TRELLIS_START_TIMEOUT: "900" # Last-resort guard for every OpenAI-compatible client. Without a # request limit llama.cpp uses n_predict=-1 and a reasoning loop can # consume the complete context before yielding visible output. @@ -621,19 +544,21 @@ services: IMAGE_DIR: /data/images IMAGE_WORKER_URL: http://image-worker:8086 IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" - IMAGE_MODEL_NAME: FLUX.2-klein-4B + IMAGE_MODEL_NAME: FLUX.2-klein-9B-fp8-beta CHAT_IMAGE_ALLOW_REMOTE_URLS: "false" ENABLE_IMAGE_GENERATION: "true" ENABLE_TTS: "true" - # Stable OpenAI compatibility names remain piper/alloy because an - # existing Open WebUI database persists those values. The gateway maps - # alloy to XTTS speaker Annmarie Nele and automatically falls back to - # Piper if XTTS is unavailable, busy or returns an error. + # The gateway keeps text normalization, output conversion and native + # PCM streaming in one stable API in front of Qwen3-TTS. TTS_WORKER_URL: http://tts-gateway:8085 - TTS_MODEL: piper + TTS_MODEL: qwen3-tts TTS_VOICES: alloy TTS_DEFAULT_VOICE: alloy ENABLE_STT: "true" + ENABLE_MUSIC_MODE: "true" + MUSIC_START_TIMEOUT: "600" + VOICE_CHANGE_START_TIMEOUT: "600" + APPLIO_START_TIMEOUT: "900" STT_WORKER_URL: http://whisper:8084 STT_TIMEOUT: "300" networks: [frontend, control, inference] @@ -655,8 +580,6 @@ services: condition: service_healthy profile-controller: condition: service_healthy - piper: - condition: service_healthy tts-gateway: condition: service_healthy whisper: @@ -680,13 +603,15 @@ services: read_only: true tmpfs: ["/tmp:size=1g,mode=1777"] volumes: - - "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro" + - "${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}:/models/components:ro" + - "${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}:/models/fp8:ro" - router-images:/data/images environment: - NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1} + NVIDIA_VISIBLE_DEVICES: all NVIDIA_DRIVER_CAPABILITIES: compute,utility WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" - FLUX_MODEL_DIR: /models/FLUX.2-klein-4B + FLUX_COMPONENT_DIR: /models/components + FLUX_TRANSFORMER_FILE: /models/fp8/flux-2-klein-9b-fp8.safetensors IMAGE_DIR: /data/images networks: [inference] security_opt: ["no-new-privileges:true"] @@ -697,41 +622,12 @@ services: timeout: 3s retries: 12 - piper: - build: - context: platform/docker/piper - args: - PIPER_TTS_VERSION: ${PIPER_TTS_VERSION:-1.6.0} - image: mike-ai/piper:local - container_name: mike-ai-piper - restart: unless-stopped - read_only: true - tmpfs: - - /tmp:size=256m,mode=1777 - volumes: - - piper-data:/data - environment: - PIPER_DATA_DIR: /data - PIPER_VOICE: ${PIPER_VOICE:-de_DE-thorsten-high} - PIPER_VOICE_ALIAS: alloy - PIPER_HOST: 0.0.0.0 - PIPER_PORT: "8085" - PIPER_MAX_TEXT_CHARS: "8000" - networks: [frontend] - security_opt: ["no-new-privileges:true"] - cap_drop: [ALL] - cap_add: [CHOWN, SETUID, SETGID] - healthcheck: - test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"] - interval: 10s - timeout: 5s - retries: 30 - start_period: 120s - qwen3-tts: image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98} container_name: mike-ai-qwen3-tts restart: unless-stopped + labels: + com.mike-ai.tts-worker: qwen3 deploy: resources: reservations: @@ -783,17 +679,12 @@ services: QWEN_TTS_VOICE: serena QWEN_TTS_LANGUAGE: German QWEN_TTS_TIMEOUT: "120" - PIPER_URL: http://piper:8085 TTS_VOICE_ALIAS: alloy TTS_DEFAULT_LANGUAGE: de # Mixed-language clip stitching caused long pauses and unintelligible # transitions. Keep full sentences in one stable German voice. TTS_CODE_SWITCH_ENABLED: "false" - PIPER_TIMEOUT: "120" networks: [frontend] - depends_on: - piper: - condition: service_healthy security_opt: ["no-new-privileges:true"] cap_drop: [ALL] healthcheck: @@ -843,7 +734,7 @@ services: image: mike-ai/llama-dashboard:local container_name: mike-ai-llama-dashboard restart: unless-stopped - network_mode: "service:wireguard-gateway" + networks: [frontend] gpus: all read_only: true tmpfs: @@ -852,15 +743,26 @@ services: - /proc:/host/proc:ro - /data:/host/data:ro - /data/models:/host/models:ro + - /data/emergency-backups:/backups:ro - /data/llama-dashboard:/var/lib/llama-dashboard environment: DASHBOARD_HOST: 0.0.0.0 DASHBOARD_PORT: "8099" ROUTER_URL: http://router:8081 ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" + MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}" + MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}" + SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}" + VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}" + VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}" + APPLIO_UI_URL: "${APPLIO_UI_URL:-http://192.168.1.212:8011/}" + MIKES_APPLIO_UI_URL: "${MIKES_APPLIO_UI_URL:-http://192.168.1.212:8012/}" + TRELLIS_UI_URL: "${TRELLIS_UI_URL:-http://192.168.1.212:8013/}" + YUE2_UI_URL: "${YUE2_UI_URL:-http://192.168.1.212:8014/}" HOST_PROC: /host/proc HOST_DATA: /host/data HOST_MODELS: /host/models + DASHBOARD_BACKUP_DIR: /backups DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3 DASHBOARD_HISTORY_INTERVAL: "15" DASHBOARD_DETAIL_RETENTION_DAYS: "21" @@ -884,7 +786,7 @@ services: image: ${PORTAINER_IMAGE:-portainer/portainer-ce@sha256:511f3f06c96fe3b993ebeaafde311c1959cae73a7ef825dba6397d51b450dffa} container_name: mike-ai-portainer restart: unless-stopped - network_mode: "service:wireguard-gateway" + networks: [frontend] command: [--no-setup-token] volumes: - /var/run/docker.sock:/var/run/docker.sock @@ -908,8 +810,9 @@ services: - /var/run/docker.sock:/var/run/docker.sock:ro - /data/docker-backups:/archive - /etc/mike-ai:/backup/etc-mike-ai:ro - - /opt/mike-ai/stack:/backup/stack:ro - - piper-data:/backup/volumes/piper-data:ro + # Include every deployed specialist UI/worker source tree, not just the + # core checkout. Images themselves remain reproducible and are rebuilt. + - /opt/mike-ai:/backup/opt-mike-ai:ro - router-state:/backup/volumes/router-state:ro - router-images:/backup/volumes/router-images:ro - portainer-data:/backup/volumes/portainer-data:ro @@ -936,7 +839,6 @@ networks: name: mike-ai-tools-egress volumes: - piper-data: whisper-data: router-state: router-images: diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index 2a4efab..4fb3660 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -13,6 +13,7 @@ import logging import os import socket import threading +import time import urllib.parse from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer @@ -25,6 +26,22 @@ ALLOWED = tuple(x.strip() for x in os.environ.get( LABEL_KEY = "com.mike-ai.llama-profile" IMAGE_LABEL_KEY = "com.mike-ai.image-worker" IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image") +RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore") +TTS_LABEL_KEY = "com.mike-ai.tts-worker" +TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3") +MUSIC_LABEL_KEY = "com.mike-ai.music-worker" +MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip() +YUE2_WORKER = os.environ.get("YUE2_WORKER", "").strip() +SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator" +SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip() +VOICE_LABEL_KEY = "com.mike-ai.voice-worker" +VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip() +VOICE_CHANGE_LABEL_KEY = "com.mike-ai.voice-change-worker" +VOICE_CHANGE_WORKER = os.environ.get("VOICE_CHANGE_WORKER", "").strip() +APPLIO_LABEL_KEY = "com.mike-ai.applio-worker" +APPLIO_WORKER = os.environ.get("APPLIO_WORKER", "").strip() +TRELLIS_LABEL_KEY = "com.mike-ai.trellis-worker" +TRELLIS_WORKER = os.environ.get("TRELLIS_WORKER", "").strip() LOCK = threading.Lock() log = logging.getLogger("profile-controller") @@ -68,15 +85,152 @@ def labelled_containers(label: str) -> list[dict]: return json.loads(body) -def image_container() -> dict: +def image_container(kind: str = IMAGE_WORKER) -> dict: matches = [item for item in labelled_containers(IMAGE_LABEL_KEY) - if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == IMAGE_WORKER] + if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == kind] if len(matches) != 1: raise RuntimeError( - f"expected exactly one image worker {IMAGE_WORKER!r}, found {len(matches)}") + f"expected exactly one image worker {kind!r}, found {len(matches)}") return matches[0] +def image_containers() -> list[dict]: + """All allowlisted GPU workers that must never overlap an LLM.""" + allowed = {IMAGE_WORKER, RESTORE_WORKER} + return [item for item in labelled_containers(IMAGE_LABEL_KEY) + if item.get("Labels", {}).get(IMAGE_LABEL_KEY) in allowed] + + +def tts_container() -> dict: + matches = [item for item in labelled_containers(TTS_LABEL_KEY) + if item.get("Labels", {}).get(TTS_LABEL_KEY) == TTS_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one TTS worker {TTS_WORKER!r}, found {len(matches)}") + return matches[0] + + +def music_container() -> dict: + if not MUSIC_WORKER: + raise RuntimeError("music worker is not configured") + matches = [item for item in labelled_containers(MUSIC_LABEL_KEY) + if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == MUSIC_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one music worker {MUSIC_WORKER!r}, found {len(matches)}") + return matches[0] + + +def yue2_container() -> dict: + if not YUE2_WORKER: + raise RuntimeError("YuE2 worker is not configured") + matches = [item for item in labelled_containers(MUSIC_LABEL_KEY) + if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == YUE2_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one YuE2 worker {YUE2_WORKER!r}, found {len(matches)}") + return matches[0] + + +def separator_container() -> dict: + if not SEPARATOR_WORKER: + raise RuntimeError("stem separator is not configured") + matches = [item for item in labelled_containers(SEPARATOR_LABEL_KEY) + if item.get("Labels", {}).get(SEPARATOR_LABEL_KEY) == SEPARATOR_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one stem separator {SEPARATOR_WORKER!r}, found {len(matches)}") + return matches[0] + + +def voice_container() -> dict: + if not VOICE_WORKER: + raise RuntimeError("voice worker is not configured") + matches = [item for item in labelled_containers(VOICE_LABEL_KEY) + if item.get("Labels", {}).get(VOICE_LABEL_KEY) == VOICE_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one voice worker {VOICE_WORKER!r}, found {len(matches)}") + return matches[0] + + +def voice_change_container() -> dict: + if not VOICE_CHANGE_WORKER: + raise RuntimeError("voice-change worker is not configured") + matches = [item for item in labelled_containers(VOICE_CHANGE_LABEL_KEY) + if item.get("Labels", {}).get(VOICE_CHANGE_LABEL_KEY) == VOICE_CHANGE_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one voice-change worker {VOICE_CHANGE_WORKER!r}, found {len(matches)}") + return matches[0] + + +def applio_container() -> dict: + if not APPLIO_WORKER: + raise RuntimeError("Applio worker is not configured") + matches = [item for item in labelled_containers(APPLIO_LABEL_KEY) + if item.get("Labels", {}).get(APPLIO_LABEL_KEY) == APPLIO_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one Applio worker {APPLIO_WORKER!r}, found {len(matches)}") + return matches[0] + + +def trellis_container() -> dict: + if not TRELLIS_WORKER: + raise RuntimeError("TRELLIS worker is not configured") + matches = [item for item in labelled_containers(TRELLIS_LABEL_KEY) + if item.get("Labels", {}).get(TRELLIS_LABEL_KEY) == TRELLIS_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one TRELLIS worker {TRELLIS_WORKER!r}, found {len(matches)}") + return matches[0] + + +def stop_music_if_configured() -> None: + if MUSIC_WORKER: + stop_container(music_container(), timeout=30) + + +def stop_yue2_if_configured() -> None: + if YUE2_WORKER: + stop_container(yue2_container(), timeout=30) + + +def stop_separator_if_configured() -> None: + if SEPARATOR_WORKER: + stop_container(separator_container(), timeout=30) + + +def stop_voice_if_configured() -> None: + if VOICE_WORKER: + stop_container(voice_container(), timeout=30) + + +def stop_voice_change_if_configured() -> None: + if VOICE_CHANGE_WORKER: + stop_container(voice_change_container(), timeout=30) + + +def stop_applio_if_configured() -> None: + if APPLIO_WORKER: + stop_container(applio_container(), timeout=30) + + +def stop_trellis_if_configured() -> None: + if TRELLIS_WORKER: + stop_container(trellis_container(), timeout=30) + + +def stop_voice_tools(except_kind: str | None = None) -> None: + if except_kind != "voice": + stop_voice_if_configured() + if except_kind != "voicechange": + stop_voice_change_if_configured() + if except_kind != "applio": + stop_applio_if_configured() + + def stop_container(item: dict, timeout: int = 120) -> None: if item.get("State") != "running": return @@ -85,6 +239,31 @@ def stop_container(item: dict, timeout: int = 120) -> None: raise RuntimeError(f"failed to stop container: HTTP {status}") +def start_container(item: dict) -> None: + status, _ = docker_request("POST", f"/containers/{item['Id']}/start") + if status not in (204, 304): + raise RuntimeError(f"failed to start container: HTTP {status}") + + +def wait_container_healthy(item: dict, timeout: int) -> None: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status, body = docker_request("GET", f"/containers/{item['Id']}/json") + if status != 200: + raise RuntimeError(f"failed to inspect container: HTTP {status}") + state = json.loads(body).get("State", {}) + health = state.get("Health", {}).get("Status") + if state.get("Status") == "running" and health == "healthy": + return + if state.get("Status") in {"dead", "exited"} or health == "unhealthy": + raise RuntimeError( + f"container {item.get('Names', ['unknown'])[0]} failed: " + f"state={state.get('Status')} health={health}") + time.sleep(1) + raise RuntimeError( + f"container {item.get('Names', ['unknown'])[0]} did not become healthy") + + def stop_inference() -> dict: with LOCK: items = containers() @@ -94,22 +273,186 @@ def stop_inference() -> dict: return {"active_profile": None, "previous_profile": previous} -def set_image_worker(running: bool) -> dict: +def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: + if kind not in {IMAGE_WORKER, RESTORE_WORKER}: + raise ValueError("worker is not allowlisted") with LOCK: - item = image_container() + item = image_container(kind) if running: # The image worker may never overlap a llama profile on the 5080. for profile_item in containers().values(): stop_container(profile_item) - if item.get("State") != "running": - status, _ = docker_request("POST", f"/containers/{item['Id']}/start") - if status not in (204, 304): - raise RuntimeError(f"failed to start image worker: HTTP {status}") + # The 9B beta text encoder temporarily borrows the RTX 3060 from + # Qwen3-TTS. TTS is unavailable during this exclusive GPU phase. + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_yue2_if_configured() + stop_separator_if_configured() + stop_voice_tools() + stop_trellis_if_configured() + for other in image_containers(): + if other["Id"] != item["Id"]: + stop_container(other, timeout=20) + start_container(item) else: # CUDA/PyTorch may not react promptly to SIGTERM after an OOM. # Bound recovery time and let Docker issue SIGKILL afterwards. stop_container(item, timeout=20) - return {"image_worker": "running" if running else "stopped"} + # TTS is restored by the following profile activation. Keeping it + # stopped here lets the router verify that both GPUs really + # released the image model before Qwen and TTS are reloaded. + return {"image_worker": kind, + "state": "running" if running else "stopped"} + + +def set_music_worker(running: bool) -> dict: + """Start ACE-Step exclusively, or stop it before LLM restoration.""" + with LOCK: + item = music_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_separator_if_configured() + stop_voice_tools() + stop_yue2_if_configured() + stop_trellis_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"music_worker": MUSIC_WORKER, + "state": "running" if running else "stopped"} + + +def set_yue2_worker(running: bool) -> dict: + """Start YuE2 exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = yue2_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_separator_if_configured() + stop_voice_tools() + stop_trellis_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"yue2_worker": YUE2_WORKER, + "state": "running" if running else "stopped"} + + +def set_separator_worker(running: bool) -> dict: + """Start vocal separation exclusively, or stop it before LLM restoration.""" + with LOCK: + item = separator_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_yue2_if_configured() + stop_voice_tools() + stop_trellis_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"separator_worker": SEPARATOR_WORKER, + "state": "running" if running else "stopped"} + + +def set_voice_worker(running: bool) -> dict: + """Start OmniVoice exclusively, or stop it before LLM restoration.""" + with LOCK: + item = voice_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_yue2_if_configured() + stop_separator_if_configured() + stop_voice_tools("voice") + stop_trellis_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"voice_worker": VOICE_WORKER, + "state": "running" if running else "stopped"} + + +def set_voice_change_worker(running: bool) -> dict: + """Start X-VC exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = voice_change_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_yue2_if_configured() + stop_separator_if_configured() + stop_voice_tools("voicechange") + stop_trellis_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"voice_change_worker": VOICE_CHANGE_WORKER, + "state": "running" if running else "stopped"} + + +def set_applio_worker(running: bool) -> dict: + """Start Applio exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = applio_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_yue2_if_configured() + stop_separator_if_configured() + stop_voice_tools("applio") + stop_trellis_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"applio_worker": APPLIO_WORKER, + "state": "running" if running else "stopped"} + + +def set_trellis_worker(running: bool) -> dict: + """Start TRELLIS.2 exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = trellis_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_yue2_if_configured() + stop_separator_if_configured() + stop_voice_tools() + start_container(item) + else: + stop_container(item, timeout=30) + return {"trellis_worker": TRELLIS_WORKER, + "state": "running" if running else "stopped"} def active_profile(items: dict[str, dict] | None = None) -> str | None: @@ -125,7 +468,14 @@ def activate(profile: str) -> dict: raise ValueError("profile is not allowlisted") with LOCK: # Defensive mutual exclusion even if a caller bypasses the router. - stop_container(image_container()) + for worker in image_containers(): + stop_container(worker) + stop_music_if_configured() + stop_yue2_if_configured() + stop_separator_if_configured() + stop_voice_tools() + stop_trellis_if_configured() + start_container(tts_container()) items = containers() missing = [name for name in ALLOWED if name not in items] if missing: @@ -189,7 +539,70 @@ class Handler(BaseHTTPRequestHandler): return try: items = containers() + music = music_container() if MUSIC_WORKER else {} + music_status = music.get("Status", "") + music_health = ("disabled" if not MUSIC_WORKER else + "healthy" if "(healthy)" in music_status else + "unhealthy" if "(unhealthy)" in music_status else + "starting" if music.get("State") == "running" else + "stopped") + yue2 = yue2_container() if YUE2_WORKER else {} + yue2_status = yue2.get("Status", "") + yue2_health = ("disabled" if not YUE2_WORKER else + "healthy" if "(healthy)" in yue2_status else + "unhealthy" if "(unhealthy)" in yue2_status else + "starting" if yue2.get("State") == "running" else + "stopped") + separator = separator_container() if SEPARATOR_WORKER else {} + separator_status = separator.get("Status", "") + separator_health = ("disabled" if not SEPARATOR_WORKER else + "healthy" if "(healthy)" in separator_status else + "unhealthy" if "(unhealthy)" in separator_status else + "starting" if separator.get("State") == "running" else + "stopped") + voice = voice_container() if VOICE_WORKER else {} + voice_status = voice.get("Status", "") + voice_health = ("disabled" if not VOICE_WORKER else + "healthy" if "(healthy)" in voice_status else + "unhealthy" if "(unhealthy)" in voice_status else + "starting" if voice.get("State") == "running" else + "stopped") + voice_change = voice_change_container() if VOICE_CHANGE_WORKER else {} + voice_change_status = voice_change.get("Status", "") + voice_change_health = ("disabled" if not VOICE_CHANGE_WORKER else + "healthy" if "(healthy)" in voice_change_status else + "unhealthy" if "(unhealthy)" in voice_change_status else + "starting" if voice_change.get("State") == "running" else + "stopped") + applio = applio_container() if APPLIO_WORKER else {} + applio_status = applio.get("Status", "") + applio_health = ("disabled" if not APPLIO_WORKER else + "healthy" if "(healthy)" in applio_status else + "unhealthy" if "(unhealthy)" in applio_status else + "starting" if applio.get("State") == "running" else + "stopped") + trellis = trellis_container() if TRELLIS_WORKER else {} + trellis_status = trellis.get("Status", "") + trellis_health = ("disabled" if not TRELLIS_WORKER else + "healthy" if "(healthy)" in trellis_status else + "unhealthy" if "(unhealthy)" in trellis_status else + "starting" if trellis.get("State") == "running" else + "stopped") self.reply(200, {"active_profile": active_profile(items), + "music_worker": music.get("State", "disabled"), + "music_health": music_health, + "yue2_worker": yue2.get("State", "disabled"), + "yue2_health": yue2_health, + "separator_worker": separator.get("State", "disabled"), + "separator_health": separator_health, + "voice_worker": voice.get("State", "disabled"), + "voice_health": voice_health, + "voice_change_worker": voice_change.get("State", "disabled"), + "voice_change_health": voice_change_health, + "applio_worker": applio.get("State", "disabled"), + "applio_health": applio_health, + "trellis_worker": trellis.get("State", "disabled"), + "trellis_health": trellis_health, "profiles": {name: items.get(name, {}).get( "State", "missing") for name in ALLOWED}}) except Exception as exc: @@ -207,9 +620,65 @@ class Handler(BaseHTTPRequestHandler): log.exception("stopping inference failed") self.reply(503, {"error": str(exc)}) return - if self.path in ("/workers/image/start", "/workers/image/stop"): + if self.path in {"/workers/music/start", "/workers/music/stop"}: try: - self.reply(200, set_image_worker(self.path.endswith("/start"))) + self.reply(200, set_music_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("music worker transition failed") + self.reply(503, {"error": str(exc)}) + return + if self.path in {"/workers/yue2/start", "/workers/yue2/stop"}: + try: + self.reply(200, set_yue2_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("YuE2 worker transition failed") + self.reply(503, {"error": str(exc)}) + return + if self.path in {"/workers/separator/start", "/workers/separator/stop"}: + try: + self.reply(200, set_separator_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("stem separator transition failed") + self.reply(503, {"error": str(exc)}) + return + if self.path in {"/workers/voice/start", "/workers/voice/stop"}: + try: + self.reply(200, set_voice_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("voice worker transition failed") + self.reply(503, {"error": str(exc)}) + return + if self.path in {"/workers/voice-change/start", "/workers/voice-change/stop"}: + try: + self.reply(200, set_voice_change_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("voice-change worker transition failed") + self.reply(503, {"error": str(exc)}) + return + if self.path in {"/workers/applio/start", "/workers/applio/stop"}: + try: + self.reply(200, set_applio_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("Applio worker transition failed") + self.reply(503, {"error": str(exc)}) + return + if self.path in {"/workers/trellis/start", "/workers/trellis/stop"}: + try: + self.reply(200, set_trellis_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("TRELLIS worker transition failed") + self.reply(503, {"error": str(exc)}) + return + worker_paths = { + "/workers/image/start": (IMAGE_WORKER, True), + "/workers/image/stop": (IMAGE_WORKER, False), + "/workers/restore/start": (RESTORE_WORKER, True), + "/workers/restore/stop": (RESTORE_WORKER, False), + } + if self.path in worker_paths: + try: + kind, running = worker_paths[self.path] + self.reply(200, set_image_worker(running, kind)) except Exception as exc: log.exception("image worker transition failed") self.reply(503, {"error": str(exc)}) diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh index 0dabbb5..34c2043 100644 --- a/platform/docker/wireguard-gateway/entrypoint.sh +++ b/platform/docker/wireguard-gateway/entrypoint.sh @@ -76,7 +76,17 @@ start_proxy() { start_proxy 22 172.30.10.1:22 start_proxy 8081 router:8081 start_proxy 8085 tts-gateway:8085 -start_proxy 8091 piper:8085 +start_proxy 8099 llama-dashboard:8099 +start_proxy 7861 music-ui:3000 +start_proxy 7862 music-worker:7860 +start_proxy 8007 stem-separator:8080 +start_proxy 8008 voice-studio:8008 +start_proxy 8009 xvc-studio:8009 +start_proxy 8011 applio-studio:6969 +start_proxy 8012 mikes-applio-ui:8012 +start_proxy 8013 trellis-studio:8080 +start_proxy 8014 yue2-studio:8014 start_proxy 8202 mcp-athena-operator:8000 +start_proxy 9443 portainer:9443 wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}') diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py index 4026ef6..2f5a925 100644 --- a/platform/llama-dashboard/app.py +++ b/platform/llama-dashboard/app.py @@ -20,15 +20,59 @@ HOST = os.getenv("DASHBOARD_HOST", "0.0.0.0") PORT = int(os.getenv("DASHBOARD_PORT", "8099")) ROUTER_URL = os.getenv("ROUTER_URL", "http://router:8081").rstrip("/") ROUTER_API_KEY = os.getenv("ROUTER_API_KEY", "") +MUSIC_COMMUNITY_UI_URL = os.getenv( + "MUSIC_COMMUNITY_UI_URL", + os.getenv("MUSIC_UI_URL", "http://192.168.1.212:7861/"), +) +MUSIC_ORIGINAL_UI_URL = os.getenv( + "MUSIC_ORIGINAL_UI_URL", "http://192.168.1.212:7862/" +) +SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/") +VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/") +VOICE_CHANGE_UI_URL = os.getenv("VOICE_CHANGE_UI_URL", "http://192.168.1.212:8009/") +APPLIO_UI_URL = os.getenv("APPLIO_UI_URL", "http://192.168.1.212:8011/") +MIKES_APPLIO_UI_URL = os.getenv( + "MIKES_APPLIO_UI_URL", "http://192.168.1.212:8012/" +) +TRELLIS_UI_URL = os.getenv("TRELLIS_UI_URL", "http://192.168.1.212:8013/") +YUE2_UI_URL = os.getenv("YUE2_UI_URL", "http://192.168.1.212:8014/") HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) HOST_DATA = os.getenv("HOST_DATA", "/host/data") HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) +BACKUP_DIR = Path(os.getenv("DASHBOARD_BACKUP_DIR", "/host/data/emergency-backups")) STARTED = time.time() HISTORY_DB = Path(os.getenv("DASHBOARD_HISTORY_DB", "/var/lib/llama-dashboard/history.sqlite3")) HISTORY_INTERVAL = max(5, int(os.getenv("DASHBOARD_HISTORY_INTERVAL", "15"))) DETAIL_RETENTION_DAYS = max(1, int(os.getenv("DASHBOARD_DETAIL_RETENTION_DAYS", "21"))) +def backup_inventory() -> list[dict[str, Any]]: + try: + candidates = sorted( + BACKUP_DIR.glob("athena-portable-*.tar.zst.age"), + key=lambda item: item.stat().st_mtime, + reverse=True, + )[:5] + except OSError: + return [] + result = [] + for path in candidates: + try: + stat = path.stat() + checksum_file = path.with_name(path.name + ".sha256") + checksum = _read_text(checksum_file).split(maxsplit=1)[0] + result.append({ + "name": path.name, + "size": stat.st_size, + "modified": stat.st_mtime, + "sha256": checksum if len(checksum) == 64 else None, + "download_url": "/api/backups/download/" + urllib.parse.quote(path.name), + }) + except OSError: + continue + return result + + def _number(value: str) -> int | float | None: value = value.strip() if not value or value.lower() in {"n/a", "[n/a]", "not supported"}: @@ -268,6 +312,28 @@ def router_status() -> tuple[dict[str, Any], str | None]: return {}, str(exc) +def change_mode(mode: str) -> tuple[int, dict[str, Any]]: + if mode not in {"llm", "music", "yue2", "separation", "voice", + "voicechange", "applio", "trellis"}: + return 400, {"error": "invalid mode"} + headers = {"Accept": "application/json", "Content-Type": "application/json"} + if ROUTER_API_KEY: + headers["Authorization"] = f"Bearer {ROUTER_API_KEY}" + request = urllib.request.Request( + f"{ROUTER_URL}/mode", data=json.dumps({"mode": mode}).encode(), + headers=headers, method="POST") + try: + with urllib.request.urlopen(request, timeout=15) as response: + return response.status, json.load(response) + except urllib.error.HTTPError as exc: + try: + return exc.code, json.loads(exc.read()) + except (ValueError, json.JSONDecodeError): + return exc.code, {"error": str(exc)} + except (OSError, urllib.error.URLError) as exc: + return 503, {"error": str(exc)} + + _MODEL_LOCK = threading.Lock() _MODEL_AT = 0.0 _MODEL_CACHE: tuple[list[dict[str, Any]], dict[str, Any]] = ([], {"count": 0, "total_size": 0}) @@ -602,9 +668,12 @@ HTML = r''' .span4 .metrics{grid-template-columns:repeat(2,1fr)} .history-controls{display:flex;flex-wrap:wrap;gap:7px;margin:10px 0 14px}.history-controls button{border:1px solid var(--line);background:#09111b;color:var(--muted);padding:6px 10px;border-radius:8px;cursor:pointer}.history-controls button.active{color:var(--cyan);border-color:var(--cyan)}.chart-legend{display:flex;flex-wrap:wrap;gap:8px;margin:2px 0 10px}.chart-legend button{border:1px solid var(--series);background:#09111b;color:var(--text);padding:6px 10px;border-radius:8px;cursor:pointer}.chart-legend button::before{content:'';display:inline-block;width:10px;height:3px;background:var(--series);margin:0 7px 3px 0}.chart-legend button.off{opacity:.4;text-decoration:line-through}.chart{width:100%;height:250px;display:block}.token-total{font-size:24px;font-weight:750;margin-top:5px}.history-note{color:var(--muted);font-size:11px;margin-top:8px} .usage-list{display:grid;gap:13px;margin-top:8px}.usage-head{display:flex;justify-content:space-between;gap:14px;align-items:baseline}.usage-head b{font-size:16px}.usage-head span{color:var(--muted)}.usage-meta{display:flex;justify-content:space-between;gap:12px;color:var(--muted);font-size:11px;margin-top:5px} +.mode-row{display:flex;align-items:center;justify-content:space-between;gap:18px;flex-wrap:wrap}.mode-buttons,.mode-buttons span{display:flex;gap:9px;flex-wrap:wrap}.mode-buttons button,.mode-buttons a{border:1px solid var(--line);background:#09111b;color:var(--text);padding:9px 14px;border-radius:9px;cursor:pointer;text-decoration:none;font:inherit}.mode-buttons button.active{border-color:var(--cyan);color:var(--cyan);box-shadow:inset 0 0 0 1px #45d7ff33}.mode-buttons button:disabled{opacity:.45;cursor:wait}.mode-buttons span[hidden]{display:none}.mode-buttons a.stable{border-color:#66e3a466;color:var(--green)}.mode-buttons a.experimental{border-color:#ffc65c66;color:var(--amber)}.mode-error{color:var(--red)} +.download{display:inline-block;border:1px solid #66e3a466;color:var(--green);padding:6px 10px;border-radius:8px;text-decoration:none}.hash{font:11px ui-monospace,SFMono-Regular,monospace;color:var(--muted);word-break:break-all}
Mike AI · Live Telemetry

Athena llama.cpp Dashboard

verbinde …
+
Aktives Profil
–
Router wird abgefragt
Modell
–
–
CPU
–
–
@@ -636,14 +705,28 @@ HTML = r'''
Ereignisse seit Dashboard-Start
Noch keine Zustandsänderung
llama.cpp Laufzeitkonfiguration
–wird gelesen
Verfügbare GGUF-Dateien
–
DateiPfadGrößeGeändert
–
+
Notfall-Backups · herunterladen
+
Verschlüsselte portable Sicherungen
Unersetzliche Daten ohne erneut ladbare Modellgewichte · maximal fünf Generationen
ErstelltGrößeSHA256
Backups werden geladen …
Zum Wiederherstellen wird der separat verwahrte Age-Schlüssel benötigt.
-
+
''' +let modeBusy=false; +async function setMode(mode){if(modeBusy)return;modeBusy=true;for(const id of ['llmMode','musicMode','yue2Mode','separationMode','voiceMode','voiceChangeMode','applioMode','trellisMode'])$(id).disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}} +async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=({llm:'LLM-Betrieb',music:'ACE-Step Studio',yue2:'YuE2 Studio',separation:'Stimmtrennung',voice:'Voice Studio',voicechange:'X-VC Voice Changer',applio:'Applio / RVC',trellis:'3D Studio'})[md.active]||'Unbekannt';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`ACE-Step: ${md.music_worker||'–'} · YuE2: ${md.yue2_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'} · 3D: ${md.trellis_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('yue2Mode').classList.toggle('active',md.active==='yue2');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('voiceChangeMode').classList.toggle('active',md.active==='voicechange');$('applioMode').classList.toggle('active',md.active==='applio');$('trellisMode').classList.toggle('active',md.active==='trellis');let modeControlsBusy=modeBusy||switchingMode||!md.enabled;for(const id of ['llmMode','musicMode','yue2Mode','separationMode','voiceMode','voiceChangeMode','applioMode','trellisMode'])$(id).disabled=modeControlsBusy;$('musicOpen').hidden=md.active!=='music';$('yue2Open').hidden=md.active!=='yue2';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('voiceChangeOpen').hidden=md.active!=='voicechange';$('applioOpen').hidden=md.active!=='applio';$('trellisOpen').hidden=md.active!=='trellis';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'
Keine GPU-Daten verfügbar
';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}${p.name}${p.pid}${p.memory_mib??'–'} MiB`).join('')||'Keine Compute-Prozesse gemeldet';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`
${v??'–'}${k}
`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000); +async function refreshBackups(){try{let r=await fetch('/api/backups',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),rows=d.backups||[];$('backupFiles').innerHTML=rows.map(b=>`${new Date(b.modified*1000).toLocaleString('de-DE')}${gib(b.size)}${b.sha256||'Prüfsumme fehlt'}Herunterladen`).join('')||'Noch kein portables Backup vorhanden';$('backupNote').textContent=rows.length?`${rows.length} von maximal 5 Generationen · verschlüsselt mit Age`:'Der erste Lauf startet spätestens fünf Stunden nach Aktivierung.'}catch(e){$('backupNote').textContent=`Backup-Liste nicht verfügbar: ${e.message}`}}refreshBackups();setInterval(refreshBackups,60000); +'''.replace( + "__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL +).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL +).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL +).replace("__VOICE_UI_URL__", VOICE_UI_URL +).replace("__VOICE_CHANGE_UI_URL__", VOICE_CHANGE_UI_URL +).replace("__APPLIO_UI_URL__", APPLIO_UI_URL +).replace("__MIKES_APPLIO_UI_URL__", MIKES_APPLIO_UI_URL).replace( + "__TRELLIS_UI_URL__", TRELLIS_UI_URL +).replace("__YUE2_UI_URL__", YUE2_UI_URL) FULL_JS = r''' @@ -764,6 +847,62 @@ class Handler(BaseHTTPRequestHandler): self.end_headers() self.wfile.write(body) + def _send_backup(self, name: str) -> None: + # Generated names are deliberately strict; never expose an arbitrary + # host path through the dashboard. + if not name.startswith("athena-portable-") or not name.endswith(".tar.zst.age"): + self._send(404, b'{"error":"not found"}', "application/json") + return + if Path(name).name != name: + self._send(404, b'{"error":"not found"}', "application/json") + return + path = BACKUP_DIR / name + try: + size = path.stat().st_size + except OSError: + self._send(404, b'{"error":"not found"}', "application/json") + return + + start, end = 0, size - 1 + status = 200 + range_header = self.headers.get("Range", "") + if range_header: + try: + unit, raw = range_header.split("=", 1) + first, last = raw.split("-", 1) + if unit != "bytes" or "," in raw or not first: + raise ValueError + start = int(first) + end = min(size - 1, int(last)) if last else size - 1 + if start < 0 or start > end or start >= size: + raise ValueError + status = 206 + except ValueError: + self.send_response(416) + self.send_header("Content-Range", f"bytes */{size}") + self.end_headers() + return + + self.send_response(status) + self.send_header("Content-Type", "application/octet-stream") + self.send_header("Content-Disposition", f'attachment; filename="{name}"') + self.send_header("Accept-Ranges", "bytes") + self.send_header("Content-Length", str(end - start + 1)) + if status == 206: + self.send_header("Content-Range", f"bytes {start}-{end}/{size}") + self.send_header("Cache-Control", "no-store") + self.send_header("X-Content-Type-Options", "nosniff") + self.end_headers() + remaining = end - start + 1 + with path.open("rb") as source: + source.seek(start) + while remaining: + chunk = source.read(min(1024 * 1024, remaining)) + if not chunk: + break + self.wfile.write(chunk) + remaining -= len(chunk) + def do_GET(self) -> None: path = self.path.split("?", 1)[0] if path == "/": @@ -782,9 +921,35 @@ class Handler(BaseHTTPRequestHandler): range_name = query.get("range", ["24h"])[0] body = json.dumps(HISTORY.query(range_name), ensure_ascii=False, separators=(",", ":")).encode() self._send(200, body, "application/json; charset=utf-8") + elif path == "/api/backups": + body = json.dumps({"backups": backup_inventory()}, ensure_ascii=False, + separators=(",", ":")).encode() + self._send(200, body, "application/json; charset=utf-8") + elif path.startswith("/api/backups/download/"): + name = urllib.parse.unquote(path.removeprefix("/api/backups/download/")) + self._send_backup(name) else: self._send(404, b'{"error":"not found"}', "application/json") + def do_POST(self) -> None: + path = self.path.split("?", 1)[0] + if path != "/api/mode": + self._send(404, b'{"error":"not found"}', "application/json") + return + try: + length = int(self.headers.get("Content-Length", "0")) + if length <= 0 or length > 1024: + raise ValueError("invalid body size") + payload = json.loads(self.rfile.read(length)) + mode = payload.get("mode") if isinstance(payload, dict) else None + except (ValueError, json.JSONDecodeError): + self._send(400, b'{"error":"invalid request"}', "application/json") + return + status, response = change_mode(mode) + body = json.dumps(response, ensure_ascii=False, + separators=(",", ":")).encode() + self._send(status, body, "application/json; charset=utf-8") + if __name__ == "__main__": threading.Thread(target=history_collector, name="history-collector", daemon=True).start() diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index 4277d2d..8b087f7 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -2,31 +2,30 @@ """AI Profile Router – OpenAI-kompatibler Proxy vor llama.cpp. Leitet OpenAI-kompatible Requests transparent an den lokalen llama.cpp-Server -weiter (Streaming, Tool Calls, JSON) und schaltet zwischen sechs festen +weiter (Streaming, Tool Calls, JSON) und schaltet zwischen fünf festen Profilen um: Profil Kontext ------ -------- fast 76800 medium 160000 - beta1 192000 large 192000 ultra 262144 uncensored 80000 -Virtuelle Modelle: qwen-fast, qwen-medium, qwen-beta-1, qwen-large, +Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra, qwen-uncensored -Kommandos: POST /fast, /medium, /beta1, /large, /ultra, +Kommandos: POST /fast, /medium, /large, /ultra, /uncensored GET /status (Zustand) -Bildgenerierung und Editing (FLUX.2-klein-4B): +Bildgenerierung und Editing (FLUX.2 Klein 9B FP8 beta): POST /v1/images/generations (OpenAI-kompatibel) POST /v1/images/edits (lokal, Referenzbilder) GET /images (Liste) GET /images/ (PNG-Download) -Sprachausgabe (XTTS-v2, multilingual, CPU-only): +Sprachausgabe (Qwen3-TTS auf RTX 3060): POST /v1/audio/speech (OpenAI-kompatibel) GET /v1/audio/voices (verfügbare Stimmen) @@ -40,7 +39,8 @@ Der Router leitet /v1/audio/speech und /v1/audio/transcriptions per HTTP an die Worker weiter. Der Router agiert als Modell-Orchestrator: vor der Generierung wird -llama.cpp gestoppt, der Bild-Worker lädt FLUX.2, generiert/bearbeitet und entlädt +llama.cpp und Qwen3-TTS gestoppt, der Bild-Worker lädt FLUX.2 und den +Text-Encoder auf getrennte GPUs, generiert/bearbeitet und entlädt das Modell wieder; danach wird das vorherige Qwen-Profil wiederher- gestellt und erst dann geantwortet (try/finally – Qwen wird auch bei Fehlgeschlagener Generierung wiederhergestellt). @@ -105,6 +105,15 @@ PROFILE_DIR = os.environ.get( PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/") PROFILE_CONTROL_TOKEN_FILE = os.environ.get( "PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token") +ENABLE_MUSIC_MODE = os.environ.get( + "ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"} +MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600")) +YUE2_START_TIMEOUT = float(os.environ.get("YUE2_START_TIMEOUT", "600")) +SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600")) +VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600")) +VOICE_CHANGE_START_TIMEOUT = float(os.environ.get("VOICE_CHANGE_START_TIMEOUT", "600")) +APPLIO_START_TIMEOUT = float(os.environ.get("APPLIO_START_TIMEOUT", "900")) +TRELLIS_START_TIMEOUT = float(os.environ.get("TRELLIS_START_TIMEOUT", "900")) # Optional worker APIs. The clean Docker baseline deliberately ships only # text/multimodal chat; absent workers must fail explicitly instead of trying @@ -126,7 +135,7 @@ DEFAULT_REASONING_EFFORT = os.environ.get( GLOBAL_SYSTEM_POLICY_FILE = os.environ.get( "GLOBAL_SYSTEM_POLICY_FILE", "").strip() -# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 4B) --- +# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 9B FP8) --- LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service") SYSTEMCTL_BIN = os.environ.get("SYSTEMCTL_BIN", "systemctl") IMAGE_WORKER = os.environ.get( @@ -135,7 +144,8 @@ IMAGE_PYTHON = os.environ.get( "IMAGE_PYTHON", "/opt/mike-ai/ai-profile-router/venv/bin/python") IMAGE_WORKER_URL = os.environ.get("IMAGE_WORKER_URL", "").rstrip("/") IMAGE_WORKER_TOKEN = os.environ.get("IMAGE_WORKER_TOKEN", "").strip() -IMAGE_MODEL_NAME = os.environ.get("IMAGE_MODEL_NAME", "FLUX.2-klein-4B") +IMAGE_MODEL_NAME = os.environ.get( + "IMAGE_MODEL_NAME", "FLUX.2-klein-9B-fp8-beta") IMAGE_DIR = os.environ.get( "IMAGE_DIR", "/opt/mike-ai/ai-profile-router/images") IMAGE_WORKER_LOG = os.environ.get( @@ -163,20 +173,20 @@ IMAGE_SIZES = { "1920x1088": (1920, 1088), "1088x1920": (1088, 1920), } -# Das destillierte FLUX.2-klein-4B ist auf vier Schritte ausgelegt. +# Das destillierte FLUX.2 Klein 9B ist auf vier Schritte ausgelegt. IMAGE_QUALITY = {"standard": 4, "high": 4} IMAGE_DEFAULT_QUALITY = "standard" IMAGE_MAX_N = 4 -# --- Sprachausgabe (austauschbarer interner TTS-Worker, CPU-only) --- +# --- Sprachausgabe (Qwen3-TTS über das interne Normalisierungs-Gateway) --- TTS_WORKER_URL = os.environ.get("TTS_WORKER_URL", "http://127.0.0.1:8085") TTS_TIMEOUT = float(os.environ.get("TTS_TIMEOUT", "300")) # s, pro Synthese TTS_CONNECT_TIMEOUT = float(os.environ.get("TTS_CONNECT_TIMEOUT", "5")) -TTS_MODEL = os.environ.get("TTS_MODEL", "xtts-v2") +TTS_MODEL = os.environ.get("TTS_MODEL", "qwen3-tts") TTS_VOICES = tuple(v.strip() for v in os.environ.get( - "TTS_VOICES", "claribel").split(",") if v.strip()) + "TTS_VOICES", "alloy").split(",") if v.strip()) TTS_DEFAULT_VOICE = os.environ.get( - "TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel") + "TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "alloy") TTS_FORMATS = ("mp3", "wav", "pcm") TTS_DEFAULT_FORMAT = "mp3" @@ -254,6 +264,7 @@ class _ImageState: self.last_error: str | None = None self.last_image: str | None = None self.last_seconds: float | None = None + self.current_model: str | None = None IMAGE_PHASES = ( @@ -279,6 +290,9 @@ class _State: self.qwen_unavailable = True self.active_chats = 0 self.avail_lock = threading.Lock() + self.mode = "llm" + self.mode_phase = "ready" + self.mode_error: str | None = None STATE = _State() @@ -309,6 +323,329 @@ def _set_qwen_unavailable(unavailable: bool) -> None: STATE.qwen_unavailable = unavailable +def _music_worker_state() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "music_worker", "missing")) + except Exception as exc: + log.warning("Musik-Worker-Status nicht verfügbar: %s", exc) + return "unknown" + + +def _music_worker_health() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "music_health", "unknown")) + except Exception: + return "unknown" + + +def _separator_worker_state() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "separator_worker", "missing")) + except Exception as exc: + log.warning("Stem-Separator-Status nicht verfügbar: %s", exc) + return "unknown" + + +def _separator_worker_health() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "separator_health", "unknown")) + except Exception: + return "unknown" + + +def _voice_worker_state() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "voice_worker", "missing")) + except Exception as exc: + log.warning("Voice-Worker-Status nicht verfügbar: %s", exc) + return "unknown" + + +def _voice_worker_health() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "voice_health", "unknown")) + except Exception: + return "unknown" + + +def _voice_change_worker_state() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "voice_change_worker", "missing")) + except Exception as exc: + log.warning("Voice-Change-Worker-Status nicht verfügbar: %s", exc) + return "unknown" + + +def _voice_change_worker_health() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "voice_change_health", "unknown")) + except Exception: + return "unknown" + + +def _worker_field(field: str) -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get(field, "missing")) + except Exception as exc: + log.warning("Spezial-Worker-Status %s nicht verfügbar: %s", field, exc) + return "unknown" + + +def _wait_music_ready() -> None: + deadline = time.monotonic() + MUSIC_START_TIMEOUT + while time.monotonic() < deadline: + status = _profile_controller_request("GET", "/status") + if (status.get("music_worker") == "running" + and status.get("music_health") == "healthy"): + return + if status.get("music_health") == "unhealthy": + raise RuntimeError("ACE-Step-Container ist unhealthy") + time.sleep(POLL_INTERVAL) + raise RuntimeError( + f"ACE-Step nach {MUSIC_START_TIMEOUT:.0f} s nicht bereit") + + +def _wait_yue2_ready() -> None: + _wait_aux_voice_ready("yue2_worker", "yue2_health", + "YuE2", YUE2_START_TIMEOUT) + + +def _wait_separator_ready() -> None: + deadline = time.monotonic() + SEPARATOR_START_TIMEOUT + while time.monotonic() < deadline: + status = _profile_controller_request("GET", "/status") + if (status.get("separator_worker") == "running" + and status.get("separator_health") == "healthy"): + return + if status.get("separator_health") == "unhealthy": + raise RuntimeError("BS-RoFormer-Container ist unhealthy") + time.sleep(POLL_INTERVAL) + raise RuntimeError( + f"BS-RoFormer nach {SEPARATOR_START_TIMEOUT:.0f} s nicht bereit") + + +def _wait_voice_ready() -> None: + deadline = time.monotonic() + VOICE_START_TIMEOUT + while time.monotonic() < deadline: + status = _profile_controller_request("GET", "/status") + if (status.get("voice_worker") == "running" + and status.get("voice_health") == "healthy"): + return + if status.get("voice_health") == "unhealthy": + raise RuntimeError("OmniVoice-Container ist unhealthy") + time.sleep(POLL_INTERVAL) + raise RuntimeError( + f"OmniVoice nach {VOICE_START_TIMEOUT:.0f} s nicht bereit") + + +def _wait_voice_change_ready() -> None: + deadline = time.monotonic() + VOICE_CHANGE_START_TIMEOUT + while time.monotonic() < deadline: + status = _profile_controller_request("GET", "/status") + if (status.get("voice_change_worker") == "running" + and status.get("voice_change_health") == "healthy"): + return + if status.get("voice_change_health") == "unhealthy": + raise RuntimeError("X-VC-Container ist unhealthy") + time.sleep(POLL_INTERVAL) + raise RuntimeError( + f"X-VC nach {VOICE_CHANGE_START_TIMEOUT:.0f} s nicht bereit") + + +def _wait_aux_voice_ready(worker_field: str, health_field: str, + label: str, timeout: float) -> None: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + status = _profile_controller_request("GET", "/status") + if (status.get(worker_field) == "running" + and status.get(health_field) == "healthy"): + return + if status.get(health_field) == "unhealthy": + raise RuntimeError(f"{label}-Container ist unhealthy") + time.sleep(POLL_INTERVAL) + raise RuntimeError(f"{label} nach {timeout:.0f} s nicht bereit") + + +def _wait_applio_ready() -> None: + _wait_aux_voice_ready("applio_worker", "applio_health", + "Applio", APPLIO_START_TIMEOUT) + + +def _wait_trellis_ready() -> None: + _wait_aux_voice_ready("trellis_worker", "trellis_health", + "TRELLIS.2", TRELLIS_START_TIMEOUT) + + +def _special_worker(mode: str) -> tuple[str, str, callable]: + if mode == "music": + return "/workers/music/start", _music_worker_state(), _wait_music_ready + if mode == "yue2": + return ("/workers/yue2/start", _worker_field("yue2_worker"), + _wait_yue2_ready) + if mode == "separation": + return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready + if mode == "voice": + return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready + if mode == "voicechange": + return ("/workers/voice-change/start", _voice_change_worker_state(), + _wait_voice_change_ready) + if mode == "applio": + return ("/workers/applio/start", _worker_field("applio_worker"), + _wait_applio_ready) + if mode == "trellis": + return ("/workers/trellis/start", _worker_field("trellis_worker"), + _wait_trellis_ready) + raise ValueError(f"unbekannter Spezialmodus: {mode}") + + +def set_operating_mode(mode: str) -> dict: + """Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen.""" + if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL: + raise RuntimeError("Musikmodus ist nicht konfiguriert") + special_modes = {"music", "yue2", "separation", "voice", "voicechange", + "applio", "trellis"} + if mode not in {"llm", *special_modes}: + raise ValueError("unbekannter Betriebsmodus") + with STATE.lock: + STATE.mode_error = None + if mode in special_modes: + path, worker_state, wait_ready = _special_worker(mode) + if STATE.mode == mode and worker_state == "running": + return {"status": "ok", "mode": mode, "changed": False} + profile = current_profile() + previous = RUNTIME.load() + saved = previous.get("return_profile") or previous.get("last_profile") + return_profile = profile if profile in PROFILES else saved + if return_profile not in PROFILES: + return_profile = next(iter(PROFILES)) + STATE.mode_phase = f"starting-{mode}" + _set_qwen_unavailable(True) + try: + # Persist intent before stopping anything so a router restart + # during ACE-Step loading can resume the same transition. + RUNTIME.save(mode=mode, return_profile=return_profile, + last_profile=return_profile, + phase=f"starting-{mode}") + _wait_chats_drained() + _profile_controller_request( + "POST", path, + timeout=120) + wait_ready() + STATE.mode = mode + STATE.mode_phase = "ready" + RUNTIME.save(mode=mode, return_profile=return_profile, + last_profile=return_profile, phase=mode) + return {"status": "ok", "mode": mode, "changed": True, + "return_profile": return_profile} + except Exception as exc: + STATE.mode_error = str(exc) + STATE.mode_phase = "error" + raise + + previous = RUNTIME.load() + profile = previous.get("return_profile") or previous.get("last_profile") + if profile not in PROFILES: + profile = next(iter(PROFILES)) + STATE.mode_phase = "restoring-llm" + _set_qwen_unavailable(True) + try: + _profile_controller_request("POST", "/workers/music/stop") + _profile_controller_request("POST", "/workers/yue2/stop") + _profile_controller_request("POST", "/workers/separator/stop") + _profile_controller_request("POST", "/workers/voice/stop") + _profile_controller_request("POST", "/workers/voice-change/stop") + _profile_controller_request("POST", "/workers/applio/stop") + _profile_controller_request("POST", "/workers/trellis/stop") + _restore_qwen(profile) + STATE.mode = "llm" + STATE.mode_phase = "ready" + _set_qwen_unavailable(False) + RUNTIME.save(mode="llm", return_profile=None, + last_profile=profile, phase="idle") + return {"status": "ok", "mode": "llm", "changed": True, + "profile": profile} + except Exception as exc: + STATE.mode_error = str(exc) + STATE.mode_phase = "error" + raise + + +def schedule_operating_mode(mode: str) -> tuple[bool, str]: + """Start a transition in the background so chat/UI acknowledgement is instant.""" + if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL: + raise RuntimeError("Musikmodus ist nicht konfiguriert") + if mode not in {"llm", "music", "yue2", "separation", "voice", + "voicechange", "applio", "trellis"}: + raise ValueError("unbekannter Betriebsmodus") + with STATE.lock: + if STATE.mode_phase not in {"ready", "error"}: + return False, STATE.mode_phase + if STATE.mode == mode and STATE.mode_phase == "ready": + return False, "ready" + STATE.mode_phase = f"starting-{mode}" if mode != "llm" else "restoring-llm" + + def transition() -> None: + try: + set_operating_mode(mode) + log.info("Betriebsmodus ist jetzt %s", mode) + except Exception: + log.exception("Betriebsmodus-Wechsel zu %s fehlgeschlagen", mode) + + threading.Thread(target=transition, name=f"mode-{mode}", daemon=True).start() + return True, STATE.mode_phase + + +def _control_command(data: dict, path: str) -> str | None: + """Recognise exact local commands without invoking an LLM.""" + text: object = None + if path == "/v1/chat/completions": + messages = data.get("messages") + if isinstance(messages, list): + for item in reversed(messages): + if isinstance(item, dict) and item.get("role") == "user": + text = item.get("content") + break + elif path == "/v1/responses": + text = data.get("input") + if not isinstance(text, str): + return None + command = text.strip().casefold() + return command if command in {"/athena music", "/athena yue2", + "/athena stems", + "/athena separation", "/athena llm", + "/athena voice", + "/athena voicechange", "/athena changer", + "/athena applio", + "/athena 3d", "/athena trellis", + "/athena status"} else None + + # --------------------------------------------------------------------------- # Upstream (llama.cpp) # --------------------------------------------------------------------------- @@ -588,7 +925,8 @@ def _read(path: str) -> str: return f.read().strip() -def _profile_controller_request(method: str, path: str) -> dict: +def _profile_controller_request(method: str, path: str, + timeout: float = 120) -> dict: token = os.environ.get("PROFILE_CONTROL_TOKEN", "").strip() if not token: token = _read(PROFILE_CONTROL_TOKEN_FILE) @@ -600,7 +938,7 @@ def _profile_controller_request(method: str, path: str) -> dict: headers={"Authorization": f"Bearer {token}"}, ) try: - with urllib.request.urlopen(request, timeout=120) as response: + with urllib.request.urlopen(request, timeout=timeout) as response: return json.load(response) except urllib.error.HTTPError as exc: body = exc.read(500).decode(errors="replace") @@ -750,7 +1088,10 @@ def switch_profile(profile: str, implicit: bool = False) -> None: f"Profildatei wurde nicht gesetzt (erwartet: {profile})") log.info("Warte, bis llama.cpp das Profil geladen hat ...") _wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT) - RUNTIME.save(last_profile=profile, phase="idle") + STATE.mode = "llm" + STATE.mode_phase = "ready" + RUNTIME.save(last_profile=profile, mode="llm", + return_profile=None, phase="idle") finally: # Nach einem fehlgeschlagenen Skript/Timeout darf der Router # Qwen nicht blind freigeben. Nur ein semantisch verifiziertes @@ -765,7 +1106,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None: # --------------------------------------------------------------------------- -# Bildgenerierung und Editing (FLUX.2-klein-4B) +# Bildgenerierung und Editing (FLUX.2 Klein 9B FP8) # --------------------------------------------------------------------------- class _Worker: @@ -861,7 +1202,10 @@ def _worker() -> _Worker: if not img.worker or not img.worker.alive(): if img.worker: img.worker.stop() - img.worker = _RemoteWorker() if IMAGE_WORKER_URL else _Worker() + img.worker = (_RemoteWorker(kind="image", url=IMAGE_WORKER_URL, + token=IMAGE_WORKER_TOKEN, + endpoint="/generate") + if IMAGE_WORKER_URL else _Worker()) img.worker.start() return img.worker @@ -871,7 +1215,12 @@ class _RemoteWorker: model_loaded = False - def __init__(self) -> None: + def __init__(self, *, kind: str, url: str, token: str, + endpoint: str) -> None: + self.kind = kind + self.url = url + self.token = token + self.endpoint = endpoint self.running = False def alive(self) -> bool: @@ -880,10 +1229,10 @@ class _RemoteWorker: def _request(self, method: str, path: str, payload: dict | None = None, timeout: float = 120) -> dict: body = None if payload is None else json.dumps(payload).encode() - headers = {"Authorization": f"Bearer {IMAGE_WORKER_TOKEN}"} + headers = {"Authorization": f"Bearer {self.token}"} if body is not None: headers["Content-Type"] = "application/json" - req = urllib.request.Request(IMAGE_WORKER_URL + path, data=body, + req = urllib.request.Request(self.url + path, data=body, method=method, headers=headers) try: with urllib.request.urlopen(req, timeout=timeout) as response: @@ -898,9 +1247,9 @@ class _RemoteWorker: raise RuntimeError(f"Bild-Worker nicht erreichbar: {exc}") from exc def start(self) -> None: - if not IMAGE_WORKER_TOKEN or len(IMAGE_WORKER_TOKEN) < 32: + if not self.token or len(self.token) < 32: raise RuntimeError("Bild-Worker-Token fehlt oder ist zu kurz") - _profile_controller_request("POST", "/workers/image/start") + _profile_controller_request("POST", f"/workers/{self.kind}/start") deadline = time.monotonic() + IMAGE_START_TIMEOUT while time.monotonic() < deadline: try: @@ -920,11 +1269,11 @@ class _RemoteWorker: clean.pop("cmd", None) output = clean.pop("output", "") clean["filename"] = os.path.basename(output) - return self._request("POST", "/generate", clean, timeout) + return self._request("POST", self.endpoint, clean, timeout) def stop(self) -> None: try: - _profile_controller_request("POST", "/workers/image/stop") + _profile_controller_request("POST", f"/workers/{self.kind}/stop") finally: self.running = False self.model_loaded = False @@ -1011,6 +1360,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int, guidance: float, seed: int | None, n: int, quality: str = "standard", source_files: list[str] | None = None, + model: str = IMAGE_MODEL_NAME, ) -> tuple[list[str], str | None]: """Orchestriert die Bildgenerierung inkl. Qwen-Hotswap. @@ -1030,6 +1380,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int, results: list[str] = [] warning: str | None = None img.last_error = None + img.current_model = model # Qwen wird gestoppt → für Chats nicht verfügbar (die warten). _set_qwen_unavailable(True) try: @@ -1062,7 +1413,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int, filename = time.strftime("%Y%m%d-%H%M%S") + \ f"-{os.urandom(2).hex()}.png" output = os.path.join(IMAGE_DIR, filename) - resp = worker.request({ + worker_payload = { "cmd": "generate", "prompt": prompt, "width": width, @@ -1072,7 +1423,8 @@ def generate_image(prompt: str, width: int, height: int, steps: int, "seed": seed, "output": output, "source_files": source_files or [], - }, timeout=IMAGE_GEN_TIMEOUT) + } + resp = worker.request(worker_payload, timeout=IMAGE_GEN_TIMEOUT) if resp.get("status") != "ok": raise RuntimeError( resp.get("message", "Bildgenerierung fehlgeschlagen")) @@ -1090,10 +1442,10 @@ def generate_image(prompt: str, width: int, height: int, steps: int, "steps": steps, "guidance": guidance, "quality": quality, - "mode": "image-edit" if source_files else "text-to-image", + "mode": ("image-edit" if source_files else "text-to-image"), "reference_images": len(source_files or []), "seconds": resp.get("seconds"), - "model": IMAGE_MODEL_NAME, + "model": model, "created": time.strftime("%Y-%m-%dT%H:%M:%S"), } meta_path = os.path.join(IMAGE_DIR, filename[:-4] + ".json") @@ -1139,6 +1491,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int, log.error(warning) # Qwen ist down → qwen_unavailable bleibt True. img.phase = "idle" + img.current_model = None return results, warning @@ -1497,6 +1850,10 @@ class Handler(BaseHTTPRequestHandler): self._send_json(200, self._models_payload()) elif path == "/status" and self.command == "GET": self._send_json(200, self._status_payload()) + elif path == "/mode" and self.command == "GET": + self._send_json(200, self._mode_payload()) + elif path == "/mode" and self.command == "POST": + self._mode_change() elif path == "/v1/audio/models" and self.command == "GET": self._send_json(200, self._audio_models_payload()) elif path == "/v1/audio/voices" and self.command == "GET": @@ -1519,6 +1876,13 @@ class Handler(BaseHTTPRequestHandler): else: self._send_error(503, "Sprachausgabe ist nicht installiert", "server_error", "feature_disabled") + elif (path == "/v1/audio/speech/pcm-stream" + and self.command == "POST"): + if ENABLE_TTS: + self._speech_pcm_stream() + else: + self._send_error(503, "Sprachausgabe ist nicht installiert", + "server_error", "feature_disabled") elif path == "/v1/audio/transcriptions" and self.command == "POST": if ENABLE_STT: self._transcribe() @@ -1713,6 +2077,7 @@ class Handler(BaseHTTPRequestHandler): "current_profile": current_profile(), "switching": STATE.switching, "profiles": PROFILES, + "mode": self._mode_payload(), "upstream": { "url": UPSTREAM_URL, "reachable": up["reachable"], @@ -1734,7 +2099,7 @@ class Handler(BaseHTTPRequestHandler): "phase": img.phase, "worker": "running" if (img.worker and img.worker.alive()) else "stopped", - "model": IMAGE_MODEL_NAME if img.phase != "idle" else None, + "model": img.current_model if img.phase != "idle" else None, "model_loaded": bool(img.worker and img.worker.model_loaded), "last_image": img.last_image, "last_seconds": img.last_seconds, @@ -1744,6 +2109,49 @@ class Handler(BaseHTTPRequestHandler): "stt": stt_status(), } + @staticmethod + def _mode_payload() -> dict: + state = RUNTIME.load() + return { + "active": STATE.mode, + "phase": STATE.mode_phase, + "music_worker": _music_worker_state(), + "music_health": _music_worker_health(), + "yue2_worker": _worker_field("yue2_worker"), + "yue2_health": _worker_field("yue2_health"), + "separator_worker": _separator_worker_state(), + "separator_health": _separator_worker_health(), + "voice_worker": _voice_worker_state(), + "voice_health": _voice_worker_health(), + "voice_change_worker": _voice_change_worker_state(), + "voice_change_health": _voice_change_worker_health(), + "applio_worker": _worker_field("applio_worker"), + "applio_health": _worker_field("applio_health"), + "trellis_worker": _worker_field("trellis_worker"), + "trellis_health": _worker_field("trellis_health"), + "return_profile": state.get("return_profile"), + "last_error": STATE.mode_error, + "enabled": ENABLE_MUSIC_MODE, + } + + def _mode_change(self) -> None: + try: + data = json.loads(self._read_body() or b"{}") + mode = data.get("mode") if isinstance(data, dict) else None + if mode not in {"llm", "music", "yue2", "separation", "voice", + "voicechange", "applio", "trellis"}: + raise ValueError("Feld 'mode' enthält einen unbekannten Betriebsmodus") + started, phase = schedule_operating_mode(mode) + self._send_json(202 if started else 200, { + "status": "accepted" if started else "ok", + "requested_mode": mode, + "phase": phase, + }) + except ValueError as exc: + self._send_error(400, str(exc), "invalid_request_error", "invalid_mode") + except RuntimeError as exc: + self._send_error(503, str(exc), "server_error", "mode_unavailable") + # ---------- Bildgenerierung ---------- def _image_generate(self) -> None: @@ -1841,6 +2249,13 @@ class Handler(BaseHTTPRequestHandler): "invalid_request_error", "prompt_too_long") return + model = data.get("model", IMAGE_MODEL_NAME) + if model != IMAGE_MODEL_NAME: + self._send_error( + 400, f"unbekanntes Bildmodell: {model!r}", + "invalid_request_error", "invalid_model") + return + # Größe size = data.get("size", "1024x1024") if size not in IMAGE_SIZES: @@ -1857,7 +2272,6 @@ class Handler(BaseHTTPRequestHandler): self._send_error(400, f"'n' muss eine Ganzzahl 1..{IMAGE_MAX_N} sein", "invalid_request_error", "invalid_n") return - # Qualität / Schritte / Guidance quality = data.get("quality", IMAGE_DEFAULT_QUALITY) if quality not in IMAGE_QUALITY: @@ -1866,8 +2280,9 @@ class Handler(BaseHTTPRequestHandler): "invalid_request_error", "invalid_quality") return steps = data.get("steps", IMAGE_QUALITY[quality]) - if not isinstance(steps, int) or isinstance(steps, bool) or steps != 4: - self._send_error(400, "FLUX.2-klein-4B erfordert 'steps'=4", + if (not isinstance(steps, int) or isinstance(steps, bool) + or steps != 4): + self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'steps'=4", "invalid_request_error", "invalid_steps") return guidance = data.get("guidance", 1.0) @@ -1878,7 +2293,7 @@ class Handler(BaseHTTPRequestHandler): "invalid_request_error", "invalid_guidance") return if guidance != 1.0: - self._send_error(400, "FLUX.2-klein-4B erfordert 'guidance'=1.0", + self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'guidance'=1.0", "invalid_request_error", "invalid_guidance") return @@ -1906,7 +2321,7 @@ class Handler(BaseHTTPRequestHandler): try: results, warning = generate_image( prompt.strip(), width, height, steps, guidance, seed, n, - quality, source_files) + quality, source_files, model) except (ValueError, RuntimeError) as e: self._send_error(503, str(e), "server_error", "image_generation_failed") return @@ -1979,7 +2394,7 @@ class Handler(BaseHTTPRequestHandler): self.end_headers() self.wfile.write(data) - # ---------- Sprachausgabe (XTTS-v2) ---------- + # ---------- Sprachausgabe (Qwen3-TTS) ---------- def _speech(self) -> None: try: @@ -2038,7 +2453,7 @@ class Handler(BaseHTTPRequestHandler): "invalid_request_error", "invalid_speed") return - # Modell-Name optional; falls angegeben, muss es xtts-v2 sein. + # Modell-Name optional; falls angegeben, muss es Qwen3-TTS sein. model = data.get("model") if model is not None and model != TTS_MODEL: self._send_error(400, f"unbekanntes Modell: {model!r} " @@ -2062,6 +2477,74 @@ class Handler(BaseHTTPRequestHandler): self.end_headers() self.wfile.write(audio) + def _speech_pcm_stream(self) -> None: + """Pass through Qwen's native 24 kHz PCM stream without buffering.""" + try: + body = self._read_body() + data = json.loads(body) + except ValueError as exc: + self._send_error(400, str(exc) or "ungültiges JSON", + "invalid_request_error", "invalid_body") + return + if not isinstance(data, dict): + self._send_error(400, "Request muss ein JSON-Objekt sein", + "invalid_request_error", "invalid_request") + return + text = data.get("input", data.get("text")) + if not isinstance(text, str) or not text.strip() or len(text) > 8000: + self._send_error(400, "'input' fehlt, ist leer oder zu lang", + "invalid_request_error", "invalid_input") + return + voice = data.get("voice", TTS_DEFAULT_VOICE) + if voice not in TTS_VOICES: + self._send_error(400, f"ungültige Stimme: {voice!r}", + "invalid_request_error", "invalid_voice") + return + + hostport = TTS_WORKER_URL.split("://", 1)[-1] + host, _, port = hostport.partition(":") + connection = None + headers_sent = False + try: + connection = http.client.HTTPConnection( + host, int(port) if port else 80, timeout=TTS_CONNECT_TIMEOUT) + connection.request( + "POST", "/tts/pcm-stream", body=body, + headers={"Content-Type": "application/json", + "Accept": "application/octet-stream"}) + connection.sock.settimeout(TTS_TIMEOUT) + response = connection.getresponse() + if response.status != 200: + message = response.read(512).decode(errors="replace") + self._send_error(503, + f"TTS-Stream fehlgeschlagen: {message}", + "server_error", "tts_failed") + return + self._last_code = 200 + self.send_response(200) + self.send_header("Content-Type", "application/octet-stream") + self.send_header("Cache-Control", "no-store") + self.send_header("Connection", "close") + self.end_headers() + headers_sent = True + while True: + chunk = response.read1(16384) + if not chunk: + break + self.wfile.write(chunk) + self.wfile.flush() + except (OSError, http.client.HTTPException) as exc: + if not headers_sent: + try: + self._send_error(502, f"TTS-Worker nicht erreichbar: {exc}", + "server_error", "tts_unavailable") + except (OSError, BrokenPipeError): + pass + finally: + if connection is not None: + connection.close() + self.close_connection = True + # ---------- Audio-Discovery ---------- def _audio_models_payload(self) -> dict: @@ -2080,7 +2563,7 @@ class Handler(BaseHTTPRequestHandler): models.append({ "id": TTS_MODEL, "object": "model", - "owned_by": "coqui-xtts", + "owned_by": "qwen", "type": "speech", }) return {"object": "list", "data": models} @@ -2269,6 +2752,12 @@ class Handler(BaseHTTPRequestHandler): data = json.loads(body) except ValueError: data = None + if (isinstance(data, dict) + and path in {"/v1/chat/completions", "/v1/responses"}): + command = _control_command(data, path) + if command is not None: + self._control_response(command, data, path) + return model = data.get("model") if isinstance(data, dict) else None if (isinstance(model, str) and REVIEW_UPSTREAM_URL and model == REVIEW_MODEL_NAME): @@ -2306,6 +2795,97 @@ class Handler(BaseHTTPRequestHandler): # An llama.cpp weiterleiten (mit Chat-Waiting, Streaming bleibt erhalten). self._proxy_with_wait(body) + def _control_response(self, command: str, data: dict, path: str) -> None: + """Return OpenAI-compatible local replies for Athena control commands.""" + if command == "/athena status": + mode = self._mode_payload() + profile = current_profile() + text = (f"Athena läuft im {mode['active'].upper()}-Modus. " + f"Phase: {mode['phase']}. Musik-Worker: " + f"{mode['music_worker']}. YuE2: " + f"{mode['yue2_worker']}. Stem-Separator: " + f"{mode['separator_worker']}. Voice Studio: " + f"{mode['voice_worker']}. Voice Changer: " + f"{mode['voice_change_worker']}. 3D Studio: " + f"{mode['trellis_worker']}. LLM-Profil: {profile or 'entladen'}.") + else: + target = ("music" if command == "/athena music" else + "yue2" if command == "/athena yue2" else + "separation" if command in {"/athena stems", "/athena separation"} + else "voice" if command == "/athena voice" + else "voicechange" if command in {"/athena voicechange", "/athena changer"} + else "applio" if command == "/athena applio" + else "trellis" if command in {"/athena 3d", "/athena trellis"} + else "llm") + try: + started, phase = schedule_operating_mode(target) + if started: + text = ("Musikstudio wird gestartet. LLM und TTS werden entladen." + if target == "music" else + "YuE2 Studio wird gestartet. LLM und TTS werden entladen." + if target == "yue2" else + "Stimmtrennung wird gestartet. LLM und TTS werden entladen." + if target == "separation" else + "Voice Studio wird gestartet. LLM und TTS werden entladen." + if target == "voice" else + "Voice Changer wird gestartet. LLM und TTS werden entladen." + if target == "voicechange" else + "Applio wird gestartet. LLM und TTS werden entladen." + if target == "applio" else + "3D Studio wird gestartet. LLM und TTS werden entladen." + if target == "trellis" else + "Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.") + else: + text = (f"Athena ist bereits im {target.upper()}-Modus " + f"oder wechselt gerade ({phase}).") + except RuntimeError as exc: + self._send_error(503, str(exc), "server_error", "mode_unavailable") + return + + model = str(data.get("model") or "athena-control") + created = int(time.time()) + request_id = f"athena-mode-{uuid.uuid4().hex[:16]}" + if path == "/v1/responses": + self._send_json(200, { + "id": request_id, "object": "response", "created_at": created, + "status": "completed", "model": model, + "output": [{"type": "message", "role": "assistant", + "content": [{"type": "output_text", "text": text}]}], + "output_text": text, + "usage": {"input_tokens": 0, "output_tokens": 0, + "total_tokens": 0}, + }) + return + if data.get("stream") is True: + chunks = [ + {"id": request_id, "object": "chat.completion.chunk", + "created": created, "model": model, + "choices": [{"index": 0, "delta": {"role": "assistant", + "content": text}, "finish_reason": None}]}, + {"id": request_id, "object": "chat.completion.chunk", + "created": created, "model": model, + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}, + ] + body = "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + body += "data: [DONE]\n\n" + encoded = body.encode() + self._last_code = 200 + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Content-Length", str(len(encoded))) + self.send_header("Connection", "close") + self.end_headers() + self.wfile.write(encoded) + return + self._send_json(200, { + "id": request_id, "object": "chat.completion", "created": created, + "model": model, + "choices": [{"index": 0, "message": {"role": "assistant", + "content": text}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 0, "completion_tokens": 0, + "total_tokens": 0}, + }) + def _acquire_model_lease(self, profile: str | None = None) -> dict: """Atomar Profil sicherstellen und einen aktiven Request registrieren.""" with STATE.lock: @@ -2545,6 +3125,29 @@ def _startup_reconcile() -> None: if removed: log.info("Startup-Retention: %d alte Bilder entfernt", len(removed)) + special_mode = previous.get("mode") + if ENABLE_MUSIC_MODE and special_mode in {"music", "yue2", "separation", + "voice", "voicechange", "applio", + "trellis"}: + STATE.mode = special_mode + STATE.mode_phase = f"starting-{special_mode}" + _set_qwen_unavailable(True) + try: + path, _worker_state, wait_ready = _special_worker(special_mode) + _profile_controller_request( + "POST", path, + timeout=120) + wait_ready() + STATE.mode_phase = "ready" + RUNTIME.save(mode=special_mode, phase=special_mode) + log.info("Recovery: Spezialmodus %s wiederhergestellt", special_mode) + except Exception as exc: + STATE.mode_error = str(exc) + STATE.mode_phase = "error" + log.error("Recovery: Spezialmodus %s konnte nicht gestartet werden: %s", + special_mode, exc) + return + profile = current_profile() if profile is None: saved = previous.get("last_profile") @@ -2568,7 +3171,9 @@ def _startup_reconcile() -> None: and (not EXPECTED_MODELS.get(profile) or up.get("model") == EXPECTED_MODELS[profile])): _set_qwen_unavailable(False) - RUNTIME.save(last_profile=profile, phase="idle") + STATE.mode = "llm" + STATE.mode_phase = "ready" + RUNTIME.save(last_profile=profile, mode="llm", phase="idle") log.info("Recovery: Profil %s ist bereits bereit", profile) return _set_qwen_unavailable(True)