Synchronize Athena operating modes with live deployment

This commit is contained in:
Mikei386
2026-09-13 18:50:34 +02:00
parent 9fe51390f6
commit 040a2df48b
5 changed files with 1349 additions and 198 deletions
+43 -141
View File
@@ -50,11 +50,6 @@ services:
- /tmp:size=16m,mode=1777
volumes:
- "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro"
# Namespace-sharing services cannot publish ports themselves. The owner
# must keep these bindings so a gateway recreation cannot hide their UIs.
ports:
- "8099:8099"
- "9443:9443"
networks:
frontend:
ipv4_address: 172.30.10.254
@@ -237,89 +232,6 @@ services:
- --spec-draft-p-min
- "0.05"
# Beta 1 keeps the complete GSQ-RCO text model and KV cache on the RTX 5080.
# Only the multimodal projector runs on the RTX 3060. The measured hard
# boundary is 196608 tokens; 192K deliberately retains runtime headroom.
llama-beta1:
<<: *llama-common
container_name: mike-ai-llama-beta1
labels:
com.mike-ai.llama-profile: beta1
environment:
NVIDIA_VISIBLE_DEVICES: ${BETA1_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${BETA1_MODEL_FILE:?BETA1_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --mmproj-offload
- --mmproj-device
- CUDA1
- --alias
- qwen-beta-1
- --ctx-size
- "${BETA1_CONTEXT:-192000}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-32768}"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "${BETA1_BATCH_SIZE:-2048}"
- --ubatch-size
- "${BETA1_UBATCH_SIZE:-128}"
- --parallel
- "${BETA1_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja
- --reasoning
- auto
- --reasoning-preserve
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --load-mode
- none
- --no-ui
- --temperature
- "1.0"
- --top-p
- "0.95"
- --top-k
- "20"
- --device
- CUDA0
- --main-gpu
- "0"
- --split-mode
- none
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "3"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
- --spec-draft-p-min
- "0.05"
llama-large:
<<: *llama-common
container_name: mike-ai-llama-large
@@ -575,8 +487,17 @@ services:
- /var/run/docker.sock:/var/run/docker.sock
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,beta1,large,ultra,uncensored
ALLOWED_PROFILES: fast,medium,large,ultra,uncensored
IMAGE_WORKER: image
RESTORE_WORKER: restore
TTS_WORKER: qwen3
MUSIC_WORKER: acestep
YUE2_WORKER: yue2
SEPARATOR_WORKER: bs-roformer
VOICE_WORKER: vevo2
VOICE_CHANGE_WORKER: xvc
APPLIO_WORKER: applio
TRELLIS_WORKER: trellis2-q8
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
@@ -612,6 +533,8 @@ services:
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
SWITCH_TIMEOUT: "600"
REQUEST_TIMEOUT: "600"
YUE2_START_TIMEOUT: "600"
TRELLIS_START_TIMEOUT: "900"
# Last-resort guard for every OpenAI-compatible client. Without a
# request limit llama.cpp uses n_predict=-1 and a reasoning loop can
# consume the complete context before yielding visible output.
@@ -621,19 +544,21 @@ services:
IMAGE_DIR: /data/images
IMAGE_WORKER_URL: http://image-worker:8086
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
IMAGE_MODEL_NAME: FLUX.2-klein-4B
IMAGE_MODEL_NAME: FLUX.2-klein-9B-fp8-beta
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "true"
ENABLE_TTS: "true"
# Stable OpenAI compatibility names remain piper/alloy because an
# existing Open WebUI database persists those values. The gateway maps
# alloy to XTTS speaker Annmarie Nele and automatically falls back to
# Piper if XTTS is unavailable, busy or returns an error.
# The gateway keeps text normalization, output conversion and native
# PCM streaming in one stable API in front of Qwen3-TTS.
TTS_WORKER_URL: http://tts-gateway:8085
TTS_MODEL: piper
TTS_MODEL: qwen3-tts
TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy
ENABLE_STT: "true"
ENABLE_MUSIC_MODE: "true"
MUSIC_START_TIMEOUT: "600"
VOICE_CHANGE_START_TIMEOUT: "600"
APPLIO_START_TIMEOUT: "900"
STT_WORKER_URL: http://whisper:8084
STT_TIMEOUT: "300"
networks: [frontend, control, inference]
@@ -655,8 +580,6 @@ services:
condition: service_healthy
profile-controller:
condition: service_healthy
piper:
condition: service_healthy
tts-gateway:
condition: service_healthy
whisper:
@@ -680,13 +603,15 @@ services:
read_only: true
tmpfs: ["/tmp:size=1g,mode=1777"]
volumes:
- "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro"
- "${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}:/models/components:ro"
- "${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}:/models/fp8:ro"
- router-images:/data/images
environment:
NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1}
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
FLUX_MODEL_DIR: /models/FLUX.2-klein-4B
FLUX_COMPONENT_DIR: /models/components
FLUX_TRANSFORMER_FILE: /models/fp8/flux-2-klein-9b-fp8.safetensors
IMAGE_DIR: /data/images
networks: [inference]
security_opt: ["no-new-privileges:true"]
@@ -697,41 +622,12 @@ services:
timeout: 3s
retries: 12
piper:
build:
context: platform/docker/piper
args:
PIPER_TTS_VERSION: ${PIPER_TTS_VERSION:-1.6.0}
image: mike-ai/piper:local
container_name: mike-ai-piper
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
volumes:
- piper-data:/data
environment:
PIPER_DATA_DIR: /data
PIPER_VOICE: ${PIPER_VOICE:-de_DE-thorsten-high}
PIPER_VOICE_ALIAS: alloy
PIPER_HOST: 0.0.0.0
PIPER_PORT: "8085"
PIPER_MAX_TEXT_CHARS: "8000"
networks: [frontend]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
cap_add: [CHOWN, SETUID, SETGID]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
interval: 10s
timeout: 5s
retries: 30
start_period: 120s
qwen3-tts:
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
container_name: mike-ai-qwen3-tts
restart: unless-stopped
labels:
com.mike-ai.tts-worker: qwen3
deploy:
resources:
reservations:
@@ -783,17 +679,12 @@ services:
QWEN_TTS_VOICE: serena
QWEN_TTS_LANGUAGE: German
QWEN_TTS_TIMEOUT: "120"
PIPER_URL: http://piper:8085
TTS_VOICE_ALIAS: alloy
TTS_DEFAULT_LANGUAGE: de
# Mixed-language clip stitching caused long pauses and unintelligible
# transitions. Keep full sentences in one stable German voice.
TTS_CODE_SWITCH_ENABLED: "false"
PIPER_TIMEOUT: "120"
networks: [frontend]
depends_on:
piper:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
@@ -843,7 +734,7 @@ services:
image: mike-ai/llama-dashboard:local
container_name: mike-ai-llama-dashboard
restart: unless-stopped
network_mode: "service:wireguard-gateway"
networks: [frontend]
gpus: all
read_only: true
tmpfs:
@@ -852,15 +743,26 @@ services:
- /proc:/host/proc:ro
- /data:/host/data:ro
- /data/models:/host/models:ro
- /data/emergency-backups:/backups:ro
- /data/llama-dashboard:/var/lib/llama-dashboard
environment:
DASHBOARD_HOST: 0.0.0.0
DASHBOARD_PORT: "8099"
ROUTER_URL: http://router:8081
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}"
APPLIO_UI_URL: "${APPLIO_UI_URL:-http://192.168.1.212:8011/}"
MIKES_APPLIO_UI_URL: "${MIKES_APPLIO_UI_URL:-http://192.168.1.212:8012/}"
TRELLIS_UI_URL: "${TRELLIS_UI_URL:-http://192.168.1.212:8013/}"
YUE2_UI_URL: "${YUE2_UI_URL:-http://192.168.1.212:8014/}"
HOST_PROC: /host/proc
HOST_DATA: /host/data
HOST_MODELS: /host/models
DASHBOARD_BACKUP_DIR: /backups
DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3
DASHBOARD_HISTORY_INTERVAL: "15"
DASHBOARD_DETAIL_RETENTION_DAYS: "21"
@@ -884,7 +786,7 @@ services:
image: ${PORTAINER_IMAGE:-portainer/portainer-ce@sha256:511f3f06c96fe3b993ebeaafde311c1959cae73a7ef825dba6397d51b450dffa}
container_name: mike-ai-portainer
restart: unless-stopped
network_mode: "service:wireguard-gateway"
networks: [frontend]
command: [--no-setup-token]
volumes:
- /var/run/docker.sock:/var/run/docker.sock
@@ -908,8 +810,9 @@ services:
- /var/run/docker.sock:/var/run/docker.sock:ro
- /data/docker-backups:/archive
- /etc/mike-ai:/backup/etc-mike-ai:ro
- /opt/mike-ai/stack:/backup/stack:ro
- piper-data:/backup/volumes/piper-data:ro
# Include every deployed specialist UI/worker source tree, not just the
# core checkout. Images themselves remain reproducible and are rebuilt.
- /opt/mike-ai:/backup/opt-mike-ai:ro
- router-state:/backup/volumes/router-state:ro
- router-images:/backup/volumes/router-images:ro
- portainer-data:/backup/volumes/portainer-data:ro
@@ -936,7 +839,6 @@ networks:
name: mike-ai-tools-egress
volumes:
piper-data:
whisper-data:
router-state:
router-images:
@@ -13,6 +13,7 @@ import logging
import os
import socket
import threading
import time
import urllib.parse
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
@@ -25,6 +26,22 @@ ALLOWED = tuple(x.strip() for x in os.environ.get(
LABEL_KEY = "com.mike-ai.llama-profile"
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore")
TTS_LABEL_KEY = "com.mike-ai.tts-worker"
TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3")
MUSIC_LABEL_KEY = "com.mike-ai.music-worker"
MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip()
YUE2_WORKER = os.environ.get("YUE2_WORKER", "").strip()
SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator"
SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
VOICE_CHANGE_LABEL_KEY = "com.mike-ai.voice-change-worker"
VOICE_CHANGE_WORKER = os.environ.get("VOICE_CHANGE_WORKER", "").strip()
APPLIO_LABEL_KEY = "com.mike-ai.applio-worker"
APPLIO_WORKER = os.environ.get("APPLIO_WORKER", "").strip()
TRELLIS_LABEL_KEY = "com.mike-ai.trellis-worker"
TRELLIS_WORKER = os.environ.get("TRELLIS_WORKER", "").strip()
LOCK = threading.Lock()
log = logging.getLogger("profile-controller")
@@ -68,15 +85,152 @@ def labelled_containers(label: str) -> list[dict]:
return json.loads(body)
def image_container() -> dict:
def image_container(kind: str = IMAGE_WORKER) -> dict:
matches = [item for item in labelled_containers(IMAGE_LABEL_KEY)
if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == IMAGE_WORKER]
if item.get("Labels", {}).get(IMAGE_LABEL_KEY) == kind]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one image worker {IMAGE_WORKER!r}, found {len(matches)}")
f"expected exactly one image worker {kind!r}, found {len(matches)}")
return matches[0]
def image_containers() -> list[dict]:
"""All allowlisted GPU workers that must never overlap an LLM."""
allowed = {IMAGE_WORKER, RESTORE_WORKER}
return [item for item in labelled_containers(IMAGE_LABEL_KEY)
if item.get("Labels", {}).get(IMAGE_LABEL_KEY) in allowed]
def tts_container() -> dict:
matches = [item for item in labelled_containers(TTS_LABEL_KEY)
if item.get("Labels", {}).get(TTS_LABEL_KEY) == TTS_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one TTS worker {TTS_WORKER!r}, found {len(matches)}")
return matches[0]
def music_container() -> dict:
if not MUSIC_WORKER:
raise RuntimeError("music worker is not configured")
matches = [item for item in labelled_containers(MUSIC_LABEL_KEY)
if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == MUSIC_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one music worker {MUSIC_WORKER!r}, found {len(matches)}")
return matches[0]
def yue2_container() -> dict:
if not YUE2_WORKER:
raise RuntimeError("YuE2 worker is not configured")
matches = [item for item in labelled_containers(MUSIC_LABEL_KEY)
if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == YUE2_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one YuE2 worker {YUE2_WORKER!r}, found {len(matches)}")
return matches[0]
def separator_container() -> dict:
if not SEPARATOR_WORKER:
raise RuntimeError("stem separator is not configured")
matches = [item for item in labelled_containers(SEPARATOR_LABEL_KEY)
if item.get("Labels", {}).get(SEPARATOR_LABEL_KEY) == SEPARATOR_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one stem separator {SEPARATOR_WORKER!r}, found {len(matches)}")
return matches[0]
def voice_container() -> dict:
if not VOICE_WORKER:
raise RuntimeError("voice worker is not configured")
matches = [item for item in labelled_containers(VOICE_LABEL_KEY)
if item.get("Labels", {}).get(VOICE_LABEL_KEY) == VOICE_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one voice worker {VOICE_WORKER!r}, found {len(matches)}")
return matches[0]
def voice_change_container() -> dict:
if not VOICE_CHANGE_WORKER:
raise RuntimeError("voice-change worker is not configured")
matches = [item for item in labelled_containers(VOICE_CHANGE_LABEL_KEY)
if item.get("Labels", {}).get(VOICE_CHANGE_LABEL_KEY) == VOICE_CHANGE_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one voice-change worker {VOICE_CHANGE_WORKER!r}, found {len(matches)}")
return matches[0]
def applio_container() -> dict:
if not APPLIO_WORKER:
raise RuntimeError("Applio worker is not configured")
matches = [item for item in labelled_containers(APPLIO_LABEL_KEY)
if item.get("Labels", {}).get(APPLIO_LABEL_KEY) == APPLIO_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one Applio worker {APPLIO_WORKER!r}, found {len(matches)}")
return matches[0]
def trellis_container() -> dict:
if not TRELLIS_WORKER:
raise RuntimeError("TRELLIS worker is not configured")
matches = [item for item in labelled_containers(TRELLIS_LABEL_KEY)
if item.get("Labels", {}).get(TRELLIS_LABEL_KEY) == TRELLIS_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one TRELLIS worker {TRELLIS_WORKER!r}, found {len(matches)}")
return matches[0]
def stop_music_if_configured() -> None:
if MUSIC_WORKER:
stop_container(music_container(), timeout=30)
def stop_yue2_if_configured() -> None:
if YUE2_WORKER:
stop_container(yue2_container(), timeout=30)
def stop_separator_if_configured() -> None:
if SEPARATOR_WORKER:
stop_container(separator_container(), timeout=30)
def stop_voice_if_configured() -> None:
if VOICE_WORKER:
stop_container(voice_container(), timeout=30)
def stop_voice_change_if_configured() -> None:
if VOICE_CHANGE_WORKER:
stop_container(voice_change_container(), timeout=30)
def stop_applio_if_configured() -> None:
if APPLIO_WORKER:
stop_container(applio_container(), timeout=30)
def stop_trellis_if_configured() -> None:
if TRELLIS_WORKER:
stop_container(trellis_container(), timeout=30)
def stop_voice_tools(except_kind: str | None = None) -> None:
if except_kind != "voice":
stop_voice_if_configured()
if except_kind != "voicechange":
stop_voice_change_if_configured()
if except_kind != "applio":
stop_applio_if_configured()
def stop_container(item: dict, timeout: int = 120) -> None:
if item.get("State") != "running":
return
@@ -85,6 +239,31 @@ def stop_container(item: dict, timeout: int = 120) -> None:
raise RuntimeError(f"failed to stop container: HTTP {status}")
def start_container(item: dict) -> None:
status, _ = docker_request("POST", f"/containers/{item['Id']}/start")
if status not in (204, 304):
raise RuntimeError(f"failed to start container: HTTP {status}")
def wait_container_healthy(item: dict, timeout: int) -> None:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
status, body = docker_request("GET", f"/containers/{item['Id']}/json")
if status != 200:
raise RuntimeError(f"failed to inspect container: HTTP {status}")
state = json.loads(body).get("State", {})
health = state.get("Health", {}).get("Status")
if state.get("Status") == "running" and health == "healthy":
return
if state.get("Status") in {"dead", "exited"} or health == "unhealthy":
raise RuntimeError(
f"container {item.get('Names', ['unknown'])[0]} failed: "
f"state={state.get('Status')} health={health}")
time.sleep(1)
raise RuntimeError(
f"container {item.get('Names', ['unknown'])[0]} did not become healthy")
def stop_inference() -> dict:
with LOCK:
items = containers()
@@ -94,22 +273,186 @@ def stop_inference() -> dict:
return {"active_profile": None, "previous_profile": previous}
def set_image_worker(running: bool) -> dict:
def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
if kind not in {IMAGE_WORKER, RESTORE_WORKER}:
raise ValueError("worker is not allowlisted")
with LOCK:
item = image_container()
item = image_container(kind)
if running:
# The image worker may never overlap a llama profile on the 5080.
for profile_item in containers().values():
stop_container(profile_item)
if item.get("State") != "running":
status, _ = docker_request("POST", f"/containers/{item['Id']}/start")
if status not in (204, 304):
raise RuntimeError(f"failed to start image worker: HTTP {status}")
# The 9B beta text encoder temporarily borrows the RTX 3060 from
# Qwen3-TTS. TTS is unavailable during this exclusive GPU phase.
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_yue2_if_configured()
stop_separator_if_configured()
stop_voice_tools()
stop_trellis_if_configured()
for other in image_containers():
if other["Id"] != item["Id"]:
stop_container(other, timeout=20)
start_container(item)
else:
# CUDA/PyTorch may not react promptly to SIGTERM after an OOM.
# Bound recovery time and let Docker issue SIGKILL afterwards.
stop_container(item, timeout=20)
return {"image_worker": "running" if running else "stopped"}
# TTS is restored by the following profile activation. Keeping it
# stopped here lets the router verify that both GPUs really
# released the image model before Qwen and TTS are reloaded.
return {"image_worker": kind,
"state": "running" if running else "stopped"}
def set_music_worker(running: bool) -> dict:
"""Start ACE-Step exclusively, or stop it before LLM restoration."""
with LOCK:
item = music_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_separator_if_configured()
stop_voice_tools()
stop_yue2_if_configured()
stop_trellis_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"music_worker": MUSIC_WORKER,
"state": "running" if running else "stopped"}
def set_yue2_worker(running: bool) -> dict:
"""Start YuE2 exclusively, or stop it before another mode is loaded."""
with LOCK:
item = yue2_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_separator_if_configured()
stop_voice_tools()
stop_trellis_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"yue2_worker": YUE2_WORKER,
"state": "running" if running else "stopped"}
def set_separator_worker(running: bool) -> dict:
"""Start vocal separation exclusively, or stop it before LLM restoration."""
with LOCK:
item = separator_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_yue2_if_configured()
stop_voice_tools()
stop_trellis_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"separator_worker": SEPARATOR_WORKER,
"state": "running" if running else "stopped"}
def set_voice_worker(running: bool) -> dict:
"""Start OmniVoice exclusively, or stop it before LLM restoration."""
with LOCK:
item = voice_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_yue2_if_configured()
stop_separator_if_configured()
stop_voice_tools("voice")
stop_trellis_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"voice_worker": VOICE_WORKER,
"state": "running" if running else "stopped"}
def set_voice_change_worker(running: bool) -> dict:
"""Start X-VC exclusively, or stop it before another mode is loaded."""
with LOCK:
item = voice_change_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_yue2_if_configured()
stop_separator_if_configured()
stop_voice_tools("voicechange")
stop_trellis_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"voice_change_worker": VOICE_CHANGE_WORKER,
"state": "running" if running else "stopped"}
def set_applio_worker(running: bool) -> dict:
"""Start Applio exclusively, or stop it before another mode is loaded."""
with LOCK:
item = applio_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_yue2_if_configured()
stop_separator_if_configured()
stop_voice_tools("applio")
stop_trellis_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"applio_worker": APPLIO_WORKER,
"state": "running" if running else "stopped"}
def set_trellis_worker(running: bool) -> dict:
"""Start TRELLIS.2 exclusively, or stop it before another mode is loaded."""
with LOCK:
item = trellis_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_yue2_if_configured()
stop_separator_if_configured()
stop_voice_tools()
start_container(item)
else:
stop_container(item, timeout=30)
return {"trellis_worker": TRELLIS_WORKER,
"state": "running" if running else "stopped"}
def active_profile(items: dict[str, dict] | None = None) -> str | None:
@@ -125,7 +468,14 @@ def activate(profile: str) -> dict:
raise ValueError("profile is not allowlisted")
with LOCK:
# Defensive mutual exclusion even if a caller bypasses the router.
stop_container(image_container())
for worker in image_containers():
stop_container(worker)
stop_music_if_configured()
stop_yue2_if_configured()
stop_separator_if_configured()
stop_voice_tools()
stop_trellis_if_configured()
start_container(tts_container())
items = containers()
missing = [name for name in ALLOWED if name not in items]
if missing:
@@ -189,7 +539,70 @@ class Handler(BaseHTTPRequestHandler):
return
try:
items = containers()
music = music_container() if MUSIC_WORKER else {}
music_status = music.get("Status", "")
music_health = ("disabled" if not MUSIC_WORKER else
"healthy" if "(healthy)" in music_status else
"unhealthy" if "(unhealthy)" in music_status else
"starting" if music.get("State") == "running" else
"stopped")
yue2 = yue2_container() if YUE2_WORKER else {}
yue2_status = yue2.get("Status", "")
yue2_health = ("disabled" if not YUE2_WORKER else
"healthy" if "(healthy)" in yue2_status else
"unhealthy" if "(unhealthy)" in yue2_status else
"starting" if yue2.get("State") == "running" else
"stopped")
separator = separator_container() if SEPARATOR_WORKER else {}
separator_status = separator.get("Status", "")
separator_health = ("disabled" if not SEPARATOR_WORKER else
"healthy" if "(healthy)" in separator_status else
"unhealthy" if "(unhealthy)" in separator_status else
"starting" if separator.get("State") == "running" else
"stopped")
voice = voice_container() if VOICE_WORKER else {}
voice_status = voice.get("Status", "")
voice_health = ("disabled" if not VOICE_WORKER else
"healthy" if "(healthy)" in voice_status else
"unhealthy" if "(unhealthy)" in voice_status else
"starting" if voice.get("State") == "running" else
"stopped")
voice_change = voice_change_container() if VOICE_CHANGE_WORKER else {}
voice_change_status = voice_change.get("Status", "")
voice_change_health = ("disabled" if not VOICE_CHANGE_WORKER else
"healthy" if "(healthy)" in voice_change_status else
"unhealthy" if "(unhealthy)" in voice_change_status else
"starting" if voice_change.get("State") == "running" else
"stopped")
applio = applio_container() if APPLIO_WORKER else {}
applio_status = applio.get("Status", "")
applio_health = ("disabled" if not APPLIO_WORKER else
"healthy" if "(healthy)" in applio_status else
"unhealthy" if "(unhealthy)" in applio_status else
"starting" if applio.get("State") == "running" else
"stopped")
trellis = trellis_container() if TRELLIS_WORKER else {}
trellis_status = trellis.get("Status", "")
trellis_health = ("disabled" if not TRELLIS_WORKER else
"healthy" if "(healthy)" in trellis_status else
"unhealthy" if "(unhealthy)" in trellis_status else
"starting" if trellis.get("State") == "running" else
"stopped")
self.reply(200, {"active_profile": active_profile(items),
"music_worker": music.get("State", "disabled"),
"music_health": music_health,
"yue2_worker": yue2.get("State", "disabled"),
"yue2_health": yue2_health,
"separator_worker": separator.get("State", "disabled"),
"separator_health": separator_health,
"voice_worker": voice.get("State", "disabled"),
"voice_health": voice_health,
"voice_change_worker": voice_change.get("State", "disabled"),
"voice_change_health": voice_change_health,
"applio_worker": applio.get("State", "disabled"),
"applio_health": applio_health,
"trellis_worker": trellis.get("State", "disabled"),
"trellis_health": trellis_health,
"profiles": {name: items.get(name, {}).get(
"State", "missing") for name in ALLOWED}})
except Exception as exc:
@@ -207,9 +620,65 @@ class Handler(BaseHTTPRequestHandler):
log.exception("stopping inference failed")
self.reply(503, {"error": str(exc)})
return
if self.path in ("/workers/image/start", "/workers/image/stop"):
if self.path in {"/workers/music/start", "/workers/music/stop"}:
try:
self.reply(200, set_image_worker(self.path.endswith("/start")))
self.reply(200, set_music_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("music worker transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/yue2/start", "/workers/yue2/stop"}:
try:
self.reply(200, set_yue2_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("YuE2 worker transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/separator/start", "/workers/separator/stop"}:
try:
self.reply(200, set_separator_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("stem separator transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/voice/start", "/workers/voice/stop"}:
try:
self.reply(200, set_voice_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("voice worker transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/voice-change/start", "/workers/voice-change/stop"}:
try:
self.reply(200, set_voice_change_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("voice-change worker transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/applio/start", "/workers/applio/stop"}:
try:
self.reply(200, set_applio_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("Applio worker transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/trellis/start", "/workers/trellis/stop"}:
try:
self.reply(200, set_trellis_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("TRELLIS worker transition failed")
self.reply(503, {"error": str(exc)})
return
worker_paths = {
"/workers/image/start": (IMAGE_WORKER, True),
"/workers/image/stop": (IMAGE_WORKER, False),
"/workers/restore/start": (RESTORE_WORKER, True),
"/workers/restore/stop": (RESTORE_WORKER, False),
}
if self.path in worker_paths:
try:
kind, running = worker_paths[self.path]
self.reply(200, set_image_worker(running, kind))
except Exception as exc:
log.exception("image worker transition failed")
self.reply(503, {"error": str(exc)})
@@ -76,7 +76,17 @@ start_proxy() {
start_proxy 22 172.30.10.1:22
start_proxy 8081 router:8081
start_proxy 8085 tts-gateway:8085
start_proxy 8091 piper:8085
start_proxy 8099 llama-dashboard:8099
start_proxy 7861 music-ui:3000
start_proxy 7862 music-worker:7860
start_proxy 8007 stem-separator:8080
start_proxy 8008 voice-studio:8008
start_proxy 8009 xvc-studio:8009
start_proxy 8011 applio-studio:6969
start_proxy 8012 mikes-applio-ui:8012
start_proxy 8013 trellis-studio:8080
start_proxy 8014 yue2-studio:8014
start_proxy 8202 mcp-athena-operator:8000
start_proxy 9443 portainer:9443
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')
+168 -3
View File
@@ -20,15 +20,59 @@ HOST = os.getenv("DASHBOARD_HOST", "0.0.0.0")
PORT = int(os.getenv("DASHBOARD_PORT", "8099"))
ROUTER_URL = os.getenv("ROUTER_URL", "http://router:8081").rstrip("/")
ROUTER_API_KEY = os.getenv("ROUTER_API_KEY", "")
MUSIC_COMMUNITY_UI_URL = os.getenv(
"MUSIC_COMMUNITY_UI_URL",
os.getenv("MUSIC_UI_URL", "http://192.168.1.212:7861/"),
)
MUSIC_ORIGINAL_UI_URL = os.getenv(
"MUSIC_ORIGINAL_UI_URL", "http://192.168.1.212:7862/"
)
SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
VOICE_CHANGE_UI_URL = os.getenv("VOICE_CHANGE_UI_URL", "http://192.168.1.212:8009/")
APPLIO_UI_URL = os.getenv("APPLIO_UI_URL", "http://192.168.1.212:8011/")
MIKES_APPLIO_UI_URL = os.getenv(
"MIKES_APPLIO_UI_URL", "http://192.168.1.212:8012/"
)
TRELLIS_UI_URL = os.getenv("TRELLIS_UI_URL", "http://192.168.1.212:8013/")
YUE2_UI_URL = os.getenv("YUE2_UI_URL", "http://192.168.1.212:8014/")
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
BACKUP_DIR = Path(os.getenv("DASHBOARD_BACKUP_DIR", "/host/data/emergency-backups"))
STARTED = time.time()
HISTORY_DB = Path(os.getenv("DASHBOARD_HISTORY_DB", "/var/lib/llama-dashboard/history.sqlite3"))
HISTORY_INTERVAL = max(5, int(os.getenv("DASHBOARD_HISTORY_INTERVAL", "15")))
DETAIL_RETENTION_DAYS = max(1, int(os.getenv("DASHBOARD_DETAIL_RETENTION_DAYS", "21")))
def backup_inventory() -> list[dict[str, Any]]:
try:
candidates = sorted(
BACKUP_DIR.glob("athena-portable-*.tar.zst.age"),
key=lambda item: item.stat().st_mtime,
reverse=True,
)[:5]
except OSError:
return []
result = []
for path in candidates:
try:
stat = path.stat()
checksum_file = path.with_name(path.name + ".sha256")
checksum = _read_text(checksum_file).split(maxsplit=1)[0]
result.append({
"name": path.name,
"size": stat.st_size,
"modified": stat.st_mtime,
"sha256": checksum if len(checksum) == 64 else None,
"download_url": "/api/backups/download/" + urllib.parse.quote(path.name),
})
except OSError:
continue
return result
def _number(value: str) -> int | float | None:
value = value.strip()
if not value or value.lower() in {"n/a", "[n/a]", "not supported"}:
@@ -268,6 +312,28 @@ def router_status() -> tuple[dict[str, Any], str | None]:
return {}, str(exc)
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
if mode not in {"llm", "music", "yue2", "separation", "voice",
"voicechange", "applio", "trellis"}:
return 400, {"error": "invalid mode"}
headers = {"Accept": "application/json", "Content-Type": "application/json"}
if ROUTER_API_KEY:
headers["Authorization"] = f"Bearer {ROUTER_API_KEY}"
request = urllib.request.Request(
f"{ROUTER_URL}/mode", data=json.dumps({"mode": mode}).encode(),
headers=headers, method="POST")
try:
with urllib.request.urlopen(request, timeout=15) as response:
return response.status, json.load(response)
except urllib.error.HTTPError as exc:
try:
return exc.code, json.loads(exc.read())
except (ValueError, json.JSONDecodeError):
return exc.code, {"error": str(exc)}
except (OSError, urllib.error.URLError) as exc:
return 503, {"error": str(exc)}
_MODEL_LOCK = threading.Lock()
_MODEL_AT = 0.0
_MODEL_CACHE: tuple[list[dict[str, Any]], dict[str, Any]] = ([], {"count": 0, "total_size": 0})
@@ -602,9 +668,12 @@ HTML = r'''<!doctype html>
.span4 .metrics{grid-template-columns:repeat(2,1fr)}
.history-controls{display:flex;flex-wrap:wrap;gap:7px;margin:10px 0 14px}.history-controls button{border:1px solid var(--line);background:#09111b;color:var(--muted);padding:6px 10px;border-radius:8px;cursor:pointer}.history-controls button.active{color:var(--cyan);border-color:var(--cyan)}.chart-legend{display:flex;flex-wrap:wrap;gap:8px;margin:2px 0 10px}.chart-legend button{border:1px solid var(--series);background:#09111b;color:var(--text);padding:6px 10px;border-radius:8px;cursor:pointer}.chart-legend button::before{content:'';display:inline-block;width:10px;height:3px;background:var(--series);margin:0 7px 3px 0}.chart-legend button.off{opacity:.4;text-decoration:line-through}.chart{width:100%;height:250px;display:block}.token-total{font-size:24px;font-weight:750;margin-top:5px}.history-note{color:var(--muted);font-size:11px;margin-top:8px}
.usage-list{display:grid;gap:13px;margin-top:8px}.usage-head{display:flex;justify-content:space-between;gap:14px;align-items:baseline}.usage-head b{font-size:16px}.usage-head span{color:var(--muted)}.usage-meta{display:flex;justify-content:space-between;gap:12px;color:var(--muted);font-size:11px;margin-top:5px}
.mode-row{display:flex;align-items:center;justify-content:space-between;gap:18px;flex-wrap:wrap}.mode-buttons,.mode-buttons span{display:flex;gap:9px;flex-wrap:wrap}.mode-buttons button,.mode-buttons a{border:1px solid var(--line);background:#09111b;color:var(--text);padding:9px 14px;border-radius:9px;cursor:pointer;text-decoration:none;font:inherit}.mode-buttons button.active{border-color:var(--cyan);color:var(--cyan);box-shadow:inset 0 0 0 1px #45d7ff33}.mode-buttons button:disabled{opacity:.45;cursor:wait}.mode-buttons span[hidden]{display:none}.mode-buttons a.stable{border-color:#66e3a466;color:var(--green)}.mode-buttons a.experimental{border-color:#ffc65c66;color:var(--amber)}.mode-error{color:var(--red)}
.download{display:inline-block;border:1px solid #66e3a466;color:var(--green);padding:6px 10px;border-radius:8px;text-decoration:none}.hash{font:11px ui-monospace,SFMono-Regular,monospace;color:var(--muted);word-break:break-all}
</style></head><body><main>
<div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div>
<section class="grid">
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">ACE-Step Studio</button><button id="yue2Mode" onclick="setMode('yue2')">YuE2 Studio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><button id="voiceMode" onclick="setMode('voice')">Voice Studio</button><button id="voiceChangeMode" onclick="setMode('voicechange')">X-VC</button><button id="applioMode" onclick="setMode('applio')">Applio / RVC · Modell nötig</button><button id="trellisMode" onclick="setMode('trellis')">3D Studio</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="yue2Open" hidden><a class="stable" href="__YUE2_UI_URL__" target="_blank" rel="noopener">YuE2 Studio öffnen</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span><span id="voiceOpen" hidden><a class="stable" href="__VOICE_UI_URL__" target="_blank" rel="noopener">Voice Studio öffnen</a></span><span id="voiceChangeOpen" hidden><a class="experimental" href="__VOICE_CHANGE_UI_URL__" target="_blank" rel="noopener">X-VC öffnen</a></span><span id="applioOpen" hidden><a class="stable" href="__APPLIO_UI_URL__" target="_blank" rel="noopener">Original Applio UI</a><a class="stable" href="__MIKES_APPLIO_UI_URL__" target="_blank" rel="noopener">Mikes Applio UI</a></span><span id="trellisOpen" hidden><a class="stable" href="__TRELLIS_UI_URL__" target="_blank" rel="noopener">3D Studio öffnen</a></span></div></div></article>
<article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article>
<article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article>
<article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article>
@@ -636,14 +705,28 @@ HTML = r'''<!doctype html>
<article class="card span6"><div class="label">Ereignisse seit Dashboard-Start</div><div id="events"><div class="sub">Noch keine Zustandsänderung</div></div></article>
<article class="card span12"><div class="label">llama.cpp Laufzeitkonfiguration</div><div class="metrics" id="runtime"><div class="metric"><b>–</b><span>wird gelesen</span></div></div></article>
<article class="card span12"><div class="row"><div><div class="label">Verfügbare GGUF-Dateien</div><div class="sub" id="modelSummary">–</div></div></div><div class="scroll"><table><thead><tr><th>Datei</th><th>Pfad</th><th>Größe</th><th>Geändert</th></tr></thead><tbody id="modelFiles"><tr><td colspan="4">–</td></tr></tbody></table></div></article>
<div class="section-title">Notfall-Backups · herunterladen</div>
<article class="card span12"><div class="row"><div><div class="label">Verschlüsselte portable Sicherungen</div><div class="sub">Unersetzliche Daten ohne erneut ladbare Modellgewichte · maximal fünf Generationen</div></div></div><div class="scroll"><table><thead><tr><th>Erstellt</th><th>Größe</th><th>SHA256</th><th></th></tr></thead><tbody id="backupFiles"><tr><td colspan="4">Backups werden geladen …</td></tr></tbody></table></div><div class="sub" id="backupNote">Zum Wiederherstellen wird der separat verwahrte Age-Schlüssel benötigt.</div></article>
<article class="card span12 error" id="errors" hidden></article>
</section><div class="footer">Aktualisierung jede Sekunde · Nur lesende Telemetrie</div>
</section><div class="footer">Aktualisierung jede Sekunde · Umschaltung über Athena Router</div>
</main><script>
const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed?.(1)??n}%`; const gib=b=>b?`${(b/1073741824).toFixed(1)} GiB`:'–'; const dur=s=>{if(s==null)return'–';let d=Math.floor(s/86400),h=Math.floor(s%86400/3600),m=Math.floor(s%3600/60);return d?`${d}d ${h}h`:`${h}h ${m}m`};
function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `<article class="card span6"><div class="gpu-title"><div><div class="label">GPU ${g.index}</div><div class="value">${g.name}</div></div><span class="badge">${g.pstate||'–'}</span></div><div class="metrics"><div class="metric"><b>${pct(g.gpu_percent)}</b><span>GPU-Kern</span></div><div class="metric"><b>${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiB</b><span>VRAM</span></div><div class="metric"><b>${g.temperature_c??'–'} °C</b><span>Temperatur</span></div><div class="metric"><b>${g.power_w??'–'} / ${g.power_limit_w??'–'} W</b><span>Leistung</span></div><div class="metric"><b>${g.graphics_clock_mhz??'–'} MHz</b><span>Grafiktakt</span></div><div class="metric"><b>${g.memory_clock_mhz??'–'} MHz</b><span>Speichertakt</span></div><div class="metric"><b>${pct(g.memory_controller_percent)}</b><span>Memory Controller</span></div><div class="metric"><b>${pct(g.fan_percent)}</b><span>Lüfter</span></div></div><div class="bar-label"><span>GPU-Auslastung</span><span>${load.toFixed(1)} %</span></div><div class="bar"><div class="fill gpu-load" style="width:${load}%"></div></div><div class="bar-label"><span>VRAM-Belegung</span><span>${p.toFixed(1)} %</span></div><div class="bar"><div class="fill" style="width:${Math.min(100,p)}%"></div></div><div class="sub">${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei</div></article>`}
const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit');
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''
let modeBusy=false;
async function setMode(mode){if(modeBusy)return;modeBusy=true;for(const id of ['llmMode','musicMode','yue2Mode','separationMode','voiceMode','voiceChangeMode','applioMode','trellisMode'])$(id).disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=({llm:'LLM-Betrieb',music:'ACE-Step Studio',yue2:'YuE2 Studio',separation:'Stimmtrennung',voice:'Voice Studio',voicechange:'X-VC Voice Changer',applio:'Applio / RVC',trellis:'3D Studio'})[md.active]||'Unbekannt';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`ACE-Step: ${md.music_worker||'–'} · YuE2: ${md.yue2_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'} · 3D: ${md.trellis_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('yue2Mode').classList.toggle('active',md.active==='yue2');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('voiceChangeMode').classList.toggle('active',md.active==='voicechange');$('applioMode').classList.toggle('active',md.active==='applio');$('trellisMode').classList.toggle('active',md.active==='trellis');let modeControlsBusy=modeBusy||switchingMode||!md.enabled;for(const id of ['llmMode','musicMode','yue2Mode','separationMode','voiceMode','voiceChangeMode','applioMode','trellisMode'])$(id).disabled=modeControlsBusy;$('musicOpen').hidden=md.active!=='music';$('yue2Open').hidden=md.active!=='yue2';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('voiceChangeOpen').hidden=md.active!=='voicechange';$('applioOpen').hidden=md.active!=='applio';$('trellisOpen').hidden=md.active!=='trellis';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
async function refreshBackups(){try{let r=await fetch('/api/backups',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),rows=d.backups||[];$('backupFiles').innerHTML=rows.map(b=>`<tr><td>${new Date(b.modified*1000).toLocaleString('de-DE')}</td><td>${gib(b.size)}</td><td class="hash">${b.sha256||'Prüfsumme fehlt'}</td><td><a class="download" href="${b.download_url}">Herunterladen</a></td></tr>`).join('')||'<tr><td colspan="4">Noch kein portables Backup vorhanden</td></tr>';$('backupNote').textContent=rows.length?`${rows.length} von maximal 5 Generationen · verschlüsselt mit Age`:'Der erste Lauf startet spätestens fünf Stunden nach Aktivierung.'}catch(e){$('backupNote').textContent=`Backup-Liste nicht verfügbar: ${e.message}`}}refreshBackups();setInterval(refreshBackups,60000);
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''.replace(
"__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL
).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL
).replace("__VOICE_UI_URL__", VOICE_UI_URL
).replace("__VOICE_CHANGE_UI_URL__", VOICE_CHANGE_UI_URL
).replace("__APPLIO_UI_URL__", APPLIO_UI_URL
).replace("__MIKES_APPLIO_UI_URL__", MIKES_APPLIO_UI_URL).replace(
"__TRELLIS_UI_URL__", TRELLIS_UI_URL
).replace("__YUE2_UI_URL__", YUE2_UI_URL)
FULL_JS = r'''
@@ -764,6 +847,62 @@ class Handler(BaseHTTPRequestHandler):
self.end_headers()
self.wfile.write(body)
def _send_backup(self, name: str) -> None:
# Generated names are deliberately strict; never expose an arbitrary
# host path through the dashboard.
if not name.startswith("athena-portable-") or not name.endswith(".tar.zst.age"):
self._send(404, b'{"error":"not found"}', "application/json")
return
if Path(name).name != name:
self._send(404, b'{"error":"not found"}', "application/json")
return
path = BACKUP_DIR / name
try:
size = path.stat().st_size
except OSError:
self._send(404, b'{"error":"not found"}', "application/json")
return
start, end = 0, size - 1
status = 200
range_header = self.headers.get("Range", "")
if range_header:
try:
unit, raw = range_header.split("=", 1)
first, last = raw.split("-", 1)
if unit != "bytes" or "," in raw or not first:
raise ValueError
start = int(first)
end = min(size - 1, int(last)) if last else size - 1
if start < 0 or start > end or start >= size:
raise ValueError
status = 206
except ValueError:
self.send_response(416)
self.send_header("Content-Range", f"bytes */{size}")
self.end_headers()
return
self.send_response(status)
self.send_header("Content-Type", "application/octet-stream")
self.send_header("Content-Disposition", f'attachment; filename="{name}"')
self.send_header("Accept-Ranges", "bytes")
self.send_header("Content-Length", str(end - start + 1))
if status == 206:
self.send_header("Content-Range", f"bytes {start}-{end}/{size}")
self.send_header("Cache-Control", "no-store")
self.send_header("X-Content-Type-Options", "nosniff")
self.end_headers()
remaining = end - start + 1
with path.open("rb") as source:
source.seek(start)
while remaining:
chunk = source.read(min(1024 * 1024, remaining))
if not chunk:
break
self.wfile.write(chunk)
remaining -= len(chunk)
def do_GET(self) -> None:
path = self.path.split("?", 1)[0]
if path == "/":
@@ -782,9 +921,35 @@ class Handler(BaseHTTPRequestHandler):
range_name = query.get("range", ["24h"])[0]
body = json.dumps(HISTORY.query(range_name), ensure_ascii=False, separators=(",", ":")).encode()
self._send(200, body, "application/json; charset=utf-8")
elif path == "/api/backups":
body = json.dumps({"backups": backup_inventory()}, ensure_ascii=False,
separators=(",", ":")).encode()
self._send(200, body, "application/json; charset=utf-8")
elif path.startswith("/api/backups/download/"):
name = urllib.parse.unquote(path.removeprefix("/api/backups/download/"))
self._send_backup(name)
else:
self._send(404, b'{"error":"not found"}', "application/json")
def do_POST(self) -> None:
path = self.path.split("?", 1)[0]
if path != "/api/mode":
self._send(404, b'{"error":"not found"}', "application/json")
return
try:
length = int(self.headers.get("Content-Length", "0"))
if length <= 0 or length > 1024:
raise ValueError("invalid body size")
payload = json.loads(self.rfile.read(length))
mode = payload.get("mode") if isinstance(payload, dict) else None
except (ValueError, json.JSONDecodeError):
self._send(400, b'{"error":"invalid request"}', "application/json")
return
status, response = change_mode(mode)
body = json.dumps(response, ensure_ascii=False,
separators=(",", ":")).encode()
self._send(status, body, "application/json; charset=utf-8")
if __name__ == "__main__":
threading.Thread(target=history_collector, name="history-collector", daemon=True).start()
+645 -40
View File
@@ -2,31 +2,30 @@
"""AI Profile Router – OpenAI-kompatibler Proxy vor llama.cpp.
Leitet OpenAI-kompatible Requests transparent an den lokalen llama.cpp-Server
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen sechs festen
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen fünf festen
Profilen um:
Profil Kontext
------ --------
fast 76800
medium 160000
beta1 192000
large 192000
ultra 262144
uncensored 80000
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-beta-1, qwen-large,
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large,
qwen-ultra, qwen-uncensored
Kommandos: POST /fast, /medium, /beta1, /large, /ultra,
Kommandos: POST /fast, /medium, /large, /ultra,
/uncensored
GET /status (Zustand)
Bildgenerierung und Editing (FLUX.2-klein-4B):
Bildgenerierung und Editing (FLUX.2 Klein 9B FP8 beta):
POST /v1/images/generations (OpenAI-kompatibel)
POST /v1/images/edits (lokal, Referenzbilder)
GET /images (Liste)
GET /images/<datei> (PNG-Download)
Sprachausgabe (XTTS-v2, multilingual, CPU-only):
Sprachausgabe (Qwen3-TTS auf RTX 3060):
POST /v1/audio/speech (OpenAI-kompatibel)
GET /v1/audio/voices (verfügbare Stimmen)
@@ -40,7 +39,8 @@ Der Router leitet /v1/audio/speech und /v1/audio/transcriptions
per HTTP an die Worker weiter.
Der Router agiert als Modell-Orchestrator: vor der Generierung wird
llama.cpp gestoppt, der Bild-Worker lädt FLUX.2, generiert/bearbeitet und entlädt
llama.cpp und Qwen3-TTS gestoppt, der Bild-Worker lädt FLUX.2 und den
Text-Encoder auf getrennte GPUs, generiert/bearbeitet und entlädt
das Modell wieder; danach wird das vorherige Qwen-Profil wiederher-
gestellt und erst dann geantwortet (try/finally – Qwen wird auch bei
Fehlgeschlagener Generierung wiederhergestellt).
@@ -105,6 +105,15 @@ PROFILE_DIR = os.environ.get(
PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/")
PROFILE_CONTROL_TOKEN_FILE = os.environ.get(
"PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token")
ENABLE_MUSIC_MODE = os.environ.get(
"ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"}
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
YUE2_START_TIMEOUT = float(os.environ.get("YUE2_START_TIMEOUT", "600"))
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
VOICE_CHANGE_START_TIMEOUT = float(os.environ.get("VOICE_CHANGE_START_TIMEOUT", "600"))
APPLIO_START_TIMEOUT = float(os.environ.get("APPLIO_START_TIMEOUT", "900"))
TRELLIS_START_TIMEOUT = float(os.environ.get("TRELLIS_START_TIMEOUT", "900"))
# Optional worker APIs. The clean Docker baseline deliberately ships only
# text/multimodal chat; absent workers must fail explicitly instead of trying
@@ -126,7 +135,7 @@ DEFAULT_REASONING_EFFORT = os.environ.get(
GLOBAL_SYSTEM_POLICY_FILE = os.environ.get(
"GLOBAL_SYSTEM_POLICY_FILE", "").strip()
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 4B) ---
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 9B FP8) ---
LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service")
SYSTEMCTL_BIN = os.environ.get("SYSTEMCTL_BIN", "systemctl")
IMAGE_WORKER = os.environ.get(
@@ -135,7 +144,8 @@ IMAGE_PYTHON = os.environ.get(
"IMAGE_PYTHON", "/opt/mike-ai/ai-profile-router/venv/bin/python")
IMAGE_WORKER_URL = os.environ.get("IMAGE_WORKER_URL", "").rstrip("/")
IMAGE_WORKER_TOKEN = os.environ.get("IMAGE_WORKER_TOKEN", "").strip()
IMAGE_MODEL_NAME = os.environ.get("IMAGE_MODEL_NAME", "FLUX.2-klein-4B")
IMAGE_MODEL_NAME = os.environ.get(
"IMAGE_MODEL_NAME", "FLUX.2-klein-9B-fp8-beta")
IMAGE_DIR = os.environ.get(
"IMAGE_DIR", "/opt/mike-ai/ai-profile-router/images")
IMAGE_WORKER_LOG = os.environ.get(
@@ -163,20 +173,20 @@ IMAGE_SIZES = {
"1920x1088": (1920, 1088),
"1088x1920": (1088, 1920),
}
# Das destillierte FLUX.2-klein-4B ist auf vier Schritte ausgelegt.
# Das destillierte FLUX.2 Klein 9B ist auf vier Schritte ausgelegt.
IMAGE_QUALITY = {"standard": 4, "high": 4}
IMAGE_DEFAULT_QUALITY = "standard"
IMAGE_MAX_N = 4
# --- Sprachausgabe (austauschbarer interner TTS-Worker, CPU-only) ---
# --- Sprachausgabe (Qwen3-TTS über das interne Normalisierungs-Gateway) ---
TTS_WORKER_URL = os.environ.get("TTS_WORKER_URL", "http://127.0.0.1:8085")
TTS_TIMEOUT = float(os.environ.get("TTS_TIMEOUT", "300")) # s, pro Synthese
TTS_CONNECT_TIMEOUT = float(os.environ.get("TTS_CONNECT_TIMEOUT", "5"))
TTS_MODEL = os.environ.get("TTS_MODEL", "xtts-v2")
TTS_MODEL = os.environ.get("TTS_MODEL", "qwen3-tts")
TTS_VOICES = tuple(v.strip() for v in os.environ.get(
"TTS_VOICES", "claribel").split(",") if v.strip())
"TTS_VOICES", "alloy").split(",") if v.strip())
TTS_DEFAULT_VOICE = os.environ.get(
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "claribel")
"TTS_DEFAULT_VOICE", TTS_VOICES[0] if TTS_VOICES else "alloy")
TTS_FORMATS = ("mp3", "wav", "pcm")
TTS_DEFAULT_FORMAT = "mp3"
@@ -254,6 +264,7 @@ class _ImageState:
self.last_error: str | None = None
self.last_image: str | None = None
self.last_seconds: float | None = None
self.current_model: str | None = None
IMAGE_PHASES = (
@@ -279,6 +290,9 @@ class _State:
self.qwen_unavailable = True
self.active_chats = 0
self.avail_lock = threading.Lock()
self.mode = "llm"
self.mode_phase = "ready"
self.mode_error: str | None = None
STATE = _State()
@@ -309,6 +323,329 @@ def _set_qwen_unavailable(unavailable: bool) -> None:
STATE.qwen_unavailable = unavailable
def _music_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"music_worker", "missing"))
except Exception as exc:
log.warning("Musik-Worker-Status nicht verfügbar: %s", exc)
return "unknown"
def _music_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"music_health", "unknown"))
except Exception:
return "unknown"
def _separator_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"separator_worker", "missing"))
except Exception as exc:
log.warning("Stem-Separator-Status nicht verfügbar: %s", exc)
return "unknown"
def _separator_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"separator_health", "unknown"))
except Exception:
return "unknown"
def _voice_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_worker", "missing"))
except Exception as exc:
log.warning("Voice-Worker-Status nicht verfügbar: %s", exc)
return "unknown"
def _voice_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_health", "unknown"))
except Exception:
return "unknown"
def _voice_change_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_change_worker", "missing"))
except Exception as exc:
log.warning("Voice-Change-Worker-Status nicht verfügbar: %s", exc)
return "unknown"
def _voice_change_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_change_health", "unknown"))
except Exception:
return "unknown"
def _worker_field(field: str) -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(field, "missing"))
except Exception as exc:
log.warning("Spezial-Worker-Status %s nicht verfügbar: %s", field, exc)
return "unknown"
def _wait_music_ready() -> None:
deadline = time.monotonic() + MUSIC_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("music_worker") == "running"
and status.get("music_health") == "healthy"):
return
if status.get("music_health") == "unhealthy":
raise RuntimeError("ACE-Step-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"ACE-Step nach {MUSIC_START_TIMEOUT:.0f} s nicht bereit")
def _wait_yue2_ready() -> None:
_wait_aux_voice_ready("yue2_worker", "yue2_health",
"YuE2", YUE2_START_TIMEOUT)
def _wait_separator_ready() -> None:
deadline = time.monotonic() + SEPARATOR_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("separator_worker") == "running"
and status.get("separator_health") == "healthy"):
return
if status.get("separator_health") == "unhealthy":
raise RuntimeError("BS-RoFormer-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"BS-RoFormer nach {SEPARATOR_START_TIMEOUT:.0f} s nicht bereit")
def _wait_voice_ready() -> None:
deadline = time.monotonic() + VOICE_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("voice_worker") == "running"
and status.get("voice_health") == "healthy"):
return
if status.get("voice_health") == "unhealthy":
raise RuntimeError("OmniVoice-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"OmniVoice nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
def _wait_voice_change_ready() -> None:
deadline = time.monotonic() + VOICE_CHANGE_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("voice_change_worker") == "running"
and status.get("voice_change_health") == "healthy"):
return
if status.get("voice_change_health") == "unhealthy":
raise RuntimeError("X-VC-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"X-VC nach {VOICE_CHANGE_START_TIMEOUT:.0f} s nicht bereit")
def _wait_aux_voice_ready(worker_field: str, health_field: str,
label: str, timeout: float) -> None:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get(worker_field) == "running"
and status.get(health_field) == "healthy"):
return
if status.get(health_field) == "unhealthy":
raise RuntimeError(f"{label}-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(f"{label} nach {timeout:.0f} s nicht bereit")
def _wait_applio_ready() -> None:
_wait_aux_voice_ready("applio_worker", "applio_health",
"Applio", APPLIO_START_TIMEOUT)
def _wait_trellis_ready() -> None:
_wait_aux_voice_ready("trellis_worker", "trellis_health",
"TRELLIS.2", TRELLIS_START_TIMEOUT)
def _special_worker(mode: str) -> tuple[str, str, callable]:
if mode == "music":
return "/workers/music/start", _music_worker_state(), _wait_music_ready
if mode == "yue2":
return ("/workers/yue2/start", _worker_field("yue2_worker"),
_wait_yue2_ready)
if mode == "separation":
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
if mode == "voice":
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
if mode == "voicechange":
return ("/workers/voice-change/start", _voice_change_worker_state(),
_wait_voice_change_ready)
if mode == "applio":
return ("/workers/applio/start", _worker_field("applio_worker"),
_wait_applio_ready)
if mode == "trellis":
return ("/workers/trellis/start", _worker_field("trellis_worker"),
_wait_trellis_ready)
raise ValueError(f"unbekannter Spezialmodus: {mode}")
def set_operating_mode(mode: str) -> dict:
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
special_modes = {"music", "yue2", "separation", "voice", "voicechange",
"applio", "trellis"}
if mode not in {"llm", *special_modes}:
raise ValueError("unbekannter Betriebsmodus")
with STATE.lock:
STATE.mode_error = None
if mode in special_modes:
path, worker_state, wait_ready = _special_worker(mode)
if STATE.mode == mode and worker_state == "running":
return {"status": "ok", "mode": mode, "changed": False}
profile = current_profile()
previous = RUNTIME.load()
saved = previous.get("return_profile") or previous.get("last_profile")
return_profile = profile if profile in PROFILES else saved
if return_profile not in PROFILES:
return_profile = next(iter(PROFILES))
STATE.mode_phase = f"starting-{mode}"
_set_qwen_unavailable(True)
try:
# Persist intent before stopping anything so a router restart
# during ACE-Step loading can resume the same transition.
RUNTIME.save(mode=mode, return_profile=return_profile,
last_profile=return_profile,
phase=f"starting-{mode}")
_wait_chats_drained()
_profile_controller_request(
"POST", path,
timeout=120)
wait_ready()
STATE.mode = mode
STATE.mode_phase = "ready"
RUNTIME.save(mode=mode, return_profile=return_profile,
last_profile=return_profile, phase=mode)
return {"status": "ok", "mode": mode, "changed": True,
"return_profile": return_profile}
except Exception as exc:
STATE.mode_error = str(exc)
STATE.mode_phase = "error"
raise
previous = RUNTIME.load()
profile = previous.get("return_profile") or previous.get("last_profile")
if profile not in PROFILES:
profile = next(iter(PROFILES))
STATE.mode_phase = "restoring-llm"
_set_qwen_unavailable(True)
try:
_profile_controller_request("POST", "/workers/music/stop")
_profile_controller_request("POST", "/workers/yue2/stop")
_profile_controller_request("POST", "/workers/separator/stop")
_profile_controller_request("POST", "/workers/voice/stop")
_profile_controller_request("POST", "/workers/voice-change/stop")
_profile_controller_request("POST", "/workers/applio/stop")
_profile_controller_request("POST", "/workers/trellis/stop")
_restore_qwen(profile)
STATE.mode = "llm"
STATE.mode_phase = "ready"
_set_qwen_unavailable(False)
RUNTIME.save(mode="llm", return_profile=None,
last_profile=profile, phase="idle")
return {"status": "ok", "mode": "llm", "changed": True,
"profile": profile}
except Exception as exc:
STATE.mode_error = str(exc)
STATE.mode_phase = "error"
raise
def schedule_operating_mode(mode: str) -> tuple[bool, str]:
"""Start a transition in the background so chat/UI acknowledgement is instant."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music", "yue2", "separation", "voice",
"voicechange", "applio", "trellis"}:
raise ValueError("unbekannter Betriebsmodus")
with STATE.lock:
if STATE.mode_phase not in {"ready", "error"}:
return False, STATE.mode_phase
if STATE.mode == mode and STATE.mode_phase == "ready":
return False, "ready"
STATE.mode_phase = f"starting-{mode}" if mode != "llm" else "restoring-llm"
def transition() -> None:
try:
set_operating_mode(mode)
log.info("Betriebsmodus ist jetzt %s", mode)
except Exception:
log.exception("Betriebsmodus-Wechsel zu %s fehlgeschlagen", mode)
threading.Thread(target=transition, name=f"mode-{mode}", daemon=True).start()
return True, STATE.mode_phase
def _control_command(data: dict, path: str) -> str | None:
"""Recognise exact local commands without invoking an LLM."""
text: object = None
if path == "/v1/chat/completions":
messages = data.get("messages")
if isinstance(messages, list):
for item in reversed(messages):
if isinstance(item, dict) and item.get("role") == "user":
text = item.get("content")
break
elif path == "/v1/responses":
text = data.get("input")
if not isinstance(text, str):
return None
command = text.strip().casefold()
return command if command in {"/athena music", "/athena yue2",
"/athena stems",
"/athena separation", "/athena llm",
"/athena voice",
"/athena voicechange", "/athena changer",
"/athena applio",
"/athena 3d", "/athena trellis",
"/athena status"} else None
# ---------------------------------------------------------------------------
# Upstream (llama.cpp)
# ---------------------------------------------------------------------------
@@ -588,7 +925,8 @@ def _read(path: str) -> str:
return f.read().strip()
def _profile_controller_request(method: str, path: str) -> dict:
def _profile_controller_request(method: str, path: str,
timeout: float = 120) -> dict:
token = os.environ.get("PROFILE_CONTROL_TOKEN", "").strip()
if not token:
token = _read(PROFILE_CONTROL_TOKEN_FILE)
@@ -600,7 +938,7 @@ def _profile_controller_request(method: str, path: str) -> dict:
headers={"Authorization": f"Bearer {token}"},
)
try:
with urllib.request.urlopen(request, timeout=120) as response:
with urllib.request.urlopen(request, timeout=timeout) as response:
return json.load(response)
except urllib.error.HTTPError as exc:
body = exc.read(500).decode(errors="replace")
@@ -750,7 +1088,10 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
log.info("Warte, bis llama.cpp das Profil geladen hat ...")
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
RUNTIME.save(last_profile=profile, phase="idle")
STATE.mode = "llm"
STATE.mode_phase = "ready"
RUNTIME.save(last_profile=profile, mode="llm",
return_profile=None, phase="idle")
finally:
# Nach einem fehlgeschlagenen Skript/Timeout darf der Router
# Qwen nicht blind freigeben. Nur ein semantisch verifiziertes
@@ -765,7 +1106,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
# ---------------------------------------------------------------------------
# Bildgenerierung und Editing (FLUX.2-klein-4B)
# Bildgenerierung und Editing (FLUX.2 Klein 9B FP8)
# ---------------------------------------------------------------------------
class _Worker:
@@ -861,7 +1202,10 @@ def _worker() -> _Worker:
if not img.worker or not img.worker.alive():
if img.worker:
img.worker.stop()
img.worker = _RemoteWorker() if IMAGE_WORKER_URL else _Worker()
img.worker = (_RemoteWorker(kind="image", url=IMAGE_WORKER_URL,
token=IMAGE_WORKER_TOKEN,
endpoint="/generate")
if IMAGE_WORKER_URL else _Worker())
img.worker.start()
return img.worker
@@ -871,7 +1215,12 @@ class _RemoteWorker:
model_loaded = False
def __init__(self) -> None:
def __init__(self, *, kind: str, url: str, token: str,
endpoint: str) -> None:
self.kind = kind
self.url = url
self.token = token
self.endpoint = endpoint
self.running = False
def alive(self) -> bool:
@@ -880,10 +1229,10 @@ class _RemoteWorker:
def _request(self, method: str, path: str, payload: dict | None = None,
timeout: float = 120) -> dict:
body = None if payload is None else json.dumps(payload).encode()
headers = {"Authorization": f"Bearer {IMAGE_WORKER_TOKEN}"}
headers = {"Authorization": f"Bearer {self.token}"}
if body is not None:
headers["Content-Type"] = "application/json"
req = urllib.request.Request(IMAGE_WORKER_URL + path, data=body,
req = urllib.request.Request(self.url + path, data=body,
method=method, headers=headers)
try:
with urllib.request.urlopen(req, timeout=timeout) as response:
@@ -898,9 +1247,9 @@ class _RemoteWorker:
raise RuntimeError(f"Bild-Worker nicht erreichbar: {exc}") from exc
def start(self) -> None:
if not IMAGE_WORKER_TOKEN or len(IMAGE_WORKER_TOKEN) < 32:
if not self.token or len(self.token) < 32:
raise RuntimeError("Bild-Worker-Token fehlt oder ist zu kurz")
_profile_controller_request("POST", "/workers/image/start")
_profile_controller_request("POST", f"/workers/{self.kind}/start")
deadline = time.monotonic() + IMAGE_START_TIMEOUT
while time.monotonic() < deadline:
try:
@@ -920,11 +1269,11 @@ class _RemoteWorker:
clean.pop("cmd", None)
output = clean.pop("output", "")
clean["filename"] = os.path.basename(output)
return self._request("POST", "/generate", clean, timeout)
return self._request("POST", self.endpoint, clean, timeout)
def stop(self) -> None:
try:
_profile_controller_request("POST", "/workers/image/stop")
_profile_controller_request("POST", f"/workers/{self.kind}/stop")
finally:
self.running = False
self.model_loaded = False
@@ -1011,6 +1360,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
guidance: float, seed: int | None, n: int,
quality: str = "standard",
source_files: list[str] | None = None,
model: str = IMAGE_MODEL_NAME,
) -> tuple[list[str], str | None]:
"""Orchestriert die Bildgenerierung inkl. Qwen-Hotswap.
@@ -1030,6 +1380,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
results: list[str] = []
warning: str | None = None
img.last_error = None
img.current_model = model
# Qwen wird gestoppt → für Chats nicht verfügbar (die warten).
_set_qwen_unavailable(True)
try:
@@ -1062,7 +1413,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
filename = time.strftime("%Y%m%d-%H%M%S") + \
f"-{os.urandom(2).hex()}.png"
output = os.path.join(IMAGE_DIR, filename)
resp = worker.request({
worker_payload = {
"cmd": "generate",
"prompt": prompt,
"width": width,
@@ -1072,7 +1423,8 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
"seed": seed,
"output": output,
"source_files": source_files or [],
}, timeout=IMAGE_GEN_TIMEOUT)
}
resp = worker.request(worker_payload, timeout=IMAGE_GEN_TIMEOUT)
if resp.get("status") != "ok":
raise RuntimeError(
resp.get("message", "Bildgenerierung fehlgeschlagen"))
@@ -1090,10 +1442,10 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
"steps": steps,
"guidance": guidance,
"quality": quality,
"mode": "image-edit" if source_files else "text-to-image",
"mode": ("image-edit" if source_files else "text-to-image"),
"reference_images": len(source_files or []),
"seconds": resp.get("seconds"),
"model": IMAGE_MODEL_NAME,
"model": model,
"created": time.strftime("%Y-%m-%dT%H:%M:%S"),
}
meta_path = os.path.join(IMAGE_DIR, filename[:-4] + ".json")
@@ -1139,6 +1491,7 @@ def generate_image(prompt: str, width: int, height: int, steps: int,
log.error(warning)
# Qwen ist down → qwen_unavailable bleibt True.
img.phase = "idle"
img.current_model = None
return results, warning
@@ -1497,6 +1850,10 @@ class Handler(BaseHTTPRequestHandler):
self._send_json(200, self._models_payload())
elif path == "/status" and self.command == "GET":
self._send_json(200, self._status_payload())
elif path == "/mode" and self.command == "GET":
self._send_json(200, self._mode_payload())
elif path == "/mode" and self.command == "POST":
self._mode_change()
elif path == "/v1/audio/models" and self.command == "GET":
self._send_json(200, self._audio_models_payload())
elif path == "/v1/audio/voices" and self.command == "GET":
@@ -1519,6 +1876,13 @@ class Handler(BaseHTTPRequestHandler):
else:
self._send_error(503, "Sprachausgabe ist nicht installiert",
"server_error", "feature_disabled")
elif (path == "/v1/audio/speech/pcm-stream"
and self.command == "POST"):
if ENABLE_TTS:
self._speech_pcm_stream()
else:
self._send_error(503, "Sprachausgabe ist nicht installiert",
"server_error", "feature_disabled")
elif path == "/v1/audio/transcriptions" and self.command == "POST":
if ENABLE_STT:
self._transcribe()
@@ -1713,6 +2077,7 @@ class Handler(BaseHTTPRequestHandler):
"current_profile": current_profile(),
"switching": STATE.switching,
"profiles": PROFILES,
"mode": self._mode_payload(),
"upstream": {
"url": UPSTREAM_URL,
"reachable": up["reachable"],
@@ -1734,7 +2099,7 @@ class Handler(BaseHTTPRequestHandler):
"phase": img.phase,
"worker": "running" if (img.worker and img.worker.alive())
else "stopped",
"model": IMAGE_MODEL_NAME if img.phase != "idle" else None,
"model": img.current_model if img.phase != "idle" else None,
"model_loaded": bool(img.worker and img.worker.model_loaded),
"last_image": img.last_image,
"last_seconds": img.last_seconds,
@@ -1744,6 +2109,49 @@ class Handler(BaseHTTPRequestHandler):
"stt": stt_status(),
}
@staticmethod
def _mode_payload() -> dict:
state = RUNTIME.load()
return {
"active": STATE.mode,
"phase": STATE.mode_phase,
"music_worker": _music_worker_state(),
"music_health": _music_worker_health(),
"yue2_worker": _worker_field("yue2_worker"),
"yue2_health": _worker_field("yue2_health"),
"separator_worker": _separator_worker_state(),
"separator_health": _separator_worker_health(),
"voice_worker": _voice_worker_state(),
"voice_health": _voice_worker_health(),
"voice_change_worker": _voice_change_worker_state(),
"voice_change_health": _voice_change_worker_health(),
"applio_worker": _worker_field("applio_worker"),
"applio_health": _worker_field("applio_health"),
"trellis_worker": _worker_field("trellis_worker"),
"trellis_health": _worker_field("trellis_health"),
"return_profile": state.get("return_profile"),
"last_error": STATE.mode_error,
"enabled": ENABLE_MUSIC_MODE,
}
def _mode_change(self) -> None:
try:
data = json.loads(self._read_body() or b"{}")
mode = data.get("mode") if isinstance(data, dict) else None
if mode not in {"llm", "music", "yue2", "separation", "voice",
"voicechange", "applio", "trellis"}:
raise ValueError("Feld 'mode' enthält einen unbekannten Betriebsmodus")
started, phase = schedule_operating_mode(mode)
self._send_json(202 if started else 200, {
"status": "accepted" if started else "ok",
"requested_mode": mode,
"phase": phase,
})
except ValueError as exc:
self._send_error(400, str(exc), "invalid_request_error", "invalid_mode")
except RuntimeError as exc:
self._send_error(503, str(exc), "server_error", "mode_unavailable")
# ---------- Bildgenerierung ----------
def _image_generate(self) -> None:
@@ -1841,6 +2249,13 @@ class Handler(BaseHTTPRequestHandler):
"invalid_request_error", "prompt_too_long")
return
model = data.get("model", IMAGE_MODEL_NAME)
if model != IMAGE_MODEL_NAME:
self._send_error(
400, f"unbekanntes Bildmodell: {model!r}",
"invalid_request_error", "invalid_model")
return
# Größe
size = data.get("size", "1024x1024")
if size not in IMAGE_SIZES:
@@ -1857,7 +2272,6 @@ class Handler(BaseHTTPRequestHandler):
self._send_error(400, f"'n' muss eine Ganzzahl 1..{IMAGE_MAX_N} sein",
"invalid_request_error", "invalid_n")
return
# Qualität / Schritte / Guidance
quality = data.get("quality", IMAGE_DEFAULT_QUALITY)
if quality not in IMAGE_QUALITY:
@@ -1866,8 +2280,9 @@ class Handler(BaseHTTPRequestHandler):
"invalid_request_error", "invalid_quality")
return
steps = data.get("steps", IMAGE_QUALITY[quality])
if not isinstance(steps, int) or isinstance(steps, bool) or steps != 4:
self._send_error(400, "FLUX.2-klein-4B erfordert 'steps'=4",
if (not isinstance(steps, int) or isinstance(steps, bool)
or steps != 4):
self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'steps'=4",
"invalid_request_error", "invalid_steps")
return
guidance = data.get("guidance", 1.0)
@@ -1878,7 +2293,7 @@ class Handler(BaseHTTPRequestHandler):
"invalid_request_error", "invalid_guidance")
return
if guidance != 1.0:
self._send_error(400, "FLUX.2-klein-4B erfordert 'guidance'=1.0",
self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'guidance'=1.0",
"invalid_request_error", "invalid_guidance")
return
@@ -1906,7 +2321,7 @@ class Handler(BaseHTTPRequestHandler):
try:
results, warning = generate_image(
prompt.strip(), width, height, steps, guidance, seed, n,
quality, source_files)
quality, source_files, model)
except (ValueError, RuntimeError) as e:
self._send_error(503, str(e), "server_error", "image_generation_failed")
return
@@ -1979,7 +2394,7 @@ class Handler(BaseHTTPRequestHandler):
self.end_headers()
self.wfile.write(data)
# ---------- Sprachausgabe (XTTS-v2) ----------
# ---------- Sprachausgabe (Qwen3-TTS) ----------
def _speech(self) -> None:
try:
@@ -2038,7 +2453,7 @@ class Handler(BaseHTTPRequestHandler):
"invalid_request_error", "invalid_speed")
return
# Modell-Name optional; falls angegeben, muss es xtts-v2 sein.
# Modell-Name optional; falls angegeben, muss es Qwen3-TTS sein.
model = data.get("model")
if model is not None and model != TTS_MODEL:
self._send_error(400, f"unbekanntes Modell: {model!r} "
@@ -2062,6 +2477,74 @@ class Handler(BaseHTTPRequestHandler):
self.end_headers()
self.wfile.write(audio)
def _speech_pcm_stream(self) -> None:
"""Pass through Qwen's native 24 kHz PCM stream without buffering."""
try:
body = self._read_body()
data = json.loads(body)
except ValueError as exc:
self._send_error(400, str(exc) or "ungültiges JSON",
"invalid_request_error", "invalid_body")
return
if not isinstance(data, dict):
self._send_error(400, "Request muss ein JSON-Objekt sein",
"invalid_request_error", "invalid_request")
return
text = data.get("input", data.get("text"))
if not isinstance(text, str) or not text.strip() or len(text) > 8000:
self._send_error(400, "'input' fehlt, ist leer oder zu lang",
"invalid_request_error", "invalid_input")
return
voice = data.get("voice", TTS_DEFAULT_VOICE)
if voice not in TTS_VOICES:
self._send_error(400, f"ungültige Stimme: {voice!r}",
"invalid_request_error", "invalid_voice")
return
hostport = TTS_WORKER_URL.split("://", 1)[-1]
host, _, port = hostport.partition(":")
connection = None
headers_sent = False
try:
connection = http.client.HTTPConnection(
host, int(port) if port else 80, timeout=TTS_CONNECT_TIMEOUT)
connection.request(
"POST", "/tts/pcm-stream", body=body,
headers={"Content-Type": "application/json",
"Accept": "application/octet-stream"})
connection.sock.settimeout(TTS_TIMEOUT)
response = connection.getresponse()
if response.status != 200:
message = response.read(512).decode(errors="replace")
self._send_error(503,
f"TTS-Stream fehlgeschlagen: {message}",
"server_error", "tts_failed")
return
self._last_code = 200
self.send_response(200)
self.send_header("Content-Type", "application/octet-stream")
self.send_header("Cache-Control", "no-store")
self.send_header("Connection", "close")
self.end_headers()
headers_sent = True
while True:
chunk = response.read1(16384)
if not chunk:
break
self.wfile.write(chunk)
self.wfile.flush()
except (OSError, http.client.HTTPException) as exc:
if not headers_sent:
try:
self._send_error(502, f"TTS-Worker nicht erreichbar: {exc}",
"server_error", "tts_unavailable")
except (OSError, BrokenPipeError):
pass
finally:
if connection is not None:
connection.close()
self.close_connection = True
# ---------- Audio-Discovery ----------
def _audio_models_payload(self) -> dict:
@@ -2080,7 +2563,7 @@ class Handler(BaseHTTPRequestHandler):
models.append({
"id": TTS_MODEL,
"object": "model",
"owned_by": "coqui-xtts",
"owned_by": "qwen",
"type": "speech",
})
return {"object": "list", "data": models}
@@ -2269,6 +2752,12 @@ class Handler(BaseHTTPRequestHandler):
data = json.loads(body)
except ValueError:
data = None
if (isinstance(data, dict)
and path in {"/v1/chat/completions", "/v1/responses"}):
command = _control_command(data, path)
if command is not None:
self._control_response(command, data, path)
return
model = data.get("model") if isinstance(data, dict) else None
if (isinstance(model, str) and REVIEW_UPSTREAM_URL
and model == REVIEW_MODEL_NAME):
@@ -2306,6 +2795,97 @@ class Handler(BaseHTTPRequestHandler):
# An llama.cpp weiterleiten (mit Chat-Waiting, Streaming bleibt erhalten).
self._proxy_with_wait(body)
def _control_response(self, command: str, data: dict, path: str) -> None:
"""Return OpenAI-compatible local replies for Athena control commands."""
if command == "/athena status":
mode = self._mode_payload()
profile = current_profile()
text = (f"Athena läuft im {mode['active'].upper()}-Modus. "
f"Phase: {mode['phase']}. Musik-Worker: "
f"{mode['music_worker']}. YuE2: "
f"{mode['yue2_worker']}. Stem-Separator: "
f"{mode['separator_worker']}. Voice Studio: "
f"{mode['voice_worker']}. Voice Changer: "
f"{mode['voice_change_worker']}. 3D Studio: "
f"{mode['trellis_worker']}. LLM-Profil: {profile or 'entladen'}.")
else:
target = ("music" if command == "/athena music" else
"yue2" if command == "/athena yue2" else
"separation" if command in {"/athena stems", "/athena separation"}
else "voice" if command == "/athena voice"
else "voicechange" if command in {"/athena voicechange", "/athena changer"}
else "applio" if command == "/athena applio"
else "trellis" if command in {"/athena 3d", "/athena trellis"}
else "llm")
try:
started, phase = schedule_operating_mode(target)
if started:
text = ("Musikstudio wird gestartet. LLM und TTS werden entladen."
if target == "music" else
"YuE2 Studio wird gestartet. LLM und TTS werden entladen."
if target == "yue2" else
"Stimmtrennung wird gestartet. LLM und TTS werden entladen."
if target == "separation" else
"Voice Studio wird gestartet. LLM und TTS werden entladen."
if target == "voice" else
"Voice Changer wird gestartet. LLM und TTS werden entladen."
if target == "voicechange" else
"Applio wird gestartet. LLM und TTS werden entladen."
if target == "applio" else
"3D Studio wird gestartet. LLM und TTS werden entladen."
if target == "trellis" else
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
else:
text = (f"Athena ist bereits im {target.upper()}-Modus "
f"oder wechselt gerade ({phase}).")
except RuntimeError as exc:
self._send_error(503, str(exc), "server_error", "mode_unavailable")
return
model = str(data.get("model") or "athena-control")
created = int(time.time())
request_id = f"athena-mode-{uuid.uuid4().hex[:16]}"
if path == "/v1/responses":
self._send_json(200, {
"id": request_id, "object": "response", "created_at": created,
"status": "completed", "model": model,
"output": [{"type": "message", "role": "assistant",
"content": [{"type": "output_text", "text": text}]}],
"output_text": text,
"usage": {"input_tokens": 0, "output_tokens": 0,
"total_tokens": 0},
})
return
if data.get("stream") is True:
chunks = [
{"id": request_id, "object": "chat.completion.chunk",
"created": created, "model": model,
"choices": [{"index": 0, "delta": {"role": "assistant",
"content": text}, "finish_reason": None}]},
{"id": request_id, "object": "chat.completion.chunk",
"created": created, "model": model,
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]},
]
body = "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks)
body += "data: [DONE]\n\n"
encoded = body.encode()
self._last_code = 200
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Content-Length", str(len(encoded)))
self.send_header("Connection", "close")
self.end_headers()
self.wfile.write(encoded)
return
self._send_json(200, {
"id": request_id, "object": "chat.completion", "created": created,
"model": model,
"choices": [{"index": 0, "message": {"role": "assistant",
"content": text}, "finish_reason": "stop"}],
"usage": {"prompt_tokens": 0, "completion_tokens": 0,
"total_tokens": 0},
})
def _acquire_model_lease(self, profile: str | None = None) -> dict:
"""Atomar Profil sicherstellen und einen aktiven Request registrieren."""
with STATE.lock:
@@ -2545,6 +3125,29 @@ def _startup_reconcile() -> None:
if removed:
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
special_mode = previous.get("mode")
if ENABLE_MUSIC_MODE and special_mode in {"music", "yue2", "separation",
"voice", "voicechange", "applio",
"trellis"}:
STATE.mode = special_mode
STATE.mode_phase = f"starting-{special_mode}"
_set_qwen_unavailable(True)
try:
path, _worker_state, wait_ready = _special_worker(special_mode)
_profile_controller_request(
"POST", path,
timeout=120)
wait_ready()
STATE.mode_phase = "ready"
RUNTIME.save(mode=special_mode, phase=special_mode)
log.info("Recovery: Spezialmodus %s wiederhergestellt", special_mode)
except Exception as exc:
STATE.mode_error = str(exc)
STATE.mode_phase = "error"
log.error("Recovery: Spezialmodus %s konnte nicht gestartet werden: %s",
special_mode, exc)
return
profile = current_profile()
if profile is None:
saved = previous.get("last_profile")
@@ -2568,7 +3171,9 @@ def _startup_reconcile() -> None:
and (not EXPECTED_MODELS.get(profile)
or up.get("model") == EXPECTED_MODELS[profile])):
_set_qwen_unavailable(False)
RUNTIME.save(last_profile=profile, phase="idle")
STATE.mode = "llm"
STATE.mode_phase = "ready"
RUNTIME.save(last_profile=profile, mode="llm", phase="idle")
log.info("Recovery: Profil %s ist bereits bereit", profile)
return
_set_qwen_unavailable(True)