From f58d61140e632e87ca59a0f08649380beae12ad0 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Tue, 8 Sep 2026 17:32:04 +0200 Subject: [PATCH] Add persistent Athena music mode switching --- ATHENA.md | 4 + README.md | 5 + compose.yaml | 4 + dev/test_profile_controller.py | 31 ++ docs/OPERATING_MODES.md | 45 +++ experiments/acestep15-xl-sft/compose.yaml | 2 + .../profile-controller/profile_controller.py | 53 ++++ platform/llama-dashboard/app.py | 51 +++- router/ai_profile_router.py | 277 +++++++++++++++++- 9 files changed, 467 insertions(+), 5 deletions(-) create mode 100644 docs/OPERATING_MODES.md diff --git a/ATHENA.md b/ATHENA.md index 224db31..c193d79 100644 --- a/ATHENA.md +++ b/ATHENA.md @@ -12,6 +12,7 @@ Sie betreibt: - den OpenAI-kompatiblen Profile Router, - FLUX.2 Klein 9B FP8 Beta für Textbilder und Referenzbild-Bearbeitung, - Qwen3-TTS und Piper für Sprache, +- ACE-Step 1.5 XL-SFT als exklusiven Musikstudio-Modus, - das Athena-Dashboard, - Portainer CE als optionale Ansicht auf die laufenden Docker-Container, - WireGuard-Gateway und Datenbackup, @@ -68,6 +69,9 @@ Qwen-Profil wird vom Profile Controller verwaltet. - Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback. Der Router reicht zusätzlich natives 24-kHz-PCM für den optionalen Hermes-Streaming-Adapter unter `integrations/hermes-qwen3-stream` durch. +- ACE-Step 1.5 XL-SFT: exklusiver Musikmodus auf der RTX 5080. Dashboard und + die Routerbefehle `/athena music`, `/athena llm`, `/athena status` bedienen + dieselbe persistente Zustandsmaschine; siehe `docs/OPERATING_MODES.md`. Die verbindlichen Werte stehen in `config/profile-matrix.json` und `docs/STANDARD_PROFILE_MATRIX.md`. diff --git a/README.md b/README.md index 0a66f1f..a11cc37 100644 --- a/README.md +++ b/README.md @@ -15,6 +15,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.** - Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback - Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung - Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 +- Dashboard-Umschaltung zwischen LLM-Betrieb und ACE-Step-Musikstudio - Portainer CE als optionale Container-Ansicht auf Port 9443 - WireGuard-Gateway, Datenbackup und Athena-Operator - keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz @@ -121,6 +122,10 @@ Details, Installation, Prüfung und Rollback stehen in - Router: `http://192.168.1.212:8081/v1` - Athena-Dashboard: `http://192.168.1.212:8099` +Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren +außerdem `/athena music`, `/athena llm` und `/athena status`; Details stehen in +[docs/OPERATING_MODES.md](docs/OPERATING_MODES.md). + Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über `/v1/audio/speech`, natives Qwen-PCM-Streaming über `/v1/audio/speech/pcm-stream` und Spracherkennung über diff --git a/compose.yaml b/compose.yaml index 30cb741..97919ad 100644 --- a/compose.yaml +++ b/compose.yaml @@ -569,6 +569,7 @@ services: IMAGE_WORKER: image RESTORE_WORKER: restore TTS_WORKER: qwen3 + MUSIC_WORKER: acestep networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: @@ -625,6 +626,8 @@ services: TTS_VOICES: alloy TTS_DEFAULT_VOICE: alloy ENABLE_STT: "true" + ENABLE_MUSIC_MODE: "true" + MUSIC_START_TIMEOUT: "600" STT_WORKER_URL: http://whisper:8084 STT_TIMEOUT: "300" networks: [frontend, control, inference] @@ -853,6 +856,7 @@ services: DASHBOARD_PORT: "8099" ROUTER_URL: http://router:8081 ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" + MUSIC_UI_URL: "${MUSIC_UI_URL:-http://127.0.0.1:7861/}" HOST_PROC: /host/proc HOST_DATA: /host/data HOST_MODELS: /host/models diff --git a/dev/test_profile_controller.py b/dev/test_profile_controller.py index 96d9ab8..c84106d 100644 --- a/dev/test_profile_controller.py +++ b/dev/test_profile_controller.py @@ -36,7 +36,38 @@ def tts_item(state="running"): "Labels": {controller.TTS_LABEL_KEY: controller.TTS_WORKER}} +def music_item(state="exited"): + return {"Id": "id-music", "State": state, + "Labels": {controller.MUSIC_LABEL_KEY: "acestep"}} + + class ProfileControllerTests(unittest.TestCase): + def test_music_start_exclusively_stops_gpu_workers(self): + profiles = {name: item(name) for name in controller.ALLOWED} + profiles["ultra"] = item("ultra", "running") + calls = [] + + def request(method, path): + calls.append((method, path)) + return 204, b"" + + with patch.object(controller, "MUSIC_WORKER", "acestep"), \ + patch.object(controller, "containers", return_value=profiles), \ + patch.object(controller, "music_container", return_value=music_item()), \ + patch.object(controller, "image_containers", + return_value=[image_item("running")]), \ + patch.object(controller, "tts_container", return_value=tts_item()), \ + patch.object(controller, "docker_request", side_effect=request): + result = controller.set_music_worker(True) + + self.assertEqual(result, {"music_worker": "acestep", "state": "running"}) + self.assertEqual(calls, [ + ("POST", "/containers/id-ultra/stop?t=120"), + ("POST", "/containers/id-flux/stop?t=20"), + ("POST", "/containers/id-tts/stop?t=30"), + ("POST", "/containers/id-music/start"), + ]) + def test_rejects_unknown_profile_before_docker_call(self): with patch.object(controller, "docker_request") as request: with self.assertRaises(ValueError): diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md new file mode 100644 index 0000000..f917bd7 --- /dev/null +++ b/docs/OPERATING_MODES.md @@ -0,0 +1,45 @@ +# Athena-Betriebsmodi + +Athena besitzt zwei gegenseitig exklusive Betriebsmodi: + +- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; ACE-Step ist gestoppt. +- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild- und TTS-Worker sind gestoppt. + +Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie +Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus +wird persistent gespeichert und beim Verlassen des Musikmodus wieder geladen. + +## Bedienung + +Im Athena-Dashboard stehen die Schaltflächen **LLM-Betrieb** und +**Musikstudio** bereit. Nach dem Start des Musikmodus öffnet **Studio öffnen** +die über den SSH-Tunnel bereitgestellte ACE-Step-Oberfläche. + +Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom +Router lokal beantwortet, auch wenn gerade kein LLM geladen ist: + +```text +/athena music +/athena llm +/athena status +``` + +Die HTTP-Schnittstelle verwendet authentifizierte Requests: + +```text +GET /mode +POST /mode {"mode":"music"} +POST /mode {"mode":"llm"} +``` + +Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in +`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit +`com.mike-ai.music-worker=acestep` markierten Container; freie Container- oder +Docker-Befehle werden nicht entgegengenommen. + +## Wiederanlauf + +Der Router speichert `mode`, `last_profile` und `return_profile` atomar. War +beim Router-Neustart der Musikmodus aktiv, startet er ACE-Step erneut. Beim +Wechsel zurück wird das gespeicherte LLM-Profil semantisch auf Alias und +Kontextfenster geprüft, bevor Chat-Anfragen wieder freigegeben werden. diff --git a/experiments/acestep15-xl-sft/compose.yaml b/experiments/acestep15-xl-sft/compose.yaml index ba7f2fa..c11ec1e 100644 --- a/experiments/acestep15-xl-sft/compose.yaml +++ b/experiments/acestep15-xl-sft/compose.yaml @@ -2,6 +2,8 @@ services: music-worker: image: ghcr.io/ace-step/ace-step-1.5:latest@sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567 container_name: mike-ai-music-acestep-test + labels: + com.mike-ai.music-worker: "acestep" profiles: ["music-test"] environment: ACESTEP_MODE: gradio diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index 057d571..900b080 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -28,6 +28,8 @@ IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image") RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore") TTS_LABEL_KEY = "com.mike-ai.tts-worker" TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3") +MUSIC_LABEL_KEY = "com.mike-ai.music-worker" +MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip() LOCK = threading.Lock() log = logging.getLogger("profile-controller") @@ -96,6 +98,22 @@ def tts_container() -> dict: return matches[0] +def music_container() -> dict: + if not MUSIC_WORKER: + raise RuntimeError("music worker is not configured") + matches = [item for item in labelled_containers(MUSIC_LABEL_KEY) + if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == MUSIC_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one music worker {MUSIC_WORKER!r}, found {len(matches)}") + return matches[0] + + +def stop_music_if_configured() -> None: + if MUSIC_WORKER: + stop_container(music_container(), timeout=30) + + def stop_container(item: dict, timeout: int = 120) -> None: if item.get("State") != "running": return @@ -133,6 +151,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: # The 9B beta text encoder temporarily borrows the RTX 3060 from # Qwen3-TTS. The gateway retains Piper as a fallback meanwhile. stop_container(tts_container(), timeout=30) + stop_music_if_configured() for other in image_containers(): if other["Id"] != item["Id"]: stop_container(other, timeout=20) @@ -148,6 +167,23 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: "state": "running" if running else "stopped"} +def set_music_worker(running: bool) -> dict: + """Start ACE-Step exclusively, or stop it before LLM restoration.""" + with LOCK: + item = music_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + start_container(item) + else: + stop_container(item, timeout=30) + return {"music_worker": MUSIC_WORKER, + "state": "running" if running else "stopped"} + + def active_profile(items: dict[str, dict] | None = None) -> str | None: items = items or containers() active = [name for name, item in items.items() if item.get("State") == "running"] @@ -163,6 +199,7 @@ def activate(profile: str) -> dict: # Defensive mutual exclusion even if a caller bypasses the router. for worker in image_containers(): stop_container(worker) + stop_music_if_configured() start_container(tts_container()) items = containers() missing = [name for name in ALLOWED if name not in items] @@ -227,7 +264,16 @@ class Handler(BaseHTTPRequestHandler): return try: items = containers() + music = music_container() if MUSIC_WORKER else {} + music_status = music.get("Status", "") + music_health = ("disabled" if not MUSIC_WORKER else + "healthy" if "(healthy)" in music_status else + "unhealthy" if "(unhealthy)" in music_status else + "starting" if music.get("State") == "running" else + "stopped") self.reply(200, {"active_profile": active_profile(items), + "music_worker": music.get("State", "disabled"), + "music_health": music_health, "profiles": {name: items.get(name, {}).get( "State", "missing") for name in ALLOWED}}) except Exception as exc: @@ -245,6 +291,13 @@ class Handler(BaseHTTPRequestHandler): log.exception("stopping inference failed") self.reply(503, {"error": str(exc)}) return + if self.path in {"/workers/music/start", "/workers/music/stop"}: + try: + self.reply(200, set_music_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("music worker transition failed") + self.reply(503, {"error": str(exc)}) + return worker_paths = { "/workers/image/start": (IMAGE_WORKER, True), "/workers/image/stop": (IMAGE_WORKER, False), diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py index 4026ef6..4ba42f5 100644 --- a/platform/llama-dashboard/app.py +++ b/platform/llama-dashboard/app.py @@ -20,6 +20,7 @@ HOST = os.getenv("DASHBOARD_HOST", "0.0.0.0") PORT = int(os.getenv("DASHBOARD_PORT", "8099")) ROUTER_URL = os.getenv("ROUTER_URL", "http://router:8081").rstrip("/") ROUTER_API_KEY = os.getenv("ROUTER_API_KEY", "") +MUSIC_UI_URL = os.getenv("MUSIC_UI_URL", "http://127.0.0.1:7861/") HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) HOST_DATA = os.getenv("HOST_DATA", "/host/data") HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) @@ -268,6 +269,27 @@ def router_status() -> tuple[dict[str, Any], str | None]: return {}, str(exc) +def change_mode(mode: str) -> tuple[int, dict[str, Any]]: + if mode not in {"llm", "music"}: + return 400, {"error": "invalid mode"} + headers = {"Accept": "application/json", "Content-Type": "application/json"} + if ROUTER_API_KEY: + headers["Authorization"] = f"Bearer {ROUTER_API_KEY}" + request = urllib.request.Request( + f"{ROUTER_URL}/mode", data=json.dumps({"mode": mode}).encode(), + headers=headers, method="POST") + try: + with urllib.request.urlopen(request, timeout=15) as response: + return response.status, json.load(response) + except urllib.error.HTTPError as exc: + try: + return exc.code, json.loads(exc.read()) + except (ValueError, json.JSONDecodeError): + return exc.code, {"error": str(exc)} + except (OSError, urllib.error.URLError) as exc: + return 503, {"error": str(exc)} + + _MODEL_LOCK = threading.Lock() _MODEL_AT = 0.0 _MODEL_CACHE: tuple[list[dict[str, Any]], dict[str, Any]] = ([], {"count": 0, "total_size": 0}) @@ -602,9 +624,11 @@ HTML = r''' .span4 .metrics{grid-template-columns:repeat(2,1fr)} .history-controls{display:flex;flex-wrap:wrap;gap:7px;margin:10px 0 14px}.history-controls button{border:1px solid var(--line);background:#09111b;color:var(--muted);padding:6px 10px;border-radius:8px;cursor:pointer}.history-controls button.active{color:var(--cyan);border-color:var(--cyan)}.chart-legend{display:flex;flex-wrap:wrap;gap:8px;margin:2px 0 10px}.chart-legend button{border:1px solid var(--series);background:#09111b;color:var(--text);padding:6px 10px;border-radius:8px;cursor:pointer}.chart-legend button::before{content:'';display:inline-block;width:10px;height:3px;background:var(--series);margin:0 7px 3px 0}.chart-legend button.off{opacity:.4;text-decoration:line-through}.chart{width:100%;height:250px;display:block}.token-total{font-size:24px;font-weight:750;margin-top:5px}.history-note{color:var(--muted);font-size:11px;margin-top:8px} .usage-list{display:grid;gap:13px;margin-top:8px}.usage-head{display:flex;justify-content:space-between;gap:14px;align-items:baseline}.usage-head b{font-size:16px}.usage-head span{color:var(--muted)}.usage-meta{display:flex;justify-content:space-between;gap:12px;color:var(--muted);font-size:11px;margin-top:5px} +.mode-row{display:flex;align-items:center;justify-content:space-between;gap:18px;flex-wrap:wrap}.mode-buttons{display:flex;gap:9px;flex-wrap:wrap}.mode-buttons button,.mode-buttons a{border:1px solid var(--line);background:#09111b;color:var(--text);padding:9px 14px;border-radius:9px;cursor:pointer;text-decoration:none;font:inherit}.mode-buttons button.active{border-color:var(--cyan);color:var(--cyan);box-shadow:inset 0 0 0 1px #45d7ff33}.mode-buttons button:disabled{opacity:.45;cursor:wait}.mode-buttons a[hidden]{display:none}.mode-error{color:var(--red)}
Mike AI · Live Telemetry

Athena llama.cpp Dashboard

verbinde …
+
Athena Betriebsmodus
–
Status wird geladen …
Aktives Profil
–
Router wird abgefragt
Modell
–
–
CPU
–
–
@@ -637,13 +661,15 @@ HTML = r'''
llama.cpp Laufzeitkonfiguration
–wird gelesen
Verfügbare GGUF-Dateien
–
DateiPfadGrößeGeändert
–
-
+
''' +let modeBusy=false; +async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}} +async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready';$('operatingMode').textContent=md.active==='music'?'Musikstudio':'LLM-Betrieb';$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik-Worker: ${md.music_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('llmMode').disabled=$('musicMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'
Keine GPU-Daten verfügbar
';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}${p.name}${p.pid}${p.memory_mib??'–'} MiB`).join('')||'Keine Compute-Prozesse gemeldet';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`
${v??'–'}${k}
`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000); +'''.replace("__MUSIC_UI_URL__", MUSIC_UI_URL) FULL_JS = r''' @@ -785,6 +811,25 @@ class Handler(BaseHTTPRequestHandler): else: self._send(404, b'{"error":"not found"}', "application/json") + def do_POST(self) -> None: + path = self.path.split("?", 1)[0] + if path != "/api/mode": + self._send(404, b'{"error":"not found"}', "application/json") + return + try: + length = int(self.headers.get("Content-Length", "0")) + if length <= 0 or length > 1024: + raise ValueError("invalid body size") + payload = json.loads(self.rfile.read(length)) + mode = payload.get("mode") if isinstance(payload, dict) else None + except (ValueError, json.JSONDecodeError): + self._send(400, b'{"error":"invalid request"}', "application/json") + return + status, response = change_mode(mode) + body = json.dumps(response, ensure_ascii=False, + separators=(",", ":")).encode() + self._send(status, body, "application/json; charset=utf-8") + if __name__ == "__main__": threading.Thread(target=history_collector, name="history-collector", daemon=True).start() diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index 0bbd3a1..13fcc24 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -106,6 +106,9 @@ PROFILE_DIR = os.environ.get( PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/") PROFILE_CONTROL_TOKEN_FILE = os.environ.get( "PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token") +ENABLE_MUSIC_MODE = os.environ.get( + "ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"} +MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600")) # Optional worker APIs. The clean Docker baseline deliberately ships only # text/multimodal chat; absent workers must fail explicitly instead of trying @@ -282,6 +285,9 @@ class _State: self.qwen_unavailable = True self.active_chats = 0 self.avail_lock = threading.Lock() + self.mode = "llm" + self.mode_phase = "ready" + self.mode_error: str | None = None STATE = _State() @@ -312,6 +318,143 @@ def _set_qwen_unavailable(unavailable: bool) -> None: STATE.qwen_unavailable = unavailable +def _music_worker_state() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "music_worker", "missing")) + except Exception as exc: + log.warning("Musik-Worker-Status nicht verfügbar: %s", exc) + return "unknown" + + +def _music_worker_health() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "music_health", "unknown")) + except Exception: + return "unknown" + + +def _wait_music_ready() -> None: + deadline = time.monotonic() + MUSIC_START_TIMEOUT + while time.monotonic() < deadline: + status = _profile_controller_request("GET", "/status") + if (status.get("music_worker") == "running" + and status.get("music_health") == "healthy"): + return + if status.get("music_health") == "unhealthy": + raise RuntimeError("ACE-Step-Container ist unhealthy") + time.sleep(POLL_INTERVAL) + raise RuntimeError( + f"ACE-Step nach {MUSIC_START_TIMEOUT:.0f} s nicht bereit") + + +def set_operating_mode(mode: str) -> dict: + """Atomarer Wechsel zwischen llama.cpp/TTS und ACE-Step Studio.""" + if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL: + raise RuntimeError("Musikmodus ist nicht konfiguriert") + if mode not in {"llm", "music"}: + raise ValueError("Modus muss 'llm' oder 'music' sein") + with STATE.lock: + STATE.mode_error = None + if mode == "music": + if STATE.mode == "music" and _music_worker_state() == "running": + return {"status": "ok", "mode": "music", "changed": False} + profile = current_profile() + saved = RUNTIME.load().get("last_profile") + return_profile = profile if profile in PROFILES else saved + if return_profile not in PROFILES: + return_profile = next(iter(PROFILES)) + STATE.mode_phase = "starting-music" + _set_qwen_unavailable(True) + try: + # Persist intent before stopping anything so a router restart + # during ACE-Step loading can resume the same transition. + RUNTIME.save(mode="music", return_profile=return_profile, + last_profile=return_profile, + phase="starting-music") + _wait_chats_drained() + _profile_controller_request("POST", "/workers/music/start") + _wait_music_ready() + STATE.mode = "music" + STATE.mode_phase = "ready" + RUNTIME.save(mode="music", return_profile=return_profile, + last_profile=return_profile, phase="music") + return {"status": "ok", "mode": "music", "changed": True, + "return_profile": return_profile} + except Exception as exc: + STATE.mode_error = str(exc) + STATE.mode_phase = "error" + raise + + previous = RUNTIME.load() + profile = previous.get("return_profile") or previous.get("last_profile") + if profile not in PROFILES: + profile = next(iter(PROFILES)) + STATE.mode_phase = "restoring-llm" + _set_qwen_unavailable(True) + try: + _profile_controller_request("POST", "/workers/music/stop") + _restore_qwen(profile) + STATE.mode = "llm" + STATE.mode_phase = "ready" + _set_qwen_unavailable(False) + RUNTIME.save(mode="llm", return_profile=None, + last_profile=profile, phase="idle") + return {"status": "ok", "mode": "llm", "changed": True, + "profile": profile} + except Exception as exc: + STATE.mode_error = str(exc) + STATE.mode_phase = "error" + raise + + +def schedule_operating_mode(mode: str) -> tuple[bool, str]: + """Start a transition in the background so chat/UI acknowledgement is instant.""" + if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL: + raise RuntimeError("Musikmodus ist nicht konfiguriert") + if mode not in {"llm", "music"}: + raise ValueError("Modus muss 'llm' oder 'music' sein") + with STATE.lock: + if STATE.mode_phase not in {"ready", "error"}: + return False, STATE.mode_phase + if STATE.mode == mode and STATE.mode_phase == "ready": + return False, "ready" + STATE.mode_phase = "starting-music" if mode == "music" else "restoring-llm" + + def transition() -> None: + try: + set_operating_mode(mode) + log.info("Betriebsmodus ist jetzt %s", mode) + except Exception: + log.exception("Betriebsmodus-Wechsel zu %s fehlgeschlagen", mode) + + threading.Thread(target=transition, name=f"mode-{mode}", daemon=True).start() + return True, STATE.mode_phase + + +def _control_command(data: dict, path: str) -> str | None: + """Recognise exact local commands without invoking an LLM.""" + text: object = None + if path == "/v1/chat/completions": + messages = data.get("messages") + if isinstance(messages, list): + for item in reversed(messages): + if isinstance(item, dict) and item.get("role") == "user": + text = item.get("content") + break + elif path == "/v1/responses": + text = data.get("input") + if not isinstance(text, str): + return None + command = text.strip().casefold() + return command if command in {"/athena music", "/athena llm", "/athena status"} else None + + # --------------------------------------------------------------------------- # Upstream (llama.cpp) # --------------------------------------------------------------------------- @@ -753,7 +896,10 @@ def switch_profile(profile: str, implicit: bool = False) -> None: f"Profildatei wurde nicht gesetzt (erwartet: {profile})") log.info("Warte, bis llama.cpp das Profil geladen hat ...") _wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT) - RUNTIME.save(last_profile=profile, phase="idle") + STATE.mode = "llm" + STATE.mode_phase = "ready" + RUNTIME.save(last_profile=profile, mode="llm", + return_profile=None, phase="idle") finally: # Nach einem fehlgeschlagenen Skript/Timeout darf der Router # Qwen nicht blind freigeben. Nur ein semantisch verifiziertes @@ -1512,6 +1658,10 @@ class Handler(BaseHTTPRequestHandler): self._send_json(200, self._models_payload()) elif path == "/status" and self.command == "GET": self._send_json(200, self._status_payload()) + elif path == "/mode" and self.command == "GET": + self._send_json(200, self._mode_payload()) + elif path == "/mode" and self.command == "POST": + self._mode_change() elif path == "/v1/audio/models" and self.command == "GET": self._send_json(200, self._audio_models_payload()) elif path == "/v1/audio/voices" and self.command == "GET": @@ -1735,6 +1885,7 @@ class Handler(BaseHTTPRequestHandler): "current_profile": current_profile(), "switching": STATE.switching, "profiles": PROFILES, + "mode": self._mode_payload(), "upstream": { "url": UPSTREAM_URL, "reachable": up["reachable"], @@ -1766,6 +1917,36 @@ class Handler(BaseHTTPRequestHandler): "stt": stt_status(), } + @staticmethod + def _mode_payload() -> dict: + state = RUNTIME.load() + return { + "active": STATE.mode, + "phase": STATE.mode_phase, + "music_worker": _music_worker_state(), + "music_health": _music_worker_health(), + "return_profile": state.get("return_profile"), + "last_error": STATE.mode_error, + "enabled": ENABLE_MUSIC_MODE, + } + + def _mode_change(self) -> None: + try: + data = json.loads(self._read_body() or b"{}") + mode = data.get("mode") if isinstance(data, dict) else None + if mode not in {"llm", "music"}: + raise ValueError("Feld 'mode' muss 'llm' oder 'music' sein") + started, phase = schedule_operating_mode(mode) + self._send_json(202 if started else 200, { + "status": "accepted" if started else "ok", + "requested_mode": mode, + "phase": phase, + }) + except ValueError as exc: + self._send_error(400, str(exc), "invalid_request_error", "invalid_mode") + except RuntimeError as exc: + self._send_error(503, str(exc), "server_error", "mode_unavailable") + # ---------- Bildgenerierung ---------- def _image_generate(self) -> None: @@ -2366,6 +2547,12 @@ class Handler(BaseHTTPRequestHandler): data = json.loads(body) except ValueError: data = None + if (isinstance(data, dict) + and path in {"/v1/chat/completions", "/v1/responses"}): + command = _control_command(data, path) + if command is not None: + self._control_response(command, data, path) + return model = data.get("model") if isinstance(data, dict) else None if (isinstance(model, str) and REVIEW_UPSTREAM_URL and model == REVIEW_MODEL_NAME): @@ -2403,6 +2590,74 @@ class Handler(BaseHTTPRequestHandler): # An llama.cpp weiterleiten (mit Chat-Waiting, Streaming bleibt erhalten). self._proxy_with_wait(body) + def _control_response(self, command: str, data: dict, path: str) -> None: + """Return OpenAI-compatible local replies for Athena control commands.""" + if command == "/athena status": + mode = self._mode_payload() + profile = current_profile() + text = (f"Athena läuft im {mode['active'].upper()}-Modus. " + f"Phase: {mode['phase']}. Musik-Worker: " + f"{mode['music_worker']}. LLM-Profil: {profile or 'entladen'}.") + else: + target = "music" if command == "/athena music" else "llm" + try: + started, phase = schedule_operating_mode(target) + if started: + text = ("Musikstudio wird gestartet. Das LLM und TTS werden " + "entladen; der Fortschritt ist im Athena-Dashboard sichtbar." + if target == "music" else + "Musikstudio wird beendet und das vorherige LLM-Profil wird wiederhergestellt.") + else: + text = (f"Athena ist bereits im {target.upper()}-Modus " + f"oder wechselt gerade ({phase}).") + except RuntimeError as exc: + self._send_error(503, str(exc), "server_error", "mode_unavailable") + return + + model = str(data.get("model") or "athena-control") + created = int(time.time()) + request_id = f"athena-mode-{uuid.uuid4().hex[:16]}" + if path == "/v1/responses": + self._send_json(200, { + "id": request_id, "object": "response", "created_at": created, + "status": "completed", "model": model, + "output": [{"type": "message", "role": "assistant", + "content": [{"type": "output_text", "text": text}]}], + "output_text": text, + "usage": {"input_tokens": 0, "output_tokens": 0, + "total_tokens": 0}, + }) + return + if data.get("stream") is True: + chunks = [ + {"id": request_id, "object": "chat.completion.chunk", + "created": created, "model": model, + "choices": [{"index": 0, "delta": {"role": "assistant", + "content": text}, "finish_reason": None}]}, + {"id": request_id, "object": "chat.completion.chunk", + "created": created, "model": model, + "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]}, + ] + body = "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks) + body += "data: [DONE]\n\n" + encoded = body.encode() + self._last_code = 200 + self.send_response(200) + self.send_header("Content-Type", "text/event-stream") + self.send_header("Content-Length", str(len(encoded))) + self.send_header("Connection", "close") + self.end_headers() + self.wfile.write(encoded) + return + self._send_json(200, { + "id": request_id, "object": "chat.completion", "created": created, + "model": model, + "choices": [{"index": 0, "message": {"role": "assistant", + "content": text}, "finish_reason": "stop"}], + "usage": {"prompt_tokens": 0, "completion_tokens": 0, + "total_tokens": 0}, + }) + def _acquire_model_lease(self, profile: str | None = None) -> dict: """Atomar Profil sicherstellen und einen aktiven Request registrieren.""" with STATE.lock: @@ -2642,6 +2897,22 @@ def _startup_reconcile() -> None: if removed: log.info("Startup-Retention: %d alte Bilder entfernt", len(removed)) + if ENABLE_MUSIC_MODE and previous.get("mode") == "music": + STATE.mode = "music" + STATE.mode_phase = "starting-music" + _set_qwen_unavailable(True) + try: + _profile_controller_request("POST", "/workers/music/start") + _wait_music_ready() + STATE.mode_phase = "ready" + RUNTIME.save(mode="music", phase="music") + log.info("Recovery: Musikmodus wiederhergestellt") + except Exception as exc: + STATE.mode_error = str(exc) + STATE.mode_phase = "error" + log.error("Recovery: Musikmodus konnte nicht gestartet werden: %s", exc) + return + profile = current_profile() if profile is None: saved = previous.get("last_profile") @@ -2665,7 +2936,9 @@ def _startup_reconcile() -> None: and (not EXPECTED_MODELS.get(profile) or up.get("model") == EXPECTED_MODELS[profile])): _set_qwen_unavailable(False) - RUNTIME.save(last_profile=profile, phase="idle") + STATE.mode = "llm" + STATE.mode_phase = "ready" + RUNTIME.save(last_profile=profile, mode="llm", phase="idle") log.info("Recovery: Profil %s ist bereits bereit", profile) return _set_qwen_unavailable(True)