Add persistent Athena music mode switching

This commit is contained in:
Mikei386
2026-09-08 17:32:04 +02:00
parent 56c382f71f
commit f58d61140e
9 changed files with 467 additions and 5 deletions
+4
View File
@@ -12,6 +12,7 @@ Sie betreibt:
- den OpenAI-kompatiblen Profile Router, - den OpenAI-kompatiblen Profile Router,
- FLUX.2 Klein 9B FP8 Beta für Textbilder und Referenzbild-Bearbeitung, - FLUX.2 Klein 9B FP8 Beta für Textbilder und Referenzbild-Bearbeitung,
- Qwen3-TTS und Piper für Sprache, - Qwen3-TTS und Piper für Sprache,
- ACE-Step 1.5 XL-SFT als exklusiven Musikstudio-Modus,
- das Athena-Dashboard, - das Athena-Dashboard,
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container, - Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
- WireGuard-Gateway und Datenbackup, - WireGuard-Gateway und Datenbackup,
@@ -68,6 +69,9 @@ Qwen-Profil wird vom Profile Controller verwaltet.
- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback. Der Router reicht - Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback. Der Router reicht
zusätzlich natives 24-kHz-PCM für den optionalen Hermes-Streaming-Adapter zusätzlich natives 24-kHz-PCM für den optionalen Hermes-Streaming-Adapter
unter `integrations/hermes-qwen3-stream` durch. unter `integrations/hermes-qwen3-stream` durch.
- ACE-Step 1.5 XL-SFT: exklusiver Musikmodus auf der RTX 5080. Dashboard und
die Routerbefehle `/athena music`, `/athena llm`, `/athena status` bedienen
dieselbe persistente Zustandsmaschine; siehe `docs/OPERATING_MODES.md`.
Die verbindlichen Werte stehen in `config/profile-matrix.json` und Die verbindlichen Werte stehen in `config/profile-matrix.json` und
`docs/STANDARD_PROFILE_MATRIX.md`. `docs/STANDARD_PROFILE_MATRIX.md`.
+5
View File
@@ -15,6 +15,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback - Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung - Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 - Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
- Dashboard-Umschaltung zwischen LLM-Betrieb und ACE-Step-Musikstudio
- Portainer CE als optionale Container-Ansicht auf Port 9443 - Portainer CE als optionale Container-Ansicht auf Port 9443
- WireGuard-Gateway, Datenbackup und Athena-Operator - WireGuard-Gateway, Datenbackup und Athena-Operator
- keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz - keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz
@@ -121,6 +122,10 @@ Details, Installation, Prüfung und Rollback stehen in
- Router: `http://192.168.1.212:8081/v1` - Router: `http://192.168.1.212:8081/v1`
- Athena-Dashboard: `http://192.168.1.212:8099` - Athena-Dashboard: `http://192.168.1.212:8099`
Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren
außerdem `/athena music`, `/athena llm` und `/athena status`; Details stehen in
[docs/OPERATING_MODES.md](docs/OPERATING_MODES.md).
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
`/v1/audio/speech`, natives Qwen-PCM-Streaming über `/v1/audio/speech`, natives Qwen-PCM-Streaming über
`/v1/audio/speech/pcm-stream` und Spracherkennung über `/v1/audio/speech/pcm-stream` und Spracherkennung über
+4
View File
@@ -569,6 +569,7 @@ services:
IMAGE_WORKER: image IMAGE_WORKER: image
RESTORE_WORKER: restore RESTORE_WORKER: restore
TTS_WORKER: qwen3 TTS_WORKER: qwen3
MUSIC_WORKER: acestep
networks: [control] networks: [control]
security_opt: ["no-new-privileges:true"] security_opt: ["no-new-privileges:true"]
healthcheck: healthcheck:
@@ -625,6 +626,8 @@ services:
TTS_VOICES: alloy TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy TTS_DEFAULT_VOICE: alloy
ENABLE_STT: "true" ENABLE_STT: "true"
ENABLE_MUSIC_MODE: "true"
MUSIC_START_TIMEOUT: "600"
STT_WORKER_URL: http://whisper:8084 STT_WORKER_URL: http://whisper:8084
STT_TIMEOUT: "300" STT_TIMEOUT: "300"
networks: [frontend, control, inference] networks: [frontend, control, inference]
@@ -853,6 +856,7 @@ services:
DASHBOARD_PORT: "8099" DASHBOARD_PORT: "8099"
ROUTER_URL: http://router:8081 ROUTER_URL: http://router:8081
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
MUSIC_UI_URL: "${MUSIC_UI_URL:-http://127.0.0.1:7861/}"
HOST_PROC: /host/proc HOST_PROC: /host/proc
HOST_DATA: /host/data HOST_DATA: /host/data
HOST_MODELS: /host/models HOST_MODELS: /host/models
+31
View File
@@ -36,7 +36,38 @@ def tts_item(state="running"):
"Labels": {controller.TTS_LABEL_KEY: controller.TTS_WORKER}} "Labels": {controller.TTS_LABEL_KEY: controller.TTS_WORKER}}
def music_item(state="exited"):
return {"Id": "id-music", "State": state,
"Labels": {controller.MUSIC_LABEL_KEY: "acestep"}}
class ProfileControllerTests(unittest.TestCase): class ProfileControllerTests(unittest.TestCase):
def test_music_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["ultra"] = item("ultra", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "music_container", return_value=music_item()), \
patch.object(controller, "image_containers",
return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_music_worker(True)
self.assertEqual(result, {"music_worker": "acestep", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-ultra/stop?t=120"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-music/start"),
])
def test_rejects_unknown_profile_before_docker_call(self): def test_rejects_unknown_profile_before_docker_call(self):
with patch.object(controller, "docker_request") as request: with patch.object(controller, "docker_request") as request:
with self.assertRaises(ValueError): with self.assertRaises(ValueError):
+45
View File
@@ -0,0 +1,45 @@
# Athena-Betriebsmodi
Athena besitzt zwei gegenseitig exklusive Betriebsmodi:
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; ACE-Step ist gestoppt.
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild- und TTS-Worker sind gestoppt.
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
wird persistent gespeichert und beim Verlassen des Musikmodus wieder geladen.
## Bedienung
Im Athena-Dashboard stehen die Schaltflächen **LLM-Betrieb** und
**Musikstudio** bereit. Nach dem Start des Musikmodus öffnet **Studio öffnen**
die über den SSH-Tunnel bereitgestellte ACE-Step-Oberfläche.
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
```text
/athena music
/athena llm
/athena status
```
Die HTTP-Schnittstelle verwendet authentifizierte Requests:
```text
GET /mode
POST /mode {"mode":"music"}
POST /mode {"mode":"llm"}
```
Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
`com.mike-ai.music-worker=acestep` markierten Container; freie Container- oder
Docker-Befehle werden nicht entgegengenommen.
## Wiederanlauf
Der Router speichert `mode`, `last_profile` und `return_profile` atomar. War
beim Router-Neustart der Musikmodus aktiv, startet er ACE-Step erneut. Beim
Wechsel zurück wird das gespeicherte LLM-Profil semantisch auf Alias und
Kontextfenster geprüft, bevor Chat-Anfragen wieder freigegeben werden.
@@ -2,6 +2,8 @@ services:
music-worker: music-worker:
image: ghcr.io/ace-step/ace-step-1.5:latest@sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567 image: ghcr.io/ace-step/ace-step-1.5:latest@sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567
container_name: mike-ai-music-acestep-test container_name: mike-ai-music-acestep-test
labels:
com.mike-ai.music-worker: "acestep"
profiles: ["music-test"] profiles: ["music-test"]
environment: environment:
ACESTEP_MODE: gradio ACESTEP_MODE: gradio
@@ -28,6 +28,8 @@ IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore") RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore")
TTS_LABEL_KEY = "com.mike-ai.tts-worker" TTS_LABEL_KEY = "com.mike-ai.tts-worker"
TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3") TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3")
MUSIC_LABEL_KEY = "com.mike-ai.music-worker"
MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip()
LOCK = threading.Lock() LOCK = threading.Lock()
log = logging.getLogger("profile-controller") log = logging.getLogger("profile-controller")
@@ -96,6 +98,22 @@ def tts_container() -> dict:
return matches[0] return matches[0]
def music_container() -> dict:
if not MUSIC_WORKER:
raise RuntimeError("music worker is not configured")
matches = [item for item in labelled_containers(MUSIC_LABEL_KEY)
if item.get("Labels", {}).get(MUSIC_LABEL_KEY) == MUSIC_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one music worker {MUSIC_WORKER!r}, found {len(matches)}")
return matches[0]
def stop_music_if_configured() -> None:
if MUSIC_WORKER:
stop_container(music_container(), timeout=30)
def stop_container(item: dict, timeout: int = 120) -> None: def stop_container(item: dict, timeout: int = 120) -> None:
if item.get("State") != "running": if item.get("State") != "running":
return return
@@ -133,6 +151,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
# The 9B beta text encoder temporarily borrows the RTX 3060 from # The 9B beta text encoder temporarily borrows the RTX 3060 from
# Qwen3-TTS. The gateway retains Piper as a fallback meanwhile. # Qwen3-TTS. The gateway retains Piper as a fallback meanwhile.
stop_container(tts_container(), timeout=30) stop_container(tts_container(), timeout=30)
stop_music_if_configured()
for other in image_containers(): for other in image_containers():
if other["Id"] != item["Id"]: if other["Id"] != item["Id"]:
stop_container(other, timeout=20) stop_container(other, timeout=20)
@@ -148,6 +167,23 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
"state": "running" if running else "stopped"} "state": "running" if running else "stopped"}
def set_music_worker(running: bool) -> dict:
"""Start ACE-Step exclusively, or stop it before LLM restoration."""
with LOCK:
item = music_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
start_container(item)
else:
stop_container(item, timeout=30)
return {"music_worker": MUSIC_WORKER,
"state": "running" if running else "stopped"}
def active_profile(items: dict[str, dict] | None = None) -> str | None: def active_profile(items: dict[str, dict] | None = None) -> str | None:
items = items or containers() items = items or containers()
active = [name for name, item in items.items() if item.get("State") == "running"] active = [name for name, item in items.items() if item.get("State") == "running"]
@@ -163,6 +199,7 @@ def activate(profile: str) -> dict:
# Defensive mutual exclusion even if a caller bypasses the router. # Defensive mutual exclusion even if a caller bypasses the router.
for worker in image_containers(): for worker in image_containers():
stop_container(worker) stop_container(worker)
stop_music_if_configured()
start_container(tts_container()) start_container(tts_container())
items = containers() items = containers()
missing = [name for name in ALLOWED if name not in items] missing = [name for name in ALLOWED if name not in items]
@@ -227,7 +264,16 @@ class Handler(BaseHTTPRequestHandler):
return return
try: try:
items = containers() items = containers()
music = music_container() if MUSIC_WORKER else {}
music_status = music.get("Status", "")
music_health = ("disabled" if not MUSIC_WORKER else
"healthy" if "(healthy)" in music_status else
"unhealthy" if "(unhealthy)" in music_status else
"starting" if music.get("State") == "running" else
"stopped")
self.reply(200, {"active_profile": active_profile(items), self.reply(200, {"active_profile": active_profile(items),
"music_worker": music.get("State", "disabled"),
"music_health": music_health,
"profiles": {name: items.get(name, {}).get( "profiles": {name: items.get(name, {}).get(
"State", "missing") for name in ALLOWED}}) "State", "missing") for name in ALLOWED}})
except Exception as exc: except Exception as exc:
@@ -245,6 +291,13 @@ class Handler(BaseHTTPRequestHandler):
log.exception("stopping inference failed") log.exception("stopping inference failed")
self.reply(503, {"error": str(exc)}) self.reply(503, {"error": str(exc)})
return return
if self.path in {"/workers/music/start", "/workers/music/stop"}:
try:
self.reply(200, set_music_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("music worker transition failed")
self.reply(503, {"error": str(exc)})
return
worker_paths = { worker_paths = {
"/workers/image/start": (IMAGE_WORKER, True), "/workers/image/start": (IMAGE_WORKER, True),
"/workers/image/stop": (IMAGE_WORKER, False), "/workers/image/stop": (IMAGE_WORKER, False),
+48 -3
View File
@@ -20,6 +20,7 @@ HOST = os.getenv("DASHBOARD_HOST", "0.0.0.0")
PORT = int(os.getenv("DASHBOARD_PORT", "8099")) PORT = int(os.getenv("DASHBOARD_PORT", "8099"))
ROUTER_URL = os.getenv("ROUTER_URL", "http://router:8081").rstrip("/") ROUTER_URL = os.getenv("ROUTER_URL", "http://router:8081").rstrip("/")
ROUTER_API_KEY = os.getenv("ROUTER_API_KEY", "") ROUTER_API_KEY = os.getenv("ROUTER_API_KEY", "")
MUSIC_UI_URL = os.getenv("MUSIC_UI_URL", "http://127.0.0.1:7861/")
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
HOST_DATA = os.getenv("HOST_DATA", "/host/data") HOST_DATA = os.getenv("HOST_DATA", "/host/data")
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
@@ -268,6 +269,27 @@ def router_status() -> tuple[dict[str, Any], str | None]:
return {}, str(exc) return {}, str(exc)
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
if mode not in {"llm", "music"}:
return 400, {"error": "invalid mode"}
headers = {"Accept": "application/json", "Content-Type": "application/json"}
if ROUTER_API_KEY:
headers["Authorization"] = f"Bearer {ROUTER_API_KEY}"
request = urllib.request.Request(
f"{ROUTER_URL}/mode", data=json.dumps({"mode": mode}).encode(),
headers=headers, method="POST")
try:
with urllib.request.urlopen(request, timeout=15) as response:
return response.status, json.load(response)
except urllib.error.HTTPError as exc:
try:
return exc.code, json.loads(exc.read())
except (ValueError, json.JSONDecodeError):
return exc.code, {"error": str(exc)}
except (OSError, urllib.error.URLError) as exc:
return 503, {"error": str(exc)}
_MODEL_LOCK = threading.Lock() _MODEL_LOCK = threading.Lock()
_MODEL_AT = 0.0 _MODEL_AT = 0.0
_MODEL_CACHE: tuple[list[dict[str, Any]], dict[str, Any]] = ([], {"count": 0, "total_size": 0}) _MODEL_CACHE: tuple[list[dict[str, Any]], dict[str, Any]] = ([], {"count": 0, "total_size": 0})
@@ -602,9 +624,11 @@ HTML = r'''<!doctype html>
.span4 .metrics{grid-template-columns:repeat(2,1fr)} .span4 .metrics{grid-template-columns:repeat(2,1fr)}
.history-controls{display:flex;flex-wrap:wrap;gap:7px;margin:10px 0 14px}.history-controls button{border:1px solid var(--line);background:#09111b;color:var(--muted);padding:6px 10px;border-radius:8px;cursor:pointer}.history-controls button.active{color:var(--cyan);border-color:var(--cyan)}.chart-legend{display:flex;flex-wrap:wrap;gap:8px;margin:2px 0 10px}.chart-legend button{border:1px solid var(--series);background:#09111b;color:var(--text);padding:6px 10px;border-radius:8px;cursor:pointer}.chart-legend button::before{content:'';display:inline-block;width:10px;height:3px;background:var(--series);margin:0 7px 3px 0}.chart-legend button.off{opacity:.4;text-decoration:line-through}.chart{width:100%;height:250px;display:block}.token-total{font-size:24px;font-weight:750;margin-top:5px}.history-note{color:var(--muted);font-size:11px;margin-top:8px} .history-controls{display:flex;flex-wrap:wrap;gap:7px;margin:10px 0 14px}.history-controls button{border:1px solid var(--line);background:#09111b;color:var(--muted);padding:6px 10px;border-radius:8px;cursor:pointer}.history-controls button.active{color:var(--cyan);border-color:var(--cyan)}.chart-legend{display:flex;flex-wrap:wrap;gap:8px;margin:2px 0 10px}.chart-legend button{border:1px solid var(--series);background:#09111b;color:var(--text);padding:6px 10px;border-radius:8px;cursor:pointer}.chart-legend button::before{content:'';display:inline-block;width:10px;height:3px;background:var(--series);margin:0 7px 3px 0}.chart-legend button.off{opacity:.4;text-decoration:line-through}.chart{width:100%;height:250px;display:block}.token-total{font-size:24px;font-weight:750;margin-top:5px}.history-note{color:var(--muted);font-size:11px;margin-top:8px}
.usage-list{display:grid;gap:13px;margin-top:8px}.usage-head{display:flex;justify-content:space-between;gap:14px;align-items:baseline}.usage-head b{font-size:16px}.usage-head span{color:var(--muted)}.usage-meta{display:flex;justify-content:space-between;gap:12px;color:var(--muted);font-size:11px;margin-top:5px} .usage-list{display:grid;gap:13px;margin-top:8px}.usage-head{display:flex;justify-content:space-between;gap:14px;align-items:baseline}.usage-head b{font-size:16px}.usage-head span{color:var(--muted)}.usage-meta{display:flex;justify-content:space-between;gap:12px;color:var(--muted);font-size:11px;margin-top:5px}
.mode-row{display:flex;align-items:center;justify-content:space-between;gap:18px;flex-wrap:wrap}.mode-buttons{display:flex;gap:9px;flex-wrap:wrap}.mode-buttons button,.mode-buttons a{border:1px solid var(--line);background:#09111b;color:var(--text);padding:9px 14px;border-radius:9px;cursor:pointer;text-decoration:none;font:inherit}.mode-buttons button.active{border-color:var(--cyan);color:var(--cyan);box-shadow:inset 0 0 0 1px #45d7ff33}.mode-buttons button:disabled{opacity:.45;cursor:wait}.mode-buttons a[hidden]{display:none}.mode-error{color:var(--red)}
</style></head><body><main> </style></head><body><main>
<div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div> <div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div>
<section class="grid"> <section class="grid">
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><a id="musicOpen" href="__MUSIC_UI_URL__" target="_blank" rel="noopener" hidden>Studio öffnen</a></div></div></article>
<article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article> <article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article>
<article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article> <article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article>
<article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article> <article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article>
@@ -637,13 +661,15 @@ HTML = r'''<!doctype html>
<article class="card span12"><div class="label">llama.cpp Laufzeitkonfiguration</div><div class="metrics" id="runtime"><div class="metric"><b>–</b><span>wird gelesen</span></div></div></article> <article class="card span12"><div class="label">llama.cpp Laufzeitkonfiguration</div><div class="metrics" id="runtime"><div class="metric"><b>–</b><span>wird gelesen</span></div></div></article>
<article class="card span12"><div class="row"><div><div class="label">Verfügbare GGUF-Dateien</div><div class="sub" id="modelSummary">–</div></div></div><div class="scroll"><table><thead><tr><th>Datei</th><th>Pfad</th><th>Größe</th><th>Geändert</th></tr></thead><tbody id="modelFiles"><tr><td colspan="4">–</td></tr></tbody></table></div></article> <article class="card span12"><div class="row"><div><div class="label">Verfügbare GGUF-Dateien</div><div class="sub" id="modelSummary">–</div></div></div><div class="scroll"><table><thead><tr><th>Datei</th><th>Pfad</th><th>Größe</th><th>Geändert</th></tr></thead><tbody id="modelFiles"><tr><td colspan="4">–</td></tr></tbody></table></div></article>
<article class="card span12 error" id="errors" hidden></article> <article class="card span12 error" id="errors" hidden></article>
</section><div class="footer">Aktualisierung jede Sekunde · Nur lesende Telemetrie</div> </section><div class="footer">Aktualisierung jede Sekunde · Umschaltung über Athena Router</div>
</main><script> </main><script>
const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed?.(1)??n}%`; const gib=b=>b?`${(b/1073741824).toFixed(1)} GiB`:'–'; const dur=s=>{if(s==null)return'–';let d=Math.floor(s/86400),h=Math.floor(s%86400/3600),m=Math.floor(s%3600/60);return d?`${d}d ${h}h`:`${h}h ${m}m`}; const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed?.(1)??n}%`; const gib=b=>b?`${(b/1073741824).toFixed(1)} GiB`:'–'; const dur=s=>{if(s==null)return'–';let d=Math.floor(s/86400),h=Math.floor(s%86400/3600),m=Math.floor(s%3600/60);return d?`${d}d ${h}h`:`${h}h ${m}m`};
function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `<article class="card span6"><div class="gpu-title"><div><div class="label">GPU ${g.index}</div><div class="value">${g.name}</div></div><span class="badge">${g.pstate||'–'}</span></div><div class="metrics"><div class="metric"><b>${pct(g.gpu_percent)}</b><span>GPU-Kern</span></div><div class="metric"><b>${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiB</b><span>VRAM</span></div><div class="metric"><b>${g.temperature_c??'–'} °C</b><span>Temperatur</span></div><div class="metric"><b>${g.power_w??'–'} / ${g.power_limit_w??'–'} W</b><span>Leistung</span></div><div class="metric"><b>${g.graphics_clock_mhz??'–'} MHz</b><span>Grafiktakt</span></div><div class="metric"><b>${g.memory_clock_mhz??'–'} MHz</b><span>Speichertakt</span></div><div class="metric"><b>${pct(g.memory_controller_percent)}</b><span>Memory Controller</span></div><div class="metric"><b>${pct(g.fan_percent)}</b><span>Lüfter</span></div></div><div class="bar-label"><span>GPU-Auslastung</span><span>${load.toFixed(1)} %</span></div><div class="bar"><div class="fill gpu-load" style="width:${load}%"></div></div><div class="bar-label"><span>VRAM-Belegung</span><span>${p.toFixed(1)} %</span></div><div class="bar"><div class="fill" style="width:${Math.min(100,p)}%"></div></div><div class="sub">${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei</div></article>`} function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `<article class="card span6"><div class="gpu-title"><div><div class="label">GPU ${g.index}</div><div class="value">${g.name}</div></div><span class="badge">${g.pstate||'–'}</span></div><div class="metrics"><div class="metric"><b>${pct(g.gpu_percent)}</b><span>GPU-Kern</span></div><div class="metric"><b>${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiB</b><span>VRAM</span></div><div class="metric"><b>${g.temperature_c??'–'} °C</b><span>Temperatur</span></div><div class="metric"><b>${g.power_w??'–'} / ${g.power_limit_w??'–'} W</b><span>Leistung</span></div><div class="metric"><b>${g.graphics_clock_mhz??'–'} MHz</b><span>Grafiktakt</span></div><div class="metric"><b>${g.memory_clock_mhz??'–'} MHz</b><span>Speichertakt</span></div><div class="metric"><b>${pct(g.memory_controller_percent)}</b><span>Memory Controller</span></div><div class="metric"><b>${pct(g.fan_percent)}</b><span>Lüfter</span></div></div><div class="bar-label"><span>GPU-Auslastung</span><span>${load.toFixed(1)} %</span></div><div class="bar"><div class="fill gpu-load" style="width:${load}%"></div></div><div class="bar-label"><span>VRAM-Belegung</span><span>${p.toFixed(1)} %</span></div><div class="bar"><div class="fill" style="width:${Math.min(100,p)}%"></div></div><div class="sub">${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei</div></article>`}
const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit'); const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit');
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000); let modeBusy=false;
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>''' async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready';$('operatingMode').textContent=md.active==='music'?'Musikstudio':'LLM-Betrieb';$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik-Worker: ${md.music_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('llmMode').disabled=$('musicMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''.replace("__MUSIC_UI_URL__", MUSIC_UI_URL)
FULL_JS = r''' FULL_JS = r'''
@@ -785,6 +811,25 @@ class Handler(BaseHTTPRequestHandler):
else: else:
self._send(404, b'{"error":"not found"}', "application/json") self._send(404, b'{"error":"not found"}', "application/json")
def do_POST(self) -> None:
path = self.path.split("?", 1)[0]
if path != "/api/mode":
self._send(404, b'{"error":"not found"}', "application/json")
return
try:
length = int(self.headers.get("Content-Length", "0"))
if length <= 0 or length > 1024:
raise ValueError("invalid body size")
payload = json.loads(self.rfile.read(length))
mode = payload.get("mode") if isinstance(payload, dict) else None
except (ValueError, json.JSONDecodeError):
self._send(400, b'{"error":"invalid request"}', "application/json")
return
status, response = change_mode(mode)
body = json.dumps(response, ensure_ascii=False,
separators=(",", ":")).encode()
self._send(status, body, "application/json; charset=utf-8")
if __name__ == "__main__": if __name__ == "__main__":
threading.Thread(target=history_collector, name="history-collector", daemon=True).start() threading.Thread(target=history_collector, name="history-collector", daemon=True).start()
+275 -2
View File
@@ -106,6 +106,9 @@ PROFILE_DIR = os.environ.get(
PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/") PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/")
PROFILE_CONTROL_TOKEN_FILE = os.environ.get( PROFILE_CONTROL_TOKEN_FILE = os.environ.get(
"PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token") "PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token")
ENABLE_MUSIC_MODE = os.environ.get(
"ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"}
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
# Optional worker APIs. The clean Docker baseline deliberately ships only # Optional worker APIs. The clean Docker baseline deliberately ships only
# text/multimodal chat; absent workers must fail explicitly instead of trying # text/multimodal chat; absent workers must fail explicitly instead of trying
@@ -282,6 +285,9 @@ class _State:
self.qwen_unavailable = True self.qwen_unavailable = True
self.active_chats = 0 self.active_chats = 0
self.avail_lock = threading.Lock() self.avail_lock = threading.Lock()
self.mode = "llm"
self.mode_phase = "ready"
self.mode_error: str | None = None
STATE = _State() STATE = _State()
@@ -312,6 +318,143 @@ def _set_qwen_unavailable(unavailable: bool) -> None:
STATE.qwen_unavailable = unavailable STATE.qwen_unavailable = unavailable
def _music_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"music_worker", "missing"))
except Exception as exc:
log.warning("Musik-Worker-Status nicht verfügbar: %s", exc)
return "unknown"
def _music_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"music_health", "unknown"))
except Exception:
return "unknown"
def _wait_music_ready() -> None:
deadline = time.monotonic() + MUSIC_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("music_worker") == "running"
and status.get("music_health") == "healthy"):
return
if status.get("music_health") == "unhealthy":
raise RuntimeError("ACE-Step-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"ACE-Step nach {MUSIC_START_TIMEOUT:.0f} s nicht bereit")
def set_operating_mode(mode: str) -> dict:
"""Atomarer Wechsel zwischen llama.cpp/TTS und ACE-Step Studio."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music"}:
raise ValueError("Modus muss 'llm' oder 'music' sein")
with STATE.lock:
STATE.mode_error = None
if mode == "music":
if STATE.mode == "music" and _music_worker_state() == "running":
return {"status": "ok", "mode": "music", "changed": False}
profile = current_profile()
saved = RUNTIME.load().get("last_profile")
return_profile = profile if profile in PROFILES else saved
if return_profile not in PROFILES:
return_profile = next(iter(PROFILES))
STATE.mode_phase = "starting-music"
_set_qwen_unavailable(True)
try:
# Persist intent before stopping anything so a router restart
# during ACE-Step loading can resume the same transition.
RUNTIME.save(mode="music", return_profile=return_profile,
last_profile=return_profile,
phase="starting-music")
_wait_chats_drained()
_profile_controller_request("POST", "/workers/music/start")
_wait_music_ready()
STATE.mode = "music"
STATE.mode_phase = "ready"
RUNTIME.save(mode="music", return_profile=return_profile,
last_profile=return_profile, phase="music")
return {"status": "ok", "mode": "music", "changed": True,
"return_profile": return_profile}
except Exception as exc:
STATE.mode_error = str(exc)
STATE.mode_phase = "error"
raise
previous = RUNTIME.load()
profile = previous.get("return_profile") or previous.get("last_profile")
if profile not in PROFILES:
profile = next(iter(PROFILES))
STATE.mode_phase = "restoring-llm"
_set_qwen_unavailable(True)
try:
_profile_controller_request("POST", "/workers/music/stop")
_restore_qwen(profile)
STATE.mode = "llm"
STATE.mode_phase = "ready"
_set_qwen_unavailable(False)
RUNTIME.save(mode="llm", return_profile=None,
last_profile=profile, phase="idle")
return {"status": "ok", "mode": "llm", "changed": True,
"profile": profile}
except Exception as exc:
STATE.mode_error = str(exc)
STATE.mode_phase = "error"
raise
def schedule_operating_mode(mode: str) -> tuple[bool, str]:
"""Start a transition in the background so chat/UI acknowledgement is instant."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music"}:
raise ValueError("Modus muss 'llm' oder 'music' sein")
with STATE.lock:
if STATE.mode_phase not in {"ready", "error"}:
return False, STATE.mode_phase
if STATE.mode == mode and STATE.mode_phase == "ready":
return False, "ready"
STATE.mode_phase = "starting-music" if mode == "music" else "restoring-llm"
def transition() -> None:
try:
set_operating_mode(mode)
log.info("Betriebsmodus ist jetzt %s", mode)
except Exception:
log.exception("Betriebsmodus-Wechsel zu %s fehlgeschlagen", mode)
threading.Thread(target=transition, name=f"mode-{mode}", daemon=True).start()
return True, STATE.mode_phase
def _control_command(data: dict, path: str) -> str | None:
"""Recognise exact local commands without invoking an LLM."""
text: object = None
if path == "/v1/chat/completions":
messages = data.get("messages")
if isinstance(messages, list):
for item in reversed(messages):
if isinstance(item, dict) and item.get("role") == "user":
text = item.get("content")
break
elif path == "/v1/responses":
text = data.get("input")
if not isinstance(text, str):
return None
command = text.strip().casefold()
return command if command in {"/athena music", "/athena llm", "/athena status"} else None
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
# Upstream (llama.cpp) # Upstream (llama.cpp)
# --------------------------------------------------------------------------- # ---------------------------------------------------------------------------
@@ -753,7 +896,10 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
f"Profildatei wurde nicht gesetzt (erwartet: {profile})") f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
log.info("Warte, bis llama.cpp das Profil geladen hat ...") log.info("Warte, bis llama.cpp das Profil geladen hat ...")
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT) _wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
RUNTIME.save(last_profile=profile, phase="idle") STATE.mode = "llm"
STATE.mode_phase = "ready"
RUNTIME.save(last_profile=profile, mode="llm",
return_profile=None, phase="idle")
finally: finally:
# Nach einem fehlgeschlagenen Skript/Timeout darf der Router # Nach einem fehlgeschlagenen Skript/Timeout darf der Router
# Qwen nicht blind freigeben. Nur ein semantisch verifiziertes # Qwen nicht blind freigeben. Nur ein semantisch verifiziertes
@@ -1512,6 +1658,10 @@ class Handler(BaseHTTPRequestHandler):
self._send_json(200, self._models_payload()) self._send_json(200, self._models_payload())
elif path == "/status" and self.command == "GET": elif path == "/status" and self.command == "GET":
self._send_json(200, self._status_payload()) self._send_json(200, self._status_payload())
elif path == "/mode" and self.command == "GET":
self._send_json(200, self._mode_payload())
elif path == "/mode" and self.command == "POST":
self._mode_change()
elif path == "/v1/audio/models" and self.command == "GET": elif path == "/v1/audio/models" and self.command == "GET":
self._send_json(200, self._audio_models_payload()) self._send_json(200, self._audio_models_payload())
elif path == "/v1/audio/voices" and self.command == "GET": elif path == "/v1/audio/voices" and self.command == "GET":
@@ -1735,6 +1885,7 @@ class Handler(BaseHTTPRequestHandler):
"current_profile": current_profile(), "current_profile": current_profile(),
"switching": STATE.switching, "switching": STATE.switching,
"profiles": PROFILES, "profiles": PROFILES,
"mode": self._mode_payload(),
"upstream": { "upstream": {
"url": UPSTREAM_URL, "url": UPSTREAM_URL,
"reachable": up["reachable"], "reachable": up["reachable"],
@@ -1766,6 +1917,36 @@ class Handler(BaseHTTPRequestHandler):
"stt": stt_status(), "stt": stt_status(),
} }
@staticmethod
def _mode_payload() -> dict:
state = RUNTIME.load()
return {
"active": STATE.mode,
"phase": STATE.mode_phase,
"music_worker": _music_worker_state(),
"music_health": _music_worker_health(),
"return_profile": state.get("return_profile"),
"last_error": STATE.mode_error,
"enabled": ENABLE_MUSIC_MODE,
}
def _mode_change(self) -> None:
try:
data = json.loads(self._read_body() or b"{}")
mode = data.get("mode") if isinstance(data, dict) else None
if mode not in {"llm", "music"}:
raise ValueError("Feld 'mode' muss 'llm' oder 'music' sein")
started, phase = schedule_operating_mode(mode)
self._send_json(202 if started else 200, {
"status": "accepted" if started else "ok",
"requested_mode": mode,
"phase": phase,
})
except ValueError as exc:
self._send_error(400, str(exc), "invalid_request_error", "invalid_mode")
except RuntimeError as exc:
self._send_error(503, str(exc), "server_error", "mode_unavailable")
# ---------- Bildgenerierung ---------- # ---------- Bildgenerierung ----------
def _image_generate(self) -> None: def _image_generate(self) -> None:
@@ -2366,6 +2547,12 @@ class Handler(BaseHTTPRequestHandler):
data = json.loads(body) data = json.loads(body)
except ValueError: except ValueError:
data = None data = None
if (isinstance(data, dict)
and path in {"/v1/chat/completions", "/v1/responses"}):
command = _control_command(data, path)
if command is not None:
self._control_response(command, data, path)
return
model = data.get("model") if isinstance(data, dict) else None model = data.get("model") if isinstance(data, dict) else None
if (isinstance(model, str) and REVIEW_UPSTREAM_URL if (isinstance(model, str) and REVIEW_UPSTREAM_URL
and model == REVIEW_MODEL_NAME): and model == REVIEW_MODEL_NAME):
@@ -2403,6 +2590,74 @@ class Handler(BaseHTTPRequestHandler):
# An llama.cpp weiterleiten (mit Chat-Waiting, Streaming bleibt erhalten). # An llama.cpp weiterleiten (mit Chat-Waiting, Streaming bleibt erhalten).
self._proxy_with_wait(body) self._proxy_with_wait(body)
def _control_response(self, command: str, data: dict, path: str) -> None:
"""Return OpenAI-compatible local replies for Athena control commands."""
if command == "/athena status":
mode = self._mode_payload()
profile = current_profile()
text = (f"Athena läuft im {mode['active'].upper()}-Modus. "
f"Phase: {mode['phase']}. Musik-Worker: "
f"{mode['music_worker']}. LLM-Profil: {profile or 'entladen'}.")
else:
target = "music" if command == "/athena music" else "llm"
try:
started, phase = schedule_operating_mode(target)
if started:
text = ("Musikstudio wird gestartet. Das LLM und TTS werden "
"entladen; der Fortschritt ist im Athena-Dashboard sichtbar."
if target == "music" else
"Musikstudio wird beendet und das vorherige LLM-Profil wird wiederhergestellt.")
else:
text = (f"Athena ist bereits im {target.upper()}-Modus "
f"oder wechselt gerade ({phase}).")
except RuntimeError as exc:
self._send_error(503, str(exc), "server_error", "mode_unavailable")
return
model = str(data.get("model") or "athena-control")
created = int(time.time())
request_id = f"athena-mode-{uuid.uuid4().hex[:16]}"
if path == "/v1/responses":
self._send_json(200, {
"id": request_id, "object": "response", "created_at": created,
"status": "completed", "model": model,
"output": [{"type": "message", "role": "assistant",
"content": [{"type": "output_text", "text": text}]}],
"output_text": text,
"usage": {"input_tokens": 0, "output_tokens": 0,
"total_tokens": 0},
})
return
if data.get("stream") is True:
chunks = [
{"id": request_id, "object": "chat.completion.chunk",
"created": created, "model": model,
"choices": [{"index": 0, "delta": {"role": "assistant",
"content": text}, "finish_reason": None}]},
{"id": request_id, "object": "chat.completion.chunk",
"created": created, "model": model,
"choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}]},
]
body = "".join(f"data: {json.dumps(chunk)}\n\n" for chunk in chunks)
body += "data: [DONE]\n\n"
encoded = body.encode()
self._last_code = 200
self.send_response(200)
self.send_header("Content-Type", "text/event-stream")
self.send_header("Content-Length", str(len(encoded)))
self.send_header("Connection", "close")
self.end_headers()
self.wfile.write(encoded)
return
self._send_json(200, {
"id": request_id, "object": "chat.completion", "created": created,
"model": model,
"choices": [{"index": 0, "message": {"role": "assistant",
"content": text}, "finish_reason": "stop"}],
"usage": {"prompt_tokens": 0, "completion_tokens": 0,
"total_tokens": 0},
})
def _acquire_model_lease(self, profile: str | None = None) -> dict: def _acquire_model_lease(self, profile: str | None = None) -> dict:
"""Atomar Profil sicherstellen und einen aktiven Request registrieren.""" """Atomar Profil sicherstellen und einen aktiven Request registrieren."""
with STATE.lock: with STATE.lock:
@@ -2642,6 +2897,22 @@ def _startup_reconcile() -> None:
if removed: if removed:
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed)) log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
if ENABLE_MUSIC_MODE and previous.get("mode") == "music":
STATE.mode = "music"
STATE.mode_phase = "starting-music"
_set_qwen_unavailable(True)
try:
_profile_controller_request("POST", "/workers/music/start")
_wait_music_ready()
STATE.mode_phase = "ready"
RUNTIME.save(mode="music", phase="music")
log.info("Recovery: Musikmodus wiederhergestellt")
except Exception as exc:
STATE.mode_error = str(exc)
STATE.mode_phase = "error"
log.error("Recovery: Musikmodus konnte nicht gestartet werden: %s", exc)
return
profile = current_profile() profile = current_profile()
if profile is None: if profile is None:
saved = previous.get("last_profile") saved = previous.get("last_profile")
@@ -2665,7 +2936,9 @@ def _startup_reconcile() -> None:
and (not EXPECTED_MODELS.get(profile) and (not EXPECTED_MODELS.get(profile)
or up.get("model") == EXPECTED_MODELS[profile])): or up.get("model") == EXPECTED_MODELS[profile])):
_set_qwen_unavailable(False) _set_qwen_unavailable(False)
RUNTIME.save(last_profile=profile, phase="idle") STATE.mode = "llm"
STATE.mode_phase = "ready"
RUNTIME.save(last_profile=profile, mode="llm", phase="idle")
log.info("Recovery: Profil %s ist bereits bereit", profile) log.info("Recovery: Profil %s ist bereits bereit", profile)
return return
_set_qwen_unavailable(True) _set_qwen_unavailable(True)