Add private Vevo2 voice studio mode

This commit is contained in:
Mikei386
2026-09-09 13:38:00 +02:00
parent 535bd751b5
commit 68d02f32bd
15 changed files with 592 additions and 28 deletions
+2
View File
@@ -571,6 +571,7 @@ services:
TTS_WORKER: qwen3
MUSIC_WORKER: acestep
SEPARATOR_WORKER: bs-roformer
VOICE_WORKER: vevo2
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
@@ -860,6 +861,7 @@ services:
MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
HOST_PROC: /host/proc
HOST_DATA: /host/data
HOST_MODELS: /host/models
+9
View File
@@ -52,6 +52,15 @@ class DashboardModeTests(unittest.TestCase):
self.assertEqual(json.loads(request.data), {"mode": "separation"})
self.assertEqual(request.get_header("Authorization"), "Bearer test-key")
def test_voice_mode_is_forwarded_to_router(self):
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
status, body = self.dashboard.change_mode("voice")
self.assertEqual(status, 202)
self.assertEqual(body, {"status": "accepted"})
request = urlopen.call_args.args[0]
self.assertEqual(json.loads(request.data), {"mode": "voice"})
def test_unknown_mode_is_rejected_without_router_request(self):
with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen:
status, body = self.dashboard.change_mode("unknown")
+36
View File
@@ -46,7 +46,43 @@ def separator_item(state="exited"):
"Labels": {controller.SEPARATOR_LABEL_KEY: "bs-roformer"}}
def voice_item(state="exited"):
return {"Id": "id-voice", "State": state,
"Labels": {controller.VOICE_LABEL_KEY: "vevo2"}}
class ProfileControllerTests(unittest.TestCase):
def test_voice_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["medium"] = item("medium", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "VOICE_WORKER", "vevo2"), \
patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "voice_container", return_value=voice_item()), \
patch.object(controller, "music_container", return_value=music_item("running")), \
patch.object(controller, "separator_container", return_value=separator_item("running")), \
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_voice_worker(True)
self.assertEqual(result, {"voice_worker": "vevo2", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-medium/stop?t=120"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-music/stop?t=30"),
("POST", "/containers/id-separator/stop?t=30"),
("POST", "/containers/id-voice/start"),
])
def test_separator_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["large"] = item("large", "running")
+18 -4
View File
@@ -1,11 +1,13 @@
# Athena-Betriebsmodi
Athena besitzt drei gegenseitig exklusive Betriebsmodi:
Athena besitzt vier gegenseitig exklusive Betriebsmodi:
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt.
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt.
- `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt
Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt.
- `voice`: Vevo2 überträgt Sprache oder Gesang auf eine gespeicherte
Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt.
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
@@ -13,8 +15,8 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen
## Bedienung
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio** und
**Audio trennen** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**
und **Voice Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
- **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende
Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der
@@ -48,12 +50,22 @@ nicht eine reine Synthesizer-Spur. Sprache nutzt das 48-kHz-Modell
`audio-separator` 0.47.0. Die ältere API-Auswahl kompletter 2-/4-/6-Stem-Sätze
bleibt rückwärtskompatibel.
Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter
`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter
`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem
Speichern eine Bestätigung der Nutzungsberechtigung. Vevo2 gibt
unkomprimiertes 24-kHz-WAV aus und wandelt sowohl Sprache als auch Gesang;
dies ist noch kein Echtzeit-Live-Voice-Changer. Die Gewichte sind
CC BY-NC-ND 4.0 lizenziert und werden auf Athena ausschließlich privat und
nichtkommerziell eingesetzt.
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
```text
/athena music
/athena stems
/athena voice
/athena llm
/athena status
```
@@ -64,13 +76,15 @@ Die HTTP-Schnittstelle verwendet authentifizierte Requests:
GET /mode
POST /mode {"mode":"music"}
POST /mode {"mode":"separation"}
POST /mode {"mode":"voice"}
POST /mode {"mode":"llm"}
```
Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
`com.mike-ai.music-worker=acestep` beziehungsweise
`com.mike-ai.stem-separator=bs-roformer` markierten Container; freie
`com.mike-ai.stem-separator=bs-roformer` oder
`com.mike-ai.voice-worker=vevo2` markierten Container; freie
Container- oder Docker-Befehle werden nicht entgegengenommen.
## Wiederanlauf
+2 -1
View File
@@ -1,6 +1,6 @@
# Register getesteter Modelle
Stand: 8. September 2026
Stand: 9. September 2026
Dieses Dokument ist die zentrale Sperrliste gegen doppelte Modelltests. Vor
jedem Download müssen Repository, Dateiname, Basismodell, Fine-Tune und
@@ -56,6 +56,7 @@ Titelgenerierung und Kontextkompression in Hermes.
| 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** |
| 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback |
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Produktions-HTTP-API mit offizieller 8,6-s-Sprachprobe: Warmstart 12,216 s, Wandlung 2,342 s, Spitzen-VRAM 5.605,6 MiB; gültiges 24-kHz-WAV mit 412.878 Byte; Profilanlage, Download und automatische Job-Bereinigung geprüft | **technischer Ende-zu-Ende-Test bestanden**, privater Voice-Studio-Betatest; Gewichte CC BY-NC-ND 4.0 und daher nicht kommerziell einsetzen |
## Musikgenerierung
@@ -0,0 +1,64 @@
FROM mike-ai/bs-roformer-separator:0.47.0
ARG AMPHION_COMMIT=26f6883110181f1dbfe95c70a7c7dbaf4de5f42a
USER root
RUN apt-get update \
&& apt-get install -y --no-install-recommends git espeak-ng \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/open-mmlab/Amphion.git /opt/amphion \
&& cd /opt/amphion \
&& git checkout "${AMPHION_COMMIT}" \
&& rm -rf .git
# Vevo2 is tested on Athena with the CUDA/PyTorch stack inherited from the
# existing audio worker. Do not install Amphion's historical torch 2.0/cu118
# pins: Blackwell requires the newer cu128 runtime already present here.
RUN python -m pip install --no-cache-dir \
accelerate==1.10.1 \
diffusers==0.35.1 \
einops==0.8.1 \
easydict==1.13 \
g2p_en==2.1.0 \
humanfriendly==10.0 \
huggingface-hub==0.34.4 \
hydra-core==1.3.2 \
inflect==7.5.0 \
ipython==9.5.0 \
json5==0.12.1 \
librosa==0.11.0 \
loguru==0.7.3 \
matplotlib==3.10.6 \
munch==4.0.0 \
omegaconf==2.3.0 \
openai-whisper==20250625 \
phonemizer==3.3.0 \
python-multipart==0.0.20 \
praat-parselmouth==0.4.6 \
pypinyin==0.55.0 \
pyworld==0.3.5 \
ruamel.yaml==0.18.15 \
safetensors==0.6.2 \
tabulate==0.9.0 \
tgt==1.5 \
torchcrepe==0.0.24 \
transformers==4.56.1 \
typeguard==4.4.4 \
unidecode==1.4.0 \
vector-quantize-pytorch==1.12.5 \
vocos==0.1.0
WORKDIR /opt/amphion
ENV PYTHONPATH=/opt/amphion \
HF_HOME=/models/huggingface \
PYTHONUNBUFFERED=1
COPY run_fm_test.py /usr/local/bin/run_fm_test.py
COPY app.py /app/app.py
COPY index.html /app/index.html
EXPOSE 8008
HEALTHCHECK --interval=5s --timeout=3s --start-period=90s --retries=3 \
CMD curl -fsS http://127.0.0.1:8008/health || exit 1
ENTRYPOINT ["python", "/app/app.py"]
+30
View File
@@ -0,0 +1,30 @@
# Vevo2 Voice Conversion on Athena
This directory contains Athena's private Vevo2 voice-conversion studio.
- Code: `open-mmlab/Amphion` commit
`26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` (MIT)
- Weights: `RMSnow/Vevo2` (CC BY-NC-ND 4.0)
- Runtime: PyTorch 2.7.1 + CUDA 12.8 inherited from Athena's tested audio
separator image; the historical Amphion torch 2.0/cu118 pins are not used.
- Storage: `/data/voice/vevo2`; removing that directory and the test image
removes all downloaded artifacts.
- Private UI: `http://192.168.1.212:8008` through WireGuard only.
- Output: uncompressed mono WAV, 24 kHz.
The 9 September technical gate converted the official 8.6-second speech sample
through the production HTTP API in 2.342 seconds. A warm service start loaded
the model in 12.216 seconds, and peak CUDA allocation during conversion was
5605.6 MiB. The returned 24-kHz WAV was 412878 bytes and 8.6 seconds long. The
service stores named reference voices, accepts a source clip, and returns a
transient WAV download. Jobs and generated outputs are removed after delivery.
It is deliberately not exposed on the university interface.
The installed Compose project is named `mike-ai-voice`. Rebuild or recreate it
with an explicit `VOICE_GPU_UUID` so Docker keeps the service attached to the
RTX 5080. The profile controller starts and stops the existing container; it
does not rebuild it during a mode switch.
The weights are licensed CC BY-NC-ND 4.0. This deployment is for Mike's
private, non-commercial use only. Do not use or expose it as a public or
commercial voice-cloning service.
+202
View File
@@ -0,0 +1,202 @@
#!/usr/bin/env python3
"""Small local-only Vevo2 voice-conversion studio for Athena."""
from __future__ import annotations
import asyncio
import os
import re
import shutil
import subprocess
import threading
import time
import uuid
from contextlib import asynccontextmanager
from pathlib import Path
import torch
from fastapi import FastAPI, File, Form, HTTPException, UploadFile
from fastapi.responses import FileResponse, HTMLResponse
from starlette.background import BackgroundTask
import models.svc.vevo2.infer_vevo2_fm as vevo
DATA_DIR = Path(os.environ.get("VOICE_DATA_DIR", "/data"))
PROFILE_DIR = DATA_DIR / "profiles"
JOB_DIR = DATA_DIR / "jobs"
INDEX = Path("/app/index.html")
MAX_UPLOAD_BYTES = int(os.environ.get("MAX_UPLOAD_BYTES", str(100 * 1024 * 1024)))
MODEL_LOCK = threading.Lock()
PIPELINE = None
MODEL_LOAD_SECONDS: float | None = None
def safe_name(value: str) -> str:
value = re.sub(r"[^A-Za-z0-9_. -]+", "-", value.strip())
value = re.sub(r"\s+", "-", value).strip("-.")
return value[:64] or "voice"
def load_pipeline() -> None:
global PIPELINE, MODEL_LOAD_SECONDS
if PIPELINE is not None:
return
with MODEL_LOCK:
if PIPELINE is not None:
return
started = time.monotonic()
PIPELINE = vevo.load_inference_pipeline()
vevo.inference_pipeline = PIPELINE
MODEL_LOAD_SECONDS = time.monotonic() - started
def to_wav(source: Path, target: Path) -> None:
completed = subprocess.run(
["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
"-ac", "1", "-ar", "24000", "-c:a", "pcm_s16le", str(target)],
capture_output=True,
text=True,
timeout=180,
check=False,
)
if completed.returncode:
raise ValueError(completed.stderr.strip() or "Audio konnte nicht gelesen werden")
async def save_upload(upload: UploadFile, target: Path) -> None:
size = 0
with target.open("wb") as handle:
while chunk := await upload.read(1024 * 1024):
size += len(chunk)
if size > MAX_UPLOAD_BYTES:
raise HTTPException(413, "Audiodatei ist zu groß")
handle.write(chunk)
def profile_path(name: str) -> Path:
target = PROFILE_DIR / f"{safe_name(name)}.wav"
if not target.is_file():
raise HTTPException(404, "Referenzstimme wurde nicht gefunden")
return target
def run_conversion(source: Path, reference: Path, output: Path, pitch_shift: bool) -> tuple[float, float]:
"""Run the GPU-bound conversion off the API event loop."""
load_pipeline()
with MODEL_LOCK:
torch.cuda.reset_peak_memory_stats()
started = time.monotonic()
vevo.vevo2_fm(str(source), str(reference), str(output), shifted_src=pitch_shift)
elapsed = time.monotonic() - started
peak = torch.cuda.max_memory_allocated() / 1048576
return elapsed, peak
@asynccontextmanager
async def lifespan(_app: FastAPI):
PROFILE_DIR.mkdir(parents=True, exist_ok=True)
JOB_DIR.mkdir(parents=True, exist_ok=True)
load_pipeline()
yield
app = FastAPI(title="Athena Voice Studio", lifespan=lifespan)
@app.get("/", response_class=HTMLResponse)
def index() -> str:
return INDEX.read_text(encoding="utf-8")
@app.get("/health")
def health() -> dict:
return {
"status": "ok" if PIPELINE is not None else "starting",
"model": "RMSnow/Vevo2",
"sample_rate": 24000,
"model_load_seconds": MODEL_LOAD_SECONDS,
}
@app.get("/api/profiles")
def profiles() -> dict:
return {"profiles": [path.stem for path in sorted(PROFILE_DIR.glob("*.wav"))]}
@app.post("/api/profiles")
async def create_profile(
name: str = Form(...),
consent: bool = Form(False),
audio: UploadFile = File(...),
) -> dict:
if not consent:
raise HTTPException(400, "Bestätige, dass du die Stimme verwenden darfst")
clean = safe_name(name)
job = JOB_DIR / f"profile-{uuid.uuid4().hex}"
job.mkdir(parents=True)
raw = job / f"upload-{safe_name(audio.filename or 'reference.audio')}"
try:
await save_upload(audio, raw)
target = PROFILE_DIR / f"{clean}.wav"
temporary = job / "reference.wav"
to_wav(raw, temporary)
os.replace(temporary, target)
return {"status": "ok", "profile": clean}
except HTTPException:
raise
except Exception as exc:
raise HTTPException(400, str(exc)) from exc
finally:
shutil.rmtree(job, ignore_errors=True)
@app.delete("/api/profiles/{name}")
def delete_profile(name: str) -> dict:
target = profile_path(name)
target.unlink()
return {"status": "ok", "profile": target.stem}
@app.post("/api/convert")
async def convert(
source: UploadFile = File(...),
profile: str = Form(...),
pitch_shift: bool = Form(True),
) -> FileResponse:
reference = profile_path(profile)
job_id = uuid.uuid4().hex
job = JOB_DIR / f"convert-{job_id}"
job.mkdir(parents=True)
raw = job / f"upload-{safe_name(source.filename or 'source.audio')}"
source_wav = job / "source.wav"
output = JOB_DIR / f"voice-{job_id}.wav"
try:
await save_upload(source, raw)
to_wav(raw, source_wav)
elapsed, peak = await asyncio.to_thread(
run_conversion, source_wav, reference, output, pitch_shift
)
return FileResponse(
output,
media_type="audio/wav",
filename=f"{safe_name(profile)}-{job_id[:8]}.wav",
headers={
"X-Conversion-Seconds": f"{elapsed:.3f}",
"X-Peak-VRAM-MiB": f"{peak:.1f}",
},
background=BackgroundTask(output.unlink, missing_ok=True),
)
except HTTPException:
raise
except Exception as exc:
output.unlink(missing_ok=True)
raise HTTPException(500, f"Stimmenwandlung fehlgeschlagen: {exc}") from exc
finally:
shutil.rmtree(job, ignore_errors=True)
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host="0.0.0.0", port=8008)
@@ -0,0 +1,39 @@
services:
voice-studio:
build: .
image: mike-ai/vevo2-voice-studio:0.1
container_name: mike-ai-voice-studio
restart: "no"
labels:
com.mike-ai.voice-worker: vevo2
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
VOICE_DATA_DIR: /data
ports:
- "127.0.0.1:8008:8008"
volumes:
- /data/voice/vevo2/huggingface/hub/Vevo2:/opt/amphion/ckpts/Vevo2
- /data/voice/vevo2/huggingface:/models/huggingface
- /data/voice/vevo2/whisper:/root/.cache/whisper
- /data/voice/studio:/data
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/health | grep -q '\"status\":\"ok\"'"]
interval: 5s
timeout: 3s
start_period: 600s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
- frontend
networks:
frontend:
name: mike-ai_frontend
external: true
@@ -0,0 +1,10 @@
<!doctype html>
<html lang="de"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
<title>Athena Voice Studio</title><style>
:root{color-scheme:dark;--bg:#07101a;--panel:#0d1926;--line:#20374d;--text:#edf6ff;--muted:#91a7bb;--cyan:#45d7ff;--green:#66e3a4;--red:#ff7185}*{box-sizing:border-box}body{margin:0;background:radial-gradient(circle at top left,#12283d,var(--bg) 45%);color:var(--text);font:16px system-ui,sans-serif}main{max-width:960px;margin:auto;padding:32px 18px}h1{margin:.2rem 0}p{color:var(--muted)}.grid{display:grid;grid-template-columns:1fr 1fr;gap:16px}.card{background:var(--panel);border:1px solid var(--line);border-radius:16px;padding:20px}label{display:block;margin:13px 0 6px;color:var(--muted)}input,select,button{width:100%;border:1px solid var(--line);border-radius:9px;padding:11px;background:#07111d;color:var(--text);font:inherit}button{margin-top:15px;background:#10334a;border-color:var(--cyan);cursor:pointer}button:disabled{opacity:.5;cursor:wait}.row{display:flex;align-items:center;gap:9px}.row input{width:auto}.status{min-height:24px;margin-top:12px;color:var(--green)}.error{color:var(--red)}audio{width:100%;margin-top:12px}@media(max-width:700px){.grid{grid-template-columns:1fr}}</style></head>
<body><main><div style="color:var(--cyan);letter-spacing:.14em;text-transform:uppercase;font-size:12px">Mike AI · lokal auf Athena</div><h1>Voice Studio</h1><p>Eine Stimme als Referenz speichern und eine vorhandene Sprach- oder Gesangsaufnahme in diese Stimme übertragen. Ausgabe: unkomprimiertes WAV mit 24 kHz.</p>
<div class="grid"><section class="card"><h2>1 · Referenzstimme</h2><form id="profileForm"><label>Name</label><input name="name" required placeholder="z. B. Meine Stimme"><label>Saubere Referenzaufnahme</label><input name="audio" type="file" accept="audio/*" required><label class="row"><input name="consent" type="checkbox" value="true" required><span>Ich darf diese Stimme verwenden.</span></label><button>Stimme speichern</button></form><div class="status" id="profileStatus"></div></section>
<section class="card"><h2>2 · Stimme übertragen</h2><form id="convertForm"><label>Zielstimme</label><select name="profile" id="profiles" required></select><label>Quellaufnahme</label><input name="source" type="file" accept="audio/*" required><label class="row"><input name="pitch_shift" type="checkbox" value="true" checked><span>Tonlage automatisch an Zielstimme anpassen</span></label><button>WAV erzeugen</button></form><div class="status" id="convertStatus"></div><audio id="player" controls hidden></audio><a id="download" hidden>WAV herunterladen</a></section></div>
<p>Hinweis: Nur privat und mit Zustimmung verwenden. Das Vevo2-Gewicht ist nicht für kommerzielle Nutzung lizenziert.</p></main><script>
const $=id=>document.getElementById(id);async function loadProfiles(){let r=await fetch('/api/profiles'),d=await r.json();$('profiles').innerHTML=d.profiles.length?d.profiles.map(x=>`<option>${x}</option>`).join(''):'<option value="">Noch keine Stimme gespeichert</option>'}function state(id,text,error=false){let e=$(id);e.textContent=text;e.classList.toggle('error',error)}$('profileForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('profileStatus','Referenz wird geprüft und gespeichert …');try{let r=await fetch('/api/profiles',{method:'POST',body:new FormData(e.target)}),d=await r.json();if(!r.ok)throw Error(d.detail||`HTTP ${r.status}`);state('profileStatus',`„${d.profile}“ ist gespeichert.`);await loadProfiles()}catch(x){state('profileStatus',x.message,true)}finally{b.disabled=false}};$('convertForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('convertStatus','Modell arbeitet …');$('player').hidden=true;$('download').hidden=true;try{let f=new FormData(e.target);if(!f.has('pitch_shift'))f.append('pitch_shift','false');let r=await fetch('/api/convert',{method:'POST',body:f});if(!r.ok){let d=await r.json();throw Error(d.detail||`HTTP ${r.status}`)}let blob=await r.blob(),url=URL.createObjectURL(blob);$('player').src=url;$('player').hidden=false;$('download').href=url;$('download').download='voice-conversion.wav';$('download').hidden=false;state('convertStatus',`Fertig in ${r.headers.get('X-Conversion-Seconds')||'?'} s · Spitzen-VRAM ${r.headers.get('X-Peak-VRAM-MiB')||'?'} MiB`)}catch(x){state('convertStatus',x.message,true)}finally{b.disabled=false}};loadProfiles();
</script></body></html>
@@ -0,0 +1,47 @@
#!/usr/bin/env python3
"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
from __future__ import annotations
import argparse
import os
import time
import torch
import models.svc.vevo2.infer_vevo2_fm as vevo
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--source", required=True)
parser.add_argument("--reference", required=True)
parser.add_argument("--output", required=True)
parser.add_argument("--no-pitch-shift", action="store_true")
args = parser.parse_args()
os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
started = time.monotonic()
vevo.inference_pipeline = vevo.load_inference_pipeline()
loaded = time.monotonic()
vevo.vevo2_fm(
args.source,
args.reference,
args.output,
shifted_src=not args.no_pitch_shift,
)
finished = time.monotonic()
print(
{
"model_load_seconds": round(loaded - started, 3),
"conversion_seconds": round(finished - loaded, 3),
"total_seconds": round(finished - started, 3),
"peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
"output": args.output,
},
flush=True,
)
if __name__ == "__main__":
main()
@@ -32,6 +32,8 @@ MUSIC_LABEL_KEY = "com.mike-ai.music-worker"
MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip()
SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator"
SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
LOCK = threading.Lock()
log = logging.getLogger("profile-controller")
@@ -122,6 +124,17 @@ def separator_container() -> dict:
return matches[0]
def voice_container() -> dict:
if not VOICE_WORKER:
raise RuntimeError("voice worker is not configured")
matches = [item for item in labelled_containers(VOICE_LABEL_KEY)
if item.get("Labels", {}).get(VOICE_LABEL_KEY) == VOICE_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one voice worker {VOICE_WORKER!r}, found {len(matches)}")
return matches[0]
def stop_music_if_configured() -> None:
if MUSIC_WORKER:
stop_container(music_container(), timeout=30)
@@ -132,6 +145,11 @@ def stop_separator_if_configured() -> None:
stop_container(separator_container(), timeout=30)
def stop_voice_if_configured() -> None:
if VOICE_WORKER:
stop_container(voice_container(), timeout=30)
def stop_container(item: dict, timeout: int = 120) -> None:
if item.get("State") != "running":
return
@@ -171,6 +189,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_separator_if_configured()
stop_voice_if_configured()
for other in image_containers():
if other["Id"] != item["Id"]:
stop_container(other, timeout=20)
@@ -197,6 +216,7 @@ def set_music_worker(running: bool) -> dict:
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_separator_if_configured()
stop_voice_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
@@ -215,6 +235,7 @@ def set_separator_worker(running: bool) -> dict:
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_voice_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
@@ -222,6 +243,25 @@ def set_separator_worker(running: bool) -> dict:
"state": "running" if running else "stopped"}
def set_voice_worker(running: bool) -> dict:
"""Start Vevo2 exclusively, or stop it before LLM restoration."""
with LOCK:
item = voice_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_separator_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"voice_worker": VOICE_WORKER,
"state": "running" if running else "stopped"}
def active_profile(items: dict[str, dict] | None = None) -> str | None:
items = items or containers()
active = [name for name, item in items.items() if item.get("State") == "running"]
@@ -239,6 +279,7 @@ def activate(profile: str) -> dict:
stop_container(worker)
stop_music_if_configured()
stop_separator_if_configured()
stop_voice_if_configured()
start_container(tts_container())
items = containers()
missing = [name for name in ALLOWED if name not in items]
@@ -317,11 +358,20 @@ class Handler(BaseHTTPRequestHandler):
"unhealthy" if "(unhealthy)" in separator_status else
"starting" if separator.get("State") == "running" else
"stopped")
voice = voice_container() if VOICE_WORKER else {}
voice_status = voice.get("Status", "")
voice_health = ("disabled" if not VOICE_WORKER else
"healthy" if "(healthy)" in voice_status else
"unhealthy" if "(unhealthy)" in voice_status else
"starting" if voice.get("State") == "running" else
"stopped")
self.reply(200, {"active_profile": active_profile(items),
"music_worker": music.get("State", "disabled"),
"music_health": music_health,
"separator_worker": separator.get("State", "disabled"),
"separator_health": separator_health,
"voice_worker": voice.get("State", "disabled"),
"voice_health": voice_health,
"profiles": {name: items.get(name, {}).get(
"State", "missing") for name in ALLOWED}})
except Exception as exc:
@@ -353,6 +403,13 @@ class Handler(BaseHTTPRequestHandler):
log.exception("stem separator transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/voice/start", "/workers/voice/stop"}:
try:
self.reply(200, set_voice_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("voice worker transition failed")
self.reply(503, {"error": str(exc)})
return
worker_paths = {
"/workers/image/start": (IMAGE_WORKER, True),
"/workers/image/stop": (IMAGE_WORKER, False),
@@ -81,6 +81,7 @@ start_proxy 8099 llama-dashboard:8099
start_proxy 7861 music-ui:3000
start_proxy 7862 music-worker:7860
start_proxy 8007 stem-separator:8080
start_proxy 8008 voice-studio:8008
start_proxy 8202 mcp-athena-operator:8000
start_proxy 9443 portainer:9443
+7 -5
View File
@@ -28,6 +28,7 @@ MUSIC_ORIGINAL_UI_URL = os.getenv(
"MUSIC_ORIGINAL_UI_URL", "http://192.168.1.212:7862/"
)
SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
@@ -277,7 +278,7 @@ def router_status() -> tuple[dict[str, Any], str | None]:
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
if mode not in {"llm", "music", "separation"}:
if mode not in {"llm", "music", "separation", "voice"}:
return 400, {"error": "invalid mode"}
headers = {"Accept": "application/json", "Content-Type": "application/json"}
if ROUTER_API_KEY:
@@ -635,7 +636,7 @@ HTML = r'''<!doctype html>
</style></head><body><main>
<div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div>
<section class="grid">
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span></div></div></article>
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><button id="voiceMode" onclick="setMode('voice')">Voice Studio</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span><span id="voiceOpen" hidden><a class="stable" href="__VOICE_UI_URL__" target="_blank" rel="noopener">Voice Studio öffnen</a></span></div></div></article>
<article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article>
<article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article>
<article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article>
@@ -674,12 +675,13 @@ const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed
function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `<article class="card span6"><div class="gpu-title"><div><div class="label">GPU ${g.index}</div><div class="value">${g.name}</div></div><span class="badge">${g.pstate||'–'}</span></div><div class="metrics"><div class="metric"><b>${pct(g.gpu_percent)}</b><span>GPU-Kern</span></div><div class="metric"><b>${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiB</b><span>VRAM</span></div><div class="metric"><b>${g.temperature_c??'–'} °C</b><span>Temperatur</span></div><div class="metric"><b>${g.power_w??'–'} / ${g.power_limit_w??'–'} W</b><span>Leistung</span></div><div class="metric"><b>${g.graphics_clock_mhz??'–'} MHz</b><span>Grafiktakt</span></div><div class="metric"><b>${g.memory_clock_mhz??'–'} MHz</b><span>Speichertakt</span></div><div class="metric"><b>${pct(g.memory_controller_percent)}</b><span>Memory Controller</span></div><div class="metric"><b>${pct(g.fan_percent)}</b><span>Lüfter</span></div></div><div class="bar-label"><span>GPU-Auslastung</span><span>${load.toFixed(1)} %</span></div><div class="bar"><div class="fill gpu-load" style="width:${load}%"></div></div><div class="bar-label"><span>VRAM-Belegung</span><span>${p.toFixed(1)} %</span></div><div class="bar"><div class="fill" style="width:${Math.min(100,p)}%"></div></div><div class="sub">${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei</div></article>`}
const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit');
let modeBusy=false;
async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''.replace(
"__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL
).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL)
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL
).replace("__VOICE_UI_URL__", VOICE_UI_URL)
FULL_JS = r'''
+68 -18
View File
@@ -110,6 +110,7 @@ ENABLE_MUSIC_MODE = os.environ.get(
"ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"}
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
# Optional worker APIs. The clean Docker baseline deliberately ships only
# text/multimodal chat; absent workers must fail explicitly instead of trying
@@ -361,6 +362,27 @@ def _separator_worker_health() -> str:
return "unknown"
def _voice_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_worker", "missing"))
except Exception as exc:
log.warning("Voice-Worker-Status nicht verfügbar: %s", exc)
return "unknown"
def _voice_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_health", "unknown"))
except Exception:
return "unknown"
def _wait_music_ready() -> None:
deadline = time.monotonic() + MUSIC_START_TIMEOUT
while time.monotonic() < deadline:
@@ -389,17 +411,40 @@ def _wait_separator_ready() -> None:
f"BS-RoFormer nach {SEPARATOR_START_TIMEOUT:.0f} s nicht bereit")
def _wait_voice_ready() -> None:
deadline = time.monotonic() + VOICE_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("voice_worker") == "running"
and status.get("voice_health") == "healthy"):
return
if status.get("voice_health") == "unhealthy":
raise RuntimeError("Vevo2-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"Vevo2 nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
def _special_worker(mode: str) -> tuple[str, str, callable]:
if mode == "music":
return "/workers/music/start", _music_worker_state(), _wait_music_ready
if mode == "separation":
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
if mode == "voice":
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
raise ValueError(f"unbekannter Spezialmodus: {mode}")
def set_operating_mode(mode: str) -> dict:
"""Atomarer Wechsel zwischen LLM, ACE-Step und Stem-Separation."""
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music", "separation"}:
raise ValueError("Modus muss 'llm', 'music' oder 'separation' sein")
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
with STATE.lock:
STATE.mode_error = None
if mode in {"music", "separation"}:
worker_state = (_music_worker_state() if mode == "music"
else _separator_worker_state())
if mode in {"music", "separation", "voice"}:
path, worker_state, wait_ready = _special_worker(mode)
if STATE.mode == mode and worker_state == "running":
return {"status": "ok", "mode": mode, "changed": False}
profile = current_profile()
@@ -417,10 +462,8 @@ def set_operating_mode(mode: str) -> dict:
last_profile=return_profile,
phase=f"starting-{mode}")
_wait_chats_drained()
path = ("/workers/music/start" if mode == "music"
else "/workers/separator/start")
_profile_controller_request("POST", path)
_wait_music_ready() if mode == "music" else _wait_separator_ready()
wait_ready()
STATE.mode = mode
STATE.mode_phase = "ready"
RUNTIME.save(mode=mode, return_profile=return_profile,
@@ -441,6 +484,7 @@ def set_operating_mode(mode: str) -> dict:
try:
_profile_controller_request("POST", "/workers/music/stop")
_profile_controller_request("POST", "/workers/separator/stop")
_profile_controller_request("POST", "/workers/voice/stop")
_restore_qwen(profile)
STATE.mode = "llm"
STATE.mode_phase = "ready"
@@ -459,8 +503,8 @@ def schedule_operating_mode(mode: str) -> tuple[bool, str]:
"""Start a transition in the background so chat/UI acknowledgement is instant."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music", "separation"}:
raise ValueError("Modus muss 'llm', 'music' oder 'separation' sein")
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
with STATE.lock:
if STATE.mode_phase not in {"ready", "error"}:
return False, STATE.mode_phase
@@ -496,6 +540,7 @@ def _control_command(data: dict, path: str) -> str | None:
command = text.strip().casefold()
return command if command in {"/athena music", "/athena stems",
"/athena separation", "/athena llm",
"/athena voice",
"/athena status"} else None
@@ -1971,6 +2016,8 @@ class Handler(BaseHTTPRequestHandler):
"music_health": _music_worker_health(),
"separator_worker": _separator_worker_state(),
"separator_health": _separator_worker_health(),
"voice_worker": _voice_worker_state(),
"voice_health": _voice_worker_health(),
"return_profile": state.get("return_profile"),
"last_error": STATE.mode_error,
"enabled": ENABLE_MUSIC_MODE,
@@ -1980,8 +2027,8 @@ class Handler(BaseHTTPRequestHandler):
try:
data = json.loads(self._read_body() or b"{}")
mode = data.get("mode") if isinstance(data, dict) else None
if mode not in {"llm", "music", "separation"}:
raise ValueError("Feld 'mode' muss 'llm', 'music' oder 'separation' sein")
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation' oder 'voice' sein")
started, phase = schedule_operating_mode(mode)
self._send_json(202 if started else 200, {
"status": "accepted" if started else "ok",
@@ -2644,10 +2691,12 @@ class Handler(BaseHTTPRequestHandler):
text = (f"Athena läuft im {mode['active'].upper()}-Modus. "
f"Phase: {mode['phase']}. Musik-Worker: "
f"{mode['music_worker']}. Stem-Separator: "
f"{mode['separator_worker']}. LLM-Profil: {profile or 'entladen'}.")
f"{mode['separator_worker']}. Voice Studio: "
f"{mode['voice_worker']}. LLM-Profil: {profile or 'entladen'}.")
else:
target = ("music" if command == "/athena music" else
"separation" if command in {"/athena stems", "/athena separation"}
else "voice" if command == "/athena voice"
else "llm")
try:
started, phase = schedule_operating_mode(target)
@@ -2656,6 +2705,8 @@ class Handler(BaseHTTPRequestHandler):
if target == "music" else
"Stimmtrennung wird gestartet. LLM und TTS werden entladen."
if target == "separation" else
"Voice Studio wird gestartet. LLM und TTS werden entladen."
if target == "voice" else
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
else:
text = (f"Athena ist bereits im {target.upper()}-Modus "
@@ -2948,15 +2999,14 @@ def _startup_reconcile() -> None:
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
special_mode = previous.get("mode")
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation"}:
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice"}:
STATE.mode = special_mode
STATE.mode_phase = f"starting-{special_mode}"
_set_qwen_unavailable(True)
try:
path = ("/workers/music/start" if special_mode == "music"
else "/workers/separator/start")
path, _worker_state, wait_ready = _special_worker(special_mode)
_profile_controller_request("POST", path)
_wait_music_ready() if special_mode == "music" else _wait_separator_ready()
wait_ready()
STATE.mode_phase = "ready"
RUNTIME.save(mode=special_mode, phase=special_mode)
log.info("Recovery: Spezialmodus %s wiederhergestellt", special_mode)