From 68d02f32bddb1bee77d9b3e53b9b6b01aea3f5eb Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Wed, 9 Sep 2026 13:38:00 +0200 Subject: [PATCH] Add private Vevo2 voice studio mode --- compose.yaml | 2 + dev/test_dashboard_modes.py | 9 + dev/test_profile_controller.py | 36 ++++ docs/OPERATING_MODES.md | 22 +- docs/TESTED_MODELS.md | 3 +- experiments/vevo2-voice-cloning/Dockerfile | 64 ++++++ experiments/vevo2-voice-cloning/README.md | 30 +++ experiments/vevo2-voice-cloning/app.py | 202 ++++++++++++++++++ experiments/vevo2-voice-cloning/compose.yaml | 39 ++++ experiments/vevo2-voice-cloning/index.html | 10 + .../vevo2-voice-cloning/run_fm_test.py | 47 ++++ .../profile-controller/profile_controller.py | 57 +++++ .../docker/wireguard-gateway/entrypoint.sh | 1 + platform/llama-dashboard/app.py | 12 +- router/ai_profile_router.py | 86 ++++++-- 15 files changed, 592 insertions(+), 28 deletions(-) create mode 100644 experiments/vevo2-voice-cloning/Dockerfile create mode 100644 experiments/vevo2-voice-cloning/README.md create mode 100644 experiments/vevo2-voice-cloning/app.py create mode 100644 experiments/vevo2-voice-cloning/compose.yaml create mode 100644 experiments/vevo2-voice-cloning/index.html create mode 100644 experiments/vevo2-voice-cloning/run_fm_test.py diff --git a/compose.yaml b/compose.yaml index c22ce74..731722e 100644 --- a/compose.yaml +++ b/compose.yaml @@ -571,6 +571,7 @@ services: TTS_WORKER: qwen3 MUSIC_WORKER: acestep SEPARATOR_WORKER: bs-roformer + VOICE_WORKER: vevo2 networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: @@ -860,6 +861,7 @@ services: MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}" MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}" SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}" + VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}" HOST_PROC: /host/proc HOST_DATA: /host/data HOST_MODELS: /host/models diff --git a/dev/test_dashboard_modes.py b/dev/test_dashboard_modes.py index ccadbfb..b6f546e 100644 --- a/dev/test_dashboard_modes.py +++ b/dev/test_dashboard_modes.py @@ -52,6 +52,15 @@ class DashboardModeTests(unittest.TestCase): self.assertEqual(json.loads(request.data), {"mode": "separation"}) self.assertEqual(request.get_header("Authorization"), "Bearer test-key") + def test_voice_mode_is_forwarded_to_router(self): + with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen: + status, body = self.dashboard.change_mode("voice") + + self.assertEqual(status, 202) + self.assertEqual(body, {"status": "accepted"}) + request = urlopen.call_args.args[0] + self.assertEqual(json.loads(request.data), {"mode": "voice"}) + def test_unknown_mode_is_rejected_without_router_request(self): with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen: status, body = self.dashboard.change_mode("unknown") diff --git a/dev/test_profile_controller.py b/dev/test_profile_controller.py index 95d0f06..c44a845 100644 --- a/dev/test_profile_controller.py +++ b/dev/test_profile_controller.py @@ -46,7 +46,43 @@ def separator_item(state="exited"): "Labels": {controller.SEPARATOR_LABEL_KEY: "bs-roformer"}} +def voice_item(state="exited"): + return {"Id": "id-voice", "State": state, + "Labels": {controller.VOICE_LABEL_KEY: "vevo2"}} + + class ProfileControllerTests(unittest.TestCase): + def test_voice_start_exclusively_stops_gpu_workers(self): + profiles = {name: item(name) for name in controller.ALLOWED} + profiles["medium"] = item("medium", "running") + calls = [] + + def request(method, path): + calls.append((method, path)) + return 204, b"" + + with patch.object(controller, "VOICE_WORKER", "vevo2"), \ + patch.object(controller, "MUSIC_WORKER", "acestep"), \ + patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \ + patch.object(controller, "containers", return_value=profiles), \ + patch.object(controller, "voice_container", return_value=voice_item()), \ + patch.object(controller, "music_container", return_value=music_item("running")), \ + patch.object(controller, "separator_container", return_value=separator_item("running")), \ + patch.object(controller, "image_containers", return_value=[image_item("running")]), \ + patch.object(controller, "tts_container", return_value=tts_item()), \ + patch.object(controller, "docker_request", side_effect=request): + result = controller.set_voice_worker(True) + + self.assertEqual(result, {"voice_worker": "vevo2", "state": "running"}) + self.assertEqual(calls, [ + ("POST", "/containers/id-medium/stop?t=120"), + ("POST", "/containers/id-flux/stop?t=20"), + ("POST", "/containers/id-tts/stop?t=30"), + ("POST", "/containers/id-music/stop?t=30"), + ("POST", "/containers/id-separator/stop?t=30"), + ("POST", "/containers/id-voice/start"), + ]) + def test_separator_start_exclusively_stops_gpu_workers(self): profiles = {name: item(name) for name in controller.ALLOWED} profiles["large"] = item("large", "running") diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md index 0bc9fd6..8d9b1f3 100644 --- a/docs/OPERATING_MODES.md +++ b/docs/OPERATING_MODES.md @@ -1,11 +1,13 @@ # Athena-Betriebsmodi -Athena besitzt drei gegenseitig exklusive Betriebsmodi: +Athena besitzt vier gegenseitig exklusive Betriebsmodi: - `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt. - `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt. - `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt. +- `voice`: Vevo2 überträgt Sprache oder Gesang auf eine gespeicherte + Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt. Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus @@ -13,8 +15,8 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen ## Bedienung -Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio** und -**Audio trennen** bereit. Im Musikmodus werden zwei Oberflächen angeboten: +Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen** +und **Voice Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten: - **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der @@ -48,12 +50,22 @@ nicht eine reine Synthesizer-Spur. Sprache nutzt das 48-kHz-Modell `audio-separator` 0.47.0. Die ältere API-Auswahl kompletter 2-/4-/6-Stem-Sätze bleibt rückwärtskompatibel. +Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter +`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter +`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem +Speichern eine Bestätigung der Nutzungsberechtigung. Vevo2 gibt +unkomprimiertes 24-kHz-WAV aus und wandelt sowohl Sprache als auch Gesang; +dies ist noch kein Echtzeit-Live-Voice-Changer. Die Gewichte sind +CC BY-NC-ND 4.0 lizenziert und werden auf Athena ausschließlich privat und +nichtkommerziell eingesetzt. + Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom Router lokal beantwortet, auch wenn gerade kein LLM geladen ist: ```text /athena music /athena stems +/athena voice /athena llm /athena status ``` @@ -64,13 +76,15 @@ Die HTTP-Schnittstelle verwendet authentifizierte Requests: GET /mode POST /mode {"mode":"music"} POST /mode {"mode":"separation"} +POST /mode {"mode":"voice"} POST /mode {"mode":"llm"} ``` Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in `GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit `com.mike-ai.music-worker=acestep` beziehungsweise -`com.mike-ai.stem-separator=bs-roformer` markierten Container; freie +`com.mike-ai.stem-separator=bs-roformer` oder +`com.mike-ai.voice-worker=vevo2` markierten Container; freie Container- oder Docker-Befehle werden nicht entgegengenommen. ## Wiederanlauf diff --git a/docs/TESTED_MODELS.md b/docs/TESTED_MODELS.md index 1d1ef28..01a148e 100644 --- a/docs/TESTED_MODELS.md +++ b/docs/TESTED_MODELS.md @@ -1,6 +1,6 @@ # Register getesteter Modelle -Stand: 8. September 2026 +Stand: 9. September 2026 Dieses Dokument ist die zentrale Sperrliste gegen doppelte Modelltests. Vor jedem Download müssen Repository, Dateiname, Basismodell, Fine-Tune und @@ -56,6 +56,7 @@ Titelgenerierung und Kontextkompression in Hermes. | 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** | | 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback | | 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** | +| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Produktions-HTTP-API mit offizieller 8,6-s-Sprachprobe: Warmstart 12,216 s, Wandlung 2,342 s, Spitzen-VRAM 5.605,6 MiB; gültiges 24-kHz-WAV mit 412.878 Byte; Profilanlage, Download und automatische Job-Bereinigung geprüft | **technischer Ende-zu-Ende-Test bestanden**, privater Voice-Studio-Betatest; Gewichte CC BY-NC-ND 4.0 und daher nicht kommerziell einsetzen | ## Musikgenerierung diff --git a/experiments/vevo2-voice-cloning/Dockerfile b/experiments/vevo2-voice-cloning/Dockerfile new file mode 100644 index 0000000..827316f --- /dev/null +++ b/experiments/vevo2-voice-cloning/Dockerfile @@ -0,0 +1,64 @@ +FROM mike-ai/bs-roformer-separator:0.47.0 + +ARG AMPHION_COMMIT=26f6883110181f1dbfe95c70a7c7dbaf4de5f42a + +USER root +RUN apt-get update \ + && apt-get install -y --no-install-recommends git espeak-ng \ + && rm -rf /var/lib/apt/lists/* + +RUN git clone https://github.com/open-mmlab/Amphion.git /opt/amphion \ + && cd /opt/amphion \ + && git checkout "${AMPHION_COMMIT}" \ + && rm -rf .git + +# Vevo2 is tested on Athena with the CUDA/PyTorch stack inherited from the +# existing audio worker. Do not install Amphion's historical torch 2.0/cu118 +# pins: Blackwell requires the newer cu128 runtime already present here. +RUN python -m pip install --no-cache-dir \ + accelerate==1.10.1 \ + diffusers==0.35.1 \ + einops==0.8.1 \ + easydict==1.13 \ + g2p_en==2.1.0 \ + humanfriendly==10.0 \ + huggingface-hub==0.34.4 \ + hydra-core==1.3.2 \ + inflect==7.5.0 \ + ipython==9.5.0 \ + json5==0.12.1 \ + librosa==0.11.0 \ + loguru==0.7.3 \ + matplotlib==3.10.6 \ + munch==4.0.0 \ + omegaconf==2.3.0 \ + openai-whisper==20250625 \ + phonemizer==3.3.0 \ + python-multipart==0.0.20 \ + praat-parselmouth==0.4.6 \ + pypinyin==0.55.0 \ + pyworld==0.3.5 \ + ruamel.yaml==0.18.15 \ + safetensors==0.6.2 \ + tabulate==0.9.0 \ + tgt==1.5 \ + torchcrepe==0.0.24 \ + transformers==4.56.1 \ + typeguard==4.4.4 \ + unidecode==1.4.0 \ + vector-quantize-pytorch==1.12.5 \ + vocos==0.1.0 + +WORKDIR /opt/amphion +ENV PYTHONPATH=/opt/amphion \ + HF_HOME=/models/huggingface \ + PYTHONUNBUFFERED=1 + +COPY run_fm_test.py /usr/local/bin/run_fm_test.py +COPY app.py /app/app.py +COPY index.html /app/index.html + +EXPOSE 8008 +HEALTHCHECK --interval=5s --timeout=3s --start-period=90s --retries=3 \ + CMD curl -fsS http://127.0.0.1:8008/health || exit 1 +ENTRYPOINT ["python", "/app/app.py"] diff --git a/experiments/vevo2-voice-cloning/README.md b/experiments/vevo2-voice-cloning/README.md new file mode 100644 index 0000000..2718379 --- /dev/null +++ b/experiments/vevo2-voice-cloning/README.md @@ -0,0 +1,30 @@ +# Vevo2 Voice Conversion on Athena + +This directory contains Athena's private Vevo2 voice-conversion studio. + +- Code: `open-mmlab/Amphion` commit + `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` (MIT) +- Weights: `RMSnow/Vevo2` (CC BY-NC-ND 4.0) +- Runtime: PyTorch 2.7.1 + CUDA 12.8 inherited from Athena's tested audio + separator image; the historical Amphion torch 2.0/cu118 pins are not used. +- Storage: `/data/voice/vevo2`; removing that directory and the test image + removes all downloaded artifacts. +- Private UI: `http://192.168.1.212:8008` through WireGuard only. +- Output: uncompressed mono WAV, 24 kHz. + +The 9 September technical gate converted the official 8.6-second speech sample +through the production HTTP API in 2.342 seconds. A warm service start loaded +the model in 12.216 seconds, and peak CUDA allocation during conversion was +5605.6 MiB. The returned 24-kHz WAV was 412878 bytes and 8.6 seconds long. The +service stores named reference voices, accepts a source clip, and returns a +transient WAV download. Jobs and generated outputs are removed after delivery. +It is deliberately not exposed on the university interface. + +The installed Compose project is named `mike-ai-voice`. Rebuild or recreate it +with an explicit `VOICE_GPU_UUID` so Docker keeps the service attached to the +RTX 5080. The profile controller starts and stops the existing container; it +does not rebuild it during a mode switch. + +The weights are licensed CC BY-NC-ND 4.0. This deployment is for Mike's +private, non-commercial use only. Do not use or expose it as a public or +commercial voice-cloning service. diff --git a/experiments/vevo2-voice-cloning/app.py b/experiments/vevo2-voice-cloning/app.py new file mode 100644 index 0000000..40288e8 --- /dev/null +++ b/experiments/vevo2-voice-cloning/app.py @@ -0,0 +1,202 @@ +#!/usr/bin/env python3 +"""Small local-only Vevo2 voice-conversion studio for Athena.""" + +from __future__ import annotations + +import asyncio +import os +import re +import shutil +import subprocess +import threading +import time +import uuid +from contextlib import asynccontextmanager +from pathlib import Path + +import torch +from fastapi import FastAPI, File, Form, HTTPException, UploadFile +from fastapi.responses import FileResponse, HTMLResponse +from starlette.background import BackgroundTask + +import models.svc.vevo2.infer_vevo2_fm as vevo + + +DATA_DIR = Path(os.environ.get("VOICE_DATA_DIR", "/data")) +PROFILE_DIR = DATA_DIR / "profiles" +JOB_DIR = DATA_DIR / "jobs" +INDEX = Path("/app/index.html") +MAX_UPLOAD_BYTES = int(os.environ.get("MAX_UPLOAD_BYTES", str(100 * 1024 * 1024))) +MODEL_LOCK = threading.Lock() +PIPELINE = None +MODEL_LOAD_SECONDS: float | None = None + + +def safe_name(value: str) -> str: + value = re.sub(r"[^A-Za-z0-9_. -]+", "-", value.strip()) + value = re.sub(r"\s+", "-", value).strip("-.") + return value[:64] or "voice" + + +def load_pipeline() -> None: + global PIPELINE, MODEL_LOAD_SECONDS + if PIPELINE is not None: + return + with MODEL_LOCK: + if PIPELINE is not None: + return + started = time.monotonic() + PIPELINE = vevo.load_inference_pipeline() + vevo.inference_pipeline = PIPELINE + MODEL_LOAD_SECONDS = time.monotonic() - started + + +def to_wav(source: Path, target: Path) -> None: + completed = subprocess.run( + ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source), + "-ac", "1", "-ar", "24000", "-c:a", "pcm_s16le", str(target)], + capture_output=True, + text=True, + timeout=180, + check=False, + ) + if completed.returncode: + raise ValueError(completed.stderr.strip() or "Audio konnte nicht gelesen werden") + + +async def save_upload(upload: UploadFile, target: Path) -> None: + size = 0 + with target.open("wb") as handle: + while chunk := await upload.read(1024 * 1024): + size += len(chunk) + if size > MAX_UPLOAD_BYTES: + raise HTTPException(413, "Audiodatei ist zu groß") + handle.write(chunk) + + +def profile_path(name: str) -> Path: + target = PROFILE_DIR / f"{safe_name(name)}.wav" + if not target.is_file(): + raise HTTPException(404, "Referenzstimme wurde nicht gefunden") + return target + + +def run_conversion(source: Path, reference: Path, output: Path, pitch_shift: bool) -> tuple[float, float]: + """Run the GPU-bound conversion off the API event loop.""" + load_pipeline() + with MODEL_LOCK: + torch.cuda.reset_peak_memory_stats() + started = time.monotonic() + vevo.vevo2_fm(str(source), str(reference), str(output), shifted_src=pitch_shift) + elapsed = time.monotonic() - started + peak = torch.cuda.max_memory_allocated() / 1048576 + return elapsed, peak + + +@asynccontextmanager +async def lifespan(_app: FastAPI): + PROFILE_DIR.mkdir(parents=True, exist_ok=True) + JOB_DIR.mkdir(parents=True, exist_ok=True) + load_pipeline() + yield + + +app = FastAPI(title="Athena Voice Studio", lifespan=lifespan) + + +@app.get("/", response_class=HTMLResponse) +def index() -> str: + return INDEX.read_text(encoding="utf-8") + + +@app.get("/health") +def health() -> dict: + return { + "status": "ok" if PIPELINE is not None else "starting", + "model": "RMSnow/Vevo2", + "sample_rate": 24000, + "model_load_seconds": MODEL_LOAD_SECONDS, + } + + +@app.get("/api/profiles") +def profiles() -> dict: + return {"profiles": [path.stem for path in sorted(PROFILE_DIR.glob("*.wav"))]} + + +@app.post("/api/profiles") +async def create_profile( + name: str = Form(...), + consent: bool = Form(False), + audio: UploadFile = File(...), +) -> dict: + if not consent: + raise HTTPException(400, "Bestätige, dass du die Stimme verwenden darfst") + clean = safe_name(name) + job = JOB_DIR / f"profile-{uuid.uuid4().hex}" + job.mkdir(parents=True) + raw = job / f"upload-{safe_name(audio.filename or 'reference.audio')}" + try: + await save_upload(audio, raw) + target = PROFILE_DIR / f"{clean}.wav" + temporary = job / "reference.wav" + to_wav(raw, temporary) + os.replace(temporary, target) + return {"status": "ok", "profile": clean} + except HTTPException: + raise + except Exception as exc: + raise HTTPException(400, str(exc)) from exc + finally: + shutil.rmtree(job, ignore_errors=True) + + +@app.delete("/api/profiles/{name}") +def delete_profile(name: str) -> dict: + target = profile_path(name) + target.unlink() + return {"status": "ok", "profile": target.stem} + + +@app.post("/api/convert") +async def convert( + source: UploadFile = File(...), + profile: str = Form(...), + pitch_shift: bool = Form(True), +) -> FileResponse: + reference = profile_path(profile) + job_id = uuid.uuid4().hex + job = JOB_DIR / f"convert-{job_id}" + job.mkdir(parents=True) + raw = job / f"upload-{safe_name(source.filename or 'source.audio')}" + source_wav = job / "source.wav" + output = JOB_DIR / f"voice-{job_id}.wav" + try: + await save_upload(source, raw) + to_wav(raw, source_wav) + elapsed, peak = await asyncio.to_thread( + run_conversion, source_wav, reference, output, pitch_shift + ) + return FileResponse( + output, + media_type="audio/wav", + filename=f"{safe_name(profile)}-{job_id[:8]}.wav", + headers={ + "X-Conversion-Seconds": f"{elapsed:.3f}", + "X-Peak-VRAM-MiB": f"{peak:.1f}", + }, + background=BackgroundTask(output.unlink, missing_ok=True), + ) + except HTTPException: + raise + except Exception as exc: + output.unlink(missing_ok=True) + raise HTTPException(500, f"Stimmenwandlung fehlgeschlagen: {exc}") from exc + finally: + shutil.rmtree(job, ignore_errors=True) + + +if __name__ == "__main__": + import uvicorn + + uvicorn.run(app, host="0.0.0.0", port=8008) diff --git a/experiments/vevo2-voice-cloning/compose.yaml b/experiments/vevo2-voice-cloning/compose.yaml new file mode 100644 index 0000000..9ad91f6 --- /dev/null +++ b/experiments/vevo2-voice-cloning/compose.yaml @@ -0,0 +1,39 @@ +services: + voice-studio: + build: . + image: mike-ai/vevo2-voice-studio:0.1 + container_name: mike-ai-voice-studio + restart: "no" + labels: + com.mike-ai.voice-worker: vevo2 + environment: + NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID} + NVIDIA_DRIVER_CAPABILITIES: compute,utility + VOICE_DATA_DIR: /data + ports: + - "127.0.0.1:8008:8008" + volumes: + - /data/voice/vevo2/huggingface/hub/Vevo2:/opt/amphion/ckpts/Vevo2 + - /data/voice/vevo2/huggingface:/models/huggingface + - /data/voice/vevo2/whisper:/root/.cache/whisper + - /data/voice/studio:/data + healthcheck: + test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/health | grep -q '\"status\":\"ok\"'"] + interval: 5s + timeout: 3s + start_period: 600s + retries: 3 + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"] + capabilities: [gpu] + networks: + - frontend + +networks: + frontend: + name: mike-ai_frontend + external: true diff --git a/experiments/vevo2-voice-cloning/index.html b/experiments/vevo2-voice-cloning/index.html new file mode 100644 index 0000000..af579be --- /dev/null +++ b/experiments/vevo2-voice-cloning/index.html @@ -0,0 +1,10 @@ + + +Athena Voice Studio +
Mike AI · lokal auf Athena

Voice Studio

Eine Stimme als Referenz speichern und eine vorhandene Sprach- oder Gesangsaufnahme in diese Stimme übertragen. Ausgabe: unkomprimiertes WAV mit 24 kHz.

+

1 · Referenzstimme

+

2 · Stimme übertragen

+

Hinweis: Nur privat und mit Zustimmung verwenden. Das Vevo2-Gewicht ist nicht für kommerzielle Nutzung lizenziert.

diff --git a/experiments/vevo2-voice-cloning/run_fm_test.py b/experiments/vevo2-voice-cloning/run_fm_test.py new file mode 100644 index 0000000..8e31d93 --- /dev/null +++ b/experiments/vevo2-voice-cloning/run_fm_test.py @@ -0,0 +1,47 @@ +#!/usr/bin/env python3 +"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test.""" + +from __future__ import annotations + +import argparse +import os +import time + +import torch + +import models.svc.vevo2.infer_vevo2_fm as vevo + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("--source", required=True) + parser.add_argument("--reference", required=True) + parser.add_argument("--output", required=True) + parser.add_argument("--no-pitch-shift", action="store_true") + args = parser.parse_args() + + os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True) + started = time.monotonic() + vevo.inference_pipeline = vevo.load_inference_pipeline() + loaded = time.monotonic() + vevo.vevo2_fm( + args.source, + args.reference, + args.output, + shifted_src=not args.no_pitch_shift, + ) + finished = time.monotonic() + print( + { + "model_load_seconds": round(loaded - started, 3), + "conversion_seconds": round(finished - loaded, 3), + "total_seconds": round(finished - started, 3), + "peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1), + "output": args.output, + }, + flush=True, + ) + + +if __name__ == "__main__": + main() diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index 341726a..c3b367d 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -32,6 +32,8 @@ MUSIC_LABEL_KEY = "com.mike-ai.music-worker" MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip() SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator" SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip() +VOICE_LABEL_KEY = "com.mike-ai.voice-worker" +VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip() LOCK = threading.Lock() log = logging.getLogger("profile-controller") @@ -122,6 +124,17 @@ def separator_container() -> dict: return matches[0] +def voice_container() -> dict: + if not VOICE_WORKER: + raise RuntimeError("voice worker is not configured") + matches = [item for item in labelled_containers(VOICE_LABEL_KEY) + if item.get("Labels", {}).get(VOICE_LABEL_KEY) == VOICE_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one voice worker {VOICE_WORKER!r}, found {len(matches)}") + return matches[0] + + def stop_music_if_configured() -> None: if MUSIC_WORKER: stop_container(music_container(), timeout=30) @@ -132,6 +145,11 @@ def stop_separator_if_configured() -> None: stop_container(separator_container(), timeout=30) +def stop_voice_if_configured() -> None: + if VOICE_WORKER: + stop_container(voice_container(), timeout=30) + + def stop_container(item: dict, timeout: int = 120) -> None: if item.get("State") != "running": return @@ -171,6 +189,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: stop_container(tts_container(), timeout=30) stop_music_if_configured() stop_separator_if_configured() + stop_voice_if_configured() for other in image_containers(): if other["Id"] != item["Id"]: stop_container(other, timeout=20) @@ -197,6 +216,7 @@ def set_music_worker(running: bool) -> dict: stop_container(worker, timeout=20) stop_container(tts_container(), timeout=30) stop_separator_if_configured() + stop_voice_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -215,6 +235,7 @@ def set_separator_worker(running: bool) -> dict: stop_container(worker, timeout=20) stop_container(tts_container(), timeout=30) stop_music_if_configured() + stop_voice_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -222,6 +243,25 @@ def set_separator_worker(running: bool) -> dict: "state": "running" if running else "stopped"} +def set_voice_worker(running: bool) -> dict: + """Start Vevo2 exclusively, or stop it before LLM restoration.""" + with LOCK: + item = voice_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_separator_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"voice_worker": VOICE_WORKER, + "state": "running" if running else "stopped"} + + def active_profile(items: dict[str, dict] | None = None) -> str | None: items = items or containers() active = [name for name, item in items.items() if item.get("State") == "running"] @@ -239,6 +279,7 @@ def activate(profile: str) -> dict: stop_container(worker) stop_music_if_configured() stop_separator_if_configured() + stop_voice_if_configured() start_container(tts_container()) items = containers() missing = [name for name in ALLOWED if name not in items] @@ -317,11 +358,20 @@ class Handler(BaseHTTPRequestHandler): "unhealthy" if "(unhealthy)" in separator_status else "starting" if separator.get("State") == "running" else "stopped") + voice = voice_container() if VOICE_WORKER else {} + voice_status = voice.get("Status", "") + voice_health = ("disabled" if not VOICE_WORKER else + "healthy" if "(healthy)" in voice_status else + "unhealthy" if "(unhealthy)" in voice_status else + "starting" if voice.get("State") == "running" else + "stopped") self.reply(200, {"active_profile": active_profile(items), "music_worker": music.get("State", "disabled"), "music_health": music_health, "separator_worker": separator.get("State", "disabled"), "separator_health": separator_health, + "voice_worker": voice.get("State", "disabled"), + "voice_health": voice_health, "profiles": {name: items.get(name, {}).get( "State", "missing") for name in ALLOWED}}) except Exception as exc: @@ -353,6 +403,13 @@ class Handler(BaseHTTPRequestHandler): log.exception("stem separator transition failed") self.reply(503, {"error": str(exc)}) return + if self.path in {"/workers/voice/start", "/workers/voice/stop"}: + try: + self.reply(200, set_voice_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("voice worker transition failed") + self.reply(503, {"error": str(exc)}) + return worker_paths = { "/workers/image/start": (IMAGE_WORKER, True), "/workers/image/stop": (IMAGE_WORKER, False), diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh index f0b4b3d..ef2633c 100644 --- a/platform/docker/wireguard-gateway/entrypoint.sh +++ b/platform/docker/wireguard-gateway/entrypoint.sh @@ -81,6 +81,7 @@ start_proxy 8099 llama-dashboard:8099 start_proxy 7861 music-ui:3000 start_proxy 7862 music-worker:7860 start_proxy 8007 stem-separator:8080 +start_proxy 8008 voice-studio:8008 start_proxy 8202 mcp-athena-operator:8000 start_proxy 9443 portainer:9443 diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py index 2581b4c..6f87968 100644 --- a/platform/llama-dashboard/app.py +++ b/platform/llama-dashboard/app.py @@ -28,6 +28,7 @@ MUSIC_ORIGINAL_UI_URL = os.getenv( "MUSIC_ORIGINAL_UI_URL", "http://192.168.1.212:7862/" ) SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/") +VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/") HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) HOST_DATA = os.getenv("HOST_DATA", "/host/data") HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) @@ -277,7 +278,7 @@ def router_status() -> tuple[dict[str, Any], str | None]: def change_mode(mode: str) -> tuple[int, dict[str, Any]]: - if mode not in {"llm", "music", "separation"}: + if mode not in {"llm", "music", "separation", "voice"}: return 400, {"error": "invalid mode"} headers = {"Accept": "application/json", "Content-Type": "application/json"} if ROUTER_API_KEY: @@ -635,7 +636,7 @@ HTML = r'''
Mike AI · Live Telemetry

Athena llama.cpp Dashboard

verbinde …
- +
Aktives Profil
–
Router wird abgefragt
Modell
–
–
CPU
–
–
@@ -674,12 +675,13 @@ const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `
GPU ${g.index}
${g.name}
${g.pstate||'–'}
${pct(g.gpu_percent)}GPU-Kern
${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiBVRAM
${g.temperature_c??'–'} °CTemperatur
${g.power_w??'–'} / ${g.power_limit_w??'–'} WLeistung
${g.graphics_clock_mhz??'–'} MHzGrafiktakt
${g.memory_clock_mhz??'–'} MHzSpeichertakt
${pct(g.memory_controller_percent)}Memory Controller
${pct(g.fan_percent)}Lüfter
GPU-Auslastung${load.toFixed(1)} %
VRAM-Belegung${p.toFixed(1)} %
${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei
`} const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit'); let modeBusy=false; -async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}} -async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'
Keine GPU-Daten verfügbar
';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}${p.name}${p.pid}${p.memory_mib??'–'} MiB`).join('')||'Keine Compute-Prozesse gemeldet';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`
${v??'–'}${k}
`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000); +async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}} +async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'
Keine GPU-Daten verfügbar
';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}${p.name}${p.pid}${p.memory_mib??'–'} MiB`).join('')||'Keine Compute-Prozesse gemeldet';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`
${v??'–'}${k}
`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000); '''.replace( "__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL ).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL -).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL) +).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL +).replace("__VOICE_UI_URL__", VOICE_UI_URL) FULL_JS = r''' diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index fd365c8..cab4955 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -110,6 +110,7 @@ ENABLE_MUSIC_MODE = os.environ.get( "ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"} MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600")) SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600")) +VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600")) # Optional worker APIs. The clean Docker baseline deliberately ships only # text/multimodal chat; absent workers must fail explicitly instead of trying @@ -361,6 +362,27 @@ def _separator_worker_health() -> str: return "unknown" +def _voice_worker_state() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "voice_worker", "missing")) + except Exception as exc: + log.warning("Voice-Worker-Status nicht verfügbar: %s", exc) + return "unknown" + + +def _voice_worker_health() -> str: + if not PROFILE_CONTROL_URL: + return "unsupported" + try: + return str(_profile_controller_request("GET", "/status").get( + "voice_health", "unknown")) + except Exception: + return "unknown" + + def _wait_music_ready() -> None: deadline = time.monotonic() + MUSIC_START_TIMEOUT while time.monotonic() < deadline: @@ -389,17 +411,40 @@ def _wait_separator_ready() -> None: f"BS-RoFormer nach {SEPARATOR_START_TIMEOUT:.0f} s nicht bereit") +def _wait_voice_ready() -> None: + deadline = time.monotonic() + VOICE_START_TIMEOUT + while time.monotonic() < deadline: + status = _profile_controller_request("GET", "/status") + if (status.get("voice_worker") == "running" + and status.get("voice_health") == "healthy"): + return + if status.get("voice_health") == "unhealthy": + raise RuntimeError("Vevo2-Container ist unhealthy") + time.sleep(POLL_INTERVAL) + raise RuntimeError( + f"Vevo2 nach {VOICE_START_TIMEOUT:.0f} s nicht bereit") + + +def _special_worker(mode: str) -> tuple[str, str, callable]: + if mode == "music": + return "/workers/music/start", _music_worker_state(), _wait_music_ready + if mode == "separation": + return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready + if mode == "voice": + return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready + raise ValueError(f"unbekannter Spezialmodus: {mode}") + + def set_operating_mode(mode: str) -> dict: - """Atomarer Wechsel zwischen LLM, ACE-Step und Stem-Separation.""" + """Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen.""" if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL: raise RuntimeError("Musikmodus ist nicht konfiguriert") - if mode not in {"llm", "music", "separation"}: - raise ValueError("Modus muss 'llm', 'music' oder 'separation' sein") + if mode not in {"llm", "music", "separation", "voice"}: + raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein") with STATE.lock: STATE.mode_error = None - if mode in {"music", "separation"}: - worker_state = (_music_worker_state() if mode == "music" - else _separator_worker_state()) + if mode in {"music", "separation", "voice"}: + path, worker_state, wait_ready = _special_worker(mode) if STATE.mode == mode and worker_state == "running": return {"status": "ok", "mode": mode, "changed": False} profile = current_profile() @@ -417,10 +462,8 @@ def set_operating_mode(mode: str) -> dict: last_profile=return_profile, phase=f"starting-{mode}") _wait_chats_drained() - path = ("/workers/music/start" if mode == "music" - else "/workers/separator/start") _profile_controller_request("POST", path) - _wait_music_ready() if mode == "music" else _wait_separator_ready() + wait_ready() STATE.mode = mode STATE.mode_phase = "ready" RUNTIME.save(mode=mode, return_profile=return_profile, @@ -441,6 +484,7 @@ def set_operating_mode(mode: str) -> dict: try: _profile_controller_request("POST", "/workers/music/stop") _profile_controller_request("POST", "/workers/separator/stop") + _profile_controller_request("POST", "/workers/voice/stop") _restore_qwen(profile) STATE.mode = "llm" STATE.mode_phase = "ready" @@ -459,8 +503,8 @@ def schedule_operating_mode(mode: str) -> tuple[bool, str]: """Start a transition in the background so chat/UI acknowledgement is instant.""" if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL: raise RuntimeError("Musikmodus ist nicht konfiguriert") - if mode not in {"llm", "music", "separation"}: - raise ValueError("Modus muss 'llm', 'music' oder 'separation' sein") + if mode not in {"llm", "music", "separation", "voice"}: + raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein") with STATE.lock: if STATE.mode_phase not in {"ready", "error"}: return False, STATE.mode_phase @@ -496,6 +540,7 @@ def _control_command(data: dict, path: str) -> str | None: command = text.strip().casefold() return command if command in {"/athena music", "/athena stems", "/athena separation", "/athena llm", + "/athena voice", "/athena status"} else None @@ -1971,6 +2016,8 @@ class Handler(BaseHTTPRequestHandler): "music_health": _music_worker_health(), "separator_worker": _separator_worker_state(), "separator_health": _separator_worker_health(), + "voice_worker": _voice_worker_state(), + "voice_health": _voice_worker_health(), "return_profile": state.get("return_profile"), "last_error": STATE.mode_error, "enabled": ENABLE_MUSIC_MODE, @@ -1980,8 +2027,8 @@ class Handler(BaseHTTPRequestHandler): try: data = json.loads(self._read_body() or b"{}") mode = data.get("mode") if isinstance(data, dict) else None - if mode not in {"llm", "music", "separation"}: - raise ValueError("Feld 'mode' muss 'llm', 'music' oder 'separation' sein") + if mode not in {"llm", "music", "separation", "voice"}: + raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation' oder 'voice' sein") started, phase = schedule_operating_mode(mode) self._send_json(202 if started else 200, { "status": "accepted" if started else "ok", @@ -2644,10 +2691,12 @@ class Handler(BaseHTTPRequestHandler): text = (f"Athena läuft im {mode['active'].upper()}-Modus. " f"Phase: {mode['phase']}. Musik-Worker: " f"{mode['music_worker']}. Stem-Separator: " - f"{mode['separator_worker']}. LLM-Profil: {profile or 'entladen'}.") + f"{mode['separator_worker']}. Voice Studio: " + f"{mode['voice_worker']}. LLM-Profil: {profile or 'entladen'}.") else: target = ("music" if command == "/athena music" else "separation" if command in {"/athena stems", "/athena separation"} + else "voice" if command == "/athena voice" else "llm") try: started, phase = schedule_operating_mode(target) @@ -2656,6 +2705,8 @@ class Handler(BaseHTTPRequestHandler): if target == "music" else "Stimmtrennung wird gestartet. LLM und TTS werden entladen." if target == "separation" else + "Voice Studio wird gestartet. LLM und TTS werden entladen." + if target == "voice" else "Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.") else: text = (f"Athena ist bereits im {target.upper()}-Modus " @@ -2948,15 +2999,14 @@ def _startup_reconcile() -> None: log.info("Startup-Retention: %d alte Bilder entfernt", len(removed)) special_mode = previous.get("mode") - if ENABLE_MUSIC_MODE and special_mode in {"music", "separation"}: + if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice"}: STATE.mode = special_mode STATE.mode_phase = f"starting-{special_mode}" _set_qwen_unavailable(True) try: - path = ("/workers/music/start" if special_mode == "music" - else "/workers/separator/start") + path, _worker_state, wait_ready = _special_worker(special_mode) _profile_controller_request("POST", path) - _wait_music_ready() if special_mode == "music" else _wait_separator_ready() + wait_ready() STATE.mode_phase = "ready" RUNTIME.save(mode=special_mode, phase=special_mode) log.info("Recovery: Spezialmodus %s wiederhergestellt", special_mode)