diff --git a/compose.yaml b/compose.yaml
index c22ce74..731722e 100644
--- a/compose.yaml
+++ b/compose.yaml
@@ -571,6 +571,7 @@ services:
TTS_WORKER: qwen3
MUSIC_WORKER: acestep
SEPARATOR_WORKER: bs-roformer
+ VOICE_WORKER: vevo2
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
@@ -860,6 +861,7 @@ services:
MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
+ VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
HOST_PROC: /host/proc
HOST_DATA: /host/data
HOST_MODELS: /host/models
diff --git a/dev/test_dashboard_modes.py b/dev/test_dashboard_modes.py
index ccadbfb..b6f546e 100644
--- a/dev/test_dashboard_modes.py
+++ b/dev/test_dashboard_modes.py
@@ -52,6 +52,15 @@ class DashboardModeTests(unittest.TestCase):
self.assertEqual(json.loads(request.data), {"mode": "separation"})
self.assertEqual(request.get_header("Authorization"), "Bearer test-key")
+ def test_voice_mode_is_forwarded_to_router(self):
+ with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
+ status, body = self.dashboard.change_mode("voice")
+
+ self.assertEqual(status, 202)
+ self.assertEqual(body, {"status": "accepted"})
+ request = urlopen.call_args.args[0]
+ self.assertEqual(json.loads(request.data), {"mode": "voice"})
+
def test_unknown_mode_is_rejected_without_router_request(self):
with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen:
status, body = self.dashboard.change_mode("unknown")
diff --git a/dev/test_profile_controller.py b/dev/test_profile_controller.py
index 95d0f06..c44a845 100644
--- a/dev/test_profile_controller.py
+++ b/dev/test_profile_controller.py
@@ -46,7 +46,43 @@ def separator_item(state="exited"):
"Labels": {controller.SEPARATOR_LABEL_KEY: "bs-roformer"}}
+def voice_item(state="exited"):
+ return {"Id": "id-voice", "State": state,
+ "Labels": {controller.VOICE_LABEL_KEY: "vevo2"}}
+
+
class ProfileControllerTests(unittest.TestCase):
+ def test_voice_start_exclusively_stops_gpu_workers(self):
+ profiles = {name: item(name) for name in controller.ALLOWED}
+ profiles["medium"] = item("medium", "running")
+ calls = []
+
+ def request(method, path):
+ calls.append((method, path))
+ return 204, b""
+
+ with patch.object(controller, "VOICE_WORKER", "vevo2"), \
+ patch.object(controller, "MUSIC_WORKER", "acestep"), \
+ patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
+ patch.object(controller, "containers", return_value=profiles), \
+ patch.object(controller, "voice_container", return_value=voice_item()), \
+ patch.object(controller, "music_container", return_value=music_item("running")), \
+ patch.object(controller, "separator_container", return_value=separator_item("running")), \
+ patch.object(controller, "image_containers", return_value=[image_item("running")]), \
+ patch.object(controller, "tts_container", return_value=tts_item()), \
+ patch.object(controller, "docker_request", side_effect=request):
+ result = controller.set_voice_worker(True)
+
+ self.assertEqual(result, {"voice_worker": "vevo2", "state": "running"})
+ self.assertEqual(calls, [
+ ("POST", "/containers/id-medium/stop?t=120"),
+ ("POST", "/containers/id-flux/stop?t=20"),
+ ("POST", "/containers/id-tts/stop?t=30"),
+ ("POST", "/containers/id-music/stop?t=30"),
+ ("POST", "/containers/id-separator/stop?t=30"),
+ ("POST", "/containers/id-voice/start"),
+ ])
+
def test_separator_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["large"] = item("large", "running")
diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md
index 0bc9fd6..8d9b1f3 100644
--- a/docs/OPERATING_MODES.md
+++ b/docs/OPERATING_MODES.md
@@ -1,11 +1,13 @@
# Athena-Betriebsmodi
-Athena besitzt drei gegenseitig exklusive Betriebsmodi:
+Athena besitzt vier gegenseitig exklusive Betriebsmodi:
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt.
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt.
- `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt
Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt.
+- `voice`: Vevo2 überträgt Sprache oder Gesang auf eine gespeicherte
+ Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt.
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
@@ -13,8 +15,8 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen
## Bedienung
-Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio** und
-**Audio trennen** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
+Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**
+und **Voice Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
- **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende
Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der
@@ -48,12 +50,22 @@ nicht eine reine Synthesizer-Spur. Sprache nutzt das 48-kHz-Modell
`audio-separator` 0.47.0. Die ältere API-Auswahl kompletter 2-/4-/6-Stem-Sätze
bleibt rückwärtskompatibel.
+Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter
+`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter
+`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem
+Speichern eine Bestätigung der Nutzungsberechtigung. Vevo2 gibt
+unkomprimiertes 24-kHz-WAV aus und wandelt sowohl Sprache als auch Gesang;
+dies ist noch kein Echtzeit-Live-Voice-Changer. Die Gewichte sind
+CC BY-NC-ND 4.0 lizenziert und werden auf Athena ausschließlich privat und
+nichtkommerziell eingesetzt.
+
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
```text
/athena music
/athena stems
+/athena voice
/athena llm
/athena status
```
@@ -64,13 +76,15 @@ Die HTTP-Schnittstelle verwendet authentifizierte Requests:
GET /mode
POST /mode {"mode":"music"}
POST /mode {"mode":"separation"}
+POST /mode {"mode":"voice"}
POST /mode {"mode":"llm"}
```
Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
`com.mike-ai.music-worker=acestep` beziehungsweise
-`com.mike-ai.stem-separator=bs-roformer` markierten Container; freie
+`com.mike-ai.stem-separator=bs-roformer` oder
+`com.mike-ai.voice-worker=vevo2` markierten Container; freie
Container- oder Docker-Befehle werden nicht entgegengenommen.
## Wiederanlauf
diff --git a/docs/TESTED_MODELS.md b/docs/TESTED_MODELS.md
index 1d1ef28..01a148e 100644
--- a/docs/TESTED_MODELS.md
+++ b/docs/TESTED_MODELS.md
@@ -1,6 +1,6 @@
# Register getesteter Modelle
-Stand: 8. September 2026
+Stand: 9. September 2026
Dieses Dokument ist die zentrale Sperrliste gegen doppelte Modelltests. Vor
jedem Download müssen Repository, Dateiname, Basismodell, Fine-Tune und
@@ -56,6 +56,7 @@ Titelgenerierung und Kontextkompression in Hermes.
| 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** |
| 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback |
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
+| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Produktions-HTTP-API mit offizieller 8,6-s-Sprachprobe: Warmstart 12,216 s, Wandlung 2,342 s, Spitzen-VRAM 5.605,6 MiB; gültiges 24-kHz-WAV mit 412.878 Byte; Profilanlage, Download und automatische Job-Bereinigung geprüft | **technischer Ende-zu-Ende-Test bestanden**, privater Voice-Studio-Betatest; Gewichte CC BY-NC-ND 4.0 und daher nicht kommerziell einsetzen |
## Musikgenerierung
diff --git a/experiments/vevo2-voice-cloning/Dockerfile b/experiments/vevo2-voice-cloning/Dockerfile
new file mode 100644
index 0000000..827316f
--- /dev/null
+++ b/experiments/vevo2-voice-cloning/Dockerfile
@@ -0,0 +1,64 @@
+FROM mike-ai/bs-roformer-separator:0.47.0
+
+ARG AMPHION_COMMIT=26f6883110181f1dbfe95c70a7c7dbaf4de5f42a
+
+USER root
+RUN apt-get update \
+ && apt-get install -y --no-install-recommends git espeak-ng \
+ && rm -rf /var/lib/apt/lists/*
+
+RUN git clone https://github.com/open-mmlab/Amphion.git /opt/amphion \
+ && cd /opt/amphion \
+ && git checkout "${AMPHION_COMMIT}" \
+ && rm -rf .git
+
+# Vevo2 is tested on Athena with the CUDA/PyTorch stack inherited from the
+# existing audio worker. Do not install Amphion's historical torch 2.0/cu118
+# pins: Blackwell requires the newer cu128 runtime already present here.
+RUN python -m pip install --no-cache-dir \
+ accelerate==1.10.1 \
+ diffusers==0.35.1 \
+ einops==0.8.1 \
+ easydict==1.13 \
+ g2p_en==2.1.0 \
+ humanfriendly==10.0 \
+ huggingface-hub==0.34.4 \
+ hydra-core==1.3.2 \
+ inflect==7.5.0 \
+ ipython==9.5.0 \
+ json5==0.12.1 \
+ librosa==0.11.0 \
+ loguru==0.7.3 \
+ matplotlib==3.10.6 \
+ munch==4.0.0 \
+ omegaconf==2.3.0 \
+ openai-whisper==20250625 \
+ phonemizer==3.3.0 \
+ python-multipart==0.0.20 \
+ praat-parselmouth==0.4.6 \
+ pypinyin==0.55.0 \
+ pyworld==0.3.5 \
+ ruamel.yaml==0.18.15 \
+ safetensors==0.6.2 \
+ tabulate==0.9.0 \
+ tgt==1.5 \
+ torchcrepe==0.0.24 \
+ transformers==4.56.1 \
+ typeguard==4.4.4 \
+ unidecode==1.4.0 \
+ vector-quantize-pytorch==1.12.5 \
+ vocos==0.1.0
+
+WORKDIR /opt/amphion
+ENV PYTHONPATH=/opt/amphion \
+ HF_HOME=/models/huggingface \
+ PYTHONUNBUFFERED=1
+
+COPY run_fm_test.py /usr/local/bin/run_fm_test.py
+COPY app.py /app/app.py
+COPY index.html /app/index.html
+
+EXPOSE 8008
+HEALTHCHECK --interval=5s --timeout=3s --start-period=90s --retries=3 \
+ CMD curl -fsS http://127.0.0.1:8008/health || exit 1
+ENTRYPOINT ["python", "/app/app.py"]
diff --git a/experiments/vevo2-voice-cloning/README.md b/experiments/vevo2-voice-cloning/README.md
new file mode 100644
index 0000000..2718379
--- /dev/null
+++ b/experiments/vevo2-voice-cloning/README.md
@@ -0,0 +1,30 @@
+# Vevo2 Voice Conversion on Athena
+
+This directory contains Athena's private Vevo2 voice-conversion studio.
+
+- Code: `open-mmlab/Amphion` commit
+ `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` (MIT)
+- Weights: `RMSnow/Vevo2` (CC BY-NC-ND 4.0)
+- Runtime: PyTorch 2.7.1 + CUDA 12.8 inherited from Athena's tested audio
+ separator image; the historical Amphion torch 2.0/cu118 pins are not used.
+- Storage: `/data/voice/vevo2`; removing that directory and the test image
+ removes all downloaded artifacts.
+- Private UI: `http://192.168.1.212:8008` through WireGuard only.
+- Output: uncompressed mono WAV, 24 kHz.
+
+The 9 September technical gate converted the official 8.6-second speech sample
+through the production HTTP API in 2.342 seconds. A warm service start loaded
+the model in 12.216 seconds, and peak CUDA allocation during conversion was
+5605.6 MiB. The returned 24-kHz WAV was 412878 bytes and 8.6 seconds long. The
+service stores named reference voices, accepts a source clip, and returns a
+transient WAV download. Jobs and generated outputs are removed after delivery.
+It is deliberately not exposed on the university interface.
+
+The installed Compose project is named `mike-ai-voice`. Rebuild or recreate it
+with an explicit `VOICE_GPU_UUID` so Docker keeps the service attached to the
+RTX 5080. The profile controller starts and stops the existing container; it
+does not rebuild it during a mode switch.
+
+The weights are licensed CC BY-NC-ND 4.0. This deployment is for Mike's
+private, non-commercial use only. Do not use or expose it as a public or
+commercial voice-cloning service.
diff --git a/experiments/vevo2-voice-cloning/app.py b/experiments/vevo2-voice-cloning/app.py
new file mode 100644
index 0000000..40288e8
--- /dev/null
+++ b/experiments/vevo2-voice-cloning/app.py
@@ -0,0 +1,202 @@
+#!/usr/bin/env python3
+"""Small local-only Vevo2 voice-conversion studio for Athena."""
+
+from __future__ import annotations
+
+import asyncio
+import os
+import re
+import shutil
+import subprocess
+import threading
+import time
+import uuid
+from contextlib import asynccontextmanager
+from pathlib import Path
+
+import torch
+from fastapi import FastAPI, File, Form, HTTPException, UploadFile
+from fastapi.responses import FileResponse, HTMLResponse
+from starlette.background import BackgroundTask
+
+import models.svc.vevo2.infer_vevo2_fm as vevo
+
+
+DATA_DIR = Path(os.environ.get("VOICE_DATA_DIR", "/data"))
+PROFILE_DIR = DATA_DIR / "profiles"
+JOB_DIR = DATA_DIR / "jobs"
+INDEX = Path("/app/index.html")
+MAX_UPLOAD_BYTES = int(os.environ.get("MAX_UPLOAD_BYTES", str(100 * 1024 * 1024)))
+MODEL_LOCK = threading.Lock()
+PIPELINE = None
+MODEL_LOAD_SECONDS: float | None = None
+
+
+def safe_name(value: str) -> str:
+ value = re.sub(r"[^A-Za-z0-9_. -]+", "-", value.strip())
+ value = re.sub(r"\s+", "-", value).strip("-.")
+ return value[:64] or "voice"
+
+
+def load_pipeline() -> None:
+ global PIPELINE, MODEL_LOAD_SECONDS
+ if PIPELINE is not None:
+ return
+ with MODEL_LOCK:
+ if PIPELINE is not None:
+ return
+ started = time.monotonic()
+ PIPELINE = vevo.load_inference_pipeline()
+ vevo.inference_pipeline = PIPELINE
+ MODEL_LOAD_SECONDS = time.monotonic() - started
+
+
+def to_wav(source: Path, target: Path) -> None:
+ completed = subprocess.run(
+ ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
+ "-ac", "1", "-ar", "24000", "-c:a", "pcm_s16le", str(target)],
+ capture_output=True,
+ text=True,
+ timeout=180,
+ check=False,
+ )
+ if completed.returncode:
+ raise ValueError(completed.stderr.strip() or "Audio konnte nicht gelesen werden")
+
+
+async def save_upload(upload: UploadFile, target: Path) -> None:
+ size = 0
+ with target.open("wb") as handle:
+ while chunk := await upload.read(1024 * 1024):
+ size += len(chunk)
+ if size > MAX_UPLOAD_BYTES:
+ raise HTTPException(413, "Audiodatei ist zu groß")
+ handle.write(chunk)
+
+
+def profile_path(name: str) -> Path:
+ target = PROFILE_DIR / f"{safe_name(name)}.wav"
+ if not target.is_file():
+ raise HTTPException(404, "Referenzstimme wurde nicht gefunden")
+ return target
+
+
+def run_conversion(source: Path, reference: Path, output: Path, pitch_shift: bool) -> tuple[float, float]:
+ """Run the GPU-bound conversion off the API event loop."""
+ load_pipeline()
+ with MODEL_LOCK:
+ torch.cuda.reset_peak_memory_stats()
+ started = time.monotonic()
+ vevo.vevo2_fm(str(source), str(reference), str(output), shifted_src=pitch_shift)
+ elapsed = time.monotonic() - started
+ peak = torch.cuda.max_memory_allocated() / 1048576
+ return elapsed, peak
+
+
+@asynccontextmanager
+async def lifespan(_app: FastAPI):
+ PROFILE_DIR.mkdir(parents=True, exist_ok=True)
+ JOB_DIR.mkdir(parents=True, exist_ok=True)
+ load_pipeline()
+ yield
+
+
+app = FastAPI(title="Athena Voice Studio", lifespan=lifespan)
+
+
+@app.get("/", response_class=HTMLResponse)
+def index() -> str:
+ return INDEX.read_text(encoding="utf-8")
+
+
+@app.get("/health")
+def health() -> dict:
+ return {
+ "status": "ok" if PIPELINE is not None else "starting",
+ "model": "RMSnow/Vevo2",
+ "sample_rate": 24000,
+ "model_load_seconds": MODEL_LOAD_SECONDS,
+ }
+
+
+@app.get("/api/profiles")
+def profiles() -> dict:
+ return {"profiles": [path.stem for path in sorted(PROFILE_DIR.glob("*.wav"))]}
+
+
+@app.post("/api/profiles")
+async def create_profile(
+ name: str = Form(...),
+ consent: bool = Form(False),
+ audio: UploadFile = File(...),
+) -> dict:
+ if not consent:
+ raise HTTPException(400, "Bestätige, dass du die Stimme verwenden darfst")
+ clean = safe_name(name)
+ job = JOB_DIR / f"profile-{uuid.uuid4().hex}"
+ job.mkdir(parents=True)
+ raw = job / f"upload-{safe_name(audio.filename or 'reference.audio')}"
+ try:
+ await save_upload(audio, raw)
+ target = PROFILE_DIR / f"{clean}.wav"
+ temporary = job / "reference.wav"
+ to_wav(raw, temporary)
+ os.replace(temporary, target)
+ return {"status": "ok", "profile": clean}
+ except HTTPException:
+ raise
+ except Exception as exc:
+ raise HTTPException(400, str(exc)) from exc
+ finally:
+ shutil.rmtree(job, ignore_errors=True)
+
+
+@app.delete("/api/profiles/{name}")
+def delete_profile(name: str) -> dict:
+ target = profile_path(name)
+ target.unlink()
+ return {"status": "ok", "profile": target.stem}
+
+
+@app.post("/api/convert")
+async def convert(
+ source: UploadFile = File(...),
+ profile: str = Form(...),
+ pitch_shift: bool = Form(True),
+) -> FileResponse:
+ reference = profile_path(profile)
+ job_id = uuid.uuid4().hex
+ job = JOB_DIR / f"convert-{job_id}"
+ job.mkdir(parents=True)
+ raw = job / f"upload-{safe_name(source.filename or 'source.audio')}"
+ source_wav = job / "source.wav"
+ output = JOB_DIR / f"voice-{job_id}.wav"
+ try:
+ await save_upload(source, raw)
+ to_wav(raw, source_wav)
+ elapsed, peak = await asyncio.to_thread(
+ run_conversion, source_wav, reference, output, pitch_shift
+ )
+ return FileResponse(
+ output,
+ media_type="audio/wav",
+ filename=f"{safe_name(profile)}-{job_id[:8]}.wav",
+ headers={
+ "X-Conversion-Seconds": f"{elapsed:.3f}",
+ "X-Peak-VRAM-MiB": f"{peak:.1f}",
+ },
+ background=BackgroundTask(output.unlink, missing_ok=True),
+ )
+ except HTTPException:
+ raise
+ except Exception as exc:
+ output.unlink(missing_ok=True)
+ raise HTTPException(500, f"Stimmenwandlung fehlgeschlagen: {exc}") from exc
+ finally:
+ shutil.rmtree(job, ignore_errors=True)
+
+
+if __name__ == "__main__":
+ import uvicorn
+
+ uvicorn.run(app, host="0.0.0.0", port=8008)
diff --git a/experiments/vevo2-voice-cloning/compose.yaml b/experiments/vevo2-voice-cloning/compose.yaml
new file mode 100644
index 0000000..9ad91f6
--- /dev/null
+++ b/experiments/vevo2-voice-cloning/compose.yaml
@@ -0,0 +1,39 @@
+services:
+ voice-studio:
+ build: .
+ image: mike-ai/vevo2-voice-studio:0.1
+ container_name: mike-ai-voice-studio
+ restart: "no"
+ labels:
+ com.mike-ai.voice-worker: vevo2
+ environment:
+ NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
+ NVIDIA_DRIVER_CAPABILITIES: compute,utility
+ VOICE_DATA_DIR: /data
+ ports:
+ - "127.0.0.1:8008:8008"
+ volumes:
+ - /data/voice/vevo2/huggingface/hub/Vevo2:/opt/amphion/ckpts/Vevo2
+ - /data/voice/vevo2/huggingface:/models/huggingface
+ - /data/voice/vevo2/whisper:/root/.cache/whisper
+ - /data/voice/studio:/data
+ healthcheck:
+ test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/health | grep -q '\"status\":\"ok\"'"]
+ interval: 5s
+ timeout: 3s
+ start_period: 600s
+ retries: 3
+ deploy:
+ resources:
+ reservations:
+ devices:
+ - driver: nvidia
+ device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
+ capabilities: [gpu]
+ networks:
+ - frontend
+
+networks:
+ frontend:
+ name: mike-ai_frontend
+ external: true
diff --git a/experiments/vevo2-voice-cloning/index.html b/experiments/vevo2-voice-cloning/index.html
new file mode 100644
index 0000000..af579be
--- /dev/null
+++ b/experiments/vevo2-voice-cloning/index.html
@@ -0,0 +1,10 @@
+
+
+Athena Voice Studio
+Mike AI · lokal auf Athena
Voice Studio
Eine Stimme als Referenz speichern und eine vorhandene Sprach- oder Gesangsaufnahme in diese Stimme übertragen. Ausgabe: unkomprimiertes WAV mit 24 kHz.
+
+Hinweis: Nur privat und mit Zustimmung verwenden. Das Vevo2-Gewicht ist nicht für kommerzielle Nutzung lizenziert.
diff --git a/experiments/vevo2-voice-cloning/run_fm_test.py b/experiments/vevo2-voice-cloning/run_fm_test.py
new file mode 100644
index 0000000..8e31d93
--- /dev/null
+++ b/experiments/vevo2-voice-cloning/run_fm_test.py
@@ -0,0 +1,47 @@
+#!/usr/bin/env python3
+"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
+
+from __future__ import annotations
+
+import argparse
+import os
+import time
+
+import torch
+
+import models.svc.vevo2.infer_vevo2_fm as vevo
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser()
+ parser.add_argument("--source", required=True)
+ parser.add_argument("--reference", required=True)
+ parser.add_argument("--output", required=True)
+ parser.add_argument("--no-pitch-shift", action="store_true")
+ args = parser.parse_args()
+
+ os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
+ started = time.monotonic()
+ vevo.inference_pipeline = vevo.load_inference_pipeline()
+ loaded = time.monotonic()
+ vevo.vevo2_fm(
+ args.source,
+ args.reference,
+ args.output,
+ shifted_src=not args.no_pitch_shift,
+ )
+ finished = time.monotonic()
+ print(
+ {
+ "model_load_seconds": round(loaded - started, 3),
+ "conversion_seconds": round(finished - loaded, 3),
+ "total_seconds": round(finished - started, 3),
+ "peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
+ "output": args.output,
+ },
+ flush=True,
+ )
+
+
+if __name__ == "__main__":
+ main()
diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py
index 341726a..c3b367d 100644
--- a/platform/docker/profile-controller/profile_controller.py
+++ b/platform/docker/profile-controller/profile_controller.py
@@ -32,6 +32,8 @@ MUSIC_LABEL_KEY = "com.mike-ai.music-worker"
MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip()
SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator"
SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
+VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
+VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
LOCK = threading.Lock()
log = logging.getLogger("profile-controller")
@@ -122,6 +124,17 @@ def separator_container() -> dict:
return matches[0]
+def voice_container() -> dict:
+ if not VOICE_WORKER:
+ raise RuntimeError("voice worker is not configured")
+ matches = [item for item in labelled_containers(VOICE_LABEL_KEY)
+ if item.get("Labels", {}).get(VOICE_LABEL_KEY) == VOICE_WORKER]
+ if len(matches) != 1:
+ raise RuntimeError(
+ f"expected exactly one voice worker {VOICE_WORKER!r}, found {len(matches)}")
+ return matches[0]
+
+
def stop_music_if_configured() -> None:
if MUSIC_WORKER:
stop_container(music_container(), timeout=30)
@@ -132,6 +145,11 @@ def stop_separator_if_configured() -> None:
stop_container(separator_container(), timeout=30)
+def stop_voice_if_configured() -> None:
+ if VOICE_WORKER:
+ stop_container(voice_container(), timeout=30)
+
+
def stop_container(item: dict, timeout: int = 120) -> None:
if item.get("State") != "running":
return
@@ -171,6 +189,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_separator_if_configured()
+ stop_voice_if_configured()
for other in image_containers():
if other["Id"] != item["Id"]:
stop_container(other, timeout=20)
@@ -197,6 +216,7 @@ def set_music_worker(running: bool) -> dict:
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_separator_if_configured()
+ stop_voice_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
@@ -215,6 +235,7 @@ def set_separator_worker(running: bool) -> dict:
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
+ stop_voice_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
@@ -222,6 +243,25 @@ def set_separator_worker(running: bool) -> dict:
"state": "running" if running else "stopped"}
+def set_voice_worker(running: bool) -> dict:
+ """Start Vevo2 exclusively, or stop it before LLM restoration."""
+ with LOCK:
+ item = voice_container()
+ if running:
+ for profile_item in containers().values():
+ stop_container(profile_item)
+ for worker in image_containers():
+ stop_container(worker, timeout=20)
+ stop_container(tts_container(), timeout=30)
+ stop_music_if_configured()
+ stop_separator_if_configured()
+ start_container(item)
+ else:
+ stop_container(item, timeout=30)
+ return {"voice_worker": VOICE_WORKER,
+ "state": "running" if running else "stopped"}
+
+
def active_profile(items: dict[str, dict] | None = None) -> str | None:
items = items or containers()
active = [name for name, item in items.items() if item.get("State") == "running"]
@@ -239,6 +279,7 @@ def activate(profile: str) -> dict:
stop_container(worker)
stop_music_if_configured()
stop_separator_if_configured()
+ stop_voice_if_configured()
start_container(tts_container())
items = containers()
missing = [name for name in ALLOWED if name not in items]
@@ -317,11 +358,20 @@ class Handler(BaseHTTPRequestHandler):
"unhealthy" if "(unhealthy)" in separator_status else
"starting" if separator.get("State") == "running" else
"stopped")
+ voice = voice_container() if VOICE_WORKER else {}
+ voice_status = voice.get("Status", "")
+ voice_health = ("disabled" if not VOICE_WORKER else
+ "healthy" if "(healthy)" in voice_status else
+ "unhealthy" if "(unhealthy)" in voice_status else
+ "starting" if voice.get("State") == "running" else
+ "stopped")
self.reply(200, {"active_profile": active_profile(items),
"music_worker": music.get("State", "disabled"),
"music_health": music_health,
"separator_worker": separator.get("State", "disabled"),
"separator_health": separator_health,
+ "voice_worker": voice.get("State", "disabled"),
+ "voice_health": voice_health,
"profiles": {name: items.get(name, {}).get(
"State", "missing") for name in ALLOWED}})
except Exception as exc:
@@ -353,6 +403,13 @@ class Handler(BaseHTTPRequestHandler):
log.exception("stem separator transition failed")
self.reply(503, {"error": str(exc)})
return
+ if self.path in {"/workers/voice/start", "/workers/voice/stop"}:
+ try:
+ self.reply(200, set_voice_worker(self.path.endswith("/start")))
+ except Exception as exc:
+ log.exception("voice worker transition failed")
+ self.reply(503, {"error": str(exc)})
+ return
worker_paths = {
"/workers/image/start": (IMAGE_WORKER, True),
"/workers/image/stop": (IMAGE_WORKER, False),
diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh
index f0b4b3d..ef2633c 100644
--- a/platform/docker/wireguard-gateway/entrypoint.sh
+++ b/platform/docker/wireguard-gateway/entrypoint.sh
@@ -81,6 +81,7 @@ start_proxy 8099 llama-dashboard:8099
start_proxy 7861 music-ui:3000
start_proxy 7862 music-worker:7860
start_proxy 8007 stem-separator:8080
+start_proxy 8008 voice-studio:8008
start_proxy 8202 mcp-athena-operator:8000
start_proxy 9443 portainer:9443
diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py
index 2581b4c..6f87968 100644
--- a/platform/llama-dashboard/app.py
+++ b/platform/llama-dashboard/app.py
@@ -28,6 +28,7 @@ MUSIC_ORIGINAL_UI_URL = os.getenv(
"MUSIC_ORIGINAL_UI_URL", "http://192.168.1.212:7862/"
)
SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
+VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
@@ -277,7 +278,7 @@ def router_status() -> tuple[dict[str, Any], str | None]:
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
- if mode not in {"llm", "music", "separation"}:
+ if mode not in {"llm", "music", "separation", "voice"}:
return 400, {"error": "invalid mode"}
headers = {"Accept": "application/json", "Content-Type": "application/json"}
if ROUTER_API_KEY:
@@ -635,7 +636,7 @@ HTML = r'''
Mike AI · Live Telemetry
Athena llama.cpp Dashboard
verbinde …
- Athena Betriebsmodus
–
Status wird geladen …
+ Athena Betriebsmodus
–
Status wird geladen …
Aktives Profil
–
Router wird abgefragt
Modell
–
–
CPU
–
–
@@ -674,12 +675,13 @@ const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed
function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `${pct(g.gpu_percent)}GPU-Kern
${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiBVRAM
${g.temperature_c??'–'} °CTemperatur
${g.power_w??'–'} / ${g.power_limit_w??'–'} WLeistung
${g.graphics_clock_mhz??'–'} MHzGrafiktakt
${g.memory_clock_mhz??'–'} MHzSpeichertakt
${pct(g.memory_controller_percent)}Memory Controller
${pct(g.fan_percent)}Lüfter
GPU-Auslastung${load.toFixed(1)} %
VRAM-Belegung${p.toFixed(1)} %
${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei
`}
const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit');
let modeBusy=false;
-async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
-async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'Keine GPU-Daten verfügbar';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`| ${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'} | ${p.name} | ${p.pid} | ${p.memory_mib??'–'} MiB |
`).join('')||'| Keine Compute-Prozesse gemeldet |
';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`${v??'–'}${k}
`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
+async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
+async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'Keine GPU-Daten verfügbar';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`| ${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'} | ${p.name} | ${p.pid} | ${p.memory_mib??'–'} MiB |
`).join('')||'| Keine Compute-Prozesse gemeldet |
';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`${v??'–'}${k}
`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);