From 87a2ae570451662a3b260b02c9160d2c8872c8cf Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Wed, 9 Sep 2026 16:14:10 +0200 Subject: [PATCH] Add X-VC voice conversion mode --- README.md | 9 +- compose.yaml | 3 + dev/test_dashboard_modes.py | 9 + dev/test_profile_controller.py | 39 ++++ docs/OPERATING_MODES.md | 31 ++- docs/TESTED_MODELS.md | 1 + experiments/xvc-voice-conversion/Dockerfile | 43 ++++ experiments/xvc-voice-conversion/README.md | 24 +++ experiments/xvc-voice-conversion/app.py | 188 ++++++++++++++++++ experiments/xvc-voice-conversion/compose.yaml | 39 ++++ .../xvc-voice-conversion/inference_log.py | 29 +++ .../profile-controller/profile_controller.py | 61 +++++- .../docker/wireguard-gateway/entrypoint.sh | 1 + platform/llama-dashboard/app.py | 26 ++- router/ai_profile_router.py | 69 ++++++- 15 files changed, 544 insertions(+), 28 deletions(-) create mode 100644 experiments/xvc-voice-conversion/Dockerfile create mode 100644 experiments/xvc-voice-conversion/README.md create mode 100644 experiments/xvc-voice-conversion/app.py create mode 100644 experiments/xvc-voice-conversion/compose.yaml create mode 100644 experiments/xvc-voice-conversion/inference_log.py diff --git a/README.md b/README.md index 480189e..22b3a1b 100644 --- a/README.md +++ b/README.md @@ -15,8 +15,8 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.** - Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback - Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung - Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 -- Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio und - BS-RoFormer-Stimmtrennung +- Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio, + BS-RoFormer-Stimmtrennung, OmniVoice und X-VC - Portainer CE als optionale Container-Ansicht auf Port 9443 - WireGuard-Gateway, Datenbackup und Athena-Operator - keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz @@ -125,9 +125,12 @@ Details, Installation, Prüfung und Rollback stehen in - Musikstudio, Original UI (stabil): `http://192.168.1.212:7862` - Musikstudio, Community UI (experimentell): `http://192.168.1.212:7861` - Spuren trennen (BS-RoFormer + Demucs, 2/4/6 Stems): `http://192.168.1.212:8007` +- Voice Studio (OmniVoice, Text zu Stimme): `http://192.168.1.212:8008` +- Voice Changer (X-VC, Audio zu Audio): `http://192.168.1.212:8009` Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren -außerdem `/athena music`, `/athena stems`, `/athena llm` und `/athena status`; Details stehen in +außerdem `/athena music`, `/athena stems`, `/athena voice`, +`/athena voicechange`, `/athena llm` und `/athena status`; Details stehen in [docs/OPERATING_MODES.md](docs/OPERATING_MODES.md). Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über diff --git a/compose.yaml b/compose.yaml index 731722e..9899dff 100644 --- a/compose.yaml +++ b/compose.yaml @@ -572,6 +572,7 @@ services: MUSIC_WORKER: acestep SEPARATOR_WORKER: bs-roformer VOICE_WORKER: vevo2 + VOICE_CHANGE_WORKER: xvc networks: [control] security_opt: ["no-new-privileges:true"] healthcheck: @@ -630,6 +631,7 @@ services: ENABLE_STT: "true" ENABLE_MUSIC_MODE: "true" MUSIC_START_TIMEOUT: "600" + VOICE_CHANGE_START_TIMEOUT: "600" STT_WORKER_URL: http://whisper:8084 STT_TIMEOUT: "300" networks: [frontend, control, inference] @@ -862,6 +864,7 @@ services: MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}" SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}" VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}" + VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}" HOST_PROC: /host/proc HOST_DATA: /host/data HOST_MODELS: /host/models diff --git a/dev/test_dashboard_modes.py b/dev/test_dashboard_modes.py index b6f546e..168b1bf 100644 --- a/dev/test_dashboard_modes.py +++ b/dev/test_dashboard_modes.py @@ -61,6 +61,15 @@ class DashboardModeTests(unittest.TestCase): request = urlopen.call_args.args[0] self.assertEqual(json.loads(request.data), {"mode": "voice"}) + def test_voice_change_mode_is_forwarded_to_router(self): + with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen: + status, body = self.dashboard.change_mode("voicechange") + + self.assertEqual(status, 202) + self.assertEqual(body, {"status": "accepted"}) + request = urlopen.call_args.args[0] + self.assertEqual(json.loads(request.data), {"mode": "voicechange"}) + def test_unknown_mode_is_rejected_without_router_request(self): with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen: status, body = self.dashboard.change_mode("unknown") diff --git a/dev/test_profile_controller.py b/dev/test_profile_controller.py index c44a845..b9c6d00 100644 --- a/dev/test_profile_controller.py +++ b/dev/test_profile_controller.py @@ -51,7 +51,46 @@ def voice_item(state="exited"): "Labels": {controller.VOICE_LABEL_KEY: "vevo2"}} +def voice_change_item(state="exited"): + return {"Id": "id-xvc", "State": state, + "Labels": {controller.VOICE_CHANGE_LABEL_KEY: "xvc"}} + + class ProfileControllerTests(unittest.TestCase): + def test_voice_change_start_exclusively_stops_gpu_workers(self): + profiles = {name: item(name) for name in controller.ALLOWED} + profiles["medium"] = item("medium", "running") + calls = [] + + def request(method, path): + calls.append((method, path)) + return 204, b"" + + with patch.object(controller, "VOICE_CHANGE_WORKER", "xvc"), \ + patch.object(controller, "VOICE_WORKER", "vevo2"), \ + patch.object(controller, "MUSIC_WORKER", "acestep"), \ + patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \ + patch.object(controller, "containers", return_value=profiles), \ + patch.object(controller, "voice_change_container", return_value=voice_change_item()), \ + patch.object(controller, "voice_container", return_value=voice_item("running")), \ + patch.object(controller, "music_container", return_value=music_item("running")), \ + patch.object(controller, "separator_container", return_value=separator_item("running")), \ + patch.object(controller, "image_containers", return_value=[image_item("running")]), \ + patch.object(controller, "tts_container", return_value=tts_item()), \ + patch.object(controller, "docker_request", side_effect=request): + result = controller.set_voice_change_worker(True) + + self.assertEqual(result, {"voice_change_worker": "xvc", "state": "running"}) + self.assertEqual(calls, [ + ("POST", "/containers/id-medium/stop?t=120"), + ("POST", "/containers/id-flux/stop?t=20"), + ("POST", "/containers/id-tts/stop?t=30"), + ("POST", "/containers/id-music/stop?t=30"), + ("POST", "/containers/id-separator/stop?t=30"), + ("POST", "/containers/id-voice/stop?t=30"), + ("POST", "/containers/id-xvc/start"), + ]) + def test_voice_start_exclusively_stops_gpu_workers(self): profiles = {name: item(name) for name in controller.ALLOWED} profiles["medium"] = item("medium", "running") diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md index 8d9b1f3..8a49da8 100644 --- a/docs/OPERATING_MODES.md +++ b/docs/OPERATING_MODES.md @@ -1,13 +1,16 @@ # Athena-Betriebsmodi -Athena besitzt vier gegenseitig exklusive Betriebsmodi: +Athena besitzt sechs gegenseitig exklusive Betriebsmodi: - `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt. - `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt. - `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt. -- `voice`: Vevo2 überträgt Sprache oder Gesang auf eine gespeicherte +- `voice`: OmniVoice erzeugt Sprache aus Text mit einer gewählten Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt. +- `voicechange`: X-VC überträgt eine vorhandene Sprachaufnahme auf eine + Referenzstimme und bewahrt dabei Inhalt und Timing. Alle anderen + GPU-Dienste sind gestoppt. Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus @@ -15,8 +18,8 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen ## Bedienung -Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen** -und **Voice Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten: +Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**, +**Voice Studio** und **Voice Changer** bereit. Im Musikmodus werden zwei Oberflächen angeboten: - **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der @@ -53,11 +56,16 @@ bleibt rückwärtskompatibel. Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter `http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter `/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem -Speichern eine Bestätigung der Nutzungsberechtigung. Vevo2 gibt -unkomprimiertes 24-kHz-WAV aus und wandelt sowohl Sprache als auch Gesang; -dies ist noch kein Echtzeit-Live-Voice-Changer. Die Gewichte sind -CC BY-NC-ND 4.0 lizenziert und werden auf Athena ausschließlich privat und -nichtkommerziell eingesetzt. +Speichern eine Bestätigung der Nutzungsberechtigung. OmniVoice gibt +unkomprimiertes WAV aus und erzeugt Sprache aus Text; es verarbeitet keine +bereits eingesprochene Quellaufnahme. + +Der X-VC Voice Changer ist ausschließlich unter +`http://192.168.1.212:8009` erreichbar. Er nimmt eine Quellaufnahme und eine +Referenzstimme an und gibt 16-kHz-PCM-WAV aus. Die dokumentierte Sprachbasis +des verwendeten GLM-4-Voice-Tokenizers ist Chinesisch und Englisch; Deutsch +bleibt deshalb bis zur Hörabnahme ein Qualitätstest und kein zugesagter +Produktionspfad. X-VC läuft ausschließlich auf der RTX 5080. Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom Router lokal beantwortet, auch wenn gerade kein LLM geladen ist: @@ -66,6 +74,7 @@ Router lokal beantwortet, auch wenn gerade kein LLM geladen ist: /athena music /athena stems /athena voice +/athena voicechange /athena llm /athena status ``` @@ -77,6 +86,7 @@ GET /mode POST /mode {"mode":"music"} POST /mode {"mode":"separation"} POST /mode {"mode":"voice"} +POST /mode {"mode":"voicechange"} POST /mode {"mode":"llm"} ``` @@ -84,7 +94,8 @@ Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in `GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit `com.mike-ai.music-worker=acestep` beziehungsweise `com.mike-ai.stem-separator=bs-roformer` oder -`com.mike-ai.voice-worker=vevo2` markierten Container; freie +`com.mike-ai.voice-worker=vevo2` beziehungsweise +`com.mike-ai.voice-change-worker=xvc` markierten Container; freie Container- oder Docker-Befehle werden nicht entgegengenommen. ## Wiederanlauf diff --git a/docs/TESTED_MODELS.md b/docs/TESTED_MODELS.md index 2d56ebd..98242ae 100644 --- a/docs/TESTED_MODELS.md +++ b/docs/TESTED_MODELS.md @@ -58,6 +58,7 @@ Titelgenerierung und Kontextkompression in Hermes. | 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** | | 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen**; Container ersetzt, Image und Daten vorerst nur als Rollback erhalten | | 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen | +| 09.09.2026 | `chenxie95/X-VC`, Code `49df8c591eafc48b096e466d96f9839f9c0dd739`, UI-Basis `d761cd6421e85376b2656dfefd8471d7f35a42be` | Offizielles Beispiel Ende-zu-Ende gewandelt: 5,20 s Audio in 1,07 s (RTF 0,21), gültiges 16-kHz-Mono-PCM-WAV; Modell belegt rund 2,9 GiB auf der RTX 5080. GLM-4-Voice-Tokenizer dokumentiert Chinesisch und Englisch | **technischer Start- und Konvertierungstest bestanden**; deutsche Hörabnahme offen | ## Musikgenerierung diff --git a/experiments/xvc-voice-conversion/Dockerfile b/experiments/xvc-voice-conversion/Dockerfile new file mode 100644 index 0000000..f11df4a --- /dev/null +++ b/experiments/xvc-voice-conversion/Dockerfile @@ -0,0 +1,43 @@ +FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04 + +ARG XVC_COMMIT=49df8c591eafc48b096e466d96f9839f9c0dd739 + +RUN apt-get update \ + && apt-get install -y --no-install-recommends \ + ca-certificates curl ffmpeg git libsndfile1 python3 python3-pip python3-venv \ + && rm -rf /var/lib/apt/lists/* + +RUN git clone https://github.com/Jerrister/X-VC.git /opt/xvc \ + && cd /opt/xvc \ + && git checkout "${XVC_COMMIT}" \ + && rm -rf .git + +RUN python3 -m venv /opt/venv +ENV PATH="/opt/venv/bin:${PATH}" + +# RTX 5080: use a CUDA 12.8 PyTorch build instead of X-VC's older training pin. +RUN python -m pip install --no-cache-dir --upgrade pip \ + && python -m pip install --no-cache-dir \ + --index-url https://download.pytorch.org/whl/cu128 \ + "torch==2.8.0" "torchaudio==2.8.0" \ + && python -m pip install --no-cache-dir \ + "gradio>=5.49,<6" "huggingface_hub>=0.34,<2" \ + "transformers>=4.44,<5" "hydra-core>=1.3,<2" "omegaconf>=2.3,<3" \ + "x-transformers>=1.40,<3" "einops>=0.8,<1" "einx>=0.3,<1" \ + "numpy>=1.26,<3" "scipy>=1.13,<2" "soundfile>=0.12,<1" \ + "soxr>=0.5,<1" "tqdm>=4.66,<5" "wandb>=0.18,<1" + +RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2" + +COPY app.py /opt/xvc/local_webui.py +COPY inference_log.py /opt/xvc/utils/log.py + +ENV HF_HOME=/models/huggingface \ + PYTHONUNBUFFERED=1 + +EXPOSE 8009 + +HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \ + CMD curl -fsS http://127.0.0.1:8009/ >/dev/null || exit 1 + +CMD ["python", "/opt/xvc/local_webui.py"] diff --git a/experiments/xvc-voice-conversion/README.md b/experiments/xvc-voice-conversion/README.md new file mode 100644 index 0000000..d22028e --- /dev/null +++ b/experiments/xvc-voice-conversion/README.md @@ -0,0 +1,24 @@ +# X-VC voice conversion on Athena + +Isolated quality gate for `chenxie95/X-VC`, using the official inference path +from commit `49df8c591eafc48b096e466d96f9839f9c0dd739`. The UI is adapted from the +public Hugging Face Space at commit +`d761cd6421e85376b2656dfefd8471d7f35a42be` and runs locally without +ZeroGPU. + +- Private URL: `http://192.168.1.212:8009` +- Source clip: speech content and timing to preserve +- Reference clip: target speaker identity +- Output: 16 kHz PCM WAV +- GPU: RTX 5080 only +- Persistent cache: `/data/voice/xvc/huggingface` +- Code and model license: MIT + +The semantic tokenizer documents Chinese and English. German is therefore a +quality gate, not an assumed supported language. Keep OmniVoice installed: it +does text-to-speech cloning, while X-VC tests true audio-to-audio conversion. + +Technical acceptance on 9 September 2026 used the repository's source and +target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The +result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about +2.9 GiB on the RTX 5080. German listening quality remains open. diff --git a/experiments/xvc-voice-conversion/app.py b/experiments/xvc-voice-conversion/app.py new file mode 100644 index 0000000..8c8dd8a --- /dev/null +++ b/experiments/xvc-voice-conversion/app.py @@ -0,0 +1,188 @@ +"""Local Athena adaptation of the public X-VC Gradio demo.""" + +import logging +import os +import sys +import tempfile +import time +from typing import Tuple + +import gradio as gr +import numpy as np +import soundfile as sf +import torch +from huggingface_hub import hf_hub_download +from omegaconf import OmegaConf + +HERE = "/opt/xvc" +sys.path.insert(0, HERE) + +from bins.infer_utils import precompute_conditions, run_offline, run_streaming, to_numpy_audio +from models.codec.sac.model import XVC +from utils.audio import audio_highpass_filter, audio_volume_normalize, load_audio + +logging.basicConfig(level=logging.INFO) +log = logging.getLogger("xvc-local") + +MODEL_REPO = "chenxie95/X-VC" +SPACE_REPO = "hugging-apps/x-vc-voice-conversion" +SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k" +SAMPLE_RATE = 16000 +LATENT_HOP_LENGTH = 1280 +MAX_SECONDS = 20.0 +MODE_OFFLINE = "Offline (höchste Qualität)" +MODE_STREAMING = "Streaming (simulierte Echtzeit)" + + +def _load_model() -> XVC: + speaker_config = hf_hub_download( + repo_id=SPACE_REPO, + repo_type="space", + filename=f"{SPEAKER_SUBDIR}/configuration.json", + ) + hf_hub_download( + repo_id=SPACE_REPO, + repo_type="space", + filename=f"{SPEAKER_SUBDIR}/pretrained_eres2net.ckpt", + ) + checkpoint = hf_hub_download(repo_id=MODEL_REPO, filename="xvc.pt") + + cfg = OmegaConf.load(os.path.join(HERE, "configs", "xvc.yaml")) + cfg["model"]["generator"].pop("loss_config", None) + cfg["model"].pop("discriminator", None) + cfg["model"]["generator"]["speaker_encoder"]["pretrained_dir"] = os.path.dirname(speaker_config) + infer_cfg = os.path.join(tempfile.gettempdir(), "xvc_inference.yaml") + OmegaConf.save(cfg, infer_cfg) + + loaded = XVC.load_from_checkpoint(infer_cfg, checkpoint, device=torch.device("cuda")) + loaded.remove_weight_norm() + loaded = loaded.eval().to("cuda") + log.info("X-VC model ready on %s", torch.cuda.get_device_name(0)) + return loaded + + +MODEL = _load_model() + + +def _prepare_wav(path: str) -> np.ndarray: + wav = load_audio(path, sampling_rate=SAMPLE_RATE, volume_normalize=False) + if wav is None or len(wav) == 0: + raise gr.Error("Die Audiodatei konnte nicht gelesen werden.") + wav = wav[: int(MAX_SECONDS * SAMPLE_RATE)] + wav = audio_volume_normalize(wav) + wav = audio_highpass_filter(wav, SAMPLE_RATE, 40) + remainder = len(wav) % LATENT_HOP_LENGTH + if remainder: + wav = np.pad(wav, (0, LATENT_HOP_LENGTH - remainder), mode="constant") + return wav.astype(np.float32) + + +def _tensor(wav: np.ndarray) -> torch.Tensor: + return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda") + + +def _write_wav(audio: np.ndarray) -> str: + os.makedirs("/output", exist_ok=True) + path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav") + sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16") + return path + + +@torch.inference_mode() +def convert( + source_audio: str, + reference_audio: str, + mode: str = MODE_OFFLINE, + chunk_ms: int = 2400, + current_ms: int = 120, + future_ms: int = 100, + smooth_ms: int = 20, + progress=gr.Progress(track_tqdm=True), +) -> Tuple[str, str]: + if not source_audio: + raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.") + if not reference_audio: + raise gr.Error("Bitte eine Referenzdatei mit der Zielstimme hochladen.") + + source_np = _prepare_wav(source_audio) + reference_np = _prepare_wav(reference_audio) + source_wav = _tensor(source_np) + target_wav = _tensor(reference_np) + seconds = len(source_np) / SAMPLE_RATE + started = time.time() + + if mode == MODE_STREAMING: + history_ms = int(chunk_ms) - int(current_ms) - int(smooth_ms) - int(future_ms) + if history_ms < 0: + raise gr.Error("Fenster muss mindestens Current + Lookahead + Crossfade umfassen.") + speaker_condition, frame_condition = precompute_conditions(MODEL, target_wav, target_wav) + recon, latency_ms = run_streaming( + model=MODEL, + source_wav=source_wav, + speaker_condition=speaker_condition, + frame_condition=frame_condition, + sample_rate=SAMPLE_RATE, + chunk_ms=int(chunk_ms), + current_ms=int(current_ms), + future_ms=int(future_ms), + smooth_ms=int(smooth_ms), + ) + elapsed = time.time() - started + latency = np.asarray(latency_ms, dtype=np.float64) + report = ( + f"**Streaming** · {len(latency)} Chunks · Mittel **{latency.mean():.0f} ms**, " + f"P95 **{np.percentile(latency, 95):.0f} ms** · insgesamt {elapsed:.2f} s " + f"für {seconds:.2f} s Audio (RTF {elapsed / seconds:.2f})" + ) + else: + recon = run_offline(MODEL, source_wav, target_wav, target_wav) + elapsed = time.time() - started + report = ( + f"**Offline** · {seconds:.2f} s Audio in {elapsed:.2f} s " + f"(RTF {elapsed / seconds:.2f})" + ) + + return _write_wav(to_numpy_audio(recon)), report + + +CSS = "#col-container { max-width: 1100px; margin: 0 auto; }" +HEADER = """# X-VC — Voice Changer + +Die **Quelle** liefert Text, Aussprache und Timing. Die **Referenz** liefert die +Zielstimme. X-VC arbeitet Audio-zu-Audio ohne Transkript oder Training. + +[Paper](https://arxiv.org/abs/2604.12456) · [Modell](https://huggingface.co/chenxie95/X-VC) · +[Code](https://github.com/Jerrister/X-VC) +""" + +with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as demo: + with gr.Column(elem_id="col-container"): + gr.Markdown(HEADER) + with gr.Row(): + source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"]) + reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"]) + run = gr.Button("Stimme umwandeln", variant="primary") + output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False) + report = gr.Markdown() + with gr.Accordion("Erweiterte Einstellungen", open=False): + mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus") + with gr.Row(): + current_ms = gr.Slider(40, 640, value=120, step=40, label="Aktueller Chunk (ms)") + chunk_ms = gr.Slider(800, 4800, value=2400, step=200, label="Gesamtfenster (ms)") + with gr.Row(): + future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)") + smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)") + gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.") + run.click( + convert, + inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms], + outputs=[output, report], + api_name="convert", + ) + +if __name__ == "__main__": + demo.queue(default_concurrency_limit=1).launch( + server_name="0.0.0.0", + server_port=8009, + show_error=True, + ) diff --git a/experiments/xvc-voice-conversion/compose.yaml b/experiments/xvc-voice-conversion/compose.yaml new file mode 100644 index 0000000..23d15f3 --- /dev/null +++ b/experiments/xvc-voice-conversion/compose.yaml @@ -0,0 +1,39 @@ +services: + xvc-studio: + build: . + image: mike-ai/xvc-studio:2026-09-09 + container_name: mike-ai-xvc-studio + restart: "no" + labels: + com.mike-ai.voice-change-worker: xvc + environment: + NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID} + NVIDIA_DRIVER_CAPABILITIES: compute,utility + HF_HOME: /models/huggingface + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + ports: + - "127.0.0.1:8009:8009" + volumes: + - /data/voice/xvc/huggingface:/models/huggingface + - /data/voice/xvc/output:/output + healthcheck: + test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8009/ >/dev/null"] + interval: 5s + timeout: 3s + start_period: 600s + retries: 3 + deploy: + resources: + reservations: + devices: + - driver: nvidia + device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"] + capabilities: [gpu] + networks: + frontend: + aliases: [xvc-studio] + +networks: + frontend: + name: mike-ai_frontend + external: true diff --git a/experiments/xvc-voice-conversion/inference_log.py b/experiments/xvc-voice-conversion/inference_log.py new file mode 100644 index 0000000..af4a2d9 --- /dev/null +++ b/experiments/xvc-voice-conversion/inference_log.py @@ -0,0 +1,29 @@ +"""Small inference-only replacement for X-VC's training logger. + +The upstream logger imports WandB, Matplotlib and TensorBoard at module import +time although model inference only uses the normal logging functions. +""" + +import logging + + +logging.basicConfig(level=logging.INFO) +_logger = logging.getLogger("xvc") + +debug = _logger.debug +info = _logger.info +warn = _logger.warning +warning = _logger.warning +error = _logger.error + + +def init(*_args, **_kwargs): + return None + + +def write_audio(*_args, **_kwargs): + return None + + +def write_loss(*_args, **_kwargs): + return None diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index c3b367d..fd6e644 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -34,6 +34,8 @@ SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator" SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip() VOICE_LABEL_KEY = "com.mike-ai.voice-worker" VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip() +VOICE_CHANGE_LABEL_KEY = "com.mike-ai.voice-change-worker" +VOICE_CHANGE_WORKER = os.environ.get("VOICE_CHANGE_WORKER", "").strip() LOCK = threading.Lock() log = logging.getLogger("profile-controller") @@ -135,6 +137,17 @@ def voice_container() -> dict: return matches[0] +def voice_change_container() -> dict: + if not VOICE_CHANGE_WORKER: + raise RuntimeError("voice-change worker is not configured") + matches = [item for item in labelled_containers(VOICE_CHANGE_LABEL_KEY) + if item.get("Labels", {}).get(VOICE_CHANGE_LABEL_KEY) == VOICE_CHANGE_WORKER] + if len(matches) != 1: + raise RuntimeError( + f"expected exactly one voice-change worker {VOICE_CHANGE_WORKER!r}, found {len(matches)}") + return matches[0] + + def stop_music_if_configured() -> None: if MUSIC_WORKER: stop_container(music_container(), timeout=30) @@ -150,6 +163,11 @@ def stop_voice_if_configured() -> None: stop_container(voice_container(), timeout=30) +def stop_voice_change_if_configured() -> None: + if VOICE_CHANGE_WORKER: + stop_container(voice_change_container(), timeout=30) + + def stop_container(item: dict, timeout: int = 120) -> None: if item.get("State") != "running": return @@ -190,6 +208,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: stop_music_if_configured() stop_separator_if_configured() stop_voice_if_configured() + stop_voice_change_if_configured() for other in image_containers(): if other["Id"] != item["Id"]: stop_container(other, timeout=20) @@ -217,6 +236,7 @@ def set_music_worker(running: bool) -> dict: stop_container(tts_container(), timeout=30) stop_separator_if_configured() stop_voice_if_configured() + stop_voice_change_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -236,6 +256,7 @@ def set_separator_worker(running: bool) -> dict: stop_container(tts_container(), timeout=30) stop_music_if_configured() stop_voice_if_configured() + stop_voice_change_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -244,7 +265,7 @@ def set_separator_worker(running: bool) -> dict: def set_voice_worker(running: bool) -> dict: - """Start Vevo2 exclusively, or stop it before LLM restoration.""" + """Start OmniVoice exclusively, or stop it before LLM restoration.""" with LOCK: item = voice_container() if running: @@ -255,6 +276,7 @@ def set_voice_worker(running: bool) -> dict: stop_container(tts_container(), timeout=30) stop_music_if_configured() stop_separator_if_configured() + stop_voice_change_if_configured() start_container(item) else: stop_container(item, timeout=30) @@ -262,6 +284,26 @@ def set_voice_worker(running: bool) -> dict: "state": "running" if running else "stopped"} +def set_voice_change_worker(running: bool) -> dict: + """Start X-VC exclusively, or stop it before another mode is loaded.""" + with LOCK: + item = voice_change_container() + if running: + for profile_item in containers().values(): + stop_container(profile_item) + for worker in image_containers(): + stop_container(worker, timeout=20) + stop_container(tts_container(), timeout=30) + stop_music_if_configured() + stop_separator_if_configured() + stop_voice_if_configured() + start_container(item) + else: + stop_container(item, timeout=30) + return {"voice_change_worker": VOICE_CHANGE_WORKER, + "state": "running" if running else "stopped"} + + def active_profile(items: dict[str, dict] | None = None) -> str | None: items = items or containers() active = [name for name, item in items.items() if item.get("State") == "running"] @@ -280,6 +322,7 @@ def activate(profile: str) -> dict: stop_music_if_configured() stop_separator_if_configured() stop_voice_if_configured() + stop_voice_change_if_configured() start_container(tts_container()) items = containers() missing = [name for name in ALLOWED if name not in items] @@ -365,6 +408,13 @@ class Handler(BaseHTTPRequestHandler): "unhealthy" if "(unhealthy)" in voice_status else "starting" if voice.get("State") == "running" else "stopped") + voice_change = voice_change_container() if VOICE_CHANGE_WORKER else {} + voice_change_status = voice_change.get("Status", "") + voice_change_health = ("disabled" if not VOICE_CHANGE_WORKER else + "healthy" if "(healthy)" in voice_change_status else + "unhealthy" if "(unhealthy)" in voice_change_status else + "starting" if voice_change.get("State") == "running" else + "stopped") self.reply(200, {"active_profile": active_profile(items), "music_worker": music.get("State", "disabled"), "music_health": music_health, @@ -372,6 +422,8 @@ class Handler(BaseHTTPRequestHandler): "separator_health": separator_health, "voice_worker": voice.get("State", "disabled"), "voice_health": voice_health, + "voice_change_worker": voice_change.get("State", "disabled"), + "voice_change_health": voice_change_health, "profiles": {name: items.get(name, {}).get( "State", "missing") for name in ALLOWED}}) except Exception as exc: @@ -410,6 +462,13 @@ class Handler(BaseHTTPRequestHandler): log.exception("voice worker transition failed") self.reply(503, {"error": str(exc)}) return + if self.path in {"/workers/voice-change/start", "/workers/voice-change/stop"}: + try: + self.reply(200, set_voice_change_worker(self.path.endswith("/start"))) + except Exception as exc: + log.exception("voice-change worker transition failed") + self.reply(503, {"error": str(exc)}) + return worker_paths = { "/workers/image/start": (IMAGE_WORKER, True), "/workers/image/stop": (IMAGE_WORKER, False), diff --git a/platform/docker/wireguard-gateway/entrypoint.sh b/platform/docker/wireguard-gateway/entrypoint.sh index ef2633c..9c6e5b6 100644 --- a/platform/docker/wireguard-gateway/entrypoint.sh +++ b/platform/docker/wireguard-gateway/entrypoint.sh @@ -82,6 +82,7 @@ start_proxy 7861 music-ui:3000 start_proxy 7862 music-worker:7860 start_proxy 8007 stem-separator:8080 start_proxy 8008 voice-studio:8008 +start_proxy 8009 xvc-studio:8009 start_proxy 8202 mcp-athena-operator:8000 start_proxy 9443 portainer:9443 diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py index 6f87968..0a7606b 100644 --- a/platform/llama-dashboard/app.py +++ b/platform/llama-dashboard/app.py @@ -29,6 +29,7 @@ MUSIC_ORIGINAL_UI_URL = os.getenv( ) SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/") VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/") +VOICE_CHANGE_UI_URL = os.getenv("VOICE_CHANGE_UI_URL", "http://192.168.1.212:8009/") HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc")) HOST_DATA = os.getenv("HOST_DATA", "/host/data") HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models")) @@ -278,7 +279,7 @@ def router_status() -> tuple[dict[str, Any], str | None]: def change_mode(mode: str) -> tuple[int, dict[str, Any]]: - if mode not in {"llm", "music", "separation", "voice"}: + if mode not in {"llm", "music", "separation", "voice", "voicechange"}: return 400, {"error": "invalid mode"} headers = {"Accept": "application/json", "Content-Type": "application/json"} if ROUTER_API_KEY: @@ -636,7 +637,7 @@ HTML = r'''