From 51ed201f6c57f670014e8a24851462736a33a022 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Wed, 9 Sep 2026 17:21:20 +0200 Subject: [PATCH] Add optional 44.1 kHz X-VC enhancement --- README.md | 2 +- docs/CONTAINER_INVENTORY.md | 2 +- docs/OPERATING_MODES.md | 6 +- docs/TESTED_MODELS.md | 1 + experiments/xvc-voice-conversion/Dockerfile | 13 ++++ experiments/xvc-voice-conversion/README.md | 8 ++- experiments/xvc-voice-conversion/app.py | 68 ++++++++++++++++--- experiments/xvc-voice-conversion/compose.yaml | 2 +- .../patch_resemble_enhance.py | 40 +++++++++++ 9 files changed, 128 insertions(+), 14 deletions(-) create mode 100644 experiments/xvc-voice-conversion/patch_resemble_enhance.py diff --git a/README.md b/README.md index 811eb05..5f4a074 100644 --- a/README.md +++ b/README.md @@ -126,7 +126,7 @@ Details, Installation, Prüfung und Rollback stehen in - Musikstudio, Community UI (experimentell): `http://192.168.1.212:7861` - Spuren trennen (BS-RoFormer + Demucs, 2/4/6 Stems): `http://192.168.1.212:8007` - Voice Studio (OmniVoice, Text zu Stimme): `http://192.168.1.212:8008` -- Voice Changer (X-VC, Audio zu Audio): `http://192.168.1.212:8009` +- Voice Changer (X-VC, Audio zu Audio; native 16 kHz plus optional restaurierte 44,1 kHz): `http://192.168.1.212:8009` Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren außerdem `/athena music`, `/athena stems`, `/athena voice`, diff --git a/docs/CONTAINER_INVENTORY.md b/docs/CONTAINER_INVENTORY.md index dde2137..49ef421 100644 --- a/docs/CONTAINER_INVENTORY.md +++ b/docs/CONTAINER_INVENTORY.md @@ -32,7 +32,7 @@ nicht automatisch ein ungenutzter Rest. | `mike-ai-voice-studio` | `k2-fsa/OmniVoice` 0.2.1 mit Whisper-ASR | Erzeugt Text-to-Speech mit einer Referenzstimme; kein Audio-to-Audio-Voice-Changer. | | `mike-ai-whisper` | Whisper.cpp `ggml-small` | Lokale deutsche Spracherkennung auf der CPU über `/v1/audio/transcriptions`. | | `mike-ai-wireguard-gateway` | kein Modell | Veröffentlicht Dashboard und Fachoberflächen ausschließlich über den privaten WireGuard-Pfad. | -| `mike-ai-xvc-studio` | `chenxie95/X-VC` und GLM-4-Voice-Tokenizer | Wandelt eine vorhandene Sprachaufnahme anhand einer Referenzstimme in Audio zu Audio um. | +| `mike-ai-xvc-studio` | `chenxie95/X-VC`, GLM-4-Voice-Tokenizer und optional Resemble Enhance | Wandelt eine vorhandene Sprachaufnahme anhand einer Referenzstimme in Audio zu Audio um; gibt das native 16-kHz-Ergebnis und optional eine neural restaurierte 44,1-kHz-Fassung aus. | ## Aufräumregel diff --git a/docs/OPERATING_MODES.md b/docs/OPERATING_MODES.md index 523bf1a..a4c5541 100644 --- a/docs/OPERATING_MODES.md +++ b/docs/OPERATING_MODES.md @@ -62,7 +62,11 @@ bereits eingesprochene Quellaufnahme. Der X-VC Voice Changer ist ausschließlich unter `http://192.168.1.212:8009` erreichbar. Er nimmt eine Quellaufnahme und eine -Referenzstimme an und gibt 16-kHz-PCM-WAV aus. Die dokumentierte Sprachbasis +Referenzstimme an. Die Oberfläche behält immer das native 16-kHz-PCM-WAV und +erzeugt auf Wunsch zusätzlich mit Resemble Enhance eine neural restaurierte +44,1-kHz-Fassung. Diese zweite Datei rekonstruiert fehlende Sprachbandbreite; +sie stellt keine im 16-kHz-Signal tatsächlich erhaltenen Originaldetails wieder +her und bleibt deshalb direkt mit dem nativen Ergebnis vergleichbar. Die dokumentierte Sprachbasis des verwendeten GLM-4-Voice-Tokenizers ist Chinesisch und Englisch; Deutsch bleibt deshalb bis zur Hörabnahme ein Qualitätstest und kein zugesagter Produktionspfad. X-VC läuft ausschließlich auf der RTX 5080. diff --git a/docs/TESTED_MODELS.md b/docs/TESTED_MODELS.md index 64df19e..b7d6a28 100644 --- a/docs/TESTED_MODELS.md +++ b/docs/TESTED_MODELS.md @@ -60,6 +60,7 @@ Titelgenerierung und Kontextkompression in Hermes. | 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen und entfernt**; Ergebnis bleibt hier dokumentiert, Images, Daten und altes Projekt wurden am 09.09. bereinigt | | 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen | | 09.09.2026 | `chenxie95/X-VC`, Code `49df8c591eafc48b096e466d96f9839f9c0dd739`, UI-Basis `d761cd6421e85376b2656dfefd8471d7f35a42be` | Offizielles Beispiel Ende-zu-Ende gewandelt: 5,20 s Audio in 1,07 s (RTF 0,21), gültiges 16-kHz-Mono-PCM-WAV; Modell belegt rund 2,9 GiB auf der RTX 5080. GLM-4-Voice-Tokenizer dokumentiert Chinesisch und Englisch | **technischer Start- und Konvertierungstest bestanden**; deutsche Hörabnahme offen | +| 09.09.2026 | Resemble Enhance 0.0.1, Modellrevision `4e3510ce4a8391159f665903544c5150bee7b2cb` | 14,56 s native X-VC-Ausgabe bei 16 kHz wurden auf der RTX 5080 in 3,55 s zu 44,1-kHz-PCM-WAV restauriert. 3,27 % der gemessenen Signalenergie lagen danach oberhalb 8 kHz; damit ist der Pfad keine bloße Neuabtastung. Wegen der alten Upstream-Pins läuft die reine Inferenz mit NumPy 1.26.4/SciPy 1.11.4 auf dem bestehenden Torch-2.8/CUDA-12.8-Unterbau | **technisch produktiv als optionaler A/B-Pfad**; Hörabnahme entscheidet, ob die rekonstruierten Höhen subjektiv besser oder künstlicher klingen | ## Musikgenerierung diff --git a/experiments/xvc-voice-conversion/Dockerfile b/experiments/xvc-voice-conversion/Dockerfile index f11df4a..455102f 100644 --- a/experiments/xvc-voice-conversion/Dockerfile +++ b/experiments/xvc-voice-conversion/Dockerfile @@ -29,6 +29,19 @@ RUN python -m pip install --no-cache-dir --upgrade pip \ RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2" +# Resemble Enhance declares old training-time pins for PyTorch, Gradio and +# DeepSpeed. Install only its inference code plus the small modules imported by +# that path, then remove the two unnecessary training imports. X-VC keeps the +# CUDA 12.8 / PyTorch 2.8 runtime required by the RTX 5080. +RUN python -m pip install --no-cache-dir --no-deps "resemble-enhance==0.0.1" \ + && python -m pip install --no-cache-dir \ + "numpy==1.26.4" "scipy==1.11.4" \ + "librosa==0.10.1" "soundfile==0.12.1" \ + "matplotlib>=3.8,<4" "pandas>=2.1,<3" "rich>=13,<15" "tabulate>=0.9,<1" + +COPY patch_resemble_enhance.py /tmp/patch_resemble_enhance.py +RUN python /tmp/patch_resemble_enhance.py && rm /tmp/patch_resemble_enhance.py + COPY app.py /opt/xvc/local_webui.py COPY inference_log.py /opt/xvc/utils/log.py diff --git a/experiments/xvc-voice-conversion/README.md b/experiments/xvc-voice-conversion/README.md index d22028e..227f3a0 100644 --- a/experiments/xvc-voice-conversion/README.md +++ b/experiments/xvc-voice-conversion/README.md @@ -9,7 +9,7 @@ ZeroGPU. - Private URL: `http://192.168.1.212:8009` - Source clip: speech content and timing to preserve - Reference clip: target speaker identity -- Output: 16 kHz PCM WAV +- Output: native 16 kHz PCM WAV plus optional Resemble-Enhance restoration at 44.1 kHz - GPU: RTX 5080 only - Persistent cache: `/data/voice/xvc/huggingface` - Code and model license: MIT @@ -22,3 +22,9 @@ Technical acceptance on 9 September 2026 used the repository's source and target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about 2.9 GiB on the RTX 5080. German listening quality remains open. + +The optional high-quality path uses `resemble-enhance` 0.0.1 with model +revision `4e3510ce4a8391159f665903544c5150bee7b2cb`. It does not change X-VC's +native 16-kHz architecture. Instead, it reconstructs missing speech bandwidth +after conversion and writes a second 44.1-kHz WAV. The UI always retains the +native output for an honest A/B comparison. diff --git a/experiments/xvc-voice-conversion/app.py b/experiments/xvc-voice-conversion/app.py index 8c8dd8a..0b7c832 100644 --- a/experiments/xvc-voice-conversion/app.py +++ b/experiments/xvc-voice-conversion/app.py @@ -5,6 +5,7 @@ import os import sys import tempfile import time +from pathlib import Path from typing import Tuple import gradio as gr @@ -28,8 +29,11 @@ MODEL_REPO = "chenxie95/X-VC" SPACE_REPO = "hugging-apps/x-vc-voice-conversion" SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k" SAMPLE_RATE = 16000 +ENHANCED_SAMPLE_RATE = 44100 LATENT_HOP_LENGTH = 1280 MAX_SECONDS = 20.0 +RESEMBLE_REVISION = "4e3510ce4a8391159f665903544c5150bee7b2cb" +RESEMBLE_RUN_DIR = Path("/models/huggingface/resemble-enhance/enhancer_stage2") MODE_OFFLINE = "Offline (höchste Qualität)" MODE_STREAMING = "Streaming (simulierte Echtzeit)" @@ -81,24 +85,49 @@ def _tensor(wav: np.ndarray) -> torch.Tensor: return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda") -def _write_wav(audio: np.ndarray) -> str: +def _write_wav(audio: np.ndarray, sample_rate: int = SAMPLE_RATE, suffix: str = "native-16k") -> str: os.makedirs("/output", exist_ok=True) - path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav") - sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16") + path = os.path.join("/output", f"xvc-{suffix}-{int(time.time() * 1000)}.wav") + sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), sample_rate, subtype="PCM_16") return path +@torch.inference_mode() +def _enhance_wav(audio: np.ndarray) -> tuple[str, float]: + if not (RESEMBLE_RUN_DIR / "hparams.yaml").is_file(): + raise gr.Error("Resemble-Enhance-Gewichte fehlen; bitte den Athena-Operator prüfen lassen.") + + from resemble_enhance.enhancer.inference import enhance + + started = time.time() + source = torch.from_numpy(np.asarray(audio, dtype=np.float32).reshape(-1)) + restored, sample_rate = enhance( + source, + SAMPLE_RATE, + "cuda", + nfe=32, + solver="midpoint", + lambd=0.1, + tau=0.5, + run_dir=RESEMBLE_RUN_DIR, + ) + if int(sample_rate) != ENHANCED_SAMPLE_RATE: + raise RuntimeError(f"Unerwartete Resemble-Enhance-Abtastrate: {sample_rate}") + return _write_wav(restored.cpu().numpy(), int(sample_rate), "enhanced-44k1"), time.time() - started + + @torch.inference_mode() def convert( source_audio: str, reference_audio: str, + enhance_44k1: bool = True, mode: str = MODE_OFFLINE, chunk_ms: int = 2400, current_ms: int = 120, future_ms: int = 100, smooth_ms: int = 20, progress=gr.Progress(track_tqdm=True), -) -> Tuple[str, str]: +) -> Tuple[str, str | None, str]: if not source_audio: raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.") if not reference_audio: @@ -142,7 +171,18 @@ def convert( f"(RTF {elapsed / seconds:.2f})" ) - return _write_wav(to_numpy_audio(recon)), report + recon_np = to_numpy_audio(recon) + native_path = _write_wav(recon_np) + enhanced_path = None + if enhance_44k1: + del source_wav, target_wav, recon + torch.cuda.empty_cache() + enhanced_path, enhancement_seconds = _enhance_wav(recon_np) + report += ( + f" · Resemble Enhance **{enhancement_seconds:.2f} s**, " + f"Ausgabe **44,1 kHz** (rekonstruierte Bandbreite)" + ) + return native_path, enhanced_path, report CSS = "#col-container { max-width: 1100px; margin: 0 auto; }" @@ -162,7 +202,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"]) reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"]) run = gr.Button("Stimme umwandeln", variant="primary") - output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False) + enhance_44k1 = gr.Checkbox( + value=True, + label="Zusätzlich mit Resemble Enhance auf 44,1 kHz restaurieren", + info="Rekonstruiert fehlende Höhen per KI; das native 16-kHz-Ergebnis bleibt zum Vergleich erhalten.", + ) + with gr.Row(): + output_native = gr.Audio(label="X-VC Original · 16 kHz", type="filepath", autoplay=False) + output_enhanced = gr.Audio(label="Enhanced · 44,1 kHz", type="filepath", autoplay=False) report = gr.Markdown() with gr.Accordion("Erweiterte Einstellungen", open=False): mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus") @@ -172,11 +219,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as with gr.Row(): future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)") smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)") - gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.") + gr.Markdown( + "Die ersten 20 Sekunden jeder Datei werden verarbeitet. X-VC bleibt nativ bei 16 kHz; " + "die optionale zweite Datei rekonstruiert die Sprachbandbreite auf 44,1 kHz." + ) run.click( convert, - inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms], - outputs=[output, report], + inputs=[source, reference, enhance_44k1, mode, chunk_ms, current_ms, future_ms, smooth_ms], + outputs=[output_native, output_enhanced, report], api_name="convert", ) diff --git a/experiments/xvc-voice-conversion/compose.yaml b/experiments/xvc-voice-conversion/compose.yaml index 23d15f3..802494c 100644 --- a/experiments/xvc-voice-conversion/compose.yaml +++ b/experiments/xvc-voice-conversion/compose.yaml @@ -1,7 +1,7 @@ services: xvc-studio: build: . - image: mike-ai/xvc-studio:2026-09-09 + image: mike-ai/xvc-studio:2026-09-09-enhance container_name: mike-ai-xvc-studio restart: "no" labels: diff --git a/experiments/xvc-voice-conversion/patch_resemble_enhance.py b/experiments/xvc-voice-conversion/patch_resemble_enhance.py new file mode 100644 index 0000000..7701d4c --- /dev/null +++ b/experiments/xvc-voice-conversion/patch_resemble_enhance.py @@ -0,0 +1,40 @@ +"""Remove training-only imports from Resemble Enhance's inference path.""" + +from pathlib import Path + +import resemble_enhance + + +root = Path(resemble_enhance.__file__).parent + +replacements = { + root / "enhancer" / "inference.py": { + "from .train import Enhancer, HParams": ( + "from .enhancer import Enhancer\nfrom .hparams import HParams" + ), + }, + root / "denoiser" / "inference.py": { + "from .train import Denoiser, HParams": ( + "from .denoiser import Denoiser\nfrom .hparams import HParams" + ), + }, + root / "enhancer" / "enhancer.py": { + "from ..utils.distributed import global_leader_only\n" + "from ..utils.train_loop import TrainLoop": ( + "def global_leader_only(fn):\n" + " return fn\n\n" + "class TrainLoop:\n" + " @classmethod\n" + " def get_running_loop(cls):\n" + " return None" + ), + }, +} + +for path, edits in replacements.items(): + text = path.read_text(encoding="utf-8") + for old, new in edits.items(): + if old not in text: + raise RuntimeError(f"Expected Resemble Enhance source not found in {path}: {old!r}") + text = text.replace(old, new, 1) + path.write_text(text, encoding="utf-8")