Add optional 44.1 kHz X-VC enhancement

This commit is contained in:
Mikei386 committed 2026-09-09 17:21:20 +02:00
1 parent b4f4bf37fd
commit 51ed201f6c
9 files changed
+128 -14

No files matched your search

@@ -29,6 +29,19 @@ RUN python -m pip install --no-cache-dir --upgrade pip \
RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2"
# Resemble Enhance declares old training-time pins for PyTorch, Gradio and
# DeepSpeed. Install only its inference code plus the small modules imported by
# that path, then remove the two unnecessary training imports. X-VC keeps the
# CUDA 12.8 / PyTorch 2.8 runtime required by the RTX 5080.
RUN python -m pip install --no-cache-dir --no-deps "resemble-enhance==0.0.1" \
&& python -m pip install --no-cache-dir \
"numpy==1.26.4" "scipy==1.11.4" \
"librosa==0.10.1" "soundfile==0.12.1" \
"matplotlib>=3.8,<4" "pandas>=2.1,<3" "rich>=13,<15" "tabulate>=0.9,<1"
COPY patch_resemble_enhance.py /tmp/patch_resemble_enhance.py
RUN python /tmp/patch_resemble_enhance.py && rm /tmp/patch_resemble_enhance.py
COPY app.py /opt/xvc/local_webui.py
COPY inference_log.py /opt/xvc/utils/log.py
+7 -1
View File
@@ -9,7 +9,7 @@ ZeroGPU.
- Private URL: `http://192.168.1.212:8009`
- Source clip: speech content and timing to preserve
- Reference clip: target speaker identity
- Output: 16 kHz PCM WAV
- Output: native 16 kHz PCM WAV plus optional Resemble-Enhance restoration at 44.1 kHz
- GPU: RTX 5080 only
- Persistent cache: `/data/voice/xvc/huggingface`
- Code and model license: MIT
@@ -22,3 +22,9 @@ Technical acceptance on 9 September 2026 used the repository's source and
target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The
result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about
2.9 GiB on the RTX 5080. German listening quality remains open.
The optional high-quality path uses `resemble-enhance` 0.0.1 with model
revision `4e3510ce4a8391159f665903544c5150bee7b2cb`. It does not change X-VC's
native 16-kHz architecture. Instead, it reconstructs missing speech bandwidth
after conversion and writes a second 44.1-kHz WAV. The UI always retains the
native output for an honest A/B comparison.
+59 -9
View File
@@ -5,6 +5,7 @@ import os
import sys
import tempfile
import time
from pathlib import Path
from typing import Tuple
import gradio as gr
@@ -28,8 +29,11 @@ MODEL_REPO = "chenxie95/X-VC"
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
SAMPLE_RATE = 16000
ENHANCED_SAMPLE_RATE = 44100
LATENT_HOP_LENGTH = 1280
MAX_SECONDS = 20.0
RESEMBLE_REVISION = "4e3510ce4a8391159f665903544c5150bee7b2cb"
RESEMBLE_RUN_DIR = Path("/models/huggingface/resemble-enhance/enhancer_stage2")
MODE_OFFLINE = "Offline (höchste Qualität)"
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
@@ -81,24 +85,49 @@ def _tensor(wav: np.ndarray) -> torch.Tensor:
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
def _write_wav(audio: np.ndarray) -> str:
def _write_wav(audio: np.ndarray, sample_rate: int = SAMPLE_RATE, suffix: str = "native-16k") -> str:
os.makedirs("/output", exist_ok=True)
path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav")
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16")
path = os.path.join("/output", f"xvc-{suffix}-{int(time.time() * 1000)}.wav")
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), sample_rate, subtype="PCM_16")
return path
@torch.inference_mode()
def _enhance_wav(audio: np.ndarray) -> tuple[str, float]:
if not (RESEMBLE_RUN_DIR / "hparams.yaml").is_file():
raise gr.Error("Resemble-Enhance-Gewichte fehlen; bitte den Athena-Operator prüfen lassen.")
from resemble_enhance.enhancer.inference import enhance
started = time.time()
source = torch.from_numpy(np.asarray(audio, dtype=np.float32).reshape(-1))
restored, sample_rate = enhance(
source,
SAMPLE_RATE,
"cuda",
nfe=32,
solver="midpoint",
lambd=0.1,
tau=0.5,
run_dir=RESEMBLE_RUN_DIR,
)
if int(sample_rate) != ENHANCED_SAMPLE_RATE:
raise RuntimeError(f"Unerwartete Resemble-Enhance-Abtastrate: {sample_rate}")
return _write_wav(restored.cpu().numpy(), int(sample_rate), "enhanced-44k1"), time.time() - started
@torch.inference_mode()
def convert(
source_audio: str,
reference_audio: str,
enhance_44k1: bool = True,
mode: str = MODE_OFFLINE,
chunk_ms: int = 2400,
current_ms: int = 120,
future_ms: int = 100,
smooth_ms: int = 20,
progress=gr.Progress(track_tqdm=True),
) -> Tuple[str, str]:
) -> Tuple[str, str | None, str]:
if not source_audio:
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
if not reference_audio:
@@ -142,7 +171,18 @@ def convert(
f"(RTF {elapsed / seconds:.2f})"
)
return _write_wav(to_numpy_audio(recon)), report
recon_np = to_numpy_audio(recon)
native_path = _write_wav(recon_np)
enhanced_path = None
if enhance_44k1:
del source_wav, target_wav, recon
torch.cuda.empty_cache()
enhanced_path, enhancement_seconds = _enhance_wav(recon_np)
report += (
f" · Resemble Enhance **{enhancement_seconds:.2f} s**, "
f"Ausgabe **44,1 kHz** (rekonstruierte Bandbreite)"
)
return native_path, enhanced_path, report
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
@@ -162,7 +202,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
run = gr.Button("Stimme umwandeln", variant="primary")
output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False)
enhance_44k1 = gr.Checkbox(
value=True,
label="Zusätzlich mit Resemble Enhance auf 44,1 kHz restaurieren",
info="Rekonstruiert fehlende Höhen per KI; das native 16-kHz-Ergebnis bleibt zum Vergleich erhalten.",
)
with gr.Row():
output_native = gr.Audio(label="X-VC Original · 16 kHz", type="filepath", autoplay=False)
output_enhanced = gr.Audio(label="Enhanced · 44,1 kHz", type="filepath", autoplay=False)
report = gr.Markdown()
with gr.Accordion("Erweiterte Einstellungen", open=False):
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
@@ -172,11 +219,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as
with gr.Row():
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.")
gr.Markdown(
"Die ersten 20 Sekunden jeder Datei werden verarbeitet. X-VC bleibt nativ bei 16 kHz; "
"die optionale zweite Datei rekonstruiert die Sprachbandbreite auf 44,1 kHz."
)
run.click(
convert,
inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms],
outputs=[output, report],
inputs=[source, reference, enhance_44k1, mode, chunk_ms, current_ms, future_ms, smooth_ms],
outputs=[output_native, output_enhanced, report],
api_name="convert",
)
@@ -1,7 +1,7 @@
services:
xvc-studio:
build: .
image: mike-ai/xvc-studio:2026-09-09
image: mike-ai/xvc-studio:2026-09-09-enhance
container_name: mike-ai-xvc-studio
restart: "no"
labels:
@@ -0,0 +1,40 @@
"""Remove training-only imports from Resemble Enhance's inference path."""
from pathlib import Path
import resemble_enhance
root = Path(resemble_enhance.__file__).parent
replacements = {
root / "enhancer" / "inference.py": {
"from .train import Enhancer, HParams": (
"from .enhancer import Enhancer\nfrom .hparams import HParams"
),
},
root / "denoiser" / "inference.py": {
"from .train import Denoiser, HParams": (
"from .denoiser import Denoiser\nfrom .hparams import HParams"
),
},
root / "enhancer" / "enhancer.py": {
"from ..utils.distributed import global_leader_only\n"
"from ..utils.train_loop import TrainLoop": (
"def global_leader_only(fn):\n"
" return fn\n\n"
"class TrainLoop:\n"
" @classmethod\n"
" def get_running_loop(cls):\n"
" return None"
),
},
}
for path, edits in replacements.items():
text = path.read_text(encoding="utf-8")
for old, new in edits.items():
if old not in text:
raise RuntimeError(f"Expected Resemble Enhance source not found in {path}: {old!r}")
text = text.replace(old, new, 1)
path.write_text(text, encoding="utf-8")