Add optional 44.1 kHz X-VC enhancement

This commit is contained in:
Mikei386
2026-09-09 17:21:20 +02:00
parent b4f4bf37fd
commit 51ed201f6c
9 changed files with 128 additions and 14 deletions
+59 -9
View File
@@ -5,6 +5,7 @@ import os
import sys
import tempfile
import time
from pathlib import Path
from typing import Tuple
import gradio as gr
@@ -28,8 +29,11 @@ MODEL_REPO = "chenxie95/X-VC"
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
SAMPLE_RATE = 16000
ENHANCED_SAMPLE_RATE = 44100
LATENT_HOP_LENGTH = 1280
MAX_SECONDS = 20.0
RESEMBLE_REVISION = "4e3510ce4a8391159f665903544c5150bee7b2cb"
RESEMBLE_RUN_DIR = Path("/models/huggingface/resemble-enhance/enhancer_stage2")
MODE_OFFLINE = "Offline (höchste Qualität)"
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
@@ -81,24 +85,49 @@ def _tensor(wav: np.ndarray) -> torch.Tensor:
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
def _write_wav(audio: np.ndarray) -> str:
def _write_wav(audio: np.ndarray, sample_rate: int = SAMPLE_RATE, suffix: str = "native-16k") -> str:
os.makedirs("/output", exist_ok=True)
path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav")
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16")
path = os.path.join("/output", f"xvc-{suffix}-{int(time.time() * 1000)}.wav")
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), sample_rate, subtype="PCM_16")
return path
@torch.inference_mode()
def _enhance_wav(audio: np.ndarray) -> tuple[str, float]:
if not (RESEMBLE_RUN_DIR / "hparams.yaml").is_file():
raise gr.Error("Resemble-Enhance-Gewichte fehlen; bitte den Athena-Operator prüfen lassen.")
from resemble_enhance.enhancer.inference import enhance
started = time.time()
source = torch.from_numpy(np.asarray(audio, dtype=np.float32).reshape(-1))
restored, sample_rate = enhance(
source,
SAMPLE_RATE,
"cuda",
nfe=32,
solver="midpoint",
lambd=0.1,
tau=0.5,
run_dir=RESEMBLE_RUN_DIR,
)
if int(sample_rate) != ENHANCED_SAMPLE_RATE:
raise RuntimeError(f"Unerwartete Resemble-Enhance-Abtastrate: {sample_rate}")
return _write_wav(restored.cpu().numpy(), int(sample_rate), "enhanced-44k1"), time.time() - started
@torch.inference_mode()
def convert(
source_audio: str,
reference_audio: str,
enhance_44k1: bool = True,
mode: str = MODE_OFFLINE,
chunk_ms: int = 2400,
current_ms: int = 120,
future_ms: int = 100,
smooth_ms: int = 20,
progress=gr.Progress(track_tqdm=True),
) -> Tuple[str, str]:
) -> Tuple[str, str | None, str]:
if not source_audio:
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
if not reference_audio:
@@ -142,7 +171,18 @@ def convert(
f"(RTF {elapsed / seconds:.2f})"
)
return _write_wav(to_numpy_audio(recon)), report
recon_np = to_numpy_audio(recon)
native_path = _write_wav(recon_np)
enhanced_path = None
if enhance_44k1:
del source_wav, target_wav, recon
torch.cuda.empty_cache()
enhanced_path, enhancement_seconds = _enhance_wav(recon_np)
report += (
f" · Resemble Enhance **{enhancement_seconds:.2f} s**, "
f"Ausgabe **44,1 kHz** (rekonstruierte Bandbreite)"
)
return native_path, enhanced_path, report
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
@@ -162,7 +202,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
run = gr.Button("Stimme umwandeln", variant="primary")
output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False)
enhance_44k1 = gr.Checkbox(
value=True,
label="Zusätzlich mit Resemble Enhance auf 44,1 kHz restaurieren",
info="Rekonstruiert fehlende Höhen per KI; das native 16-kHz-Ergebnis bleibt zum Vergleich erhalten.",
)
with gr.Row():
output_native = gr.Audio(label="X-VC Original · 16 kHz", type="filepath", autoplay=False)
output_enhanced = gr.Audio(label="Enhanced · 44,1 kHz", type="filepath", autoplay=False)
report = gr.Markdown()
with gr.Accordion("Erweiterte Einstellungen", open=False):
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
@@ -172,11 +219,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as
with gr.Row():
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.")
gr.Markdown(
"Die ersten 20 Sekunden jeder Datei werden verarbeitet. X-VC bleibt nativ bei 16 kHz; "
"die optionale zweite Datei rekonstruiert die Sprachbandbreite auf 44,1 kHz."
)
run.click(
convert,
inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms],
outputs=[output, report],
inputs=[source, reference, enhance_44k1, mode, chunk_ms, current_ms, future_ms, smooth_ms],
outputs=[output_native, output_enhanced, report],
api_name="convert",
)