Add optional 44.1 kHz X-VC enhancement
This commit is contained in:
@@ -5,6 +5,7 @@ import os
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from pathlib import Path
|
||||
from typing import Tuple
|
||||
|
||||
import gradio as gr
|
||||
@@ -28,8 +29,11 @@ MODEL_REPO = "chenxie95/X-VC"
|
||||
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
|
||||
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
|
||||
SAMPLE_RATE = 16000
|
||||
ENHANCED_SAMPLE_RATE = 44100
|
||||
LATENT_HOP_LENGTH = 1280
|
||||
MAX_SECONDS = 20.0
|
||||
RESEMBLE_REVISION = "4e3510ce4a8391159f665903544c5150bee7b2cb"
|
||||
RESEMBLE_RUN_DIR = Path("/models/huggingface/resemble-enhance/enhancer_stage2")
|
||||
MODE_OFFLINE = "Offline (höchste Qualität)"
|
||||
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
|
||||
|
||||
@@ -81,24 +85,49 @@ def _tensor(wav: np.ndarray) -> torch.Tensor:
|
||||
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
|
||||
|
||||
|
||||
def _write_wav(audio: np.ndarray) -> str:
|
||||
def _write_wav(audio: np.ndarray, sample_rate: int = SAMPLE_RATE, suffix: str = "native-16k") -> str:
|
||||
os.makedirs("/output", exist_ok=True)
|
||||
path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav")
|
||||
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16")
|
||||
path = os.path.join("/output", f"xvc-{suffix}-{int(time.time() * 1000)}.wav")
|
||||
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), sample_rate, subtype="PCM_16")
|
||||
return path
|
||||
|
||||
|
||||
@torch.inference_mode()
|
||||
def _enhance_wav(audio: np.ndarray) -> tuple[str, float]:
|
||||
if not (RESEMBLE_RUN_DIR / "hparams.yaml").is_file():
|
||||
raise gr.Error("Resemble-Enhance-Gewichte fehlen; bitte den Athena-Operator prüfen lassen.")
|
||||
|
||||
from resemble_enhance.enhancer.inference import enhance
|
||||
|
||||
started = time.time()
|
||||
source = torch.from_numpy(np.asarray(audio, dtype=np.float32).reshape(-1))
|
||||
restored, sample_rate = enhance(
|
||||
source,
|
||||
SAMPLE_RATE,
|
||||
"cuda",
|
||||
nfe=32,
|
||||
solver="midpoint",
|
||||
lambd=0.1,
|
||||
tau=0.5,
|
||||
run_dir=RESEMBLE_RUN_DIR,
|
||||
)
|
||||
if int(sample_rate) != ENHANCED_SAMPLE_RATE:
|
||||
raise RuntimeError(f"Unerwartete Resemble-Enhance-Abtastrate: {sample_rate}")
|
||||
return _write_wav(restored.cpu().numpy(), int(sample_rate), "enhanced-44k1"), time.time() - started
|
||||
|
||||
|
||||
@torch.inference_mode()
|
||||
def convert(
|
||||
source_audio: str,
|
||||
reference_audio: str,
|
||||
enhance_44k1: bool = True,
|
||||
mode: str = MODE_OFFLINE,
|
||||
chunk_ms: int = 2400,
|
||||
current_ms: int = 120,
|
||||
future_ms: int = 100,
|
||||
smooth_ms: int = 20,
|
||||
progress=gr.Progress(track_tqdm=True),
|
||||
) -> Tuple[str, str]:
|
||||
) -> Tuple[str, str | None, str]:
|
||||
if not source_audio:
|
||||
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
|
||||
if not reference_audio:
|
||||
@@ -142,7 +171,18 @@ def convert(
|
||||
f"(RTF {elapsed / seconds:.2f})"
|
||||
)
|
||||
|
||||
return _write_wav(to_numpy_audio(recon)), report
|
||||
recon_np = to_numpy_audio(recon)
|
||||
native_path = _write_wav(recon_np)
|
||||
enhanced_path = None
|
||||
if enhance_44k1:
|
||||
del source_wav, target_wav, recon
|
||||
torch.cuda.empty_cache()
|
||||
enhanced_path, enhancement_seconds = _enhance_wav(recon_np)
|
||||
report += (
|
||||
f" · Resemble Enhance **{enhancement_seconds:.2f} s**, "
|
||||
f"Ausgabe **44,1 kHz** (rekonstruierte Bandbreite)"
|
||||
)
|
||||
return native_path, enhanced_path, report
|
||||
|
||||
|
||||
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
|
||||
@@ -162,7 +202,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as
|
||||
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
|
||||
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
|
||||
run = gr.Button("Stimme umwandeln", variant="primary")
|
||||
output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False)
|
||||
enhance_44k1 = gr.Checkbox(
|
||||
value=True,
|
||||
label="Zusätzlich mit Resemble Enhance auf 44,1 kHz restaurieren",
|
||||
info="Rekonstruiert fehlende Höhen per KI; das native 16-kHz-Ergebnis bleibt zum Vergleich erhalten.",
|
||||
)
|
||||
with gr.Row():
|
||||
output_native = gr.Audio(label="X-VC Original · 16 kHz", type="filepath", autoplay=False)
|
||||
output_enhanced = gr.Audio(label="Enhanced · 44,1 kHz", type="filepath", autoplay=False)
|
||||
report = gr.Markdown()
|
||||
with gr.Accordion("Erweiterte Einstellungen", open=False):
|
||||
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
|
||||
@@ -172,11 +219,14 @@ with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as
|
||||
with gr.Row():
|
||||
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
|
||||
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
|
||||
gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.")
|
||||
gr.Markdown(
|
||||
"Die ersten 20 Sekunden jeder Datei werden verarbeitet. X-VC bleibt nativ bei 16 kHz; "
|
||||
"die optionale zweite Datei rekonstruiert die Sprachbandbreite auf 44,1 kHz."
|
||||
)
|
||||
run.click(
|
||||
convert,
|
||||
inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms],
|
||||
outputs=[output, report],
|
||||
inputs=[source, reference, enhance_44k1, mode, chunk_ms, current_ms, future_ms, smooth_ms],
|
||||
outputs=[output_native, output_enhanced, report],
|
||||
api_name="convert",
|
||||
)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user