Add X-VC voice conversion mode

This commit is contained in:
Mikei386 committed 2026-09-09 16:14:10 +02:00
1 parent 17f1a08d7d
commit 87a2ae5704
15 files changed
+544 -28

No files matched your search

@@ -0,0 +1,43 @@
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
ARG XVC_COMMIT=49df8c591eafc48b096e466d96f9839f9c0dd739
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
ca-certificates curl ffmpeg git libsndfile1 python3 python3-pip python3-venv \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/Jerrister/X-VC.git /opt/xvc \
&& cd /opt/xvc \
&& git checkout "${XVC_COMMIT}" \
&& rm -rf .git
RUN python3 -m venv /opt/venv
ENV PATH="/opt/venv/bin:${PATH}"
# RTX 5080: use a CUDA 12.8 PyTorch build instead of X-VC's older training pin.
RUN python -m pip install --no-cache-dir --upgrade pip \
&& python -m pip install --no-cache-dir \
--index-url https://download.pytorch.org/whl/cu128 \
"torch==2.8.0" "torchaudio==2.8.0" \
&& python -m pip install --no-cache-dir \
"gradio>=5.49,<6" "huggingface_hub>=0.34,<2" \
"transformers>=4.44,<5" "hydra-core>=1.3,<2" "omegaconf>=2.3,<3" \
"x-transformers>=1.40,<3" "einops>=0.8,<1" "einx>=0.3,<1" \
"numpy>=1.26,<3" "scipy>=1.13,<2" "soundfile>=0.12,<1" \
"soxr>=0.5,<1" "tqdm>=4.66,<5" "wandb>=0.18,<1"
RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2"
COPY app.py /opt/xvc/local_webui.py
COPY inference_log.py /opt/xvc/utils/log.py
ENV HF_HOME=/models/huggingface \
PYTHONUNBUFFERED=1
EXPOSE 8009
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
CMD curl -fsS http://127.0.0.1:8009/ >/dev/null || exit 1
CMD ["python", "/opt/xvc/local_webui.py"]
@@ -0,0 +1,24 @@
# X-VC voice conversion on Athena
Isolated quality gate for `chenxie95/X-VC`, using the official inference path
from commit `49df8c591eafc48b096e466d96f9839f9c0dd739`. The UI is adapted from the
public Hugging Face Space at commit
`d761cd6421e85376b2656dfefd8471d7f35a42be` and runs locally without
ZeroGPU.
- Private URL: `http://192.168.1.212:8009`
- Source clip: speech content and timing to preserve
- Reference clip: target speaker identity
- Output: 16 kHz PCM WAV
- GPU: RTX 5080 only
- Persistent cache: `/data/voice/xvc/huggingface`
- Code and model license: MIT
The semantic tokenizer documents Chinese and English. German is therefore a
quality gate, not an assumed supported language. Keep OmniVoice installed: it
does text-to-speech cloning, while X-VC tests true audio-to-audio conversion.
Technical acceptance on 9 September 2026 used the repository's source and
target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The
result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about
2.9 GiB on the RTX 5080. German listening quality remains open.
+188
View File
@@ -0,0 +1,188 @@
"""Local Athena adaptation of the public X-VC Gradio demo."""
import logging
import os
import sys
import tempfile
import time
from typing import Tuple
import gradio as gr
import numpy as np
import soundfile as sf
import torch
from huggingface_hub import hf_hub_download
from omegaconf import OmegaConf
HERE = "/opt/xvc"
sys.path.insert(0, HERE)
from bins.infer_utils import precompute_conditions, run_offline, run_streaming, to_numpy_audio
from models.codec.sac.model import XVC
from utils.audio import audio_highpass_filter, audio_volume_normalize, load_audio
logging.basicConfig(level=logging.INFO)
log = logging.getLogger("xvc-local")
MODEL_REPO = "chenxie95/X-VC"
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
SAMPLE_RATE = 16000
LATENT_HOP_LENGTH = 1280
MAX_SECONDS = 20.0
MODE_OFFLINE = "Offline (höchste Qualität)"
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
def _load_model() -> XVC:
speaker_config = hf_hub_download(
repo_id=SPACE_REPO,
repo_type="space",
filename=f"{SPEAKER_SUBDIR}/configuration.json",
)
hf_hub_download(
repo_id=SPACE_REPO,
repo_type="space",
filename=f"{SPEAKER_SUBDIR}/pretrained_eres2net.ckpt",
)
checkpoint = hf_hub_download(repo_id=MODEL_REPO, filename="xvc.pt")
cfg = OmegaConf.load(os.path.join(HERE, "configs", "xvc.yaml"))
cfg["model"]["generator"].pop("loss_config", None)
cfg["model"].pop("discriminator", None)
cfg["model"]["generator"]["speaker_encoder"]["pretrained_dir"] = os.path.dirname(speaker_config)
infer_cfg = os.path.join(tempfile.gettempdir(), "xvc_inference.yaml")
OmegaConf.save(cfg, infer_cfg)
loaded = XVC.load_from_checkpoint(infer_cfg, checkpoint, device=torch.device("cuda"))
loaded.remove_weight_norm()
loaded = loaded.eval().to("cuda")
log.info("X-VC model ready on %s", torch.cuda.get_device_name(0))
return loaded
MODEL = _load_model()
def _prepare_wav(path: str) -> np.ndarray:
wav = load_audio(path, sampling_rate=SAMPLE_RATE, volume_normalize=False)
if wav is None or len(wav) == 0:
raise gr.Error("Die Audiodatei konnte nicht gelesen werden.")
wav = wav[: int(MAX_SECONDS * SAMPLE_RATE)]
wav = audio_volume_normalize(wav)
wav = audio_highpass_filter(wav, SAMPLE_RATE, 40)
remainder = len(wav) % LATENT_HOP_LENGTH
if remainder:
wav = np.pad(wav, (0, LATENT_HOP_LENGTH - remainder), mode="constant")
return wav.astype(np.float32)
def _tensor(wav: np.ndarray) -> torch.Tensor:
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
def _write_wav(audio: np.ndarray) -> str:
os.makedirs("/output", exist_ok=True)
path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav")
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16")
return path
@torch.inference_mode()
def convert(
source_audio: str,
reference_audio: str,
mode: str = MODE_OFFLINE,
chunk_ms: int = 2400,
current_ms: int = 120,
future_ms: int = 100,
smooth_ms: int = 20,
progress=gr.Progress(track_tqdm=True),
) -> Tuple[str, str]:
if not source_audio:
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
if not reference_audio:
raise gr.Error("Bitte eine Referenzdatei mit der Zielstimme hochladen.")
source_np = _prepare_wav(source_audio)
reference_np = _prepare_wav(reference_audio)
source_wav = _tensor(source_np)
target_wav = _tensor(reference_np)
seconds = len(source_np) / SAMPLE_RATE
started = time.time()
if mode == MODE_STREAMING:
history_ms = int(chunk_ms) - int(current_ms) - int(smooth_ms) - int(future_ms)
if history_ms < 0:
raise gr.Error("Fenster muss mindestens Current + Lookahead + Crossfade umfassen.")
speaker_condition, frame_condition = precompute_conditions(MODEL, target_wav, target_wav)
recon, latency_ms = run_streaming(
model=MODEL,
source_wav=source_wav,
speaker_condition=speaker_condition,
frame_condition=frame_condition,
sample_rate=SAMPLE_RATE,
chunk_ms=int(chunk_ms),
current_ms=int(current_ms),
future_ms=int(future_ms),
smooth_ms=int(smooth_ms),
)
elapsed = time.time() - started
latency = np.asarray(latency_ms, dtype=np.float64)
report = (
f"**Streaming** · {len(latency)} Chunks · Mittel **{latency.mean():.0f} ms**, "
f"P95 **{np.percentile(latency, 95):.0f} ms** · insgesamt {elapsed:.2f} s "
f"für {seconds:.2f} s Audio (RTF {elapsed / seconds:.2f})"
)
else:
recon = run_offline(MODEL, source_wav, target_wav, target_wav)
elapsed = time.time() - started
report = (
f"**Offline** · {seconds:.2f} s Audio in {elapsed:.2f} s "
f"(RTF {elapsed / seconds:.2f})"
)
return _write_wav(to_numpy_audio(recon)), report
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
HEADER = """# X-VC — Voice Changer
Die **Quelle** liefert Text, Aussprache und Timing. Die **Referenz** liefert die
Zielstimme. X-VC arbeitet Audio-zu-Audio ohne Transkript oder Training.
[Paper](https://arxiv.org/abs/2604.12456) · [Modell](https://huggingface.co/chenxie95/X-VC) ·
[Code](https://github.com/Jerrister/X-VC)
"""
with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(HEADER)
with gr.Row():
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
run = gr.Button("Stimme umwandeln", variant="primary")
output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False)
report = gr.Markdown()
with gr.Accordion("Erweiterte Einstellungen", open=False):
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
with gr.Row():
current_ms = gr.Slider(40, 640, value=120, step=40, label="Aktueller Chunk (ms)")
chunk_ms = gr.Slider(800, 4800, value=2400, step=200, label="Gesamtfenster (ms)")
with gr.Row():
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.")
run.click(
convert,
inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms],
outputs=[output, report],
api_name="convert",
)
if __name__ == "__main__":
demo.queue(default_concurrency_limit=1).launch(
server_name="0.0.0.0",
server_port=8009,
show_error=True,
)
@@ -0,0 +1,39 @@
services:
xvc-studio:
build: .
image: mike-ai/xvc-studio:2026-09-09
container_name: mike-ai-xvc-studio
restart: "no"
labels:
com.mike-ai.voice-change-worker: xvc
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
HF_HOME: /models/huggingface
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
ports:
- "127.0.0.1:8009:8009"
volumes:
- /data/voice/xvc/huggingface:/models/huggingface
- /data/voice/xvc/output:/output
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8009/ >/dev/null"]
interval: 5s
timeout: 3s
start_period: 600s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
frontend:
aliases: [xvc-studio]
networks:
frontend:
name: mike-ai_frontend
external: true
@@ -0,0 +1,29 @@
"""Small inference-only replacement for X-VC's training logger.
The upstream logger imports WandB, Matplotlib and TensorBoard at module import
time although model inference only uses the normal logging functions.
"""
import logging
logging.basicConfig(level=logging.INFO)
_logger = logging.getLogger("xvc")
debug = _logger.debug
info = _logger.info
warn = _logger.warning
warning = _logger.warning
error = _logger.error
def init(*_args, **_kwargs):
return None
def write_audio(*_args, **_kwargs):
return None
def write_loss(*_args, **_kwargs):
return None