Add X-VC voice conversion mode
This commit is contained in:
1 parent
17f1a08d7d
commit
87a2ae5704
15 files changed
+544
-28
No files matched your search
@@ -0,0 +1,43 @@
|
||||
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
|
||||
|
||||
ARG XVC_COMMIT=49df8c591eafc48b096e466d96f9839f9c0dd739
|
||||
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl ffmpeg git libsndfile1 python3 python3-pip python3-venv \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN git clone https://github.com/Jerrister/X-VC.git /opt/xvc \
|
||||
&& cd /opt/xvc \
|
||||
&& git checkout "${XVC_COMMIT}" \
|
||||
&& rm -rf .git
|
||||
|
||||
RUN python3 -m venv /opt/venv
|
||||
ENV PATH="/opt/venv/bin:${PATH}"
|
||||
|
||||
# RTX 5080: use a CUDA 12.8 PyTorch build instead of X-VC's older training pin.
|
||||
RUN python -m pip install --no-cache-dir --upgrade pip \
|
||||
&& python -m pip install --no-cache-dir \
|
||||
--index-url https://download.pytorch.org/whl/cu128 \
|
||||
"torch==2.8.0" "torchaudio==2.8.0" \
|
||||
&& python -m pip install --no-cache-dir \
|
||||
"gradio>=5.49,<6" "huggingface_hub>=0.34,<2" \
|
||||
"transformers>=4.44,<5" "hydra-core>=1.3,<2" "omegaconf>=2.3,<3" \
|
||||
"x-transformers>=1.40,<3" "einops>=0.8,<1" "einx>=0.3,<1" \
|
||||
"numpy>=1.26,<3" "scipy>=1.13,<2" "soundfile>=0.12,<1" \
|
||||
"soxr>=0.5,<1" "tqdm>=4.66,<5" "wandb>=0.18,<1"
|
||||
|
||||
RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2"
|
||||
|
||||
COPY app.py /opt/xvc/local_webui.py
|
||||
COPY inference_log.py /opt/xvc/utils/log.py
|
||||
|
||||
ENV HF_HOME=/models/huggingface \
|
||||
PYTHONUNBUFFERED=1
|
||||
|
||||
EXPOSE 8009
|
||||
|
||||
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
|
||||
CMD curl -fsS http://127.0.0.1:8009/ >/dev/null || exit 1
|
||||
|
||||
CMD ["python", "/opt/xvc/local_webui.py"]
|
||||
@@ -0,0 +1,24 @@
|
||||
# X-VC voice conversion on Athena
|
||||
|
||||
Isolated quality gate for `chenxie95/X-VC`, using the official inference path
|
||||
from commit `49df8c591eafc48b096e466d96f9839f9c0dd739`. The UI is adapted from the
|
||||
public Hugging Face Space at commit
|
||||
`d761cd6421e85376b2656dfefd8471d7f35a42be` and runs locally without
|
||||
ZeroGPU.
|
||||
|
||||
- Private URL: `http://192.168.1.212:8009`
|
||||
- Source clip: speech content and timing to preserve
|
||||
- Reference clip: target speaker identity
|
||||
- Output: 16 kHz PCM WAV
|
||||
- GPU: RTX 5080 only
|
||||
- Persistent cache: `/data/voice/xvc/huggingface`
|
||||
- Code and model license: MIT
|
||||
|
||||
The semantic tokenizer documents Chinese and English. German is therefore a
|
||||
quality gate, not an assumed supported language. Keep OmniVoice installed: it
|
||||
does text-to-speech cloning, while X-VC tests true audio-to-audio conversion.
|
||||
|
||||
Technical acceptance on 9 September 2026 used the repository's source and
|
||||
target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The
|
||||
result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about
|
||||
2.9 GiB on the RTX 5080. German listening quality remains open.
|
||||
@@ -0,0 +1,188 @@
|
||||
"""Local Athena adaptation of the public X-VC Gradio demo."""
|
||||
|
||||
import logging
|
||||
import os
|
||||
import sys
|
||||
import tempfile
|
||||
import time
|
||||
from typing import Tuple
|
||||
|
||||
import gradio as gr
|
||||
import numpy as np
|
||||
import soundfile as sf
|
||||
import torch
|
||||
from huggingface_hub import hf_hub_download
|
||||
from omegaconf import OmegaConf
|
||||
|
||||
HERE = "/opt/xvc"
|
||||
sys.path.insert(0, HERE)
|
||||
|
||||
from bins.infer_utils import precompute_conditions, run_offline, run_streaming, to_numpy_audio
|
||||
from models.codec.sac.model import XVC
|
||||
from utils.audio import audio_highpass_filter, audio_volume_normalize, load_audio
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
log = logging.getLogger("xvc-local")
|
||||
|
||||
MODEL_REPO = "chenxie95/X-VC"
|
||||
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
|
||||
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
|
||||
SAMPLE_RATE = 16000
|
||||
LATENT_HOP_LENGTH = 1280
|
||||
MAX_SECONDS = 20.0
|
||||
MODE_OFFLINE = "Offline (höchste Qualität)"
|
||||
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
|
||||
|
||||
|
||||
def _load_model() -> XVC:
|
||||
speaker_config = hf_hub_download(
|
||||
repo_id=SPACE_REPO,
|
||||
repo_type="space",
|
||||
filename=f"{SPEAKER_SUBDIR}/configuration.json",
|
||||
)
|
||||
hf_hub_download(
|
||||
repo_id=SPACE_REPO,
|
||||
repo_type="space",
|
||||
filename=f"{SPEAKER_SUBDIR}/pretrained_eres2net.ckpt",
|
||||
)
|
||||
checkpoint = hf_hub_download(repo_id=MODEL_REPO, filename="xvc.pt")
|
||||
|
||||
cfg = OmegaConf.load(os.path.join(HERE, "configs", "xvc.yaml"))
|
||||
cfg["model"]["generator"].pop("loss_config", None)
|
||||
cfg["model"].pop("discriminator", None)
|
||||
cfg["model"]["generator"]["speaker_encoder"]["pretrained_dir"] = os.path.dirname(speaker_config)
|
||||
infer_cfg = os.path.join(tempfile.gettempdir(), "xvc_inference.yaml")
|
||||
OmegaConf.save(cfg, infer_cfg)
|
||||
|
||||
loaded = XVC.load_from_checkpoint(infer_cfg, checkpoint, device=torch.device("cuda"))
|
||||
loaded.remove_weight_norm()
|
||||
loaded = loaded.eval().to("cuda")
|
||||
log.info("X-VC model ready on %s", torch.cuda.get_device_name(0))
|
||||
return loaded
|
||||
|
||||
|
||||
MODEL = _load_model()
|
||||
|
||||
|
||||
def _prepare_wav(path: str) -> np.ndarray:
|
||||
wav = load_audio(path, sampling_rate=SAMPLE_RATE, volume_normalize=False)
|
||||
if wav is None or len(wav) == 0:
|
||||
raise gr.Error("Die Audiodatei konnte nicht gelesen werden.")
|
||||
wav = wav[: int(MAX_SECONDS * SAMPLE_RATE)]
|
||||
wav = audio_volume_normalize(wav)
|
||||
wav = audio_highpass_filter(wav, SAMPLE_RATE, 40)
|
||||
remainder = len(wav) % LATENT_HOP_LENGTH
|
||||
if remainder:
|
||||
wav = np.pad(wav, (0, LATENT_HOP_LENGTH - remainder), mode="constant")
|
||||
return wav.astype(np.float32)
|
||||
|
||||
|
||||
def _tensor(wav: np.ndarray) -> torch.Tensor:
|
||||
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
|
||||
|
||||
|
||||
def _write_wav(audio: np.ndarray) -> str:
|
||||
os.makedirs("/output", exist_ok=True)
|
||||
path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav")
|
||||
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16")
|
||||
return path
|
||||
|
||||
|
||||
@torch.inference_mode()
|
||||
def convert(
|
||||
source_audio: str,
|
||||
reference_audio: str,
|
||||
mode: str = MODE_OFFLINE,
|
||||
chunk_ms: int = 2400,
|
||||
current_ms: int = 120,
|
||||
future_ms: int = 100,
|
||||
smooth_ms: int = 20,
|
||||
progress=gr.Progress(track_tqdm=True),
|
||||
) -> Tuple[str, str]:
|
||||
if not source_audio:
|
||||
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
|
||||
if not reference_audio:
|
||||
raise gr.Error("Bitte eine Referenzdatei mit der Zielstimme hochladen.")
|
||||
|
||||
source_np = _prepare_wav(source_audio)
|
||||
reference_np = _prepare_wav(reference_audio)
|
||||
source_wav = _tensor(source_np)
|
||||
target_wav = _tensor(reference_np)
|
||||
seconds = len(source_np) / SAMPLE_RATE
|
||||
started = time.time()
|
||||
|
||||
if mode == MODE_STREAMING:
|
||||
history_ms = int(chunk_ms) - int(current_ms) - int(smooth_ms) - int(future_ms)
|
||||
if history_ms < 0:
|
||||
raise gr.Error("Fenster muss mindestens Current + Lookahead + Crossfade umfassen.")
|
||||
speaker_condition, frame_condition = precompute_conditions(MODEL, target_wav, target_wav)
|
||||
recon, latency_ms = run_streaming(
|
||||
model=MODEL,
|
||||
source_wav=source_wav,
|
||||
speaker_condition=speaker_condition,
|
||||
frame_condition=frame_condition,
|
||||
sample_rate=SAMPLE_RATE,
|
||||
chunk_ms=int(chunk_ms),
|
||||
current_ms=int(current_ms),
|
||||
future_ms=int(future_ms),
|
||||
smooth_ms=int(smooth_ms),
|
||||
)
|
||||
elapsed = time.time() - started
|
||||
latency = np.asarray(latency_ms, dtype=np.float64)
|
||||
report = (
|
||||
f"**Streaming** · {len(latency)} Chunks · Mittel **{latency.mean():.0f} ms**, "
|
||||
f"P95 **{np.percentile(latency, 95):.0f} ms** · insgesamt {elapsed:.2f} s "
|
||||
f"für {seconds:.2f} s Audio (RTF {elapsed / seconds:.2f})"
|
||||
)
|
||||
else:
|
||||
recon = run_offline(MODEL, source_wav, target_wav, target_wav)
|
||||
elapsed = time.time() - started
|
||||
report = (
|
||||
f"**Offline** · {seconds:.2f} s Audio in {elapsed:.2f} s "
|
||||
f"(RTF {elapsed / seconds:.2f})"
|
||||
)
|
||||
|
||||
return _write_wav(to_numpy_audio(recon)), report
|
||||
|
||||
|
||||
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
|
||||
HEADER = """# X-VC — Voice Changer
|
||||
|
||||
Die **Quelle** liefert Text, Aussprache und Timing. Die **Referenz** liefert die
|
||||
Zielstimme. X-VC arbeitet Audio-zu-Audio ohne Transkript oder Training.
|
||||
|
||||
[Paper](https://arxiv.org/abs/2604.12456) · [Modell](https://huggingface.co/chenxie95/X-VC) ·
|
||||
[Code](https://github.com/Jerrister/X-VC)
|
||||
"""
|
||||
|
||||
with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as demo:
|
||||
with gr.Column(elem_id="col-container"):
|
||||
gr.Markdown(HEADER)
|
||||
with gr.Row():
|
||||
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
|
||||
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
|
||||
run = gr.Button("Stimme umwandeln", variant="primary")
|
||||
output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False)
|
||||
report = gr.Markdown()
|
||||
with gr.Accordion("Erweiterte Einstellungen", open=False):
|
||||
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
|
||||
with gr.Row():
|
||||
current_ms = gr.Slider(40, 640, value=120, step=40, label="Aktueller Chunk (ms)")
|
||||
chunk_ms = gr.Slider(800, 4800, value=2400, step=200, label="Gesamtfenster (ms)")
|
||||
with gr.Row():
|
||||
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
|
||||
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
|
||||
gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.")
|
||||
run.click(
|
||||
convert,
|
||||
inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms],
|
||||
outputs=[output, report],
|
||||
api_name="convert",
|
||||
)
|
||||
|
||||
if __name__ == "__main__":
|
||||
demo.queue(default_concurrency_limit=1).launch(
|
||||
server_name="0.0.0.0",
|
||||
server_port=8009,
|
||||
show_error=True,
|
||||
)
|
||||
@@ -0,0 +1,39 @@
|
||||
services:
|
||||
xvc-studio:
|
||||
build: .
|
||||
image: mike-ai/xvc-studio:2026-09-09
|
||||
container_name: mike-ai-xvc-studio
|
||||
restart: "no"
|
||||
labels:
|
||||
com.mike-ai.voice-change-worker: xvc
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
HF_HOME: /models/huggingface
|
||||
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
|
||||
ports:
|
||||
- "127.0.0.1:8009:8009"
|
||||
volumes:
|
||||
- /data/voice/xvc/huggingface:/models/huggingface
|
||||
- /data/voice/xvc/output:/output
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8009/ >/dev/null"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
start_period: 600s
|
||||
retries: 3
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
|
||||
capabilities: [gpu]
|
||||
networks:
|
||||
frontend:
|
||||
aliases: [xvc-studio]
|
||||
|
||||
networks:
|
||||
frontend:
|
||||
name: mike-ai_frontend
|
||||
external: true
|
||||
@@ -0,0 +1,29 @@
|
||||
"""Small inference-only replacement for X-VC's training logger.
|
||||
|
||||
The upstream logger imports WandB, Matplotlib and TensorBoard at module import
|
||||
time although model inference only uses the normal logging functions.
|
||||
"""
|
||||
|
||||
import logging
|
||||
|
||||
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
_logger = logging.getLogger("xvc")
|
||||
|
||||
debug = _logger.debug
|
||||
info = _logger.info
|
||||
warn = _logger.warning
|
||||
warning = _logger.warning
|
||||
error = _logger.error
|
||||
|
||||
|
||||
def init(*_args, **_kwargs):
|
||||
return None
|
||||
|
||||
|
||||
def write_audio(*_args, **_kwargs):
|
||||
return None
|
||||
|
||||
|
||||
def write_loss(*_args, **_kwargs):
|
||||
return None
|
||||
Reference in new issue
Block a user