Add private Vevo2 voice studio mode

This commit is contained in:
Mikei386
2026-09-09 13:38:00 +02:00
parent 535bd751b5
commit 68d02f32bd
15 changed files with 592 additions and 28 deletions
@@ -0,0 +1,64 @@
FROM mike-ai/bs-roformer-separator:0.47.0
ARG AMPHION_COMMIT=26f6883110181f1dbfe95c70a7c7dbaf4de5f42a
USER root
RUN apt-get update \
&& apt-get install -y --no-install-recommends git espeak-ng \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/open-mmlab/Amphion.git /opt/amphion \
&& cd /opt/amphion \
&& git checkout "${AMPHION_COMMIT}" \
&& rm -rf .git
# Vevo2 is tested on Athena with the CUDA/PyTorch stack inherited from the
# existing audio worker. Do not install Amphion's historical torch 2.0/cu118
# pins: Blackwell requires the newer cu128 runtime already present here.
RUN python -m pip install --no-cache-dir \
accelerate==1.10.1 \
diffusers==0.35.1 \
einops==0.8.1 \
easydict==1.13 \
g2p_en==2.1.0 \
humanfriendly==10.0 \
huggingface-hub==0.34.4 \
hydra-core==1.3.2 \
inflect==7.5.0 \
ipython==9.5.0 \
json5==0.12.1 \
librosa==0.11.0 \
loguru==0.7.3 \
matplotlib==3.10.6 \
munch==4.0.0 \
omegaconf==2.3.0 \
openai-whisper==20250625 \
phonemizer==3.3.0 \
python-multipart==0.0.20 \
praat-parselmouth==0.4.6 \
pypinyin==0.55.0 \
pyworld==0.3.5 \
ruamel.yaml==0.18.15 \
safetensors==0.6.2 \
tabulate==0.9.0 \
tgt==1.5 \
torchcrepe==0.0.24 \
transformers==4.56.1 \
typeguard==4.4.4 \
unidecode==1.4.0 \
vector-quantize-pytorch==1.12.5 \
vocos==0.1.0
WORKDIR /opt/amphion
ENV PYTHONPATH=/opt/amphion \
HF_HOME=/models/huggingface \
PYTHONUNBUFFERED=1
COPY run_fm_test.py /usr/local/bin/run_fm_test.py
COPY app.py /app/app.py
COPY index.html /app/index.html
EXPOSE 8008
HEALTHCHECK --interval=5s --timeout=3s --start-period=90s --retries=3 \
CMD curl -fsS http://127.0.0.1:8008/health || exit 1
ENTRYPOINT ["python", "/app/app.py"]
+30
View File
@@ -0,0 +1,30 @@
# Vevo2 Voice Conversion on Athena
This directory contains Athena's private Vevo2 voice-conversion studio.
- Code: `open-mmlab/Amphion` commit
`26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` (MIT)
- Weights: `RMSnow/Vevo2` (CC BY-NC-ND 4.0)
- Runtime: PyTorch 2.7.1 + CUDA 12.8 inherited from Athena's tested audio
separator image; the historical Amphion torch 2.0/cu118 pins are not used.
- Storage: `/data/voice/vevo2`; removing that directory and the test image
removes all downloaded artifacts.
- Private UI: `http://192.168.1.212:8008` through WireGuard only.
- Output: uncompressed mono WAV, 24 kHz.
The 9 September technical gate converted the official 8.6-second speech sample
through the production HTTP API in 2.342 seconds. A warm service start loaded
the model in 12.216 seconds, and peak CUDA allocation during conversion was
5605.6 MiB. The returned 24-kHz WAV was 412878 bytes and 8.6 seconds long. The
service stores named reference voices, accepts a source clip, and returns a
transient WAV download. Jobs and generated outputs are removed after delivery.
It is deliberately not exposed on the university interface.
The installed Compose project is named `mike-ai-voice`. Rebuild or recreate it
with an explicit `VOICE_GPU_UUID` so Docker keeps the service attached to the
RTX 5080. The profile controller starts and stops the existing container; it
does not rebuild it during a mode switch.
The weights are licensed CC BY-NC-ND 4.0. This deployment is for Mike's
private, non-commercial use only. Do not use or expose it as a public or
commercial voice-cloning service.
+202
View File
@@ -0,0 +1,202 @@
#!/usr/bin/env python3
"""Small local-only Vevo2 voice-conversion studio for Athena."""
from __future__ import annotations
import asyncio
import os
import re
import shutil
import subprocess
import threading
import time
import uuid
from contextlib import asynccontextmanager
from pathlib import Path
import torch
from fastapi import FastAPI, File, Form, HTTPException, UploadFile
from fastapi.responses import FileResponse, HTMLResponse
from starlette.background import BackgroundTask
import models.svc.vevo2.infer_vevo2_fm as vevo
DATA_DIR = Path(os.environ.get("VOICE_DATA_DIR", "/data"))
PROFILE_DIR = DATA_DIR / "profiles"
JOB_DIR = DATA_DIR / "jobs"
INDEX = Path("/app/index.html")
MAX_UPLOAD_BYTES = int(os.environ.get("MAX_UPLOAD_BYTES", str(100 * 1024 * 1024)))
MODEL_LOCK = threading.Lock()
PIPELINE = None
MODEL_LOAD_SECONDS: float | None = None
def safe_name(value: str) -> str:
value = re.sub(r"[^A-Za-z0-9_. -]+", "-", value.strip())
value = re.sub(r"\s+", "-", value).strip("-.")
return value[:64] or "voice"
def load_pipeline() -> None:
global PIPELINE, MODEL_LOAD_SECONDS
if PIPELINE is not None:
return
with MODEL_LOCK:
if PIPELINE is not None:
return
started = time.monotonic()
PIPELINE = vevo.load_inference_pipeline()
vevo.inference_pipeline = PIPELINE
MODEL_LOAD_SECONDS = time.monotonic() - started
def to_wav(source: Path, target: Path) -> None:
completed = subprocess.run(
["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
"-ac", "1", "-ar", "24000", "-c:a", "pcm_s16le", str(target)],
capture_output=True,
text=True,
timeout=180,
check=False,
)
if completed.returncode:
raise ValueError(completed.stderr.strip() or "Audio konnte nicht gelesen werden")
async def save_upload(upload: UploadFile, target: Path) -> None:
size = 0
with target.open("wb") as handle:
while chunk := await upload.read(1024 * 1024):
size += len(chunk)
if size > MAX_UPLOAD_BYTES:
raise HTTPException(413, "Audiodatei ist zu groß")
handle.write(chunk)
def profile_path(name: str) -> Path:
target = PROFILE_DIR / f"{safe_name(name)}.wav"
if not target.is_file():
raise HTTPException(404, "Referenzstimme wurde nicht gefunden")
return target
def run_conversion(source: Path, reference: Path, output: Path, pitch_shift: bool) -> tuple[float, float]:
"""Run the GPU-bound conversion off the API event loop."""
load_pipeline()
with MODEL_LOCK:
torch.cuda.reset_peak_memory_stats()
started = time.monotonic()
vevo.vevo2_fm(str(source), str(reference), str(output), shifted_src=pitch_shift)
elapsed = time.monotonic() - started
peak = torch.cuda.max_memory_allocated() / 1048576
return elapsed, peak
@asynccontextmanager
async def lifespan(_app: FastAPI):
PROFILE_DIR.mkdir(parents=True, exist_ok=True)
JOB_DIR.mkdir(parents=True, exist_ok=True)
load_pipeline()
yield
app = FastAPI(title="Athena Voice Studio", lifespan=lifespan)
@app.get("/", response_class=HTMLResponse)
def index() -> str:
return INDEX.read_text(encoding="utf-8")
@app.get("/health")
def health() -> dict:
return {
"status": "ok" if PIPELINE is not None else "starting",
"model": "RMSnow/Vevo2",
"sample_rate": 24000,
"model_load_seconds": MODEL_LOAD_SECONDS,
}
@app.get("/api/profiles")
def profiles() -> dict:
return {"profiles": [path.stem for path in sorted(PROFILE_DIR.glob("*.wav"))]}
@app.post("/api/profiles")
async def create_profile(
name: str = Form(...),
consent: bool = Form(False),
audio: UploadFile = File(...),
) -> dict:
if not consent:
raise HTTPException(400, "Bestätige, dass du die Stimme verwenden darfst")
clean = safe_name(name)
job = JOB_DIR / f"profile-{uuid.uuid4().hex}"
job.mkdir(parents=True)
raw = job / f"upload-{safe_name(audio.filename or 'reference.audio')}"
try:
await save_upload(audio, raw)
target = PROFILE_DIR / f"{clean}.wav"
temporary = job / "reference.wav"
to_wav(raw, temporary)
os.replace(temporary, target)
return {"status": "ok", "profile": clean}
except HTTPException:
raise
except Exception as exc:
raise HTTPException(400, str(exc)) from exc
finally:
shutil.rmtree(job, ignore_errors=True)
@app.delete("/api/profiles/{name}")
def delete_profile(name: str) -> dict:
target = profile_path(name)
target.unlink()
return {"status": "ok", "profile": target.stem}
@app.post("/api/convert")
async def convert(
source: UploadFile = File(...),
profile: str = Form(...),
pitch_shift: bool = Form(True),
) -> FileResponse:
reference = profile_path(profile)
job_id = uuid.uuid4().hex
job = JOB_DIR / f"convert-{job_id}"
job.mkdir(parents=True)
raw = job / f"upload-{safe_name(source.filename or 'source.audio')}"
source_wav = job / "source.wav"
output = JOB_DIR / f"voice-{job_id}.wav"
try:
await save_upload(source, raw)
to_wav(raw, source_wav)
elapsed, peak = await asyncio.to_thread(
run_conversion, source_wav, reference, output, pitch_shift
)
return FileResponse(
output,
media_type="audio/wav",
filename=f"{safe_name(profile)}-{job_id[:8]}.wav",
headers={
"X-Conversion-Seconds": f"{elapsed:.3f}",
"X-Peak-VRAM-MiB": f"{peak:.1f}",
},
background=BackgroundTask(output.unlink, missing_ok=True),
)
except HTTPException:
raise
except Exception as exc:
output.unlink(missing_ok=True)
raise HTTPException(500, f"Stimmenwandlung fehlgeschlagen: {exc}") from exc
finally:
shutil.rmtree(job, ignore_errors=True)
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host="0.0.0.0", port=8008)
@@ -0,0 +1,39 @@
services:
voice-studio:
build: .
image: mike-ai/vevo2-voice-studio:0.1
container_name: mike-ai-voice-studio
restart: "no"
labels:
com.mike-ai.voice-worker: vevo2
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
VOICE_DATA_DIR: /data
ports:
- "127.0.0.1:8008:8008"
volumes:
- /data/voice/vevo2/huggingface/hub/Vevo2:/opt/amphion/ckpts/Vevo2
- /data/voice/vevo2/huggingface:/models/huggingface
- /data/voice/vevo2/whisper:/root/.cache/whisper
- /data/voice/studio:/data
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/health | grep -q '\"status\":\"ok\"'"]
interval: 5s
timeout: 3s
start_period: 600s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
- frontend
networks:
frontend:
name: mike-ai_frontend
external: true
@@ -0,0 +1,10 @@
<!doctype html>
<html lang="de"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
<title>Athena Voice Studio</title><style>
:root{color-scheme:dark;--bg:#07101a;--panel:#0d1926;--line:#20374d;--text:#edf6ff;--muted:#91a7bb;--cyan:#45d7ff;--green:#66e3a4;--red:#ff7185}*{box-sizing:border-box}body{margin:0;background:radial-gradient(circle at top left,#12283d,var(--bg) 45%);color:var(--text);font:16px system-ui,sans-serif}main{max-width:960px;margin:auto;padding:32px 18px}h1{margin:.2rem 0}p{color:var(--muted)}.grid{display:grid;grid-template-columns:1fr 1fr;gap:16px}.card{background:var(--panel);border:1px solid var(--line);border-radius:16px;padding:20px}label{display:block;margin:13px 0 6px;color:var(--muted)}input,select,button{width:100%;border:1px solid var(--line);border-radius:9px;padding:11px;background:#07111d;color:var(--text);font:inherit}button{margin-top:15px;background:#10334a;border-color:var(--cyan);cursor:pointer}button:disabled{opacity:.5;cursor:wait}.row{display:flex;align-items:center;gap:9px}.row input{width:auto}.status{min-height:24px;margin-top:12px;color:var(--green)}.error{color:var(--red)}audio{width:100%;margin-top:12px}@media(max-width:700px){.grid{grid-template-columns:1fr}}</style></head>
<body><main><div style="color:var(--cyan);letter-spacing:.14em;text-transform:uppercase;font-size:12px">Mike AI · lokal auf Athena</div><h1>Voice Studio</h1><p>Eine Stimme als Referenz speichern und eine vorhandene Sprach- oder Gesangsaufnahme in diese Stimme übertragen. Ausgabe: unkomprimiertes WAV mit 24 kHz.</p>
<div class="grid"><section class="card"><h2>1 · Referenzstimme</h2><form id="profileForm"><label>Name</label><input name="name" required placeholder="z. B. Meine Stimme"><label>Saubere Referenzaufnahme</label><input name="audio" type="file" accept="audio/*" required><label class="row"><input name="consent" type="checkbox" value="true" required><span>Ich darf diese Stimme verwenden.</span></label><button>Stimme speichern</button></form><div class="status" id="profileStatus"></div></section>
<section class="card"><h2>2 · Stimme übertragen</h2><form id="convertForm"><label>Zielstimme</label><select name="profile" id="profiles" required></select><label>Quellaufnahme</label><input name="source" type="file" accept="audio/*" required><label class="row"><input name="pitch_shift" type="checkbox" value="true" checked><span>Tonlage automatisch an Zielstimme anpassen</span></label><button>WAV erzeugen</button></form><div class="status" id="convertStatus"></div><audio id="player" controls hidden></audio><a id="download" hidden>WAV herunterladen</a></section></div>
<p>Hinweis: Nur privat und mit Zustimmung verwenden. Das Vevo2-Gewicht ist nicht für kommerzielle Nutzung lizenziert.</p></main><script>
const $=id=>document.getElementById(id);async function loadProfiles(){let r=await fetch('/api/profiles'),d=await r.json();$('profiles').innerHTML=d.profiles.length?d.profiles.map(x=>`<option>${x}</option>`).join(''):'<option value="">Noch keine Stimme gespeichert</option>'}function state(id,text,error=false){let e=$(id);e.textContent=text;e.classList.toggle('error',error)}$('profileForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('profileStatus','Referenz wird geprüft und gespeichert …');try{let r=await fetch('/api/profiles',{method:'POST',body:new FormData(e.target)}),d=await r.json();if(!r.ok)throw Error(d.detail||`HTTP ${r.status}`);state('profileStatus',`„${d.profile}“ ist gespeichert.`);await loadProfiles()}catch(x){state('profileStatus',x.message,true)}finally{b.disabled=false}};$('convertForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('convertStatus','Modell arbeitet …');$('player').hidden=true;$('download').hidden=true;try{let f=new FormData(e.target);if(!f.has('pitch_shift'))f.append('pitch_shift','false');let r=await fetch('/api/convert',{method:'POST',body:f});if(!r.ok){let d=await r.json();throw Error(d.detail||`HTTP ${r.status}`)}let blob=await r.blob(),url=URL.createObjectURL(blob);$('player').src=url;$('player').hidden=false;$('download').href=url;$('download').download='voice-conversion.wav';$('download').hidden=false;state('convertStatus',`Fertig in ${r.headers.get('X-Conversion-Seconds')||'?'} s · Spitzen-VRAM ${r.headers.get('X-Peak-VRAM-MiB')||'?'} MiB`)}catch(x){state('convertStatus',x.message,true)}finally{b.disabled=false}};loadProfiles();
</script></body></html>
@@ -0,0 +1,47 @@
#!/usr/bin/env python3
"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
from __future__ import annotations
import argparse
import os
import time
import torch
import models.svc.vevo2.infer_vevo2_fm as vevo
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--source", required=True)
parser.add_argument("--reference", required=True)
parser.add_argument("--output", required=True)
parser.add_argument("--no-pitch-shift", action="store_true")
args = parser.parse_args()
os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
started = time.monotonic()
vevo.inference_pipeline = vevo.load_inference_pipeline()
loaded = time.monotonic()
vevo.vevo2_fm(
args.source,
args.reference,
args.output,
shifted_src=not args.no_pitch_shift,
)
finished = time.monotonic()
print(
{
"model_load_seconds": round(loaded - started, 3),
"conversion_seconds": round(finished - loaded, 3),
"total_seconds": round(finished - started, 3),
"peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
"output": args.output,
},
flush=True,
)
if __name__ == "__main__":
main()