Add private Vevo2 voice studio mode
This commit is contained in:
@@ -0,0 +1,64 @@
|
||||
FROM mike-ai/bs-roformer-separator:0.47.0
|
||||
|
||||
ARG AMPHION_COMMIT=26f6883110181f1dbfe95c70a7c7dbaf4de5f42a
|
||||
|
||||
USER root
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends git espeak-ng \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN git clone https://github.com/open-mmlab/Amphion.git /opt/amphion \
|
||||
&& cd /opt/amphion \
|
||||
&& git checkout "${AMPHION_COMMIT}" \
|
||||
&& rm -rf .git
|
||||
|
||||
# Vevo2 is tested on Athena with the CUDA/PyTorch stack inherited from the
|
||||
# existing audio worker. Do not install Amphion's historical torch 2.0/cu118
|
||||
# pins: Blackwell requires the newer cu128 runtime already present here.
|
||||
RUN python -m pip install --no-cache-dir \
|
||||
accelerate==1.10.1 \
|
||||
diffusers==0.35.1 \
|
||||
einops==0.8.1 \
|
||||
easydict==1.13 \
|
||||
g2p_en==2.1.0 \
|
||||
humanfriendly==10.0 \
|
||||
huggingface-hub==0.34.4 \
|
||||
hydra-core==1.3.2 \
|
||||
inflect==7.5.0 \
|
||||
ipython==9.5.0 \
|
||||
json5==0.12.1 \
|
||||
librosa==0.11.0 \
|
||||
loguru==0.7.3 \
|
||||
matplotlib==3.10.6 \
|
||||
munch==4.0.0 \
|
||||
omegaconf==2.3.0 \
|
||||
openai-whisper==20250625 \
|
||||
phonemizer==3.3.0 \
|
||||
python-multipart==0.0.20 \
|
||||
praat-parselmouth==0.4.6 \
|
||||
pypinyin==0.55.0 \
|
||||
pyworld==0.3.5 \
|
||||
ruamel.yaml==0.18.15 \
|
||||
safetensors==0.6.2 \
|
||||
tabulate==0.9.0 \
|
||||
tgt==1.5 \
|
||||
torchcrepe==0.0.24 \
|
||||
transformers==4.56.1 \
|
||||
typeguard==4.4.4 \
|
||||
unidecode==1.4.0 \
|
||||
vector-quantize-pytorch==1.12.5 \
|
||||
vocos==0.1.0
|
||||
|
||||
WORKDIR /opt/amphion
|
||||
ENV PYTHONPATH=/opt/amphion \
|
||||
HF_HOME=/models/huggingface \
|
||||
PYTHONUNBUFFERED=1
|
||||
|
||||
COPY run_fm_test.py /usr/local/bin/run_fm_test.py
|
||||
COPY app.py /app/app.py
|
||||
COPY index.html /app/index.html
|
||||
|
||||
EXPOSE 8008
|
||||
HEALTHCHECK --interval=5s --timeout=3s --start-period=90s --retries=3 \
|
||||
CMD curl -fsS http://127.0.0.1:8008/health || exit 1
|
||||
ENTRYPOINT ["python", "/app/app.py"]
|
||||
@@ -0,0 +1,30 @@
|
||||
# Vevo2 Voice Conversion on Athena
|
||||
|
||||
This directory contains Athena's private Vevo2 voice-conversion studio.
|
||||
|
||||
- Code: `open-mmlab/Amphion` commit
|
||||
`26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` (MIT)
|
||||
- Weights: `RMSnow/Vevo2` (CC BY-NC-ND 4.0)
|
||||
- Runtime: PyTorch 2.7.1 + CUDA 12.8 inherited from Athena's tested audio
|
||||
separator image; the historical Amphion torch 2.0/cu118 pins are not used.
|
||||
- Storage: `/data/voice/vevo2`; removing that directory and the test image
|
||||
removes all downloaded artifacts.
|
||||
- Private UI: `http://192.168.1.212:8008` through WireGuard only.
|
||||
- Output: uncompressed mono WAV, 24 kHz.
|
||||
|
||||
The 9 September technical gate converted the official 8.6-second speech sample
|
||||
through the production HTTP API in 2.342 seconds. A warm service start loaded
|
||||
the model in 12.216 seconds, and peak CUDA allocation during conversion was
|
||||
5605.6 MiB. The returned 24-kHz WAV was 412878 bytes and 8.6 seconds long. The
|
||||
service stores named reference voices, accepts a source clip, and returns a
|
||||
transient WAV download. Jobs and generated outputs are removed after delivery.
|
||||
It is deliberately not exposed on the university interface.
|
||||
|
||||
The installed Compose project is named `mike-ai-voice`. Rebuild or recreate it
|
||||
with an explicit `VOICE_GPU_UUID` so Docker keeps the service attached to the
|
||||
RTX 5080. The profile controller starts and stops the existing container; it
|
||||
does not rebuild it during a mode switch.
|
||||
|
||||
The weights are licensed CC BY-NC-ND 4.0. This deployment is for Mike's
|
||||
private, non-commercial use only. Do not use or expose it as a public or
|
||||
commercial voice-cloning service.
|
||||
@@ -0,0 +1,202 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Small local-only Vevo2 voice-conversion studio for Athena."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from contextlib import asynccontextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from fastapi import FastAPI, File, Form, HTTPException, UploadFile
|
||||
from fastapi.responses import FileResponse, HTMLResponse
|
||||
from starlette.background import BackgroundTask
|
||||
|
||||
import models.svc.vevo2.infer_vevo2_fm as vevo
|
||||
|
||||
|
||||
DATA_DIR = Path(os.environ.get("VOICE_DATA_DIR", "/data"))
|
||||
PROFILE_DIR = DATA_DIR / "profiles"
|
||||
JOB_DIR = DATA_DIR / "jobs"
|
||||
INDEX = Path("/app/index.html")
|
||||
MAX_UPLOAD_BYTES = int(os.environ.get("MAX_UPLOAD_BYTES", str(100 * 1024 * 1024)))
|
||||
MODEL_LOCK = threading.Lock()
|
||||
PIPELINE = None
|
||||
MODEL_LOAD_SECONDS: float | None = None
|
||||
|
||||
|
||||
def safe_name(value: str) -> str:
|
||||
value = re.sub(r"[^A-Za-z0-9_. -]+", "-", value.strip())
|
||||
value = re.sub(r"\s+", "-", value).strip("-.")
|
||||
return value[:64] or "voice"
|
||||
|
||||
|
||||
def load_pipeline() -> None:
|
||||
global PIPELINE, MODEL_LOAD_SECONDS
|
||||
if PIPELINE is not None:
|
||||
return
|
||||
with MODEL_LOCK:
|
||||
if PIPELINE is not None:
|
||||
return
|
||||
started = time.monotonic()
|
||||
PIPELINE = vevo.load_inference_pipeline()
|
||||
vevo.inference_pipeline = PIPELINE
|
||||
MODEL_LOAD_SECONDS = time.monotonic() - started
|
||||
|
||||
|
||||
def to_wav(source: Path, target: Path) -> None:
|
||||
completed = subprocess.run(
|
||||
["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
|
||||
"-ac", "1", "-ar", "24000", "-c:a", "pcm_s16le", str(target)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=180,
|
||||
check=False,
|
||||
)
|
||||
if completed.returncode:
|
||||
raise ValueError(completed.stderr.strip() or "Audio konnte nicht gelesen werden")
|
||||
|
||||
|
||||
async def save_upload(upload: UploadFile, target: Path) -> None:
|
||||
size = 0
|
||||
with target.open("wb") as handle:
|
||||
while chunk := await upload.read(1024 * 1024):
|
||||
size += len(chunk)
|
||||
if size > MAX_UPLOAD_BYTES:
|
||||
raise HTTPException(413, "Audiodatei ist zu groß")
|
||||
handle.write(chunk)
|
||||
|
||||
|
||||
def profile_path(name: str) -> Path:
|
||||
target = PROFILE_DIR / f"{safe_name(name)}.wav"
|
||||
if not target.is_file():
|
||||
raise HTTPException(404, "Referenzstimme wurde nicht gefunden")
|
||||
return target
|
||||
|
||||
|
||||
def run_conversion(source: Path, reference: Path, output: Path, pitch_shift: bool) -> tuple[float, float]:
|
||||
"""Run the GPU-bound conversion off the API event loop."""
|
||||
load_pipeline()
|
||||
with MODEL_LOCK:
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
started = time.monotonic()
|
||||
vevo.vevo2_fm(str(source), str(reference), str(output), shifted_src=pitch_shift)
|
||||
elapsed = time.monotonic() - started
|
||||
peak = torch.cuda.max_memory_allocated() / 1048576
|
||||
return elapsed, peak
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(_app: FastAPI):
|
||||
PROFILE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
JOB_DIR.mkdir(parents=True, exist_ok=True)
|
||||
load_pipeline()
|
||||
yield
|
||||
|
||||
|
||||
app = FastAPI(title="Athena Voice Studio", lifespan=lifespan)
|
||||
|
||||
|
||||
@app.get("/", response_class=HTMLResponse)
|
||||
def index() -> str:
|
||||
return INDEX.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
@app.get("/health")
|
||||
def health() -> dict:
|
||||
return {
|
||||
"status": "ok" if PIPELINE is not None else "starting",
|
||||
"model": "RMSnow/Vevo2",
|
||||
"sample_rate": 24000,
|
||||
"model_load_seconds": MODEL_LOAD_SECONDS,
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/profiles")
|
||||
def profiles() -> dict:
|
||||
return {"profiles": [path.stem for path in sorted(PROFILE_DIR.glob("*.wav"))]}
|
||||
|
||||
|
||||
@app.post("/api/profiles")
|
||||
async def create_profile(
|
||||
name: str = Form(...),
|
||||
consent: bool = Form(False),
|
||||
audio: UploadFile = File(...),
|
||||
) -> dict:
|
||||
if not consent:
|
||||
raise HTTPException(400, "Bestätige, dass du die Stimme verwenden darfst")
|
||||
clean = safe_name(name)
|
||||
job = JOB_DIR / f"profile-{uuid.uuid4().hex}"
|
||||
job.mkdir(parents=True)
|
||||
raw = job / f"upload-{safe_name(audio.filename or 'reference.audio')}"
|
||||
try:
|
||||
await save_upload(audio, raw)
|
||||
target = PROFILE_DIR / f"{clean}.wav"
|
||||
temporary = job / "reference.wav"
|
||||
to_wav(raw, temporary)
|
||||
os.replace(temporary, target)
|
||||
return {"status": "ok", "profile": clean}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise HTTPException(400, str(exc)) from exc
|
||||
finally:
|
||||
shutil.rmtree(job, ignore_errors=True)
|
||||
|
||||
|
||||
@app.delete("/api/profiles/{name}")
|
||||
def delete_profile(name: str) -> dict:
|
||||
target = profile_path(name)
|
||||
target.unlink()
|
||||
return {"status": "ok", "profile": target.stem}
|
||||
|
||||
|
||||
@app.post("/api/convert")
|
||||
async def convert(
|
||||
source: UploadFile = File(...),
|
||||
profile: str = Form(...),
|
||||
pitch_shift: bool = Form(True),
|
||||
) -> FileResponse:
|
||||
reference = profile_path(profile)
|
||||
job_id = uuid.uuid4().hex
|
||||
job = JOB_DIR / f"convert-{job_id}"
|
||||
job.mkdir(parents=True)
|
||||
raw = job / f"upload-{safe_name(source.filename or 'source.audio')}"
|
||||
source_wav = job / "source.wav"
|
||||
output = JOB_DIR / f"voice-{job_id}.wav"
|
||||
try:
|
||||
await save_upload(source, raw)
|
||||
to_wav(raw, source_wav)
|
||||
elapsed, peak = await asyncio.to_thread(
|
||||
run_conversion, source_wav, reference, output, pitch_shift
|
||||
)
|
||||
return FileResponse(
|
||||
output,
|
||||
media_type="audio/wav",
|
||||
filename=f"{safe_name(profile)}-{job_id[:8]}.wav",
|
||||
headers={
|
||||
"X-Conversion-Seconds": f"{elapsed:.3f}",
|
||||
"X-Peak-VRAM-MiB": f"{peak:.1f}",
|
||||
},
|
||||
background=BackgroundTask(output.unlink, missing_ok=True),
|
||||
)
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as exc:
|
||||
output.unlink(missing_ok=True)
|
||||
raise HTTPException(500, f"Stimmenwandlung fehlgeschlagen: {exc}") from exc
|
||||
finally:
|
||||
shutil.rmtree(job, ignore_errors=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import uvicorn
|
||||
|
||||
uvicorn.run(app, host="0.0.0.0", port=8008)
|
||||
@@ -0,0 +1,39 @@
|
||||
services:
|
||||
voice-studio:
|
||||
build: .
|
||||
image: mike-ai/vevo2-voice-studio:0.1
|
||||
container_name: mike-ai-voice-studio
|
||||
restart: "no"
|
||||
labels:
|
||||
com.mike-ai.voice-worker: vevo2
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
VOICE_DATA_DIR: /data
|
||||
ports:
|
||||
- "127.0.0.1:8008:8008"
|
||||
volumes:
|
||||
- /data/voice/vevo2/huggingface/hub/Vevo2:/opt/amphion/ckpts/Vevo2
|
||||
- /data/voice/vevo2/huggingface:/models/huggingface
|
||||
- /data/voice/vevo2/whisper:/root/.cache/whisper
|
||||
- /data/voice/studio:/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/health | grep -q '\"status\":\"ok\"'"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
start_period: 600s
|
||||
retries: 3
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
|
||||
capabilities: [gpu]
|
||||
networks:
|
||||
- frontend
|
||||
|
||||
networks:
|
||||
frontend:
|
||||
name: mike-ai_frontend
|
||||
external: true
|
||||
@@ -0,0 +1,10 @@
|
||||
<!doctype html>
|
||||
<html lang="de"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
|
||||
<title>Athena Voice Studio</title><style>
|
||||
:root{color-scheme:dark;--bg:#07101a;--panel:#0d1926;--line:#20374d;--text:#edf6ff;--muted:#91a7bb;--cyan:#45d7ff;--green:#66e3a4;--red:#ff7185}*{box-sizing:border-box}body{margin:0;background:radial-gradient(circle at top left,#12283d,var(--bg) 45%);color:var(--text);font:16px system-ui,sans-serif}main{max-width:960px;margin:auto;padding:32px 18px}h1{margin:.2rem 0}p{color:var(--muted)}.grid{display:grid;grid-template-columns:1fr 1fr;gap:16px}.card{background:var(--panel);border:1px solid var(--line);border-radius:16px;padding:20px}label{display:block;margin:13px 0 6px;color:var(--muted)}input,select,button{width:100%;border:1px solid var(--line);border-radius:9px;padding:11px;background:#07111d;color:var(--text);font:inherit}button{margin-top:15px;background:#10334a;border-color:var(--cyan);cursor:pointer}button:disabled{opacity:.5;cursor:wait}.row{display:flex;align-items:center;gap:9px}.row input{width:auto}.status{min-height:24px;margin-top:12px;color:var(--green)}.error{color:var(--red)}audio{width:100%;margin-top:12px}@media(max-width:700px){.grid{grid-template-columns:1fr}}</style></head>
|
||||
<body><main><div style="color:var(--cyan);letter-spacing:.14em;text-transform:uppercase;font-size:12px">Mike AI · lokal auf Athena</div><h1>Voice Studio</h1><p>Eine Stimme als Referenz speichern und eine vorhandene Sprach- oder Gesangsaufnahme in diese Stimme übertragen. Ausgabe: unkomprimiertes WAV mit 24 kHz.</p>
|
||||
<div class="grid"><section class="card"><h2>1 · Referenzstimme</h2><form id="profileForm"><label>Name</label><input name="name" required placeholder="z. B. Meine Stimme"><label>Saubere Referenzaufnahme</label><input name="audio" type="file" accept="audio/*" required><label class="row"><input name="consent" type="checkbox" value="true" required><span>Ich darf diese Stimme verwenden.</span></label><button>Stimme speichern</button></form><div class="status" id="profileStatus"></div></section>
|
||||
<section class="card"><h2>2 · Stimme übertragen</h2><form id="convertForm"><label>Zielstimme</label><select name="profile" id="profiles" required></select><label>Quellaufnahme</label><input name="source" type="file" accept="audio/*" required><label class="row"><input name="pitch_shift" type="checkbox" value="true" checked><span>Tonlage automatisch an Zielstimme anpassen</span></label><button>WAV erzeugen</button></form><div class="status" id="convertStatus"></div><audio id="player" controls hidden></audio><a id="download" hidden>WAV herunterladen</a></section></div>
|
||||
<p>Hinweis: Nur privat und mit Zustimmung verwenden. Das Vevo2-Gewicht ist nicht für kommerzielle Nutzung lizenziert.</p></main><script>
|
||||
const $=id=>document.getElementById(id);async function loadProfiles(){let r=await fetch('/api/profiles'),d=await r.json();$('profiles').innerHTML=d.profiles.length?d.profiles.map(x=>`<option>${x}</option>`).join(''):'<option value="">Noch keine Stimme gespeichert</option>'}function state(id,text,error=false){let e=$(id);e.textContent=text;e.classList.toggle('error',error)}$('profileForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('profileStatus','Referenz wird geprüft und gespeichert …');try{let r=await fetch('/api/profiles',{method:'POST',body:new FormData(e.target)}),d=await r.json();if(!r.ok)throw Error(d.detail||`HTTP ${r.status}`);state('profileStatus',`„${d.profile}“ ist gespeichert.`);await loadProfiles()}catch(x){state('profileStatus',x.message,true)}finally{b.disabled=false}};$('convertForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('convertStatus','Modell arbeitet …');$('player').hidden=true;$('download').hidden=true;try{let f=new FormData(e.target);if(!f.has('pitch_shift'))f.append('pitch_shift','false');let r=await fetch('/api/convert',{method:'POST',body:f});if(!r.ok){let d=await r.json();throw Error(d.detail||`HTTP ${r.status}`)}let blob=await r.blob(),url=URL.createObjectURL(blob);$('player').src=url;$('player').hidden=false;$('download').href=url;$('download').download='voice-conversion.wav';$('download').hidden=false;state('convertStatus',`Fertig in ${r.headers.get('X-Conversion-Seconds')||'?'} s · Spitzen-VRAM ${r.headers.get('X-Peak-VRAM-MiB')||'?'} MiB`)}catch(x){state('convertStatus',x.message,true)}finally{b.disabled=false}};loadProfiles();
|
||||
</script></body></html>
|
||||
@@ -0,0 +1,47 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import time
|
||||
|
||||
import torch
|
||||
|
||||
import models.svc.vevo2.infer_vevo2_fm as vevo
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--source", required=True)
|
||||
parser.add_argument("--reference", required=True)
|
||||
parser.add_argument("--output", required=True)
|
||||
parser.add_argument("--no-pitch-shift", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
|
||||
started = time.monotonic()
|
||||
vevo.inference_pipeline = vevo.load_inference_pipeline()
|
||||
loaded = time.monotonic()
|
||||
vevo.vevo2_fm(
|
||||
args.source,
|
||||
args.reference,
|
||||
args.output,
|
||||
shifted_src=not args.no_pitch_shift,
|
||||
)
|
||||
finished = time.monotonic()
|
||||
print(
|
||||
{
|
||||
"model_load_seconds": round(loaded - started, 3),
|
||||
"conversion_seconds": round(finished - loaded, 3),
|
||||
"total_seconds": round(finished - started, 3),
|
||||
"peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
|
||||
"output": args.output,
|
||||
},
|
||||
flush=True,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user