Add BS-RoFormer vocal separation mode

This commit is contained in:
Mikei386 committed 2026-09-08 19:23:29 +02:00
1 parent a4e894fe70
commit 0069b61dbb
15 files changed
+421 -52

No files matched your search

@@ -0,0 +1,22 @@
FROM pytorch/pytorch:2.7.1-cuda12.8-cudnn9-runtime@sha256:c16f4c749e2d9e96878875cdf6cc45cddda1d1a36fddd371dd6f2360f1b6e2a2
RUN apt-get update \
&& DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends build-essential curl ffmpeg libsndfile1 \
&& rm -rf /var/lib/apt/lists/*
RUN python -m pip install --no-cache-dir \
"audio-separator[gpu]==0.47.0" \
"onnxruntime-gpu==1.22.0" \
"fastapi==0.116.1" \
"python-multipart==0.0.20" \
"uvicorn[standard]==0.35.0"
WORKDIR /app
COPY app.py index.html ./
ENV MODEL_FILENAME=model_bs_roformer_ep_317_sdr_12.9755.ckpt \
MODEL_DIR=/models \
JOB_DIR=/data/jobs
EXPOSE 8080
CMD ["sh", "-c", "audio-separator --model_filename \"$MODEL_FILENAME\" --model_file_dir \"$MODEL_DIR\" --download_model_only && exec uvicorn app:app --host 0.0.0.0 --port 8080 --workers 1"]
@@ -0,0 +1,16 @@
# Athena Vocal Separator
Exklusiver dritter Athena-Betriebsmodus für lokale Zwei-Spur-Trennung in
`vocals.flac` und `instrumental.flac`.
- Engine: `audio-separator` 0.47.0 (MIT)
- Modell: BS-RoFormer Viperx 1297,
`model_bs_roformer_ep_317_sdr_12.9755.ckpt`
- Modellbewertung im audio-separator-Katalog: Vocal SDR 12,9,
Instrumental SDR 17,0
- GPU: RTX 5080; LLM, Bildmodelle, TTS und ACE-Step sind dabei verriegelt.
- Privat erreichbar: `http://192.168.1.212:8007/`
Das Modell wird beim ersten Start nach `/data/models/audio-separator`
heruntergeladen. Temporäre Jobs liegen unter `/data/audio/separation` und
werden nach dem ZIP-Download entfernt.
@@ -0,0 +1,124 @@
from __future__ import annotations
import asyncio
import os
import shutil
import subprocess
import tempfile
import time
import zipfile
from pathlib import Path
from fastapi import FastAPI, File, HTTPException, UploadFile
from fastapi.responses import FileResponse, HTMLResponse
from starlette.background import BackgroundTask
MODEL = os.getenv("MODEL_FILENAME", "model_bs_roformer_ep_317_sdr_12.9755.ckpt")
MODEL_DIR = Path(os.getenv("MODEL_DIR", "/models"))
JOB_DIR = Path(os.getenv("JOB_DIR", "/data/jobs"))
MAX_UPLOAD = int(os.getenv("MAX_UPLOAD_BYTES", str(1024 ** 3)))
ALLOWED = {".wav", ".flac", ".mp3", ".m4a", ".aac", ".ogg", ".opus", ".wma"}
SEPARATION_LOCK = asyncio.Lock()
STARTED = time.time()
app = FastAPI(title="Athena Vocal Separator", version="1.0")
@app.get("/", response_class=HTMLResponse)
def index() -> str:
return Path("/app/index.html").read_text(encoding="utf-8")
@app.get("/health")
def health() -> dict:
checkpoint = MODEL_DIR / MODEL
return {
"status": "ok" if checkpoint.exists() else "starting",
"model": MODEL,
"model_ready": checkpoint.exists(),
"busy": SEPARATION_LOCK.locked(),
"uptime_seconds": round(time.time() - STARTED, 1),
}
def _cleanup(path: Path) -> None:
shutil.rmtree(path, ignore_errors=True)
def _run_separator(input_path: Path, output_dir: Path) -> None:
args = [
"audio-separator", str(input_path),
"--model_filename", MODEL,
"--model_file_dir", str(MODEL_DIR),
"--output_dir", str(output_dir),
"--output_format", "FLAC",
"--sample_rate", "44100",
"--use_soundfile",
"--use_autocast",
"--mdxc_segment_size", "256",
"--mdxc_overlap", "8",
"--mdxc_batch_size", "1",
]
completed = subprocess.run(args, capture_output=True, text=True, timeout=7200)
if completed.returncode:
detail = (completed.stderr or completed.stdout or "unknown error")[-4000:]
raise RuntimeError(detail)
@app.post("/v1/separate")
async def separate(file: UploadFile = File(...)) -> FileResponse:
suffix = Path(file.filename or "upload.wav").suffix.lower()
if suffix not in ALLOWED:
raise HTTPException(415, "Dieses Audioformat wird nicht unterstützt.")
if SEPARATION_LOCK.locked():
raise HTTPException(409, "Eine Trennung läuft bereits.")
job = Path(tempfile.mkdtemp(prefix="separate-", dir=JOB_DIR))
input_path = job / f"input{suffix}"
output_dir = job / "output"
output_dir.mkdir()
size = 0
try:
with input_path.open("wb") as handle:
while chunk := await file.read(1024 * 1024):
size += len(chunk)
if size > MAX_UPLOAD:
raise HTTPException(413, "Datei ist größer als 1 GiB.")
handle.write(chunk)
async with SEPARATION_LOCK:
await asyncio.to_thread(_run_separator, input_path, output_dir)
stems = sorted(output_dir.glob("*.flac"))
if len(stems) != 2:
raise RuntimeError(f"Erwartet wurden zwei FLAC-Dateien, gefunden: {len(stems)}")
archive = job / "athena-vocals-instrumental.zip"
with zipfile.ZipFile(archive, "w", compression=zipfile.ZIP_STORED) as bundle:
for stem in stems:
lower = stem.name.lower()
target = "vocals.flac" if "vocal" in lower else "instrumental.flac"
bundle.write(stem, target)
return FileResponse(
archive,
media_type="application/zip",
filename="athena-vocals-instrumental.zip",
background=BackgroundTask(_cleanup, job),
)
except HTTPException:
_cleanup(job)
raise
except subprocess.TimeoutExpired:
_cleanup(job)
raise HTTPException(504, "Die Trennung hat das Zeitlimit überschritten.")
except Exception as exc:
_cleanup(job)
raise HTTPException(500, f"Trennung fehlgeschlagen: {exc}")
@app.on_event("startup")
def prepare() -> None:
JOB_DIR.mkdir(parents=True, exist_ok=True)
MODEL_DIR.mkdir(parents=True, exist_ok=True)
for old in JOB_DIR.glob("separate-*"):
if old.is_dir() and time.time() - old.stat().st_mtime > 86400:
_cleanup(old)
@@ -0,0 +1,38 @@
services:
stem-separator:
build: .
image: mike-ai/bs-roformer-separator:0.47.0
container_name: mike-ai-stem-separator
labels:
com.mike-ai.stem-separator: "bs-roformer"
environment:
NVIDIA_VISIBLE_DEVICES: ${SEPARATOR_GPU_UUID:?set SEPARATOR_GPU_UUID to the RTX 5080 UUID}
MODEL_FILENAME: model_bs_roformer_ep_317_sdr_12.9755.ckpt
MODEL_DIR: /models
JOB_DIR: /data/jobs
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${SEPARATOR_GPU_UUID:?set SEPARATOR_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
ports:
- "127.0.0.1:${SEPARATOR_PORT:-8007}:8080"
volumes:
- ${SEPARATOR_MODEL_DIR:-/data/models/audio-separator}:/models
- ${SEPARATOR_DATA_DIR:-/data/audio/separation}:/data
shm_size: "2gb"
restart: "no"
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8080/health | grep -q '\"status\":\"ok\"'"]
interval: 15s
timeout: 5s
start_period: 600s
retries: 3
networks: [frontend]
networks:
frontend:
external: true
name: mike-ai_frontend
@@ -0,0 +1,4 @@
<!doctype html><html lang="de"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1"><title>Athena · Stimmen trennen</title><style>
:root{color-scheme:dark;--bg:#07111c;--card:#101d2b;--line:#26384b;--cyan:#48d7f5;--mint:#63e6be;--text:#ecf5ff;--muted:#91a4b7}*{box-sizing:border-box}body{margin:0;background:radial-gradient(circle at 20% 0,#142a42 0,#07111c 42%);font:16px system-ui,sans-serif;color:var(--text);min-height:100vh;display:grid;place-items:center;padding:24px}.card{width:min(760px,100%);padding:32px;border:1px solid var(--line);border-radius:22px;background:rgba(16,29,43,.96);box-shadow:0 25px 70px #0008}.eyebrow{color:var(--cyan);font-weight:800;letter-spacing:.14em;text-transform:uppercase;font-size:12px}h1{font-size:clamp(30px,5vw,52px);margin:.3em 0 .15em}p{color:var(--muted);line-height:1.6}.drop{display:block;margin:28px 0;padding:44px 24px;border:2px dashed #3f5a72;border-radius:18px;text-align:center;cursor:pointer;transition:.2s}.drop:hover,.drop.drag{border-color:var(--cyan);background:#48d7f50b}input{display:none}.file{color:var(--mint);font-weight:700;margin-top:8px}button{width:100%;border:0;border-radius:13px;padding:15px;font-weight:800;font-size:16px;background:linear-gradient(90deg,var(--cyan),var(--mint));color:#05202a;cursor:pointer}button:disabled{opacity:.45;cursor:not-allowed}.status{min-height:28px;margin-top:18px;color:var(--muted)}.bar{height:7px;background:#07111c;border-radius:9px;overflow:hidden;margin-top:12px}.fill{height:100%;width:0;background:linear-gradient(90deg,var(--cyan),var(--mint));transition:.4s}.run .fill{width:85%;animation:pulse 1.5s infinite alternate}@keyframes pulse{to{opacity:.45}}small{display:block;color:#71879a;margin-top:20px}</style></head><body><main class="card"><div class="eyebrow">Athena Audio Lab</div><h1>Stimmen sauber trennen</h1><p>BS‑RoFormer Viperx 1297 zerlegt deinen Titel in zwei verlustfreie FLAC-Spuren: Gesang und Instrumental.</p><label class="drop" id="drop">Audio auswählen oder hier ablegen<input id="file" type="file" accept="audio/*"><div class="file" id="name">Noch keine Datei gewählt</div></label><button id="start" disabled>Vocals und Instrumental erzeugen</button><div class="status" id="status">Bereit.</div><div class="bar" id="bar"><div class="fill"></div></div><small>Die Verarbeitung läuft lokal auf Athena. Nichts wird in eine Cloud hochgeladen.</small></main><script>
const file=document.querySelector('#file'),drop=document.querySelector('#drop'),name=document.querySelector('#name'),start=document.querySelector('#start'),status=document.querySelector('#status'),bar=document.querySelector('#bar');let selected;function choose(f){selected=f;name.textContent=f?`${f.name} · ${(f.size/1048576).toFixed(1)} MiB`:'Noch keine Datei gewählt';start.disabled=!f}file.onchange=()=>choose(file.files[0]);drop.ondragover=e=>{e.preventDefault();drop.classList.add('drag')};drop.ondragleave=()=>drop.classList.remove('drag');drop.ondrop=e=>{e.preventDefault();drop.classList.remove('drag');choose(e.dataTransfer.files[0])};start.onclick=async()=>{start.disabled=true;bar.classList.add('run');status.textContent='Modell trennt den Titel – das kann einige Minuten dauern …';let body=new FormData();body.append('file',selected);try{let r=await fetch('/v1/separate',{method:'POST',body});if(!r.ok)throw Error((await r.json()).detail||`HTTP ${r.status}`);let blob=await r.blob(),a=document.createElement('a');a.href=URL.createObjectURL(blob);a.download='athena-vocals-instrumental.zip';a.click();setTimeout(()=>URL.revokeObjectURL(a.href),5000);status.textContent='Fertig – ZIP mit vocals.flac und instrumental.flac wurde geladen.'}catch(e){status.textContent=`Fehler: ${e.message}`}finally{bar.classList.remove('run');start.disabled=false}};
</script></body></html>