feat: add speech noise separation

This commit is contained in:
Mikei386 committed 2026-09-09 08:24:12 +02:00
1 parent 0fd1966ae3
commit 805228abfd
7 files changed
+135 -14

No files matched your search

@@ -26,6 +26,7 @@ MODES = {
"vocals": {"model": MODEL, "stems": ("vocals", "instrumental"), "archive": "athena-vocals-instrumental.zip", "engine": "mdxc"},
"four_stem": {"model": "htdemucs_ft.yaml", "stems": ("vocals", "drums", "bass", "other"), "archive": "athena-4-stems.zip", "engine": "demucs"},
"six_stem": {"model": "htdemucs_6s.yaml", "stems": ("vocals", "drums", "bass", "guitar", "piano", "other"), "archive": "athena-6-stems-experimental.zip", "engine": "demucs"},
"speech": {"model": "MossFormer2_SE_48K", "stems": ("speech", "noise"), "archive": "athena-sprache-und-hintergrund.zip", "engine": "clearvoice"},
}
TARGETS = {
"vocals": {"mode": "vocals", "stem": "vocals", "remainder": "instrumental", "archive": "athena-gesang-und-rest.zip", "rest_file": "instrumental.flac"},
@@ -34,6 +35,7 @@ TARGETS = {
"guitar": {"mode": "six_stem", "stem": "guitar", "archive": "athena-gitarre-und-rest.zip", "rest_file": "rest-ohne-gitarre.flac"},
"piano": {"mode": "six_stem", "stem": "piano", "archive": "athena-piano-und-rest.zip", "rest_file": "rest-ohne-piano.flac"},
"other": {"mode": "six_stem", "stem": "other", "archive": "athena-sonstiges-und-rest.zip", "rest_file": "rest-ohne-sonstiges.flac"},
"speech": {"mode": "speech", "stem": "speech", "remainder": "noise", "archive": "athena-sprache-und-hintergrund.zip", "rest_file": "hintergrund-ohne-sprache.flac"},
}
app = FastAPI(title="Athena Stem Separator", version="2.0")
@@ -46,7 +48,14 @@ def index() -> str:
@app.get("/health")
def health() -> dict:
available = {name: (MODEL_DIR / mode["model"]).exists() for name, mode in MODES.items()}
available = {
name: (
(MODEL_DIR / "clearvoice" / mode["model"] / "last_best_checkpoint").exists()
if mode["engine"] == "clearvoice"
else (MODEL_DIR / mode["model"]).exists()
)
for name, mode in MODES.items()
}
return {
"status": "ok" if all(available.values()) else "starting",
"models": {name: mode["model"] for name, mode in MODES.items()},
@@ -62,6 +71,20 @@ def _cleanup(path: Path) -> None:
def _run_separator(input_path: Path, output_dir: Path, mode: dict) -> None:
if mode["engine"] == "clearvoice":
completed = subprocess.run(
[
"/opt/clearvoice-venv/bin/python", "/app/speech_enhance.py", str(input_path),
str(output_dir / "speech.flac"), str(output_dir / "noise.flac"),
],
capture_output=True,
text=True,
timeout=7200,
)
if completed.returncode:
detail = (completed.stderr or completed.stdout or "unknown ClearVoice error")[-4000:]
raise RuntimeError(detail)
return
args = [
"audio-separator", str(input_path),
"--model_filename", mode["model"],