Add private Vevo2 voice studio mode
This commit is contained in:
@@ -571,6 +571,7 @@ services:
|
||||
TTS_WORKER: qwen3
|
||||
MUSIC_WORKER: acestep
|
||||
SEPARATOR_WORKER: bs-roformer
|
||||
VOICE_WORKER: vevo2
|
||||
networks: [control]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
healthcheck:
|
||||
@@ -860,6 +861,7 @@ services:
|
||||
MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
|
||||
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
|
||||
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
|
||||
VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
|
||||
HOST_PROC: /host/proc
|
||||
HOST_DATA: /host/data
|
||||
HOST_MODELS: /host/models
|
||||
|
||||
@@ -52,6 +52,15 @@ class DashboardModeTests(unittest.TestCase):
|
||||
self.assertEqual(json.loads(request.data), {"mode": "separation"})
|
||||
self.assertEqual(request.get_header("Authorization"), "Bearer test-key")
|
||||
|
||||
def test_voice_mode_is_forwarded_to_router(self):
|
||||
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
|
||||
status, body = self.dashboard.change_mode("voice")
|
||||
|
||||
self.assertEqual(status, 202)
|
||||
self.assertEqual(body, {"status": "accepted"})
|
||||
request = urlopen.call_args.args[0]
|
||||
self.assertEqual(json.loads(request.data), {"mode": "voice"})
|
||||
|
||||
def test_unknown_mode_is_rejected_without_router_request(self):
|
||||
with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen:
|
||||
status, body = self.dashboard.change_mode("unknown")
|
||||
|
||||
@@ -46,7 +46,43 @@ def separator_item(state="exited"):
|
||||
"Labels": {controller.SEPARATOR_LABEL_KEY: "bs-roformer"}}
|
||||
|
||||
|
||||
def voice_item(state="exited"):
|
||||
return {"Id": "id-voice", "State": state,
|
||||
"Labels": {controller.VOICE_LABEL_KEY: "vevo2"}}
|
||||
|
||||
|
||||
class ProfileControllerTests(unittest.TestCase):
|
||||
def test_voice_start_exclusively_stops_gpu_workers(self):
|
||||
profiles = {name: item(name) for name in controller.ALLOWED}
|
||||
profiles["medium"] = item("medium", "running")
|
||||
calls = []
|
||||
|
||||
def request(method, path):
|
||||
calls.append((method, path))
|
||||
return 204, b""
|
||||
|
||||
with patch.object(controller, "VOICE_WORKER", "vevo2"), \
|
||||
patch.object(controller, "MUSIC_WORKER", "acestep"), \
|
||||
patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
|
||||
patch.object(controller, "containers", return_value=profiles), \
|
||||
patch.object(controller, "voice_container", return_value=voice_item()), \
|
||||
patch.object(controller, "music_container", return_value=music_item("running")), \
|
||||
patch.object(controller, "separator_container", return_value=separator_item("running")), \
|
||||
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
|
||||
patch.object(controller, "tts_container", return_value=tts_item()), \
|
||||
patch.object(controller, "docker_request", side_effect=request):
|
||||
result = controller.set_voice_worker(True)
|
||||
|
||||
self.assertEqual(result, {"voice_worker": "vevo2", "state": "running"})
|
||||
self.assertEqual(calls, [
|
||||
("POST", "/containers/id-medium/stop?t=120"),
|
||||
("POST", "/containers/id-flux/stop?t=20"),
|
||||
("POST", "/containers/id-tts/stop?t=30"),
|
||||
("POST", "/containers/id-music/stop?t=30"),
|
||||
("POST", "/containers/id-separator/stop?t=30"),
|
||||
("POST", "/containers/id-voice/start"),
|
||||
])
|
||||
|
||||
def test_separator_start_exclusively_stops_gpu_workers(self):
|
||||
profiles = {name: item(name) for name in controller.ALLOWED}
|
||||
profiles["large"] = item("large", "running")
|
||||
|
||||
+18
-4
@@ -1,11 +1,13 @@
|
||||
# Athena-Betriebsmodi
|
||||
|
||||
Athena besitzt drei gegenseitig exklusive Betriebsmodi:
|
||||
Athena besitzt vier gegenseitig exklusive Betriebsmodi:
|
||||
|
||||
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt.
|
||||
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt.
|
||||
- `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt
|
||||
Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt.
|
||||
- `voice`: Vevo2 überträgt Sprache oder Gesang auf eine gespeicherte
|
||||
Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt.
|
||||
|
||||
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
|
||||
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
|
||||
@@ -13,8 +15,8 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen
|
||||
|
||||
## Bedienung
|
||||
|
||||
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio** und
|
||||
**Audio trennen** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
|
||||
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**
|
||||
und **Voice Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
|
||||
|
||||
- **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende
|
||||
Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der
|
||||
@@ -48,12 +50,22 @@ nicht eine reine Synthesizer-Spur. Sprache nutzt das 48-kHz-Modell
|
||||
`audio-separator` 0.47.0. Die ältere API-Auswahl kompletter 2-/4-/6-Stem-Sätze
|
||||
bleibt rückwärtskompatibel.
|
||||
|
||||
Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter
|
||||
`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter
|
||||
`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem
|
||||
Speichern eine Bestätigung der Nutzungsberechtigung. Vevo2 gibt
|
||||
unkomprimiertes 24-kHz-WAV aus und wandelt sowohl Sprache als auch Gesang;
|
||||
dies ist noch kein Echtzeit-Live-Voice-Changer. Die Gewichte sind
|
||||
CC BY-NC-ND 4.0 lizenziert und werden auf Athena ausschließlich privat und
|
||||
nichtkommerziell eingesetzt.
|
||||
|
||||
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
|
||||
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
|
||||
|
||||
```text
|
||||
/athena music
|
||||
/athena stems
|
||||
/athena voice
|
||||
/athena llm
|
||||
/athena status
|
||||
```
|
||||
@@ -64,13 +76,15 @@ Die HTTP-Schnittstelle verwendet authentifizierte Requests:
|
||||
GET /mode
|
||||
POST /mode {"mode":"music"}
|
||||
POST /mode {"mode":"separation"}
|
||||
POST /mode {"mode":"voice"}
|
||||
POST /mode {"mode":"llm"}
|
||||
```
|
||||
|
||||
Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in
|
||||
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
|
||||
`com.mike-ai.music-worker=acestep` beziehungsweise
|
||||
`com.mike-ai.stem-separator=bs-roformer` markierten Container; freie
|
||||
`com.mike-ai.stem-separator=bs-roformer` oder
|
||||
`com.mike-ai.voice-worker=vevo2` markierten Container; freie
|
||||
Container- oder Docker-Befehle werden nicht entgegengenommen.
|
||||
|
||||
## Wiederanlauf
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Register getesteter Modelle
|
||||
|
||||
Stand: 8. September 2026
|
||||
Stand: 9. September 2026
|
||||
|
||||
Dieses Dokument ist die zentrale Sperrliste gegen doppelte Modelltests. Vor
|
||||
jedem Download müssen Repository, Dateiname, Basismodell, Fine-Tune und
|
||||
@@ -56,6 +56,7 @@ Titelgenerierung und Kontextkompression in Hermes.
|
||||
| 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** |
|
||||
| 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** mit Piper-Fallback |
|
||||
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
|
||||
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Produktions-HTTP-API mit offizieller 8,6-s-Sprachprobe: Warmstart 12,216 s, Wandlung 2,342 s, Spitzen-VRAM 5.605,6 MiB; gültiges 24-kHz-WAV mit 412.878 Byte; Profilanlage, Download und automatische Job-Bereinigung geprüft | **technischer Ende-zu-Ende-Test bestanden**, privater Voice-Studio-Betatest; Gewichte CC BY-NC-ND 4.0 und daher nicht kommerziell einsetzen |
|
||||
|
||||
## Musikgenerierung
|
||||
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
FROM mike-ai/bs-roformer-separator:0.47.0
|
||||
|
||||
ARG AMPHION_COMMIT=26f6883110181f1dbfe95c70a7c7dbaf4de5f42a
|
||||
|
||||
USER root
|
||||
RUN apt-get update \
|
||||
&& apt-get install -y --no-install-recommends git espeak-ng \
|
||||
&& rm -rf /var/lib/apt/lists/*
|
||||
|
||||
RUN git clone https://github.com/open-mmlab/Amphion.git /opt/amphion \
|
||||
&& cd /opt/amphion \
|
||||
&& git checkout "${AMPHION_COMMIT}" \
|
||||
&& rm -rf .git
|
||||
|
||||
# Vevo2 is tested on Athena with the CUDA/PyTorch stack inherited from the
|
||||
# existing audio worker. Do not install Amphion's historical torch 2.0/cu118
|
||||
# pins: Blackwell requires the newer cu128 runtime already present here.
|
||||
RUN python -m pip install --no-cache-dir \
|
||||
accelerate==1.10.1 \
|
||||
diffusers==0.35.1 \
|
||||
einops==0.8.1 \
|
||||
easydict==1.13 \
|
||||
g2p_en==2.1.0 \
|
||||
humanfriendly==10.0 \
|
||||
huggingface-hub==0.34.4 \
|
||||
hydra-core==1.3.2 \
|
||||
inflect==7.5.0 \
|
||||
ipython==9.5.0 \
|
||||
json5==0.12.1 \
|
||||
librosa==0.11.0 \
|
||||
loguru==0.7.3 \
|
||||
matplotlib==3.10.6 \
|
||||
munch==4.0.0 \
|
||||
omegaconf==2.3.0 \
|
||||
openai-whisper==20250625 \
|
||||
phonemizer==3.3.0 \
|
||||
python-multipart==0.0.20 \
|
||||
praat-parselmouth==0.4.6 \
|
||||
pypinyin==0.55.0 \
|
||||
pyworld==0.3.5 \
|
||||
ruamel.yaml==0.18.15 \
|
||||
safetensors==0.6.2 \
|
||||
tabulate==0.9.0 \
|
||||
tgt==1.5 \
|
||||
torchcrepe==0.0.24 \
|
||||
transformers==4.56.1 \
|
||||
typeguard==4.4.4 \
|
||||
unidecode==1.4.0 \
|
||||
vector-quantize-pytorch==1.12.5 \
|
||||
vocos==0.1.0
|
||||
|
||||
WORKDIR /opt/amphion
|
||||
ENV PYTHONPATH=/opt/amphion \
|
||||
HF_HOME=/models/huggingface \
|
||||
PYTHONUNBUFFERED=1
|
||||
|
||||
COPY run_fm_test.py /usr/local/bin/run_fm_test.py
|
||||
COPY app.py /app/app.py
|
||||
COPY index.html /app/index.html
|
||||
|
||||
EXPOSE 8008
|
||||
HEALTHCHECK --interval=5s --timeout=3s --start-period=90s --retries=3 \
|
||||
CMD curl -fsS http://127.0.0.1:8008/health || exit 1
|
||||
ENTRYPOINT ["python", "/app/app.py"]
|
||||
@@ -0,0 +1,30 @@
|
||||
# Vevo2 Voice Conversion on Athena
|
||||
|
||||
This directory contains Athena's private Vevo2 voice-conversion studio.
|
||||
|
||||
- Code: `open-mmlab/Amphion` commit
|
||||
`26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` (MIT)
|
||||
- Weights: `RMSnow/Vevo2` (CC BY-NC-ND 4.0)
|
||||
- Runtime: PyTorch 2.7.1 + CUDA 12.8 inherited from Athena's tested audio
|
||||
separator image; the historical Amphion torch 2.0/cu118 pins are not used.
|
||||
- Storage: `/data/voice/vevo2`; removing that directory and the test image
|
||||
removes all downloaded artifacts.
|
||||
- Private UI: `http://192.168.1.212:8008` through WireGuard only.
|
||||
- Output: uncompressed mono WAV, 24 kHz.
|
||||
|
||||
The 9 September technical gate converted the official 8.6-second speech sample
|
||||
through the production HTTP API in 2.342 seconds. A warm service start loaded
|
||||
the model in 12.216 seconds, and peak CUDA allocation during conversion was
|
||||
5605.6 MiB. The returned 24-kHz WAV was 412878 bytes and 8.6 seconds long. The
|
||||
service stores named reference voices, accepts a source clip, and returns a
|
||||
transient WAV download. Jobs and generated outputs are removed after delivery.
|
||||
It is deliberately not exposed on the university interface.
|
||||
|
||||
The installed Compose project is named `mike-ai-voice`. Rebuild or recreate it
|
||||
with an explicit `VOICE_GPU_UUID` so Docker keeps the service attached to the
|
||||
RTX 5080. The profile controller starts and stops the existing container; it
|
||||
does not rebuild it during a mode switch.
|
||||
|
||||
The weights are licensed CC BY-NC-ND 4.0. This deployment is for Mike's
|
||||
private, non-commercial use only. Do not use or expose it as a public or
|
||||
commercial voice-cloning service.
|
||||
@@ -0,0 +1,202 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Small local-only Vevo2 voice-conversion studio for Athena."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import threading
|
||||
import time
|
||||
import uuid
|
||||
from contextlib import asynccontextmanager
|
||||
from pathlib import Path
|
||||
|
||||
import torch
|
||||
from fastapi import FastAPI, File, Form, HTTPException, UploadFile
|
||||
from fastapi.responses import FileResponse, HTMLResponse
|
||||
from starlette.background import BackgroundTask
|
||||
|
||||
import models.svc.vevo2.infer_vevo2_fm as vevo
|
||||
|
||||
|
||||
DATA_DIR = Path(os.environ.get("VOICE_DATA_DIR", "/data"))
|
||||
PROFILE_DIR = DATA_DIR / "profiles"
|
||||
JOB_DIR = DATA_DIR / "jobs"
|
||||
INDEX = Path("/app/index.html")
|
||||
MAX_UPLOAD_BYTES = int(os.environ.get("MAX_UPLOAD_BYTES", str(100 * 1024 * 1024)))
|
||||
MODEL_LOCK = threading.Lock()
|
||||
PIPELINE = None
|
||||
MODEL_LOAD_SECONDS: float | None = None
|
||||
|
||||
|
||||
def safe_name(value: str) -> str:
|
||||
value = re.sub(r"[^A-Za-z0-9_. -]+", "-", value.strip())
|
||||
value = re.sub(r"\s+", "-", value).strip("-.")
|
||||
return value[:64] or "voice"
|
||||
|
||||
|
||||
def load_pipeline() -> None:
|
||||
global PIPELINE, MODEL_LOAD_SECONDS
|
||||
if PIPELINE is not None:
|
||||
return
|
||||
with MODEL_LOCK:
|
||||
if PIPELINE is not None:
|
||||
return
|
||||
started = time.monotonic()
|
||||
PIPELINE = vevo.load_inference_pipeline()
|
||||
vevo.inference_pipeline = PIPELINE
|
||||
MODEL_LOAD_SECONDS = time.monotonic() - started
|
||||
|
||||
|
||||
def to_wav(source: Path, target: Path) -> None:
|
||||
completed = subprocess.run(
|
||||
["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
|
||||
"-ac", "1", "-ar", "24000", "-c:a", "pcm_s16le", str(target)],
|
||||
capture_output=True,
|
||||
text=True,
|
||||
timeout=180,
|
||||
check=False,
|
||||
)
|
||||
if completed.returncode:
|
||||
raise ValueError(completed.stderr.strip() or "Audio konnte nicht gelesen werden")
|
||||
|
||||
|
||||
async def save_upload(upload: UploadFile, target: Path) -> None:
|
||||
size = 0
|
||||
with target.open("wb") as handle:
|
||||
while chunk := await upload.read(1024 * 1024):
|
||||
size += len(chunk)
|
||||
if size > MAX_UPLOAD_BYTES:
|
||||
raise HTTPException(413, "Audiodatei ist zu groß")
|
||||
handle.write(chunk)
|
||||
|
||||
|
||||
def profile_path(name: str) -> Path:
|
||||
target = PROFILE_DIR / f"{safe_name(name)}.wav"
|
||||
if not target.is_file():
|
||||
raise HTTPException(404, "Referenzstimme wurde nicht gefunden")
|
||||
return target
|
||||
|
||||
|
||||
def run_conversion(source: Path, reference: Path, output: Path, pitch_shift: bool) -> tuple[float, float]:
|
||||
"""Run the GPU-bound conversion off the API event loop."""
|
||||
load_pipeline()
|
||||
with MODEL_LOCK:
|
||||
torch.cuda.reset_peak_memory_stats()
|
||||
started = time.monotonic()
|
||||
vevo.vevo2_fm(str(source), str(reference), str(output), shifted_src=pitch_shift)
|
||||
elapsed = time.monotonic() - started
|
||||
peak = torch.cuda.max_memory_allocated() / 1048576
|
||||
return elapsed, peak
|
||||
|
||||
|
||||
@asynccontextmanager
|
||||
async def lifespan(_app: FastAPI):
|
||||
PROFILE_DIR.mkdir(parents=True, exist_ok=True)
|
||||
JOB_DIR.mkdir(parents=True, exist_ok=True)
|
||||
load_pipeline()
|
||||
yield
|
||||
|
||||
|
||||
app = FastAPI(title="Athena Voice Studio", lifespan=lifespan)
|
||||
|
||||
|
||||
@app.get("/", response_class=HTMLResponse)
|
||||
def index() -> str:
|
||||
return INDEX.read_text(encoding="utf-8")
|
||||
|
||||
|
||||
@app.get("/health")
|
||||
def health() -> dict:
|
||||
return {
|
||||
"status": "ok" if PIPELINE is not None else "starting",
|
||||
"model": "RMSnow/Vevo2",
|
||||
"sample_rate": 24000,
|
||||
"model_load_seconds": MODEL_LOAD_SECONDS,
|
||||
}
|
||||
|
||||
|
||||
@app.get("/api/profiles")
|
||||
def profiles() -> dict:
|
||||
return {"profiles": [path.stem for path in sorted(PROFILE_DIR.glob("*.wav"))]}
|
||||
|
||||
|
||||
@app.post("/api/profiles")
|
||||
async def create_profile(
|
||||
name: str = Form(...),
|
||||
consent: bool = Form(False),
|
||||
audio: UploadFile = File(...),
|
||||
) -> dict:
|
||||
if not consent:
|
||||
raise HTTPException(400, "Bestätige, dass du die Stimme verwenden darfst")
|
||||
clean = safe_name(name)
|
||||
job = JOB_DIR / f"profile-{uuid.uuid4().hex}"
|
||||
job.mkdir(parents=True)
|
||||
raw = job / f"upload-{safe_name(audio.filename or 'reference.audio')}"
|
||||
try:
|
||||
await save_upload(audio, raw)
|
||||
target = PROFILE_DIR / f"{clean}.wav"
|
||||
temporary = job / "reference.wav"
|
||||
to_wav(raw, temporary)
|
||||
os.replace(temporary, target)
|
||||
return {"status": "ok", "profile": clean}
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise HTTPException(400, str(exc)) from exc
|
||||
finally:
|
||||
shutil.rmtree(job, ignore_errors=True)
|
||||
|
||||
|
||||
@app.delete("/api/profiles/{name}")
|
||||
def delete_profile(name: str) -> dict:
|
||||
target = profile_path(name)
|
||||
target.unlink()
|
||||
return {"status": "ok", "profile": target.stem}
|
||||
|
||||
|
||||
@app.post("/api/convert")
|
||||
async def convert(
|
||||
source: UploadFile = File(...),
|
||||
profile: str = Form(...),
|
||||
pitch_shift: bool = Form(True),
|
||||
) -> FileResponse:
|
||||
reference = profile_path(profile)
|
||||
job_id = uuid.uuid4().hex
|
||||
job = JOB_DIR / f"convert-{job_id}"
|
||||
job.mkdir(parents=True)
|
||||
raw = job / f"upload-{safe_name(source.filename or 'source.audio')}"
|
||||
source_wav = job / "source.wav"
|
||||
output = JOB_DIR / f"voice-{job_id}.wav"
|
||||
try:
|
||||
await save_upload(source, raw)
|
||||
to_wav(raw, source_wav)
|
||||
elapsed, peak = await asyncio.to_thread(
|
||||
run_conversion, source_wav, reference, output, pitch_shift
|
||||
)
|
||||
return FileResponse(
|
||||
output,
|
||||
media_type="audio/wav",
|
||||
filename=f"{safe_name(profile)}-{job_id[:8]}.wav",
|
||||
headers={
|
||||
"X-Conversion-Seconds": f"{elapsed:.3f}",
|
||||
"X-Peak-VRAM-MiB": f"{peak:.1f}",
|
||||
},
|
||||
background=BackgroundTask(output.unlink, missing_ok=True),
|
||||
)
|
||||
except HTTPException:
|
||||
raise
|
||||
except Exception as exc:
|
||||
output.unlink(missing_ok=True)
|
||||
raise HTTPException(500, f"Stimmenwandlung fehlgeschlagen: {exc}") from exc
|
||||
finally:
|
||||
shutil.rmtree(job, ignore_errors=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
import uvicorn
|
||||
|
||||
uvicorn.run(app, host="0.0.0.0", port=8008)
|
||||
@@ -0,0 +1,39 @@
|
||||
services:
|
||||
voice-studio:
|
||||
build: .
|
||||
image: mike-ai/vevo2-voice-studio:0.1
|
||||
container_name: mike-ai-voice-studio
|
||||
restart: "no"
|
||||
labels:
|
||||
com.mike-ai.voice-worker: vevo2
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
VOICE_DATA_DIR: /data
|
||||
ports:
|
||||
- "127.0.0.1:8008:8008"
|
||||
volumes:
|
||||
- /data/voice/vevo2/huggingface/hub/Vevo2:/opt/amphion/ckpts/Vevo2
|
||||
- /data/voice/vevo2/huggingface:/models/huggingface
|
||||
- /data/voice/vevo2/whisper:/root/.cache/whisper
|
||||
- /data/voice/studio:/data
|
||||
healthcheck:
|
||||
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/health | grep -q '\"status\":\"ok\"'"]
|
||||
interval: 5s
|
||||
timeout: 3s
|
||||
start_period: 600s
|
||||
retries: 3
|
||||
deploy:
|
||||
resources:
|
||||
reservations:
|
||||
devices:
|
||||
- driver: nvidia
|
||||
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
|
||||
capabilities: [gpu]
|
||||
networks:
|
||||
- frontend
|
||||
|
||||
networks:
|
||||
frontend:
|
||||
name: mike-ai_frontend
|
||||
external: true
|
||||
@@ -0,0 +1,10 @@
|
||||
<!doctype html>
|
||||
<html lang="de"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
|
||||
<title>Athena Voice Studio</title><style>
|
||||
:root{color-scheme:dark;--bg:#07101a;--panel:#0d1926;--line:#20374d;--text:#edf6ff;--muted:#91a7bb;--cyan:#45d7ff;--green:#66e3a4;--red:#ff7185}*{box-sizing:border-box}body{margin:0;background:radial-gradient(circle at top left,#12283d,var(--bg) 45%);color:var(--text);font:16px system-ui,sans-serif}main{max-width:960px;margin:auto;padding:32px 18px}h1{margin:.2rem 0}p{color:var(--muted)}.grid{display:grid;grid-template-columns:1fr 1fr;gap:16px}.card{background:var(--panel);border:1px solid var(--line);border-radius:16px;padding:20px}label{display:block;margin:13px 0 6px;color:var(--muted)}input,select,button{width:100%;border:1px solid var(--line);border-radius:9px;padding:11px;background:#07111d;color:var(--text);font:inherit}button{margin-top:15px;background:#10334a;border-color:var(--cyan);cursor:pointer}button:disabled{opacity:.5;cursor:wait}.row{display:flex;align-items:center;gap:9px}.row input{width:auto}.status{min-height:24px;margin-top:12px;color:var(--green)}.error{color:var(--red)}audio{width:100%;margin-top:12px}@media(max-width:700px){.grid{grid-template-columns:1fr}}</style></head>
|
||||
<body><main><div style="color:var(--cyan);letter-spacing:.14em;text-transform:uppercase;font-size:12px">Mike AI · lokal auf Athena</div><h1>Voice Studio</h1><p>Eine Stimme als Referenz speichern und eine vorhandene Sprach- oder Gesangsaufnahme in diese Stimme übertragen. Ausgabe: unkomprimiertes WAV mit 24 kHz.</p>
|
||||
<div class="grid"><section class="card"><h2>1 · Referenzstimme</h2><form id="profileForm"><label>Name</label><input name="name" required placeholder="z. B. Meine Stimme"><label>Saubere Referenzaufnahme</label><input name="audio" type="file" accept="audio/*" required><label class="row"><input name="consent" type="checkbox" value="true" required><span>Ich darf diese Stimme verwenden.</span></label><button>Stimme speichern</button></form><div class="status" id="profileStatus"></div></section>
|
||||
<section class="card"><h2>2 · Stimme übertragen</h2><form id="convertForm"><label>Zielstimme</label><select name="profile" id="profiles" required></select><label>Quellaufnahme</label><input name="source" type="file" accept="audio/*" required><label class="row"><input name="pitch_shift" type="checkbox" value="true" checked><span>Tonlage automatisch an Zielstimme anpassen</span></label><button>WAV erzeugen</button></form><div class="status" id="convertStatus"></div><audio id="player" controls hidden></audio><a id="download" hidden>WAV herunterladen</a></section></div>
|
||||
<p>Hinweis: Nur privat und mit Zustimmung verwenden. Das Vevo2-Gewicht ist nicht für kommerzielle Nutzung lizenziert.</p></main><script>
|
||||
const $=id=>document.getElementById(id);async function loadProfiles(){let r=await fetch('/api/profiles'),d=await r.json();$('profiles').innerHTML=d.profiles.length?d.profiles.map(x=>`<option>${x}</option>`).join(''):'<option value="">Noch keine Stimme gespeichert</option>'}function state(id,text,error=false){let e=$(id);e.textContent=text;e.classList.toggle('error',error)}$('profileForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('profileStatus','Referenz wird geprüft und gespeichert …');try{let r=await fetch('/api/profiles',{method:'POST',body:new FormData(e.target)}),d=await r.json();if(!r.ok)throw Error(d.detail||`HTTP ${r.status}`);state('profileStatus',`„${d.profile}“ ist gespeichert.`);await loadProfiles()}catch(x){state('profileStatus',x.message,true)}finally{b.disabled=false}};$('convertForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('convertStatus','Modell arbeitet …');$('player').hidden=true;$('download').hidden=true;try{let f=new FormData(e.target);if(!f.has('pitch_shift'))f.append('pitch_shift','false');let r=await fetch('/api/convert',{method:'POST',body:f});if(!r.ok){let d=await r.json();throw Error(d.detail||`HTTP ${r.status}`)}let blob=await r.blob(),url=URL.createObjectURL(blob);$('player').src=url;$('player').hidden=false;$('download').href=url;$('download').download='voice-conversion.wav';$('download').hidden=false;state('convertStatus',`Fertig in ${r.headers.get('X-Conversion-Seconds')||'?'} s · Spitzen-VRAM ${r.headers.get('X-Peak-VRAM-MiB')||'?'} MiB`)}catch(x){state('convertStatus',x.message,true)}finally{b.disabled=false}};loadProfiles();
|
||||
</script></body></html>
|
||||
@@ -0,0 +1,47 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import os
|
||||
import time
|
||||
|
||||
import torch
|
||||
|
||||
import models.svc.vevo2.infer_vevo2_fm as vevo
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("--source", required=True)
|
||||
parser.add_argument("--reference", required=True)
|
||||
parser.add_argument("--output", required=True)
|
||||
parser.add_argument("--no-pitch-shift", action="store_true")
|
||||
args = parser.parse_args()
|
||||
|
||||
os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
|
||||
started = time.monotonic()
|
||||
vevo.inference_pipeline = vevo.load_inference_pipeline()
|
||||
loaded = time.monotonic()
|
||||
vevo.vevo2_fm(
|
||||
args.source,
|
||||
args.reference,
|
||||
args.output,
|
||||
shifted_src=not args.no_pitch_shift,
|
||||
)
|
||||
finished = time.monotonic()
|
||||
print(
|
||||
{
|
||||
"model_load_seconds": round(loaded - started, 3),
|
||||
"conversion_seconds": round(finished - loaded, 3),
|
||||
"total_seconds": round(finished - started, 3),
|
||||
"peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
|
||||
"output": args.output,
|
||||
},
|
||||
flush=True,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -32,6 +32,8 @@ MUSIC_LABEL_KEY = "com.mike-ai.music-worker"
|
||||
MUSIC_WORKER = os.environ.get("MUSIC_WORKER", "").strip()
|
||||
SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator"
|
||||
SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
|
||||
VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
|
||||
VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
|
||||
LOCK = threading.Lock()
|
||||
log = logging.getLogger("profile-controller")
|
||||
|
||||
@@ -122,6 +124,17 @@ def separator_container() -> dict:
|
||||
return matches[0]
|
||||
|
||||
|
||||
def voice_container() -> dict:
|
||||
if not VOICE_WORKER:
|
||||
raise RuntimeError("voice worker is not configured")
|
||||
matches = [item for item in labelled_containers(VOICE_LABEL_KEY)
|
||||
if item.get("Labels", {}).get(VOICE_LABEL_KEY) == VOICE_WORKER]
|
||||
if len(matches) != 1:
|
||||
raise RuntimeError(
|
||||
f"expected exactly one voice worker {VOICE_WORKER!r}, found {len(matches)}")
|
||||
return matches[0]
|
||||
|
||||
|
||||
def stop_music_if_configured() -> None:
|
||||
if MUSIC_WORKER:
|
||||
stop_container(music_container(), timeout=30)
|
||||
@@ -132,6 +145,11 @@ def stop_separator_if_configured() -> None:
|
||||
stop_container(separator_container(), timeout=30)
|
||||
|
||||
|
||||
def stop_voice_if_configured() -> None:
|
||||
if VOICE_WORKER:
|
||||
stop_container(voice_container(), timeout=30)
|
||||
|
||||
|
||||
def stop_container(item: dict, timeout: int = 120) -> None:
|
||||
if item.get("State") != "running":
|
||||
return
|
||||
@@ -171,6 +189,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
|
||||
stop_container(tts_container(), timeout=30)
|
||||
stop_music_if_configured()
|
||||
stop_separator_if_configured()
|
||||
stop_voice_if_configured()
|
||||
for other in image_containers():
|
||||
if other["Id"] != item["Id"]:
|
||||
stop_container(other, timeout=20)
|
||||
@@ -197,6 +216,7 @@ def set_music_worker(running: bool) -> dict:
|
||||
stop_container(worker, timeout=20)
|
||||
stop_container(tts_container(), timeout=30)
|
||||
stop_separator_if_configured()
|
||||
stop_voice_if_configured()
|
||||
start_container(item)
|
||||
else:
|
||||
stop_container(item, timeout=30)
|
||||
@@ -215,6 +235,7 @@ def set_separator_worker(running: bool) -> dict:
|
||||
stop_container(worker, timeout=20)
|
||||
stop_container(tts_container(), timeout=30)
|
||||
stop_music_if_configured()
|
||||
stop_voice_if_configured()
|
||||
start_container(item)
|
||||
else:
|
||||
stop_container(item, timeout=30)
|
||||
@@ -222,6 +243,25 @@ def set_separator_worker(running: bool) -> dict:
|
||||
"state": "running" if running else "stopped"}
|
||||
|
||||
|
||||
def set_voice_worker(running: bool) -> dict:
|
||||
"""Start Vevo2 exclusively, or stop it before LLM restoration."""
|
||||
with LOCK:
|
||||
item = voice_container()
|
||||
if running:
|
||||
for profile_item in containers().values():
|
||||
stop_container(profile_item)
|
||||
for worker in image_containers():
|
||||
stop_container(worker, timeout=20)
|
||||
stop_container(tts_container(), timeout=30)
|
||||
stop_music_if_configured()
|
||||
stop_separator_if_configured()
|
||||
start_container(item)
|
||||
else:
|
||||
stop_container(item, timeout=30)
|
||||
return {"voice_worker": VOICE_WORKER,
|
||||
"state": "running" if running else "stopped"}
|
||||
|
||||
|
||||
def active_profile(items: dict[str, dict] | None = None) -> str | None:
|
||||
items = items or containers()
|
||||
active = [name for name, item in items.items() if item.get("State") == "running"]
|
||||
@@ -239,6 +279,7 @@ def activate(profile: str) -> dict:
|
||||
stop_container(worker)
|
||||
stop_music_if_configured()
|
||||
stop_separator_if_configured()
|
||||
stop_voice_if_configured()
|
||||
start_container(tts_container())
|
||||
items = containers()
|
||||
missing = [name for name in ALLOWED if name not in items]
|
||||
@@ -317,11 +358,20 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"unhealthy" if "(unhealthy)" in separator_status else
|
||||
"starting" if separator.get("State") == "running" else
|
||||
"stopped")
|
||||
voice = voice_container() if VOICE_WORKER else {}
|
||||
voice_status = voice.get("Status", "")
|
||||
voice_health = ("disabled" if not VOICE_WORKER else
|
||||
"healthy" if "(healthy)" in voice_status else
|
||||
"unhealthy" if "(unhealthy)" in voice_status else
|
||||
"starting" if voice.get("State") == "running" else
|
||||
"stopped")
|
||||
self.reply(200, {"active_profile": active_profile(items),
|
||||
"music_worker": music.get("State", "disabled"),
|
||||
"music_health": music_health,
|
||||
"separator_worker": separator.get("State", "disabled"),
|
||||
"separator_health": separator_health,
|
||||
"voice_worker": voice.get("State", "disabled"),
|
||||
"voice_health": voice_health,
|
||||
"profiles": {name: items.get(name, {}).get(
|
||||
"State", "missing") for name in ALLOWED}})
|
||||
except Exception as exc:
|
||||
@@ -353,6 +403,13 @@ class Handler(BaseHTTPRequestHandler):
|
||||
log.exception("stem separator transition failed")
|
||||
self.reply(503, {"error": str(exc)})
|
||||
return
|
||||
if self.path in {"/workers/voice/start", "/workers/voice/stop"}:
|
||||
try:
|
||||
self.reply(200, set_voice_worker(self.path.endswith("/start")))
|
||||
except Exception as exc:
|
||||
log.exception("voice worker transition failed")
|
||||
self.reply(503, {"error": str(exc)})
|
||||
return
|
||||
worker_paths = {
|
||||
"/workers/image/start": (IMAGE_WORKER, True),
|
||||
"/workers/image/stop": (IMAGE_WORKER, False),
|
||||
|
||||
@@ -81,6 +81,7 @@ start_proxy 8099 llama-dashboard:8099
|
||||
start_proxy 7861 music-ui:3000
|
||||
start_proxy 7862 music-worker:7860
|
||||
start_proxy 8007 stem-separator:8080
|
||||
start_proxy 8008 voice-studio:8008
|
||||
start_proxy 8202 mcp-athena-operator:8000
|
||||
start_proxy 9443 portainer:9443
|
||||
|
||||
|
||||
@@ -28,6 +28,7 @@ MUSIC_ORIGINAL_UI_URL = os.getenv(
|
||||
"MUSIC_ORIGINAL_UI_URL", "http://192.168.1.212:7862/"
|
||||
)
|
||||
SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
|
||||
VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
|
||||
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
|
||||
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
|
||||
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
|
||||
@@ -277,7 +278,7 @@ def router_status() -> tuple[dict[str, Any], str | None]:
|
||||
|
||||
|
||||
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
|
||||
if mode not in {"llm", "music", "separation"}:
|
||||
if mode not in {"llm", "music", "separation", "voice"}:
|
||||
return 400, {"error": "invalid mode"}
|
||||
headers = {"Accept": "application/json", "Content-Type": "application/json"}
|
||||
if ROUTER_API_KEY:
|
||||
@@ -635,7 +636,7 @@ HTML = r'''<!doctype html>
|
||||
</style></head><body><main>
|
||||
<div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div>
|
||||
<section class="grid">
|
||||
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span></div></div></article>
|
||||
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><button id="voiceMode" onclick="setMode('voice')">Voice Studio</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span><span id="voiceOpen" hidden><a class="stable" href="__VOICE_UI_URL__" target="_blank" rel="noopener">Voice Studio öffnen</a></span></div></div></article>
|
||||
<article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article>
|
||||
<article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article>
|
||||
<article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article>
|
||||
@@ -674,12 +675,13 @@ const $=id=>document.getElementById(id); const pct=n=>n==null?'–':`${n.toFixed
|
||||
function gpuCard(g){let total=g.memory_total_mib||0,used=g.memory_used_mib||0,p=total?used/total*100:0,load=Math.max(0,Math.min(100,g.gpu_percent||0));return `<article class="card span6"><div class="gpu-title"><div><div class="label">GPU ${g.index}</div><div class="value">${g.name}</div></div><span class="badge">${g.pstate||'–'}</span></div><div class="metrics"><div class="metric"><b>${pct(g.gpu_percent)}</b><span>GPU-Kern</span></div><div class="metric"><b>${(used/1024).toFixed(1)} / ${(total/1024).toFixed(1)} GiB</b><span>VRAM</span></div><div class="metric"><b>${g.temperature_c??'–'} °C</b><span>Temperatur</span></div><div class="metric"><b>${g.power_w??'–'} / ${g.power_limit_w??'–'} W</b><span>Leistung</span></div><div class="metric"><b>${g.graphics_clock_mhz??'–'} MHz</b><span>Grafiktakt</span></div><div class="metric"><b>${g.memory_clock_mhz??'–'} MHz</b><span>Speichertakt</span></div><div class="metric"><b>${pct(g.memory_controller_percent)}</b><span>Memory Controller</span></div><div class="metric"><b>${pct(g.fan_percent)}</b><span>Lüfter</span></div></div><div class="bar-label"><span>GPU-Auslastung</span><span>${load.toFixed(1)} %</span></div><div class="bar"><div class="fill gpu-load" style="width:${load}%"></div></div><div class="bar-label"><span>VRAM-Belegung</span><span>${p.toFixed(1)} %</span></div><div class="bar"><div class="fill" style="width:${Math.min(100,p)}%"></div></div><div class="sub">${(g.memory_free_mib/1024).toFixed(1)} GiB VRAM frei</div></article>`}
|
||||
const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":"Bildmodell wird geladen","generating":"Bild wird generiert","unloading-image":"Bildmodell wird entladen","restoring-qwen":"Qwen wird wiederhergestellt"}[p]||p||'bereit');
|
||||
let modeBusy=false;
|
||||
async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
|
||||
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
|
||||
async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
|
||||
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
|
||||
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''.replace(
|
||||
"__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL
|
||||
).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL
|
||||
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL)
|
||||
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL
|
||||
).replace("__VOICE_UI_URL__", VOICE_UI_URL)
|
||||
|
||||
|
||||
FULL_JS = r'''
|
||||
|
||||
+68
-18
@@ -110,6 +110,7 @@ ENABLE_MUSIC_MODE = os.environ.get(
|
||||
"ENABLE_MUSIC_MODE", "false").lower() in {"1", "true", "yes"}
|
||||
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
|
||||
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
|
||||
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
|
||||
|
||||
# Optional worker APIs. The clean Docker baseline deliberately ships only
|
||||
# text/multimodal chat; absent workers must fail explicitly instead of trying
|
||||
@@ -361,6 +362,27 @@ def _separator_worker_health() -> str:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _voice_worker_state() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"voice_worker", "missing"))
|
||||
except Exception as exc:
|
||||
log.warning("Voice-Worker-Status nicht verfügbar: %s", exc)
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _voice_worker_health() -> str:
|
||||
if not PROFILE_CONTROL_URL:
|
||||
return "unsupported"
|
||||
try:
|
||||
return str(_profile_controller_request("GET", "/status").get(
|
||||
"voice_health", "unknown"))
|
||||
except Exception:
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _wait_music_ready() -> None:
|
||||
deadline = time.monotonic() + MUSIC_START_TIMEOUT
|
||||
while time.monotonic() < deadline:
|
||||
@@ -389,17 +411,40 @@ def _wait_separator_ready() -> None:
|
||||
f"BS-RoFormer nach {SEPARATOR_START_TIMEOUT:.0f} s nicht bereit")
|
||||
|
||||
|
||||
def _wait_voice_ready() -> None:
|
||||
deadline = time.monotonic() + VOICE_START_TIMEOUT
|
||||
while time.monotonic() < deadline:
|
||||
status = _profile_controller_request("GET", "/status")
|
||||
if (status.get("voice_worker") == "running"
|
||||
and status.get("voice_health") == "healthy"):
|
||||
return
|
||||
if status.get("voice_health") == "unhealthy":
|
||||
raise RuntimeError("Vevo2-Container ist unhealthy")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
raise RuntimeError(
|
||||
f"Vevo2 nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
|
||||
|
||||
|
||||
def _special_worker(mode: str) -> tuple[str, str, callable]:
|
||||
if mode == "music":
|
||||
return "/workers/music/start", _music_worker_state(), _wait_music_ready
|
||||
if mode == "separation":
|
||||
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
|
||||
if mode == "voice":
|
||||
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
|
||||
raise ValueError(f"unbekannter Spezialmodus: {mode}")
|
||||
|
||||
|
||||
def set_operating_mode(mode: str) -> dict:
|
||||
"""Atomarer Wechsel zwischen LLM, ACE-Step und Stem-Separation."""
|
||||
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
|
||||
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
||||
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
||||
if mode not in {"llm", "music", "separation"}:
|
||||
raise ValueError("Modus muss 'llm', 'music' oder 'separation' sein")
|
||||
if mode not in {"llm", "music", "separation", "voice"}:
|
||||
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
|
||||
with STATE.lock:
|
||||
STATE.mode_error = None
|
||||
if mode in {"music", "separation"}:
|
||||
worker_state = (_music_worker_state() if mode == "music"
|
||||
else _separator_worker_state())
|
||||
if mode in {"music", "separation", "voice"}:
|
||||
path, worker_state, wait_ready = _special_worker(mode)
|
||||
if STATE.mode == mode and worker_state == "running":
|
||||
return {"status": "ok", "mode": mode, "changed": False}
|
||||
profile = current_profile()
|
||||
@@ -417,10 +462,8 @@ def set_operating_mode(mode: str) -> dict:
|
||||
last_profile=return_profile,
|
||||
phase=f"starting-{mode}")
|
||||
_wait_chats_drained()
|
||||
path = ("/workers/music/start" if mode == "music"
|
||||
else "/workers/separator/start")
|
||||
_profile_controller_request("POST", path)
|
||||
_wait_music_ready() if mode == "music" else _wait_separator_ready()
|
||||
wait_ready()
|
||||
STATE.mode = mode
|
||||
STATE.mode_phase = "ready"
|
||||
RUNTIME.save(mode=mode, return_profile=return_profile,
|
||||
@@ -441,6 +484,7 @@ def set_operating_mode(mode: str) -> dict:
|
||||
try:
|
||||
_profile_controller_request("POST", "/workers/music/stop")
|
||||
_profile_controller_request("POST", "/workers/separator/stop")
|
||||
_profile_controller_request("POST", "/workers/voice/stop")
|
||||
_restore_qwen(profile)
|
||||
STATE.mode = "llm"
|
||||
STATE.mode_phase = "ready"
|
||||
@@ -459,8 +503,8 @@ def schedule_operating_mode(mode: str) -> tuple[bool, str]:
|
||||
"""Start a transition in the background so chat/UI acknowledgement is instant."""
|
||||
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
||||
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
||||
if mode not in {"llm", "music", "separation"}:
|
||||
raise ValueError("Modus muss 'llm', 'music' oder 'separation' sein")
|
||||
if mode not in {"llm", "music", "separation", "voice"}:
|
||||
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
|
||||
with STATE.lock:
|
||||
if STATE.mode_phase not in {"ready", "error"}:
|
||||
return False, STATE.mode_phase
|
||||
@@ -496,6 +540,7 @@ def _control_command(data: dict, path: str) -> str | None:
|
||||
command = text.strip().casefold()
|
||||
return command if command in {"/athena music", "/athena stems",
|
||||
"/athena separation", "/athena llm",
|
||||
"/athena voice",
|
||||
"/athena status"} else None
|
||||
|
||||
|
||||
@@ -1971,6 +2016,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"music_health": _music_worker_health(),
|
||||
"separator_worker": _separator_worker_state(),
|
||||
"separator_health": _separator_worker_health(),
|
||||
"voice_worker": _voice_worker_state(),
|
||||
"voice_health": _voice_worker_health(),
|
||||
"return_profile": state.get("return_profile"),
|
||||
"last_error": STATE.mode_error,
|
||||
"enabled": ENABLE_MUSIC_MODE,
|
||||
@@ -1980,8 +2027,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
try:
|
||||
data = json.loads(self._read_body() or b"{}")
|
||||
mode = data.get("mode") if isinstance(data, dict) else None
|
||||
if mode not in {"llm", "music", "separation"}:
|
||||
raise ValueError("Feld 'mode' muss 'llm', 'music' oder 'separation' sein")
|
||||
if mode not in {"llm", "music", "separation", "voice"}:
|
||||
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation' oder 'voice' sein")
|
||||
started, phase = schedule_operating_mode(mode)
|
||||
self._send_json(202 if started else 200, {
|
||||
"status": "accepted" if started else "ok",
|
||||
@@ -2644,10 +2691,12 @@ class Handler(BaseHTTPRequestHandler):
|
||||
text = (f"Athena läuft im {mode['active'].upper()}-Modus. "
|
||||
f"Phase: {mode['phase']}. Musik-Worker: "
|
||||
f"{mode['music_worker']}. Stem-Separator: "
|
||||
f"{mode['separator_worker']}. LLM-Profil: {profile or 'entladen'}.")
|
||||
f"{mode['separator_worker']}. Voice Studio: "
|
||||
f"{mode['voice_worker']}. LLM-Profil: {profile or 'entladen'}.")
|
||||
else:
|
||||
target = ("music" if command == "/athena music" else
|
||||
"separation" if command in {"/athena stems", "/athena separation"}
|
||||
else "voice" if command == "/athena voice"
|
||||
else "llm")
|
||||
try:
|
||||
started, phase = schedule_operating_mode(target)
|
||||
@@ -2656,6 +2705,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
if target == "music" else
|
||||
"Stimmtrennung wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "separation" else
|
||||
"Voice Studio wird gestartet. LLM und TTS werden entladen."
|
||||
if target == "voice" else
|
||||
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
|
||||
else:
|
||||
text = (f"Athena ist bereits im {target.upper()}-Modus "
|
||||
@@ -2948,15 +2999,14 @@ def _startup_reconcile() -> None:
|
||||
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
|
||||
|
||||
special_mode = previous.get("mode")
|
||||
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation"}:
|
||||
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice"}:
|
||||
STATE.mode = special_mode
|
||||
STATE.mode_phase = f"starting-{special_mode}"
|
||||
_set_qwen_unavailable(True)
|
||||
try:
|
||||
path = ("/workers/music/start" if special_mode == "music"
|
||||
else "/workers/separator/start")
|
||||
path, _worker_state, wait_ready = _special_worker(special_mode)
|
||||
_profile_controller_request("POST", path)
|
||||
_wait_music_ready() if special_mode == "music" else _wait_separator_ready()
|
||||
wait_ready()
|
||||
STATE.mode_phase = "ready"
|
||||
RUNTIME.save(mode=special_mode, phase=special_mode)
|
||||
log.info("Recovery: Spezialmodus %s wiederhergestellt", special_mode)
|
||||
|
||||
Reference in New Issue
Block a user