Add X-VC voice conversion mode

This commit is contained in:
Mikei386
2026-09-09 16:14:10 +02:00
parent 17f1a08d7d
commit 87a2ae5704
15 changed files with 544 additions and 28 deletions
+6 -3
View File
@@ -15,8 +15,8 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
- Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio und
BS-RoFormer-Stimmtrennung
- Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio,
BS-RoFormer-Stimmtrennung, OmniVoice und X-VC
- Portainer CE als optionale Container-Ansicht auf Port 9443
- WireGuard-Gateway, Datenbackup und Athena-Operator
- keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz
@@ -125,9 +125,12 @@ Details, Installation, Prüfung und Rollback stehen in
- Musikstudio, Original UI (stabil): `http://192.168.1.212:7862`
- Musikstudio, Community UI (experimentell): `http://192.168.1.212:7861`
- Spuren trennen (BS-RoFormer + Demucs, 2/4/6 Stems): `http://192.168.1.212:8007`
- Voice Studio (OmniVoice, Text zu Stimme): `http://192.168.1.212:8008`
- Voice Changer (X-VC, Audio zu Audio): `http://192.168.1.212:8009`
Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren
außerdem `/athena music`, `/athena stems`, `/athena llm` und `/athena status`; Details stehen in
außerdem `/athena music`, `/athena stems`, `/athena voice`,
`/athena voicechange`, `/athena llm` und `/athena status`; Details stehen in
[docs/OPERATING_MODES.md](docs/OPERATING_MODES.md).
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
+3
View File
@@ -572,6 +572,7 @@ services:
MUSIC_WORKER: acestep
SEPARATOR_WORKER: bs-roformer
VOICE_WORKER: vevo2
VOICE_CHANGE_WORKER: xvc
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
@@ -630,6 +631,7 @@ services:
ENABLE_STT: "true"
ENABLE_MUSIC_MODE: "true"
MUSIC_START_TIMEOUT: "600"
VOICE_CHANGE_START_TIMEOUT: "600"
STT_WORKER_URL: http://whisper:8084
STT_TIMEOUT: "300"
networks: [frontend, control, inference]
@@ -862,6 +864,7 @@ services:
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}"
HOST_PROC: /host/proc
HOST_DATA: /host/data
HOST_MODELS: /host/models
+9
View File
@@ -61,6 +61,15 @@ class DashboardModeTests(unittest.TestCase):
request = urlopen.call_args.args[0]
self.assertEqual(json.loads(request.data), {"mode": "voice"})
def test_voice_change_mode_is_forwarded_to_router(self):
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
status, body = self.dashboard.change_mode("voicechange")
self.assertEqual(status, 202)
self.assertEqual(body, {"status": "accepted"})
request = urlopen.call_args.args[0]
self.assertEqual(json.loads(request.data), {"mode": "voicechange"})
def test_unknown_mode_is_rejected_without_router_request(self):
with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen:
status, body = self.dashboard.change_mode("unknown")
+39
View File
@@ -51,7 +51,46 @@ def voice_item(state="exited"):
"Labels": {controller.VOICE_LABEL_KEY: "vevo2"}}
def voice_change_item(state="exited"):
return {"Id": "id-xvc", "State": state,
"Labels": {controller.VOICE_CHANGE_LABEL_KEY: "xvc"}}
class ProfileControllerTests(unittest.TestCase):
def test_voice_change_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["medium"] = item("medium", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "VOICE_CHANGE_WORKER", "xvc"), \
patch.object(controller, "VOICE_WORKER", "vevo2"), \
patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "voice_change_container", return_value=voice_change_item()), \
patch.object(controller, "voice_container", return_value=voice_item("running")), \
patch.object(controller, "music_container", return_value=music_item("running")), \
patch.object(controller, "separator_container", return_value=separator_item("running")), \
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_voice_change_worker(True)
self.assertEqual(result, {"voice_change_worker": "xvc", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-medium/stop?t=120"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-music/stop?t=30"),
("POST", "/containers/id-separator/stop?t=30"),
("POST", "/containers/id-voice/stop?t=30"),
("POST", "/containers/id-xvc/start"),
])
def test_voice_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["medium"] = item("medium", "running")
+21 -10
View File
@@ -1,13 +1,16 @@
# Athena-Betriebsmodi
Athena besitzt vier gegenseitig exklusive Betriebsmodi:
Athena besitzt sechs gegenseitig exklusive Betriebsmodi:
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt.
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt.
- `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt
Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt.
- `voice`: Vevo2 überträgt Sprache oder Gesang auf eine gespeicherte
- `voice`: OmniVoice erzeugt Sprache aus Text mit einer gewählten
Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt.
- `voicechange`: X-VC überträgt eine vorhandene Sprachaufnahme auf eine
Referenzstimme und bewahrt dabei Inhalt und Timing. Alle anderen
GPU-Dienste sind gestoppt.
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
@@ -15,8 +18,8 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen
## Bedienung
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**
und **Voice Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**,
**Voice Studio** und **Voice Changer** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
- **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende
Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der
@@ -53,11 +56,16 @@ bleibt rückwärtskompatibel.
Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter
`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter
`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem
Speichern eine Bestätigung der Nutzungsberechtigung. Vevo2 gibt
unkomprimiertes 24-kHz-WAV aus und wandelt sowohl Sprache als auch Gesang;
dies ist noch kein Echtzeit-Live-Voice-Changer. Die Gewichte sind
CC BY-NC-ND 4.0 lizenziert und werden auf Athena ausschließlich privat und
nichtkommerziell eingesetzt.
Speichern eine Bestätigung der Nutzungsberechtigung. OmniVoice gibt
unkomprimiertes WAV aus und erzeugt Sprache aus Text; es verarbeitet keine
bereits eingesprochene Quellaufnahme.
Der X-VC Voice Changer ist ausschließlich unter
`http://192.168.1.212:8009` erreichbar. Er nimmt eine Quellaufnahme und eine
Referenzstimme an und gibt 16-kHz-PCM-WAV aus. Die dokumentierte Sprachbasis
des verwendeten GLM-4-Voice-Tokenizers ist Chinesisch und Englisch; Deutsch
bleibt deshalb bis zur Hörabnahme ein Qualitätstest und kein zugesagter
Produktionspfad. X-VC läuft ausschließlich auf der RTX 5080.
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
@@ -66,6 +74,7 @@ Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
/athena music
/athena stems
/athena voice
/athena voicechange
/athena llm
/athena status
```
@@ -77,6 +86,7 @@ GET /mode
POST /mode {"mode":"music"}
POST /mode {"mode":"separation"}
POST /mode {"mode":"voice"}
POST /mode {"mode":"voicechange"}
POST /mode {"mode":"llm"}
```
@@ -84,7 +94,8 @@ Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
`com.mike-ai.music-worker=acestep` beziehungsweise
`com.mike-ai.stem-separator=bs-roformer` oder
`com.mike-ai.voice-worker=vevo2` markierten Container; freie
`com.mike-ai.voice-worker=vevo2` beziehungsweise
`com.mike-ai.voice-change-worker=xvc` markierten Container; freie
Container- oder Docker-Befehle werden nicht entgegengenommen.
## Wiederanlauf
+1
View File
@@ -58,6 +58,7 @@ Titelgenerierung und Kontextkompression in Hermes.
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen**; Container ersetzt, Image und Daten vorerst nur als Rollback erhalten |
| 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen |
| 09.09.2026 | `chenxie95/X-VC`, Code `49df8c591eafc48b096e466d96f9839f9c0dd739`, UI-Basis `d761cd6421e85376b2656dfefd8471d7f35a42be` | Offizielles Beispiel Ende-zu-Ende gewandelt: 5,20 s Audio in 1,07 s (RTF 0,21), gültiges 16-kHz-Mono-PCM-WAV; Modell belegt rund 2,9 GiB auf der RTX 5080. GLM-4-Voice-Tokenizer dokumentiert Chinesisch und Englisch | **technischer Start- und Konvertierungstest bestanden**; deutsche Hörabnahme offen |
## Musikgenerierung
@@ -0,0 +1,43 @@
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
ARG XVC_COMMIT=49df8c591eafc48b096e466d96f9839f9c0dd739
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
ca-certificates curl ffmpeg git libsndfile1 python3 python3-pip python3-venv \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/Jerrister/X-VC.git /opt/xvc \
&& cd /opt/xvc \
&& git checkout "${XVC_COMMIT}" \
&& rm -rf .git
RUN python3 -m venv /opt/venv
ENV PATH="/opt/venv/bin:${PATH}"
# RTX 5080: use a CUDA 12.8 PyTorch build instead of X-VC's older training pin.
RUN python -m pip install --no-cache-dir --upgrade pip \
&& python -m pip install --no-cache-dir \
--index-url https://download.pytorch.org/whl/cu128 \
"torch==2.8.0" "torchaudio==2.8.0" \
&& python -m pip install --no-cache-dir \
"gradio>=5.49,<6" "huggingface_hub>=0.34,<2" \
"transformers>=4.44,<5" "hydra-core>=1.3,<2" "omegaconf>=2.3,<3" \
"x-transformers>=1.40,<3" "einops>=0.8,<1" "einx>=0.3,<1" \
"numpy>=1.26,<3" "scipy>=1.13,<2" "soundfile>=0.12,<1" \
"soxr>=0.5,<1" "tqdm>=4.66,<5" "wandb>=0.18,<1"
RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2"
COPY app.py /opt/xvc/local_webui.py
COPY inference_log.py /opt/xvc/utils/log.py
ENV HF_HOME=/models/huggingface \
PYTHONUNBUFFERED=1
EXPOSE 8009
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
CMD curl -fsS http://127.0.0.1:8009/ >/dev/null || exit 1
CMD ["python", "/opt/xvc/local_webui.py"]
@@ -0,0 +1,24 @@
# X-VC voice conversion on Athena
Isolated quality gate for `chenxie95/X-VC`, using the official inference path
from commit `49df8c591eafc48b096e466d96f9839f9c0dd739`. The UI is adapted from the
public Hugging Face Space at commit
`d761cd6421e85376b2656dfefd8471d7f35a42be` and runs locally without
ZeroGPU.
- Private URL: `http://192.168.1.212:8009`
- Source clip: speech content and timing to preserve
- Reference clip: target speaker identity
- Output: 16 kHz PCM WAV
- GPU: RTX 5080 only
- Persistent cache: `/data/voice/xvc/huggingface`
- Code and model license: MIT
The semantic tokenizer documents Chinese and English. German is therefore a
quality gate, not an assumed supported language. Keep OmniVoice installed: it
does text-to-speech cloning, while X-VC tests true audio-to-audio conversion.
Technical acceptance on 9 September 2026 used the repository's source and
target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The
result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about
2.9 GiB on the RTX 5080. German listening quality remains open.
+188
View File
@@ -0,0 +1,188 @@
"""Local Athena adaptation of the public X-VC Gradio demo."""
import logging
import os
import sys
import tempfile
import time
from typing import Tuple
import gradio as gr
import numpy as np
import soundfile as sf
import torch
from huggingface_hub import hf_hub_download
from omegaconf import OmegaConf
HERE = "/opt/xvc"
sys.path.insert(0, HERE)
from bins.infer_utils import precompute_conditions, run_offline, run_streaming, to_numpy_audio
from models.codec.sac.model import XVC
from utils.audio import audio_highpass_filter, audio_volume_normalize, load_audio
logging.basicConfig(level=logging.INFO)
log = logging.getLogger("xvc-local")
MODEL_REPO = "chenxie95/X-VC"
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
SAMPLE_RATE = 16000
LATENT_HOP_LENGTH = 1280
MAX_SECONDS = 20.0
MODE_OFFLINE = "Offline (höchste Qualität)"
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
def _load_model() -> XVC:
speaker_config = hf_hub_download(
repo_id=SPACE_REPO,
repo_type="space",
filename=f"{SPEAKER_SUBDIR}/configuration.json",
)
hf_hub_download(
repo_id=SPACE_REPO,
repo_type="space",
filename=f"{SPEAKER_SUBDIR}/pretrained_eres2net.ckpt",
)
checkpoint = hf_hub_download(repo_id=MODEL_REPO, filename="xvc.pt")
cfg = OmegaConf.load(os.path.join(HERE, "configs", "xvc.yaml"))
cfg["model"]["generator"].pop("loss_config", None)
cfg["model"].pop("discriminator", None)
cfg["model"]["generator"]["speaker_encoder"]["pretrained_dir"] = os.path.dirname(speaker_config)
infer_cfg = os.path.join(tempfile.gettempdir(), "xvc_inference.yaml")
OmegaConf.save(cfg, infer_cfg)
loaded = XVC.load_from_checkpoint(infer_cfg, checkpoint, device=torch.device("cuda"))
loaded.remove_weight_norm()
loaded = loaded.eval().to("cuda")
log.info("X-VC model ready on %s", torch.cuda.get_device_name(0))
return loaded
MODEL = _load_model()
def _prepare_wav(path: str) -> np.ndarray:
wav = load_audio(path, sampling_rate=SAMPLE_RATE, volume_normalize=False)
if wav is None or len(wav) == 0:
raise gr.Error("Die Audiodatei konnte nicht gelesen werden.")
wav = wav[: int(MAX_SECONDS * SAMPLE_RATE)]
wav = audio_volume_normalize(wav)
wav = audio_highpass_filter(wav, SAMPLE_RATE, 40)
remainder = len(wav) % LATENT_HOP_LENGTH
if remainder:
wav = np.pad(wav, (0, LATENT_HOP_LENGTH - remainder), mode="constant")
return wav.astype(np.float32)
def _tensor(wav: np.ndarray) -> torch.Tensor:
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
def _write_wav(audio: np.ndarray) -> str:
os.makedirs("/output", exist_ok=True)
path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav")
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16")
return path
@torch.inference_mode()
def convert(
source_audio: str,
reference_audio: str,
mode: str = MODE_OFFLINE,
chunk_ms: int = 2400,
current_ms: int = 120,
future_ms: int = 100,
smooth_ms: int = 20,
progress=gr.Progress(track_tqdm=True),
) -> Tuple[str, str]:
if not source_audio:
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
if not reference_audio:
raise gr.Error("Bitte eine Referenzdatei mit der Zielstimme hochladen.")
source_np = _prepare_wav(source_audio)
reference_np = _prepare_wav(reference_audio)
source_wav = _tensor(source_np)
target_wav = _tensor(reference_np)
seconds = len(source_np) / SAMPLE_RATE
started = time.time()
if mode == MODE_STREAMING:
history_ms = int(chunk_ms) - int(current_ms) - int(smooth_ms) - int(future_ms)
if history_ms < 0:
raise gr.Error("Fenster muss mindestens Current + Lookahead + Crossfade umfassen.")
speaker_condition, frame_condition = precompute_conditions(MODEL, target_wav, target_wav)
recon, latency_ms = run_streaming(
model=MODEL,
source_wav=source_wav,
speaker_condition=speaker_condition,
frame_condition=frame_condition,
sample_rate=SAMPLE_RATE,
chunk_ms=int(chunk_ms),
current_ms=int(current_ms),
future_ms=int(future_ms),
smooth_ms=int(smooth_ms),
)
elapsed = time.time() - started
latency = np.asarray(latency_ms, dtype=np.float64)
report = (
f"**Streaming** · {len(latency)} Chunks · Mittel **{latency.mean():.0f} ms**, "
f"P95 **{np.percentile(latency, 95):.0f} ms** · insgesamt {elapsed:.2f} s "
f"für {seconds:.2f} s Audio (RTF {elapsed / seconds:.2f})"
)
else:
recon = run_offline(MODEL, source_wav, target_wav, target_wav)
elapsed = time.time() - started
report = (
f"**Offline** · {seconds:.2f} s Audio in {elapsed:.2f} s "
f"(RTF {elapsed / seconds:.2f})"
)
return _write_wav(to_numpy_audio(recon)), report
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
HEADER = """# X-VC — Voice Changer
Die **Quelle** liefert Text, Aussprache und Timing. Die **Referenz** liefert die
Zielstimme. X-VC arbeitet Audio-zu-Audio ohne Transkript oder Training.
[Paper](https://arxiv.org/abs/2604.12456) · [Modell](https://huggingface.co/chenxie95/X-VC) ·
[Code](https://github.com/Jerrister/X-VC)
"""
with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(HEADER)
with gr.Row():
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
run = gr.Button("Stimme umwandeln", variant="primary")
output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False)
report = gr.Markdown()
with gr.Accordion("Erweiterte Einstellungen", open=False):
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
with gr.Row():
current_ms = gr.Slider(40, 640, value=120, step=40, label="Aktueller Chunk (ms)")
chunk_ms = gr.Slider(800, 4800, value=2400, step=200, label="Gesamtfenster (ms)")
with gr.Row():
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.")
run.click(
convert,
inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms],
outputs=[output, report],
api_name="convert",
)
if __name__ == "__main__":
demo.queue(default_concurrency_limit=1).launch(
server_name="0.0.0.0",
server_port=8009,
show_error=True,
)
@@ -0,0 +1,39 @@
services:
xvc-studio:
build: .
image: mike-ai/xvc-studio:2026-09-09
container_name: mike-ai-xvc-studio
restart: "no"
labels:
com.mike-ai.voice-change-worker: xvc
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
HF_HOME: /models/huggingface
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
ports:
- "127.0.0.1:8009:8009"
volumes:
- /data/voice/xvc/huggingface:/models/huggingface
- /data/voice/xvc/output:/output
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8009/ >/dev/null"]
interval: 5s
timeout: 3s
start_period: 600s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
frontend:
aliases: [xvc-studio]
networks:
frontend:
name: mike-ai_frontend
external: true
@@ -0,0 +1,29 @@
"""Small inference-only replacement for X-VC's training logger.
The upstream logger imports WandB, Matplotlib and TensorBoard at module import
time although model inference only uses the normal logging functions.
"""
import logging
logging.basicConfig(level=logging.INFO)
_logger = logging.getLogger("xvc")
debug = _logger.debug
info = _logger.info
warn = _logger.warning
warning = _logger.warning
error = _logger.error
def init(*_args, **_kwargs):
return None
def write_audio(*_args, **_kwargs):
return None
def write_loss(*_args, **_kwargs):
return None
@@ -34,6 +34,8 @@ SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator"
SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
VOICE_CHANGE_LABEL_KEY = "com.mike-ai.voice-change-worker"
VOICE_CHANGE_WORKER = os.environ.get("VOICE_CHANGE_WORKER", "").strip()
LOCK = threading.Lock()
log = logging.getLogger("profile-controller")
@@ -135,6 +137,17 @@ def voice_container() -> dict:
return matches[0]
def voice_change_container() -> dict:
if not VOICE_CHANGE_WORKER:
raise RuntimeError("voice-change worker is not configured")
matches = [item for item in labelled_containers(VOICE_CHANGE_LABEL_KEY)
if item.get("Labels", {}).get(VOICE_CHANGE_LABEL_KEY) == VOICE_CHANGE_WORKER]
if len(matches) != 1:
raise RuntimeError(
f"expected exactly one voice-change worker {VOICE_CHANGE_WORKER!r}, found {len(matches)}")
return matches[0]
def stop_music_if_configured() -> None:
if MUSIC_WORKER:
stop_container(music_container(), timeout=30)
@@ -150,6 +163,11 @@ def stop_voice_if_configured() -> None:
stop_container(voice_container(), timeout=30)
def stop_voice_change_if_configured() -> None:
if VOICE_CHANGE_WORKER:
stop_container(voice_change_container(), timeout=30)
def stop_container(item: dict, timeout: int = 120) -> None:
if item.get("State") != "running":
return
@@ -190,6 +208,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
stop_music_if_configured()
stop_separator_if_configured()
stop_voice_if_configured()
stop_voice_change_if_configured()
for other in image_containers():
if other["Id"] != item["Id"]:
stop_container(other, timeout=20)
@@ -217,6 +236,7 @@ def set_music_worker(running: bool) -> dict:
stop_container(tts_container(), timeout=30)
stop_separator_if_configured()
stop_voice_if_configured()
stop_voice_change_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
@@ -236,6 +256,7 @@ def set_separator_worker(running: bool) -> dict:
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_voice_if_configured()
stop_voice_change_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
@@ -244,7 +265,7 @@ def set_separator_worker(running: bool) -> dict:
def set_voice_worker(running: bool) -> dict:
"""Start Vevo2 exclusively, or stop it before LLM restoration."""
"""Start OmniVoice exclusively, or stop it before LLM restoration."""
with LOCK:
item = voice_container()
if running:
@@ -255,6 +276,7 @@ def set_voice_worker(running: bool) -> dict:
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_separator_if_configured()
stop_voice_change_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
@@ -262,6 +284,26 @@ def set_voice_worker(running: bool) -> dict:
"state": "running" if running else "stopped"}
def set_voice_change_worker(running: bool) -> dict:
"""Start X-VC exclusively, or stop it before another mode is loaded."""
with LOCK:
item = voice_change_container()
if running:
for profile_item in containers().values():
stop_container(profile_item)
for worker in image_containers():
stop_container(worker, timeout=20)
stop_container(tts_container(), timeout=30)
stop_music_if_configured()
stop_separator_if_configured()
stop_voice_if_configured()
start_container(item)
else:
stop_container(item, timeout=30)
return {"voice_change_worker": VOICE_CHANGE_WORKER,
"state": "running" if running else "stopped"}
def active_profile(items: dict[str, dict] | None = None) -> str | None:
items = items or containers()
active = [name for name, item in items.items() if item.get("State") == "running"]
@@ -280,6 +322,7 @@ def activate(profile: str) -> dict:
stop_music_if_configured()
stop_separator_if_configured()
stop_voice_if_configured()
stop_voice_change_if_configured()
start_container(tts_container())
items = containers()
missing = [name for name in ALLOWED if name not in items]
@@ -365,6 +408,13 @@ class Handler(BaseHTTPRequestHandler):
"unhealthy" if "(unhealthy)" in voice_status else
"starting" if voice.get("State") == "running" else
"stopped")
voice_change = voice_change_container() if VOICE_CHANGE_WORKER else {}
voice_change_status = voice_change.get("Status", "")
voice_change_health = ("disabled" if not VOICE_CHANGE_WORKER else
"healthy" if "(healthy)" in voice_change_status else
"unhealthy" if "(unhealthy)" in voice_change_status else
"starting" if voice_change.get("State") == "running" else
"stopped")
self.reply(200, {"active_profile": active_profile(items),
"music_worker": music.get("State", "disabled"),
"music_health": music_health,
@@ -372,6 +422,8 @@ class Handler(BaseHTTPRequestHandler):
"separator_health": separator_health,
"voice_worker": voice.get("State", "disabled"),
"voice_health": voice_health,
"voice_change_worker": voice_change.get("State", "disabled"),
"voice_change_health": voice_change_health,
"profiles": {name: items.get(name, {}).get(
"State", "missing") for name in ALLOWED}})
except Exception as exc:
@@ -410,6 +462,13 @@ class Handler(BaseHTTPRequestHandler):
log.exception("voice worker transition failed")
self.reply(503, {"error": str(exc)})
return
if self.path in {"/workers/voice-change/start", "/workers/voice-change/stop"}:
try:
self.reply(200, set_voice_change_worker(self.path.endswith("/start")))
except Exception as exc:
log.exception("voice-change worker transition failed")
self.reply(503, {"error": str(exc)})
return
worker_paths = {
"/workers/image/start": (IMAGE_WORKER, True),
"/workers/image/stop": (IMAGE_WORKER, False),
@@ -82,6 +82,7 @@ start_proxy 7861 music-ui:3000
start_proxy 7862 music-worker:7860
start_proxy 8007 stem-separator:8080
start_proxy 8008 voice-studio:8008
start_proxy 8009 xvc-studio:8009
start_proxy 8202 mcp-athena-operator:8000
start_proxy 9443 portainer:9443
+23 -3
View File
@@ -29,6 +29,7 @@ MUSIC_ORIGINAL_UI_URL = os.getenv(
)
SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
VOICE_CHANGE_UI_URL = os.getenv("VOICE_CHANGE_UI_URL", "http://192.168.1.212:8009/")
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
@@ -278,7 +279,7 @@ def router_status() -> tuple[dict[str, Any], str | None]:
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
if mode not in {"llm", "music", "separation", "voice"}:
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
return 400, {"error": "invalid mode"}
headers = {"Accept": "application/json", "Content-Type": "application/json"}
if ROUTER_API_KEY:
@@ -636,7 +637,7 @@ HTML = r'''<!doctype html>
</style></head><body><main>
<div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div>
<section class="grid">
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><button id="voiceMode" onclick="setMode('voice')">Voice Studio</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span><span id="voiceOpen" hidden><a class="stable" href="__VOICE_UI_URL__" target="_blank" rel="noopener">Voice Studio öffnen</a></span></div></div></article>
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><button id="voiceMode" onclick="setMode('voice')">Voice Studio</button><button id="voiceChangeMode" onclick="setMode('voicechange')">Voice Changer</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span><span id="voiceOpen" hidden><a class="stable" href="__VOICE_UI_URL__" target="_blank" rel="noopener">Voice Studio öffnen</a></span><span id="voiceChangeOpen" hidden><a class="experimental" href="__VOICE_CHANGE_UI_URL__" target="_blank" rel="noopener">X-VC öffnen</a></span></div></div></article>
<article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article>
<article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article>
<article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article>
@@ -677,11 +678,30 @@ const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":
let modeBusy=false;
async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
</script><script>
const baseSetMode=setMode;
setMode=async function(mode){
if(modeBusy)return;
for(const id of ['llmMode','musicMode','separationMode','voiceMode','voiceChangeMode'])$(id).disabled=true;
await baseSetMode(mode);
};
async function refreshVoiceChange(){
try{
const response=await fetch('/api/status',{cache:'no-store'}); if(!response.ok)return;
const data=await response.json(),mode=data.router?.mode||{},active=mode.active==='voicechange';
$('voiceChangeMode').classList.toggle('active',active);
$('voiceChangeMode').disabled=modeBusy||(mode.phase&&mode.phase!=='ready')||!mode.enabled;
$('voiceChangeOpen').hidden=!active;
if(active)$('operatingMode').textContent='Voice Changer';
}catch(_error){}
}
setTimeout(()=>{refreshVoiceChange();setInterval(refreshVoiceChange,1000)},150);
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''.replace(
"__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL
).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL
).replace("__VOICE_UI_URL__", VOICE_UI_URL)
).replace("__VOICE_UI_URL__", VOICE_UI_URL
).replace("__VOICE_CHANGE_UI_URL__", VOICE_CHANGE_UI_URL)
FULL_JS = r'''
+58 -11
View File
@@ -111,6 +111,7 @@ ENABLE_MUSIC_MODE = os.environ.get(
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
VOICE_CHANGE_START_TIMEOUT = float(os.environ.get("VOICE_CHANGE_START_TIMEOUT", "600"))
# Optional worker APIs. The clean Docker baseline deliberately ships only
# text/multimodal chat; absent workers must fail explicitly instead of trying
@@ -383,6 +384,27 @@ def _voice_worker_health() -> str:
return "unknown"
def _voice_change_worker_state() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_change_worker", "missing"))
except Exception as exc:
log.warning("Voice-Change-Worker-Status nicht verfügbar: %s", exc)
return "unknown"
def _voice_change_worker_health() -> str:
if not PROFILE_CONTROL_URL:
return "unsupported"
try:
return str(_profile_controller_request("GET", "/status").get(
"voice_change_health", "unknown"))
except Exception:
return "unknown"
def _wait_music_ready() -> None:
deadline = time.monotonic() + MUSIC_START_TIMEOUT
while time.monotonic() < deadline:
@@ -419,10 +441,24 @@ def _wait_voice_ready() -> None:
and status.get("voice_health") == "healthy"):
return
if status.get("voice_health") == "unhealthy":
raise RuntimeError("Vevo2-Container ist unhealthy")
raise RuntimeError("OmniVoice-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"Vevo2 nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
f"OmniVoice nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
def _wait_voice_change_ready() -> None:
deadline = time.monotonic() + VOICE_CHANGE_START_TIMEOUT
while time.monotonic() < deadline:
status = _profile_controller_request("GET", "/status")
if (status.get("voice_change_worker") == "running"
and status.get("voice_change_health") == "healthy"):
return
if status.get("voice_change_health") == "unhealthy":
raise RuntimeError("X-VC-Container ist unhealthy")
time.sleep(POLL_INTERVAL)
raise RuntimeError(
f"X-VC nach {VOICE_CHANGE_START_TIMEOUT:.0f} s nicht bereit")
def _special_worker(mode: str) -> tuple[str, str, callable]:
@@ -432,6 +468,9 @@ def _special_worker(mode: str) -> tuple[str, str, callable]:
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
if mode == "voice":
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
if mode == "voicechange":
return ("/workers/voice-change/start", _voice_change_worker_state(),
_wait_voice_change_ready)
raise ValueError(f"unbekannter Spezialmodus: {mode}")
@@ -439,11 +478,11 @@ def set_operating_mode(mode: str) -> dict:
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
raise ValueError("Modus muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
with STATE.lock:
STATE.mode_error = None
if mode in {"music", "separation", "voice"}:
if mode in {"music", "separation", "voice", "voicechange"}:
path, worker_state, wait_ready = _special_worker(mode)
if STATE.mode == mode and worker_state == "running":
return {"status": "ok", "mode": mode, "changed": False}
@@ -485,6 +524,7 @@ def set_operating_mode(mode: str) -> dict:
_profile_controller_request("POST", "/workers/music/stop")
_profile_controller_request("POST", "/workers/separator/stop")
_profile_controller_request("POST", "/workers/voice/stop")
_profile_controller_request("POST", "/workers/voice-change/stop")
_restore_qwen(profile)
STATE.mode = "llm"
STATE.mode_phase = "ready"
@@ -503,8 +543,8 @@ def schedule_operating_mode(mode: str) -> tuple[bool, str]:
"""Start a transition in the background so chat/UI acknowledgement is instant."""
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
raise RuntimeError("Musikmodus ist nicht konfiguriert")
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
raise ValueError("Modus muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
with STATE.lock:
if STATE.mode_phase not in {"ready", "error"}:
return False, STATE.mode_phase
@@ -541,6 +581,7 @@ def _control_command(data: dict, path: str) -> str | None:
return command if command in {"/athena music", "/athena stems",
"/athena separation", "/athena llm",
"/athena voice",
"/athena voicechange", "/athena changer",
"/athena status"} else None
@@ -2018,6 +2059,8 @@ class Handler(BaseHTTPRequestHandler):
"separator_health": _separator_worker_health(),
"voice_worker": _voice_worker_state(),
"voice_health": _voice_worker_health(),
"voice_change_worker": _voice_change_worker_state(),
"voice_change_health": _voice_change_worker_health(),
"return_profile": state.get("return_profile"),
"last_error": STATE.mode_error,
"enabled": ENABLE_MUSIC_MODE,
@@ -2027,8 +2070,8 @@ class Handler(BaseHTTPRequestHandler):
try:
data = json.loads(self._read_body() or b"{}")
mode = data.get("mode") if isinstance(data, dict) else None
if mode not in {"llm", "music", "separation", "voice"}:
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation' oder 'voice' sein")
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
started, phase = schedule_operating_mode(mode)
self._send_json(202 if started else 200, {
"status": "accepted" if started else "ok",
@@ -2692,11 +2735,13 @@ class Handler(BaseHTTPRequestHandler):
f"Phase: {mode['phase']}. Musik-Worker: "
f"{mode['music_worker']}. Stem-Separator: "
f"{mode['separator_worker']}. Voice Studio: "
f"{mode['voice_worker']}. LLM-Profil: {profile or 'entladen'}.")
f"{mode['voice_worker']}. Voice Changer: "
f"{mode['voice_change_worker']}. LLM-Profil: {profile or 'entladen'}.")
else:
target = ("music" if command == "/athena music" else
"separation" if command in {"/athena stems", "/athena separation"}
else "voice" if command == "/athena voice"
else "voicechange" if command in {"/athena voicechange", "/athena changer"}
else "llm")
try:
started, phase = schedule_operating_mode(target)
@@ -2707,6 +2752,8 @@ class Handler(BaseHTTPRequestHandler):
if target == "separation" else
"Voice Studio wird gestartet. LLM und TTS werden entladen."
if target == "voice" else
"Voice Changer wird gestartet. LLM und TTS werden entladen."
if target == "voicechange" else
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
else:
text = (f"Athena ist bereits im {target.upper()}-Modus "
@@ -2999,7 +3046,7 @@ def _startup_reconcile() -> None:
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
special_mode = previous.get("mode")
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice"}:
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice", "voicechange"}:
STATE.mode = special_mode
STATE.mode_phase = f"starting-{special_mode}"
_set_qwen_unavailable(True)