Add X-VC voice conversion mode
This commit is contained in:
@@ -15,8 +15,8 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
|
|||||||
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
|
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
|
||||||
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
||||||
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
||||||
- Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio und
|
- Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio,
|
||||||
BS-RoFormer-Stimmtrennung
|
BS-RoFormer-Stimmtrennung, OmniVoice und X-VC
|
||||||
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
- Portainer CE als optionale Container-Ansicht auf Port 9443
|
||||||
- WireGuard-Gateway, Datenbackup und Athena-Operator
|
- WireGuard-Gateway, Datenbackup und Athena-Operator
|
||||||
- keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz
|
- keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz
|
||||||
@@ -125,9 +125,12 @@ Details, Installation, Prüfung und Rollback stehen in
|
|||||||
- Musikstudio, Original UI (stabil): `http://192.168.1.212:7862`
|
- Musikstudio, Original UI (stabil): `http://192.168.1.212:7862`
|
||||||
- Musikstudio, Community UI (experimentell): `http://192.168.1.212:7861`
|
- Musikstudio, Community UI (experimentell): `http://192.168.1.212:7861`
|
||||||
- Spuren trennen (BS-RoFormer + Demucs, 2/4/6 Stems): `http://192.168.1.212:8007`
|
- Spuren trennen (BS-RoFormer + Demucs, 2/4/6 Stems): `http://192.168.1.212:8007`
|
||||||
|
- Voice Studio (OmniVoice, Text zu Stimme): `http://192.168.1.212:8008`
|
||||||
|
- Voice Changer (X-VC, Audio zu Audio): `http://192.168.1.212:8009`
|
||||||
|
|
||||||
Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren
|
Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren
|
||||||
außerdem `/athena music`, `/athena stems`, `/athena llm` und `/athena status`; Details stehen in
|
außerdem `/athena music`, `/athena stems`, `/athena voice`,
|
||||||
|
`/athena voicechange`, `/athena llm` und `/athena status`; Details stehen in
|
||||||
[docs/OPERATING_MODES.md](docs/OPERATING_MODES.md).
|
[docs/OPERATING_MODES.md](docs/OPERATING_MODES.md).
|
||||||
|
|
||||||
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
|
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
|
||||||
|
|||||||
@@ -572,6 +572,7 @@ services:
|
|||||||
MUSIC_WORKER: acestep
|
MUSIC_WORKER: acestep
|
||||||
SEPARATOR_WORKER: bs-roformer
|
SEPARATOR_WORKER: bs-roformer
|
||||||
VOICE_WORKER: vevo2
|
VOICE_WORKER: vevo2
|
||||||
|
VOICE_CHANGE_WORKER: xvc
|
||||||
networks: [control]
|
networks: [control]
|
||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
healthcheck:
|
healthcheck:
|
||||||
@@ -630,6 +631,7 @@ services:
|
|||||||
ENABLE_STT: "true"
|
ENABLE_STT: "true"
|
||||||
ENABLE_MUSIC_MODE: "true"
|
ENABLE_MUSIC_MODE: "true"
|
||||||
MUSIC_START_TIMEOUT: "600"
|
MUSIC_START_TIMEOUT: "600"
|
||||||
|
VOICE_CHANGE_START_TIMEOUT: "600"
|
||||||
STT_WORKER_URL: http://whisper:8084
|
STT_WORKER_URL: http://whisper:8084
|
||||||
STT_TIMEOUT: "300"
|
STT_TIMEOUT: "300"
|
||||||
networks: [frontend, control, inference]
|
networks: [frontend, control, inference]
|
||||||
@@ -862,6 +864,7 @@ services:
|
|||||||
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
|
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
|
||||||
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
|
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
|
||||||
VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
|
VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
|
||||||
|
VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}"
|
||||||
HOST_PROC: /host/proc
|
HOST_PROC: /host/proc
|
||||||
HOST_DATA: /host/data
|
HOST_DATA: /host/data
|
||||||
HOST_MODELS: /host/models
|
HOST_MODELS: /host/models
|
||||||
|
|||||||
@@ -61,6 +61,15 @@ class DashboardModeTests(unittest.TestCase):
|
|||||||
request = urlopen.call_args.args[0]
|
request = urlopen.call_args.args[0]
|
||||||
self.assertEqual(json.loads(request.data), {"mode": "voice"})
|
self.assertEqual(json.loads(request.data), {"mode": "voice"})
|
||||||
|
|
||||||
|
def test_voice_change_mode_is_forwarded_to_router(self):
|
||||||
|
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
|
||||||
|
status, body = self.dashboard.change_mode("voicechange")
|
||||||
|
|
||||||
|
self.assertEqual(status, 202)
|
||||||
|
self.assertEqual(body, {"status": "accepted"})
|
||||||
|
request = urlopen.call_args.args[0]
|
||||||
|
self.assertEqual(json.loads(request.data), {"mode": "voicechange"})
|
||||||
|
|
||||||
def test_unknown_mode_is_rejected_without_router_request(self):
|
def test_unknown_mode_is_rejected_without_router_request(self):
|
||||||
with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen:
|
with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen:
|
||||||
status, body = self.dashboard.change_mode("unknown")
|
status, body = self.dashboard.change_mode("unknown")
|
||||||
|
|||||||
@@ -51,7 +51,46 @@ def voice_item(state="exited"):
|
|||||||
"Labels": {controller.VOICE_LABEL_KEY: "vevo2"}}
|
"Labels": {controller.VOICE_LABEL_KEY: "vevo2"}}
|
||||||
|
|
||||||
|
|
||||||
|
def voice_change_item(state="exited"):
|
||||||
|
return {"Id": "id-xvc", "State": state,
|
||||||
|
"Labels": {controller.VOICE_CHANGE_LABEL_KEY: "xvc"}}
|
||||||
|
|
||||||
|
|
||||||
class ProfileControllerTests(unittest.TestCase):
|
class ProfileControllerTests(unittest.TestCase):
|
||||||
|
def test_voice_change_start_exclusively_stops_gpu_workers(self):
|
||||||
|
profiles = {name: item(name) for name in controller.ALLOWED}
|
||||||
|
profiles["medium"] = item("medium", "running")
|
||||||
|
calls = []
|
||||||
|
|
||||||
|
def request(method, path):
|
||||||
|
calls.append((method, path))
|
||||||
|
return 204, b""
|
||||||
|
|
||||||
|
with patch.object(controller, "VOICE_CHANGE_WORKER", "xvc"), \
|
||||||
|
patch.object(controller, "VOICE_WORKER", "vevo2"), \
|
||||||
|
patch.object(controller, "MUSIC_WORKER", "acestep"), \
|
||||||
|
patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
|
||||||
|
patch.object(controller, "containers", return_value=profiles), \
|
||||||
|
patch.object(controller, "voice_change_container", return_value=voice_change_item()), \
|
||||||
|
patch.object(controller, "voice_container", return_value=voice_item("running")), \
|
||||||
|
patch.object(controller, "music_container", return_value=music_item("running")), \
|
||||||
|
patch.object(controller, "separator_container", return_value=separator_item("running")), \
|
||||||
|
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
|
||||||
|
patch.object(controller, "tts_container", return_value=tts_item()), \
|
||||||
|
patch.object(controller, "docker_request", side_effect=request):
|
||||||
|
result = controller.set_voice_change_worker(True)
|
||||||
|
|
||||||
|
self.assertEqual(result, {"voice_change_worker": "xvc", "state": "running"})
|
||||||
|
self.assertEqual(calls, [
|
||||||
|
("POST", "/containers/id-medium/stop?t=120"),
|
||||||
|
("POST", "/containers/id-flux/stop?t=20"),
|
||||||
|
("POST", "/containers/id-tts/stop?t=30"),
|
||||||
|
("POST", "/containers/id-music/stop?t=30"),
|
||||||
|
("POST", "/containers/id-separator/stop?t=30"),
|
||||||
|
("POST", "/containers/id-voice/stop?t=30"),
|
||||||
|
("POST", "/containers/id-xvc/start"),
|
||||||
|
])
|
||||||
|
|
||||||
def test_voice_start_exclusively_stops_gpu_workers(self):
|
def test_voice_start_exclusively_stops_gpu_workers(self):
|
||||||
profiles = {name: item(name) for name in controller.ALLOWED}
|
profiles = {name: item(name) for name in controller.ALLOWED}
|
||||||
profiles["medium"] = item("medium", "running")
|
profiles["medium"] = item("medium", "running")
|
||||||
|
|||||||
+21
-10
@@ -1,13 +1,16 @@
|
|||||||
# Athena-Betriebsmodi
|
# Athena-Betriebsmodi
|
||||||
|
|
||||||
Athena besitzt vier gegenseitig exklusive Betriebsmodi:
|
Athena besitzt sechs gegenseitig exklusive Betriebsmodi:
|
||||||
|
|
||||||
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt.
|
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt.
|
||||||
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt.
|
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt.
|
||||||
- `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt
|
- `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt
|
||||||
Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt.
|
Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt.
|
||||||
- `voice`: Vevo2 überträgt Sprache oder Gesang auf eine gespeicherte
|
- `voice`: OmniVoice erzeugt Sprache aus Text mit einer gewählten
|
||||||
Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt.
|
Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt.
|
||||||
|
- `voicechange`: X-VC überträgt eine vorhandene Sprachaufnahme auf eine
|
||||||
|
Referenzstimme und bewahrt dabei Inhalt und Timing. Alle anderen
|
||||||
|
GPU-Dienste sind gestoppt.
|
||||||
|
|
||||||
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
|
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
|
||||||
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
|
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
|
||||||
@@ -15,8 +18,8 @@ wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen
|
|||||||
|
|
||||||
## Bedienung
|
## Bedienung
|
||||||
|
|
||||||
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**
|
Im Athena-Dashboard stehen **LLM-Betrieb**, **Musikstudio**, **Audio trennen**,
|
||||||
und **Voice Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
|
**Voice Studio** und **Voice Changer** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
|
||||||
|
|
||||||
- **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende
|
- **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende
|
||||||
Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der
|
Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der
|
||||||
@@ -53,11 +56,16 @@ bleibt rückwärtskompatibel.
|
|||||||
Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter
|
Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter
|
||||||
`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter
|
`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter
|
||||||
`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem
|
`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem
|
||||||
Speichern eine Bestätigung der Nutzungsberechtigung. Vevo2 gibt
|
Speichern eine Bestätigung der Nutzungsberechtigung. OmniVoice gibt
|
||||||
unkomprimiertes 24-kHz-WAV aus und wandelt sowohl Sprache als auch Gesang;
|
unkomprimiertes WAV aus und erzeugt Sprache aus Text; es verarbeitet keine
|
||||||
dies ist noch kein Echtzeit-Live-Voice-Changer. Die Gewichte sind
|
bereits eingesprochene Quellaufnahme.
|
||||||
CC BY-NC-ND 4.0 lizenziert und werden auf Athena ausschließlich privat und
|
|
||||||
nichtkommerziell eingesetzt.
|
Der X-VC Voice Changer ist ausschließlich unter
|
||||||
|
`http://192.168.1.212:8009` erreichbar. Er nimmt eine Quellaufnahme und eine
|
||||||
|
Referenzstimme an und gibt 16-kHz-PCM-WAV aus. Die dokumentierte Sprachbasis
|
||||||
|
des verwendeten GLM-4-Voice-Tokenizers ist Chinesisch und Englisch; Deutsch
|
||||||
|
bleibt deshalb bis zur Hörabnahme ein Qualitätstest und kein zugesagter
|
||||||
|
Produktionspfad. X-VC läuft ausschließlich auf der RTX 5080.
|
||||||
|
|
||||||
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
|
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
|
||||||
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
|
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
|
||||||
@@ -66,6 +74,7 @@ Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
|
|||||||
/athena music
|
/athena music
|
||||||
/athena stems
|
/athena stems
|
||||||
/athena voice
|
/athena voice
|
||||||
|
/athena voicechange
|
||||||
/athena llm
|
/athena llm
|
||||||
/athena status
|
/athena status
|
||||||
```
|
```
|
||||||
@@ -77,6 +86,7 @@ GET /mode
|
|||||||
POST /mode {"mode":"music"}
|
POST /mode {"mode":"music"}
|
||||||
POST /mode {"mode":"separation"}
|
POST /mode {"mode":"separation"}
|
||||||
POST /mode {"mode":"voice"}
|
POST /mode {"mode":"voice"}
|
||||||
|
POST /mode {"mode":"voicechange"}
|
||||||
POST /mode {"mode":"llm"}
|
POST /mode {"mode":"llm"}
|
||||||
```
|
```
|
||||||
|
|
||||||
@@ -84,7 +94,8 @@ Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in
|
|||||||
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
|
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
|
||||||
`com.mike-ai.music-worker=acestep` beziehungsweise
|
`com.mike-ai.music-worker=acestep` beziehungsweise
|
||||||
`com.mike-ai.stem-separator=bs-roformer` oder
|
`com.mike-ai.stem-separator=bs-roformer` oder
|
||||||
`com.mike-ai.voice-worker=vevo2` markierten Container; freie
|
`com.mike-ai.voice-worker=vevo2` beziehungsweise
|
||||||
|
`com.mike-ai.voice-change-worker=xvc` markierten Container; freie
|
||||||
Container- oder Docker-Befehle werden nicht entgegengenommen.
|
Container- oder Docker-Befehle werden nicht entgegengenommen.
|
||||||
|
|
||||||
## Wiederanlauf
|
## Wiederanlauf
|
||||||
|
|||||||
@@ -58,6 +58,7 @@ Titelgenerierung und Kontextkompression in Hermes.
|
|||||||
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
|
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich | **produktiv** |
|
||||||
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen**; Container ersetzt, Image und Daten vorerst nur als Rollback erhalten |
|
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen**; Container ersetzt, Image und Daten vorerst nur als Rollback erhalten |
|
||||||
| 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen |
|
| 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen |
|
||||||
|
| 09.09.2026 | `chenxie95/X-VC`, Code `49df8c591eafc48b096e466d96f9839f9c0dd739`, UI-Basis `d761cd6421e85376b2656dfefd8471d7f35a42be` | Offizielles Beispiel Ende-zu-Ende gewandelt: 5,20 s Audio in 1,07 s (RTF 0,21), gültiges 16-kHz-Mono-PCM-WAV; Modell belegt rund 2,9 GiB auf der RTX 5080. GLM-4-Voice-Tokenizer dokumentiert Chinesisch und Englisch | **technischer Start- und Konvertierungstest bestanden**; deutsche Hörabnahme offen |
|
||||||
|
|
||||||
## Musikgenerierung
|
## Musikgenerierung
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,43 @@
|
|||||||
|
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
|
||||||
|
|
||||||
|
ARG XVC_COMMIT=49df8c591eafc48b096e466d96f9839f9c0dd739
|
||||||
|
|
||||||
|
RUN apt-get update \
|
||||||
|
&& apt-get install -y --no-install-recommends \
|
||||||
|
ca-certificates curl ffmpeg git libsndfile1 python3 python3-pip python3-venv \
|
||||||
|
&& rm -rf /var/lib/apt/lists/*
|
||||||
|
|
||||||
|
RUN git clone https://github.com/Jerrister/X-VC.git /opt/xvc \
|
||||||
|
&& cd /opt/xvc \
|
||||||
|
&& git checkout "${XVC_COMMIT}" \
|
||||||
|
&& rm -rf .git
|
||||||
|
|
||||||
|
RUN python3 -m venv /opt/venv
|
||||||
|
ENV PATH="/opt/venv/bin:${PATH}"
|
||||||
|
|
||||||
|
# RTX 5080: use a CUDA 12.8 PyTorch build instead of X-VC's older training pin.
|
||||||
|
RUN python -m pip install --no-cache-dir --upgrade pip \
|
||||||
|
&& python -m pip install --no-cache-dir \
|
||||||
|
--index-url https://download.pytorch.org/whl/cu128 \
|
||||||
|
"torch==2.8.0" "torchaudio==2.8.0" \
|
||||||
|
&& python -m pip install --no-cache-dir \
|
||||||
|
"gradio>=5.49,<6" "huggingface_hub>=0.34,<2" \
|
||||||
|
"transformers>=4.44,<5" "hydra-core>=1.3,<2" "omegaconf>=2.3,<3" \
|
||||||
|
"x-transformers>=1.40,<3" "einops>=0.8,<1" "einx>=0.3,<1" \
|
||||||
|
"numpy>=1.26,<3" "scipy>=1.13,<2" "soundfile>=0.12,<1" \
|
||||||
|
"soxr>=0.5,<1" "tqdm>=4.66,<5" "wandb>=0.18,<1"
|
||||||
|
|
||||||
|
RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2"
|
||||||
|
|
||||||
|
COPY app.py /opt/xvc/local_webui.py
|
||||||
|
COPY inference_log.py /opt/xvc/utils/log.py
|
||||||
|
|
||||||
|
ENV HF_HOME=/models/huggingface \
|
||||||
|
PYTHONUNBUFFERED=1
|
||||||
|
|
||||||
|
EXPOSE 8009
|
||||||
|
|
||||||
|
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
|
||||||
|
CMD curl -fsS http://127.0.0.1:8009/ >/dev/null || exit 1
|
||||||
|
|
||||||
|
CMD ["python", "/opt/xvc/local_webui.py"]
|
||||||
@@ -0,0 +1,24 @@
|
|||||||
|
# X-VC voice conversion on Athena
|
||||||
|
|
||||||
|
Isolated quality gate for `chenxie95/X-VC`, using the official inference path
|
||||||
|
from commit `49df8c591eafc48b096e466d96f9839f9c0dd739`. The UI is adapted from the
|
||||||
|
public Hugging Face Space at commit
|
||||||
|
`d761cd6421e85376b2656dfefd8471d7f35a42be` and runs locally without
|
||||||
|
ZeroGPU.
|
||||||
|
|
||||||
|
- Private URL: `http://192.168.1.212:8009`
|
||||||
|
- Source clip: speech content and timing to preserve
|
||||||
|
- Reference clip: target speaker identity
|
||||||
|
- Output: 16 kHz PCM WAV
|
||||||
|
- GPU: RTX 5080 only
|
||||||
|
- Persistent cache: `/data/voice/xvc/huggingface`
|
||||||
|
- Code and model license: MIT
|
||||||
|
|
||||||
|
The semantic tokenizer documents Chinese and English. German is therefore a
|
||||||
|
quality gate, not an assumed supported language. Keep OmniVoice installed: it
|
||||||
|
does text-to-speech cloning, while X-VC tests true audio-to-audio conversion.
|
||||||
|
|
||||||
|
Technical acceptance on 9 September 2026 used the repository's source and
|
||||||
|
target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The
|
||||||
|
result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about
|
||||||
|
2.9 GiB on the RTX 5080. German listening quality remains open.
|
||||||
@@ -0,0 +1,188 @@
|
|||||||
|
"""Local Athena adaptation of the public X-VC Gradio demo."""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import os
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import time
|
||||||
|
from typing import Tuple
|
||||||
|
|
||||||
|
import gradio as gr
|
||||||
|
import numpy as np
|
||||||
|
import soundfile as sf
|
||||||
|
import torch
|
||||||
|
from huggingface_hub import hf_hub_download
|
||||||
|
from omegaconf import OmegaConf
|
||||||
|
|
||||||
|
HERE = "/opt/xvc"
|
||||||
|
sys.path.insert(0, HERE)
|
||||||
|
|
||||||
|
from bins.infer_utils import precompute_conditions, run_offline, run_streaming, to_numpy_audio
|
||||||
|
from models.codec.sac.model import XVC
|
||||||
|
from utils.audio import audio_highpass_filter, audio_volume_normalize, load_audio
|
||||||
|
|
||||||
|
logging.basicConfig(level=logging.INFO)
|
||||||
|
log = logging.getLogger("xvc-local")
|
||||||
|
|
||||||
|
MODEL_REPO = "chenxie95/X-VC"
|
||||||
|
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
|
||||||
|
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
|
||||||
|
SAMPLE_RATE = 16000
|
||||||
|
LATENT_HOP_LENGTH = 1280
|
||||||
|
MAX_SECONDS = 20.0
|
||||||
|
MODE_OFFLINE = "Offline (höchste Qualität)"
|
||||||
|
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
|
||||||
|
|
||||||
|
|
||||||
|
def _load_model() -> XVC:
|
||||||
|
speaker_config = hf_hub_download(
|
||||||
|
repo_id=SPACE_REPO,
|
||||||
|
repo_type="space",
|
||||||
|
filename=f"{SPEAKER_SUBDIR}/configuration.json",
|
||||||
|
)
|
||||||
|
hf_hub_download(
|
||||||
|
repo_id=SPACE_REPO,
|
||||||
|
repo_type="space",
|
||||||
|
filename=f"{SPEAKER_SUBDIR}/pretrained_eres2net.ckpt",
|
||||||
|
)
|
||||||
|
checkpoint = hf_hub_download(repo_id=MODEL_REPO, filename="xvc.pt")
|
||||||
|
|
||||||
|
cfg = OmegaConf.load(os.path.join(HERE, "configs", "xvc.yaml"))
|
||||||
|
cfg["model"]["generator"].pop("loss_config", None)
|
||||||
|
cfg["model"].pop("discriminator", None)
|
||||||
|
cfg["model"]["generator"]["speaker_encoder"]["pretrained_dir"] = os.path.dirname(speaker_config)
|
||||||
|
infer_cfg = os.path.join(tempfile.gettempdir(), "xvc_inference.yaml")
|
||||||
|
OmegaConf.save(cfg, infer_cfg)
|
||||||
|
|
||||||
|
loaded = XVC.load_from_checkpoint(infer_cfg, checkpoint, device=torch.device("cuda"))
|
||||||
|
loaded.remove_weight_norm()
|
||||||
|
loaded = loaded.eval().to("cuda")
|
||||||
|
log.info("X-VC model ready on %s", torch.cuda.get_device_name(0))
|
||||||
|
return loaded
|
||||||
|
|
||||||
|
|
||||||
|
MODEL = _load_model()
|
||||||
|
|
||||||
|
|
||||||
|
def _prepare_wav(path: str) -> np.ndarray:
|
||||||
|
wav = load_audio(path, sampling_rate=SAMPLE_RATE, volume_normalize=False)
|
||||||
|
if wav is None or len(wav) == 0:
|
||||||
|
raise gr.Error("Die Audiodatei konnte nicht gelesen werden.")
|
||||||
|
wav = wav[: int(MAX_SECONDS * SAMPLE_RATE)]
|
||||||
|
wav = audio_volume_normalize(wav)
|
||||||
|
wav = audio_highpass_filter(wav, SAMPLE_RATE, 40)
|
||||||
|
remainder = len(wav) % LATENT_HOP_LENGTH
|
||||||
|
if remainder:
|
||||||
|
wav = np.pad(wav, (0, LATENT_HOP_LENGTH - remainder), mode="constant")
|
||||||
|
return wav.astype(np.float32)
|
||||||
|
|
||||||
|
|
||||||
|
def _tensor(wav: np.ndarray) -> torch.Tensor:
|
||||||
|
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
|
||||||
|
|
||||||
|
|
||||||
|
def _write_wav(audio: np.ndarray) -> str:
|
||||||
|
os.makedirs("/output", exist_ok=True)
|
||||||
|
path = os.path.join("/output", f"xvc-{int(time.time() * 1000)}.wav")
|
||||||
|
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), SAMPLE_RATE, subtype="PCM_16")
|
||||||
|
return path
|
||||||
|
|
||||||
|
|
||||||
|
@torch.inference_mode()
|
||||||
|
def convert(
|
||||||
|
source_audio: str,
|
||||||
|
reference_audio: str,
|
||||||
|
mode: str = MODE_OFFLINE,
|
||||||
|
chunk_ms: int = 2400,
|
||||||
|
current_ms: int = 120,
|
||||||
|
future_ms: int = 100,
|
||||||
|
smooth_ms: int = 20,
|
||||||
|
progress=gr.Progress(track_tqdm=True),
|
||||||
|
) -> Tuple[str, str]:
|
||||||
|
if not source_audio:
|
||||||
|
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
|
||||||
|
if not reference_audio:
|
||||||
|
raise gr.Error("Bitte eine Referenzdatei mit der Zielstimme hochladen.")
|
||||||
|
|
||||||
|
source_np = _prepare_wav(source_audio)
|
||||||
|
reference_np = _prepare_wav(reference_audio)
|
||||||
|
source_wav = _tensor(source_np)
|
||||||
|
target_wav = _tensor(reference_np)
|
||||||
|
seconds = len(source_np) / SAMPLE_RATE
|
||||||
|
started = time.time()
|
||||||
|
|
||||||
|
if mode == MODE_STREAMING:
|
||||||
|
history_ms = int(chunk_ms) - int(current_ms) - int(smooth_ms) - int(future_ms)
|
||||||
|
if history_ms < 0:
|
||||||
|
raise gr.Error("Fenster muss mindestens Current + Lookahead + Crossfade umfassen.")
|
||||||
|
speaker_condition, frame_condition = precompute_conditions(MODEL, target_wav, target_wav)
|
||||||
|
recon, latency_ms = run_streaming(
|
||||||
|
model=MODEL,
|
||||||
|
source_wav=source_wav,
|
||||||
|
speaker_condition=speaker_condition,
|
||||||
|
frame_condition=frame_condition,
|
||||||
|
sample_rate=SAMPLE_RATE,
|
||||||
|
chunk_ms=int(chunk_ms),
|
||||||
|
current_ms=int(current_ms),
|
||||||
|
future_ms=int(future_ms),
|
||||||
|
smooth_ms=int(smooth_ms),
|
||||||
|
)
|
||||||
|
elapsed = time.time() - started
|
||||||
|
latency = np.asarray(latency_ms, dtype=np.float64)
|
||||||
|
report = (
|
||||||
|
f"**Streaming** · {len(latency)} Chunks · Mittel **{latency.mean():.0f} ms**, "
|
||||||
|
f"P95 **{np.percentile(latency, 95):.0f} ms** · insgesamt {elapsed:.2f} s "
|
||||||
|
f"für {seconds:.2f} s Audio (RTF {elapsed / seconds:.2f})"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
recon = run_offline(MODEL, source_wav, target_wav, target_wav)
|
||||||
|
elapsed = time.time() - started
|
||||||
|
report = (
|
||||||
|
f"**Offline** · {seconds:.2f} s Audio in {elapsed:.2f} s "
|
||||||
|
f"(RTF {elapsed / seconds:.2f})"
|
||||||
|
)
|
||||||
|
|
||||||
|
return _write_wav(to_numpy_audio(recon)), report
|
||||||
|
|
||||||
|
|
||||||
|
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
|
||||||
|
HEADER = """# X-VC — Voice Changer
|
||||||
|
|
||||||
|
Die **Quelle** liefert Text, Aussprache und Timing. Die **Referenz** liefert die
|
||||||
|
Zielstimme. X-VC arbeitet Audio-zu-Audio ohne Transkript oder Training.
|
||||||
|
|
||||||
|
[Paper](https://arxiv.org/abs/2604.12456) · [Modell](https://huggingface.co/chenxie95/X-VC) ·
|
||||||
|
[Code](https://github.com/Jerrister/X-VC)
|
||||||
|
"""
|
||||||
|
|
||||||
|
with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as demo:
|
||||||
|
with gr.Column(elem_id="col-container"):
|
||||||
|
gr.Markdown(HEADER)
|
||||||
|
with gr.Row():
|
||||||
|
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
|
||||||
|
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
|
||||||
|
run = gr.Button("Stimme umwandeln", variant="primary")
|
||||||
|
output = gr.Audio(label="Umgewandelte Sprache", type="filepath", autoplay=False)
|
||||||
|
report = gr.Markdown()
|
||||||
|
with gr.Accordion("Erweiterte Einstellungen", open=False):
|
||||||
|
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
|
||||||
|
with gr.Row():
|
||||||
|
current_ms = gr.Slider(40, 640, value=120, step=40, label="Aktueller Chunk (ms)")
|
||||||
|
chunk_ms = gr.Slider(800, 4800, value=2400, step=200, label="Gesamtfenster (ms)")
|
||||||
|
with gr.Row():
|
||||||
|
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
|
||||||
|
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
|
||||||
|
gr.Markdown("Die ersten 20 Sekunden jeder Datei werden verarbeitet. Ausgabe: 16-kHz-WAV.")
|
||||||
|
run.click(
|
||||||
|
convert,
|
||||||
|
inputs=[source, reference, mode, chunk_ms, current_ms, future_ms, smooth_ms],
|
||||||
|
outputs=[output, report],
|
||||||
|
api_name="convert",
|
||||||
|
)
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
demo.queue(default_concurrency_limit=1).launch(
|
||||||
|
server_name="0.0.0.0",
|
||||||
|
server_port=8009,
|
||||||
|
show_error=True,
|
||||||
|
)
|
||||||
@@ -0,0 +1,39 @@
|
|||||||
|
services:
|
||||||
|
xvc-studio:
|
||||||
|
build: .
|
||||||
|
image: mike-ai/xvc-studio:2026-09-09
|
||||||
|
container_name: mike-ai-xvc-studio
|
||||||
|
restart: "no"
|
||||||
|
labels:
|
||||||
|
com.mike-ai.voice-change-worker: xvc
|
||||||
|
environment:
|
||||||
|
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
|
||||||
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
|
HF_HOME: /models/huggingface
|
||||||
|
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
|
||||||
|
ports:
|
||||||
|
- "127.0.0.1:8009:8009"
|
||||||
|
volumes:
|
||||||
|
- /data/voice/xvc/huggingface:/models/huggingface
|
||||||
|
- /data/voice/xvc/output:/output
|
||||||
|
healthcheck:
|
||||||
|
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8009/ >/dev/null"]
|
||||||
|
interval: 5s
|
||||||
|
timeout: 3s
|
||||||
|
start_period: 600s
|
||||||
|
retries: 3
|
||||||
|
deploy:
|
||||||
|
resources:
|
||||||
|
reservations:
|
||||||
|
devices:
|
||||||
|
- driver: nvidia
|
||||||
|
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
|
||||||
|
capabilities: [gpu]
|
||||||
|
networks:
|
||||||
|
frontend:
|
||||||
|
aliases: [xvc-studio]
|
||||||
|
|
||||||
|
networks:
|
||||||
|
frontend:
|
||||||
|
name: mike-ai_frontend
|
||||||
|
external: true
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
"""Small inference-only replacement for X-VC's training logger.
|
||||||
|
|
||||||
|
The upstream logger imports WandB, Matplotlib and TensorBoard at module import
|
||||||
|
time although model inference only uses the normal logging functions.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
|
||||||
|
|
||||||
|
logging.basicConfig(level=logging.INFO)
|
||||||
|
_logger = logging.getLogger("xvc")
|
||||||
|
|
||||||
|
debug = _logger.debug
|
||||||
|
info = _logger.info
|
||||||
|
warn = _logger.warning
|
||||||
|
warning = _logger.warning
|
||||||
|
error = _logger.error
|
||||||
|
|
||||||
|
|
||||||
|
def init(*_args, **_kwargs):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def write_audio(*_args, **_kwargs):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def write_loss(*_args, **_kwargs):
|
||||||
|
return None
|
||||||
@@ -34,6 +34,8 @@ SEPARATOR_LABEL_KEY = "com.mike-ai.stem-separator"
|
|||||||
SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
|
SEPARATOR_WORKER = os.environ.get("SEPARATOR_WORKER", "").strip()
|
||||||
VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
|
VOICE_LABEL_KEY = "com.mike-ai.voice-worker"
|
||||||
VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
|
VOICE_WORKER = os.environ.get("VOICE_WORKER", "").strip()
|
||||||
|
VOICE_CHANGE_LABEL_KEY = "com.mike-ai.voice-change-worker"
|
||||||
|
VOICE_CHANGE_WORKER = os.environ.get("VOICE_CHANGE_WORKER", "").strip()
|
||||||
LOCK = threading.Lock()
|
LOCK = threading.Lock()
|
||||||
log = logging.getLogger("profile-controller")
|
log = logging.getLogger("profile-controller")
|
||||||
|
|
||||||
@@ -135,6 +137,17 @@ def voice_container() -> dict:
|
|||||||
return matches[0]
|
return matches[0]
|
||||||
|
|
||||||
|
|
||||||
|
def voice_change_container() -> dict:
|
||||||
|
if not VOICE_CHANGE_WORKER:
|
||||||
|
raise RuntimeError("voice-change worker is not configured")
|
||||||
|
matches = [item for item in labelled_containers(VOICE_CHANGE_LABEL_KEY)
|
||||||
|
if item.get("Labels", {}).get(VOICE_CHANGE_LABEL_KEY) == VOICE_CHANGE_WORKER]
|
||||||
|
if len(matches) != 1:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"expected exactly one voice-change worker {VOICE_CHANGE_WORKER!r}, found {len(matches)}")
|
||||||
|
return matches[0]
|
||||||
|
|
||||||
|
|
||||||
def stop_music_if_configured() -> None:
|
def stop_music_if_configured() -> None:
|
||||||
if MUSIC_WORKER:
|
if MUSIC_WORKER:
|
||||||
stop_container(music_container(), timeout=30)
|
stop_container(music_container(), timeout=30)
|
||||||
@@ -150,6 +163,11 @@ def stop_voice_if_configured() -> None:
|
|||||||
stop_container(voice_container(), timeout=30)
|
stop_container(voice_container(), timeout=30)
|
||||||
|
|
||||||
|
|
||||||
|
def stop_voice_change_if_configured() -> None:
|
||||||
|
if VOICE_CHANGE_WORKER:
|
||||||
|
stop_container(voice_change_container(), timeout=30)
|
||||||
|
|
||||||
|
|
||||||
def stop_container(item: dict, timeout: int = 120) -> None:
|
def stop_container(item: dict, timeout: int = 120) -> None:
|
||||||
if item.get("State") != "running":
|
if item.get("State") != "running":
|
||||||
return
|
return
|
||||||
@@ -190,6 +208,7 @@ def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
|
|||||||
stop_music_if_configured()
|
stop_music_if_configured()
|
||||||
stop_separator_if_configured()
|
stop_separator_if_configured()
|
||||||
stop_voice_if_configured()
|
stop_voice_if_configured()
|
||||||
|
stop_voice_change_if_configured()
|
||||||
for other in image_containers():
|
for other in image_containers():
|
||||||
if other["Id"] != item["Id"]:
|
if other["Id"] != item["Id"]:
|
||||||
stop_container(other, timeout=20)
|
stop_container(other, timeout=20)
|
||||||
@@ -217,6 +236,7 @@ def set_music_worker(running: bool) -> dict:
|
|||||||
stop_container(tts_container(), timeout=30)
|
stop_container(tts_container(), timeout=30)
|
||||||
stop_separator_if_configured()
|
stop_separator_if_configured()
|
||||||
stop_voice_if_configured()
|
stop_voice_if_configured()
|
||||||
|
stop_voice_change_if_configured()
|
||||||
start_container(item)
|
start_container(item)
|
||||||
else:
|
else:
|
||||||
stop_container(item, timeout=30)
|
stop_container(item, timeout=30)
|
||||||
@@ -236,6 +256,7 @@ def set_separator_worker(running: bool) -> dict:
|
|||||||
stop_container(tts_container(), timeout=30)
|
stop_container(tts_container(), timeout=30)
|
||||||
stop_music_if_configured()
|
stop_music_if_configured()
|
||||||
stop_voice_if_configured()
|
stop_voice_if_configured()
|
||||||
|
stop_voice_change_if_configured()
|
||||||
start_container(item)
|
start_container(item)
|
||||||
else:
|
else:
|
||||||
stop_container(item, timeout=30)
|
stop_container(item, timeout=30)
|
||||||
@@ -244,7 +265,7 @@ def set_separator_worker(running: bool) -> dict:
|
|||||||
|
|
||||||
|
|
||||||
def set_voice_worker(running: bool) -> dict:
|
def set_voice_worker(running: bool) -> dict:
|
||||||
"""Start Vevo2 exclusively, or stop it before LLM restoration."""
|
"""Start OmniVoice exclusively, or stop it before LLM restoration."""
|
||||||
with LOCK:
|
with LOCK:
|
||||||
item = voice_container()
|
item = voice_container()
|
||||||
if running:
|
if running:
|
||||||
@@ -255,6 +276,7 @@ def set_voice_worker(running: bool) -> dict:
|
|||||||
stop_container(tts_container(), timeout=30)
|
stop_container(tts_container(), timeout=30)
|
||||||
stop_music_if_configured()
|
stop_music_if_configured()
|
||||||
stop_separator_if_configured()
|
stop_separator_if_configured()
|
||||||
|
stop_voice_change_if_configured()
|
||||||
start_container(item)
|
start_container(item)
|
||||||
else:
|
else:
|
||||||
stop_container(item, timeout=30)
|
stop_container(item, timeout=30)
|
||||||
@@ -262,6 +284,26 @@ def set_voice_worker(running: bool) -> dict:
|
|||||||
"state": "running" if running else "stopped"}
|
"state": "running" if running else "stopped"}
|
||||||
|
|
||||||
|
|
||||||
|
def set_voice_change_worker(running: bool) -> dict:
|
||||||
|
"""Start X-VC exclusively, or stop it before another mode is loaded."""
|
||||||
|
with LOCK:
|
||||||
|
item = voice_change_container()
|
||||||
|
if running:
|
||||||
|
for profile_item in containers().values():
|
||||||
|
stop_container(profile_item)
|
||||||
|
for worker in image_containers():
|
||||||
|
stop_container(worker, timeout=20)
|
||||||
|
stop_container(tts_container(), timeout=30)
|
||||||
|
stop_music_if_configured()
|
||||||
|
stop_separator_if_configured()
|
||||||
|
stop_voice_if_configured()
|
||||||
|
start_container(item)
|
||||||
|
else:
|
||||||
|
stop_container(item, timeout=30)
|
||||||
|
return {"voice_change_worker": VOICE_CHANGE_WORKER,
|
||||||
|
"state": "running" if running else "stopped"}
|
||||||
|
|
||||||
|
|
||||||
def active_profile(items: dict[str, dict] | None = None) -> str | None:
|
def active_profile(items: dict[str, dict] | None = None) -> str | None:
|
||||||
items = items or containers()
|
items = items or containers()
|
||||||
active = [name for name, item in items.items() if item.get("State") == "running"]
|
active = [name for name, item in items.items() if item.get("State") == "running"]
|
||||||
@@ -280,6 +322,7 @@ def activate(profile: str) -> dict:
|
|||||||
stop_music_if_configured()
|
stop_music_if_configured()
|
||||||
stop_separator_if_configured()
|
stop_separator_if_configured()
|
||||||
stop_voice_if_configured()
|
stop_voice_if_configured()
|
||||||
|
stop_voice_change_if_configured()
|
||||||
start_container(tts_container())
|
start_container(tts_container())
|
||||||
items = containers()
|
items = containers()
|
||||||
missing = [name for name in ALLOWED if name not in items]
|
missing = [name for name in ALLOWED if name not in items]
|
||||||
@@ -365,6 +408,13 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
"unhealthy" if "(unhealthy)" in voice_status else
|
"unhealthy" if "(unhealthy)" in voice_status else
|
||||||
"starting" if voice.get("State") == "running" else
|
"starting" if voice.get("State") == "running" else
|
||||||
"stopped")
|
"stopped")
|
||||||
|
voice_change = voice_change_container() if VOICE_CHANGE_WORKER else {}
|
||||||
|
voice_change_status = voice_change.get("Status", "")
|
||||||
|
voice_change_health = ("disabled" if not VOICE_CHANGE_WORKER else
|
||||||
|
"healthy" if "(healthy)" in voice_change_status else
|
||||||
|
"unhealthy" if "(unhealthy)" in voice_change_status else
|
||||||
|
"starting" if voice_change.get("State") == "running" else
|
||||||
|
"stopped")
|
||||||
self.reply(200, {"active_profile": active_profile(items),
|
self.reply(200, {"active_profile": active_profile(items),
|
||||||
"music_worker": music.get("State", "disabled"),
|
"music_worker": music.get("State", "disabled"),
|
||||||
"music_health": music_health,
|
"music_health": music_health,
|
||||||
@@ -372,6 +422,8 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
"separator_health": separator_health,
|
"separator_health": separator_health,
|
||||||
"voice_worker": voice.get("State", "disabled"),
|
"voice_worker": voice.get("State", "disabled"),
|
||||||
"voice_health": voice_health,
|
"voice_health": voice_health,
|
||||||
|
"voice_change_worker": voice_change.get("State", "disabled"),
|
||||||
|
"voice_change_health": voice_change_health,
|
||||||
"profiles": {name: items.get(name, {}).get(
|
"profiles": {name: items.get(name, {}).get(
|
||||||
"State", "missing") for name in ALLOWED}})
|
"State", "missing") for name in ALLOWED}})
|
||||||
except Exception as exc:
|
except Exception as exc:
|
||||||
@@ -410,6 +462,13 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
log.exception("voice worker transition failed")
|
log.exception("voice worker transition failed")
|
||||||
self.reply(503, {"error": str(exc)})
|
self.reply(503, {"error": str(exc)})
|
||||||
return
|
return
|
||||||
|
if self.path in {"/workers/voice-change/start", "/workers/voice-change/stop"}:
|
||||||
|
try:
|
||||||
|
self.reply(200, set_voice_change_worker(self.path.endswith("/start")))
|
||||||
|
except Exception as exc:
|
||||||
|
log.exception("voice-change worker transition failed")
|
||||||
|
self.reply(503, {"error": str(exc)})
|
||||||
|
return
|
||||||
worker_paths = {
|
worker_paths = {
|
||||||
"/workers/image/start": (IMAGE_WORKER, True),
|
"/workers/image/start": (IMAGE_WORKER, True),
|
||||||
"/workers/image/stop": (IMAGE_WORKER, False),
|
"/workers/image/stop": (IMAGE_WORKER, False),
|
||||||
|
|||||||
@@ -82,6 +82,7 @@ start_proxy 7861 music-ui:3000
|
|||||||
start_proxy 7862 music-worker:7860
|
start_proxy 7862 music-worker:7860
|
||||||
start_proxy 8007 stem-separator:8080
|
start_proxy 8007 stem-separator:8080
|
||||||
start_proxy 8008 voice-studio:8008
|
start_proxy 8008 voice-studio:8008
|
||||||
|
start_proxy 8009 xvc-studio:8009
|
||||||
start_proxy 8202 mcp-athena-operator:8000
|
start_proxy 8202 mcp-athena-operator:8000
|
||||||
start_proxy 9443 portainer:9443
|
start_proxy 9443 portainer:9443
|
||||||
|
|
||||||
|
|||||||
@@ -29,6 +29,7 @@ MUSIC_ORIGINAL_UI_URL = os.getenv(
|
|||||||
)
|
)
|
||||||
SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
|
SEPARATOR_UI_URL = os.getenv("SEPARATOR_UI_URL", "http://192.168.1.212:8007/")
|
||||||
VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
|
VOICE_UI_URL = os.getenv("VOICE_UI_URL", "http://192.168.1.212:8008/")
|
||||||
|
VOICE_CHANGE_UI_URL = os.getenv("VOICE_CHANGE_UI_URL", "http://192.168.1.212:8009/")
|
||||||
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
|
HOST_PROC = Path(os.getenv("HOST_PROC", "/host/proc"))
|
||||||
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
|
HOST_DATA = os.getenv("HOST_DATA", "/host/data")
|
||||||
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
|
HOST_MODELS = Path(os.getenv("HOST_MODELS", "/host/models"))
|
||||||
@@ -278,7 +279,7 @@ def router_status() -> tuple[dict[str, Any], str | None]:
|
|||||||
|
|
||||||
|
|
||||||
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
|
def change_mode(mode: str) -> tuple[int, dict[str, Any]]:
|
||||||
if mode not in {"llm", "music", "separation", "voice"}:
|
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
|
||||||
return 400, {"error": "invalid mode"}
|
return 400, {"error": "invalid mode"}
|
||||||
headers = {"Accept": "application/json", "Content-Type": "application/json"}
|
headers = {"Accept": "application/json", "Content-Type": "application/json"}
|
||||||
if ROUTER_API_KEY:
|
if ROUTER_API_KEY:
|
||||||
@@ -636,7 +637,7 @@ HTML = r'''<!doctype html>
|
|||||||
</style></head><body><main>
|
</style></head><body><main>
|
||||||
<div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div>
|
<div class="top"><div><div class="eyebrow">Mike AI · Live Telemetry</div><h1>Athena llama.cpp Dashboard</h1></div><div class="live"><span class="dot" id="dot"></span><span id="updated">verbinde …</span></div></div>
|
||||||
<section class="grid">
|
<section class="grid">
|
||||||
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><button id="voiceMode" onclick="setMode('voice')">Voice Studio</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span><span id="voiceOpen" hidden><a class="stable" href="__VOICE_UI_URL__" target="_blank" rel="noopener">Voice Studio öffnen</a></span></div></div></article>
|
<article class="card span12"><div class="mode-row"><div><div class="label">Athena Betriebsmodus</div><div class="value" id="operatingMode">–</div><div class="sub" id="modeStatus">Status wird geladen …</div></div><div class="mode-buttons"><button id="llmMode" onclick="setMode('llm')">LLM-Betrieb</button><button id="musicMode" onclick="setMode('music')">Musikstudio</button><button id="separationMode" onclick="setMode('separation')">Audio trennen</button><button id="voiceMode" onclick="setMode('voice')">Voice Studio</button><button id="voiceChangeMode" onclick="setMode('voicechange')">Voice Changer</button><span id="musicOpen" hidden><a class="stable" href="__MUSIC_ORIGINAL_UI_URL__" target="_blank" rel="noopener">Original UI · stabil</a><a class="experimental" href="__MUSIC_COMMUNITY_UI_URL__" target="_blank" rel="noopener">Community UI · experimentell</a></span><span id="separatorOpen" hidden><a class="stable" href="__SEPARATOR_UI_URL__" target="_blank" rel="noopener">Separator öffnen</a></span><span id="voiceOpen" hidden><a class="stable" href="__VOICE_UI_URL__" target="_blank" rel="noopener">Voice Studio öffnen</a></span><span id="voiceChangeOpen" hidden><a class="experimental" href="__VOICE_CHANGE_UI_URL__" target="_blank" rel="noopener">X-VC öffnen</a></span></div></div></article>
|
||||||
<article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article>
|
<article class="card span3"><div class="label">Aktives Profil</div><div class="value" id="profile">–</div><div class="sub" id="profileSub">Router wird abgefragt</div></article>
|
||||||
<article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article>
|
<article class="card span3"><div class="label">Modell</div><div class="value" id="model">–</div><div class="sub" id="modelSub">–</div></article>
|
||||||
<article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article>
|
<article class="card span3"><div class="label">CPU</div><div class="value" id="cpu">–</div><div class="bar"><div class="fill" id="cpuBar"></div></div><div class="sub" id="load">–</div></article>
|
||||||
@@ -677,11 +678,30 @@ const imagePhaseLabel=p=>({"stopping-qwen":"Qwen wird entladen","loading-image":
|
|||||||
let modeBusy=false;
|
let modeBusy=false;
|
||||||
async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
|
async function setMode(mode){if(modeBusy)return;modeBusy=true;$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=true;$('modeStatus').textContent='Umschaltung angefordert …';try{let r=await fetch('/api/mode',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify({mode})});let d=await r.json();if(!r.ok)throw Error(d?.error?.message||d?.error||`HTTP ${r.status}`);$('modeStatus').textContent='Umschaltung läuft …'}catch(e){$('modeStatus').textContent=e.message;$('modeStatus').classList.add('mode-error')}finally{modeBusy=false;setTimeout(refresh,250)}}
|
||||||
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
|
async function refresh(){try{let r=await fetch('/api/status',{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),c=d.cpu||{},m=c.memory||{},rt=d.router||{},up=rt.upstream||{},q=rt.qwen||{},lr=d.llama_runtime||{},img=rt.image||{},imageActive=img.phase&&img.phase!=='idle';let md=rt.mode||{},switchingMode=md.phase&&md.phase!=='ready',modeName=md.active==='music'?'Musikstudio':md.active==='separation'?'Stimmtrennung':md.active==='voice'?'Voice Studio':'LLM-Betrieb';$('operatingMode').textContent=modeName;$('modeStatus').textContent=switchingMode?`Umschaltung: ${md.phase}`:(md.last_error||`Musik: ${md.music_worker||'–'} · Separator: ${md.separator_worker||'–'} · Voice: ${md.voice_worker||'–'}${md.return_profile?` · Rückkehr zu ${md.return_profile}`:''}`);$('modeStatus').classList.toggle('mode-error',!!md.last_error);$('llmMode').classList.toggle('active',md.active==='llm');$('musicMode').classList.toggle('active',md.active==='music');$('separationMode').classList.toggle('active',md.active==='separation');$('voiceMode').classList.toggle('active',md.active==='voice');$('llmMode').disabled=$('musicMode').disabled=$('separationMode').disabled=$('voiceMode').disabled=modeBusy||switchingMode||!md.enabled;$('musicOpen').hidden=md.active!=='music';$('separatorOpen').hidden=md.active!=='separation';$('voiceOpen').hidden=md.active!=='voice';$('profile').textContent=imageActive?'Bildgenerierung':(rt.current_profile||'nicht geladen');$('profileSub').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching?`Wechsel zu ${rt.switching}`:`Kontext: ${up.ctx?up.ctx.toLocaleString('de-DE'):'–'} Token`);$('model').textContent=imageActive?(img.model||'Bildmodell'):(up.model||'–');$('modelSub').textContent=imageActive?`${img.model_loaded?'geladen':'wird vorbereitet'} · Worker ${img.worker||'–'}`:(lr.model_file|| (up.reachable?'llama.cpp erreichbar':'llama.cpp nicht erreichbar'));$('cpu').textContent=pct(c.usage_percent);$('cpuBar').style.width=`${c.usage_percent||0}%`;$('load').textContent=`${c.logical_cpus||'–'} Threads · Load ${(c.load||[]).join(' / ')}`;let rp=m.total?m.used/m.total*100:0;$('ram').textContent=pct(rp);$('ramBar').style.width=`${rp}%`;$('ramSub').textContent=`${gib(m.used)} / ${gib(m.total)}`;$('gpuCards').innerHTML=(d.gpus||[]).map(gpuCard).join('')||'<article class="card span12 error">Keine GPU-Daten verfügbar</article>';$('availability').textContent=imageActive?imagePhaseLabel(img.phase):(q.available?'bereit':'nicht bereit');$('availability').className=`value status ${(imageActive||q.available)?'':'bad'}`;$('activeChats').textContent=q.active_chats??'–';$('routerUptime').textContent=dur(rt.uptime_seconds);$('switching').textContent=imageActive?imagePhaseLabel(img.phase):(rt.switching||'nein');let disk=c.disk_data||{},dp=disk.total?disk.used/disk.total*100:null;$('dataDisk').textContent=pct(dp);$('processes').innerHTML=(d.gpu_processes||[]).map(p=>`<tr><td>${(d.gpus||[]).find(g=>g.uuid===p.gpu_uuid)?.index??'–'}</td><td>${p.name}</td><td>${p.pid}</td><td>${p.memory_mib??'–'} MiB</td></tr>`).join('')||'<tr><td colspan="4">Keine Compute-Prozesse gemeldet</td></tr>';let runtime=[['Modell-Datei',lr.model_file],['PID',lr.pid],['Kontext',lr.context_size?lr.context_size.toLocaleString('de-DE'):'–'],['Batch / µBatch',`${lr.batch_size??'–'} / ${lr.ubatch_size??'–'}`],['Parallel',lr.parallel],['Threads',`${lr.threads??'–'} / ${lr.threads_batch??'–'}`],['Geräte',lr.device],['Tensor-Split',lr.tensor_split],['KV-Cache',`${lr.cache_k??'–'} / ${lr.cache_v??'–'}`],['Flash Attention',lr.flash_attention?'an':'aus'],['Prompt-Cache',lr.prompt_cache?'an':'aus'],['MTP Draft',lr.mtp_draft_tokens]];$('runtime').innerHTML=runtime.map(([k,v])=>`<div class="metric"><b>${v??'–'}</b><span>${k}</span></div>`).join('');let es=Object.entries(d.errors||{}).filter(([,v])=>v);$('errors').hidden=!es.length;$('errors').textContent=es.map(([k,v])=>`${k}: ${v}`).join('\n');$('updated').textContent=`Live · ${new Date(d.timestamp*1000).toLocaleTimeString('de-DE')}`;$('dot').style.background='var(--green)'}catch(e){$('updated').textContent=`Verbindung gestört: ${e.message}`;$('dot').style.background='var(--red)'}}refresh();setInterval(refresh,1000);
|
||||||
|
</script><script>
|
||||||
|
const baseSetMode=setMode;
|
||||||
|
setMode=async function(mode){
|
||||||
|
if(modeBusy)return;
|
||||||
|
for(const id of ['llmMode','musicMode','separationMode','voiceMode','voiceChangeMode'])$(id).disabled=true;
|
||||||
|
await baseSetMode(mode);
|
||||||
|
};
|
||||||
|
async function refreshVoiceChange(){
|
||||||
|
try{
|
||||||
|
const response=await fetch('/api/status',{cache:'no-store'}); if(!response.ok)return;
|
||||||
|
const data=await response.json(),mode=data.router?.mode||{},active=mode.active==='voicechange';
|
||||||
|
$('voiceChangeMode').classList.toggle('active',active);
|
||||||
|
$('voiceChangeMode').disabled=modeBusy||(mode.phase&&mode.phase!=='ready')||!mode.enabled;
|
||||||
|
$('voiceChangeOpen').hidden=!active;
|
||||||
|
if(active)$('operatingMode').textContent='Voice Changer';
|
||||||
|
}catch(_error){}
|
||||||
|
}
|
||||||
|
setTimeout(()=>{refreshVoiceChange();setInterval(refreshVoiceChange,1000)},150);
|
||||||
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''.replace(
|
</script><script src="/full.js"></script><script src="/history.js"></script></body></html>'''.replace(
|
||||||
"__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL
|
"__MUSIC_ORIGINAL_UI_URL__", MUSIC_ORIGINAL_UI_URL
|
||||||
).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL
|
).replace("__MUSIC_COMMUNITY_UI_URL__", MUSIC_COMMUNITY_UI_URL
|
||||||
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL
|
).replace("__SEPARATOR_UI_URL__", SEPARATOR_UI_URL
|
||||||
).replace("__VOICE_UI_URL__", VOICE_UI_URL)
|
).replace("__VOICE_UI_URL__", VOICE_UI_URL
|
||||||
|
).replace("__VOICE_CHANGE_UI_URL__", VOICE_CHANGE_UI_URL)
|
||||||
|
|
||||||
|
|
||||||
FULL_JS = r'''
|
FULL_JS = r'''
|
||||||
|
|||||||
+58
-11
@@ -111,6 +111,7 @@ ENABLE_MUSIC_MODE = os.environ.get(
|
|||||||
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
|
MUSIC_START_TIMEOUT = float(os.environ.get("MUSIC_START_TIMEOUT", "600"))
|
||||||
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
|
SEPARATOR_START_TIMEOUT = float(os.environ.get("SEPARATOR_START_TIMEOUT", "600"))
|
||||||
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
|
VOICE_START_TIMEOUT = float(os.environ.get("VOICE_START_TIMEOUT", "600"))
|
||||||
|
VOICE_CHANGE_START_TIMEOUT = float(os.environ.get("VOICE_CHANGE_START_TIMEOUT", "600"))
|
||||||
|
|
||||||
# Optional worker APIs. The clean Docker baseline deliberately ships only
|
# Optional worker APIs. The clean Docker baseline deliberately ships only
|
||||||
# text/multimodal chat; absent workers must fail explicitly instead of trying
|
# text/multimodal chat; absent workers must fail explicitly instead of trying
|
||||||
@@ -383,6 +384,27 @@ def _voice_worker_health() -> str:
|
|||||||
return "unknown"
|
return "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def _voice_change_worker_state() -> str:
|
||||||
|
if not PROFILE_CONTROL_URL:
|
||||||
|
return "unsupported"
|
||||||
|
try:
|
||||||
|
return str(_profile_controller_request("GET", "/status").get(
|
||||||
|
"voice_change_worker", "missing"))
|
||||||
|
except Exception as exc:
|
||||||
|
log.warning("Voice-Change-Worker-Status nicht verfügbar: %s", exc)
|
||||||
|
return "unknown"
|
||||||
|
|
||||||
|
|
||||||
|
def _voice_change_worker_health() -> str:
|
||||||
|
if not PROFILE_CONTROL_URL:
|
||||||
|
return "unsupported"
|
||||||
|
try:
|
||||||
|
return str(_profile_controller_request("GET", "/status").get(
|
||||||
|
"voice_change_health", "unknown"))
|
||||||
|
except Exception:
|
||||||
|
return "unknown"
|
||||||
|
|
||||||
|
|
||||||
def _wait_music_ready() -> None:
|
def _wait_music_ready() -> None:
|
||||||
deadline = time.monotonic() + MUSIC_START_TIMEOUT
|
deadline = time.monotonic() + MUSIC_START_TIMEOUT
|
||||||
while time.monotonic() < deadline:
|
while time.monotonic() < deadline:
|
||||||
@@ -419,10 +441,24 @@ def _wait_voice_ready() -> None:
|
|||||||
and status.get("voice_health") == "healthy"):
|
and status.get("voice_health") == "healthy"):
|
||||||
return
|
return
|
||||||
if status.get("voice_health") == "unhealthy":
|
if status.get("voice_health") == "unhealthy":
|
||||||
raise RuntimeError("Vevo2-Container ist unhealthy")
|
raise RuntimeError("OmniVoice-Container ist unhealthy")
|
||||||
time.sleep(POLL_INTERVAL)
|
time.sleep(POLL_INTERVAL)
|
||||||
raise RuntimeError(
|
raise RuntimeError(
|
||||||
f"Vevo2 nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
|
f"OmniVoice nach {VOICE_START_TIMEOUT:.0f} s nicht bereit")
|
||||||
|
|
||||||
|
|
||||||
|
def _wait_voice_change_ready() -> None:
|
||||||
|
deadline = time.monotonic() + VOICE_CHANGE_START_TIMEOUT
|
||||||
|
while time.monotonic() < deadline:
|
||||||
|
status = _profile_controller_request("GET", "/status")
|
||||||
|
if (status.get("voice_change_worker") == "running"
|
||||||
|
and status.get("voice_change_health") == "healthy"):
|
||||||
|
return
|
||||||
|
if status.get("voice_change_health") == "unhealthy":
|
||||||
|
raise RuntimeError("X-VC-Container ist unhealthy")
|
||||||
|
time.sleep(POLL_INTERVAL)
|
||||||
|
raise RuntimeError(
|
||||||
|
f"X-VC nach {VOICE_CHANGE_START_TIMEOUT:.0f} s nicht bereit")
|
||||||
|
|
||||||
|
|
||||||
def _special_worker(mode: str) -> tuple[str, str, callable]:
|
def _special_worker(mode: str) -> tuple[str, str, callable]:
|
||||||
@@ -432,6 +468,9 @@ def _special_worker(mode: str) -> tuple[str, str, callable]:
|
|||||||
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
|
return "/workers/separator/start", _separator_worker_state(), _wait_separator_ready
|
||||||
if mode == "voice":
|
if mode == "voice":
|
||||||
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
|
return "/workers/voice/start", _voice_worker_state(), _wait_voice_ready
|
||||||
|
if mode == "voicechange":
|
||||||
|
return ("/workers/voice-change/start", _voice_change_worker_state(),
|
||||||
|
_wait_voice_change_ready)
|
||||||
raise ValueError(f"unbekannter Spezialmodus: {mode}")
|
raise ValueError(f"unbekannter Spezialmodus: {mode}")
|
||||||
|
|
||||||
|
|
||||||
@@ -439,11 +478,11 @@ def set_operating_mode(mode: str) -> dict:
|
|||||||
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
|
"""Atomarer Wechsel zwischen LLM und den exklusiven GPU-Werkzeugen."""
|
||||||
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
||||||
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
||||||
if mode not in {"llm", "music", "separation", "voice"}:
|
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
|
||||||
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
|
raise ValueError("Modus muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
|
||||||
with STATE.lock:
|
with STATE.lock:
|
||||||
STATE.mode_error = None
|
STATE.mode_error = None
|
||||||
if mode in {"music", "separation", "voice"}:
|
if mode in {"music", "separation", "voice", "voicechange"}:
|
||||||
path, worker_state, wait_ready = _special_worker(mode)
|
path, worker_state, wait_ready = _special_worker(mode)
|
||||||
if STATE.mode == mode and worker_state == "running":
|
if STATE.mode == mode and worker_state == "running":
|
||||||
return {"status": "ok", "mode": mode, "changed": False}
|
return {"status": "ok", "mode": mode, "changed": False}
|
||||||
@@ -485,6 +524,7 @@ def set_operating_mode(mode: str) -> dict:
|
|||||||
_profile_controller_request("POST", "/workers/music/stop")
|
_profile_controller_request("POST", "/workers/music/stop")
|
||||||
_profile_controller_request("POST", "/workers/separator/stop")
|
_profile_controller_request("POST", "/workers/separator/stop")
|
||||||
_profile_controller_request("POST", "/workers/voice/stop")
|
_profile_controller_request("POST", "/workers/voice/stop")
|
||||||
|
_profile_controller_request("POST", "/workers/voice-change/stop")
|
||||||
_restore_qwen(profile)
|
_restore_qwen(profile)
|
||||||
STATE.mode = "llm"
|
STATE.mode = "llm"
|
||||||
STATE.mode_phase = "ready"
|
STATE.mode_phase = "ready"
|
||||||
@@ -503,8 +543,8 @@ def schedule_operating_mode(mode: str) -> tuple[bool, str]:
|
|||||||
"""Start a transition in the background so chat/UI acknowledgement is instant."""
|
"""Start a transition in the background so chat/UI acknowledgement is instant."""
|
||||||
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
if not ENABLE_MUSIC_MODE or not PROFILE_CONTROL_URL:
|
||||||
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
raise RuntimeError("Musikmodus ist nicht konfiguriert")
|
||||||
if mode not in {"llm", "music", "separation", "voice"}:
|
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
|
||||||
raise ValueError("Modus muss 'llm', 'music', 'separation' oder 'voice' sein")
|
raise ValueError("Modus muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
|
||||||
with STATE.lock:
|
with STATE.lock:
|
||||||
if STATE.mode_phase not in {"ready", "error"}:
|
if STATE.mode_phase not in {"ready", "error"}:
|
||||||
return False, STATE.mode_phase
|
return False, STATE.mode_phase
|
||||||
@@ -541,6 +581,7 @@ def _control_command(data: dict, path: str) -> str | None:
|
|||||||
return command if command in {"/athena music", "/athena stems",
|
return command if command in {"/athena music", "/athena stems",
|
||||||
"/athena separation", "/athena llm",
|
"/athena separation", "/athena llm",
|
||||||
"/athena voice",
|
"/athena voice",
|
||||||
|
"/athena voicechange", "/athena changer",
|
||||||
"/athena status"} else None
|
"/athena status"} else None
|
||||||
|
|
||||||
|
|
||||||
@@ -2018,6 +2059,8 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
"separator_health": _separator_worker_health(),
|
"separator_health": _separator_worker_health(),
|
||||||
"voice_worker": _voice_worker_state(),
|
"voice_worker": _voice_worker_state(),
|
||||||
"voice_health": _voice_worker_health(),
|
"voice_health": _voice_worker_health(),
|
||||||
|
"voice_change_worker": _voice_change_worker_state(),
|
||||||
|
"voice_change_health": _voice_change_worker_health(),
|
||||||
"return_profile": state.get("return_profile"),
|
"return_profile": state.get("return_profile"),
|
||||||
"last_error": STATE.mode_error,
|
"last_error": STATE.mode_error,
|
||||||
"enabled": ENABLE_MUSIC_MODE,
|
"enabled": ENABLE_MUSIC_MODE,
|
||||||
@@ -2027,8 +2070,8 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
try:
|
try:
|
||||||
data = json.loads(self._read_body() or b"{}")
|
data = json.loads(self._read_body() or b"{}")
|
||||||
mode = data.get("mode") if isinstance(data, dict) else None
|
mode = data.get("mode") if isinstance(data, dict) else None
|
||||||
if mode not in {"llm", "music", "separation", "voice"}:
|
if mode not in {"llm", "music", "separation", "voice", "voicechange"}:
|
||||||
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation' oder 'voice' sein")
|
raise ValueError("Feld 'mode' muss 'llm', 'music', 'separation', 'voice' oder 'voicechange' sein")
|
||||||
started, phase = schedule_operating_mode(mode)
|
started, phase = schedule_operating_mode(mode)
|
||||||
self._send_json(202 if started else 200, {
|
self._send_json(202 if started else 200, {
|
||||||
"status": "accepted" if started else "ok",
|
"status": "accepted" if started else "ok",
|
||||||
@@ -2692,11 +2735,13 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
f"Phase: {mode['phase']}. Musik-Worker: "
|
f"Phase: {mode['phase']}. Musik-Worker: "
|
||||||
f"{mode['music_worker']}. Stem-Separator: "
|
f"{mode['music_worker']}. Stem-Separator: "
|
||||||
f"{mode['separator_worker']}. Voice Studio: "
|
f"{mode['separator_worker']}. Voice Studio: "
|
||||||
f"{mode['voice_worker']}. LLM-Profil: {profile or 'entladen'}.")
|
f"{mode['voice_worker']}. Voice Changer: "
|
||||||
|
f"{mode['voice_change_worker']}. LLM-Profil: {profile or 'entladen'}.")
|
||||||
else:
|
else:
|
||||||
target = ("music" if command == "/athena music" else
|
target = ("music" if command == "/athena music" else
|
||||||
"separation" if command in {"/athena stems", "/athena separation"}
|
"separation" if command in {"/athena stems", "/athena separation"}
|
||||||
else "voice" if command == "/athena voice"
|
else "voice" if command == "/athena voice"
|
||||||
|
else "voicechange" if command in {"/athena voicechange", "/athena changer"}
|
||||||
else "llm")
|
else "llm")
|
||||||
try:
|
try:
|
||||||
started, phase = schedule_operating_mode(target)
|
started, phase = schedule_operating_mode(target)
|
||||||
@@ -2707,6 +2752,8 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
if target == "separation" else
|
if target == "separation" else
|
||||||
"Voice Studio wird gestartet. LLM und TTS werden entladen."
|
"Voice Studio wird gestartet. LLM und TTS werden entladen."
|
||||||
if target == "voice" else
|
if target == "voice" else
|
||||||
|
"Voice Changer wird gestartet. LLM und TTS werden entladen."
|
||||||
|
if target == "voicechange" else
|
||||||
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
|
"Spezialmodus wird beendet und das vorherige LLM-Profil wiederhergestellt.")
|
||||||
else:
|
else:
|
||||||
text = (f"Athena ist bereits im {target.upper()}-Modus "
|
text = (f"Athena ist bereits im {target.upper()}-Modus "
|
||||||
@@ -2999,7 +3046,7 @@ def _startup_reconcile() -> None:
|
|||||||
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
|
log.info("Startup-Retention: %d alte Bilder entfernt", len(removed))
|
||||||
|
|
||||||
special_mode = previous.get("mode")
|
special_mode = previous.get("mode")
|
||||||
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice"}:
|
if ENABLE_MUSIC_MODE and special_mode in {"music", "separation", "voice", "voicechange"}:
|
||||||
STATE.mode = special_mode
|
STATE.mode = special_mode
|
||||||
STATE.mode_phase = f"starting-{special_mode}"
|
STATE.mode_phase = f"starting-{special_mode}"
|
||||||
_set_qwen_unavailable(True)
|
_set_qwen_unavailable(True)
|
||||||
|
|||||||
Reference in New Issue
Block a user