56 Commits
Author SHA1 Message Date
Mikei386 f2052adb78 Integrate YuE2 as an exclusive Athena mode 2026-09-10 23:53:19 +02:00
Mikei386 5c34afa7fa Enable SheetSage2 audio remix in YuE2 UI 2026-09-10 22:56:24 +02:00
Mikei386 af25425aee Integrate pinned YuE2 community WebUI 2026-09-10 22:13:32 +02:00
Mikei386 72c9d8c485 Fix YuE2 playground output path 2026-09-10 21:52:21 +02:00
Mikei386 f8b1b19d4a Add isolated YuE2 playground 2026-09-10 21:49:16 +02:00
Mikei386 1904f2104f Prepare isolated YuE2 music evaluation 2026-09-10 21:25:59 +02:00
Mikei386 6bd6c48a95 Document HeartMuLa test routing and timing 2026-09-10 20:24:20 +02:00
Mikei386 fe2a93eeb1 Add isolated HeartMuLa 3B quality test 2026-09-10 20:20:51 +02:00
Mikei386 7c95dab324 Handle negated vocal prompts 2026-09-10 19:52:16 +02:00
Mikei386 436dee6f1e Tune ACE-Step XL-SFT test defaults 2026-09-10 19:49:04 +02:00
Mikei386 1d5a113158 Fix ACE-Step community UI parameter wiring 2026-09-10 19:15:33 +02:00
Mikei386 43321b6797 Document Athena 3D mode and agent operations 2026-09-10 18:31:44 +02:00
Mikei386 795101b746 Enable Applio realtime WebSockets 2026-09-10 14:47:12 +02:00
Mikei386 2a278bd5bd Expose encrypted backups in dashboard 2026-09-10 14:18:03 +02:00
Mikei386 bc2ca9af7d Add three-scenario disaster recovery 2026-09-10 14:04:56 +02:00
Mikei386 e5e5d7fa4d Document current Athena stack and storage 2026-09-10 13:32:35 +02:00
Mikei386 08ff3d7c4e Remove Beta 1 and Piper fallback 2026-09-10 13:17:57 +02:00
Mikei386 52482627be Add both Applio frontends to dashboard 2026-09-10 11:37:04 +02:00
Mikei386 d3534627f0 feat: expose Mikes Applio UI through WireGuard 2026-09-10 07:10:08 +02:00
Mikei386 0e45518e6e Remove Seed-VC after failed German quality test 2026-09-09 22:03:31 +02:00
Mikei386 4cdd837482 Fix Seed-VC startup readiness and clarify Applio models 2026-09-09 21:29:06 +02:00
Mikei386 13a6714b2b Add Seed-VC and Applio studio modes 2026-09-09 20:58:32 +02:00
Mikei386 51ed201f6c Add optional 44.1 kHz X-VC enhancement 2026-09-09 17:21:20 +02:00
Mikei386 b4f4bf37fd Document Athena containers and expand Hermes operator skill 2026-09-09 16:39:34 +02:00
Mikei386 87a2ae5704 Add X-VC voice conversion mode 2026-09-09 16:14:10 +02:00
Mikei386 17f1a08d7d Document OmniVoice cloning gate 2026-09-09 15:15:10 +02:00
Mikei386 68d02f32bd Add private Vevo2 voice studio mode 2026-09-09 13:38:00 +02:00
Mikei386 535bd751b5 Update llama.cpp runtime to build 10872 2026-09-09 12:27:36 +02:00
Mikei386 805228abfd feat: add speech noise separation 2026-09-09 08:24:12 +02:00
Mikei386 0fd1966ae3 feat: add other stem extraction target 2026-09-09 07:48:26 +02:00
Mikei386 c5bbebecc6 fix: use ffmpeg export for all MP3 stem jobs 2026-09-09 01:03:39 +02:00
Mikei386 2342495d24 feat: add target-and-remainder stem extraction 2026-09-09 01:00:01 +02:00
Mikei386 0f0e77928a fix: export Demucs stems through ffmpeg 2026-09-09 00:46:06 +02:00
Mikei386 e2f35517f8 feat: add multi-stem audio separation modes 2026-09-09 00:30:08 +02:00
Mikei386 30203bf13b Fix dashboard separation mode switch 2026-09-08 23:08:58 +02:00
Mikei386 0069b61dbb Add BS-RoFormer vocal separation mode 2026-09-08 19:23:29 +02:00
Mikei386 a4e894fe70 Add dual ACE-Step music studio interfaces 2026-09-08 18:46:54 +02:00
Mikei386 f58d61140e Add persistent Athena music mode switching 2026-09-08 17:32:04 +02:00
Mikei386 56c382f71f Persist ACE-Step XL-SFT quality defaults 2026-09-08 16:37:36 +02:00
Mikei386 eeebbd06eb Fix UI networking across gateway recreation 2026-09-08 15:56:12 +02:00
Mikei386 5e18b7776b Add ACE-Step XL SFT music experiment 2026-09-08 15:16:15 +02:00
Mikei386 2724861224 Speak aspect ratios correctly in TTS 2026-09-08 14:42:21 +02:00
Mikei386 f1ed51a302 Track tested models centrally 2026-09-08 14:26:46 +02:00
Mikei386 30fdbd4b7a Document max-VRAM IQ3 speed tests 2026-09-08 14:14:46 +02:00
Mikei386 3e3fbbe9bd Document profile-wide GSQ-RCO A-B test 2026-09-08 13:48:47 +02:00
Mikei386 edb845eb19 Document GSQ-RCO IQ3_S A-B results 2026-09-08 12:10:08 +02:00
Mikei386 f7ff14a1ce Upgrade Beta 1 to GSQ-RCO IQ3_S 2026-09-08 11:46:52 +02:00
Mikei386 5746ac0e2c Document removal of photo restoration experiment 2026-09-08 10:56:30 +02:00
Mikei386 e82e0340e4 Remove ineffective photo restoration pipeline 2026-09-08 09:40:43 +02:00
Mikei386 f82dc081c9 Disable text conditioning for faithful restoration 2026-09-08 08:52:49 +02:00
Mikei386 3220a67f1b Deduplicate Hermes image references 2026-09-08 08:45:57 +02:00
Mikei386 636e48ce93 Merge restoration policy into system message 2026-09-08 08:37:53 +02:00
Mikei386 118e32005e Add explicit HYPIR restoration profile 2026-09-07 22:52:59 +02:00
Mikei386 2ae61baec7 Prepare isolated Qwen3.8 IQ4_XS A/B tests 2026-09-07 16:22:51 +02:00
Mikei386 1844551534 Add dual-GPU FLUX 9B image pipeline 2026-09-07 15:42:45 +02:00
Mikei386 61aa20cb52 Add native Qwen3 TTS streaming for Hermes 2026-09-05 16:06:27 +02:00
126 changed files with 9487 additions and 717 deletions
+2 -9
View File
@@ -3,10 +3,9 @@ AI_BIND_ADDRESS=10.77.0.2
MODEL_DIR=/data/models MODEL_DIR=/data/models
ROUTER_API_KEY=GENERATED_BY_INSTALLER ROUTER_API_KEY=GENERATED_BY_INSTALLER
CONTROLLER_TOKEN=GENERATED_BY_INSTALLER CONTROLLER_TOKEN=GENERATED_BY_INSTALLER
FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B FLUX_COMPONENT_DIR=/data/models/FLUX.2-klein-9B-components
FLUX_TRANSFORMER_DIR=/data/models/FLUX.2-klein-9B-fp8
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98 QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
@@ -16,7 +15,6 @@ DEFAULT_REASONING_EFFORT=off
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
BETA1_MODEL_FILE=qwen3.8-27b-gsq-rco-test/Qwen3.8-27B-GSQ-RCO-IQ3_XXS-mtp.gguf
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf ULTRA_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
UNCENSORED_MODEL_FILE=qwen3.8-27b-abliterated/Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf UNCENSORED_MODEL_FILE=qwen3.8-27b-abliterated/Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf
@@ -29,9 +27,6 @@ FAST_UBATCH_SIZE=32
MEDIUM_CONTEXT=160000 MEDIUM_CONTEXT=160000
MEDIUM_BATCH_SIZE=2048 MEDIUM_BATCH_SIZE=2048
MEDIUM_UBATCH_SIZE=128 MEDIUM_UBATCH_SIZE=128
BETA1_CONTEXT=192000
BETA1_BATCH_SIZE=2048
BETA1_UBATCH_SIZE=128
LARGE_CONTEXT=192000 LARGE_CONTEXT=192000
LARGE_BATCH_SIZE=2048 LARGE_BATCH_SIZE=2048
LARGE_UBATCH_SIZE=128 LARGE_UBATCH_SIZE=128
@@ -44,7 +39,6 @@ UNCENSORED_UBATCH_SIZE=128
FAST_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b FAST_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
MEDIUM_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b MEDIUM_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
MEDIUM_TENSOR_SPLIT=85,15 MEDIUM_TENSOR_SPLIT=85,15
BETA1_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
LARGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b LARGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
LARGE_TENSOR_SPLIT=86,14 LARGE_TENSOR_SPLIT=86,14
ULTRA_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b ULTRA_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
@@ -57,7 +51,6 @@ LLAMA_THREADS_BATCH=6
FAST_PARALLEL_SLOTS=1 FAST_PARALLEL_SLOTS=1
LLAMA_CACHE_RAM_MIB=32768 LLAMA_CACHE_RAM_MIB=32768
MEDIUM_PARALLEL_SLOTS=1 MEDIUM_PARALLEL_SLOTS=1
BETA1_PARALLEL_SLOTS=1
LARGE_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1
ULTRA_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1
UNCENSORED_PARALLEL_SLOTS=1 UNCENSORED_PARALLEL_SLOTS=1
+15
View File
@@ -0,0 +1,15 @@
# Repository-Regeln für Modelltests
- Vor jedem Download, Benchmark oder neuen Profil zuerst
`docs/TESTED_MODELS.md` vollständig prüfen.
- Ein bereits verworfenes oder ersetztes Artefakt nicht erneut testen, sofern
sich nicht mindestens Runtime, Hardware, Quantisierung oder Modellrevision
konkret geändert hat. Den neuen Grund im Testbericht festhalten.
- Nach jedem Modelltest `docs/TESTED_MODELS.md` im selben Commit aktualisieren:
Datum, exaktes Repository, exakter Dateiname beziehungsweise Ollama-Tag,
Quantisierung, Kontext, Ergebnis, Entscheidung und Belegpfad.
- Ein heruntergeladenes, aber nicht belastbar getestetes Modell als
`unvollständig` eintragen; nicht stillschweigend als verworfen behandeln.
- Verworfene Gewichte erst löschen, nachdem die entscheidenden Resultate
dauerhaft dokumentiert sind.
+35 -9
View File
@@ -10,8 +10,11 @@ Sie betreibt:
- llama.cpp mit genau einem aktiven Qwen-Profil, - llama.cpp mit genau einem aktiven Qwen-Profil,
- den OpenAI-kompatiblen Profile Router, - den OpenAI-kompatiblen Profile Router,
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung, - FLUX.2 Klein 9B FP8 Beta für Textbilder und Referenzbild-Bearbeitung,
- Qwen3-TTS und Piper für Sprache, - Qwen3-TTS für Sprache,
- ACE-Step 1.5 XL-SFT als exklusiven Musikstudio-Modus,
- YuE2-3B mit Ladypoly-WebUI als zweiten, getrennten Musikstudio-Modus,
- TRELLIS.2 4B Q8 als exklusives Bild-zu-3D-Studio,
- das Athena-Dashboard, - das Athena-Dashboard,
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container, - Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
- WireGuard-Gateway und Datenbackup, - WireGuard-Gateway und Datenbackup,
@@ -28,13 +31,19 @@ werden keine zweiten Instanzen dieser Dienste angelegt.
| `/data/models` | Modellgewichte | | `/data/models` | Modellgewichte |
| `/data/llama-dashboard` | historische Dashboard-Messwerte | | `/data/llama-dashboard` | historische Dashboard-Messwerte |
| `/data/docker-backups` | automatische Athena-Backups | | `/data/docker-backups` | automatische Athena-Backups |
| `/data/trellis-studio` | trellis.cpp-Runtime und erzeugte 3D-Modelle |
| `/data/models/yue2`, `/data/music/yue2` | YuE2-Gewichte und dauerhafte Ergebnisse |
| `/etc/mike-ai` | lokale Konfiguration und Secrets, niemals Git | | `/etc/mike-ai` | lokale Konfiguration und Secrets, niemals Git |
Portainer läuft als separater, optionaler Verwaltungscontainer Portainer läuft als separater, optionaler Verwaltungscontainer
`mike-ai-portainer`, teilt den Netzwerk-Namespace des WireGuard-Gateways und `mike-ai-portainer` im internen Frontend-Netz und ist ausschließlich über den
ist unter `https://192.168.1.212:9443` erreichbar. Seine Einstellungen liegen namensbasierten Proxy des WireGuard-Gateways unter
im Docker-Volume `portainer_data`, das vom Athena-Backup mitgesichert wird. `https://192.168.1.212:9443` erreichbar. Das Dashboard verwendet denselben
Portainer beobachtet beziehungsweise stabilen Aufbau auf Port 8099. Beide teilen ausdrücklich nicht den
Netzwerk-Namespace des Gateway-Containers: Ein Recreate des Gateways kann sie
dadurch nicht mehr in einem veralteten Namespace zurücklassen. Portainers
Einstellungen liegen im Docker-Volume `portainer_data`, das vom Athena-Backup
mitgesichert wird. Portainer beobachtet beziehungsweise
verwaltet Docker, ist aber keine Abhängigkeit des Inferenz-Stacks. verwaltet Docker, ist aber keine Abhängigkeit des Inferenz-Stacks.
## Standardbefehle ## Standardbefehle
@@ -56,9 +65,23 @@ Qwen-Profil wird vom Profile Controller verwaltet.
- Fast: kurze, interaktive Aufgaben - Fast: kurze, interaktive Aufgaben
- Medium/Large/Ultra: steigende Kontextgrößen desselben lokalen Qwen-Modells - Medium/Large/Ultra: steigende Kontextgrößen desselben lokalen Qwen-Modells
- Uncensored: separates lokales Profil - Uncensored: separates lokales Profil
- FLUX.2-klein-4B: Bildgenerierung und Editing; Qwen wird dafür kurz entladen und danach - FLUX.2 Klein 9B FP8 Beta: Der Transformer läuft auf der RTX 5080, der
automatisch wiederhergestellt Qwen3-8B-NF4-Textencoder vorübergehend auf der RTX 3060. Das aktive
- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback llama.cpp-Profil und Qwen3-TTS werden dafür gestoppt und danach automatisch
wiederhergestellt. Die Beta arbeitet mit 1024 × 1024 Pixeln, vier Schritten
und Guidance 1,0.
- Qwen3-TTS 1.7B: RTX 3060. Der Router reicht
zusätzlich natives 24-kHz-PCM für den optionalen Hermes-Streaming-Adapter
unter `integrations/hermes-qwen3-stream` durch.
- ACE-Step 1.5 XL-SFT: exklusiver Musikmodus auf der RTX 5080. Dashboard und
die Routerbefehle `/athena music`, `/athena llm`, `/athena status` bedienen
dieselbe persistente Zustandsmaschine; siehe `docs/OPERATING_MODES.md`.
- YuE2-3B: eigener exklusiver Musikmodus mit Score- und Remix-Funktionen unter
`http://192.168.1.212:8014`. Der Routerbefehl lautet `/athena yue2`.
- TRELLIS.2 4B Q8: exklusives Bild-zu-3D-Profil auf der RTX 5080. Die
browserbasierte Oberfläche läuft unter `http://192.168.1.212:8013`, erzeugt
GLB und verwendet standardmäßig `1024 · cascade`. Der 1536er Pfad kann die
16 GiB VRAM überschreiten.
Die verbindlichen Werte stehen in `config/profile-matrix.json` und Die verbindlichen Werte stehen in `config/profile-matrix.json` und
`docs/STANDARD_PROFILE_MATRIX.md`. `docs/STANDARD_PROFILE_MATRIX.md`.
@@ -87,6 +110,9 @@ Bootloader, Partitionen, Mounts, SSH, LAN, WireGuard oder Firewall ändern.
Secrets dürfen lokal verwendet, aber nie in Git, Logs oder Chatantworten Secrets dürfen lokal verwendet, aber nie in Git, Logs oder Chatantworten
veröffentlicht werden. veröffentlicht werden.
Vor Änderungen durch einen Agenten ist [for_ki.md](for_ki.md) vollständig zu
lesen. Dort stehen insbesondere Modus-, Label-, Netzwerk- und Aufräumregeln.
## Fertig bedeutet ## Fertig bedeutet
- Änderung ist im kanonischen Git-Checkout, - Änderung ist im kanonischen Git-Checkout,
+342
View File
@@ -0,0 +1,342 @@
ATHENA – AUFBAU VON UNTEN NACH OBEN
====================================
Stand: 10.09.2026 nach Entfernung von Beta 1 und Piper sowie Integration von
TRELLIS.2 als 3D-Studio.
Athena besitzt derzeit 23 Container, fünf auswählbare LLM-Profile und vier
verwendete Docker-Volumes. Verwaiste Docker-Volumes gibt es nicht.
+--------------------------------------+
| PHYSISCHER RECHNER: ATHENA |
| |
| CPU, RAM, Systemplatte, Netzwerk |
| NVIDIA RTX 5080 + NVIDIA RTX 3060 |
+------------------+-------------------+
|
v
+--------------------------------------+
| DEBIAN-HOSTSYSTEM |
| |
| - startet den Rechner |
| - verwaltet Netzwerk und Datenträger |
| - stellt NVIDIA-Treiber bereit |
| - führt Docker aus |
+------------------+-------------------+
|
+-------------------+-------------------+
| |
v v
+----------------------------------+ +----------------------------------+
| DOCKER-STACK | | DAUERHAFTE DATEN AUF DEM HOST |
| | | |
| - Router und Profilsteuerung | | - Modelle und Modellgewichte |
| - llama.cpp-Modellserver | | - Trainingsdatensätze |
| - Athena-Dashboard | | - trainierte Stimmen |
| - Bild-, Musik- und Audiodienste | | - Checkpoints und Ergebnisse |
| - Applio und Mikes Applio UI | | - Konfigurationen und Logs |
| - Hilfs- und Netzwerkdienste | | |
+----------------+-----------------+ | Hauptpfade: |
| | /data |
| liest und schreibt | /etc/mike-ai |
+-------------------->| |
+----------------+-----------------+
|
v
+----------------------------------+
| BACKUP |
| |
| Sichert ausgewählte dauerhafte |
| Daten und Konfigurationen. |
+----------------------------------+
DOCKER-STACK: CONTAINER-INVENTAR
================================
Bestandsaufnahme vom 10.09.2026. "Gestoppt/bereit" bedeutet hier nicht
automatisch defekt: GPU-intensive Dienste werden absichtlich nur im passenden
Betriebsmodus gestartet. Zum Zeitpunkt der Aufnahme war Applio/RVC aktiv.
+-----------------------------------+-------------------+----------------------------------------------+
| Container | Zustand | Aufgabe |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-router | läuft | Zentrale API; leitet Text-, Bild-, Audio- |
| | | und Profilanfragen an den passenden Dienst. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-profile-controller | läuft | Schaltet Profile und Betriebsmodi und sorgt |
| | | dafür, dass sich GPU-Dienste nicht stören. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-llama-fast | gestoppt/bereit | llama.cpp-Textmodell mit kleinem Kontext und |
| | | hoher Geschwindigkeit. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-llama-medium | gestoppt/bereit | llama.cpp-Textmodell mit mittlerem Kontext. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-llama-large | gestoppt/bereit | llama.cpp-Textmodell mit großem Kontext. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-llama-ultra | gestoppt/bereit | llama.cpp-Textmodell mit maximalem Kontext. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-llama-uncensored | gestoppt/bereit | Separates ungefiltertes llama.cpp-Profil. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-image-worker | gestoppt/bereit | Lokale Bildgenerierung und Bildbearbeitung; |
| | | wird nur für Bildaufträge geladen. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-qwen3-tts | gestoppt/bereit | Hochwertige GPU-Sprachausgabe mit Qwen3-TTS. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-tts-gateway | läuft | Normalisiert Text, wandelt Audioformate und |
| | | streamt die Ausgabe von Qwen3-TTS. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-whisper | läuft | Lokale Spracherkennung: Sprache zu Text. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-music-acestep-test | gestoppt/bereit | ACE-Step 1.5: erzeugt und bearbeitet Musik. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-music-ui | läuft | Community-Weboberfläche für ACE-Step; das |
| | | eigentliche Musikmodell wird separat geladen.|
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-yue2-playground | gestoppt/bereit | Eigenständiges YuE2-Musikstudio mit Score-, |
| | | Generierungs- und Remix-Funktionen auf :8014.|
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-stem-separator | gestoppt/bereit | Trennt Gesang, Begleitung und Instrumente |
| | | mit BS-RoFormer und Demucs. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-trellis-studio | läuft/bedarfsgest.| TRELLIS.2 4B Q8 erzeugt aus einem Bild ein |
| | | texturiertes GLB-Modell auf der RTX 5080. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-voice-studio | gestoppt/bereit | Voice Studio für Text-zu-Stimme und |
| | | referenzbasierte Stimmerzeugung. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-xvc-studio | gestoppt/bereit | X-VC für direkte Stimme-zu-Stimme-Umwandlung |
| | | ohne vorheriges RVC-Training. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-applio-studio | läuft | Applio/RVC-Backend: Training, Modelle, |
| | | Sprachumwandlung und Original-Weboberfläche. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-mikes-applio-ui | läuft | Eigene geführte Oberfläche für das Applio- |
| | | Backend; enthält selbst kein RVC-Modell. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-llama-dashboard | läuft | Athena-Dashboard: Zustand, Telemetrie und |
| | | Umschaltung der Betriebsmodi. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-portainer | läuft | Allgemeine Webverwaltung und Einsicht für |
| | | Docker-Container, Images, Netze und Volumes. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-wireguard-gateway | läuft | Stellt die Athena-Webdienste ausschließlich |
| | | über den privaten WireGuard-Zugang bereit. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-mcp-athena-operator | läuft | Kontrollierte Verwaltungswerkzeuge für |
| | | Athena, unter anderem für Hermes. |
+-----------------------------------+-------------------+----------------------------------------------+
| mike-ai-backup | läuft | Sichert regelmäßig die dauerhaften Daten und |
| | | Konfigurationen von Athena. |
+-----------------------------------+-------------------+----------------------------------------------+
Die Container gehören technisch zu mehreren Compose-Projekten, werden hier
aber gemeinsam als Athena-Docker-Stack betrachtet:
- Kernsystem: /opt/mike-ai/stack
- Applio/RVC: /opt/mike-ai/stack/experiments/applio-rvc
- Mikes Applio UI: /opt/mike-ai/Mikes-Applio-UI
- ACE-Step-Musik: /opt/mike-ai/acestep-test
- Spurentrennung: /opt/mike-ai/stem-separator
- Voice Studio: /opt/mike-ai/omnivoice-studio
- X-VC: /opt/mike-ai/xvc-studio
- 3D Studio: /opt/mike-ai/trellis-studio
LLM-PROFILE
===========
Es läuft immer höchstens eines dieser Profile. Fast, Medium, Large und
Uncensored können zusätzlich den Vision-Projektor verwenden. Ultra reserviert
den verfügbaren Speicher für den maximalen Textkontext und läuft ohne Vision.
+------------+-------------------+----------------+-----------------------------+
| Profil | API-Modell | Kontext | Zweck |
+------------+-------------------+----------------+-----------------------------+
| Fast | qwen-fast | 76.800 Token | Hohe Geschwindigkeit und |
| | | | kurze bis mittlere Aufgaben.|
+------------+-------------------+----------------+-----------------------------+
| Medium | qwen-medium | 160.000 Token | Ausgewogenes Standardprofil.|
+------------+-------------------+----------------+-----------------------------+
| Large | qwen-large | 192.000 Token | Umfangreiche Dokumente und |
| | | | lange technische Arbeiten. |
+------------+-------------------+----------------+-----------------------------+
| Ultra | qwen-ultra | 262.144 Token | Maximaler Textkontext; ohne |
| | | | Vision-Projektor. |
+------------+-------------------+----------------+-----------------------------+
| Uncensored | qwen-uncensored | 80.000 Token | Weniger restriktives |
| | | | Spezialprofil. |
+------------+-------------------+----------------+-----------------------------+
Die produktiven Standardprofile verwenden Qwen3.8-27B in Q4-Quantisierung.
Das frühere Beta-1-Profil mit GSQ-RCO IQ3_S wurde entfernt: Es benötigte zwar
weniger Speicher, war im gemessenen Betrieb aber überwiegend langsamer und
brachte keinen belastbaren Qualitäts- oder Geschwindigkeitsvorteil.
BETRIEBSMODI UND GPU-UMSCHALTUNG
===============================
Die großen GPU-Dienste laufen gegenseitig exklusiv. Der Router speichert den
gewählten Zustand und die Profilsteuerung entlädt vor einem Wechsel die nicht
benötigten Modelle.
+---------------+------------------------------------------------------------+
| Modus | Geladener Hauptdienst |
+---------------+------------------------------------------------------------+
| LLM | Ein Qwen-LLM-Profil und Qwen3-TTS. |
+---------------+------------------------------------------------------------+
| Musik | ACE-Step 1.5 für Musikgenerierung. |
+---------------+------------------------------------------------------------+
| YuE2 Studio | YuE2-3B für Musik, Score-Steuerung und Audio-Remix. |
+---------------+------------------------------------------------------------+
| Audio trennen | BS-RoFormer, Demucs oder MossFormer2. |
+---------------+------------------------------------------------------------+
| Voice Studio | OmniVoice für referenzbasierte Text-zu-Sprache-Ausgabe. |
+---------------+------------------------------------------------------------+
| X-VC | Direkte Stimme-zu-Stimme-Umwandlung. |
+---------------+------------------------------------------------------------+
| Applio / RVC | RVC-Inferenz, Modellverwaltung und Stimmtraining. |
+---------------+------------------------------------------------------------+
| 3D Studio | TRELLIS.2 4B Q8 über trellis.cpp auf der RTX 5080. |
+---------------+------------------------------------------------------------+
Qwen3-TTS läuft nur im LLM-Modus. In einem exklusiven Spezialmodus bleibt das
leichte TTS-Gateway als API-Dienst gesund, meldet aber "ready: false", weil das
eigentliche Qwen3-TTS-Modell absichtlich entladen ist.
Das 3D-Studio ist im privaten WireGuard-Netz unter
http://192.168.1.212:8013 erreichbar. Es erzeugt GLB-Dateien; empfohlen ist
1024 · cascade. Runtime und Ausgaben liegen unter /data/trellis-studio, die
Q8-Gewichte unter /data/models/trellis2-q8.
TTS-AUFBAU
===========
+-----------------------------+
| Router / OpenAI-TTS-Endpunkt|
+--------------+--------------+
|
v
+-----------------------------+
| mike-ai-tts-gateway |
| - Text normalisieren |
| - Ausgabeformat umwandeln |
| - PCM-Streaming |
+--------------+--------------+
|
v
+-----------------------------+
| mike-ai-qwen3-tts |
| Qwen3-TTS 1.7B / Serena |
+-----------------------------+
Piper und sein CPU-Fallback wurden vollständig entfernt. Das TTS-Gateway
bleibt notwendig, weil es die stabile Schnittstelle und die Verarbeitung um
Qwen3-TTS herum bereitstellt. Wenn Qwen3-TTS nicht geladen ist, steht keine
Sprachausgabe zur Verfügung; es wird nicht mehr auf ein zweites Modell
zurückgegriffen.
DAUERHAFTE DOCKER-VOLUMES
=========================
Bestandsprüfung vom 10.09.2026: Alle vier Volumes sind einem vorhandenen
Container zugeordnet. "docker volume ls -f dangling=true" liefert keine
Treffer.
+-------------------------+-----------------------------------------------+
| Volume | Verwendung |
+-------------------------+-----------------------------------------------+
| mike-ai_router-state | Persistenter Routerzustand und Betriebsmodus. |
+-------------------------+-----------------------------------------------+
| mike-ai_router-images | Vom Router und Bilddienst erzeugte Bilder. |
+-------------------------+-----------------------------------------------+
| mike-ai_whisper-data | Lokales Whisper-Modell für Sprache-zu-Text. |
+-------------------------+-----------------------------------------------+
| portainer_data | Einstellungen und Daten von Portainer. |
+-------------------------+-----------------------------------------------+
Das frühere Volume "mike-ai_piper-data" wurde zusammen mit Piper gelöscht.
Beta 1 besaß kein eigenes Docker-Volume; seine rund 12 GB Modellgewichte lagen
als Hostverzeichnis unter /data/models und wurden ebenfalls gelöscht.
Viele Fachdienste verwenden statt Docker-Volumes direkte Hostverzeichnisse.
Die wichtigsten davon sind:
- /data/models Modellgewichte und Modell-Caches
- /data/voice/applio Applio-Datensätze, Logs und Stimmenmodelle
- /data/music Musikprojekte und generierte Titel
- /data/audio/separation Ergebnisse der Audio- und Spurentrennung
- /data/trellis-studio trellis.cpp-Runtime und erzeugte GLB-Dateien
- /data/llama-dashboard Verlauf und Zustandsdaten des Dashboards
- /etc/mike-ai betriebliche Konfiguration und Geheimnisse
- /data/docker-backups erzeugte Sicherungsarchive
Diese Verzeichnisse sind keine Docker-Volumes. Ein leerer Docker-Volume-Check
beweist deshalb nicht automatisch, dass unter /data keine alten Experiment-
oder Modelldateien mehr liegen.
BACKUP UND DISASTER RECOVERY
============================
Athena verwendet zwei Sicherungsebenen:
1. mike-ai-backup schreibt alle fünf Stunden ein lokales Schnellbackup nach
/data/docker-backups. Darin liegen /etc/mike-ai, ganz /opt/mike-ai sowie
Router- und Portainer-Zustand. Dieses Backup deckt den Ausfall der
Systemplatte ab, solange /data erhalten bleibt.
2. athena-disaster-backup schreibt nachts ein verschlüsseltes und
dedupliziertes Restic-Backup auf einen physisch anderen Speicher. Es enthält
zusätzlich eigene Stimmen, Trainingsdatensätze, Musik, Audioergebnisse,
Dashboard- und Projektdaten. Dieses Backup deckt den Ausfall der Datenplatte
und den gleichzeitigen Ausfall beider Platten ab.
Die reproduzierbaren Modellgewichte unter /data/models werden nicht extern
doppelt gespeichert. Bei Verlust der Datenplatte werden sie aus den
versionierten Quellen neu geladen. Das Whisper-Volume wird ebenfalls neu
erzeugt.
Nach Debian-Installation und dem Einhängen einer eigenen /data-Partition führt
disaster-recovery.sh den passenden Wiederaufbau aus:
- --scenario system: Systemplatte neu, alte Datenplatte vorhanden
- --scenario data: Datenplatte neu, Systemplatte vorhanden
- --scenario all: beide Platten neu
Das Skript formatiert keine Platten, führt keinen Neustart aus und beendet den
Wiederaufbau im sicheren LLM-Standardmodus. Details stehen in docs/RECOVERY.md.
Zusätzlich entstehen alle fünf Stunden unter /data/emergency-backups bis zu
fünf verschlüsselte Notfallpakete. Sie können mit Prüfsumme direkt aus dem
Athena-Dashboard heruntergeladen werden. Ein auf einen anderen Rechner
heruntergeladenes Paket kann statt des externen Restic-Speichers als Quelle
für den Daten- oder Totalausfall dienen. Auf /data verbliebene Pakete schützen
nicht gegen den Ausfall genau dieser Datenplatte.
ENTFERNTE KOMPONENTEN
=====================
- Beta 1 / qwen-beta-1: Profil, Containerdefinition, Container und
GSQ-RCO-IQ3_S-Modellgewichte entfernt. Der historische Testbericht bleibt
erhalten, damit das Modell nicht versehentlich erneut getestet wird.
- Piper: Containerdefinition, Container, Image, Datenvolume, Konfiguration und
WireGuard-Port 8091 entfernt.
WICHTIGES GRUNDPRINZIP
======================
Die Anwendungen laufen überwiegend in Docker-Containern. Container selbst
sind austauschbar und können aus den versionierten Stack-Dateien neu gebaut
werden. Modelle, Trainingsmaterial, Ergebnisse und betriebliche Einstellungen
liegen dagegen dauerhaft auf dem Debian-Host und werden in die Container
eingebunden.
Ein neu gebauter Container darf deshalb keine Nutzdaten vernichten. Für eine
vollständige Wiederherstellung werden jedoch sowohl das Git-Repository mit dem
Stack als auch eine Sicherung der dauerhaften Hostdaten benötigt.
+67 -10
View File
@@ -10,10 +10,15 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored - genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
- Profile Router auf Port 8081 - Profile Router auf Port 8081
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080 - FLUX.2 Klein 9B FP8 Beta für Textbilder und Referenzbild-Bearbeitung:
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback Transformer auf RTX 5080, Qwen3-8B-NF4-Textencoder auf RTX 3060
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung - Qwen3-TTS 1.7B auf der RTX 3060 hinter einem Normalisierungs- und Streaming-Gateway
- Whisper.cpp `ggml-small` auf der CPU für lokale deutsche Spracherkennung
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 - Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
- Dashboard-Umschaltung zwischen LLM-Betrieb, ACE-Step-Musikstudio,
BS-RoFormer-Stimmtrennung, OmniVoice, X-VC, Applio/RVC und TRELLIS.2
- TRELLIS.2 4B Q8 über trellis.cpp 0.6.0 für lokale Bild-zu-3D-Erzeugung
auf der RTX 5080
- Portainer CE als optionale Container-Ansicht auf Port 9443 - Portainer CE als optionale Container-Ansicht auf Port 9443
- WireGuard-Gateway, Datenbackup und Athena-Operator - WireGuard-Gateway, Datenbackup und Athena-Operator
- keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz - keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz
@@ -40,6 +45,11 @@ sudo ./install.sh --config /root/mike-ai-install.env
Das Installationsskript baut llama.cpp und die lokalen Images, lädt die Das Installationsskript baut llama.cpp und die lokalen Images, lädt die
versionierten Modellartefakte und startet ausschließlich den Athena-Kern. versionierten Modellartefakte und startet ausschließlich den Athena-Kern.
FLUX.2 Klein 9B ist bei Hugging Face zugriffsbeschränkt. Vor der Installation
müssen die Bedingungen beider BFL-Repositories akzeptiert und ein Token in der
unter `HF_TOKEN_FILE` konfigurierten, nur für root lesbaren Datei abgelegt sein.
Der Token wird ausschließlich als Read-only-Datei in den Download-Container
eingehängt und weder in `stack.env` noch in Git kopiert.
## Betrieb ## Betrieb
@@ -95,13 +105,45 @@ Zwei Slots wurden direkt am Router erfolgreich getestet; Hermes verwaltete zwei
gleichzeitig aktive Chats jedoch nicht zuverlässig. Deshalb bleibt ein Slot der gleichzeitig aktive Chats jedoch nicht zuverlässig. Deshalb bleibt ein Slot der
Standard, bis Hermes' Sitzungsfehler behoben ist. Standard, bis Hermes' Sitzungsfehler behoben ist.
### Bildgenerierung mit FLUX.2 Klein 9B FP8 Beta
Ein Bildauftrag verwendet beide GPUs exklusiv. Der Profile Controller stoppt
zuerst das aktive llama.cpp-Profil und Qwen3-TTS. Anschließend läuft der
FP8-Transformer auf der RTX 5080 und der in NF4 geladene Qwen3-8B-Textencoder
auf der RTX 3060. Vor dem VAE-Decoding werden Transformer und Textencoder
freigegeben. Nach dem Bildauftrag stoppt der Router den Bild-Worker und stellt
Qwen3-TTS sowie das zuvor aktive Textprofil automatisch wieder her. Während
der exklusiven Nutzung der RTX 3060 ist TTS vorübergehend nicht verfügbar.
Die Beta ist derzeit bewusst auf `1024x1024`, vier Schritte, Guidance `1.0`,
einen parallelen Auftrag und maximal vier lokale Referenzbilder begrenzt.
Details, Installation, Prüfung und Rollback stehen in
[docs/FLUX_9B_BETA.md](docs/FLUX_9B_BETA.md).
## Endpunkte ## Endpunkte
- Router: `http://192.168.1.212:8081/v1` - Router: `http://192.168.1.212:8081/v1`
- Athena-Dashboard: `http://192.168.1.212:8099` - Athena-Dashboard: `http://192.168.1.212:8099`
- Musikstudio, Original UI (stabil): `http://192.168.1.212:7862`
- Musikstudio, Community UI (experimentell): `http://192.168.1.212:7861`
- Spuren trennen (BS-RoFormer + Demucs, 2/4/6 Stems): `http://192.168.1.212:8007`
- Voice Studio (OmniVoice, Text zu Stimme): `http://192.168.1.212:8008`
- Voice Changer (X-VC, Audio zu Audio; native 16 kHz plus optional restaurierte 44,1 kHz): `http://192.168.1.212:8009`
- Applio (RVC-Inferenz, Modelle und Training): `http://192.168.1.212:8011`
- Mikes Applio UI (geführte RVC-Oberfläche): `http://192.168.1.212:8012`
- 3D Studio (TRELLIS.2 Q8, GLB-Ausgabe): `http://192.168.1.212:8013`
- YuE2 Studio (YuE2-3B, Generierung und Audio-Remix): `http://192.168.1.212:8014`
Der Betriebsmodus lässt sich dort direkt umschalten. In Hermes funktionieren
außerdem `/athena music`, `/athena stems`, `/athena voice`,
`/athena voicechange`, `/athena applio`, `/athena 3d`, `/athena llm` und
`/athena status`; Details stehen in
[docs/OPERATING_MODES.md](docs/OPERATING_MODES.md).
Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über Der Router stellt Sprache OpenAI-kompatibel bereit: Sprachausgabe über
`/v1/audio/speech` und Spracherkennung über `/v1/audio/transcriptions`. Das `/v1/audio/speech`, natives Qwen-PCM-Streaming über
`/v1/audio/speech/pcm-stream` und Spracherkennung über
`/v1/audio/transcriptions`. Das
Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten Whisper-Modell liegt persistent im Docker-Volume `whisper-data`; Audiodaten
werden lokal auf Athena verarbeitet. Für OpenClaw Talk liegt der lokale werden lokal auf Athena verarbeitet. Für OpenClaw Talk liegt der lokale
Realtime-Provider unter Realtime-Provider unter
@@ -110,6 +152,15 @@ verbindet Mikrofon → Athena Whisper → normalen OpenClaw-Agenten → aktives
Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten Athena-TTS, sodass Modell, Werkzeuge und Memory auch im Sprachmodus erhalten
bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis bleiben. Die Installation landet in OpenClaws persistentem Datenverzeichnis
und bleibt deshalb bei normalen Container-Updates bestehen. und bleibt deshalb bei normalen Container-Updates bestehen.
Für Hermes liegt unter `integrations/hermes-qwen3-stream` ein optionales,
persistentes Backend-Plugin. Es nutzt den nativen PCM-Strom und verkürzt den
Beginn der Sprachausgabe, ohne den Modellrouter oder die Textprofile zu ändern.
Bildgenerierung läuft über `/v1/images/generations`; Hermes verwendet dafür den
persistenten Benutzer-Provider `athena-local` mit dem Modellnamen
`FLUX.2-klein-9B-fp8-beta`. Seine versionierte Quelle und Installationshinweise
liegen unter
[`integrations/hermes-athena-image`](integrations/hermes-athena-image).
- Portainer: `https://192.168.1.212:9443` - Portainer: `https://192.168.1.212:9443`
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119` - Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
@@ -127,22 +178,28 @@ Unraid-DockerMan-Templates. Details stehen in
## Wiederherstellung ## Wiederherstellung
Nach einer frischen Debian-Installation und erneut eingehängtem `/data`: Nach einer frischen Debian-Installation und separat eingehängtem `/data`
übernimmt ein Orchestrator den vollständigen Wiederaufbau. Beispiel bei
erhaltener Datenplatte:
```bash ```bash
sudo ./install.sh --config /root/mike-ai-install.env sudo ./disaster-recovery.sh --scenario system \
sudo ./restore.sh --check /data/docker-backups/athena-latest.tar.gz --archive /data/docker-backups/athena-latest.tar.gz
sudo ./restore.sh /data/docker-backups/athena-latest.tar.gz
sudo ./smoke-test.sh
``` ```
Der genaue Sicherungsumfang steht in [docs/RECOVERY.md](docs/RECOVERY.md). Für den Ausfall der Datenplatte oder beider Platten wird das verschlüsselte
externe Restic-Backup verwendet. Der genaue Sicherungsumfang und alle drei
Szenarien stehen in [docs/RECOVERY.md](docs/RECOVERY.md).
## Verbindliche Dokumentation ## Verbindliche Dokumentation
- [ATHENA.md](ATHENA.md) – kurze Betriebsanleitung - [ATHENA.md](ATHENA.md) – kurze Betriebsanleitung
- [for_ki.md](for_ki.md) – verbindlicher System- und Änderungsleitfaden für KI-Agenten
- [docs/STANDARD_PROFILE_MATRIX.md](docs/STANDARD_PROFILE_MATRIX.md) – Profile - [docs/STANDARD_PROFILE_MATRIX.md](docs/STANDARD_PROFILE_MATRIX.md) – Profile
- [docs/CONTAINER_INVENTORY.md](docs/CONTAINER_INVENTORY.md) – alle Container, Modelle und Aufgaben
- [docs/TESTED_MODELS.md](docs/TESTED_MODELS.md) – zentrale Testhistorie und Sperrliste gegen Doppeltests
- [docs/MCP_SERVERS.md](docs/MCP_SERVERS.md) – produktive Werkzeuge - [docs/MCP_SERVERS.md](docs/MCP_SERVERS.md) – produktive Werkzeuge
- [docs/RECOVERY.md](docs/RECOVERY.md) – Backup und Neuaufbau - [docs/RECOVERY.md](docs/RECOVERY.md) – Backup und Neuaufbau
- [docs/FLUX_9B_BETA.md](docs/FLUX_9B_BETA.md) – 9B-Bildpfad, Test und Rollback
Git enthält keine Secrets, Chatdaten oder Modellgewichte. Git enthält keine Secrets, Chatdaten oder Modellgewichte.
+43 -140
View File
@@ -50,11 +50,6 @@ services:
- /tmp:size=16m,mode=1777 - /tmp:size=16m,mode=1777
volumes: volumes:
- "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro" - "${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}:/run/secrets/fritz-athena.conf:ro"
# Namespace-sharing services cannot publish ports themselves. The owner
# must keep these bindings so a gateway recreation cannot hide their UIs.
ports:
- "8099:8099"
- "9443:9443"
networks: networks:
frontend: frontend:
ipv4_address: 172.30.10.254 ipv4_address: 172.30.10.254
@@ -235,88 +230,6 @@ services:
- --spec-draft-p-min - --spec-draft-p-min
- "0.05" - "0.05"
# Beta 1 keeps the complete GSQ-RCO text model and KV cache on the RTX 5080.
# Only the multimodal projector runs on the RTX 3060. The measured hard
# boundary is 196608 tokens; 192K deliberately retains runtime headroom.
llama-beta1:
<<: *llama-common
container_name: mike-ai-llama-beta1
labels:
com.mike-ai.llama-profile: beta1
environment:
NVIDIA_VISIBLE_DEVICES: ${BETA1_GPU_DEVICES:-0,1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
MTMD_BACKEND_DEVICE: CUDA1
command:
- --model
- "/models/${BETA1_MODEL_FILE:?BETA1_MODEL_FILE is required}"
- --mmproj
- "/models/${VISION_PROJECTOR_FILE:?VISION_PROJECTOR_FILE is required}"
- --mmproj-offload
- --mmproj-device
- CUDA1
- --alias
- qwen-beta-1
- --ctx-size
- "${BETA1_CONTEXT:-192000}"
- --flash-attn
- "on"
- --cache-type-k
- q4_0
- --cache-type-v
- q4_0
- --cache-prompt
- --cache-ram
- "${LLAMA_CACHE_RAM_MIB:-32768}"
- --threads
- "${LLAMA_THREADS:-6}"
- --threads-batch
- "${LLAMA_THREADS_BATCH:-6}"
- --batch-size
- "${BETA1_BATCH_SIZE:-2048}"
- --ubatch-size
- "${BETA1_UBATCH_SIZE:-128}"
- --parallel
- "${BETA1_PARALLEL_SLOTS:-1}"
- --kv-unified
- --jinja
- --reasoning
- auto
- --reasoning-preserve
- --host
- 0.0.0.0
- --port
- "8080"
- --metrics
- --fit
- "off"
- --n-gpu-layers
- all
- --no-mmap
- --no-ui
- --temperature
- "1.0"
- --top-p
- "0.95"
- --top-k
- "20"
- --device
- CUDA0
- --main-gpu
- "0"
- --split-mode
- none
- --spec-type
- draft-mtp
- --spec-draft-n-max
- "3"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
- f16
- --spec-draft-p-min
- "0.05"
llama-large: llama-large:
<<: *llama-common <<: *llama-common
container_name: mike-ai-llama-large container_name: mike-ai-llama-large
@@ -569,8 +482,17 @@ services:
- /var/run/docker.sock:/var/run/docker.sock - /var/run/docker.sock:/var/run/docker.sock
environment: environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,beta1,large,ultra,uncensored ALLOWED_PROFILES: fast,medium,large,ultra,uncensored
IMAGE_WORKER: image IMAGE_WORKER: image
RESTORE_WORKER: restore
TTS_WORKER: qwen3
MUSIC_WORKER: acestep
YUE2_WORKER: yue2
SEPARATOR_WORKER: bs-roformer
VOICE_WORKER: vevo2
VOICE_CHANGE_WORKER: xvc
APPLIO_WORKER: applio
TRELLIS_WORKER: trellis2-q8
networks: [control] networks: [control]
security_opt: ["no-new-privileges:true"] security_opt: ["no-new-privileges:true"]
healthcheck: healthcheck:
@@ -606,6 +528,8 @@ services:
PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" PROFILE_CONTROL_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
SWITCH_TIMEOUT: "600" SWITCH_TIMEOUT: "600"
REQUEST_TIMEOUT: "600" REQUEST_TIMEOUT: "600"
YUE2_START_TIMEOUT: "600"
TRELLIS_START_TIMEOUT: "900"
# Last-resort guard for every OpenAI-compatible client. Without a # Last-resort guard for every OpenAI-compatible client. Without a
# request limit llama.cpp uses n_predict=-1 and a reasoning loop can # request limit llama.cpp uses n_predict=-1 and a reasoning loop can
# consume the complete context before yielding visible output. # consume the complete context before yielding visible output.
@@ -615,19 +539,21 @@ services:
IMAGE_DIR: /data/images IMAGE_DIR: /data/images
IMAGE_WORKER_URL: http://image-worker:8086 IMAGE_WORKER_URL: http://image-worker:8086
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
IMAGE_MODEL_NAME: FLUX.2-klein-4B IMAGE_MODEL_NAME: FLUX.2-klein-9B-fp8-beta
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false" CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "true" ENABLE_IMAGE_GENERATION: "true"
ENABLE_TTS: "true" ENABLE_TTS: "true"
# Stable OpenAI compatibility names remain piper/alloy because an # The gateway keeps text normalization, output conversion and native
# existing Open WebUI database persists those values. The gateway maps # PCM streaming in one stable API in front of Qwen3-TTS.
# alloy to XTTS speaker Annmarie Nele and automatically falls back to
# Piper if XTTS is unavailable, busy or returns an error.
TTS_WORKER_URL: http://tts-gateway:8085 TTS_WORKER_URL: http://tts-gateway:8085
TTS_MODEL: piper TTS_MODEL: qwen3-tts
TTS_VOICES: alloy TTS_VOICES: alloy
TTS_DEFAULT_VOICE: alloy TTS_DEFAULT_VOICE: alloy
ENABLE_STT: "true" ENABLE_STT: "true"
ENABLE_MUSIC_MODE: "true"
MUSIC_START_TIMEOUT: "600"
VOICE_CHANGE_START_TIMEOUT: "600"
APPLIO_START_TIMEOUT: "900"
STT_WORKER_URL: http://whisper:8084 STT_WORKER_URL: http://whisper:8084
STT_TIMEOUT: "300" STT_TIMEOUT: "300"
networks: [frontend, control, inference] networks: [frontend, control, inference]
@@ -649,8 +575,6 @@ services:
condition: service_healthy condition: service_healthy
profile-controller: profile-controller:
condition: service_healthy condition: service_healthy
piper:
condition: service_healthy
tts-gateway: tts-gateway:
condition: service_healthy condition: service_healthy
whisper: whisper:
@@ -674,13 +598,15 @@ services:
read_only: true read_only: true
tmpfs: ["/tmp:size=1g,mode=1777"] tmpfs: ["/tmp:size=1g,mode=1777"]
volumes: volumes:
- "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro" - "${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}:/models/components:ro"
- "${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}:/models/fp8:ro"
- router-images:/data/images - router-images:/data/images
environment: environment:
NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1} NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility NVIDIA_DRIVER_CAPABILITIES: compute,utility
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
FLUX_MODEL_DIR: /models/FLUX.2-klein-4B FLUX_COMPONENT_DIR: /models/components
FLUX_TRANSFORMER_FILE: /models/fp8/flux-2-klein-9b-fp8.safetensors
IMAGE_DIR: /data/images IMAGE_DIR: /data/images
networks: [inference] networks: [inference]
security_opt: ["no-new-privileges:true"] security_opt: ["no-new-privileges:true"]
@@ -691,41 +617,12 @@ services:
timeout: 3s timeout: 3s
retries: 12 retries: 12
piper:
build:
context: platform/docker/piper
args:
PIPER_TTS_VERSION: ${PIPER_TTS_VERSION:-1.6.0}
image: mike-ai/piper:local
container_name: mike-ai-piper
restart: unless-stopped
read_only: true
tmpfs:
- /tmp:size=256m,mode=1777
volumes:
- piper-data:/data
environment:
PIPER_DATA_DIR: /data
PIPER_VOICE: ${PIPER_VOICE:-de_DE-thorsten-high}
PIPER_VOICE_ALIAS: alloy
PIPER_HOST: 0.0.0.0
PIPER_PORT: "8085"
PIPER_MAX_TEXT_CHARS: "8000"
networks: [frontend]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
cap_add: [CHOWN, SETUID, SETGID]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8085/status"]
interval: 10s
timeout: 5s
retries: 30
start_period: 120s
qwen3-tts: qwen3-tts:
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98} image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
container_name: mike-ai-qwen3-tts container_name: mike-ai-qwen3-tts
restart: unless-stopped restart: unless-stopped
labels:
com.mike-ai.tts-worker: qwen3
deploy: deploy:
resources: resources:
reservations: reservations:
@@ -777,17 +674,12 @@ services:
QWEN_TTS_VOICE: serena QWEN_TTS_VOICE: serena
QWEN_TTS_LANGUAGE: German QWEN_TTS_LANGUAGE: German
QWEN_TTS_TIMEOUT: "120" QWEN_TTS_TIMEOUT: "120"
PIPER_URL: http://piper:8085
TTS_VOICE_ALIAS: alloy TTS_VOICE_ALIAS: alloy
TTS_DEFAULT_LANGUAGE: de TTS_DEFAULT_LANGUAGE: de
# Mixed-language clip stitching caused long pauses and unintelligible # Mixed-language clip stitching caused long pauses and unintelligible
# transitions. Keep full sentences in one stable German voice. # transitions. Keep full sentences in one stable German voice.
TTS_CODE_SWITCH_ENABLED: "false" TTS_CODE_SWITCH_ENABLED: "false"
PIPER_TIMEOUT: "120"
networks: [frontend] networks: [frontend]
depends_on:
piper:
condition: service_healthy
security_opt: ["no-new-privileges:true"] security_opt: ["no-new-privileges:true"]
cap_drop: [ALL] cap_drop: [ALL]
healthcheck: healthcheck:
@@ -837,7 +729,7 @@ services:
image: mike-ai/llama-dashboard:local image: mike-ai/llama-dashboard:local
container_name: mike-ai-llama-dashboard container_name: mike-ai-llama-dashboard
restart: unless-stopped restart: unless-stopped
network_mode: "service:wireguard-gateway" networks: [frontend]
gpus: all gpus: all
read_only: true read_only: true
tmpfs: tmpfs:
@@ -846,15 +738,26 @@ services:
- /proc:/host/proc:ro - /proc:/host/proc:ro
- /data:/host/data:ro - /data:/host/data:ro
- /data/models:/host/models:ro - /data/models:/host/models:ro
- /data/emergency-backups:/backups:ro
- /data/llama-dashboard:/var/lib/llama-dashboard - /data/llama-dashboard:/var/lib/llama-dashboard
environment: environment:
DASHBOARD_HOST: 0.0.0.0 DASHBOARD_HOST: 0.0.0.0
DASHBOARD_PORT: "8099" DASHBOARD_PORT: "8099"
ROUTER_URL: http://router:8081 ROUTER_URL: http://router:8081
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}" ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
MUSIC_COMMUNITY_UI_URL: "${MUSIC_COMMUNITY_UI_URL:-http://192.168.1.212:7861/}"
MUSIC_ORIGINAL_UI_URL: "${MUSIC_ORIGINAL_UI_URL:-http://192.168.1.212:7862/}"
SEPARATOR_UI_URL: "${SEPARATOR_UI_URL:-http://192.168.1.212:8007/}"
VOICE_UI_URL: "${VOICE_UI_URL:-http://192.168.1.212:8008/}"
VOICE_CHANGE_UI_URL: "${VOICE_CHANGE_UI_URL:-http://192.168.1.212:8009/}"
APPLIO_UI_URL: "${APPLIO_UI_URL:-http://192.168.1.212:8011/}"
MIKES_APPLIO_UI_URL: "${MIKES_APPLIO_UI_URL:-http://192.168.1.212:8012/}"
TRELLIS_UI_URL: "${TRELLIS_UI_URL:-http://192.168.1.212:8013/}"
YUE2_UI_URL: "${YUE2_UI_URL:-http://192.168.1.212:8014/}"
HOST_PROC: /host/proc HOST_PROC: /host/proc
HOST_DATA: /host/data HOST_DATA: /host/data
HOST_MODELS: /host/models HOST_MODELS: /host/models
DASHBOARD_BACKUP_DIR: /backups
DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3 DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3
DASHBOARD_HISTORY_INTERVAL: "15" DASHBOARD_HISTORY_INTERVAL: "15"
DASHBOARD_DETAIL_RETENTION_DAYS: "21" DASHBOARD_DETAIL_RETENTION_DAYS: "21"
@@ -878,7 +781,7 @@ services:
image: ${PORTAINER_IMAGE:-portainer/portainer-ce@sha256:511f3f06c96fe3b993ebeaafde311c1959cae73a7ef825dba6397d51b450dffa} image: ${PORTAINER_IMAGE:-portainer/portainer-ce@sha256:511f3f06c96fe3b993ebeaafde311c1959cae73a7ef825dba6397d51b450dffa}
container_name: mike-ai-portainer container_name: mike-ai-portainer
restart: unless-stopped restart: unless-stopped
network_mode: "service:wireguard-gateway" networks: [frontend]
command: [--no-setup-token] command: [--no-setup-token]
volumes: volumes:
- /var/run/docker.sock:/var/run/docker.sock - /var/run/docker.sock:/var/run/docker.sock
@@ -902,8 +805,9 @@ services:
- /var/run/docker.sock:/var/run/docker.sock:ro - /var/run/docker.sock:/var/run/docker.sock:ro
- /data/docker-backups:/archive - /data/docker-backups:/archive
- /etc/mike-ai:/backup/etc-mike-ai:ro - /etc/mike-ai:/backup/etc-mike-ai:ro
- /opt/mike-ai/stack:/backup/stack:ro # Include every deployed specialist UI/worker source tree, not just the
- piper-data:/backup/volumes/piper-data:ro # core checkout. Images themselves remain reproducible and are rebuilt.
- /opt/mike-ai:/backup/opt-mike-ai:ro
- router-state:/backup/volumes/router-state:ro - router-state:/backup/volumes/router-state:ro
- router-images:/backup/volumes/router-images:ro - router-images:/backup/volumes/router-images:ro
- portainer-data:/backup/volumes/portainer-data:ro - portainer-data:/backup/volumes/portainer-data:ro
@@ -930,7 +834,6 @@ networks:
name: mike-ai-tools-egress name: mike-ai-tools-egress
volumes: volumes:
piper-data:
whisper-data: whisper-data:
router-state: router-state:
router-images: router-images:
+18
View File
@@ -0,0 +1,18 @@
# Root-only configuration for the encrypted off-host Restic repository.
# Copy to /etc/mike-ai/disaster-backup.env and chmod 600.
#
# Recommended: mount an Unraid backup share at /mnt/athena-offsite and use:
RESTIC_REPOSITORY=/mnt/athena-offsite/restic
RESTIC_REQUIRE_MOUNT=/mnt/athena-offsite
# The password file must ALSO exist outside Athena (password manager/offline
# recovery USB). Without it a total-loss backup cannot be decrypted.
RESTIC_PASSWORD_FILE=/root/athena-restic-password
RESTIC_TAG=athena-disaster
RESTIC_KEEP_DAILY=14
RESTIC_KEEP_WEEKLY=8
RESTIC_KEEP_MONTHLY=12
# Set true only after the repository and credentials have been tested.
DISASTER_BACKUP_ENABLED=false
+5 -11
View File
@@ -18,7 +18,11 @@ NVIDIA_MIN_DRIVER_MAJOR=570
TEXT_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe TEXT_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
SECONDARY_GPU_DEVICES=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b SECONDARY_GPU_DEVICES=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B # FLUX.2 Klein 9B is gated. Accept both BFL model licenses first, then store
# the Hugging Face token in this root-readable file (never in this config).
HF_TOKEN_FILE=/root/.cache/huggingface/token
FLUX_COMPONENT_DIR=/data/models/FLUX.2-klein-9B-components
FLUX_TRANSFORMER_DIR=/data/models/FLUX.2-klein-9B-fp8
# Headless remote reachability. Firmware power-loss recovery is configured # Headless remote reachability. Firmware power-loss recovery is configured
# separately once at the physical machine. # separately once at the physical machine.
@@ -56,9 +60,6 @@ FAST_MODEL_SHA256=54879ae8738d5938f46cb3b8cbf16bf42b8c85b7d68d7c73f062b612ec183e
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
MEDIUM_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf MEDIUM_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
MEDIUM_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675 MEDIUM_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
BETA1_MODEL_FILE=qwen3.8-27b-gsq-rco-test/Qwen3.8-27B-GSQ-RCO-IQ3_XXS-mtp.gguf
BETA1_MODEL_URL=https://huggingface.co/ISTA-DASLab/Qwen3.8-27B-GSQ-RCO-GGUF/resolve/main/Qwen3.8-27B-GSQ-RCO-IQ3_XXS-mtp.gguf
BETA1_MODEL_SHA256=63f29a2189a6b4cc31f81e093d3856ad293a5583114439f41ae1ed7af4093262
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
LARGE_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf LARGE_MODEL_URL=https://huggingface.co/jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF/resolve/main/qwen3.8-27b-IQ4_XS-pure.gguf
LARGE_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675 LARGE_MODEL_SHA256=ea5a3c45d407f9b9e5d2c0d647f0ea600f486f6b86b92b56d0823ba073dae675
@@ -85,11 +86,6 @@ MEDIUM_CONTEXT=160000
MEDIUM_BATCH_SIZE=2048 MEDIUM_BATCH_SIZE=2048
MEDIUM_UBATCH_SIZE=128 MEDIUM_UBATCH_SIZE=128
MEDIUM_TENSOR_SPLIT=85,15 MEDIUM_TENSOR_SPLIT=85,15
BETA1_CONTEXT=192000
BETA1_BATCH_SIZE=2048
BETA1_UBATCH_SIZE=128
BETA1_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe,GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
BETA1_PARALLEL_SLOTS=1
LARGE_CONTEXT=192000 LARGE_CONTEXT=192000
LARGE_BATCH_SIZE=2048 LARGE_BATCH_SIZE=2048
LARGE_UBATCH_SIZE=128 LARGE_UBATCH_SIZE=128
@@ -111,8 +107,6 @@ MEDIUM_PARALLEL_SLOTS=1
LARGE_PARALLEL_SLOTS=1 LARGE_PARALLEL_SLOTS=1
ULTRA_PARALLEL_SLOTS=1 ULTRA_PARALLEL_SLOTS=1
UNCENSORED_PARALLEL_SLOTS=1 UNCENSORED_PARALLEL_SLOTS=1
PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high
QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98 QWEN3_TTS_IMAGE=ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98
QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache QWEN3_TTS_CACHE_DIR=/data/models/qwen3-tts-cache
QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices QWEN3_TTS_VOICES_DIR=/data/models/qwen3-tts-voices
-12
View File
@@ -27,18 +27,6 @@
"mtp": 3, "mtp": 3,
"description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben." "description": "Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben."
}, },
{
"id": "beta1",
"alias": "qwen-beta-1",
"context": 192000,
"parallel_slots": 1,
"model_env": "BETA1_MODEL_FILE",
"model_family": "Qwen3.8-27B GSQ-RCO IQ3_XXS MTP",
"gpu_split": "5080 model / 3060 vision",
"vision": true,
"mtp": 3,
"description": "Beta 1: schnelles GSQ-RCO-Testprofil mit 192K Kontext und Vision-Projektor auf der RTX 3060."
},
{ {
"id": "large", "id": "large",
"alias": "qwen-large", "alias": "qwen-large",
+2 -2
View File
@@ -71,8 +71,8 @@ class Handler(BaseHTTPRequestHandler):
self._send_json(200, { self._send_json(200, {
"status": "ok", "status": "ok",
"ready": True, "ready": True,
"voices": ["claribel"], "voices": ["alloy"],
"default_voice": "claribel", "default_voice": "alloy",
"load_errors": [], "load_errors": [],
"sample_rate": SAMPLE_RATE, "sample_rate": SAMPLE_RATE,
"uptime_seconds": 1.0, "uptime_seconds": 1.0,
+139
View File
@@ -0,0 +1,139 @@
import importlib.util
import json
import os
import tempfile
import unittest
from pathlib import Path
from unittest.mock import patch
ROOT = Path(__file__).resolve().parents[1]
MODULE_PATH = ROOT / "platform/llama-dashboard/app.py"
class _Response:
status = 202
def __enter__(self):
return self
def __exit__(self, *_args):
return False
def read(self):
return b'{"status":"accepted"}'
class DashboardModeTests(unittest.TestCase):
@classmethod
def setUpClass(cls):
cls.tempdir = tempfile.TemporaryDirectory()
with patch.dict(os.environ, {
"DASHBOARD_HISTORY_DB": str(Path(cls.tempdir.name) / "history.sqlite3"),
"DASHBOARD_BACKUP_DIR": str(Path(cls.tempdir.name) / "backups"),
"ROUTER_URL": "http://router.test:8081",
"ROUTER_API_KEY": "test-key",
}):
spec = importlib.util.spec_from_file_location("dashboard_app_test", MODULE_PATH)
cls.dashboard = importlib.util.module_from_spec(spec)
assert spec.loader is not None
spec.loader.exec_module(cls.dashboard)
cls.backup_dir = Path(cls.tempdir.name) / "backups"
cls.backup_dir.mkdir()
@classmethod
def tearDownClass(cls):
cls.tempdir.cleanup()
def test_separation_mode_is_forwarded_to_router(self):
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
status, body = self.dashboard.change_mode("separation")
self.assertEqual(status, 202)
self.assertEqual(body, {"status": "accepted"})
request = urlopen.call_args.args[0]
self.assertEqual(json.loads(request.data), {"mode": "separation"})
self.assertEqual(request.get_header("Authorization"), "Bearer test-key")
def test_voice_mode_is_forwarded_to_router(self):
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
status, body = self.dashboard.change_mode("voice")
self.assertEqual(status, 202)
self.assertEqual(body, {"status": "accepted"})
request = urlopen.call_args.args[0]
self.assertEqual(json.loads(request.data), {"mode": "voice"})
def test_voice_change_mode_is_forwarded_to_router(self):
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
status, body = self.dashboard.change_mode("voicechange")
self.assertEqual(status, 202)
self.assertEqual(body, {"status": "accepted"})
request = urlopen.call_args.args[0]
self.assertEqual(json.loads(request.data), {"mode": "voicechange"})
def test_applio_mode_is_forwarded_to_router(self):
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
status, body = self.dashboard.change_mode("applio")
self.assertEqual(status, 202)
self.assertEqual(body, {"status": "accepted"})
self.assertEqual(json.loads(urlopen.call_args.args[0].data), {"mode": "applio"})
def test_yue2_mode_is_forwarded_to_router(self):
with patch.object(self.dashboard.urllib.request, "urlopen", return_value=_Response()) as urlopen:
status, body = self.dashboard.change_mode("yue2")
self.assertEqual(status, 202)
self.assertEqual(body, {"status": "accepted"})
self.assertEqual(json.loads(urlopen.call_args.args[0].data), {"mode": "yue2"})
def test_unknown_mode_is_rejected_without_router_request(self):
with patch.object(self.dashboard.urllib.request, "urlopen") as urlopen:
status, body = self.dashboard.change_mode("unknown")
self.assertEqual(status, 400)
self.assertEqual(body, {"error": "invalid mode"})
urlopen.assert_not_called()
def test_dashboard_uses_one_status_poll_for_all_mode_labels(self):
html = self.dashboard.HTML
self.assertEqual(html.count("fetch('/api/status'"), 1)
self.assertNotIn("refreshVoiceChange", html)
self.assertIn("voicechange:'X-VC Voice Changer'", html)
self.assertIn("applio:'Applio / RVC'", html)
def test_dashboard_offers_both_applio_frontends(self):
html = self.dashboard.HTML
self.assertIn("Original Applio UI", html)
self.assertIn("Mikes Applio UI", html)
self.assertIn("http://192.168.1.212:8011/", html)
self.assertIn("http://192.168.1.212:8012/", html)
def test_dashboard_keeps_ace_step_and_offers_yue2_separately(self):
html = self.dashboard.HTML
self.assertIn("ACE-Step Studio", html)
self.assertIn("YuE2 Studio", html)
self.assertIn("setMode('music')", html)
self.assertIn("setMode('yue2')", html)
self.assertIn("http://192.168.1.212:8014/", html)
def test_dashboard_lists_only_portable_encrypted_backups(self):
valid = self.backup_dir / "athena-portable-2026-09-10T10-00-00Z.tar.zst.age"
valid.write_bytes(b"encrypted")
valid.with_name(valid.name + ".sha256").write_text(
"a" * 64 + " " + valid.name + "\n", encoding="utf-8"
)
(self.backup_dir / "unrelated.txt").write_text("ignore", encoding="utf-8")
backups = self.dashboard.backup_inventory()
self.assertEqual([item["name"] for item in backups], [valid.name])
self.assertEqual(backups[0]["sha256"], "a" * 64)
self.assertTrue(backups[0]["download_url"].startswith("/api/backups/download/"))
if __name__ == "__main__":
unittest.main()
+13 -13
View File
@@ -561,14 +561,14 @@ d=json.load(sys.stdin)
tts=d["tts"] tts=d["tts"]
assert tts["reachable"] is True, tts assert tts["reachable"] is True, tts
assert tts["ready"] is True, tts assert tts["ready"] is True, tts
assert set(tts["voices"])=={"claribel"}, tts assert set(tts["voices"])=={"alloy"}, tts
' && ok "Status: TTS erreichbar, bereit, 2 Stimmen" || bad "Status tts-Section" ' && ok "Status: TTS erreichbar und bereit" || bad "Status tts-Section"
# --- 28. TTS: POST /v1/audio/speech (wav) --------------------------------------------------------------- # --- 28. TTS: POST /v1/audio/speech (wav) ---------------------------------------------------------------
echo "== Test 28: POST /v1/audio/speech (wav)" echo "== Test 28: POST /v1/audio/speech (wav)"
CODE=$(curl -s -o /tmp/tts28.wav -w "%{http_code}" -D /tmp/hdr28.txt \ CODE=$(curl -s -o /tmp/tts28.wav -w "%{http_code}" -D /tmp/hdr28.txt \
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \ "$BASE/v1/audio/speech" -H "Content-Type: application/json" \
-d '{"model":"xtts-v2","input":"Hallo Welt","voice":"claribel","response_format":"wav"}') -d '{"model":"qwen3-tts","input":"Hallo Welt","voice":"alloy","response_format":"wav"}')
CTYPE=$(grep -i content-type /tmp/hdr28.txt | tr -d "\r") CTYPE=$(grep -i content-type /tmp/hdr28.txt | tr -d "\r")
[ "$CODE" = "200" ] && [ -s /tmp/tts28.wav ] && echo "$CTYPE" | grep -qi "audio/wav" \ [ "$CODE" = "200" ] && [ -s /tmp/tts28.wav ] && echo "$CTYPE" | grep -qi "audio/wav" \
&& ok "TTS wav (200, $CTYPE, $(stat -f%z /tmp/tts28.wav 2>/dev/null || stat -c%s /tmp/tts28.wav) Bytes)" \ && ok "TTS wav (200, $CTYPE, $(stat -f%z /tmp/tts28.wav 2>/dev/null || stat -c%s /tmp/tts28.wav) Bytes)" \
@@ -578,7 +578,7 @@ CTYPE=$(grep -i content-type /tmp/hdr28.txt | tr -d "\r")
echo "== Test 29: POST /v1/audio/speech (mp3, Default)" echo "== Test 29: POST /v1/audio/speech (mp3, Default)"
CODE=$(curl -s -o /tmp/tts29.mp3 -w "%{http_code}" -D /tmp/hdr29.txt \ CODE=$(curl -s -o /tmp/tts29.mp3 -w "%{http_code}" -D /tmp/hdr29.txt \
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \ "$BASE/v1/audio/speech" -H "Content-Type: application/json" \
-d '{"input":"Guten Tag","voice":"claribel"}') -d '{"input":"Guten Tag","voice":"alloy"}')
CTYPE=$(grep -i content-type /tmp/hdr29.txt | tr -d "\r") CTYPE=$(grep -i content-type /tmp/hdr29.txt | tr -d "\r")
[ "$CODE" = "200" ] && [ -s /tmp/tts29.mp3 ] && echo "$CTYPE" | grep -qi "audio/mpeg" \ [ "$CODE" = "200" ] && [ -s /tmp/tts29.mp3 ] && echo "$CTYPE" | grep -qi "audio/mpeg" \
&& ok "TTS mp3 (200, $CTYPE)" || bad "TTS mp3 (Code $CODE, $CTYPE)" && ok "TTS mp3 (200, $CTYPE)" || bad "TTS mp3 (Code $CODE, $CTYPE)"
@@ -586,7 +586,7 @@ CTYPE=$(grep -i content-type /tmp/hdr29.txt | tr -d "\r")
# --- 30. TTS: Validierung -------------------------------------------------------------------------------- # --- 30. TTS: Validierung --------------------------------------------------------------------------------
echo "== Test 30: TTS-Validierung" echo "== Test 30: TTS-Validierung"
CODE=$(curl -s -o /tmp/err30a.json -w "%{http_code}" "$BASE/v1/audio/speech" \ CODE=$(curl -s -o /tmp/err30a.json -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"voice":"claribel"}') -H "Content-Type: application/json" -d '{"voice":"alloy"}')
cat /tmp/err30a.json; echo cat /tmp/err30a.json; echo
[ "$CODE" = "400" ] && ok "400 bei fehlendem input" || bad "erwartet 400, bekam $CODE" [ "$CODE" = "400" ] && ok "400 bei fehlendem input" || bad "erwartet 400, bekam $CODE"
@@ -608,7 +608,7 @@ cat /tmp/err30d.json; echo
# --- 31. TTS: Worker-Fehler → 503 ------------------------------------------------------------------------ # --- 31. TTS: Worker-Fehler → 503 ------------------------------------------------------------------------
echo "== Test 31: TTS-Worker-Fehler → 503" echo "== Test 31: TTS-Worker-Fehler → 503"
CODE=$(curl -s -o /tmp/err31.json -w "%{http_code}" "$BASE/v1/audio/speech" \ CODE=$(curl -s -o /tmp/err31.json -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"input":"FAIL","voice":"claribel"}') -H "Content-Type: application/json" -d '{"input":"FAIL","voice":"alloy"}')
cat /tmp/err31.json; echo cat /tmp/err31.json; echo
[ "$CODE" = "503" ] && ok "503 bei TTS-Worker-Fehler" || bad "erwartet 503, bekam $CODE" [ "$CODE" = "503" ] && ok "503 bei TTS-Worker-Fehler" || bad "erwartet 503, bekam $CODE"
@@ -617,7 +617,7 @@ echo "== Test 32: TTS-Worker down → 503"
kill "$TTS_PID" 2>/dev/null || true kill "$TTS_PID" 2>/dev/null || true
sleep 0.5 sleep 0.5
CODE=$(curl -s -o /tmp/err32.json -w "%{http_code}" "$BASE/v1/audio/speech" \ CODE=$(curl -s -o /tmp/err32.json -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"input":"Hallo","voice":"claribel"}') -H "Content-Type: application/json" -d '{"input":"Hallo","voice":"alloy"}')
cat /tmp/err32.json; echo cat /tmp/err32.json; echo
[ "$CODE" = "503" ] && ok "503 bei downem TTS-Worker" || bad "erwartet 503, bekam $CODE" [ "$CODE" = "503" ] && ok "503 bei downem TTS-Worker" || bad "erwartet 503, bekam $CODE"
RESP=$(curl -sf "$BASE/status") RESP=$(curl -sf "$BASE/status")
@@ -634,7 +634,7 @@ MOCK_TTS_PORT="$TTS_PORT" MOCK_TTS_DELAY=0.1 \
TTS_PID=$! TTS_PID=$!
sleep 0.5 sleep 0.5
CODE=$(curl -s -o /tmp/tts33.wav -w "%{http_code}" "$BASE/v1/audio/speech" \ CODE=$(curl -s -o /tmp/tts33.wav -w "%{http_code}" "$BASE/v1/audio/speech" \
-H "Content-Type: application/json" -d '{"input":"Wieder da","voice":"claribel","response_format":"wav"}') -H "Content-Type: application/json" -d '{"input":"Wieder da","voice":"alloy","response_format":"wav"}')
[ "$CODE" = "200" ] && [ -s /tmp/tts33.wav ] \ [ "$CODE" = "200" ] && [ -s /tmp/tts33.wav ] \
&& ok "TTS nach Neustart wieder verfügbar" || bad "TTS-Recovery (Code $CODE)" && ok "TTS nach Neustart wieder verfügbar" || bad "TTS-Recovery (Code $CODE)"
@@ -727,8 +727,8 @@ import json,sys
d=json.load(sys.stdin) d=json.load(sys.stdin)
ids={m["id"] for m in d["data"]} ids={m["id"] for m in d["data"]}
assert "whisper-1" in ids, ids assert "whisper-1" in ids, ids
assert "xtts-v2" in ids, ids assert "qwen3-tts" in ids, ids
' && ok "Audio-Modelle: whisper-1 + xtts-v2" || bad "Audio-Modelle" ' && ok "Audio-Modelle: whisper-1 + qwen3-tts" || bad "Audio-Modelle"
# --- 41. /v1/audio/voices ------------------------------------------------------------------------------------------ # --- 41. /v1/audio/voices ------------------------------------------------------------------------------------------
echo "== Test 41: GET /v1/audio/voices" echo "== Test 41: GET /v1/audio/voices"
@@ -738,8 +738,8 @@ echo "$RESP" | python3 -c '
import json,sys import json,sys
d=json.load(sys.stdin) d=json.load(sys.stdin)
ids={v["id"] for v in d["data"]} ids={v["id"] for v in d["data"]}
assert "claribel" in ids, ids assert "alloy" in ids, ids
' && ok "Audio-Voices: claribel" || bad "Audio-Voices" ' && ok "Audio-Voices: alloy" || bad "Audio-Voices"
# --- 42. STT + Qwen parallel ---------------------------------------------------------------------------------------- # --- 42. STT + Qwen parallel ----------------------------------------------------------------------------------------
echo "== Test 42: STT + Qwen parallel" echo "== Test 42: STT + Qwen parallel"
@@ -770,7 +770,7 @@ sleep 0.2
# TTS-Request # TTS-Request
CODE=$(curl -s -o /tmp/tts43.mp3 -w "%{http_code}" \ CODE=$(curl -s -o /tmp/tts43.mp3 -w "%{http_code}" \
"$BASE/v1/audio/speech" -H "Content-Type: application/json" \ "$BASE/v1/audio/speech" -H "Content-Type: application/json" \
-d '{"input":"Hallo","voice":"claribel"}') -d '{"input":"Hallo","voice":"alloy"}')
wait $STT_PID43 wait $STT_PID43
[ "$CODE" = "200" ] && [ -s /tmp/tts43.mp3 ] \ [ "$CODE" = "200" ] && [ -s /tmp/tts43.mp3 ] \
&& ok "STT + TTS parallel (beide 200)" || bad "STT + TTS parallel (TTS Code $CODE)" && ok "STT + TTS parallel (beide 200)" || bad "STT + TTS parallel (TTS Code $CODE)"
+210 -4
View File
@@ -26,7 +26,185 @@ def image_item(state="exited"):
"Labels": {controller.IMAGE_LABEL_KEY: controller.IMAGE_WORKER}} "Labels": {controller.IMAGE_LABEL_KEY: controller.IMAGE_WORKER}}
def restore_item(state="exited"):
return {"Id": "id-restore", "State": state,
"Labels": {controller.IMAGE_LABEL_KEY: controller.RESTORE_WORKER}}
def tts_item(state="running"):
return {"Id": "id-tts", "State": state,
"Labels": {controller.TTS_LABEL_KEY: controller.TTS_WORKER}}
def music_item(state="exited"):
return {"Id": "id-music", "State": state,
"Labels": {controller.MUSIC_LABEL_KEY: "acestep"}}
def yue2_item(state="exited"):
return {"Id": "id-yue2", "State": state,
"Labels": {controller.MUSIC_LABEL_KEY: "yue2"}}
def separator_item(state="exited"):
return {"Id": "id-separator", "State": state,
"Labels": {controller.SEPARATOR_LABEL_KEY: "bs-roformer"}}
def voice_item(state="exited"):
return {"Id": "id-voice", "State": state,
"Labels": {controller.VOICE_LABEL_KEY: "vevo2"}}
def voice_change_item(state="exited"):
return {"Id": "id-xvc", "State": state,
"Labels": {controller.VOICE_CHANGE_LABEL_KEY: "xvc"}}
class ProfileControllerTests(unittest.TestCase): class ProfileControllerTests(unittest.TestCase):
def test_yue2_start_exclusively_stops_llm_and_ace_step(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["ultra"] = item("ultra", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "YUE2_WORKER", "yue2"), \
patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "yue2_container", return_value=yue2_item()), \
patch.object(controller, "music_container", return_value=music_item("running")), \
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_yue2_worker(True)
self.assertEqual(result, {"yue2_worker": "yue2", "state": "running"})
self.assertIn(("POST", "/containers/id-ultra/stop?t=120"), calls)
self.assertIn(("POST", "/containers/id-music/stop?t=30"), calls)
self.assertEqual(calls[-1], ("POST", "/containers/id-yue2/start"))
def test_voice_change_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["medium"] = item("medium", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "VOICE_CHANGE_WORKER", "xvc"), \
patch.object(controller, "VOICE_WORKER", "vevo2"), \
patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "voice_change_container", return_value=voice_change_item()), \
patch.object(controller, "voice_container", return_value=voice_item("running")), \
patch.object(controller, "music_container", return_value=music_item("running")), \
patch.object(controller, "separator_container", return_value=separator_item("running")), \
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_voice_change_worker(True)
self.assertEqual(result, {"voice_change_worker": "xvc", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-medium/stop?t=120"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-music/stop?t=30"),
("POST", "/containers/id-separator/stop?t=30"),
("POST", "/containers/id-voice/stop?t=30"),
("POST", "/containers/id-xvc/start"),
])
def test_voice_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["medium"] = item("medium", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "VOICE_WORKER", "vevo2"), \
patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "voice_container", return_value=voice_item()), \
patch.object(controller, "music_container", return_value=music_item("running")), \
patch.object(controller, "separator_container", return_value=separator_item("running")), \
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_voice_worker(True)
self.assertEqual(result, {"voice_worker": "vevo2", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-medium/stop?t=120"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-music/stop?t=30"),
("POST", "/containers/id-separator/stop?t=30"),
("POST", "/containers/id-voice/start"),
])
def test_separator_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["large"] = item("large", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "SEPARATOR_WORKER", "bs-roformer"), \
patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "separator_container", return_value=separator_item()), \
patch.object(controller, "music_container", return_value=music_item("running")), \
patch.object(controller, "image_containers", return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_separator_worker(True)
self.assertEqual(result, {"separator_worker": "bs-roformer", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-large/stop?t=120"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-music/stop?t=30"),
("POST", "/containers/id-separator/start"),
])
def test_music_start_exclusively_stops_gpu_workers(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["ultra"] = item("ultra", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "MUSIC_WORKER", "acestep"), \
patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "music_container", return_value=music_item()), \
patch.object(controller, "image_containers",
return_value=[image_item("running")]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_music_worker(True)
self.assertEqual(result, {"music_worker": "acestep", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-ultra/stop?t=120"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-music/start"),
])
def test_rejects_unknown_profile_before_docker_call(self): def test_rejects_unknown_profile_before_docker_call(self):
with patch.object(controller, "docker_request") as request: with patch.object(controller, "docker_request") as request:
with self.assertRaises(ValueError): with self.assertRaises(ValueError):
@@ -43,7 +221,8 @@ class ProfileControllerTests(unittest.TestCase):
return 204, b"" return 204, b""
with patch.object(controller, "containers", return_value=profiles), \ with patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "image_container", return_value=image_item()), \ patch.object(controller, "image_containers", return_value=[image_item()]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request): patch.object(controller, "docker_request", side_effect=request):
result = controller.activate("medium") result = controller.activate("medium")
@@ -56,7 +235,8 @@ class ProfileControllerTests(unittest.TestCase):
def test_fails_if_profile_container_is_missing(self): def test_fails_if_profile_container_is_missing(self):
profiles = {name: item(name) for name in controller.ALLOWED[:-1]} profiles = {name: item(name) for name in controller.ALLOWED[:-1]}
with patch.object(controller, "containers", return_value=profiles), \ with patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "image_container", return_value=image_item()): patch.object(controller, "image_containers", return_value=[image_item()]), \
patch.object(controller, "tts_container", return_value=tts_item()):
with self.assertRaisesRegex(RuntimeError, "missing"): with self.assertRaisesRegex(RuntimeError, "missing"):
controller.activate("fast") controller.activate("fast")
@@ -71,13 +251,38 @@ class ProfileControllerTests(unittest.TestCase):
with patch.object(controller, "containers", return_value=profiles), \ with patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "image_container", return_value=image_item()), \ patch.object(controller, "image_container", return_value=image_item()), \
patch.object(controller, "image_containers",
return_value=[image_item(), restore_item()]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request): patch.object(controller, "docker_request", side_effect=request):
controller.set_image_worker(True) controller.set_image_worker(True)
self.assertEqual(calls, [ self.assertEqual(calls, [
("POST", "/containers/id-medium/stop?t=120"), ("POST", "/containers/id-medium/stop?t=120"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-flux/start"), ("POST", "/containers/id-flux/start"),
]) ])
def test_restore_start_stops_flux_and_starts_restore(self):
profiles = {name: item(name) for name in controller.ALLOWED}
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "image_container", return_value=restore_item()), \
patch.object(controller, "image_containers",
return_value=[image_item("running"), restore_item()]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
controller.set_image_worker(True, controller.RESTORE_WORKER)
self.assertEqual(calls, [
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-restore/start"),
])
def test_profile_activation_stops_image_worker_first(self): def test_profile_activation_stops_image_worker_first(self):
profiles = {name: item(name) for name in controller.ALLOWED} profiles = {name: item(name) for name in controller.ALLOWED}
calls = [] calls = []
@@ -87,8 +292,9 @@ class ProfileControllerTests(unittest.TestCase):
return 204, b"" return 204, b""
with patch.object(controller, "containers", return_value=profiles), \ with patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "image_container", patch.object(controller, "image_containers",
return_value=image_item("running")), \ return_value=[image_item("running"), restore_item()]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request): patch.object(controller, "docker_request", side_effect=request):
controller.activate("fast") controller.activate("fast")
self.assertEqual(calls, [ self.assertEqual(calls, [
+34 -8
View File
@@ -82,19 +82,45 @@ class RecoveryScriptTests(unittest.TestCase):
for profile in ("fast", "medium", "large", "ultra", "uncensored"): for profile in ("fast", "medium", "large", "ultra", "uncensored"):
self.assertIn(f"llama-{profile}", installer) self.assertIn(f"llama-{profile}", installer)
def test_gateway_consumers_are_stopped_before_gateway_recreation(self) -> None: def test_gateway_consumers_do_not_require_rebinding(self) -> None:
manager = (ROOT / "manage.sh").read_text(encoding="utf-8") manager = (ROOT / "manage.sh").read_text(encoding="utf-8")
stop = 'stop llama-dashboard portainer' installer = (ROOT / "install.sh").read_text(encoding="utf-8")
deploy = 'up -d --build' self.assertNotIn('stop llama-dashboard portainer', manager)
self.assertIn(stop, manager) self.assertNotIn('stop llama-dashboard portainer', installer)
self.assertLess(manager.index(stop), manager.index(deploy)) self.assertNotIn('force-recreate llama-dashboard portainer', manager)
def test_gateway_owns_shared_ui_ports_and_portainer_backup(self) -> None: def test_gateway_proxies_stable_ui_services_and_portainer_backup(self) -> None:
compose = (ROOT / "compose.yaml").read_text(encoding="utf-8") compose = (ROOT / "compose.yaml").read_text(encoding="utf-8")
self.assertIn('"8099:8099"', compose) gateway = (ROOT / "platform/docker/wireguard-gateway/entrypoint.sh").read_text(
self.assertIn('"9443:9443"', compose) encoding="utf-8"
)
self.assertNotIn('network_mode: "service:wireguard-gateway"', compose)
self.assertNotIn('"8099:8099"', compose)
self.assertNotIn('"9443:9443"', compose)
self.assertIn('start_proxy 8099 llama-dashboard:8099', gateway)
self.assertIn('start_proxy 9443 portainer:9443', gateway)
self.assertIn('portainer-data:/backup/volumes/portainer-data:ro', compose) self.assertIn('portainer-data:/backup/volumes/portainer-data:ro', compose)
def test_disaster_recovery_covers_all_three_scenarios_without_formatting(self) -> None:
script = (ROOT / "disaster-recovery.sh").read_text(encoding="utf-8")
for scenario in ("system", "data", "all"):
self.assertIn(scenario, script)
self.assertIn("mountpoint -q /data", script)
self.assertIn("--portable", script)
for destructive in ("mkfs", "fdisk", "parted", "reboot", "shutdown"):
self.assertNotIn(f"{destructive} ", script)
def test_backup_layers_include_code_and_irreplaceable_data(self) -> None:
compose = (ROOT / "compose.yaml").read_text(encoding="utf-8")
export = (ROOT / "platform/backup/athena-export-backup").read_text(
encoding="utf-8"
)
self.assertIn("/opt/mike-ai:/backup/opt-mike-ai:ro", compose)
self.assertIn("/data/voice/applio/logs", export)
self.assertIn("/data/voice/applio/datasets", export)
self.assertIn("ATHENA_EXPORT_KEEP:-5", export)
self.assertNotIn("add_path /data/models", export)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()
+172
View File
@@ -0,0 +1,172 @@
#!/usr/bin/env bash
# One-shot recovery orchestrator for system-disk, data-disk and total loss.
# It never partitions, formats, reboots or shuts down the host.
set -Eeuo pipefail
umask 077
ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SCENARIO=""
ARCHIVE=/data/docker-backups/athena-latest.tar.gz
RECOVERY_CONFIG=""
INSTALL_CONFIG=""
SNAPSHOT=latest
PORTABLE=""
AGE_IDENTITY=""
WORK=""
log() { printf '\n==> %s\n' "$*"; }
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
cleanup() { [[ -z $WORK ]] || rm -rf "$WORK"; }
trap cleanup EXIT
usage() {
cat <<'EOF'
Systemplatte defekt, vorhandene /data-Platte:
sudo ./disaster-recovery.sh --scenario system \
--archive /data/docker-backups/athena-latest.tar.gz
Datenplatte defekt, Systemplatte vorhanden:
sudo ./disaster-recovery.sh --scenario data \
--config /root/athena-recovery.env
Beide Platten neu:
sudo ./disaster-recovery.sh --scenario all \
--config /root/athena-recovery.env
Alternativ bei Daten-/Totalausfall mit einem zuvor heruntergeladenen Paket:
sudo ./disaster-recovery.sh --scenario all \
--portable /pfad/athena-portable-....tar.zst.age \
--identity /root/athena-recovery-key.txt
Voraussetzung: Debian ist installiert und die richtige, bereits formatierte
Datenpartition ist separat unter /data eingehängt. Dieses Skript formatiert
keine Datenträger und führt niemals selbst einen Neustart aus.
EOF
}
while [[ $# -gt 0 ]]; do
case "$1" in
--scenario) SCENARIO=${2:-}; shift 2 ;;
--archive) ARCHIVE=${2:-}; shift 2 ;;
--config) RECOVERY_CONFIG=${2:-}; shift 2 ;;
--install-config) INSTALL_CONFIG=${2:-}; shift 2 ;;
--snapshot) SNAPSHOT=${2:-}; shift 2 ;;
--portable) PORTABLE=${2:-}; shift 2 ;;
--identity) AGE_IDENTITY=${2:-}; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) die "Unbekanntes Argument: $1" ;;
esac
done
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
[[ $SCENARIO == system || $SCENARIO == data || $SCENARIO == all ]] || \
die "--scenario muss system, data oder all sein."
mountpoint -q /data || die "/data ist kein eigener Mountpoint. Abbruch zum Schutz der Systemplatte."
install_bootstrap_packages() {
apt-get update
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
ca-certificates gzip rsync tar restic
}
copy_tree() {
local source=$1 target=$2
[[ -d $source ]] || return 0
install -d -m 0755 "$target"
rsync -a "$source/" "$target/"
}
restore_local_bootstrap() {
[[ -s $ARCHIVE ]] || die "Lokales Backup fehlt: $ARCHIVE"
gzip -t "$ARCHIVE" || die "Lokales Backup ist beschädigt."
WORK=$(mktemp -d /tmp/athena-system-recovery.XXXXXX)
tar -xzf "$ARCHIVE" -C "$WORK"
[[ -d $WORK/backup/etc-mike-ai ]] || die "Backup enthält /etc/mike-ai nicht."
copy_tree "$WORK/backup/etc-mike-ai" /etc/mike-ai
if [[ -d $WORK/backup/opt-mike-ai ]]; then
copy_tree "$WORK/backup/opt-mike-ai" /opt/mike-ai
elif [[ -d $WORK/backup/stack ]]; then
copy_tree "$WORK/backup/stack" /opt/mike-ai/stack
fi
install -d -m 0700 /var/lib/mike-ai-disaster-backup/latest
install -m 0600 "$ARCHIVE" \
/var/lib/mike-ai-disaster-backup/latest/docker-state.tar.gz
}
restore_external_snapshot() {
[[ -r $RECOVERY_CONFIG ]] || die "Externe Recovery-Konfiguration fehlt: $RECOVERY_CONFIG"
# shellcheck disable=SC1090
source "$RECOVERY_CONFIG"
[[ -n ${RESTIC_REPOSITORY:-} ]] || die "RESTIC_REPOSITORY fehlt."
[[ -n ${RESTIC_PASSWORD_FILE:-} && -r $RESTIC_PASSWORD_FILE ]] || die \
"Der separat aufzubewahrende Restic-Schlüssel fehlt."
WORK=$(mktemp -d /tmp/athena-offsite-recovery.XXXXXX)
restic restore "$SNAPSHOT" --tag "${RESTIC_TAG:-athena-disaster}" --target "$WORK"
[[ -d $WORK/data ]] || die "Snapshot enthält keine Athena-Daten."
copy_tree "$WORK/data" /data
if [[ $SCENARIO == all ]]; then
copy_tree "$WORK/etc/mike-ai" /etc/mike-ai
copy_tree "$WORK/opt/mike-ai" /opt/mike-ai
fi
}
restore_portable_archive() {
[[ -s $PORTABLE ]] || die "Portables Backup fehlt: $PORTABLE"
[[ -r $AGE_IDENTITY ]] || die "Age-Identität fehlt: $AGE_IDENTITY"
WORK=$(mktemp -d /tmp/athena-portable-recovery.XXXXXX)
age --decrypt -i "$AGE_IDENTITY" "$PORTABLE" | zstd -d | tar -xf - -C "$WORK"
[[ -d $WORK/data ]] || die "Portables Backup enthält keine Athena-Daten."
copy_tree "$WORK/data" /data
if [[ $SCENARIO == all ]]; then
copy_tree "$WORK/etc/mike-ai" /etc/mike-ai
copy_tree "$WORK/opt/mike-ai" /opt/mike-ai
fi
}
run_installer() {
local config=${INSTALL_CONFIG:-/etc/mike-ai/install.env}
[[ -r $config ]] || die \
"Installationskonfiguration fehlt: $config (alternativ --install-config angeben)."
chmod 0600 "$config"
[[ -x /opt/mike-ai/stack/install.sh ]] || die "Wiederhergestellter Stack fehlt."
set +e
/opt/mike-ai/stack/install.sh --config "$config"
local rc=$?
set -e
if [[ $rc == 20 || $rc == 21 ]]; then
printf '\nEin kontrollierter Neustart ist für Treiber/Netzwerk nötig.\n'
printf 'Danach exakt denselben Disaster-Recovery-Befehl erneut ausführen.\n'
exit "$rc"
fi
[[ $rc == 0 ]] || die "Installer fehlgeschlagen (Exit $rc)."
}
restore_docker_state() {
local state=/var/lib/mike-ai-disaster-backup/latest/docker-state.tar.gz
if [[ ! -s $state && -n $WORK ]]; then
state=$(find "$WORK/var/lib/mike-ai-disaster-backup/latest" \
-maxdepth 1 -name docker-state.tar.gz -type f -print -quit 2>/dev/null || true)
fi
[[ -s $state ]] || die "Docker-Zustandsarchiv fehlt im Backup."
/opt/mike-ai/stack/restore.sh --check "$state"
/opt/mike-ai/stack/restore.sh "$state"
}
install_bootstrap_packages
if [[ $SCENARIO == system ]]; then
restore_local_bootstrap
elif [[ -n $PORTABLE ]]; then
restore_portable_archive
else
restore_external_snapshot
fi
# In the data-only case the source/configuration remain on the system disk.
[[ -x /opt/mike-ai/stack/install.sh ]] || die "/opt/mike-ai/stack fehlt."
run_installer
restore_docker_state
/opt/mike-ai/stack/platform/recovery/rebuild-specialized.sh
/opt/mike-ai/stack/smoke-test.sh
printf '\nATHENA_DISASTER_RECOVERY_OK scenario=%s\n' "$SCENARIO"
printf 'Athena läuft im LLM-Standardmodus; Spezial-GPU-Worker bleiben gestoppt.\n'
+52 -10
View File
@@ -6,9 +6,12 @@ flowchart LR
H -->|OpenAI API| R[Profile Router<br/>Athena :8081] H -->|OpenAI API| R[Profile Router<br/>Athena :8081]
R --> P[Profile Controller] R --> P[Profile Controller]
P --> Q[genau ein llama.cpp-Profil<br/>Qwen Fast / Medium / Large / Ultra / Uncensored] P --> Q[genau ein llama.cpp-Profil<br/>Qwen Fast / Medium / Large / Ultra / Uncensored]
R --> I[FLUX.2-klein-4B<br/>RTX 5080, Text + Editing] R --> I[FLUX.2 Klein 9B FP8 Beta<br/>RTX 5080 Transformer]
R --> T[XTTS RTX 3060<br/>Piper CPU-Fallback] I --> E[Qwen3-8B NF4 Textencoder<br/>RTX 3060 während Bildauftrag]
R --> STT[Whisper.cpp large-v3-turbo<br/>CPU, lokale Spracherkennung] R --> T[Qwen3-TTS RTX 3060<br/>Normalisierungs- und Streaming-Gateway]
R --> STT[Whisper.cpp ggml-small<br/>CPU, lokale Spracherkennung]
P --> SP[Exklusive Spezialworker<br/>Musik / Trennung / Voice / RVC / 3D]
SP --> TR[TRELLIS.2 4B Q8<br/>trellis.cpp, RTX 5080]
H --> U[MUA / Unraid MCP] H --> U[MUA / Unraid MCP]
H --> A[ARR-MCP] H --> A[ARR-MCP]
@@ -16,12 +19,13 @@ flowchart LR
H --> N[Navidrome-MCP] H --> N[Navidrome-MCP]
H --> S[STRATO-MCP] H --> S[STRATO-MCP]
H --> X[Nginx-Proxy-Manager-MCP] H --> X[Nginx-Proxy-Manager-MCP]
U --> M[Media-Tools<br/>ffmpeg / ffprobe / yt-dlp] U --> MT[Media-Tools<br/>ffmpeg / ffprobe / yt-dlp]
W[WireGuard-Gateway<br/>Athena] --- R W[WireGuard-Gateway<br/>Athena] -->|DNS-Proxy| R
W --- B[Athena Dashboard :8099] W -->|DNS-Proxy :8099| B[Athena Dashboard<br/>internes Frontend-Netz]
W --- PRT[Portainer :9443<br/>optionale Docker-Ansicht] W -->|DNS-Proxy :9443| PRT[Portainer<br/>internes Frontend-Netz]
W --- O[Athena Operator] W -->|DNS-Proxy :8013| TRUI[Trellis Studio<br/>Bild zu GLB]
W -->|DNS-Proxy| O[Athena Operator]
K[Backup alle 5 Stunden] --> DATA[/data und /etc/mike-ai] K[Backup alle 5 Stunden] --> DATA[/data und /etc/mike-ai]
``` ```
@@ -33,6 +37,12 @@ flowchart LR
- **MUA** verwaltet Unraid. **Athena Operator** bleibt auf den Athena-Host - **MUA** verwaltet Unraid. **Athena Operator** bleibt auf den Athena-Host
begrenzt. begrenzt.
- Der Router ist die einzige Modelladresse, die Hermes kennen muss. - Der Router ist die einzige Modelladresse, die Hermes kennen muss.
- GPU-intensive Spezialdienste sind gegenseitig exklusiv. Der Router speichert
Modus und Rückkehrprofil; der Profile Controller startet nur eindeutig
gelabelte Worker.
- Dashboard und Portainer besitzen eigene Netzwerk-Namespaces. Das
WireGuard-Gateway löst ihre stabilen Compose-Dienstnamen bei jeder
Verbindung neu auf; seine konkrete Container-ID ist damit irrelevant.
## Dynamische Qwen-Profile ## Dynamische Qwen-Profile
@@ -48,7 +58,39 @@ Kontextgröße:
| Ultra | 262.144 Token | | Ultra | 262.144 Token |
| Uncensored | 80.000 Token | | Uncensored | 80.000 Token |
## Exklusiver Bildmodus
Text- und Bildinferenz teilen sich dieselben GPUs und laufen deshalb nicht
gleichzeitig. Der Wechsel ist transaktional:
1. Router merkt sich das aktive Textprofil.
2. Profile Controller stoppt alle llama.cpp-Profile und Qwen3-TTS.
3. Bild-Worker lädt Qwen3-8B als NF4-Textencoder auf die RTX 3060 und den
FLUX.2-Klein-9B-FP8-Transformer auf die RTX 5080.
4. Nach dem Prompt-Encoding werden die Embeddings zur RTX 5080 übertragen.
5. Vor dem VAE-Decoding werden Textencoder und Transformer freigegeben.
6. Der Worker wird gestoppt; anschließend starten Qwen3-TTS und das vorherige
Textprofil wieder. Während der exklusiven Bildphase steht kein TTS bereit.
Der Bild-Worker ist lazy und besitzt `restart: "no"`; im normalen Textbetrieb
belegt er daher keinen VRAM. Container werden über eindeutige Docker-Labels
gefunden, nicht über zufällige Container-IDs.
Die visuelle Fassung liegt als `athena-architecture-map.png` neben dieser Datei. Die visuelle Fassung liegt als `athena-architecture-map.png` neben dieser Datei.
Eine zweite Detailkarte, `athena-gpu-allocation-map.png`, zeigt die Eine zweite Detailkarte, `athena-gpu-allocation-map.png`, zeigt die
profilabhängige Layer-Verteilung auf RTX 5080 und RTX 3060 sowie die festen profilabhängige Layer-Verteilung auf RTX 5080 und RTX 3060. Die PNG-Karten
GPU-Zuordnungen von FLUX.2, Vision-Projektor und XTTS. zeigen noch den Stand vor dem 9B-Bildpfad; die aktuelle textuelle Beschreibung
in diesem Dokument ist verbindlich.
## TRELLIS.2 3D-Modus
Der Modus `trellis` stoppt die anderen GPU-Worker und startet genau den mit
`com.mike-ai.trellis-worker=trellis2-q8` markierten Container. trellis.cpp
0.6.0 sieht ausschließlich die Host-GPU 1, die RTX 5080. Q8-Gewichte liegen
unter `/data/models/trellis2-q8`, Runtime und Ausgaben unter
`/data/trellis-studio`. Die UI ist intern `trellis-studio:8080` und wird vom
WireGuard-Gateway auf `192.168.1.212:8013` weitergeleitet. Sie erzeugt GLB;
regulärer Qualitätsmodus ist 1024 Pixel.
Die vollständigen Regeln für Erweiterungen, Rückbau und Fehlersuche stehen in
[`../for_ki.md`](../for_ki.md).
+50
View File
@@ -0,0 +1,50 @@
# Container-Inventar auf Athena
Stand: 10. September 2026
Athena besteht nach der Bereinigung aus 23 Docker-Containern. Nicht jeder Container enthält
ein KI-Modell: Router, Oberflächen, Netzwerk, Steuerung und Sicherung sind
gewöhnliche Dienste. Die rechenintensiven GPU-Worker werden absichtlich nur bei
Bedarf gestartet. Ein Container im Zustand `Created` oder `Exited (0)` ist daher
nicht automatisch ein ungenutzter Rest.
| Container | Modell oder wesentliche Komponente | Aufgabe |
|---|---|---|
| `mike-ai-backup` | kein Modell; Offen Docker Volume Backup | Sichert `/data`, `/etc/mike-ai`, den Stack und die persistenten Docker-Volumes im Fünf-Stunden-Takt. |
| `mike-ai-applio-studio` | Applio/RVC; Stimmenmodelle werden nutzerseitig ergänzt | Vollständige RVC-Oberfläche für Inferenz, Modellverwaltung und Training auf der RTX 5080. Für eine Konvertierung ist ein importiertes oder trainiertes `.pth`-Modell nötig; eine Referenzaufnahme allein reicht nicht. |
| `mike-ai-image-worker` | FLUX.2 Klein 9B FP8, Qwen3-8B NF4 Textencoder und VAE | Erzeugt und bearbeitet Bilder transaktional; nutzt während eines Auftrags RTX 5080 und RTX 3060. |
| `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung und Betriebsarten an und bietet die Modusumschaltung. |
| `mike-ai-llama-fast` | Qwen3.8-27B `IQ4-MIX`, Qwen-MMProj BF16 | Schnelles Q4-Text-/Vision-Profil mit 76.800 Token Kontext. |
| `mike-ai-llama-large` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Q4-Text-/Vision-Profil mit 192.000 Token Kontext und Verteilung auf beide GPUs. |
| `mike-ai-llama-medium` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Standard-Q4-Text-/Vision-Profil mit 160.000 Token Kontext und Verteilung auf beide GPUs. |
| `mike-ai-llama-ultra` | Qwen3.8-27B `IQ4_XS-pure`, ohne Vision-Projektor | Maximales Langkontextprofil mit 262.144 Token Kontext und Verteilung auf beide GPUs. |
| `mike-ai-llama-uncensored` | Qwen3.8-27B Abliterated `Q4_K_M`, eigener MMProj F16 | Spezialprofil mit 80.000 Token Kontext und gelockerten Modellgrenzen. |
| `mike-ai-mcp-athena-operator` | kein Modell | Stellt Hermes begrenzte Werkzeuge zum Prüfen, Ändern, Testen, Sichern und Versionieren von Athena bereit. |
| `mike-ai-music-acestep-test` | ACE-Step 1.5 XL-SFT und `acestep-5Hz-lm-1.7B` | Generiert Musik im exklusiven Musikmodus auf der RTX 5080. |
| `mike-ai-music-ui` | kein Modell; `fspecii/ace-step-ui` | Community-Oberfläche für ACE-Step; bleibt als leichte UI verfügbar, während der GPU-Worker bedarfsgesteuert läuft. |
| `mike-ai-portainer` | kein Modell; Portainer CE | Optionale Docker-Verwaltungsoberfläche. |
| `mike-ai-profile-controller` | kein Modell | Startet und stoppt ausschließlich freigegebene Modellprofile und Spezialworker in einer sicheren Reihenfolge. |
| `mike-ai-qwen3-tts` | `Qwen/Qwen3-TTS-12Hz-1.7B-Base`, Stimme Serena | Hochwertige deutsche Sprachausgabe auf der RTX 3060 im LLM-Betrieb. |
| `mike-ai-router` | kein eigenes Modell | Einzige OpenAI-kompatible Modelladresse; koordiniert Profile, Bildaufträge, Sprache und Betriebsarten. |
| `mike-ai-stem-separator` | BS-RoFormer Viperx 1297, `htdemucs_ft`, `htdemucs_6s`, `MossFormer2_SE_48K` | Trennt Gesang, Instrumente oder Sprache/Hintergrundgeräusche im exklusiven Separationsmodus. |
| `mike-ai-trellis-studio` | TRELLIS.2 4B Q8 über trellis.cpp 0.6.0 | Erzeugt im exklusiven 3D-Modus aus einem Bild ein texturiertes, geschlossen aufbereitetes GLB-Mesh. Nutzt ausschließlich die RTX 5080 und wird über Port 8013 bedient. |
| `mike-ai-yue2-playground` | YuE2-3B mit Ladypoly `YuE2_WebUI` | Eigenständiges Musikstudio für Generierung, Score-basierte Steuerung und SheetSage2-Audioanalyse/Remix. Läuft exklusiv zu ACE-Step und allen übrigen GPU-Diensten; Zugriff über Port 8014. |
| `mike-ai-tts-gateway` | kein eigenes Modell | Normalisiert Text, konvertiert Ausgabeformate und stellt Qwen3-TTS sowie natives PCM-Streaming über eine stabile interne API bereit. |
| `mike-ai-voice-studio` | `k2-fsa/OmniVoice` 0.2.1 mit Whisper-ASR | Erzeugt Text-to-Speech mit einer Referenzstimme; kein Audio-to-Audio-Voice-Changer. |
| `mike-ai-whisper` | Whisper.cpp `ggml-small` | Lokale deutsche Spracherkennung auf der CPU über `/v1/audio/transcriptions`. |
| `mike-ai-wireguard-gateway` | kein Modell | Veröffentlicht Dashboard und Fachoberflächen ausschließlich über den privaten WireGuard-Pfad. |
| `mike-ai-xvc-studio` | `chenxie95/X-VC`, GLM-4-Voice-Tokenizer und optional Resemble Enhance | Wandelt eine vorhandene Sprachaufnahme anhand einer Referenzstimme in Audio zu Audio um; gibt das native 16-kHz-Ergebnis und optional eine neural restaurierte 44,1-kHz-Fassung aus. |
## Aufräumregel
Vor dem Löschen muss ein Kandidat gegen Compose-Dateien, Docker-Labels,
Mounts, Router-/Controller-Verweise und `/data` geprüft werden. Entfernt werden
nur nachweislich abgelöste Images, Gewichte, Versuchsdaten und Build-Caches.
Gewollt gestoppte Profilcontainer, persistente Modell-Caches und die letzte
funktionierende Produktionsvariante bleiben erhalten.
Am 9. September wurden die verworfenen Vevo2-Images und -Daten, das alte
Vevo2-Projektverzeichnis, ein leeres Test-Lab sowie der Docker-Build-Cache
entfernt. Der Build-Cache allein gab 74,43 GB frei; `/data` besitzt danach rund
562 GB freien Speicher. Kein produktiver oder bedarfsgesteuerter Container
wurde gelöscht.
+54 -10
View File
@@ -1,10 +1,55 @@
# Aktueller produktiver Laufzustand # Aktueller produktiver Laufzustand
Stand: 3. September 2026 Stand: 10. September 2026
## TRELLIS.2 3D-Studio
TRELLIS.2 4B läuft über trellis.cpp 0.6.0 als exklusiver Q8-Worker auf der
RTX 5080. Runtime und Q8-Gewichte liegen getrennt unter
`/data/trellis-studio` und `/data/models/trellis2-q8`; die Browseroberfläche
ist im WireGuard-Netz unter `http://192.168.1.212:8013` erreichbar. Ein realer
512er Ende-zu-Ende-Test erzeugte in 54,2 Sekunden ein gültiges 4,4-MB-GLB.
Für reguläre Qualitätsläufe ist `1024 · cascade` vorgesehen.
## Fotorestaurierung verworfen
Der versuchsweise HYPIR-SD2-Restaurationspfad wurde vollständig aus Router,
Compose und Hermes entfernt. HYPIR glättete beziehungsweise erfand beim realen
Testfoto Details; ein isolierter SeedVR2-7B-FP8-Test bewahrte das Motiv besser,
lieferte bei der starken Bewegungsunschärfe aber keinen ausreichenden
Qualitätsgewinn. Athena veröffentlicht deshalb kein Modell `restauration` und
Hermes besitzt keinen entsprechenden Skill mehr.
FLUX.2 Klein 9B bleibt für Bildgenerierung und kreative Referenzbild-Edits
aktiv. Details und Abnahmekriterien stehen in
[IMAGE_RESTORATION.md](IMAGE_RESTORATION.md).
## FLUX.2 Klein 9B FP8 Beta
Die bisherige 4B-Bildinferenz wurde testweise durch FLUX.2 Klein 9B FP8
ersetzt. Athenas Profile Controller stellt dafür einen exklusiven Zwei-GPU-Pfad
bereit:
- RTX 5080: 9B-FP8-Diffusionstransformer und VAE-Decoding
- RTX 3060: Qwen3-8B-Textencoder in NF4
- Qwen3-TTS und aktives llama.cpp-Profil werden für den Bildauftrag pausiert
- TTS ist währenddessen vorübergehend nicht verfügbar
- nach Abschluss werden TTS und das vorherige Textprofil wiederhergestellt
Ein vollständiger Aufruf über Athenas OpenAI-kompatiblen Router wurde mit
HTTP 200, einem korrekt gespeicherten 1024×1024-PNG und anschließender
Wiederherstellung von Qwen3-TTS und `qwen-fast` erfolgreich geprüft. Ein
isolierter Vergleich ergab ungefähr 14,6 Sekunden Bildlaufzeit mit dem
GPU-Textencoder gegenüber 102,1 Sekunden mit CPU-Textencoder. Diese Werte sind
eine lokale Einzelmessung und keine allgemeine Modellgarantie.
Das Modell ist nicht kommerziell lizenziert. Die Bedingungen der beiden
zugriffsbeschränkten Black-Forest-Labs-Repositories müssen vor dem Download
akzeptiert werden. Details stehen in [FLUX_9B_BETA.md](FLUX_9B_BETA.md).
## Lokale Spracherkennung ## Lokale Spracherkennung
Athena betreibt Whisper.cpp v1.9.1 mit `large-v3-turbo` als CPU-Dienst. Der Athena betreibt Whisper.cpp v1.9.1 mit `ggml-small` als CPU-Dienst. Der
Profile Router veröffentlicht ihn als OpenAI-kompatiblen Endpunkt Profile Router veröffentlicht ihn als OpenAI-kompatiblen Endpunkt
`/v1/audio/transcriptions`; Standardsprache ist Deutsch. Modell und Download `/v1/audio/transcriptions`; Standardsprache ist Deutsch. Modell und Download
bleiben im persistenten Docker-Volume `whisper-data` erhalten. bleiben im persistenten Docker-Volume `whisper-data` erhalten.
@@ -21,15 +66,14 @@ Alle Textprofile verwenden llama.cpp Build 10781,
Commit `c7bda030e7faee594dbe7550185e857351ad405d`. Der Stand enthält die ab Commit `c7bda030e7faee594dbe7550185e857351ad405d`. Der Stand enthält die ab
Build 10751 verfügbare Korrektur für eine zwischenzeitliche Build 10751 verfügbare Korrektur für eine zwischenzeitliche
MTP-/KV-Cache-Initialisierungsregression. Der vorherige produktive Stand war MTP-/KV-Cache-Initialisierungsregression. Der vorherige produktive Stand war
Build 10718, Commit `41ef91f7c8046087cdfbb276b79bff311ecf1c6d`, und bleibt über das alte Build 10718, Commit `41ef91f7c8046087cdfbb276b79bff311ecf1c6d`. Dessen lokales
lokale Image `mike-ai/llama.cpp:b10718-fallback` als unmittelbarer Fallback-Image wurde am 8. September 2026 beim gezielten Aufräumen entfernt;
Rückfallpunkt erhalten. ein Rückfall erfordert daher einen Neubau dieses Commits.
Build 10781 wurde nach dem Bau produktiv verifiziert: Alle fünf Build 10781 wurde nach dem Bau produktiv verifiziert. Alle fünf
Profilcontainer verwenden dasselbe neue Image, ausschließlich Medium läuft, Profildefinitionen verwenden dasselbe neue Image. Am 8. September lief Ultra
Qwen Medium ist mit 160.000 Tokens Kontext gesund, der MTP-Kontext wurde mit 262.144 Tokens Kontext gesund; MTP und eine lokale Textprobe wurden
erfolgreich initialisiert, der Vision-Projektor geladen und eine lokale erfolgreich geprüft.
Textprobe korrekt beantwortet.
## Qwen Medium: Vision-Projektor wieder aktiviert ## Qwen Medium: Vision-Projektor wieder aktiviert
+151
View File
@@ -0,0 +1,151 @@
# FLUX.2 Klein 9B FP8 Beta auf Athena
Stand: 7. September 2026
## Zweck und Status
Der Bildpfad ersetzt testweise FLUX.2 Klein 4B durch das größere
FLUX.2-Klein-9B-Modell. Ziel sind bessere Prompttreue, räumliche Beziehungen,
Objektkonsistenz und Referenzbild-Bearbeitung. Der Pfad ist technisch
funktionsfähig, bleibt aber bis zu weiteren Qualitäts- und Editing-Tests als
Beta bezeichnet.
Der OpenAI-kompatible Modellname lautet:
```text
FLUX.2-klein-9B-fp8-beta
```
## Modellartefakte und Lizenz
Verwendet werden zwei gepinnte, zugriffsbeschränkte Hugging-Face-Repositories:
| Zweck | Repository | Revision | Lokaler Pfad |
|---|---|---|---|
| Pipeline-Komponenten, Qwen3-Textencoder und VAE | `black-forest-labs/FLUX.2-klein-9B` | `92196c8e11f7b6cf2b7493e037d8c5345c559216` | `/data/models/FLUX.2-klein-9B-components` |
| FP8-Transformer | `black-forest-labs/FLUX.2-klein-9b-fp8` | `902d9d510b51533e07729f19211414a3648b77d2` | `/data/models/FLUX.2-klein-9B-fp8` |
FLUX.2 Klein 9B steht unter der FLUX Non-Commercial License. Vor dem Download
müssen die Bedingungen beider Repositories im verwendeten Hugging-Face-Konto
akzeptiert werden. Ein Token gehört ausschließlich in die durch
`HF_TOKEN_FILE` angegebene, für root lesbare Datei; niemals in Git oder
`stack.env`.
## GPU-Aufteilung
| Phase | RTX 5080, 16 GB | RTX 3060, 12 GB |
|---|---|---|
| Text-/Sprachbetrieb | aktives Qwen3.8-27B-Profil | Qwen3-TTS; Vision je nach Profil |
| Prompt-Encoding | FLUX-Transformer und VAE | Qwen3-8B-Textencoder, NF4 |
| Denoising | FLUX-Transformer | Textencoder wird nicht mehr benötigt |
| VAE-Decoding | VAE; Transformer zuvor freigegeben | Textencoder zuvor freigegeben |
Der Profile Controller stoppt vor dem Start des Bild-Workers alle
llama.cpp-Profile und den mit `com.mike-ai.tts-worker=qwen3` markierten
Qwen3-TTS-Container. Dadurch bleibt genügend VRAM für beide Bildkomponenten.
Nach dem Bildauftrag startet er Qwen3-TTS und das zuvor aktive Textprofil
wieder. Während des exklusiven GPU-Wechsels ist TTS vorübergehend nicht verfügbar.
## Aktuelle Grenzen
- genau 1024 × 1024 Pixel
- genau vier Inferenzschritte
- Guidance Scale 1,0
- ein Bildauftrag gleichzeitig
- höchstens vier bereits lokal gespeicherte Referenzbilder
- Textencoder-Maximum 128 Token
- Bildbearbeitung wird vom Worker angenommen, ist aber noch gesondert
Ende-zu-Ende zu qualifizieren
## Installation und Aktualisierung
In `/root/mike-ai-install.env` müssen diese Werte gesetzt sein:
```bash
HF_TOKEN_FILE=/root/.cache/huggingface/token
FLUX_COMPONENT_DIR=/data/models/FLUX.2-klein-9B-components
FLUX_TRANSFORMER_DIR=/data/models/FLUX.2-klein-9B-fp8
```
Anschließend lädt der normale Installer nur die benötigten Komponenten und die
gepinnten FP8-Gewichte. Bestehende, vollständige Dateien werden nicht erneut
geladen:
```bash
cd /opt/mike-ai/stack
sudo ./install.sh --config /root/mike-ai-install.env
```
## Funktionsprobe
Der Router ist nur über das private Netz erreichbar. Ein minimaler Test lautet:
```bash
curl -fsS http://192.168.1.212:8081/v1/images/generations \
-H "Authorization: Bearer $ROUTER_API_KEY" \
-H 'Content-Type: application/json' \
-d '{
"model":"FLUX.2-klein-9B-fp8-beta",
"prompt":"A yellow toy excavator on the left and a red toy truck on the right, studio photo",
"size":"1024x1024",
"steps":4,
"guidance":1.0,
"seed":9072026
}'
```
Danach müssen folgende Zustände wiederhergestellt sein:
```bash
docker ps --format '{{.Names}} {{.Status}}' \
--filter name=mike-ai-router \
--filter name=mike-ai-qwen3-tts \
--filter name=mike-ai-llama
docker ps -a --filter name=mike-ai-image-worker \
--format '{{.Names}} {{.Status}}'
nvidia-smi
```
Erwartet werden ein gesunder Router, gesundes Qwen3-TTS, genau ein gesundes
llama.cpp-Profil und ein mit Exit-Code 0 beendeter Bild-Worker.
## Hermes
Hermes auf Unraid verwendet einen persistenten Benutzer-Provider
`athena-local`. Seine Konfiguration muss auf denselben Modellnamen zeigen:
```yaml
image_gen:
provider: athena-local
model: FLUX.2-klein-9B-fp8-beta
max_parallel_requests: 1
```
Der Provider lebt in Hermes-Appdata und bleibt bei normalen Container-Updates
erhalten. Er gehört nicht in die Desktop-App und muss auf weiteren Clients
nicht erneut installiert werden. Die versionierte Quellfassung liegt unter
[`integrations/hermes-athena-image`](../integrations/hermes-athena-image).
## Rollback
Die lokalen 4B-Gewichte und die kurzfristigen Rückfall-Images wurden am
8. September 2026 nach erfolgreicher 9B-Abnahme gezielt entfernt. Ein Rollback
auf 4B ist deshalb weiterhin reproduzierbar, aber nicht mehr unmittelbar: Das
4B-Modell muss erneut geladen und die ältere Stack-Fassung neu gebaut werden.
Die zugehörige Deployment-Sicherung liegt auf Athena unter:
```text
/data/deploy-backups/20260907-flux9b-beta
```
Die vorherige Hermes-Konfiguration und der alte Provider liegen auf Unraid
unter:
```text
/mnt/nvme-storage/appdata/Hermes-Agent/backups/flux9b-beta-20260907
```
Ein Rollback darf nicht blind erfolgen: Zuerst aktives Profil, laufende
Anfragen und vorhandene Image-Tags prüfen, dann nur Image-Worker,
Profile Controller und Hermes-Provider auf den gesicherten Stand zurücksetzen.
+124 -8
View File
@@ -1,19 +1,24 @@
# Qwen Beta 1 – GSQ-RCO # Qwen Beta 1 – GSQ-RCO
`qwen-beta-1` ist ein zusätzliches, nicht standardmäßig aktives Router-Profil. > Historischer Testbericht. Das Beta-1-Profil wurde am 10. September 2026
Die bestehenden Profile und das Standardprofil `qwen-medium` bleiben unverändert. > vollständig aus dem produktiven Router entfernt, weil die kleinere
> Quantisierung gegenüber den Q4-Profilen keinen belastbaren Vorteil brachte.
`qwen-beta-1` war ein zusätzliches, nicht standardmäßig aktives Router-Profil.
Der Bericht bleibt erhalten, damit diese Quantisierung nicht versehentlich
erneut getestet wird.
## Laufzeitkonfiguration ## Laufzeitkonfiguration
- Modell: Qwen3.8-27B GSQ-RCO IQ3_XXS MTP - Modell: Qwen3.8-27B GSQ-RCO IQ3_S MTP
- Kontext: 192.000 Token - Kontext: 112.000 Token als konservativer Startwert
- Textmodell und KV-Cache: vollständig RTX 5080 - Textmodell und KV-Cache: vollständig RTX 5080
- Vision-Projektor: RTX 3060 - Vision-Projektor: RTX 3060
- KV-Quantisierung: Q4_0 für K und V - KV-Quantisierung: Q4_0 für K und V
- MTP: 3 Draft-Token - MTP: 3 Draft-Token
- Batch / Micro-Batch: 2048 / 128 - Batch / Micro-Batch: 2048 / 128
## Gemessene Kontextgrenze ## Vorherige IQ3_XXS-Kontextgrenze
Eine echte Bildanfrage mit einer 2,3-MB-JPEG-Datei wurde zur Bestimmung der Eine echte Bildanfrage mit einer 2,3-MB-JPEG-Datei wurde zur Bestimmung der
VRAM-Grenze verwendet. VRAM-Grenze verwendet.
@@ -24,9 +29,120 @@ VRAM-Grenze verwendet.
| 196.608 | bestanden, harte Kante | ca. 9 MiB | | 196.608 | bestanden, harte Kante | ca. 9 MiB |
| 197.120 | CUDA Out of Memory | ca. 1 MiB vor Abbruch | | 197.120 | CUDA Out of Memory | ca. 1 MiB vor Abbruch |
Der produktive Beta-Modus verwendet deshalb 192.000 Token. 196.608 ist nur Diese Werte gelten ausschließlich für die frühere, kleinere
die gemessene technische Obergrenze und besitzt keine ausreichende Reserve `IQ3_XXS-MTP`-Datei. Sie dürfen nicht als Grenze der größeren
für einen verlässlichen Dauerbetrieb. `IQ3_S-MTP`-Datei interpretiert werden. Das neue Profil startet bei 112.000
Token; seine technische und betrieblich sichere Grenze wird neu vermessen.
Beim erfolgreichen 196.608-Test erreichte die Bildanfrage rund 366 Prompt- Beim erfolgreichen 196.608-Test erreichte die Bildanfrage rund 366 Prompt-
Token/s und 85 Ausgabe-Token/s. Das erkannte Bild wurde korrekt beschrieben. Token/s und 85 Ausgabe-Token/s. Das erkannte Bild wurde korrekt beschrieben.
## Austausch am 08.09.2026
Die bisherige `IQ3_XXS-MTP`-Datei wurde durch `IQ3_S-MTP` ersetzt. ISTA
berichtet für die 3,5-bpw-Variante gegenüber BF16 identische Ergebnisse auf
AIME25 und LiveCodeBench v6 sowie 0,51 Punkte Abstand auf GPQA-Diamond. Diese
Herstellermessungen rechtfertigen den A/B-Test, ersetzen aber keine lokale
Prüfung mit Hermes-, Werkzeug- und Langkontextaufgaben.
## Lokaler A/B-Test am 08.09.2026
Beide Dateien liefen mit 112.000 Kontext, Q4_0-K/V-Cache, MTP 3, identischem
Sampling und einem 85:15-Layer-Split über RTX 5080 und RTX 3060.
| Messung | IQ4_XS Pure | GSQ-RCO IQ3_S MTP |
|---|---:|---:|
| deterministische Kurzaufgaben | 24/25 | 24/25 |
| Decode, 512 Token | 65,6 Token/s | 59,3 Token/s |
| Prefill, 30 Token | 184,5 Token/s | 251,6 Token/s |
Beide Modelle machten denselben einzelnen Fehler bei `2^100 modulo 13`. Im
lokalen Kurztest war damit kein Qualitätsverlust der neuen Quantisierung
messbar. Der kurze Prefill-Wert ist nur ein Laufzeitindikator und kein
Langkontext-Benchmark.
In der produktiven Beta-1-Verteilung liegt das komplette Textmodell auf der
RTX 5080 und nur der Vision-Projektor auf der RTX 3060. Dort wurden 89,9
Token/s Decode gemessen; nach dem Lauf blieben etwa 1.051 MiB auf der RTX 5080
frei. Ein realer Bildtest beschrieb Motiv und sichtbaren Text korrekt. Das
Profil war anschließend gesund. Die frühere IQ3_XXS-GGUF wurde erst nach diesen
Prüfungen entfernt; die JSON-Ergebnisse liegen auf Athena unter
`/data/model-benchmarks/gsq-rco-iq3s-ab-20260908/`.
## Profilweiter A/B-Härtetest am 08.09.2026
Ein zweiter Test verglich GSQ-RCO IQ3_S mit den jeweils heute verwendeten
Q4-Modellen unter den echten Kontext-, GPU-, MTP- und Batch-Einstellungen der
Profile. Medium und Large luden dabei auch den Vision-Projektor auf der RTX
3060; der dort bereits laufende TTS-Dienst blieb unangetastet. Alle acht
Varianten fanden drei synthetische Nadeln bei 70 Prozent des jeweiligen
Kontextfensters.
| Profil | Q4 kurzer Prefill | IQ3_S kurzer Prefill | Delta | Q4 Decode | IQ3_S Decode | Delta | Q4 Lang-Prefill | IQ3_S Lang-Prefill | Delta | Q4 Lang-Decode | IQ3_S Lang-Decode | Delta |
|---|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|---:|
| Fast 76,8K | 938,2 | 860,5 | -8,3 % | 115,3 | 109,2 | -5,2 % | 816,7 | 757,4 | -7,3 % | 74,7 | 76,4 | +2,3 % |
| Medium 160K | 1.449,7 | 1.342,3 | -7,4 % | 104,0 | 86,9 | -16,5 % | 948,3 | 897,3 | -5,4 % | 56,3 | 48,9 | -13,1 % |
| Large 192K | 1.456,5 | 1.342,6 | -7,8 % | 104,3 | 86,9 | -16,7 % | 849,5 | 808,9 | -4,8 % | 52,2 | 45,5 | -12,8 % |
| Ultra 262K | 1.728,3 | 1.288,2 | -25,5 % | 83,8 | 73,5 | -12,3 % | 808,2 | 677,8 | -16,1 % | 33,3 | 32,0 | -4,0 % |
Alle Geschwindigkeiten sind Token/s. `Fast` vergleicht das produktive
IQ4-MIX mit IQ3_S; die übrigen Profile vergleichen IQ4_XS Pure mit IQ3_S.
Die langen Prompts enthielten rund 53,8K, 112K, 134,5K beziehungsweise 183,6K
synthetische Token.
Der komplexere Qualitätstest bestand aus neun deutschsprachigen Aufgaben zu
Logik, evidenzgebundener Diagnose, nebenläufigem Python, Kapazitätsplanung,
Prompt-Injection, Laufzeit- gegenüber Konfigurationszustand und sicherem
Adminverhalten sowie einem nativen Tool-Call. Acht Aufgaben waren inhaltlich
gleichwertig; beide Modelle hatten beim nebenläufigen Python-Code denselben
subtilen Restfehler. Bei der Kapazitätsplanung ermittelten beide intern korrekt,
dass die Migration unmöglich ist. Beide erreichten jedoch das 4K-Ausgabelimit:
Q4 gab die Schlussfolgerung und fast den ganzen Beweis sichtbar aus, IQ3_S
verbrauchte das Limit vollständig im Reasoning und lieferte keinen sichtbaren
Antworttext. Beide nativen Tool-Calls waren korrekt.
Über alle neun Aufgaben benötigte Q4 223,0 Sekunden und IQ3_S 278,9 Sekunden;
IQ3_S war damit 25,0 Prozent länger beschäftigt. Zusammen mit der überwiegend
niedrigeren Inferenzgeschwindigkeit ist kein profilweiter Vorteil belegt.
Entscheidung: Die produktiven Q4-Profile werden nicht durch IQ3_S ersetzt und
es werden keine vollständigen Q3-Doppelprofile angelegt. `beta1` bleibt als
gezielter 112K-Versuch erhalten: Dort passt das gesamte Textmodell auf die RTX
5080, während der Projektor auf der RTX 3060 liegt. Dieser besondere
Platzierungsvorteil gilt nicht automatisch für die größeren Profile.
Die vollständigen JSON-Ergebnisse liegen auf Athena unter
`/data/model-benchmarks/gsq-rco-iq3s-ab-v2-20260908/`.
## Nachtest mit maximaler RTX-5080-Belegung am 08.09.2026
Der vorige Vergleich übernahm absichtlich die produktiven Q4-Tensor-Splits.
Dadurch nutzte IQ3_S seinen geringeren Platzbedarf nicht aus. In einem weiteren
reinen Geschwindigkeitstest wurde deshalb pro Profil der größtmögliche unter
echter Last stabile Anteil auf der RTX 5080 gesucht. TTS blieb auf der RTX 3060
geladen. Ein Split galt erst dann als stabil, wenn Modellstart, kurzer Test und
ein Prompt mit rund 70 Prozent des Kontextfensters vollständig durchliefen.
| Profil | stabiler IQ3_S-Split 5080:3060 | kurzer Prefill vs. Q4 | Decode vs. Q4 | Lang-Prefill vs. Q4 | Lang-Decode vs. Q4 |
|---|---:|---:|---:|---:|---:|
| Medium 160K | 96:4 | +1,1 % | -8,2 % | +2,9 % | -2,8 % |
| Large 192K | 96:4 | +0,8 % | -8,3 % | +1,9 % | -2,1 % |
| Ultra 262K | 88:12 | -20,1 % | -4,5 % | -12,4 % | +3,2 % |
Medium lief mit 96:4 stabil. 98:2 ließ sich zwar laden, stürzte jedoch beim
ersten langen Prompt ab; 99:1 scheiterte bereits beim Laden. Large lief mit
96:4 stabil, während 97:3 beim Laden des MTP-KV-Caches keinen ausreichenden
VRAM mehr hatte. Ultra lief mit 88:12 stabil. 92:8 und 90:10 ließen sich laden,
stürzten aber beim langen Prompt ab; 94:6 scheiterte bereits an den benötigten
Compute-Puffern. Die scheinbar nicht streng monotone Belegung entsteht durch
die diskrete Verteilung ganzer Tensoren beziehungsweise Layer und zusätzliche
KV-, MTP- und Compute-Puffer.
Alle drei stabilen Grenzläufe fanden erneut sämtliche drei Nadeln. Das stärkere
Ausreizen der RTX 5080 macht IQ3_S bei Medium und Large im Prefill knapp
schneller, beseitigt den Decode-Nachteil aber nicht. Bei Ultra steht einem
kleinen Vorteil von 3,2 Prozent im langen Decode ein deutlicher
Prompt-Verarbeitungsverlust gegenüber. Auch nach optimaler Platzierung ergibt
sich daher kein Geschwindigkeitsgrund, die produktiven Q4-Profile zu ersetzen.
Die optimierten JSON-Ergebnisse liegen im selben Benchmark-Verzeichnis und
tragen das Suffix `opt96-4` beziehungsweise `opt88-12`.
+49
View File
@@ -0,0 +1,49 @@
# Bewertung lokaler Fotorestaurierung
Stand: 8. September 2026
## Entscheidung
Athena betreibt derzeit **keinen separaten Fotorestaurationspfad**. Das
virtuelle Hermes-/Router-Modell `restauration`, der HYPIR-Worker und der dafür
angelegte Hermes-Skill wurden nach Ende-zu-Ende-Tests wieder entfernt.
Die normale Bildgenerierung und kreative Referenzbildbearbeitung mit
`FLUX.2-klein-9B-fp8-beta` bleiben davon unberührt. Sie sind jedoch kein Ersatz
für eine originalgetreue Restaurierung beschädigter oder stark unscharfer
Fotos.
## Getestete Ansätze
### HYPIR-SD2
HYPIR lief technisch als eigener Worker und war über den Athena-Router sowie
Hermes aufrufbar. Beim realen Testfoto wurden jedoch Strukturen geglättet oder
neu gezeichnet, statt vorhandene Details zuverlässig wiederherzustellen. Die
Identität und Geometrie kleiner Bildbereiche konnten driften. Das Ergebnis
erfüllte damit die Anforderung „gleiches Foto, nur sauberer und schärfer“
nicht.
### SeedVR2 7B FP8
SeedVR2 wurde isoliert auf Athena getestet, ohne es in Hermes oder den
produktiven Router einzubauen. Der Lauf bei 2048 × 1536 Pixeln war technisch
erfolgreich und bewahrte Komposition und Identität besser als HYPIR. Bei stark
verrauschtem und bewegungsunscharfem Ausgangsmaterial stellte das Modell aber
keine wesentlich brauchbareren Details her; Unschärfe und Rauschen blieben zu
großen Teilen bestehen.
## Konsequenz für die Architektur
- kein `restauration`-Modell in `/v1/models`
- kein Restaurationszweig im Profile Router
- kein `restoration-worker` in Docker Compose
- keine HYPIR- oder SeedVR2-Gewichte auf Athena
- kein `image-restoration`-Skill und kein Restaurationsmodell in Hermes
- Referenzbilder gehen weiterhin ausschließlich an FLUX und gelten als
kreative Bildbearbeitung
Ein neuer Restaurationspfad soll erst wieder aufgenommen werden, wenn ein
Kandidat am realen Testfoto einen klaren Qualitätsgewinn zeigt, Identität und
Geometrie zuverlässig bewahrt und auf Athenas RTX 5080/RTX 3060-Konfiguration
reproduzierbar läuft. Ein bloß technisch erfolgreicher Lauf reicht nicht.
+152
View File
@@ -0,0 +1,152 @@
# Athena-Betriebsmodi
Athena besitzt acht gegenseitig exklusive Betriebsmodi:
- `llm`: ein llama.cpp-Profil und Qwen3-TTS laufen; Spezialdienste sind gestoppt.
- `music`: ACE-Step 1.5 XL-SFT läuft; alle LLM-, Bild-, TTS- und Separator-Worker sind gestoppt.
- `yue2`: YuE2-3B und die angepasste `YuE2_WebUI` laufen; alle anderen
GPU-Dienste einschließlich ACE-Step sind gestoppt.
- `separation`: BS-RoFormer und Demucs trennen Musikspuren; ClearVoice trennt
Sprache von Hintergrundgeräuschen. LLM, Bild, TTS und ACE-Step sind gestoppt.
- `voice`: OmniVoice erzeugt Sprache aus Text mit einer gewählten
Referenzstimme. LLM, Bild, TTS, ACE-Step und Separator sind gestoppt.
- `voicechange`: X-VC überträgt eine vorhandene Sprachaufnahme auf eine
Referenzstimme und bewahrt dabei Inhalt und Timing. Alle anderen
GPU-Dienste sind gestoppt.
- `applio`: Applio stellt RVC-Inferenz, Modellverwaltung und Training bereit.
Alle anderen GPU-Dienste sind gestoppt.
- `trellis`: TRELLIS.2 4B Q8 erzeugt über trellis.cpp aus einem Eingabebild ein
texturiertes GLB. Der Worker läuft ausschließlich auf der RTX 5080; alle
anderen GPU-Dienste sind gestoppt.
Die Zustandsmaschine lebt im Athena-Router. Das Dashboard und Chat-Clients wie
Hermes sind nur Bedienoberflächen derselben API. Der zuletzt aktive LLM-Modus
wird persistent gespeichert und beim Verlassen eines Spezialmodus wieder geladen.
## Bedienung
Im Athena-Dashboard stehen **LLM-Betrieb**, **ACE-Step Studio**, **YuE2 Studio**, **Audio trennen**,
**Voice Studio**, **X-VC**, **Applio / RVC** und **3D Studio** bereit. Im Musikmodus werden zwei Oberflächen angeboten:
- **Original UI · stabil** öffnet die zum laufenden ACE-Step-Image gehörende
Gradio-Oberfläche. Sie ist für Cover, Remix und erweiterte Workflows der
verbindliche Produktionspfad.
- **Community UI · experimentell** öffnet `fspecii/ace-step-ui`. Die
CPU-leichte React/Express-Anwendung hält Bibliothek, Playlists und
Einstellungen in `/data/music/ace-step-ui`. Sie verwendet die offizielle
`/release_task`-API mit benannten Parametern und ist damit unabhängig von der
Reihenfolge der Gradio-Felder. Normale Generierung funktioniert; Cover und
Remix gelten bis zu eigenen Ende-zu-Ende-Tests weiterhin als experimentell.
Vor dem Start zeigt sie die übertragenen Werte an. Referenzaudio beeinflusst
nur Klang und Produktion, während Quellaudio Melodie, Rhythmus und Akkorde
erhält.
Die Community-Oberfläche ist im WireGuard-Netz unter
`http://192.168.1.212:7861`, die originale Gradio-Oberfläche unter
`http://192.168.1.212:7862` erreichbar. Beide Host-Ports bleiben zusätzlich
auf `127.0.0.1` gebunden und werden auf der Universitäts-Schnittstelle nicht
veröffentlicht. `ace-step-ui` ist reproduzierbar auf Commit
`a1fdf91829ec6f7b98844f80e323529cd155dbf2` fixiert und greift intern über das
Docker-Netz `mike-ai-music` auf `http://music-worker:7860` zu.
Das getrennte **YuE2 Studio** ist über WireGuard unter
`http://192.168.1.212:8014` erreichbar. Es verwendet YuE2-3B und die auf einen
festen Commit gesetzte `YuE2_WebUI` von Ladypoly. Analyse/Remix über
SheetSage2, Score-Übernahme und freie Generierung bleiben damit unabhängig vom
ACE-Step-Stack. Der Container trägt das Router-Label
`com.mike-ai.music-worker=yue2` und hängt als `yue2-studio` im privaten
Frontend-Netz. Ein Moduswechsel stoppt ihn zuverlässig, bevor LLM,
Audio-Trenner oder ein anderer GPU-Dienst gestartet werden.
Im Trennmodus öffnet das Dashboard die private Athena-Oberfläche unter
`http://192.168.1.212:8007`. Sie nimmt WAV, FLAC, MP3, M4A und weitere
übliche Formate an. Gewählt wird die herauszulösende Quelle: Gesang,
Schlagzeug, Bass, Gitarre, Piano, Sonstiges oder gereinigte Sprache. Das ZIP enthält genau diese Zielspur und
eine zweite FLAC-Datei mit dem vollständigen Rest ohne die Zielspur. Gesang
nutzt BS-RoFormer Viperx 1297, Schlagzeug/Bass `htdemucs_ft` und
Gitarre/Piano/Sonstiges experimentell `htdemucs_6s`. „Sonstiges“ ist dessen
gemischter `other`-Stem (unter anderem Synthesizer, Streicher, Bläser und Effekte),
nicht eine reine Synthesizer-Spur. Sprache nutzt das 48-kHz-Modell
`MossFormer2_SE_48K`; der Download enthält `speech.flac` und
`hintergrund-ohne-sprache.flac`. Die Musiktrennung basiert auf
`audio-separator` 0.47.0. Die ältere API-Auswahl kompletter 2-/4-/6-Stem-Sätze
bleibt rückwärtskompatibel.
Das Voice Studio ist ausschließlich über den privaten WireGuard-Pfad unter
`http://192.168.1.212:8008` erreichbar. Referenzstimmen werden unter
`/data/voice/studio/profiles` gespeichert. Die Oberfläche verlangt vor dem
Speichern eine Bestätigung der Nutzungsberechtigung. OmniVoice gibt
unkomprimiertes WAV aus und erzeugt Sprache aus Text; es verarbeitet keine
bereits eingesprochene Quellaufnahme.
Der X-VC Voice Changer ist ausschließlich unter
`http://192.168.1.212:8009` erreichbar. Er nimmt eine Quellaufnahme und eine
Referenzstimme an. Die Oberfläche behält immer das native 16-kHz-PCM-WAV und
erzeugt auf Wunsch zusätzlich mit Resemble Enhance eine neural restaurierte
44,1-kHz-Fassung. Diese zweite Datei rekonstruiert fehlende Sprachbandbreite;
sie stellt keine im 16-kHz-Signal tatsächlich erhaltenen Originaldetails wieder
her und bleibt deshalb direkt mit dem nativen Ergebnis vergleichbar. Die dokumentierte Sprachbasis
des verwendeten GLM-4-Voice-Tokenizers ist Chinesisch und Englisch; Deutsch
bleibt deshalb bis zur Hörabnahme ein Qualitätstest und kein zugesagter
Produktionspfad. X-VC läuft ausschließlich auf der RTX 5080.
Applio ist unter `http://192.168.1.212:8011` erreichbar. Der RVC-Pfad besitzt
eine eigene Modellbibliothek, Inferenz und Training. Hochwertige Inferenz
benötigt zwingend ein zuvor importiertes oder trainiertes RVC-Stimmenmodell
(`.pth`, optional `.index`). Eine bloße Referenzaufnahme genügt bei Applio
nicht. Der Code ist auf Commit
`7fa68ec2166ab1331c539704159fa14901e94e5a` fixiert.
Das TRELLIS.2-3D-Studio ist unter `http://192.168.1.212:8013` erreichbar. Es
verwendet trellis.cpp 0.6.0 und die Q8-Variante von TRELLIS.2 4B. Das Modell
läuft ausschließlich auf der RTX 5080; `1024 · cascade`, automatische
Hintergrundentfernung und `xatlas` sind die empfohlenen Standardwerte. Die UI
exportiert GLB. Ein nachgelagerter STL-/3MF-Export ist noch nicht Bestandteil
der Oberfläche.
Hermes benötigt dafür kein Plugin. Exakt eingegebene Steuerbefehle werden vom
Router lokal beantwortet, auch wenn gerade kein LLM geladen ist:
```text
/athena music
/athena yue2
/athena stems
/athena voice
/athena voicechange
/athena applio
/athena 3d
/athena trellis
/athena llm
/athena status
```
Die HTTP-Schnittstelle verwendet authentifizierte Requests:
```text
GET /mode
POST /mode {"mode":"music"}
POST /mode {"mode":"yue2"}
POST /mode {"mode":"separation"}
POST /mode {"mode":"voice"}
POST /mode {"mode":"voicechange"}
POST /mode {"mode":"applio"}
POST /mode {"mode":"trellis"}
POST /mode {"mode":"llm"}
```
Der Wechsel läuft asynchron. Fortschritt und Fehler stehen unter `mode` in
`GET /status`. Der Profile-Controller akzeptiert ausschließlich den mit
`com.mike-ai.music-worker=acestep` oder `com.mike-ai.music-worker=yue2` beziehungsweise
`com.mike-ai.stem-separator=bs-roformer` oder
`com.mike-ai.voice-worker=vevo2` beziehungsweise
`com.mike-ai.voice-change-worker=xvc` oder
`com.mike-ai.applio-worker=applio` oder
`com.mike-ai.trellis-worker=trellis2-q8` markierten Container. Freie
Container- oder Docker-Befehle werden nicht entgegengenommen.
## Wiederanlauf
Der Router speichert `mode`, `last_profile` und `return_profile` atomar. War
beim Router-Neustart ein Spezialmodus aktiv, startet er den passenden Worker erneut. Beim
Wechsel zurück wird das gespeicherte LLM-Profil semantisch auf Alias und
Kontextfenster geprüft, bevor Chat-Anfragen wieder freigegeben werden.
+197 -79
View File
@@ -1,97 +1,215 @@
# Backup und Wiederherstellung # Backup und vollständige Wiederherstellung
## Athena Athena besitzt zwei voneinander unabhängige Sicherungsebenen. Nur gemeinsam
decken sie Systemplatten-, Datenplatten- und Totalausfall ab.
`mike-ai-backup` erzeugt alle fünf Stunden ein Archiv unter ## Sicherungsebenen
`/data/docker-backups` und behält 14 Tage. Gesichert werden:
- `/etc/mike-ai` mit lokaler Konfiguration, | Ebene | Ziel | Takt | Zweck |
- Router-Zustand und erzeugte Bilder, |---|---|---:|---|
- Piper-Daten, | Lokales Schnellbackup | `/data/docker-backups` | alle 5 Stunden | schneller Wiederaufbau, wenn nur die Systemplatte stirbt |
- der kanonische Stack als zusätzlicher Snapshot. | Verschlüsseltes Disaster-Backup | externes Restic-Repository, bevorzugt Unraid | nachts | Wiederaufbau, wenn `/data` oder beide Platten sterben |
Nicht in das Archiv gehören die großen Modellgewichte unter `/data/models`. Ein Backup, das ausschließlich auf `/data` liegt, schützt ausdrücklich nicht
Sie bleiben auf der Daten-SSD oder werden anhand der gepinnten Angaben in vor dem Ausfall der Datenplatte.
`config/install.env.example` erneut geladen. Die Dashboard-Historie liegt
dauerhaft unter `/data/llama-dashboard`.
Portainers lokale Konfiguration liegt im Docker-Volume `portainer_data` und ### Lokales Schnellbackup
wird zusammen mit den übrigen nicht reproduzierbaren Volumes gesichert und
wiederhergestellt.
### Neuaufbau `mike-ai-backup` sichert:
1. Debian installieren und `/data` wieder am bisherigen Pfad einhängen. - `/etc/mike-ai`, einschließlich `install.env`, WireGuard und Geheimnissen,
2. Dieses Repository klonen. - ganz `/opt/mike-ai`, einschließlich aller bereitgestellten Spezialprojekte,
3. Installationsdatei ausfüllen und Installation starten: - Router-Zustand und Router-Bilder,
- Portainer-Daten.
Das Whisper-Volume ist reproduzierbar und wird bei Bedarf erneut geladen.
Ein vorhandener Hugging-Face-Token wird als root-only
`/etc/mike-ai/huggingface-token` mitgesichert, damit auch zugriffsbeschränkte
FLUX-Gewichte nach einem Datenverlust automatisch erneut geladen werden
können. Er steht niemals im Git-Repository.
### Externes Disaster-Backup
`athena-disaster-backup.timer` startet nachts ein verschlüsseltes,
dedupliziertes Restic-Backup. Vor jedem Lauf erzeugt es ein konsistentes
Docker-Schnellbackup und nimmt dieses in den externen Snapshot auf. Gesichert
werden außerdem:
- `/etc/mike-ai` und `/opt/mike-ai`,
- eigene Stimmen, Applio-Datasets und Trainingsstände unter `/data/voice`,
- Musikprojekte und Ausgaben unter `/data/music`,
- Audio-Trennungen unter `/data/audio`,
- Dashboard-, Operator-, Benchmark- und Projektdaten.
### Aktuelle TRELLIS-Lücke
Das am 10.09.2026 ergänzte 3D-Studio speichert seine Ausgaben unter
`/data/trellis-studio/output`. Dieser Pfad ist im derzeit ausgerollten
Export- und Disaster-Backup **noch nicht enthalten**. Wichtige GLB-Dateien
müssen bis zur Erweiterung der Backup-Skripte zusätzlich extern gesichert
werden. Runtime und Q8-Gewichte sind erneut ladbar; die vom Benutzer erzeugten
GLB-Dateien sind es nicht. Der Live-Code unter `/opt/mike-ai/trellis-studio`
wird vom lokalen Schnellbackup über `/opt/mike-ai` erfasst.
Die rund 100 GB reproduzierbaren Modellgewichte unter `/data/models` werden
nicht extern dupliziert. Kerngewichte lädt `install.sh` anhand URL und SHA256
neu. Spezialmodelle laden ihre gepinnten Container beim ersten Start erneut.
### Herunterladbare Notfallpakete
Zusätzlich erzeugt `athena-export-backup.timer` alle fünf Stunden ein mit Age
verschlüsseltes Komplettpaket der unersetzlichen Daten unter
`/data/emergency-backups`. Das Dashboard zeigt die letzten fünf Generationen
mit Größe, SHA256-Prüfsumme und einem fortsetzbaren Download an. Enthalten sind
insbesondere Applio-Logs und -Checkpoints, Datasets, eigene Stimmen,
Musikprojekte, Audioergebnisse, Konfiguration, Docker-Zustand und sämtliche
bereitgestellten Quellstände. Erneut ladbare Modell- und Hugging-Face-Caches
sind ausgeschlossen.
Bei aktuellem Datenbestand ist mit ungefähr 16 bis 20 GB je Generation zu
rechnen. Fünf Generationen benötigen daher grob 80 bis 100 GB auf `/data`.
Diese Pakete schützen nur dann vor einem Datenplattenausfall, wenn mindestens
eine Generation tatsächlich auf einen anderen Rechner oder Datenträger
heruntergeladen wurde. Die Pakete auf `/data` selbst sterben mit `/data`.
Der zu `recovery.age-recipient` gehörende private Age-Schlüssel darf nicht auf
Athena verbleiben. Ohne ihn können die Pakete absichtlich nicht entschlüsselt
werden.
## Einmalige Einrichtung des externen Backups
1. Ein physisch anderes Backupziel bereitstellen, vorzugsweise einen
ausschließlich über WireGuard erreichbaren Unraid-Share, und zum Beispiel
unter `/mnt/athena-offsite` einhängen.
2. Eine starke Restic-Passphrase erzeugen und **zusätzlich außerhalb Athenas**
in einem Passwortmanager oder auf einem Recovery-USB verwahren.
3. Konfiguration anlegen:
```bash ```bash
sudo ./install.sh --config /root/mike-ai-install.env cp config/disaster-backup.env.example /etc/mike-ai/disaster-backup.env
chmod 600 /etc/mike-ai/disaster-backup.env
# Repository, Mountpoint und Passwortdatei eintragen; danach:
sed -i 's/^DISASTER_BACKUP_ENABLED=false/DISASTER_BACKUP_ENABLED=true/' \
/etc/mike-ai/disaster-backup.env
``` ```
4. Letztes Datenarchiv einspielen: 4. Ersten Lauf und Snapshot prüfen:
```bash ```bash
systemctl start athena-disaster-backup.service
journalctl -u athena-disaster-backup.service --no-pager
restic snapshots --tag athena-disaster
```
Die externe Recovery-Konfiguration und die Passphrase bilden den kleinen
Recovery-Schlüssel. Eine Kopie davon muss außerhalb beider Athena-Platten
liegen. Ohne extern erreichbares Repository und dessen Schlüssel ist ein
Totalausfall mathematisch nicht wiederherstellbar.
## Gemeinsame Voraussetzung aller drei Fälle
Debian 13 ist frisch beziehungsweise weiterhin vorhanden. Die korrekte
Datenpartition ist formatiert und als **eigener Mountpoint** `/data`
eingehängt. `disaster-recovery.sh` partitioniert und formatiert absichtlich
nichts und bricht ab, wenn `/data` nur ein Verzeichnis auf der Systemplatte
ist. Dadurch kann es nicht versehentlich die falsche Platte überschreiben.
Der Installer darf einen kontrollierten Neustart für NVIDIA-Treiber oder die
stabile Netzwerkschnittstelle verlangen. Das Recovery-Skript startet Athena
niemals selbst neu. Nach dem manuellen Neustart wird derselbe Befehl erneut
ausgeführt; alle Schritte sind idempotent.
## Fall 1: Systemplatte defekt, Datenplatte erhalten
Nach Debian-Installation und Einhängen der alten `/data`-Platte:
```bash
sudo ./disaster-recovery.sh --scenario system \
--archive /data/docker-backups/athena-latest.tar.gz
```
Das Skript birgt Konfiguration und sämtliche `/opt/mike-ai`-Projekte aus dem
lokalen Archiv, installiert Docker/NVIDIA, verwendet die vorhandenen Modelle,
stellt die Docker-Volumes wieder her, baut Spezialcontainer und führt den
Smoke-Test aus.
Ältere Archive vor Einführung von `/etc/mike-ai/install.env` bleiben lesbar.
Bei einem solchen Archiv muss die Installationsdatei einmal separat angegeben
werden:
```bash
sudo ./disaster-recovery.sh --scenario system \
--archive /data/docker-backups/athena-latest.tar.gz \
--install-config /root/mike-ai-install.env
```
## Fall 2: Datenplatte defekt, Systemplatte erhalten
Neue Datenpartition unter `/data` einhängen und den extern aufbewahrten
Recovery-Schlüssel bereitstellen:
```bash
sudo ./disaster-recovery.sh --scenario data \
--config /root/athena-recovery.env
```
Eigene Daten und der letzte Docker-Zustand kommen aus Restic. Modellgewichte
werden anschließend automatisch neu geladen. Je nach Internetverbindung ist
dies der längste Teil der Wiederherstellung.
## Fall 3: Beide Platten defekt
Debian auf der neuen Systemplatte installieren, neue Datenpartition als
`/data` einhängen, dieses Git-Repository klonen und den externen
Recovery-Schlüssel bereitstellen:
```bash
sudo ./disaster-recovery.sh --scenario all \
--config /root/athena-recovery.env
```
Der externe Snapshot liefert Installationskonfiguration, Schlüssel,
Anwendungsquellen, Spezial-UIs, eigene Daten und Docker-Zustand. Danach werden
Pakete, Images und Modellgewichte reproduzierbar neu aufgebaut.
Alternativ kann ein zuvor aus dem Dashboard heruntergeladenes Notfallpaket
direkt verwendet werden:
```bash
sudo ./disaster-recovery.sh --scenario all \
--portable /mnt/usb/athena-portable-2026-09-10T15-00-00Z.tar.zst.age \
--identity /mnt/usb/athena-recovery-key.txt
```
## Ergebnis und Sicherheitsverhalten
Nach erfolgreichem Lauf gilt:
- Kernstack und Dashboard laufen,
- Medium ist das aktive LLM-Standardprofil,
- Spezialcontainer und ihre Oberflächen sind gebaut beziehungsweise erstellt,
- GPU-intensive Spezialworker bleiben gestoppt,
- keine automatische Umschaltung in Musik-, Bild-, Voice- oder Applio-Modus,
- `smoke-test.sh` hat den Kern geprüft.
Erst danach wird der gewünschte Spezialmodus über das Dashboard aktiviert.
## Regelmäßige Prüfung
Mindestens vierteljährlich einen Restore in eine leere Test-VM beziehungsweise
auf Testdatenträger durchführen. Ein grünes Backup-Log beweist nur, dass Daten
geschrieben wurden; erst ein Restore-Test beweist Wiederherstellbarkeit.
```bash
systemctl status athena-disaster-backup.timer
journalctl -u athena-disaster-backup.service --since '2 days ago'
restic snapshots --tag athena-disaster
sudo ./restore.sh --check /data/docker-backups/athena-latest.tar.gz sudo ./restore.sh --check /data/docker-backups/athena-latest.tar.gz
sudo ./restore.sh /data/docker-backups/athena-latest.tar.gz
sudo ./smoke-test.sh
``` ```
`--check` liest das komplette gzip-Archiv und prüft dessen sichere
`/backup`-Struktur sowie die benötigten Konfigurations- und Volume-Bäume, ohne
Container oder Dateien zu verändern. Der reguläre Restore extrahiert und
verwendet anschließend ausschließlich diesen einen geprüften Baum.
Das Restore verändert weder SSH noch LAN, WireGuard, Kernel, Partitionen oder
Mounts.
## Unraid ## Unraid
Hermes und die Fach-MCPs sind kein Bestandteil des Athena-Backups. Sie werden Hermes und die Fach-MCPs laufen auf Unraid und sind kein Bestandteil des
durch das vorhandene Unraid-Appdata-Backup gesichert: Athena-Restores. Sie werden weiterhin über das Unraid-Appdata-Backup gesichert.
Das Athena-Disaster-Repository muss auf einem anderen Datenträger beziehungsweise
- `/mnt/nvme-storage/appdata/Hermes-Agent` Storage-Pool als das zu schützende Athena-System liegen.
- die jeweiligen Appdata-Verzeichnisse der MCP-Container
- DockerMan-Templates unter
`/boot/config/plugins/dockerMan/templates-user/`
Container-Images stammen aus den dokumentierten Registries beziehungsweise den
eigenen Gitea-Repositories. Damit besteht die Wiederherstellung aus
Appdata-Restore plus Neuerstellung über die jeweilige Template-XML.
### Hermes Cron/Bot-Chat auf Unraid
Der offizielle Hermes-Build `0.21.0` mit Upstream-Stand `4b30b917` entfernt im
Cron-Zustellprozess fälschlich `HERMES_HOME`. Bei einem Docker-Datenverzeichnis
unter `/opt/data` findet `deliver=bot-chat:<profil>` dadurch vorhandene Profile
nicht. Bis zur Übernahme des Upstream-Fixes bindet die Unraid-Vorlage dieses
idempotente Startskript ein:
- Host: `/mnt/nvme-storage/appdata/Hermes-Agent/patches/025-cron-profile-root-fix`
- Container: `/etc/cont-init.d/025-cron-profile-root-fix` (read-only)
- Quelle: `platform/hermes/025-cron-profile-root-fix`
Das Skript entfernt nur die bekannte fehlerhafte Zeile. Ist sie in einem neuen
Image nicht mehr vorhanden, bleibt der Workaround automatisch wirkungslos. Es
stellt außerdem `/usr/local/bin/hermes` wieder her, weil der offizielle
Container den vom eigenen Doctor erwarteten CLI-Link derzeit nicht anlegt.
Nach einem Restore die Datei mit Modus `0755` ins Appdata kopieren, den Mount in
der DockerMan-Vorlage kontrollieren und den Container neu erstellen. Prüfung:
```bash
docker logs Hermes-Agent 2>&1 | grep cron-profile-root-fix
docker exec Hermes-Agent hermes cron doctor
```
## Kontrolle
```bash
docker compose --env-file /etc/mike-ai/stack.env ps
sudo ./restore.sh --check /data/docker-backups/athena-latest.tar.gz
curl -fsS http://192.168.1.212:8099/health
sudo ./smoke-test.sh
```
Anschließend einen Hermes-Chat, einen Router-Aufruf und je eine kleine
read-only-Abfrage der benötigten MCPs testen.
+63
View File
@@ -0,0 +1,63 @@
# Roadmap fuer spezialisierte lokale KI-Dienste
Stand: 10. September 2026
Diese Liste sammelt Nischenmodelle, die wir auf Athena nacheinander testen.
Ein Eintrag ist erst produktiv, wenn er auf der realen Hardware abgenommen und
im zentralen Register `TESTED_MODELS.md` dokumentiert wurde.
| Prioritaet | Aufgabe | Kandidat | Geplanter Betrieb | Status |
|---:|---|---|---|---|
| 1 | Musik erzeugen und bearbeiten | `ACE-Step 1.5 XL SFT` mit `acestep-5Hz-lm-1.7B` | exklusives On-Demand-Profil auf der RTX 5080; CPU-Offload; Qwen, Vision und TTS werden waehrenddessen entladen | **integriert; Klangabnahme laeuft** |
| 2 | Gesang und Instrumente trennen | BS-RoFormer Viperx 1297, `ep_317` | exklusiver Audio-Worker auf der RTX 5080; FLAC-Ausgabe; eigener Dashboard-Modus | **integriert; Qualitätstest läuft** |
| 3 | Voice Cloning | vorhandenes `Qwen3-TTS-12Hz-1.7B-Base` | bestehender TTS-Worker auf der RTX 3060; zunaechst den eingebauten 3-Sekunden-Klonpfad freilegen | offen |
| 4 | Objekte lokalisieren und zaehlen | Grounding DINO oder RF-DETR | optionaler Vision-Worker; normales Erkennen bleibt beim vorhandenen Qwen-Vision-Projektor | offen |
| 5 | Bildort schaetzen | GeoAgent 8B | exklusives Vision-Profil; Ergebnis nur als Wahrscheinlichkeitsrangliste | offen |
| 6 | Eigene Orte/Bilder wiederfinden | AnyLoc oder GME-Qwen2-VL-7B | Embedding-Index mit eigener Referenzdatenbank | offen |
| 7 | Bild zu texturiertem 3D-Modell | TRELLIS.2 4B Q8 über trellis.cpp 0.6.0 | exklusiver Worker auf RTX 5080; GLB; Standard 1024 | **integriert; technischer Ende-zu-Ende-Test bestanden** |
## Grundsaetze
- Qualitaet geht vor Dauerbetrieb: schwere Spezialmodelle duerfen die regulaeren
Profile voruebergehend entladen.
- Die RTX 5080 und RTX 3060 besitzen zusammen 28 GiB physischen VRAM, bilden
aber keinen gemeinsamen Speicherpool. Mehrkartenbetrieb muss vom jeweiligen
Modell beziehungsweise Backend ausdruecklich unterstuetzt werden.
- Jeder Test bleibt isoliert und entfernbar. Abgelehnte Images, Gewichte, Caches
und Integrationsreste werden nach der Dokumentation entfernt.
- Neue Kandidaten werden vor dem Download mit `TESTED_MODELS.md` abgeglichen.
## Erster Test: ACE-Step 1.5 XL SFT
Der erste Durchlauf nutzt nur die RTX 5080. Laut offiziellem Projekt benoetigt
XL mindestens 12 GiB mit Offload und empfiehlt mindestens 20 GiB ohne Offload.
Auf der 16-GiB-5080 wird deshalb der offiziell vorgesehene Offload-Pfad mit dem
1,7B-Musikplaner getestet. Die 3060 bleibt zunaechst frei; eine Verteilung ueber
beide Karten wird erst erwogen, wenn ACE-Step dafuer einen belastbaren
Inferenzpfad anbietet.
Abnahmekriterien:
1. Dienst startet reproduzierbar und belegt keine GPU im Ruhezustand.
2. Ein 30-Sekunden-Stueck wird ohne OOM erzeugt.
3. Laufzeit, Spitzen-VRAM, RAM-Nutzung und Ausgabedatei werden protokolliert.
4. Danach werden Ultra und TTS wiederhergestellt.
5. Erst nach bestandener Abnahme folgt die Hermes-Integration.
Erster Messlauf am 8. September 2026: Ein 30-Sekunden-Instrumental wurde in
15,39 Sekunden erzeugt (LM 8,00 s, DiT 7,39 s, MP3-Encoding 0,82 s). Die
gemeldete maximale CUDA-Allokation lag bei 9,38 GiB. Es gab weder OOM noch
CUDA-Fehler. Der technische Test ist damit bestanden; die subjektive
Die originale ACE-Step-Gradio-Oberflaeche ist der stabile Produktionspfad. Die
persistente `fspecii/ace-step-ui`-Oberflaeche bleibt bis zur Abnahme aller
Audio-zu-Audio-Modi experimentell. Am 10. September 2026 wurde der zerbrechliche
Aufruf der positionsabhaengigen `/generation_wrapper`-Schnittstelle entfernt.
Die Community-UI nutzt nun `/release_task` mit benannten Parametern; das
versionierte Worker-Derivat reicht dabei auch Referenz-/Quellaudio, Cover-
Staerke, Thinking, AI Enhance und die XL-SFT-Werte weiter.
Ein zehnsekündiger FLAC-Textauftrag lief am 8. September 2026 erfolgreich durch
UI, Express-Backend und Gradio-API und wurde in der persistenten Bibliothek
gespeichert; dieser Test belegt Cover und Remix ausdrücklich noch nicht.
Offizielle Referenzen: [ACE-Step 1.5](https://github.com/ace-step/ACE-Step-1.5)
und [REST-API](https://github.com/ace-step/ACE-Step-1.5/blob/main/docs/en/API.md).
-2
View File
@@ -8,7 +8,6 @@ Standardprofil: **medium** · globales Ausgabelimit: **8192 Token**
|---|---|---:|---:|---|---|---|---:| |---|---|---:|---:|---|---|---|---:|
| fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 | | fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 |
| medium | `qwen-medium` | 160,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 3 | | medium | `qwen-medium` | 160,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 3 |
| beta1 | `qwen-beta-1` | 192,000 | 1 | Qwen3.8-27B GSQ-RCO IQ3_XXS MTP | 5080 model / 3060 vision | ja | 3 |
| large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 | | large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 |
| ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 | | ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 |
| uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 | | uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 |
@@ -17,7 +16,6 @@ Standardprofil: **medium** · globales Ausgabelimit: **8192 Token**
- **fast**: Schnelles Profil für kurze Chats und zügige Werkzeugaufgaben. - **fast**: Schnelles Profil für kurze Chats und zügige Werkzeugaufgaben.
- **medium**: Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben. - **medium**: Ausgewogenes Standardprofil für Alltag und lange agentische Aufgaben.
- **beta1**: Beta 1: schnelles GSQ-RCO-Testprofil mit 192K Kontext und Vision-Projektor auf der RTX 3060.
- **large**: Großes Profil für umfangreiche Dokumente und lange technische Arbeiten. - **large**: Großes Profil für umfangreiche Dokumente und lange technische Arbeiten.
- **ultra**: Maximaler Textkontext; bewusst ohne Vision-Projektor. - **ultra**: Maximaler Textkontext; bewusst ohne Vision-Projektor.
- **uncensored**: Weniger restriktives Spezialprofil; Werkzeugrechte bleiben unverändert. - **uncensored**: Weniger restriktives Spezialprofil; Werkzeugrechte bleiben unverändert.
+96
View File
@@ -0,0 +1,96 @@
# Register getesteter Modelle
Stand: 10. September 2026
Dieses Dokument ist die zentrale Sperrliste gegen doppelte Modelltests. Vor
jedem Download müssen Repository, Dateiname, Basismodell, Fine-Tune und
Quantisierung hier geprüft werden. Unterschiedliche Quantisierungen desselben
Basismodells gelten als eigene Kandidaten.
Statuswerte:
- **produktiv**: wird von mindestens einem regulären Profil verwendet
- **Beta**: bleibt gezielt verfügbar, ersetzt aber nicht den Standard
- **verworfen**: getestet und ohne ausreichenden Gesamtvorteil
- **ersetzt**: früher genutzt oder getestet, inzwischen abgelöst
- **unvollständig**: Artefakt vorbereitet, aber kein belastbarer Abnahmetest
## Textmodelle auf Athena
| Datum | Exaktes Modell beziehungsweise Artefakt | Kontext im Test | Ergebnis | Status / Entscheidung | Beleg |
|---|---|---:|---|---|---|
| 22.08.2026 | `jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF` / `qwen3.8-27b-IQ4_XS-pure.gguf` | 160K–262K | beste ausgewogene Q4-Referenz; Langkontext, Tool-Call und Vision geprüft | **produktiv** für Medium, Large und Ultra | `benchmarks/qwen38-final-pre-move-20260822/` |
| 22.08.2026 | `vmarcelo/Qwen3.8-27B-MIX_GGUF` / `Qwen3.8-27B-IQ4-MIX.gguf` | 76,8K | schnellstes vollständig auf der RTX 5080 liegendes Q4-Profil | **produktiv** für Fast | `benchmarks/qwen38-final-pre-move-20260822/` |
| 22.08.2026 | Qwen3.8-27B NVFP4 `Q4_K_M` mit eingebettetem beziehungsweise separatem MTP | 72K | eingebettete Variante scheiterte beim Laden; Split-MTP lief, bot aber keinen ausreichenden Vorteil | **verworfen** | `benchmarks/qwen38-final-pre-move-20260822/qwen38-final-acceptance-20260822/` |
| 22.08.2026 | `Blackfrost-AI/Qwen3.8-27B-ABLITERATED-GGUF` / `Q4_K_M` | 80K | stabiler Spezialpfad mit Vision und MTP2 | **produktiv** für Uncensored | `benchmarks/qwen38-final-pre-move-20260822/qwen38-abliterated-final-20260822/` |
| 01.09.2026 | `peculiar-ragdoll/Dirk-Qwen3.8-27B-GGUF` / `UD-Q4_K_XL` | 80K–262K | korrekt und teils knapper, bei 160K aber 29–38 % langsamer im Decode als Pure | **verworfen** | [DIRK_QWEN38_AB_20260901.md](DIRK_QWEN38_AB_20260901.md) |
| 04.09.2026 | ISTA-DASLab Qwen3.8-27B GSQ-RCO `IQ3_XXS-MTP` | bis 196.608 | sehr platzsparend und bis 196.608 technisch lauffähig; später durch IQ3_S ersetzt | **ersetzt** | [GSQ_RCO_BETA1_20260904.md](GSQ_RCO_BETA1_20260904.md) |
| 07.09.2026 | `Jackrong/Qwopus3.8-27B-Flash-GGUF` / `Qwopus3.8-27B-Flash-MTP-IQ4_XS.gguf` | 160K | Recall 3/3; Decode 87,2 statt 105,3 Token/s, Lang-Decode 51,0 statt 56,7 Token/s; kein Gesamtvorteil | **verworfen** | Athena: `/data/benchmarks/qwen38-ab-20260907/` |
| 07.09.2026 | `bartowski/Qwen3.8-27B-GGUF` / `Qwen3.8-27B-IQ4_XS.gguf` | 160K | Recall 3/3; Decode 90,4 statt 105,3 Token/s, Lang-Decode 51,9 statt 56,7 Token/s; kein Gesamtvorteil | **verworfen** | Athena: `/data/benchmarks/qwen38-ab-20260907/` |
| 08.09.2026 | `ISTA-DASLab/Qwen3.8-27B-GSQ-RCO-GGUF` / `IQ3_S-MTP` | 76,8K–262K | Qualität im lokalen Test praktisch gleich, trotz optimierter GPU-Splits überwiegend langsamer als Q4 | **entfernt**; kein Ersatz für Q4 | [GSQ_RCO_BETA1_20260904.md](GSQ_RCO_BETA1_20260904.md) |
| 08.09.2026 | `Tiel-Coder-35B-A3B-UD-IQ4_XS.gguf` | 160K vorgesehen | Testcontainer und Gewichte vorhanden gewesen, aber kein versionierter, belastbarer Abnahmebericht | **unvollständig**; nicht als getesteter Sieger behandeln | kein Ergebnisartefakt vorhanden |
## Externe CPU-Helfermodelle
Diese Versuche liefen nicht als Athena-Hauptprofil, sind aber relevant für
Titelgenerierung und Kontextkompression in Hermes.
| Datum | Modell | Beobachtung | Entscheidung |
|---|---|---|---|
| 06.09.2026 | Ollama `qwen3:8b` | ungefähr 8,5–9,2 Token/s auf dem alten Dual-Xeon-Server | technisch brauchbar, aber für synchrone Hermes-Hilfsaufrufe langsam |
| 06.09.2026 | Ollama `gemma4:e4b` Q4 | ungefähr 4,7 Token/s auf dem HP EliteDesk, 10,2 auf dem alten Dual-Xeon und 15,1 auf dem neueren Proxmox-Host; mit Thinking liefen Hilfsaufrufe in Hermes in den 30-s-Timeout | nur mit `think:false` sinnvoll; nicht produktiv als Hermes-Auxiliary belegt |
## Bildmodelle und Restaurierung
| Datum | Modell | Ergebnis | Status / Entscheidung | Beleg |
|---|---|---|---|---|
| bis 07.09.2026 | FLUX.2 Klein 4B | funktional, aber schwächere räumliche und motivische Konsistenz | **ersetzt** durch 9B FP8 | [FLUX_9B_BETA.md](FLUX_9B_BETA.md) |
| 07.09.2026 | FLUX.2 Klein 9B FP8 | bessere Prompttreue; produktiver Zwei-GPU-Pfad, derzeit auf 1024 × 1024 begrenzt | **produktiv als Beta** | [FLUX_9B_BETA.md](FLUX_9B_BETA.md) |
| 08.09.2026 | HYPIR-SD2 | glättete oder erfand Details und veränderte kleine Strukturen | **verworfen** | [IMAGE_RESTORATION.md](IMAGE_RESTORATION.md) |
| 08.09.2026 | SeedVR2 7B FP8 | bewahrte Identität besser als HYPIR, brachte beim realen unscharfen Foto aber kaum nutzbare Details zurück | **verworfen** | [IMAGE_RESTORATION.md](IMAGE_RESTORATION.md) |
## Sprache
| Datum | Modell | Ergebnis | Status / Entscheidung |
|---|---|---|---|
| 05.09.2026 | Coqui XTTS v2 | deutsche Satzzeichen, Enden und Streaming-Chunks erzeugten Halluzinationen und unnatürliche Prosodie | **verworfen und entfernt** |
| 05.09.2026 | `Qwen/Qwen3-TTS-12Hz-1.7B-Base` | deutlich natürlichere deutsche Ausgabe ohne die XTTS-Endhalluzinationen | **produktiv** als einziges TTS-Backend |
| 06.–08.09.2026 | Whisper.cpp `large-v3-turbo` | lokaler TTS→STT-Rundlauf und OpenClaw-Transkription erfolgreich; später zugunsten des kleineren Laufzeitmodells entfernt | **ersetzt** |
| seit 03.09.2026 | Whisper.cpp `ggml-small` | tatsächlich im Compose-Stack und im laufenden Container verwendetes CPU-STT-Modell | **produktiv** |
| 09.09.2026 | `RMSnow/Vevo2`, Amphion `26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` | Technik und Geschwindigkeit funktionierten, reale deutsche Sprachwandlung mit kurzer und langer Referenz war jedoch unverständlich, halluzinierend oder musikalisch | **qualitativ verworfen und entfernt**; Ergebnis bleibt hier dokumentiert, Images, Daten und altes Projekt wurden am 09.09. bereinigt |
| 09.09.2026 | `k2-fsa/OmniVoice` 0.2.1 | Offizielle Gradio-UI auf RTX 5080 gestartet; Modell plus Whisper-ASR belegen rund 3,7 GiB VRAM. `omnivoice-triton` 0.1.0 ist kompatibel im Image vorhanden, für den ersten Hörtest aber bewusst noch nicht aktiviert | **technischer Starttest bestanden**, Hörabnahme und Basis-vs.-Triton-Messung offen; Gewichte CC BY-NC und daher nur nichtkommerziell einsetzen |
| 09.09.2026 | `chenxie95/X-VC`, Code `49df8c591eafc48b096e466d96f9839f9c0dd739`, UI-Basis `d761cd6421e85376b2656dfefd8471d7f35a42be` | Offizielles Beispiel Ende-zu-Ende gewandelt: 5,20 s Audio in 1,07 s (RTF 0,21), gültiges 16-kHz-Mono-PCM-WAV; Modell belegt rund 2,9 GiB auf der RTX 5080. GLM-4-Voice-Tokenizer dokumentiert Chinesisch und Englisch | **technischer Start- und Konvertierungstest bestanden**; deutsche Hörabnahme offen |
| 09.09.2026 | Resemble Enhance 0.0.1, Modellrevision `4e3510ce4a8391159f665903544c5150bee7b2cb` | 14,56 s native X-VC-Ausgabe bei 16 kHz wurden auf der RTX 5080 in 3,55 s zu 44,1-kHz-PCM-WAV restauriert. 3,27 % der gemessenen Signalenergie lagen danach oberhalb 8 kHz; damit ist der Pfad keine bloße Neuabtastung. Wegen der alten Upstream-Pins läuft die reine Inferenz mit NumPy 1.26.4/SciPy 1.11.4 auf dem bestehenden Torch-2.8/CUDA-12.8-Unterbau | **technisch produktiv als optionaler A/B-Pfad**; Hörabnahme entscheidet, ob die rekonstruierten Höhen subjektiv besser oder künstlicher klingen |
| 09.09.2026 | `Plachtaa/seed-vc` V1, Code `51383efd921027683c89e5348211d93ff12ac2a8` | Technisch vollständig lauffähig: gepinntes CUDA-Image, persistente Gewichte und reale WAV-Konvertierung mit etwa 3,6 GiB VRAM. Im deutschen Hörtest erhielt die Ausgabe jedoch einen deutlich chinesischen Akzent | **qualitativ verworfen und vollständig entfernt**; nicht erneut für deutsche Sprachwandlung einplanen |
| 09.09.2026 | `IAHispano/Applio`, Code `7fa68ec2166ab1331c539704159fa14901e94e5a` | Gepinntes CUDA-12.8-fähiges Image auf RTX 5080 gestartet; vollständige Applio/RVC-Oberfläche antwortet und CUDA ist verfügbar. Rund 1,8 GiB Basisgewichte und die Konfiguration wurden persistent ausgelagert. Es ist kein Zielstimmenmodell installiert; Applio kann aus einer Referenzaufnahme allein kein Modell ableiten | **technischer Start- und Persistenztest bestanden**; Konvertierung erst nach Import oder Training einer `.pth`-Stimme möglich |
## 3D-Erzeugung
| Datum | Modell | Test | Ergebnis | Status / Entscheidung | Beleg |
|---|---|---|---|---|---|
| 10.09.2026 | TRELLIS.2 4B Q8, zehn GGUF-Komponenten, trellis.cpp 0.6.0 | Bild-zu-3D bei 512, ausschließlich RTX 5080, Hintergrundentfernung `auto`, UV `xatlas` | HTTP 200 nach 54,2 s; gültiges GLB 2 mit 4,4 MB; Container und Browseroberfläche gesund | **technisch integriert**; 1024 ist der vorgesehene Qualitätsstandard, Druck- und subjektive Geometrieabnahme noch offen | Athena: `/opt/mike-ai/trellis-studio`, Gewichte: `/data/models/trellis2-q8` |
## Musikgenerierung
| Datum | Modell | Test | Ergebnis | Status / Entscheidung | Beleg |
|---|---|---|---|---|---|
| 08.–10.09.2026 | `ACE-Step/acestep-v15-xl-sft` mit `acestep-5Hz-lm-1.7B`, offizielles ACE-Step-1.5-Image `sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567` | Mehrere Instrumentaltests bis 244 s; abschließender Kontrolllauf mit geladenem 1.7B-Planer, `thinking=True`, XL-SFT 4B, 80 Schritten, Guidance 8 und Shift 3 | technisch vollständig und schnell, aber wiederholt nur Geräusche/Krach oder musikalisch chaotische Ergebnisse; der letzte Lauf schließt einen bloß fehlenden Planer als Ursache aus | **qualitativ verworfen**; nicht als Qualitätslösung weiterverfolgen | Athena: `/data/music/acestep/`; [SPECIALIZED_MODEL_ROADMAP.md](SPECIALIZED_MODEL_ROADMAP.md) |
| 10.09.2026 | `HeartMuLa/HeartMuLa-oss-3B-happy-new-year` mit `HeartMuLa/HeartCodec-oss-20260123`, heartlib `3783bdb8441f2c298b1e64c8651173aac200361c` | Mehrere Instrumentalversuche mit offiziellen Samplingwerten; HeartMuLa BF16 auf RTX 5080, HeartCodec FP32 auf RTX 3060 | technisch stabil und schnell, klanglich jedoch fast so unbrauchbar wie ACE-Step: Fantasiesprache trotz Instrumentalwunsch und gravierende Missachtung der Synthwave-/Synthpop-Stilvorgabe. Die lokale Test-UI hatte zusätzlich einen nicht upstream dokumentierten `[Instrumental]`-Marker verwendet | **qualitativ verworfen und vollständig entfernt**; Container, Image, Gewichte und Ausgaben am 10.09.2026 gelöscht | diese Tabelle; keine Laufzeitreste auf Athena |
| 10.09.2026 | `m-a-p/YuE2-3B` mit `m-a-p/YuE2-Vae`, offizieller Release `yue2-v0.1.6` | Unquantisiertes BF16 ausschließlich auf RTX 5080; leere Lyrics, expliziter Instrumentalstil, 118 BPM, Seed 831001 und vollständige symbolische Planung | 189,5 s Musik in 83,8 s erzeugt; 48 kHz, Stereo, 24-Bit-FLAC, keine Kürzung und kein OOM. Der erste Hörtest war im deutlichen Gegensatz zu ACE-Step und HeartMuLa musikalisch überzeugend. Entstehung über neuen ABC-Plan, 4.739 semantische Tokens und neue akustische Latents verifiziert; keine mitgelieferte Demo-Datei | **technischer Test und erste Hörabnahme bestanden**; isolierter Playground bereit, breitere Stil-/Gesangsprüfung und spätere Routerentscheidung noch offen | [yue2-3b](../experiments/yue2-3b/README.md); Athena: `/data/music/yue2/instrumental_synthwave_control/` |
## Audio-Trennung
| Datum | Modell | Test | Ergebnis | Status / Entscheidung | Beleg |
|---|---|---|---|---|---|
| 08.09.2026 | BS-RoFormer Viperx 1297, `model_bs_roformer_ep_317_sdr_12.9755.ckpt`, `audio-separator` 0.47.0 | 20-s-FLAC eines vorhandenen ACE-Step-Titels, RTX 5080, CUDA 12.8, ONNX Runtime GPU 1.22.0 | zwei gültige FLAC-Spuren mit jeweils exakt 20,0 s; Verarbeitung 19 s; Vocal-Datei 1,45 MB, Instrumental-Datei 3,84 MB | **technischer Ende-zu-Ende-Test bestanden**; Hörabnahme durch Nutzer offen | [bs-roformer-vocal-separation](../experiments/bs-roformer-vocal-separation/README.md) |
## Ablauf für zukünftige Kandidaten
1. Exakten Hugging-Face-/Ollama-Namen und Dateinamen in diesem Dokument suchen.
2. Bei einem Treffer zuerst den vorhandenen Beleg lesen; kein erneuter Download
ohne einen konkret neuen Grund wie Runtime, Quantisierung oder Hardware.
3. Neue Tests isoliert gegen das aktuelle Produktionsmodell mit identischem
Kontext, KV-Cache, MTP, Sampling und Promptset ausführen.
4. Unmittelbar danach hier Datum, exaktes Artefakt, Ergebnis, Entscheidung und
Pfad zum Detailbericht ergänzen.
5. Verworfene Gewichte nach gesichertem Ergebnis wieder löschen.
+90
View File
@@ -0,0 +1,90 @@
# ACE-Step 1.5 XL SFT: isolierter Athena-Test
Der GPU-Worker startet nur im Musikmodus und bindet seine rohe
Gradio-Oberflaeche nur an localhost. Die separate `fspecii/ace-step-ui`
bleibt als leichte React/Express-Oberflaeche aktiv; ihre SQLite-Datenbank,
Bibliothek und Uploads liegen persistent unter `/data/music/ace-step-ui`.
Das offizielle Image ist auf den am 8. September 2026 geladenen Digest
`sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567`
fixiert.
## Persistente Qualitaetsvorgaben
Die Weboberflaeche besitzt keine einzelne INI-Datei. Ihre Vorgaben kommen aus
Python-Modulen und teilweise aus dem Browser-`localStorage`. Deshalb bindet der
Compose-Dienst vier kleine, versionierte Overrides aus `./overrides` read-only
in den Container ein. Sie setzen fuer das XL-SFT-Modell:
- 80 DiT-Schritte, Guidance 8, Shift 3, ODE/Euler und CFG-Intervall 0 bis 1
- reine Stilreferenzen werden entsprechend der ACE-Step-API-Empfehlung automatisch mit Stärke 0,2 übertragen; Cover-/Quellaudio behält seine eigene Stärke
- ADG aus, keine benutzerdefinierten Timesteps
- FLAC als verlustfreie Standardausgabe
- 320 kbit/s als MP3-Ausweichwert
- Batchgroesse 1 fuer einen einzelnen Qualitaetslauf
- Normalisierung an bei -1 dB, kein Fade, Latent Shift 0, Latent Rescale 1
Die Preference-Schema-Version wurde auf 2 angehoben. Alte, im Browser
gespeicherte MP3/128-kbit/s-Werte werden dadurch einmalig verworfen; danach
bleiben bewusst vorgenommene Aenderungen wieder im jeweiligen Browser erhalten.
Beim Wechsel des gepinnten Image-Digests muessen die Overrides gegen die neue
Upstream-Fassung geprueft werden.
## Zwei Musikoberflaechen
Die originale Gradio-Oberflaeche aus demselben ACE-Step-Image ist der stabile
Produktionspfad fuer Simple, Custom, Cover, Remix und Repaint. Sie ist im
WireGuard-Netz unter `http://192.168.1.212:7862` erreichbar.
`music-ui` baut [fspecii/ace-step-ui](https://github.com/fspecii/ace-step-ui)
reproduzierbar von Commit `a1fdf91829ec6f7b98844f80e323529cd155dbf2`.
Die Community-Oberflaeche verwendet nicht mehr das positionsabhaengige
Gradio-Schema. Ihr Express-Dienst ruft die offizielle `/release_task`-API mit
benannten Feldern auf; das kleine Worker-Derivat erweitert diese Route um die
im installierten `GenerationParams` bereits vorhandenen Felder fuer Referenz-,
Quell- und Coveraudio sowie XL-SFT-Parameter. Athena-spezifisch sind ausserdem
die persistente Ablage, die XL-SFT-Anzeige und die gemeldeten Laufzeitlimits.
Schlaegt bei einer spaeteren Upstream-Fassung ein Patch-Anker fehl, bricht der
Image-Build ab. Die Community-UI unter `http://192.168.1.212:7861` ist bis zu
vollstaendigen Ende-zu-Ende-Tests von Cover und Remix als experimentell
gekennzeichnet.
Die Community-UI startet XL-SFT mit 80 Schritten, Guidance 8, Shift 3, FLAC
und aktivem Thinking ueber den 1,7B-Planer. `AI Enhance` wird als `use_format`
uebertragen. Vor jedem Auftrag zeigt sie die tatsaechlich gesendeten Parameter
und faengt offensichtliche Widersprueche ab. Referenzaudio steuert nur Klang
und Produktion; nur **Quellaudio / Cover** erhaelt Melodie, Rhythmus und
Akkorde.
## Start
Vor dem Start muessen das aktive llama.cpp-Profil und Qwen3-TTS beendet sein.
Die RTX 5080 wird ueber ihre UUID exklusiv an den Container uebergeben.
```bash
export ACESTEP_GPU_UUID="GPU-..."
export ACESTEP_UI_JWT_SECRET="$(openssl rand -hex 32)"
docker compose up -d music-ui
docker compose --profile music-test up -d music-worker
docker compose logs -f music-worker
```
Alternativ zum WireGuard-Zugriff lassen sich beide Oberflaechen per SSH-Tunnel
erreichen:
```bash
ssh -L 7861:127.0.0.1:7861 -L 7862:127.0.0.1:7862 root@athena.scc.kit.edu
```
`ace-step-ui` spricht den Worker ausschliesslich ueber das interne Docker-Netz
an. Dafuer startet ACE-Step mit aktivierten, benannten API-Endpunkten
(`--enable-api`).
## Beenden
```bash
docker compose --profile music-test stop music-worker
```
Der Befehl stoppt nur den GPU-Worker. Die Musikoberflaeche, ihre Bibliothek,
Caches, Modellgewichte und Ausgaben bleiben erhalten.
@@ -0,0 +1,38 @@
FROM node:22-bookworm AS build
ARG ACE_STEP_UI_COMMIT
RUN test -n "$ACE_STEP_UI_COMMIT"
RUN apt-get update \
&& apt-get install -y --no-install-recommends git python3 make g++ \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/fspecii/ace-step-ui.git /src \
&& cd /src \
&& git checkout --detach "$ACE_STEP_UI_COMMIT"
COPY patch-source.mjs /tmp/patch-source.mjs
RUN node /tmp/patch-source.mjs /src
RUN cd /src \
&& npm ci \
&& npm run build
RUN cd /src/server \
&& npm ci \
&& npm run build \
&& npm prune --omit=dev
FROM node:22-bookworm-slim AS runtime
RUN apt-get update \
&& apt-get install -y --no-install-recommends nginx curl ca-certificates ffmpeg \
&& rm -rf /var/lib/apt/lists/*
COPY --from=build /src/dist /usr/share/nginx/html
COPY --from=build /src/server/dist /app/server/dist
COPY --from=build /src/server/node_modules /app/server/node_modules
COPY --from=build /src/server/package.json /app/server/package.json
COPY --from=build /src/server/public /app/server/public
COPY --from=build /src/server/audio-editor /app/server/audio-editor
COPY nginx.conf /etc/nginx/nginx.conf
COPY entrypoint.sh /usr/local/bin/ace-step-ui-entrypoint
RUN chmod 0755 /usr/local/bin/ace-step-ui-entrypoint \
&& mkdir -p /data/audio /data/datasets/uploads
EXPOSE 3000 3001
ENTRYPOINT ["/usr/local/bin/ace-step-ui-entrypoint"]
@@ -0,0 +1,18 @@
#!/bin/sh
set -eu
node /app/server/dist/index.js &
server_pid=$!
trap 'kill "$server_pid" 2>/dev/null || true' INT TERM EXIT
nginx -g 'daemon off;' &
nginx_pid=$!
while kill -0 "$server_pid" 2>/dev/null && kill -0 "$nginx_pid" 2>/dev/null; do
sleep 1
done
kill "$server_pid" "$nginx_pid" 2>/dev/null || true
wait "$server_pid" 2>/dev/null || true
wait "$nginx_pid" 2>/dev/null || true
exit 1
@@ -0,0 +1,35 @@
worker_processes auto;
pid /tmp/nginx.pid;
events {
worker_connections 1024;
}
http {
include /etc/nginx/mime.types;
default_type application/octet-stream;
sendfile on;
client_max_body_size 512m;
server {
listen 3000;
server_name _;
root /usr/share/nginx/html;
index index.html;
location ~ ^/(api|audio|editor|blog|demucs-web)(/|$) {
proxy_pass http://127.0.0.1:3001;
proxy_http_version 1.1;
proxy_set_header Host $host;
proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for;
proxy_set_header X-Forwarded-Proto $scheme;
proxy_buffering off;
proxy_read_timeout 1800s;
proxy_send_timeout 1800s;
}
location / {
try_files $uri $uri/ /index.html;
}
}
}
@@ -0,0 +1,417 @@
import fs from 'node:fs';
import path from 'node:path';
const root = process.argv[2];
if (!root) throw new Error('source root argument is required');
function patch(relativePath, transform) {
const filename = path.join(root, relativePath);
const before = fs.readFileSync(filename, 'utf8');
const after = transform(before);
if (after === before) throw new Error(`patch made no change: ${relativePath}`);
fs.writeFileSync(filename, after);
}
function replaceOnce(text, before, after, label) {
const first = text.indexOf(before);
if (first < 0) throw new Error(`patch anchor missing: ${label}`);
if (text.indexOf(before, first + 1) >= 0) throw new Error(`patch anchor repeated: ${label}`);
return text.slice(0, first) + after + text.slice(first + before.length);
}
patch('server/src/services/acestep.ts', (text) => {
text = replaceOnce(
text,
"const AUDIO_DIR = path.join(__dirname, '../../public/audio');",
'const AUDIO_DIR = config.storage.audioDir;',
'persistent generated audio',
);
text = text.replace("import { handle_file } from '@gradio/client';\n", '');
text = text.replace('getGradioClient, ', '');
const helperStart = text.indexOf('// Gradio generation: map params');
const helperEnd = text.indexOf('/**\n * Download a Gradio audio result file', helperStart);
if (helperStart < 0 || helperEnd < 0) throw new Error('legacy Gradio helper anchors missing');
const helperCommentStart = text.lastIndexOf('// ---------------------------------------------------------------------------', helperStart);
const namedHelpers = `// ---------------------------------------------------------------------------
// Named REST generation through ACE-Step's official /release_task endpoint
// ---------------------------------------------------------------------------
function resolveAudioPath(audioUrl: string): string {
if (audioUrl.startsWith('/audio/')) {
return path.join(AUDIO_DIR, audioUrl.replace('/audio/', ''));
}
if (audioUrl.startsWith('http')) {
try {
const parsed = new URL(audioUrl);
if (parsed.pathname.startsWith('/audio/')) {
return path.join(AUDIO_DIR, parsed.pathname.replace('/audio/', ''));
}
} catch { /* fall through */ }
}
return audioUrl;
}
function resolveWorkerAudioPath(audioUrl: string | undefined): string | undefined {
if (!audioUrl) return undefined;
const localPath = resolveAudioPath(audioUrl);
if (!existsSync(localPath)) {
throw new Error(\`Uploaded audio is missing: \${localPath}\`);
}
const relativePath = path.relative(AUDIO_DIR, localPath);
if (relativePath.startsWith('..') || path.isAbsolute(relativePath)) {
throw new Error('Audio path is outside the shared Community UI storage');
}
return path.posix.join('/data/community-audio', relativePath.split(path.sep).join('/'));
}
function buildReleaseTaskPayload(params: GenerationParams): Record<string, unknown> {
const caption = params.style || 'pop music';
const prompt = params.customMode ? caption : (params.songDescription || caption);
const thinking = params.thinking ?? true;
const enhance = params.enhance ?? false;
const taskType = params.taskType === 'audio2audio' ? 'cover' : (params.taskType || 'text2music');
// A reference in text-to-music mode is only a global style/timbre guide.
// ACE-Step's own API guide recommends a low value (~0.2) for style transfer.
// Cover/source-audio jobs retain the explicitly selected cover strength.
const isStyleReference = taskType === 'text2music' && Boolean(params.referenceAudioUrl) && !params.sourceAudioUrl;
const effectiveAudioStrength = isStyleReference ? 0.2 : (params.audioCoverStrength ?? 1.0);
return {
prompt,
lyrics: params.instrumental ? '[Instrumental]' : (params.lyrics || ''),
instrumental: params.instrumental,
vocal_language: params.vocalLanguage || 'en',
bpm: params.bpm && params.bpm > 0 ? params.bpm : 0,
key_scale: params.keyScale || '',
time_signature: params.timeSignature || '',
audio_duration: params.duration && params.duration > 0 ? params.duration : -1,
inference_steps: params.inferenceSteps ?? 80,
guidance_scale: params.guidanceScale ?? 8.0,
shift: params.shift ?? 3.0,
infer_method: params.inferMethod || 'ode',
batch_size: Math.min(Math.max(params.batchSize ?? 1, 1), 16),
use_random_seed: params.randomSeed !== false,
seed: params.seed ?? -1,
thinking,
use_format: enhance,
lm_temperature: params.lmTemperature ?? 0.85,
lm_cfg_scale: params.lmCfgScale ?? 2.0,
lm_top_k: params.lmTopK ?? 0,
lm_top_p: params.lmTopP ?? 0.9,
lm_negative_prompt: params.lmNegativePrompt || 'NO USER INPUT',
use_cot_metas: thinking ? (params.useCotMetas ?? true) : false,
use_cot_caption: thinking ? (params.useCotCaption ?? true) : false,
use_cot_language: thinking ? (params.useCotLanguage ?? true) : false,
allow_lm_batch: params.allowLmBatch ?? true,
constrained_decoding_debug: params.constrainedDecodingDebug ?? false,
lm_batch_chunk_size: params.lmBatchChunkSize ?? 8,
task_type: taskType,
instruction: params.instruction || 'Fill the audio semantic mask based on the given conditions:',
reference_audio_path: resolveWorkerAudioPath(params.referenceAudioUrl),
src_audio_path: resolveWorkerAudioPath(params.sourceAudioUrl),
audio_codes: params.audioCodes || '',
repainting_start: params.repaintingStart ?? 0.0,
repainting_end: params.repaintingEnd ?? -1,
audio_cover_strength: effectiveAudioStrength,
use_adg: params.useAdg ?? false,
cfg_interval_start: params.cfgIntervalStart ?? 0.0,
cfg_interval_end: params.cfgIntervalEnd ?? 1.0,
audio_format: params.audioFormat || 'flac',
mp3_bitrate: '320k',
mp3_sample_rate: 48000,
};
}
`;
text = text.slice(0, helperCommentStart) + namedHelpers + text.slice(helperEnd);
const processStart = text.indexOf('// processGeneration — Gradio primary');
const processEnd = text.indexOf('function isAudioFile', processStart);
if (processStart < 0 || processEnd < 0) throw new Error('legacy generation anchors missing');
const processCommentStart = text.lastIndexOf('// ---------------------------------------------------------------------------', processStart);
const namedProcess = `// ---------------------------------------------------------------------------
// processGeneration — official named REST API only
// ---------------------------------------------------------------------------
async function processGeneration(
jobId: string,
params: GenerationParams,
job: ActiveJob,
): Promise<void> {
job.status = 'running';
job.stage = 'Preparing named ACE-Step request...';
if ((params.taskType === 'cover' || params.taskType === 'audio2audio') && !params.sourceAudioUrl && !params.audioCodes) {
job.status = 'failed';
job.error = \`task_type='\${params.taskType}' requires source audio or audio codes\`;
return;
}
try {
if (params.ditModel) {
job.stage = \`Loading model \${params.ditModel}...\`;
await switchModelIfNeeded(params.ditModel);
}
const payload = buildReleaseTaskPayload(params);
console.log(\`Job \${jobId}: POST /release_task with named parameters\`, payload);
job.stage = 'Generating music via named ACE-Step API...';
const releaseResponse = await fetch(\`\${ACESTEP_API}/release_task\`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify(payload),
});
const releaseText = await releaseResponse.text();
if (!releaseResponse.ok) {
throw new Error(\`ACE-Step /release_task failed (\${releaseResponse.status}): \${releaseText}\`);
}
const release = JSON.parse(releaseText) as any;
if (release.code !== 200 || !release.data?.task_id) {
throw new Error(release.error || 'ACE-Step returned no task_id');
}
const taskId = String(release.data.task_id);
job.taskId = taskId;
const queryResponse = await fetch(\`\${ACESTEP_API}/query_result\`, {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ task_id_list: [taskId] }),
});
const queryText = await queryResponse.text();
if (!queryResponse.ok) {
throw new Error(\`ACE-Step /query_result failed (\${queryResponse.status}): \${queryText}\`);
}
const query = JSON.parse(queryText) as any;
const taskResult = query.data?.[0];
const audioItems = taskResult?.result ? JSON.parse(taskResult.result) : [];
if (!Array.isArray(audioItems) || audioItems.length === 0) {
throw new Error('ACE-Step completed without downloadable audio results');
}
const audioUrls: string[] = [];
let actualDuration = 0;
for (const item of audioItems) {
if (!item?.url) continue;
const remoteUrl = new URL(item.url, ACESTEP_API).toString();
const remoteName = String(item.file || item.url);
const ext = path.extname(remoteName) || \`.\${params.audioFormat || 'flac'}\`;
const filename = \`\${jobId}_\${audioUrls.length}\${ext}\`;
const destPath = path.join(AUDIO_DIR, filename);
await downloadGradioAudioFile({ url: remoteUrl, orig_name: remoteName }, destPath);
if (audioUrls.length === 0) actualDuration = getAudioDuration(destPath);
audioUrls.push(\`/audio/\${filename}\`);
}
if (audioUrls.length === 0) throw new Error('ACE-Step returned no supported audio files');
const first = audioItems[0] || {};
job.status = 'succeeded';
job.result = {
audioUrls,
duration: actualDuration || Number(first.duration) || params.duration || 0,
bpm: Number(first.bpm) || params.bpm,
keyScale: first.keyscale || params.keyScale,
timeSignature: first.timesignature || params.timeSignature,
status: 'succeeded',
};
job.rawResponse = { release, query, transmittedParameters: payload };
console.log(\`Job \${jobId}: Completed via named REST API with \${audioUrls.length} audio files\`);
} catch (error) {
job.status = 'failed';
job.error = error instanceof Error ? error.message : String(error);
console.error(\`Job \${jobId}: Named REST generation failed\`, error);
}
}
`;
text = text.slice(0, processCommentStart) + namedProcess + text.slice(processEnd);
return text;
});
patch('server/src/services/storage/local.ts', (text) => {
text = replaceOnce(
text,
"import type { StorageProvider } from './index.js';",
"import type { StorageProvider } from './index.js';\nimport { config } from '../../config/index.js';",
'storage config import',
);
return replaceOnce(
text,
"const AUDIO_DIR = path.join(__dirname, '../../../public/audio');",
'const AUDIO_DIR = config.storage.audioDir;',
'persistent uploaded audio',
);
});
patch('server/src/index.ts', (text) => replaceOnce(
text,
"app.use('/audio', express.static(path.join(__dirname, '../public/audio')));",
"app.use('/audio', express.static(config.storage.audioDir));",
'persistent audio static route',
));
patch('server/src/routes/referenceTrack.ts', (text) => {
text = replaceOnce(
text,
"import { spawn } from 'child_process';",
"import { spawn } from 'child_process';\nimport { config } from '../config/index.js';",
'reference audio config import',
);
return replaceOnce(
text,
"const AUDIO_DIR = path.join(__dirname, '../../public/audio');",
'const AUDIO_DIR = config.storage.audioDir;',
'persistent reference audio',
);
});
patch('server/src/routes/generate.ts', (text) => {
text = replaceOnce(
text,
" thinking?: boolean;\n audioFormat?: 'mp3' | 'flac';",
" thinking?: boolean;\n enhance?: boolean;\n audioFormat?: 'mp3' | 'flac';",
'server enhance type',
);
const enhanceAnchor = ' thinking,\n audioFormat,';
if (text.split(enhanceAnchor).length - 1 !== 2) {
throw new Error('expected enhance anchor in destructuring and forwarding');
}
text = text.replaceAll(enhanceAnchor, ' thinking,\n enhance,\n audioFormat,');
text = replaceOnce(
text,
" const ALL_DIT_MODELS = [\n 'acestep-v15-turbo',",
" const ALL_DIT_MODELS = [\n 'acestep-v15-xl-sft', // Athena production model\n 'acestep-v15-turbo',",
'XL-SFT model list',
);
const start = text.indexOf("router.get('/limits'");
const end = text.indexOf("router.get('/debug/", start);
if (start < 0 || end < 0) throw new Error('limits route anchors missing');
const limits = `router.get('/limits', async (_req, res: Response) => {
// The UI container intentionally has no CUDA or ACE-Step Python runtime.
// These are the limits reported by Athena's dedicated RTX 5080 worker.
res.json({
tier: process.env.ACESTEP_TIER || 'tier5',
gpu_memory_gb: Number(process.env.ACESTEP_GPU_MEMORY_GB || 15.5),
max_duration_with_lm: Number(process.env.ACESTEP_MAX_DURATION_WITH_LM || 480),
max_duration_without_lm: Number(process.env.ACESTEP_MAX_DURATION_WITHOUT_LM || 600),
max_batch_size_with_lm: Number(process.env.ACESTEP_MAX_BATCH_WITH_LM || 4),
max_batch_size_without_lm: Number(process.env.ACESTEP_MAX_BATCH_WITHOUT_LM || 4),
});
});
`;
return text.slice(0, start) + limits + text.slice(end);
});
patch('services/api.ts', (text) => replaceOnce(
text,
" thinking?: boolean;\n audioFormat?: 'mp3' | 'flac';",
" thinking?: boolean;\n enhance?: boolean;\n audioFormat?: 'mp3' | 'flac';",
'client enhance type',
));
patch('App.tsx', (text) => {
text = replaceOnce(
text,
' thinking: params.thinking,\n audioFormat: params.audioFormat,',
' thinking: params.thinking,\n enhance: params.enhance,\n audioFormat: params.audioFormat,',
'client enhance forwarding',
);
return replaceOnce(
text,
' title: params.title,\n instrumental: params.instrumental,',
' title: params.title,\n ditModel: params.ditModel,\n instrumental: params.instrumental,',
'client model forwarding',
);
});
patch('components/CreatePanel.tsx', (text) => {
text = replaceOnce(text, 'useState(9.0);', 'useState(8.0);', 'guidance default');
text = replaceOnce(
text,
'useState(false); // Default false for GPU compatibility',
'useState(true); // Athena default: use the 1.7B planner for coherent structure',
'thinking default',
);
text = replaceOnce(text, "useState<'mp3' | 'flac'>('mp3');", "useState<'mp3' | 'flac'>('flac');", 'lossless default');
text = replaceOnce(text, 'useState(12);', 'useState(80);', 'XL-SFT steps default');
text = replaceOnce(text, "localStorage.getItem('ace-lmModel') || 'acestep-5Hz-lm-0.6B'", "localStorage.getItem('ace-lmModel') || 'acestep-5Hz-lm-1.7B'", 'planner model default');
// Upstream already defaults to Shift 3. Keep it instead of replacing it.
text = replaceOnce(
text,
' // Bulk generation: loop bulkCount times\n for (let i = 0; i < bulkCount; i++) {',
` const requestedText = customMode ? styleWithGender : songDescription;
const vocalRequestText = \`\${requestedText || ''}\\n\${lyrics}\`;
const explicitlyNoVocals = /\\b(no vocals?|without vocals?|instrumental only|kein(?:e[rs]?)? gesang|ohne gesang|keine stimme|ohne stimme)\\b/i.test(vocalRequestText);
const asksForVocals = !explicitlyNoVocals && /\\b(vocals?|singer|singing|male voice|female voice|gesang|stimme|sänger(?:in)?|singt)\\b/i.test(vocalRequestText);
if (!instrumental && asksForVocals && !lyrics.trim()) {
window.alert('Widerspruch: Der Auftrag verlangt Gesang, aber das Liedtextfeld ist leer. Bitte Text eintragen oder „Instrumental“ wählen.');
return;
}
if (instrumental && asksForVocals) {
window.alert('Widerspruch: „Instrumental“ ist aktiv, aber die Beschreibung verlangt Gesang. Bitte Gesangsbegriffe aus der Beschreibung entfernen oder „Instrumental“ deaktivieren und einen Liedtext eintragen.');
return;
}
if ((taskType === 'cover' || taskType === 'audio2audio') && !sourceAudioUrl.trim() && !audioCodes.trim()) {
window.alert('Für einen Cover-Auftrag fehlt das Quellaudio. Bitte unter „Quellaudio / Cover“ eine Datei auswählen.');
return;
}
const taskLabel = taskType === 'cover' || taskType === 'audio2audio' ? 'Cover / Audio-zu-Audio' : taskType;
const summary = [
'Folgende Parameter werden tatsächlich an Athena übertragen:',
'',
\`Aufgabe: \${taskLabel}\`,
\`Modell: \${selectedModel}\`,
\`Dauer: \${duration > 0 ? \`\${duration} Sekunden\` : 'automatisch'}\`,
\`Tempo: \${bpm > 0 ? \`\${bpm} BPM\` : 'automatisch'}\`,
\`Tonart: \${keyScale || 'automatisch'}\`,
\`Taktart: \${timeSignature || 'automatisch'}\`,
\`Thinking/Planung: \${thinking ? 'AN' : 'AUS'}\`,
\`AI Enhance: \${enhance ? 'AN' : 'AUS'}\`,
\`XL-SFT: \${inferenceSteps} Schritte, Guidance \${guidanceScale}, Shift \${shift}\`,
\`Gesang: \${instrumental ? 'nein (Instrumental)' : 'ja'}\`,
\`Referenzaudio (nur Klang/Produktion): \${referenceAudioUrl ? 'vorhanden' : 'keines'}\`,
\`Quellaudio (Melodie/Rhythmus/Akkorde): \${sourceAudioUrl ? 'vorhanden' : 'keines'}\`,
\`Audio-Einfluss: \${referenceAudioUrl && !sourceAudioUrl && taskType === 'text2music' ? '0,2 (sichere Stilreferenz)' : audioCoverStrength}\`,
\`Ausgabe: \${audioFormat.toUpperCase()}, \${batchSize} Variation(en), \${bulkCount} Auftrag/Aufträge\`,
'',
'Auftrag jetzt starten?',
].join('\\n');
if (!window.confirm(summary)) return;
// Bulk generation: loop bulkCount times
for (let i = 0; i < bulkCount; i++) {`,
'validation and transmitted parameter summary',
);
text = replaceOnce(
text,
" {t('reference')}\n </button>",
" Referenzaudio\n </button>",
'reference tab label',
);
text = replaceOnce(
text,
" {t('cover')}\n </button>",
" Quellaudio / Cover\n </button>",
'source tab label',
);
text = replaceOnce(
text,
' {/* Audio Content */}\n <div className="p-3 space-y-2">',
` {/* Audio Content */}
<div className="p-3 space-y-2">
<p className="text-[11px] leading-relaxed text-zinc-500 dark:text-zinc-400">
{audioTab === 'reference'
? 'Referenzaudio beeinflusst nur Klang, Instrumentierung und Produktion – nicht die Melodie.'
: 'Quellaudio / Cover erhält Melodie, Rhythmus und Akkorde des hochgeladenen Titels.'}
</p>`,
'audio semantics explanation',
);
return text;
});
+95
View File
@@ -0,0 +1,95 @@
services:
music-worker:
build:
context: ./worker
image: mike-ai/ace-step-1.5:named-api-v1
container_name: mike-ai-music-acestep-test
labels:
com.mike-ai.music-worker: "acestep"
profiles: ["music-test"]
environment:
ACESTEP_MODE: gradio
ACESTEP_CONFIG_PATH: acestep-v15-xl-sft
ACESTEP_LM_MODEL_PATH: acestep-5Hz-lm-1.7B
ACESTEP_LLM_BACKEND: pt
ACESTEP_INIT_SERVICE: "true"
ACESTEP_INIT_LLM: "true"
ACESTEP_DEVICE: cuda
# The image entrypoint forwards only ACESTEP_EXTRA_ARGS to the UI CLI.
# One result per run avoids the batch=2 VRAM/time penalty.
# Named Gradio endpoints are consumed by the separate ace-step-ui service.
ACESTEP_EXTRA_ARGS: "--batch_size 1 --enable-api"
TOKENIZERS_PARALLELISM: "false"
NVIDIA_VISIBLE_DEVICES: ${ACESTEP_GPU_UUID:?set ACESTEP_GPU_UUID to the RTX 5080 UUID}
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${ACESTEP_GPU_UUID:?set ACESTEP_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
ports:
# Raw Gradio remains available for diagnostics; users open music-ui below.
- "127.0.0.1:${ACESTEP_GRADIO_PORT:-7862}:7860"
volumes:
- ${ACESTEP_CHECKPOINTS_DIR:-/data/models/acestep/checkpoints}:/app/checkpoints
- ${ACESTEP_HF_CACHE_DIR:-/data/models/acestep/hf-cache}:/root/.cache/huggingface
- ${ACESTEP_OUTPUT_DIR:-/data/music/acestep}:/app/gradio_outputs
- ${ACESTEP_OUTPUT_DIR:-/data/music/acestep}:/app/output
# Uploaded Community-UI audio is shared read-only with the named REST API.
- ${ACESTEP_UI_DATA_DIR:-/data/music/ace-step-ui}/audio:/data/community-audio:ro
# Version-pinned UI defaults for XL-SFT quality and lossless output.
- ./overrides/model_config.py:/app/acestep/ui/gradio/events/generation/model_config.py:ro
- ./overrides/generation_advanced_output_controls.py:/app/acestep/ui/gradio/interfaces/generation_advanced_output_controls.py:ro
- ./overrides/user_preferences.py:/app/acestep/ui/gradio/interfaces/user_preferences.py:ro
- ./overrides/user_preferences.js:/app/acestep/ui/gradio/interfaces/user_preferences.js:ro
shm_size: "2gb"
restart: "no"
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:7860/ >/dev/null"]
interval: 30s
timeout: 10s
start_period: 300s
retries: 3
networks:
- music
- frontend
music-ui:
build:
context: ./ace-step-ui
args:
ACE_STEP_UI_COMMIT: a1fdf91829ec6f7b98844f80e323529cd155dbf2
image: mike-ai/ace-step-ui:a1fdf918
container_name: mike-ai-music-ui
environment:
NODE_ENV: production
PORT: "3001"
FRONTEND_URL: ${ACESTEP_UI_PUBLIC_URL:-http://192.168.1.212:7861}
ACESTEP_API_URL: http://music-worker:7860
DATABASE_PATH: /data/acestep.db
AUDIO_DIR: /data/audio
DATASETS_DIR: /data/datasets
DATASETS_UPLOADS_DIR: /data/datasets/uploads
JWT_SECRET: ${ACESTEP_UI_JWT_SECRET:?set ACESTEP_UI_JWT_SECRET}
ports:
- "127.0.0.1:${ACESTEP_UI_PORT:-7861}:3000"
volumes:
- ${ACESTEP_UI_DATA_DIR:-/data/music/ace-step-ui}:/data
restart: unless-stopped
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:3000/ >/dev/null && curl -fsS http://127.0.0.1:3001/health >/dev/null"]
interval: 30s
timeout: 10s
start_period: 30s
retries: 3
networks:
- music
- frontend
networks:
music:
name: mike-ai-music
frontend:
external: true
name: mike-ai_frontend
@@ -0,0 +1,216 @@
"""Output and automation controls for generation advanced settings."""
from typing import Any
import gradio as gr
from acestep.ui.gradio.i18n import t
_MP3_BITRATE_CHOICES = [("128 kbps", "128k"), ("192 kbps", "192k"), ("256 kbps", "256k"), ("320 kbps", "320k")]
_MP3_SAMPLE_RATE_CHOICES = [("48 kHz", 48000), ("44.1 kHz", 44100)]
def _update_mp3_control_visibility(audio_format: str, service_mode: bool = False):
"""Return visibility and interactivity updates for MP3-only controls."""
visible = audio_format == "mp3"
interactive = visible and not service_mode
return (
gr.update(visible=visible),
gr.update(choices=_MP3_BITRATE_CHOICES, visible=visible, interactive=interactive),
gr.update(choices=_MP3_SAMPLE_RATE_CHOICES, visible=visible, interactive=interactive),
)
def build_output_controls(
service_pre_initialized: bool,
service_mode: bool,
init_params: dict[str, Any] | None,
) -> dict[str, Any]:
"""Create audio-output and post-processing controls for advanced settings.
Args:
service_pre_initialized: Whether existing init params should prefill values.
service_mode: Whether the UI is running in service mode (disables some controls).
init_params: Optional startup state containing persisted output values.
Returns:
A component map containing format, scoring, normalization, and latent controls.
"""
params = init_params or {}
# Keep the master lossless. MP3 is only an optional sharing export.
initial_audio_format = params.get("audio_format", "flac")
initial_mp3_visible = initial_audio_format == "mp3"
with gr.Accordion(t("generation.advanced_output_section"), open=False, elem_classes=["has-info-container"]):
with gr.Row():
with gr.Column(scale=1):
audio_format = gr.Dropdown(
choices=[
("FLAC", "flac"),
("MP3", "mp3"),
("Opus", "opus"),
("AAC", "aac"),
("WAV (16-bit)", "wav"),
("WAV (32-bit Float)", "wav32"),
],
value=initial_audio_format,
label=t("generation.audio_format_label"),
info=t("generation.audio_format_info"),
elem_id="acestep-audio-format",
elem_classes=["has-info-container"],
interactive=not service_mode,
)
with gr.Row(visible=initial_mp3_visible) as mp3_controls_row:
mp3_bitrate = gr.Dropdown(
choices=[
("128 kbps", "128k"),
("192 kbps", "192k"),
("256 kbps", "256k"),
("320 kbps", "320k"),
],
value=params.get("mp3_bitrate", "320k"),
label=t("generation.mp3_bitrate_label"),
info=t("generation.mp3_bitrate_info"),
elem_id="acestep-mp3-bitrate",
elem_classes=["has-info-container"],
visible=initial_mp3_visible,
interactive=initial_mp3_visible and not service_mode,
scale=1,
)
mp3_sample_rate = gr.Dropdown(
choices=[
("48 kHz", 48000),
("44.1 kHz", 44100),
],
value=params.get("mp3_sample_rate", 48000),
label=t("generation.mp3_sample_rate_label"),
info=t("generation.mp3_sample_rate_info"),
elem_id="acestep-mp3-sample-rate",
elem_classes=["has-info-container"],
visible=initial_mp3_visible,
interactive=initial_mp3_visible and not service_mode,
scale=1,
)
with gr.Column(scale=1):
score_scale = gr.Slider(
minimum=0.01,
maximum=1.0,
value=0.5,
step=0.01,
label=t("generation.score_sensitivity_label"),
info=t("generation.score_sensitivity_info"),
elem_id="acestep-score-scale",
elem_classes=["has-info-container"],
scale=1,
visible=not service_mode,
)
audio_format.change(
fn=lambda value: _update_mp3_control_visibility(value, service_mode),
inputs=[audio_format],
outputs=[mp3_controls_row, mp3_bitrate, mp3_sample_rate],
)
with gr.Row():
enable_normalization = gr.Checkbox(
label=t("generation.enable_normalization"),
value=params.get("enable_normalization", True) if service_pre_initialized else True,
info=t("generation.enable_normalization_info"),
elem_id="acestep-enable-normalization",
elem_classes=["has-info-container"],
)
normalization_db = gr.Slider(
label=t("generation.normalization_db"),
minimum=-10.0,
maximum=0.0,
step=0.1,
value=params.get("normalization_db", -1.0) if service_pre_initialized else -1.0,
info=t("generation.normalization_db_info"),
elem_id="acestep-normalization-db",
elem_classes=["has-info-container"],
)
with gr.Row():
fade_in_duration = gr.Slider(
label=t("generation.fade_in_duration"),
minimum=0.0,
maximum=10.0,
step=0.1,
value=params.get("fade_in_duration", 0.0) if service_pre_initialized else 0.0,
info=t("generation.fade_in_duration_info"),
elem_id="acestep-fade-in-duration",
elem_classes=["has-info-container"],
)
fade_out_duration = gr.Slider(
label=t("generation.fade_out_duration"),
minimum=0.0,
maximum=10.0,
step=0.1,
value=params.get("fade_out_duration", 0.0) if service_pre_initialized else 0.0,
info=t("generation.fade_out_duration_info"),
elem_id="acestep-fade-out-duration",
elem_classes=["has-info-container"],
)
with gr.Row():
latent_shift = gr.Slider(
label=t("generation.latent_shift"),
minimum=-0.2,
maximum=0.2,
step=0.01,
value=params.get("latent_shift", 0.0) if service_pre_initialized else 0.0,
info=t("generation.latent_shift_info"),
elem_id="acestep-latent-shift",
elem_classes=["has-info-container"],
)
latent_rescale = gr.Slider(
label=t("generation.latent_rescale"),
minimum=0.5,
maximum=1.5,
step=0.01,
value=params.get("latent_rescale", 1.0) if service_pre_initialized else 1.0,
info=t("generation.latent_rescale_info"),
elem_id="acestep-latent-rescale",
elem_classes=["has-info-container"],
)
return {
"audio_format": audio_format,
"mp3_controls_row": mp3_controls_row,
"mp3_bitrate": mp3_bitrate,
"mp3_sample_rate": mp3_sample_rate,
"score_scale": score_scale,
"enable_normalization": enable_normalization,
"normalization_db": normalization_db,
"fade_in_duration": fade_in_duration,
"fade_out_duration": fade_out_duration,
"latent_shift": latent_shift,
"latent_rescale": latent_rescale,
}
def build_automation_controls(service_mode: bool) -> dict[str, Any]:
"""Create automation controls for LM batch chunking.
Args:
service_mode: Whether the UI is running in service mode (disables some controls).
Returns:
A component map containing ``lm_batch_chunk_size``.
"""
with gr.Accordion(
t("generation.advanced_automation_section"),
open=False,
elem_classes=["has-info-container"],
):
with gr.Row():
lm_batch_chunk_size = gr.Number(
label=t("generation.lm_batch_chunk_label"),
value=8,
minimum=1,
maximum=32,
step=1,
info=t("generation.lm_batch_chunk_info"),
scale=1,
interactive=not service_mode,
elem_id="acestep-lm-batch-chunk-size",
elem_classes=["has-info-container"],
)
return {"lm_batch_chunk_size": lm_batch_chunk_size}
@@ -0,0 +1,201 @@
"""Model configuration and UI control settings for generation handlers.
Contains functions for determining model type (turbo/base/pure-base),
producing UI control configurations, and computing gr.update() tuples
for model-type-dependent controls.
"""
import re
import gradio as gr
from acestep.constants import (
TASK_TYPES_TURBO,
TASK_TYPES_BASE,
GENERATION_MODES_TURBO,
GENERATION_MODES_BASE,
)
def _has_token(token: str, path: str) -> bool:
"""Check if *token* appears as a delimited word in *path*.
Matches when *token* is bounded by start/end of string or a common
path delimiter (``/``, ``\\``, ``.``, ``_``, ``-``).
"""
return re.search(rf"(^|[\\\\/._-]){token}($|[\\\\/._-])", path) is not None
def is_pure_base_model(config_path_lower: str) -> bool:
"""Check whether a model path refers to a pure base model.
Args:
config_path_lower: Lowercased model config path string.
Returns:
``True`` when the path contains ``"base"`` and excludes ``"sft"`` and ``"turbo"``.
"""
return (
_has_token("base", config_path_lower)
and not _has_token("sft", config_path_lower)
and not _has_token("turbo", config_path_lower)
)
def update_model_type_settings(config_path: str | None, current_mode: str | None = None) -> tuple:
"""Update UI settings based on model type (fallback when handler not initialized yet).
Args:
config_path: Model config path string.
current_mode: Current generation mode value to preserve across choices update.
Returns:
Ten-element tuple of ``gr.update()`` dicts for inference_steps,
guidance_scale, use_adg, shift, cfg_interval_start, cfg_interval_end,
task_type, generation_mode, init_llm_checkbox, and dcw_enabled.
"""
if config_path is None:
config_path = ""
config_path_lower = config_path.lower()
# Precedence: turbo > SFT > pure base > fallback.
# Detection functions enforce mutual exclusivity.
is_turbo = _has_token("turbo", config_path_lower)
is_pure_base = is_pure_base_model(config_path_lower)
is_sft = is_sft_model(config_path_lower)
return get_model_type_ui_settings(is_turbo, current_mode=current_mode, is_pure_base=is_pure_base, is_sft=is_sft)
def is_sft_model(config_path_lower: str) -> bool:
"""Check whether a model path refers to an SFT (supervised fine-tuned) model.
Args:
config_path_lower: Lowercased model config path string.
Returns:
``True`` when the path contains ``"sft"`` and excludes ``"turbo"``.
"""
return _has_token("sft", config_path_lower) and not _has_token("turbo", config_path_lower)
def is_xl_model(config_path_lower: str) -> bool:
"""Check whether a model path refers to an XL (4B DiT) variant.
Args:
config_path_lower: Lowercased model config path string.
Returns:
``True`` when the path contains ``"xl"`` as a delimited token.
"""
return _has_token("xl", config_path_lower)
def get_ui_control_config(is_turbo: bool, is_pure_base: bool = False, is_sft: bool = False) -> dict:
"""Return UI control configuration (values, limits, visibility) for model type.
Args:
is_turbo: Whether the model is a turbo variant.
is_pure_base: Whether the model is a pure base model.
is_sft: Whether the model is an SFT (supervised fine-tuned) variant.
SFT models are optimized for 50 inference steps, matching the
training defaults in model_discovery._BASE_DEFAULTS.
Used by both interactive init and service-mode startup so controls stay consistent.
"""
# Precedence: turbo > SFT > pure base > fallback.
if is_pure_base:
task_choices = TASK_TYPES_BASE
mode_choices = GENERATION_MODES_BASE
else:
task_choices = TASK_TYPES_TURBO
mode_choices = GENERATION_MODES_TURBO
if is_turbo:
return {
"inference_steps_value": 8,
"inference_steps_maximum": 20,
"inference_steps_minimum": 1,
"guidance_scale_visible": False,
"use_adg_visible": False,
"shift_value": 3.0,
"shift_visible": True,
"dcw_enabled_value": True,
"cfg_interval_start_visible": False,
"cfg_interval_end_visible": False,
"task_type_choices": task_choices,
"generation_mode_choices": mode_choices,
}
else:
# SFT models use 50 steps; pure base / unknown models use 32.
steps = 50 if is_sft else 32
return {
"inference_steps_value": steps,
"inference_steps_maximum": 200,
"inference_steps_minimum": 1,
"guidance_scale_visible": True,
"use_adg_visible": True,
# ACE-Step XL-SFT was trained/recommended with shift=1.0.
# Keep 3.0 only for non-SFT base/unknown variants.
"shift_value": 1.0 if is_sft else 3.0,
"shift_visible": True,
"dcw_enabled_value": False,
"cfg_interval_start_visible": True,
"cfg_interval_end_visible": True,
"task_type_choices": task_choices,
"generation_mode_choices": mode_choices,
}
def get_model_type_ui_settings(is_turbo: bool, current_mode: str | None = None, is_pure_base: bool = False, is_sft: bool = False):
"""Get gr.update() tuple for model-type controls.
Args:
is_turbo: Whether the model is a turbo variant.
current_mode: Current generation mode value to preserve.
is_pure_base: Whether the model is a pure base model.
is_sft: Whether the model is an SFT variant.
Returns:
Tuple of updates for inference_steps, guidance_scale, use_adg,
shift, cfg_interval_start, cfg_interval_end, task_type,
generation_mode, init_llm_checkbox, and dcw_enabled.
"""
cfg = get_ui_control_config(is_turbo, is_pure_base=is_pure_base, is_sft=is_sft)
new_choices = cfg["generation_mode_choices"]
if current_mode and current_mode in new_choices:
mode_update = gr.update(choices=new_choices, value=current_mode)
else:
mode_update = gr.update(choices=new_choices)
init_llm_update = gr.update(value=False) if is_pure_base else gr.update()
return (
gr.update(
value=cfg["inference_steps_value"],
maximum=cfg["inference_steps_maximum"],
minimum=cfg["inference_steps_minimum"],
),
gr.update(visible=cfg["guidance_scale_visible"]),
gr.update(visible=cfg["use_adg_visible"]),
gr.update(value=cfg["shift_value"], visible=cfg["shift_visible"]),
gr.update(visible=cfg["cfg_interval_start_visible"]),
gr.update(visible=cfg["cfg_interval_end_visible"]),
gr.skip(), # task_type (gr.State — no-op on model config change)
mode_update,
init_llm_update,
gr.update(value=cfg["dcw_enabled_value"]),
)
def get_generation_mode_choices(is_pure_base: bool = False) -> list:
"""Get the list of generation mode choices based on model type.
Args:
is_pure_base: Whether the model is a pure base model.
Returns:
List of mode choice strings.
"""
if is_pure_base:
return GENERATION_MODES_BASE
else:
return GENERATION_MODES_TURBO
@@ -0,0 +1,164 @@
/**
* User preferences persistence – SAVE side only.
*
* Listens for user changes on Gradio UI controls and persists the current
* values to browser localStorage. Restoration is handled on the Python side
* via ``gr.Blocks.load()`` so Gradio's own Svelte reactivity updates every
* component correctly.
*
* Storage schema:
* key = "acestep.ui.user_preferences"
* value = JSON { _version: 2, audio_format: "flac", … }
*/
(() => {
const STORAGE_KEY = "acestep.ui.user_preferences";
const SCHEMA_VERSION = 2;
const DEBOUNCE_MS = 500;
/**
* Map of preference key → { elemId, type }.
* elemId : the HTML elem_id set in Gradio
* type : "dropdown" | "slider" | "checkbox" | "number"
*/
const PREFS = {
audio_format: { elemId: "acestep-audio-format", type: "dropdown" },
mp3_bitrate: { elemId: "acestep-mp3-bitrate", type: "dropdown" },
mp3_sample_rate: { elemId: "acestep-mp3-sample-rate", type: "dropdown" },
score_scale: { elemId: "acestep-score-scale", type: "slider" },
enable_normalization:{ elemId: "acestep-enable-normalization", type: "checkbox" },
normalization_db: { elemId: "acestep-normalization-db", type: "slider" },
fade_in_duration: { elemId: "acestep-fade-in-duration", type: "slider" },
fade_out_duration: { elemId: "acestep-fade-out-duration", type: "slider" },
latent_shift: { elemId: "acestep-latent-shift", type: "slider" },
latent_rescale: { elemId: "acestep-latent-rescale", type: "slider" },
lm_batch_chunk_size: { elemId: "acestep-lm-batch-chunk-size", type: "number" },
};
let saveTimer = null;
const wiredElements = new WeakSet();
// ── Storage helpers ──────────────────────────────────────────────
const saveAll = (prefs) => {
try {
window.localStorage.setItem(STORAGE_KEY, JSON.stringify(prefs));
} catch (_e) {
// Private browsing or quota exceeded – silently ignore.
}
};
// ── DOM helpers ──────────────────────────────────────────────────
const findInput = (elemId, type) => {
const wrapper = document.getElementById(elemId);
if (!wrapper) return null;
if (type === "dropdown") {
return wrapper.querySelector("input");
}
if (type === "slider") {
return wrapper.querySelector("input[type='range']")
|| wrapper.querySelector("input[type='number']");
}
if (type === "checkbox") {
return wrapper.querySelector("input[type='checkbox']");
}
if (type === "number") {
return wrapper.querySelector("input[type='number']");
}
return null;
};
const readValue = (key) => {
const spec = PREFS[key];
if (!spec) return undefined;
const el = findInput(spec.elemId, spec.type);
if (!el) return undefined;
if (spec.type === "checkbox") return el.checked;
if (spec.type === "slider" || spec.type === "number") {
const v = Number(el.value);
return Number.isFinite(v) ? v : undefined;
}
return el.value || undefined;
};
// ── Save (debounced) ─────────────────────────────────────────────
const scheduleSave = () => {
if (saveTimer !== null) {
clearTimeout(saveTimer);
}
saveTimer = setTimeout(() => {
saveTimer = null;
const prefs = { _version: SCHEMA_VERSION };
for (const key of Object.keys(PREFS)) {
const v = readValue(key);
if (v !== undefined) {
prefs[key] = v;
}
}
saveAll(prefs);
}, DEBOUNCE_MS);
};
// ── Wire listeners (re-entrant – safe to call on re-renders) ─────
const wireListeners = () => {
for (const key of Object.keys(PREFS)) {
const spec = PREFS[key];
const el = findInput(spec.elemId, spec.type);
if (!el || wiredElements.has(el)) continue;
wiredElements.add(el);
el.addEventListener("input", scheduleSave, { passive: true });
el.addEventListener("change", scheduleSave, { passive: true });
}
};
// ── MutationObserver – re-wire after Gradio re-renders ───────────
const startObserver = () => {
const target = document.getElementById("acestep-audio-format")
|| document.body;
const root = target.closest(".gradio-container") || document.body;
let rafPending = false;
new MutationObserver(() => {
if (rafPending) return;
rafPending = true;
requestAnimationFrame(() => {
rafPending = false;
wireListeners();
});
}).observe(root, { childList: true, subtree: true });
};
// ── Boot ─────────────────────────────────────────────────────────
const BOOT_POLL_MS = 200;
const BOOT_TIMEOUT_MS = 10000;
const boot = () => {
const started = Date.now();
const poll = () => {
const probe = document.getElementById(
PREFS.audio_format.elemId
);
if (!probe) {
if (Date.now() - started < BOOT_TIMEOUT_MS) {
setTimeout(poll, BOOT_POLL_MS);
}
return;
}
wireListeners();
startObserver();
};
poll();
};
if (document.readyState === "loading") {
document.addEventListener("DOMContentLoaded", boot, { once: true });
} else {
boot();
}
})();
@@ -0,0 +1,258 @@
"""Frontend user-preference persistence helpers for the Gradio UI.
Save side: A ``<script>`` injected via ``Blocks(head=…)`` listens for DOM
changes and writes the current preference values to ``localStorage``.
Restore side: ``wire_preference_restore`` attaches a ``demo.load()`` handler
whose *js* parameter reads ``localStorage`` on page load and feeds the saved
values straight into the Gradio component outputs. Because Gradio itself
applies the updates through its own Svelte reactivity, every component type
(dropdown, slider, checkbox, number) is updated correctly—no fragile DOM
hacking required.
"""
from __future__ import annotations
import json
from functools import partial
from pathlib import Path
from typing import Any
_ASSET_FILENAME = "user_preferences.js"
_STORAGE_KEY = "acestep.ui.user_preferences"
_SCHEMA_VERSION = 2
# Ordered list of preference keys. The order here MUST match the order of
# *outputs* passed to ``demo.load()`` in ``wire_preference_restore``.
PREF_KEYS: list[str] = [
"audio_format",
"mp3_bitrate",
"mp3_sample_rate",
"score_scale",
"enable_normalization",
"normalization_db",
"fade_in_duration",
"fade_out_duration",
"latent_shift",
"latent_rescale",
"lm_batch_chunk_size",
]
# Default values used when localStorage is empty or the schema version has
# changed. Keys must match ``PREF_KEYS``.
_DEFAULTS: dict[str, Any] = {
"audio_format": "flac",
"mp3_bitrate": "320k",
"mp3_sample_rate": 48000,
"score_scale": 0.5,
"enable_normalization": True,
"normalization_db": -1.0,
"fade_in_duration": 0.0,
"fade_out_duration": 0.0,
"latent_shift": 0.0,
"latent_rescale": 1.0,
"lm_batch_chunk_size": 8,
}
# ── Save-side: head script injection ────────────────────────────────────
def _load_preferences_script() -> str:
"""Load the external save-preferences JavaScript asset."""
asset_path = Path(__file__).with_name(_ASSET_FILENAME)
return asset_path.read_text(encoding="utf-8").strip()
def get_user_preferences_head() -> str:
"""Return Gradio head HTML that injects save-side preference persistence."""
script_source = _load_preferences_script()
return f"<script>\n{script_source}\n</script>"
# ── Restore-side: Gradio .load() wiring ─────────────────────────────────
def _build_restore_js(num_outputs: int) -> str:
"""Build the client-side JS that reads localStorage and returns values.
The returned function is passed as the ``js`` parameter to
``demo.load()``. It returns an array whose element order matches
``PREF_KEYS`` (and therefore the *outputs* list).
When localStorage has no saved preferences (first visit, cleared
storage, private browsing), the function returns an array of ``null``
sentinels so the Python side can skip the update and preserve whatever
values were already rendered from ``init_params``.
Args:
num_outputs: Total number of output components (preference keys
plus any extra outputs like ``mp3_controls_row``).
"""
keys_json = json.dumps(PREF_KEYS)
# Build a type map so the restore JS can validate each value.
type_map: dict[str, str] = {}
for k in PREF_KEYS:
v = _DEFAULTS[k]
if isinstance(v, bool):
type_map[k] = "boolean"
elif isinstance(v, (int, float)):
type_map[k] = "number"
else:
type_map[k] = "string"
type_map_json = json.dumps(type_map, ensure_ascii=False)
# Keys whose Gradio Dropdown choices are integers stored as strings in
# localStorage. Only actual dropdown keys with numeric defaults need
# coercion; sliders/numbers are already stored as numbers.
numeric_dropdown_keys_json = json.dumps(["mp3_sample_rate"])
# Sentinel array returned when there is nothing to restore. Using null
# lets the Python fn detect "no stored prefs" and return gr.update()
# for every output, preserving the values already rendered on the page.
skip_sentinel = f"new Array({num_outputs}).fill(null)"
return f"""() => {{
const STORAGE_KEY = {json.dumps(_STORAGE_KEY)};
const SCHEMA_VERSION = {_SCHEMA_VERSION};
const KEYS = {keys_json};
const TYPE_MAP = {type_map_json};
const NUMERIC_COERCE_KEYS = new Set({numeric_dropdown_keys_json});
const SKIP = {skip_sentinel};
try {{
const raw = window.localStorage.getItem(STORAGE_KEY);
if (!raw) return SKIP;
const prefs = JSON.parse(raw);
// Only reset on downgrade; forward-compatible additions of new
// keys are handled by skipping (preserving init_params).
if (prefs._version !== SCHEMA_VERSION) {{
return SKIP;
}}
const result = KEYS.map(k => {{
if (!(k in prefs)) return null;
let v = prefs[k];
// Type-check: fall back to null (skip) if the stored type
// does not match what the Gradio component expects.
const expected = TYPE_MAP[k];
if (expected && typeof v !== expected) {{
// Allow stringified numbers for dropdown coercion below.
if (!(NUMERIC_COERCE_KEYS.has(k) && typeof v === "string")) {{
return null;
}}
}}
// Coerce stringified numbers back for Dropdown choices that
// expect integers (e.g. mp3_sample_rate: 48000 not "48000").
if (NUMERIC_COERCE_KEYS.has(k) && typeof v === "string") {{
const n = Number(v);
if (Number.isFinite(n)) v = n;
else return null;
}}
return v;
}});
// If none of the keys had stored values, skip entirely.
if (result.every(v => v === null)) return SKIP;
// Compute mp3 control visibility from audio_format (index 0).
// Push 3 extra values: mp3_controls_row, mp3_bitrate, mp3_sample_rate
// matching the outputs of _update_mp3_control_visibility().
// When audioFormat is null (no stored value), push nulls so Python
// emits gr.update() and preserves whatever init_params set.
const audioFormat = result[0];
const mp3 = audioFormat === null ? null : audioFormat === "mp3";
result.push(mp3, mp3, mp3);
return result;
}} catch (_e) {{
return SKIP;
}}
}}"""
def restore_preferences(
*values: Any, _num_outputs: int = 0
) -> tuple[Any, ...]:
"""Map JS restore results into Gradio output values.
The JS function reads localStorage and produces an array:
- First ``len(PREF_KEYS)`` elements are preference values (or null).
- Next 3 elements are mp3 visibility booleans (or null):
[mp3_controls_row, mp3_bitrate, mp3_sample_rate].
``None`` (JSON ``null``) → ``gr.update()`` (no-op, preserves current).
Booleans beyond PREF_KEYS → visibility/interactivity updates matching
``_update_mp3_control_visibility()`` from the output controls module.
When the JS side returns no values (e.g. certain Gradio versions do not
forward the JS return value to the Python ``fn`` when ``inputs=None``),
``_num_outputs`` is used to produce the correct number of no-op updates
so Gradio does not raise a ``ValueError`` about mismatched output count.
"""
import gradio as gr
if not values:
return tuple(gr.update() for _ in range(_num_outputs))
n_prefs = len(PREF_KEYS)
results: list[Any] = []
for i, v in enumerate(values):
if v is None:
results.append(gr.update())
elif i == n_prefs and isinstance(v, bool):
# mp3_controls_row: visibility only.
results.append(gr.update(visible=v))
elif i > n_prefs and isinstance(v, bool):
# mp3_bitrate, mp3_sample_rate: visibility + interactivity.
results.append(gr.update(visible=v, interactive=v))
else:
results.append(v)
return tuple(results)
def wire_preference_restore(
demo: Any,
generation_section: dict[str, Any],
*,
service_mode: bool = False,
) -> None:
"""Attach a ``demo.load()`` handler that restores saved preferences.
Must be called **inside** the ``with gr.Blocks() as demo:`` context,
after all generation components have been created.
In service mode the function is a no-op: service-mode sessions use
server-side ``init_params`` and controls are locked
(``interactive=False``), so localStorage values must not override them.
Args:
demo: The ``gr.Blocks`` instance.
generation_section: Merged component dict that includes the output
control components (``audio_format``, ``mp3_bitrate``, etc.).
service_mode: When ``True``, skip wiring entirely so that
localStorage cannot override server-configured values.
"""
if service_mode:
return
outputs = []
for key in PREF_KEYS:
component = generation_section.get(key)
if component is None:
raise KeyError(
f"wire_preference_restore: missing component {key!r} in "
f"generation_section (available: {sorted(generation_section)})"
)
outputs.append(component)
# Also update mp3 control visibility so it stays in sync when the
# restored audio_format differs from the server-rendered default.
# Gradio does not fire .change() for load-time value assignments, so
# without this the MP3 row and its children could be visible/hidden
# incorrectly. The three extra outputs mirror the return of
# _update_mp3_control_visibility(): [row, bitrate, sample_rate].
for mp3_key in ("mp3_controls_row", "mp3_bitrate", "mp3_sample_rate"):
comp = generation_section.get(mp3_key)
if comp is not None:
outputs.append(comp)
demo.load(
fn=partial(restore_preferences, _num_outputs=len(outputs)),
inputs=None,
outputs=outputs,
js=_build_restore_js(num_outputs=len(outputs)),
)
@@ -0,0 +1,6 @@
FROM ghcr.io/ace-step/ace-step-1.5:latest@sha256:95652cd780c78a1b1a7f6f0335530430f0ae53d96c7c12d59f9f39fa23d38567
COPY patch-api-routes.py /tmp/patch-api-routes.py
RUN /usr/bin/python3 /tmp/patch-api-routes.py \
/app/acestep/ui/gradio/api/api_routes.py \
&& rm /tmp/patch-api-routes.py
@@ -0,0 +1,137 @@
"""Extend ACE-Step's official /release_task route with named generation inputs.
The base image already provides the route. This build-time patch only exposes
the parameters supported by its installed GenerationParams/GenerationConfig
dataclasses, so the separate Community UI never has to depend on Gradio's
positional component order.
"""
from pathlib import Path
import sys
target = Path(sys.argv[1])
source = target.read_text(encoding="utf-8")
def replace_once(old: str, new: str, label: str) -> None:
global source
count = source.count(old)
if count != 1:
raise RuntimeError(f"{label}: expected one anchor, found {count}")
source = source.replace(old, new, 1)
old_params = ''' # Build generation params with alias support
params = GenerationParams(
task_type=get_param("task_type", default="text2music"),
caption=caption,
lyrics=lyrics,
bpm=sample_bpm or get_param("bpm"),
keyscale=sample_keyscale or get_param("key_scale", "keyscale", "key", default=""),
timesignature=sample_timesignature or get_param("time_signature", "timesignature", default=""),
duration=sample_duration or get_param("audio_duration", "duration", default=-1),
vocal_language=sample_language,
inference_steps=get_param("inference_steps", default=8),
guidance_scale=float(get_param("guidance_scale", default=7.0) or 7.0),
seed=int(get_param("seed", default=-1) or -1),
thinking=to_bool(get_param("thinking"), False),
lm_temperature=lm_temperature,
lm_cfg_scale=float(get_param("lm_cfg_scale", default=2.0) or 2.0),
lm_negative_prompt=get_param("lm_negative_prompt", default="NO USER INPUT") or "NO USER INPUT",
repaint_latent_crossfade_frames=int(
get_param("repaint_latent_crossfade_frames", default=10) or 10,
),
repaint_wav_crossfade_sec=float(
get_param("repaint_wav_crossfade_sec", default=0.0) or 0.0,
),
repaint_mode=get_param("repaint_mode", default="balanced") or "balanced",
repaint_strength=float(
get_param("repaint_strength", default=0.5) or 0.5,
),
)
'''
new_params = ''' # Build generation params with alias support. Keep this
# mapping explicit: every public API field below is named and independent
# from the order of components in the Gradio interface.
raw_bpm = sample_bpm or get_param("bpm")
params = GenerationParams(
task_type=get_param("task_type", default="text2music") or "text2music",
instruction=get_param("instruction", default="Fill the audio semantic mask based on the given conditions:") or "Fill the audio semantic mask based on the given conditions:",
reference_audio=get_param("reference_audio_path", "reference_audio"),
src_audio=get_param("src_audio_path", "src_audio", "source_audio"),
audio_codes=get_param("audio_codes", default="") or "",
caption=caption,
lyrics=lyrics,
instrumental=to_bool(get_param("instrumental"), False),
bpm=int(float(raw_bpm)) if raw_bpm not in (None, "", 0, "0") else None,
keyscale=sample_keyscale or get_param("key_scale", "keyscale", "key", default=""),
timesignature=sample_timesignature or get_param("time_signature", "timesignature", default=""),
duration=float(sample_duration or get_param("audio_duration", "duration", default=-1) or -1),
vocal_language=sample_language,
inference_steps=int(get_param("inference_steps", default=50) or 50),
guidance_scale=float(get_param("guidance_scale", default=7.0) or 7.0),
seed=int(get_param("seed", default=-1) or -1),
use_adg=to_bool(get_param("use_adg"), False),
cfg_interval_start=float(get_param("cfg_interval_start", default=0.0) or 0.0),
cfg_interval_end=float(get_param("cfg_interval_end", default=1.0) or 1.0),
shift=float(get_param("shift", default=1.0) or 1.0),
infer_method=get_param("infer_method", default="ode") or "ode",
sampler_mode=get_param("sampler_mode", default="euler") or "euler",
repainting_start=float(get_param("repainting_start", default=0.0) or 0.0),
repainting_end=float(get_param("repainting_end", default=-1.0) or -1.0),
chunk_mask_mode=get_param("chunk_mask_mode", default="auto") or "auto",
audio_cover_strength=float(get_param("audio_cover_strength", default=1.0) or 1.0),
cover_noise_strength=float(get_param("cover_noise_strength", default=0.0) or 0.0),
thinking=to_bool(get_param("thinking"), True),
lm_temperature=lm_temperature,
lm_cfg_scale=float(get_param("lm_cfg_scale", default=2.0) or 2.0),
lm_top_k=int(get_param("lm_top_k", default=0) or 0),
lm_top_p=float(get_param("lm_top_p", default=0.9) or 0.9),
lm_negative_prompt=get_param("lm_negative_prompt", default="NO USER INPUT") or "NO USER INPUT",
use_cot_metas=to_bool(get_param("use_cot_metas"), True),
use_cot_caption=to_bool(get_param("use_cot_caption"), True),
use_cot_lyrics=to_bool(get_param("use_cot_lyrics"), False),
use_cot_language=to_bool(get_param("use_cot_language"), True),
use_constrained_decoding=to_bool(get_param("use_constrained_decoding"), True),
enable_normalization=to_bool(get_param("enable_normalization"), True),
normalization_db=float(get_param("normalization_db", default=-1.0) or -1.0),
fade_in_duration=float(get_param("fade_in_duration", default=0.0) or 0.0),
fade_out_duration=float(get_param("fade_out_duration", default=0.0) or 0.0),
latent_shift=float(get_param("latent_shift", default=0.0) or 0.0),
latent_rescale=float(get_param("latent_rescale", default=1.0) or 1.0),
repaint_latent_crossfade_frames=int(get_param("repaint_latent_crossfade_frames", default=10) or 10),
repaint_wav_crossfade_sec=float(get_param("repaint_wav_crossfade_sec", default=0.0) or 0.0),
repaint_mode=get_param("repaint_mode", default="balanced") or "balanced",
repaint_strength=float(get_param("repaint_strength", default=0.5) or 0.5),
)
'''
replace_once(old_params, new_params, "GenerationParams mapping")
old_config = ''' config = GenerationConfig(
batch_size=get_param("batch_size", default=2),
use_random_seed=use_random_seed,
seeds=resolved_seeds,
audio_format=get_param("audio_format", default="flac"),
mp3_bitrate=get_param("mp3_bitrate", default="128k"),
mp3_sample_rate=get_param("mp3_sample_rate", default=48000),
)
'''
new_config = ''' config = GenerationConfig(
batch_size=int(get_param("batch_size", default=1) or 1),
allow_lm_batch=to_bool(get_param("allow_lm_batch"), True),
use_random_seed=to_bool(use_random_seed, True),
seeds=resolved_seeds,
lm_batch_chunk_size=int(get_param("lm_batch_chunk_size", default=8) or 8),
constrained_decoding_debug=to_bool(get_param("constrained_decoding_debug"), False),
audio_format=get_param("audio_format", default="flac") or "flac",
mp3_bitrate=get_param("mp3_bitrate", default="320k") or "320k",
mp3_sample_rate=int(get_param("mp3_sample_rate", default=48000) or 48000),
)
'''
replace_once(old_config, new_config, "GenerationConfig mapping")
target.write_text(source, encoding="utf-8")
+26
View File
@@ -0,0 +1,26 @@
# syntax=docker/dockerfile:1
FROM python:3.12-trixie
ARG APPLIO_COMMIT=7fa68ec2166ab1331c539704159fa14901e94e5a
ENV PATH=/app/.venv/bin:$PATH \
HF_HOME=/models/huggingface \
PIP_DISABLE_PIP_VERSION_CHECK=1
RUN apt-get update && apt-get install -y --no-install-recommends \
ca-certificates curl ffmpeg git libportaudio2 \
&& rm -rf /var/lib/apt/lists/*
WORKDIR /app
RUN git clone https://github.com/IAHispano/Applio.git . \
&& git checkout "$APPLIO_COMMIT" \
&& python3 -m venv /app/.venv \
&& pip install --no-cache-dir --upgrade pip \
&& pip install --no-cache-dir python-ffmpeg \
&& pip install --no-cache-dir torch==2.7.1 torchvision==0.22.1 torchaudio==2.7.1 \
--index-url https://download.pytorch.org/whl/cu128 \
&& sed -i '/^torch==/d;/^torchvision==/d;/^torchaudio==/d' requirements.txt \
&& pip install --no-cache-dir -r requirements.txt \
&& pip install --no-cache-dir "websockets>=13.0"
EXPOSE 6969
CMD ["python3", "app.py", "--server-name", "0.0.0.0", "--port", "6969"]
+14
View File
@@ -0,0 +1,14 @@
# Applio / RVC Studio
Reproduzierbarer, experimenteller Applio-Worker mit der offiziellen
Weboberfläche. Der Build ist auf Upstream-Commit
`7fa68ec2166ab1331c539704159fa14901e94e5a` festgeschrieben.
- Dashboard-Modus: `Applio / RVC`
- WireGuard-URL: `http://192.168.1.212:8011/`
- GPU: RTX 5080, exklusiv zu LLM, Musik- und anderen Voice-Modi
- Persistenz: Basisgewichte, importierte/trainierte Modelle, Konfiguration,
Logs und Hugging-Face-Cache unter `/data/voice/applio`
Applio stellt die RVC-Werkzeuge und deren Oberfläche bereit. Eine konkrete
Zielstimme wird anschließend in der Oberfläche importiert oder trainiert.
+41
View File
@@ -0,0 +1,41 @@
services:
applio-studio:
build: .
image: mike-ai/applio-studio:7fa68ec
container_name: mike-ai-applio-studio
restart: "no"
labels:
com.mike-ai.applio-worker: applio
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
HF_HOME: /models/huggingface
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
ports:
- "127.0.0.1:8011:6969"
volumes:
- /data/voice/applio/huggingface:/models/huggingface
- /data/voice/applio/logs:/app/logs
- /data/voice/applio/models:/app/rvc/models
- /data/voice/applio/config.json:/app/assets/config.json
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:6969/ >/dev/null"]
interval: 5s
timeout: 3s
start_period: 900s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
frontend:
aliases: [applio-studio]
networks:
frontend:
name: mike-ai_frontend
external: true
@@ -0,0 +1,30 @@
FROM pytorch/pytorch:2.7.1-cuda12.8-cudnn9-runtime@sha256:c16f4c749e2d9e96878875cdf6cc45cddda1d1a36fddd371dd6f2360f1b6e2a2
RUN apt-get update \
&& DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends build-essential curl ffmpeg libsndfile1 \
&& rm -rf /var/lib/apt/lists/*
RUN python -m pip install --no-cache-dir \
"audio-separator[gpu]==0.47.0" \
"onnxruntime-gpu==1.22.0" \
"fastapi==0.116.1" \
"python-multipart==0.0.20" \
"uvicorn[standard]==0.35.0"
# audio-separator 0.47 requires NumPy 2 while ClearVoice 0.1.2 still pins
# NumPy 1.x. Keep ClearVoice in a small overlay venv but share the image's
# CUDA-enabled PyTorch installation instead of duplicating it.
RUN python -m venv --system-site-packages /opt/clearvoice-venv \
&& /opt/clearvoice-venv/bin/python -m pip install --no-cache-dir \
"clearvoice==0.1.2" \
"numpy>=1.24.3,<2.0"
WORKDIR /app
COPY app.py index.html speech_enhance.py ./
ENV MODEL_FILENAME=model_bs_roformer_ep_317_sdr_12.9755.ckpt \
MODEL_DIR=/models \
JOB_DIR=/data/jobs
EXPOSE 8080
CMD ["sh", "-c", "mkdir -p \"$MODEL_DIR/clearvoice\" && ln -sfn \"$MODEL_DIR/clearvoice\" /app/checkpoints && for model in \"$MODEL_FILENAME\" htdemucs_ft.yaml htdemucs_6s.yaml; do audio-separator --model_filename \"$model\" --model_file_dir \"$MODEL_DIR\" --download_model_only || exit 1; done; /opt/clearvoice-venv/bin/python /app/speech_enhance.py --download-only || exit 1; exec uvicorn app:app --host 0.0.0.0 --port 8080 --workers 1"]
@@ -0,0 +1,35 @@
# Athena Stem Separator
Exklusiver dritter Athena-Betriebsmodus zum gezielten Herauslösen einer Quelle.
Der Download enthält immer die Zielspur und eine zweite Spur mit dem kompletten
Rest ohne dieses Ziel.
- Engine: `audio-separator` 0.47.0 (MIT)
- **Gesang / Instrumental:** BS-RoFormer Viperx 1297,
`model_bs_roformer_ep_317_sdr_12.9755.ckpt`; Vocal SDR 12,9,
Instrumental SDR 17,0.
- **Schlagzeug oder Bass:** `htdemucs_ft.yaml`; die nicht gewählten Stems werden
zu einer gemeinsamen Restspur summiert.
- **Gitarre oder Piano (experimentell):** `htdemucs_6s.yaml`; auch hier werden
alle übrigen Stems wieder zur Restspur zusammengesetzt. Die Instrumentqualität
liegt unter der spezialisierten Gesangstrennung.
- **Sonstiges:** der `other`-Stem von `htdemucs_6s.yaml`. Er bündelt unter anderem
Synthesizer, Streicher, Bläser und Effekte und ist keine reine Synthesizer-Spur.
- **Sprache / Hintergrund:** ClearVoice `MossFormer2_SE_48K` (Apache-2.0)
verbessert Sprache bei 48 kHz. Die zweite Spur ist das vom Originalsignal
abgezogene Sprachsignal und enthält den verbleibenden Hintergrund. Stereo wird
kanalweise verarbeitet und anschließend wieder zusammengesetzt.
- GPU: RTX 5080; LLM, Bildmodelle, TTS und ACE-Step sind dabei verriegelt.
- Privat erreichbar: `http://192.168.1.212:8007/`
Die Modelle werden beim ersten Start nach `/data/models/audio-separator`
heruntergeladen. Temporäre Jobs liegen unter `/data/audio/separation` und
werden nach dem ZIP-Download entfernt. Eigene Spuren für E-/Akustikgitarre,
Synthesizer und Streicher sind bewusst noch nicht angeboten: Dafür braucht es
weitere Zielmodelle. Die Oberfläche bietet stattdessen den ehrlich benannten,
gemischten `other`-Stem als **Sonstiges** an.
Die API erwartet `multipart/form-data` mit `file` und optional `target`:
`vocals` (Standard), `drums`, `bass`, `guitar`, `piano`, `other` oder `speech`. Das ältere Feld
`mode` mit `vocals`, `four_stem` oder `six_stem` bleibt für vorhandene Clients
erhalten und liefert weiterhin alle Modell-Stems.
@@ -0,0 +1,215 @@
from __future__ import annotations
import asyncio
import os
import shutil
import subprocess
import tempfile
import time
import zipfile
from pathlib import Path
from fastapi import FastAPI, File, Form, HTTPException, UploadFile
from fastapi.responses import FileResponse, HTMLResponse
from starlette.background import BackgroundTask
MODEL = os.getenv("MODEL_FILENAME", "model_bs_roformer_ep_317_sdr_12.9755.ckpt")
MODEL_DIR = Path(os.getenv("MODEL_DIR", "/models"))
JOB_DIR = Path(os.getenv("JOB_DIR", "/data/jobs"))
MAX_UPLOAD = int(os.getenv("MAX_UPLOAD_BYTES", str(1024 ** 3)))
ALLOWED = {".wav", ".flac", ".mp3", ".m4a", ".aac", ".ogg", ".opus", ".wma"}
SEPARATION_LOCK = asyncio.Lock()
STARTED = time.time()
MODES = {
"vocals": {"model": MODEL, "stems": ("vocals", "instrumental"), "archive": "athena-vocals-instrumental.zip", "engine": "mdxc"},
"four_stem": {"model": "htdemucs_ft.yaml", "stems": ("vocals", "drums", "bass", "other"), "archive": "athena-4-stems.zip", "engine": "demucs"},
"six_stem": {"model": "htdemucs_6s.yaml", "stems": ("vocals", "drums", "bass", "guitar", "piano", "other"), "archive": "athena-6-stems-experimental.zip", "engine": "demucs"},
"speech": {"model": "MossFormer2_SE_48K", "stems": ("speech", "noise"), "archive": "athena-sprache-und-hintergrund.zip", "engine": "clearvoice"},
}
TARGETS = {
"vocals": {"mode": "vocals", "stem": "vocals", "remainder": "instrumental", "archive": "athena-gesang-und-rest.zip", "rest_file": "instrumental.flac"},
"drums": {"mode": "four_stem", "stem": "drums", "archive": "athena-schlagzeug-und-rest.zip", "rest_file": "rest-ohne-schlagzeug.flac"},
"bass": {"mode": "four_stem", "stem": "bass", "archive": "athena-bass-und-rest.zip", "rest_file": "rest-ohne-bass.flac"},
"guitar": {"mode": "six_stem", "stem": "guitar", "archive": "athena-gitarre-und-rest.zip", "rest_file": "rest-ohne-gitarre.flac"},
"piano": {"mode": "six_stem", "stem": "piano", "archive": "athena-piano-und-rest.zip", "rest_file": "rest-ohne-piano.flac"},
"other": {"mode": "six_stem", "stem": "other", "archive": "athena-sonstiges-und-rest.zip", "rest_file": "rest-ohne-sonstiges.flac"},
"speech": {"mode": "speech", "stem": "speech", "remainder": "noise", "archive": "athena-sprache-und-hintergrund.zip", "rest_file": "hintergrund-ohne-sprache.flac"},
}
app = FastAPI(title="Athena Stem Separator", version="2.0")
@app.get("/", response_class=HTMLResponse)
def index() -> str:
return Path("/app/index.html").read_text(encoding="utf-8")
@app.get("/health")
def health() -> dict:
available = {
name: (
(MODEL_DIR / "clearvoice" / mode["model"] / "last_best_checkpoint").exists()
if mode["engine"] == "clearvoice"
else (MODEL_DIR / mode["model"]).exists()
)
for name, mode in MODES.items()
}
return {
"status": "ok" if all(available.values()) else "starting",
"models": {name: mode["model"] for name, mode in MODES.items()},
"models_ready": available,
"targets": list(TARGETS),
"busy": SEPARATION_LOCK.locked(),
"uptime_seconds": round(time.time() - STARTED, 1),
}
def _cleanup(path: Path) -> None:
shutil.rmtree(path, ignore_errors=True)
def _run_separator(input_path: Path, output_dir: Path, mode: dict) -> None:
if mode["engine"] == "clearvoice":
completed = subprocess.run(
[
"/opt/clearvoice-venv/bin/python", "/app/speech_enhance.py", str(input_path),
str(output_dir / "speech.flac"), str(output_dir / "noise.flac"),
],
capture_output=True,
text=True,
timeout=7200,
)
if completed.returncode:
detail = (completed.stderr or completed.stdout or "unknown ClearVoice error")[-4000:]
raise RuntimeError(detail)
return
args = [
"audio-separator", str(input_path),
"--model_filename", mode["model"],
"--model_file_dir", str(MODEL_DIR),
"--output_dir", str(output_dir),
"--output_format", "FLAC",
"--sample_rate", "44100",
"--use_autocast",
]
if mode["engine"] == "mdxc":
args.extend(["--mdxc_segment_size", "256", "--mdxc_overlap", "8", "--mdxc_batch_size", "1"])
else:
args.extend(["--demucs_segment_size", "40", "--demucs_shifts", "2", "--demucs_overlap", "0.25"])
completed = subprocess.run(args, capture_output=True, text=True, timeout=7200)
if completed.returncode:
detail = (completed.stderr or completed.stdout or "unknown error")[-4000:]
raise RuntimeError(detail)
def _stem_name(path: Path, expected: tuple[str, ...]) -> str | None:
lower = path.stem.lower()
for stem in sorted(expected, key=len, reverse=True):
if stem in lower:
return stem
return None
def _mix_remainder(stems: list[Path], output_path: Path) -> None:
args = ["ffmpeg", "-hide_banner", "-loglevel", "error", "-y"]
for stem in stems:
args.extend(["-i", str(stem)])
inputs = "".join(f"[{index}:a]" for index in range(len(stems)))
args.extend([
"-filter_complex", f"{inputs}amix=inputs={len(stems)}:normalize=0:dropout_transition=0[rest]",
"-map", "[rest]", "-ar", "44100", "-c:a", "flac", str(output_path),
])
completed = subprocess.run(args, capture_output=True, text=True, timeout=1800)
if completed.returncode:
detail = (completed.stderr or completed.stdout or "unknown ffmpeg error")[-4000:]
raise RuntimeError(f"Restspur konnte nicht erzeugt werden: {detail}")
@app.post("/v1/separate")
async def separate(
file: UploadFile = File(...),
target: str | None = Form(None),
mode: str | None = Form(None),
) -> FileResponse:
selected_target = TARGETS.get(target) if target else None
if target and selected_target is None:
raise HTTPException(422, f"Unbekannte Zielspur: {target}")
selected_mode_name = selected_target["mode"] if selected_target else (mode or "vocals")
selected_mode = MODES.get(selected_mode_name)
if selected_mode is None:
raise HTTPException(422, f"Unbekannter Trennmodus: {selected_mode_name}")
suffix = Path(file.filename or "upload.wav").suffix.lower()
if suffix not in ALLOWED:
raise HTTPException(415, "Dieses Audioformat wird nicht unterstützt.")
if SEPARATION_LOCK.locked():
raise HTTPException(409, "Eine Trennung läuft bereits.")
job = Path(tempfile.mkdtemp(prefix="separate-", dir=JOB_DIR))
input_path = job / f"input{suffix}"
output_dir = job / "output"
output_dir.mkdir()
size = 0
try:
with input_path.open("wb") as handle:
while chunk := await file.read(1024 * 1024):
size += len(chunk)
if size > MAX_UPLOAD:
raise HTTPException(413, "Datei ist größer als 1 GiB.")
handle.write(chunk)
async with SEPARATION_LOCK:
await asyncio.to_thread(_run_separator, input_path, output_dir, selected_mode)
stems = sorted(output_dir.glob("*.flac"))
expected = selected_mode["stems"]
recognized = {_stem_name(stem, expected): stem for stem in stems}
recognized.pop(None, None)
missing = [stem for stem in expected if stem not in recognized]
if missing:
found = ", ".join(stem.name for stem in stems) or "keine"
raise RuntimeError(f"Fehlende Spuren: {', '.join(missing)}; gefunden: {found}")
archive_name = selected_target["archive"] if selected_target else selected_mode["archive"]
archive = job / archive_name
with zipfile.ZipFile(archive, "w", compression=zipfile.ZIP_STORED) as bundle:
if selected_target:
target_stem = selected_target["stem"]
bundle.write(recognized[target_stem], f"{target_stem}.flac")
if "remainder" in selected_target:
remainder = recognized[selected_target["remainder"]]
else:
remainder = job / selected_target["rest_file"]
await asyncio.to_thread(
_mix_remainder,
[recognized[stem] for stem in expected if stem != target_stem],
remainder,
)
bundle.write(remainder, selected_target["rest_file"])
else:
# Rückwärtskompatibilität für bestehende API-Clients.
for stem in expected:
bundle.write(recognized[stem], f"{stem}.flac")
return FileResponse(
archive,
media_type="application/zip",
filename=archive_name,
background=BackgroundTask(_cleanup, job),
)
except HTTPException:
_cleanup(job)
raise
except subprocess.TimeoutExpired:
_cleanup(job)
raise HTTPException(504, "Die Trennung hat das Zeitlimit überschritten.")
except Exception as exc:
_cleanup(job)
raise HTTPException(500, f"Trennung fehlgeschlagen: {exc}")
@app.on_event("startup")
def prepare() -> None:
JOB_DIR.mkdir(parents=True, exist_ok=True)
MODEL_DIR.mkdir(parents=True, exist_ok=True)
for old in JOB_DIR.glob("separate-*"):
if old.is_dir() and time.time() - old.stat().st_mtime > 86400:
_cleanup(old)
@@ -0,0 +1,38 @@
services:
stem-separator:
build: .
image: mike-ai/bs-roformer-separator:0.47.0
container_name: mike-ai-stem-separator
labels:
com.mike-ai.stem-separator: "bs-roformer"
environment:
NVIDIA_VISIBLE_DEVICES: ${SEPARATOR_GPU_UUID:?set SEPARATOR_GPU_UUID to the RTX 5080 UUID}
MODEL_FILENAME: model_bs_roformer_ep_317_sdr_12.9755.ckpt
MODEL_DIR: /models
JOB_DIR: /data/jobs
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${SEPARATOR_GPU_UUID:?set SEPARATOR_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
ports:
- "127.0.0.1:${SEPARATOR_PORT:-8007}:8080"
volumes:
- ${SEPARATOR_MODEL_DIR:-/data/models/audio-separator}:/models
- ${SEPARATOR_DATA_DIR:-/data/audio/separation}:/data
shm_size: "2gb"
restart: "no"
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8080/health | grep -q '\"status\":\"ok\"'"]
interval: 15s
timeout: 5s
start_period: 600s
retries: 3
networks: [frontend]
networks:
frontend:
external: true
name: mike-ai_frontend
@@ -0,0 +1,19 @@
<!doctype html>
<html lang="de"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
<title>Athena · Spuren herauslösen</title><style>
:root{color-scheme:dark;--bg:#07111c;--card:#101d2b;--line:#26384b;--cyan:#48d7f5;--mint:#63e6be;--text:#ecf5ff;--muted:#91a4b7}*{box-sizing:border-box}body{margin:0;background:radial-gradient(circle at 20% 0,#142a42 0,#07111c 42%);font:16px system-ui,sans-serif;color:var(--text);min-height:100vh;display:grid;place-items:center;padding:24px}.card{width:min(880px,100%);padding:32px;border:1px solid var(--line);border-radius:22px;background:rgba(16,29,43,.96);box-shadow:0 25px 70px #0008}.eyebrow{color:var(--cyan);font-weight:800;letter-spacing:.14em;text-transform:uppercase;font-size:12px}h1{font-size:clamp(30px,5vw,52px);margin:.3em 0 .15em}p{color:var(--muted);line-height:1.6}.targets{display:grid;grid-template-columns:repeat(3,1fr);gap:9px;margin:24px 0}.target{display:block;border:1px solid var(--line);border-radius:14px;padding:14px 10px;text-align:center;cursor:pointer}.target:has(input:checked){border-color:var(--cyan);background:#48d7f510;box-shadow:0 0 0 1px #48d7f528}.target input{display:none}.target b,.target span{display:block}.target span{color:var(--muted);font-size:12px;margin-top:5px;line-height:1.35}.section{grid-column:1/-1;color:var(--cyan);font-size:12px;font-weight:800;letter-spacing:.12em;text-transform:uppercase;margin-top:8px}.drop{display:block;margin:20px 0;padding:36px 24px;border:2px dashed #3f5a72;border-radius:18px;text-align:center;cursor:pointer;transition:.2s}.drop:hover,.drop.drag{border-color:var(--cyan);background:#48d7f50b}.drop input{display:none}.file{color:var(--mint);font-weight:700;margin-top:8px}button{width:100%;border:0;border-radius:13px;padding:15px;font-weight:800;font-size:16px;background:linear-gradient(90deg,var(--cyan),var(--mint));color:#05202a;cursor:pointer}button:disabled{opacity:.45;cursor:not-allowed}.status{min-height:28px;margin-top:18px;color:var(--muted)}.bar{height:7px;background:#07111c;border-radius:9px;overflow:hidden;margin-top:12px}.fill{height:100%;width:0;background:linear-gradient(90deg,var(--cyan),var(--mint));transition:.4s}.run .fill{width:85%;animation:pulse 1.5s infinite alternate}@keyframes pulse{to{opacity:.45}}small{display:block;color:#71879a;margin-top:20px}@media(max-width:760px){.targets{grid-template-columns:repeat(2,1fr)}}
</style></head><body><main class="card"><div class="eyebrow">Athena Audio Lab</div><h1>Was möchtest du herauslösen?</h1><p>Der Download enthält immer die gewählte Spur separat und zusätzlich den vollständigen Rest ohne diese Spur.</p>
<div class="targets">
<div class="section">Musik</div>
<label class="target"><input type="radio" name="target" value="vocals" checked><b>Gesang</b><span>BS‑RoFormer<br>beste Qualität</span></label>
<label class="target"><input type="radio" name="target" value="drums"><b>Schlagzeug</b><span>HTDemucs FT</span></label>
<label class="target"><input type="radio" name="target" value="bass"><b>Bass</b><span>HTDemucs FT</span></label>
<label class="target"><input type="radio" name="target" value="guitar"><b>Gitarre</b><span>HTDemucs 6s<br>experimentell</span></label>
<label class="target"><input type="radio" name="target" value="piano"><b>Piano</b><span>HTDemucs 6s<br>experimentell</span></label>
<label class="target"><input type="radio" name="target" value="other"><b>Sonstiges</b><span>Synths, Streicher etc.<br>gemischte Spur</span></label>
<div class="section">Sprache und Geräusche</div>
<label class="target"><input type="radio" name="target" value="speech"><b>Sprache reinigen</b><span>MossFormer2 · 48 kHz<br>Sprache + Hintergrund</span></label>
</div>
<label class="drop" id="drop">Audio auswählen oder hier ablegen<input id="file" type="file" accept="audio/*"><div class="file" id="name">Noch keine Datei gewählt</div></label><button id="start" disabled>Ausgewählte Spur und Rest erzeugen</button><div class="status" id="status">Bereit.</div><div class="bar" id="bar"><div class="fill"></div></div><small>Alles läuft lokal auf Athena. Synthesizer, Streicher sowie elektrische und akustische Gitarre separat benötigen zusätzliche Spezialmodelle.</small></main><script>
const file=document.querySelector('#file'),drop=document.querySelector('#drop'),name=document.querySelector('#name'),start=document.querySelector('#start'),status=document.querySelector('#status'),bar=document.querySelector('#bar');let selected;const names={vocals:'gesang',drums:'schlagzeug',bass:'bass',guitar:'gitarre',piano:'piano',other:'sonstiges',speech:'sprache-und-hintergrund'};function choose(f){selected=f;name.textContent=f?`${f.name} · ${(f.size/1048576).toFixed(1)} MiB`:'Noch keine Datei gewählt';start.disabled=!f}file.onchange=()=>choose(file.files[0]);drop.ondragover=e=>{e.preventDefault();drop.classList.add('drag')};drop.ondragleave=()=>drop.classList.remove('drag');drop.ondrop=e=>{e.preventDefault();drop.classList.remove('drag');choose(e.dataTransfer.files[0])};start.onclick=async()=>{const target=document.querySelector('input[name=target]:checked').value;start.disabled=true;bar.classList.add('run');status.textContent=target==='speech'?'MossFormer2 trennt Sprache und Hintergrund – das kann einige Minuten dauern …':'Modell löst die gewählte Spur heraus – das kann einige Minuten dauern …';let body=new FormData();body.append('file',selected);body.append('target',target);try{let r=await fetch('/v1/separate',{method:'POST',body});if(!r.ok)throw Error((await r.json()).detail||`HTTP ${r.status}`);let blob=await r.blob(),a=document.createElement('a');a.href=URL.createObjectURL(blob);a.download=`athena-${names[target]}-und-rest.zip`;a.click();setTimeout(()=>URL.revokeObjectURL(a.href),5000);status.textContent='Fertig – ZIP mit der ausgewählten Spur und dem Rest wurde geladen.'}catch(e){status.textContent=`Fehler: ${e.message}`}finally{bar.classList.remove('run');start.disabled=false}};
</script></body></html>
@@ -0,0 +1,81 @@
from __future__ import annotations
import argparse
import subprocess
import tempfile
from pathlib import Path
import numpy as np
import soundfile as sf
from clearvoice import ClearVoice
MODEL = "MossFormer2_SE_48K"
SAMPLE_RATE = 48_000
def convert_input(source: Path, target: Path) -> None:
completed = subprocess.run(
[
"ffmpeg", "-hide_banner", "-loglevel", "error", "-y",
"-i", str(source), "-vn", "-ar", str(SAMPLE_RATE),
"-c:a", "pcm_f32le", str(target),
],
capture_output=True,
text=True,
timeout=1800,
)
if completed.returncode:
raise RuntimeError(completed.stderr[-4000:] or "ffmpeg input conversion failed")
def enhance(source: Path, speech_path: Path, noise_path: Path) -> None:
with tempfile.TemporaryDirectory(prefix="clearvoice-") as temp_dir:
converted = Path(temp_dir) / "input-48k.wav"
convert_input(source, converted)
audio, sample_rate = sf.read(converted, dtype="float32", always_2d=True)
if sample_rate != SAMPLE_RATE:
raise RuntimeError(f"unexpected sample rate: {sample_rate}")
model = ClearVoice(task="speech_enhancement", model_names=[MODEL])
# Use ClearVoice's file-I/O path so recordings longer than its 20-second
# one-pass window are segmented correctly. Run each channel separately
# because the enhancement network itself is mono, then restore stereo.
channels = []
for channel_index in range(audio.shape[1]):
channel_path = Path(temp_dir) / f"channel-{channel_index}.wav"
sf.write(channel_path, audio[:, channel_index], SAMPLE_RATE, subtype="FLOAT")
result = np.asarray(model(str(channel_path), False), dtype=np.float32).squeeze()
if result.ndim != 1:
raise RuntimeError(f"unexpected ClearVoice output shape: {result.shape}")
channels.append(result)
enhanced = np.column_stack(channels)
length = min(len(audio), len(enhanced))
original = audio[:length]
speech = enhanced[:length]
noise = original - speech
# FLAC does not support floating-point samples. PCM_24 retains ample
# headroom and avoids the invalid FLOAT/FLAC combination in libsndfile.
sf.write(speech_path, np.clip(speech, -1.0, 1.0), SAMPLE_RATE, format="FLAC", subtype="PCM_24")
sf.write(noise_path, np.clip(noise, -1.0, 1.0), SAMPLE_RATE, format="FLAC", subtype="PCM_24")
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("input", nargs="?", type=Path)
parser.add_argument("speech", nargs="?", type=Path)
parser.add_argument("noise", nargs="?", type=Path)
parser.add_argument("--download-only", action="store_true")
args = parser.parse_args()
if args.download_only:
ClearVoice(task="speech_enhancement", model_names=[MODEL])
return
if not all((args.input, args.speech, args.noise)):
parser.error("input, speech and noise output paths are required")
enhance(args.input, args.speech, args.noise)
if __name__ == "__main__":
main()
+18
View File
@@ -0,0 +1,18 @@
# GSQ-RCO IQ3_S profile A/B test
`run-case.sh` starts one isolated llama.cpp test container for either the
current Q4 reference or GSQ-RCO IQ3_S. It refuses to start while a production
LLM container is running. Context, GPU split, vision projector, MTP depth and
batch sizes are explicit command-line arguments so every candidate can use the
same settings as its corresponding production profile.
The 2026-09-08 run used the existing dependency-free benchmark programs:
- `experiments/dirk-qwen38/bench-case.py`
- `experiments/dirk-qwen38/quality-ab.py`
- `dev/QWEN38-FINAL-ACCEPTANCE-v1.json`
Results are retained on Athena in
`/data/model-benchmarks/gsq-rco-iq3s-ab-v2-20260908/`. The conclusions and
aggregate measurements are documented in
`docs/GSQ_RCO_BETA1_20260904.md`.
+130
View File
@@ -0,0 +1,130 @@
#!/usr/bin/env bash
set -Eeuo pipefail
MODEL=${1:-}
CONTEXT=${2:-}
SPLIT=${3:-}
VISION=${4:-no}
MTP=${5:-3}
BATCH=${6:-2048}
UBATCH=${7:-128}
case "$MODEL" in
q4-pure)
MODEL_FILE=/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf
;;
q4-mix)
MODEL_FILE=/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
;;
iq3s)
MODEL_FILE=/models/qwen3.8-27b-gsq-rco-iq3s/Qwen3.8-27B-GSQ-RCO-IQ3_S-mtp.gguf
;;
*)
echo "Usage: $0 {q4-pure|q4-mix|iq3s} CONTEXT {none|PERCENT,PERCENT} {yes|no}" >&2
exit 2
;;
esac
[[ $CONTEXT =~ ^[0-9]+$ ]] || { echo "Invalid context" >&2; exit 2; }
[[ $SPLIT == none || $SPLIT =~ ^[0-9]+,[0-9]+$ ]] || { echo "Invalid split" >&2; exit 2; }
[[ $VISION == yes || $VISION == no ]] || { echo "Invalid vision setting" >&2; exit 2; }
[[ $MTP =~ ^[0-9]+$ ]] || { echo "Invalid MTP setting" >&2; exit 2; }
[[ $BATCH =~ ^[0-9]+$ ]] || { echo "Invalid batch setting" >&2; exit 2; }
[[ $UBATCH =~ ^[0-9]+$ ]] || { echo "Invalid ubatch setting" >&2; exit 2; }
NAME=mike-ai-llama-gsq-v2
IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
if docker ps --format '{{.Names}}' | grep -qx "$NAME"; then
echo "A/B container already running" >&2
exit 1
fi
mapfile -t blockers < <(
docker ps --format '{{.Names}}' |
grep -E '^mike-ai-llama-' |
grep -vE '^mike-ai-llama-dashboard$' || true
)
if ((${#blockers[@]})); then
printf 'Production model still running: %s\n' "${blockers[*]}" >&2
exit 1
fi
args=(
--model "$MODEL_FILE"
--alias benchmark
--ctx-size "$CONTEXT"
--flash-attn on
--cache-type-k q4_0
--cache-type-v q4_0
--cache-prompt
--cache-reuse 256
--cache-ram 8192
--threads 6
--threads-batch 6
--batch-size "$BATCH"
--ubatch-size "$UBATCH"
--parallel 1
--kv-unified
--jinja
--reasoning auto
--reasoning-preserve
--host 127.0.0.1
--port 5005
--metrics
--fit off
--n-gpu-layers all
--no-mmap
--no-ui
--temperature 0.2
--top-p 0.8
--top-k 20
--spec-type draft-mtp
--spec-draft-n-max "$MTP"
--spec-draft-type-k f16
--spec-draft-type-v f16
)
env_args=()
if [[ $VISION == yes ]]; then
env_args=(-e MTMD_BACKEND_DEVICE=CUDA1)
args+=(--mmproj /models/qwen/mmproj-BF16.gguf --mmproj-device CUDA1)
fi
if [[ $SPLIT == none ]]; then
args+=(--device CUDA0 --main-gpu 0 --split-mode none)
else
args+=(--device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split "$SPLIT")
fi
docker run -d \
--name "$NAME" \
--gpus all \
--network host \
--read-only \
--tmpfs /tmp:rw,noexec,nosuid,nodev,size=256m \
--security-opt no-new-privileges:true \
--cap-drop ALL \
--pids-limit 1024 \
--log-opt max-size=20m \
--log-opt max-file=2 \
-v /data/models:/models:ro \
"${env_args[@]}" \
--label mike-ai.experiment=gsq-rco-iq3s-ab-v2 \
"$IMAGE" "${args[@]}"
deadline=$((SECONDS + 900))
until curl -fsS --max-time 3 http://127.0.0.1:5005/health >/dev/null; do
if [[ $(docker inspect -f '{{.State.Running}}' "$NAME" 2>/dev/null || true) != true ]]; then
docker logs --tail 100 "$NAME" >&2 || true
exit 1
fi
if ((SECONDS >= deadline)); then
docker logs --tail 100 "$NAME" >&2 || true
exit 1
fi
sleep 2
done
curl -fsS http://127.0.0.1:5005/props
printf '\nReady: %s, context %s, split %s, vision %s\n' "$MODEL" "$CONTEXT" "$SPLIT" "$VISION"
+28
View File
@@ -0,0 +1,28 @@
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
ARG OMNIVOICE_VERSION=0.2.1
ARG OMNIVOICE_TRITON_VERSION=0.1.0
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
ca-certificates curl ffmpeg libsndfile1 python3 python3-pip python3-venv \
&& rm -rf /var/lib/apt/lists/*
RUN python3 -m venv /opt/venv
ENV PATH="/opt/venv/bin:${PATH}"
RUN python -m pip install --no-cache-dir --upgrade pip \
&& python -m pip install --no-cache-dir \
--index-url https://download.pytorch.org/whl/cu128 \
"torch==2.8.0" "torchaudio==2.8.0" \
&& python -m pip install --no-cache-dir \
"omnivoice==${OMNIVOICE_VERSION}" \
"omnivoice-triton==${OMNIVOICE_TRITON_VERSION}" \
"num2words>=0.5.14"
EXPOSE 8008
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
CMD curl -fsS http://127.0.0.1:8008/ >/dev/null || exit 1
CMD ["omnivoice-demo", "--ip", "0.0.0.0", "--port", "8008"]
+18
View File
@@ -0,0 +1,18 @@
# OmniVoice cloning gate on Athena
This is the isolated quality gate for `k2-fsa/OmniVoice` 0.2.1. It exposes
the upstream Gradio demo through Athena's existing private Voice Studio route.
The image also contains `omnivoice-triton` 0.1.0 for a later measured
base-versus-optimized benchmark; the upstream UI deliberately starts in the
unmodified reference mode so kernel changes cannot contaminate the first
listening test.
- Model weights: CC-BY-NC
- Code: Apache-2.0
- Private URL: `http://192.168.1.212:8008`
- Persistent cache: `/data/voice/omnivoice/huggingface`
- GPU allocator workaround: `expandable_segments:True`
Use a clean 3–10 second reference and provide its exact transcript. German
target text should be written out normally; avoid raw abbreviations and digits
in the first quality test.
@@ -0,0 +1,40 @@
services:
voice-studio:
build: .
image: mike-ai/omnivoice-studio:0.2.1
container_name: mike-ai-voice-studio
restart: "no"
labels:
# Kept compatible with the current controller during the A/B gate.
com.mike-ai.voice-worker: vevo2
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
HF_HOME: /models/huggingface
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
ports:
- "127.0.0.1:8008:8008"
volumes:
- /data/voice/omnivoice/huggingface:/models/huggingface
- /data/voice/omnivoice/output:/output
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/ >/dev/null"]
interval: 5s
timeout: 3s
start_period: 600s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
frontend:
aliases: [voice-studio]
networks:
frontend:
name: mike-ai_frontend
external: true
+82
View File
@@ -0,0 +1,82 @@
# Qwen3.8 September 2026 A/B preparation
This directory prepares an isolated comparison of two Qwen3.8-27B IQ4_XS
artifacts without adding router profiles or changing the running Athena stack.
## Candidates
| ID | Artifact | Purpose | Pinned revision | Size |
| --- | --- | --- | --- | ---: |
| `qwopus` | `Jackrong/Qwopus3.8-27B-Flash-GGUF` / `Qwopus3.8-27B-Flash-MTP-IQ4_XS.gguf` | Efficiency fine-tune | `e146d61e88782677805b3b68ad3adf8674dde80d` | 15,420,445,792 B |
| `bartowski` | `bartowski/Qwen3.8-27B-GGUF` / `Qwen3.8-27B-IQ4_XS.gguf` | Standard Qwen, alternative IQ4_XS quant | `f0eec4a4bb4975114a030d048952d83c0a53c034` | 15,567,824,480 B |
The production reference remains
`jpetrina/Qwen3.8-27B-IQ4_XS-pure-GGUF`. The candidates intentionally use the
same IQ4_XS quantization class so that the first comparison does not mix a
fine-tune difference with a quantization-class difference.
## Safety boundary
- Nothing in this directory is called by installation, Compose or the router.
- No production profile is added.
- Downloads happen only after explicitly running `download-candidate.sh`.
- `run-case.sh` refuses to start while a production `mike-ai-llama-*` model is
running. It never stops production itself.
- The test server binds to `127.0.0.1:5005` and is not exposed to the LAN.
- Cleanup is a dry run unless an explicit deletion flag is supplied. It only
addresses the exact test container, result directory and two pinned files.
## Later test sequence
Run these commands on Athena only after the active coding task has finished and
the GPUs have deliberately been released:
```sh
cd /opt/mike-ai/stack/experiments/qwen38-20260907-ab
./inventory.sh
./download-candidate.sh qwopus
./download-candidate.sh bartowski
./run-case.sh qwopus 160000 85,15
./wait-ready.sh
./run-benchmark.sh qwopus-160k
./stop-case.sh
./run-case.sh bartowski 160000 85,15
./wait-ready.sh
./run-benchmark.sh bartowski-160k
./stop-case.sh
```
Only after both candidates pass the 160K quality and tool-call tests should
192K and 262144 be attempted. Promotion into the router is a separate decision
and is deliberately not implemented here.
## Cleanup
Preview everything owned by this experiment:
```sh
./cleanup.sh
```
Remove only the test container and benchmark results:
```sh
./cleanup.sh --results
```
Remove only the two downloaded candidate files and their now-empty directory:
```sh
./cleanup.sh --models
```
Remove both:
```sh
./cleanup.sh --all
```
The script never touches the production Pure, Mix, Beta 1 or uncensored model
directories.
+32
View File
@@ -0,0 +1,32 @@
#!/usr/bin/env bash
candidate_config() {
case "${1:-}" in
qwopus)
CANDIDATE_ID=qwopus
REPOSITORY=Jackrong/Qwopus3.8-27B-Flash-GGUF
REVISION=e146d61e88782677805b3b68ad3adf8674dde80d
MODEL_FILE=Qwopus3.8-27B-Flash-MTP-IQ4_XS.gguf
MODEL_SHA256=88848920fd069ecfe509afd60d4b1f2192327f1edde5e10491a48f7f80c16fb8
MODEL_SIZE=15420445792
MODEL_ALIAS=qwen-qwopus-flash-ab
;;
bartowski)
CANDIDATE_ID=bartowski
REPOSITORY=bartowski/Qwen3.8-27B-GGUF
REVISION=f0eec4a4bb4975114a030d048952d83c0a53c034
MODEL_FILE=Qwen3.8-27B-IQ4_XS.gguf
MODEL_SHA256=c2ae2b018f967370087c196c86d6811b2340ec19138a3752252ade5fbd1f4786
MODEL_SIZE=15567824480
MODEL_ALIAS=qwen-bartowski-iq4-xs-ab
;;
*)
echo "Candidate must be qwopus or bartowski" >&2
return 2
;;
esac
MODEL_ROOT=${MODEL_ROOT:-/data/models/experiments/qwen38-ab-20260907}
MODEL_PATH="$MODEL_ROOT/$CANDIDATE_ID/$MODEL_FILE"
MODEL_URL="https://huggingface.co/$REPOSITORY/resolve/$REVISION/$MODEL_FILE"
}
+42
View File
@@ -0,0 +1,42 @@
#!/usr/bin/env bash
set -Eeuo pipefail
MODE=${1:-preview}
MODEL_ROOT=${MODEL_ROOT:-/data/models/experiments/qwen38-ab-20260907}
RESULT_DIR=${RESULT_DIR:-/data/benchmarks/qwen38-ab-20260907}
NAME=mike-ai-llama-qwen38-ab
QWOPUS="$MODEL_ROOT/qwopus/Qwopus3.8-27B-Flash-MTP-IQ4_XS.gguf"
BARTOWSKI="$MODEL_ROOT/bartowski/Qwen3.8-27B-IQ4_XS.gguf"
case "$MODE" in
preview)
echo "Dry run only. Exact owned targets:"
printf ' container: %s\n results: %s\n model: %s\n model: %s\n' \
"$NAME" "$RESULT_DIR" "$QWOPUS" "$BARTOWSKI"
echo "Use --results, --models or --all to delete these exact targets."
exit 0
;;
--results|--models|--all) ;;
*)
echo "Usage: $0 [--results|--models|--all]" >&2
exit 2
;;
esac
docker rm -f "$NAME" >/dev/null 2>&1 || true
if [[ $MODE == --results || $MODE == --all ]]; then
if [[ -d $RESULT_DIR ]]; then
find "$RESULT_DIR" -maxdepth 1 -type f -name '*.json' -delete
rmdir "$RESULT_DIR" 2>/dev/null || true
fi
fi
if [[ $MODE == --models || $MODE == --all ]]; then
rm -f -- "$QWOPUS" "$QWOPUS.part" "$BARTOWSKI" "$BARTOWSKI.part"
rmdir "$MODEL_ROOT/qwopus" "$MODEL_ROOT/bartowski" 2>/dev/null || true
rmdir "$MODEL_ROOT" 2>/dev/null || true
fi
echo "Cleanup complete for mode $MODE. No production model path was addressed."
+32
View File
@@ -0,0 +1,32 @@
#!/usr/bin/env bash
set -Eeuo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=candidates.sh
source "$SCRIPT_DIR/candidates.sh"
candidate_config "${1:-}"
target_dir="$MODEL_ROOT/$CANDIDATE_ID"
target="$target_dir/$MODEL_FILE"
partial="$target.part"
install -d -m 0755 "$target_dir"
if [[ -s $target ]]; then
printf '%s %s\n' "$MODEL_SHA256" "$target" | sha256sum -c -
echo "Already prepared: $target"
exit 0
fi
echo "Downloading pinned $CANDIDATE_ID artifact ($MODEL_SIZE bytes)"
curl --fail --location --retry 8 --retry-delay 5 --continue-at - \
--output "$partial" "$MODEL_URL"
actual_size=$(stat -c %s "$partial")
if [[ $actual_size != "$MODEL_SIZE" ]]; then
echo "Size mismatch: expected $MODEL_SIZE, got $actual_size; keeping $partial for inspection" >&2
exit 1
fi
printf '%s %s\n' "$MODEL_SHA256" "$partial" | sha256sum -c -
mv -f "$partial" "$target"
echo "Prepared and verified: $target"
+32
View File
@@ -0,0 +1,32 @@
#!/usr/bin/env bash
set -Eeuo pipefail
MODEL_ROOT=${MODEL_ROOT:-/data/models/experiments/qwen38-ab-20260907}
RESULT_DIR=${RESULT_DIR:-/data/benchmarks/qwen38-ab-20260907}
echo "== Containers owned by this experiment =="
docker ps -a --filter label=mike-ai.experiment=qwen38-ab-20260907 \
--format 'table {{.Names}}\t{{.Status}}\t{{.Label "mike-ai.candidate"}}' || true
echo "== Candidate model files and partial downloads =="
if [[ -d $MODEL_ROOT ]]; then
find "$MODEL_ROOT" -maxdepth 2 -type f -exec ls -lh {} +
else
echo "Not present: $MODEL_ROOT"
fi
echo "== Benchmark results =="
if [[ -d $RESULT_DIR ]]; then
find "$RESULT_DIR" -maxdepth 1 -type f -exec ls -lh {} +
else
echo "Not present: $RESULT_DIR"
fi
echo "== Known older experimental model directories (read-only report) =="
for path in \
/data/models/qwen3.8-27b-dirk \
/data/models/qwen3.8-27b-gsq-rco-test \
/data/models/qwen3.8-27b-iq4-mix \
/data/models/qwen3.8-27b-iq4-xs-pure; do
[[ -d $path ]] && du -sh "$path"
done
+24
View File
@@ -0,0 +1,24 @@
#!/usr/bin/env bash
set -Eeuo pipefail
LABEL=${1:-}
CONTEXT=${2:-160000}
[[ -n $LABEL && $LABEL =~ ^[a-zA-Z0-9._-]+$ ]] || {
echo "Usage: $0 SAFE_LABEL [CONTEXT]" >&2
exit 2
}
[[ $CONTEXT =~ ^[0-9]+$ ]] || {
echo "CONTEXT must be an integer" >&2
exit 2
}
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
REFERENCE_DIR="$SCRIPT_DIR/../dirk-qwen38"
RESULT_DIR=${RESULT_DIR:-/data/benchmarks/qwen38-ab-20260907}
TASKS=${TASKS:-/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json}
curl -fsS --max-time 3 http://127.0.0.1:5005/health >/dev/null
python3 "$REFERENCE_DIR/bench-case.py" "$LABEL" "$CONTEXT" \
--base http://127.0.0.1:5005 --output "$RESULT_DIR"
python3 "$REFERENCE_DIR/quality-ab.py" "$LABEL" \
--base http://127.0.0.1:5005 --tasks "$TASKS" --output "$RESULT_DIR"
+103
View File
@@ -0,0 +1,103 @@
#!/usr/bin/env bash
set -Eeuo pipefail
CANDIDATE=${1:-}
CONTEXT=${2:-}
SPLIT=${3:-}
if [[ -z $CANDIDATE || -z $CONTEXT || -z $SPLIT || ! $CONTEXT =~ ^[0-9]+$ || ! $SPLIT =~ ^[0-9]+,[0-9]+$ ]]; then
echo "Usage: $0 {qwopus|bartowski} CONTEXT TENSOR_SPLIT" >&2
exit 2
fi
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
# shellcheck source=candidates.sh
source "$SCRIPT_DIR/candidates.sh"
candidate_config "$CANDIDATE"
HOST_MODEL_FILE=$MODEL_PATH
CONTAINER_MODEL="/models/$CANDIDATE_ID/$MODEL_FILE"
IMAGE=${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
NAME=mike-ai-llama-qwen38-ab
RESULT_DIR=${RESULT_DIR:-/data/benchmarks/qwen38-ab-20260907}
[[ -s $HOST_MODEL_FILE ]] || {
echo "Candidate is not downloaded: $HOST_MODEL_FILE" >&2
exit 1
}
printf '%s %s\n' "$MODEL_SHA256" "$HOST_MODEL_FILE" | sha256sum -c -
mapfile -t blockers < <(
docker ps --format '{{.Names}}' |
grep -E '^mike-ai-llama-' |
grep -vE '^mike-ai-llama-dashboard$|^mike-ai-llama-qwen38-ab$' || true
)
if ((${#blockers[@]})); then
printf 'Refusing to start: production model container(s) still running: %s\n' "${blockers[*]}" >&2
exit 1
fi
if docker ps --format '{{.Names}}' | grep -qx "$NAME"; then
echo "Refusing to replace a running A/B container; run ./stop-case.sh first" >&2
exit 1
fi
install -d -m 0755 "$RESULT_DIR"
args=(
--model "$CONTAINER_MODEL"
--alias "$MODEL_ALIAS"
--ctx-size "$CONTEXT"
--flash-attn on
--cache-type-k q4_0
--cache-type-v q4_0
--cache-prompt
--cache-ram 8192
--threads 6
--threads-batch 6
--batch-size 64
--ubatch-size 32
--parallel 1
--jinja
--reasoning auto
--reasoning-budget 8192
--reasoning-preserve
--host 127.0.0.1
--port 5005
--metrics
--fit off
--n-gpu-layers all
--no-mmap
--temperature 1.0
--top-p 0.95
--top-k 20
--device CUDA0,CUDA1
--main-gpu 0
--split-mode layer
--tensor-split "$SPLIT"
--spec-type draft-mtp
--spec-draft-n-max 3
--spec-draft-type-k f16
--spec-draft-type-v f16
)
label="$CANDIDATE-ctx${CONTEXT}-split${SPLIT//,/-}"
docker run -d \
--name "$NAME" \
--gpus all \
--network host \
--read-only \
--tmpfs /tmp:rw,noexec,nosuid,nodev,size=256m \
--security-opt no-new-privileges:true \
--cap-drop ALL \
--pids-limit 1024 \
--log-opt max-size=20m \
--log-opt max-file=2 \
-v "$MODEL_ROOT":/models:ro \
-v "$RESULT_DIR":/results \
--label mike-ai.experiment=qwen38-ab-20260907 \
--label mike-ai.candidate="$CANDIDATE" \
--label mike-ai.case="$label" \
"$IMAGE" "${args[@]}"
printf 'Started isolated case %s on http://127.0.0.1:5005\n' "$label"
+5
View File
@@ -0,0 +1,5 @@
#!/usr/bin/env bash
set -Eeuo pipefail
docker rm -f mike-ai-llama-qwen38-ab >/dev/null 2>&1 || true
echo "A/B test container removed. Production was not changed."
+19
View File
@@ -0,0 +1,19 @@
#!/usr/bin/env bash
set -Eeuo pipefail
NAME=mike-ai-llama-qwen38-ab
deadline=$((SECONDS + ${READY_TIMEOUT:-900}))
until curl -fsS --max-time 3 http://127.0.0.1:5005/health >/dev/null; do
state=$(docker inspect -f '{{.State.Running}}' "$NAME" 2>/dev/null || true)
if [[ $state != true ]]; then
docker logs --tail 100 "$NAME" >&2 || true
exit 1
fi
if ((SECONDS >= deadline)); then
docker logs --tail 100 "$NAME" >&2 || true
exit 1
fi
sleep 2
done
curl -fsS http://127.0.0.1:5005/props
printf '\nA/B test server is ready.\n'
@@ -0,0 +1,64 @@
FROM mike-ai/bs-roformer-separator:0.47.0
ARG AMPHION_COMMIT=26f6883110181f1dbfe95c70a7c7dbaf4de5f42a
USER root
RUN apt-get update \
&& apt-get install -y --no-install-recommends git espeak-ng \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/open-mmlab/Amphion.git /opt/amphion \
&& cd /opt/amphion \
&& git checkout "${AMPHION_COMMIT}" \
&& rm -rf .git
# Vevo2 is tested on Athena with the CUDA/PyTorch stack inherited from the
# existing audio worker. Do not install Amphion's historical torch 2.0/cu118
# pins: Blackwell requires the newer cu128 runtime already present here.
RUN python -m pip install --no-cache-dir \
accelerate==1.10.1 \
diffusers==0.35.1 \
einops==0.8.1 \
easydict==1.13 \
g2p_en==2.1.0 \
humanfriendly==10.0 \
huggingface-hub==0.34.4 \
hydra-core==1.3.2 \
inflect==7.5.0 \
ipython==9.5.0 \
json5==0.12.1 \
librosa==0.11.0 \
loguru==0.7.3 \
matplotlib==3.10.6 \
munch==4.0.0 \
omegaconf==2.3.0 \
openai-whisper==20250625 \
phonemizer==3.3.0 \
python-multipart==0.0.20 \
praat-parselmouth==0.4.6 \
pypinyin==0.55.0 \
pyworld==0.3.5 \
ruamel.yaml==0.18.15 \
safetensors==0.6.2 \
tabulate==0.9.0 \
tgt==1.5 \
torchcrepe==0.0.24 \
transformers==4.56.1 \
typeguard==4.4.4 \
unidecode==1.4.0 \
vector-quantize-pytorch==1.12.5 \
vocos==0.1.0
WORKDIR /opt/amphion
ENV PYTHONPATH=/opt/amphion \
HF_HOME=/models/huggingface \
PYTHONUNBUFFERED=1
COPY run_fm_test.py /usr/local/bin/run_fm_test.py
COPY app.py /app/app.py
COPY index.html /app/index.html
EXPOSE 8008
HEALTHCHECK --interval=5s --timeout=3s --start-period=90s --retries=3 \
CMD curl -fsS http://127.0.0.1:8008/health || exit 1
ENTRYPOINT ["python", "/app/app.py"]
+30
View File
@@ -0,0 +1,30 @@
# Vevo2 Voice Conversion on Athena
This directory contains Athena's private Vevo2 voice-conversion studio.
- Code: `open-mmlab/Amphion` commit
`26f6883110181f1dbfe95c70a7c7dbaf4de5f42a` (MIT)
- Weights: `RMSnow/Vevo2` (CC BY-NC-ND 4.0)
- Runtime: PyTorch 2.7.1 + CUDA 12.8 inherited from Athena's tested audio
separator image; the historical Amphion torch 2.0/cu118 pins are not used.
- Storage: `/data/voice/vevo2`; removing that directory and the test image
removes all downloaded artifacts.
- Private UI: `http://192.168.1.212:8008` through WireGuard only.
- Output: uncompressed mono WAV, 24 kHz.
The 9 September technical gate converted the official 8.6-second speech sample
through the production HTTP API in 2.342 seconds. A warm service start loaded
the model in 12.216 seconds, and peak CUDA allocation during conversion was
5605.6 MiB. The returned 24-kHz WAV was 412878 bytes and 8.6 seconds long. The
service stores named reference voices, accepts a source clip, and returns a
transient WAV download. Jobs and generated outputs are removed after delivery.
It is deliberately not exposed on the university interface.
The installed Compose project is named `mike-ai-voice`. Rebuild or recreate it
with an explicit `VOICE_GPU_UUID` so Docker keeps the service attached to the
RTX 5080. The profile controller starts and stops the existing container; it
does not rebuild it during a mode switch.
The weights are licensed CC BY-NC-ND 4.0. This deployment is for Mike's
private, non-commercial use only. Do not use or expose it as a public or
commercial voice-cloning service.
+202
View File
@@ -0,0 +1,202 @@
#!/usr/bin/env python3
"""Small local-only Vevo2 voice-conversion studio for Athena."""
from __future__ import annotations
import asyncio
import os
import re
import shutil
import subprocess
import threading
import time
import uuid
from contextlib import asynccontextmanager
from pathlib import Path
import torch
from fastapi import FastAPI, File, Form, HTTPException, UploadFile
from fastapi.responses import FileResponse, HTMLResponse
from starlette.background import BackgroundTask
import models.svc.vevo2.infer_vevo2_fm as vevo
DATA_DIR = Path(os.environ.get("VOICE_DATA_DIR", "/data"))
PROFILE_DIR = DATA_DIR / "profiles"
JOB_DIR = DATA_DIR / "jobs"
INDEX = Path("/app/index.html")
MAX_UPLOAD_BYTES = int(os.environ.get("MAX_UPLOAD_BYTES", str(100 * 1024 * 1024)))
MODEL_LOCK = threading.Lock()
PIPELINE = None
MODEL_LOAD_SECONDS: float | None = None
def safe_name(value: str) -> str:
value = re.sub(r"[^A-Za-z0-9_. -]+", "-", value.strip())
value = re.sub(r"\s+", "-", value).strip("-.")
return value[:64] or "voice"
def load_pipeline() -> None:
global PIPELINE, MODEL_LOAD_SECONDS
if PIPELINE is not None:
return
with MODEL_LOCK:
if PIPELINE is not None:
return
started = time.monotonic()
PIPELINE = vevo.load_inference_pipeline()
vevo.inference_pipeline = PIPELINE
MODEL_LOAD_SECONDS = time.monotonic() - started
def to_wav(source: Path, target: Path) -> None:
completed = subprocess.run(
["ffmpeg", "-hide_banner", "-loglevel", "error", "-y", "-i", str(source),
"-ac", "1", "-ar", "24000", "-c:a", "pcm_s16le", str(target)],
capture_output=True,
text=True,
timeout=180,
check=False,
)
if completed.returncode:
raise ValueError(completed.stderr.strip() or "Audio konnte nicht gelesen werden")
async def save_upload(upload: UploadFile, target: Path) -> None:
size = 0
with target.open("wb") as handle:
while chunk := await upload.read(1024 * 1024):
size += len(chunk)
if size > MAX_UPLOAD_BYTES:
raise HTTPException(413, "Audiodatei ist zu groß")
handle.write(chunk)
def profile_path(name: str) -> Path:
target = PROFILE_DIR / f"{safe_name(name)}.wav"
if not target.is_file():
raise HTTPException(404, "Referenzstimme wurde nicht gefunden")
return target
def run_conversion(source: Path, reference: Path, output: Path, pitch_shift: bool) -> tuple[float, float]:
"""Run the GPU-bound conversion off the API event loop."""
load_pipeline()
with MODEL_LOCK:
torch.cuda.reset_peak_memory_stats()
started = time.monotonic()
vevo.vevo2_fm(str(source), str(reference), str(output), shifted_src=pitch_shift)
elapsed = time.monotonic() - started
peak = torch.cuda.max_memory_allocated() / 1048576
return elapsed, peak
@asynccontextmanager
async def lifespan(_app: FastAPI):
PROFILE_DIR.mkdir(parents=True, exist_ok=True)
JOB_DIR.mkdir(parents=True, exist_ok=True)
load_pipeline()
yield
app = FastAPI(title="Athena Voice Studio", lifespan=lifespan)
@app.get("/", response_class=HTMLResponse)
def index() -> str:
return INDEX.read_text(encoding="utf-8")
@app.get("/health")
def health() -> dict:
return {
"status": "ok" if PIPELINE is not None else "starting",
"model": "RMSnow/Vevo2",
"sample_rate": 24000,
"model_load_seconds": MODEL_LOAD_SECONDS,
}
@app.get("/api/profiles")
def profiles() -> dict:
return {"profiles": [path.stem for path in sorted(PROFILE_DIR.glob("*.wav"))]}
@app.post("/api/profiles")
async def create_profile(
name: str = Form(...),
consent: bool = Form(False),
audio: UploadFile = File(...),
) -> dict:
if not consent:
raise HTTPException(400, "Bestätige, dass du die Stimme verwenden darfst")
clean = safe_name(name)
job = JOB_DIR / f"profile-{uuid.uuid4().hex}"
job.mkdir(parents=True)
raw = job / f"upload-{safe_name(audio.filename or 'reference.audio')}"
try:
await save_upload(audio, raw)
target = PROFILE_DIR / f"{clean}.wav"
temporary = job / "reference.wav"
to_wav(raw, temporary)
os.replace(temporary, target)
return {"status": "ok", "profile": clean}
except HTTPException:
raise
except Exception as exc:
raise HTTPException(400, str(exc)) from exc
finally:
shutil.rmtree(job, ignore_errors=True)
@app.delete("/api/profiles/{name}")
def delete_profile(name: str) -> dict:
target = profile_path(name)
target.unlink()
return {"status": "ok", "profile": target.stem}
@app.post("/api/convert")
async def convert(
source: UploadFile = File(...),
profile: str = Form(...),
pitch_shift: bool = Form(True),
) -> FileResponse:
reference = profile_path(profile)
job_id = uuid.uuid4().hex
job = JOB_DIR / f"convert-{job_id}"
job.mkdir(parents=True)
raw = job / f"upload-{safe_name(source.filename or 'source.audio')}"
source_wav = job / "source.wav"
output = JOB_DIR / f"voice-{job_id}.wav"
try:
await save_upload(source, raw)
to_wav(raw, source_wav)
elapsed, peak = await asyncio.to_thread(
run_conversion, source_wav, reference, output, pitch_shift
)
return FileResponse(
output,
media_type="audio/wav",
filename=f"{safe_name(profile)}-{job_id[:8]}.wav",
headers={
"X-Conversion-Seconds": f"{elapsed:.3f}",
"X-Peak-VRAM-MiB": f"{peak:.1f}",
},
background=BackgroundTask(output.unlink, missing_ok=True),
)
except HTTPException:
raise
except Exception as exc:
output.unlink(missing_ok=True)
raise HTTPException(500, f"Stimmenwandlung fehlgeschlagen: {exc}") from exc
finally:
shutil.rmtree(job, ignore_errors=True)
if __name__ == "__main__":
import uvicorn
uvicorn.run(app, host="0.0.0.0", port=8008)
@@ -0,0 +1,39 @@
services:
voice-studio:
build: .
image: mike-ai/vevo2-voice-studio:0.1
container_name: mike-ai-voice-studio
restart: "no"
labels:
com.mike-ai.voice-worker: vevo2
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
VOICE_DATA_DIR: /data
ports:
- "127.0.0.1:8008:8008"
volumes:
- /data/voice/vevo2/huggingface/hub/Vevo2:/opt/amphion/ckpts/Vevo2
- /data/voice/vevo2/huggingface:/models/huggingface
- /data/voice/vevo2/whisper:/root/.cache/whisper
- /data/voice/studio:/data
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8008/health | grep -q '\"status\":\"ok\"'"]
interval: 5s
timeout: 3s
start_period: 600s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
- frontend
networks:
frontend:
name: mike-ai_frontend
external: true
@@ -0,0 +1,10 @@
<!doctype html>
<html lang="de"><head><meta charset="utf-8"><meta name="viewport" content="width=device-width,initial-scale=1">
<title>Athena Voice Studio</title><style>
:root{color-scheme:dark;--bg:#07101a;--panel:#0d1926;--line:#20374d;--text:#edf6ff;--muted:#91a7bb;--cyan:#45d7ff;--green:#66e3a4;--red:#ff7185}*{box-sizing:border-box}body{margin:0;background:radial-gradient(circle at top left,#12283d,var(--bg) 45%);color:var(--text);font:16px system-ui,sans-serif}main{max-width:960px;margin:auto;padding:32px 18px}h1{margin:.2rem 0}p{color:var(--muted)}.grid{display:grid;grid-template-columns:1fr 1fr;gap:16px}.card{background:var(--panel);border:1px solid var(--line);border-radius:16px;padding:20px}label{display:block;margin:13px 0 6px;color:var(--muted)}input,select,button{width:100%;border:1px solid var(--line);border-radius:9px;padding:11px;background:#07111d;color:var(--text);font:inherit}button{margin-top:15px;background:#10334a;border-color:var(--cyan);cursor:pointer}button:disabled{opacity:.5;cursor:wait}.row{display:flex;align-items:center;gap:9px}.row input{width:auto}.status{min-height:24px;margin-top:12px;color:var(--green)}.error{color:var(--red)}audio{width:100%;margin-top:12px}@media(max-width:700px){.grid{grid-template-columns:1fr}}</style></head>
<body><main><div style="color:var(--cyan);letter-spacing:.14em;text-transform:uppercase;font-size:12px">Mike AI · lokal auf Athena</div><h1>Voice Studio</h1><p>Eine Stimme als Referenz speichern und eine vorhandene Sprach- oder Gesangsaufnahme in diese Stimme übertragen. Ausgabe: unkomprimiertes WAV mit 24 kHz.</p>
<div class="grid"><section class="card"><h2>1 · Referenzstimme</h2><form id="profileForm"><label>Name</label><input name="name" required placeholder="z. B. Meine Stimme"><label>Saubere Referenzaufnahme</label><input name="audio" type="file" accept="audio/*" required><label class="row"><input name="consent" type="checkbox" value="true" required><span>Ich darf diese Stimme verwenden.</span></label><button>Stimme speichern</button></form><div class="status" id="profileStatus"></div></section>
<section class="card"><h2>2 · Stimme übertragen</h2><form id="convertForm"><label>Zielstimme</label><select name="profile" id="profiles" required></select><label>Quellaufnahme</label><input name="source" type="file" accept="audio/*" required><label class="row"><input name="pitch_shift" type="checkbox" value="true" checked><span>Tonlage automatisch an Zielstimme anpassen</span></label><button>WAV erzeugen</button></form><div class="status" id="convertStatus"></div><audio id="player" controls hidden></audio><a id="download" hidden>WAV herunterladen</a></section></div>
<p>Hinweis: Nur privat und mit Zustimmung verwenden. Das Vevo2-Gewicht ist nicht für kommerzielle Nutzung lizenziert.</p></main><script>
const $=id=>document.getElementById(id);async function loadProfiles(){let r=await fetch('/api/profiles'),d=await r.json();$('profiles').innerHTML=d.profiles.length?d.profiles.map(x=>`<option>${x}</option>`).join(''):'<option value="">Noch keine Stimme gespeichert</option>'}function state(id,text,error=false){let e=$(id);e.textContent=text;e.classList.toggle('error',error)}$('profileForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('profileStatus','Referenz wird geprüft und gespeichert …');try{let r=await fetch('/api/profiles',{method:'POST',body:new FormData(e.target)}),d=await r.json();if(!r.ok)throw Error(d.detail||`HTTP ${r.status}`);state('profileStatus',`„${d.profile}“ ist gespeichert.`);await loadProfiles()}catch(x){state('profileStatus',x.message,true)}finally{b.disabled=false}};$('convertForm').onsubmit=async e=>{e.preventDefault();let b=e.submitter;b.disabled=true;state('convertStatus','Modell arbeitet …');$('player').hidden=true;$('download').hidden=true;try{let f=new FormData(e.target);if(!f.has('pitch_shift'))f.append('pitch_shift','false');let r=await fetch('/api/convert',{method:'POST',body:f});if(!r.ok){let d=await r.json();throw Error(d.detail||`HTTP ${r.status}`)}let blob=await r.blob(),url=URL.createObjectURL(blob);$('player').src=url;$('player').hidden=false;$('download').href=url;$('download').download='voice-conversion.wav';$('download').hidden=false;state('convertStatus',`Fertig in ${r.headers.get('X-Conversion-Seconds')||'?'} s · Spitzen-VRAM ${r.headers.get('X-Peak-VRAM-MiB')||'?'} MiB`)}catch(x){state('convertStatus',x.message,true)}finally{b.disabled=false}};loadProfiles();
</script></body></html>
@@ -0,0 +1,47 @@
#!/usr/bin/env python3
"""Minimal reproducible Vevo2 style-preserving voice-conversion smoke test."""
from __future__ import annotations
import argparse
import os
import time
import torch
import models.svc.vevo2.infer_vevo2_fm as vevo
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("--source", required=True)
parser.add_argument("--reference", required=True)
parser.add_argument("--output", required=True)
parser.add_argument("--no-pitch-shift", action="store_true")
args = parser.parse_args()
os.makedirs(os.path.dirname(os.path.abspath(args.output)), exist_ok=True)
started = time.monotonic()
vevo.inference_pipeline = vevo.load_inference_pipeline()
loaded = time.monotonic()
vevo.vevo2_fm(
args.source,
args.reference,
args.output,
shifted_src=not args.no_pitch_shift,
)
finished = time.monotonic()
print(
{
"model_load_seconds": round(loaded - started, 3),
"conversion_seconds": round(finished - loaded, 3),
"total_seconds": round(finished - started, 3),
"peak_vram_mib": round(torch.cuda.max_memory_allocated() / 1048576, 1),
"output": args.output,
},
flush=True,
)
if __name__ == "__main__":
main()
@@ -0,0 +1,56 @@
FROM nvidia/cuda:12.8.1-cudnn-runtime-ubuntu24.04
ARG XVC_COMMIT=49df8c591eafc48b096e466d96f9839f9c0dd739
RUN apt-get update \
&& apt-get install -y --no-install-recommends \
ca-certificates curl ffmpeg git libsndfile1 python3 python3-pip python3-venv \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/Jerrister/X-VC.git /opt/xvc \
&& cd /opt/xvc \
&& git checkout "${XVC_COMMIT}" \
&& rm -rf .git
RUN python3 -m venv /opt/venv
ENV PATH="/opt/venv/bin:${PATH}"
# RTX 5080: use a CUDA 12.8 PyTorch build instead of X-VC's older training pin.
RUN python -m pip install --no-cache-dir --upgrade pip \
&& python -m pip install --no-cache-dir \
--index-url https://download.pytorch.org/whl/cu128 \
"torch==2.8.0" "torchaudio==2.8.0" \
&& python -m pip install --no-cache-dir \
"gradio>=5.49,<6" "huggingface_hub>=0.34,<2" \
"transformers>=4.44,<5" "hydra-core>=1.3,<2" "omegaconf>=2.3,<3" \
"x-transformers>=1.40,<3" "einops>=0.8,<1" "einx>=0.3,<1" \
"numpy>=1.26,<3" "scipy>=1.13,<2" "soundfile>=0.12,<1" \
"soxr>=0.5,<1" "tqdm>=4.66,<5" "wandb>=0.18,<1"
RUN python -m pip install --no-cache-dir "descript-audiotools==0.7.2"
# Resemble Enhance declares old training-time pins for PyTorch, Gradio and
# DeepSpeed. Install only its inference code plus the small modules imported by
# that path, then remove the two unnecessary training imports. X-VC keeps the
# CUDA 12.8 / PyTorch 2.8 runtime required by the RTX 5080.
RUN python -m pip install --no-cache-dir --no-deps "resemble-enhance==0.0.1" \
&& python -m pip install --no-cache-dir \
"numpy==1.26.4" "scipy==1.11.4" \
"librosa==0.10.1" "soundfile==0.12.1" \
"matplotlib>=3.8,<4" "pandas>=2.1,<3" "rich>=13,<15" "tabulate>=0.9,<1"
COPY patch_resemble_enhance.py /tmp/patch_resemble_enhance.py
RUN python /tmp/patch_resemble_enhance.py && rm /tmp/patch_resemble_enhance.py
COPY app.py /opt/xvc/local_webui.py
COPY inference_log.py /opt/xvc/utils/log.py
ENV HF_HOME=/models/huggingface \
PYTHONUNBUFFERED=1
EXPOSE 8009
HEALTHCHECK --interval=5s --timeout=3s --start-period=600s --retries=3 \
CMD curl -fsS http://127.0.0.1:8009/ >/dev/null || exit 1
CMD ["python", "/opt/xvc/local_webui.py"]
@@ -0,0 +1,30 @@
# X-VC voice conversion on Athena
Isolated quality gate for `chenxie95/X-VC`, using the official inference path
from commit `49df8c591eafc48b096e466d96f9839f9c0dd739`. The UI is adapted from the
public Hugging Face Space at commit
`d761cd6421e85376b2656dfefd8471d7f35a42be` and runs locally without
ZeroGPU.
- Private URL: `http://192.168.1.212:8009`
- Source clip: speech content and timing to preserve
- Reference clip: target speaker identity
- Output: native 16 kHz PCM WAV plus optional Resemble-Enhance restoration at 44.1 kHz
- GPU: RTX 5080 only
- Persistent cache: `/data/voice/xvc/huggingface`
- Code and model license: MIT
The semantic tokenizer documents Chinese and English. German is therefore a
quality gate, not an assumed supported language. Keep OmniVoice installed: it
does text-to-speech cloning, while X-VC tests true audio-to-audio conversion.
Technical acceptance on 9 September 2026 used the repository's source and
target examples: 5.20 seconds were converted in 1.07 seconds (RTF 0.21). The
result was valid mono PCM WAV at 16 kHz, and the loaded process occupied about
2.9 GiB on the RTX 5080. German listening quality remains open.
The optional high-quality path uses `resemble-enhance` 0.0.1 with model
revision `4e3510ce4a8391159f665903544c5150bee7b2cb`. It does not change X-VC's
native 16-kHz architecture. Instead, it reconstructs missing speech bandwidth
after conversion and writes a second 44.1-kHz WAV. The UI always retains the
native output for an honest A/B comparison.
+238
View File
@@ -0,0 +1,238 @@
"""Local Athena adaptation of the public X-VC Gradio demo."""
import logging
import os
import sys
import tempfile
import time
from pathlib import Path
from typing import Tuple
import gradio as gr
import numpy as np
import soundfile as sf
import torch
from huggingface_hub import hf_hub_download
from omegaconf import OmegaConf
HERE = "/opt/xvc"
sys.path.insert(0, HERE)
from bins.infer_utils import precompute_conditions, run_offline, run_streaming, to_numpy_audio
from models.codec.sac.model import XVC
from utils.audio import audio_highpass_filter, audio_volume_normalize, load_audio
logging.basicConfig(level=logging.INFO)
log = logging.getLogger("xvc-local")
MODEL_REPO = "chenxie95/X-VC"
SPACE_REPO = "hugging-apps/x-vc-voice-conversion"
SPEAKER_SUBDIR = "pretrained/speech_eres2net_sv_en_voxceleb_16k"
SAMPLE_RATE = 16000
ENHANCED_SAMPLE_RATE = 44100
LATENT_HOP_LENGTH = 1280
MAX_SECONDS = 20.0
RESEMBLE_REVISION = "4e3510ce4a8391159f665903544c5150bee7b2cb"
RESEMBLE_RUN_DIR = Path("/models/huggingface/resemble-enhance/enhancer_stage2")
MODE_OFFLINE = "Offline (höchste Qualität)"
MODE_STREAMING = "Streaming (simulierte Echtzeit)"
def _load_model() -> XVC:
speaker_config = hf_hub_download(
repo_id=SPACE_REPO,
repo_type="space",
filename=f"{SPEAKER_SUBDIR}/configuration.json",
)
hf_hub_download(
repo_id=SPACE_REPO,
repo_type="space",
filename=f"{SPEAKER_SUBDIR}/pretrained_eres2net.ckpt",
)
checkpoint = hf_hub_download(repo_id=MODEL_REPO, filename="xvc.pt")
cfg = OmegaConf.load(os.path.join(HERE, "configs", "xvc.yaml"))
cfg["model"]["generator"].pop("loss_config", None)
cfg["model"].pop("discriminator", None)
cfg["model"]["generator"]["speaker_encoder"]["pretrained_dir"] = os.path.dirname(speaker_config)
infer_cfg = os.path.join(tempfile.gettempdir(), "xvc_inference.yaml")
OmegaConf.save(cfg, infer_cfg)
loaded = XVC.load_from_checkpoint(infer_cfg, checkpoint, device=torch.device("cuda"))
loaded.remove_weight_norm()
loaded = loaded.eval().to("cuda")
log.info("X-VC model ready on %s", torch.cuda.get_device_name(0))
return loaded
MODEL = _load_model()
def _prepare_wav(path: str) -> np.ndarray:
wav = load_audio(path, sampling_rate=SAMPLE_RATE, volume_normalize=False)
if wav is None or len(wav) == 0:
raise gr.Error("Die Audiodatei konnte nicht gelesen werden.")
wav = wav[: int(MAX_SECONDS * SAMPLE_RATE)]
wav = audio_volume_normalize(wav)
wav = audio_highpass_filter(wav, SAMPLE_RATE, 40)
remainder = len(wav) % LATENT_HOP_LENGTH
if remainder:
wav = np.pad(wav, (0, LATENT_HOP_LENGTH - remainder), mode="constant")
return wav.astype(np.float32)
def _tensor(wav: np.ndarray) -> torch.Tensor:
return torch.from_numpy(wav).unsqueeze(0).unsqueeze(1).float().to("cuda")
def _write_wav(audio: np.ndarray, sample_rate: int = SAMPLE_RATE, suffix: str = "native-16k") -> str:
os.makedirs("/output", exist_ok=True)
path = os.path.join("/output", f"xvc-{suffix}-{int(time.time() * 1000)}.wav")
sf.write(path, np.clip(np.asarray(audio, dtype=np.float32), -1.0, 1.0), sample_rate, subtype="PCM_16")
return path
@torch.inference_mode()
def _enhance_wav(audio: np.ndarray) -> tuple[str, float]:
if not (RESEMBLE_RUN_DIR / "hparams.yaml").is_file():
raise gr.Error("Resemble-Enhance-Gewichte fehlen; bitte den Athena-Operator prüfen lassen.")
from resemble_enhance.enhancer.inference import enhance
started = time.time()
source = torch.from_numpy(np.asarray(audio, dtype=np.float32).reshape(-1))
restored, sample_rate = enhance(
source,
SAMPLE_RATE,
"cuda",
nfe=32,
solver="midpoint",
lambd=0.1,
tau=0.5,
run_dir=RESEMBLE_RUN_DIR,
)
if int(sample_rate) != ENHANCED_SAMPLE_RATE:
raise RuntimeError(f"Unerwartete Resemble-Enhance-Abtastrate: {sample_rate}")
return _write_wav(restored.cpu().numpy(), int(sample_rate), "enhanced-44k1"), time.time() - started
@torch.inference_mode()
def convert(
source_audio: str,
reference_audio: str,
enhance_44k1: bool = True,
mode: str = MODE_OFFLINE,
chunk_ms: int = 2400,
current_ms: int = 120,
future_ms: int = 100,
smooth_ms: int = 20,
progress=gr.Progress(track_tqdm=True),
) -> Tuple[str, str | None, str]:
if not source_audio:
raise gr.Error("Bitte eine Quelldatei mit dem zu erhaltenden Inhalt hochladen.")
if not reference_audio:
raise gr.Error("Bitte eine Referenzdatei mit der Zielstimme hochladen.")
source_np = _prepare_wav(source_audio)
reference_np = _prepare_wav(reference_audio)
source_wav = _tensor(source_np)
target_wav = _tensor(reference_np)
seconds = len(source_np) / SAMPLE_RATE
started = time.time()
if mode == MODE_STREAMING:
history_ms = int(chunk_ms) - int(current_ms) - int(smooth_ms) - int(future_ms)
if history_ms < 0:
raise gr.Error("Fenster muss mindestens Current + Lookahead + Crossfade umfassen.")
speaker_condition, frame_condition = precompute_conditions(MODEL, target_wav, target_wav)
recon, latency_ms = run_streaming(
model=MODEL,
source_wav=source_wav,
speaker_condition=speaker_condition,
frame_condition=frame_condition,
sample_rate=SAMPLE_RATE,
chunk_ms=int(chunk_ms),
current_ms=int(current_ms),
future_ms=int(future_ms),
smooth_ms=int(smooth_ms),
)
elapsed = time.time() - started
latency = np.asarray(latency_ms, dtype=np.float64)
report = (
f"**Streaming** · {len(latency)} Chunks · Mittel **{latency.mean():.0f} ms**, "
f"P95 **{np.percentile(latency, 95):.0f} ms** · insgesamt {elapsed:.2f} s "
f"für {seconds:.2f} s Audio (RTF {elapsed / seconds:.2f})"
)
else:
recon = run_offline(MODEL, source_wav, target_wav, target_wav)
elapsed = time.time() - started
report = (
f"**Offline** · {seconds:.2f} s Audio in {elapsed:.2f} s "
f"(RTF {elapsed / seconds:.2f})"
)
recon_np = to_numpy_audio(recon)
native_path = _write_wav(recon_np)
enhanced_path = None
if enhance_44k1:
del source_wav, target_wav, recon
torch.cuda.empty_cache()
enhanced_path, enhancement_seconds = _enhance_wav(recon_np)
report += (
f" · Resemble Enhance **{enhancement_seconds:.2f} s**, "
f"Ausgabe **44,1 kHz** (rekonstruierte Bandbreite)"
)
return native_path, enhanced_path, report
CSS = "#col-container { max-width: 1100px; margin: 0 auto; }"
HEADER = """# X-VC — Voice Changer
Die **Quelle** liefert Text, Aussprache und Timing. Die **Referenz** liefert die
Zielstimme. X-VC arbeitet Audio-zu-Audio ohne Transkript oder Training.
[Paper](https://arxiv.org/abs/2604.12456) · [Modell](https://huggingface.co/chenxie95/X-VC) ·
[Code](https://github.com/Jerrister/X-VC)
"""
with gr.Blocks(title="X-VC Voice Changer", theme=gr.themes.Citrus(), css=CSS) as demo:
with gr.Column(elem_id="col-container"):
gr.Markdown(HEADER)
with gr.Row():
source = gr.Audio(label="Quelle — Inhalt und Sprechweise", type="filepath", sources=["upload", "microphone"])
reference = gr.Audio(label="Referenz — gewünschte Zielstimme", type="filepath", sources=["upload", "microphone"])
run = gr.Button("Stimme umwandeln", variant="primary")
enhance_44k1 = gr.Checkbox(
value=True,
label="Zusätzlich mit Resemble Enhance auf 44,1 kHz restaurieren",
info="Rekonstruiert fehlende Höhen per KI; das native 16-kHz-Ergebnis bleibt zum Vergleich erhalten.",
)
with gr.Row():
output_native = gr.Audio(label="X-VC Original · 16 kHz", type="filepath", autoplay=False)
output_enhanced = gr.Audio(label="Enhanced · 44,1 kHz", type="filepath", autoplay=False)
report = gr.Markdown()
with gr.Accordion("Erweiterte Einstellungen", open=False):
mode = gr.Radio([MODE_OFFLINE, MODE_STREAMING], value=MODE_OFFLINE, label="Verarbeitungsmodus")
with gr.Row():
current_ms = gr.Slider(40, 640, value=120, step=40, label="Aktueller Chunk (ms)")
chunk_ms = gr.Slider(800, 4800, value=2400, step=200, label="Gesamtfenster (ms)")
with gr.Row():
future_ms = gr.Slider(0, 400, value=100, step=20, label="Lookahead (ms)")
smooth_ms = gr.Slider(0, 100, value=20, step=10, label="Crossfade (ms)")
gr.Markdown(
"Die ersten 20 Sekunden jeder Datei werden verarbeitet. X-VC bleibt nativ bei 16 kHz; "
"die optionale zweite Datei rekonstruiert die Sprachbandbreite auf 44,1 kHz."
)
run.click(
convert,
inputs=[source, reference, enhance_44k1, mode, chunk_ms, current_ms, future_ms, smooth_ms],
outputs=[output_native, output_enhanced, report],
api_name="convert",
)
if __name__ == "__main__":
demo.queue(default_concurrency_limit=1).launch(
server_name="0.0.0.0",
server_port=8009,
show_error=True,
)
@@ -0,0 +1,39 @@
services:
xvc-studio:
build: .
image: mike-ai/xvc-studio:2026-09-09-enhance
container_name: mike-ai-xvc-studio
restart: "no"
labels:
com.mike-ai.voice-change-worker: xvc
environment:
NVIDIA_VISIBLE_DEVICES: ${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
HF_HOME: /models/huggingface
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
ports:
- "127.0.0.1:8009:8009"
volumes:
- /data/voice/xvc/huggingface:/models/huggingface
- /data/voice/xvc/output:/output
healthcheck:
test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8009/ >/dev/null"]
interval: 5s
timeout: 3s
start_period: 600s
retries: 3
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["${VOICE_GPU_UUID:?set VOICE_GPU_UUID to the RTX 5080 UUID}"]
capabilities: [gpu]
networks:
frontend:
aliases: [xvc-studio]
networks:
frontend:
name: mike-ai_frontend
external: true
@@ -0,0 +1,29 @@
"""Small inference-only replacement for X-VC's training logger.
The upstream logger imports WandB, Matplotlib and TensorBoard at module import
time although model inference only uses the normal logging functions.
"""
import logging
logging.basicConfig(level=logging.INFO)
_logger = logging.getLogger("xvc")
debug = _logger.debug
info = _logger.info
warn = _logger.warning
warning = _logger.warning
error = _logger.error
def init(*_args, **_kwargs):
return None
def write_audio(*_args, **_kwargs):
return None
def write_loss(*_args, **_kwargs):
return None
@@ -0,0 +1,40 @@
"""Remove training-only imports from Resemble Enhance's inference path."""
from pathlib import Path
import resemble_enhance
root = Path(resemble_enhance.__file__).parent
replacements = {
root / "enhancer" / "inference.py": {
"from .train import Enhancer, HParams": (
"from .enhancer import Enhancer\nfrom .hparams import HParams"
),
},
root / "denoiser" / "inference.py": {
"from .train import Denoiser, HParams": (
"from .denoiser import Denoiser\nfrom .hparams import HParams"
),
},
root / "enhancer" / "enhancer.py": {
"from ..utils.distributed import global_leader_only\n"
"from ..utils.train_loop import TrainLoop": (
"def global_leader_only(fn):\n"
" return fn\n\n"
"class TrainLoop:\n"
" @classmethod\n"
" def get_running_loop(cls):\n"
" return None"
),
},
}
for path, edits in replacements.items():
text = path.read_text(encoding="utf-8")
for old, new in edits.items():
if old not in text:
raise RuntimeError(f"Expected Resemble Enhance source not found in {path}: {old!r}")
text = text.replace(old, new, 1)
path.write_text(text, encoding="utf-8")
+47
View File
@@ -0,0 +1,47 @@
FROM python:3.12-slim-bookworm
ARG YUE2_COMMIT=9c6c4b349be978b06a9d0d958471a07a6cdeff4d
ARG YUE2_WEBUI_COMMIT=8fc05609bde5dcd345d7c6d57fff3da0839164c5
RUN apt-get update \
&& apt-get install -y --no-install-recommends ca-certificates ffmpeg git libsndfile1 \
&& rm -rf /var/lib/apt/lists/*
RUN git clone https://github.com/multimodal-art-projection/YuE.git /opt/yue2 \
&& git -C /opt/yue2 checkout "${YUE2_COMMIT}" \
&& python -m pip install --no-cache-dir /opt/yue2
# The community console is intentionally layered onto the already verified,
# pinned YuE2 runtime. We use only its webui/ directory; its bundled YuE2 fork
# and Windows installers never become part of the Athena image.
RUN git clone https://github.com/Ladypoly/YuE2_WebUI.git /tmp/yue2-webui \
&& git -C /tmp/yue2-webui checkout "${YUE2_WEBUI_COMMIT}" \
&& cp -a /tmp/yue2-webui/webui /opt/yue2/webui \
&& cp /tmp/yue2-webui/LICENSE /opt/yue2/YuE2_WebUI-LICENSE \
&& python -m pip install --no-cache-dir -r /opt/yue2/webui/requirements.txt \
&& rm -rf /tmp/yue2-webui
# SheetSage2 converts an uploaded recording into the ABC score YuE2 consumes.
# Keep its pinned Transformers/Numpy stack in a small virtual environment, but
# reuse the image's Blackwell-capable torch 2.10 + CUDA 12.8 installation.
# The upstream torch 2.8+cu126 recipe cannot execute RTX 50-series kernels.
RUN python -m venv --system-site-packages /opt/yue2/.venv-sheetsage2 \
&& /opt/yue2/.venv-sheetsage2/bin/pip install --no-cache-dir \
"torchaudio==2.10.0" --index-url https://download.pytorch.org/whl/cu128 \
&& /opt/yue2/.venv-sheetsage2/bin/pip install --no-cache-dir \
"transformers==4.45.2" "huggingface-hub==0.36.0" \
"safetensors==0.5.3" "numpy==1.26.4" "scipy==1.13.1" \
"mir_eval==0.8.2" "pretty_midi==0.2.10" "mido==1.3.3" \
"setuptools==78.1.1"
COPY community-webui-patches /tmp/community-webui-patches
RUN python /tmp/community-webui-patches/patch_webui.py /opt/yue2/webui \
&& rm -rf /tmp/community-webui-patches
COPY ui /opt/yue2-playground
WORKDIR /workspace
ENV PYTHONUNBUFFERED=1 \
YUE2_KIT=/workspace
ENTRYPOINT ["yue2"]
+102
View File
@@ -0,0 +1,102 @@
# YuE2 3B isolated quality test
Prepared, non-starting evaluation of `m-a-p/YuE2-3B` with the standard
`m-a-p/YuE2-Vae` listening decoder. The source is pinned to the official
`yue2-v0.1.6` commit `9c6c4b349be978b06a9d0d958471a07a6cdeff4d`.
Preparation on Athena is complete. The model and VAE files were checked
against their published `weights_manifest.json` SHA-256 values. The Docker
image is built, but no YuE2 container has been created or started.
## Safety and isolation
- This experiment is not part of the profile controller or dashboard.
- The Compose service uses the `manual` profile, has no restart policy and
cannot start through an ordinary `docker compose up`.
- Only the RTX 5080 is exposed to the container.
- Building and downloading do not load the model or use a GPU.
- Do not start it while another Athena GPU job is active.
## Persistent files
```text
/data/models/yue2/
├── YuE2-3B/
└── YuE2-Vae/
/data/music/yue2/
```
The initial control request is a true empty-lyrics instrumental request. No
invented `[Instrumental]` lyrics marker is used.
## Community WebUI
The `community-ui` profile runs Ladypoly's YuE2 WebUI at the pinned commit
`8fc05609bde5dcd345d7c6d57fff3da0839164c5`. Only the WebUI layer is copied
from that repository. Athena continues to use the verified official YuE2
`0.1.6` runtime and the existing model cache; the WebUI fork's bundled model
code and Windows installers are not used.
The UI is bound only to Athena's localhost on port 8014 and stores complete
takes below `/data/music/yue2`. Optional Windows-only installers for llama.cpp
and stable-diffusion.cpp are not part of the Athena setup. Manual composition,
generation, result playback, score editing and the take library work without
those optional components.
### Uploaded-audio remix (SheetSage2)
The **Remix a take** drawer also accepts WAV, FLAC, MP3, M4A, OGG, Opus and
AAC uploads. SheetSage2 transcribes the recording into an editable ABC melody
and chord plan, then YuE2 renders that structure in a newly selected style.
It does not preserve the original samples, singer or production verbatim.
SheetSage2 is kept in `/opt/yue2/.venv-sheetsage2`, while its persistent model
files live below `/data/models/yue2`. The environment intentionally reuses the
image's PyTorch 2.10/CUDA 12.8 runtime: the upstream cu126 recipe is not
Blackwell-capable. Required persistent directories are:
```text
/data/models/yue2/
├── SheetSage2/
└── MERT-v2-FullSong/
```
The local `SheetSage2/config.json` must point `base_model_name_or_path` to
`/opt/yue2/models/MERT-v2-FullSong`, allowing the complete transcription path
to run offline. Analysis and generation share the RTX 5080 and therefore run
sequentially; the UI parks YuE2 before starting SheetSage2.
The small integration patch under `community-webui-patches/` fixes the
community release's missing `refreshArt()` function on Linux and permits the
native YuE2 empty-lyrics request for true instrumentals. It deliberately does
not modify the model runtime.
```sh
docker compose --profile community-ui up -d yue2-ui
```
The original small German playground remains available as a stopped fallback
on localhost port 8016 through the `playground-fallback` profile. Do not run
both frontends concurrently because both can submit work to the same GPU.
## Manual test (only after GPU availability was checked)
From `/opt/mike-ai/yue2-3b` on Athena:
```sh
docker compose --profile manual run --rm yue2-test generate \
--offline \
--device cuda:0 \
--budget 16 \
--request /workspace/requests/instrumental-synthwave.json \
--output /workspace/runs
```
Start with the official unquantized BF16 path. If and only if this fails from
VRAM pressure, repeat with `--quantization fp8 --offload-ar`; keep the outputs
separate because that is a different inference configuration.
YuE2 is newly released and officially specifies a 24-GB BF16 GPU. Readiness of
this image and the downloaded weights is not evidence that the 16-GB RTX 5080
run will fit or that its audio quality is acceptable.
@@ -0,0 +1,167 @@
"""Small Linux integration fixes for the pinned community WebUI.
Keep these transformations explicit and fail the image build if upstream moves
the expected code. That prevents a future upstream update from silently
producing a half-patched console.
"""
from pathlib import Path
import sys
root = Path(sys.argv[1])
app_js = root / "static" / "app.js"
index_html = root / "static" / "index.html"
server_py = root / "server.py"
def replace_once(path: Path, old: str, new: str) -> None:
text = path.read_text(encoding="utf-8")
if text.count(old) != 1:
raise RuntimeError(f"expected exactly one patch marker in {path}: {old[:80]!r}")
path.write_text(text.replace(old, new, 1), encoding="utf-8")
# The published UI calls refreshArt() during startup, but does not define it.
# Athena does not install the Windows-only stable-diffusion.cpp helper, so show
# that state honestly and keep the optional controls inert instead of crashing
# the whole page.
art_marker = ''' $("artDirAdd").addEventListener("click", function () {
'''
art_fix = ''' function refreshArt() {
return api("/api/art/status").then(function (data) {
STATE.art = data;
var installed = !!data.installed;
var supported = !!data.supported;
$("artState").textContent = data.busy || (installed ? "installed" :
(supported ? "not installed" : "not available on Linux"));
$("artState").dataset.s = data.busy ? "busy" : (installed ? "ready" : "missing");
$("artInstall").disabled = !!data.busy || !supported;
$("artInstall").textContent = installed ? "Reinstall" : "Install";
$("setArtAuto").checked = !!data.auto;
$("setArtAuto").disabled = !installed;
$("setArtModel").innerHTML = (data.models || []).length
? data.models.map(function (m) {
return '<option value="' + escape(m.path) + '">' + escape(m.name || m.file) + '</option>';
}).join("")
: '<option value="">no art model available</option>';
if (data.selected) $("setArtModel").value = data.selected;
$("setArtModel").disabled = !installed;
$("artDir").disabled = !supported;
$("artDirAdd").disabled = !supported;
$("artDirs").textContent = supported
? ((data.dirs || []).length ? "Also scanning: " + data.dirs.join(" · ") : "No extra folders configured.")
: "The community release only provides the cover-art installer for Windows; song generation is unaffected.";
$("artCatalog").innerHTML = "";
return data;
}).catch(function (error) {
$("artState").textContent = "unavailable";
$("artInstall").disabled = true;
$("artDirs").textContent = error.message;
});
}
$("artDirAdd").addEventListener("click", function () {
'''
replace_once(app_js, art_marker, art_fix)
# YuE2 natively supports an empty lyric string for true instrumentals. The
# community server rejected that valid request even though the browser did not.
replace_once(
server_py,
''' if not body.style.strip() or not body.lyrics.strip():
raise HTTPException(400, "A style prompt and lyrics are both required")
''',
''' if not body.style.strip():
raise HTTPException(400, "A style prompt is required")
''',
)
replace_once(
index_html,
'''<span class="label">Lyrics <em>section tags on their own line</em></span>''',
'''<span class="label">Lyrics <em>section tags on their own line · leave empty for a true instrumental</em></span>''',
)
# The backend already ships a complete SheetSage2 upload endpoint but hides it
# from the published console. Surface it beside the existing take-to-take
# remix controls so uploaded songs can seed a new melody/arrangement.
replace_once(
index_html,
''' <p class="row-hint" id="coverStatus"></p>
</div>
</details>
''',
''' <p class="row-hint" id="coverStatus"></p>
<h4>Remix an uploaded recording</h4>
<p class="hint">SheetSage2 listens to an uploaded song and writes an editable melody
and chord score for YuE2. This is structural transcription, not a sample or a copy of
the original sound. After analysis, choose a new style and generate normally.</p>
<div class="field-row">
<label class="field grow">
<span class="label">Source audio</span>
<input id="coverAudio" type="file" accept=".wav,.flac,.mp3,.m4a,.ogg,.opus,.aac,audio/*" />
</label>
<div class="field reset-cell">
<button type="button" class="btn ghost" id="coverFromAudio">Analyse uploaded melody</button>
</div>
</div>
<label class="check"><input type="checkbox" id="coverAudioMelodyOnly" checked />
<span>Keep the melody, but let the new style rebuild the harmony</span></label>
<p class="row-hint" id="coverAudioStatus">Checking SheetSage2…</p>
</div>
</details>
''',
)
replace_once(
app_js,
''' $("coverStatus").textContent = takes.length
? takes.length + " takes carry a score you can remix."
: "Make a song in Full plan or Melody only mode first; Direct mode keeps no score.";
}).catch(function () {});
''',
''' $("coverStatus").textContent = takes.length
? takes.length + " takes carry a score you can remix."
: "Make a song in Full plan or Melody only mode first; Direct mode keeps no score.";
var sheetsage = data.sheetsage || {};
$("coverFromAudio").disabled = !sheetsage.available;
$("coverAudioStatus").textContent = sheetsage.available
? "SheetSage2 is ready. Analysis temporarily parks YuE2 because both use the GPU."
: "SheetSage2 is not installed yet; uploaded-audio remix is unavailable.";
}).catch(function (error) {
$("coverFromAudio").disabled = true;
$("coverAudioStatus").textContent = error.message;
});
''',
)
replace_once(
app_js,
''' /* -------------------------------------------------------------- compose */
''',
''' $("coverFromAudio").addEventListener("click", function () {
var file = $("coverAudio").files[0];
if (!file) return toast("Choose an audio file first", "bad");
var button = $("coverFromAudio");
var body = new FormData();
body.append("file", file, file.name);
body.append("melody_only", $("coverAudioMelodyOnly").checked ? "true" : "false");
button.disabled = true;
button.textContent = "Analysing…";
$("coverAudioStatus").textContent = "Listening to “" + file.name + "”… this can take several minutes.";
api("/api/cover/from-audio", { method: "POST", body: body }).then(function (data) {
applyCoverScore(data.abc, "Melody analysed from “" + data.source + "” in " + data.seconds + " s. Choose the new style and generate.");
$("coverAudioStatus").textContent = "Analysis complete in " + data.seconds + " s. The editable score is loaded below.";
}).catch(function (error) {
$("coverAudioStatus").textContent = error.message;
toast(error.message, "bad");
}).then(function () {
button.textContent = "Analyse uploaded melody";
refreshCover();
});
});
/* -------------------------------------------------------------- compose */
''',
)
+104
View File
@@ -0,0 +1,104 @@
services:
yue2-test:
profiles: ["manual"]
build:
context: .
args:
YUE2_COMMIT: 9c6c4b349be978b06a9d0d958471a07a6cdeff4d
YUE2_WEBUI_COMMIT: 8fc05609bde5dcd345d7c6d57fff3da0839164c5
image: mike-ai/yue2:3b-0.1.6
restart: "no"
environment:
NVIDIA_VISIBLE_DEVICES: "GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"
NVIDIA_DRIVER_CAPABILITIES: compute,utility
volumes:
- /data/models/yue2:/workspace/models:ro
- /data/music/yue2:/workspace/runs
- ./requests:/workspace/requests:ro
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"]
capabilities: [gpu]
yue2-ui:
profiles: ["community-ui"]
build:
context: .
args:
YUE2_COMMIT: 9c6c4b349be978b06a9d0d958471a07a6cdeff4d
YUE2_WEBUI_COMMIT: 8fc05609bde5dcd345d7c6d57fff3da0839164c5
image: mike-ai/yue2:3b-0.1.6
container_name: mike-ai-yue2-playground
restart: "no"
labels:
com.mike-ai.music-worker: yue2
entrypoint: ["python", "/opt/yue2/webui/server.py"]
environment:
NVIDIA_VISIBLE_DEVICES: "GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"
NVIDIA_DRIVER_CAPABILITIES: compute,utility
YUE2_HOST: "0.0.0.0"
YUE2_PORT: "8014"
YUE2_OUTPUTS: "/workspace/runs"
HF_HUB_OFFLINE: "1"
TRANSFORMERS_OFFLINE: "1"
ports:
- "127.0.0.1:8014:8014"
networks:
frontend:
aliases: [yue2-studio]
volumes:
- /data/models/yue2:/opt/yue2/models:ro
- /data/music/yue2:/workspace/runs
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"]
capabilities: [gpu]
healthcheck:
test: ["CMD", "python", "-c", "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8014/api/state', timeout=2)"]
interval: 5s
timeout: 3s
retries: 12
start_period: 5s
# Kept as a deliberately non-default fallback until the community console
# has completed a real generation on Athena.
yue2-playground-fallback:
profiles: ["playground-fallback"]
build:
context: .
args:
YUE2_COMMIT: 9c6c4b349be978b06a9d0d958471a07a6cdeff4d
YUE2_WEBUI_COMMIT: 8fc05609bde5dcd345d7c6d57fff3da0839164c5
image: mike-ai/yue2:3b-0.1.6
container_name: mike-ai-yue2-playground-fallback
restart: "no"
entrypoint: ["python", "/opt/yue2-playground/server.py"]
environment:
NVIDIA_VISIBLE_DEVICES: "GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"
NVIDIA_DRIVER_CAPABILITIES: compute,utility
YUE2_UI_HOST: "0.0.0.0"
YUE2_UI_PORT: "8016"
ports:
- "127.0.0.1:8016:8016"
volumes:
- /data/models/yue2:/workspace/models:ro
- /data/music/yue2:/workspace/runs
deploy:
resources:
reservations:
devices:
- driver: nvidia
device_ids: ["GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe"]
capabilities: [gpu]
networks:
frontend:
external: true
name: mike-ai_frontend
@@ -0,0 +1,7 @@
{
"id": "instrumental_synthwave_control",
"style": "Instrumental, synthwave, synth-pop, energetic, memorable lead melody, arpeggiated synthesizer, analog synthesizer, drum machine, driving bass, 118 BPM, no vocals",
"lyrics": "",
"cot": "full",
"seed": 831001
}
+45
View File
@@ -0,0 +1,45 @@
<!doctype html>
<html lang="de">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<title>YuE2 Playground</title>
<style>
:root{color-scheme:dark;--bg:#090d14;--panel:#121925;--line:#293344;--text:#eef3fb;--muted:#98a6b9;--blue:#58a6ff;--green:#42d392;--red:#ff6b7a}
*{box-sizing:border-box} body{margin:0;background:radial-gradient(circle at 20% 0,#16243b 0,transparent 35%),var(--bg);font:15px/1.5 system-ui,sans-serif;color:var(--text)}
main{max-width:1100px;margin:auto;padding:36px 22px 80px} h1{font-size:36px;margin:0} h2{margin:0 0 16px}.lead{color:var(--muted);margin:4px 0 28px}
.grid{display:grid;grid-template-columns:1.15fr .85fr;gap:22px}@media(max-width:800px){.grid{grid-template-columns:1fr}}
.card{background:rgba(18,25,37,.94);border:1px solid var(--line);border-radius:16px;padding:22px;box-shadow:0 18px 60px #0005}
label{display:block;font-weight:650;margin:14px 0 6px}textarea,input,select{width:100%;background:#090e17;border:1px solid #344157;border-radius:9px;color:var(--text);padding:11px;font:inherit}textarea{resize:vertical;min-height:120px}
.row{display:grid;grid-template-columns:1fr 1fr;gap:12px}.check{display:flex;align-items:center;gap:10px;margin:14px 0}.check input{width:auto}
button{border:0;border-radius:10px;padding:12px 16px;font-weight:750;cursor:pointer;background:var(--blue);color:#05101d}button:disabled{opacity:.45;cursor:not-allowed}.ghost{background:#242e3d;color:var(--text);padding:7px 10px}
.notice{padding:12px;border-radius:9px;background:#0c2630;color:#a9edda;margin-top:14px}.error{background:#371a23;color:#ffc0c7}.job{border-top:1px solid var(--line);padding:15px 0}.job:first-child{border-top:0;padding-top:0}.meta{color:var(--muted);font-size:13px}.state{font-weight:750;color:var(--green)}.state.failed{color:var(--red)}audio{width:100%;margin-top:10px}.job-head{display:flex;justify-content:space-between;gap:12px}.style{white-space:pre-wrap;margin:5px 0}.empty{color:var(--muted)}
</style>
</head>
<body><main>
<h1>YuE2 Playground 🎵</h1><p class="lead">Lokale Musikgenerierung auf Athena · 48 kHz Stereo · RTX 5080</p>
<div class="grid">
<section class="card"><h2>Neuen Song erzeugen</h2>
<form id="form">
<label for="style">Stil und musikalische Vorgaben</label>
<textarea id="style" required>Instrumental, synthwave, synth-pop, energetic, memorable lead melody, arpeggiated synthesizer, analog synthesizer, drum machine, driving bass, 118 BPM, no vocals</textarea>
<label class="check"><input id="instrumental" type="checkbox" checked> Instrumental – ohne Gesang</label>
<div id="lyrics-wrap" hidden><label for="lyrics">Liedtext</label><textarea id="lyrics" placeholder="[Verse]\n...\n\n[Chorus]\n..."></textarea></div>
<div class="row"><div><label for="seed">Seed (leer = Zufall)</label><input id="seed" type="number" min="0" max="4294967295" placeholder="zufällig"></div>
<div><label for="cot">Kompositionsplanung</label><select id="cot"><option value="full">Vollständig – Melodie und Akkorde</option><option value="off">Direkt – ohne editierbaren Plan</option></select></div></div>
<div id="message" class="notice" hidden></div>
<button id="submit" type="submit" style="margin-top:18px;width:100%">Song generieren</button>
</form>
</section>
<section class="card"><h2>Ergebnisse</h2><div id="jobs" class="empty">Wird geladen …</div></section>
</div>
</main><script>
const $=s=>document.querySelector(s), form=$('#form'), inst=$('#instrumental'), lyricsWrap=$('#lyrics-wrap'), submit=$('#submit'), msg=$('#message'), jobs=$('#jobs');
inst.onchange=()=>lyricsWrap.hidden=inst.checked;
function esc(v){return String(v??'').replace(/[&<>"']/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;',"'":'&#39;'}[c]))}
function duration(v){if(!v)return '';const m=Math.floor(v/60),s=Math.round(v%60);return `${m}:${String(s).padStart(2,'0')} min`}
async function refresh(){try{const r=await fetch('/api/jobs',{cache:'no-store'}),d=await r.json();submit.disabled=!!d.active;submit.textContent=d.active?'YuE2 arbeitet …':'Song generieren';jobs.className='';jobs.innerHTML=d.jobs.length?d.jobs.map(j=>`<article class="job"><div class="job-head"><span class="state ${j.state}">${j.state==='running'?'Wird erzeugt …':j.state==='complete'?'Fertig':'Fehlgeschlagen'}</span>${j.state!=='running'?`<button class="ghost" onclick="removeJob('${esc(j.id)}')">Löschen</button>`:''}</div><div class="style">${esc(j.style)}</div><div class="meta">Seed ${esc(j.seed)} · ${j.instrumental?'Instrumental':'Gesang'}${j.audio_seconds?` · ${duration(j.audio_seconds)}`:''}${j.elapsed_seconds?` · erzeugt in ${duration(j.elapsed_seconds)}`:''}</div>${j.audio_url?`<audio controls preload="metadata" src="${j.audio_url}"></audio><p><a href="${j.audio_url}" download="${esc(j.id)}.flac">FLAC herunterladen</a></p>`:''}${j.error?`<div class="notice error">${esc(j.error)}<pre>${esc(j.log||'')}</pre></div>`:''}</article>`).join(''):'<p class="empty">Noch keine Songs vorhanden.</p>'}catch(e){jobs.innerHTML=`<div class="notice error">${esc(e)}</div>`}}
async function removeJob(id){if(!confirm('Diesen Song und alle Zwischenartefakte wirklich löschen?'))return;await fetch('/api/jobs/'+encodeURIComponent(id),{method:'DELETE'});refresh()}
form.onsubmit=async e=>{e.preventDefault();msg.hidden=true;submit.disabled=true;try{const payload={style:$('#style').value,lyrics:$('#lyrics').value,instrumental:inst.checked,seed:$('#seed').value,cot:$('#cot').value};const r=await fetch('/api/jobs',{method:'POST',headers:{'Content-Type':'application/json'},body:JSON.stringify(payload)}),d=await r.json();if(!r.ok)throw new Error(d.error||'Start fehlgeschlagen');msg.className='notice';msg.textContent=`Auftrag ${d.id} gestartet. Die Seite aktualisiert sich automatisch.`;msg.hidden=false;refresh()}catch(e){msg.className='notice error';msg.textContent=e.message;msg.hidden=false;submit.disabled=false}};
refresh();setInterval(refresh,2000);
</script></body></html>
+261
View File
@@ -0,0 +1,261 @@
#!/usr/bin/env python3
from __future__ import annotations
import json
import mimetypes
import os
import random
import re
import shutil
import subprocess
import threading
import time
from http import HTTPStatus
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from urllib.parse import unquote, urlparse
HOST = os.getenv("YUE2_UI_HOST", "0.0.0.0")
PORT = int(os.getenv("YUE2_UI_PORT", "8014"))
ROOT = Path("/workspace/runs")
REQUESTS = ROOT / ".playground_requests"
LOGS = ROOT / ".playground_logs"
INDEX = Path(__file__).with_name("index.html")
SAFE_ID = re.compile(r"^[a-z0-9][a-z0-9_-]{0,63}$")
LOCK = threading.Lock()
JOBS: dict[str, dict] = {}
ACTIVE: str | None = None
def now_ms() -> int:
return int(time.time() * 1000)
def existing_jobs() -> list[dict]:
found: list[dict] = []
for result_file in ROOT.glob("*/result.json"):
try:
result = json.loads(result_file.read_text(encoding="utf-8"))
request_file = result_file.parent / "request.json"
request = json.loads(request_file.read_text(encoding="utf-8"))
found.append({
"id": result_file.parent.name,
"state": "complete",
"style": request.get("style", ""),
"seed": request.get("seed"),
"instrumental": not bool(request.get("lyrics")),
"audio_seconds": result.get("audio_seconds"),
"elapsed_seconds": (result.get("timing") or {}).get("e2e_seconds"),
"audio_url": f"/audio/{result_file.parent.name}",
"created": int(result_file.stat().st_mtime * 1000),
})
except (OSError, ValueError, TypeError):
continue
return sorted(found, key=lambda item: item["created"], reverse=True)
def snapshot() -> dict:
with LOCK:
live = [dict(item) for item in JOBS.values()]
active = ACTIVE
known = {item["id"] for item in live}
live.extend(item for item in existing_jobs() if item["id"] not in known)
return {"active": active, "jobs": sorted(live, key=lambda item: item["created"], reverse=True)}
def run_job(job_id: str, request: dict) -> None:
global ACTIVE
output = ROOT / job_id
request_file = REQUESTS / f"{job_id}.json"
log_file = LOGS / f"{job_id}.log"
command = [
"yue2", "generate", "--offline", "--device", "cuda:0", "--budget", "16",
# YuE2 creates a child directory from request["id"] itself.
"--request", str(request_file), "--output", str(ROOT),
]
started = time.monotonic()
try:
with log_file.open("w", encoding="utf-8") as log:
process = subprocess.Popen(command, stdout=log, stderr=subprocess.STDOUT, text=True)
with LOCK:
JOBS[job_id]["pid"] = process.pid
code = process.wait()
if code != 0:
raise RuntimeError(f"YuE2 wurde mit Exit-Code {code} beendet")
result = json.loads((output / "result.json").read_text(encoding="utf-8"))
update = {
"state": "complete",
"audio_seconds": result.get("audio_seconds"),
"elapsed_seconds": round(time.monotonic() - started, 1),
"audio_url": f"/audio/{job_id}",
}
except Exception as exc:
tail = ""
try:
tail = "\n".join(log_file.read_text(encoding="utf-8", errors="replace").splitlines()[-30:])
except OSError:
pass
update = {"state": "failed", "error": str(exc), "log": tail,
"elapsed_seconds": round(time.monotonic() - started, 1)}
with LOCK:
JOBS[job_id].update(update)
ACTIVE = None
class Handler(BaseHTTPRequestHandler):
server_version = "YuE2Playground/1.0"
def log_message(self, fmt: str, *args: object) -> None:
print(f"{self.address_string()} - {fmt % args}", flush=True)
def json_response(self, status: int, payload: object) -> None:
body = json.dumps(payload, ensure_ascii=False).encode("utf-8")
self.send_response(status)
self.send_header("Content-Type", "application/json; charset=utf-8")
self.send_header("Content-Length", str(len(body)))
self.send_header("Cache-Control", "no-store")
self.end_headers()
self.wfile.write(body)
def do_GET(self) -> None: # noqa: N802
path = urlparse(self.path).path
if path == "/health":
self.json_response(200, {"status": "ok", "active": ACTIVE})
return
if path == "/api/jobs":
self.json_response(200, snapshot())
return
if path.startswith("/audio/"):
self.send_audio(unquote(path.removeprefix("/audio/")))
return
if path in {"/", "/index.html"}:
body = INDEX.read_bytes()
self.send_response(200)
self.send_header("Content-Type", "text/html; charset=utf-8")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
return
self.send_error(404)
def do_POST(self) -> None: # noqa: N802
global ACTIVE
if urlparse(self.path).path != "/api/jobs":
self.send_error(404)
return
try:
length = int(self.headers.get("Content-Length", "0"))
if length <= 0 or length > 65536:
raise ValueError("Ungültige Anfragegröße")
data = json.loads(self.rfile.read(length))
style = str(data.get("style", "")).strip()
lyrics = str(data.get("lyrics", "")).strip()
instrumental = bool(data.get("instrumental", False))
cot = str(data.get("cot", "full"))
if not style or len(style) > 3000:
raise ValueError("Bitte eine Stilbeschreibung mit höchstens 3000 Zeichen eingeben")
if len(lyrics) > 20000:
raise ValueError("Der Liedtext ist zu lang")
if cot not in {"full", "off"}:
raise ValueError("Unbekannter Planungsmodus")
if not instrumental and not lyrics:
raise ValueError("Für einen Song mit Gesang fehlt der Liedtext")
if instrumental:
lyrics = ""
if "instrumental" not in style.casefold():
style = "Instrumental, no vocals, " + style
raw_seed = data.get("seed")
seed = int(raw_seed) if str(raw_seed).strip() else random.SystemRandom().randrange(1, 2**31)
if not 0 <= seed < 2**32:
raise ValueError("Seed muss zwischen 0 und 4294967295 liegen")
except (ValueError, TypeError, json.JSONDecodeError) as exc:
self.json_response(400, {"error": str(exc)})
return
with LOCK:
if ACTIVE is not None:
self.json_response(409, {"error": f"Auftrag {ACTIVE} läuft bereits"})
return
job_id = time.strftime("song-%Y%m%d-%H%M%S") + f"-{seed % 10000:04d}"
request = {"id": job_id, "style": style, "lyrics": lyrics, "cot": cot, "seed": seed}
REQUESTS.mkdir(parents=True, exist_ok=True)
LOGS.mkdir(parents=True, exist_ok=True)
request_file = REQUESTS / f"{job_id}.json"
request_file.write_text(json.dumps(request, ensure_ascii=False, indent=2) + "\n", encoding="utf-8")
JOBS[job_id] = {"id": job_id, "state": "running", "style": style,
"seed": seed, "instrumental": instrumental,
"created": now_ms(), "elapsed_seconds": 0}
ACTIVE = job_id
threading.Thread(target=run_job, args=(job_id, request), daemon=True).start()
self.json_response(HTTPStatus.ACCEPTED, JOBS[job_id])
def do_DELETE(self) -> None: # noqa: N802
path = urlparse(self.path).path
job_id = unquote(path.removeprefix("/api/jobs/"))
if not path.startswith("/api/jobs/") or not SAFE_ID.fullmatch(job_id):
self.send_error(404)
return
with LOCK:
if ACTIVE == job_id:
self.json_response(409, {"error": "Ein laufender Auftrag kann nicht gelöscht werden"})
return
JOBS.pop(job_id, None)
shutil.rmtree(ROOT / job_id, ignore_errors=True)
for directory, suffix in ((REQUESTS, ".json"), (LOGS, ".log")):
try:
(directory / f"{job_id}{suffix}").unlink()
except FileNotFoundError:
pass
self.json_response(200, {"status": "deleted", "id": job_id})
def send_audio(self, job_id: str) -> None:
if not SAFE_ID.fullmatch(job_id):
self.send_error(404)
return
path = ROOT / job_id / "audio.flac"
if not path.is_file():
self.send_error(404)
return
size = path.stat().st_size
start, end = 0, size - 1
status = 200
range_header = self.headers.get("Range", "")
if range_header.startswith("bytes="):
try:
left, right = range_header[6:].split("-", 1)
start = int(left) if left else 0
end = min(int(right), size - 1) if right else size - 1
if start < 0 or start > end:
raise ValueError
status = 206
except ValueError:
self.send_error(416)
return
self.send_response(status)
self.send_header("Content-Type", mimetypes.guess_type(path.name)[0] or "audio/flac")
self.send_header("Accept-Ranges", "bytes")
self.send_header("Content-Length", str(end - start + 1))
if status == 206:
self.send_header("Content-Range", f"bytes {start}-{end}/{size}")
self.end_headers()
with path.open("rb") as source:
source.seek(start)
remaining = end - start + 1
while remaining:
chunk = source.read(min(1024 * 1024, remaining))
if not chunk:
break
try:
self.wfile.write(chunk)
except (BrokenPipeError, ConnectionResetError):
# Browsers routinely close an old range request after a
# seek or metadata probe. This is not a server failure.
break
remaining -= len(chunk)
if __name__ == "__main__":
ROOT.mkdir(parents=True, exist_ok=True)
print(f"YuE2 Playground listening on {HOST}:{PORT}", flush=True)
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
+264
View File
@@ -0,0 +1,264 @@
# Athena: verbindlicher Kontext für KI-Agenten
Stand: 10. September 2026, nach Integration von TRELLIS.2 als 3D-Studio.
Diese Datei ist die erste Lektüre für jede KI, die Athena prüfen oder ändern
soll. Sie beschreibt den realen Aufbau, die Zuständigkeiten und die Regeln für
sichere Erweiterungen. Bei Abweichungen zwischen Annahmen und Live-System gilt:
erst lesend prüfen, dann die Dokumentation und den Code gemeinsam korrigieren.
## Unverhandelbare Sicherheitsregeln
1. **Athena niemals herunterfahren oder neu starten.** Der Rechner steht in
einer anderen Stadt und ist nicht kurzfristig physisch erreichbar.
2. Ohne ausdrücklichen aktuellen Auftrag weder Kernel, Bootloader, BIOS,
Partitionen, Mounts, SSH, LAN, WireGuard noch Firewall verändern.
3. Secrets dürfen lokal benutzt, aber niemals ausgegeben, geloggt oder in Git
aufgenommen werden. Das betrifft besonders `/etc/mike-ai`.
4. Keine laufende Modellarbeit abbrechen. Vor Änderungen Betriebsmodus,
Containerzustand und GPU-Prozesse prüfen.
5. Keine pauschalen Docker-Bereinigungen ausführen. Ein gestoppter Worker ist
meistens gewollt und kein Müll.
6. Keine Container anhand zufälliger IDs verdrahten. Stabile Dienstnamen,
Compose-Netze und eindeutige `com.mike-ai.*`-Labels verwenden.
7. Änderungen klein und reversibel halten. Nie den gesamten Stack neu erstellen,
wenn ein einzelner Dienst aktualisiert werden kann.
## Physischer und logischer Aufbau
```text
Athena: ASUS PRIME B550-PLUS
├── Debian 13 (trixie), Kernel 6.12
├── AMD Ryzen 5 5600, 6 Kerne / 12 Threads
├── 46 GiB nutzbarer RAM + 47 GiB Swap
├── System: Samsung 980 PRO 1 TB, ext4 auf /
├── Daten: WD Blue SN580 1 TB, ext4 auf /data
├── GPU 0: RTX 3060, 12.288 MiB
├── GPU 1: RTX 5080, 16.303 MiB
└── Docker
├── Kernprojekt /opt/mike-ai/stack
│ ├── Router, Profile Controller und Dashboard
│ ├── fünf llama.cpp-Profile
│ ├── Bild, Qwen3-TTS, TTS-Gateway und Whisper
│ ├── WireGuard-Gateway, Portainer, Backup
│ └── Athena-Operator
├── /opt/mike-ai/acestep-test Musik
├── /opt/mike-ai/stem-separator Audio-Trennung
├── /opt/mike-ai/omnivoice-studio Voice Studio
├── /opt/mike-ai/xvc-studio Voice Changer
├── /opt/mike-ai/stack/experiments/applio-rvc
│ Applio/RVC
├── /opt/mike-ai/Mikes-Applio-UI geführte Applio-UI
└── /opt/mike-ai/trellis-studio 3D Studio
```
Die beiden GPUs bilden **keinen gemeinsamen VRAM-Pool**. Ein Backend muss
Mehrkartenbetrieb ausdrücklich unterstützen. Die Nummern oben sind Hostnummern;
wenn ein Container nur `NVIDIA_VISIBLE_DEVICES=1` erhält, sieht er die RTX 5080
innerhalb des Containers üblicherweise als GPU 0.
## Rollen der dauerhaften Kerndienste
| Dienst | Rolle |
|---|---|
| `mike-ai-router` | Einzige OpenAI-kompatible Modelladresse; besitzt die Zustandsmaschine für Profile und Betriebsmodi. |
| `mike-ai-profile-controller` | Darf ausschließlich freigegebene, eindeutig markierte Worker starten und stoppen. |
| `mike-ai-llama-dashboard` | Telemetrie, Modusumschaltung und Download der portablen Backups. |
| `mike-ai-wireguard-gateway` | Veröffentlicht interne Dienste an der privaten Adresse `192.168.1.212`; keine öffentliche/LAN-Bindung. |
| `mike-ai-tts-gateway` | Stabile TTS-API, Textnormalisierung, Formatumwandlung und PCM-Streaming; enthält kein Ersatzmodell. |
| `mike-ai-whisper` | Dauerhafte CPU-Spracherkennung mit Whisper.cpp `ggml-small`. |
| `mike-ai-mcp-athena-operator` | Begrenzte Verwaltungsfunktionen für Agenten; kein allgemeiner Root-Ersatz. |
| `mike-ai-backup` | Lokales Schnellbackup; externe Disaster-Sicherung läuft zusätzlich über systemd-Timer. |
## LLM-Profile
Es läuft höchstens ein llama.cpp-Profil. Die Standardprofile nutzen
Qwen3.8-27B in Q4-Quantisierung.
| Profil | API-Name | Kontext | Vision |
|---|---|---:|---|
| Fast | `qwen-fast` | 76.800 | ja |
| Medium | `qwen-medium` | 160.000 | ja |
| Large | `qwen-large` | 192.000 | ja |
| Ultra | `qwen-ultra` | 262.144 | nein |
| Uncensored | `qwen-uncensored` | 80.000 | ja, eigener Projektor |
Die verbindlichen Parameter stehen in `config/profile-matrix.json`,
`router/router_profiles.json`, `platform/profiles/` und
`docs/STANDARD_PROFILE_MATRIX.md`. Diese Quellen dürfen sich nicht
widersprechen.
## Exklusive Betriebsmodi
Große GPU-Worker sind gegenseitig exklusiv. Der Router speichert
`mode`, `last_profile` und `return_profile` persistent. Beim Wechsel in einen
Spezialmodus werden LLM, Bildworker und Qwen3-TTS soweit nötig gestoppt; beim
Wechsel zu `llm` wird das zuvor gemerkte Profil wiederhergestellt.
| Modus | Worker / Modell | GPU-Nutzung | Oberfläche |
|---|---|---|---|
| `llm` | ein Qwen-Profil + Qwen3-TTS | profilabhängig beide GPUs; TTS RTX 3060 | Router `:8081` |
| Bildauftrag | FLUX.2 Klein 9B FP8 + Qwen3-8B NF4 | RTX 5080 + RTX 3060, transaktional | über Router |
| `music` | ACE-Step 1.5 XL-SFT | RTX 5080 | `:7862` original, `:7861` Community |
| `yue2` | YuE2-3B + Ladypoly `YuE2_WebUI` | RTX 5080 | `:8014` |
| `separation` | BS-RoFormer, Demucs, MossFormer2 | RTX 5080 | `:8007` |
| `voice` | OmniVoice | RTX 5080 | `:8008` |
| `voicechange` | X-VC + optional Resemble Enhance | RTX 5080 | `:8009` |
| `applio` | Applio/RVC | RTX 5080 | `:8011`, eigene UI `:8012` |
| `trellis` | TRELLIS.2 4B Q8 über trellis.cpp 0.6.0 | ausschließlich RTX 5080 | `:8013` |
YuE2 liegt unter `/opt/mike-ai/yue2-3b`, seine Gewichte unter
`/data/models/yue2` und Ergebnisse unter `/data/music/yue2`. Der Container
trägt `com.mike-ai.music-worker=yue2` und muss mit dem Alias `yue2-studio` am
externen Netz `mike-ai_frontend` hängen. YuE2 niemals außerhalb der
Router-Zustandsmaschine dauerhaft starten: Sonst bleibt sein VRAM belegt und
der nächste LLM- oder Separator-Start kann mit OOM scheitern.
TRELLIS liegt unter `/opt/mike-ai/trellis-studio`. Seine Q8-Gewichte liegen
unter `/data/models/trellis2-q8`, die Runtime und Ausgaben unter
`/data/trellis-studio`. Die Oberfläche liefert GLB. `1024 · cascade` ist der
Qualitätsstandard für die 16-GiB-RTX-5080; 1536 kann den VRAM überschreiten.
Ein 512er Ende-zu-Ende-Test erzeugte am 10.09.2026 in 54,2 Sekunden ein
gültiges 4,4-MB-GLB.
## Steuerbefehle und Status
Im Dashboard wird über die Modus-API geschaltet. Hermes kann dieselbe
Zustandsmaschine mit exakten Befehlen bedienen:
```text
/athena music
/athena stems
/athena voice
/athena voicechange
/athena applio
/athena 3d
/athena trellis
/athena llm
/athena status
```
Ein Moduswechsel ist asynchron. Eine angenommene Anfrage bedeutet noch nicht,
dass der Worker bereit ist. Immer warten, bis `GET /status` beziehungsweise das
Dashboard `phase: ready`, den richtigen `active`-Modus und einen gesunden
Worker meldet. Bei Fehlern nicht blind erneut starten, sondern `last_error`,
Containerstatus und Logs lesen.
## Netzwerkmodell
Anwendungscontainer veröffentlichen ihre Host-Ports nur auf `127.0.0.1` oder
gar nicht. Das WireGuard-Gateway sitzt im externen Docker-Netz
`mike-ai_frontend`, bindet die private WireGuard-Adresse `192.168.1.212` und
leitet mit `socat` auf Compose-Dienstnamen weiter.
Wichtige Regeln:
- Gateway und Anwendung **nicht** über `network_mode: container:...` koppeln.
- Ziel ist zum Beispiel `trellis-studio:8080`, niemals eine Container-IP.
- Der Zielcontainer muss im selben externen Frontend-Netz liegen.
- Beim Hinzufügen eines Ports den Proxy-Eintrag im Gateway, das Dashboard und
die Endpunkt-Dokumentation gemeinsam ergänzen.
- Ein Gateway-Recreate kann eine bestehende SSH-Verbindung unterbrechen. Nur
kontrolliert und mit automatisch verzögertem Wiederanlauf durchführen.
- Nach einem Recreate DNS-Auflösung, Listener, Ziel-Healthcheck und Zugriff
über den WireGuard-Pfad prüfen.
## Daten und Sicherung
| Pfad | Inhalt |
|---|---|
| `/opt/mike-ai` | Deployments, Compose-Projekte und lokale Quellstände |
| `/etc/mike-ai` | Konfiguration, Schlüssel und Tokens; geheim |
| `/data/models` | erneut ladbare Modellgewichte und Caches |
| `/data/voice` | Trainingsdaten, Checkpoints und trainierte Stimmen |
| `/data/music` | Musikprojekte und Ausgaben |
| `/data/audio` | Audio-Trennungen |
| `/data/trellis-studio` | trellis.cpp-Runtime und 3D-Ausgaben |
| `/data/llama-dashboard` | Telemetriehistorie |
| `/data/docker-backups` | lokale Schnellbackups |
Docker-Volumes: `mike-ai_router-state`, `mike-ai_router-images`,
`mike-ai_whisper-data`, `portainer_data`.
Das lokale Exportbackup läuft etwa alle fünf Stunden, das verschlüsselte
Disaster-Backup nachts. Ein Backup auf `/data` schützt nicht vor dem Ausfall
der Datenplatte. Details und alle drei Ausfallszenarien stehen in
`docs/RECOVERY.md`. **Aktuelle Lücke:** `/data/trellis-studio/output` ist im
ausgerollten Export- und Disaster-Backup noch nicht enthalten. Wichtige GLB-
Ausgaben daher zusätzlich extern sichern, bis die Backup-Skripte erweitert und
getestet wurden.
## Neuen GPU-Dienst korrekt hinzufügen
1. `docs/TESTED_MODELS.md` vollständig prüfen, damit kein verworfener Kandidat
erneut geladen wird.
2. Lizenz, Modellrevision, Runtime-Revision, VRAM, RAM, Ausgabeformat und
Hardwareunterstützung dokumentieren.
3. Eigenes Compose-Projekt oder klar abgegrenzten Kernservice anlegen. Image
und Upstream-Commit pinnen; nicht dauerhaft `latest` als einzige
Wiederherstellungsinformation verwenden.
4. Gewichte unter einem eindeutigen Verzeichnis in `/data/models` speichern,
veränderliche Ergebnisse separat unter `/data`.
5. `restart: "no"` für exklusive GPU-Worker verwenden. Dauerhafte UIs dürfen
laufen, dürfen aber im Leerlauf kein großes Modell laden.
6. Genau ein eindeutiges Label vergeben, zum Beispiel
`com.mike-ai.trellis-worker=trellis2-q8`. Der Controller muss bei null oder
mehreren Treffern absichtlich abbrechen.
7. Worker in **Controller, Router, Dashboard, Compose-Umgebung,
WireGuard-Proxy, Tests und Dokumentation** ergänzen.
8. Alle anderen exklusiven Worker sowohl beim Eintritt als auch beim Verlassen
des neuen Modus behandeln. Den Rückweg zum gespeicherten LLM-Profil testen.
9. Healthcheck-Werkzeuge tatsächlich im Image installieren. Ein Backendprozess
kann laufen, während ein fehlerhafter Healthcheck den Modus blockiert.
10. Bei Web-UIs korrekte MIME-Typen ausliefern. ES-Module benötigen
`application/javascript`, CSS `text/css`; Browsermodus muss denselben
Ursprung oder eine sauber konfigurierte API-Adresse verwenden.
11. Compose validieren, Syntax prüfen, nur den betroffenen Dienst bauen und
einen echten Ende-zu-Ende-Auftrag ausführen. Danach Rückschaltung testen.
12. Quellcode, Installer, Wiederaufbau und Dokumentation im selben Git-Stand
versionieren. Erst dann ist die Erweiterung wiederherstellbar.
## Dienst vollständig entfernen
1. Belegen, dass der Dienst nicht aktiv ist und keine laufende Arbeit besitzt.
2. Testergebnis und Ablehnungsgrund zuerst in `docs/TESTED_MODELS.md` sichern.
3. Routerbefehle, Zustandsfelder, Controller-Labelsuche, Dashboard-Schalter,
Proxy-Port, Compose-Projekt, Tests und Dokumentation entfernen.
4. Container und Image gezielt anhand exakter Namen entfernen.
5. Gewichte, Cache, Ausgaben und Volumes einzeln klassifizieren: reproduzierbar,
ersetzbar oder unersetzlich. Unersetzliche Daten sichern; keine Globs oder
pauschalen Prune-Befehle benutzen.
6. Prüfen, dass kein Labelduplikat, verwaister Proxy, unbenutztes Netz oder
verwaistes Volume übrig ist.
7. LLM-Modus wiederherstellen und einen Smoke-Test ausführen.
## Häufige Fehlerbilder
- **Controller meldet zwei Worker:** Während `docker compose up
--force-recreate` können alter und neuer Container kurz dasselbe Label
tragen. Recreate beenden lassen, danach exakt gelabelte Container prüfen und
erst dann den Modus erneut anfordern.
- **Webseite ist unformatiert und bleibt auf „connecting“:** MIME-Typen oder
Asset-Cache prüfen; nicht automatisch das KI-Backend beschuldigen.
- **Dashboard oder Port fehlt nach Recreate:** Listener im Gateway,
DNS-Auflösung des Dienstnamens und gemeinsames Frontend-Netz prüfen.
- **Worker gesund, Modus trotzdem fehlerhaft:** Routerzustand und
`last_error` können noch den vorherigen fehlgeschlagenen Übergang zeigen;
nach Beseitigung der Ursache Modus kontrolliert erneut anfordern.
- **VRAM scheinbar leer:** Manche Runtime lädt Gewichte erst beim ersten
Auftrag und gibt Speicher anschließend wieder frei. Ein Healthcheck allein
ist daher kein vollständiger GPU-Test.
- **Compose verwendet falsche Werte:** Der Kernstack benötigt
`--env-file /etc/mike-ai/stack.env`.
## Definition von „fertig“
Eine Änderung ist erst fertig, wenn sie im kanonischen Git-Stand liegt,
reproduzierbar gebaut werden kann, Compose/Syntax valide sind, der Dienst gesund
ist, ein echter kleiner Funktionsauftrag erfolgreich war, die Rückschaltung
funktioniert, Backup und WireGuard-Zugriff gesund geblieben sind und Commit
sowie Push erfolgt sind.
Weiterführend: `ATHENA.md`, `docs/ARCHITECTURE.md`,
`docs/OPERATING_MODES.md`, `docs/CONTAINER_INVENTORY.md`,
`docs/TESTED_MODELS.md` und `docs/RECOVERY.md`.
+67 -22
View File
@@ -37,11 +37,14 @@ if (( (8#$config_mode & 077) != 0 )); then
fi fi
# shellcheck disable=SC1090 # shellcheck disable=SC1090
source "$CONFIG" source "$CONFIG"
if [[ -n ${HF_TOKEN_FILE:-} && ! -r ${HF_TOKEN_FILE:-} && \
-r /etc/mike-ai/huggingface-token ]]; then
HF_TOKEN_FILE=/etc/mike-ai/huggingface-token
fi
required=(AI_HOSTNAME ADMIN_USER MODEL_DIR FAST_MODEL_FILE required=(AI_HOSTNAME ADMIN_USER MODEL_DIR FAST_MODEL_FILE
FAST_MODEL_URL FAST_MODEL_SHA256 MEDIUM_MODEL_FILE MEDIUM_MODEL_URL FAST_MODEL_URL FAST_MODEL_SHA256 MEDIUM_MODEL_FILE MEDIUM_MODEL_URL
MEDIUM_MODEL_SHA256 LARGE_MODEL_FILE LARGE_MODEL_URL LARGE_MODEL_SHA256 MEDIUM_MODEL_SHA256 LARGE_MODEL_FILE LARGE_MODEL_URL LARGE_MODEL_SHA256
BETA1_MODEL_FILE BETA1_MODEL_URL BETA1_MODEL_SHA256
ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256 ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256
UNCENSORED_MODEL_FILE UNCENSORED_MODEL_URL UNCENSORED_MODEL_SHA256 UNCENSORED_MODEL_FILE UNCENSORED_MODEL_URL UNCENSORED_MODEL_SHA256
UNCENSORED_PROJECTOR_FILE UNCENSORED_PROJECTOR_URL UNCENSORED_PROJECTOR_FILE UNCENSORED_PROJECTOR_URL
@@ -70,7 +73,7 @@ install_base_packages() {
apt-get update apt-get update
DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \ DEBIAN_FRONTEND=noninteractive apt-get install -y --no-install-recommends \
ca-certificates curl git gnupg jq openssl wireguard-tools iptables \ ca-certificates curl git gnupg jq openssl wireguard-tools iptables \
iproute2 pciutils rsync unattended-upgrades ethtool age iproute2 pciutils rsync unattended-upgrades ethtool age restic zstd
} }
setup_stable_network_name() { setup_stable_network_name() {
@@ -304,6 +307,19 @@ install_stack_files() {
printf '%s\n' "$source_head" >"$STACK_DIR/.mike-ai-source-commit" printf '%s\n' "$source_head" >"$STACK_DIR/.mike-ai-source-commit"
fi fi
install -d -m 0700 "$SECRETS_DIR" install -d -m 0700 "$SECRETS_DIR"
# Keep the exact host bootstrap inputs with the protected system
# configuration. This breaks the former recovery cycle in which a fresh
# host needed a lost /root-only install file before it could restore backup.
if [[ $(realpath "$CONFIG") != $(realpath -m "$SECRETS_DIR/install.env") ]]; then
install -m 0600 "$CONFIG" "$SECRETS_DIR/install.env"
else
chmod 0600 "$SECRETS_DIR/install.env"
fi
if [[ -n ${HF_TOKEN_FILE:-} && -r $HF_TOKEN_FILE && \
$(realpath "$HF_TOKEN_FILE") != $(realpath -m "$SECRETS_DIR/huggingface-token") ]]; then
install -m 0600 "$HF_TOKEN_FILE" "$SECRETS_DIR/huggingface-token"
HF_TOKEN_FILE=$SECRETS_DIR/huggingface-token
fi
[[ -s $SECRETS_DIR/router-api-key ]] || openssl rand -base64 48 >$SECRETS_DIR/router-api-key [[ -s $SECRETS_DIR/router-api-key ]] || openssl rand -base64 48 >$SECRETS_DIR/router-api-key
[[ -s $SECRETS_DIR/controller-token ]] || openssl rand -base64 48 >$SECRETS_DIR/controller-token [[ -s $SECRETS_DIR/controller-token ]] || openssl rand -base64 48 >$SECRETS_DIR/controller-token
chmod 0600 "$SECRETS_DIR"/* chmod 0600 "$SECRETS_DIR"/*
@@ -314,8 +330,6 @@ MODEL_DIR=$MODEL_DIR
WIREGUARD_CONFIG_FILE=${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf} WIREGUARD_CONFIG_FILE=${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athena.conf}
ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key) ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key)
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token) CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
PIPER_VOICE=${PIPER_VOICE:-de_DE-thorsten-high}
QWEN3_TTS_IMAGE=${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98} QWEN3_TTS_IMAGE=${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
QWEN3_TTS_CACHE_DIR=$QWEN3_TTS_CACHE_DIR QWEN3_TTS_CACHE_DIR=$QWEN3_TTS_CACHE_DIR
QWEN3_TTS_VOICES_DIR=$QWEN3_TTS_VOICES_DIR QWEN3_TTS_VOICES_DIR=$QWEN3_TTS_VOICES_DIR
@@ -323,7 +337,6 @@ QWEN3_TTS_GPU_DEVICE=${QWEN3_TTS_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d4
AI_DNS=${WG_DNS:-1.1.1.1} AI_DNS=${WG_DNS:-1.1.1.1}
FAST_MODEL_FILE=$FAST_MODEL_FILE FAST_MODEL_FILE=$FAST_MODEL_FILE
MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE MEDIUM_MODEL_FILE=$MEDIUM_MODEL_FILE
BETA1_MODEL_FILE=$BETA1_MODEL_FILE
LARGE_MODEL_FILE=$LARGE_MODEL_FILE LARGE_MODEL_FILE=$LARGE_MODEL_FILE
ULTRA_MODEL_FILE=$ULTRA_MODEL_FILE ULTRA_MODEL_FILE=$ULTRA_MODEL_FILE
UNCENSORED_MODEL_FILE=$UNCENSORED_MODEL_FILE UNCENSORED_MODEL_FILE=$UNCENSORED_MODEL_FILE
@@ -337,10 +350,6 @@ MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000}
MEDIUM_BATCH_SIZE=${MEDIUM_BATCH_SIZE:-2048} MEDIUM_BATCH_SIZE=${MEDIUM_BATCH_SIZE:-2048}
MEDIUM_UBATCH_SIZE=${MEDIUM_UBATCH_SIZE:-128} MEDIUM_UBATCH_SIZE=${MEDIUM_UBATCH_SIZE:-128}
MEDIUM_PARALLEL_SLOTS=${MEDIUM_PARALLEL_SLOTS:-1} MEDIUM_PARALLEL_SLOTS=${MEDIUM_PARALLEL_SLOTS:-1}
BETA1_CONTEXT=${BETA1_CONTEXT:-192000}
BETA1_BATCH_SIZE=${BETA1_BATCH_SIZE:-2048}
BETA1_UBATCH_SIZE=${BETA1_UBATCH_SIZE:-128}
BETA1_PARALLEL_SLOTS=${BETA1_PARALLEL_SLOTS:-1}
LARGE_CONTEXT=${LARGE_CONTEXT:-192000} LARGE_CONTEXT=${LARGE_CONTEXT:-192000}
LARGE_BATCH_SIZE=${LARGE_BATCH_SIZE:-2048} LARGE_BATCH_SIZE=${LARGE_BATCH_SIZE:-2048}
LARGE_UBATCH_SIZE=${LARGE_UBATCH_SIZE:-128} LARGE_UBATCH_SIZE=${LARGE_UBATCH_SIZE:-128}
@@ -356,7 +365,6 @@ UNCENSORED_PARALLEL_SLOTS=${UNCENSORED_PARALLEL_SLOTS:-1}
FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES} FAST_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES} MEDIUM_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
MEDIUM_TENSOR_SPLIT=${MEDIUM_TENSOR_SPLIT:-90,10} MEDIUM_TENSOR_SPLIT=${MEDIUM_TENSOR_SPLIT:-90,10}
BETA1_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
LARGE_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES} LARGE_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
LARGE_TENSOR_SPLIT=${LARGE_TENSOR_SPLIT:-86,14} LARGE_TENSOR_SPLIT=${LARGE_TENSOR_SPLIT:-86,14}
ULTRA_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES} ULTRA_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDARY_GPU_DEVICES}
@@ -365,7 +373,8 @@ UNCENSORED_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDAR
UNCENSORED_TENSOR_SPLIT=${UNCENSORED_TENSOR_SPLIT:-90,10} UNCENSORED_TENSOR_SPLIT=${UNCENSORED_TENSOR_SPLIT:-90,10}
UNCENSORED_MTP_MAX=${UNCENSORED_MTP_MAX:-2} UNCENSORED_MTP_MAX=${UNCENSORED_MTP_MAX:-2}
IMAGE_GPU_DEVICES=${IMAGE_GPU_DEVICES:-${TEXT_GPU_DEVICES:-0}} IMAGE_GPU_DEVICES=${IMAGE_GPU_DEVICES:-${TEXT_GPU_DEVICES:-0}}
FLUX_MODEL_DIR=${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B} FLUX_COMPONENT_DIR=${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}
FLUX_TRANSFORMER_DIR=${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}
LLAMA_THREADS=${LLAMA_THREADS:-6} LLAMA_THREADS=${LLAMA_THREADS:-6}
LLAMA_THREADS_BATCH=${LLAMA_THREADS_BATCH:-6} LLAMA_THREADS_BATCH=${LLAMA_THREADS_BATCH:-6}
LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-32768} LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-32768}
@@ -374,6 +383,29 @@ EOF
chmod 0600 $SECRETS_DIR/stack.env chmod 0600 $SECRETS_DIR/stack.env
} }
install_disaster_backup() {
log "Externes Disaster-Backup installieren"
install -m 0755 "$ROOT_DIR/platform/backup/athena-disaster-backup" \
/usr/local/sbin/athena-disaster-backup
install -m 0644 "$ROOT_DIR/platform/backup/athena-disaster-backup.service" \
/etc/systemd/system/athena-disaster-backup.service
install -m 0644 "$ROOT_DIR/platform/backup/athena-disaster-backup.timer" \
/etc/systemd/system/athena-disaster-backup.timer
install -m 0755 "$ROOT_DIR/platform/backup/athena-export-backup" \
/usr/local/sbin/athena-export-backup
install -m 0644 "$ROOT_DIR/platform/backup/athena-export-backup.service" \
/etc/systemd/system/athena-export-backup.service
install -m 0644 "$ROOT_DIR/platform/backup/athena-export-backup.timer" \
/etc/systemd/system/athena-export-backup.timer
if [[ ! -e $SECRETS_DIR/disaster-backup.env ]]; then
install -m 0600 "$ROOT_DIR/config/disaster-backup.env.example" \
"$SECRETS_DIR/disaster-backup.env.example"
fi
systemctl daemon-reload
systemctl enable --now athena-disaster-backup.timer
systemctl enable --now athena-export-backup.timer
}
download_one() { download_one() {
local relative=$1 url=$2 expected=$3 target="$MODEL_DIR/$1" local relative=$1 url=$2 expected=$3 target="$MODEL_DIR/$1"
install -d -m 0755 "$(dirname "$target")" install -d -m 0755 "$(dirname "$target")"
@@ -404,7 +436,6 @@ download_models() {
done <<EOF done <<EOF
$FAST_MODEL_FILE|$FAST_MODEL_URL|$FAST_MODEL_SHA256 $FAST_MODEL_FILE|$FAST_MODEL_URL|$FAST_MODEL_SHA256
$MEDIUM_MODEL_FILE|$MEDIUM_MODEL_URL|$MEDIUM_MODEL_SHA256 $MEDIUM_MODEL_FILE|$MEDIUM_MODEL_URL|$MEDIUM_MODEL_SHA256
$BETA1_MODEL_FILE|$BETA1_MODEL_URL|$BETA1_MODEL_SHA256
$LARGE_MODEL_FILE|$LARGE_MODEL_URL|$LARGE_MODEL_SHA256 $LARGE_MODEL_FILE|$LARGE_MODEL_URL|$LARGE_MODEL_SHA256
$ULTRA_MODEL_FILE|$ULTRA_MODEL_URL|$ULTRA_MODEL_SHA256 $ULTRA_MODEL_FILE|$ULTRA_MODEL_URL|$ULTRA_MODEL_SHA256
$UNCENSORED_MODEL_FILE|$UNCENSORED_MODEL_URL|$UNCENSORED_MODEL_SHA256 $UNCENSORED_MODEL_FILE|$UNCENSORED_MODEL_URL|$UNCENSORED_MODEL_SHA256
@@ -486,25 +517,38 @@ build_and_start() {
docker build --progress=plain --build-arg LLAMA_CPP_COMMIT="$commit" \ docker build --progress=plain --build-arg LLAMA_CPP_COMMIT="$commit" \
-f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local . -f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local .
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image build image-worker docker compose --env-file "$SECRETS_DIR/stack.env" --profile image build image-worker
if [[ ! -s ${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}/model_index.json ]]; then local flux_components=${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}
log "FLUX.2-klein-4B laden" local flux_transformer=${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}
install -d -m 0755 "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}" local hf_token_file=${HF_TOKEN_FILE:-/root/.cache/huggingface/token}
docker run --rm --entrypoint python \ if [[ ! -s $flux_components/model_index.json || \
-v "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/download" \ ! -s $flux_transformer/flux-2-klein-9b-fp8.safetensors ]]; then
[[ -r $hf_token_file ]] || die \
"Hugging-Face-Token fehlt: $hf_token_file (FLUX.2 Klein 9B ist gated)"
log "FLUX.2 Klein 9B Komponenten und FP8-Transformer laden"
install -d -m 0755 "$flux_components" "$flux_transformer"
docker run --rm --entrypoint /opt/image-venv/bin/python \
-e HF_TOKEN_PATH=/run/secrets/hf-token \
-v "$hf_token_file:/run/secrets/hf-token:ro" \
-v "$flux_components:/download" \
mike-ai/image-worker:local -c \ mike-ai/image-worker:local -c \
"from huggingface_hub import snapshot_download; snapshot_download('black-forest-labs/FLUX.2-klein-4B', revision='e7b7dc27f91deacad38e78976d1f2b499d76a294', local_dir='/download')" "from huggingface_hub import snapshot_download; snapshot_download('black-forest-labs/FLUX.2-klein-9B', revision='92196c8e11f7b6cf2b7493e037d8c5345c559216', local_dir='/download', allow_patterns=['model_index.json', 'scheduler/*', 'text_encoder/*', 'tokenizer/*', 'transformer/config.json', 'vae/*'])"
chmod -R a-w "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}" docker run --rm --entrypoint /opt/image-venv/bin/python \
-e HF_TOKEN_PATH=/run/secrets/hf-token \
-v "$hf_token_file:/run/secrets/hf-token:ro" \
-v "$flux_transformer:/download" \
mike-ai/image-worker:local -c \
"from huggingface_hub import snapshot_download; snapshot_download('black-forest-labs/FLUX.2-klein-9b-fp8', revision='902d9d510b51533e07729f19211414a3648b77d2', local_dir='/download', allow_patterns=['flux-2-klein-9b-fp8.safetensors', 'README.md', 'LICENSE.md'])"
chmod -R a-w "$flux_components" "$flux_transformer"
fi fi
# Creates the tools network and deploys the only host-bound MCP: Operator. # Creates the tools network and deploys the only host-bound MCP: Operator.
# Portable MCPs and Hermes live on Unraid and are restored through Appdata. # Portable MCPs and Hermes live on Unraid and are restored through Appdata.
"$STACK_DIR/platform/mcp/install-tools.sh" "$STACK_DIR/platform/mcp/install-tools.sh"
docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \ docker compose --env-file "$SECRETS_DIR/stack.env" --profile inference create \
llama-fast llama-medium llama-beta1 llama-large llama-ultra llama-uncensored llama-fast llama-medium llama-large llama-ultra llama-uncensored
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create image-worker docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create image-worker
docker volume inspect portainer_data >/dev/null 2>&1 || docker volume create portainer_data >/dev/null docker volume inspect portainer_data >/dev/null 2>&1 || docker volume create portainer_data >/dev/null
docker compose --env-file "$SECRETS_DIR/stack.env" stop llama-dashboard portainer
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \ docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
wireguard-gateway qwen3-tts piper tts-gateway profile-controller router llama-dashboard portainer backup wireguard-gateway qwen3-tts tts-gateway profile-controller router llama-dashboard portainer backup
if [[ ${WIREGUARD_MODE:-container} == container ]]; then if [[ ${WIREGUARD_MODE:-container} == container ]]; then
systemctl restart mike-ai-container-vpn-guard.service systemctl restart mike-ai-container-vpn-guard.service
fi fi
@@ -565,6 +609,7 @@ install_stack_files
download_models download_models
install_routing_guard install_routing_guard
build_and_start build_and_start
install_disaster_backup
log "Installation abgeschlossen" log "Installation abgeschlossen"
if [[ ${WIREGUARD_MODE:-container} == container ]]; then if [[ ${WIREGUARD_MODE:-container} == container ]]; then
@@ -0,0 +1,55 @@
# Hermes Athena image provider
Hermes backend plugin for the OpenAI-compatible image API exposed by the
Athena profile router. The router starts the local FLUX worker on demand,
unloads the active LLM and Qwen3-TTS, and restores both after generation.
Reference-image requests use FLUX for creative edits. No second Hermes
provider or desktop installation is required.
There is deliberately no photo-restoration model or restoration skill in this
provider. The former HYPIR experiment was removed after it redrew and smoothed
details instead of preserving the source faithfully. See
[`docs/IMAGE_RESTORATION.md`](../../docs/IMAGE_RESTORATION.md) for the recorded
decision and the isolated SeedVR2 comparison.
## Gateway installation
Install this directory on the Hermes gateway, not on each Desktop client:
```text
$HERMES_HOME/plugins/image_gen/athena-local/
__init__.py
plugin.yaml
```
Set these secrets or environment variables on the gateway:
```text
ATHENA_IMAGE_BASE_URL=http://192.168.1.212:8081/v1
ATHENA_IMAGE_API_KEY=<same API key accepted by the Athena router>
```
Then enable and select the provider:
```yaml
plugins:
enabled:
- image_gen/athena-local
image_gen:
provider: athena-local
model: FLUX.2-klein-9B-fp8-beta
max_parallel_requests: 1
```
Restart the Hermes gateway after changing plugin files or configuration. A
second computer connected to the same gateway needs no plugin installation.
Optional overrides:
- `ATHENA_IMAGE_MODEL` defaults to `FLUX.2-klein-9B-fp8-beta`.
- `ROUTER_API_KEY` is accepted as a migration fallback.
- An existing `HERMES_CUSTOM_192_168_1_212_8081_API_KEY` is accepted as the
final fallback, so an existing Athena chat-provider setup needs no duplicate
secret.
@@ -0,0 +1,263 @@
"""Hermes image generation/edit provider for the local Athena router.
The provider deliberately rejects public destinations. Prompts and generated
images may only travel to a loopback or private-network address.
"""
from __future__ import annotations
import base64
import ipaddress
import json
import os
import urllib.error
import urllib.parse
import urllib.request
from typing import Any, Dict, List, Optional
from agent.image_gen_provider import (
DEFAULT_ASPECT_RATIO,
ImageGenProvider,
error_response,
normalize_reference_images,
resolve_aspect_ratio,
save_b64_image,
success_response,
)
from agent.secret_scope import get_secret
_SIZES = {
"landscape": "1536x1024",
"square": "1024x1024",
"portrait": "1024x1536",
}
_DEFAULT_BASE_URL = "http://192.168.1.212:8081/v1"
_DEFAULT_MODEL = "FLUX.2-klein-9B-fp8-beta"
_MAX_IMAGE_BYTES = 20 * 1024 * 1024
def _base_url() -> str:
"""Use the dedicated image URL and never inherit an unrelated chat URL."""
override = os.environ.get("ATHENA_IMAGE_BASE_URL", "").strip()
return (override or _DEFAULT_BASE_URL).rstrip("/")
def _model() -> str:
return os.environ.get("ATHENA_IMAGE_MODEL", "").strip() or _DEFAULT_MODEL
def _api_key() -> str:
"""Prefer a scoped key; accept the existing router key for migration."""
return (
get_secret("ATHENA_IMAGE_API_KEY", "")
or get_secret("ROUTER_API_KEY", "")
or get_secret("HERMES_CUSTOM_192_168_1_212_8081_API_KEY", "")
or ""
).strip()
def _private_destination(url: str) -> bool:
"""Fail closed unless the configured endpoint is local/private."""
try:
parsed = urllib.parse.urlparse(url)
if parsed.scheme not in {"http", "https"} or not parsed.hostname:
return False
if parsed.hostname == "localhost":
return True
address = ipaddress.ip_address(parsed.hostname)
return address.is_private or address.is_loopback
except (ValueError, TypeError):
return False
def _load_private_image(ref: str) -> bytes:
"""Load a local/data/private-LAN image without contacting public hosts."""
ref = ref.strip()
lower = ref.lower()
if lower.startswith("data:image/"):
_, separator, payload = ref.partition(",")
if not separator:
raise ValueError("invalid image data URI")
data = base64.b64decode(payload, validate=True)
elif lower.startswith(("http://", "https://")):
if not _private_destination(ref):
raise ValueError("public reference-image URLs are blocked")
request = urllib.request.Request(
ref, headers={"User-Agent": "Hermes-Athena-Image/2.0"})
with urllib.request.urlopen(request, timeout=60) as response:
data = response.read(_MAX_IMAGE_BYTES + 1)
else:
from agent.file_safety import raise_if_read_blocked
raise_if_read_blocked(ref)
with open(ref, "rb") as image_file:
data = image_file.read(_MAX_IMAGE_BYTES + 1)
if not data or len(data) > _MAX_IMAGE_BYTES:
raise ValueError("reference image is empty or exceeds 20 MiB")
return data
class AthenaLocalImageProvider(ImageGenProvider):
@property
def name(self) -> str:
return "athena-local"
@property
def display_name(self) -> str:
return "Athena Local (FLUX.2 Klein)"
def is_available(self) -> bool:
return bool(_api_key()) and _private_destination(_base_url())
def list_models(self) -> List[Dict[str, Any]]:
return [{
"id": _model(),
"display": "FLUX.2 Klein 9B FP8 Beta on Athena",
"speed": "local",
"strengths": "Private local generation and multi-reference editing",
"price": "local / no cloud",
}]
def default_model(self) -> Optional[str]:
return _model()
def capabilities(self) -> Dict[str, Any]:
return {"modalities": ["text", "image"], "max_reference_images": 3}
def get_setup_schema(self) -> Dict[str, Any]:
return {
"name": "Athena Local (FLUX.2 Klein)",
"badge": "local",
"tag": "Private image generation on Athena; public endpoints are rejected",
"env_vars": [
{"key": "ATHENA_IMAGE_API_KEY", "prompt": "Athena router API key"},
{
"key": "ATHENA_IMAGE_BASE_URL",
"prompt": "Athena image API base URL",
"default": _DEFAULT_BASE_URL,
},
],
}
def generate(
self,
prompt: str,
aspect_ratio: str = DEFAULT_ASPECT_RATIO,
*,
image_url: Optional[str] = None,
reference_image_urls: Optional[List[str]] = None,
**kwargs: Any,
) -> Dict[str, Any]:
clean_prompt = (prompt or "").strip()
aspect = resolve_aspect_ratio(aspect_ratio)
base_url = _base_url()
if not clean_prompt:
return error_response(
error="Prompt is required.", error_type="invalid_argument",
provider=self.name, aspect_ratio=aspect)
if not _private_destination(base_url):
return error_response(
error=("Athena image endpoint is not a private-network "
"destination; request blocked."),
error_type="unsafe_destination", provider=self.name,
prompt=clean_prompt, aspect_ratio=aspect)
model = _model()
api_key = _api_key()
if not api_key:
return error_response(
error=("No Athena router key is configured. Set "
"ATHENA_IMAGE_API_KEY or reuse "
"HERMES_CUSTOM_192_168_1_212_8081_API_KEY."),
error_type="auth_required", provider=self.name, model=model,
prompt=clean_prompt, aspect_ratio=aspect)
sources: List[str] = []
if isinstance(image_url, str) and image_url.strip():
sources.append(image_url.strip())
sources.extend(normalize_reference_images(reference_image_urls) or [])
# Hermes may expose the primary upload through both ``image_url`` and
# ``reference_image_urls``. Preserve order while removing duplicates.
sources = list(dict.fromkeys(sources))[:4]
model = _model()
try:
encoded_sources = [
base64.b64encode(_load_private_image(source)).decode("ascii")
for source in sources
]
except Exception as exc:
return error_response(
error=f"Reference image could not be loaded locally: {exc}",
error_type="io_error", provider=self.name, model=model,
prompt=clean_prompt, aspect_ratio=aspect)
request_data = {
"model": model,
"prompt": clean_prompt,
"size": _SIZES[aspect],
"n": 1,
"quality": "standard",
"steps": 4,
"guidance": 1.0,
"response_format": "b64_json",
}
endpoint = "generations"
if encoded_sources:
endpoint = "edits"
request_data["image_b64"] = encoded_sources[0]
request_data["reference_images_b64"] = encoded_sources[1:]
request = urllib.request.Request(
f"{base_url}/images/{endpoint}",
data=json.dumps(request_data).encode("utf-8"), method="POST",
headers={
"Authorization": f"Bearer {api_key}",
"Content-Type": "application/json",
"Accept": "application/json",
})
try:
with urllib.request.urlopen(request, timeout=900) as response:
result = json.load(response)
except urllib.error.HTTPError as exc:
try:
detail = exc.read(4096).decode("utf-8", errors="replace")
except Exception:
detail = ""
return error_response(
error=f"Athena image request failed (HTTP {exc.code}): {detail[:500]}",
error_type="api_error", provider=self.name, model=model,
prompt=clean_prompt, aspect_ratio=aspect)
except (OSError, TimeoutError, ValueError, json.JSONDecodeError) as exc:
return error_response(
error=f"Athena image request failed: {exc}",
error_type="connection_error", provider=self.name, model=model,
prompt=clean_prompt, aspect_ratio=aspect)
items = result.get("data") if isinstance(result, dict) else None
first = items[0] if isinstance(items, list) and items else None
b64_data = first.get("b64_json") if isinstance(first, dict) else None
if not isinstance(b64_data, str) or not b64_data:
return error_response(
error="Athena returned no image data.",
error_type="empty_response", provider=self.name, model=model,
prompt=clean_prompt, aspect_ratio=aspect)
try:
saved = save_b64_image(b64_data, prefix="athena_flux2")
except Exception as exc:
return error_response(
error=f"Generated image could not be saved: {exc}",
error_type="io_error", provider=self.name, model=model,
prompt=clean_prompt, aspect_ratio=aspect)
return success_response(
image=str(saved), model=model, prompt=clean_prompt,
aspect_ratio=aspect, provider=self.name,
modality="image" if encoded_sources else "text",
extra={"size": _SIZES[aspect], "local_only": True,
"reference_images": len(encoded_sources)})
def register(ctx) -> None:
ctx.register_image_gen_provider(AthenaLocalImageProvider())
@@ -0,0 +1,7 @@
name: athena-local
version: 2.2.0
description: "Local-only FLUX.2 Klein 9B FP8 beta generation and editing through Athena."
author: Michael
kind: backend
requires_env:
- HERMES_CUSTOM_192_168_1_212_8081_API_KEY
@@ -0,0 +1,23 @@
# Hermes Qwen3-TTS PCM streaming adapter
This optional Hermes backend plugin uses Athena's native
`/v1/audio/speech/pcm-stream` route. It starts playback while Qwen3-TTS is
still synthesizing the current sentence instead of waiting for a complete
audio file.
Install this directory as `${HERMES_HOME}/plugins/qwen3-stream`, enable the
plugin and set:
```yaml
tts:
provider: qwen3-stream
streaming:
provider: qwen3-stream
```
The adapter reuses `tts.openai.base_url`, `tts.openai.api_key`, model, voice
and language unless an explicit `tts.qwen3-stream` section overrides them.
This avoids copying the Athena credential into another file.
Rollback is immediate: restore `tts.provider` and `tts.streaming.provider` to
`openai`, disable the plugin and restart the Hermes gateway.
@@ -0,0 +1,153 @@
from __future__ import annotations
from pathlib import Path
from typing import Any, Dict, Iterator, List, Optional
import requests
from agent.tts_provider import TTSProvider
from tools.tool_backend_helpers import resolve_openai_audio_api_key
from tools.tts_streaming import StreamingTTSProvider, register as register_streamer
from tools.tts_tool import _load_tts_config
NAME = "qwen3-stream"
SAMPLE_RATE = 24000
def _settings() -> Dict[str, Any]:
config = _load_tts_config()
own = dict(config.get(NAME) or {})
fallback = dict(config.get("openai") or {})
own.setdefault("base_url", fallback.get("base_url", ""))
own.setdefault("api_key", fallback.get("api_key", ""))
own.setdefault("model", fallback.get("model", "tts-1"))
own.setdefault("voice", fallback.get("voice", "alloy"))
own.setdefault("language", fallback.get("language", "German"))
own.setdefault("chunk_size", 4)
return own
def _url(path: str, section: Optional[Dict[str, Any]] = None) -> str:
cfg = section or _settings()
base = str(cfg.get("base_url") or "").rstrip("/")
if not base:
raise RuntimeError("tts.qwen3-stream.base_url is not configured")
if not base.endswith("/v1"):
base += "/v1"
return base + path
def _headers(section: Optional[Dict[str, Any]] = None) -> Dict[str, str]:
cfg = section or _settings()
key = str(cfg.get("api_key") or resolve_openai_audio_api_key() or "").strip()
headers = {"Accept": "application/octet-stream"}
if key:
headers["Authorization"] = f"Bearer {key}"
return headers
def _payload(text: str, section: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
cfg = section or _settings()
payload: Dict[str, Any] = {
"input": text,
"model": cfg.get("model") or "tts-1",
"voice": cfg.get("voice") or "alloy",
"language": cfg.get("language") or "German",
}
instruct = str(cfg.get("instruct") or "").strip()
if instruct:
payload["instruct"] = instruct
return payload
class Qwen3PCMStreamer(StreamingTTSProvider):
sample_rate = SAMPLE_RATE
channels = 1
sample_width = 2
@staticmethod
def available() -> bool:
try:
return bool(_settings().get("base_url"))
except Exception:
return False
def stream(self, text: str) -> Iterator[bytes]:
cfg = dict(_settings())
cfg.update(self.section or {})
payload = _payload(text, cfg)
payload["chunk_size"] = max(1, int(cfg.get("chunk_size", 4)))
with requests.post(
_url("/audio/speech/pcm-stream", cfg),
json=payload,
headers=_headers(cfg),
stream=True,
timeout=(5, 120),
) as response:
response.raise_for_status()
pending = b""
# An explicit read size prevents urllib3 from buffering the
# unknown-length response until connection close.
for chunk in response.iter_content(chunk_size=4096):
if not chunk:
continue
data = pending + chunk
even = len(data) & ~1
if even:
yield data[:even]
pending = data[even:]
class Qwen3TTSProvider(TTSProvider):
@property
def name(self) -> str:
return NAME
@property
def display_name(self) -> str:
return "Athena Qwen3-TTS Streaming"
def is_available(self) -> bool:
return Qwen3PCMStreamer.available()
def list_voices(self) -> List[Dict[str, Any]]:
voice = str(_settings().get("voice") or "alloy")
return [{"id": voice, "display": voice, "language": "de"}]
def synthesize(
self,
text: str,
output_path: str,
*,
voice: Optional[str] = None,
model: Optional[str] = None,
speed: Optional[float] = None,
format: str = "mp3",
**extra: Any,
) -> str:
cfg = _settings()
payload = _payload(text, cfg)
payload["response_format"] = format
if voice:
payload["voice"] = voice
if model:
payload["model"] = model
if speed is not None:
payload["speed"] = speed
response = requests.post(
_url("/audio/speech", cfg),
json=payload,
headers=_headers(cfg),
timeout=(5, 120),
)
response.raise_for_status()
Path(output_path).write_bytes(response.content)
return output_path
register_streamer(NAME)(Qwen3PCMStreamer)
def register(ctx) -> None:
ctx.register_tts_provider(Qwen3TTSProvider())
@@ -0,0 +1,5 @@
name: qwen3-stream
version: 0.1.1
description: Native PCM streaming adapter for the local Athena Qwen3-TTS service
author: Mike AI local stack
kind: backend
+1 -1
View File
@@ -7,7 +7,7 @@ Athena speech stack:
2. Athena Whisper transcribes it, 2. Athena Whisper transcribes it,
3. OpenClaw's normal agent-consult path answers with its configured model and 3. OpenClaw's normal agent-consult path answers with its configured model and
tools, tools,
4. Athena XTTS/Piper returns PCM audio to the Talk client. 4. Athena Qwen3-TTS returns PCM audio to the Talk client.
Long replies are synthesized incrementally. The first short phrase starts Long replies are synthesized incrementally. The first short phrase starts
playing as soon as it is ready while the next phrase is generated in parallel. playing as soon as it is ready while the next phrase is generated in parallel.
+1 -1
View File
@@ -324,7 +324,7 @@ export default definePluginEntry({
label: "Athena Local Talk", label: "Athena Local Talk",
aliases: ["athena", "local-athena"], aliases: ["athena", "local-athena"],
defaultModel: "athena-local", defaultModel: "athena-local",
voices: ["alloy", "claribel"], voices: ["alloy"],
autoSelectOrder: 1, autoSelectOrder: 1,
capabilities: { capabilities: {
transports: ["gateway-relay"], transports: ["gateway-relay"],
+2 -2
View File
@@ -307,7 +307,7 @@ class AthenaTalkBridge {
const response = await fetch(`${this.cfg.baseUrl}/audio/speech`, { const response = await fetch(`${this.cfg.baseUrl}/audio/speech`, {
method: "POST", method: "POST",
headers: { "Content-Type": "application/json", ...this.authHeaders() }, headers: { "Content-Type": "application/json", ...this.authHeaders() },
// Let Athena select its currently active local backend (XTTS or Piper). // Athena normalizes the request and serves it through Qwen3-TTS.
body: JSON.stringify({ voice: this.cfg.voice, input: text, response_format: "wav" }), body: JSON.stringify({ voice: this.cfg.voice, input: text, response_format: "wav" }),
signal, signal,
}); });
@@ -330,7 +330,7 @@ export default definePluginEntry({
label: "Athena Local Talk", label: "Athena Local Talk",
aliases: ["athena", "local-athena"], aliases: ["athena", "local-athena"],
defaultModel: "athena-local", defaultModel: "athena-local",
voices: ["alloy", "claribel"], voices: ["alloy"],
autoSelectOrder: 1, autoSelectOrder: 1,
capabilities: { capabilities: {
transports: ["gateway-relay"], transports: ["gateway-relay"],
+1 -15
View File
@@ -51,24 +51,10 @@ case "$command" in
shift shift
[[ $# -eq 0 ]] || { echo "core akzeptiert keine weiteren Services" >&2; exit 2; } [[ $# -eq 0 ]] || { echo "core akzeptiert keine weiteren Services" >&2; exit 2; }
run "$ROOT_DIR/platform/mcp/install-tools.sh" run "$ROOT_DIR/platform/mcp/install-tools.sh"
# Release the old gateway namespace (and its fixed network addresses)
# before replacing its owner. Otherwise Docker can strand the host with
# the old namespace still held by these two consumers.
run "${compose[@]}" stop llama-dashboard portainer
run "${compose[@]}" up -d --build \ run "${compose[@]}" up -d --build \
wireguard-gateway piper xtts tts-gateway profile-controller router llama-dashboard portainer backup wireguard-gateway qwen3-tts tts-gateway profile-controller router llama-dashboard portainer backup
else else
rebind_gateway=false
for service in "$@"; do
[[ $service == wireguard-gateway ]] && rebind_gateway=true
done
if [[ $rebind_gateway == true ]]; then
run "${compose[@]}" stop llama-dashboard portainer
fi
run "${compose[@]}" up -d --build --no-deps "$@" run "${compose[@]}" up -d --build --no-deps "$@"
if [[ $rebind_gateway == true ]]; then
run "${compose[@]}" up -d --no-deps --force-recreate llama-dashboard portainer
fi
fi fi
;; ;;
purge-legacy) purge-legacy)
+78
View File
@@ -0,0 +1,78 @@
#!/usr/bin/env bash
# Encrypted off-host backup for data-disk and total-loss recovery.
set -Eeuo pipefail
umask 077
CONFIG=${DISASTER_BACKUP_CONFIG:-/etc/mike-ai/disaster-backup.env}
STATE=/var/lib/mike-ai-disaster-backup
log() { printf '\n==> %s\n' "$*"; }
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
[[ -r $CONFIG ]] || die "Konfiguration fehlt: $CONFIG"
# shellcheck disable=SC1090
source "$CONFIG"
[[ ${DISASTER_BACKUP_ENABLED:-false} == true ]] || die \
"Externes Backup ist noch nicht freigeschaltet (DISASTER_BACKUP_ENABLED=true)."
[[ -n ${RESTIC_REPOSITORY:-} ]] || die "RESTIC_REPOSITORY fehlt."
if [[ -n ${RESTIC_REQUIRE_MOUNT:-} ]]; then
mountpoint -q "$RESTIC_REQUIRE_MOUNT" || die \
"Externes Backupziel ist nicht eingehängt: $RESTIC_REQUIRE_MOUNT"
fi
[[ -n ${RESTIC_PASSWORD_FILE:-} && -r $RESTIC_PASSWORD_FILE ]] || die \
"RESTIC_PASSWORD_FILE fehlt oder ist nicht lesbar."
command -v restic >/dev/null || die "restic ist nicht installiert."
command -v docker >/dev/null || die "Docker ist nicht installiert."
exec 9>/run/lock/athena-disaster-backup.lock
flock -n 9 || die "Ein Disaster-Backup läuft bereits."
install -d -m 0700 "$STATE/latest"
log "Konsistentes Docker-Schnellbackup erzeugen"
docker inspect mike-ai-backup >/dev/null 2>&1 || die "mike-ai-backup fehlt."
docker exec mike-ai-backup backup
latest=$(readlink -f /data/docker-backups/athena-latest.tar.gz)
[[ -s $latest ]] || die "Lokales Docker-Backup wurde nicht erzeugt."
gzip -t "$latest" || die "Lokales Docker-Backup ist beschädigt."
install -m 0600 "$latest" "$STATE/latest/docker-state.tar.gz"
sha256sum "$STATE/latest/docker-state.tar.gz" >"$STATE/latest/docker-state.tar.gz.sha256"
log "Wiederaufbau-Metadaten erfassen"
{
printf 'created_utc=%s\n' "$(date -u +%FT%TZ)"
printf 'hostname=%s\n' "$(hostname)"
printf 'source_commit=%s\n' "$(git -C /opt/mike-ai/stack rev-parse HEAD 2>/dev/null || printf unknown)"
findmnt -rn -o SOURCE,UUID,FSTYPE,TARGET / /data 2>/dev/null || true
} >"$STATE/latest/manifest.txt"
find /data/models -type f -printf '%P\t%s\n' 2>/dev/null | sort \
>"$STATE/latest/model-manifest.tsv"
docker ps -a --format '{{.Names}}\t{{.Image}}\t{{.Status}}' \
>"$STATE/latest/container-manifest.tsv"
paths=(/etc/mike-ai /opt/mike-ai "$STATE/latest")
for path in \
/data/voice /data/music /data/audio /data/llama-dashboard \
/data/mike-ai-operator /data/benchmarks /data/model-benchmarks \
/data/image-comparison /data/backups /data/deploy-backups; do
[[ ! -e $path ]] || paths+=("$path")
done
tag=${RESTIC_TAG:-athena-disaster}
log "Verschlüsseltes externes Backup schreiben"
if ! restic snapshots >/dev/null 2>&1; then
log "Neues Restic-Repository initialisieren"
restic init
fi
restic backup --tag "$tag" "${paths[@]}"
log "Aufbewahrung anwenden"
restic forget --tag "$tag" \
--keep-daily "${RESTIC_KEEP_DAILY:-14}" \
--keep-weekly "${RESTIC_KEEP_WEEKLY:-8}" \
--keep-monthly "${RESTIC_KEEP_MONTHLY:-12}" --prune
log "Letzten Snapshot verifizieren"
restic snapshots --tag "$tag" --latest 1
restic check
printf 'ATHENA_DISASTER_BACKUP_OK\n'
@@ -0,0 +1,12 @@
[Unit]
Description=Encrypted off-host disaster backup for Athena
After=docker.service network-online.target
Wants=network-online.target
ConditionPathExists=/etc/mike-ai/disaster-backup.env
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/athena-disaster-backup
Nice=10
IOSchedulingClass=best-effort
IOSchedulingPriority=7
@@ -0,0 +1,11 @@
[Unit]
Description=Nightly Athena off-host disaster backup
[Timer]
OnCalendar=*-*-* 03:15:00
Persistent=true
RandomizedDelaySec=30m
Unit=athena-disaster-backup.service
[Install]
WantedBy=timers.target
+82
View File
@@ -0,0 +1,82 @@
#!/usr/bin/env bash
# Build a browser-downloadable, encrypted archive of irreplaceable Athena data.
set -Eeuo pipefail
umask 077
OUTPUT_DIR=${ATHENA_EXPORT_DIR:-/data/emergency-backups}
RECIPIENT_FILE=${ATHENA_AGE_RECIPIENT_FILE:-/etc/mike-ai/recovery.age-recipient}
STATE=/var/lib/mike-ai-disaster-backup/latest
KEEP=${ATHENA_EXPORT_KEEP:-5}
log() { printf '\n==> %s\n' "$*"; }
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
[[ -s $RECIPIENT_FILE ]] || die "Age-Empfänger fehlt: $RECIPIENT_FILE"
[[ $KEEP =~ ^[1-9][0-9]*$ ]] || die "ATHENA_EXPORT_KEEP muss positiv sein."
for command in age zstd tar docker sha256sum flock; do
command -v "$command" >/dev/null || die "$command fehlt."
done
exec 9>/run/lock/athena-export-backup.lock
flock -n 9 || die "Ein exportierbares Backup läuft bereits."
install -d -m 0755 "$OUTPUT_DIR"
install -d -m 0700 "$STATE"
log "Aktuellen Docker-Zustand sichern"
docker exec mike-ai-backup backup
latest=$(readlink -f /data/docker-backups/athena-latest.tar.gz)
[[ -s $latest ]] || die "Docker-Zustandsbackup fehlt."
gzip -t "$latest" || die "Docker-Zustandsbackup ist beschädigt."
install -m 0600 "$latest" "$STATE/docker-state.tar.gz"
stamp=$(date -u +%Y-%m-%dT%H-%M-%SZ)
name="athena-portable-$stamp.tar.zst.age"
partial="$OUTPUT_DIR/.$name.partial"
target="$OUTPUT_DIR/$name"
list=$(mktemp /tmp/athena-export-list.XXXXXX)
trap 'rm -f "$list" "$partial"' EXIT
add_path() {
local path=${1#/}
[[ ! -e /$path ]] || printf '%s\0' "$path" >>"$list"
}
# Reproducible model/HF caches are deliberately omitted. Everything below is
# either a host configuration, project source, user input or generated result.
add_path /etc/mike-ai
add_path /opt/mike-ai
add_path /var/lib/mike-ai-disaster-backup/latest
add_path /data/voice/applio/logs
add_path /data/voice/applio/datasets
add_path /data/voice/applio/config.json
add_path /data/voice/omnivoice/output
add_path /data/voice/xvc/output
add_path /data/voice/studio
add_path /data/music
add_path /data/audio
add_path /data/llama-dashboard
add_path /data/mike-ai-operator
add_path /data/benchmarks
add_path /data/model-benchmarks
add_path /data/image-comparison
add_path /data/backups
add_path /data/deploy-backups
log "Portables, verschlüsseltes Backup erzeugen"
tar --create --numeric-owner --acls --xattrs -C / --null --files-from="$list" \
| zstd -T0 -3 \
| age -R "$RECIPIENT_FILE" -o "$partial"
chmod 0644 "$partial"
mv "$partial" "$target"
sha256sum "$target" >"$target.sha256"
chmod 0644 "$target.sha256"
log "Nur die letzten $KEEP Generationen behalten"
mapfile -t old < <(find "$OUTPUT_DIR" -maxdepth 1 -type f \
-name 'athena-portable-*.tar.zst.age' -printf '%T@ %p\n' | sort -rn | tail -n +$((KEEP + 1)) | cut -d' ' -f2-)
for archive in "${old[@]}"; do
rm -f -- "$archive" "$archive.sha256"
done
printf 'ATHENA_EXPORT_BACKUP_OK file=%s bytes=%s\n' "$target" "$(stat -c %s "$target")"
@@ -0,0 +1,12 @@
[Unit]
Description=Create encrypted downloadable Athena recovery package
After=docker.service
Requires=docker.service
ConditionPathExists=/etc/mike-ai/recovery.age-recipient
[Service]
Type=oneshot
ExecStart=/usr/local/sbin/athena-export-backup
Nice=10
IOSchedulingClass=best-effort
IOSchedulingPriority=7
@@ -0,0 +1,12 @@
[Unit]
Description=Create an Athena recovery package every five hours
[Timer]
OnBootSec=45m
OnUnitActiveSec=5h
Persistent=true
RandomizedDelaySec=10m
Unit=athena-export-backup.service
[Install]
WantedBy=timers.target
+3 -1
View File
@@ -13,9 +13,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends python3.12-venv
"transformers==${TRANSFORMERS_VERSION}" \ "transformers==${TRANSFORMERS_VERSION}" \
"accelerate==${ACCELERATE_VERSION}" \ "accelerate==${ACCELERATE_VERSION}" \
"huggingface-hub==${HF_HUB_VERSION}" \ "huggingface-hub==${HF_HUB_VERSION}" \
"nvidia-modelopt==0.46.0" bitsandbytes \
sentencepiece protobuf safetensors pillow && \ sentencepiece protobuf safetensors pillow && \
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker
COPY image_worker.py /app/image_worker.py COPY image_worker.py /app/image_worker.py
COPY image_worker_9b.py /app/image_worker_9b.py
USER 10002:10002 USER 10002:10002
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker.py"] ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker_9b.py"]
@@ -0,0 +1,282 @@
#!/usr/bin/env python3
"""Private FLUX.2 Klein 9B FP8 beta worker for Athena's two GPUs.
The FP8 diffusion transformer runs on the RTX 5080. A Qwen3-8B NF4 text
encoder runs on the RTX 3060 while the profile controller temporarily pauses
Qwen3-TTS. The transformer and encoder are released before VAE decoding so
the 1024px decoder has sufficient workspace on the RTX 5080.
"""
from __future__ import annotations
import gc
import json
import os
import signal
import time
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
from pathlib import Path
from types import MethodType
HOST = os.environ.get("WORKER_HOST", "0.0.0.0")
PORT = int(os.environ.get("WORKER_PORT", "8086"))
TOKEN = os.environ.get("WORKER_TOKEN", "").strip()
COMPONENT_DIR = os.environ.get("FLUX_COMPONENT_DIR", "/models/components")
TRANSFORMER_FILE = os.environ.get(
"FLUX_TRANSFORMER_FILE", "/models/fp8/flux-2-klein-9b-fp8.safetensors")
OUTPUT_DIR = Path(os.environ.get("IMAGE_DIR", "/data/images")).resolve()
ACTIVE = False
os.environ.setdefault("DIFFUSERS_VERBOSITY", "error")
os.environ.setdefault("TRANSFORMERS_VERBOSITY", "error")
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
if len(TOKEN) < 32:
raise RuntimeError("WORKER_TOKEN is missing or too short")
signal.signal(signal.SIGTERM, lambda *_: os._exit(0))
def _devices(torch):
if torch.cuda.device_count() != 2:
raise RuntimeError("FLUX 9B beta requires exactly two visible CUDA GPUs")
totals = {i: torch.cuda.get_device_properties(i).total_memory
for i in range(torch.cuda.device_count())}
transformer_index = max(totals, key=totals.get)
encoder_index = min(totals, key=totals.get)
return (transformer_index, encoder_index,
torch.device(f"cuda:{transformer_index}"),
torch.device(f"cuda:{encoder_index}"))
def _install_fp8_converter():
import diffusers.loaders.single_file_model as single_file_model
original = single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"]
scales = {}
double_map = {
"img_attn.proj": "attn.to_out.0",
"img_mlp.0": "ff.linear_in",
"img_mlp.2": "ff.linear_out",
"txt_attn.proj": "attn.to_add_out",
"txt_mlp.0": "ff_context.linear_in",
"txt_mlp.2": "ff_context.linear_out",
}
single_map = {
"linear1": "attn.to_qkv_mlp_proj",
"linear2": "attn.to_out",
}
def record(key, value):
parts = key.split(".")
scale_name, block = parts[-1], parts[1]
within = ".".join(parts[2:-1])
if parts[0] == "double_blocks":
if within == "img_attn.qkv":
targets = ("attn.to_q", "attn.to_k", "attn.to_v")
elif within == "txt_attn.qkv":
targets = ("attn.add_q_proj", "attn.add_k_proj",
"attn.add_v_proj")
else:
targets = (double_map[within],)
prefix = f"transformer_blocks.{block}"
elif parts[0] == "single_blocks":
targets = (single_map[within],)
prefix = f"single_transformer_blocks.{block}"
else:
raise ValueError(f"unexpected FP8 scale key: {key}")
for target in targets:
scales.setdefault(f"{prefix}.{target}", {})[scale_name] = value.clone()
def convert(checkpoint, **kwargs):
scales.clear()
for key in list(checkpoint):
if key.endswith((".input_scale", ".weight_scale")):
record(key, checkpoint.pop(key))
return original(checkpoint=checkpoint, **kwargs)
single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"] = convert
return scales
def _fp8_forward(torch, module, inputs):
shape = inputs.shape
input_fp8 = ((inputs / module._fp8_input_scale)
.clamp(torch.finfo(torch.float8_e4m3fn).min,
torch.finfo(torch.float8_e4m3fn).max)
.to(torch.float8_e4m3fn).reshape(-1, shape[-1]))
output = torch._scaled_mm(
input_fp8,
module.weight.reshape(-1, module.weight.shape[-1]).t(),
scale_a=module._fp8_input_scale,
scale_b=module._fp8_weight_scale,
bias=module.bias,
out_dtype=inputs.dtype,
use_fast_accum=True,
)
return output.reshape(*shape[:-1], output.shape[-1])
def generate(data: dict) -> dict:
global ACTIVE
import torch
from diffusers import (Flux2KleinPipeline, Flux2Transformer2DModel,
NVIDIAModelOptConfig)
from modelopt.torch.opt import enable_huggingface_checkpointing
from modelopt.torch.quantization.config import FP8_DEFAULT_CFG
from PIL import Image
from transformers import BitsAndBytesConfig, Qwen3ForCausalLM
prompt, filename = data.get("prompt"), data.get("filename")
if not isinstance(prompt, str) or not prompt.strip() or len(prompt) > 8000:
raise ValueError("invalid prompt")
if (not isinstance(filename, str) or Path(filename).name != filename
or not filename.endswith(".png")):
raise ValueError("invalid filename")
width, height = int(data.get("width", 1024)), int(data.get("height", 1024))
if (width, height) != (1024, 1024):
raise ValueError("FLUX 9B beta currently supports only 1024x1024")
if int(data.get("steps", 4)) != 4 or float(data.get("guidance", 1.0)) != 1.0:
raise ValueError("FLUX 9B beta requires steps=4 and guidance=1.0")
source_files = data.get("source_files") or []
if not isinstance(source_files, list) or len(source_files) > 4:
raise ValueError("invalid source image list")
source_images = []
for name in source_files:
if not isinstance(name, str) or Path(name).name != name:
raise ValueError("invalid source image filename")
source = (OUTPUT_DIR / name).resolve()
if source.parent != OUTPUT_DIR or not source.is_file():
raise ValueError("source image not found")
with Image.open(source) as opened:
source_images.append(opened.convert("RGB"))
started = time.monotonic()
ACTIVE = True
transformer = text_encoder = pipe = latent = decoded = image = None
try:
enable_huggingface_checkpointing()
scales = _install_fp8_converter()
tx_index, enc_index, tx_device, enc_device = _devices(torch)
quantization = NVIDIAModelOptConfig(
quant_type="FP8", weight_only=False,
modelopt_config=FP8_DEFAULT_CFG)
transformer = Flux2Transformer2DModel.from_single_file(
TRANSFORMER_FILE, config=COMPONENT_DIR, subfolder="transformer",
quantization_config=quantization, torch_dtype=torch.bfloat16,
device_map={"": tx_index}, local_files_only=True)
patched = 0
for module_name, module in transformer.named_modules():
if module_name not in scales:
continue
module.register_buffer("_fp8_input_scale",
scales[module_name]["input_scale"])
module.register_buffer("_fp8_weight_scale",
scales[module_name]["weight_scale"])
module.forward = MethodType(
lambda self, inputs: _fp8_forward(torch, self, inputs), module)
patched += 1
if patched != len(scales):
raise RuntimeError(f"patched only {patched} of {len(scales)} FP8 layers")
transformer.to(tx_device)
text_encoder = Qwen3ForCausalLM.from_pretrained(
os.path.join(COMPONENT_DIR, "text_encoder"),
torch_dtype=torch.bfloat16, low_cpu_mem_usage=True,
quantization_config=BitsAndBytesConfig(
load_in_4bit=True, bnb_4bit_quant_type="nf4",
bnb_4bit_compute_dtype=torch.bfloat16,
bnb_4bit_use_double_quant=True),
device_map={"": enc_index}, local_files_only=True)
pipe = Flux2KleinPipeline.from_pretrained(
COMPONENT_DIR, transformer=transformer, text_encoder=text_encoder,
torch_dtype=torch.bfloat16, local_files_only=True)
pipe.vae.enable_slicing()
pipe.vae.enable_tiling()
pipe.vae.to(tx_device)
loaded = time.monotonic() - started
prompt_embeds, _ = pipe.encode_prompt(
prompt.strip(), device=enc_device, max_sequence_length=128)
prompt_embeds = prompt_embeds.to(tx_device)
pipe.text_encoder = None
seed = data.get("seed")
generator = None if seed is None else torch.Generator(
device=tx_device).manual_seed(int(seed))
kwargs = {
"prompt": None, "prompt_embeds": prompt_embeds,
"height": height, "width": width, "num_inference_steps": 4,
"guidance_scale": 1.0, "generator": generator,
"output_type": "latent",
}
if source_images:
kwargs["image"] = (source_images[0] if len(source_images) == 1
else source_images)
latent = pipe(**kwargs).images
pipe.transformer = None
del transformer, text_encoder, prompt_embeds, generator
transformer = text_encoder = None
gc.collect()
torch.cuda.empty_cache()
latent = latent.to(device=tx_device, dtype=pipe.vae.dtype)
decoded = pipe.vae.decode(latent, return_dict=False)[0]
image = pipe.image_processor.postprocess(
decoded.detach(), output_type="pil")[0]
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
image.save(OUTPUT_DIR / filename)
return {"status": "ok", "filename": filename,
"seconds": round(time.monotonic() - started, 3),
"load_seconds": round(loaded, 3),
"model": "FLUX.2-klein-9B-fp8-beta"}
finally:
for value in (image, decoded, latent, pipe, text_encoder, transformer):
if value is not None:
del value
gc.collect()
torch.cuda.empty_cache()
ACTIVE = False
class Handler(BaseHTTPRequestHandler):
def log_message(self, fmt: str, *args: object) -> None:
print(f"[flux9b-beta] {self.client_address[0]} {fmt % args}", flush=True)
def reply(self, status: int, payload: dict) -> None:
body = json.dumps(payload, separators=(",", ":")).encode()
self.send_response(status)
self.send_header("Content-Type", "application/json")
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def do_GET(self) -> None: # noqa: N802
if self.path == "/health":
self.reply(200, {"status": "ok", "model_loaded": ACTIVE,
"model": "FLUX.2-klein-9B-fp8-beta"})
else:
self.reply(404, {"error": "not found"})
def do_POST(self) -> None: # noqa: N802
if self.headers.get("Authorization", "") != f"Bearer {TOKEN}":
self.reply(401, {"error": "unauthorized"})
return
if self.path != "/generate":
self.reply(404, {"error": "not found"})
return
try:
length = int(self.headers.get("Content-Length", "0"))
if length < 2 or length > 16384:
raise ValueError("invalid request size")
self.reply(200, generate(json.loads(self.rfile.read(length))))
except Exception as exc:
print(f"[flux9b-beta] generation failed: {type(exc).__name__}: "
f"{str(exc)[:1000]}", flush=True)
self.reply(500, {"status": "error", "message": str(exc)})
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()

Some files were not shown because too many files have changed in this diff Show More