Add dual-GPU FLUX 9B image pipeline
This commit is contained in:
+2
-1
@@ -3,7 +3,8 @@ AI_BIND_ADDRESS=10.77.0.2
|
|||||||
MODEL_DIR=/data/models
|
MODEL_DIR=/data/models
|
||||||
ROUTER_API_KEY=GENERATED_BY_INSTALLER
|
ROUTER_API_KEY=GENERATED_BY_INSTALLER
|
||||||
CONTROLLER_TOKEN=GENERATED_BY_INSTALLER
|
CONTROLLER_TOKEN=GENERATED_BY_INSTALLER
|
||||||
FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B
|
FLUX_COMPONENT_DIR=/data/models/FLUX.2-klein-9B-components
|
||||||
|
FLUX_TRANSFORMER_DIR=/data/models/FLUX.2-klein-9B-fp8
|
||||||
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ Sie betreibt:
|
|||||||
|
|
||||||
- llama.cpp mit genau einem aktiven Qwen-Profil,
|
- llama.cpp mit genau einem aktiven Qwen-Profil,
|
||||||
- den OpenAI-kompatiblen Profile Router,
|
- den OpenAI-kompatiblen Profile Router,
|
||||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung,
|
- FLUX.2 Klein 9B FP8 Beta für Textbilder und Referenzbild-Bearbeitung,
|
||||||
- Qwen3-TTS und Piper für Sprache,
|
- Qwen3-TTS und Piper für Sprache,
|
||||||
- das Athena-Dashboard,
|
- das Athena-Dashboard,
|
||||||
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
|
- Portainer CE als optionale Ansicht auf die laufenden Docker-Container,
|
||||||
@@ -56,8 +56,11 @@ Qwen-Profil wird vom Profile Controller verwaltet.
|
|||||||
- Fast: kurze, interaktive Aufgaben
|
- Fast: kurze, interaktive Aufgaben
|
||||||
- Medium/Large/Ultra: steigende Kontextgrößen desselben lokalen Qwen-Modells
|
- Medium/Large/Ultra: steigende Kontextgrößen desselben lokalen Qwen-Modells
|
||||||
- Uncensored: separates lokales Profil
|
- Uncensored: separates lokales Profil
|
||||||
- FLUX.2-klein-4B: Bildgenerierung und Editing; Qwen wird dafür kurz entladen und danach
|
- FLUX.2 Klein 9B FP8 Beta: Der Transformer läuft auf der RTX 5080, der
|
||||||
automatisch wiederhergestellt
|
Qwen3-8B-NF4-Textencoder vorübergehend auf der RTX 3060. Das aktive
|
||||||
|
llama.cpp-Profil und Qwen3-TTS werden dafür gestoppt und danach automatisch
|
||||||
|
wiederhergestellt. Die Beta arbeitet mit 1024 × 1024 Pixeln, vier Schritten
|
||||||
|
und Guidance 1,0.
|
||||||
- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback. Der Router reicht
|
- Qwen3-TTS 1.7B: RTX 3060; Piper bleibt CPU-Fallback. Der Router reicht
|
||||||
zusätzlich natives 24-kHz-PCM für den optionalen Hermes-Streaming-Adapter
|
zusätzlich natives 24-kHz-PCM für den optionalen Hermes-Streaming-Adapter
|
||||||
unter `integrations/hermes-qwen3-stream` durch.
|
unter `integrations/hermes-qwen3-stream` durch.
|
||||||
|
|||||||
@@ -10,7 +10,8 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
|
|||||||
|
|
||||||
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
|
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
|
||||||
- Profile Router auf Port 8081
|
- Profile Router auf Port 8081
|
||||||
- FLUX.2-klein-4B für Textbilder und Referenzbild-Bearbeitung auf der RTX 5080
|
- FLUX.2 Klein 9B FP8 Beta für Textbilder und Referenzbild-Bearbeitung:
|
||||||
|
Transformer auf RTX 5080, Qwen3-8B-NF4-Textencoder auf RTX 3060
|
||||||
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
|
- Qwen3-TTS 1.7B auf der RTX 3060 mit Piper als CPU-Fallback
|
||||||
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
- Whisper.cpp `large-v3-turbo` auf der CPU für lokale deutsche Spracherkennung
|
||||||
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
|
||||||
@@ -40,6 +41,11 @@ sudo ./install.sh --config /root/mike-ai-install.env
|
|||||||
|
|
||||||
Das Installationsskript baut llama.cpp und die lokalen Images, lädt die
|
Das Installationsskript baut llama.cpp und die lokalen Images, lädt die
|
||||||
versionierten Modellartefakte und startet ausschließlich den Athena-Kern.
|
versionierten Modellartefakte und startet ausschließlich den Athena-Kern.
|
||||||
|
FLUX.2 Klein 9B ist bei Hugging Face zugriffsbeschränkt. Vor der Installation
|
||||||
|
müssen die Bedingungen beider BFL-Repositories akzeptiert und ein Token in der
|
||||||
|
unter `HF_TOKEN_FILE` konfigurierten, nur für root lesbaren Datei abgelegt sein.
|
||||||
|
Der Token wird ausschließlich als Read-only-Datei in den Download-Container
|
||||||
|
eingehängt und weder in `stack.env` noch in Git kopiert.
|
||||||
|
|
||||||
## Betrieb
|
## Betrieb
|
||||||
|
|
||||||
@@ -95,6 +101,21 @@ Zwei Slots wurden direkt am Router erfolgreich getestet; Hermes verwaltete zwei
|
|||||||
gleichzeitig aktive Chats jedoch nicht zuverlässig. Deshalb bleibt ein Slot der
|
gleichzeitig aktive Chats jedoch nicht zuverlässig. Deshalb bleibt ein Slot der
|
||||||
Standard, bis Hermes' Sitzungsfehler behoben ist.
|
Standard, bis Hermes' Sitzungsfehler behoben ist.
|
||||||
|
|
||||||
|
### Bildgenerierung mit FLUX.2 Klein 9B FP8 Beta
|
||||||
|
|
||||||
|
Ein Bildauftrag verwendet beide GPUs exklusiv. Der Profile Controller stoppt
|
||||||
|
zuerst das aktive llama.cpp-Profil und Qwen3-TTS. Anschließend läuft der
|
||||||
|
FP8-Transformer auf der RTX 5080 und der in NF4 geladene Qwen3-8B-Textencoder
|
||||||
|
auf der RTX 3060. Vor dem VAE-Decoding werden Transformer und Textencoder
|
||||||
|
freigegeben. Nach dem Bildauftrag stoppt der Router den Bild-Worker und stellt
|
||||||
|
Qwen3-TTS sowie das zuvor aktive Textprofil automatisch wieder her. Piper
|
||||||
|
bleibt währenddessen als CPU-Fallback verfügbar.
|
||||||
|
|
||||||
|
Die Beta ist derzeit bewusst auf `1024x1024`, vier Schritte, Guidance `1.0`,
|
||||||
|
einen parallelen Auftrag und maximal vier lokale Referenzbilder begrenzt.
|
||||||
|
Details, Installation, Prüfung und Rollback stehen in
|
||||||
|
[docs/FLUX_9B_BETA.md](docs/FLUX_9B_BETA.md).
|
||||||
|
|
||||||
## Endpunkte
|
## Endpunkte
|
||||||
|
|
||||||
- Router: `http://192.168.1.212:8081/v1`
|
- Router: `http://192.168.1.212:8081/v1`
|
||||||
@@ -116,6 +137,11 @@ und bleibt deshalb bei normalen Container-Updates bestehen.
|
|||||||
Für Hermes liegt unter `integrations/hermes-qwen3-stream` ein optionales,
|
Für Hermes liegt unter `integrations/hermes-qwen3-stream` ein optionales,
|
||||||
persistentes Backend-Plugin. Es nutzt den nativen PCM-Strom und verkürzt den
|
persistentes Backend-Plugin. Es nutzt den nativen PCM-Strom und verkürzt den
|
||||||
Beginn der Sprachausgabe, ohne den Modellrouter oder die Textprofile zu ändern.
|
Beginn der Sprachausgabe, ohne den Modellrouter oder die Textprofile zu ändern.
|
||||||
|
Bildgenerierung läuft über `/v1/images/generations`; Hermes verwendet dafür den
|
||||||
|
persistenten Benutzer-Provider `athena-local` mit dem Modellnamen
|
||||||
|
`FLUX.2-klein-9B-fp8-beta`. Seine versionierte Quelle und Installationshinweise
|
||||||
|
liegen unter
|
||||||
|
[`integrations/hermes-athena-image`](integrations/hermes-athena-image).
|
||||||
- Portainer: `https://192.168.1.212:9443`
|
- Portainer: `https://192.168.1.212:9443`
|
||||||
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
|
- Hermes-Dashboard auf Unraid: `http://192.168.1.2:9119`
|
||||||
|
|
||||||
@@ -150,5 +176,6 @@ Der genaue Sicherungsumfang steht in [docs/RECOVERY.md](docs/RECOVERY.md).
|
|||||||
- [docs/STANDARD_PROFILE_MATRIX.md](docs/STANDARD_PROFILE_MATRIX.md) – Profile
|
- [docs/STANDARD_PROFILE_MATRIX.md](docs/STANDARD_PROFILE_MATRIX.md) – Profile
|
||||||
- [docs/MCP_SERVERS.md](docs/MCP_SERVERS.md) – produktive Werkzeuge
|
- [docs/MCP_SERVERS.md](docs/MCP_SERVERS.md) – produktive Werkzeuge
|
||||||
- [docs/RECOVERY.md](docs/RECOVERY.md) – Backup und Neuaufbau
|
- [docs/RECOVERY.md](docs/RECOVERY.md) – Backup und Neuaufbau
|
||||||
|
- [docs/FLUX_9B_BETA.md](docs/FLUX_9B_BETA.md) – 9B-Bildpfad, Test und Rollback
|
||||||
|
|
||||||
Git enthält keine Secrets, Chatdaten oder Modellgewichte.
|
Git enthält keine Secrets, Chatdaten oder Modellgewichte.
|
||||||
|
|||||||
+12
-8
@@ -571,6 +571,7 @@ services:
|
|||||||
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||||
ALLOWED_PROFILES: fast,medium,beta1,large,ultra,uncensored
|
ALLOWED_PROFILES: fast,medium,beta1,large,ultra,uncensored
|
||||||
IMAGE_WORKER: image
|
IMAGE_WORKER: image
|
||||||
|
TTS_WORKER: qwen3
|
||||||
networks: [control]
|
networks: [control]
|
||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
healthcheck:
|
healthcheck:
|
||||||
@@ -615,14 +616,13 @@ services:
|
|||||||
IMAGE_DIR: /data/images
|
IMAGE_DIR: /data/images
|
||||||
IMAGE_WORKER_URL: http://image-worker:8086
|
IMAGE_WORKER_URL: http://image-worker:8086
|
||||||
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||||
IMAGE_MODEL_NAME: FLUX.2-klein-4B
|
IMAGE_MODEL_NAME: FLUX.2-klein-9B-fp8-beta
|
||||||
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
|
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
|
||||||
ENABLE_IMAGE_GENERATION: "true"
|
ENABLE_IMAGE_GENERATION: "true"
|
||||||
ENABLE_TTS: "true"
|
ENABLE_TTS: "true"
|
||||||
# Stable OpenAI compatibility names remain piper/alloy because an
|
# Stable OpenAI compatibility names remain piper/alloy for existing
|
||||||
# existing Open WebUI database persists those values. The gateway maps
|
# clients. The gateway maps them to Qwen3-TTS and falls back to Piper
|
||||||
# alloy to XTTS speaker Annmarie Nele and automatically falls back to
|
# while Qwen3-TTS is unavailable, busy or paused for image generation.
|
||||||
# Piper if XTTS is unavailable, busy or returns an error.
|
|
||||||
TTS_WORKER_URL: http://tts-gateway:8085
|
TTS_WORKER_URL: http://tts-gateway:8085
|
||||||
TTS_MODEL: piper
|
TTS_MODEL: piper
|
||||||
TTS_VOICES: alloy
|
TTS_VOICES: alloy
|
||||||
@@ -674,13 +674,15 @@ services:
|
|||||||
read_only: true
|
read_only: true
|
||||||
tmpfs: ["/tmp:size=1g,mode=1777"]
|
tmpfs: ["/tmp:size=1g,mode=1777"]
|
||||||
volumes:
|
volumes:
|
||||||
- "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro"
|
- "${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}:/models/components:ro"
|
||||||
|
- "${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}:/models/fp8:ro"
|
||||||
- router-images:/data/images
|
- router-images:/data/images
|
||||||
environment:
|
environment:
|
||||||
NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1}
|
NVIDIA_VISIBLE_DEVICES: all
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||||
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||||
FLUX_MODEL_DIR: /models/FLUX.2-klein-4B
|
FLUX_COMPONENT_DIR: /models/components
|
||||||
|
FLUX_TRANSFORMER_FILE: /models/fp8/flux-2-klein-9b-fp8.safetensors
|
||||||
IMAGE_DIR: /data/images
|
IMAGE_DIR: /data/images
|
||||||
networks: [inference]
|
networks: [inference]
|
||||||
security_opt: ["no-new-privileges:true"]
|
security_opt: ["no-new-privileges:true"]
|
||||||
@@ -726,6 +728,8 @@ services:
|
|||||||
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
|
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
|
||||||
container_name: mike-ai-qwen3-tts
|
container_name: mike-ai-qwen3-tts
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
labels:
|
||||||
|
com.mike-ai.tts-worker: qwen3
|
||||||
deploy:
|
deploy:
|
||||||
resources:
|
resources:
|
||||||
reservations:
|
reservations:
|
||||||
|
|||||||
@@ -18,7 +18,11 @@ NVIDIA_MIN_DRIVER_MAJOR=570
|
|||||||
TEXT_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
TEXT_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
||||||
SECONDARY_GPU_DEVICES=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
SECONDARY_GPU_DEVICES=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||||
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
IMAGE_GPU_DEVICES=GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe
|
||||||
FLUX_MODEL_DIR=/data/models/FLUX.2-klein-4B
|
# FLUX.2 Klein 9B is gated. Accept both BFL model licenses first, then store
|
||||||
|
# the Hugging Face token in this root-readable file (never in this config).
|
||||||
|
HF_TOKEN_FILE=/root/.cache/huggingface/token
|
||||||
|
FLUX_COMPONENT_DIR=/data/models/FLUX.2-klein-9B-components
|
||||||
|
FLUX_TRANSFORMER_DIR=/data/models/FLUX.2-klein-9B-fp8
|
||||||
|
|
||||||
# Headless remote reachability. Firmware power-loss recovery is configured
|
# Headless remote reachability. Firmware power-loss recovery is configured
|
||||||
# separately once at the physical machine.
|
# separately once at the physical machine.
|
||||||
|
|||||||
@@ -26,6 +26,11 @@ def image_item(state="exited"):
|
|||||||
"Labels": {controller.IMAGE_LABEL_KEY: controller.IMAGE_WORKER}}
|
"Labels": {controller.IMAGE_LABEL_KEY: controller.IMAGE_WORKER}}
|
||||||
|
|
||||||
|
|
||||||
|
def tts_item(state="running"):
|
||||||
|
return {"Id": "id-tts", "State": state,
|
||||||
|
"Labels": {controller.TTS_LABEL_KEY: controller.TTS_WORKER}}
|
||||||
|
|
||||||
|
|
||||||
class ProfileControllerTests(unittest.TestCase):
|
class ProfileControllerTests(unittest.TestCase):
|
||||||
def test_rejects_unknown_profile_before_docker_call(self):
|
def test_rejects_unknown_profile_before_docker_call(self):
|
||||||
with patch.object(controller, "docker_request") as request:
|
with patch.object(controller, "docker_request") as request:
|
||||||
@@ -44,6 +49,7 @@ class ProfileControllerTests(unittest.TestCase):
|
|||||||
|
|
||||||
with patch.object(controller, "containers", return_value=profiles), \
|
with patch.object(controller, "containers", return_value=profiles), \
|
||||||
patch.object(controller, "image_container", return_value=image_item()), \
|
patch.object(controller, "image_container", return_value=image_item()), \
|
||||||
|
patch.object(controller, "tts_container", return_value=tts_item()), \
|
||||||
patch.object(controller, "docker_request", side_effect=request):
|
patch.object(controller, "docker_request", side_effect=request):
|
||||||
result = controller.activate("medium")
|
result = controller.activate("medium")
|
||||||
|
|
||||||
@@ -56,7 +62,8 @@ class ProfileControllerTests(unittest.TestCase):
|
|||||||
def test_fails_if_profile_container_is_missing(self):
|
def test_fails_if_profile_container_is_missing(self):
|
||||||
profiles = {name: item(name) for name in controller.ALLOWED[:-1]}
|
profiles = {name: item(name) for name in controller.ALLOWED[:-1]}
|
||||||
with patch.object(controller, "containers", return_value=profiles), \
|
with patch.object(controller, "containers", return_value=profiles), \
|
||||||
patch.object(controller, "image_container", return_value=image_item()):
|
patch.object(controller, "image_container", return_value=image_item()), \
|
||||||
|
patch.object(controller, "tts_container", return_value=tts_item()):
|
||||||
with self.assertRaisesRegex(RuntimeError, "missing"):
|
with self.assertRaisesRegex(RuntimeError, "missing"):
|
||||||
controller.activate("fast")
|
controller.activate("fast")
|
||||||
|
|
||||||
@@ -71,10 +78,12 @@ class ProfileControllerTests(unittest.TestCase):
|
|||||||
|
|
||||||
with patch.object(controller, "containers", return_value=profiles), \
|
with patch.object(controller, "containers", return_value=profiles), \
|
||||||
patch.object(controller, "image_container", return_value=image_item()), \
|
patch.object(controller, "image_container", return_value=image_item()), \
|
||||||
|
patch.object(controller, "tts_container", return_value=tts_item()), \
|
||||||
patch.object(controller, "docker_request", side_effect=request):
|
patch.object(controller, "docker_request", side_effect=request):
|
||||||
controller.set_image_worker(True)
|
controller.set_image_worker(True)
|
||||||
self.assertEqual(calls, [
|
self.assertEqual(calls, [
|
||||||
("POST", "/containers/id-medium/stop?t=120"),
|
("POST", "/containers/id-medium/stop?t=120"),
|
||||||
|
("POST", "/containers/id-tts/stop?t=30"),
|
||||||
("POST", "/containers/id-flux/start"),
|
("POST", "/containers/id-flux/start"),
|
||||||
])
|
])
|
||||||
|
|
||||||
@@ -89,6 +98,7 @@ class ProfileControllerTests(unittest.TestCase):
|
|||||||
with patch.object(controller, "containers", return_value=profiles), \
|
with patch.object(controller, "containers", return_value=profiles), \
|
||||||
patch.object(controller, "image_container",
|
patch.object(controller, "image_container",
|
||||||
return_value=image_item("running")), \
|
return_value=image_item("running")), \
|
||||||
|
patch.object(controller, "tts_container", return_value=tts_item()), \
|
||||||
patch.object(controller, "docker_request", side_effect=request):
|
patch.object(controller, "docker_request", side_effect=request):
|
||||||
controller.activate("fast")
|
controller.activate("fast")
|
||||||
self.assertEqual(calls, [
|
self.assertEqual(calls, [
|
||||||
|
|||||||
+25
-4
@@ -6,8 +6,9 @@ flowchart LR
|
|||||||
H -->|OpenAI API| R[Profile Router<br/>Athena :8081]
|
H -->|OpenAI API| R[Profile Router<br/>Athena :8081]
|
||||||
R --> P[Profile Controller]
|
R --> P[Profile Controller]
|
||||||
P --> Q[genau ein llama.cpp-Profil<br/>Qwen Fast / Medium / Large / Ultra / Uncensored]
|
P --> Q[genau ein llama.cpp-Profil<br/>Qwen Fast / Medium / Large / Ultra / Uncensored]
|
||||||
R --> I[FLUX.2-klein-4B<br/>RTX 5080, Text + Editing]
|
R --> I[FLUX.2 Klein 9B FP8 Beta<br/>RTX 5080 Transformer]
|
||||||
R --> T[XTTS RTX 3060<br/>Piper CPU-Fallback]
|
I --> E[Qwen3-8B NF4 Textencoder<br/>RTX 3060 während Bildauftrag]
|
||||||
|
R --> T[Qwen3-TTS RTX 3060<br/>Piper CPU-Fallback]
|
||||||
R --> STT[Whisper.cpp large-v3-turbo<br/>CPU, lokale Spracherkennung]
|
R --> STT[Whisper.cpp large-v3-turbo<br/>CPU, lokale Spracherkennung]
|
||||||
|
|
||||||
H --> U[MUA / Unraid MCP]
|
H --> U[MUA / Unraid MCP]
|
||||||
@@ -46,9 +47,29 @@ Kontextgröße:
|
|||||||
| Medium | 160.000 Token |
|
| Medium | 160.000 Token |
|
||||||
| Large | 192.000 Token |
|
| Large | 192.000 Token |
|
||||||
| Ultra | 262.144 Token |
|
| Ultra | 262.144 Token |
|
||||||
|
| Beta 1 | 192.000 Token |
|
||||||
| Uncensored | 80.000 Token |
|
| Uncensored | 80.000 Token |
|
||||||
|
|
||||||
|
## Exklusiver Bildmodus
|
||||||
|
|
||||||
|
Text- und Bildinferenz teilen sich dieselben GPUs und laufen deshalb nicht
|
||||||
|
gleichzeitig. Der Wechsel ist transaktional:
|
||||||
|
|
||||||
|
1. Router merkt sich das aktive Textprofil.
|
||||||
|
2. Profile Controller stoppt alle llama.cpp-Profile und Qwen3-TTS.
|
||||||
|
3. Bild-Worker lädt Qwen3-8B als NF4-Textencoder auf die RTX 3060 und den
|
||||||
|
FLUX.2-Klein-9B-FP8-Transformer auf die RTX 5080.
|
||||||
|
4. Nach dem Prompt-Encoding werden die Embeddings zur RTX 5080 übertragen.
|
||||||
|
5. Vor dem VAE-Decoding werden Textencoder und Transformer freigegeben.
|
||||||
|
6. Der Worker wird gestoppt; anschließend starten Qwen3-TTS und das vorherige
|
||||||
|
Textprofil wieder. Piper bleibt als CPU-Fallback verfügbar.
|
||||||
|
|
||||||
|
Der Bild-Worker ist lazy und besitzt `restart: "no"`; im normalen Textbetrieb
|
||||||
|
belegt er daher keinen VRAM. Container werden über eindeutige Docker-Labels
|
||||||
|
gefunden, nicht über zufällige Container-IDs.
|
||||||
|
|
||||||
Die visuelle Fassung liegt als `athena-architecture-map.png` neben dieser Datei.
|
Die visuelle Fassung liegt als `athena-architecture-map.png` neben dieser Datei.
|
||||||
Eine zweite Detailkarte, `athena-gpu-allocation-map.png`, zeigt die
|
Eine zweite Detailkarte, `athena-gpu-allocation-map.png`, zeigt die
|
||||||
profilabhängige Layer-Verteilung auf RTX 5080 und RTX 3060 sowie die festen
|
profilabhängige Layer-Verteilung auf RTX 5080 und RTX 3060. Die PNG-Karten
|
||||||
GPU-Zuordnungen von FLUX.2, Vision-Projektor und XTTS.
|
zeigen noch den Stand vor dem 9B-Bildpfad; die aktuelle textuelle Beschreibung
|
||||||
|
in diesem Dokument ist verbindlich.
|
||||||
|
|||||||
@@ -1,6 +1,29 @@
|
|||||||
# Aktueller produktiver Laufzustand
|
# Aktueller produktiver Laufzustand
|
||||||
|
|
||||||
Stand: 3. September 2026
|
Stand: 7. September 2026
|
||||||
|
|
||||||
|
## FLUX.2 Klein 9B FP8 Beta
|
||||||
|
|
||||||
|
Die bisherige 4B-Bildinferenz wurde testweise durch FLUX.2 Klein 9B FP8
|
||||||
|
ersetzt. Athenas Profile Controller stellt dafür einen exklusiven Zwei-GPU-Pfad
|
||||||
|
bereit:
|
||||||
|
|
||||||
|
- RTX 5080: 9B-FP8-Diffusionstransformer und VAE-Decoding
|
||||||
|
- RTX 3060: Qwen3-8B-Textencoder in NF4
|
||||||
|
- Qwen3-TTS und aktives llama.cpp-Profil werden für den Bildauftrag pausiert
|
||||||
|
- Piper bleibt währenddessen als CPU-TTS verfügbar
|
||||||
|
- nach Abschluss werden TTS und das vorherige Textprofil wiederhergestellt
|
||||||
|
|
||||||
|
Ein vollständiger Aufruf über Athenas OpenAI-kompatiblen Router wurde mit
|
||||||
|
HTTP 200, einem korrekt gespeicherten 1024×1024-PNG und anschließender
|
||||||
|
Wiederherstellung von Qwen3-TTS und `qwen-fast` erfolgreich geprüft. Ein
|
||||||
|
isolierter Vergleich ergab ungefähr 14,6 Sekunden Bildlaufzeit mit dem
|
||||||
|
GPU-Textencoder gegenüber 102,1 Sekunden mit CPU-Textencoder. Diese Werte sind
|
||||||
|
eine lokale Einzelmessung und keine allgemeine Modellgarantie.
|
||||||
|
|
||||||
|
Das Modell ist nicht kommerziell lizenziert. Die Bedingungen der beiden
|
||||||
|
zugriffsbeschränkten Black-Forest-Labs-Repositories müssen vor dem Download
|
||||||
|
akzeptiert werden. Details stehen in [FLUX_9B_BETA.md](FLUX_9B_BETA.md).
|
||||||
|
|
||||||
## Lokale Spracherkennung
|
## Lokale Spracherkennung
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,153 @@
|
|||||||
|
# FLUX.2 Klein 9B FP8 Beta auf Athena
|
||||||
|
|
||||||
|
Stand: 7. September 2026
|
||||||
|
|
||||||
|
## Zweck und Status
|
||||||
|
|
||||||
|
Der Bildpfad ersetzt testweise FLUX.2 Klein 4B durch das größere
|
||||||
|
FLUX.2-Klein-9B-Modell. Ziel sind bessere Prompttreue, räumliche Beziehungen,
|
||||||
|
Objektkonsistenz und Referenzbild-Bearbeitung. Der Pfad ist technisch
|
||||||
|
funktionsfähig, bleibt aber bis zu weiteren Qualitäts- und Editing-Tests als
|
||||||
|
Beta bezeichnet.
|
||||||
|
|
||||||
|
Der OpenAI-kompatible Modellname lautet:
|
||||||
|
|
||||||
|
```text
|
||||||
|
FLUX.2-klein-9B-fp8-beta
|
||||||
|
```
|
||||||
|
|
||||||
|
## Modellartefakte und Lizenz
|
||||||
|
|
||||||
|
Verwendet werden zwei gepinnte, zugriffsbeschränkte Hugging-Face-Repositories:
|
||||||
|
|
||||||
|
| Zweck | Repository | Revision | Lokaler Pfad |
|
||||||
|
|---|---|---|---|
|
||||||
|
| Pipeline-Komponenten, Qwen3-Textencoder und VAE | `black-forest-labs/FLUX.2-klein-9B` | `92196c8e11f7b6cf2b7493e037d8c5345c559216` | `/data/models/FLUX.2-klein-9B-components` |
|
||||||
|
| FP8-Transformer | `black-forest-labs/FLUX.2-klein-9b-fp8` | `902d9d510b51533e07729f19211414a3648b77d2` | `/data/models/FLUX.2-klein-9B-fp8` |
|
||||||
|
|
||||||
|
FLUX.2 Klein 9B steht unter der FLUX Non-Commercial License. Vor dem Download
|
||||||
|
müssen die Bedingungen beider Repositories im verwendeten Hugging-Face-Konto
|
||||||
|
akzeptiert werden. Ein Token gehört ausschließlich in die durch
|
||||||
|
`HF_TOKEN_FILE` angegebene, für root lesbare Datei; niemals in Git oder
|
||||||
|
`stack.env`.
|
||||||
|
|
||||||
|
## GPU-Aufteilung
|
||||||
|
|
||||||
|
| Phase | RTX 5080, 16 GB | RTX 3060, 12 GB |
|
||||||
|
|---|---|---|
|
||||||
|
| Text-/Sprachbetrieb | aktives Qwen3.8-27B-Profil | Qwen3-TTS; Vision je nach Profil |
|
||||||
|
| Prompt-Encoding | FLUX-Transformer und VAE | Qwen3-8B-Textencoder, NF4 |
|
||||||
|
| Denoising | FLUX-Transformer | Textencoder wird nicht mehr benötigt |
|
||||||
|
| VAE-Decoding | VAE; Transformer zuvor freigegeben | Textencoder zuvor freigegeben |
|
||||||
|
|
||||||
|
Der Profile Controller stoppt vor dem Start des Bild-Workers alle
|
||||||
|
llama.cpp-Profile und den mit `com.mike-ai.tts-worker=qwen3` markierten
|
||||||
|
Qwen3-TTS-Container. Dadurch bleibt genügend VRAM für beide Bildkomponenten.
|
||||||
|
Nach dem Bildauftrag startet er Qwen3-TTS und das zuvor aktive Textprofil
|
||||||
|
wieder. Der TTS-Gateway kann während des Wechsels auf Piper ausweichen.
|
||||||
|
|
||||||
|
## Aktuelle Grenzen
|
||||||
|
|
||||||
|
- genau 1024 × 1024 Pixel
|
||||||
|
- genau vier Inferenzschritte
|
||||||
|
- Guidance Scale 1,0
|
||||||
|
- ein Bildauftrag gleichzeitig
|
||||||
|
- höchstens vier bereits lokal gespeicherte Referenzbilder
|
||||||
|
- Textencoder-Maximum 128 Token
|
||||||
|
- Bildbearbeitung wird vom Worker angenommen, ist aber noch gesondert
|
||||||
|
Ende-zu-Ende zu qualifizieren
|
||||||
|
|
||||||
|
## Installation und Aktualisierung
|
||||||
|
|
||||||
|
In `/root/mike-ai-install.env` müssen diese Werte gesetzt sein:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
HF_TOKEN_FILE=/root/.cache/huggingface/token
|
||||||
|
FLUX_COMPONENT_DIR=/data/models/FLUX.2-klein-9B-components
|
||||||
|
FLUX_TRANSFORMER_DIR=/data/models/FLUX.2-klein-9B-fp8
|
||||||
|
```
|
||||||
|
|
||||||
|
Anschließend lädt der normale Installer nur die benötigten Komponenten und die
|
||||||
|
gepinnten FP8-Gewichte. Bestehende, vollständige Dateien werden nicht erneut
|
||||||
|
geladen:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd /opt/mike-ai/stack
|
||||||
|
sudo ./install.sh --config /root/mike-ai-install.env
|
||||||
|
```
|
||||||
|
|
||||||
|
## Funktionsprobe
|
||||||
|
|
||||||
|
Der Router ist nur über das private Netz erreichbar. Ein minimaler Test lautet:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
curl -fsS http://192.168.1.212:8081/v1/images/generations \
|
||||||
|
-H "Authorization: Bearer $ROUTER_API_KEY" \
|
||||||
|
-H 'Content-Type: application/json' \
|
||||||
|
-d '{
|
||||||
|
"model":"FLUX.2-klein-9B-fp8-beta",
|
||||||
|
"prompt":"A yellow toy excavator on the left and a red toy truck on the right, studio photo",
|
||||||
|
"size":"1024x1024",
|
||||||
|
"steps":4,
|
||||||
|
"guidance":1.0,
|
||||||
|
"seed":9072026
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
Danach müssen folgende Zustände wiederhergestellt sein:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker ps --format '{{.Names}} {{.Status}}' \
|
||||||
|
--filter name=mike-ai-router \
|
||||||
|
--filter name=mike-ai-qwen3-tts \
|
||||||
|
--filter name=mike-ai-llama
|
||||||
|
docker ps -a --filter name=mike-ai-image-worker \
|
||||||
|
--format '{{.Names}} {{.Status}}'
|
||||||
|
nvidia-smi
|
||||||
|
```
|
||||||
|
|
||||||
|
Erwartet werden ein gesunder Router, gesundes Qwen3-TTS, genau ein gesundes
|
||||||
|
llama.cpp-Profil und ein mit Exit-Code 0 beendeter Bild-Worker.
|
||||||
|
|
||||||
|
## Hermes
|
||||||
|
|
||||||
|
Hermes auf Unraid verwendet einen persistenten Benutzer-Provider
|
||||||
|
`athena-local`. Seine Konfiguration muss auf denselben Modellnamen zeigen:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
image_gen:
|
||||||
|
provider: athena-local
|
||||||
|
model: FLUX.2-klein-9B-fp8-beta
|
||||||
|
max_parallel_requests: 1
|
||||||
|
```
|
||||||
|
|
||||||
|
Der Provider lebt in Hermes-Appdata und bleibt bei normalen Container-Updates
|
||||||
|
erhalten. Er gehört nicht in die Desktop-App und muss auf weiteren Clients
|
||||||
|
nicht erneut installiert werden. Die versionierte Quellfassung liegt unter
|
||||||
|
[`integrations/hermes-athena-image`](../integrations/hermes-athena-image).
|
||||||
|
|
||||||
|
## Rollback
|
||||||
|
|
||||||
|
Vor der Beta-Bereitstellung wurden auf Athena diese Rückfall-Images angelegt:
|
||||||
|
|
||||||
|
```text
|
||||||
|
mike-ai/image-worker:4b-rollback-20260907
|
||||||
|
mike-ai/profile-controller:rollback-20260907
|
||||||
|
```
|
||||||
|
|
||||||
|
Die zugehörige Deployment-Sicherung liegt auf Athena unter:
|
||||||
|
|
||||||
|
```text
|
||||||
|
/data/deploy-backups/20260907-flux9b-beta
|
||||||
|
```
|
||||||
|
|
||||||
|
Die vorherige Hermes-Konfiguration und der alte Provider liegen auf Unraid
|
||||||
|
unter:
|
||||||
|
|
||||||
|
```text
|
||||||
|
/mnt/nvme-storage/appdata/Hermes-Agent/backups/flux9b-beta-20260907
|
||||||
|
```
|
||||||
|
|
||||||
|
Ein Rollback darf nicht blind erfolgen: Zuerst aktives Profil, laufende
|
||||||
|
Anfragen und vorhandene Image-Tags prüfen, dann nur Image-Worker,
|
||||||
|
Profile Controller und Hermes-Provider auf den gesicherten Stand zurücksetzen.
|
||||||
@@ -15,6 +15,12 @@ Sie bleiben auf der Daten-SSD oder werden anhand der gepinnten Angaben in
|
|||||||
`config/install.env.example` erneut geladen. Die Dashboard-Historie liegt
|
`config/install.env.example` erneut geladen. Die Dashboard-Historie liegt
|
||||||
dauerhaft unter `/data/llama-dashboard`.
|
dauerhaft unter `/data/llama-dashboard`.
|
||||||
|
|
||||||
|
Für FLUX.2 Klein 9B müssen vor einem erneuten Download die Bedingungen der
|
||||||
|
beiden Black-Forest-Labs-Repositories im Hugging-Face-Konto akzeptiert sein.
|
||||||
|
Außerdem muss die in `HF_TOKEN_FILE` angegebene Token-Datei wiederhergestellt
|
||||||
|
oder neu erzeugt werden. Der Token selbst ist absichtlich nicht Bestandteil
|
||||||
|
des Git-Repositories oder des Athena-Backups.
|
||||||
|
|
||||||
Portainers lokale Konfiguration liegt im Docker-Volume `portainer_data` und
|
Portainers lokale Konfiguration liegt im Docker-Volume `portainer_data` und
|
||||||
wird zusammen mit den übrigen nicht reproduzierbaren Volumes gesichert und
|
wird zusammen mit den übrigen nicht reproduzierbaren Volumes gesichert und
|
||||||
wiederhergestellt.
|
wiederhergestellt.
|
||||||
|
|||||||
+23
-8
@@ -365,7 +365,8 @@ UNCENSORED_GPU_DEVICES=${TEXT_GPU_DEVICES:-0}${SECONDARY_GPU_DEVICES:+,$SECONDAR
|
|||||||
UNCENSORED_TENSOR_SPLIT=${UNCENSORED_TENSOR_SPLIT:-90,10}
|
UNCENSORED_TENSOR_SPLIT=${UNCENSORED_TENSOR_SPLIT:-90,10}
|
||||||
UNCENSORED_MTP_MAX=${UNCENSORED_MTP_MAX:-2}
|
UNCENSORED_MTP_MAX=${UNCENSORED_MTP_MAX:-2}
|
||||||
IMAGE_GPU_DEVICES=${IMAGE_GPU_DEVICES:-${TEXT_GPU_DEVICES:-0}}
|
IMAGE_GPU_DEVICES=${IMAGE_GPU_DEVICES:-${TEXT_GPU_DEVICES:-0}}
|
||||||
FLUX_MODEL_DIR=${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}
|
FLUX_COMPONENT_DIR=${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}
|
||||||
|
FLUX_TRANSFORMER_DIR=${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}
|
||||||
LLAMA_THREADS=${LLAMA_THREADS:-6}
|
LLAMA_THREADS=${LLAMA_THREADS:-6}
|
||||||
LLAMA_THREADS_BATCH=${LLAMA_THREADS_BATCH:-6}
|
LLAMA_THREADS_BATCH=${LLAMA_THREADS_BATCH:-6}
|
||||||
LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-32768}
|
LLAMA_CACHE_RAM_MIB=${LLAMA_CACHE_RAM_MIB:-32768}
|
||||||
@@ -486,14 +487,28 @@ build_and_start() {
|
|||||||
docker build --progress=plain --build-arg LLAMA_CPP_COMMIT="$commit" \
|
docker build --progress=plain --build-arg LLAMA_CPP_COMMIT="$commit" \
|
||||||
-f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local .
|
-f platform/docker/llama-cpp/Dockerfile -t mike-ai/llama.cpp:local .
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image build image-worker
|
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image build image-worker
|
||||||
if [[ ! -s ${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}/model_index.json ]]; then
|
local flux_components=${FLUX_COMPONENT_DIR:-/data/models/FLUX.2-klein-9B-components}
|
||||||
log "FLUX.2-klein-4B laden"
|
local flux_transformer=${FLUX_TRANSFORMER_DIR:-/data/models/FLUX.2-klein-9B-fp8}
|
||||||
install -d -m 0755 "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}"
|
local hf_token_file=${HF_TOKEN_FILE:-/root/.cache/huggingface/token}
|
||||||
docker run --rm --entrypoint python \
|
if [[ ! -s $flux_components/model_index.json || \
|
||||||
-v "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/download" \
|
! -s $flux_transformer/flux-2-klein-9b-fp8.safetensors ]]; then
|
||||||
|
[[ -r $hf_token_file ]] || die \
|
||||||
|
"Hugging-Face-Token fehlt: $hf_token_file (FLUX.2 Klein 9B ist gated)"
|
||||||
|
log "FLUX.2 Klein 9B Komponenten und FP8-Transformer laden"
|
||||||
|
install -d -m 0755 "$flux_components" "$flux_transformer"
|
||||||
|
docker run --rm --entrypoint /opt/image-venv/bin/python \
|
||||||
|
-e HF_TOKEN_PATH=/run/secrets/hf-token \
|
||||||
|
-v "$hf_token_file:/run/secrets/hf-token:ro" \
|
||||||
|
-v "$flux_components:/download" \
|
||||||
mike-ai/image-worker:local -c \
|
mike-ai/image-worker:local -c \
|
||||||
"from huggingface_hub import snapshot_download; snapshot_download('black-forest-labs/FLUX.2-klein-4B', revision='e7b7dc27f91deacad38e78976d1f2b499d76a294', local_dir='/download')"
|
"from huggingface_hub import snapshot_download; snapshot_download('black-forest-labs/FLUX.2-klein-9B', revision='92196c8e11f7b6cf2b7493e037d8c5345c559216', local_dir='/download', allow_patterns=['model_index.json', 'scheduler/*', 'text_encoder/*', 'tokenizer/*', 'transformer/config.json', 'vae/*'])"
|
||||||
chmod -R a-w "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}"
|
docker run --rm --entrypoint /opt/image-venv/bin/python \
|
||||||
|
-e HF_TOKEN_PATH=/run/secrets/hf-token \
|
||||||
|
-v "$hf_token_file:/run/secrets/hf-token:ro" \
|
||||||
|
-v "$flux_transformer:/download" \
|
||||||
|
mike-ai/image-worker:local -c \
|
||||||
|
"from huggingface_hub import snapshot_download; snapshot_download('black-forest-labs/FLUX.2-klein-9b-fp8', revision='902d9d510b51533e07729f19211414a3648b77d2', local_dir='/download', allow_patterns=['flux-2-klein-9b-fp8.safetensors', 'README.md', 'LICENSE.md'])"
|
||||||
|
chmod -R a-w "$flux_components" "$flux_transformer"
|
||||||
fi
|
fi
|
||||||
# Creates the tools network and deploys the only host-bound MCP: Operator.
|
# Creates the tools network and deploys the only host-bound MCP: Operator.
|
||||||
# Portable MCPs and Hermes live on Unraid and are restored through Appdata.
|
# Portable MCPs and Hermes live on Unraid and are restored through Appdata.
|
||||||
|
|||||||
@@ -0,0 +1,46 @@
|
|||||||
|
# Hermes Athena image provider
|
||||||
|
|
||||||
|
Hermes backend plugin for the OpenAI-compatible image API exposed by the
|
||||||
|
Athena profile router. The router starts the local FLUX worker on demand,
|
||||||
|
unloads the active LLM and Qwen3-TTS, and restores both after generation.
|
||||||
|
|
||||||
|
## Gateway installation
|
||||||
|
|
||||||
|
Install this directory on the Hermes gateway, not on each Desktop client:
|
||||||
|
|
||||||
|
```text
|
||||||
|
$HERMES_HOME/plugins/image_gen/athena-local/
|
||||||
|
__init__.py
|
||||||
|
plugin.yaml
|
||||||
|
```
|
||||||
|
|
||||||
|
Set these secrets or environment variables on the gateway:
|
||||||
|
|
||||||
|
```text
|
||||||
|
ATHENA_IMAGE_BASE_URL=http://192.168.1.212:8081/v1
|
||||||
|
ATHENA_IMAGE_API_KEY=<same API key accepted by the Athena router>
|
||||||
|
```
|
||||||
|
|
||||||
|
Then enable and select the provider:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
plugins:
|
||||||
|
enabled:
|
||||||
|
- image_gen/athena-local
|
||||||
|
|
||||||
|
image_gen:
|
||||||
|
provider: athena-local
|
||||||
|
model: FLUX.2-klein-9B-fp8-beta
|
||||||
|
max_parallel_requests: 1
|
||||||
|
```
|
||||||
|
|
||||||
|
Restart the Hermes gateway after changing plugin files or configuration. A
|
||||||
|
second computer connected to the same gateway needs no plugin installation.
|
||||||
|
|
||||||
|
Optional overrides:
|
||||||
|
|
||||||
|
- `ATHENA_IMAGE_MODEL` defaults to `FLUX.2-klein-9B-fp8-beta`.
|
||||||
|
- `ROUTER_API_KEY` is accepted as a migration fallback.
|
||||||
|
- An existing `HERMES_CUSTOM_192_168_1_212_8081_API_KEY` is accepted as the
|
||||||
|
final fallback, so an existing Athena chat-provider setup needs no duplicate
|
||||||
|
secret.
|
||||||
@@ -0,0 +1,260 @@
|
|||||||
|
"""Hermes image generation/edit provider for the local Athena router.
|
||||||
|
|
||||||
|
The provider deliberately rejects public destinations. Prompts and generated
|
||||||
|
images may only travel to a loopback or private-network address.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import base64
|
||||||
|
import ipaddress
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import urllib.error
|
||||||
|
import urllib.parse
|
||||||
|
import urllib.request
|
||||||
|
from typing import Any, Dict, List, Optional
|
||||||
|
|
||||||
|
from agent.image_gen_provider import (
|
||||||
|
DEFAULT_ASPECT_RATIO,
|
||||||
|
ImageGenProvider,
|
||||||
|
error_response,
|
||||||
|
normalize_reference_images,
|
||||||
|
resolve_aspect_ratio,
|
||||||
|
save_b64_image,
|
||||||
|
success_response,
|
||||||
|
)
|
||||||
|
from agent.secret_scope import get_secret
|
||||||
|
|
||||||
|
|
||||||
|
_SIZES = {
|
||||||
|
"landscape": "1536x1024",
|
||||||
|
"square": "1024x1024",
|
||||||
|
"portrait": "1024x1536",
|
||||||
|
}
|
||||||
|
_DEFAULT_BASE_URL = "http://192.168.1.212:8081/v1"
|
||||||
|
_DEFAULT_MODEL = "FLUX.2-klein-9B-fp8-beta"
|
||||||
|
_MAX_IMAGE_BYTES = 20 * 1024 * 1024
|
||||||
|
|
||||||
|
|
||||||
|
def _base_url() -> str:
|
||||||
|
"""Use the dedicated image URL and never inherit an unrelated chat URL."""
|
||||||
|
override = os.environ.get("ATHENA_IMAGE_BASE_URL", "").strip()
|
||||||
|
return (override or _DEFAULT_BASE_URL).rstrip("/")
|
||||||
|
|
||||||
|
|
||||||
|
def _model() -> str:
|
||||||
|
return os.environ.get("ATHENA_IMAGE_MODEL", "").strip() or _DEFAULT_MODEL
|
||||||
|
|
||||||
|
|
||||||
|
def _api_key() -> str:
|
||||||
|
"""Prefer a scoped key; accept the existing router key for migration."""
|
||||||
|
return (
|
||||||
|
get_secret("ATHENA_IMAGE_API_KEY", "")
|
||||||
|
or get_secret("ROUTER_API_KEY", "")
|
||||||
|
or get_secret("HERMES_CUSTOM_192_168_1_212_8081_API_KEY", "")
|
||||||
|
or ""
|
||||||
|
).strip()
|
||||||
|
|
||||||
|
|
||||||
|
def _private_destination(url: str) -> bool:
|
||||||
|
"""Fail closed unless the configured endpoint is local/private."""
|
||||||
|
try:
|
||||||
|
parsed = urllib.parse.urlparse(url)
|
||||||
|
if parsed.scheme not in {"http", "https"} or not parsed.hostname:
|
||||||
|
return False
|
||||||
|
if parsed.hostname == "localhost":
|
||||||
|
return True
|
||||||
|
address = ipaddress.ip_address(parsed.hostname)
|
||||||
|
return address.is_private or address.is_loopback
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def _load_private_image(ref: str) -> bytes:
|
||||||
|
"""Load a local/data/private-LAN image without contacting public hosts."""
|
||||||
|
ref = ref.strip()
|
||||||
|
lower = ref.lower()
|
||||||
|
if lower.startswith("data:image/"):
|
||||||
|
_, separator, payload = ref.partition(",")
|
||||||
|
if not separator:
|
||||||
|
raise ValueError("invalid image data URI")
|
||||||
|
data = base64.b64decode(payload, validate=True)
|
||||||
|
elif lower.startswith(("http://", "https://")):
|
||||||
|
if not _private_destination(ref):
|
||||||
|
raise ValueError("public reference-image URLs are blocked")
|
||||||
|
request = urllib.request.Request(
|
||||||
|
ref, headers={"User-Agent": "Hermes-Athena-Image/2.0"})
|
||||||
|
with urllib.request.urlopen(request, timeout=60) as response:
|
||||||
|
data = response.read(_MAX_IMAGE_BYTES + 1)
|
||||||
|
else:
|
||||||
|
from agent.file_safety import raise_if_read_blocked
|
||||||
|
raise_if_read_blocked(ref)
|
||||||
|
with open(ref, "rb") as image_file:
|
||||||
|
data = image_file.read(_MAX_IMAGE_BYTES + 1)
|
||||||
|
if not data or len(data) > _MAX_IMAGE_BYTES:
|
||||||
|
raise ValueError("reference image is empty or exceeds 20 MiB")
|
||||||
|
return data
|
||||||
|
|
||||||
|
|
||||||
|
class AthenaLocalImageProvider(ImageGenProvider):
|
||||||
|
@property
|
||||||
|
def name(self) -> str:
|
||||||
|
return "athena-local"
|
||||||
|
|
||||||
|
@property
|
||||||
|
def display_name(self) -> str:
|
||||||
|
return "Athena Local (FLUX.2 Klein)"
|
||||||
|
|
||||||
|
def is_available(self) -> bool:
|
||||||
|
return bool(_api_key()) and _private_destination(_base_url())
|
||||||
|
|
||||||
|
def list_models(self) -> List[Dict[str, Any]]:
|
||||||
|
return [{
|
||||||
|
"id": _model(),
|
||||||
|
"display": "FLUX.2 Klein 9B FP8 Beta on Athena",
|
||||||
|
"speed": "local",
|
||||||
|
"strengths": "Private local generation and multi-reference editing",
|
||||||
|
"price": "local / no cloud",
|
||||||
|
}]
|
||||||
|
|
||||||
|
def default_model(self) -> Optional[str]:
|
||||||
|
return _model()
|
||||||
|
|
||||||
|
def capabilities(self) -> Dict[str, Any]:
|
||||||
|
return {"modalities": ["text", "image"], "max_reference_images": 3}
|
||||||
|
|
||||||
|
def get_setup_schema(self) -> Dict[str, Any]:
|
||||||
|
return {
|
||||||
|
"name": "Athena Local (FLUX.2 Klein)",
|
||||||
|
"badge": "local",
|
||||||
|
"tag": "Private image generation on Athena; public endpoints are rejected",
|
||||||
|
"env_vars": [
|
||||||
|
{"key": "ATHENA_IMAGE_API_KEY", "prompt": "Athena router API key"},
|
||||||
|
{
|
||||||
|
"key": "ATHENA_IMAGE_BASE_URL",
|
||||||
|
"prompt": "Athena image API base URL",
|
||||||
|
"default": _DEFAULT_BASE_URL,
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
|
||||||
|
def generate(
|
||||||
|
self,
|
||||||
|
prompt: str,
|
||||||
|
aspect_ratio: str = DEFAULT_ASPECT_RATIO,
|
||||||
|
*,
|
||||||
|
image_url: Optional[str] = None,
|
||||||
|
reference_image_urls: Optional[List[str]] = None,
|
||||||
|
**kwargs: Any,
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
clean_prompt = (prompt or "").strip()
|
||||||
|
aspect = resolve_aspect_ratio(aspect_ratio)
|
||||||
|
base_url = _base_url()
|
||||||
|
|
||||||
|
if not clean_prompt:
|
||||||
|
return error_response(
|
||||||
|
error="Prompt is required.", error_type="invalid_argument",
|
||||||
|
provider=self.name, aspect_ratio=aspect)
|
||||||
|
if not _private_destination(base_url):
|
||||||
|
return error_response(
|
||||||
|
error=("Athena image endpoint is not a private-network "
|
||||||
|
"destination; request blocked."),
|
||||||
|
error_type="unsafe_destination", provider=self.name,
|
||||||
|
prompt=clean_prompt, aspect_ratio=aspect)
|
||||||
|
|
||||||
|
model = _model()
|
||||||
|
api_key = _api_key()
|
||||||
|
if not api_key:
|
||||||
|
return error_response(
|
||||||
|
error=("No Athena router key is configured. Set "
|
||||||
|
"ATHENA_IMAGE_API_KEY or reuse "
|
||||||
|
"HERMES_CUSTOM_192_168_1_212_8081_API_KEY."),
|
||||||
|
error_type="auth_required", provider=self.name, model=model,
|
||||||
|
prompt=clean_prompt, aspect_ratio=aspect)
|
||||||
|
|
||||||
|
sources: List[str] = []
|
||||||
|
if isinstance(image_url, str) and image_url.strip():
|
||||||
|
sources.append(image_url.strip())
|
||||||
|
sources.extend(normalize_reference_images(reference_image_urls) or [])
|
||||||
|
sources = sources[:4]
|
||||||
|
try:
|
||||||
|
encoded_sources = [
|
||||||
|
base64.b64encode(_load_private_image(source)).decode("ascii")
|
||||||
|
for source in sources
|
||||||
|
]
|
||||||
|
except Exception as exc:
|
||||||
|
return error_response(
|
||||||
|
error=f"Reference image could not be loaded locally: {exc}",
|
||||||
|
error_type="io_error", provider=self.name, model=model,
|
||||||
|
prompt=clean_prompt, aspect_ratio=aspect)
|
||||||
|
|
||||||
|
request_data = {
|
||||||
|
"model": model,
|
||||||
|
"prompt": clean_prompt,
|
||||||
|
"size": _SIZES[aspect],
|
||||||
|
"n": 1,
|
||||||
|
"quality": "standard",
|
||||||
|
"steps": 4,
|
||||||
|
"guidance": 1.0,
|
||||||
|
"response_format": "b64_json",
|
||||||
|
}
|
||||||
|
endpoint = "generations"
|
||||||
|
if encoded_sources:
|
||||||
|
endpoint = "edits"
|
||||||
|
request_data["image_b64"] = encoded_sources[0]
|
||||||
|
request_data["reference_images_b64"] = encoded_sources[1:]
|
||||||
|
request = urllib.request.Request(
|
||||||
|
f"{base_url}/images/{endpoint}",
|
||||||
|
data=json.dumps(request_data).encode("utf-8"), method="POST",
|
||||||
|
headers={
|
||||||
|
"Authorization": f"Bearer {api_key}",
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
"Accept": "application/json",
|
||||||
|
})
|
||||||
|
|
||||||
|
try:
|
||||||
|
with urllib.request.urlopen(request, timeout=900) as response:
|
||||||
|
result = json.load(response)
|
||||||
|
except urllib.error.HTTPError as exc:
|
||||||
|
try:
|
||||||
|
detail = exc.read(4096).decode("utf-8", errors="replace")
|
||||||
|
except Exception:
|
||||||
|
detail = ""
|
||||||
|
return error_response(
|
||||||
|
error=f"Athena image request failed (HTTP {exc.code}): {detail[:500]}",
|
||||||
|
error_type="api_error", provider=self.name, model=model,
|
||||||
|
prompt=clean_prompt, aspect_ratio=aspect)
|
||||||
|
except (OSError, TimeoutError, ValueError, json.JSONDecodeError) as exc:
|
||||||
|
return error_response(
|
||||||
|
error=f"Athena image request failed: {exc}",
|
||||||
|
error_type="connection_error", provider=self.name, model=model,
|
||||||
|
prompt=clean_prompt, aspect_ratio=aspect)
|
||||||
|
|
||||||
|
items = result.get("data") if isinstance(result, dict) else None
|
||||||
|
first = items[0] if isinstance(items, list) and items else None
|
||||||
|
b64_data = first.get("b64_json") if isinstance(first, dict) else None
|
||||||
|
if not isinstance(b64_data, str) or not b64_data:
|
||||||
|
return error_response(
|
||||||
|
error="Athena returned no image data.",
|
||||||
|
error_type="empty_response", provider=self.name, model=model,
|
||||||
|
prompt=clean_prompt, aspect_ratio=aspect)
|
||||||
|
|
||||||
|
try:
|
||||||
|
saved = save_b64_image(b64_data, prefix="athena_flux2")
|
||||||
|
except Exception as exc:
|
||||||
|
return error_response(
|
||||||
|
error=f"Generated image could not be saved: {exc}",
|
||||||
|
error_type="io_error", provider=self.name, model=model,
|
||||||
|
prompt=clean_prompt, aspect_ratio=aspect)
|
||||||
|
|
||||||
|
return success_response(
|
||||||
|
image=str(saved), model=model, prompt=clean_prompt,
|
||||||
|
aspect_ratio=aspect, provider=self.name,
|
||||||
|
modality="image" if encoded_sources else "text",
|
||||||
|
extra={"size": _SIZES[aspect], "local_only": True,
|
||||||
|
"reference_images": len(encoded_sources)})
|
||||||
|
|
||||||
|
|
||||||
|
def register(ctx) -> None:
|
||||||
|
ctx.register_image_gen_provider(AthenaLocalImageProvider())
|
||||||
@@ -0,0 +1,7 @@
|
|||||||
|
name: athena-local
|
||||||
|
version: 2.2.0
|
||||||
|
description: "Local-only FLUX.2 Klein 9B FP8 beta generation and editing through Athena."
|
||||||
|
author: Michael
|
||||||
|
kind: backend
|
||||||
|
requires_env:
|
||||||
|
- HERMES_CUSTOM_192_168_1_212_8081_API_KEY
|
||||||
@@ -13,9 +13,11 @@ RUN apt-get update && apt-get install -y --no-install-recommends python3.12-venv
|
|||||||
"transformers==${TRANSFORMERS_VERSION}" \
|
"transformers==${TRANSFORMERS_VERSION}" \
|
||||||
"accelerate==${ACCELERATE_VERSION}" \
|
"accelerate==${ACCELERATE_VERSION}" \
|
||||||
"huggingface-hub==${HF_HUB_VERSION}" \
|
"huggingface-hub==${HF_HUB_VERSION}" \
|
||||||
|
"nvidia-modelopt==0.46.0" bitsandbytes \
|
||||||
sentencepiece protobuf safetensors pillow && \
|
sentencepiece protobuf safetensors pillow && \
|
||||||
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker
|
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker
|
||||||
|
|
||||||
COPY image_worker.py /app/image_worker.py
|
COPY image_worker.py /app/image_worker.py
|
||||||
|
COPY image_worker_9b.py /app/image_worker_9b.py
|
||||||
USER 10002:10002
|
USER 10002:10002
|
||||||
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker.py"]
|
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker_9b.py"]
|
||||||
|
|||||||
@@ -0,0 +1,282 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Private FLUX.2 Klein 9B FP8 beta worker for Athena's two GPUs.
|
||||||
|
|
||||||
|
The FP8 diffusion transformer runs on the RTX 5080. A Qwen3-8B NF4 text
|
||||||
|
encoder runs on the RTX 3060 while the profile controller temporarily pauses
|
||||||
|
Qwen3-TTS. The transformer and encoder are released before VAE decoding so
|
||||||
|
the 1024px decoder has sufficient workspace on the RTX 5080.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import gc
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import signal
|
||||||
|
import time
|
||||||
|
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||||
|
from pathlib import Path
|
||||||
|
from types import MethodType
|
||||||
|
|
||||||
|
HOST = os.environ.get("WORKER_HOST", "0.0.0.0")
|
||||||
|
PORT = int(os.environ.get("WORKER_PORT", "8086"))
|
||||||
|
TOKEN = os.environ.get("WORKER_TOKEN", "").strip()
|
||||||
|
COMPONENT_DIR = os.environ.get("FLUX_COMPONENT_DIR", "/models/components")
|
||||||
|
TRANSFORMER_FILE = os.environ.get(
|
||||||
|
"FLUX_TRANSFORMER_FILE", "/models/fp8/flux-2-klein-9b-fp8.safetensors")
|
||||||
|
OUTPUT_DIR = Path(os.environ.get("IMAGE_DIR", "/data/images")).resolve()
|
||||||
|
ACTIVE = False
|
||||||
|
|
||||||
|
os.environ.setdefault("DIFFUSERS_VERBOSITY", "error")
|
||||||
|
os.environ.setdefault("TRANSFORMERS_VERBOSITY", "error")
|
||||||
|
os.environ.setdefault("HF_HUB_DISABLE_PROGRESS_BARS", "1")
|
||||||
|
os.environ.setdefault("TOKENIZERS_PARALLELISM", "false")
|
||||||
|
|
||||||
|
if len(TOKEN) < 32:
|
||||||
|
raise RuntimeError("WORKER_TOKEN is missing or too short")
|
||||||
|
|
||||||
|
signal.signal(signal.SIGTERM, lambda *_: os._exit(0))
|
||||||
|
|
||||||
|
|
||||||
|
def _devices(torch):
|
||||||
|
if torch.cuda.device_count() != 2:
|
||||||
|
raise RuntimeError("FLUX 9B beta requires exactly two visible CUDA GPUs")
|
||||||
|
totals = {i: torch.cuda.get_device_properties(i).total_memory
|
||||||
|
for i in range(torch.cuda.device_count())}
|
||||||
|
transformer_index = max(totals, key=totals.get)
|
||||||
|
encoder_index = min(totals, key=totals.get)
|
||||||
|
return (transformer_index, encoder_index,
|
||||||
|
torch.device(f"cuda:{transformer_index}"),
|
||||||
|
torch.device(f"cuda:{encoder_index}"))
|
||||||
|
|
||||||
|
|
||||||
|
def _install_fp8_converter():
|
||||||
|
import diffusers.loaders.single_file_model as single_file_model
|
||||||
|
|
||||||
|
original = single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
|
||||||
|
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"]
|
||||||
|
scales = {}
|
||||||
|
double_map = {
|
||||||
|
"img_attn.proj": "attn.to_out.0",
|
||||||
|
"img_mlp.0": "ff.linear_in",
|
||||||
|
"img_mlp.2": "ff.linear_out",
|
||||||
|
"txt_attn.proj": "attn.to_add_out",
|
||||||
|
"txt_mlp.0": "ff_context.linear_in",
|
||||||
|
"txt_mlp.2": "ff_context.linear_out",
|
||||||
|
}
|
||||||
|
single_map = {
|
||||||
|
"linear1": "attn.to_qkv_mlp_proj",
|
||||||
|
"linear2": "attn.to_out",
|
||||||
|
}
|
||||||
|
|
||||||
|
def record(key, value):
|
||||||
|
parts = key.split(".")
|
||||||
|
scale_name, block = parts[-1], parts[1]
|
||||||
|
within = ".".join(parts[2:-1])
|
||||||
|
if parts[0] == "double_blocks":
|
||||||
|
if within == "img_attn.qkv":
|
||||||
|
targets = ("attn.to_q", "attn.to_k", "attn.to_v")
|
||||||
|
elif within == "txt_attn.qkv":
|
||||||
|
targets = ("attn.add_q_proj", "attn.add_k_proj",
|
||||||
|
"attn.add_v_proj")
|
||||||
|
else:
|
||||||
|
targets = (double_map[within],)
|
||||||
|
prefix = f"transformer_blocks.{block}"
|
||||||
|
elif parts[0] == "single_blocks":
|
||||||
|
targets = (single_map[within],)
|
||||||
|
prefix = f"single_transformer_blocks.{block}"
|
||||||
|
else:
|
||||||
|
raise ValueError(f"unexpected FP8 scale key: {key}")
|
||||||
|
for target in targets:
|
||||||
|
scales.setdefault(f"{prefix}.{target}", {})[scale_name] = value.clone()
|
||||||
|
|
||||||
|
def convert(checkpoint, **kwargs):
|
||||||
|
scales.clear()
|
||||||
|
for key in list(checkpoint):
|
||||||
|
if key.endswith((".input_scale", ".weight_scale")):
|
||||||
|
record(key, checkpoint.pop(key))
|
||||||
|
return original(checkpoint=checkpoint, **kwargs)
|
||||||
|
|
||||||
|
single_file_model.SINGLE_FILE_LOADABLE_CLASSES[
|
||||||
|
"Flux2Transformer2DModel"]["checkpoint_mapping_fn"] = convert
|
||||||
|
return scales
|
||||||
|
|
||||||
|
|
||||||
|
def _fp8_forward(torch, module, inputs):
|
||||||
|
shape = inputs.shape
|
||||||
|
input_fp8 = ((inputs / module._fp8_input_scale)
|
||||||
|
.clamp(torch.finfo(torch.float8_e4m3fn).min,
|
||||||
|
torch.finfo(torch.float8_e4m3fn).max)
|
||||||
|
.to(torch.float8_e4m3fn).reshape(-1, shape[-1]))
|
||||||
|
output = torch._scaled_mm(
|
||||||
|
input_fp8,
|
||||||
|
module.weight.reshape(-1, module.weight.shape[-1]).t(),
|
||||||
|
scale_a=module._fp8_input_scale,
|
||||||
|
scale_b=module._fp8_weight_scale,
|
||||||
|
bias=module.bias,
|
||||||
|
out_dtype=inputs.dtype,
|
||||||
|
use_fast_accum=True,
|
||||||
|
)
|
||||||
|
return output.reshape(*shape[:-1], output.shape[-1])
|
||||||
|
|
||||||
|
|
||||||
|
def generate(data: dict) -> dict:
|
||||||
|
global ACTIVE
|
||||||
|
import torch
|
||||||
|
from diffusers import (Flux2KleinPipeline, Flux2Transformer2DModel,
|
||||||
|
NVIDIAModelOptConfig)
|
||||||
|
from modelopt.torch.opt import enable_huggingface_checkpointing
|
||||||
|
from modelopt.torch.quantization.config import FP8_DEFAULT_CFG
|
||||||
|
from PIL import Image
|
||||||
|
from transformers import BitsAndBytesConfig, Qwen3ForCausalLM
|
||||||
|
|
||||||
|
prompt, filename = data.get("prompt"), data.get("filename")
|
||||||
|
if not isinstance(prompt, str) or not prompt.strip() or len(prompt) > 8000:
|
||||||
|
raise ValueError("invalid prompt")
|
||||||
|
if (not isinstance(filename, str) or Path(filename).name != filename
|
||||||
|
or not filename.endswith(".png")):
|
||||||
|
raise ValueError("invalid filename")
|
||||||
|
width, height = int(data.get("width", 1024)), int(data.get("height", 1024))
|
||||||
|
if (width, height) != (1024, 1024):
|
||||||
|
raise ValueError("FLUX 9B beta currently supports only 1024x1024")
|
||||||
|
if int(data.get("steps", 4)) != 4 or float(data.get("guidance", 1.0)) != 1.0:
|
||||||
|
raise ValueError("FLUX 9B beta requires steps=4 and guidance=1.0")
|
||||||
|
|
||||||
|
source_files = data.get("source_files") or []
|
||||||
|
if not isinstance(source_files, list) or len(source_files) > 4:
|
||||||
|
raise ValueError("invalid source image list")
|
||||||
|
source_images = []
|
||||||
|
for name in source_files:
|
||||||
|
if not isinstance(name, str) or Path(name).name != name:
|
||||||
|
raise ValueError("invalid source image filename")
|
||||||
|
source = (OUTPUT_DIR / name).resolve()
|
||||||
|
if source.parent != OUTPUT_DIR or not source.is_file():
|
||||||
|
raise ValueError("source image not found")
|
||||||
|
with Image.open(source) as opened:
|
||||||
|
source_images.append(opened.convert("RGB"))
|
||||||
|
|
||||||
|
started = time.monotonic()
|
||||||
|
ACTIVE = True
|
||||||
|
transformer = text_encoder = pipe = latent = decoded = image = None
|
||||||
|
try:
|
||||||
|
enable_huggingface_checkpointing()
|
||||||
|
scales = _install_fp8_converter()
|
||||||
|
tx_index, enc_index, tx_device, enc_device = _devices(torch)
|
||||||
|
quantization = NVIDIAModelOptConfig(
|
||||||
|
quant_type="FP8", weight_only=False,
|
||||||
|
modelopt_config=FP8_DEFAULT_CFG)
|
||||||
|
transformer = Flux2Transformer2DModel.from_single_file(
|
||||||
|
TRANSFORMER_FILE, config=COMPONENT_DIR, subfolder="transformer",
|
||||||
|
quantization_config=quantization, torch_dtype=torch.bfloat16,
|
||||||
|
device_map={"": tx_index}, local_files_only=True)
|
||||||
|
patched = 0
|
||||||
|
for module_name, module in transformer.named_modules():
|
||||||
|
if module_name not in scales:
|
||||||
|
continue
|
||||||
|
module.register_buffer("_fp8_input_scale",
|
||||||
|
scales[module_name]["input_scale"])
|
||||||
|
module.register_buffer("_fp8_weight_scale",
|
||||||
|
scales[module_name]["weight_scale"])
|
||||||
|
module.forward = MethodType(
|
||||||
|
lambda self, inputs: _fp8_forward(torch, self, inputs), module)
|
||||||
|
patched += 1
|
||||||
|
if patched != len(scales):
|
||||||
|
raise RuntimeError(f"patched only {patched} of {len(scales)} FP8 layers")
|
||||||
|
transformer.to(tx_device)
|
||||||
|
|
||||||
|
text_encoder = Qwen3ForCausalLM.from_pretrained(
|
||||||
|
os.path.join(COMPONENT_DIR, "text_encoder"),
|
||||||
|
torch_dtype=torch.bfloat16, low_cpu_mem_usage=True,
|
||||||
|
quantization_config=BitsAndBytesConfig(
|
||||||
|
load_in_4bit=True, bnb_4bit_quant_type="nf4",
|
||||||
|
bnb_4bit_compute_dtype=torch.bfloat16,
|
||||||
|
bnb_4bit_use_double_quant=True),
|
||||||
|
device_map={"": enc_index}, local_files_only=True)
|
||||||
|
pipe = Flux2KleinPipeline.from_pretrained(
|
||||||
|
COMPONENT_DIR, transformer=transformer, text_encoder=text_encoder,
|
||||||
|
torch_dtype=torch.bfloat16, local_files_only=True)
|
||||||
|
pipe.vae.enable_slicing()
|
||||||
|
pipe.vae.enable_tiling()
|
||||||
|
pipe.vae.to(tx_device)
|
||||||
|
loaded = time.monotonic() - started
|
||||||
|
|
||||||
|
prompt_embeds, _ = pipe.encode_prompt(
|
||||||
|
prompt.strip(), device=enc_device, max_sequence_length=128)
|
||||||
|
prompt_embeds = prompt_embeds.to(tx_device)
|
||||||
|
pipe.text_encoder = None
|
||||||
|
seed = data.get("seed")
|
||||||
|
generator = None if seed is None else torch.Generator(
|
||||||
|
device=tx_device).manual_seed(int(seed))
|
||||||
|
kwargs = {
|
||||||
|
"prompt": None, "prompt_embeds": prompt_embeds,
|
||||||
|
"height": height, "width": width, "num_inference_steps": 4,
|
||||||
|
"guidance_scale": 1.0, "generator": generator,
|
||||||
|
"output_type": "latent",
|
||||||
|
}
|
||||||
|
if source_images:
|
||||||
|
kwargs["image"] = (source_images[0] if len(source_images) == 1
|
||||||
|
else source_images)
|
||||||
|
latent = pipe(**kwargs).images
|
||||||
|
|
||||||
|
pipe.transformer = None
|
||||||
|
del transformer, text_encoder, prompt_embeds, generator
|
||||||
|
transformer = text_encoder = None
|
||||||
|
gc.collect()
|
||||||
|
torch.cuda.empty_cache()
|
||||||
|
latent = latent.to(device=tx_device, dtype=pipe.vae.dtype)
|
||||||
|
decoded = pipe.vae.decode(latent, return_dict=False)[0]
|
||||||
|
image = pipe.image_processor.postprocess(
|
||||||
|
decoded.detach(), output_type="pil")[0]
|
||||||
|
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||||
|
image.save(OUTPUT_DIR / filename)
|
||||||
|
return {"status": "ok", "filename": filename,
|
||||||
|
"seconds": round(time.monotonic() - started, 3),
|
||||||
|
"load_seconds": round(loaded, 3),
|
||||||
|
"model": "FLUX.2-klein-9B-fp8-beta"}
|
||||||
|
finally:
|
||||||
|
for value in (image, decoded, latent, pipe, text_encoder, transformer):
|
||||||
|
if value is not None:
|
||||||
|
del value
|
||||||
|
gc.collect()
|
||||||
|
torch.cuda.empty_cache()
|
||||||
|
ACTIVE = False
|
||||||
|
|
||||||
|
|
||||||
|
class Handler(BaseHTTPRequestHandler):
|
||||||
|
def log_message(self, fmt: str, *args: object) -> None:
|
||||||
|
print(f"[flux9b-beta] {self.client_address[0]} {fmt % args}", flush=True)
|
||||||
|
|
||||||
|
def reply(self, status: int, payload: dict) -> None:
|
||||||
|
body = json.dumps(payload, separators=(",", ":")).encode()
|
||||||
|
self.send_response(status)
|
||||||
|
self.send_header("Content-Type", "application/json")
|
||||||
|
self.send_header("Content-Length", str(len(body)))
|
||||||
|
self.end_headers()
|
||||||
|
self.wfile.write(body)
|
||||||
|
|
||||||
|
def do_GET(self) -> None: # noqa: N802
|
||||||
|
if self.path == "/health":
|
||||||
|
self.reply(200, {"status": "ok", "model_loaded": ACTIVE,
|
||||||
|
"model": "FLUX.2-klein-9B-fp8-beta"})
|
||||||
|
else:
|
||||||
|
self.reply(404, {"error": "not found"})
|
||||||
|
|
||||||
|
def do_POST(self) -> None: # noqa: N802
|
||||||
|
if self.headers.get("Authorization", "") != f"Bearer {TOKEN}":
|
||||||
|
self.reply(401, {"error": "unauthorized"})
|
||||||
|
return
|
||||||
|
if self.path != "/generate":
|
||||||
|
self.reply(404, {"error": "not found"})
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
length = int(self.headers.get("Content-Length", "0"))
|
||||||
|
if length < 2 or length > 16384:
|
||||||
|
raise ValueError("invalid request size")
|
||||||
|
self.reply(200, generate(json.loads(self.rfile.read(length))))
|
||||||
|
except Exception as exc:
|
||||||
|
print(f"[flux9b-beta] generation failed: {type(exc).__name__}: "
|
||||||
|
f"{str(exc)[:1000]}", flush=True)
|
||||||
|
self.reply(500, {"status": "error", "message": str(exc)})
|
||||||
|
|
||||||
|
|
||||||
|
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||||
@@ -25,6 +25,8 @@ ALLOWED = tuple(x.strip() for x in os.environ.get(
|
|||||||
LABEL_KEY = "com.mike-ai.llama-profile"
|
LABEL_KEY = "com.mike-ai.llama-profile"
|
||||||
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
|
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
|
||||||
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
|
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
|
||||||
|
TTS_LABEL_KEY = "com.mike-ai.tts-worker"
|
||||||
|
TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3")
|
||||||
LOCK = threading.Lock()
|
LOCK = threading.Lock()
|
||||||
log = logging.getLogger("profile-controller")
|
log = logging.getLogger("profile-controller")
|
||||||
|
|
||||||
@@ -77,6 +79,15 @@ def image_container() -> dict:
|
|||||||
return matches[0]
|
return matches[0]
|
||||||
|
|
||||||
|
|
||||||
|
def tts_container() -> dict:
|
||||||
|
matches = [item for item in labelled_containers(TTS_LABEL_KEY)
|
||||||
|
if item.get("Labels", {}).get(TTS_LABEL_KEY) == TTS_WORKER]
|
||||||
|
if len(matches) != 1:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"expected exactly one TTS worker {TTS_WORKER!r}, found {len(matches)}")
|
||||||
|
return matches[0]
|
||||||
|
|
||||||
|
|
||||||
def stop_container(item: dict, timeout: int = 120) -> None:
|
def stop_container(item: dict, timeout: int = 120) -> None:
|
||||||
if item.get("State") != "running":
|
if item.get("State") != "running":
|
||||||
return
|
return
|
||||||
@@ -85,6 +96,14 @@ def stop_container(item: dict, timeout: int = 120) -> None:
|
|||||||
raise RuntimeError(f"failed to stop container: HTTP {status}")
|
raise RuntimeError(f"failed to stop container: HTTP {status}")
|
||||||
|
|
||||||
|
|
||||||
|
def start_container(item: dict) -> None:
|
||||||
|
if item.get("State") == "running":
|
||||||
|
return
|
||||||
|
status, _ = docker_request("POST", f"/containers/{item['Id']}/start")
|
||||||
|
if status not in (204, 304):
|
||||||
|
raise RuntimeError(f"failed to start container: HTTP {status}")
|
||||||
|
|
||||||
|
|
||||||
def stop_inference() -> dict:
|
def stop_inference() -> dict:
|
||||||
with LOCK:
|
with LOCK:
|
||||||
items = containers()
|
items = containers()
|
||||||
@@ -101,14 +120,17 @@ def set_image_worker(running: bool) -> dict:
|
|||||||
# The image worker may never overlap a llama profile on the 5080.
|
# The image worker may never overlap a llama profile on the 5080.
|
||||||
for profile_item in containers().values():
|
for profile_item in containers().values():
|
||||||
stop_container(profile_item)
|
stop_container(profile_item)
|
||||||
if item.get("State") != "running":
|
# The 9B beta text encoder temporarily borrows the RTX 3060 from
|
||||||
status, _ = docker_request("POST", f"/containers/{item['Id']}/start")
|
# Qwen3-TTS. The gateway retains Piper as a fallback meanwhile.
|
||||||
if status not in (204, 304):
|
stop_container(tts_container(), timeout=30)
|
||||||
raise RuntimeError(f"failed to start image worker: HTTP {status}")
|
start_container(item)
|
||||||
else:
|
else:
|
||||||
# CUDA/PyTorch may not react promptly to SIGTERM after an OOM.
|
# CUDA/PyTorch may not react promptly to SIGTERM after an OOM.
|
||||||
# Bound recovery time and let Docker issue SIGKILL afterwards.
|
# Bound recovery time and let Docker issue SIGKILL afterwards.
|
||||||
stop_container(item, timeout=20)
|
stop_container(item, timeout=20)
|
||||||
|
# TTS is restored by the following profile activation. Keeping it
|
||||||
|
# stopped here lets the router verify that both GPUs really
|
||||||
|
# released the image model before Qwen and TTS are reloaded.
|
||||||
return {"image_worker": "running" if running else "stopped"}
|
return {"image_worker": "running" if running else "stopped"}
|
||||||
|
|
||||||
|
|
||||||
@@ -126,6 +148,7 @@ def activate(profile: str) -> dict:
|
|||||||
with LOCK:
|
with LOCK:
|
||||||
# Defensive mutual exclusion even if a caller bypasses the router.
|
# Defensive mutual exclusion even if a caller bypasses the router.
|
||||||
stop_container(image_container())
|
stop_container(image_container())
|
||||||
|
start_container(tts_container())
|
||||||
items = containers()
|
items = containers()
|
||||||
missing = [name for name in ALLOWED if name not in items]
|
missing = [name for name in ALLOWED if name not in items]
|
||||||
if missing:
|
if missing:
|
||||||
|
|||||||
@@ -28,12 +28,17 @@ models:
|
|||||||
sha256: "REPLACE_AFTER_VERIFICATION"
|
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||||
image:
|
image:
|
||||||
role: image-generation
|
role: image-generation
|
||||||
source: black-forest-labs/FLUX.2-klein-4B
|
source: black-forest-labs/FLUX.2-klein-9B
|
||||||
revision: e7b7dc27f91deacad38e78976d1f2b499d76a294
|
revision: 92196c8e11f7b6cf2b7493e037d8c5345c559216
|
||||||
target: /data/models/FLUX.2-klein-4B
|
target: /data/models/FLUX.2-klein-9B-components
|
||||||
revision: "f332072aa78be7aecdf3ee76d5c247082da564a6"
|
include: model_index.json,scheduler/*,text_encoder/*,tokenizer/*,transformer/config.json,vae/*
|
||||||
xtts:
|
image_transformer:
|
||||||
|
role: image-generation-fp8-transformer
|
||||||
|
source: black-forest-labs/FLUX.2-klein-9b-fp8
|
||||||
|
revision: 902d9d510b51533e07729f19211414a3648b77d2
|
||||||
|
file: flux-2-klein-9b-fp8.safetensors
|
||||||
|
target: /data/models/FLUX.2-klein-9B-fp8
|
||||||
|
qwen3_tts:
|
||||||
role: text-to-speech
|
role: text-to-speech
|
||||||
source: coqui/XTTS-v2
|
source: Qwen/Qwen3-TTS-12Hz-1.7B-Base
|
||||||
target: /opt/mike-ai/xtts/.cache
|
target: /data/models/qwen3-tts-cache
|
||||||
revision: "PIN_EXACT_REVISION"
|
|
||||||
|
|||||||
+12
-10
@@ -20,13 +20,13 @@ Kommandos: POST /fast, /medium, /beta1, /large, /ultra,
|
|||||||
/uncensored
|
/uncensored
|
||||||
GET /status (Zustand)
|
GET /status (Zustand)
|
||||||
|
|
||||||
Bildgenerierung und Editing (FLUX.2-klein-4B):
|
Bildgenerierung und Editing (FLUX.2 Klein 9B FP8 beta):
|
||||||
POST /v1/images/generations (OpenAI-kompatibel)
|
POST /v1/images/generations (OpenAI-kompatibel)
|
||||||
POST /v1/images/edits (lokal, Referenzbilder)
|
POST /v1/images/edits (lokal, Referenzbilder)
|
||||||
GET /images (Liste)
|
GET /images (Liste)
|
||||||
GET /images/<datei> (PNG-Download)
|
GET /images/<datei> (PNG-Download)
|
||||||
|
|
||||||
Sprachausgabe (XTTS-v2, multilingual, CPU-only):
|
Sprachausgabe (Qwen3-TTS auf RTX 3060, Piper als CPU-Fallback):
|
||||||
POST /v1/audio/speech (OpenAI-kompatibel)
|
POST /v1/audio/speech (OpenAI-kompatibel)
|
||||||
GET /v1/audio/voices (verfügbare Stimmen)
|
GET /v1/audio/voices (verfügbare Stimmen)
|
||||||
|
|
||||||
@@ -40,7 +40,8 @@ Der Router leitet /v1/audio/speech und /v1/audio/transcriptions
|
|||||||
per HTTP an die Worker weiter.
|
per HTTP an die Worker weiter.
|
||||||
|
|
||||||
Der Router agiert als Modell-Orchestrator: vor der Generierung wird
|
Der Router agiert als Modell-Orchestrator: vor der Generierung wird
|
||||||
llama.cpp gestoppt, der Bild-Worker lädt FLUX.2, generiert/bearbeitet und entlädt
|
llama.cpp und Qwen3-TTS gestoppt, der Bild-Worker lädt FLUX.2 und den
|
||||||
|
Text-Encoder auf getrennte GPUs, generiert/bearbeitet und entlädt
|
||||||
das Modell wieder; danach wird das vorherige Qwen-Profil wiederher-
|
das Modell wieder; danach wird das vorherige Qwen-Profil wiederher-
|
||||||
gestellt und erst dann geantwortet (try/finally – Qwen wird auch bei
|
gestellt und erst dann geantwortet (try/finally – Qwen wird auch bei
|
||||||
Fehlgeschlagener Generierung wiederhergestellt).
|
Fehlgeschlagener Generierung wiederhergestellt).
|
||||||
@@ -126,7 +127,7 @@ DEFAULT_REASONING_EFFORT = os.environ.get(
|
|||||||
GLOBAL_SYSTEM_POLICY_FILE = os.environ.get(
|
GLOBAL_SYSTEM_POLICY_FILE = os.environ.get(
|
||||||
"GLOBAL_SYSTEM_POLICY_FILE", "").strip()
|
"GLOBAL_SYSTEM_POLICY_FILE", "").strip()
|
||||||
|
|
||||||
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 4B) ---
|
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 9B FP8) ---
|
||||||
LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service")
|
LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service")
|
||||||
SYSTEMCTL_BIN = os.environ.get("SYSTEMCTL_BIN", "systemctl")
|
SYSTEMCTL_BIN = os.environ.get("SYSTEMCTL_BIN", "systemctl")
|
||||||
IMAGE_WORKER = os.environ.get(
|
IMAGE_WORKER = os.environ.get(
|
||||||
@@ -135,7 +136,8 @@ IMAGE_PYTHON = os.environ.get(
|
|||||||
"IMAGE_PYTHON", "/opt/mike-ai/ai-profile-router/venv/bin/python")
|
"IMAGE_PYTHON", "/opt/mike-ai/ai-profile-router/venv/bin/python")
|
||||||
IMAGE_WORKER_URL = os.environ.get("IMAGE_WORKER_URL", "").rstrip("/")
|
IMAGE_WORKER_URL = os.environ.get("IMAGE_WORKER_URL", "").rstrip("/")
|
||||||
IMAGE_WORKER_TOKEN = os.environ.get("IMAGE_WORKER_TOKEN", "").strip()
|
IMAGE_WORKER_TOKEN = os.environ.get("IMAGE_WORKER_TOKEN", "").strip()
|
||||||
IMAGE_MODEL_NAME = os.environ.get("IMAGE_MODEL_NAME", "FLUX.2-klein-4B")
|
IMAGE_MODEL_NAME = os.environ.get(
|
||||||
|
"IMAGE_MODEL_NAME", "FLUX.2-klein-9B-fp8-beta")
|
||||||
IMAGE_DIR = os.environ.get(
|
IMAGE_DIR = os.environ.get(
|
||||||
"IMAGE_DIR", "/opt/mike-ai/ai-profile-router/images")
|
"IMAGE_DIR", "/opt/mike-ai/ai-profile-router/images")
|
||||||
IMAGE_WORKER_LOG = os.environ.get(
|
IMAGE_WORKER_LOG = os.environ.get(
|
||||||
@@ -163,7 +165,7 @@ IMAGE_SIZES = {
|
|||||||
"1920x1088": (1920, 1088),
|
"1920x1088": (1920, 1088),
|
||||||
"1088x1920": (1088, 1920),
|
"1088x1920": (1088, 1920),
|
||||||
}
|
}
|
||||||
# Das destillierte FLUX.2-klein-4B ist auf vier Schritte ausgelegt.
|
# Das destillierte FLUX.2 Klein 9B ist auf vier Schritte ausgelegt.
|
||||||
IMAGE_QUALITY = {"standard": 4, "high": 4}
|
IMAGE_QUALITY = {"standard": 4, "high": 4}
|
||||||
IMAGE_DEFAULT_QUALITY = "standard"
|
IMAGE_DEFAULT_QUALITY = "standard"
|
||||||
IMAGE_MAX_N = 4
|
IMAGE_MAX_N = 4
|
||||||
@@ -765,7 +767,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
|||||||
|
|
||||||
|
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
# Bildgenerierung und Editing (FLUX.2-klein-4B)
|
# Bildgenerierung und Editing (FLUX.2 Klein 9B FP8)
|
||||||
# ---------------------------------------------------------------------------
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
class _Worker:
|
class _Worker:
|
||||||
@@ -1874,7 +1876,7 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
return
|
return
|
||||||
steps = data.get("steps", IMAGE_QUALITY[quality])
|
steps = data.get("steps", IMAGE_QUALITY[quality])
|
||||||
if not isinstance(steps, int) or isinstance(steps, bool) or steps != 4:
|
if not isinstance(steps, int) or isinstance(steps, bool) or steps != 4:
|
||||||
self._send_error(400, "FLUX.2-klein-4B erfordert 'steps'=4",
|
self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'steps'=4",
|
||||||
"invalid_request_error", "invalid_steps")
|
"invalid_request_error", "invalid_steps")
|
||||||
return
|
return
|
||||||
guidance = data.get("guidance", 1.0)
|
guidance = data.get("guidance", 1.0)
|
||||||
@@ -1885,7 +1887,7 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
"invalid_request_error", "invalid_guidance")
|
"invalid_request_error", "invalid_guidance")
|
||||||
return
|
return
|
||||||
if guidance != 1.0:
|
if guidance != 1.0:
|
||||||
self._send_error(400, "FLUX.2-klein-4B erfordert 'guidance'=1.0",
|
self._send_error(400, f"{IMAGE_MODEL_NAME} erfordert 'guidance'=1.0",
|
||||||
"invalid_request_error", "invalid_guidance")
|
"invalid_request_error", "invalid_guidance")
|
||||||
return
|
return
|
||||||
|
|
||||||
@@ -1986,7 +1988,7 @@ class Handler(BaseHTTPRequestHandler):
|
|||||||
self.end_headers()
|
self.end_headers()
|
||||||
self.wfile.write(data)
|
self.wfile.write(data)
|
||||||
|
|
||||||
# ---------- Sprachausgabe (XTTS-v2) ----------
|
# ---------- Sprachausgabe (Qwen3-TTS mit Piper-Fallback) ----------
|
||||||
|
|
||||||
def _speech(self) -> None:
|
def _speech(self) -> None:
|
||||||
try:
|
try:
|
||||||
|
|||||||
Reference in New Issue
Block a user