Simplify Athena runtime and document current architecture

This commit is contained in:
Mikei386 committed 2026-08-30 22:11:26 +02:00
1 parent 9bc7d9803a
commit 0b927d47b9
56 files changed
+1276 -4712

No files matched your search

@@ -7,15 +7,15 @@ ARG HF_HUB_VERSION=1.28.0
RUN apt-get update && apt-get install -y --no-install-recommends python3.12-venv && \
rm -rf /var/lib/apt/lists/* && \
python -m venv --system-site-packages /opt/flux-venv && \
/opt/flux-venv/bin/pip install --no-cache-dir \
python -m venv --system-site-packages /opt/image-venv && \
/opt/image-venv/bin/pip install --no-cache-dir \
"diffusers==${DIFFUSERS_VERSION}" \
"transformers==${TRANSFORMERS_VERSION}" \
"accelerate==${ACCELERATE_VERSION}" \
"huggingface-hub==${HF_HUB_VERSION}" \
sentencepiece protobuf safetensors pillow && \
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin flux
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker
COPY flux_worker.py /app/flux_worker.py
COPY image_worker.py /app/image_worker.py
USER 10002:10002
ENTRYPOINT ["/opt/flux-venv/bin/python", "/app/flux_worker.py"]
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker.py"]
@@ -1,5 +1,5 @@
#!/usr/bin/env python3
"""Private FLUX.2 Klein Distilled worker used only during a GPU hot swap."""
"""Private Z-Image-Turbo worker used only during a GPU hot swap."""
from __future__ import annotations
@@ -14,7 +14,7 @@ from pathlib import Path
HOST = os.environ.get("WORKER_HOST", "0.0.0.0")
PORT = int(os.environ.get("WORKER_PORT", "8086"))
TOKEN = os.environ.get("WORKER_TOKEN", "").strip()
MODEL_DIR = os.environ.get("FLUX_MODEL_DIR", "/models/FLUX.2-klein-4B")
MODEL_DIR = os.environ.get("Z_IMAGE_MODEL_DIR", "/models/Z-Image-Turbo")
OUTPUT_DIR = Path(os.environ.get("IMAGE_DIR", "/data/images")).resolve()
PIPE = None
LOAD_SECONDS = 0.0
@@ -33,14 +33,15 @@ def load_pipeline() -> None:
if PIPE is not None:
return
import torch
from diffusers import DiffusionPipeline
from diffusers import ZImagePipeline
started = time.monotonic()
PIPE = DiffusionPipeline.from_pretrained(
MODEL_DIR, torch_dtype=torch.bfloat16)
# The full pipeline leaves too little activation headroom on a 16 GiB
# RTX 5080. Model CPU offload keeps each active component on CUDA while
# parking inactive components in system RAM between the four steps.
PIPE.enable_model_cpu_offload()
PIPE = ZImagePipeline.from_pretrained(
MODEL_DIR, torch_dtype=torch.bfloat16, low_cpu_mem_usage=False)
# The Qwen text encoder and the DiT do not fit together in the usable
# 16 GiB of the RTX 5080. Sequential offload keeps only the active
# submodule on CUDA. This is slower than a fully resident pipeline, but
# deterministic and leaves the RTX 3060 available for XTTS.
PIPE.enable_sequential_cpu_offload()
if hasattr(PIPE, "enable_vae_slicing"):
PIPE.enable_vae_slicing()
if hasattr(PIPE, "enable_vae_tiling"):
@@ -61,16 +62,16 @@ def generate(data: dict) -> dict:
if (width, height) not in {(1024, 1024), (1536, 1024), (1024, 1536),
(1920, 1088), (1088, 1920)}:
raise ValueError("unsupported image size")
steps = int(data.get("steps", 4))
guidance = float(data.get("guidance", 1.0))
if steps != 4 or guidance != 1.0:
raise ValueError("distilled FLUX.2 Klein requires steps=4 and guidance=1.0")
steps = int(data.get("steps", 9))
guidance = float(data.get("guidance", 0.0))
if steps != 9 or guidance != 0.0:
raise ValueError("Z-Image-Turbo requires steps=9 and guidance=0.0")
seed = data.get("seed")
generator = None if seed is None else torch.Generator(device="cuda").manual_seed(int(seed))
load_pipeline()
started = time.monotonic()
image = PIPE(prompt=prompt, height=height, width=width,
num_inference_steps=4, guidance_scale=1.0,
num_inference_steps=9, guidance_scale=0.0,
generator=generator).images[0]
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
output = OUTPUT_DIR / filename
@@ -83,7 +84,7 @@ def generate(data: dict) -> dict:
class Handler(BaseHTTPRequestHandler):
def log_message(self, fmt: str, *args: object) -> None:
# Never log request bodies/prompts.
print(f"[flux-worker] {self.client_address[0]} {fmt % args}", flush=True)
print(f"[z-image-worker] {self.client_address[0]} {fmt % args}", flush=True)
def reply(self, status: int, payload: dict) -> None:
body = json.dumps(payload, separators=(",", ":")).encode()
@@ -112,7 +113,7 @@ class Handler(BaseHTTPRequestHandler):
raise ValueError("invalid request size")
self.reply(200, generate(json.loads(self.rfile.read(length))))
except Exception as exc:
print(f"[flux-worker] generation failed: "
print(f"[z-image-worker] generation failed: "
f"{type(exc).__name__}: {str(exc)[:1000]}", flush=True)
self.reply(400, {"status": "error", "message": str(exc)})
@@ -24,7 +24,7 @@ ALLOWED = tuple(x.strip() for x in os.environ.get(
"ALLOWED_PROFILES", "fast,medium,large,ultra,uncensored,experimental").split(",") if x.strip())
LABEL_KEY = "com.mike-ai.llama-profile"
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "flux")
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
LOCK = threading.Lock()
log = logging.getLogger("profile-controller")
@@ -98,7 +98,7 @@ def set_image_worker(running: bool) -> dict:
with LOCK:
item = image_container()
if running:
# A FLUX worker may never overlap a llama profile on the 5080.
# The image worker may never overlap a llama profile on the 5080.
for profile_item in containers().values():
stop_container(profile_item)
if item.get("State") != "running":
@@ -64,7 +64,7 @@ wg_ipv4=$(ip -4 -o address show dev wg0 | awk 'NR == 1 { split($4, address, "/")
# The VPN is Athena's normal application network. Nothing below is published
# on the physical university interface: every listener is bound inside this
# namespace to the Fritzbox-assigned WireGuard address. Clients on the home
# VPN may use OpenWebUI, the router and every useful MCP directly.
# VPN clients may use the router, speech services and Athena operator directly.
proxy_pids=""
start_proxy() {
listen_port=$1
@@ -74,29 +74,10 @@ start_proxy() {
}
start_proxy 22 172.30.10.1:22
start_proxy 8080 open-webui:8080
start_proxy 8081 router:8081
start_proxy 8085 tts-gateway:8085
start_proxy 8091 piper:8085
start_proxy 8092 xtts:80
start_proxy 9119 hermes:9119
start_proxy 8642 hermes:8642
# MCP endpoints. Optional services keep their listener even while stopped and
# begin working automatically as soon as their container is started.
start_proxy 8202 mcp-athena-operator:8000
# Portable general web MCP for Pi, Hermes and other clients. OpenWebUI uses
# its native broad search by default; both paths are site-agnostic.
start_proxy 8203 tinysearch:8000
start_proxy 8204 mcp-github:8000
start_proxy 8205 mcp-homeassistant:8000
start_proxy 8206 mcp-arr:8000
start_proxy 8207 mcp-navidrome:3000
start_proxy 8208 mcp-unraid-ssh:8000
# Search backends are also directly available for diagnostics and alternative
# clients. Normal chat clients should prefer the MCP endpoint on 8203.
start_proxy 8210 searxng:8080
start_proxy 8211 tinysearch:8000
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')