Simplify Athena runtime and document current architecture
This commit is contained in:
@@ -7,15 +7,15 @@ ARG HF_HUB_VERSION=1.28.0
|
||||
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends python3.12-venv && \
|
||||
rm -rf /var/lib/apt/lists/* && \
|
||||
python -m venv --system-site-packages /opt/flux-venv && \
|
||||
/opt/flux-venv/bin/pip install --no-cache-dir \
|
||||
python -m venv --system-site-packages /opt/image-venv && \
|
||||
/opt/image-venv/bin/pip install --no-cache-dir \
|
||||
"diffusers==${DIFFUSERS_VERSION}" \
|
||||
"transformers==${TRANSFORMERS_VERSION}" \
|
||||
"accelerate==${ACCELERATE_VERSION}" \
|
||||
"huggingface-hub==${HF_HUB_VERSION}" \
|
||||
sentencepiece protobuf safetensors pillow && \
|
||||
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin flux
|
||||
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin image-worker
|
||||
|
||||
COPY flux_worker.py /app/flux_worker.py
|
||||
COPY image_worker.py /app/image_worker.py
|
||||
USER 10002:10002
|
||||
ENTRYPOINT ["/opt/flux-venv/bin/python", "/app/flux_worker.py"]
|
||||
ENTRYPOINT ["/opt/image-venv/bin/python", "/app/image_worker.py"]
|
||||
+17
-16
@@ -1,5 +1,5 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Private FLUX.2 Klein Distilled worker used only during a GPU hot swap."""
|
||||
"""Private Z-Image-Turbo worker used only during a GPU hot swap."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -14,7 +14,7 @@ from pathlib import Path
|
||||
HOST = os.environ.get("WORKER_HOST", "0.0.0.0")
|
||||
PORT = int(os.environ.get("WORKER_PORT", "8086"))
|
||||
TOKEN = os.environ.get("WORKER_TOKEN", "").strip()
|
||||
MODEL_DIR = os.environ.get("FLUX_MODEL_DIR", "/models/FLUX.2-klein-4B")
|
||||
MODEL_DIR = os.environ.get("Z_IMAGE_MODEL_DIR", "/models/Z-Image-Turbo")
|
||||
OUTPUT_DIR = Path(os.environ.get("IMAGE_DIR", "/data/images")).resolve()
|
||||
PIPE = None
|
||||
LOAD_SECONDS = 0.0
|
||||
@@ -33,14 +33,15 @@ def load_pipeline() -> None:
|
||||
if PIPE is not None:
|
||||
return
|
||||
import torch
|
||||
from diffusers import DiffusionPipeline
|
||||
from diffusers import ZImagePipeline
|
||||
started = time.monotonic()
|
||||
PIPE = DiffusionPipeline.from_pretrained(
|
||||
MODEL_DIR, torch_dtype=torch.bfloat16)
|
||||
# The full pipeline leaves too little activation headroom on a 16 GiB
|
||||
# RTX 5080. Model CPU offload keeps each active component on CUDA while
|
||||
# parking inactive components in system RAM between the four steps.
|
||||
PIPE.enable_model_cpu_offload()
|
||||
PIPE = ZImagePipeline.from_pretrained(
|
||||
MODEL_DIR, torch_dtype=torch.bfloat16, low_cpu_mem_usage=False)
|
||||
# The Qwen text encoder and the DiT do not fit together in the usable
|
||||
# 16 GiB of the RTX 5080. Sequential offload keeps only the active
|
||||
# submodule on CUDA. This is slower than a fully resident pipeline, but
|
||||
# deterministic and leaves the RTX 3060 available for XTTS.
|
||||
PIPE.enable_sequential_cpu_offload()
|
||||
if hasattr(PIPE, "enable_vae_slicing"):
|
||||
PIPE.enable_vae_slicing()
|
||||
if hasattr(PIPE, "enable_vae_tiling"):
|
||||
@@ -61,16 +62,16 @@ def generate(data: dict) -> dict:
|
||||
if (width, height) not in {(1024, 1024), (1536, 1024), (1024, 1536),
|
||||
(1920, 1088), (1088, 1920)}:
|
||||
raise ValueError("unsupported image size")
|
||||
steps = int(data.get("steps", 4))
|
||||
guidance = float(data.get("guidance", 1.0))
|
||||
if steps != 4 or guidance != 1.0:
|
||||
raise ValueError("distilled FLUX.2 Klein requires steps=4 and guidance=1.0")
|
||||
steps = int(data.get("steps", 9))
|
||||
guidance = float(data.get("guidance", 0.0))
|
||||
if steps != 9 or guidance != 0.0:
|
||||
raise ValueError("Z-Image-Turbo requires steps=9 and guidance=0.0")
|
||||
seed = data.get("seed")
|
||||
generator = None if seed is None else torch.Generator(device="cuda").manual_seed(int(seed))
|
||||
load_pipeline()
|
||||
started = time.monotonic()
|
||||
image = PIPE(prompt=prompt, height=height, width=width,
|
||||
num_inference_steps=4, guidance_scale=1.0,
|
||||
num_inference_steps=9, guidance_scale=0.0,
|
||||
generator=generator).images[0]
|
||||
OUTPUT_DIR.mkdir(parents=True, exist_ok=True)
|
||||
output = OUTPUT_DIR / filename
|
||||
@@ -83,7 +84,7 @@ def generate(data: dict) -> dict:
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
def log_message(self, fmt: str, *args: object) -> None:
|
||||
# Never log request bodies/prompts.
|
||||
print(f"[flux-worker] {self.client_address[0]} {fmt % args}", flush=True)
|
||||
print(f"[z-image-worker] {self.client_address[0]} {fmt % args}", flush=True)
|
||||
|
||||
def reply(self, status: int, payload: dict) -> None:
|
||||
body = json.dumps(payload, separators=(",", ":")).encode()
|
||||
@@ -112,7 +113,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
raise ValueError("invalid request size")
|
||||
self.reply(200, generate(json.loads(self.rfile.read(length))))
|
||||
except Exception as exc:
|
||||
print(f"[flux-worker] generation failed: "
|
||||
print(f"[z-image-worker] generation failed: "
|
||||
f"{type(exc).__name__}: {str(exc)[:1000]}", flush=True)
|
||||
self.reply(400, {"status": "error", "message": str(exc)})
|
||||
|
||||
@@ -24,7 +24,7 @@ ALLOWED = tuple(x.strip() for x in os.environ.get(
|
||||
"ALLOWED_PROFILES", "fast,medium,large,ultra,uncensored,experimental").split(",") if x.strip())
|
||||
LABEL_KEY = "com.mike-ai.llama-profile"
|
||||
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
|
||||
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "flux")
|
||||
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
|
||||
LOCK = threading.Lock()
|
||||
log = logging.getLogger("profile-controller")
|
||||
|
||||
@@ -98,7 +98,7 @@ def set_image_worker(running: bool) -> dict:
|
||||
with LOCK:
|
||||
item = image_container()
|
||||
if running:
|
||||
# A FLUX worker may never overlap a llama profile on the 5080.
|
||||
# The image worker may never overlap a llama profile on the 5080.
|
||||
for profile_item in containers().values():
|
||||
stop_container(profile_item)
|
||||
if item.get("State") != "running":
|
||||
|
||||
@@ -64,7 +64,7 @@ wg_ipv4=$(ip -4 -o address show dev wg0 | awk 'NR == 1 { split($4, address, "/")
|
||||
# The VPN is Athena's normal application network. Nothing below is published
|
||||
# on the physical university interface: every listener is bound inside this
|
||||
# namespace to the Fritzbox-assigned WireGuard address. Clients on the home
|
||||
# VPN may use OpenWebUI, the router and every useful MCP directly.
|
||||
# VPN clients may use the router, speech services and Athena operator directly.
|
||||
proxy_pids=""
|
||||
start_proxy() {
|
||||
listen_port=$1
|
||||
@@ -74,29 +74,10 @@ start_proxy() {
|
||||
}
|
||||
|
||||
start_proxy 22 172.30.10.1:22
|
||||
start_proxy 8080 open-webui:8080
|
||||
start_proxy 8081 router:8081
|
||||
start_proxy 8085 tts-gateway:8085
|
||||
start_proxy 8091 piper:8085
|
||||
start_proxy 8092 xtts:80
|
||||
start_proxy 9119 hermes:9119
|
||||
start_proxy 8642 hermes:8642
|
||||
|
||||
# MCP endpoints. Optional services keep their listener even while stopped and
|
||||
# begin working automatically as soon as their container is started.
|
||||
start_proxy 8202 mcp-athena-operator:8000
|
||||
# Portable general web MCP for Pi, Hermes and other clients. OpenWebUI uses
|
||||
# its native broad search by default; both paths are site-agnostic.
|
||||
start_proxy 8203 tinysearch:8000
|
||||
start_proxy 8204 mcp-github:8000
|
||||
start_proxy 8205 mcp-homeassistant:8000
|
||||
start_proxy 8206 mcp-arr:8000
|
||||
start_proxy 8207 mcp-navidrome:3000
|
||||
start_proxy 8208 mcp-unraid-ssh:8000
|
||||
|
||||
# Search backends are also directly available for diagnostics and alternative
|
||||
# clients. Normal chat clients should prefer the MCP endpoint on 8203.
|
||||
start_proxy 8210 searxng:8080
|
||||
start_proxy 8211 tinysearch:8000
|
||||
|
||||
wait $(printf '%s\n' "$proxy_pids" | awk '{print $2}')
|
||||
|
||||
Reference in New Issue
Block a user