Prepare isolated Qwen Image 2.1 evaluation

This commit is contained in:
Mikei386
2026-09-21 08:52:36 +02:00
parent 302b08e051
commit 1a660c20ba
8 changed files with 351 additions and 1 deletions
+2
View File
@@ -22,6 +22,8 @@ dokumentieren den ausgerollten Stand, reduzierte Statuslatenzen und Tests.
- genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored
- Profile Router auf Port 8081
- FLUX.2 Klein 9B FP8 Beta: Transformer/VAE auf RTX 5080, Textencoder auf RTX 3060
- Qwen-Image-2.1 als isolierter, standardmäßig gestoppter Vergleichsworker
([Testablauf](docs/QWEN_IMAGE_21_TEST.md)); nicht produktiv an den Router angebunden
- Qwen3-TTS 1.7B auf der RTX 3060 hinter dem TTS-Gateway; kein Piper-Fallback
- Whisper.cpp `small` auf der CPU für lokale deutsche Spracherkennung
- EmbeddingGemma 300M Q8 auf der CPU für OpenClaws hybride Memory-Suche
+64
View File
@@ -572,6 +572,7 @@ services:
ALLOWED_PROFILES: fast,medium,large,ultra,uncensored
IMAGE_WORKER: image
RESTORE_WORKER: restore
QWEN_IMAGE_TEST_WORKER: qwen-image-2.1-test
TTS_WORKER: qwen3
MUSIC_WORKER: acestep
YUE2_WORKER: yue2
@@ -708,6 +709,69 @@ services:
timeout: 3s
retries: 12
# Isolated evaluation target. It is created in a stopped state by the
# preparation script and can only be started through the controller's
# allowlisted, GPU-exclusive test endpoint.
qwen-image-21-test:
build:
context: platform/docker/qwen-image-21-test
args:
COMFYUI_COMMIT: b0f4b7b294ce482a2e071d9d762c133d38c7aa07
image: mike-ai/qwen-image-2.1-test:local
container_name: mike-ai-qwen-image-2.1-test
restart: "no"
profiles: [qwen-image-test]
labels:
com.mike-ai.image-worker: qwen-image-2.1-test
gpus: all
read_only: true
tmpfs: ["/tmp:size=4g,mode=1777"]
volumes:
- "${QWEN_IMAGE_21_MODEL_DIR:-/data/models/Qwen-Image-2.1-ComfyUI}:/opt/ComfyUI/models:ro"
- "${QWEN_IMAGE_21_OUTPUT_DIR:-/data/qwen-image-2.1-test-output}:/opt/ComfyUI/output"
environment:
# Athena enumerates RTX 3060 as GPU 0 and RTX 5080 as GPU 1.
NVIDIA_VISIBLE_DEVICES: "${QWEN_IMAGE_21_GPU:-1}"
NVIDIA_DRIVER_CAPABILITIES: compute,utility
PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True
command:
- python
- main.py
- --listen
- 0.0.0.0
- --port
- "8188"
- --lowvram
- --preview-method
- none
networks: [inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8188/system_stats', timeout=2)"]
interval: 5s
timeout: 3s
retries: 60
start_period: 10s
qwen-image-21-test-runner:
image: python:3.12-slim
profiles: [qwen-image-test]
read_only: true
tmpfs: ["/tmp:size=32m,mode=1777"]
volumes:
- ./scripts/qwen-image-21-test.py:/opt/test/run.py:ro
- "${QWEN_IMAGE_21_OUTPUT_DIR:-/data/qwen-image-2.1-test-output}:/output"
environment:
CONTROLLER_URL: http://profile-controller:8090
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
COMFY_URL: http://qwen-image-21-test:8188
OUTPUT_DIR: /output
command: [python, /opt/test/run.py]
networks: [control, inference]
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
qwen3-tts:
image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98}
container_name: mike-ai-qwen3-tts
+32
View File
@@ -32,6 +32,11 @@ def restore_item(state="exited"):
"Labels": {controller.IMAGE_LABEL_KEY: controller.RESTORE_WORKER}}
def qwen_image_test_item(state="exited"):
return {"Id": "id-qwen-image-test", "State": state,
"Labels": {controller.IMAGE_LABEL_KEY: controller.QWEN_IMAGE_TEST_WORKER}}
def tts_item(state="running"):
return {"Id": "id-tts", "State": state,
"Labels": {controller.TTS_LABEL_KEY: controller.TTS_WORKER}}
@@ -177,6 +182,33 @@ class ProfileControllerTests(unittest.TestCase):
("POST", "/containers/id-restore/start"),
])
def test_qwen_image_test_is_allowlisted_and_exclusive(self):
profiles = {name: item(name) for name in controller.ALLOWED}
profiles["medium"] = item("medium", "running")
calls = []
def request(method, path):
calls.append((method, path))
return 204, b""
with patch.object(controller, "containers", return_value=profiles), \
patch.object(controller, "image_container", return_value=qwen_image_test_item()), \
patch.object(controller, "image_containers", return_value=[
image_item("running"), restore_item(), qwen_image_test_item()]), \
patch.object(controller, "tts_container", return_value=tts_item()), \
patch.object(controller, "docker_request", side_effect=request):
result = controller.set_image_worker(
True, controller.QWEN_IMAGE_TEST_WORKER)
self.assertEqual(result, {
"image_worker": "qwen-image-2.1-test", "state": "running"})
self.assertEqual(calls, [
("POST", "/containers/id-medium/stop?t=120"),
("POST", "/containers/id-tts/stop?t=30"),
("POST", "/containers/id-flux/stop?t=20"),
("POST", "/containers/id-qwen-image-test/start"),
])
def test_profile_activation_stops_image_worker_first(self):
profiles = {name: item(name) for name in controller.ALLOWED}
calls = []
+60
View File
@@ -0,0 +1,60 @@
# Qwen-Image-2.1 – isolierter Test
Stand: 21. September 2026. Dieser Pfad dient nur dem Vergleich mit dem
produktiven FLUX.2-Worker. Er ersetzt FLUX nicht und ist nicht an den Router
angebunden.
## Reproduzierbarer Stand
- ComfyUI: Commit `b0f4b7b294ce482a2e071d9d762c133d38c7aa07`
- Modell-Repository: `Comfy-Org/Qwen-Image-2.1`, Revision
`ace0edeb3791a594ddfa36ed5f41a178a394e921`
- Transformer: `qwen_image_2.1_int8_convrot.safetensors` (7.256.783.064 Byte)
- Textencoder: `qwen3vl_8b_int8_convrot.safetensors` (9.350.798.360 Byte)
- VAE: `qwen_image_2.1_vae_bf16.safetensors` (675.509.688 Byte)
- erster Vergleich: 1024 × 1024, 25 Schritte, CFG 1, Euler/Simple
Die Gewichte sind die von Qwen verlinkte offizielle ComfyUI-Aufbereitung. Der
Worker verwendet `--lowvram` auf der RTX 5080. So bleibt der Test mit 16 GB
VRAM möglich; der Preis ist CPU-Offload und damit eine längere Laufzeit.
## Vorbereitung ohne GPU-Last
```bash
cd /opt/mike-ai/stack
sudo ./scripts/prepare-qwen-image-21-test.sh
```
Das Skript prüft feste SHA-256-Summen, baut das Image, aktualisiert den
Profile Controller und erstellt den Testcontainer **gestoppt**. Es lädt kein
Modell in die GPU.
## Einmaliger Test nach ausdrücklichem „Go“
```bash
cd /opt/mike-ai/stack
docker compose --env-file /etc/mike-ai/stack.env \
--profile qwen-image-test run --rm qwen-image-21-test-runner \
python /opt/test/run.py 'HIER DEN VEREINBARTEN PROMPT EINSETZEN'
```
Der Runner merkt sich das aktive LLM-Profil, startet den Qwen-Worker über den
allowlist-beschränkten Controller, erzeugt genau ein Bild und stellt danach
auch bei einem Fehler das vorherige Profil wieder her. Das Ergebnis liegt in
`/data/qwen-image-2.1-test-output/`. FLUX-Konfiguration und FLUX-Gewichte
werden nicht verändert.
## Vollständig entfernen
```bash
cd /opt/mike-ai/stack
docker compose --env-file /etc/mike-ai/stack.env \
--profile qwen-image-test rm -sf qwen-image-21-test
docker image rm mike-ai/qwen-image-2.1-test:local
rm -rf /data/models/Qwen-Image-2.1-ComfyUI \
/data/qwen-image-2.1-test-output
```
Anschließend kann die Controller-Erweiterung bei Bedarf aus Git zurückgenommen
und nur `profile-controller` neu gebaut werden. Die Entfernung ist nicht Teil
der Vorbereitung und wird niemals automatisch ausgeführt.
@@ -27,6 +27,8 @@ LABEL_KEY = "com.mike-ai.llama-profile"
IMAGE_LABEL_KEY = "com.mike-ai.image-worker"
IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image")
RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore")
QWEN_IMAGE_TEST_WORKER = os.environ.get(
"QWEN_IMAGE_TEST_WORKER", "qwen-image-2.1-test").strip()
TTS_LABEL_KEY = "com.mike-ai.tts-worker"
TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3")
MUSIC_LABEL_KEY = "com.mike-ai.music-worker"
@@ -101,6 +103,8 @@ def image_container(kind: str = IMAGE_WORKER) -> dict:
def image_containers() -> list[dict]:
"""All allowlisted GPU workers that must never overlap an LLM."""
allowed = {IMAGE_WORKER, RESTORE_WORKER}
if QWEN_IMAGE_TEST_WORKER:
allowed.add(QWEN_IMAGE_TEST_WORKER)
return [item for item in labelled_containers(IMAGE_LABEL_KEY)
if item.get("Labels", {}).get(IMAGE_LABEL_KEY) in allowed]
@@ -359,7 +363,7 @@ def stop_inference() -> dict:
def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict:
if kind not in {IMAGE_WORKER, RESTORE_WORKER}:
if kind not in {IMAGE_WORKER, RESTORE_WORKER, QWEN_IMAGE_TEST_WORKER}:
raise ValueError("worker is not allowlisted")
with LOCK:
item = image_container(kind)
@@ -797,6 +801,8 @@ class Handler(BaseHTTPRequestHandler):
"/workers/image/stop": (IMAGE_WORKER, False),
"/workers/restore/start": (RESTORE_WORKER, True),
"/workers/restore/stop": (RESTORE_WORKER, False),
"/workers/qwen-image-test/start": (QWEN_IMAGE_TEST_WORKER, True),
"/workers/qwen-image-test/stop": (QWEN_IMAGE_TEST_WORKER, False),
}
if self.path in worker_paths:
try:
@@ -0,0 +1,15 @@
ARG PYTORCH_IMAGE=pytorch/pytorch:2.11.0-cuda12.8-cudnn9-runtime
FROM ${PYTORCH_IMAGE}
ARG COMFYUI_COMMIT
RUN test -n "$COMFYUI_COMMIT" \
&& apt-get update \
&& apt-get install -y --no-install-recommends git ca-certificates \
&& git clone --filter=blob:none https://github.com/Comfy-Org/ComfyUI.git /opt/ComfyUI \
&& git -C /opt/ComfyUI checkout "$COMFYUI_COMMIT" \
&& python -m pip install --no-cache-dir -r /opt/ComfyUI/requirements.txt \
&& apt-get purge -y --auto-remove git \
&& rm -rf /var/lib/apt/lists/* /root/.cache
WORKDIR /opt/ComfyUI
EXPOSE 8188
+35
View File
@@ -0,0 +1,35 @@
#!/usr/bin/env bash
# Download pinned official weights, build the isolated worker and leave it stopped.
set -Eeuo pipefail
ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
ENV_FILE=${STACK_ENV:-/etc/mike-ai/stack.env}
MODEL_DIR=${QWEN_IMAGE_21_MODEL_DIR:-/data/models/Qwen-Image-2.1-ComfyUI}
OUTPUT_DIR=${QWEN_IMAGE_21_OUTPUT_DIR:-/data/qwen-image-2.1-test-output}
BASE=https://huggingface.co/Comfy-Org/Qwen-Image-2.1/resolve/ace0edeb3791a594ddfa36ed5f41a178a394e921
download() {
local relative=$1 expected=$2 target="$MODEL_DIR/$1"
install -d -m 0755 "$(dirname "$target")"
if [[ -f $target ]] && echo "$expected $target" | sha256sum -c --status; then
echo "OK: $relative"
return
fi
curl -fL --retry 5 --retry-all-errors -C - -o "$target.part" "$BASE/$relative"
echo "$expected $target.part" | sha256sum -c --status
mv "$target.part" "$target"
}
download diffusion_models/qwen_image_2.1_int8_convrot.safetensors cb74113cb03faecd79611b01fd7fd642f0aa60d6f0b95086abee214d75eaa57d
download text_encoders/qwen3vl_8b_int8_convrot.safetensors 8bfd0f6e12abf2d2d697ecc888e5e90b0d6741d6708f05799f53afa560452e8f
download vae/qwen_image_2.1_vae_bf16.safetensors bb21f7473051e1ac368515dd3f2e15cd44d7a11748ee8823e1ddca3e4876b7c9
install -d -m 0755 "$OUTPUT_DIR"
compose=(docker compose --env-file "$ENV_FILE" -f "$ROOT_DIR/compose.yaml" --profile qwen-image-test)
"${compose[@]}" build qwen-image-21-test
"${compose[@]}" up -d --build --no-deps profile-controller
"${compose[@]}" create qwen-image-21-test
"${compose[@]}" stop qwen-image-21-test
state=$(docker inspect -f '{{.State.Status}}' mike-ai-qwen-image-2.1-test)
[[ $state == exited || $state == created ]]
echo "Qwen-Image-2.1 test is prepared and stopped. No GPU model was loaded."
+136
View File
@@ -0,0 +1,136 @@
#!/usr/bin/env python3
"""One-shot Qwen-Image-2.1 evaluation with guaranteed profile restoration."""
from __future__ import annotations
import argparse
import json
import os
import pathlib
import random
import time
import urllib.parse
import urllib.request
CONTROLLER = os.environ.get("CONTROLLER_URL", "http://profile-controller:8090")
TOKEN = os.environ["CONTROLLER_TOKEN"]
COMFY = os.environ.get("COMFY_URL", "http://qwen-image-21-test:8188")
OUTPUT = pathlib.Path(os.environ.get("OUTPUT_DIR", "/output"))
def request_json(url: str, *, payload: dict | None = None,
authenticated: bool = False, timeout: float = 30) -> dict:
body = None if payload is None else json.dumps(payload).encode()
headers = {"Content-Type": "application/json"}
if authenticated:
headers["Authorization"] = f"Bearer {TOKEN}"
method = "POST" if payload is not None else "GET"
req = urllib.request.Request(url, data=body, headers=headers, method=method)
with urllib.request.urlopen(req, timeout=timeout) as response:
return json.load(response)
def controller(path: str, *, post: bool = False) -> dict:
return request_json(CONTROLLER + path, payload={} if post else None,
authenticated=True, timeout=900)
def wait_comfy(timeout: int = 900) -> None:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
try:
request_json(COMFY + "/system_stats", timeout=3)
return
except Exception:
time.sleep(2)
raise TimeoutError("Qwen-Image worker did not become ready")
def workflow(prompt: str, seed: int, steps: int, width: int, height: int) -> dict:
return {
"1": {"class_type": "UNETLoader", "inputs": {
"unet_name": "qwen_image_2.1_int8_convrot.safetensors",
"weight_dtype": "default"}},
"2": {"class_type": "CLIPLoader", "inputs": {
"clip_name": "qwen3vl_8b_int8_convrot.safetensors",
"type": "qwen_image", "device": "default"}},
"3": {"class_type": "VAELoader", "inputs": {
"vae_name": "qwen_image_2.1_vae_bf16.safetensors"}},
"4": {"class_type": "TextEncodeQwenImage21", "inputs": {
"clip": ["2", 0], "prompt": prompt, "negative_prompt": "",
"resolution": max(width, height)}},
"5": {"class_type": "EmptyLatentImage", "inputs": {
"width": width, "height": height, "batch_size": 1}},
"6": {"class_type": "KSampler", "inputs": {
"model": ["1", 0], "positive": ["4", 0], "negative": ["4", 1],
"latent_image": ["5", 0], "seed": seed, "steps": steps,
"cfg": 1.0, "sampler_name": "euler", "scheduler": "simple",
"denoise": 1.0}},
"7": {"class_type": "VAEDecode", "inputs": {
"samples": ["6", 0], "vae": ["3", 0]}},
"8": {"class_type": "SaveImage", "inputs": {
"filename_prefix": "qwen-image-2.1-test", "images": ["7", 0]}},
}
def wait_result(prompt_id: str, timeout: int = 3600) -> dict:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
history = request_json(f"{COMFY}/history/{prompt_id}", timeout=10)
if prompt_id in history:
result = history[prompt_id]
status = result.get("status", {})
if status.get("status_str") == "error" or not status.get("completed", True):
raise RuntimeError("generation failed: " + json.dumps(status))
return result
time.sleep(2)
raise TimeoutError("Qwen-Image generation timed out")
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("prompt")
parser.add_argument("--seed", type=int, default=None)
parser.add_argument("--steps", type=int, default=25)
parser.add_argument("--width", type=int, default=1024)
parser.add_argument("--height", type=int, default=1024)
args = parser.parse_args()
if args.width % 32 or args.height % 32:
parser.error("width and height must be multiples of 32")
seed = args.seed if args.seed is not None else random.randrange(2**53)
previous = controller("/profiles/status").get("active_profile")
started = time.monotonic()
try:
controller("/workers/qwen-image-test/start", post=True)
wait_comfy()
queued = request_json(COMFY + "/prompt", payload={
"prompt": workflow(args.prompt, seed, args.steps, args.width, args.height),
"client_id": "athena-qwen-image-21-test"}, timeout=30)
prompt_id = queued["prompt_id"]
result = wait_result(prompt_id)
images = []
for node in result.get("outputs", {}).values():
images.extend(node.get("images", []))
if not images:
raise RuntimeError("generation completed without an image")
image = images[0]
query = urllib.parse.urlencode({
"filename": image["filename"], "subfolder": image.get("subfolder", ""),
"type": image.get("type", "output")})
target = OUTPUT / image["filename"]
target.parent.mkdir(parents=True, exist_ok=True)
with urllib.request.urlopen(COMFY + "/view?" + query, timeout=120) as src:
target.write_bytes(src.read())
print(json.dumps({"status": "ok", "file": str(target), "seed": seed,
"seconds": round(time.monotonic() - started, 1)}, indent=2))
finally:
try:
controller("/workers/qwen-image-test/stop", post=True)
finally:
if previous:
controller(f"/profiles/{previous}/activate", post=True)
if __name__ == "__main__":
main()