From 1a660c20ba5587f979d9f0de69a5114016166acd Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Mon, 21 Sep 2026 08:52:36 +0200 Subject: [PATCH] Prepare isolated Qwen Image 2.1 evaluation --- README.md | 2 + compose.yaml | 64 +++++++++ dev/test_profile_controller.py | 32 +++++ docs/QWEN_IMAGE_21_TEST.md | 60 ++++++++ .../profile-controller/profile_controller.py | 8 +- platform/docker/qwen-image-21-test/Dockerfile | 15 ++ scripts/prepare-qwen-image-21-test.sh | 35 +++++ scripts/qwen-image-21-test.py | 136 ++++++++++++++++++ 8 files changed, 351 insertions(+), 1 deletion(-) create mode 100644 docs/QWEN_IMAGE_21_TEST.md create mode 100644 platform/docker/qwen-image-21-test/Dockerfile create mode 100755 scripts/prepare-qwen-image-21-test.sh create mode 100755 scripts/qwen-image-21-test.py diff --git a/README.md b/README.md index 91713af..d533771 100644 --- a/README.md +++ b/README.md @@ -22,6 +22,8 @@ dokumentieren den ausgerollten Stand, reduzierte Statuslatenzen und Tests. - genau ein aktives llama.cpp-Profil: Fast, Medium, Large, Ultra oder Uncensored - Profile Router auf Port 8081 - FLUX.2 Klein 9B FP8 Beta: Transformer/VAE auf RTX 5080, Textencoder auf RTX 3060 +- Qwen-Image-2.1 als isolierter, standardmäßig gestoppter Vergleichsworker + ([Testablauf](docs/QWEN_IMAGE_21_TEST.md)); nicht produktiv an den Router angebunden - Qwen3-TTS 1.7B auf der RTX 3060 hinter dem TTS-Gateway; kein Piper-Fallback - Whisper.cpp `small` auf der CPU für lokale deutsche Spracherkennung - EmbeddingGemma 300M Q8 auf der CPU für OpenClaws hybride Memory-Suche diff --git a/compose.yaml b/compose.yaml index c4ff547..f339522 100644 --- a/compose.yaml +++ b/compose.yaml @@ -572,6 +572,7 @@ services: ALLOWED_PROFILES: fast,medium,large,ultra,uncensored IMAGE_WORKER: image RESTORE_WORKER: restore + QWEN_IMAGE_TEST_WORKER: qwen-image-2.1-test TTS_WORKER: qwen3 MUSIC_WORKER: acestep YUE2_WORKER: yue2 @@ -708,6 +709,69 @@ services: timeout: 3s retries: 12 + # Isolated evaluation target. It is created in a stopped state by the + # preparation script and can only be started through the controller's + # allowlisted, GPU-exclusive test endpoint. + qwen-image-21-test: + build: + context: platform/docker/qwen-image-21-test + args: + COMFYUI_COMMIT: b0f4b7b294ce482a2e071d9d762c133d38c7aa07 + image: mike-ai/qwen-image-2.1-test:local + container_name: mike-ai-qwen-image-2.1-test + restart: "no" + profiles: [qwen-image-test] + labels: + com.mike-ai.image-worker: qwen-image-2.1-test + gpus: all + read_only: true + tmpfs: ["/tmp:size=4g,mode=1777"] + volumes: + - "${QWEN_IMAGE_21_MODEL_DIR:-/data/models/Qwen-Image-2.1-ComfyUI}:/opt/ComfyUI/models:ro" + - "${QWEN_IMAGE_21_OUTPUT_DIR:-/data/qwen-image-2.1-test-output}:/opt/ComfyUI/output" + environment: + # Athena enumerates RTX 3060 as GPU 0 and RTX 5080 as GPU 1. + NVIDIA_VISIBLE_DEVICES: "${QWEN_IMAGE_21_GPU:-1}" + NVIDIA_DRIVER_CAPABILITIES: compute,utility + PYTORCH_CUDA_ALLOC_CONF: expandable_segments:True + command: + - python + - main.py + - --listen + - 0.0.0.0 + - --port + - "8188" + - --lowvram + - --preview-method + - none + networks: [inference] + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + healthcheck: + test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8188/system_stats', timeout=2)"] + interval: 5s + timeout: 3s + retries: 60 + start_period: 10s + + qwen-image-21-test-runner: + image: python:3.12-slim + profiles: [qwen-image-test] + read_only: true + tmpfs: ["/tmp:size=32m,mode=1777"] + volumes: + - ./scripts/qwen-image-21-test.py:/opt/test/run.py:ro + - "${QWEN_IMAGE_21_OUTPUT_DIR:-/data/qwen-image-2.1-test-output}:/output" + environment: + CONTROLLER_URL: http://profile-controller:8090 + CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" + COMFY_URL: http://qwen-image-21-test:8188 + OUTPUT_DIR: /output + command: [python, /opt/test/run.py] + networks: [control, inference] + security_opt: ["no-new-privileges:true"] + cap_drop: [ALL] + qwen3-tts: image: ${QWEN3_TTS_IMAGE:-ghcr.io/malaiwah/qwen3-tts-server:latest@sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98} container_name: mike-ai-qwen3-tts diff --git a/dev/test_profile_controller.py b/dev/test_profile_controller.py index 743724a..9f3c09f 100644 --- a/dev/test_profile_controller.py +++ b/dev/test_profile_controller.py @@ -32,6 +32,11 @@ def restore_item(state="exited"): "Labels": {controller.IMAGE_LABEL_KEY: controller.RESTORE_WORKER}} +def qwen_image_test_item(state="exited"): + return {"Id": "id-qwen-image-test", "State": state, + "Labels": {controller.IMAGE_LABEL_KEY: controller.QWEN_IMAGE_TEST_WORKER}} + + def tts_item(state="running"): return {"Id": "id-tts", "State": state, "Labels": {controller.TTS_LABEL_KEY: controller.TTS_WORKER}} @@ -177,6 +182,33 @@ class ProfileControllerTests(unittest.TestCase): ("POST", "/containers/id-restore/start"), ]) + def test_qwen_image_test_is_allowlisted_and_exclusive(self): + profiles = {name: item(name) for name in controller.ALLOWED} + profiles["medium"] = item("medium", "running") + calls = [] + + def request(method, path): + calls.append((method, path)) + return 204, b"" + + with patch.object(controller, "containers", return_value=profiles), \ + patch.object(controller, "image_container", return_value=qwen_image_test_item()), \ + patch.object(controller, "image_containers", return_value=[ + image_item("running"), restore_item(), qwen_image_test_item()]), \ + patch.object(controller, "tts_container", return_value=tts_item()), \ + patch.object(controller, "docker_request", side_effect=request): + result = controller.set_image_worker( + True, controller.QWEN_IMAGE_TEST_WORKER) + + self.assertEqual(result, { + "image_worker": "qwen-image-2.1-test", "state": "running"}) + self.assertEqual(calls, [ + ("POST", "/containers/id-medium/stop?t=120"), + ("POST", "/containers/id-tts/stop?t=30"), + ("POST", "/containers/id-flux/stop?t=20"), + ("POST", "/containers/id-qwen-image-test/start"), + ]) + def test_profile_activation_stops_image_worker_first(self): profiles = {name: item(name) for name in controller.ALLOWED} calls = [] diff --git a/docs/QWEN_IMAGE_21_TEST.md b/docs/QWEN_IMAGE_21_TEST.md new file mode 100644 index 0000000..93ab4cb --- /dev/null +++ b/docs/QWEN_IMAGE_21_TEST.md @@ -0,0 +1,60 @@ +# Qwen-Image-2.1 – isolierter Test + +Stand: 21. September 2026. Dieser Pfad dient nur dem Vergleich mit dem +produktiven FLUX.2-Worker. Er ersetzt FLUX nicht und ist nicht an den Router +angebunden. + +## Reproduzierbarer Stand + +- ComfyUI: Commit `b0f4b7b294ce482a2e071d9d762c133d38c7aa07` +- Modell-Repository: `Comfy-Org/Qwen-Image-2.1`, Revision + `ace0edeb3791a594ddfa36ed5f41a178a394e921` +- Transformer: `qwen_image_2.1_int8_convrot.safetensors` (7.256.783.064 Byte) +- Textencoder: `qwen3vl_8b_int8_convrot.safetensors` (9.350.798.360 Byte) +- VAE: `qwen_image_2.1_vae_bf16.safetensors` (675.509.688 Byte) +- erster Vergleich: 1024 × 1024, 25 Schritte, CFG 1, Euler/Simple + +Die Gewichte sind die von Qwen verlinkte offizielle ComfyUI-Aufbereitung. Der +Worker verwendet `--lowvram` auf der RTX 5080. So bleibt der Test mit 16 GB +VRAM möglich; der Preis ist CPU-Offload und damit eine längere Laufzeit. + +## Vorbereitung ohne GPU-Last + +```bash +cd /opt/mike-ai/stack +sudo ./scripts/prepare-qwen-image-21-test.sh +``` + +Das Skript prüft feste SHA-256-Summen, baut das Image, aktualisiert den +Profile Controller und erstellt den Testcontainer **gestoppt**. Es lädt kein +Modell in die GPU. + +## Einmaliger Test nach ausdrücklichem „Go“ + +```bash +cd /opt/mike-ai/stack +docker compose --env-file /etc/mike-ai/stack.env \ + --profile qwen-image-test run --rm qwen-image-21-test-runner \ + python /opt/test/run.py 'HIER DEN VEREINBARTEN PROMPT EINSETZEN' +``` + +Der Runner merkt sich das aktive LLM-Profil, startet den Qwen-Worker über den +allowlist-beschränkten Controller, erzeugt genau ein Bild und stellt danach +auch bei einem Fehler das vorherige Profil wieder her. Das Ergebnis liegt in +`/data/qwen-image-2.1-test-output/`. FLUX-Konfiguration und FLUX-Gewichte +werden nicht verändert. + +## Vollständig entfernen + +```bash +cd /opt/mike-ai/stack +docker compose --env-file /etc/mike-ai/stack.env \ + --profile qwen-image-test rm -sf qwen-image-21-test +docker image rm mike-ai/qwen-image-2.1-test:local +rm -rf /data/models/Qwen-Image-2.1-ComfyUI \ + /data/qwen-image-2.1-test-output +``` + +Anschließend kann die Controller-Erweiterung bei Bedarf aus Git zurückgenommen +und nur `profile-controller` neu gebaut werden. Die Entfernung ist nicht Teil +der Vorbereitung und wird niemals automatisch ausgeführt. diff --git a/platform/docker/profile-controller/profile_controller.py b/platform/docker/profile-controller/profile_controller.py index 4c3686a..89b3e13 100644 --- a/platform/docker/profile-controller/profile_controller.py +++ b/platform/docker/profile-controller/profile_controller.py @@ -27,6 +27,8 @@ LABEL_KEY = "com.mike-ai.llama-profile" IMAGE_LABEL_KEY = "com.mike-ai.image-worker" IMAGE_WORKER = os.environ.get("IMAGE_WORKER", "image") RESTORE_WORKER = os.environ.get("RESTORE_WORKER", "restore") +QWEN_IMAGE_TEST_WORKER = os.environ.get( + "QWEN_IMAGE_TEST_WORKER", "qwen-image-2.1-test").strip() TTS_LABEL_KEY = "com.mike-ai.tts-worker" TTS_WORKER = os.environ.get("TTS_WORKER", "qwen3") MUSIC_LABEL_KEY = "com.mike-ai.music-worker" @@ -101,6 +103,8 @@ def image_container(kind: str = IMAGE_WORKER) -> dict: def image_containers() -> list[dict]: """All allowlisted GPU workers that must never overlap an LLM.""" allowed = {IMAGE_WORKER, RESTORE_WORKER} + if QWEN_IMAGE_TEST_WORKER: + allowed.add(QWEN_IMAGE_TEST_WORKER) return [item for item in labelled_containers(IMAGE_LABEL_KEY) if item.get("Labels", {}).get(IMAGE_LABEL_KEY) in allowed] @@ -359,7 +363,7 @@ def stop_inference() -> dict: def set_image_worker(running: bool, kind: str = IMAGE_WORKER) -> dict: - if kind not in {IMAGE_WORKER, RESTORE_WORKER}: + if kind not in {IMAGE_WORKER, RESTORE_WORKER, QWEN_IMAGE_TEST_WORKER}: raise ValueError("worker is not allowlisted") with LOCK: item = image_container(kind) @@ -797,6 +801,8 @@ class Handler(BaseHTTPRequestHandler): "/workers/image/stop": (IMAGE_WORKER, False), "/workers/restore/start": (RESTORE_WORKER, True), "/workers/restore/stop": (RESTORE_WORKER, False), + "/workers/qwen-image-test/start": (QWEN_IMAGE_TEST_WORKER, True), + "/workers/qwen-image-test/stop": (QWEN_IMAGE_TEST_WORKER, False), } if self.path in worker_paths: try: diff --git a/platform/docker/qwen-image-21-test/Dockerfile b/platform/docker/qwen-image-21-test/Dockerfile new file mode 100644 index 0000000..c451bf4 --- /dev/null +++ b/platform/docker/qwen-image-21-test/Dockerfile @@ -0,0 +1,15 @@ +ARG PYTORCH_IMAGE=pytorch/pytorch:2.11.0-cuda12.8-cudnn9-runtime +FROM ${PYTORCH_IMAGE} + +ARG COMFYUI_COMMIT +RUN test -n "$COMFYUI_COMMIT" \ + && apt-get update \ + && apt-get install -y --no-install-recommends git ca-certificates \ + && git clone --filter=blob:none https://github.com/Comfy-Org/ComfyUI.git /opt/ComfyUI \ + && git -C /opt/ComfyUI checkout "$COMFYUI_COMMIT" \ + && python -m pip install --no-cache-dir -r /opt/ComfyUI/requirements.txt \ + && apt-get purge -y --auto-remove git \ + && rm -rf /var/lib/apt/lists/* /root/.cache + +WORKDIR /opt/ComfyUI +EXPOSE 8188 diff --git a/scripts/prepare-qwen-image-21-test.sh b/scripts/prepare-qwen-image-21-test.sh new file mode 100755 index 0000000..70155dc --- /dev/null +++ b/scripts/prepare-qwen-image-21-test.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +# Download pinned official weights, build the isolated worker and leave it stopped. +set -Eeuo pipefail + +ROOT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" +ENV_FILE=${STACK_ENV:-/etc/mike-ai/stack.env} +MODEL_DIR=${QWEN_IMAGE_21_MODEL_DIR:-/data/models/Qwen-Image-2.1-ComfyUI} +OUTPUT_DIR=${QWEN_IMAGE_21_OUTPUT_DIR:-/data/qwen-image-2.1-test-output} +BASE=https://huggingface.co/Comfy-Org/Qwen-Image-2.1/resolve/ace0edeb3791a594ddfa36ed5f41a178a394e921 + +download() { + local relative=$1 expected=$2 target="$MODEL_DIR/$1" + install -d -m 0755 "$(dirname "$target")" + if [[ -f $target ]] && echo "$expected $target" | sha256sum -c --status; then + echo "OK: $relative" + return + fi + curl -fL --retry 5 --retry-all-errors -C - -o "$target.part" "$BASE/$relative" + echo "$expected $target.part" | sha256sum -c --status + mv "$target.part" "$target" +} + +download diffusion_models/qwen_image_2.1_int8_convrot.safetensors cb74113cb03faecd79611b01fd7fd642f0aa60d6f0b95086abee214d75eaa57d +download text_encoders/qwen3vl_8b_int8_convrot.safetensors 8bfd0f6e12abf2d2d697ecc888e5e90b0d6741d6708f05799f53afa560452e8f +download vae/qwen_image_2.1_vae_bf16.safetensors bb21f7473051e1ac368515dd3f2e15cd44d7a11748ee8823e1ddca3e4876b7c9 +install -d -m 0755 "$OUTPUT_DIR" + +compose=(docker compose --env-file "$ENV_FILE" -f "$ROOT_DIR/compose.yaml" --profile qwen-image-test) +"${compose[@]}" build qwen-image-21-test +"${compose[@]}" up -d --build --no-deps profile-controller +"${compose[@]}" create qwen-image-21-test +"${compose[@]}" stop qwen-image-21-test +state=$(docker inspect -f '{{.State.Status}}' mike-ai-qwen-image-2.1-test) +[[ $state == exited || $state == created ]] +echo "Qwen-Image-2.1 test is prepared and stopped. No GPU model was loaded." diff --git a/scripts/qwen-image-21-test.py b/scripts/qwen-image-21-test.py new file mode 100755 index 0000000..bfe8d6b --- /dev/null +++ b/scripts/qwen-image-21-test.py @@ -0,0 +1,136 @@ +#!/usr/bin/env python3 +"""One-shot Qwen-Image-2.1 evaluation with guaranteed profile restoration.""" + +from __future__ import annotations + +import argparse +import json +import os +import pathlib +import random +import time +import urllib.parse +import urllib.request + + +CONTROLLER = os.environ.get("CONTROLLER_URL", "http://profile-controller:8090") +TOKEN = os.environ["CONTROLLER_TOKEN"] +COMFY = os.environ.get("COMFY_URL", "http://qwen-image-21-test:8188") +OUTPUT = pathlib.Path(os.environ.get("OUTPUT_DIR", "/output")) + + +def request_json(url: str, *, payload: dict | None = None, + authenticated: bool = False, timeout: float = 30) -> dict: + body = None if payload is None else json.dumps(payload).encode() + headers = {"Content-Type": "application/json"} + if authenticated: + headers["Authorization"] = f"Bearer {TOKEN}" + method = "POST" if payload is not None else "GET" + req = urllib.request.Request(url, data=body, headers=headers, method=method) + with urllib.request.urlopen(req, timeout=timeout) as response: + return json.load(response) + + +def controller(path: str, *, post: bool = False) -> dict: + return request_json(CONTROLLER + path, payload={} if post else None, + authenticated=True, timeout=900) + + +def wait_comfy(timeout: int = 900) -> None: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + try: + request_json(COMFY + "/system_stats", timeout=3) + return + except Exception: + time.sleep(2) + raise TimeoutError("Qwen-Image worker did not become ready") + + +def workflow(prompt: str, seed: int, steps: int, width: int, height: int) -> dict: + return { + "1": {"class_type": "UNETLoader", "inputs": { + "unet_name": "qwen_image_2.1_int8_convrot.safetensors", + "weight_dtype": "default"}}, + "2": {"class_type": "CLIPLoader", "inputs": { + "clip_name": "qwen3vl_8b_int8_convrot.safetensors", + "type": "qwen_image", "device": "default"}}, + "3": {"class_type": "VAELoader", "inputs": { + "vae_name": "qwen_image_2.1_vae_bf16.safetensors"}}, + "4": {"class_type": "TextEncodeQwenImage21", "inputs": { + "clip": ["2", 0], "prompt": prompt, "negative_prompt": "", + "resolution": max(width, height)}}, + "5": {"class_type": "EmptyLatentImage", "inputs": { + "width": width, "height": height, "batch_size": 1}}, + "6": {"class_type": "KSampler", "inputs": { + "model": ["1", 0], "positive": ["4", 0], "negative": ["4", 1], + "latent_image": ["5", 0], "seed": seed, "steps": steps, + "cfg": 1.0, "sampler_name": "euler", "scheduler": "simple", + "denoise": 1.0}}, + "7": {"class_type": "VAEDecode", "inputs": { + "samples": ["6", 0], "vae": ["3", 0]}}, + "8": {"class_type": "SaveImage", "inputs": { + "filename_prefix": "qwen-image-2.1-test", "images": ["7", 0]}}, + } + + +def wait_result(prompt_id: str, timeout: int = 3600) -> dict: + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + history = request_json(f"{COMFY}/history/{prompt_id}", timeout=10) + if prompt_id in history: + result = history[prompt_id] + status = result.get("status", {}) + if status.get("status_str") == "error" or not status.get("completed", True): + raise RuntimeError("generation failed: " + json.dumps(status)) + return result + time.sleep(2) + raise TimeoutError("Qwen-Image generation timed out") + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("prompt") + parser.add_argument("--seed", type=int, default=None) + parser.add_argument("--steps", type=int, default=25) + parser.add_argument("--width", type=int, default=1024) + parser.add_argument("--height", type=int, default=1024) + args = parser.parse_args() + if args.width % 32 or args.height % 32: + parser.error("width and height must be multiples of 32") + seed = args.seed if args.seed is not None else random.randrange(2**53) + previous = controller("/profiles/status").get("active_profile") + started = time.monotonic() + try: + controller("/workers/qwen-image-test/start", post=True) + wait_comfy() + queued = request_json(COMFY + "/prompt", payload={ + "prompt": workflow(args.prompt, seed, args.steps, args.width, args.height), + "client_id": "athena-qwen-image-21-test"}, timeout=30) + prompt_id = queued["prompt_id"] + result = wait_result(prompt_id) + images = [] + for node in result.get("outputs", {}).values(): + images.extend(node.get("images", [])) + if not images: + raise RuntimeError("generation completed without an image") + image = images[0] + query = urllib.parse.urlencode({ + "filename": image["filename"], "subfolder": image.get("subfolder", ""), + "type": image.get("type", "output")}) + target = OUTPUT / image["filename"] + target.parent.mkdir(parents=True, exist_ok=True) + with urllib.request.urlopen(COMFY + "/view?" + query, timeout=120) as src: + target.write_bytes(src.read()) + print(json.dumps({"status": "ok", "file": str(target), "seed": seed, + "seconds": round(time.monotonic() - started, 1)}, indent=2)) + finally: + try: + controller("/workers/qwen-image-test/stop", post=True) + finally: + if previous: + controller(f"/profiles/{previous}/activate", post=True) + + +if __name__ == "__main__": + main()