#!/usr/bin/env python3 """Final pre-move Qwen3.8 runtime, MTP, split-draft and ablation matrix. This wrapper reuses the proven isolated benchmark runner already deployed on Athena. It operates only on synthetic prompts, restores the Medium production profile on every exit path and keeps the candidate runtime image separate. """ from __future__ import annotations import importlib.util import pathlib import subprocess import sys BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py") SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE) if SPEC is None or SPEC.loader is None: raise RuntimeError(f"Could not load {BASE}") runner = importlib.util.module_from_spec(SPEC) sys.modules[SPEC.name] = runner SPEC.loader.exec_module(runner) CURRENT_IMAGE = "mike-ai/llama.cpp:local" LATEST_IMAGE = "mike-ai/llama.cpp:3f545bec-test" OFFICIAL_PROJECTOR = "/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf" ABLITERATED_PROJECTOR = "/models/qwen3.8-27b-abliterated/mmproj-Qwen3.8-27B-ABLITERATED-F16.gguf" SPLIT_DRAFT = "/models/qwen3.8-nvfp4-split/mtp-Qwen3.8-27B-NVFP4.gguf" runner.OUT = pathlib.Path("/data/benchmarks/qwen38-final-acceptance-20260822") runner.TASK_FILE = pathlib.Path("/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json") runner.MODELS.update({ "pure-current": runner.MODELS["iq4-pure"], "pure-latest": runner.MODELS["iq4-pure"], "nvfp4-latest": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf", "abliterated-latest": "/models/qwen3.8-27b-abliterated/Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf", }) # The case IDs intentionally encode the parameters that the legacy Case # dataclass does not know (runtime image, p-min and separate draft model). runner.CASES = [ runner.Case("01-current-pure-160k-mtp3-p000", "pure-current", 160000, (90, 10), projector="gpu", quality=True, long_fill=0.75), runner.Case("02-latest-pure-160k-mtp3-p000", "pure-latest", 160000, (90, 10), projector="gpu", quality=True, long_fill=0.75), runner.Case("03-current-fast-76k-mtp2-p000", "iq4-mix", 76800, None, projector="gpu", mtp_max=2), runner.Case("04-latest-fast-76k-mtp2-p000", "iq4-mix", 76800, None, projector="gpu", mtp_max=2), runner.Case("05-latest-pure-160k-mtp2-p010", "pure-latest", 160000, (90, 10), projector="gpu", mtp_max=2), runner.Case("06-latest-pure-160k-mtp3-p005", "pure-latest", 160000, (90, 10), projector="gpu", mtp_max=3), runner.Case("07-latest-pure-160k-mtp4-p010", "pure-latest", 160000, (90, 10), projector="gpu", mtp_max=4), runner.Case("08-latest-nvfp4-72k-embedded-mtp3-p010", "nvfp4-latest", 72000, (72, 28), mtp_max=3), runner.Case("09-latest-nvfp4-72k-split-mtp2-p010", "nvfp4-latest", 72000, (85, 15), mtp_max=2), runner.Case("10-latest-nvfp4-72k-split-mtp3-p010", "nvfp4-latest", 72000, (85, 15), mtp_max=3, quality=True), runner.Case("11-latest-abliterated-80k-mtp2-p010", "abliterated-latest", 80000, (72, 28), projector="gpu", mtp_max=2, vision=True), runner.Case("12-latest-abliterated-80k-mtp3-p010", "abliterated-latest", 80000, (72, 28), projector="gpu", mtp_max=3, quality=True, long_fill=0.75, vision=True), ] _base_command = runner.command def command(case): runner.IMAGE = CURRENT_IMAGE if "current" in case.id else LATEST_IMAGE runner.PROJECTOR = ABLITERATED_PROJECTOR if "abliterated" in case.id else OFFICIAL_PROJECTOR args = _base_command(case) # Keep the vision projector entirely on the secondary GPU. Insert the # environment variable before the image name in `docker run`. if case.projector == "gpu": image_index = args.index(runner.IMAGE) args[image_index:image_index] = ["-e", "MTMD_BACKEND_DEVICE=CUDA1"] if runner.IMAGE == LATEST_IMAGE: args += ["--mmproj-device", "CUDA1"] p_min = "0.05" if "p005" in case.id else "0.10" if case.mtp and "p000" not in case.id: args += ["--spec-draft-p-min", p_min] if "-split-" in case.id: args += [ "--model-draft", SPLIT_DRAFT, # The 6 GB high-precision MTP model belongs wholly on the 3060. # This reserves the faster 5080 for most trunk weights and KV. "--device-draft", "CUDA1", "--n-gpu-layers-draft", "999", ] return args def production(action: str) -> None: names = [ "mike-ai-open-webui", "mike-ai-router", "mike-ai-profile-controller", "mike-ai-llama-fast", "mike-ai-llama-medium", "mike-ai-llama-large", "mike-ai-llama-ultra", "mike-ai-llama-experimental", ] if action == "stop": runner.run(["docker", "stop", *names], check=False, timeout=240) return runner.run([ "docker", "compose", "-f", "/opt/mike-ai/stack/compose.yaml", "--env-file", "/etc/mike-ai/stack.env", "--profile", "inference", "up", "-d", "router", "open-webui", "profile-controller", "llama-medium", ], check=False, timeout=900) runner.command = command runner.production = production raise SystemExit(runner.main())