#!/usr/bin/env python3 """Final acceptance of the fastest stable Abliterated 80K candidate.""" from __future__ import annotations import base64 import importlib.util import pathlib import sys import time BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py") SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE) if SPEC is None or SPEC.loader is None: raise RuntimeError(f"Could not load {BASE}") runner = importlib.util.module_from_spec(SPEC) sys.modules[SPEC.name] = runner SPEC.loader.exec_module(runner) runner.IMAGE = "mike-ai/llama.cpp:3f545bec-test" runner.PROJECTOR = "/models/qwen3.8-27b-abliterated/mmproj-Qwen3.8-27B-ABLITERATED-F16.gguf" runner.OUT = pathlib.Path("/data/benchmarks/qwen38-abliterated-final-20260822") runner.TASK_FILE = pathlib.Path("/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json") runner.MODELS["abliterated"] = "/models/qwen3.8-27b-abliterated/Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf" runner.CASES = [ runner.Case("17-abliterated-80k-90x10-mtp2-p010-final", "abliterated", 80000, (90, 10), projector="gpu", mtp_max=2, quality=True, long_fill=0.75, vision=True), ] _base_command = runner.command def command(case): args = _base_command(case) image_index = args.index(runner.IMAGE) args[image_index:image_index] = ["-e", "MTMD_BACKEND_DEVICE=CUDA1"] return args + ["--mmproj-device", "CUDA1", "--spec-draft-p-min", "0.10"] def production(action: str) -> None: names = [ "mike-ai-open-webui", "mike-ai-router", "mike-ai-profile-controller", "mike-ai-llama-fast", "mike-ai-llama-medium", "mike-ai-llama-large", "mike-ai-llama-ultra", "mike-ai-llama-uncensored", "mike-ai-llama-experimental", ] if action == "stop": runner.run(["docker", "stop", *names], check=False, timeout=240) return runner.run([ "docker", "compose", "-f", "/opt/mike-ai/stack/compose.yaml", "--env-file", "/etc/mike-ai/stack.env", "--profile", "inference", "up", "-d", "router", "open-webui", "profile-controller", "llama-medium", ], check=False, timeout=900) def vision_probe(case, image: bytes) -> dict: payload = { "model": case.id, "temperature": 0.1, "max_tokens": 1200, "messages": [{"role": "user", "content": [ {"type": "text", "text": "Describe the image precisely. What animal or object is visible, and what text can you read? Do not guess."}, {"type": "image_url", "image_url": {"url": "data:image/jpeg;base64," + base64.b64encode(image).decode()}}, ]}], } started = time.monotonic() try: response = runner.api("/v1/chat/completions", payload, timeout=1200) choice = (response.get("choices") or [{}])[0] message = choice.get("message") or {} return {"ok": True, "elapsed_seconds": round(time.monotonic() - started, 3), "finish_reason": choice.get("finish_reason"), "content": message.get("content", ""), "reasoning_content": message.get("reasoning_content", ""), "timings": response.get("timings", {})} except Exception as exc: return {"ok": False, "elapsed_seconds": round(time.monotonic() - started, 3), "error": f"{type(exc).__name__}: {exc}"} runner.command = command runner.production = production runner.vision_probe = vision_probe raise SystemExit(runner.main())