Add final Qwen3.8 runtime and uncensored profile
This commit is contained in:
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Final pre-move Qwen3.8 runtime, MTP, split-draft and ablation matrix.
|
||||
|
||||
This wrapper reuses the proven isolated benchmark runner already deployed on
|
||||
Athena. It operates only on synthetic prompts, restores the Medium production
|
||||
profile on every exit path and keeps the candidate runtime image separate.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
|
||||
BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py")
|
||||
SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE)
|
||||
if SPEC is None or SPEC.loader is None:
|
||||
raise RuntimeError(f"Could not load {BASE}")
|
||||
|
||||
runner = importlib.util.module_from_spec(SPEC)
|
||||
sys.modules[SPEC.name] = runner
|
||||
SPEC.loader.exec_module(runner)
|
||||
|
||||
CURRENT_IMAGE = "mike-ai/llama.cpp:local"
|
||||
LATEST_IMAGE = "mike-ai/llama.cpp:3f545bec-test"
|
||||
OFFICIAL_PROJECTOR = "/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf"
|
||||
ABLITERATED_PROJECTOR = "/models/qwen3.8-27b-abliterated/mmproj-Qwen3.8-27B-ABLITERATED-F16.gguf"
|
||||
SPLIT_DRAFT = "/models/qwen3.8-nvfp4-split/mtp-Qwen3.8-27B-NVFP4.gguf"
|
||||
|
||||
runner.OUT = pathlib.Path("/data/benchmarks/qwen38-final-acceptance-20260822")
|
||||
runner.TASK_FILE = pathlib.Path("/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json")
|
||||
runner.MODELS.update({
|
||||
"pure-current": runner.MODELS["iq4-pure"],
|
||||
"pure-latest": runner.MODELS["iq4-pure"],
|
||||
"nvfp4-latest": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf",
|
||||
"abliterated-latest": "/models/qwen3.8-27b-abliterated/Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf",
|
||||
})
|
||||
|
||||
# The case IDs intentionally encode the parameters that the legacy Case
|
||||
# dataclass does not know (runtime image, p-min and separate draft model).
|
||||
runner.CASES = [
|
||||
runner.Case("01-current-pure-160k-mtp3-p000", "pure-current", 160000, (90, 10), projector="gpu", quality=True, long_fill=0.75),
|
||||
runner.Case("02-latest-pure-160k-mtp3-p000", "pure-latest", 160000, (90, 10), projector="gpu", quality=True, long_fill=0.75),
|
||||
runner.Case("03-current-fast-76k-mtp2-p000", "iq4-mix", 76800, None, projector="gpu", mtp_max=2),
|
||||
runner.Case("04-latest-fast-76k-mtp2-p000", "iq4-mix", 76800, None, projector="gpu", mtp_max=2),
|
||||
runner.Case("05-latest-pure-160k-mtp2-p010", "pure-latest", 160000, (90, 10), projector="gpu", mtp_max=2),
|
||||
runner.Case("06-latest-pure-160k-mtp3-p005", "pure-latest", 160000, (90, 10), projector="gpu", mtp_max=3),
|
||||
runner.Case("07-latest-pure-160k-mtp4-p010", "pure-latest", 160000, (90, 10), projector="gpu", mtp_max=4),
|
||||
runner.Case("08-latest-nvfp4-72k-embedded-mtp3-p010", "nvfp4-latest", 72000, (72, 28), mtp_max=3),
|
||||
runner.Case("09-latest-nvfp4-72k-split-mtp2-p010", "nvfp4-latest", 72000, (85, 15), mtp_max=2),
|
||||
runner.Case("10-latest-nvfp4-72k-split-mtp3-p010", "nvfp4-latest", 72000, (85, 15), mtp_max=3, quality=True),
|
||||
runner.Case("11-latest-abliterated-80k-mtp2-p010", "abliterated-latest", 80000, (72, 28), projector="gpu", mtp_max=2, vision=True),
|
||||
runner.Case("12-latest-abliterated-80k-mtp3-p010", "abliterated-latest", 80000, (72, 28), projector="gpu", mtp_max=3, quality=True, long_fill=0.75, vision=True),
|
||||
]
|
||||
|
||||
_base_command = runner.command
|
||||
|
||||
|
||||
def command(case):
|
||||
runner.IMAGE = CURRENT_IMAGE if "current" in case.id else LATEST_IMAGE
|
||||
runner.PROJECTOR = ABLITERATED_PROJECTOR if "abliterated" in case.id else OFFICIAL_PROJECTOR
|
||||
args = _base_command(case)
|
||||
|
||||
# Keep the vision projector entirely on the secondary GPU. Insert the
|
||||
# environment variable before the image name in `docker run`.
|
||||
if case.projector == "gpu":
|
||||
image_index = args.index(runner.IMAGE)
|
||||
args[image_index:image_index] = ["-e", "MTMD_BACKEND_DEVICE=CUDA1"]
|
||||
if runner.IMAGE == LATEST_IMAGE:
|
||||
args += ["--mmproj-device", "CUDA1"]
|
||||
|
||||
p_min = "0.05" if "p005" in case.id else "0.10"
|
||||
if case.mtp and "p000" not in case.id:
|
||||
args += ["--spec-draft-p-min", p_min]
|
||||
|
||||
if "-split-" in case.id:
|
||||
args += [
|
||||
"--model-draft", SPLIT_DRAFT,
|
||||
# The 6 GB high-precision MTP model belongs wholly on the 3060.
|
||||
# This reserves the faster 5080 for most trunk weights and KV.
|
||||
"--device-draft", "CUDA1",
|
||||
"--n-gpu-layers-draft", "999",
|
||||
]
|
||||
return args
|
||||
|
||||
|
||||
def production(action: str) -> None:
|
||||
names = [
|
||||
"mike-ai-open-webui", "mike-ai-router", "mike-ai-profile-controller",
|
||||
"mike-ai-llama-fast", "mike-ai-llama-medium", "mike-ai-llama-large",
|
||||
"mike-ai-llama-ultra", "mike-ai-llama-experimental",
|
||||
]
|
||||
if action == "stop":
|
||||
runner.run(["docker", "stop", *names], check=False, timeout=240)
|
||||
return
|
||||
runner.run([
|
||||
"docker", "compose", "-f", "/opt/mike-ai/stack/compose.yaml",
|
||||
"--env-file", "/etc/mike-ai/stack.env", "--profile", "inference",
|
||||
"up", "-d", "router", "open-webui", "profile-controller", "llama-medium",
|
||||
], check=False, timeout=900)
|
||||
|
||||
|
||||
runner.command = command
|
||||
runner.production = production
|
||||
|
||||
raise SystemExit(runner.main())
|
||||
Reference in New Issue
Block a user