Add final Qwen3.8 runtime and uncensored profile
This commit is contained in:
@@ -0,0 +1,112 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Tune the final Abliterated profile after the broad acceptance matrix.
|
||||
|
||||
The broad matrix deliberately starts with a conservative 72:28 layer split.
|
||||
This follow-up moves as much work as possible back to the RTX 5080, checks the
|
||||
new MTP probability threshold, and subjects the fastest stable candidate to
|
||||
the same synthetic correctness and long-context checks.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import base64
|
||||
import pathlib
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
|
||||
BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py")
|
||||
SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE)
|
||||
if SPEC is None or SPEC.loader is None:
|
||||
raise RuntimeError(f"Could not load {BASE}")
|
||||
|
||||
runner = importlib.util.module_from_spec(SPEC)
|
||||
sys.modules[SPEC.name] = runner
|
||||
SPEC.loader.exec_module(runner)
|
||||
|
||||
runner.IMAGE = "mike-ai/llama.cpp:3f545bec-test"
|
||||
runner.PROJECTOR = "/models/qwen3.8-27b-abliterated/mmproj-Qwen3.8-27B-ABLITERATED-F16.gguf"
|
||||
runner.OUT = pathlib.Path("/data/benchmarks/qwen38-abliterated-tuning-20260822")
|
||||
runner.TASK_FILE = pathlib.Path("/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json")
|
||||
runner.MODELS["abliterated"] = "/models/qwen3.8-27b-abliterated/Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf"
|
||||
|
||||
runner.CASES = [
|
||||
runner.Case("13-abliterated-80k-85x15-mtp2-p010", "abliterated", 80000, (85, 15), projector="gpu", mtp_max=2, vision=True),
|
||||
runner.Case("14-abliterated-80k-85x15-mtp3-p005", "abliterated", 80000, (85, 15), projector="gpu", mtp_max=3, vision=True),
|
||||
runner.Case("15-abliterated-80k-90x10-mtp3-p005", "abliterated", 80000, (90, 10), projector="gpu", mtp_max=3, vision=True),
|
||||
runner.Case("16-abliterated-80k-85x15-mtp3-p005-final", "abliterated", 80000, (85, 15), projector="gpu", mtp_max=3, quality=True, long_fill=0.75, vision=True),
|
||||
]
|
||||
|
||||
_base_command = runner.command
|
||||
|
||||
|
||||
def command(case):
|
||||
args = _base_command(case)
|
||||
image_index = args.index(runner.IMAGE)
|
||||
args[image_index:image_index] = ["-e", "MTMD_BACKEND_DEVICE=CUDA1"]
|
||||
args += ["--mmproj-device", "CUDA1"]
|
||||
if "p005" in case.id:
|
||||
args += ["--spec-draft-p-min", "0.05"]
|
||||
else:
|
||||
args += ["--spec-draft-p-min", "0.10"]
|
||||
return args
|
||||
|
||||
|
||||
def production(action: str) -> None:
|
||||
names = [
|
||||
"mike-ai-open-webui", "mike-ai-router", "mike-ai-profile-controller",
|
||||
"mike-ai-llama-fast", "mike-ai-llama-medium", "mike-ai-llama-large",
|
||||
"mike-ai-llama-ultra", "mike-ai-llama-uncensored",
|
||||
"mike-ai-llama-experimental",
|
||||
]
|
||||
if action == "stop":
|
||||
runner.run(["docker", "stop", *names], check=False, timeout=240)
|
||||
return
|
||||
runner.run([
|
||||
"docker", "compose", "-f", "/opt/mike-ai/stack/compose.yaml",
|
||||
"--env-file", "/etc/mike-ai/stack.env", "--profile", "inference",
|
||||
"up", "-d", "router", "open-webui", "profile-controller", "llama-medium",
|
||||
], check=False, timeout=900)
|
||||
|
||||
|
||||
runner.command = command
|
||||
runner.production = production
|
||||
|
||||
|
||||
def vision_probe(case, image: bytes) -> dict:
|
||||
"""Allow enough tokens for Qwen to finish reasoning before its answer."""
|
||||
payload = {
|
||||
"model": case.id,
|
||||
"temperature": 0.1,
|
||||
"max_tokens": 1200,
|
||||
"messages": [{"role": "user", "content": [
|
||||
{"type": "text", "text": "Describe the image precisely. What animal or object is visible, and what text can you read? Do not guess."},
|
||||
{"type": "image_url", "image_url": {"url": "data:image/jpeg;base64," + base64.b64encode(image).decode()}},
|
||||
]}],
|
||||
}
|
||||
started = time.monotonic()
|
||||
try:
|
||||
response = runner.api("/v1/chat/completions", payload, timeout=1200)
|
||||
choice = (response.get("choices") or [{}])[0]
|
||||
message = choice.get("message") or {}
|
||||
return {
|
||||
"ok": True,
|
||||
"elapsed_seconds": round(time.monotonic() - started, 3),
|
||||
"finish_reason": choice.get("finish_reason"),
|
||||
"content": message.get("content", ""),
|
||||
"reasoning_content": message.get("reasoning_content", ""),
|
||||
"timings": response.get("timings", {}),
|
||||
}
|
||||
except Exception as exc:
|
||||
return {
|
||||
"ok": False,
|
||||
"elapsed_seconds": round(time.monotonic() - started, 3),
|
||||
"error": f"{type(exc).__name__}: {exc}",
|
||||
}
|
||||
|
||||
|
||||
runner.vision_probe = vision_probe
|
||||
|
||||
raise SystemExit(runner.main())
|
||||
Reference in New Issue
Block a user