Validate Pure quant for 256K Ultra profile

This commit is contained in:
Mikei386
2026-08-22 11:01:13 +02:00
parent f52cf14069
commit 977f8f5c76
9 changed files with 85 additions and 11 deletions
+39
View File
@@ -0,0 +1,39 @@
#!/usr/bin/env python3
"""Focused 256K validation for the Qwen3.8 IQ4_XS-pure artifact.
Reuses the synthetic dual-GPU runner. It reads no chats or user data, stops
production inference for the isolated test and restores the Fast profile in
the runner's finally block.
"""
from __future__ import annotations
import importlib.util
import pathlib
import sys
BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py")
SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE)
if SPEC is None or SPEC.loader is None:
raise RuntimeError(f"Could not load {BASE}")
runner = importlib.util.module_from_spec(SPEC)
sys.modules[SPEC.name] = runner
SPEC.loader.exec_module(runner)
runner.OUT = pathlib.Path("/data/benchmarks/qwen38-pure-256k-20260822")
runner.CASES = [
runner.Case(
"pure-262k-80-20-mtp2",
"iq4-pure",
262144,
(80, 20),
mtp=True,
mtp_max=2,
quality=False,
long_fill=0.84,
),
]
raise SystemExit(runner.main())