40 lines
1.0 KiB
Python
40 lines
1.0 KiB
Python
#!/usr/bin/env python3
|
|
"""Focused 256K validation for the Qwen3.8 IQ4_XS-pure artifact.
|
|
|
|
Reuses the synthetic dual-GPU runner. It reads no chats or user data, stops
|
|
production inference for the isolated test and restores the Fast profile in
|
|
the runner's finally block.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import importlib.util
|
|
import pathlib
|
|
import sys
|
|
|
|
|
|
BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py")
|
|
SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE)
|
|
if SPEC is None or SPEC.loader is None:
|
|
raise RuntimeError(f"Could not load {BASE}")
|
|
|
|
runner = importlib.util.module_from_spec(SPEC)
|
|
sys.modules[SPEC.name] = runner
|
|
SPEC.loader.exec_module(runner)
|
|
|
|
runner.OUT = pathlib.Path("/data/benchmarks/qwen38-pure-256k-20260822")
|
|
runner.CASES = [
|
|
runner.Case(
|
|
"pure-262k-80-20-mtp2",
|
|
"iq4-pure",
|
|
262144,
|
|
(80, 20),
|
|
mtp=True,
|
|
mtp_max=2,
|
|
quality=False,
|
|
long_fill=0.84,
|
|
),
|
|
]
|
|
|
|
raise SystemExit(runner.main())
|