Files
AI-Profile-Router/dev/run_qwen38_pure_256k.py
T

40 lines
1.0 KiB
Python

#!/usr/bin/env python3
"""Focused 256K validation for the Qwen3.8 IQ4_XS-pure artifact.
Reuses the synthetic dual-GPU runner. It reads no chats or user data, stops
production inference for the isolated test and restores the Fast profile in
the runner's finally block.
"""
from __future__ import annotations
import importlib.util
import pathlib
import sys
BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py")
SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE)
if SPEC is None or SPEC.loader is None:
raise RuntimeError(f"Could not load {BASE}")
runner = importlib.util.module_from_spec(SPEC)
sys.modules[SPEC.name] = runner
SPEC.loader.exec_module(runner)
runner.OUT = pathlib.Path("/data/benchmarks/qwen38-pure-256k-20260822")
runner.CASES = [
runner.Case(
"pure-262k-80-20-mtp2",
"iq4-pure",
262144,
(80, 20),
mtp=True,
mtp_max=2,
quality=False,
long_fill=0.84,
),
]
raise SystemExit(runner.main())