#!/usr/bin/env python3 """Focused 256K validation for the Qwen3.8 IQ4_XS-pure artifact. Reuses the synthetic dual-GPU runner. It reads no chats or user data, stops production inference for the isolated test and restores the Fast profile in the runner's finally block. """ from __future__ import annotations import importlib.util import pathlib import sys BASE = pathlib.Path("/opt/mike-ai/stack/run_qwen38_dualgpu_exhaustive.py") SPEC = importlib.util.spec_from_file_location("qwen38_dualgpu", BASE) if SPEC is None or SPEC.loader is None: raise RuntimeError(f"Could not load {BASE}") runner = importlib.util.module_from_spec(SPEC) sys.modules[SPEC.name] = runner SPEC.loader.exec_module(runner) runner.OUT = pathlib.Path("/data/benchmarks/qwen38-pure-256k-20260822") runner.CASES = [ runner.Case( "pure-262k-80-20-mtp2", "iq4-pure", 262144, (80, 20), mtp=True, mtp_max=2, quality=False, long_fill=0.84, ), ] raise SystemExit(runner.main())