#!/usr/bin/env python3 """Bounded, isolated Qwen quantization benchmark. Supervisor restores production.""" import json, pathlib, subprocess, sys, time, urllib.request, threading, signal ROOT = pathlib.Path('/data/benchmarks/profile-mtp2-20260920') NAME = 'mike-ai-profile-mtp2-test' BASE = 'http://127.0.0.1:5005' GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe' GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b' IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907' MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'} def cmd(*args, check=True, timeout=90): r = subprocess.run(args, capture_output=True, text=True, timeout=timeout) if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:]) return r.stdout def api(path, data=None, timeout=900): req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'}) with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r) def save(path, data): path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n') def gpu(): rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu,pcie.link.gen.current,pcie.link.width.current','--format=csv,noheader,nounits',timeout=15) return [dict(zip(['name','used','total','temp','util','pcie_gen','pcie_width'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()] def health_check(): rows=gpu() if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit') mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines()) if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB') return rows def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None): health_check() p={'model':'qwen-medium','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False} if tools: p.update(tools=tools,tool_choice='auto') start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start health_check() return r def prefill(n, seed): import gzip payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)] r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed']) content=r['choices'][0]['message'].get('content','') r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']} return r def run_case(case): label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True) if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label) save(out/'config.json',case) print('START',label,flush=True) single=case.get('single',False) production=json.loads((ROOT/(case['profile']+'.json')).read_text()) assert production['image']==IMAGE args=list(production['args']) for flag,value in [('--ubatch-size',str(case['ubatch'])),('--tensor-split',case['split']),('--spec-draft-n-max',str(case['mtp'])),('--host','127.0.0.1'),('--port','5005')]: args[args.index(flag)+1]=value args += ['--verbosity','3'] save(out/'server-args.json',args) cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args) stop=threading.Event(); samples=[] def monitor(): while not stop.wait(2): try: rows=health_check() samples.append({'time':time.time(),'gpus':rows}) except RuntimeError as e: samples.append({'error':str(e),'aborted':True}) cmd('docker','stop','-t','10',NAME,check=False) return except Exception as e: samples.append({'error':str(e)}) thread=threading.Thread(target=monitor,daemon=True); thread.start() result={'case':case,'started':time.time()} try: for _ in range(150): try: if api('/health',timeout=3).get('status')=='ok': break except Exception: pass if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load') time.sleep(2) else: raise RuntimeError('Startup exceeded 300s') result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots') save(out/'loaded.json',result) # Added after the 88:12 trial: model loading alone can succeed while # the first real attention graph still needs more CUDA workspace. minimum=case.get('minimum_headroom_mib',512) used_devices=['5080'] if single else ['5080','3060'] for g in result['idle_gpu']: if any(device in g['name'] for device in used_devices): free=int(g['total'])-int(g['used']) if free=262144: break previous=context context=min(262144,context+growth) final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]} return run_case(final) if __name__=='__main__': for case in json.loads(pathlib.Path(sys.argv[1]).read_text()): if case.get('capacity_search'): capacity_case(case) else: run_case(case)