162 lines
11 KiB
Python
162 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""Bounded, isolated Qwen quantization benchmark. Supervisor restores production."""
|
|
import json, pathlib, subprocess, sys, time, urllib.request, threading, signal
|
|
ROOT = pathlib.Path('/data/benchmarks/dflash2-medium-20260920')
|
|
NAME = 'mike-ai-dflash2-test'
|
|
BASE = 'http://127.0.0.1:5005'
|
|
GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe'
|
|
GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b'
|
|
IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
|
|
MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'}
|
|
|
|
def cmd(*args, check=True, timeout=90):
|
|
r = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
|
|
if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:])
|
|
return r.stdout
|
|
|
|
def api(path, data=None, timeout=900):
|
|
req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'})
|
|
with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r)
|
|
|
|
def save(path, data):
|
|
path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n')
|
|
|
|
def gpu():
|
|
rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu','--format=csv,noheader,nounits',timeout=15)
|
|
return [dict(zip(['name','used','total','temp','util'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()]
|
|
|
|
def health_check():
|
|
rows=gpu()
|
|
if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit')
|
|
mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines())
|
|
if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB')
|
|
return rows
|
|
|
|
def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None):
|
|
health_check()
|
|
p={'model':'benchmark','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False}
|
|
if tools: p.update(tools=tools,tool_choice='auto')
|
|
start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start
|
|
health_check()
|
|
return r
|
|
|
|
def prefill(n, seed):
|
|
import gzip
|
|
payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)]
|
|
r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed'])
|
|
content=r['choices'][0]['message'].get('content','')
|
|
r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']}
|
|
return r
|
|
|
|
def run_case(case):
|
|
label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True)
|
|
if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label)
|
|
save(out/'config.json',case)
|
|
print('START',label,flush=True)
|
|
single=case.get('single',False)
|
|
args=['--model','/models/'+MODELS[case['model']], '--alias','benchmark','--ctx-size',str(case['ctx']), '--flash-attn','on','--cache-type-k','q4_0','--cache-type-v','q4_0','--cache-ram','0','--threads','6','--threads-batch','6','--batch-size','2048','--ubatch-size',str(case.get('ubatch',512)), '--parallel','2','--kv-unified','--jinja','--reasoning','auto','--reasoning-preserve','--host','127.0.0.1','--port','5005','--metrics','--fit','off','--n-gpu-layers','all','--load-mode','none','--no-ui','--temperature','1.0','--top-p','0.95','--top-k','20','--device','CUDA0' if single else 'CUDA0,CUDA1','--main-gpu','0','--split-mode','layer','--tensor-split',case.get('split','1,0'),'--spec-type','draft-dflash','--spec-draft-n-max','7','--spec-draft-type-k','q4_0','--spec-draft-type-v','q4_0','--spec-draft-p-min','0.05','--verbosity','3']
|
|
args += ['--spec-draft-model','/models/qwen38-dflash2/draft.gguf','--spec-draft-device','CUDA0','--spec-draft-ngl','all']
|
|
save(out/'server-args.json',args)
|
|
cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
|
|
stop=threading.Event(); samples=[]
|
|
def monitor():
|
|
while not stop.wait(2):
|
|
try:
|
|
rows=health_check()
|
|
samples.append({'time':time.time(),'gpus':rows})
|
|
except RuntimeError as e:
|
|
samples.append({'error':str(e),'aborted':True})
|
|
cmd('docker','stop','-t','10',NAME,check=False)
|
|
return
|
|
except Exception as e: samples.append({'error':str(e)})
|
|
thread=threading.Thread(target=monitor,daemon=True); thread.start()
|
|
result={'case':case,'started':time.time()}
|
|
try:
|
|
for _ in range(150):
|
|
try:
|
|
if api('/health',timeout=3).get('status')=='ok': break
|
|
except Exception: pass
|
|
if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load')
|
|
time.sleep(2)
|
|
else: raise RuntimeError('Startup exceeded 300s')
|
|
result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots')
|
|
save(out/'loaded.json',result)
|
|
# Added after the 88:12 trial: model loading alone can succeed while
|
|
# the first real attention graph still needs more CUDA workspace.
|
|
minimum=case.get('minimum_headroom_mib',512)
|
|
used_devices=['5080'] if single else ['5080','3060']
|
|
for g in result['idle_gpu']:
|
|
if any(device in g['name'] for device in used_devices):
|
|
free=int(g['total'])-int(g['used'])
|
|
if free<minimum:
|
|
raise RuntimeError(f"Insufficient loaded VRAM reserve on {g['name']}: {free} < {minimum} MiB; refusing inference")
|
|
result['smoke']=chat('Antworte nur mit OK.',8)
|
|
if case.get('quality'):
|
|
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
|
|
result['quality']=[]
|
|
for task in tasks:
|
|
ans=chat(task['prompt'],task['max_tokens'],'medium')
|
|
result['quality'].append({'id':task['id'],'response':ans})
|
|
save(out/'partial.json',result); print(label,task['id'],round(ans['wall_seconds'],1),flush=True)
|
|
tool={'type':'function','function':{'name':'read_server_status','description':'Read-only server status lookup','parameters':{'type':'object','properties':{'server':{'type':'string'}},'required':['server'],'additionalProperties':False}}}
|
|
result['tool']=chat('Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.',512,tools=[tool])
|
|
if case.get('quality_followup'):
|
|
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
|
|
result['quality_followup']=[]
|
|
for task in tasks:
|
|
if task['id'] not in ['i3_code_debugging','i4_capacity_planning','i6_state_vs_configuration']: continue
|
|
ans=chat(task['prompt'],8192,'medium',seed=43)
|
|
result['quality_followup'].append({'id':task['id'],'seed':43,'budget':8192,'response':ans})
|
|
save(out/'partial.json',result); print(label,'followup',task['id'],round(ans['wall_seconds'],1),flush=True)
|
|
if not case.get('load_only'):
|
|
result['decode']=[]
|
|
prompts=['Erkläre ausführlich auf Deutsch, wie ein Reverse Proxy funktioniert, welche Fehler bei Container-IP-Wechseln auftreten können und wie man sie anhand von Logs eingrenzt. Schreibe mindestens 600 Wörter.', 'Write a Python implementation of an asynchronous first_success function: start all awaitables concurrently, return the first successful result, cancel and await remaining tasks, collect exceptions if all fail. Include an explanation and usage example.']
|
|
for i,p in enumerate(prompts): result['decode'].append(chat(p,768,seed=42+i))
|
|
result['prefill']=[]
|
|
for n in case.get('prompts',[4096,16384]):
|
|
r=prefill(n,42); result['prefill'].append({'target':n,'response':r}); save(out/'partial.json',result)
|
|
print(label,'prefill',n,r.get('timings'),flush=True)
|
|
result['finished']=time.time()
|
|
finally:
|
|
stop.set(); thread.join(5)
|
|
r=subprocess.run(['docker','logs',NAME],capture_output=True,text=True,timeout=30)
|
|
(out/'server.log').write_text(r.stdout+r.stderr)
|
|
save(out/'gpu.json',samples); save(out/'result.json',result)
|
|
cmd('docker','rm','-f',NAME,check=False)
|
|
print('DONE',label,flush=True)
|
|
return result
|
|
|
|
def capacity_case(case):
|
|
"""Bounded growth from a previously working context, with 768 MiB reserve.
|
|
|
|
0.04 MiB/token exceeds the measured 512-ubatch steady-state slope.
|
|
A failed 114688/512 ByteShape startup revealed additional transient MTP
|
|
buffers, so reserve is deliberately larger than steady-state extrapolation.
|
|
Smaller ubatches start at an already working context, not a guessed OOM edge.
|
|
The final context is tested with an actual almost-full prompt.
|
|
"""
|
|
context=case['ctx']
|
|
previous=None
|
|
for attempt in range(8):
|
|
pilot={**case,'ctx':context,'label':case['label']+'-pilot-'+str(context),'load_only':True,'quality':False,'quality_followup':False}
|
|
result=run_case(pilot)
|
|
rows=[g for g in result['idle_gpu'] if '5080' in g['name']]
|
|
samples=json.loads((ROOT/pilot['label']/'gpu.json').read_text())
|
|
peak=max([int(rows[0]['used'])+32]+[int(g['used']) for s in samples for g in s.get('gpus',[]) if '5080' in g['name']])
|
|
free=int(rows[0]['total'])-peak
|
|
if free<768:
|
|
if previous is None: raise RuntimeError('Initial capacity pilot has insufficient reserve')
|
|
context=previous
|
|
break
|
|
growth=min(16384,int((free-768)/0.04)//1024*1024)
|
|
if growth<1024 or attempt==7 or context>=262144: break
|
|
previous=context
|
|
context=min(262144,context+growth)
|
|
final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]}
|
|
return run_case(final)
|
|
|
|
if __name__=='__main__':
|
|
for case in json.loads(pathlib.Path(sys.argv[1]).read_text()):
|
|
if case.get('capacity_search'): capacity_case(case)
|
|
else: run_case(case)
|