7NdpO7ZD{LBog_D6@
zrn!`8Zkd0L=E4at&>SJ+!5afRC%SZnu=;x-IvWvv@5E!TadM67Asd^NlTBVEyYJcN
z!jjz&PRTAj^#a*LZv;YxQ;zZu#EW!i;RU+GM}U^mRc}al#ki+@H|Xt4shmJfIn4!|
zLU(+udn^Ry-3htrO^#3W&>!gTqsx$IbjO%CoajP2NFPpC`TsbQp*lE15cs(xA0It~
zBYb6e_$R%|jxOs8$^{=SL$m8*uP3^~_KonEzL?{pTf9u)1F&E+u*52^-3YMZM
zn~iQIDaYQg-$zQ0-kYn
zkxq(66Va6I9BsJFTP5@k{ft6%L5pv_(&%33HPUO2zB<(gefWyrzPmU!EJf(Dj4RjWFI<{Bc29!4O-?Y0Q&QCpLn3z57h`x4R%b6I
zZ!}c+Mle1{{uSs$W#2F@+6Gk=T5D25!cErE^Ex%Jy9!o{1NEqB!?^NX*Ruf5u=v4&
z^&I0MnW{-fH3j|1Sl@x+?mgYU0TmnmOVvWLExCN(S6Xzk1yI$>i=S$`CApm>9hXGa
zSI82PJ1!euT_HP33Q@aT^lH-E9rgZ&Y~(x56;W#{d2xB>wVfW^^m5U7mB_7wIaS(g)u|KR?YH;eewz10)7CTCnIiH@{IGmOisfMrsZ?}PQa>Oz!QFW
zvDn-O1Fwk!Z;FARwTUida5E{Ig_GTe
zvm_7lr+J3++7WOz0GSO(HjEv@40G5&9Mp>+Nr5w>@dIa+K=p<33pn@bA0~c;7t5uo
z?=YDZcoef>Mw9=|F=7_+6fw-OQ4JX0HjPO= 85 for x in rows): raise RuntimeError('GPU temperature limit')
+ mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines())
+ if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB')
+ return rows
+
+def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None):
+ health_check()
+ p={'model':'qwen-medium','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False}
+ if tools: p.update(tools=tools,tool_choice='auto')
+ start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start
+ health_check()
+ return r
+
+def prefill(n, seed):
+ import gzip
+ payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)]
+ r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed'])
+ content=r['choices'][0]['message'].get('content','')
+ r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']}
+ return r
+
+def run_case(case):
+ label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True)
+ if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label)
+ save(out/'config.json',case)
+ print('START',label,flush=True)
+ single=case.get('single',False)
+ production=json.loads((ROOT/(case['profile']+'.json')).read_text())
+ assert production['image']==IMAGE
+ args=list(production['args'])
+ for flag,value in [('--ubatch-size',str(case['ubatch'])),('--tensor-split',case['split']),('--spec-draft-n-max',str(case['mtp'])),('--host','127.0.0.1'),('--port','5005')]:
+ args[args.index(flag)+1]=value
+ args += ['--verbosity','3']
+ save(out/'server-args.json',args)
+ cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
+ stop=threading.Event(); samples=[]
+ def monitor():
+ while not stop.wait(2):
+ try:
+ rows=health_check()
+ samples.append({'time':time.time(),'gpus':rows})
+ except RuntimeError as e:
+ samples.append({'error':str(e),'aborted':True})
+ cmd('docker','stop','-t','10',NAME,check=False)
+ return
+ except Exception as e: samples.append({'error':str(e)})
+ thread=threading.Thread(target=monitor,daemon=True); thread.start()
+ result={'case':case,'started':time.time()}
+ try:
+ for _ in range(150):
+ try:
+ if api('/health',timeout=3).get('status')=='ok': break
+ except Exception: pass
+ if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load')
+ time.sleep(2)
+ else: raise RuntimeError('Startup exceeded 300s')
+ result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots')
+ save(out/'loaded.json',result)
+ # Added after the 88:12 trial: model loading alone can succeed while
+ # the first real attention graph still needs more CUDA workspace.
+ minimum=case.get('minimum_headroom_mib',512)
+ used_devices=['5080'] if single else ['5080','3060']
+ for g in result['idle_gpu']:
+ if any(device in g['name'] for device in used_devices):
+ free=int(g['total'])-int(g['used'])
+ if free=262144: break
+ previous=context
+ context=min(262144,context+growth)
+ final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]}
+ return run_case(final)
+
+if __name__=='__main__':
+ for case in json.loads(pathlib.Path(sys.argv[1]).read_text()):
+ if case.get('capacity_search'): capacity_case(case)
+ else: run_case(case)
diff --git a/experiments/profile-mtp2-20260920/supervise.py b/experiments/profile-mtp2-20260920/supervise.py
new file mode 100644
index 0000000..ed8eac9
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/supervise.py
@@ -0,0 +1,55 @@
+#!/usr/bin/env python3
+"""Stop only existing router/controller/model; always restore the same containers."""
+import json, pathlib, subprocess, sys, time, signal
+ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
+NAMES=['mike-ai-router','mike-ai-profile-controller','mike-ai-llama-medium']
+def run(*args,check=True,timeout=90):
+ return subprocess.run(args,capture_output=True,text=True,check=check,timeout=timeout)
+def stop_signal(*_): raise RuntimeError('Supervisor interrupted')
+signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_signal)
+# Refuse if the known production state has changed, or if requests are active.
+for name in NAMES:
+ assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
+for attempt in range(60):
+ slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
+ if not any(s['is_processing'] for s in slots): break
+ if attempt==0: print('WAIT production request active; no interruption',flush=True)
+ time.sleep(3)
+else: raise RuntimeError('Production remained busy for 180s; no services stopped')
+assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').stdout.strip(),'Existing experiment'
+production=json.loads(run('docker','inspect',NAMES[-1]).stdout)[0]
+(ROOT/'production.json').write_text(json.dumps({'image':production['Image'],'args':production['Args']},indent=2)+'\n')
+for profile in ['medium','large']:
+ d=json.loads(run('docker','inspect','mike-ai-llama-'+profile).stdout)[0]
+ (ROOT/(profile+'.json')).write_text(json.dumps({'image':d['Image'],'args':d['Args']},indent=2)+'\n')
+child=None
+try:
+ run('docker','stop','-t','30',*NAMES[:2])
+ # Drain requests already handed to the model, before unloading it.
+ for _ in range(120):
+ slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
+ if not any(s['is_processing'] for s in slots): break
+ time.sleep(2)
+ else: raise RuntimeError('Model did not drain')
+ run('docker','stop','-t','30',NAMES[-1])
+ child=subprocess.Popen(['python3',str(ROOT/'run.py'),sys.argv[1]])
+ code=child.wait(timeout=600)
+ if code: raise RuntimeError('Benchmark failed: '+str(code))
+finally:
+ if child is not None and child.poll() is None:
+ child.terminate()
+ try: child.wait(timeout=20)
+ except subprocess.TimeoutExpired: child.kill(); child.wait(timeout=10)
+ run('docker','rm','-f','mike-ai-profile-mtp2-test',check=False)
+ run('docker','start',NAMES[-1])
+ for _ in range(150):
+ if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
+ time.sleep(2)
+ else: raise RuntimeError('Restored Medium did not become healthy')
+ run('docker','start',NAMES[1],NAMES[0])
+ for _ in range(60):
+ statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
+ if all(status=='healthy' for status in statuses): break
+ time.sleep(2)
+ else: raise RuntimeError('Restored router/controller did not become healthy')
+ print('RESTORED existing medium/controller/router; all healthy',flush=True)
diff --git a/experiments/profile-mtp2-20260920/verify_restore.py b/experiments/profile-mtp2-20260920/verify_restore.py
new file mode 100644
index 0000000..03c9bd9
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/verify_restore.py
@@ -0,0 +1,29 @@
+#!/usr/bin/env python3
+"""Read-only restoration verification plus a two-token model smoke request."""
+import json,pathlib,re,subprocess,time
+ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
+def run(*args):return subprocess.check_output(args,text=True,timeout=30)
+report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}}
+for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']:
+ d=json.loads(run('docker','inspect',name))[0]
+ report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']}
+ assert d['State']['Running'],name
+ if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']:
+ assert d['State'].get('Health',{}).get('Status')=='healthy',name
+assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
+assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').strip()
+probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))'
+report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
+payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0}
+r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions'))
+report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')}
+assert report['smoke']['content'].strip()=='OK',report['smoke']
+started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json'))
+journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager')
+pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I)
+report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)]
+report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip()
+report['disk']=run('df','-h','/','/data').strip()
+(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n')
+print(json.dumps(report,indent=2))
+assert not report['kernel_errors'],'Kernel/GPU errors require review'