Test Medium microbatch 256 against matched 512 control
This commit is contained in:
@@ -0,0 +1,35 @@
|
||||
# Medium microbatch quick A/B, 2026-09-20
|
||||
|
||||
User-authorized first-priority test: microbatch512 versus256, all other Medium
|
||||
model settings unchanged. Three sequential requests per arm: frozen German and
|
||||
code decode prompts (768-token caps), then frozen24576-target prefill/recall
|
||||
(24674 actual prompt tokens,512 output-token cap). No reasoning in these speed
|
||||
probes; same sampling/seeds as archived requests. Small matched512 control is
|
||||
necessary because the frozen baseline has different slot/vision/context settings;
|
||||
this is not a repeat of the broad Qwen benchmark or an overwrite of that baseline.
|
||||
|
||||
Supervisor snapshots actual production image/Args into production.json. Runner
|
||||
copies them, changes only microbatch and isolated bind address/port, and sets
|
||||
log verbosity3. Vision remains loaded on3060; context160000 shared by two slots,
|
||||
Pure model,85:15 split,MTP3, q4_0 target KV, f16 MTP KV, batch2048, six CPU threads,
|
||||
cache-RAM32768, TTS resident. Requests disable prompt cache for cold measurement.
|
||||
No concurrency/vision task/full-context validation in this small speed probe.
|
||||
|
||||
Bounded600-second child; production requests drain before stopping the existing
|
||||
router/controller/Medium. Isolated read-only model mount,26GiB memory cap with no
|
||||
additional swap allowance,512MiB loaded VRAM reserve,85C/3GiB host RAM guard,
|
||||
no core dumps. Supervisor restores the same production containers and waits for
|
||||
healthy status. Kernel/network/driver untouched. Each result directory is unique
|
||||
and cannot be overwritten. New scripts are not run automatically from checkout.
|
||||
|
||||
Check startup logs for successful allocation versus explicit fallback without
|
||||
pipeline parallelism. Absence of a fallback line alone is not proof of actual
|
||||
pipeline utilization; source/verbose logs may be needed to establish activation.
|
||||
Judge walltime, prompt/decode rates, VRAM and the three-needle recall together.
|
||||
|
||||
## Outcome
|
||||
|
||||
512 control completed;256 loaded without the earlier fallback message but had
|
||||
only461MiB free on5080, below the512MiB guard. No256 inference was executed.
|
||||
Args differ only in --ubatch-size. Production512 restored and checked. See
|
||||
../../docs/MEDIUM_MICROBATCH_20260920.md. Do not claim a256 speedup or quality result.
|
||||
@@ -0,0 +1,10 @@
|
||||
{
|
||||
"only_model_argument_change": [
|
||||
{
|
||||
"index": 27,
|
||||
"flag": "--ubatch-size",
|
||||
"before": "512",
|
||||
"after": "256"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
{
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
|
||||
"args": [
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-medium",
|
||||
"--ctx-size",
|
||||
"160000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"512",
|
||||
"--parallel",
|
||||
"2",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"8080",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"1.0",
|
||||
"--top-p",
|
||||
"0.95",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"85,15",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"3",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16",
|
||||
"--spec-draft-p-min",
|
||||
"0.05"
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
[
|
||||
{
|
||||
"label": "medium-ub512",
|
||||
"model": "pure",
|
||||
"ctx": 160000,
|
||||
"parallel": 2,
|
||||
"single": false,
|
||||
"split": "85,15",
|
||||
"ubatch": 512,
|
||||
"mtp": 3,
|
||||
"vision": true,
|
||||
"prompts": [
|
||||
24576
|
||||
],
|
||||
"minimum_headroom_mib": 512
|
||||
},
|
||||
{
|
||||
"label": "medium-ub256",
|
||||
"model": "pure",
|
||||
"ctx": 160000,
|
||||
"parallel": 2,
|
||||
"single": false,
|
||||
"split": "85,15",
|
||||
"ubatch": 256,
|
||||
"mtp": 3,
|
||||
"vision": true,
|
||||
"prompts": [
|
||||
24576
|
||||
],
|
||||
"minimum_headroom_mib": 512
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,61 @@
|
||||
{
|
||||
"checked_at": 1789932247.2439013,
|
||||
"uptime": "21:24:07 up 3 days, 10:19, 1 user, load average: 0.49, 0.39, 0.47",
|
||||
"containers": {
|
||||
"mike-ai-llama-medium": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
|
||||
"started": "2026-09-20T19:23:24.996629585Z"
|
||||
},
|
||||
"mike-ai-router": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb",
|
||||
"started": "2026-09-20T19:23:35.441398885Z"
|
||||
},
|
||||
"mike-ai-profile-controller": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f",
|
||||
"started": "2026-09-20T19:23:35.287332793Z"
|
||||
},
|
||||
"mike-ai-wireguard-gateway": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0",
|
||||
"started": "2026-09-17T09:05:07.164533904Z"
|
||||
},
|
||||
"mike-ai-qwen3-tts": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98",
|
||||
"started": "2026-09-19T13:47:00.227813668Z"
|
||||
}
|
||||
},
|
||||
"router": {
|
||||
"/health": {
|
||||
"status": "ok",
|
||||
"router": "alive"
|
||||
},
|
||||
"/ready": {
|
||||
"status": "ok",
|
||||
"router": "alive",
|
||||
"upstream": "ready"
|
||||
}
|
||||
},
|
||||
"smoke": {
|
||||
"content": "OK",
|
||||
"usage": {
|
||||
"completion_tokens": 2,
|
||||
"prompt_tokens": 19,
|
||||
"total_tokens": 21,
|
||||
"prompt_tokens_details": {
|
||||
"cached_tokens": 0
|
||||
}
|
||||
}
|
||||
},
|
||||
"kernel_errors": [],
|
||||
"gpu": "name, memory.used [MiB], memory.total [MiB], temperature.gpu\nNVIDIA GeForce RTX 3060, 10920 MiB, 12288 MiB, 44\nNVIDIA GeForce RTX 5080, 15714 MiB, 16303 MiB, 52",
|
||||
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data"
|
||||
}
|
||||
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,171 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Bounded, isolated Qwen quantization benchmark. Supervisor restores production."""
|
||||
import json, pathlib, subprocess, sys, time, urllib.request, threading, signal
|
||||
ROOT = pathlib.Path('/data/benchmarks/medium-microbatch-20260920')
|
||||
NAME = 'mike-ai-microbatch-test'
|
||||
BASE = 'http://127.0.0.1:5005'
|
||||
GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe'
|
||||
GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b'
|
||||
IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
|
||||
MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'}
|
||||
|
||||
def cmd(*args, check=True, timeout=90):
|
||||
r = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
|
||||
if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:])
|
||||
return r.stdout
|
||||
|
||||
def api(path, data=None, timeout=900):
|
||||
req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'})
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r)
|
||||
|
||||
def save(path, data):
|
||||
path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n')
|
||||
|
||||
def gpu():
|
||||
rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu','--format=csv,noheader,nounits',timeout=15)
|
||||
return [dict(zip(['name','used','total','temp','util'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()]
|
||||
|
||||
def health_check():
|
||||
rows=gpu()
|
||||
if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit')
|
||||
mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines())
|
||||
if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB')
|
||||
return rows
|
||||
|
||||
def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None):
|
||||
health_check()
|
||||
p={'model':'qwen-medium','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False}
|
||||
if tools: p.update(tools=tools,tool_choice='auto')
|
||||
start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start
|
||||
health_check()
|
||||
return r
|
||||
|
||||
def prefill(n, seed):
|
||||
import gzip
|
||||
payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)]
|
||||
r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed'])
|
||||
content=r['choices'][0]['message'].get('content','')
|
||||
r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']}
|
||||
return r
|
||||
|
||||
def run_case(case):
|
||||
label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True)
|
||||
if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label)
|
||||
save(out/'config.json',case)
|
||||
print('START',label,flush=True)
|
||||
single=case.get('single',False)
|
||||
production=json.loads((ROOT/'production.json').read_text())
|
||||
assert production['image']==IMAGE
|
||||
args=list(production['args'])
|
||||
for flag,value in [('--ubatch-size',str(case['ubatch'])),('--host','127.0.0.1'),('--port','5005')]:
|
||||
args[args.index(flag)+1]=value
|
||||
args += ['--verbosity','3']
|
||||
save(out/'server-args.json',args)
|
||||
cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
|
||||
stop=threading.Event(); samples=[]
|
||||
def monitor():
|
||||
while not stop.wait(2):
|
||||
try:
|
||||
rows=health_check()
|
||||
samples.append({'time':time.time(),'gpus':rows})
|
||||
except RuntimeError as e:
|
||||
samples.append({'error':str(e),'aborted':True})
|
||||
cmd('docker','stop','-t','10',NAME,check=False)
|
||||
return
|
||||
except Exception as e: samples.append({'error':str(e)})
|
||||
thread=threading.Thread(target=monitor,daemon=True); thread.start()
|
||||
result={'case':case,'started':time.time()}
|
||||
try:
|
||||
for _ in range(150):
|
||||
try:
|
||||
if api('/health',timeout=3).get('status')=='ok': break
|
||||
except Exception: pass
|
||||
if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load')
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Startup exceeded 300s')
|
||||
result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots')
|
||||
save(out/'loaded.json',result)
|
||||
# Added after the 88:12 trial: model loading alone can succeed while
|
||||
# the first real attention graph still needs more CUDA workspace.
|
||||
minimum=case.get('minimum_headroom_mib',512)
|
||||
used_devices=['5080'] if single else ['5080','3060']
|
||||
for g in result['idle_gpu']:
|
||||
if any(device in g['name'] for device in used_devices):
|
||||
free=int(g['total'])-int(g['used'])
|
||||
if free<minimum:
|
||||
raise RuntimeError(f"Insufficient loaded VRAM reserve on {g['name']}: {free} < {minimum} MiB; refusing inference")
|
||||
result['smoke']=chat('Antworte nur mit OK.',8)
|
||||
if case.get('quality'):
|
||||
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
|
||||
result['quality']=[]
|
||||
for task in tasks:
|
||||
ans=chat(task['prompt'],task['max_tokens'],'medium')
|
||||
result['quality'].append({'id':task['id'],'response':ans})
|
||||
save(out/'partial.json',result); print(label,task['id'],round(ans['wall_seconds'],1),flush=True)
|
||||
tool={'type':'function','function':{'name':'read_server_status','description':'Read-only server status lookup','parameters':{'type':'object','properties':{'server':{'type':'string'}},'required':['server'],'additionalProperties':False}}}
|
||||
result['tool']=chat('Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.',512,tools=[tool])
|
||||
if case.get('quality_followup'):
|
||||
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
|
||||
result['quality_followup']=[]
|
||||
for task in tasks:
|
||||
if task['id'] not in ['i3_code_debugging','i4_capacity_planning','i6_state_vs_configuration']: continue
|
||||
ans=chat(task['prompt'],8192,'medium',seed=43)
|
||||
result['quality_followup'].append({'id':task['id'],'seed':43,'budget':8192,'response':ans})
|
||||
save(out/'partial.json',result); print(label,'followup',task['id'],round(ans['wall_seconds'],1),flush=True)
|
||||
if not case.get('load_only'):
|
||||
result['decode']=[]
|
||||
prompts=['Erkläre ausführlich auf Deutsch, wie ein Reverse Proxy funktioniert, welche Fehler bei Container-IP-Wechseln auftreten können und wie man sie anhand von Logs eingrenzt. Schreibe mindestens 600 Wörter.', 'Write a Python implementation of an asynchronous first_success function: start all awaitables concurrently, return the first successful result, cancel and await remaining tasks, collect exceptions if all fail. Include an explanation and usage example.']
|
||||
for i,p in enumerate(prompts):
|
||||
result['decode'].append(chat(p,768,seed=42+i))
|
||||
save(out/'partial.json',result)
|
||||
print(label,'decode',i,result['decode'][-1].get('timings'),flush=True)
|
||||
result['prefill']=[]
|
||||
for n in case.get('prompts',[4096,16384]):
|
||||
r=prefill(n,42); result['prefill'].append({'target':n,'response':r}); save(out/'partial.json',result)
|
||||
print(label,'prefill',n,r.get('timings'),flush=True)
|
||||
result['finished']=time.time()
|
||||
except Exception as exc:
|
||||
result['error']=str(exc)
|
||||
raise
|
||||
finally:
|
||||
stop.set(); thread.join(5)
|
||||
r=subprocess.run(['docker','logs',NAME],capture_output=True,text=True,timeout=30)
|
||||
(out/'server.log').write_text(r.stdout+r.stderr)
|
||||
save(out/'gpu.json',samples); save(out/'result.json',result)
|
||||
cmd('docker','rm','-f',NAME,check=False)
|
||||
print('DONE',label,flush=True)
|
||||
return result
|
||||
|
||||
def capacity_case(case):
|
||||
"""Bounded growth from a previously working context, with 768 MiB reserve.
|
||||
|
||||
0.04 MiB/token exceeds the measured 512-ubatch steady-state slope.
|
||||
A failed 114688/512 ByteShape startup revealed additional transient MTP
|
||||
buffers, so reserve is deliberately larger than steady-state extrapolation.
|
||||
Smaller ubatches start at an already working context, not a guessed OOM edge.
|
||||
The final context is tested with an actual almost-full prompt.
|
||||
"""
|
||||
context=case['ctx']
|
||||
previous=None
|
||||
for attempt in range(8):
|
||||
pilot={**case,'ctx':context,'label':case['label']+'-pilot-'+str(context),'load_only':True,'quality':False,'quality_followup':False}
|
||||
result=run_case(pilot)
|
||||
rows=[g for g in result['idle_gpu'] if '5080' in g['name']]
|
||||
samples=json.loads((ROOT/pilot['label']/'gpu.json').read_text())
|
||||
peak=max([int(rows[0]['used'])+32]+[int(g['used']) for s in samples for g in s.get('gpus',[]) if '5080' in g['name']])
|
||||
free=int(rows[0]['total'])-peak
|
||||
if free<768:
|
||||
if previous is None: raise RuntimeError('Initial capacity pilot has insufficient reserve')
|
||||
context=previous
|
||||
break
|
||||
growth=min(16384,int((free-768)/0.04)//1024*1024)
|
||||
if growth<1024 or attempt==7 or context>=262144: break
|
||||
previous=context
|
||||
context=min(262144,context+growth)
|
||||
final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]}
|
||||
return run_case(final)
|
||||
|
||||
if __name__=='__main__':
|
||||
for case in json.loads(pathlib.Path(sys.argv[1]).read_text()):
|
||||
if case.get('capacity_search'): capacity_case(case)
|
||||
else: run_case(case)
|
||||
@@ -0,0 +1,52 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Stop only existing router/controller/model; always restore the same containers."""
|
||||
import json, pathlib, subprocess, sys, time, signal
|
||||
ROOT=pathlib.Path('/data/benchmarks/medium-microbatch-20260920')
|
||||
NAMES=['mike-ai-router','mike-ai-profile-controller','mike-ai-llama-medium']
|
||||
def run(*args,check=True,timeout=90):
|
||||
return subprocess.run(args,capture_output=True,text=True,check=check,timeout=timeout)
|
||||
def stop_signal(*_): raise RuntimeError('Supervisor interrupted')
|
||||
signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_signal)
|
||||
# Refuse if the known production state has changed, or if requests are active.
|
||||
for name in NAMES:
|
||||
assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
|
||||
for attempt in range(60):
|
||||
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
|
||||
if not any(s['is_processing'] for s in slots): break
|
||||
if attempt==0: print('WAIT production request active; no interruption',flush=True)
|
||||
time.sleep(3)
|
||||
else: raise RuntimeError('Production remained busy for 180s; no services stopped')
|
||||
assert not run('docker','ps','-q','--filter','name=^mike-ai-microbatch-test$').stdout.strip(),'Existing experiment'
|
||||
production=json.loads(run('docker','inspect',NAMES[-1]).stdout)[0]
|
||||
(ROOT/'production.json').write_text(json.dumps({'image':production['Image'],'args':production['Args']},indent=2)+'\n')
|
||||
child=None
|
||||
try:
|
||||
run('docker','stop','-t','30',*NAMES[:2])
|
||||
# Drain requests already handed to the model, before unloading it.
|
||||
for _ in range(120):
|
||||
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
|
||||
if not any(s['is_processing'] for s in slots): break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Model did not drain')
|
||||
run('docker','stop','-t','30',NAMES[-1])
|
||||
child=subprocess.Popen(['python3',str(ROOT/'run.py'),sys.argv[1]])
|
||||
code=child.wait(timeout=600)
|
||||
if code: raise RuntimeError('Benchmark failed: '+str(code))
|
||||
finally:
|
||||
if child is not None and child.poll() is None:
|
||||
child.terminate()
|
||||
try: child.wait(timeout=20)
|
||||
except subprocess.TimeoutExpired: child.kill(); child.wait(timeout=10)
|
||||
run('docker','rm','-f','mike-ai-microbatch-test',check=False)
|
||||
run('docker','start',NAMES[-1])
|
||||
for _ in range(150):
|
||||
if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Restored Medium did not become healthy')
|
||||
run('docker','start',NAMES[1],NAMES[0])
|
||||
for _ in range(60):
|
||||
statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
|
||||
if all(status=='healthy' for status in statuses): break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Restored router/controller did not become healthy')
|
||||
print('RESTORED existing medium/controller/router; all healthy',flush=True)
|
||||
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Read-only restoration verification plus a two-token model smoke request."""
|
||||
import json,pathlib,re,subprocess,time
|
||||
ROOT=pathlib.Path('/data/benchmarks/medium-microbatch-20260920')
|
||||
def run(*args):return subprocess.check_output(args,text=True,timeout=30)
|
||||
report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}}
|
||||
for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']:
|
||||
d=json.loads(run('docker','inspect',name))[0]
|
||||
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']}
|
||||
assert d['State']['Running'],name
|
||||
if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']:
|
||||
assert d['State'].get('Health',{}).get('Status')=='healthy',name
|
||||
assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
|
||||
assert not run('docker','ps','-q','--filter','name=^mike-ai-microbatch-test$').strip()
|
||||
probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))'
|
||||
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
|
||||
payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0}
|
||||
r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions'))
|
||||
report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')}
|
||||
assert report['smoke']['content'].strip()=='OK',report['smoke']
|
||||
started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json'))
|
||||
journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager')
|
||||
pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I)
|
||||
report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)]
|
||||
report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip()
|
||||
report['disk']=run('df','-h','/','/data').strip()
|
||||
(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n')
|
||||
print(json.dumps(report,indent=2))
|
||||
assert not report['kernel_errors'],'Kernel/GPU errors require review'
|
||||
Reference in New Issue
Block a user