30 lines
2.6 KiB
Python
30 lines
2.6 KiB
Python
#!/usr/bin/env python3
|
|
"""Read-only restoration verification plus a two-token model smoke request."""
|
|
import json,pathlib,re,subprocess,time
|
|
ROOT=pathlib.Path('/data/benchmarks/byteshape-20260920')
|
|
def run(*args):return subprocess.check_output(args,text=True,timeout=30)
|
|
report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}}
|
|
for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']:
|
|
d=json.loads(run('docker','inspect',name))[0]
|
|
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']}
|
|
assert d['State']['Running'],name
|
|
if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']:
|
|
assert d['State'].get('Health',{}).get('Status')=='healthy',name
|
|
assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
|
|
assert not run('docker','ps','-q','--filter','name=^mike-ai-byteshape-test$').strip()
|
|
probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))'
|
|
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
|
|
payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0}
|
|
r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions'))
|
|
report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')}
|
|
assert report['smoke']['content'].strip()=='OK',report['smoke']
|
|
started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json'))
|
|
journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager')
|
|
pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I)
|
|
report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)]
|
|
report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip()
|
|
report['disk']=run('df','-h','/','/data').strip()
|
|
(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n')
|
|
print(json.dumps(report,indent=2))
|
|
assert not report['kernel_errors'],'Kernel/GPU errors require review'
|