30 lines
2.7 KiB
Python
30 lines
2.7 KiB
Python
#!/usr/bin/env python3
|
|
"""Read-only live checks; no inference request, container restart or system change."""
|
|
import datetime,json,re,subprocess
|
|
|
|
def run(*args):return subprocess.check_output(args,text=True,timeout=30).strip()
|
|
report={'checked_at':datetime.datetime.now(datetime.timezone.utc).isoformat(),'scope':'read-only state/health/capacity checks, existing benchmark reuse; no new throughput benchmark'}
|
|
names=['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']
|
|
report['containers']={}
|
|
for name in names:
|
|
d=json.loads(run('docker','inspect',name))[0]
|
|
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'started':d['State']['StartedAt'],'image':d['Image']}
|
|
if name==names[0]:report['medium_args']=d['Args'];start=d['State']['StartedAt']
|
|
report['gpu']=run('nvidia-smi','--query-gpu=name,driver_version,memory.used,memory.total,utilization.gpu,temperature.gpu,power.draw,power.limit,pcie.link.gen.current,pcie.link.gen.max,pcie.link.width.current,pcie.link.width.max','--format=csv')
|
|
report['vmstat']=run('vmstat','1','3')
|
|
report['memory']=run('free','-m')
|
|
report['disk']=run('df','-h','/','/data')
|
|
report['uptime']=run('uptime')
|
|
probe='import urllib.request,json;print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=10)) for p in ["/health","/ready"]}))'
|
|
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
|
|
report['model_health']=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','http://127.0.0.1:8080/health'))
|
|
props=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','http://127.0.0.1:8080/props'))
|
|
report['model_properties']={k:props.get(k) for k in ['model_alias','model_ftype','model_path','total_slots','modalities','build_info']}
|
|
p=subprocess.run(['docker','logs','--since',start,names[0]],capture_output=True,text=True,timeout=30)
|
|
report['startup_findings']=[s for s in (p.stdout+p.stderr).splitlines() if re.search('failed to allocate|allocation failed|without pipeline|model loaded|initializing, n_slots',s)]
|
|
journal=run('journalctl','-k','--since','2026-09-20 00:00:00','--no-pager')
|
|
report['kernel_errors']=[s for s in journal.splitlines() if re.search('NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',s,re.I)]
|
|
report['tests']={'containers_healthy':all(x['running'] and x['health']=='healthy' for x in report['containers'].values()),'router_ready':report['router']['/ready'].get('upstream')=='ready','model_ready':report['model_health'].get('status')=='ok','no_recorded_kernel_faults':not report['kernel_errors']}
|
|
print(json.dumps(report,indent=2,ensure_ascii=False))
|
|
assert all(report['tests'].values()),report['tests']
|