Audit Athena quantization inventory and runtime efficiency limits

This commit is contained in:
Mikei386
2026-09-20 21:18:27 +02:00
parent 4007242709
commit 2425c999b0
11 changed files with 532 additions and 1 deletions
@@ -0,0 +1,29 @@
#!/usr/bin/env python3
"""Read-only live checks; no inference request, container restart or system change."""
import datetime,json,re,subprocess
def run(*args):return subprocess.check_output(args,text=True,timeout=30).strip()
report={'checked_at':datetime.datetime.now(datetime.timezone.utc).isoformat(),'scope':'read-only state/health/capacity checks, existing benchmark reuse; no new throughput benchmark'}
names=['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']
report['containers']={}
for name in names:
d=json.loads(run('docker','inspect',name))[0]
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'started':d['State']['StartedAt'],'image':d['Image']}
if name==names[0]:report['medium_args']=d['Args'];start=d['State']['StartedAt']
report['gpu']=run('nvidia-smi','--query-gpu=name,driver_version,memory.used,memory.total,utilization.gpu,temperature.gpu,power.draw,power.limit,pcie.link.gen.current,pcie.link.gen.max,pcie.link.width.current,pcie.link.width.max','--format=csv')
report['vmstat']=run('vmstat','1','3')
report['memory']=run('free','-m')
report['disk']=run('df','-h','/','/data')
report['uptime']=run('uptime')
probe='import urllib.request,json;print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=10)) for p in ["/health","/ready"]}))'
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
report['model_health']=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','http://127.0.0.1:8080/health'))
props=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','http://127.0.0.1:8080/props'))
report['model_properties']={k:props.get(k) for k in ['model_alias','model_ftype','model_path','total_slots','modalities','build_info']}
p=subprocess.run(['docker','logs','--since',start,names[0]],capture_output=True,text=True,timeout=30)
report['startup_findings']=[s for s in (p.stdout+p.stderr).splitlines() if re.search('failed to allocate|allocation failed|without pipeline|model loaded|initializing, n_slots',s)]
journal=run('journalctl','-k','--since','2026-09-20 00:00:00','--no-pager')
report['kernel_errors']=[s for s in journal.splitlines() if re.search('NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',s,re.I)]
report['tests']={'containers_healthy':all(x['running'] and x['health']=='healthy' for x in report['containers'].values()),'router_ready':report['router']['/ready'].get('upstream')=='ready','model_ready':report['model_health'].get('status')=='ok','no_recorded_kernel_faults':not report['kernel_errors']}
print(json.dumps(report,indent=2,ensure_ascii=False))
assert all(report['tests'].values()),report['tests']
@@ -0,0 +1,38 @@
[
{
"path": "/data/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"bytes": 14534384640,
"metadata": {
"general.architecture": "qwen35",
"general.name": "Qwen3.8-27B",
"general.quantization_version": 2,
"general.file_type": 30
},
"tensor_count": 866,
"tensor_types": {
"IQ4_XS": 506,
"F32": 360
},
"descriptor_end": 10996711
},
{
"path": "/data/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf",
"bytes": 14111614400,
"metadata": {
"general.architecture": "qwen35",
"general.name": "Hf_Symlink",
"general.quantization_version": 2,
"general.file_type": 30
},
"tensor_count": 866,
"tensor_types": {
"Q5_K": 1,
"F32": 360,
"IQ2_S": 1,
"IQ3_S": 96,
"IQ4_XS": 340,
"Q4_K": 68
},
"descriptor_end": 10995127
}
]
@@ -0,0 +1,71 @@
[
{
"source": "benchmarks/qwen38-final-pre-move-20260822/qwen38-final-acceptance-20260822/cases/08-latest-nvfp4-72k-embedded-mtp3-p010/result.json",
"case": {
"id": "08-latest-nvfp4-72k-embedded-mtp3-p010",
"model": "nvfp4-latest",
"context": 72000,
"split": [
72,
28
],
"projector": "off",
"mtp": true,
"mtp_max": 3,
"quality": false,
"long_fill": 0.0,
"vision": false
},
"model_path": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf",
"status": "load_failed",
"loaded": false,
"mean_predicted_tps": null,
"warning": "Historical GGUF experiment, different runtime/prompts/settings. Not current native NVIDIA NVFP4 performance."
},
{
"source": "benchmarks/qwen38-final-pre-move-20260822/qwen38-final-acceptance-20260822/cases/09-latest-nvfp4-72k-split-mtp2-p010/result.json",
"case": {
"id": "09-latest-nvfp4-72k-split-mtp2-p010",
"model": "nvfp4-latest",
"context": 72000,
"split": [
85,
15
],
"projector": "off",
"mtp": true,
"mtp_max": 2,
"quality": false,
"long_fill": 0.0,
"vision": false
},
"model_path": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf",
"status": "complete",
"loaded": true,
"mean_predicted_tps": 42.27570687556832,
"warning": "Historical GGUF experiment, different runtime/prompts/settings. Not current native NVIDIA NVFP4 performance."
},
{
"source": "benchmarks/qwen38-final-pre-move-20260822/qwen38-final-acceptance-20260822/cases/10-latest-nvfp4-72k-split-mtp3-p010/result.json",
"case": {
"id": "10-latest-nvfp4-72k-split-mtp3-p010",
"model": "nvfp4-latest",
"context": 72000,
"split": [
85,
15
],
"projector": "off",
"mtp": true,
"mtp_max": 3,
"quality": true,
"long_fill": 0.0,
"vision": false
},
"model_path": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf",
"status": "complete",
"loaded": true,
"mean_predicted_tps": 35.94553417986984,
"warning": "Historical GGUF experiment, different runtime/prompts/settings. Not current native NVIDIA NVFP4 performance."
}
]
@@ -0,0 +1,31 @@
#!/usr/bin/env python3
"""Read GGUF metadata/tensor descriptors only; never loads weights or invokes GPUs."""
import collections,json,pathlib,struct,sys
TYPES={0:'F32',1:'F16',2:'Q4_0',3:'Q4_1',6:'Q5_0',7:'Q5_1',8:'Q8_0',9:'Q8_1',10:'Q2_K',11:'Q3_K',12:'Q4_K',13:'Q5_K',14:'Q6_K',15:'Q8_K',16:'IQ2_XXS',17:'IQ2_XS',18:'IQ3_XXS',19:'IQ1_S',20:'IQ4_NL',21:'IQ3_S',22:'IQ2_S',23:'IQ4_XS',30:'BF16'}
def inspect(path):
with open(path,'rb') as f:
def read(fmt):return struct.unpack('<'+fmt,f.read(struct.calcsize('<'+fmt)))[0]
def string(keep=True):
n=read('Q')
if keep:return f.read(n).decode('utf-8')
f.seek(n,1)
def value(t,keep=False):
if t==8:return string(keep)
if t==9:
elem,n=read('I'),read('Q')
for _ in range(n):value(elem,False)
return None
fmt={0:'B',1:'b',2:'H',3:'h',4:'I',5:'i',6:'f',7:'?',10:'Q',11:'q',12:'d'}[t]
return read(fmt)
assert f.read(4)==b'GGUF'
version=read('I');assert version==3
tensors,metas=read('Q'),read('Q');meta={}
for _ in range(metas):
key=string();t=read('I');keep=key in ['general.name','general.architecture','general.file_type','general.quantization_version']
v=value(t,keep)
if keep:meta[key]=v
counts=collections.Counter()
for _ in range(tensors):
name=string();dims=read('I');shape=[read('Q') for _ in range(dims)];ty=read('I');offset=read('Q');counts[TYPES.get(ty,'type_'+str(ty))]+=1
return {'path':str(path),'bytes':pathlib.Path(path).stat().st_size,'metadata':meta,'tensor_count':tensors,'tensor_types':dict(counts),'descriptor_end':f.tell()}
print(json.dumps([inspect(p) for p in sys.argv[1:]],indent=2))
@@ -0,0 +1,154 @@
{
"checked_at": "2026-09-20T19:15:15.116699+00:00",
"scope": "read-only state/health/capacity checks, existing benchmark reuse; no new throughput benchmark",
"containers": {
"mike-ai-llama-medium": {
"running": true,
"health": "healthy",
"started": "2026-09-20T19:07:06.261497776Z",
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907"
},
"mike-ai-router": {
"running": true,
"health": "healthy",
"started": "2026-09-20T19:07:16.676662725Z",
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb"
},
"mike-ai-profile-controller": {
"running": true,
"health": "healthy",
"started": "2026-09-20T19:07:16.548271903Z",
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f"
},
"mike-ai-wireguard-gateway": {
"running": true,
"health": "healthy",
"started": "2026-09-17T09:05:07.164533904Z",
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0"
},
"mike-ai-qwen3-tts": {
"running": true,
"health": "healthy",
"started": "2026-09-19T13:47:00.227813668Z",
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98"
}
},
"medium_args": [
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"512",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"0.0.0.0",
"--port",
"8080",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05"
],
"gpu": "name, driver_version, memory.used [MiB], memory.total [MiB], utilization.gpu [%], temperature.gpu, power.draw [W], power.limit [W], pcie.link.gen.current, pcie.link.gen.max, pcie.link.width.current, pcie.link.width.max\nNVIDIA GeForce RTX 3060, 615.71.09, 10920 MiB, 12288 MiB, 0 %, 45, 11.33 W, 170.00 W, 1, 3, 4, 16\nNVIDIA GeForce RTX 5080, 615.71.09, 15714 MiB, 16303 MiB, 0 %, 44, 14.16 W, 360.00 W, 1, 4, 16, 16",
"vmstat": "procs -----------memory---------- ---swap-- -----io---- -system-- -------cpu-------\n r b swpd free buff cache si so bi bo in cs us sy id wa st gu\n 1 0 2780072 5840012 40828 41027564 222 585 7322 3314 9022 29 4 1 95 0 0 0\n 0 0 2780072 5835276 40828 41027564 0 0 0 24 1274 1982 1 0 98 0 0 0\n 0 0 2780072 5829488 40828 41027564 0 0 0 0 2129 4313 1 0 99 0 0 0",
"memory": "total used free shared buff/cache available\nMem: 48062 4423 5692 1579 40105 43638\nSwap: 49032 2714 46318",
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data",
"uptime": "21:15:17 up 3 days, 10:10, 1 user, load average: 0.12, 0.36, 0.57",
"router": {
"/health": {
"status": "ok",
"router": "alive"
},
"/ready": {
"status": "ok",
"router": "alive",
"upstream": "ready"
}
},
"model_health": {
"status": "ok"
},
"model_properties": {
"model_alias": "qwen-medium",
"model_ftype": "IQ4_XS - 4.25 bpw",
"model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"total_slots": 2,
"modalities": {
"vision": true,
"video": true,
"audio": false
},
"build_info": "b10964-b29c606e2"
},
"startup_findings": [
"0.02.693.409 E ggml_gallocr_reserve_n_impl: failed to allocate CUDA0 buffer of size 1437729280",
"0.02.693.409 E graph_reserve: failed to allocate compute buffers",
"0.02.693.412 W sched_reserve: compute buffer allocation failed, retrying without pipeline parallelism",
"0.05.408.988 I srv load_model: initializing, n_slots = 2, n_ctx_slot = 160000, kv_unified = 'true'",
"0.06.165.055 I srv llama_server: model loaded"
],
"kernel_errors": [],
"tests": {
"containers_healthy": true,
"router_ready": true,
"model_ready": true,
"no_recorded_kernel_faults": true
}
}
@@ -0,0 +1,25 @@
{
"repository": "nvidia/Qwen3.8-27B-NVFP4",
"revision": "482ca0f3832238542f8f5295dde86b5f22711d80",
"source": "https://huggingface.co/api/models/nvidia/Qwen3.8-27B-NVFP4?blobs=true",
"weight_files": [
{
"name": "model-00001-of-00003.safetensors",
"bytes": 9965652544,
"sha256": "7d0fd155118901373eb0fd13ed3aae68f95be747bc205be72ebc41739c25ee80"
},
{
"name": "model-00002-of-00003.safetensors",
"bytes": 9985757064,
"sha256": "98a7e9486baa860c792c9463a770cb9d017696bdd17606300a5b9334149f4c27"
},
{
"name": "model-00003-of-00003.safetensors",
"bytes": 1970287672,
"sha256": "0506ad35dc21469708e7813bd76c592cef08bd0317b4a1fe899745ebe2435271"
}
],
"total_weight_bytes": 21921697280,
"total_weight_gib": 20.416171550750732,
"status": "metadata/capacity review only; not downloaded or benchmarked"
}
@@ -0,0 +1,7 @@
{
"scope": "2-second sampled peaks across saved Pure/MIX baseline cases; no clock/throttling proof",
"temperature_c": {
"NVIDIA GeForce RTX 3060": 62,
"NVIDIA GeForce RTX 5080": 81
}
}