Audit Athena quantization inventory and runtime efficiency limits
This commit is contained in:
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Read-only live checks; no inference request, container restart or system change."""
|
||||
import datetime,json,re,subprocess
|
||||
|
||||
def run(*args):return subprocess.check_output(args,text=True,timeout=30).strip()
|
||||
report={'checked_at':datetime.datetime.now(datetime.timezone.utc).isoformat(),'scope':'read-only state/health/capacity checks, existing benchmark reuse; no new throughput benchmark'}
|
||||
names=['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']
|
||||
report['containers']={}
|
||||
for name in names:
|
||||
d=json.loads(run('docker','inspect',name))[0]
|
||||
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'started':d['State']['StartedAt'],'image':d['Image']}
|
||||
if name==names[0]:report['medium_args']=d['Args'];start=d['State']['StartedAt']
|
||||
report['gpu']=run('nvidia-smi','--query-gpu=name,driver_version,memory.used,memory.total,utilization.gpu,temperature.gpu,power.draw,power.limit,pcie.link.gen.current,pcie.link.gen.max,pcie.link.width.current,pcie.link.width.max','--format=csv')
|
||||
report['vmstat']=run('vmstat','1','3')
|
||||
report['memory']=run('free','-m')
|
||||
report['disk']=run('df','-h','/','/data')
|
||||
report['uptime']=run('uptime')
|
||||
probe='import urllib.request,json;print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=10)) for p in ["/health","/ready"]}))'
|
||||
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
|
||||
report['model_health']=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','http://127.0.0.1:8080/health'))
|
||||
props=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','http://127.0.0.1:8080/props'))
|
||||
report['model_properties']={k:props.get(k) for k in ['model_alias','model_ftype','model_path','total_slots','modalities','build_info']}
|
||||
p=subprocess.run(['docker','logs','--since',start,names[0]],capture_output=True,text=True,timeout=30)
|
||||
report['startup_findings']=[s for s in (p.stdout+p.stderr).splitlines() if re.search('failed to allocate|allocation failed|without pipeline|model loaded|initializing, n_slots',s)]
|
||||
journal=run('journalctl','-k','--since','2026-09-20 00:00:00','--no-pager')
|
||||
report['kernel_errors']=[s for s in journal.splitlines() if re.search('NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',s,re.I)]
|
||||
report['tests']={'containers_healthy':all(x['running'] and x['health']=='healthy' for x in report['containers'].values()),'router_ready':report['router']['/ready'].get('upstream')=='ready','model_ready':report['model_health'].get('status')=='ok','no_recorded_kernel_faults':not report['kernel_errors']}
|
||||
print(json.dumps(report,indent=2,ensure_ascii=False))
|
||||
assert all(report['tests'].values()),report['tests']
|
||||
@@ -0,0 +1,38 @@
|
||||
[
|
||||
{
|
||||
"path": "/data/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"bytes": 14534384640,
|
||||
"metadata": {
|
||||
"general.architecture": "qwen35",
|
||||
"general.name": "Qwen3.8-27B",
|
||||
"general.quantization_version": 2,
|
||||
"general.file_type": 30
|
||||
},
|
||||
"tensor_count": 866,
|
||||
"tensor_types": {
|
||||
"IQ4_XS": 506,
|
||||
"F32": 360
|
||||
},
|
||||
"descriptor_end": 10996711
|
||||
},
|
||||
{
|
||||
"path": "/data/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf",
|
||||
"bytes": 14111614400,
|
||||
"metadata": {
|
||||
"general.architecture": "qwen35",
|
||||
"general.name": "Hf_Symlink",
|
||||
"general.quantization_version": 2,
|
||||
"general.file_type": 30
|
||||
},
|
||||
"tensor_count": 866,
|
||||
"tensor_types": {
|
||||
"Q5_K": 1,
|
||||
"F32": 360,
|
||||
"IQ2_S": 1,
|
||||
"IQ3_S": 96,
|
||||
"IQ4_XS": 340,
|
||||
"Q4_K": 68
|
||||
},
|
||||
"descriptor_end": 10995127
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,71 @@
|
||||
[
|
||||
{
|
||||
"source": "benchmarks/qwen38-final-pre-move-20260822/qwen38-final-acceptance-20260822/cases/08-latest-nvfp4-72k-embedded-mtp3-p010/result.json",
|
||||
"case": {
|
||||
"id": "08-latest-nvfp4-72k-embedded-mtp3-p010",
|
||||
"model": "nvfp4-latest",
|
||||
"context": 72000,
|
||||
"split": [
|
||||
72,
|
||||
28
|
||||
],
|
||||
"projector": "off",
|
||||
"mtp": true,
|
||||
"mtp_max": 3,
|
||||
"quality": false,
|
||||
"long_fill": 0.0,
|
||||
"vision": false
|
||||
},
|
||||
"model_path": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf",
|
||||
"status": "load_failed",
|
||||
"loaded": false,
|
||||
"mean_predicted_tps": null,
|
||||
"warning": "Historical GGUF experiment, different runtime/prompts/settings. Not current native NVIDIA NVFP4 performance."
|
||||
},
|
||||
{
|
||||
"source": "benchmarks/qwen38-final-pre-move-20260822/qwen38-final-acceptance-20260822/cases/09-latest-nvfp4-72k-split-mtp2-p010/result.json",
|
||||
"case": {
|
||||
"id": "09-latest-nvfp4-72k-split-mtp2-p010",
|
||||
"model": "nvfp4-latest",
|
||||
"context": 72000,
|
||||
"split": [
|
||||
85,
|
||||
15
|
||||
],
|
||||
"projector": "off",
|
||||
"mtp": true,
|
||||
"mtp_max": 2,
|
||||
"quality": false,
|
||||
"long_fill": 0.0,
|
||||
"vision": false
|
||||
},
|
||||
"model_path": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf",
|
||||
"status": "complete",
|
||||
"loaded": true,
|
||||
"mean_predicted_tps": 42.27570687556832,
|
||||
"warning": "Historical GGUF experiment, different runtime/prompts/settings. Not current native NVIDIA NVFP4 performance."
|
||||
},
|
||||
{
|
||||
"source": "benchmarks/qwen38-final-pre-move-20260822/qwen38-final-acceptance-20260822/cases/10-latest-nvfp4-72k-split-mtp3-p010/result.json",
|
||||
"case": {
|
||||
"id": "10-latest-nvfp4-72k-split-mtp3-p010",
|
||||
"model": "nvfp4-latest",
|
||||
"context": 72000,
|
||||
"split": [
|
||||
85,
|
||||
15
|
||||
],
|
||||
"projector": "off",
|
||||
"mtp": true,
|
||||
"mtp_max": 3,
|
||||
"quality": true,
|
||||
"long_fill": 0.0,
|
||||
"vision": false
|
||||
},
|
||||
"model_path": "/models/qwen3.8-nvfp4-split/Qwen3.8-27B-NVFP4-Q4_K_M-mtp.gguf",
|
||||
"status": "complete",
|
||||
"loaded": true,
|
||||
"mean_predicted_tps": 35.94553417986984,
|
||||
"warning": "Historical GGUF experiment, different runtime/prompts/settings. Not current native NVIDIA NVFP4 performance."
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,31 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Read GGUF metadata/tensor descriptors only; never loads weights or invokes GPUs."""
|
||||
import collections,json,pathlib,struct,sys
|
||||
TYPES={0:'F32',1:'F16',2:'Q4_0',3:'Q4_1',6:'Q5_0',7:'Q5_1',8:'Q8_0',9:'Q8_1',10:'Q2_K',11:'Q3_K',12:'Q4_K',13:'Q5_K',14:'Q6_K',15:'Q8_K',16:'IQ2_XXS',17:'IQ2_XS',18:'IQ3_XXS',19:'IQ1_S',20:'IQ4_NL',21:'IQ3_S',22:'IQ2_S',23:'IQ4_XS',30:'BF16'}
|
||||
def inspect(path):
|
||||
with open(path,'rb') as f:
|
||||
def read(fmt):return struct.unpack('<'+fmt,f.read(struct.calcsize('<'+fmt)))[0]
|
||||
def string(keep=True):
|
||||
n=read('Q')
|
||||
if keep:return f.read(n).decode('utf-8')
|
||||
f.seek(n,1)
|
||||
def value(t,keep=False):
|
||||
if t==8:return string(keep)
|
||||
if t==9:
|
||||
elem,n=read('I'),read('Q')
|
||||
for _ in range(n):value(elem,False)
|
||||
return None
|
||||
fmt={0:'B',1:'b',2:'H',3:'h',4:'I',5:'i',6:'f',7:'?',10:'Q',11:'q',12:'d'}[t]
|
||||
return read(fmt)
|
||||
assert f.read(4)==b'GGUF'
|
||||
version=read('I');assert version==3
|
||||
tensors,metas=read('Q'),read('Q');meta={}
|
||||
for _ in range(metas):
|
||||
key=string();t=read('I');keep=key in ['general.name','general.architecture','general.file_type','general.quantization_version']
|
||||
v=value(t,keep)
|
||||
if keep:meta[key]=v
|
||||
counts=collections.Counter()
|
||||
for _ in range(tensors):
|
||||
name=string();dims=read('I');shape=[read('Q') for _ in range(dims)];ty=read('I');offset=read('Q');counts[TYPES.get(ty,'type_'+str(ty))]+=1
|
||||
return {'path':str(path),'bytes':pathlib.Path(path).stat().st_size,'metadata':meta,'tensor_count':tensors,'tensor_types':dict(counts),'descriptor_end':f.tell()}
|
||||
print(json.dumps([inspect(p) for p in sys.argv[1:]],indent=2))
|
||||
@@ -0,0 +1,154 @@
|
||||
{
|
||||
"checked_at": "2026-09-20T19:15:15.116699+00:00",
|
||||
"scope": "read-only state/health/capacity checks, existing benchmark reuse; no new throughput benchmark",
|
||||
"containers": {
|
||||
"mike-ai-llama-medium": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"started": "2026-09-20T19:07:06.261497776Z",
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907"
|
||||
},
|
||||
"mike-ai-router": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"started": "2026-09-20T19:07:16.676662725Z",
|
||||
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb"
|
||||
},
|
||||
"mike-ai-profile-controller": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"started": "2026-09-20T19:07:16.548271903Z",
|
||||
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f"
|
||||
},
|
||||
"mike-ai-wireguard-gateway": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"started": "2026-09-17T09:05:07.164533904Z",
|
||||
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0"
|
||||
},
|
||||
"mike-ai-qwen3-tts": {
|
||||
"running": true,
|
||||
"health": "healthy",
|
||||
"started": "2026-09-19T13:47:00.227813668Z",
|
||||
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98"
|
||||
}
|
||||
},
|
||||
"medium_args": [
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-medium",
|
||||
"--ctx-size",
|
||||
"160000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"512",
|
||||
"--parallel",
|
||||
"2",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"8080",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"1.0",
|
||||
"--top-p",
|
||||
"0.95",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"85,15",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"3",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16",
|
||||
"--spec-draft-p-min",
|
||||
"0.05"
|
||||
],
|
||||
"gpu": "name, driver_version, memory.used [MiB], memory.total [MiB], utilization.gpu [%], temperature.gpu, power.draw [W], power.limit [W], pcie.link.gen.current, pcie.link.gen.max, pcie.link.width.current, pcie.link.width.max\nNVIDIA GeForce RTX 3060, 615.71.09, 10920 MiB, 12288 MiB, 0 %, 45, 11.33 W, 170.00 W, 1, 3, 4, 16\nNVIDIA GeForce RTX 5080, 615.71.09, 15714 MiB, 16303 MiB, 0 %, 44, 14.16 W, 360.00 W, 1, 4, 16, 16",
|
||||
"vmstat": "procs -----------memory---------- ---swap-- -----io---- -system-- -------cpu-------\n r b swpd free buff cache si so bi bo in cs us sy id wa st gu\n 1 0 2780072 5840012 40828 41027564 222 585 7322 3314 9022 29 4 1 95 0 0 0\n 0 0 2780072 5835276 40828 41027564 0 0 0 24 1274 1982 1 0 98 0 0 0\n 0 0 2780072 5829488 40828 41027564 0 0 0 0 2129 4313 1 0 99 0 0 0",
|
||||
"memory": "total used free shared buff/cache available\nMem: 48062 4423 5692 1579 40105 43638\nSwap: 49032 2714 46318",
|
||||
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data",
|
||||
"uptime": "21:15:17 up 3 days, 10:10, 1 user, load average: 0.12, 0.36, 0.57",
|
||||
"router": {
|
||||
"/health": {
|
||||
"status": "ok",
|
||||
"router": "alive"
|
||||
},
|
||||
"/ready": {
|
||||
"status": "ok",
|
||||
"router": "alive",
|
||||
"upstream": "ready"
|
||||
}
|
||||
},
|
||||
"model_health": {
|
||||
"status": "ok"
|
||||
},
|
||||
"model_properties": {
|
||||
"model_alias": "qwen-medium",
|
||||
"model_ftype": "IQ4_XS - 4.25 bpw",
|
||||
"model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"total_slots": 2,
|
||||
"modalities": {
|
||||
"vision": true,
|
||||
"video": true,
|
||||
"audio": false
|
||||
},
|
||||
"build_info": "b10964-b29c606e2"
|
||||
},
|
||||
"startup_findings": [
|
||||
"0.02.693.409 E ggml_gallocr_reserve_n_impl: failed to allocate CUDA0 buffer of size 1437729280",
|
||||
"0.02.693.409 E graph_reserve: failed to allocate compute buffers",
|
||||
"0.02.693.412 W sched_reserve: compute buffer allocation failed, retrying without pipeline parallelism",
|
||||
"0.05.408.988 I srv load_model: initializing, n_slots = 2, n_ctx_slot = 160000, kv_unified = 'true'",
|
||||
"0.06.165.055 I srv llama_server: model loaded"
|
||||
],
|
||||
"kernel_errors": [],
|
||||
"tests": {
|
||||
"containers_healthy": true,
|
||||
"router_ready": true,
|
||||
"model_ready": true,
|
||||
"no_recorded_kernel_faults": true
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
{
|
||||
"repository": "nvidia/Qwen3.8-27B-NVFP4",
|
||||
"revision": "482ca0f3832238542f8f5295dde86b5f22711d80",
|
||||
"source": "https://huggingface.co/api/models/nvidia/Qwen3.8-27B-NVFP4?blobs=true",
|
||||
"weight_files": [
|
||||
{
|
||||
"name": "model-00001-of-00003.safetensors",
|
||||
"bytes": 9965652544,
|
||||
"sha256": "7d0fd155118901373eb0fd13ed3aae68f95be747bc205be72ebc41739c25ee80"
|
||||
},
|
||||
{
|
||||
"name": "model-00002-of-00003.safetensors",
|
||||
"bytes": 9985757064,
|
||||
"sha256": "98a7e9486baa860c792c9463a770cb9d017696bdd17606300a5b9334149f4c27"
|
||||
},
|
||||
{
|
||||
"name": "model-00003-of-00003.safetensors",
|
||||
"bytes": 1970287672,
|
||||
"sha256": "0506ad35dc21469708e7813bd76c592cef08bd0317b4a1fe899745ebe2435271"
|
||||
}
|
||||
],
|
||||
"total_weight_bytes": 21921697280,
|
||||
"total_weight_gib": 20.416171550750732,
|
||||
"status": "metadata/capacity review only; not downloaded or benchmarked"
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
{
|
||||
"scope": "2-second sampled peaks across saved Pure/MIX baseline cases; no clock/throttling proof",
|
||||
"temperature_c": {
|
||||
"NVIDIA GeForce RTX 3060": 62,
|
||||
"NVIDIA GeForce RTX 5080": 81
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user