Files
AI-Profile-Router/experiments/efficiency-review-20260920/live-audit.json
T

155 lines
5.0 KiB
JSON

{
"checked_at": "2026-09-20T19:15:15.116699+00:00",
"scope": "read-only state/health/capacity checks, existing benchmark reuse; no new throughput benchmark",
"containers": {
"mike-ai-llama-medium": {
"running": true,
"health": "healthy",
"started": "2026-09-20T19:07:06.261497776Z",
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907"
},
"mike-ai-router": {
"running": true,
"health": "healthy",
"started": "2026-09-20T19:07:16.676662725Z",
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb"
},
"mike-ai-profile-controller": {
"running": true,
"health": "healthy",
"started": "2026-09-20T19:07:16.548271903Z",
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f"
},
"mike-ai-wireguard-gateway": {
"running": true,
"health": "healthy",
"started": "2026-09-17T09:05:07.164533904Z",
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0"
},
"mike-ai-qwen3-tts": {
"running": true,
"health": "healthy",
"started": "2026-09-19T13:47:00.227813668Z",
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98"
}
},
"medium_args": [
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"512",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"0.0.0.0",
"--port",
"8080",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05"
],
"gpu": "name, driver_version, memory.used [MiB], memory.total [MiB], utilization.gpu [%], temperature.gpu, power.draw [W], power.limit [W], pcie.link.gen.current, pcie.link.gen.max, pcie.link.width.current, pcie.link.width.max\nNVIDIA GeForce RTX 3060, 615.71.09, 10920 MiB, 12288 MiB, 0 %, 45, 11.33 W, 170.00 W, 1, 3, 4, 16\nNVIDIA GeForce RTX 5080, 615.71.09, 15714 MiB, 16303 MiB, 0 %, 44, 14.16 W, 360.00 W, 1, 4, 16, 16",
"vmstat": "procs -----------memory---------- ---swap-- -----io---- -system-- -------cpu-------\n r b swpd free buff cache si so bi bo in cs us sy id wa st gu\n 1 0 2780072 5840012 40828 41027564 222 585 7322 3314 9022 29 4 1 95 0 0 0\n 0 0 2780072 5835276 40828 41027564 0 0 0 24 1274 1982 1 0 98 0 0 0\n 0 0 2780072 5829488 40828 41027564 0 0 0 0 2129 4313 1 0 99 0 0 0",
"memory": "total used free shared buff/cache available\nMem: 48062 4423 5692 1579 40105 43638\nSwap: 49032 2714 46318",
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data",
"uptime": "21:15:17 up 3 days, 10:10, 1 user, load average: 0.12, 0.36, 0.57",
"router": {
"/health": {
"status": "ok",
"router": "alive"
},
"/ready": {
"status": "ok",
"router": "alive",
"upstream": "ready"
}
},
"model_health": {
"status": "ok"
},
"model_properties": {
"model_alias": "qwen-medium",
"model_ftype": "IQ4_XS - 4.25 bpw",
"model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"total_slots": 2,
"modalities": {
"vision": true,
"video": true,
"audio": false
},
"build_info": "b10964-b29c606e2"
},
"startup_findings": [
"0.02.693.409 E ggml_gallocr_reserve_n_impl: failed to allocate CUDA0 buffer of size 1437729280",
"0.02.693.409 E graph_reserve: failed to allocate compute buffers",
"0.02.693.412 W sched_reserve: compute buffer allocation failed, retrying without pipeline parallelism",
"0.05.408.988 I srv load_model: initializing, n_slots = 2, n_ctx_slot = 160000, kv_unified = 'true'",
"0.06.165.055 I srv llama_server: model loaded"
],
"kernel_errors": [],
"tests": {
"containers_healthy": true,
"router_ready": true,
"model_ready": true,
"no_recorded_kernel_faults": true
}
}