{ "checked_at": "2026-09-20T19:15:15.116699+00:00", "scope": "read-only state/health/capacity checks, existing benchmark reuse; no new throughput benchmark", "containers": { "mike-ai-llama-medium": { "running": true, "health": "healthy", "started": "2026-09-20T19:07:06.261497776Z", "image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907" }, "mike-ai-router": { "running": true, "health": "healthy", "started": "2026-09-20T19:07:16.676662725Z", "image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb" }, "mike-ai-profile-controller": { "running": true, "health": "healthy", "started": "2026-09-20T19:07:16.548271903Z", "image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f" }, "mike-ai-wireguard-gateway": { "running": true, "health": "healthy", "started": "2026-09-17T09:05:07.164533904Z", "image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0" }, "mike-ai-qwen3-tts": { "running": true, "health": "healthy", "started": "2026-09-19T13:47:00.227813668Z", "image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98" } }, "medium_args": [ "--model", "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf", "--mmproj", "/models/qwen/mmproj-BF16.gguf", "--mmproj-offload", "--mmproj-device", "CUDA1", "--alias", "qwen-medium", "--ctx-size", "160000", "--flash-attn", "on", "--cache-type-k", "q4_0", "--cache-type-v", "q4_0", "--cache-prompt", "--cache-ram", "32768", "--threads", "6", "--threads-batch", "6", "--batch-size", "2048", "--ubatch-size", "512", "--parallel", "2", "--kv-unified", "--jinja", "--reasoning", "auto", "--reasoning-preserve", "--host", "0.0.0.0", "--port", "8080", "--metrics", "--fit", "off", "--n-gpu-layers", "all", "--load-mode", "none", "--no-ui", "--temperature", "1.0", "--top-p", "0.95", "--top-k", "20", "--device", "CUDA0,CUDA1", "--main-gpu", "0", "--split-mode", "layer", "--tensor-split", "85,15", "--spec-type", "draft-mtp", "--spec-draft-n-max", "3", "--spec-draft-type-k", "f16", "--spec-draft-type-v", "f16", "--spec-draft-p-min", "0.05" ], "gpu": "name, driver_version, memory.used [MiB], memory.total [MiB], utilization.gpu [%], temperature.gpu, power.draw [W], power.limit [W], pcie.link.gen.current, pcie.link.gen.max, pcie.link.width.current, pcie.link.width.max\nNVIDIA GeForce RTX 3060, 615.71.09, 10920 MiB, 12288 MiB, 0 %, 45, 11.33 W, 170.00 W, 1, 3, 4, 16\nNVIDIA GeForce RTX 5080, 615.71.09, 15714 MiB, 16303 MiB, 0 %, 44, 14.16 W, 360.00 W, 1, 4, 16, 16", "vmstat": "procs -----------memory---------- ---swap-- -----io---- -system-- -------cpu-------\n r b swpd free buff cache si so bi bo in cs us sy id wa st gu\n 1 0 2780072 5840012 40828 41027564 222 585 7322 3314 9022 29 4 1 95 0 0 0\n 0 0 2780072 5835276 40828 41027564 0 0 0 24 1274 1982 1 0 98 0 0 0\n 0 0 2780072 5829488 40828 41027564 0 0 0 0 2129 4313 1 0 99 0 0 0", "memory": "total used free shared buff/cache available\nMem: 48062 4423 5692 1579 40105 43638\nSwap: 49032 2714 46318", "disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 821G 49G 95% /data", "uptime": "21:15:17 up 3 days, 10:10, 1 user, load average: 0.12, 0.36, 0.57", "router": { "/health": { "status": "ok", "router": "alive" }, "/ready": { "status": "ok", "router": "alive", "upstream": "ready" } }, "model_health": { "status": "ok" }, "model_properties": { "model_alias": "qwen-medium", "model_ftype": "IQ4_XS - 4.25 bpw", "model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf", "total_slots": 2, "modalities": { "vision": true, "video": true, "audio": false }, "build_info": "b10964-b29c606e2" }, "startup_findings": [ "0.02.693.409 E ggml_gallocr_reserve_n_impl: failed to allocate CUDA0 buffer of size 1437729280", "0.02.693.409 E graph_reserve: failed to allocate compute buffers", "0.02.693.412 W sched_reserve: compute buffer allocation failed, retrying without pipeline parallelism", "0.05.408.988 I srv load_model: initializing, n_slots = 2, n_ctx_slot = 160000, kv_unified = 'true'", "0.06.165.055 I srv llama_server: model loaded" ], "kernel_errors": [], "tests": { "containers_healthy": true, "router_ready": true, "model_ready": true, "no_recorded_kernel_faults": true } }