Record bounded Medium GPU split and MTP comparison

This commit is contained in:
Mikei386
2026-09-20 22:42:09 +02:00
parent 3cbcf41fab
commit 635b3c4ba9
32 changed files with 5107 additions and 0 deletions
@@ -0,0 +1,10 @@
{
"label": "83-mtp3",
"ubatch": 256,
"split": "83,17",
"mtp": 3,
"prompts": [
24576
],
"minimum_headroom_mib": 448
}
@@ -0,0 +1,623 @@
[
{
"time": 1789936766.905543,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "7346",
"total": "12288",
"temp": "50",
"util": "100",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "10888",
"total": "16303",
"temp": "56",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936768.9340215,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "9024",
"total": "12288",
"temp": "50",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15228",
"total": "16303",
"temp": "55",
"util": "1",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936770.9613502,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11222",
"total": "12288",
"temp": "49",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15256",
"total": "16303",
"temp": "54",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936772.992449,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11222",
"total": "12288",
"temp": "48",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15256",
"total": "16303",
"temp": "53",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936775.0237627,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "54",
"util": "48",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "59",
"util": "44",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936777.056265,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "54",
"util": "60",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "62",
"util": "34",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936779.0883763,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "55",
"util": "59",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "59",
"util": "46",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936781.11879,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "55",
"util": "49",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "59",
"util": "41",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936783.1498864,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "56",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "61",
"util": "48",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936785.180021,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "55",
"util": "49",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "62",
"util": "34",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936787.2098007,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "56",
"util": "54",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "63",
"util": "41",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936789.240345,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "56",
"util": "52",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "63",
"util": "39",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936791.2699392,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "56",
"util": "59",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "64",
"util": "36",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936793.3014712,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "57",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "60",
"util": "46",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936795.3310409,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "57",
"util": "61",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "62",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936797.3619807,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11228",
"total": "12288",
"temp": "56",
"util": "55",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15272",
"total": "16303",
"temp": "60",
"util": "45",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936799.3926547,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "57",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "59",
"util": "15",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936801.422532,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "57",
"util": "32",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "72",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936803.4543598,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "57",
"util": "45",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "73",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936805.4877064,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "55",
"util": "70",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "73",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936807.533202,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "57",
"util": "68",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "72",
"util": "12",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936809.5705192,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "57",
"util": "70",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "75",
"util": "46",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936811.6039462,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "56",
"util": "55",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "64",
"util": "40",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936813.633929,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "57",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "62",
"util": "39",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936815.6666515,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "57",
"util": "54",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "62",
"util": "40",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936817.7020009,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "56",
"util": "46",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "62",
"util": "40",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936819.7318478,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11230",
"total": "12288",
"temp": "58",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15274",
"total": "16303",
"temp": "61",
"util": "40",
"pcie_gen": "4",
"pcie_width": "16"
}
]
}
]
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,75 @@
[
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"256",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"127.0.0.1",
"--port",
"5005",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"83,17",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05",
"--verbosity",
"3"
]
@@ -0,0 +1,10 @@
{
"label": "85-mtp2",
"ubatch": 256,
"split": "85,15",
"mtp": 2,
"prompts": [
24576
],
"minimum_headroom_mib": 448
}
@@ -0,0 +1,577 @@
[
{
"time": 1789936657.4383829,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "48",
"util": "100",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "55",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936659.4671788,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "8424",
"total": "12288",
"temp": "48",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15574",
"total": "16303",
"temp": "54",
"util": "17",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936661.4983554,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10608",
"total": "12288",
"temp": "47",
"util": "2",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15574",
"total": "16303",
"temp": "53",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936663.5305703,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "52",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "60",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936665.5614212,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "53",
"util": "43",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "63",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936667.5952804,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "54",
"util": "50",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "59",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936669.627116,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "54",
"util": "48",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "61",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936671.6623967,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "53",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "60",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936673.6930187,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "54",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "63",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936675.7236755,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "54",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "60",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936677.7546844,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "53",
"util": "45",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "64",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936679.785267,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "54",
"util": "43",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "61",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936681.8165836,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "54",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "62",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936683.8488667,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10612",
"total": "12288",
"temp": "54",
"util": "45",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15590",
"total": "16303",
"temp": "64",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936685.8805652,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "52",
"util": "53",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "71",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936687.911105,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "54",
"util": "52",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "73",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936689.9424007,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "54",
"util": "55",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "73",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936691.973218,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "55",
"util": "48",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "72",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936694.0040557,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "53",
"util": "65",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "74",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936696.0368607,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "54",
"util": "64",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "74",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936698.0835536,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "57",
"util": "88",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "63",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936700.1248791,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "55",
"util": "41",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "65",
"util": "45",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936702.155853,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "54",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "62",
"util": "60",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936704.1904633,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "55",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "62",
"util": "56",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936706.2270036,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10614",
"total": "12288",
"temp": "55",
"util": "42",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15592",
"total": "16303",
"temp": "64",
"util": "44",
"pcie_gen": "4",
"pcie_width": "16"
}
]
}
]
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,75 @@
[
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"256",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"127.0.0.1",
"--port",
"5005",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"2",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05",
"--verbosity",
"3"
]
@@ -0,0 +1,10 @@
{
"label": "85-mtp4",
"ubatch": 256,
"split": "85,15",
"mtp": 4,
"prompts": [
24576
],
"minimum_headroom_mib": 448
}
@@ -0,0 +1,623 @@
[
{
"time": 1789936709.8619335,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "49",
"util": "100",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "56",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936711.8901343,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "8230",
"total": "12288",
"temp": "48",
"util": "7",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15852",
"total": "16303",
"temp": "55",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936713.9222171,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10414",
"total": "12288",
"temp": "48",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15854",
"total": "16303",
"temp": "54",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936715.9548185,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "54",
"util": "50",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "63",
"util": "35",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936717.9897373,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "55",
"util": "63",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "60",
"util": "43",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936720.0238986,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "55",
"util": "49",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "63",
"util": "47",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936722.0577638,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "55",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "65",
"util": "35",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936724.089602,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "55",
"util": "57",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "61",
"util": "36",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936726.1215844,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "55",
"util": "63",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "60",
"util": "35",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936728.155224,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "56",
"util": "61",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "60",
"util": "49",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936730.1939929,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "55",
"util": "9",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "60",
"util": "19",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936732.233054,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "57",
"util": "58",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "60",
"util": "36",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936734.2694385,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "55",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "62",
"util": "35",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936736.3035913,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "56",
"util": "54",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "60",
"util": "36",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936738.3328967,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "56",
"util": "49",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "63",
"util": "41",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936740.3630881,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10418",
"total": "12288",
"temp": "53",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "57",
"util": "33",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936742.3925698,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "57",
"util": "1",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "59",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936744.422326,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "55",
"util": "69",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "70",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936746.4554908,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "56",
"util": "37",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "64",
"util": "12",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936748.4865391,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "56",
"util": "36",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "74",
"util": "90",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936750.516579,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "57",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "75",
"util": "91",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936752.547624,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "53",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "74",
"util": "81",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936754.5767446,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "57",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "62",
"util": "44",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936756.6081572,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "56",
"util": "50",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "64",
"util": "39",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936758.6397696,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "56",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "67",
"util": "43",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936760.67214,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "56",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "62",
"util": "43",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936762.7054853,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10420",
"total": "12288",
"temp": "57",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15872",
"total": "16303",
"temp": "61",
"util": "44",
"pcie_gen": "4",
"pcie_width": "16"
}
]
}
]
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,75 @@
[
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"256",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"127.0.0.1",
"--port",
"5005",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"4",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05",
"--verbosity",
"3"
]
@@ -0,0 +1,10 @@
{
"label": "control-85-mtp3",
"ubatch": 256,
"split": "85,15",
"mtp": 3,
"prompts": [
24576
],
"minimum_headroom_mib": 448
}
@@ -0,0 +1,669 @@
[
{
"time": 1789936597.676435,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "43",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "50",
"util": "14",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936599.706274,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "43",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "49",
"util": "14",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936601.736262,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "43",
"util": "94",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "49",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936603.763663,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "8442",
"total": "12288",
"temp": "43",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15814",
"total": "16303",
"temp": "49",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936605.7941704,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10384",
"total": "12288",
"temp": "43",
"util": "43",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15842",
"total": "16303",
"temp": "49",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936607.8243515,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10640",
"total": "12288",
"temp": "43",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15842",
"total": "16303",
"temp": "49",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936609.854398,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "49",
"util": "52",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "55",
"util": "48",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936611.8838413,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "50",
"util": "57",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "55",
"util": "52",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936613.917109,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "50",
"util": "50",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "60",
"util": "46",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936615.9517796,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "52",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "60",
"util": "40",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936617.9818773,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "51",
"util": "55",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "59",
"util": "47",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936620.0144792,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "51",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "58",
"util": "53",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936622.0455701,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "52",
"util": "49",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "61",
"util": "47",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936624.0782108,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "52",
"util": "49",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "58",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936626.1088305,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "52",
"util": "45",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "60",
"util": "40",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936628.1410763,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "53",
"util": "54",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "58",
"util": "41",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936630.1718125,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "54",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "60",
"util": "44",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936632.203748,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10644",
"total": "12288",
"temp": "53",
"util": "51",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15858",
"total": "16303",
"temp": "58",
"util": "38",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936634.2333362,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "51",
"util": "54",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "70",
"util": "95",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936636.2640924,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "54",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "71",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936638.294401,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "52",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "72",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936640.3246953,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "53",
"util": "48",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "72",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936642.3555238,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "53",
"util": "58",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "66",
"util": "73",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936644.3847382,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "54",
"util": "57",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "74",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936646.419878,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "54",
"util": "57",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "64",
"util": "45",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936648.4510555,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "54",
"util": "54",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "65",
"util": "45",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936650.4859831,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "54",
"util": "46",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "61",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936652.518545,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "53",
"util": "53",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "61",
"util": "41",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789936654.5496242,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10646",
"total": "12288",
"temp": "54",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15860",
"total": "16303",
"temp": "63",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
}
]
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,75 @@
[
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"256",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"127.0.0.1",
"--port",
"5005",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05",
"--verbosity",
"3"
]
@@ -0,0 +1,76 @@
{
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"args": [
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"512",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"0.0.0.0",
"--port",
"8080",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05"
]
}
@@ -0,0 +1,42 @@
[
{
"label": "control-85-mtp3",
"ubatch": 256,
"split": "85,15",
"mtp": 3,
"prompts": [
24576
],
"minimum_headroom_mib": 448
},
{
"label": "85-mtp2",
"ubatch": 256,
"split": "85,15",
"mtp": 2,
"prompts": [
24576
],
"minimum_headroom_mib": 448
},
{
"label": "85-mtp4",
"ubatch": 256,
"split": "85,15",
"mtp": 4,
"prompts": [
24576
],
"minimum_headroom_mib": 448
},
{
"label": "83-mtp3",
"ubatch": 256,
"split": "83,17",
"mtp": 3,
"prompts": [
24576
],
"minimum_headroom_mib": 448
}
]
@@ -0,0 +1,61 @@
{
"checked_at": 1789936855.966608,
"uptime": "22:40:55 up 3 days, 11:36, 1 user, load average: 1.59, 1.19, 1.06",
"containers": {
"mike-ai-llama-medium": {
"running": true,
"health": "healthy",
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"started": "2026-09-20T20:40:20.923606083Z"
},
"mike-ai-router": {
"running": true,
"health": "healthy",
"image": "sha256:0448758bec6968b29263bcac0f8b4682c6d3029d626f7c12334698c11ef096bb",
"started": "2026-09-20T20:40:31.403318094Z"
},
"mike-ai-profile-controller": {
"running": true,
"health": "healthy",
"image": "sha256:a5f156d94c4921e671fafa524cf9c0fe91e1cbec0113c1f19c136204d63f243f",
"started": "2026-09-20T20:40:31.237404223Z"
},
"mike-ai-wireguard-gateway": {
"running": true,
"health": "healthy",
"image": "sha256:0d24e93c85a1c420b52b17666ede5fbd4672d92ab8dc28ea4ac5ba144ebc41d0",
"started": "2026-09-17T09:05:07.164533904Z"
},
"mike-ai-qwen3-tts": {
"running": true,
"health": "healthy",
"image": "sha256:b363a01d08b1bbecbfc3ca6f585368fae2cfdc591f9ecca6643738369f9a9d98",
"started": "2026-09-20T20:16:34.144444642Z"
}
},
"router": {
"/health": {
"status": "ok",
"router": "alive"
},
"/ready": {
"status": "ok",
"router": "alive",
"upstream": "ready"
}
},
"smoke": {
"content": "OK",
"usage": {
"completion_tokens": 2,
"prompt_tokens": 19,
"total_tokens": 21,
"prompt_tokens_details": {
"cached_tokens": 0
}
}
},
"kernel_errors": [],
"gpu": "name, memory.used [MiB], memory.total [MiB], temperature.gpu\nNVIDIA GeForce RTX 3060, 10918 MiB, 12288 MiB, 46\nNVIDIA GeForce RTX 5080, 15714 MiB, 16303 MiB, 49",
"disk": "Filesystem Size Used Avail Use% Mounted on\n/dev/nvme0n1p2 868G 187G 637G 23% /\n/dev/nvme1n1p1 916G 822G 49G 95% /data"
}
@@ -0,0 +1,171 @@
#!/usr/bin/env python3
"""Bounded, isolated Qwen quantization benchmark. Supervisor restores production."""
import json, pathlib, subprocess, sys, time, urllib.request, threading, signal
ROOT = pathlib.Path('/data/benchmarks/medium-split-mtp-20260920')
NAME = 'mike-ai-split-mtp-test'
BASE = 'http://127.0.0.1:5005'
GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe'
GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b'
IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'}
def cmd(*args, check=True, timeout=90):
r = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:])
return r.stdout
def api(path, data=None, timeout=900):
req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'})
with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r)
def save(path, data):
path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n')
def gpu():
rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu,pcie.link.gen.current,pcie.link.width.current','--format=csv,noheader,nounits',timeout=15)
return [dict(zip(['name','used','total','temp','util','pcie_gen','pcie_width'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()]
def health_check():
rows=gpu()
if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit')
mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines())
if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB')
return rows
def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None):
health_check()
p={'model':'qwen-medium','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False}
if tools: p.update(tools=tools,tool_choice='auto')
start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start
health_check()
return r
def prefill(n, seed):
import gzip
payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)]
r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed'])
content=r['choices'][0]['message'].get('content','')
r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']}
return r
def run_case(case):
label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True)
if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label)
save(out/'config.json',case)
print('START',label,flush=True)
single=case.get('single',False)
production=json.loads((ROOT/'production.json').read_text())
assert production['image']==IMAGE
args=list(production['args'])
for flag,value in [('--ubatch-size',str(case['ubatch'])),('--tensor-split',case['split']),('--spec-draft-n-max',str(case['mtp'])),('--host','127.0.0.1'),('--port','5005')]:
args[args.index(flag)+1]=value
args += ['--verbosity','3']
save(out/'server-args.json',args)
cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
stop=threading.Event(); samples=[]
def monitor():
while not stop.wait(2):
try:
rows=health_check()
samples.append({'time':time.time(),'gpus':rows})
except RuntimeError as e:
samples.append({'error':str(e),'aborted':True})
cmd('docker','stop','-t','10',NAME,check=False)
return
except Exception as e: samples.append({'error':str(e)})
thread=threading.Thread(target=monitor,daemon=True); thread.start()
result={'case':case,'started':time.time()}
try:
for _ in range(150):
try:
if api('/health',timeout=3).get('status')=='ok': break
except Exception: pass
if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load')
time.sleep(2)
else: raise RuntimeError('Startup exceeded 300s')
result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots')
save(out/'loaded.json',result)
# Added after the 88:12 trial: model loading alone can succeed while
# the first real attention graph still needs more CUDA workspace.
minimum=case.get('minimum_headroom_mib',512)
used_devices=['5080'] if single else ['5080','3060']
for g in result['idle_gpu']:
if any(device in g['name'] for device in used_devices):
free=int(g['total'])-int(g['used'])
if free<minimum:
raise RuntimeError(f"Insufficient loaded VRAM reserve on {g['name']}: {free} < {minimum} MiB; refusing inference")
result['smoke']=chat('Antworte nur mit OK.',8)
if case.get('quality'):
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
result['quality']=[]
for task in tasks:
ans=chat(task['prompt'],task['max_tokens'],'medium')
result['quality'].append({'id':task['id'],'response':ans})
save(out/'partial.json',result); print(label,task['id'],round(ans['wall_seconds'],1),flush=True)
tool={'type':'function','function':{'name':'read_server_status','description':'Read-only server status lookup','parameters':{'type':'object','properties':{'server':{'type':'string'}},'required':['server'],'additionalProperties':False}}}
result['tool']=chat('Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.',512,tools=[tool])
if case.get('quality_followup'):
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
result['quality_followup']=[]
for task in tasks:
if task['id'] not in ['i3_code_debugging','i4_capacity_planning','i6_state_vs_configuration']: continue
ans=chat(task['prompt'],8192,'medium',seed=43)
result['quality_followup'].append({'id':task['id'],'seed':43,'budget':8192,'response':ans})
save(out/'partial.json',result); print(label,'followup',task['id'],round(ans['wall_seconds'],1),flush=True)
if not case.get('load_only'):
result['decode']=[]
prompts=['Erkläre ausführlich auf Deutsch, wie ein Reverse Proxy funktioniert, welche Fehler bei Container-IP-Wechseln auftreten können und wie man sie anhand von Logs eingrenzt. Schreibe mindestens 600 Wörter.', 'Write a Python implementation of an asynchronous first_success function: start all awaitables concurrently, return the first successful result, cancel and await remaining tasks, collect exceptions if all fail. Include an explanation and usage example.']
for i,p in enumerate(prompts):
result['decode'].append(chat(p,768,seed=42+i))
save(out/'partial.json',result)
print(label,'decode',i,result['decode'][-1].get('timings'),flush=True)
result['prefill']=[]
for n in case.get('prompts',[4096,16384]):
r=prefill(n,42); result['prefill'].append({'target':n,'response':r}); save(out/'partial.json',result)
print(label,'prefill',n,r.get('timings'),flush=True)
result['finished']=time.time()
except Exception as exc:
result['error']=str(exc)
raise
finally:
stop.set(); thread.join(5)
r=subprocess.run(['docker','logs',NAME],capture_output=True,text=True,timeout=30)
(out/'server.log').write_text(r.stdout+r.stderr)
save(out/'gpu.json',samples); save(out/'result.json',result)
cmd('docker','rm','-f',NAME,check=False)
print('DONE',label,flush=True)
return result
def capacity_case(case):
"""Bounded growth from a previously working context, with 768 MiB reserve.
0.04 MiB/token exceeds the measured 512-ubatch steady-state slope.
A failed 114688/512 ByteShape startup revealed additional transient MTP
buffers, so reserve is deliberately larger than steady-state extrapolation.
Smaller ubatches start at an already working context, not a guessed OOM edge.
The final context is tested with an actual almost-full prompt.
"""
context=case['ctx']
previous=None
for attempt in range(8):
pilot={**case,'ctx':context,'label':case['label']+'-pilot-'+str(context),'load_only':True,'quality':False,'quality_followup':False}
result=run_case(pilot)
rows=[g for g in result['idle_gpu'] if '5080' in g['name']]
samples=json.loads((ROOT/pilot['label']/'gpu.json').read_text())
peak=max([int(rows[0]['used'])+32]+[int(g['used']) for s in samples for g in s.get('gpus',[]) if '5080' in g['name']])
free=int(rows[0]['total'])-peak
if free<768:
if previous is None: raise RuntimeError('Initial capacity pilot has insufficient reserve')
context=previous
break
growth=min(16384,int((free-768)/0.04)//1024*1024)
if growth<1024 or attempt==7 or context>=262144: break
previous=context
context=min(262144,context+growth)
final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]}
return run_case(final)
if __name__=='__main__':
for case in json.loads(pathlib.Path(sys.argv[1]).read_text()):
if case.get('capacity_search'): capacity_case(case)
else: run_case(case)
@@ -0,0 +1,52 @@
#!/usr/bin/env python3
"""Stop only existing router/controller/model; always restore the same containers."""
import json, pathlib, subprocess, sys, time, signal
ROOT=pathlib.Path('/data/benchmarks/medium-split-mtp-20260920')
NAMES=['mike-ai-router','mike-ai-profile-controller','mike-ai-llama-medium']
def run(*args,check=True,timeout=90):
return subprocess.run(args,capture_output=True,text=True,check=check,timeout=timeout)
def stop_signal(*_): raise RuntimeError('Supervisor interrupted')
signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_signal)
# Refuse if the known production state has changed, or if requests are active.
for name in NAMES:
assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
for attempt in range(60):
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
if not any(s['is_processing'] for s in slots): break
if attempt==0: print('WAIT production request active; no interruption',flush=True)
time.sleep(3)
else: raise RuntimeError('Production remained busy for 180s; no services stopped')
assert not run('docker','ps','-q','--filter','name=^mike-ai-split-mtp-test$').stdout.strip(),'Existing experiment'
production=json.loads(run('docker','inspect',NAMES[-1]).stdout)[0]
(ROOT/'production.json').write_text(json.dumps({'image':production['Image'],'args':production['Args']},indent=2)+'\n')
child=None
try:
run('docker','stop','-t','30',*NAMES[:2])
# Drain requests already handed to the model, before unloading it.
for _ in range(120):
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
if not any(s['is_processing'] for s in slots): break
time.sleep(2)
else: raise RuntimeError('Model did not drain')
run('docker','stop','-t','30',NAMES[-1])
child=subprocess.Popen(['python3',str(ROOT/'run.py'),sys.argv[1]])
code=child.wait(timeout=600)
if code: raise RuntimeError('Benchmark failed: '+str(code))
finally:
if child is not None and child.poll() is None:
child.terminate()
try: child.wait(timeout=20)
except subprocess.TimeoutExpired: child.kill(); child.wait(timeout=10)
run('docker','rm','-f','mike-ai-split-mtp-test',check=False)
run('docker','start',NAMES[-1])
for _ in range(150):
if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
time.sleep(2)
else: raise RuntimeError('Restored Medium did not become healthy')
run('docker','start',NAMES[1],NAMES[0])
for _ in range(60):
statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
if all(status=='healthy' for status in statuses): break
time.sleep(2)
else: raise RuntimeError('Restored router/controller did not become healthy')
print('RESTORED existing medium/controller/router; all healthy',flush=True)
@@ -0,0 +1,29 @@
#!/usr/bin/env python3
"""Read-only restoration verification plus a two-token model smoke request."""
import json,pathlib,re,subprocess,time
ROOT=pathlib.Path('/data/benchmarks/medium-split-mtp-20260920')
def run(*args):return subprocess.check_output(args,text=True,timeout=30)
report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}}
for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']:
d=json.loads(run('docker','inspect',name))[0]
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']}
assert d['State']['Running'],name
if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']:
assert d['State'].get('Health',{}).get('Status')=='healthy',name
assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
assert not run('docker','ps','-q','--filter','name=^mike-ai-split-mtp-test$').strip()
probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))'
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0}
r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions'))
report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')}
assert report['smoke']['content'].strip()=='OK',report['smoke']
started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json'))
journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager')
pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I)
report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)]
report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip()
report['disk']=run('df','-h','/','/data').strip()
(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n')
print(json.dumps(report,indent=2))
assert not report['kernel_errors'],'Kernel/GPU errors require review'