Adopt tested MTP2 for Large; retain Medium MTP3 after warmup failure

This commit is contained in:
Mikei386
2026-09-20 22:59:50 +02:00
parent 9e836cf5fd
commit b352e29c89
29 changed files with 2340 additions and 3 deletions
@@ -0,0 +1,24 @@
[
{
"label": "large-mtp3",
"profile": "large",
"ubatch": 256,
"split": "86,14",
"mtp": 3,
"prompts": [
4096
],
"minimum_headroom_mib": 448
},
{
"label": "large-mtp2",
"profile": "large",
"ubatch": 256,
"split": "86,14",
"mtp": 2,
"prompts": [
4096
],
"minimum_headroom_mib": 448
}
]
@@ -0,0 +1,11 @@
{
"label": "large-mtp2",
"profile": "large",
"ubatch": 256,
"split": "86,14",
"mtp": 2,
"prompts": [
4096
],
"minimum_headroom_mib": 448
}
@@ -0,0 +1,278 @@
[
{
"time": 1789937874.460181,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "54",
"util": "100",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "53",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937876.4885523,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "8636",
"total": "12288",
"temp": "53",
"util": "24",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15852",
"total": "16303",
"temp": "54",
"util": "17",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937878.5206223,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11008",
"total": "12288",
"temp": "53",
"util": "3",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15852",
"total": "16303",
"temp": "52",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937880.5534434,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11012",
"total": "12288",
"temp": "56",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15868",
"total": "16303",
"temp": "60",
"util": "51",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937882.586797,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11012",
"total": "12288",
"temp": "57",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15868",
"total": "16303",
"temp": "60",
"util": "52",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937884.6185646,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11012",
"total": "12288",
"temp": "57",
"util": "49",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15868",
"total": "16303",
"temp": "60",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937886.6517417,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11012",
"total": "12288",
"temp": "58",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15868",
"total": "16303",
"temp": "61",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937888.6827967,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11014",
"total": "12288",
"temp": "56",
"util": "38",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "67",
"util": "98",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937890.7141511,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11014",
"total": "12288",
"temp": "57",
"util": "44",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "60",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937892.7445765,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11014",
"total": "12288",
"temp": "57",
"util": "46",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "61",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937894.7754166,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11014",
"total": "12288",
"temp": "57",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "61",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937896.8081644,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "11014",
"total": "12288",
"temp": "58",
"util": "46",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15870",
"total": "16303",
"temp": "61",
"util": "50",
"pcie_gen": "4",
"pcie_width": "16"
}
]
}
]
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,73 @@
[
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-large",
"--ctx-size",
"192000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"256",
"--parallel",
"1",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"127.0.0.1",
"--port",
"5005",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"0.2",
"--top-p",
"0.8",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"86,14",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"2",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--verbosity",
"3"
]
@@ -0,0 +1,11 @@
{
"label": "large-mtp3",
"profile": "large",
"ubatch": 256,
"split": "86,14",
"mtp": 3,
"prompts": [
4096
],
"minimum_headroom_mib": 448
}
@@ -0,0 +1,278 @@
[
{
"time": 1789937847.9726973,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "50",
"util": "100",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "48",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937850.0014837,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "8350",
"total": "12288",
"temp": "50",
"util": "4",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15682",
"total": "16303",
"temp": "49",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937852.0301683,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10722",
"total": "12288",
"temp": "50",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15684",
"total": "16303",
"temp": "48",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937854.061566,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10726",
"total": "12288",
"temp": "56",
"util": "59",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15700",
"total": "16303",
"temp": "56",
"util": "47",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937856.0915275,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10726",
"total": "12288",
"temp": "57",
"util": "53",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15700",
"total": "16303",
"temp": "58",
"util": "52",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937858.1227798,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10726",
"total": "12288",
"temp": "56",
"util": "47",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15700",
"total": "16303",
"temp": "57",
"util": "52",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937860.1527982,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10726",
"total": "12288",
"temp": "58",
"util": "56",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15700",
"total": "16303",
"temp": "58",
"util": "44",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937862.1820376,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10728",
"total": "12288",
"temp": "56",
"util": "50",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15702",
"total": "16303",
"temp": "69",
"util": "82",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937864.2163124,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10728",
"total": "12288",
"temp": "58",
"util": "54",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15702",
"total": "16303",
"temp": "60",
"util": "39",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937866.2504282,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10728",
"total": "12288",
"temp": "59",
"util": "42",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15702",
"total": "16303",
"temp": "60",
"util": "46",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937868.2800074,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10728",
"total": "12288",
"temp": "59",
"util": "48",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15702",
"total": "16303",
"temp": "60",
"util": "44",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937870.3127642,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "10728",
"total": "12288",
"temp": "59",
"util": "59",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15702",
"total": "16303",
"temp": "60",
"util": "40",
"pcie_gen": "4",
"pcie_width": "16"
}
]
}
]
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,73 @@
[
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-large",
"--ctx-size",
"192000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"256",
"--parallel",
"1",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"127.0.0.1",
"--port",
"5005",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"0.2",
"--top-p",
"0.8",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"86,14",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--verbosity",
"3"
]
@@ -0,0 +1,74 @@
{
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"args": [
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-large",
"--ctx-size",
"192000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"256",
"--parallel",
"1",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"0.0.0.0",
"--port",
"8080",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"0.2",
"--top-p",
"0.8",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"86,14",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16"
]
}
@@ -0,0 +1,11 @@
{
"label": "medium-mtp2",
"profile": "medium",
"ubatch": 512,
"split": "85,15",
"mtp": 2,
"prompts": [
4096
],
"minimum_headroom_mib": 448
}
@@ -0,0 +1,71 @@
[
{
"time": 1789937696.1053374,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "6964",
"total": "12288",
"temp": "46",
"util": "100",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "11272",
"total": "16303",
"temp": "45",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937698.1347463,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "8782",
"total": "12288",
"temp": "46",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "15918",
"total": "16303",
"temp": "45",
"util": "1",
"pcie_gen": "4",
"pcie_width": "16"
}
]
},
{
"time": 1789937700.1621327,
"gpus": [
{
"name": "NVIDIA GeForce RTX 3060",
"used": "4645",
"total": "12288",
"temp": "46",
"util": "0",
"pcie_gen": "3",
"pcie_width": "4"
},
{
"name": "NVIDIA GeForce RTX 5080",
"used": "1",
"total": "16303",
"temp": "45",
"util": "0",
"pcie_gen": "4",
"pcie_width": "16"
}
]
}
]
@@ -0,0 +1,15 @@
{
"case": {
"label": "medium-mtp2",
"profile": "medium",
"ubatch": 512,
"split": "85,15",
"mtp": 2,
"prompts": [
4096
],
"minimum_headroom_mib": 448
},
"started": 1789937694.0730917,
"error": "Test container exited during load"
}
@@ -0,0 +1,75 @@
[
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"512",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"127.0.0.1",
"--port",
"5005",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"2",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05",
"--verbosity",
"3"
]
@@ -0,0 +1,76 @@
{
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"args": [
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"512",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"0.0.0.0",
"--port",
"8080",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05"
]
}
@@ -0,0 +1,76 @@
{
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
"args": [
"--model",
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
"--mmproj",
"/models/qwen/mmproj-BF16.gguf",
"--mmproj-offload",
"--mmproj-device",
"CUDA1",
"--alias",
"qwen-medium",
"--ctx-size",
"160000",
"--flash-attn",
"on",
"--cache-type-k",
"q4_0",
"--cache-type-v",
"q4_0",
"--cache-prompt",
"--cache-ram",
"32768",
"--threads",
"6",
"--threads-batch",
"6",
"--batch-size",
"2048",
"--ubatch-size",
"512",
"--parallel",
"2",
"--kv-unified",
"--jinja",
"--reasoning",
"auto",
"--reasoning-preserve",
"--host",
"0.0.0.0",
"--port",
"8080",
"--metrics",
"--fit",
"off",
"--n-gpu-layers",
"all",
"--load-mode",
"none",
"--no-ui",
"--temperature",
"1.0",
"--top-p",
"0.95",
"--top-k",
"20",
"--device",
"CUDA0,CUDA1",
"--main-gpu",
"0",
"--split-mode",
"layer",
"--tensor-split",
"85,15",
"--spec-type",
"draft-mtp",
"--spec-draft-n-max",
"3",
"--spec-draft-type-k",
"f16",
"--spec-draft-type-v",
"f16",
"--spec-draft-p-min",
"0.05"
]
}
@@ -0,0 +1,35 @@
[
{
"label": "medium-mtp2",
"profile": "medium",
"ubatch": 512,
"split": "85,15",
"mtp": 2,
"prompts": [
4096
],
"minimum_headroom_mib": 448
},
{
"label": "large-mtp3",
"profile": "large",
"ubatch": 256,
"split": "86,14",
"mtp": 3,
"prompts": [
4096
],
"minimum_headroom_mib": 448
},
{
"label": "large-mtp2",
"profile": "large",
"ubatch": 256,
"split": "86,14",
"mtp": 2,
"prompts": [
4096
],
"minimum_headroom_mib": 448
}
]
+171
View File
@@ -0,0 +1,171 @@
#!/usr/bin/env python3
"""Bounded, isolated Qwen quantization benchmark. Supervisor restores production."""
import json, pathlib, subprocess, sys, time, urllib.request, threading, signal
ROOT = pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
NAME = 'mike-ai-profile-mtp2-test'
BASE = 'http://127.0.0.1:5005'
GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe'
GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b'
IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'}
def cmd(*args, check=True, timeout=90):
r = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:])
return r.stdout
def api(path, data=None, timeout=900):
req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'})
with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r)
def save(path, data):
path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n')
def gpu():
rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu,pcie.link.gen.current,pcie.link.width.current','--format=csv,noheader,nounits',timeout=15)
return [dict(zip(['name','used','total','temp','util','pcie_gen','pcie_width'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()]
def health_check():
rows=gpu()
if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit')
mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines())
if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB')
return rows
def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None):
health_check()
p={'model':'qwen-medium','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False}
if tools: p.update(tools=tools,tool_choice='auto')
start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start
health_check()
return r
def prefill(n, seed):
import gzip
payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)]
r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed'])
content=r['choices'][0]['message'].get('content','')
r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']}
return r
def run_case(case):
label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True)
if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label)
save(out/'config.json',case)
print('START',label,flush=True)
single=case.get('single',False)
production=json.loads((ROOT/(case['profile']+'.json')).read_text())
assert production['image']==IMAGE
args=list(production['args'])
for flag,value in [('--ubatch-size',str(case['ubatch'])),('--tensor-split',case['split']),('--spec-draft-n-max',str(case['mtp'])),('--host','127.0.0.1'),('--port','5005')]:
args[args.index(flag)+1]=value
args += ['--verbosity','3']
save(out/'server-args.json',args)
cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
stop=threading.Event(); samples=[]
def monitor():
while not stop.wait(2):
try:
rows=health_check()
samples.append({'time':time.time(),'gpus':rows})
except RuntimeError as e:
samples.append({'error':str(e),'aborted':True})
cmd('docker','stop','-t','10',NAME,check=False)
return
except Exception as e: samples.append({'error':str(e)})
thread=threading.Thread(target=monitor,daemon=True); thread.start()
result={'case':case,'started':time.time()}
try:
for _ in range(150):
try:
if api('/health',timeout=3).get('status')=='ok': break
except Exception: pass
if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load')
time.sleep(2)
else: raise RuntimeError('Startup exceeded 300s')
result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots')
save(out/'loaded.json',result)
# Added after the 88:12 trial: model loading alone can succeed while
# the first real attention graph still needs more CUDA workspace.
minimum=case.get('minimum_headroom_mib',512)
used_devices=['5080'] if single else ['5080','3060']
for g in result['idle_gpu']:
if any(device in g['name'] for device in used_devices):
free=int(g['total'])-int(g['used'])
if free<minimum:
raise RuntimeError(f"Insufficient loaded VRAM reserve on {g['name']}: {free} < {minimum} MiB; refusing inference")
result['smoke']=chat('Antworte nur mit OK.',8)
if case.get('quality'):
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
result['quality']=[]
for task in tasks:
ans=chat(task['prompt'],task['max_tokens'],'medium')
result['quality'].append({'id':task['id'],'response':ans})
save(out/'partial.json',result); print(label,task['id'],round(ans['wall_seconds'],1),flush=True)
tool={'type':'function','function':{'name':'read_server_status','description':'Read-only server status lookup','parameters':{'type':'object','properties':{'server':{'type':'string'}},'required':['server'],'additionalProperties':False}}}
result['tool']=chat('Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.',512,tools=[tool])
if case.get('quality_followup'):
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
result['quality_followup']=[]
for task in tasks:
if task['id'] not in ['i3_code_debugging','i4_capacity_planning','i6_state_vs_configuration']: continue
ans=chat(task['prompt'],8192,'medium',seed=43)
result['quality_followup'].append({'id':task['id'],'seed':43,'budget':8192,'response':ans})
save(out/'partial.json',result); print(label,'followup',task['id'],round(ans['wall_seconds'],1),flush=True)
if not case.get('load_only'):
result['decode']=[]
prompts=['Erkläre ausführlich auf Deutsch, wie ein Reverse Proxy funktioniert, welche Fehler bei Container-IP-Wechseln auftreten können und wie man sie anhand von Logs eingrenzt. Schreibe mindestens 600 Wörter.', 'Write a Python implementation of an asynchronous first_success function: start all awaitables concurrently, return the first successful result, cancel and await remaining tasks, collect exceptions if all fail. Include an explanation and usage example.']
for i,p in enumerate(prompts):
result['decode'].append(chat(p,256,seed=42+i))
save(out/'partial.json',result)
print(label,'decode',i,result['decode'][-1].get('timings'),flush=True)
result['prefill']=[]
for n in case.get('prompts',[4096,16384]):
r=prefill(n,42); result['prefill'].append({'target':n,'response':r}); save(out/'partial.json',result)
print(label,'prefill',n,r.get('timings'),flush=True)
result['finished']=time.time()
except Exception as exc:
result['error']=str(exc)
raise
finally:
stop.set(); thread.join(5)
r=subprocess.run(['docker','logs',NAME],capture_output=True,text=True,timeout=30)
(out/'server.log').write_text(r.stdout+r.stderr)
save(out/'gpu.json',samples); save(out/'result.json',result)
cmd('docker','rm','-f',NAME,check=False)
print('DONE',label,flush=True)
return result
def capacity_case(case):
"""Bounded growth from a previously working context, with 768 MiB reserve.
0.04 MiB/token exceeds the measured 512-ubatch steady-state slope.
A failed 114688/512 ByteShape startup revealed additional transient MTP
buffers, so reserve is deliberately larger than steady-state extrapolation.
Smaller ubatches start at an already working context, not a guessed OOM edge.
The final context is tested with an actual almost-full prompt.
"""
context=case['ctx']
previous=None
for attempt in range(8):
pilot={**case,'ctx':context,'label':case['label']+'-pilot-'+str(context),'load_only':True,'quality':False,'quality_followup':False}
result=run_case(pilot)
rows=[g for g in result['idle_gpu'] if '5080' in g['name']]
samples=json.loads((ROOT/pilot['label']/'gpu.json').read_text())
peak=max([int(rows[0]['used'])+32]+[int(g['used']) for s in samples for g in s.get('gpus',[]) if '5080' in g['name']])
free=int(rows[0]['total'])-peak
if free<768:
if previous is None: raise RuntimeError('Initial capacity pilot has insufficient reserve')
context=previous
break
growth=min(16384,int((free-768)/0.04)//1024*1024)
if growth<1024 or attempt==7 or context>=262144: break
previous=context
context=min(262144,context+growth)
final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]}
return run_case(final)
if __name__=='__main__':
for case in json.loads(pathlib.Path(sys.argv[1]).read_text()):
if case.get('capacity_search'): capacity_case(case)
else: run_case(case)
@@ -0,0 +1,55 @@
#!/usr/bin/env python3
"""Stop only existing router/controller/model; always restore the same containers."""
import json, pathlib, subprocess, sys, time, signal
ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
NAMES=['mike-ai-router','mike-ai-profile-controller','mike-ai-llama-medium']
def run(*args,check=True,timeout=90):
return subprocess.run(args,capture_output=True,text=True,check=check,timeout=timeout)
def stop_signal(*_): raise RuntimeError('Supervisor interrupted')
signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_signal)
# Refuse if the known production state has changed, or if requests are active.
for name in NAMES:
assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
for attempt in range(60):
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
if not any(s['is_processing'] for s in slots): break
if attempt==0: print('WAIT production request active; no interruption',flush=True)
time.sleep(3)
else: raise RuntimeError('Production remained busy for 180s; no services stopped')
assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').stdout.strip(),'Existing experiment'
production=json.loads(run('docker','inspect',NAMES[-1]).stdout)[0]
(ROOT/'production.json').write_text(json.dumps({'image':production['Image'],'args':production['Args']},indent=2)+'\n')
for profile in ['medium','large']:
d=json.loads(run('docker','inspect','mike-ai-llama-'+profile).stdout)[0]
(ROOT/(profile+'.json')).write_text(json.dumps({'image':d['Image'],'args':d['Args']},indent=2)+'\n')
child=None
try:
run('docker','stop','-t','30',*NAMES[:2])
# Drain requests already handed to the model, before unloading it.
for _ in range(120):
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
if not any(s['is_processing'] for s in slots): break
time.sleep(2)
else: raise RuntimeError('Model did not drain')
run('docker','stop','-t','30',NAMES[-1])
child=subprocess.Popen(['python3',str(ROOT/'run.py'),sys.argv[1]])
code=child.wait(timeout=600)
if code: raise RuntimeError('Benchmark failed: '+str(code))
finally:
if child is not None and child.poll() is None:
child.terminate()
try: child.wait(timeout=20)
except subprocess.TimeoutExpired: child.kill(); child.wait(timeout=10)
run('docker','rm','-f','mike-ai-profile-mtp2-test',check=False)
run('docker','start',NAMES[-1])
for _ in range(150):
if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
time.sleep(2)
else: raise RuntimeError('Restored Medium did not become healthy')
run('docker','start',NAMES[1],NAMES[0])
for _ in range(60):
statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
if all(status=='healthy' for status in statuses): break
time.sleep(2)
else: raise RuntimeError('Restored router/controller did not become healthy')
print('RESTORED existing medium/controller/router; all healthy',flush=True)
@@ -0,0 +1,29 @@
#!/usr/bin/env python3
"""Read-only restoration verification plus a two-token model smoke request."""
import json,pathlib,re,subprocess,time
ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
def run(*args):return subprocess.check_output(args,text=True,timeout=30)
report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}}
for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']:
d=json.loads(run('docker','inspect',name))[0]
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']}
assert d['State']['Running'],name
if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']:
assert d['State'].get('Health',{}).get('Status')=='healthy',name
assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').strip()
probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))'
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0}
r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions'))
report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')}
assert report['smoke']['content'].strip()=='OK',report['smoke']
started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json'))
journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager')
pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I)
report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)]
report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip()
report['disk']=run('df','-h','/','/data').strip()
(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n')
print(json.dumps(report,indent=2))
assert not report['kernel_errors'],'Kernel/GPU errors require review'