Adopt tested MTP2 for Large; retain Medium MTP3 after warmup failure
This commit is contained in:
@@ -0,0 +1,24 @@
|
||||
[
|
||||
{
|
||||
"label": "large-mtp3",
|
||||
"profile": "large",
|
||||
"ubatch": 256,
|
||||
"split": "86,14",
|
||||
"mtp": 3,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
},
|
||||
{
|
||||
"label": "large-mtp2",
|
||||
"profile": "large",
|
||||
"ubatch": 256,
|
||||
"split": "86,14",
|
||||
"mtp": 2,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"label": "large-mtp2",
|
||||
"profile": "large",
|
||||
"ubatch": 256,
|
||||
"split": "86,14",
|
||||
"mtp": 2,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
[
|
||||
{
|
||||
"time": 1789937874.460181,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "6964",
|
||||
"total": "12288",
|
||||
"temp": "54",
|
||||
"util": "100",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "11272",
|
||||
"total": "16303",
|
||||
"temp": "53",
|
||||
"util": "0",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937876.4885523,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "8636",
|
||||
"total": "12288",
|
||||
"temp": "53",
|
||||
"util": "24",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15852",
|
||||
"total": "16303",
|
||||
"temp": "54",
|
||||
"util": "17",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937878.5206223,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11008",
|
||||
"total": "12288",
|
||||
"temp": "53",
|
||||
"util": "3",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15852",
|
||||
"total": "16303",
|
||||
"temp": "52",
|
||||
"util": "0",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937880.5534434,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11012",
|
||||
"total": "12288",
|
||||
"temp": "56",
|
||||
"util": "44",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15868",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "51",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937882.586797,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11012",
|
||||
"total": "12288",
|
||||
"temp": "57",
|
||||
"util": "44",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15868",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "52",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937884.6185646,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11012",
|
||||
"total": "12288",
|
||||
"temp": "57",
|
||||
"util": "49",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15868",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "50",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937886.6517417,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11012",
|
||||
"total": "12288",
|
||||
"temp": "58",
|
||||
"util": "47",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15868",
|
||||
"total": "16303",
|
||||
"temp": "61",
|
||||
"util": "50",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937888.6827967,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11014",
|
||||
"total": "12288",
|
||||
"temp": "56",
|
||||
"util": "38",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15870",
|
||||
"total": "16303",
|
||||
"temp": "67",
|
||||
"util": "98",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937890.7141511,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11014",
|
||||
"total": "12288",
|
||||
"temp": "57",
|
||||
"util": "44",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15870",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "50",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937892.7445765,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11014",
|
||||
"total": "12288",
|
||||
"temp": "57",
|
||||
"util": "46",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15870",
|
||||
"total": "16303",
|
||||
"temp": "61",
|
||||
"util": "50",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937894.7754166,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11014",
|
||||
"total": "12288",
|
||||
"temp": "57",
|
||||
"util": "47",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15870",
|
||||
"total": "16303",
|
||||
"temp": "61",
|
||||
"util": "50",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937896.8081644,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "11014",
|
||||
"total": "12288",
|
||||
"temp": "58",
|
||||
"util": "46",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15870",
|
||||
"total": "16303",
|
||||
"temp": "61",
|
||||
"util": "50",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,73 @@
|
||||
[
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-large",
|
||||
"--ctx-size",
|
||||
"192000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"256",
|
||||
"--parallel",
|
||||
"1",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"127.0.0.1",
|
||||
"--port",
|
||||
"5005",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"0.2",
|
||||
"--top-p",
|
||||
"0.8",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"86,14",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"2",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16",
|
||||
"--verbosity",
|
||||
"3"
|
||||
]
|
||||
Binary file not shown.
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"label": "large-mtp3",
|
||||
"profile": "large",
|
||||
"ubatch": 256,
|
||||
"split": "86,14",
|
||||
"mtp": 3,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
}
|
||||
@@ -0,0 +1,278 @@
|
||||
[
|
||||
{
|
||||
"time": 1789937847.9726973,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "6964",
|
||||
"total": "12288",
|
||||
"temp": "50",
|
||||
"util": "100",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "11272",
|
||||
"total": "16303",
|
||||
"temp": "48",
|
||||
"util": "0",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937850.0014837,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "8350",
|
||||
"total": "12288",
|
||||
"temp": "50",
|
||||
"util": "4",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15682",
|
||||
"total": "16303",
|
||||
"temp": "49",
|
||||
"util": "0",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937852.0301683,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10722",
|
||||
"total": "12288",
|
||||
"temp": "50",
|
||||
"util": "0",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15684",
|
||||
"total": "16303",
|
||||
"temp": "48",
|
||||
"util": "0",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937854.061566,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10726",
|
||||
"total": "12288",
|
||||
"temp": "56",
|
||||
"util": "59",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15700",
|
||||
"total": "16303",
|
||||
"temp": "56",
|
||||
"util": "47",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937856.0915275,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10726",
|
||||
"total": "12288",
|
||||
"temp": "57",
|
||||
"util": "53",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15700",
|
||||
"total": "16303",
|
||||
"temp": "58",
|
||||
"util": "52",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937858.1227798,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10726",
|
||||
"total": "12288",
|
||||
"temp": "56",
|
||||
"util": "47",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15700",
|
||||
"total": "16303",
|
||||
"temp": "57",
|
||||
"util": "52",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937860.1527982,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10726",
|
||||
"total": "12288",
|
||||
"temp": "58",
|
||||
"util": "56",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15700",
|
||||
"total": "16303",
|
||||
"temp": "58",
|
||||
"util": "44",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937862.1820376,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10728",
|
||||
"total": "12288",
|
||||
"temp": "56",
|
||||
"util": "50",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15702",
|
||||
"total": "16303",
|
||||
"temp": "69",
|
||||
"util": "82",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937864.2163124,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10728",
|
||||
"total": "12288",
|
||||
"temp": "58",
|
||||
"util": "54",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15702",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "39",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937866.2504282,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10728",
|
||||
"total": "12288",
|
||||
"temp": "59",
|
||||
"util": "42",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15702",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "46",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937868.2800074,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10728",
|
||||
"total": "12288",
|
||||
"temp": "59",
|
||||
"util": "48",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15702",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "44",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937870.3127642,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "10728",
|
||||
"total": "12288",
|
||||
"temp": "59",
|
||||
"util": "59",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15702",
|
||||
"total": "16303",
|
||||
"temp": "60",
|
||||
"util": "40",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -0,0 +1,73 @@
|
||||
[
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-large",
|
||||
"--ctx-size",
|
||||
"192000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"256",
|
||||
"--parallel",
|
||||
"1",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"127.0.0.1",
|
||||
"--port",
|
||||
"5005",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"0.2",
|
||||
"--top-p",
|
||||
"0.8",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"86,14",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"3",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16",
|
||||
"--verbosity",
|
||||
"3"
|
||||
]
|
||||
Binary file not shown.
@@ -0,0 +1,74 @@
|
||||
{
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
|
||||
"args": [
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-large",
|
||||
"--ctx-size",
|
||||
"192000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"256",
|
||||
"--parallel",
|
||||
"1",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"8080",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"0.2",
|
||||
"--top-p",
|
||||
"0.8",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"86,14",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"3",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16"
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"label": "medium-mtp2",
|
||||
"profile": "medium",
|
||||
"ubatch": 512,
|
||||
"split": "85,15",
|
||||
"mtp": 2,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
}
|
||||
@@ -0,0 +1,71 @@
|
||||
[
|
||||
{
|
||||
"time": 1789937696.1053374,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "6964",
|
||||
"total": "12288",
|
||||
"temp": "46",
|
||||
"util": "100",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "11272",
|
||||
"total": "16303",
|
||||
"temp": "45",
|
||||
"util": "0",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937698.1347463,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "8782",
|
||||
"total": "12288",
|
||||
"temp": "46",
|
||||
"util": "0",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "15918",
|
||||
"total": "16303",
|
||||
"temp": "45",
|
||||
"util": "1",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"time": 1789937700.1621327,
|
||||
"gpus": [
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 3060",
|
||||
"used": "4645",
|
||||
"total": "12288",
|
||||
"temp": "46",
|
||||
"util": "0",
|
||||
"pcie_gen": "3",
|
||||
"pcie_width": "4"
|
||||
},
|
||||
{
|
||||
"name": "NVIDIA GeForce RTX 5080",
|
||||
"used": "1",
|
||||
"total": "16303",
|
||||
"temp": "45",
|
||||
"util": "0",
|
||||
"pcie_gen": "4",
|
||||
"pcie_width": "16"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,15 @@
|
||||
{
|
||||
"case": {
|
||||
"label": "medium-mtp2",
|
||||
"profile": "medium",
|
||||
"ubatch": 512,
|
||||
"split": "85,15",
|
||||
"mtp": 2,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
},
|
||||
"started": 1789937694.0730917,
|
||||
"error": "Test container exited during load"
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
[
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-medium",
|
||||
"--ctx-size",
|
||||
"160000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"512",
|
||||
"--parallel",
|
||||
"2",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"127.0.0.1",
|
||||
"--port",
|
||||
"5005",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"1.0",
|
||||
"--top-p",
|
||||
"0.95",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"85,15",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"2",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16",
|
||||
"--spec-draft-p-min",
|
||||
"0.05",
|
||||
"--verbosity",
|
||||
"3"
|
||||
]
|
||||
Binary file not shown.
@@ -0,0 +1,76 @@
|
||||
{
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
|
||||
"args": [
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-medium",
|
||||
"--ctx-size",
|
||||
"160000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"512",
|
||||
"--parallel",
|
||||
"2",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"8080",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"1.0",
|
||||
"--top-p",
|
||||
"0.95",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"85,15",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"3",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16",
|
||||
"--spec-draft-p-min",
|
||||
"0.05"
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
{
|
||||
"image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
|
||||
"args": [
|
||||
"--model",
|
||||
"/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
|
||||
"--mmproj",
|
||||
"/models/qwen/mmproj-BF16.gguf",
|
||||
"--mmproj-offload",
|
||||
"--mmproj-device",
|
||||
"CUDA1",
|
||||
"--alias",
|
||||
"qwen-medium",
|
||||
"--ctx-size",
|
||||
"160000",
|
||||
"--flash-attn",
|
||||
"on",
|
||||
"--cache-type-k",
|
||||
"q4_0",
|
||||
"--cache-type-v",
|
||||
"q4_0",
|
||||
"--cache-prompt",
|
||||
"--cache-ram",
|
||||
"32768",
|
||||
"--threads",
|
||||
"6",
|
||||
"--threads-batch",
|
||||
"6",
|
||||
"--batch-size",
|
||||
"2048",
|
||||
"--ubatch-size",
|
||||
"512",
|
||||
"--parallel",
|
||||
"2",
|
||||
"--kv-unified",
|
||||
"--jinja",
|
||||
"--reasoning",
|
||||
"auto",
|
||||
"--reasoning-preserve",
|
||||
"--host",
|
||||
"0.0.0.0",
|
||||
"--port",
|
||||
"8080",
|
||||
"--metrics",
|
||||
"--fit",
|
||||
"off",
|
||||
"--n-gpu-layers",
|
||||
"all",
|
||||
"--load-mode",
|
||||
"none",
|
||||
"--no-ui",
|
||||
"--temperature",
|
||||
"1.0",
|
||||
"--top-p",
|
||||
"0.95",
|
||||
"--top-k",
|
||||
"20",
|
||||
"--device",
|
||||
"CUDA0,CUDA1",
|
||||
"--main-gpu",
|
||||
"0",
|
||||
"--split-mode",
|
||||
"layer",
|
||||
"--tensor-split",
|
||||
"85,15",
|
||||
"--spec-type",
|
||||
"draft-mtp",
|
||||
"--spec-draft-n-max",
|
||||
"3",
|
||||
"--spec-draft-type-k",
|
||||
"f16",
|
||||
"--spec-draft-type-v",
|
||||
"f16",
|
||||
"--spec-draft-p-min",
|
||||
"0.05"
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
[
|
||||
{
|
||||
"label": "medium-mtp2",
|
||||
"profile": "medium",
|
||||
"ubatch": 512,
|
||||
"split": "85,15",
|
||||
"mtp": 2,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
},
|
||||
{
|
||||
"label": "large-mtp3",
|
||||
"profile": "large",
|
||||
"ubatch": 256,
|
||||
"split": "86,14",
|
||||
"mtp": 3,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
},
|
||||
{
|
||||
"label": "large-mtp2",
|
||||
"profile": "large",
|
||||
"ubatch": 256,
|
||||
"split": "86,14",
|
||||
"mtp": 2,
|
||||
"prompts": [
|
||||
4096
|
||||
],
|
||||
"minimum_headroom_mib": 448
|
||||
}
|
||||
]
|
||||
@@ -0,0 +1,171 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Bounded, isolated Qwen quantization benchmark. Supervisor restores production."""
|
||||
import json, pathlib, subprocess, sys, time, urllib.request, threading, signal
|
||||
ROOT = pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
|
||||
NAME = 'mike-ai-profile-mtp2-test'
|
||||
BASE = 'http://127.0.0.1:5005'
|
||||
GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe'
|
||||
GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b'
|
||||
IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
|
||||
MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'}
|
||||
|
||||
def cmd(*args, check=True, timeout=90):
|
||||
r = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
|
||||
if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:])
|
||||
return r.stdout
|
||||
|
||||
def api(path, data=None, timeout=900):
|
||||
req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'})
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r)
|
||||
|
||||
def save(path, data):
|
||||
path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n')
|
||||
|
||||
def gpu():
|
||||
rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu,pcie.link.gen.current,pcie.link.width.current','--format=csv,noheader,nounits',timeout=15)
|
||||
return [dict(zip(['name','used','total','temp','util','pcie_gen','pcie_width'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()]
|
||||
|
||||
def health_check():
|
||||
rows=gpu()
|
||||
if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit')
|
||||
mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines())
|
||||
if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB')
|
||||
return rows
|
||||
|
||||
def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None):
|
||||
health_check()
|
||||
p={'model':'qwen-medium','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False}
|
||||
if tools: p.update(tools=tools,tool_choice='auto')
|
||||
start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start
|
||||
health_check()
|
||||
return r
|
||||
|
||||
def prefill(n, seed):
|
||||
import gzip
|
||||
payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)]
|
||||
r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed'])
|
||||
content=r['choices'][0]['message'].get('content','')
|
||||
r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']}
|
||||
return r
|
||||
|
||||
def run_case(case):
|
||||
label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True)
|
||||
if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label)
|
||||
save(out/'config.json',case)
|
||||
print('START',label,flush=True)
|
||||
single=case.get('single',False)
|
||||
production=json.loads((ROOT/(case['profile']+'.json')).read_text())
|
||||
assert production['image']==IMAGE
|
||||
args=list(production['args'])
|
||||
for flag,value in [('--ubatch-size',str(case['ubatch'])),('--tensor-split',case['split']),('--spec-draft-n-max',str(case['mtp'])),('--host','127.0.0.1'),('--port','5005')]:
|
||||
args[args.index(flag)+1]=value
|
||||
args += ['--verbosity','3']
|
||||
save(out/'server-args.json',args)
|
||||
cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
|
||||
stop=threading.Event(); samples=[]
|
||||
def monitor():
|
||||
while not stop.wait(2):
|
||||
try:
|
||||
rows=health_check()
|
||||
samples.append({'time':time.time(),'gpus':rows})
|
||||
except RuntimeError as e:
|
||||
samples.append({'error':str(e),'aborted':True})
|
||||
cmd('docker','stop','-t','10',NAME,check=False)
|
||||
return
|
||||
except Exception as e: samples.append({'error':str(e)})
|
||||
thread=threading.Thread(target=monitor,daemon=True); thread.start()
|
||||
result={'case':case,'started':time.time()}
|
||||
try:
|
||||
for _ in range(150):
|
||||
try:
|
||||
if api('/health',timeout=3).get('status')=='ok': break
|
||||
except Exception: pass
|
||||
if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load')
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Startup exceeded 300s')
|
||||
result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots')
|
||||
save(out/'loaded.json',result)
|
||||
# Added after the 88:12 trial: model loading alone can succeed while
|
||||
# the first real attention graph still needs more CUDA workspace.
|
||||
minimum=case.get('minimum_headroom_mib',512)
|
||||
used_devices=['5080'] if single else ['5080','3060']
|
||||
for g in result['idle_gpu']:
|
||||
if any(device in g['name'] for device in used_devices):
|
||||
free=int(g['total'])-int(g['used'])
|
||||
if free<minimum:
|
||||
raise RuntimeError(f"Insufficient loaded VRAM reserve on {g['name']}: {free} < {minimum} MiB; refusing inference")
|
||||
result['smoke']=chat('Antworte nur mit OK.',8)
|
||||
if case.get('quality'):
|
||||
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
|
||||
result['quality']=[]
|
||||
for task in tasks:
|
||||
ans=chat(task['prompt'],task['max_tokens'],'medium')
|
||||
result['quality'].append({'id':task['id'],'response':ans})
|
||||
save(out/'partial.json',result); print(label,task['id'],round(ans['wall_seconds'],1),flush=True)
|
||||
tool={'type':'function','function':{'name':'read_server_status','description':'Read-only server status lookup','parameters':{'type':'object','properties':{'server':{'type':'string'}},'required':['server'],'additionalProperties':False}}}
|
||||
result['tool']=chat('Read the current status of server alpha. Use the provided tool exactly once and do not invent its result.',512,tools=[tool])
|
||||
if case.get('quality_followup'):
|
||||
tasks=json.loads(pathlib.Path('/opt/mike-ai/stack/dev/QWEN38-FINAL-ACCEPTANCE-v1.json').read_text())
|
||||
result['quality_followup']=[]
|
||||
for task in tasks:
|
||||
if task['id'] not in ['i3_code_debugging','i4_capacity_planning','i6_state_vs_configuration']: continue
|
||||
ans=chat(task['prompt'],8192,'medium',seed=43)
|
||||
result['quality_followup'].append({'id':task['id'],'seed':43,'budget':8192,'response':ans})
|
||||
save(out/'partial.json',result); print(label,'followup',task['id'],round(ans['wall_seconds'],1),flush=True)
|
||||
if not case.get('load_only'):
|
||||
result['decode']=[]
|
||||
prompts=['Erkläre ausführlich auf Deutsch, wie ein Reverse Proxy funktioniert, welche Fehler bei Container-IP-Wechseln auftreten können und wie man sie anhand von Logs eingrenzt. Schreibe mindestens 600 Wörter.', 'Write a Python implementation of an asynchronous first_success function: start all awaitables concurrently, return the first successful result, cancel and await remaining tasks, collect exceptions if all fail. Include an explanation and usage example.']
|
||||
for i,p in enumerate(prompts):
|
||||
result['decode'].append(chat(p,256,seed=42+i))
|
||||
save(out/'partial.json',result)
|
||||
print(label,'decode',i,result['decode'][-1].get('timings'),flush=True)
|
||||
result['prefill']=[]
|
||||
for n in case.get('prompts',[4096,16384]):
|
||||
r=prefill(n,42); result['prefill'].append({'target':n,'response':r}); save(out/'partial.json',result)
|
||||
print(label,'prefill',n,r.get('timings'),flush=True)
|
||||
result['finished']=time.time()
|
||||
except Exception as exc:
|
||||
result['error']=str(exc)
|
||||
raise
|
||||
finally:
|
||||
stop.set(); thread.join(5)
|
||||
r=subprocess.run(['docker','logs',NAME],capture_output=True,text=True,timeout=30)
|
||||
(out/'server.log').write_text(r.stdout+r.stderr)
|
||||
save(out/'gpu.json',samples); save(out/'result.json',result)
|
||||
cmd('docker','rm','-f',NAME,check=False)
|
||||
print('DONE',label,flush=True)
|
||||
return result
|
||||
|
||||
def capacity_case(case):
|
||||
"""Bounded growth from a previously working context, with 768 MiB reserve.
|
||||
|
||||
0.04 MiB/token exceeds the measured 512-ubatch steady-state slope.
|
||||
A failed 114688/512 ByteShape startup revealed additional transient MTP
|
||||
buffers, so reserve is deliberately larger than steady-state extrapolation.
|
||||
Smaller ubatches start at an already working context, not a guessed OOM edge.
|
||||
The final context is tested with an actual almost-full prompt.
|
||||
"""
|
||||
context=case['ctx']
|
||||
previous=None
|
||||
for attempt in range(8):
|
||||
pilot={**case,'ctx':context,'label':case['label']+'-pilot-'+str(context),'load_only':True,'quality':False,'quality_followup':False}
|
||||
result=run_case(pilot)
|
||||
rows=[g for g in result['idle_gpu'] if '5080' in g['name']]
|
||||
samples=json.loads((ROOT/pilot['label']/'gpu.json').read_text())
|
||||
peak=max([int(rows[0]['used'])+32]+[int(g['used']) for s in samples for g in s.get('gpus',[]) if '5080' in g['name']])
|
||||
free=int(rows[0]['total'])-peak
|
||||
if free<768:
|
||||
if previous is None: raise RuntimeError('Initial capacity pilot has insufficient reserve')
|
||||
context=previous
|
||||
break
|
||||
growth=min(16384,int((free-768)/0.04)//1024*1024)
|
||||
if growth<1024 or attempt==7 or context>=262144: break
|
||||
previous=context
|
||||
context=min(262144,context+growth)
|
||||
final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]}
|
||||
return run_case(final)
|
||||
|
||||
if __name__=='__main__':
|
||||
for case in json.loads(pathlib.Path(sys.argv[1]).read_text()):
|
||||
if case.get('capacity_search'): capacity_case(case)
|
||||
else: run_case(case)
|
||||
@@ -0,0 +1,55 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Stop only existing router/controller/model; always restore the same containers."""
|
||||
import json, pathlib, subprocess, sys, time, signal
|
||||
ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
|
||||
NAMES=['mike-ai-router','mike-ai-profile-controller','mike-ai-llama-medium']
|
||||
def run(*args,check=True,timeout=90):
|
||||
return subprocess.run(args,capture_output=True,text=True,check=check,timeout=timeout)
|
||||
def stop_signal(*_): raise RuntimeError('Supervisor interrupted')
|
||||
signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_signal)
|
||||
# Refuse if the known production state has changed, or if requests are active.
|
||||
for name in NAMES:
|
||||
assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
|
||||
for attempt in range(60):
|
||||
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
|
||||
if not any(s['is_processing'] for s in slots): break
|
||||
if attempt==0: print('WAIT production request active; no interruption',flush=True)
|
||||
time.sleep(3)
|
||||
else: raise RuntimeError('Production remained busy for 180s; no services stopped')
|
||||
assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').stdout.strip(),'Existing experiment'
|
||||
production=json.loads(run('docker','inspect',NAMES[-1]).stdout)[0]
|
||||
(ROOT/'production.json').write_text(json.dumps({'image':production['Image'],'args':production['Args']},indent=2)+'\n')
|
||||
for profile in ['medium','large']:
|
||||
d=json.loads(run('docker','inspect','mike-ai-llama-'+profile).stdout)[0]
|
||||
(ROOT/(profile+'.json')).write_text(json.dumps({'image':d['Image'],'args':d['Args']},indent=2)+'\n')
|
||||
child=None
|
||||
try:
|
||||
run('docker','stop','-t','30',*NAMES[:2])
|
||||
# Drain requests already handed to the model, before unloading it.
|
||||
for _ in range(120):
|
||||
slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
|
||||
if not any(s['is_processing'] for s in slots): break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Model did not drain')
|
||||
run('docker','stop','-t','30',NAMES[-1])
|
||||
child=subprocess.Popen(['python3',str(ROOT/'run.py'),sys.argv[1]])
|
||||
code=child.wait(timeout=600)
|
||||
if code: raise RuntimeError('Benchmark failed: '+str(code))
|
||||
finally:
|
||||
if child is not None and child.poll() is None:
|
||||
child.terminate()
|
||||
try: child.wait(timeout=20)
|
||||
except subprocess.TimeoutExpired: child.kill(); child.wait(timeout=10)
|
||||
run('docker','rm','-f','mike-ai-profile-mtp2-test',check=False)
|
||||
run('docker','start',NAMES[-1])
|
||||
for _ in range(150):
|
||||
if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Restored Medium did not become healthy')
|
||||
run('docker','start',NAMES[1],NAMES[0])
|
||||
for _ in range(60):
|
||||
statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
|
||||
if all(status=='healthy' for status in statuses): break
|
||||
time.sleep(2)
|
||||
else: raise RuntimeError('Restored router/controller did not become healthy')
|
||||
print('RESTORED existing medium/controller/router; all healthy',flush=True)
|
||||
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Read-only restoration verification plus a two-token model smoke request."""
|
||||
import json,pathlib,re,subprocess,time
|
||||
ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
|
||||
def run(*args):return subprocess.check_output(args,text=True,timeout=30)
|
||||
report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}}
|
||||
for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']:
|
||||
d=json.loads(run('docker','inspect',name))[0]
|
||||
report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']}
|
||||
assert d['State']['Running'],name
|
||||
if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']:
|
||||
assert d['State'].get('Health',{}).get('Status')=='healthy',name
|
||||
assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
|
||||
assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').strip()
|
||||
probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))'
|
||||
report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
|
||||
payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0}
|
||||
r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions'))
|
||||
report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')}
|
||||
assert report['smoke']['content'].strip()=='OK',report['smoke']
|
||||
started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json'))
|
||||
journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager')
|
||||
pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I)
|
||||
report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)]
|
||||
report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip()
|
||||
report['disk']=run('df','-h','/','/data').strip()
|
||||
(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n')
|
||||
print(json.dumps(report,indent=2))
|
||||
assert not report['kernel_errors'],'Kernel/GPU errors require review'
|
||||
Reference in New Issue
Block a user