diff --git a/compose.yaml b/compose.yaml
index 1832427..7019b60 100644
--- a/compose.yaml
+++ b/compose.yaml
@@ -329,7 +329,7 @@ services:
- --spec-type
- draft-mtp
- --spec-draft-n-max
- - "3"
+ - "2"
- --spec-draft-type-k
- f16
- --spec-draft-type-v
diff --git a/config/profile-matrix.json b/config/profile-matrix.json
index 27f5572..a39e2c3 100644
--- a/config/profile-matrix.json
+++ b/config/profile-matrix.json
@@ -36,7 +36,7 @@
"model_family": "Qwen3.8-27B IQ4 XS Pure",
"gpu_split": "86:14",
"vision": true,
- "mtp": 3,
+ "mtp": 2,
"description": "Großes Profil für umfangreiche Dokumente und lange technische Arbeiten."
},
{
diff --git a/docs/PROFILE_MTP2_ROLLOUT_20260920.md b/docs/PROFILE_MTP2_ROLLOUT_20260920.md
new file mode 100644
index 0000000..ff780b8
--- /dev/null
+++ b/docs/PROFILE_MTP2_ROLLOUT_20260920.md
@@ -0,0 +1,31 @@
+# Selektive MTP2-Übernahme
+
+20.09.2026. Fast, Ultra und Uncensored verwenden bereits MTP2 (Containerargumente
+geprüft, nicht neu gebenchmarkt). Medium und Large verwenden vorher MTP3.
+
+Medium mit unveränderter Microbatch512 und MTP2 scheitert beim CUDA-Warmup im
+isolierten Container. Frühere erfolgreiche MTP2-Versuche waren Microbatch256.
+Medium bleibt daher MTP3/512. Produktion automatisch wiederhergestellt,
+keine erfassten Kernel-/Xid-/OOM-Kill-Fehler.
+
+Large mit unverändert192000Kontext, Microbatch256, Split86:14:
+
+| Messung | MTP3 | MTP2 |
+|---|---:|---:|
+| Deutsch tok/s |59,67|61,36|
+| Code tok/s |77,18|77,84|
+| Prefill4196Tokens tok/s |1964,41|2016,76|
+| Decode nach4196Tokens tok/s |62,46|66,42|
+| Recall |3/3|3/3|
+
+Large wird entsprechend Nutzerauftrag auf MTP2 übernommen. Alle übrigen
+Modellargumente bleiben gleich. Je ein kurzer Lauf, Decode256Tokens,
+Recall512Tokens; keine volle192K-/Vision-/Parallelitätsfreigabe. Code-Text zwischen
+beiden Armen identisch, Deutsch unterschiedlich. Kleine Geschwindigkeitsunterschiede
+können Messrauschen/Tokenfolgen widerspiegeln. Kein breiter Qualitätsvergleich.
+
+Die erste Large-Testausführung hatte eine Dateinamenskollision zwischen Testplan
+und Profilsnapshot und führte keine Inferenz aus. Korrigierter Plan: large-cases.json.
+Rohdaten: experiments/profile-mtp2-20260920 und gleichnamiger Ordner unter
+/data/benchmarks auf Athena. Aktives Profil bleibt Medium; Large mit MTP2 wird
+beim nächsten Profilwechsel verwendet.
diff --git a/docs/STANDARD_PROFILE_MATRIX.md b/docs/STANDARD_PROFILE_MATRIX.md
index 5a9537d..0ed6dda 100644
--- a/docs/STANDARD_PROFILE_MATRIX.md
+++ b/docs/STANDARD_PROFILE_MATRIX.md
@@ -8,7 +8,7 @@ Standardprofil: **medium** · globales Ausgabelimit: **8192 Token**
|---|---|---:|---:|---|---|---|---:|
| fast | `qwen-fast` | 76,800 | 1 | Qwen3.8-27B IQ4 Mix | 5080 only | ja | 2 |
| medium | `qwen-medium` | 160,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 85:15 | ja | 3 |
-| large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 3 |
+| large | `qwen-large` | 192,000 | 1 | Qwen3.8-27B IQ4 XS Pure | 86:14 | ja | 2 |
| ultra | `qwen-ultra` | 262,144 | 1 | Qwen3.8-27B IQ4 XS Pure | 80:20 | nein | 2 |
| uncensored | `qwen-uncensored` | 80,000 | 1 | Qwen3.8-27B Abliterated Q4_K_M | 90:10 | ja | 2 |
diff --git a/experiments/profile-mtp2-20260920/large-cases.json b/experiments/profile-mtp2-20260920/large-cases.json
new file mode 100644
index 0000000..ef7be75
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-cases.json
@@ -0,0 +1,24 @@
+[
+ {
+ "label": "large-mtp3",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 3,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ {
+ "label": "large-mtp2",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ }
+]
\ No newline at end of file
diff --git a/experiments/profile-mtp2-20260920/large-mtp2/config.json b/experiments/profile-mtp2-20260920/large-mtp2/config.json
new file mode 100644
index 0000000..b6b6d97
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp2/config.json
@@ -0,0 +1,11 @@
+{
+ "label": "large-mtp2",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+}
diff --git a/experiments/profile-mtp2-20260920/large-mtp2/gpu.json b/experiments/profile-mtp2-20260920/large-mtp2/gpu.json
new file mode 100644
index 0000000..bbdb21f
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp2/gpu.json
@@ -0,0 +1,278 @@
+[
+ {
+ "time": 1789937874.460181,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "6964",
+ "total": "12288",
+ "temp": "54",
+ "util": "100",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "11272",
+ "total": "16303",
+ "temp": "53",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937876.4885523,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "8636",
+ "total": "12288",
+ "temp": "53",
+ "util": "24",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15852",
+ "total": "16303",
+ "temp": "54",
+ "util": "17",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937878.5206223,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11008",
+ "total": "12288",
+ "temp": "53",
+ "util": "3",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15852",
+ "total": "16303",
+ "temp": "52",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937880.5534434,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11012",
+ "total": "12288",
+ "temp": "56",
+ "util": "44",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15868",
+ "total": "16303",
+ "temp": "60",
+ "util": "51",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937882.586797,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11012",
+ "total": "12288",
+ "temp": "57",
+ "util": "44",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15868",
+ "total": "16303",
+ "temp": "60",
+ "util": "52",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937884.6185646,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11012",
+ "total": "12288",
+ "temp": "57",
+ "util": "49",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15868",
+ "total": "16303",
+ "temp": "60",
+ "util": "50",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937886.6517417,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11012",
+ "total": "12288",
+ "temp": "58",
+ "util": "47",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15868",
+ "total": "16303",
+ "temp": "61",
+ "util": "50",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937888.6827967,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11014",
+ "total": "12288",
+ "temp": "56",
+ "util": "38",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15870",
+ "total": "16303",
+ "temp": "67",
+ "util": "98",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937890.7141511,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11014",
+ "total": "12288",
+ "temp": "57",
+ "util": "44",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15870",
+ "total": "16303",
+ "temp": "60",
+ "util": "50",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937892.7445765,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11014",
+ "total": "12288",
+ "temp": "57",
+ "util": "46",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15870",
+ "total": "16303",
+ "temp": "61",
+ "util": "50",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937894.7754166,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11014",
+ "total": "12288",
+ "temp": "57",
+ "util": "47",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15870",
+ "total": "16303",
+ "temp": "61",
+ "util": "50",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937896.8081644,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11014",
+ "total": "12288",
+ "temp": "58",
+ "util": "46",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15870",
+ "total": "16303",
+ "temp": "61",
+ "util": "50",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ }
+]
diff --git a/experiments/profile-mtp2-20260920/large-mtp2/loaded.json b/experiments/profile-mtp2-20260920/large-mtp2/loaded.json
new file mode 100644
index 0000000..7f42189
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp2/loaded.json
@@ -0,0 +1,133 @@
+{
+ "case": {
+ "label": "large-mtp2",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ "started": 1789937872.428395,
+ "idle_gpu": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11008",
+ "total": "12288",
+ "temp": "53",
+ "util": "3",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15852",
+ "total": "16303",
+ "temp": "52",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ],
+ "props": {
+ "default_generation_settings": {
+ "params": {
+ "seed": 4294967295,
+ "temperature": 0.20000000298023224,
+ "dynatemp_range": 0.0,
+ "dynatemp_exponent": 1.0,
+ "top_k": 20,
+ "top_p": 0.800000011920929,
+ "min_p": 0.05000000074505806,
+ "top_n_sigma": -1.0,
+ "xtc_probability": 0.0,
+ "xtc_threshold": 0.10000000149011612,
+ "typical_p": 1.0,
+ "repeat_last_n": 64,
+ "repeat_penalty": 1.0,
+ "presence_penalty": 0.0,
+ "frequency_penalty": 0.0,
+ "dry_multiplier": 0.0,
+ "dry_base": 1.75,
+ "dry_allowed_length": 2,
+ "dry_penalty_last_n": 64,
+ "mirostat": 0,
+ "mirostat_tau": 5.0,
+ "mirostat_eta": 0.10000000149011612,
+ "adaptive_target": -1.0,
+ "adaptive_decay": 0.8999999761581421,
+ "max_tokens": -1,
+ "n_predict": -1,
+ "n_keep": 0,
+ "n_discard": 0,
+ "ignore_eos": false,
+ "stream": false,
+ "n_probs": 0,
+ "min_keep": 0,
+ "chat_format": "Content-only",
+ "reasoning_format": "none",
+ "reasoning_in_content": false,
+ "generation_prompt": "",
+ "samplers": [
+ "penalties",
+ "dry",
+ "top_n_sigma",
+ "top_k",
+ "typ_p",
+ "top_p",
+ "min_p",
+ "xtc",
+ "temperature"
+ ],
+ "speculative.types": "none",
+ "timings_per_token": false,
+ "post_sampling_probs": false,
+ "backend_sampling": false,
+ "lora": []
+ },
+ "n_ctx": 192000
+ },
+ "total_slots": 1,
+ "model_alias": "qwen-large",
+ "model_ftype": "IQ4_XS - 4.25 bpw",
+ "model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "modalities": {
+ "vision": true,
+ "video": true,
+ "audio": false
+ },
+ "media_marker": "<__media_RIkeCjhjPwUwHqRncySt7e9bbJNW8xW1__>",
+ "endpoint_slots": true,
+ "endpoint_props": false,
+ "endpoint_metrics": true,
+ "ui": false,
+ "ui_settings": {},
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set sysns = namespace(count=0, text='') %}\n{%- for message in messages %}\n {%- if sysns.count == loop.index0 and (message.role == 'system' or message.role == 'developer') %}\n {%- set sys_content = render_content(message.content, false, true)|trim %}\n {%- if sys_content %}\n {%- set sysns.text = sysns.text + ('\\n' if sysns.text else '') + sys_content %}\n {%- endif %}\n {%- set sysns.count = sysns.count + 1 %}\n {%- endif %}\n{%- endfor %}\n{%- set num_sys = sysns.count %}\n{%- set merged_system = sysns.text %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort == 'high' %}\n {%- set resolved_reasoning_effort = 'xhigh' %}\n {%- endif %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n\\n\\n\\nvalue_1\\n\\n\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n\\n\\n\\n\\n\\nReminder:\\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n' }}\n {%- if merged_system %}\n {{- '\\n\\n' + merged_system }}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if merged_system %}\n {{- '<|im_start|>system\\n' + (reasoning_instructions + '\\n\\n' if reasoning_instructions else '') + merged_system + '<|im_end|>\\n' }}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('') and content.endswith('')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if loop.index0 >= num_sys %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" or message.role == \"developer\" %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n\\n' + reasoning_content + '\\n\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if tool_call.name is not defined or tool_call.name is none %}\n {{- raise_exception('Tool call is missing a function name.') }}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n\\n\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is mapping %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '\\n' }}\n {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}\n {{- args_value }}\n {{- '\\n\\n' }}\n {%- endfor %}\n {%- elif tool_call.arguments is string %}\n {%- if tool_call.arguments|trim %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" were passed as a JSON string. Parse them into an object before calling apply_chat_template.') }}\n {%- endif %}\n {%- elif tool_call.arguments is defined and tool_call.arguments is not none %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" must be an object/mapping or a JSON string.') }}\n {%- endif %}\n {{- '\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- content }}\n {{- '\\n' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n' }}\n {%- endif %}\n{%- endif %}\n{#- Unsloth fixes - developer role, merged system messages, tool calling #}",
+ "chat_template_caps": {
+ "supports_object_arguments": true,
+ "supports_parallel_tool_calls": true,
+ "supports_preserve_reasoning": true,
+ "supports_reasoning_effort": true,
+ "supports_string_content": true,
+ "supports_system_role": true,
+ "supports_tool_calls": true,
+ "supports_tools": true,
+ "supports_typed_content": true
+ },
+ "bos_token": "<|endoftext|>",
+ "eos_token": "<|im_end|>",
+ "build_info": "b10964-b29c606e2",
+ "is_sleeping": false,
+ "cors_proxy_enabled": false
+ },
+ "slots": [
+ {
+ "id": 0,
+ "n_ctx": 192000,
+ "speculative": true,
+ "is_processing": false
+ }
+ ]
+}
diff --git a/experiments/profile-mtp2-20260920/large-mtp2/result.json b/experiments/profile-mtp2-20260920/large-mtp2/result.json
new file mode 100644
index 0000000..8a7ff97
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp2/result.json
@@ -0,0 +1,302 @@
+{
+ "case": {
+ "label": "large-mtp2",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ "started": 1789937872.428395,
+ "idle_gpu": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "11008",
+ "total": "12288",
+ "temp": "53",
+ "util": "3",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15852",
+ "total": "16303",
+ "temp": "52",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ],
+ "props": {
+ "default_generation_settings": {
+ "params": {
+ "seed": 4294967295,
+ "temperature": 0.20000000298023224,
+ "dynatemp_range": 0.0,
+ "dynatemp_exponent": 1.0,
+ "top_k": 20,
+ "top_p": 0.800000011920929,
+ "min_p": 0.05000000074505806,
+ "top_n_sigma": -1.0,
+ "xtc_probability": 0.0,
+ "xtc_threshold": 0.10000000149011612,
+ "typical_p": 1.0,
+ "repeat_last_n": 64,
+ "repeat_penalty": 1.0,
+ "presence_penalty": 0.0,
+ "frequency_penalty": 0.0,
+ "dry_multiplier": 0.0,
+ "dry_base": 1.75,
+ "dry_allowed_length": 2,
+ "dry_penalty_last_n": 64,
+ "mirostat": 0,
+ "mirostat_tau": 5.0,
+ "mirostat_eta": 0.10000000149011612,
+ "adaptive_target": -1.0,
+ "adaptive_decay": 0.8999999761581421,
+ "max_tokens": -1,
+ "n_predict": -1,
+ "n_keep": 0,
+ "n_discard": 0,
+ "ignore_eos": false,
+ "stream": false,
+ "n_probs": 0,
+ "min_keep": 0,
+ "chat_format": "Content-only",
+ "reasoning_format": "none",
+ "reasoning_in_content": false,
+ "generation_prompt": "",
+ "samplers": [
+ "penalties",
+ "dry",
+ "top_n_sigma",
+ "top_k",
+ "typ_p",
+ "top_p",
+ "min_p",
+ "xtc",
+ "temperature"
+ ],
+ "speculative.types": "none",
+ "timings_per_token": false,
+ "post_sampling_probs": false,
+ "backend_sampling": false,
+ "lora": []
+ },
+ "n_ctx": 192000
+ },
+ "total_slots": 1,
+ "model_alias": "qwen-large",
+ "model_ftype": "IQ4_XS - 4.25 bpw",
+ "model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "modalities": {
+ "vision": true,
+ "video": true,
+ "audio": false
+ },
+ "media_marker": "<__media_RIkeCjhjPwUwHqRncySt7e9bbJNW8xW1__>",
+ "endpoint_slots": true,
+ "endpoint_props": false,
+ "endpoint_metrics": true,
+ "ui": false,
+ "ui_settings": {},
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set sysns = namespace(count=0, text='') %}\n{%- for message in messages %}\n {%- if sysns.count == loop.index0 and (message.role == 'system' or message.role == 'developer') %}\n {%- set sys_content = render_content(message.content, false, true)|trim %}\n {%- if sys_content %}\n {%- set sysns.text = sysns.text + ('\\n' if sysns.text else '') + sys_content %}\n {%- endif %}\n {%- set sysns.count = sysns.count + 1 %}\n {%- endif %}\n{%- endfor %}\n{%- set num_sys = sysns.count %}\n{%- set merged_system = sysns.text %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort == 'high' %}\n {%- set resolved_reasoning_effort = 'xhigh' %}\n {%- endif %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n\\n\\n\\nvalue_1\\n\\n\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n\\n\\n\\n\\n\\nReminder:\\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n' }}\n {%- if merged_system %}\n {{- '\\n\\n' + merged_system }}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if merged_system %}\n {{- '<|im_start|>system\\n' + (reasoning_instructions + '\\n\\n' if reasoning_instructions else '') + merged_system + '<|im_end|>\\n' }}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('') and content.endswith('')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if loop.index0 >= num_sys %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" or message.role == \"developer\" %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n\\n' + reasoning_content + '\\n\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if tool_call.name is not defined or tool_call.name is none %}\n {{- raise_exception('Tool call is missing a function name.') }}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n\\n\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is mapping %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '\\n' }}\n {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}\n {{- args_value }}\n {{- '\\n\\n' }}\n {%- endfor %}\n {%- elif tool_call.arguments is string %}\n {%- if tool_call.arguments|trim %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" were passed as a JSON string. Parse them into an object before calling apply_chat_template.') }}\n {%- endif %}\n {%- elif tool_call.arguments is defined and tool_call.arguments is not none %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" must be an object/mapping or a JSON string.') }}\n {%- endif %}\n {{- '\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- content }}\n {{- '\\n' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n' }}\n {%- endif %}\n{%- endif %}\n{#- Unsloth fixes - developer role, merged system messages, tool calling #}",
+ "chat_template_caps": {
+ "supports_object_arguments": true,
+ "supports_parallel_tool_calls": true,
+ "supports_preserve_reasoning": true,
+ "supports_reasoning_effort": true,
+ "supports_string_content": true,
+ "supports_system_role": true,
+ "supports_tool_calls": true,
+ "supports_tools": true,
+ "supports_typed_content": true
+ },
+ "bos_token": "<|endoftext|>",
+ "eos_token": "<|im_end|>",
+ "build_info": "b10964-b29c606e2",
+ "is_sleeping": false,
+ "cors_proxy_enabled": false
+ },
+ "slots": [
+ {
+ "id": 0,
+ "n_ctx": 192000,
+ "speculative": true,
+ "is_processing": false
+ }
+ ],
+ "smoke": {
+ "choices": [
+ {
+ "finish_reason": "stop",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "OK"
+ }
+ }
+ ],
+ "created": 1789937879,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 2,
+ "prompt_tokens": 19,
+ "total_tokens": 21,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-OJFZCBwqN40FOsi9rr2CRe9ZoQRj7pL7",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 19,
+ "prompt_ms": 773.198,
+ "prompt_per_token_ms": 40.694631578947366,
+ "prompt_per_second": 24.573265838763163,
+ "predicted_n": 2,
+ "predicted_ms": 286.198,
+ "predicted_per_token_ms": 286.198,
+ "predicted_per_second": 3.494084514916247,
+ "draft_n": 2,
+ "draft_n_accepted": 2
+ },
+ "wall_seconds": 1.0617731009842828
+ },
+ "decode": [
+ {
+ "choices": [
+ {
+ "finish_reason": "length",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "Ein Reverse Proxy ist ein zentrales Element in modernen IT-Infrastrukturen, das als Vermittler zwischen Clienten und Backend-Servern agiert. Im Gegensatz zu einem Forward Proxy, der Anfragen von internen Clients nach außen vertritt, stellt der Reverse Proxy eine Eindepunkt-Schnittstelle (Single Point of Entry) für externe Anfragen dar. Typische Vertreter sind nginx, HAProxy oder Apache HTTP Server. Die Grundprinzipie des Reverse Proxys besteht darin, eingehende HTTP- oder HTTPS-Anfragen entgegenzunehmen und diese basierend auf definierten Routing-Regeln an die geeigneten Backend-Server weiterzuleiten. Diese Architektur bietet erhebliche Vorteile wie Lastverteilung, SSL/TLS-Terminierung, Caching und eine vereinfachte Verwaltung mehrerer hinterlegter Dienste. Durch die Abstraktion der Backend-Infrastruktur wird sichergestellt, dass Clients die Details der internen Topologie, wie IP-Adressen oder Server-Namen, nicht kennen müssen.\n\nIn dynamischen Umgebungen, insbesondere bei der Nutzung von Container-Orchestrierungsplattformen wie Kubernetes oder Docker Compose, ist die Stabilität der Netzwerkanbindung eine große Herausforderung. Container sind per Definition"
+ }
+ }
+ ],
+ "created": 1789937883,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 256,
+ "prompt_tokens": 59,
+ "total_tokens": 315,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-KbgSBnegXOb9jTGM0YKi9dtVhTrzsGG3",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 59,
+ "prompt_ms": 119.336,
+ "prompt_per_token_ms": 2.02264406779661,
+ "prompt_per_second": 494.40235972380503,
+ "predicted_n": 256,
+ "predicted_ms": 4155.571,
+ "predicted_per_token_ms": 16.296356862745096,
+ "predicted_per_second": 61.36340830177129,
+ "draft_n": 245,
+ "draft_n_accepted": 131
+ },
+ "wall_seconds": 4.33330046699848
+ },
+ {
+ "choices": [
+ {
+ "finish_reason": "length",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "```python\nimport asyncio\nfrom typing import Any, Awaitable, Callable, List, Optional, Union\n\n\nasync def first_success(\n awaitables: List[Awaitable],\n return_exceptions: bool = True\n) -> Any:\n \"\"\"\n Start all awaitables concurrently, return the first successful result,\n cancel and await remaining tasks, and collect exceptions if all fail.\n\n Args:\n awaitables: A list of awaitables to run concurrently.\n return_exceptions: If True, when all awaitables fail, raise an\n ExceptionGroup (Python 3.11+) or re-raise the first exception\n with others attached. If False, simply re-raise the first exception.\n\n Returns:\n The result of the first successfully completed awaitable.\n\n Raises:\n ExceptionGroup (or the first exception if return_exceptions=False):\n If all awaitables fail.\n \"\"\"\n if not awaitables:\n raise ValueError(\"awaitables list must not be empty\")\n\n # Create tasks from all awaitables\n tasks = [asyncio.ensure_future(av) for av in awaitables]\n\n try:\n # Use asyncio.wait to get the"
+ }
+ }
+ ],
+ "created": 1789937887,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 256,
+ "prompt_tokens": 54,
+ "total_tokens": 310,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-FuQzoyB89ose2K7pstCmaTLcZYaamzlC",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 54,
+ "prompt_ms": 117.583,
+ "prompt_per_token_ms": 2.177462962962963,
+ "prompt_per_second": 459.2500616585731,
+ "predicted_n": 256,
+ "predicted_ms": 3275.917,
+ "predicted_per_token_ms": 12.846733333333333,
+ "predicted_per_second": 77.8407999958485,
+ "draft_n": 195,
+ "draft_n_accepted": 156
+ },
+ "wall_seconds": 3.4525404840242118
+ }
+ ],
+ "prefill": [
+ {
+ "target": 4096,
+ "response": {
+ "choices": [
+ {
+ "finish_reason": "length",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "```json\n{\n \"alpha\": \"RAVEN-417\",\n \"beta\": \"CEDAR-928\",\n \"gamma\": \"ORBIT-563\"\n}\n```\n\nDie Verifizierung der drei identifizierten Nadel-Werte innerhalb des vorliegenden Datensatzes erfordert eine methodische und strenge Vorgehensweise, die auf der direkten textuellen Analyse basiert, ohne interpretative Überlegungen oder das Erfinden nicht vorhandener Belege hinzuzuziehen. Um zu garantieren, dass die Werte `RAVEN-417`, `CEDAR-928` und `ORBIT-563` korrekt identifiziert sind, muss man zunächst den Rohdatenstrom systematisch durchsuchen. Der Kern der Verifizierung liegt darin, die spezifischen Markierungen im Text zu lokalisieren, die durch die Präfixe `NEEDLE_ALPHA=`, `NEEDLE_BETA=` und `NEEDLE_GAMMA=` gekennzeichnet sind. \n\nIm vorliegenden Fall finden wir diese Markierungen an sehr spezifischen und unregelmäßigen Stellen in den Zeilen 000024, 000097 und 000170. Es ist entscheidend zu verstehen, dass diese Einträge nicht als vollständige, saubere Datensätze im Format der anderen Zeilen vorliegen, sondern als unterbrochene Zeilen erscheinen, in denen die Nadel-Definitionen in den laufenden Textfluss eingebettet sind. Zum Beispiel bricht Record 000024 nach \"cobalt lant\" ab, gefolgt von der Zeile \"NEEDLE_ALPHA=RAVEN-417\". Eine korrekte Verifizierung bedeutet hier, den genauen String-Wert nach dem Gleichheitszeichen zu extrahieren und ihn mit dem erwarteten Muster abzugleichen. Man darf nicht annehmen, dass der Rest der Zeile fehlt oder manipuliert wurde; man muss akzeptieren, dass die Definition genau an dieser Stelle steht. \n\nGenauso verhält es sich bei Record 000097, wo \"NEEDLE_BETA=CEDAR-928\" erscheint, und bei Record 000170 mit \"NEEDLE_GAMMA=ORBIT-563\". Um Fehler zu vermeiden, sollte man keine automatischen Korrekturmechanismen anwenden, die versuchend"
+ }
+ }
+ ],
+ "created": 1789937897,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 512,
+ "prompt_tokens": 4196,
+ "total_tokens": 4708,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-41Euh1CBxKMlhzOIvw92C6hxdNhmqIwH",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 4196,
+ "prompt_ms": 2080.562,
+ "prompt_per_token_ms": 0.4958441372735939,
+ "prompt_per_second": 2016.7627785184966,
+ "predicted_n": 512,
+ "predicted_ms": 7693.879,
+ "predicted_per_token_ms": 15.056514677103719,
+ "predicted_per_second": 66.41643311520755,
+ "draft_n": 451,
+ "draft_n_accepted": 284
+ },
+ "wall_seconds": 9.837755398999434,
+ "recall": {
+ "RAVEN-417": true,
+ "CEDAR-928": true,
+ "ORBIT-563": true
+ }
+ }
+ }
+ ],
+ "finished": 1789937897.414032
+}
diff --git a/experiments/profile-mtp2-20260920/large-mtp2/server-args.json b/experiments/profile-mtp2-20260920/large-mtp2/server-args.json
new file mode 100644
index 0000000..826e1d4
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp2/server-args.json
@@ -0,0 +1,73 @@
+[
+ "--model",
+ "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "--mmproj",
+ "/models/qwen/mmproj-BF16.gguf",
+ "--mmproj-offload",
+ "--mmproj-device",
+ "CUDA1",
+ "--alias",
+ "qwen-large",
+ "--ctx-size",
+ "192000",
+ "--flash-attn",
+ "on",
+ "--cache-type-k",
+ "q4_0",
+ "--cache-type-v",
+ "q4_0",
+ "--cache-prompt",
+ "--cache-ram",
+ "32768",
+ "--threads",
+ "6",
+ "--threads-batch",
+ "6",
+ "--batch-size",
+ "2048",
+ "--ubatch-size",
+ "256",
+ "--parallel",
+ "1",
+ "--kv-unified",
+ "--jinja",
+ "--reasoning",
+ "auto",
+ "--reasoning-preserve",
+ "--host",
+ "127.0.0.1",
+ "--port",
+ "5005",
+ "--metrics",
+ "--fit",
+ "off",
+ "--n-gpu-layers",
+ "all",
+ "--load-mode",
+ "none",
+ "--no-ui",
+ "--temperature",
+ "0.2",
+ "--top-p",
+ "0.8",
+ "--top-k",
+ "20",
+ "--device",
+ "CUDA0,CUDA1",
+ "--main-gpu",
+ "0",
+ "--split-mode",
+ "layer",
+ "--tensor-split",
+ "86,14",
+ "--spec-type",
+ "draft-mtp",
+ "--spec-draft-n-max",
+ "2",
+ "--spec-draft-type-k",
+ "f16",
+ "--spec-draft-type-v",
+ "f16",
+ "--verbosity",
+ "3"
+]
diff --git a/experiments/profile-mtp2-20260920/large-mtp2/server.log.gz b/experiments/profile-mtp2-20260920/large-mtp2/server.log.gz
new file mode 100644
index 0000000..a157b4e
Binary files /dev/null and b/experiments/profile-mtp2-20260920/large-mtp2/server.log.gz differ
diff --git a/experiments/profile-mtp2-20260920/large-mtp3/config.json b/experiments/profile-mtp2-20260920/large-mtp3/config.json
new file mode 100644
index 0000000..af153ac
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp3/config.json
@@ -0,0 +1,11 @@
+{
+ "label": "large-mtp3",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 3,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+}
diff --git a/experiments/profile-mtp2-20260920/large-mtp3/gpu.json b/experiments/profile-mtp2-20260920/large-mtp3/gpu.json
new file mode 100644
index 0000000..6bd09a2
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp3/gpu.json
@@ -0,0 +1,278 @@
+[
+ {
+ "time": 1789937847.9726973,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "6964",
+ "total": "12288",
+ "temp": "50",
+ "util": "100",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "11272",
+ "total": "16303",
+ "temp": "48",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937850.0014837,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "8350",
+ "total": "12288",
+ "temp": "50",
+ "util": "4",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15682",
+ "total": "16303",
+ "temp": "49",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937852.0301683,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10722",
+ "total": "12288",
+ "temp": "50",
+ "util": "0",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15684",
+ "total": "16303",
+ "temp": "48",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937854.061566,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10726",
+ "total": "12288",
+ "temp": "56",
+ "util": "59",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15700",
+ "total": "16303",
+ "temp": "56",
+ "util": "47",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937856.0915275,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10726",
+ "total": "12288",
+ "temp": "57",
+ "util": "53",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15700",
+ "total": "16303",
+ "temp": "58",
+ "util": "52",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937858.1227798,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10726",
+ "total": "12288",
+ "temp": "56",
+ "util": "47",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15700",
+ "total": "16303",
+ "temp": "57",
+ "util": "52",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937860.1527982,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10726",
+ "total": "12288",
+ "temp": "58",
+ "util": "56",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15700",
+ "total": "16303",
+ "temp": "58",
+ "util": "44",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937862.1820376,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10728",
+ "total": "12288",
+ "temp": "56",
+ "util": "50",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15702",
+ "total": "16303",
+ "temp": "69",
+ "util": "82",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937864.2163124,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10728",
+ "total": "12288",
+ "temp": "58",
+ "util": "54",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15702",
+ "total": "16303",
+ "temp": "60",
+ "util": "39",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937866.2504282,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10728",
+ "total": "12288",
+ "temp": "59",
+ "util": "42",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15702",
+ "total": "16303",
+ "temp": "60",
+ "util": "46",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937868.2800074,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10728",
+ "total": "12288",
+ "temp": "59",
+ "util": "48",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15702",
+ "total": "16303",
+ "temp": "60",
+ "util": "44",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937870.3127642,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10728",
+ "total": "12288",
+ "temp": "59",
+ "util": "59",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15702",
+ "total": "16303",
+ "temp": "60",
+ "util": "40",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ }
+]
diff --git a/experiments/profile-mtp2-20260920/large-mtp3/loaded.json b/experiments/profile-mtp2-20260920/large-mtp3/loaded.json
new file mode 100644
index 0000000..34f86f1
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp3/loaded.json
@@ -0,0 +1,133 @@
+{
+ "case": {
+ "label": "large-mtp3",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 3,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ "started": 1789937845.9410455,
+ "idle_gpu": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10722",
+ "total": "12288",
+ "temp": "50",
+ "util": "0",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15684",
+ "total": "16303",
+ "temp": "48",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ],
+ "props": {
+ "default_generation_settings": {
+ "params": {
+ "seed": 4294967295,
+ "temperature": 0.20000000298023224,
+ "dynatemp_range": 0.0,
+ "dynatemp_exponent": 1.0,
+ "top_k": 20,
+ "top_p": 0.800000011920929,
+ "min_p": 0.05000000074505806,
+ "top_n_sigma": -1.0,
+ "xtc_probability": 0.0,
+ "xtc_threshold": 0.10000000149011612,
+ "typical_p": 1.0,
+ "repeat_last_n": 64,
+ "repeat_penalty": 1.0,
+ "presence_penalty": 0.0,
+ "frequency_penalty": 0.0,
+ "dry_multiplier": 0.0,
+ "dry_base": 1.75,
+ "dry_allowed_length": 2,
+ "dry_penalty_last_n": 64,
+ "mirostat": 0,
+ "mirostat_tau": 5.0,
+ "mirostat_eta": 0.10000000149011612,
+ "adaptive_target": -1.0,
+ "adaptive_decay": 0.8999999761581421,
+ "max_tokens": -1,
+ "n_predict": -1,
+ "n_keep": 0,
+ "n_discard": 0,
+ "ignore_eos": false,
+ "stream": false,
+ "n_probs": 0,
+ "min_keep": 0,
+ "chat_format": "Content-only",
+ "reasoning_format": "none",
+ "reasoning_in_content": false,
+ "generation_prompt": "",
+ "samplers": [
+ "penalties",
+ "dry",
+ "top_n_sigma",
+ "top_k",
+ "typ_p",
+ "top_p",
+ "min_p",
+ "xtc",
+ "temperature"
+ ],
+ "speculative.types": "none",
+ "timings_per_token": false,
+ "post_sampling_probs": false,
+ "backend_sampling": false,
+ "lora": []
+ },
+ "n_ctx": 192000
+ },
+ "total_slots": 1,
+ "model_alias": "qwen-large",
+ "model_ftype": "IQ4_XS - 4.25 bpw",
+ "model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "modalities": {
+ "vision": true,
+ "video": true,
+ "audio": false
+ },
+ "media_marker": "<__media_hAMBKxzOCNfgC3Hs9kmmgkfFiAKv6k09__>",
+ "endpoint_slots": true,
+ "endpoint_props": false,
+ "endpoint_metrics": true,
+ "ui": false,
+ "ui_settings": {},
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set sysns = namespace(count=0, text='') %}\n{%- for message in messages %}\n {%- if sysns.count == loop.index0 and (message.role == 'system' or message.role == 'developer') %}\n {%- set sys_content = render_content(message.content, false, true)|trim %}\n {%- if sys_content %}\n {%- set sysns.text = sysns.text + ('\\n' if sysns.text else '') + sys_content %}\n {%- endif %}\n {%- set sysns.count = sysns.count + 1 %}\n {%- endif %}\n{%- endfor %}\n{%- set num_sys = sysns.count %}\n{%- set merged_system = sysns.text %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort == 'high' %}\n {%- set resolved_reasoning_effort = 'xhigh' %}\n {%- endif %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n\\n\\n\\nvalue_1\\n\\n\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n\\n\\n\\n\\n\\nReminder:\\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n' }}\n {%- if merged_system %}\n {{- '\\n\\n' + merged_system }}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if merged_system %}\n {{- '<|im_start|>system\\n' + (reasoning_instructions + '\\n\\n' if reasoning_instructions else '') + merged_system + '<|im_end|>\\n' }}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('') and content.endswith('')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if loop.index0 >= num_sys %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" or message.role == \"developer\" %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n\\n' + reasoning_content + '\\n\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if tool_call.name is not defined or tool_call.name is none %}\n {{- raise_exception('Tool call is missing a function name.') }}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n\\n\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is mapping %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '\\n' }}\n {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}\n {{- args_value }}\n {{- '\\n\\n' }}\n {%- endfor %}\n {%- elif tool_call.arguments is string %}\n {%- if tool_call.arguments|trim %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" were passed as a JSON string. Parse them into an object before calling apply_chat_template.') }}\n {%- endif %}\n {%- elif tool_call.arguments is defined and tool_call.arguments is not none %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" must be an object/mapping or a JSON string.') }}\n {%- endif %}\n {{- '\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- content }}\n {{- '\\n' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n' }}\n {%- endif %}\n{%- endif %}\n{#- Unsloth fixes - developer role, merged system messages, tool calling #}",
+ "chat_template_caps": {
+ "supports_object_arguments": true,
+ "supports_parallel_tool_calls": true,
+ "supports_preserve_reasoning": true,
+ "supports_reasoning_effort": true,
+ "supports_string_content": true,
+ "supports_system_role": true,
+ "supports_tool_calls": true,
+ "supports_tools": true,
+ "supports_typed_content": true
+ },
+ "bos_token": "<|endoftext|>",
+ "eos_token": "<|im_end|>",
+ "build_info": "b10964-b29c606e2",
+ "is_sleeping": false,
+ "cors_proxy_enabled": false
+ },
+ "slots": [
+ {
+ "id": 0,
+ "n_ctx": 192000,
+ "speculative": true,
+ "is_processing": false
+ }
+ ]
+}
diff --git a/experiments/profile-mtp2-20260920/large-mtp3/result.json b/experiments/profile-mtp2-20260920/large-mtp3/result.json
new file mode 100644
index 0000000..52f64cd
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp3/result.json
@@ -0,0 +1,302 @@
+{
+ "case": {
+ "label": "large-mtp3",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 3,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ "started": 1789937845.9410455,
+ "idle_gpu": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "10722",
+ "total": "12288",
+ "temp": "50",
+ "util": "0",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15684",
+ "total": "16303",
+ "temp": "48",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ],
+ "props": {
+ "default_generation_settings": {
+ "params": {
+ "seed": 4294967295,
+ "temperature": 0.20000000298023224,
+ "dynatemp_range": 0.0,
+ "dynatemp_exponent": 1.0,
+ "top_k": 20,
+ "top_p": 0.800000011920929,
+ "min_p": 0.05000000074505806,
+ "top_n_sigma": -1.0,
+ "xtc_probability": 0.0,
+ "xtc_threshold": 0.10000000149011612,
+ "typical_p": 1.0,
+ "repeat_last_n": 64,
+ "repeat_penalty": 1.0,
+ "presence_penalty": 0.0,
+ "frequency_penalty": 0.0,
+ "dry_multiplier": 0.0,
+ "dry_base": 1.75,
+ "dry_allowed_length": 2,
+ "dry_penalty_last_n": 64,
+ "mirostat": 0,
+ "mirostat_tau": 5.0,
+ "mirostat_eta": 0.10000000149011612,
+ "adaptive_target": -1.0,
+ "adaptive_decay": 0.8999999761581421,
+ "max_tokens": -1,
+ "n_predict": -1,
+ "n_keep": 0,
+ "n_discard": 0,
+ "ignore_eos": false,
+ "stream": false,
+ "n_probs": 0,
+ "min_keep": 0,
+ "chat_format": "Content-only",
+ "reasoning_format": "none",
+ "reasoning_in_content": false,
+ "generation_prompt": "",
+ "samplers": [
+ "penalties",
+ "dry",
+ "top_n_sigma",
+ "top_k",
+ "typ_p",
+ "top_p",
+ "min_p",
+ "xtc",
+ "temperature"
+ ],
+ "speculative.types": "none",
+ "timings_per_token": false,
+ "post_sampling_probs": false,
+ "backend_sampling": false,
+ "lora": []
+ },
+ "n_ctx": 192000
+ },
+ "total_slots": 1,
+ "model_alias": "qwen-large",
+ "model_ftype": "IQ4_XS - 4.25 bpw",
+ "model_path": "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "modalities": {
+ "vision": true,
+ "video": true,
+ "audio": false
+ },
+ "media_marker": "<__media_hAMBKxzOCNfgC3Hs9kmmgkfFiAKv6k09__>",
+ "endpoint_slots": true,
+ "endpoint_props": false,
+ "endpoint_metrics": true,
+ "ui": false,
+ "ui_settings": {},
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- set sysns = namespace(count=0, text='') %}\n{%- for message in messages %}\n {%- if sysns.count == loop.index0 and (message.role == 'system' or message.role == 'developer') %}\n {%- set sys_content = render_content(message.content, false, true)|trim %}\n {%- if sys_content %}\n {%- set sysns.text = sysns.text + ('\\n' if sysns.text else '') + sys_content %}\n {%- endif %}\n {%- set sysns.count = sysns.count + 1 %}\n {%- endif %}\n{%- endfor %}\n{%- set num_sys = sysns.count %}\n{%- set merged_system = sysns.text %}\n{%- set reasoning_instructions = '' %}\n{%- if enable_thinking is undefined or enable_thinking is true %}\n {%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}\n {%- if resolved_reasoning_effort == 'high' %}\n {%- set resolved_reasoning_effort = 'xhigh' %}\n {%- endif %}\n {%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}\n {{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}\n {%- endif %}\n {%- if resolved_reasoning_effort == 'xhigh' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}\n {%- elif resolved_reasoning_effort == 'low' %}\n {%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}\n {%- endif %}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {%- if reasoning_instructions %}\n {{- reasoning_instructions + '\\n\\n' }}\n {%- endif %}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n\\n\\n\\nvalue_1\\n\\n\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n\\n\\n\\n\\n\\nReminder:\\n- Function calls MUST follow the specified format: an inner block must be nested within XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n' }}\n {%- if merged_system %}\n {{- '\\n\\n' + merged_system }}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if merged_system %}\n {{- '<|im_start|>system\\n' + (reasoning_instructions + '\\n\\n' if reasoning_instructions else '') + merged_system + '<|im_end|>\\n' }}\n {%- elif reasoning_instructions %}\n {{- '<|im_start|>system\\n' + reasoning_instructions + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('') and content.endswith('')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- for message in messages %}\n {%- if loop.index0 >= num_sys %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" or message.role == \"developer\" %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n\\n' + reasoning_content + '\\n\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if tool_call.name is not defined or tool_call.name is none %}\n {{- raise_exception('Tool call is missing a function name.') }}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n\\n\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is mapping %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '\\n' }}\n {%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}\n {{- args_value }}\n {{- '\\n\\n' }}\n {%- endfor %}\n {%- elif tool_call.arguments is string %}\n {%- if tool_call.arguments|trim %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" were passed as a JSON string. Parse them into an object before calling apply_chat_template.') }}\n {%- endif %}\n {%- elif tool_call.arguments is defined and tool_call.arguments is not none %}\n {{- raise_exception('Tool call arguments for function \"' + (tool_call.name | string) + '\" must be an object/mapping or a JSON string.') }}\n {%- endif %}\n {{- '\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n\\n' }}\n {{- content }}\n {{- '\\n' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is false %}\n {{- '\\n\\n\\n\\n' }}\n {%- else %}\n {{- '\\n' }}\n {%- endif %}\n{%- endif %}\n{#- Unsloth fixes - developer role, merged system messages, tool calling #}",
+ "chat_template_caps": {
+ "supports_object_arguments": true,
+ "supports_parallel_tool_calls": true,
+ "supports_preserve_reasoning": true,
+ "supports_reasoning_effort": true,
+ "supports_string_content": true,
+ "supports_system_role": true,
+ "supports_tool_calls": true,
+ "supports_tools": true,
+ "supports_typed_content": true
+ },
+ "bos_token": "<|endoftext|>",
+ "eos_token": "<|im_end|>",
+ "build_info": "b10964-b29c606e2",
+ "is_sleeping": false,
+ "cors_proxy_enabled": false
+ },
+ "slots": [
+ {
+ "id": 0,
+ "n_ctx": 192000,
+ "speculative": true,
+ "is_processing": false
+ }
+ ],
+ "smoke": {
+ "choices": [
+ {
+ "finish_reason": "stop",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "OK"
+ }
+ }
+ ],
+ "created": 1789937853,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 2,
+ "prompt_tokens": 19,
+ "total_tokens": 21,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-bCJNioi2oUEITazGNAxnHtl7oOc1h5TV",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 19,
+ "prompt_ms": 775.163,
+ "prompt_per_token_ms": 40.79805263157895,
+ "prompt_per_second": 24.510973820989907,
+ "predicted_n": 2,
+ "predicted_ms": 301.346,
+ "predicted_per_token_ms": 301.346,
+ "predicted_per_second": 3.318444578657092,
+ "draft_n": 3,
+ "draft_n_accepted": 3
+ },
+ "wall_seconds": 1.0790359990205616
+ },
+ "decode": [
+ {
+ "choices": [
+ {
+ "finish_reason": "length",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "Ein Reverse Proxy ist ein zentrales Element in modernen IT-Infrastrukturen, das als Vermittler zwischen Clienten und Backend-Servern agiert. Im Gegensatz zu einem Forward Proxy, der Anfragen von internen Clients nach außen vertritt, stellt der Reverse Proxy eine Eindepunkt-Schnittstelle (Single Point of Entry) für externe Anfragen dar. Typische Vertreter sind nginx, HAProxy oder Apache HTTP Server. Die Grundprinzipie des Reverse Proxys besteht darin, eingehende HTTP- oder HTTPS-Anfragen entgegenzunehmen und diese basierend auf definierten Routing-Regeln an die geeigneten Backend-Server weiterzuleiten. Diese Architektur bietet erhebliche Vorteile wie Lastverteilung, SSL/TLS-Terminierung, Caching und eine vereinfachte Verwaltung mehrerer hinterlegter Dienste. Durch die Abstraktion der Backend-Infrastruktur wird sichergestellt, dass Clients die Details der internen Topologie, wie IP-Adressen oder Server-Namen, nicht kennen müssen.\n\nIn dynamischen Umgebungen, insbesondere bei der Nutzung von Container-Orchestrierungsplattformen wie Kubernetes oder Docker Compose, ist die Stabilität der IP-Adressen ein wesentlicher Faktor. Container sind per"
+ }
+ }
+ ],
+ "created": 1789937857,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 256,
+ "prompt_tokens": 59,
+ "total_tokens": 315,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-tEHgKUwpLCIA8UJBia6svZw83OyzrcmS",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 59,
+ "prompt_ms": 119.504,
+ "prompt_per_token_ms": 2.0254915254237287,
+ "prompt_per_second": 493.70732360423074,
+ "predicted_n": 256,
+ "predicted_ms": 4273.816,
+ "predicted_per_token_ms": 16.76006274509804,
+ "predicted_per_second": 59.6656477489906,
+ "draft_n": 328,
+ "draft_n_accepted": 144
+ },
+ "wall_seconds": 4.451622880005743
+ },
+ {
+ "choices": [
+ {
+ "finish_reason": "length",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "```python\nimport asyncio\nfrom typing import Any, Awaitable, Callable, List, Optional, Union\n\n\nasync def first_success(\n awaitables: List[Awaitable],\n return_exceptions: bool = True\n) -> Any:\n \"\"\"\n Start all awaitables concurrently, return the first successful result,\n cancel and await remaining tasks, and collect exceptions if all fail.\n\n Args:\n awaitables: A list of awaitables to run concurrently.\n return_exceptions: If True, when all awaitables fail, raise an\n ExceptionGroup (Python 3.11+) or re-raise the first exception\n with others attached. If False, simply re-raise the first exception.\n\n Returns:\n The result of the first successfully completed awaitable.\n\n Raises:\n ExceptionGroup (or the first exception if return_exceptions=False):\n If all awaitables fail.\n \"\"\"\n if not awaitables:\n raise ValueError(\"awaitables list must not be empty\")\n\n # Create tasks from all awaitables\n tasks = [asyncio.ensure_future(av) for av in awaitables]\n\n try:\n # Use asyncio.wait to get the"
+ }
+ }
+ ],
+ "created": 1789937861,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 256,
+ "prompt_tokens": 54,
+ "total_tokens": 310,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-iMNksmKOEmz5a1TEQOwMAdf0N9yfpAJA",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 54,
+ "prompt_ms": 118.276,
+ "prompt_per_token_ms": 2.1902962962962964,
+ "prompt_per_second": 456.5592343332545,
+ "predicted_n": 256,
+ "predicted_ms": 3304.062,
+ "predicted_per_token_ms": 12.957105882352941,
+ "predicted_per_second": 77.17772850509463,
+ "draft_n": 257,
+ "draft_n_accepted": 168
+ },
+ "wall_seconds": 3.481562177010346
+ }
+ ],
+ "prefill": [
+ {
+ "target": 4096,
+ "response": {
+ "choices": [
+ {
+ "finish_reason": "length",
+ "index": 0,
+ "message": {
+ "role": "assistant",
+ "content": "```json\n{\n \"alpha\": \"RAVEN-417\",\n \"beta\": \"CEDAR-928\",\n \"gamma\": \"ORBIT-563\"\n}\n```\n\nDie Verifizierung der drei identifizierten Nadel-Werte innerhalb des vorliegenden Datensatzes erfordert eine methodische und strenge Vorgehensweise, die auf der direkten textuellen Analyse basiert, ohne interpretative Überlegungen oder das Erfinden nicht vorhandener Belege hinzuzuziehen. Um zu garantieren, dass die Werte `RAVEN-417`, `CEDAR-928` und `ORBIT-563` korrekt identifiziert sind, muss man zunächst den Rohdatenstrom systematisch durchsuchen. Der Kern der Verifizierung liegt darin, die spezifischen Markierungen im Text zu lokalisieren, die durch die Präfixe `NEEDLE_ALPHA=`, `NEEDLE_BETA=` und `NEEDLE_GAMMA=` gekennzeichnet sind. \n\nIm vorliegenden Fall finden wir diese Markierungen an sehr spezifischen und unregelmäßigen Stellen in den Zeilen 000024, 000097 und 000170. Es ist entscheidend zu verstehen, dass diese Einträge nicht als vollständige, saubere Datensätze im Format der anderen Zeilen vorliegen, sondern als unterbrochene Zeilen erscheinen, in denen die Nadel-Definitionen in den laufenden Textfluss eingebettet sind. Zum Beispiel bricht Record 000024 nach \"cobalt lant\" ab, gefolgt von der Zeile \"NEEDLE_ALPHA=RAVEN-417\". Eine korrekte Verifizierung bedeutet hier, den genauen String-Wert nach dem Gleichheitszeichen zu extrahieren und ihn mit dem erwarteten Muster abzugleichen. Man darf nicht annehmen, dass der Rest der Zeile fehlt oder ergänzt werden muss; man muss genau das akzeptieren, was vorliegt.\n\nUm diese Prozesse zu objektivieren, empfiehlt es sich, automatische Suchfunktionen wie reguläre Ausdrücke (Regex) zu verwenden. Für Alpha könnte ein Suchmuster wie `NEEDLE_ALPHA=(\\w+[-\\d]+)` verwendet werden, um den Wert direkt nach dem Gleichungszeichen zu greifen. Dies eliminiert menschliche Fehler bei der manuellen Ablesung. Zusätzlich sollte man den Kont"
+ }
+ }
+ ],
+ "created": 1789937871,
+ "model": "qwen-large",
+ "system_fingerprint": "b10964-b29c606e2",
+ "object": "chat.completion",
+ "usage": {
+ "completion_tokens": 512,
+ "prompt_tokens": 4196,
+ "total_tokens": 4708,
+ "prompt_tokens_details": {
+ "cached_tokens": 0
+ }
+ },
+ "id": "chatcmpl-uRbhbuOoWzXGl3clcx5uHpkU2BXTN0NX",
+ "timings": {
+ "cache_n": 0,
+ "prompt_n": 4196,
+ "prompt_ms": 2136.011,
+ "prompt_per_token_ms": 0.5090588655862727,
+ "prompt_per_second": 1964.4093593150972,
+ "predicted_n": 512,
+ "predicted_ms": 8181.726,
+ "predicted_per_token_ms": 16.011205479452055,
+ "predicted_per_second": 62.45625922940954,
+ "draft_n": 620,
+ "draft_n_accepted": 303
+ },
+ "wall_seconds": 10.38065240799915,
+ "recall": {
+ "RAVEN-417": true,
+ "CEDAR-928": true,
+ "ORBIT-563": true
+ }
+ }
+ }
+ ],
+ "finished": 1789937871.640576
+}
diff --git a/experiments/profile-mtp2-20260920/large-mtp3/server-args.json b/experiments/profile-mtp2-20260920/large-mtp3/server-args.json
new file mode 100644
index 0000000..02ee73e
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large-mtp3/server-args.json
@@ -0,0 +1,73 @@
+[
+ "--model",
+ "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "--mmproj",
+ "/models/qwen/mmproj-BF16.gguf",
+ "--mmproj-offload",
+ "--mmproj-device",
+ "CUDA1",
+ "--alias",
+ "qwen-large",
+ "--ctx-size",
+ "192000",
+ "--flash-attn",
+ "on",
+ "--cache-type-k",
+ "q4_0",
+ "--cache-type-v",
+ "q4_0",
+ "--cache-prompt",
+ "--cache-ram",
+ "32768",
+ "--threads",
+ "6",
+ "--threads-batch",
+ "6",
+ "--batch-size",
+ "2048",
+ "--ubatch-size",
+ "256",
+ "--parallel",
+ "1",
+ "--kv-unified",
+ "--jinja",
+ "--reasoning",
+ "auto",
+ "--reasoning-preserve",
+ "--host",
+ "127.0.0.1",
+ "--port",
+ "5005",
+ "--metrics",
+ "--fit",
+ "off",
+ "--n-gpu-layers",
+ "all",
+ "--load-mode",
+ "none",
+ "--no-ui",
+ "--temperature",
+ "0.2",
+ "--top-p",
+ "0.8",
+ "--top-k",
+ "20",
+ "--device",
+ "CUDA0,CUDA1",
+ "--main-gpu",
+ "0",
+ "--split-mode",
+ "layer",
+ "--tensor-split",
+ "86,14",
+ "--spec-type",
+ "draft-mtp",
+ "--spec-draft-n-max",
+ "3",
+ "--spec-draft-type-k",
+ "f16",
+ "--spec-draft-type-v",
+ "f16",
+ "--verbosity",
+ "3"
+]
diff --git a/experiments/profile-mtp2-20260920/large-mtp3/server.log.gz b/experiments/profile-mtp2-20260920/large-mtp3/server.log.gz
new file mode 100644
index 0000000..8527e6a
Binary files /dev/null and b/experiments/profile-mtp2-20260920/large-mtp3/server.log.gz differ
diff --git a/experiments/profile-mtp2-20260920/large.json b/experiments/profile-mtp2-20260920/large.json
new file mode 100644
index 0000000..aa64063
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/large.json
@@ -0,0 +1,74 @@
+{
+ "image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
+ "args": [
+ "--model",
+ "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "--mmproj",
+ "/models/qwen/mmproj-BF16.gguf",
+ "--mmproj-offload",
+ "--mmproj-device",
+ "CUDA1",
+ "--alias",
+ "qwen-large",
+ "--ctx-size",
+ "192000",
+ "--flash-attn",
+ "on",
+ "--cache-type-k",
+ "q4_0",
+ "--cache-type-v",
+ "q4_0",
+ "--cache-prompt",
+ "--cache-ram",
+ "32768",
+ "--threads",
+ "6",
+ "--threads-batch",
+ "6",
+ "--batch-size",
+ "2048",
+ "--ubatch-size",
+ "256",
+ "--parallel",
+ "1",
+ "--kv-unified",
+ "--jinja",
+ "--reasoning",
+ "auto",
+ "--reasoning-preserve",
+ "--host",
+ "0.0.0.0",
+ "--port",
+ "8080",
+ "--metrics",
+ "--fit",
+ "off",
+ "--n-gpu-layers",
+ "all",
+ "--load-mode",
+ "none",
+ "--no-ui",
+ "--temperature",
+ "0.2",
+ "--top-p",
+ "0.8",
+ "--top-k",
+ "20",
+ "--device",
+ "CUDA0,CUDA1",
+ "--main-gpu",
+ "0",
+ "--split-mode",
+ "layer",
+ "--tensor-split",
+ "86,14",
+ "--spec-type",
+ "draft-mtp",
+ "--spec-draft-n-max",
+ "3",
+ "--spec-draft-type-k",
+ "f16",
+ "--spec-draft-type-v",
+ "f16"
+ ]
+}
diff --git a/experiments/profile-mtp2-20260920/medium-mtp2/config.json b/experiments/profile-mtp2-20260920/medium-mtp2/config.json
new file mode 100644
index 0000000..e9c2b26
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/medium-mtp2/config.json
@@ -0,0 +1,11 @@
+{
+ "label": "medium-mtp2",
+ "profile": "medium",
+ "ubatch": 512,
+ "split": "85,15",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+}
diff --git a/experiments/profile-mtp2-20260920/medium-mtp2/gpu.json b/experiments/profile-mtp2-20260920/medium-mtp2/gpu.json
new file mode 100644
index 0000000..d737289
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/medium-mtp2/gpu.json
@@ -0,0 +1,71 @@
+[
+ {
+ "time": 1789937696.1053374,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "6964",
+ "total": "12288",
+ "temp": "46",
+ "util": "100",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "11272",
+ "total": "16303",
+ "temp": "45",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937698.1347463,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "8782",
+ "total": "12288",
+ "temp": "46",
+ "util": "0",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "15918",
+ "total": "16303",
+ "temp": "45",
+ "util": "1",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ },
+ {
+ "time": 1789937700.1621327,
+ "gpus": [
+ {
+ "name": "NVIDIA GeForce RTX 3060",
+ "used": "4645",
+ "total": "12288",
+ "temp": "46",
+ "util": "0",
+ "pcie_gen": "3",
+ "pcie_width": "4"
+ },
+ {
+ "name": "NVIDIA GeForce RTX 5080",
+ "used": "1",
+ "total": "16303",
+ "temp": "45",
+ "util": "0",
+ "pcie_gen": "4",
+ "pcie_width": "16"
+ }
+ ]
+ }
+]
diff --git a/experiments/profile-mtp2-20260920/medium-mtp2/result.json b/experiments/profile-mtp2-20260920/medium-mtp2/result.json
new file mode 100644
index 0000000..8da0714
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/medium-mtp2/result.json
@@ -0,0 +1,15 @@
+{
+ "case": {
+ "label": "medium-mtp2",
+ "profile": "medium",
+ "ubatch": 512,
+ "split": "85,15",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ "started": 1789937694.0730917,
+ "error": "Test container exited during load"
+}
diff --git a/experiments/profile-mtp2-20260920/medium-mtp2/server-args.json b/experiments/profile-mtp2-20260920/medium-mtp2/server-args.json
new file mode 100644
index 0000000..9a49a91
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/medium-mtp2/server-args.json
@@ -0,0 +1,75 @@
+[
+ "--model",
+ "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "--mmproj",
+ "/models/qwen/mmproj-BF16.gguf",
+ "--mmproj-offload",
+ "--mmproj-device",
+ "CUDA1",
+ "--alias",
+ "qwen-medium",
+ "--ctx-size",
+ "160000",
+ "--flash-attn",
+ "on",
+ "--cache-type-k",
+ "q4_0",
+ "--cache-type-v",
+ "q4_0",
+ "--cache-prompt",
+ "--cache-ram",
+ "32768",
+ "--threads",
+ "6",
+ "--threads-batch",
+ "6",
+ "--batch-size",
+ "2048",
+ "--ubatch-size",
+ "512",
+ "--parallel",
+ "2",
+ "--kv-unified",
+ "--jinja",
+ "--reasoning",
+ "auto",
+ "--reasoning-preserve",
+ "--host",
+ "127.0.0.1",
+ "--port",
+ "5005",
+ "--metrics",
+ "--fit",
+ "off",
+ "--n-gpu-layers",
+ "all",
+ "--load-mode",
+ "none",
+ "--no-ui",
+ "--temperature",
+ "1.0",
+ "--top-p",
+ "0.95",
+ "--top-k",
+ "20",
+ "--device",
+ "CUDA0,CUDA1",
+ "--main-gpu",
+ "0",
+ "--split-mode",
+ "layer",
+ "--tensor-split",
+ "85,15",
+ "--spec-type",
+ "draft-mtp",
+ "--spec-draft-n-max",
+ "2",
+ "--spec-draft-type-k",
+ "f16",
+ "--spec-draft-type-v",
+ "f16",
+ "--spec-draft-p-min",
+ "0.05",
+ "--verbosity",
+ "3"
+]
diff --git a/experiments/profile-mtp2-20260920/medium-mtp2/server.log.gz b/experiments/profile-mtp2-20260920/medium-mtp2/server.log.gz
new file mode 100644
index 0000000..c361666
Binary files /dev/null and b/experiments/profile-mtp2-20260920/medium-mtp2/server.log.gz differ
diff --git a/experiments/profile-mtp2-20260920/medium.json b/experiments/profile-mtp2-20260920/medium.json
new file mode 100644
index 0000000..2f2b1fb
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/medium.json
@@ -0,0 +1,76 @@
+{
+ "image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
+ "args": [
+ "--model",
+ "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "--mmproj",
+ "/models/qwen/mmproj-BF16.gguf",
+ "--mmproj-offload",
+ "--mmproj-device",
+ "CUDA1",
+ "--alias",
+ "qwen-medium",
+ "--ctx-size",
+ "160000",
+ "--flash-attn",
+ "on",
+ "--cache-type-k",
+ "q4_0",
+ "--cache-type-v",
+ "q4_0",
+ "--cache-prompt",
+ "--cache-ram",
+ "32768",
+ "--threads",
+ "6",
+ "--threads-batch",
+ "6",
+ "--batch-size",
+ "2048",
+ "--ubatch-size",
+ "512",
+ "--parallel",
+ "2",
+ "--kv-unified",
+ "--jinja",
+ "--reasoning",
+ "auto",
+ "--reasoning-preserve",
+ "--host",
+ "0.0.0.0",
+ "--port",
+ "8080",
+ "--metrics",
+ "--fit",
+ "off",
+ "--n-gpu-layers",
+ "all",
+ "--load-mode",
+ "none",
+ "--no-ui",
+ "--temperature",
+ "1.0",
+ "--top-p",
+ "0.95",
+ "--top-k",
+ "20",
+ "--device",
+ "CUDA0,CUDA1",
+ "--main-gpu",
+ "0",
+ "--split-mode",
+ "layer",
+ "--tensor-split",
+ "85,15",
+ "--spec-type",
+ "draft-mtp",
+ "--spec-draft-n-max",
+ "3",
+ "--spec-draft-type-k",
+ "f16",
+ "--spec-draft-type-v",
+ "f16",
+ "--spec-draft-p-min",
+ "0.05"
+ ]
+}
diff --git a/experiments/profile-mtp2-20260920/production.json b/experiments/profile-mtp2-20260920/production.json
new file mode 100644
index 0000000..2f2b1fb
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/production.json
@@ -0,0 +1,76 @@
+{
+ "image": "sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907",
+ "args": [
+ "--model",
+ "/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf",
+ "--mmproj",
+ "/models/qwen/mmproj-BF16.gguf",
+ "--mmproj-offload",
+ "--mmproj-device",
+ "CUDA1",
+ "--alias",
+ "qwen-medium",
+ "--ctx-size",
+ "160000",
+ "--flash-attn",
+ "on",
+ "--cache-type-k",
+ "q4_0",
+ "--cache-type-v",
+ "q4_0",
+ "--cache-prompt",
+ "--cache-ram",
+ "32768",
+ "--threads",
+ "6",
+ "--threads-batch",
+ "6",
+ "--batch-size",
+ "2048",
+ "--ubatch-size",
+ "512",
+ "--parallel",
+ "2",
+ "--kv-unified",
+ "--jinja",
+ "--reasoning",
+ "auto",
+ "--reasoning-preserve",
+ "--host",
+ "0.0.0.0",
+ "--port",
+ "8080",
+ "--metrics",
+ "--fit",
+ "off",
+ "--n-gpu-layers",
+ "all",
+ "--load-mode",
+ "none",
+ "--no-ui",
+ "--temperature",
+ "1.0",
+ "--top-p",
+ "0.95",
+ "--top-k",
+ "20",
+ "--device",
+ "CUDA0,CUDA1",
+ "--main-gpu",
+ "0",
+ "--split-mode",
+ "layer",
+ "--tensor-split",
+ "85,15",
+ "--spec-type",
+ "draft-mtp",
+ "--spec-draft-n-max",
+ "3",
+ "--spec-draft-type-k",
+ "f16",
+ "--spec-draft-type-v",
+ "f16",
+ "--spec-draft-p-min",
+ "0.05"
+ ]
+}
diff --git a/experiments/profile-mtp2-20260920/quick.json b/experiments/profile-mtp2-20260920/quick.json
new file mode 100644
index 0000000..b7832a7
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/quick.json
@@ -0,0 +1,35 @@
+[
+ {
+ "label": "medium-mtp2",
+ "profile": "medium",
+ "ubatch": 512,
+ "split": "85,15",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ {
+ "label": "large-mtp3",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 3,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ },
+ {
+ "label": "large-mtp2",
+ "profile": "large",
+ "ubatch": 256,
+ "split": "86,14",
+ "mtp": 2,
+ "prompts": [
+ 4096
+ ],
+ "minimum_headroom_mib": 448
+ }
+]
\ No newline at end of file
diff --git a/experiments/profile-mtp2-20260920/run.py b/experiments/profile-mtp2-20260920/run.py
new file mode 100644
index 0000000..e21e067
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/run.py
@@ -0,0 +1,171 @@
+#!/usr/bin/env python3
+"""Bounded, isolated Qwen quantization benchmark. Supervisor restores production."""
+import json, pathlib, subprocess, sys, time, urllib.request, threading, signal
+ROOT = pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
+NAME = 'mike-ai-profile-mtp2-test'
+BASE = 'http://127.0.0.1:5005'
+GPU0 = 'GPU-8ad38c6c-5a01-9d8e-1dfa-ed662ad78fbe'
+GPU1 = 'GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b'
+IMAGE = 'sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
+MODELS = {'mix':'qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf', 'pure':'qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf', 'byteshape':'byteshape-qwen38-gpu5/model.gguf'}
+
+def cmd(*args, check=True, timeout=90):
+ r = subprocess.run(args, capture_output=True, text=True, timeout=timeout)
+ if check and r.returncode: raise RuntimeError(str(args[:3])+': '+r.stderr[-2000:])
+ return r.stdout
+
+def api(path, data=None, timeout=900):
+ req = urllib.request.Request(BASE+path, data=None if data is None else json.dumps(data).encode(), headers={'Content-Type':'application/json'})
+ with urllib.request.urlopen(req, timeout=timeout) as r: return json.load(r)
+
+def save(path, data):
+ path.write_text(json.dumps(data, indent=2, ensure_ascii=False)+'\n')
+
+def gpu():
+ rows = cmd('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu,utilization.gpu,pcie.link.gen.current,pcie.link.width.current','--format=csv,noheader,nounits',timeout=15)
+ return [dict(zip(['name','used','total','temp','util','pcie_gen','pcie_width'], [v.strip() for v in row.split(',')])) for row in rows.splitlines()]
+
+def health_check():
+ rows=gpu()
+ if any(int(x['temp']) >= 85 for x in rows): raise RuntimeError('GPU temperature limit')
+ mem = dict((a.split(':')[0],int(a.split()[1])) for a in pathlib.Path('/proc/meminfo').read_text().splitlines())
+ if mem['MemAvailable'] < 3*1024*1024: raise RuntimeError('Host RAM reserve below 3 GiB')
+ return rows
+
+def chat(prompt, max_tokens=512, effort='none', seed=42, tools=None):
+ health_check()
+ p={'model':'qwen-medium','messages':[{'role':'user','content':prompt}], 'max_tokens':max_tokens,'temperature':1.0,'top_p':0.95,'top_k':20,'min_p':0.0,'seed':seed,'reasoning_effort':effort,'cache_prompt':False}
+ if tools: p.update(tools=tools,tool_choice='auto')
+ start=time.monotonic(); r=api('/v1/chat/completions',p); r['wall_seconds']=time.monotonic()-start
+ health_check()
+ return r
+
+def prefill(n, seed):
+ import gzip
+ payload=json.loads(gzip.decompress(pathlib.Path('/opt/mike-ai/stack/benchmarks/athena-qwen38-reference-20260920/frozen-requests.json.gz').read_bytes()))['prefill-'+str(n)]
+ r=chat(payload['messages'][0]['content'],payload['max_tokens'],seed=payload['seed'])
+ content=r['choices'][0]['message'].get('content','')
+ r['recall']={x:x in content for x in ['RAVEN-417','CEDAR-928','ORBIT-563']}
+ return r
+
+def run_case(case):
+ label=case['label']; out=ROOT/label; out.mkdir(exist_ok=True)
+ if (out/'result.json').exists(): raise RuntimeError('Refusing to overwrite completed case '+label)
+ save(out/'config.json',case)
+ print('START',label,flush=True)
+ single=case.get('single',False)
+ production=json.loads((ROOT/(case['profile']+'.json')).read_text())
+ assert production['image']==IMAGE
+ args=list(production['args'])
+ for flag,value in [('--ubatch-size',str(case['ubatch'])),('--tensor-split',case['split']),('--spec-draft-n-max',str(case['mtp'])),('--host','127.0.0.1'),('--port','5005')]:
+ args[args.index(flag)+1]=value
+ args += ['--verbosity','3']
+ save(out/'server-args.json',args)
+ cmd('docker','run','-d','--name',NAME,'--gpus','all','--network','host','--read-only','--tmpfs','/tmp:rw,nosuid,nodev,size=256m','--security-opt','no-new-privileges:true','--cap-drop','ALL','--pids-limit','512','--ulimit','core=0','--memory','26g','--memory-swap','26g','--shm-size','1g','--log-opt','max-size=32m','--log-opt','max-file=1','-e','NVIDIA_VISIBLE_DEVICES='+GPU0+','+GPU1,'-e','NVIDIA_DRIVER_CAPABILITIES=compute,utility','-v','/data/models:/models:ro',IMAGE,*args)
+ stop=threading.Event(); samples=[]
+ def monitor():
+ while not stop.wait(2):
+ try:
+ rows=health_check()
+ samples.append({'time':time.time(),'gpus':rows})
+ except RuntimeError as e:
+ samples.append({'error':str(e),'aborted':True})
+ cmd('docker','stop','-t','10',NAME,check=False)
+ return
+ except Exception as e: samples.append({'error':str(e)})
+ thread=threading.Thread(target=monitor,daemon=True); thread.start()
+ result={'case':case,'started':time.time()}
+ try:
+ for _ in range(150):
+ try:
+ if api('/health',timeout=3).get('status')=='ok': break
+ except Exception: pass
+ if cmd('docker','inspect',NAME,'--format','{{.State.Running}}').strip()!='true': raise RuntimeError('Test container exited during load')
+ time.sleep(2)
+ else: raise RuntimeError('Startup exceeded 300s')
+ result['idle_gpu']=health_check(); result['props']=api('/props'); result['slots']=api('/slots')
+ save(out/'loaded.json',result)
+ # Added after the 88:12 trial: model loading alone can succeed while
+ # the first real attention graph still needs more CUDA workspace.
+ minimum=case.get('minimum_headroom_mib',512)
+ used_devices=['5080'] if single else ['5080','3060']
+ for g in result['idle_gpu']:
+ if any(device in g['name'] for device in used_devices):
+ free=int(g['total'])-int(g['used'])
+ if free=262144: break
+ previous=context
+ context=min(262144,context+growth)
+ final={**case,'ctx':context,'label':case['label']+'-validated-'+str(context),'load_only':False,'prompts':[49152,context-1024]}
+ return run_case(final)
+
+if __name__=='__main__':
+ for case in json.loads(pathlib.Path(sys.argv[1]).read_text()):
+ if case.get('capacity_search'): capacity_case(case)
+ else: run_case(case)
diff --git a/experiments/profile-mtp2-20260920/supervise.py b/experiments/profile-mtp2-20260920/supervise.py
new file mode 100644
index 0000000..ed8eac9
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/supervise.py
@@ -0,0 +1,55 @@
+#!/usr/bin/env python3
+"""Stop only existing router/controller/model; always restore the same containers."""
+import json, pathlib, subprocess, sys, time, signal
+ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
+NAMES=['mike-ai-router','mike-ai-profile-controller','mike-ai-llama-medium']
+def run(*args,check=True,timeout=90):
+ return subprocess.run(args,capture_output=True,text=True,check=check,timeout=timeout)
+def stop_signal(*_): raise RuntimeError('Supervisor interrupted')
+signal.signal(signal.SIGTERM,stop_signal); signal.signal(signal.SIGINT,stop_signal)
+# Refuse if the known production state has changed, or if requests are active.
+for name in NAMES:
+ assert run('docker','inspect',name,'--format','{{.State.Running}}').stdout.strip()=='true',name
+for attempt in range(60):
+ slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
+ if not any(s['is_processing'] for s in slots): break
+ if attempt==0: print('WAIT production request active; no interruption',flush=True)
+ time.sleep(3)
+else: raise RuntimeError('Production remained busy for 180s; no services stopped')
+assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').stdout.strip(),'Existing experiment'
+production=json.loads(run('docker','inspect',NAMES[-1]).stdout)[0]
+(ROOT/'production.json').write_text(json.dumps({'image':production['Image'],'args':production['Args']},indent=2)+'\n')
+for profile in ['medium','large']:
+ d=json.loads(run('docker','inspect','mike-ai-llama-'+profile).stdout)[0]
+ (ROOT/(profile+'.json')).write_text(json.dumps({'image':d['Image'],'args':d['Args']},indent=2)+'\n')
+child=None
+try:
+ run('docker','stop','-t','30',*NAMES[:2])
+ # Drain requests already handed to the model, before unloading it.
+ for _ in range(120):
+ slots=json.loads(run('docker','exec',NAMES[-1],'curl','-fsS','http://127.0.0.1:8080/slots').stdout)
+ if not any(s['is_processing'] for s in slots): break
+ time.sleep(2)
+ else: raise RuntimeError('Model did not drain')
+ run('docker','stop','-t','30',NAMES[-1])
+ child=subprocess.Popen(['python3',str(ROOT/'run.py'),sys.argv[1]])
+ code=child.wait(timeout=600)
+ if code: raise RuntimeError('Benchmark failed: '+str(code))
+finally:
+ if child is not None and child.poll() is None:
+ child.terminate()
+ try: child.wait(timeout=20)
+ except subprocess.TimeoutExpired: child.kill(); child.wait(timeout=10)
+ run('docker','rm','-f','mike-ai-profile-mtp2-test',check=False)
+ run('docker','start',NAMES[-1])
+ for _ in range(150):
+ if run('docker','inspect',NAMES[-1],'--format','{{.State.Health.Status}}').stdout.strip()=='healthy': break
+ time.sleep(2)
+ else: raise RuntimeError('Restored Medium did not become healthy')
+ run('docker','start',NAMES[1],NAMES[0])
+ for _ in range(60):
+ statuses=[run('docker','inspect',name,'--format','{{.State.Health.Status}}').stdout.strip() for name in NAMES[:2]]
+ if all(status=='healthy' for status in statuses): break
+ time.sleep(2)
+ else: raise RuntimeError('Restored router/controller did not become healthy')
+ print('RESTORED existing medium/controller/router; all healthy',flush=True)
diff --git a/experiments/profile-mtp2-20260920/verify_restore.py b/experiments/profile-mtp2-20260920/verify_restore.py
new file mode 100644
index 0000000..03c9bd9
--- /dev/null
+++ b/experiments/profile-mtp2-20260920/verify_restore.py
@@ -0,0 +1,29 @@
+#!/usr/bin/env python3
+"""Read-only restoration verification plus a two-token model smoke request."""
+import json,pathlib,re,subprocess,time
+ROOT=pathlib.Path('/data/benchmarks/profile-mtp2-20260920')
+def run(*args):return subprocess.check_output(args,text=True,timeout=30)
+report={'checked_at':time.time(),'uptime':run('uptime').strip(),'containers':{}}
+for name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway','mike-ai-qwen3-tts']:
+ d=json.loads(run('docker','inspect',name))[0]
+ report['containers'][name]={'running':d['State']['Running'],'health':d['State'].get('Health',{}).get('Status'),'image':d['Image'],'started':d['State']['StartedAt']}
+ assert d['State']['Running'],name
+ if name in ['mike-ai-llama-medium','mike-ai-router','mike-ai-profile-controller','mike-ai-wireguard-gateway']:
+ assert d['State'].get('Health',{}).get('Status')=='healthy',name
+assert report['containers']['mike-ai-llama-medium']['image']=='sha256:5e3c12c145b8045e5731b44b6b97033f24b327ae3d4a3fa85ecdd159cc844907'
+assert not run('docker','ps','-q','--filter','name=^mike-ai-profile-mtp2-test$').strip()
+probe='import urllib.request,json; print(json.dumps({p:json.load(urllib.request.urlopen("http://127.0.0.1:8081"+p,timeout=20)) for p in ["/health","/ready"]}))'
+report['router']=json.loads(run('docker','exec','mike-ai-router','python','-c',probe))
+payload={'model':'qwen-medium','messages':[{'role':'user','content':'Antworte ausschließlich mit OK.'}],'reasoning_effort':'none','max_tokens':8,'temperature':0}
+r=json.loads(run('docker','exec','mike-ai-llama-medium','curl','-fsS','--max-time','20','-H','Content-Type: application/json','--data',json.dumps(payload),'http://127.0.0.1:8080/v1/chat/completions'))
+report['smoke']={'content':r['choices'][0]['message'].get('content',''),'usage':r.get('usage')}
+assert report['smoke']['content'].strip()=='OK',report['smoke']
+started=min(json.loads(p.read_text())['started'] for p in ROOT.glob('*/result.json'))
+journal=run('journalctl','-k','--since','@'+str(int(started)-60),'--no-pager')
+pattern=re.compile(r'NVRM.*Xid|oom-kill|Out of memory: Killed process|Kernel panic|GPU has fallen off',re.I)
+report['kernel_errors']=[line for line in journal.splitlines() if pattern.search(line)]
+report['gpu']=run('nvidia-smi','--query-gpu=name,memory.used,memory.total,temperature.gpu','--format=csv').strip()
+report['disk']=run('df','-h','/','/data').strip()
+(ROOT/'restore-verification.json').write_text(json.dumps(report,indent=2)+'\n')
+print(json.dumps(report,indent=2))
+assert not report['kernel_errors'],'Kernel/GPU errors require review'