From b3e86cc7ae69afc216620ca9a4ff8fedb18284b4 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Tue, 1 Sep 2026 13:55:11 +0200 Subject: [PATCH] Make reasoning levels enforce real token budgets --- README.md | 10 ++++++++ compose.yaml | 7 ++---- dev/test_router_support.py | 35 +++++++++++++++++---------- platform/profiles/profile-medium.conf | 2 +- router/ai_profile_router.py | 19 +++++++++++++-- 5 files changed, 52 insertions(+), 21 deletions(-) diff --git a/README.md b/README.md index a84b416..637f129 100644 --- a/README.md +++ b/README.md @@ -59,6 +59,16 @@ versionierten Modellartefakte und startet ausschließlich den Athena-Kern. sudo ./smoke-test.sh ``` +### Reasoning-Stufen + +Der Router übersetzt die Auswahl eines OpenAI-kompatiblen Clients in echte, +pro Anfrage geltende llama.cpp-Denkbudgets. `Off` deaktiviert Thinking; die +aktiven Stufen sind auf 256 (Minimal), 768 (Low), 2048 (Medium), 4096 (High) +und 8192 Tokens (XHigh/Max/Ultra) begrenzt. Die Modellserver dürfen deshalb +kein festes `--reasoning-budget` setzen, da dieses die dynamischen Budgets +von llama.cpp übersteuern würde. Clients, die direkt +`thinking_budget_tokens` senden, behalten ihren expliziten Wert. + ### Ein oder zwei Modell-Slots Produktiv laufen alle Profile mit einem Slot. Damit erhält ein einzelner Chat diff --git a/compose.yaml b/compose.yaml index 96e0bab..83443ea 100644 --- a/compose.yaml +++ b/compose.yaml @@ -194,11 +194,8 @@ services: - --jinja - --reasoning - auto - # Bound each individual thinking phase. Long agent jobs can still use - # many phases around tool calls, but one degenerate reasoning loop can - # no longer consume the complete response budget indefinitely. - - --reasoning-budget - - "8192" + # No fixed --reasoning-budget here: the router supplies a real budget + # per request from the client's reasoning_effort selection. - --reasoning-preserve - --host - 0.0.0.0 diff --git a/dev/test_router_support.py b/dev/test_router_support.py index 2896cf7..1669a7e 100644 --- a/dev/test_router_support.py +++ b/dev/test_router_support.py @@ -140,26 +140,27 @@ class LlamaCppReasoningTests(unittest.TestCase): normalized["chat_template_kwargs"], {"enable_thinking": False}, ) + self.assertEqual(normalized["thinking_budget_tokens"], 0) - def test_low_and_medium_reach_chat_template(self) -> None: - for effort in ("low", "medium"): + def test_reasoning_levels_receive_real_per_request_budgets(self) -> None: + expected = { + "minimal": ("low", 256), + "low": ("low", 768), + "medium": ("medium", 2048), + "high": ("xhigh", 4096), + "xhigh": ("xhigh", 8192), + "max": ("xhigh", 8192), + "ultra": ("xhigh", 8192), + } + for effort, (template_effort, budget) in expected.items(): with self.subTest(effort=effort): request = {"reasoning_effort": effort, "messages": []} normalized = _normalize_llamacpp_reasoning(request) self.assertEqual(normalized["chat_template_kwargs"], { "enable_thinking": True, - "reasoning_effort": effort, + "reasoning_effort": template_effort, }) - - def test_unsupported_high_levels_are_clamped_to_xhigh(self) -> None: - for effort in ("high", "xhigh", "max", "ultra"): - with self.subTest(effort=effort): - request = {"reasoning_effort": effort, "messages": []} - normalized = _normalize_llamacpp_reasoning(request) - self.assertEqual( - normalized["chat_template_kwargs"]["reasoning_effort"], - "xhigh", - ) + self.assertEqual(normalized["thinking_budget_tokens"], budget) def test_existing_template_kwargs_are_preserved(self) -> None: request = { @@ -173,6 +174,7 @@ class LlamaCppReasoningTests(unittest.TestCase): "enable_thinking": True, "reasoning_effort": "low", }) + self.assertEqual(normalized["thinking_budget_tokens"], 768) def test_request_without_effort_uses_safe_off_default(self) -> None: request = {"messages": []} @@ -181,6 +183,13 @@ class LlamaCppReasoningTests(unittest.TestCase): request["chat_template_kwargs"], {"enable_thinking": False}, ) + self.assertEqual(request["thinking_budget_tokens"], 0) + + def test_native_thinking_budget_is_preserved_without_openai_effort(self) -> None: + request = {"thinking_budget_tokens": 1234, "messages": []} + self.assertIs(_normalize_llamacpp_reasoning(request), request) + self.assertEqual(request["thinking_budget_tokens"], 1234) + self.assertNotIn("chat_template_kwargs", request) class ChatGenerationLimitTests(unittest.TestCase): diff --git a/platform/profiles/profile-medium.conf b/platform/profiles/profile-medium.conf index b9f7a89..bb2ed29 100644 --- a/platform/profiles/profile-medium.conf +++ b/platform/profiles/profile-medium.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index f247bbb..b515c34 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -1270,6 +1270,15 @@ _REASONING_EFFORT_MAP = { "ultra": "xhigh", } _REASONING_OFF = {"", "none", "off", "disabled", "false"} +_REASONING_BUDGET_TOKENS = { + "minimal": 256, + "low": 768, + "medium": 2048, + "high": 4096, + "xhigh": 8192, + "max": 8192, + "ultra": 8192, +} def _load_global_system_policy() -> str: @@ -1330,6 +1339,9 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict: Top-Level-Feld. llama.cpp akzeptiert das Feld zwar, reicht es dort aber nicht an das Jinja-Chat-Template weiter. Qwen3.8 erwartet stattdessen ``chat_template_kwargs.reasoning_effort`` bzw. ``enable_thinking=false``. + Zusätzlich erhält llama.cpp mit ``thinking_budget_tokens`` eine echte, + pro Request geltende Obergrenze. Dadurch sind die in Hermes sichtbaren + Stufen nicht bloß unterschiedlich formulierte Template-Hinweise. Die Funktion verändert den übergebenen Request absichtlich in-place und entfernt das wirkungslose Top-Level-Feld. Andere Template-Argumente des @@ -1345,9 +1357,10 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict: # set to Off, so the safe default must remain Off; enabled levels are # sent explicitly by Hermes and other capable clients. existing_kwargs = data.get("chat_template_kwargs") - if isinstance(existing_kwargs, dict) and ( + if "thinking_budget_tokens" in data or ( + isinstance(existing_kwargs, dict) and ( "reasoning_effort" in existing_kwargs - or "enable_thinking" in existing_kwargs): + or "enable_thinking" in existing_kwargs)): return data raw_effort = DEFAULT_REASONING_EFFORT effort = str(raw_effort).strip().lower() if raw_effort is not None else "" @@ -1359,6 +1372,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict: if effort in _REASONING_OFF: template_kwargs.pop("reasoning_effort", None) template_kwargs["enable_thinking"] = False + data["thinking_budget_tokens"] = 0 return data mapped = _REASONING_EFFORT_MAP.get(effort) @@ -1373,6 +1387,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict: template_kwargs["enable_thinking"] = True template_kwargs["reasoning_effort"] = mapped + data["thinking_budget_tokens"] = _REASONING_BUDGET_TOKENS[effort] return data