Make reasoning levels enforce real token budgets

This commit is contained in:
Mikei386
2026-09-01 13:55:11 +02:00
parent 5f793020b0
commit b3e86cc7ae
5 changed files with 52 additions and 21 deletions
+10
View File
@@ -59,6 +59,16 @@ versionierten Modellartefakte und startet ausschließlich den Athena-Kern.
sudo ./smoke-test.sh sudo ./smoke-test.sh
``` ```
### Reasoning-Stufen
Der Router übersetzt die Auswahl eines OpenAI-kompatiblen Clients in echte,
pro Anfrage geltende llama.cpp-Denkbudgets. `Off` deaktiviert Thinking; die
aktiven Stufen sind auf 256 (Minimal), 768 (Low), 2048 (Medium), 4096 (High)
und 8192 Tokens (XHigh/Max/Ultra) begrenzt. Die Modellserver dürfen deshalb
kein festes `--reasoning-budget` setzen, da dieses die dynamischen Budgets
von llama.cpp übersteuern würde. Clients, die direkt
`thinking_budget_tokens` senden, behalten ihren expliziten Wert.
### Ein oder zwei Modell-Slots ### Ein oder zwei Modell-Slots
Produktiv laufen alle Profile mit einem Slot. Damit erhält ein einzelner Chat Produktiv laufen alle Profile mit einem Slot. Damit erhält ein einzelner Chat
+2 -5
View File
@@ -194,11 +194,8 @@ services:
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
# Bound each individual thinking phase. Long agent jobs can still use # No fixed --reasoning-budget here: the router supplies a real budget
# many phases around tool calls, but one degenerate reasoning loop can # per request from the client's reasoning_effort selection.
# no longer consume the complete response budget indefinitely.
- --reasoning-budget
- "8192"
- --reasoning-preserve - --reasoning-preserve
- --host - --host
- 0.0.0.0 - 0.0.0.0
+22 -13
View File
@@ -140,26 +140,27 @@ class LlamaCppReasoningTests(unittest.TestCase):
normalized["chat_template_kwargs"], normalized["chat_template_kwargs"],
{"enable_thinking": False}, {"enable_thinking": False},
) )
self.assertEqual(normalized["thinking_budget_tokens"], 0)
def test_low_and_medium_reach_chat_template(self) -> None: def test_reasoning_levels_receive_real_per_request_budgets(self) -> None:
for effort in ("low", "medium"): expected = {
"minimal": ("low", 256),
"low": ("low", 768),
"medium": ("medium", 2048),
"high": ("xhigh", 4096),
"xhigh": ("xhigh", 8192),
"max": ("xhigh", 8192),
"ultra": ("xhigh", 8192),
}
for effort, (template_effort, budget) in expected.items():
with self.subTest(effort=effort): with self.subTest(effort=effort):
request = {"reasoning_effort": effort, "messages": []} request = {"reasoning_effort": effort, "messages": []}
normalized = _normalize_llamacpp_reasoning(request) normalized = _normalize_llamacpp_reasoning(request)
self.assertEqual(normalized["chat_template_kwargs"], { self.assertEqual(normalized["chat_template_kwargs"], {
"enable_thinking": True, "enable_thinking": True,
"reasoning_effort": effort, "reasoning_effort": template_effort,
}) })
self.assertEqual(normalized["thinking_budget_tokens"], budget)
def test_unsupported_high_levels_are_clamped_to_xhigh(self) -> None:
for effort in ("high", "xhigh", "max", "ultra"):
with self.subTest(effort=effort):
request = {"reasoning_effort": effort, "messages": []}
normalized = _normalize_llamacpp_reasoning(request)
self.assertEqual(
normalized["chat_template_kwargs"]["reasoning_effort"],
"xhigh",
)
def test_existing_template_kwargs_are_preserved(self) -> None: def test_existing_template_kwargs_are_preserved(self) -> None:
request = { request = {
@@ -173,6 +174,7 @@ class LlamaCppReasoningTests(unittest.TestCase):
"enable_thinking": True, "enable_thinking": True,
"reasoning_effort": "low", "reasoning_effort": "low",
}) })
self.assertEqual(normalized["thinking_budget_tokens"], 768)
def test_request_without_effort_uses_safe_off_default(self) -> None: def test_request_without_effort_uses_safe_off_default(self) -> None:
request = {"messages": []} request = {"messages": []}
@@ -181,6 +183,13 @@ class LlamaCppReasoningTests(unittest.TestCase):
request["chat_template_kwargs"], request["chat_template_kwargs"],
{"enable_thinking": False}, {"enable_thinking": False},
) )
self.assertEqual(request["thinking_budget_tokens"], 0)
def test_native_thinking_budget_is_preserved_without_openai_effort(self) -> None:
request = {"thinking_budget_tokens": 1234, "messages": []}
self.assertIs(_normalize_llamacpp_reasoning(request), request)
self.assertEqual(request["thinking_budget_tokens"], 1234)
self.assertNotIn("chat_template_kwargs", request)
class ChatGenerationLimitTests(unittest.TestCase): class ChatGenerationLimitTests(unittest.TestCase):
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
[Service] [Service]
ExecStart= ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
+17 -2
View File
@@ -1270,6 +1270,15 @@ _REASONING_EFFORT_MAP = {
"ultra": "xhigh", "ultra": "xhigh",
} }
_REASONING_OFF = {"", "none", "off", "disabled", "false"} _REASONING_OFF = {"", "none", "off", "disabled", "false"}
_REASONING_BUDGET_TOKENS = {
"minimal": 256,
"low": 768,
"medium": 2048,
"high": 4096,
"xhigh": 8192,
"max": 8192,
"ultra": 8192,
}
def _load_global_system_policy() -> str: def _load_global_system_policy() -> str:
@@ -1330,6 +1339,9 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
Top-Level-Feld. llama.cpp akzeptiert das Feld zwar, reicht es dort aber Top-Level-Feld. llama.cpp akzeptiert das Feld zwar, reicht es dort aber
nicht an das Jinja-Chat-Template weiter. Qwen3.8 erwartet stattdessen nicht an das Jinja-Chat-Template weiter. Qwen3.8 erwartet stattdessen
``chat_template_kwargs.reasoning_effort`` bzw. ``enable_thinking=false``. ``chat_template_kwargs.reasoning_effort`` bzw. ``enable_thinking=false``.
Zusätzlich erhält llama.cpp mit ``thinking_budget_tokens`` eine echte,
pro Request geltende Obergrenze. Dadurch sind die in Hermes sichtbaren
Stufen nicht bloß unterschiedlich formulierte Template-Hinweise.
Die Funktion verändert den übergebenen Request absichtlich in-place und Die Funktion verändert den übergebenen Request absichtlich in-place und
entfernt das wirkungslose Top-Level-Feld. Andere Template-Argumente des entfernt das wirkungslose Top-Level-Feld. Andere Template-Argumente des
@@ -1345,9 +1357,10 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
# set to Off, so the safe default must remain Off; enabled levels are # set to Off, so the safe default must remain Off; enabled levels are
# sent explicitly by Hermes and other capable clients. # sent explicitly by Hermes and other capable clients.
existing_kwargs = data.get("chat_template_kwargs") existing_kwargs = data.get("chat_template_kwargs")
if isinstance(existing_kwargs, dict) and ( if "thinking_budget_tokens" in data or (
isinstance(existing_kwargs, dict) and (
"reasoning_effort" in existing_kwargs "reasoning_effort" in existing_kwargs
or "enable_thinking" in existing_kwargs): or "enable_thinking" in existing_kwargs)):
return data return data
raw_effort = DEFAULT_REASONING_EFFORT raw_effort = DEFAULT_REASONING_EFFORT
effort = str(raw_effort).strip().lower() if raw_effort is not None else "" effort = str(raw_effort).strip().lower() if raw_effort is not None else ""
@@ -1359,6 +1372,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
if effort in _REASONING_OFF: if effort in _REASONING_OFF:
template_kwargs.pop("reasoning_effort", None) template_kwargs.pop("reasoning_effort", None)
template_kwargs["enable_thinking"] = False template_kwargs["enable_thinking"] = False
data["thinking_budget_tokens"] = 0
return data return data
mapped = _REASONING_EFFORT_MAP.get(effort) mapped = _REASONING_EFFORT_MAP.get(effort)
@@ -1373,6 +1387,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
template_kwargs["enable_thinking"] = True template_kwargs["enable_thinking"] = True
template_kwargs["reasoning_effort"] = mapped template_kwargs["reasoning_effort"] = mapped
data["thinking_budget_tokens"] = _REASONING_BUDGET_TOKENS[effort]
return data return data