Make reasoning levels enforce real token budgets
This commit is contained in:
@@ -59,6 +59,16 @@ versionierten Modellartefakte und startet ausschließlich den Athena-Kern.
|
||||
sudo ./smoke-test.sh
|
||||
```
|
||||
|
||||
### Reasoning-Stufen
|
||||
|
||||
Der Router übersetzt die Auswahl eines OpenAI-kompatiblen Clients in echte,
|
||||
pro Anfrage geltende llama.cpp-Denkbudgets. `Off` deaktiviert Thinking; die
|
||||
aktiven Stufen sind auf 256 (Minimal), 768 (Low), 2048 (Medium), 4096 (High)
|
||||
und 8192 Tokens (XHigh/Max/Ultra) begrenzt. Die Modellserver dürfen deshalb
|
||||
kein festes `--reasoning-budget` setzen, da dieses die dynamischen Budgets
|
||||
von llama.cpp übersteuern würde. Clients, die direkt
|
||||
`thinking_budget_tokens` senden, behalten ihren expliziten Wert.
|
||||
|
||||
### Ein oder zwei Modell-Slots
|
||||
|
||||
Produktiv laufen alle Profile mit einem Slot. Damit erhält ein einzelner Chat
|
||||
|
||||
+2
-5
@@ -194,11 +194,8 @@ services:
|
||||
- --jinja
|
||||
- --reasoning
|
||||
- auto
|
||||
# Bound each individual thinking phase. Long agent jobs can still use
|
||||
# many phases around tool calls, but one degenerate reasoning loop can
|
||||
# no longer consume the complete response budget indefinitely.
|
||||
- --reasoning-budget
|
||||
- "8192"
|
||||
# No fixed --reasoning-budget here: the router supplies a real budget
|
||||
# per request from the client's reasoning_effort selection.
|
||||
- --reasoning-preserve
|
||||
- --host
|
||||
- 0.0.0.0
|
||||
|
||||
+22
-13
@@ -140,26 +140,27 @@ class LlamaCppReasoningTests(unittest.TestCase):
|
||||
normalized["chat_template_kwargs"],
|
||||
{"enable_thinking": False},
|
||||
)
|
||||
self.assertEqual(normalized["thinking_budget_tokens"], 0)
|
||||
|
||||
def test_low_and_medium_reach_chat_template(self) -> None:
|
||||
for effort in ("low", "medium"):
|
||||
def test_reasoning_levels_receive_real_per_request_budgets(self) -> None:
|
||||
expected = {
|
||||
"minimal": ("low", 256),
|
||||
"low": ("low", 768),
|
||||
"medium": ("medium", 2048),
|
||||
"high": ("xhigh", 4096),
|
||||
"xhigh": ("xhigh", 8192),
|
||||
"max": ("xhigh", 8192),
|
||||
"ultra": ("xhigh", 8192),
|
||||
}
|
||||
for effort, (template_effort, budget) in expected.items():
|
||||
with self.subTest(effort=effort):
|
||||
request = {"reasoning_effort": effort, "messages": []}
|
||||
normalized = _normalize_llamacpp_reasoning(request)
|
||||
self.assertEqual(normalized["chat_template_kwargs"], {
|
||||
"enable_thinking": True,
|
||||
"reasoning_effort": effort,
|
||||
"reasoning_effort": template_effort,
|
||||
})
|
||||
|
||||
def test_unsupported_high_levels_are_clamped_to_xhigh(self) -> None:
|
||||
for effort in ("high", "xhigh", "max", "ultra"):
|
||||
with self.subTest(effort=effort):
|
||||
request = {"reasoning_effort": effort, "messages": []}
|
||||
normalized = _normalize_llamacpp_reasoning(request)
|
||||
self.assertEqual(
|
||||
normalized["chat_template_kwargs"]["reasoning_effort"],
|
||||
"xhigh",
|
||||
)
|
||||
self.assertEqual(normalized["thinking_budget_tokens"], budget)
|
||||
|
||||
def test_existing_template_kwargs_are_preserved(self) -> None:
|
||||
request = {
|
||||
@@ -173,6 +174,7 @@ class LlamaCppReasoningTests(unittest.TestCase):
|
||||
"enable_thinking": True,
|
||||
"reasoning_effort": "low",
|
||||
})
|
||||
self.assertEqual(normalized["thinking_budget_tokens"], 768)
|
||||
|
||||
def test_request_without_effort_uses_safe_off_default(self) -> None:
|
||||
request = {"messages": []}
|
||||
@@ -181,6 +183,13 @@ class LlamaCppReasoningTests(unittest.TestCase):
|
||||
request["chat_template_kwargs"],
|
||||
{"enable_thinking": False},
|
||||
)
|
||||
self.assertEqual(request["thinking_budget_tokens"], 0)
|
||||
|
||||
def test_native_thinking_budget_is_preserved_without_openai_effort(self) -> None:
|
||||
request = {"thinking_budget_tokens": 1234, "messages": []}
|
||||
self.assertIs(_normalize_llamacpp_reasoning(request), request)
|
||||
self.assertEqual(request["thinking_budget_tokens"], 1234)
|
||||
self.assertNotIn("chat_template_kwargs", request)
|
||||
|
||||
|
||||
class ChatGenerationLimitTests(unittest.TestCase):
|
||||
|
||||
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --cache-prompt --cache-ram 8192 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
|
||||
@@ -1270,6 +1270,15 @@ _REASONING_EFFORT_MAP = {
|
||||
"ultra": "xhigh",
|
||||
}
|
||||
_REASONING_OFF = {"", "none", "off", "disabled", "false"}
|
||||
_REASONING_BUDGET_TOKENS = {
|
||||
"minimal": 256,
|
||||
"low": 768,
|
||||
"medium": 2048,
|
||||
"high": 4096,
|
||||
"xhigh": 8192,
|
||||
"max": 8192,
|
||||
"ultra": 8192,
|
||||
}
|
||||
|
||||
|
||||
def _load_global_system_policy() -> str:
|
||||
@@ -1330,6 +1339,9 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
Top-Level-Feld. llama.cpp akzeptiert das Feld zwar, reicht es dort aber
|
||||
nicht an das Jinja-Chat-Template weiter. Qwen3.8 erwartet stattdessen
|
||||
``chat_template_kwargs.reasoning_effort`` bzw. ``enable_thinking=false``.
|
||||
Zusätzlich erhält llama.cpp mit ``thinking_budget_tokens`` eine echte,
|
||||
pro Request geltende Obergrenze. Dadurch sind die in Hermes sichtbaren
|
||||
Stufen nicht bloß unterschiedlich formulierte Template-Hinweise.
|
||||
|
||||
Die Funktion verändert den übergebenen Request absichtlich in-place und
|
||||
entfernt das wirkungslose Top-Level-Feld. Andere Template-Argumente des
|
||||
@@ -1345,9 +1357,10 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
# set to Off, so the safe default must remain Off; enabled levels are
|
||||
# sent explicitly by Hermes and other capable clients.
|
||||
existing_kwargs = data.get("chat_template_kwargs")
|
||||
if isinstance(existing_kwargs, dict) and (
|
||||
if "thinking_budget_tokens" in data or (
|
||||
isinstance(existing_kwargs, dict) and (
|
||||
"reasoning_effort" in existing_kwargs
|
||||
or "enable_thinking" in existing_kwargs):
|
||||
or "enable_thinking" in existing_kwargs)):
|
||||
return data
|
||||
raw_effort = DEFAULT_REASONING_EFFORT
|
||||
effort = str(raw_effort).strip().lower() if raw_effort is not None else ""
|
||||
@@ -1359,6 +1372,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
if effort in _REASONING_OFF:
|
||||
template_kwargs.pop("reasoning_effort", None)
|
||||
template_kwargs["enable_thinking"] = False
|
||||
data["thinking_budget_tokens"] = 0
|
||||
return data
|
||||
|
||||
mapped = _REASONING_EFFORT_MAP.get(effort)
|
||||
@@ -1373,6 +1387,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
|
||||
template_kwargs["enable_thinking"] = True
|
||||
template_kwargs["reasoning_effort"] = mapped
|
||||
data["thinking_budget_tokens"] = _REASONING_BUDGET_TOKENS[effort]
|
||||
return data
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user