Make reasoning levels enforce real token budgets

This commit is contained in:
Mikei386
2026-09-01 13:55:11 +02:00
parent 5f793020b0
commit b3e86cc7ae
5 changed files with 52 additions and 21 deletions
+22 -13
View File
@@ -140,26 +140,27 @@ class LlamaCppReasoningTests(unittest.TestCase):
normalized["chat_template_kwargs"],
{"enable_thinking": False},
)
self.assertEqual(normalized["thinking_budget_tokens"], 0)
def test_low_and_medium_reach_chat_template(self) -> None:
for effort in ("low", "medium"):
def test_reasoning_levels_receive_real_per_request_budgets(self) -> None:
expected = {
"minimal": ("low", 256),
"low": ("low", 768),
"medium": ("medium", 2048),
"high": ("xhigh", 4096),
"xhigh": ("xhigh", 8192),
"max": ("xhigh", 8192),
"ultra": ("xhigh", 8192),
}
for effort, (template_effort, budget) in expected.items():
with self.subTest(effort=effort):
request = {"reasoning_effort": effort, "messages": []}
normalized = _normalize_llamacpp_reasoning(request)
self.assertEqual(normalized["chat_template_kwargs"], {
"enable_thinking": True,
"reasoning_effort": effort,
"reasoning_effort": template_effort,
})
def test_unsupported_high_levels_are_clamped_to_xhigh(self) -> None:
for effort in ("high", "xhigh", "max", "ultra"):
with self.subTest(effort=effort):
request = {"reasoning_effort": effort, "messages": []}
normalized = _normalize_llamacpp_reasoning(request)
self.assertEqual(
normalized["chat_template_kwargs"]["reasoning_effort"],
"xhigh",
)
self.assertEqual(normalized["thinking_budget_tokens"], budget)
def test_existing_template_kwargs_are_preserved(self) -> None:
request = {
@@ -173,6 +174,7 @@ class LlamaCppReasoningTests(unittest.TestCase):
"enable_thinking": True,
"reasoning_effort": "low",
})
self.assertEqual(normalized["thinking_budget_tokens"], 768)
def test_request_without_effort_uses_safe_off_default(self) -> None:
request = {"messages": []}
@@ -181,6 +183,13 @@ class LlamaCppReasoningTests(unittest.TestCase):
request["chat_template_kwargs"],
{"enable_thinking": False},
)
self.assertEqual(request["thinking_budget_tokens"], 0)
def test_native_thinking_budget_is_preserved_without_openai_effort(self) -> None:
request = {"thinking_budget_tokens": 1234, "messages": []}
self.assertIs(_normalize_llamacpp_reasoning(request), request)
self.assertEqual(request["thinking_budget_tokens"], 1234)
self.assertNotIn("chat_template_kwargs", request)
class ChatGenerationLimitTests(unittest.TestCase):