Make reasoning levels enforce real token budgets
This commit is contained in:
@@ -1270,6 +1270,15 @@ _REASONING_EFFORT_MAP = {
|
||||
"ultra": "xhigh",
|
||||
}
|
||||
_REASONING_OFF = {"", "none", "off", "disabled", "false"}
|
||||
_REASONING_BUDGET_TOKENS = {
|
||||
"minimal": 256,
|
||||
"low": 768,
|
||||
"medium": 2048,
|
||||
"high": 4096,
|
||||
"xhigh": 8192,
|
||||
"max": 8192,
|
||||
"ultra": 8192,
|
||||
}
|
||||
|
||||
|
||||
def _load_global_system_policy() -> str:
|
||||
@@ -1330,6 +1339,9 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
Top-Level-Feld. llama.cpp akzeptiert das Feld zwar, reicht es dort aber
|
||||
nicht an das Jinja-Chat-Template weiter. Qwen3.8 erwartet stattdessen
|
||||
``chat_template_kwargs.reasoning_effort`` bzw. ``enable_thinking=false``.
|
||||
Zusätzlich erhält llama.cpp mit ``thinking_budget_tokens`` eine echte,
|
||||
pro Request geltende Obergrenze. Dadurch sind die in Hermes sichtbaren
|
||||
Stufen nicht bloß unterschiedlich formulierte Template-Hinweise.
|
||||
|
||||
Die Funktion verändert den übergebenen Request absichtlich in-place und
|
||||
entfernt das wirkungslose Top-Level-Feld. Andere Template-Argumente des
|
||||
@@ -1345,9 +1357,10 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
# set to Off, so the safe default must remain Off; enabled levels are
|
||||
# sent explicitly by Hermes and other capable clients.
|
||||
existing_kwargs = data.get("chat_template_kwargs")
|
||||
if isinstance(existing_kwargs, dict) and (
|
||||
if "thinking_budget_tokens" in data or (
|
||||
isinstance(existing_kwargs, dict) and (
|
||||
"reasoning_effort" in existing_kwargs
|
||||
or "enable_thinking" in existing_kwargs):
|
||||
or "enable_thinking" in existing_kwargs)):
|
||||
return data
|
||||
raw_effort = DEFAULT_REASONING_EFFORT
|
||||
effort = str(raw_effort).strip().lower() if raw_effort is not None else ""
|
||||
@@ -1359,6 +1372,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
if effort in _REASONING_OFF:
|
||||
template_kwargs.pop("reasoning_effort", None)
|
||||
template_kwargs["enable_thinking"] = False
|
||||
data["thinking_budget_tokens"] = 0
|
||||
return data
|
||||
|
||||
mapped = _REASONING_EFFORT_MAP.get(effort)
|
||||
@@ -1373,6 +1387,7 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
|
||||
template_kwargs["enable_thinking"] = True
|
||||
template_kwargs["reasoning_effort"] = mapped
|
||||
data["thinking_budget_tokens"] = _REASONING_BUDGET_TOKENS[effort]
|
||||
return data
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user