Map Hermes reasoning controls into Qwen templates

This commit is contained in:
Mikei386
2026-08-25 14:52:43 +02:00
parent 7df28d9750
commit 63fc921988
6 changed files with 149 additions and 2 deletions
+2
View File
@@ -89,6 +89,8 @@ class Handler(BaseHTTPRequestHandler):
"total_tokens": 15},
"mock_ctx": current_ctx(),
"mock_authorization": self.headers.get("Authorization"),
"mock_reasoning_effort": body.get("reasoning_effort"),
"mock_chat_template_kwargs": body.get("chat_template_kwargs"),
}
def _stream(self, body: dict) -> None:
+19
View File
@@ -156,6 +156,25 @@ assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
assert d.get("mock_authorization") is None, d
' && ok "Request wurde weitergeleitet, Modell ersetzt" || bad "Forwarding"
echo "== Test 3b: Hermes-Reasoning erreicht das llama.cpp-Chat-Template"
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
-d '{"model":"qwen-fast","reasoning_effort":"none","messages":[{"role":"user","content":"Hallo"}]}')
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d.get("mock_reasoning_effort") is None, d
assert d.get("mock_chat_template_kwargs") == {"enable_thinking": False}, d
' && ok "none wird als enable_thinking=false weitergegeben" || bad "Reasoning none"
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
-d '{"model":"qwen-fast","reasoning_effort":"medium","messages":[{"role":"user","content":"Hallo"}]}')
echo "$RESP" | python3 -c '
import json,sys
d=json.load(sys.stdin)
assert d.get("mock_reasoning_effort") is None, d
assert d.get("mock_chat_template_kwargs") == {
"enable_thinking": True, "reasoning_effort": "medium"}, d
' && ok "medium erreicht chat_template_kwargs" || bad "Reasoning medium"
# --- 4. Streaming ----------------------------------------------------------------
echo "== Test 4: Streaming (SSE)"
RESP=$(curl -sfN "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
+50
View File
@@ -27,6 +27,7 @@ from ai_profile_router import ( # noqa: E402
_context_matches,
_normalize_chat_image,
_normalize_chat_images,
_normalize_llamacpp_reasoning,
_request_has_image,
)
@@ -126,6 +127,55 @@ class ChatImageInputTests(unittest.TestCase):
self.assertEqual(request, normalized)
class LlamaCppReasoningTests(unittest.TestCase):
def test_none_really_disables_thinking(self) -> None:
request = {"reasoning_effort": "none", "messages": []}
normalized = _normalize_llamacpp_reasoning(request)
self.assertNotIn("reasoning_effort", normalized)
self.assertEqual(
normalized["chat_template_kwargs"],
{"enable_thinking": False},
)
def test_low_and_medium_reach_chat_template(self) -> None:
for effort in ("low", "medium"):
with self.subTest(effort=effort):
request = {"reasoning_effort": effort, "messages": []}
normalized = _normalize_llamacpp_reasoning(request)
self.assertEqual(normalized["chat_template_kwargs"], {
"enable_thinking": True,
"reasoning_effort": effort,
})
def test_unsupported_high_levels_are_clamped_to_xhigh(self) -> None:
for effort in ("high", "xhigh", "max", "ultra"):
with self.subTest(effort=effort):
request = {"reasoning_effort": effort, "messages": []}
normalized = _normalize_llamacpp_reasoning(request)
self.assertEqual(
normalized["chat_template_kwargs"]["reasoning_effort"],
"xhigh",
)
def test_existing_template_kwargs_are_preserved(self) -> None:
request = {
"reasoning_effort": "low",
"chat_template_kwargs": {"preserve_thinking": True},
"messages": [],
}
normalized = _normalize_llamacpp_reasoning(request)
self.assertEqual(normalized["chat_template_kwargs"], {
"preserve_thinking": True,
"enable_thinking": True,
"reasoning_effort": "low",
})
def test_request_without_effort_is_unchanged(self) -> None:
request = {"messages": []}
self.assertIs(_normalize_llamacpp_reasoning(request), request)
self.assertNotIn("chat_template_kwargs", request)
class RetentionTests(unittest.TestCase):
def test_oldest_pairs_are_removed(self) -> None:
with tempfile.TemporaryDirectory() as temp: