Map Hermes reasoning controls into Qwen templates
This commit is contained in:
@@ -89,6 +89,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"total_tokens": 15},
|
||||
"mock_ctx": current_ctx(),
|
||||
"mock_authorization": self.headers.get("Authorization"),
|
||||
"mock_reasoning_effort": body.get("reasoning_effort"),
|
||||
"mock_chat_template_kwargs": body.get("chat_template_kwargs"),
|
||||
}
|
||||
|
||||
def _stream(self, body: dict) -> None:
|
||||
|
||||
@@ -156,6 +156,25 @@ assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
|
||||
assert d.get("mock_authorization") is None, d
|
||||
' && ok "Request wurde weitergeleitet, Modell ersetzt" || bad "Forwarding"
|
||||
|
||||
echo "== Test 3b: Hermes-Reasoning erreicht das llama.cpp-Chat-Template"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-fast","reasoning_effort":"none","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d.get("mock_reasoning_effort") is None, d
|
||||
assert d.get("mock_chat_template_kwargs") == {"enable_thinking": False}, d
|
||||
' && ok "none wird als enable_thinking=false weitergegeben" || bad "Reasoning none"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-fast","reasoning_effort":"medium","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d.get("mock_reasoning_effort") is None, d
|
||||
assert d.get("mock_chat_template_kwargs") == {
|
||||
"enable_thinking": True, "reasoning_effort": "medium"}, d
|
||||
' && ok "medium erreicht chat_template_kwargs" || bad "Reasoning medium"
|
||||
|
||||
# --- 4. Streaming ----------------------------------------------------------------
|
||||
echo "== Test 4: Streaming (SSE)"
|
||||
RESP=$(curl -sfN "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
|
||||
@@ -27,6 +27,7 @@ from ai_profile_router import ( # noqa: E402
|
||||
_context_matches,
|
||||
_normalize_chat_image,
|
||||
_normalize_chat_images,
|
||||
_normalize_llamacpp_reasoning,
|
||||
_request_has_image,
|
||||
)
|
||||
|
||||
@@ -126,6 +127,55 @@ class ChatImageInputTests(unittest.TestCase):
|
||||
self.assertEqual(request, normalized)
|
||||
|
||||
|
||||
class LlamaCppReasoningTests(unittest.TestCase):
|
||||
def test_none_really_disables_thinking(self) -> None:
|
||||
request = {"reasoning_effort": "none", "messages": []}
|
||||
normalized = _normalize_llamacpp_reasoning(request)
|
||||
self.assertNotIn("reasoning_effort", normalized)
|
||||
self.assertEqual(
|
||||
normalized["chat_template_kwargs"],
|
||||
{"enable_thinking": False},
|
||||
)
|
||||
|
||||
def test_low_and_medium_reach_chat_template(self) -> None:
|
||||
for effort in ("low", "medium"):
|
||||
with self.subTest(effort=effort):
|
||||
request = {"reasoning_effort": effort, "messages": []}
|
||||
normalized = _normalize_llamacpp_reasoning(request)
|
||||
self.assertEqual(normalized["chat_template_kwargs"], {
|
||||
"enable_thinking": True,
|
||||
"reasoning_effort": effort,
|
||||
})
|
||||
|
||||
def test_unsupported_high_levels_are_clamped_to_xhigh(self) -> None:
|
||||
for effort in ("high", "xhigh", "max", "ultra"):
|
||||
with self.subTest(effort=effort):
|
||||
request = {"reasoning_effort": effort, "messages": []}
|
||||
normalized = _normalize_llamacpp_reasoning(request)
|
||||
self.assertEqual(
|
||||
normalized["chat_template_kwargs"]["reasoning_effort"],
|
||||
"xhigh",
|
||||
)
|
||||
|
||||
def test_existing_template_kwargs_are_preserved(self) -> None:
|
||||
request = {
|
||||
"reasoning_effort": "low",
|
||||
"chat_template_kwargs": {"preserve_thinking": True},
|
||||
"messages": [],
|
||||
}
|
||||
normalized = _normalize_llamacpp_reasoning(request)
|
||||
self.assertEqual(normalized["chat_template_kwargs"], {
|
||||
"preserve_thinking": True,
|
||||
"enable_thinking": True,
|
||||
"reasoning_effort": "low",
|
||||
})
|
||||
|
||||
def test_request_without_effort_is_unchanged(self) -> None:
|
||||
request = {"messages": []}
|
||||
self.assertIs(_normalize_llamacpp_reasoning(request), request)
|
||||
self.assertNotIn("chat_template_kwargs", request)
|
||||
|
||||
|
||||
class RetentionTests(unittest.TestCase):
|
||||
def test_oldest_pairs_are_removed(self) -> None:
|
||||
with tempfile.TemporaryDirectory() as temp:
|
||||
|
||||
Reference in New Issue
Block a user