Map Hermes reasoning controls into Qwen templates
This commit is contained in:
@@ -156,6 +156,25 @@ assert "Mock-Antwort" in d["choices"][0]["message"]["content"], d
|
||||
assert d.get("mock_authorization") is None, d
|
||||
' && ok "Request wurde weitergeleitet, Modell ersetzt" || bad "Forwarding"
|
||||
|
||||
echo "== Test 3b: Hermes-Reasoning erreicht das llama.cpp-Chat-Template"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-fast","reasoning_effort":"none","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d.get("mock_reasoning_effort") is None, d
|
||||
assert d.get("mock_chat_template_kwargs") == {"enable_thinking": False}, d
|
||||
' && ok "none wird als enable_thinking=false weitergegeben" || bad "Reasoning none"
|
||||
RESP=$(curl -sf "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
-d '{"model":"qwen-fast","reasoning_effort":"medium","messages":[{"role":"user","content":"Hallo"}]}')
|
||||
echo "$RESP" | python3 -c '
|
||||
import json,sys
|
||||
d=json.load(sys.stdin)
|
||||
assert d.get("mock_reasoning_effort") is None, d
|
||||
assert d.get("mock_chat_template_kwargs") == {
|
||||
"enable_thinking": True, "reasoning_effort": "medium"}, d
|
||||
' && ok "medium erreicht chat_template_kwargs" || bad "Reasoning medium"
|
||||
|
||||
# --- 4. Streaming ----------------------------------------------------------------
|
||||
echo "== Test 4: Streaming (SSE)"
|
||||
RESP=$(curl -sfN "$BASE/v1/chat/completions" -H "Content-Type: application/json" \
|
||||
|
||||
Reference in New Issue
Block a user