Map reasoning effort aliases to supported llama.cpp values

This commit is contained in:
Mikei386
2026-09-29 21:51:57 +02:00
parent d08e194be7
commit fd2c21f208
2 changed files with 4 additions and 3 deletions
+3 -2
View File
@@ -4,8 +4,9 @@ Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
positive levels to the model template; it handles 'none' as thinking disabled.
"""
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'})
LLAMA_EFFORTS=frozenset({'none','minimal','low','medium','high','xhigh','max'})
EFFORT_ALIASES={'ultra':'max'}
LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'})
# The installed llama.cpp server rejects minimal and max. Use nearest supported hints.
EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'}
class CompatibilityError(ValueError):pass
+1 -1
View File
@@ -5,7 +5,7 @@ class CompatibilityTests(unittest.TestCase):
for value in ['none','minimal','low','medium','high','xhigh','max','ultra']:
request={'model':'any','reasoning_effort':value};body,info=normalize_chat(request)
self.assertEqual(request['reasoning_effort'],value)
self.assertEqual(body['reasoning_effort'],'max' if value=='ultra' else value)
self.assertEqual(body['reasoning_effort'],{'minimal':'low','max':'xhigh','ultra':'xhigh'}.get(value,value))
self.assertEqual(info['requested'],value)
def test_unset_and_null_do_not_invent_defaults(self):
for request in [{},{'reasoning_effort':None}]: