diff --git a/api_compat.py b/api_compat.py index 9fde386..47e6c94 100644 --- a/api_compat.py +++ b/api_compat.py @@ -4,8 +4,9 @@ Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards positive levels to the model template; it handles 'none' as thinking disabled. """ CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'}) -LLAMA_EFFORTS=frozenset({'none','minimal','low','medium','high','xhigh','max'}) -EFFORT_ALIASES={'ultra':'max'} +LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'}) +# The installed llama.cpp server rejects minimal and max. Use nearest supported hints. +EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'} class CompatibilityError(ValueError):pass diff --git a/test_api_compat.py b/test_api_compat.py index a43188b..3f15ac2 100644 --- a/test_api_compat.py +++ b/test_api_compat.py @@ -5,7 +5,7 @@ class CompatibilityTests(unittest.TestCase): for value in ['none','minimal','low','medium','high','xhigh','max','ultra']: request={'model':'any','reasoning_effort':value};body,info=normalize_chat(request) self.assertEqual(request['reasoning_effort'],value) - self.assertEqual(body['reasoning_effort'],'max' if value=='ultra' else value) + self.assertEqual(body['reasoning_effort'],{'minimal':'low','max':'xhigh','ultra':'xhigh'}.get(value,value)) self.assertEqual(info['requested'],value) def test_unset_and_null_do_not_invent_defaults(self): for request in [{},{'reasoning_effort':None}]: