Isolate client reasoning normalization in API compatibility adapter
This commit is contained in:
@@ -0,0 +1,23 @@
|
||||
"""Client-neutral API normalization, separate from profiles and inference workers.
|
||||
|
||||
Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
|
||||
positive levels to the model template; it handles 'none' as thinking disabled.
|
||||
"""
|
||||
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'})
|
||||
LLAMA_EFFORTS=frozenset({'none','minimal','low','medium','high','xhigh','max'})
|
||||
EFFORT_ALIASES={'ultra':'max'}
|
||||
|
||||
class CompatibilityError(ValueError):pass
|
||||
|
||||
def normalize_chat(data):
|
||||
"""Return a new request and non-sensitive translation metadata; never mutate input."""
|
||||
unknown=set(data)-CHAT_FIELDS
|
||||
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
|
||||
body=dict(data);requested=body.get('reasoning_effort')
|
||||
if requested is None:
|
||||
body.pop('reasoning_effort',None)
|
||||
return body,{}
|
||||
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
|
||||
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
|
||||
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
|
||||
return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|
||||
Reference in New Issue
Block a user