24 lines
1.4 KiB
Python
24 lines
1.4 KiB
Python
"""Client-neutral API normalization, separate from profiles and inference workers.
|
|
|
|
Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
|
|
positive levels to the model template; it handles 'none' as thinking disabled.
|
|
"""
|
|
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'})
|
|
LLAMA_EFFORTS=frozenset({'none','minimal','low','medium','high','xhigh','max'})
|
|
EFFORT_ALIASES={'ultra':'max'}
|
|
|
|
class CompatibilityError(ValueError):pass
|
|
|
|
def normalize_chat(data):
|
|
"""Return a new request and non-sensitive translation metadata; never mutate input."""
|
|
unknown=set(data)-CHAT_FIELDS
|
|
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
|
|
body=dict(data);requested=body.get('reasoning_effort')
|
|
if requested is None:
|
|
body.pop('reasoning_effort',None)
|
|
return body,{}
|
|
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
|
|
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
|
|
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
|
|
return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|