"""Client-neutral API normalization, separate from profiles and inference workers. Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards positive levels to the model template; it handles 'none' as thinking disabled. """ CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'}) LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'}) # The installed llama.cpp server rejects minimal and max. Use nearest supported hints. EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'} class CompatibilityError(ValueError):pass def normalize_chat(data): """Return a new request and non-sensitive translation metadata; never mutate input.""" unknown=set(data)-CHAT_FIELDS if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown))) body=dict(data);requested=body.get('reasoning_effort') if requested is None: body.pop('reasoning_effort',None) return body,{} if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys(): raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.') effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'}