101 lines
7.4 KiB
Python
101 lines
7.4 KiB
Python
"""Client-neutral API normalization, separate from profiles and inference workers.
|
|
|
|
Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
|
|
positive levels to the model template; it handles 'none' as thinking disabled.
|
|
"""
|
|
import math
|
|
import json
|
|
import re
|
|
from copy import deepcopy
|
|
|
|
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort','chat_template_kwargs'})
|
|
LLAMA_FIELDS=frozenset({'min_p','typical_p','repeat_last_n','mirostat','mirostat_tau','mirostat_eta','dynatemp_range','dynatemp_exponent','cache_prompt','ignore_eos','reasoning_format','logit_bias'})
|
|
ADAPTER_FIELDS=frozenset({'functions','function_call','extra_body','metadata','store','service_tier','safety_identifier','prompt_cache_key'})
|
|
CHAT_FIELDS=CHAT_FIELDS|LLAMA_FIELDS|ADAPTER_FIELDS
|
|
|
|
def capabilities():
|
|
return dict(version=1,chat_route='/v1/chat/completions',fields=sorted(CHAT_FIELDS),llama_extensions=sorted(LLAMA_FIELDS|{'chat_template_kwargs'}),template_options=['enable_thinking','preserve_thinking'],aliases={'max_completion_tokens':'max_tokens','functions':'tools','function_call':'tool_choice'},unsupported_routes=['/v1/responses','/v1/embeddings','/completion','/slots'],model_dependent=['tools','vision','reasoning','json_schema'],notes=['Only n=1; no stored completions or priority service tier.','Model capabilities and backend-version support are not guaranteed by this field list.'])
|
|
|
|
LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'})
|
|
# The installed llama.cpp server rejects minimal and max. Use nearest supported hints.
|
|
EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'}
|
|
|
|
class CompatibilityError(ValueError):pass
|
|
|
|
def normalize_chat(data):
|
|
"""Return a new request and non-sensitive translation metadata; never mutate input."""
|
|
if not isinstance(data,dict):raise CompatibilityError('Chat-Anfrage muss ein JSON-Objekt sein.')
|
|
data=deepcopy(data)
|
|
extra=data.pop('extra_body',{})
|
|
if not isinstance(extra,dict):raise CompatibilityError('extra_body muss ein JSON-Objekt sein.')
|
|
if 'extra_body' in extra or set(extra)&{'model','messages'}:raise CompatibilityError('extra_body darf Modell oder Nachrichten nicht überschreiben.')
|
|
for key,value in extra.items():
|
|
if key in data and data[key]!=value:raise CompatibilityError('Widersprüchliches Feld in extra_body: '+key)
|
|
data[key]=value
|
|
unknown=set(data)-CHAT_FIELDS
|
|
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
|
|
body=dict(data)
|
|
ignored=[]
|
|
for key in ('metadata','safety_identifier','prompt_cache_key'):
|
|
if key in body:
|
|
value=body.pop(key)
|
|
if key=='metadata' and value is not None and (not isinstance(value,dict) or len(value)>16 or any(not isinstance(k,str) or len(k)>64 or not isinstance(v,str) or len(v)>512 for k,v in value.items())):raise CompatibilityError('Ungültige metadata.')
|
|
if key!='metadata' and value is not None and (not isinstance(value,str) or len(value)>512):raise CompatibilityError('Ungültiger '+key+'.')
|
|
ignored.append(key)
|
|
if 'store' in body:
|
|
store=body.pop('store')
|
|
if store is not False and store is not None:raise CompatibilityError('store=true wird nicht unterstützt; Deck speichert keine Completions.')
|
|
ignored.append('store')
|
|
if 'service_tier' in body:
|
|
if body.pop('service_tier') not in (None,'auto','default'):raise CompatibilityError('Nur service_tier=auto/default wird unterstützt.')
|
|
ignored.append('service_tier')
|
|
for key in ('stop','stream_options','logprobs','top_logprobs','tools','tool_choice','response_format','logit_bias'):
|
|
if body.get(key,'absent') is None:body.pop(key,None)
|
|
if 'max_completion_tokens' in body:
|
|
value=body.pop('max_completion_tokens')
|
|
if value is not None:
|
|
if 'max_tokens' in body and body['max_tokens']!=value:raise CompatibilityError('max_tokens und max_completion_tokens widersprechen sich.')
|
|
body['max_tokens']=value
|
|
if 'functions' in body:
|
|
value=body.pop('functions')
|
|
if not isinstance(value,list) or any(not isinstance(f,dict) for f in value):raise CompatibilityError('functions muss eine Liste von Funktionsdefinitionen sein.')
|
|
translated=[{'type':'function','function':f} for f in value]
|
|
if 'tools' in body and body['tools']!=translated:raise CompatibilityError('functions und tools widersprechen sich.')
|
|
body['tools']=translated
|
|
if 'function_call' in body:
|
|
value=body.pop('function_call')
|
|
if isinstance(value,dict) and set(value)=={'name'} and isinstance(value['name'],str):value={'type':'function','function':value}
|
|
elif value not in ('auto','none'):raise CompatibilityError('Ungültiges function_call.')
|
|
if 'tool_choice' in body and body['tool_choice']!=value:raise CompatibilityError('function_call und tool_choice widersprechen sich.')
|
|
body['tool_choice']=value
|
|
for key,lo,hi in [('min_p',0,1),('typical_p',0,1),('mirostat_tau',0,100),('mirostat_eta',0,1),('dynatemp_range',0,100),('dynatemp_exponent',0,100)]:
|
|
if key in body and (type(body[key]) not in (int,float) or not math.isfinite(body[key]) or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
|
|
for key,lo,hi in [('repeat_last_n',-1,2097152),('mirostat',0,2),('max_tokens',1,2097152)]:
|
|
if key in body and (type(body[key]) is not int or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
|
|
for key in ('cache_prompt','ignore_eos','parallel_tool_calls'):
|
|
if key in body and type(body[key]) is not bool:raise CompatibilityError(key+' muss true oder false sein.')
|
|
if 'reasoning_format' in body and body['reasoning_format'] not in ('none','deepseek','auto'):raise CompatibilityError('Ungültiges reasoning_format.')
|
|
if 'logit_bias' in body:
|
|
biases=body['logit_bias']
|
|
if not isinstance(biases,dict) or len(biases)>4096 or any(not isinstance(k,str) or not re.fullmatch(r'[0-9]{1,9}',k) or type(v) not in (int,float) or not math.isfinite(v) or not -100<=v<=100 for k,v in biases.items()):raise CompatibilityError('Ungültiges logit_bias.')
|
|
if 'chat_template_kwargs' in body:
|
|
kwargs=body['chat_template_kwargs']
|
|
if not isinstance(kwargs,dict) or set(kwargs)-{'enable_thinking','preserve_thinking'}:
|
|
raise CompatibilityError('chat_template_kwargs unterstützt enable_thinking und preserve_thinking.')
|
|
if any(type(v) is not bool for v in kwargs.values()):
|
|
raise CompatibilityError('Thinking-Template-Optionen müssen true oder false sein.')
|
|
body['chat_template_kwargs']=dict(kwargs)
|
|
for key,lo,hi in [('repeat_penalty',0,2),('presence_penalty',-2,2),('frequency_penalty',-2,2)]:
|
|
if key in body:
|
|
value=body[key]
|
|
if isinstance(value,bool) or not isinstance(value,(int,float)) or not math.isfinite(value) or not lo<=value<=hi:
|
|
raise CompatibilityError('Ungültiger '+key+'-Wert.')
|
|
requested=body.get('reasoning_effort')
|
|
if requested is None:
|
|
body.pop('reasoning_effort',None)
|
|
return body,({'ignored':ignored} if ignored else {})
|
|
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
|
|
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
|
|
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
|
|
return body,{'ignored':ignored,'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|