Broaden client-neutral OpenAI and llama.cpp chat compatibility
This commit is contained in:
1 parent
c9036201f7
commit
3cd9fa5750
5 files changed
+158
-3
No files matched your search
+63
-2
@@ -4,8 +4,18 @@ Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
|
||||
positive levels to the model template; it handles 'none' as thinking disabled.
|
||||
"""
|
||||
import math
|
||||
import json
|
||||
import re
|
||||
from copy import deepcopy
|
||||
|
||||
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort','chat_template_kwargs'})
|
||||
LLAMA_FIELDS=frozenset({'min_p','typical_p','repeat_last_n','mirostat','mirostat_tau','mirostat_eta','dynatemp_range','dynatemp_exponent','cache_prompt','ignore_eos','reasoning_format','logit_bias'})
|
||||
ADAPTER_FIELDS=frozenset({'functions','function_call','extra_body','metadata','store','service_tier','safety_identifier','prompt_cache_key'})
|
||||
CHAT_FIELDS=CHAT_FIELDS|LLAMA_FIELDS|ADAPTER_FIELDS
|
||||
|
||||
def capabilities():
|
||||
return dict(version=1,chat_route='/v1/chat/completions',fields=sorted(CHAT_FIELDS),llama_extensions=sorted(LLAMA_FIELDS|{'chat_template_kwargs'}),template_options=['enable_thinking','preserve_thinking'],aliases={'max_completion_tokens':'max_tokens','functions':'tools','function_call':'tool_choice'},unsupported_routes=['/v1/responses','/v1/embeddings','/completion','/slots'],model_dependent=['tools','vision','reasoning','json_schema'],notes=['Only n=1; no stored completions or priority service tier.','Model capabilities and backend-version support are not guaranteed by this field list.'])
|
||||
|
||||
LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'})
|
||||
# The installed llama.cpp server rejects minimal and max. Use nearest supported hints.
|
||||
EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'}
|
||||
@@ -14,9 +24,60 @@ class CompatibilityError(ValueError):pass
|
||||
|
||||
def normalize_chat(data):
|
||||
"""Return a new request and non-sensitive translation metadata; never mutate input."""
|
||||
if not isinstance(data,dict):raise CompatibilityError('Chat-Anfrage muss ein JSON-Objekt sein.')
|
||||
data=deepcopy(data)
|
||||
extra=data.pop('extra_body',{})
|
||||
if not isinstance(extra,dict):raise CompatibilityError('extra_body muss ein JSON-Objekt sein.')
|
||||
if 'extra_body' in extra or set(extra)&{'model','messages'}:raise CompatibilityError('extra_body darf Modell oder Nachrichten nicht überschreiben.')
|
||||
for key,value in extra.items():
|
||||
if key in data and data[key]!=value:raise CompatibilityError('Widersprüchliches Feld in extra_body: '+key)
|
||||
data[key]=value
|
||||
unknown=set(data)-CHAT_FIELDS
|
||||
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
|
||||
body=dict(data)
|
||||
ignored=[]
|
||||
for key in ('metadata','safety_identifier','prompt_cache_key'):
|
||||
if key in body:
|
||||
value=body.pop(key)
|
||||
if key=='metadata' and value is not None and (not isinstance(value,dict) or len(value)>16 or any(not isinstance(k,str) or len(k)>64 or not isinstance(v,str) or len(v)>512 for k,v in value.items())):raise CompatibilityError('Ungültige metadata.')
|
||||
if key!='metadata' and value is not None and (not isinstance(value,str) or len(value)>512):raise CompatibilityError('Ungültiger '+key+'.')
|
||||
ignored.append(key)
|
||||
if 'store' in body:
|
||||
store=body.pop('store')
|
||||
if store is not False and store is not None:raise CompatibilityError('store=true wird nicht unterstützt; Deck speichert keine Completions.')
|
||||
ignored.append('store')
|
||||
if 'service_tier' in body:
|
||||
if body.pop('service_tier') not in (None,'auto','default'):raise CompatibilityError('Nur service_tier=auto/default wird unterstützt.')
|
||||
ignored.append('service_tier')
|
||||
for key in ('stop','stream_options','logprobs','top_logprobs','tools','tool_choice','response_format','logit_bias'):
|
||||
if body.get(key,'absent') is None:body.pop(key,None)
|
||||
if 'max_completion_tokens' in body:
|
||||
value=body.pop('max_completion_tokens')
|
||||
if value is not None:
|
||||
if 'max_tokens' in body and body['max_tokens']!=value:raise CompatibilityError('max_tokens und max_completion_tokens widersprechen sich.')
|
||||
body['max_tokens']=value
|
||||
if 'functions' in body:
|
||||
value=body.pop('functions')
|
||||
if not isinstance(value,list) or any(not isinstance(f,dict) for f in value):raise CompatibilityError('functions muss eine Liste von Funktionsdefinitionen sein.')
|
||||
translated=[{'type':'function','function':f} for f in value]
|
||||
if 'tools' in body and body['tools']!=translated:raise CompatibilityError('functions und tools widersprechen sich.')
|
||||
body['tools']=translated
|
||||
if 'function_call' in body:
|
||||
value=body.pop('function_call')
|
||||
if isinstance(value,dict) and set(value)=={'name'} and isinstance(value['name'],str):value={'type':'function','function':value}
|
||||
elif value not in ('auto','none'):raise CompatibilityError('Ungültiges function_call.')
|
||||
if 'tool_choice' in body and body['tool_choice']!=value:raise CompatibilityError('function_call und tool_choice widersprechen sich.')
|
||||
body['tool_choice']=value
|
||||
for key,lo,hi in [('min_p',0,1),('typical_p',0,1),('mirostat_tau',0,100),('mirostat_eta',0,1),('dynatemp_range',0,100),('dynatemp_exponent',0,100)]:
|
||||
if key in body and (type(body[key]) not in (int,float) or not math.isfinite(body[key]) or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
|
||||
for key,lo,hi in [('repeat_last_n',-1,2097152),('mirostat',0,2),('max_tokens',1,2097152)]:
|
||||
if key in body and (type(body[key]) is not int or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
|
||||
for key in ('cache_prompt','ignore_eos','parallel_tool_calls'):
|
||||
if key in body and type(body[key]) is not bool:raise CompatibilityError(key+' muss true oder false sein.')
|
||||
if 'reasoning_format' in body and body['reasoning_format'] not in ('none','deepseek','auto'):raise CompatibilityError('Ungültiges reasoning_format.')
|
||||
if 'logit_bias' in body:
|
||||
biases=body['logit_bias']
|
||||
if not isinstance(biases,dict) or len(biases)>4096 or any(not isinstance(k,str) or not re.fullmatch(r'[0-9]{1,9}',k) or type(v) not in (int,float) or not math.isfinite(v) or not -100<=v<=100 for k,v in biases.items()):raise CompatibilityError('Ungültiges logit_bias.')
|
||||
if 'chat_template_kwargs' in body:
|
||||
kwargs=body['chat_template_kwargs']
|
||||
if not isinstance(kwargs,dict) or set(kwargs)-{'enable_thinking','preserve_thinking'}:
|
||||
@@ -32,8 +93,8 @@ def normalize_chat(data):
|
||||
requested=body.get('reasoning_effort')
|
||||
if requested is None:
|
||||
body.pop('reasoning_effort',None)
|
||||
return body,{}
|
||||
return body,({'ignored':ignored} if ignored else {})
|
||||
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
|
||||
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
|
||||
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
|
||||
return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|
||||
return body,{'ignored':ignored,'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|
||||
Reference in new issue
Block a user