Broaden client-neutral OpenAI and llama.cpp chat compatibility
This commit is contained in:
1 parent
c9036201f7
commit
3cd9fa5750
5 files changed
+158
-3
No files matched your search
@@ -0,0 +1,47 @@
|
|||||||
|
# Client-neutral chat compatibility
|
||||||
|
|
||||||
|
Deck exposes authenticated OpenAI-style Chat Completions at `/v1/chat/completions`.
|
||||||
|
It is a router, not a complete replica of OpenAI or of every llama-server route.
|
||||||
|
The compatibility layer is `api_compat.py`; it does not modify model profiles.
|
||||||
|
|
||||||
|
## Discovery and behaviour
|
||||||
|
|
||||||
|
`GET /v1/capabilities` with the same Bearer token lists accepted request fields,
|
||||||
|
llama.cpp extensions, aliases and routes not implemented by Deck. This describes
|
||||||
|
router support, not a guarantee that every model/build supports every function.
|
||||||
|
Existing `/v1/models` remains a chat-only list. Images, speech and transcription
|
||||||
|
have their own model-list routes. Request handling and streaming share the same
|
||||||
|
normalization path; client-specific branches are not used.
|
||||||
|
|
||||||
|
- `max_completion_tokens` becomes `max_tokens`; conflicting limits fail explicitly.
|
||||||
|
- Legacy `functions` and `function_call` become `tools` and `tool_choice`.
|
||||||
|
- SDK `extra_body` is flattened, validated and cannot replace model/messages.
|
||||||
|
- Null optional stop/tools/stream options are omitted.
|
||||||
|
- `reasoning_effort` aliases retain existing nearest-supported mappings.
|
||||||
|
- Template options `enable_thinking` and `preserve_thinking` require booleans.
|
||||||
|
- Additional llama.cpp options: `min_p`, `typical_p`, `repeat_last_n`, `mirostat`,
|
||||||
|
`mirostat_tau`, `mirostat_eta`, `dynatemp_range`, `dynatemp_exponent`,
|
||||||
|
`cache_prompt`, `ignore_eos`, `reasoning_format`, and `logit_bias`.
|
||||||
|
Numeric types, finite values and bounded ranges are validated.
|
||||||
|
- Metadata, safety identifiers and prompt-cache keys are accepted as annotations,
|
||||||
|
but not persisted or used for OpenAI-hosted safety/caching semantics. Their
|
||||||
|
removal is reported in `X-Athena-Compatibility-Ignored` (also on SSE responses).
|
||||||
|
- `store=false/null` and service tier `auto/default/null` are tolerated and reported;
|
||||||
|
stored completions and priority service tiers are rejected, not simulated.
|
||||||
|
|
||||||
|
Unknown fields and conflicting aliases still produce explicit HTTP 400 errors.
|
||||||
|
Only one completion (`n=1`) is supported. Embeddings, Responses, native completion,
|
||||||
|
slot administration and arbitrary template overrides are not added by this change.
|
||||||
|
Tools, vision, reasoning and structured output depend on model weights, template
|
||||||
|
and active runtime version. Tool calls are returned to clients, not executed by Deck.
|
||||||
|
Model capability declarations must not be inferred solely from a model name.
|
||||||
|
|
||||||
|
## Verification
|
||||||
|
|
||||||
|
Tests cover alias translation, conflicts, input isolation, invalid sampling values,
|
||||||
|
metadata treatment, upstream forwarding and authenticated capability discovery.
|
||||||
|
Synthetic/fake-worker API tests do not establish inference quality or actual tool
|
||||||
|
calling for every installed model. New upstream features require review and tests.
|
||||||
|
|
||||||
|
References: [llama.cpp server API](https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md)
|
||||||
|
and [OpenAI Chat Completions](https://platform.openai.com/docs/api-reference/chat/create).
|
||||||
+63
-2
@@ -4,8 +4,18 @@ Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
|
|||||||
positive levels to the model template; it handles 'none' as thinking disabled.
|
positive levels to the model template; it handles 'none' as thinking disabled.
|
||||||
"""
|
"""
|
||||||
import math
|
import math
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
from copy import deepcopy
|
||||||
|
|
||||||
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort','chat_template_kwargs'})
|
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort','chat_template_kwargs'})
|
||||||
|
LLAMA_FIELDS=frozenset({'min_p','typical_p','repeat_last_n','mirostat','mirostat_tau','mirostat_eta','dynatemp_range','dynatemp_exponent','cache_prompt','ignore_eos','reasoning_format','logit_bias'})
|
||||||
|
ADAPTER_FIELDS=frozenset({'functions','function_call','extra_body','metadata','store','service_tier','safety_identifier','prompt_cache_key'})
|
||||||
|
CHAT_FIELDS=CHAT_FIELDS|LLAMA_FIELDS|ADAPTER_FIELDS
|
||||||
|
|
||||||
|
def capabilities():
|
||||||
|
return dict(version=1,chat_route='/v1/chat/completions',fields=sorted(CHAT_FIELDS),llama_extensions=sorted(LLAMA_FIELDS|{'chat_template_kwargs'}),template_options=['enable_thinking','preserve_thinking'],aliases={'max_completion_tokens':'max_tokens','functions':'tools','function_call':'tool_choice'},unsupported_routes=['/v1/responses','/v1/embeddings','/completion','/slots'],model_dependent=['tools','vision','reasoning','json_schema'],notes=['Only n=1; no stored completions or priority service tier.','Model capabilities and backend-version support are not guaranteed by this field list.'])
|
||||||
|
|
||||||
LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'})
|
LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'})
|
||||||
# The installed llama.cpp server rejects minimal and max. Use nearest supported hints.
|
# The installed llama.cpp server rejects minimal and max. Use nearest supported hints.
|
||||||
EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'}
|
EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'}
|
||||||
@@ -14,9 +24,60 @@ class CompatibilityError(ValueError):pass
|
|||||||
|
|
||||||
def normalize_chat(data):
|
def normalize_chat(data):
|
||||||
"""Return a new request and non-sensitive translation metadata; never mutate input."""
|
"""Return a new request and non-sensitive translation metadata; never mutate input."""
|
||||||
|
if not isinstance(data,dict):raise CompatibilityError('Chat-Anfrage muss ein JSON-Objekt sein.')
|
||||||
|
data=deepcopy(data)
|
||||||
|
extra=data.pop('extra_body',{})
|
||||||
|
if not isinstance(extra,dict):raise CompatibilityError('extra_body muss ein JSON-Objekt sein.')
|
||||||
|
if 'extra_body' in extra or set(extra)&{'model','messages'}:raise CompatibilityError('extra_body darf Modell oder Nachrichten nicht überschreiben.')
|
||||||
|
for key,value in extra.items():
|
||||||
|
if key in data and data[key]!=value:raise CompatibilityError('Widersprüchliches Feld in extra_body: '+key)
|
||||||
|
data[key]=value
|
||||||
unknown=set(data)-CHAT_FIELDS
|
unknown=set(data)-CHAT_FIELDS
|
||||||
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
|
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
|
||||||
body=dict(data)
|
body=dict(data)
|
||||||
|
ignored=[]
|
||||||
|
for key in ('metadata','safety_identifier','prompt_cache_key'):
|
||||||
|
if key in body:
|
||||||
|
value=body.pop(key)
|
||||||
|
if key=='metadata' and value is not None and (not isinstance(value,dict) or len(value)>16 or any(not isinstance(k,str) or len(k)>64 or not isinstance(v,str) or len(v)>512 for k,v in value.items())):raise CompatibilityError('Ungültige metadata.')
|
||||||
|
if key!='metadata' and value is not None and (not isinstance(value,str) or len(value)>512):raise CompatibilityError('Ungültiger '+key+'.')
|
||||||
|
ignored.append(key)
|
||||||
|
if 'store' in body:
|
||||||
|
store=body.pop('store')
|
||||||
|
if store is not False and store is not None:raise CompatibilityError('store=true wird nicht unterstützt; Deck speichert keine Completions.')
|
||||||
|
ignored.append('store')
|
||||||
|
if 'service_tier' in body:
|
||||||
|
if body.pop('service_tier') not in (None,'auto','default'):raise CompatibilityError('Nur service_tier=auto/default wird unterstützt.')
|
||||||
|
ignored.append('service_tier')
|
||||||
|
for key in ('stop','stream_options','logprobs','top_logprobs','tools','tool_choice','response_format','logit_bias'):
|
||||||
|
if body.get(key,'absent') is None:body.pop(key,None)
|
||||||
|
if 'max_completion_tokens' in body:
|
||||||
|
value=body.pop('max_completion_tokens')
|
||||||
|
if value is not None:
|
||||||
|
if 'max_tokens' in body and body['max_tokens']!=value:raise CompatibilityError('max_tokens und max_completion_tokens widersprechen sich.')
|
||||||
|
body['max_tokens']=value
|
||||||
|
if 'functions' in body:
|
||||||
|
value=body.pop('functions')
|
||||||
|
if not isinstance(value,list) or any(not isinstance(f,dict) for f in value):raise CompatibilityError('functions muss eine Liste von Funktionsdefinitionen sein.')
|
||||||
|
translated=[{'type':'function','function':f} for f in value]
|
||||||
|
if 'tools' in body and body['tools']!=translated:raise CompatibilityError('functions und tools widersprechen sich.')
|
||||||
|
body['tools']=translated
|
||||||
|
if 'function_call' in body:
|
||||||
|
value=body.pop('function_call')
|
||||||
|
if isinstance(value,dict) and set(value)=={'name'} and isinstance(value['name'],str):value={'type':'function','function':value}
|
||||||
|
elif value not in ('auto','none'):raise CompatibilityError('Ungültiges function_call.')
|
||||||
|
if 'tool_choice' in body and body['tool_choice']!=value:raise CompatibilityError('function_call und tool_choice widersprechen sich.')
|
||||||
|
body['tool_choice']=value
|
||||||
|
for key,lo,hi in [('min_p',0,1),('typical_p',0,1),('mirostat_tau',0,100),('mirostat_eta',0,1),('dynatemp_range',0,100),('dynatemp_exponent',0,100)]:
|
||||||
|
if key in body and (type(body[key]) not in (int,float) or not math.isfinite(body[key]) or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
|
||||||
|
for key,lo,hi in [('repeat_last_n',-1,2097152),('mirostat',0,2),('max_tokens',1,2097152)]:
|
||||||
|
if key in body and (type(body[key]) is not int or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
|
||||||
|
for key in ('cache_prompt','ignore_eos','parallel_tool_calls'):
|
||||||
|
if key in body and type(body[key]) is not bool:raise CompatibilityError(key+' muss true oder false sein.')
|
||||||
|
if 'reasoning_format' in body and body['reasoning_format'] not in ('none','deepseek','auto'):raise CompatibilityError('Ungültiges reasoning_format.')
|
||||||
|
if 'logit_bias' in body:
|
||||||
|
biases=body['logit_bias']
|
||||||
|
if not isinstance(biases,dict) or len(biases)>4096 or any(not isinstance(k,str) or not re.fullmatch(r'[0-9]{1,9}',k) or type(v) not in (int,float) or not math.isfinite(v) or not -100<=v<=100 for k,v in biases.items()):raise CompatibilityError('Ungültiges logit_bias.')
|
||||||
if 'chat_template_kwargs' in body:
|
if 'chat_template_kwargs' in body:
|
||||||
kwargs=body['chat_template_kwargs']
|
kwargs=body['chat_template_kwargs']
|
||||||
if not isinstance(kwargs,dict) or set(kwargs)-{'enable_thinking','preserve_thinking'}:
|
if not isinstance(kwargs,dict) or set(kwargs)-{'enable_thinking','preserve_thinking'}:
|
||||||
@@ -32,8 +93,8 @@ def normalize_chat(data):
|
|||||||
requested=body.get('reasoning_effort')
|
requested=body.get('reasoning_effort')
|
||||||
if requested is None:
|
if requested is None:
|
||||||
body.pop('reasoning_effort',None)
|
body.pop('reasoning_effort',None)
|
||||||
return body,{}
|
return body,({'ignored':ignored} if ignored else {})
|
||||||
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
|
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
|
||||||
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
|
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
|
||||||
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
|
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
|
||||||
return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|
return body,{'ignored':ignored,'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|
||||||
+6
-1
@@ -169,7 +169,9 @@ class APIHandler(BaseHTTPRequestHandler):
|
|||||||
def log_message(self,*args):pass
|
def log_message(self,*args):pass
|
||||||
def compatibility_headers(self):
|
def compatibility_headers(self):
|
||||||
info=getattr(self,'compatibility',{})
|
info=getattr(self,'compatibility',{})
|
||||||
if info:
|
if info.get('ignored'):
|
||||||
|
self.send_header('X-Athena-Compatibility-Ignored',','.join(info['ignored']))
|
||||||
|
if 'requested' in info:
|
||||||
self.send_header('X-Athena-Reasoning-Requested',info['requested'])
|
self.send_header('X-Athena-Reasoning-Requested',info['requested'])
|
||||||
self.send_header('X-Athena-Reasoning-Effective',info['effective'])
|
self.send_header('X-Athena-Reasoning-Effective',info['effective'])
|
||||||
self.send_header('X-Athena-Reasoning-Semantics',info['semantics'])
|
self.send_header('X-Athena-Reasoning-Semantics',info['semantics'])
|
||||||
@@ -209,6 +211,9 @@ class APIHandler(BaseHTTPRequestHandler):
|
|||||||
try:ep.video.relay(self) if callable(getattr(type(ep.video),'relay',None)) else relay(self,video['selected'])
|
try:ep.video.relay(self) if callable(getattr(type(ep.video),'relay',None)) else relay(self,video['selected'])
|
||||||
except ValueError as exc:raise APIError(str(exc)) from None
|
except ValueError as exc:raise APIError(str(exc)) from None
|
||||||
return
|
return
|
||||||
|
if self.command=='GET' and self.path=='/v1/capabilities':
|
||||||
|
from api_compat import capabilities
|
||||||
|
return self.send(capabilities())
|
||||||
model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt','/v1/audio/music/models':'music'}
|
model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt','/v1/audio/music/models':'music'}
|
||||||
if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path]))
|
if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path]))
|
||||||
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
|
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
|
||||||
|
|||||||
@@ -23,3 +23,31 @@ class CompatibilityTests(unittest.TestCase):
|
|||||||
self.assertFalse(original['chat_template_kwargs']['enable_thinking'])
|
self.assertFalse(original['chat_template_kwargs']['enable_thinking'])
|
||||||
for bad in [None,[],{'enable_thinking':'false'},{'template':'custom'},{'preserve_thinking':1}]:
|
for bad in [None,[],{'enable_thinking':'false'},{'template':'custom'},{'preserve_thinking':1}]:
|
||||||
with self.assertRaises(CompatibilityError):normalize_chat({'chat_template_kwargs':bad})
|
with self.assertRaises(CompatibilityError):normalize_chat({'chat_template_kwargs':bad})
|
||||||
|
|
||||||
|
def test_openai_aliases_and_legacy_tools(self):
|
||||||
|
request={'max_completion_tokens':32,'functions':[{'name':'echo','parameters':{'type':'object'}}],'function_call':{'name':'echo'},'stop':None}
|
||||||
|
body,_=normalize_chat(request)
|
||||||
|
self.assertEqual(body['max_tokens'],32)
|
||||||
|
self.assertEqual(body['tools'][0]['function']['name'],'echo')
|
||||||
|
self.assertEqual(body['tool_choice']['function']['name'],'echo')
|
||||||
|
self.assertNotIn('stop',body)
|
||||||
|
self.assertIn('functions',request)
|
||||||
|
|
||||||
|
def test_sdk_extra_body_and_conflicts(self):
|
||||||
|
body,_=normalize_chat({'extra_body':{'min_p':.05,'chat_template_kwargs':{'enable_thinking':False}},'max_tokens':8})
|
||||||
|
self.assertEqual(body['min_p'],.05)
|
||||||
|
for request in [{'extra_body':{'model':'other'}},{'extra_body':{'min_p':.1},'min_p':.2},{'max_tokens':8,'max_completion_tokens':16},{'functions':[],'tools':[{}]}]:
|
||||||
|
with self.assertRaises(CompatibilityError):normalize_chat(request)
|
||||||
|
|
||||||
|
def test_metadata_is_explicitly_reported_not_forwarded(self):
|
||||||
|
body,info=normalize_chat({'store':False,'metadata':{'project':'synthetic'},'service_tier':'default'})
|
||||||
|
self.assertEqual(body,{})
|
||||||
|
self.assertEqual(set(info['ignored']),{'store','metadata','service_tier'})
|
||||||
|
for request in [{'store':True},{'service_tier':'priority'},{'metadata':{'x':42}}]:
|
||||||
|
with self.assertRaises(CompatibilityError):normalize_chat(request)
|
||||||
|
|
||||||
|
def test_llama_validation(self):
|
||||||
|
body,_=normalize_chat({'repeat_last_n':-1,'mirostat':2,'mirostat_eta':.1,'typical_p':1,'cache_prompt':True,'reasoning_format':'auto','logit_bias':{'123':-10}})
|
||||||
|
self.assertEqual(body['logit_bias'],{'123':-10})
|
||||||
|
for request in [{'min_p':float('nan')},{'cache_prompt':1},{'repeat_last_n':.5},{'logit_bias':{'not-token':1}},{'mirostat':3},{'reasoning_format':'invented'}]:
|
||||||
|
with self.assertRaises(CompatibilityError):normalize_chat(request)
|
||||||
@@ -112,6 +112,20 @@ class EndpointTests(unittest.TestCase):
|
|||||||
self.assertEqual(self.request(route,token='bad')[0],401)
|
self.assertEqual(self.request(route,token='bad')[0],401)
|
||||||
self.rows[-1]['runnable']=False
|
self.rows[-1]['runnable']=False
|
||||||
self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[])
|
self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[])
|
||||||
|
def test_extended_compatibility_forwarding(self):
|
||||||
|
self.enable('alpha')
|
||||||
|
data=dict(model='alpha',messages=[dict(role='user',content='synthetic')],max_completion_tokens=8,extra_body={'min_p':.05},metadata={'project':'synthetic'},store=False,functions=[{'name':'echo','parameters':{'type':'object'}}],function_call='auto',chat_template_kwargs={'enable_thinking':False})
|
||||||
|
status,_=self.request('/v1/chat/completions',data)
|
||||||
|
self.assertEqual(status,200)
|
||||||
|
forwarded=self.worker.requests[-1]
|
||||||
|
self.assertEqual(forwarded['max_tokens'],8)
|
||||||
|
self.assertEqual(forwarded['min_p'],.05)
|
||||||
|
self.assertEqual(forwarded['tools'][0]['function']['name'],'echo')
|
||||||
|
self.assertNotIn('metadata',forwarded)
|
||||||
|
status,body=self.request('/v1/capabilities')
|
||||||
|
self.assertEqual(status,200)
|
||||||
|
self.assertIn('min_p',body['llama_extensions'])
|
||||||
|
|
||||||
def test_reasoning_effort_forwarded_and_validated(self):
|
def test_reasoning_effort_forwarded_and_validated(self):
|
||||||
self.enable('alpha')
|
self.enable('alpha')
|
||||||
for effort in ('none','minimal','low','medium','high','xhigh','max','ultra',None):
|
for effort in ('none','minimal','low','medium','high','xhigh','max','ultra',None):
|
||||||
|
|||||||
Reference in new issue
Block a user