Broaden client-neutral OpenAI and llama.cpp chat compatibility

This commit is contained in:
Mikei386 committed 2026-10-02 10:29:32 +02:00
1 parent c9036201f7
commit 3cd9fa5750
5 files changed
+158 -3

No files matched your search

+47
View File
@@ -0,0 +1,47 @@
# Client-neutral chat compatibility
Deck exposes authenticated OpenAI-style Chat Completions at `/v1/chat/completions`.
It is a router, not a complete replica of OpenAI or of every llama-server route.
The compatibility layer is `api_compat.py`; it does not modify model profiles.
## Discovery and behaviour
`GET /v1/capabilities` with the same Bearer token lists accepted request fields,
llama.cpp extensions, aliases and routes not implemented by Deck. This describes
router support, not a guarantee that every model/build supports every function.
Existing `/v1/models` remains a chat-only list. Images, speech and transcription
have their own model-list routes. Request handling and streaming share the same
normalization path; client-specific branches are not used.
- `max_completion_tokens` becomes `max_tokens`; conflicting limits fail explicitly.
- Legacy `functions` and `function_call` become `tools` and `tool_choice`.
- SDK `extra_body` is flattened, validated and cannot replace model/messages.
- Null optional stop/tools/stream options are omitted.
- `reasoning_effort` aliases retain existing nearest-supported mappings.
- Template options `enable_thinking` and `preserve_thinking` require booleans.
- Additional llama.cpp options: `min_p`, `typical_p`, `repeat_last_n`, `mirostat`,
`mirostat_tau`, `mirostat_eta`, `dynatemp_range`, `dynatemp_exponent`,
`cache_prompt`, `ignore_eos`, `reasoning_format`, and `logit_bias`.
Numeric types, finite values and bounded ranges are validated.
- Metadata, safety identifiers and prompt-cache keys are accepted as annotations,
but not persisted or used for OpenAI-hosted safety/caching semantics. Their
removal is reported in `X-Athena-Compatibility-Ignored` (also on SSE responses).
- `store=false/null` and service tier `auto/default/null` are tolerated and reported;
stored completions and priority service tiers are rejected, not simulated.
Unknown fields and conflicting aliases still produce explicit HTTP 400 errors.
Only one completion (`n=1`) is supported. Embeddings, Responses, native completion,
slot administration and arbitrary template overrides are not added by this change.
Tools, vision, reasoning and structured output depend on model weights, template
and active runtime version. Tool calls are returned to clients, not executed by Deck.
Model capability declarations must not be inferred solely from a model name.
## Verification
Tests cover alias translation, conflicts, input isolation, invalid sampling values,
metadata treatment, upstream forwarding and authenticated capability discovery.
Synthetic/fake-worker API tests do not establish inference quality or actual tool
calling for every installed model. New upstream features require review and tests.
References: [llama.cpp server API](https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md)
and [OpenAI Chat Completions](https://platform.openai.com/docs/api-reference/chat/create).
+63 -2
View File
@@ -4,8 +4,18 @@ Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
positive levels to the model template; it handles 'none' as thinking disabled. positive levels to the model template; it handles 'none' as thinking disabled.
""" """
import math import math
import json
import re
from copy import deepcopy
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort','chat_template_kwargs'}) CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort','chat_template_kwargs'})
LLAMA_FIELDS=frozenset({'min_p','typical_p','repeat_last_n','mirostat','mirostat_tau','mirostat_eta','dynatemp_range','dynatemp_exponent','cache_prompt','ignore_eos','reasoning_format','logit_bias'})
ADAPTER_FIELDS=frozenset({'functions','function_call','extra_body','metadata','store','service_tier','safety_identifier','prompt_cache_key'})
CHAT_FIELDS=CHAT_FIELDS|LLAMA_FIELDS|ADAPTER_FIELDS
def capabilities():
return dict(version=1,chat_route='/v1/chat/completions',fields=sorted(CHAT_FIELDS),llama_extensions=sorted(LLAMA_FIELDS|{'chat_template_kwargs'}),template_options=['enable_thinking','preserve_thinking'],aliases={'max_completion_tokens':'max_tokens','functions':'tools','function_call':'tool_choice'},unsupported_routes=['/v1/responses','/v1/embeddings','/completion','/slots'],model_dependent=['tools','vision','reasoning','json_schema'],notes=['Only n=1; no stored completions or priority service tier.','Model capabilities and backend-version support are not guaranteed by this field list.'])
LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'}) LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'})
# The installed llama.cpp server rejects minimal and max. Use nearest supported hints. # The installed llama.cpp server rejects minimal and max. Use nearest supported hints.
EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'} EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'}
@@ -14,9 +24,60 @@ class CompatibilityError(ValueError):pass
def normalize_chat(data): def normalize_chat(data):
"""Return a new request and non-sensitive translation metadata; never mutate input.""" """Return a new request and non-sensitive translation metadata; never mutate input."""
if not isinstance(data,dict):raise CompatibilityError('Chat-Anfrage muss ein JSON-Objekt sein.')
data=deepcopy(data)
extra=data.pop('extra_body',{})
if not isinstance(extra,dict):raise CompatibilityError('extra_body muss ein JSON-Objekt sein.')
if 'extra_body' in extra or set(extra)&{'model','messages'}:raise CompatibilityError('extra_body darf Modell oder Nachrichten nicht überschreiben.')
for key,value in extra.items():
if key in data and data[key]!=value:raise CompatibilityError('Widersprüchliches Feld in extra_body: '+key)
data[key]=value
unknown=set(data)-CHAT_FIELDS unknown=set(data)-CHAT_FIELDS
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown))) if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
body=dict(data) body=dict(data)
ignored=[]
for key in ('metadata','safety_identifier','prompt_cache_key'):
if key in body:
value=body.pop(key)
if key=='metadata' and value is not None and (not isinstance(value,dict) or len(value)>16 or any(not isinstance(k,str) or len(k)>64 or not isinstance(v,str) or len(v)>512 for k,v in value.items())):raise CompatibilityError('Ungültige metadata.')
if key!='metadata' and value is not None and (not isinstance(value,str) or len(value)>512):raise CompatibilityError('Ungültiger '+key+'.')
ignored.append(key)
if 'store' in body:
store=body.pop('store')
if store is not False and store is not None:raise CompatibilityError('store=true wird nicht unterstützt; Deck speichert keine Completions.')
ignored.append('store')
if 'service_tier' in body:
if body.pop('service_tier') not in (None,'auto','default'):raise CompatibilityError('Nur service_tier=auto/default wird unterstützt.')
ignored.append('service_tier')
for key in ('stop','stream_options','logprobs','top_logprobs','tools','tool_choice','response_format','logit_bias'):
if body.get(key,'absent') is None:body.pop(key,None)
if 'max_completion_tokens' in body:
value=body.pop('max_completion_tokens')
if value is not None:
if 'max_tokens' in body and body['max_tokens']!=value:raise CompatibilityError('max_tokens und max_completion_tokens widersprechen sich.')
body['max_tokens']=value
if 'functions' in body:
value=body.pop('functions')
if not isinstance(value,list) or any(not isinstance(f,dict) for f in value):raise CompatibilityError('functions muss eine Liste von Funktionsdefinitionen sein.')
translated=[{'type':'function','function':f} for f in value]
if 'tools' in body and body['tools']!=translated:raise CompatibilityError('functions und tools widersprechen sich.')
body['tools']=translated
if 'function_call' in body:
value=body.pop('function_call')
if isinstance(value,dict) and set(value)=={'name'} and isinstance(value['name'],str):value={'type':'function','function':value}
elif value not in ('auto','none'):raise CompatibilityError('Ungültiges function_call.')
if 'tool_choice' in body and body['tool_choice']!=value:raise CompatibilityError('function_call und tool_choice widersprechen sich.')
body['tool_choice']=value
for key,lo,hi in [('min_p',0,1),('typical_p',0,1),('mirostat_tau',0,100),('mirostat_eta',0,1),('dynatemp_range',0,100),('dynatemp_exponent',0,100)]:
if key in body and (type(body[key]) not in (int,float) or not math.isfinite(body[key]) or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
for key,lo,hi in [('repeat_last_n',-1,2097152),('mirostat',0,2),('max_tokens',1,2097152)]:
if key in body and (type(body[key]) is not int or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.')
for key in ('cache_prompt','ignore_eos','parallel_tool_calls'):
if key in body and type(body[key]) is not bool:raise CompatibilityError(key+' muss true oder false sein.')
if 'reasoning_format' in body and body['reasoning_format'] not in ('none','deepseek','auto'):raise CompatibilityError('Ungültiges reasoning_format.')
if 'logit_bias' in body:
biases=body['logit_bias']
if not isinstance(biases,dict) or len(biases)>4096 or any(not isinstance(k,str) or not re.fullmatch(r'[0-9]{1,9}',k) or type(v) not in (int,float) or not math.isfinite(v) or not -100<=v<=100 for k,v in biases.items()):raise CompatibilityError('Ungültiges logit_bias.')
if 'chat_template_kwargs' in body: if 'chat_template_kwargs' in body:
kwargs=body['chat_template_kwargs'] kwargs=body['chat_template_kwargs']
if not isinstance(kwargs,dict) or set(kwargs)-{'enable_thinking','preserve_thinking'}: if not isinstance(kwargs,dict) or set(kwargs)-{'enable_thinking','preserve_thinking'}:
@@ -32,8 +93,8 @@ def normalize_chat(data):
requested=body.get('reasoning_effort') requested=body.get('reasoning_effort')
if requested is None: if requested is None:
body.pop('reasoning_effort',None) body.pop('reasoning_effort',None)
return body,{} return body,({'ignored':ignored} if ignored else {})
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys(): if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.') raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'} return body,{'ignored':ignored,'requested':requested,'effective':effective,'semantics':'model-template-hint'}
+6 -1
View File
@@ -169,7 +169,9 @@ class APIHandler(BaseHTTPRequestHandler):
def log_message(self,*args):pass def log_message(self,*args):pass
def compatibility_headers(self): def compatibility_headers(self):
info=getattr(self,'compatibility',{}) info=getattr(self,'compatibility',{})
if info: if info.get('ignored'):
self.send_header('X-Athena-Compatibility-Ignored',','.join(info['ignored']))
if 'requested' in info:
self.send_header('X-Athena-Reasoning-Requested',info['requested']) self.send_header('X-Athena-Reasoning-Requested',info['requested'])
self.send_header('X-Athena-Reasoning-Effective',info['effective']) self.send_header('X-Athena-Reasoning-Effective',info['effective'])
self.send_header('X-Athena-Reasoning-Semantics',info['semantics']) self.send_header('X-Athena-Reasoning-Semantics',info['semantics'])
@@ -209,6 +211,9 @@ class APIHandler(BaseHTTPRequestHandler):
try:ep.video.relay(self) if callable(getattr(type(ep.video),'relay',None)) else relay(self,video['selected']) try:ep.video.relay(self) if callable(getattr(type(ep.video),'relay',None)) else relay(self,video['selected'])
except ValueError as exc:raise APIError(str(exc)) from None except ValueError as exc:raise APIError(str(exc)) from None
return return
if self.command=='GET' and self.path=='/v1/capabilities':
from api_compat import capabilities
return self.send(capabilities())
model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt','/v1/audio/music/models':'music'} model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt','/v1/audio/music/models':'music'}
if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path])) if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path]))
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'}) if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
+28
View File
@@ -23,3 +23,31 @@ class CompatibilityTests(unittest.TestCase):
self.assertFalse(original['chat_template_kwargs']['enable_thinking']) self.assertFalse(original['chat_template_kwargs']['enable_thinking'])
for bad in [None,[],{'enable_thinking':'false'},{'template':'custom'},{'preserve_thinking':1}]: for bad in [None,[],{'enable_thinking':'false'},{'template':'custom'},{'preserve_thinking':1}]:
with self.assertRaises(CompatibilityError):normalize_chat({'chat_template_kwargs':bad}) with self.assertRaises(CompatibilityError):normalize_chat({'chat_template_kwargs':bad})
def test_openai_aliases_and_legacy_tools(self):
request={'max_completion_tokens':32,'functions':[{'name':'echo','parameters':{'type':'object'}}],'function_call':{'name':'echo'},'stop':None}
body,_=normalize_chat(request)
self.assertEqual(body['max_tokens'],32)
self.assertEqual(body['tools'][0]['function']['name'],'echo')
self.assertEqual(body['tool_choice']['function']['name'],'echo')
self.assertNotIn('stop',body)
self.assertIn('functions',request)
def test_sdk_extra_body_and_conflicts(self):
body,_=normalize_chat({'extra_body':{'min_p':.05,'chat_template_kwargs':{'enable_thinking':False}},'max_tokens':8})
self.assertEqual(body['min_p'],.05)
for request in [{'extra_body':{'model':'other'}},{'extra_body':{'min_p':.1},'min_p':.2},{'max_tokens':8,'max_completion_tokens':16},{'functions':[],'tools':[{}]}]:
with self.assertRaises(CompatibilityError):normalize_chat(request)
def test_metadata_is_explicitly_reported_not_forwarded(self):
body,info=normalize_chat({'store':False,'metadata':{'project':'synthetic'},'service_tier':'default'})
self.assertEqual(body,{})
self.assertEqual(set(info['ignored']),{'store','metadata','service_tier'})
for request in [{'store':True},{'service_tier':'priority'},{'metadata':{'x':42}}]:
with self.assertRaises(CompatibilityError):normalize_chat(request)
def test_llama_validation(self):
body,_=normalize_chat({'repeat_last_n':-1,'mirostat':2,'mirostat_eta':.1,'typical_p':1,'cache_prompt':True,'reasoning_format':'auto','logit_bias':{'123':-10}})
self.assertEqual(body['logit_bias'],{'123':-10})
for request in [{'min_p':float('nan')},{'cache_prompt':1},{'repeat_last_n':.5},{'logit_bias':{'not-token':1}},{'mirostat':3},{'reasoning_format':'invented'}]:
with self.assertRaises(CompatibilityError):normalize_chat(request)
+14
View File
@@ -112,6 +112,20 @@ class EndpointTests(unittest.TestCase):
self.assertEqual(self.request(route,token='bad')[0],401) self.assertEqual(self.request(route,token='bad')[0],401)
self.rows[-1]['runnable']=False self.rows[-1]['runnable']=False
self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[]) self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[])
def test_extended_compatibility_forwarding(self):
self.enable('alpha')
data=dict(model='alpha',messages=[dict(role='user',content='synthetic')],max_completion_tokens=8,extra_body={'min_p':.05},metadata={'project':'synthetic'},store=False,functions=[{'name':'echo','parameters':{'type':'object'}}],function_call='auto',chat_template_kwargs={'enable_thinking':False})
status,_=self.request('/v1/chat/completions',data)
self.assertEqual(status,200)
forwarded=self.worker.requests[-1]
self.assertEqual(forwarded['max_tokens'],8)
self.assertEqual(forwarded['min_p'],.05)
self.assertEqual(forwarded['tools'][0]['function']['name'],'echo')
self.assertNotIn('metadata',forwarded)
status,body=self.request('/v1/capabilities')
self.assertEqual(status,200)
self.assertIn('min_p',body['llama_extensions'])
def test_reasoning_effort_forwarded_and_validated(self): def test_reasoning_effort_forwarded_and_validated(self):
self.enable('alpha') self.enable('alpha')
for effort in ('none','minimal','low','medium','high','xhigh','max','ultra',None): for effort in ('none','minimal','low','medium','high','xhigh','max','ultra',None):