From 3cd9fa57501ddbc80e9fc6116b814da80808cafe Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Fri, 2 Oct 2026 10:29:32 +0200 Subject: [PATCH] Broaden client-neutral OpenAI and llama.cpp chat compatibility --- API_COMPATIBILITY.md | 47 ++++++++++++++++++++++++++++++++ api_compat.py | 65 ++++++++++++++++++++++++++++++++++++++++++-- endpoint.py | 7 ++++- test_api_compat.py | 28 +++++++++++++++++++ test_endpoint.py | 14 ++++++++++ 5 files changed, 158 insertions(+), 3 deletions(-) create mode 100644 API_COMPATIBILITY.md diff --git a/API_COMPATIBILITY.md b/API_COMPATIBILITY.md new file mode 100644 index 0000000..22dd9f9 --- /dev/null +++ b/API_COMPATIBILITY.md @@ -0,0 +1,47 @@ +# Client-neutral chat compatibility + +Deck exposes authenticated OpenAI-style Chat Completions at `/v1/chat/completions`. +It is a router, not a complete replica of OpenAI or of every llama-server route. +The compatibility layer is `api_compat.py`; it does not modify model profiles. + +## Discovery and behaviour + +`GET /v1/capabilities` with the same Bearer token lists accepted request fields, +llama.cpp extensions, aliases and routes not implemented by Deck. This describes +router support, not a guarantee that every model/build supports every function. +Existing `/v1/models` remains a chat-only list. Images, speech and transcription +have their own model-list routes. Request handling and streaming share the same +normalization path; client-specific branches are not used. + +- `max_completion_tokens` becomes `max_tokens`; conflicting limits fail explicitly. +- Legacy `functions` and `function_call` become `tools` and `tool_choice`. +- SDK `extra_body` is flattened, validated and cannot replace model/messages. +- Null optional stop/tools/stream options are omitted. +- `reasoning_effort` aliases retain existing nearest-supported mappings. +- Template options `enable_thinking` and `preserve_thinking` require booleans. +- Additional llama.cpp options: `min_p`, `typical_p`, `repeat_last_n`, `mirostat`, + `mirostat_tau`, `mirostat_eta`, `dynatemp_range`, `dynatemp_exponent`, + `cache_prompt`, `ignore_eos`, `reasoning_format`, and `logit_bias`. + Numeric types, finite values and bounded ranges are validated. +- Metadata, safety identifiers and prompt-cache keys are accepted as annotations, + but not persisted or used for OpenAI-hosted safety/caching semantics. Their + removal is reported in `X-Athena-Compatibility-Ignored` (also on SSE responses). +- `store=false/null` and service tier `auto/default/null` are tolerated and reported; + stored completions and priority service tiers are rejected, not simulated. + +Unknown fields and conflicting aliases still produce explicit HTTP 400 errors. +Only one completion (`n=1`) is supported. Embeddings, Responses, native completion, +slot administration and arbitrary template overrides are not added by this change. +Tools, vision, reasoning and structured output depend on model weights, template +and active runtime version. Tool calls are returned to clients, not executed by Deck. +Model capability declarations must not be inferred solely from a model name. + +## Verification + +Tests cover alias translation, conflicts, input isolation, invalid sampling values, +metadata treatment, upstream forwarding and authenticated capability discovery. +Synthetic/fake-worker API tests do not establish inference quality or actual tool +calling for every installed model. New upstream features require review and tests. + +References: [llama.cpp server API](https://github.com/ggml-org/llama.cpp/blob/master/tools/server/README.md) +and [OpenAI Chat Completions](https://platform.openai.com/docs/api-reference/chat/create). diff --git a/api_compat.py b/api_compat.py index cf5b583..db8d174 100644 --- a/api_compat.py +++ b/api_compat.py @@ -4,8 +4,18 @@ Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards positive levels to the model template; it handles 'none' as thinking disabled. """ import math +import json +import re +from copy import deepcopy CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','repeat_penalty','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort','chat_template_kwargs'}) +LLAMA_FIELDS=frozenset({'min_p','typical_p','repeat_last_n','mirostat','mirostat_tau','mirostat_eta','dynatemp_range','dynatemp_exponent','cache_prompt','ignore_eos','reasoning_format','logit_bias'}) +ADAPTER_FIELDS=frozenset({'functions','function_call','extra_body','metadata','store','service_tier','safety_identifier','prompt_cache_key'}) +CHAT_FIELDS=CHAT_FIELDS|LLAMA_FIELDS|ADAPTER_FIELDS + +def capabilities(): + return dict(version=1,chat_route='/v1/chat/completions',fields=sorted(CHAT_FIELDS),llama_extensions=sorted(LLAMA_FIELDS|{'chat_template_kwargs'}),template_options=['enable_thinking','preserve_thinking'],aliases={'max_completion_tokens':'max_tokens','functions':'tools','function_call':'tool_choice'},unsupported_routes=['/v1/responses','/v1/embeddings','/completion','/slots'],model_dependent=['tools','vision','reasoning','json_schema'],notes=['Only n=1; no stored completions or priority service tier.','Model capabilities and backend-version support are not guaranteed by this field list.']) + LLAMA_EFFORTS=frozenset({'none','low','medium','high','xhigh'}) # The installed llama.cpp server rejects minimal and max. Use nearest supported hints. EFFORT_ALIASES={'minimal':'low','max':'xhigh','ultra':'xhigh'} @@ -14,9 +24,60 @@ class CompatibilityError(ValueError):pass def normalize_chat(data): """Return a new request and non-sensitive translation metadata; never mutate input.""" + if not isinstance(data,dict):raise CompatibilityError('Chat-Anfrage muss ein JSON-Objekt sein.') + data=deepcopy(data) + extra=data.pop('extra_body',{}) + if not isinstance(extra,dict):raise CompatibilityError('extra_body muss ein JSON-Objekt sein.') + if 'extra_body' in extra or set(extra)&{'model','messages'}:raise CompatibilityError('extra_body darf Modell oder Nachrichten nicht überschreiben.') + for key,value in extra.items(): + if key in data and data[key]!=value:raise CompatibilityError('Widersprüchliches Feld in extra_body: '+key) + data[key]=value unknown=set(data)-CHAT_FIELDS if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown))) body=dict(data) + ignored=[] + for key in ('metadata','safety_identifier','prompt_cache_key'): + if key in body: + value=body.pop(key) + if key=='metadata' and value is not None and (not isinstance(value,dict) or len(value)>16 or any(not isinstance(k,str) or len(k)>64 or not isinstance(v,str) or len(v)>512 for k,v in value.items())):raise CompatibilityError('Ungültige metadata.') + if key!='metadata' and value is not None and (not isinstance(value,str) or len(value)>512):raise CompatibilityError('Ungültiger '+key+'.') + ignored.append(key) + if 'store' in body: + store=body.pop('store') + if store is not False and store is not None:raise CompatibilityError('store=true wird nicht unterstützt; Deck speichert keine Completions.') + ignored.append('store') + if 'service_tier' in body: + if body.pop('service_tier') not in (None,'auto','default'):raise CompatibilityError('Nur service_tier=auto/default wird unterstützt.') + ignored.append('service_tier') + for key in ('stop','stream_options','logprobs','top_logprobs','tools','tool_choice','response_format','logit_bias'): + if body.get(key,'absent') is None:body.pop(key,None) + if 'max_completion_tokens' in body: + value=body.pop('max_completion_tokens') + if value is not None: + if 'max_tokens' in body and body['max_tokens']!=value:raise CompatibilityError('max_tokens und max_completion_tokens widersprechen sich.') + body['max_tokens']=value + if 'functions' in body: + value=body.pop('functions') + if not isinstance(value,list) or any(not isinstance(f,dict) for f in value):raise CompatibilityError('functions muss eine Liste von Funktionsdefinitionen sein.') + translated=[{'type':'function','function':f} for f in value] + if 'tools' in body and body['tools']!=translated:raise CompatibilityError('functions und tools widersprechen sich.') + body['tools']=translated + if 'function_call' in body: + value=body.pop('function_call') + if isinstance(value,dict) and set(value)=={'name'} and isinstance(value['name'],str):value={'type':'function','function':value} + elif value not in ('auto','none'):raise CompatibilityError('Ungültiges function_call.') + if 'tool_choice' in body and body['tool_choice']!=value:raise CompatibilityError('function_call und tool_choice widersprechen sich.') + body['tool_choice']=value + for key,lo,hi in [('min_p',0,1),('typical_p',0,1),('mirostat_tau',0,100),('mirostat_eta',0,1),('dynatemp_range',0,100),('dynatemp_exponent',0,100)]: + if key in body and (type(body[key]) not in (int,float) or not math.isfinite(body[key]) or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.') + for key,lo,hi in [('repeat_last_n',-1,2097152),('mirostat',0,2),('max_tokens',1,2097152)]: + if key in body and (type(body[key]) is not int or not lo<=body[key]<=hi):raise CompatibilityError('Ungültiger '+key+'-Wert.') + for key in ('cache_prompt','ignore_eos','parallel_tool_calls'): + if key in body and type(body[key]) is not bool:raise CompatibilityError(key+' muss true oder false sein.') + if 'reasoning_format' in body and body['reasoning_format'] not in ('none','deepseek','auto'):raise CompatibilityError('Ungültiges reasoning_format.') + if 'logit_bias' in body: + biases=body['logit_bias'] + if not isinstance(biases,dict) or len(biases)>4096 or any(not isinstance(k,str) or not re.fullmatch(r'[0-9]{1,9}',k) or type(v) not in (int,float) or not math.isfinite(v) or not -100<=v<=100 for k,v in biases.items()):raise CompatibilityError('Ungültiges logit_bias.') if 'chat_template_kwargs' in body: kwargs=body['chat_template_kwargs'] if not isinstance(kwargs,dict) or set(kwargs)-{'enable_thinking','preserve_thinking'}: @@ -32,8 +93,8 @@ def normalize_chat(data): requested=body.get('reasoning_effort') if requested is None: body.pop('reasoning_effort',None) - return body,{} + return body,({'ignored':ignored} if ignored else {}) if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys(): raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.') effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective - return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'} + return body,{'ignored':ignored,'requested':requested,'effective':effective,'semantics':'model-template-hint'} diff --git a/endpoint.py b/endpoint.py index 3a3eee6..7d4c147 100644 --- a/endpoint.py +++ b/endpoint.py @@ -169,7 +169,9 @@ class APIHandler(BaseHTTPRequestHandler): def log_message(self,*args):pass def compatibility_headers(self): info=getattr(self,'compatibility',{}) - if info: + if info.get('ignored'): + self.send_header('X-Athena-Compatibility-Ignored',','.join(info['ignored'])) + if 'requested' in info: self.send_header('X-Athena-Reasoning-Requested',info['requested']) self.send_header('X-Athena-Reasoning-Effective',info['effective']) self.send_header('X-Athena-Reasoning-Semantics',info['semantics']) @@ -209,6 +211,9 @@ class APIHandler(BaseHTTPRequestHandler): try:ep.video.relay(self) if callable(getattr(type(ep.video),'relay',None)) else relay(self,video['selected']) except ValueError as exc:raise APIError(str(exc)) from None return + if self.command=='GET' and self.path=='/v1/capabilities': + from api_compat import capabilities + return self.send(capabilities()) model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt','/v1/audio/music/models':'music'} if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path])) if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'}) diff --git a/test_api_compat.py b/test_api_compat.py index 6e3ece1..6095d34 100644 --- a/test_api_compat.py +++ b/test_api_compat.py @@ -23,3 +23,31 @@ class CompatibilityTests(unittest.TestCase): self.assertFalse(original['chat_template_kwargs']['enable_thinking']) for bad in [None,[],{'enable_thinking':'false'},{'template':'custom'},{'preserve_thinking':1}]: with self.assertRaises(CompatibilityError):normalize_chat({'chat_template_kwargs':bad}) + + def test_openai_aliases_and_legacy_tools(self): + request={'max_completion_tokens':32,'functions':[{'name':'echo','parameters':{'type':'object'}}],'function_call':{'name':'echo'},'stop':None} + body,_=normalize_chat(request) + self.assertEqual(body['max_tokens'],32) + self.assertEqual(body['tools'][0]['function']['name'],'echo') + self.assertEqual(body['tool_choice']['function']['name'],'echo') + self.assertNotIn('stop',body) + self.assertIn('functions',request) + + def test_sdk_extra_body_and_conflicts(self): + body,_=normalize_chat({'extra_body':{'min_p':.05,'chat_template_kwargs':{'enable_thinking':False}},'max_tokens':8}) + self.assertEqual(body['min_p'],.05) + for request in [{'extra_body':{'model':'other'}},{'extra_body':{'min_p':.1},'min_p':.2},{'max_tokens':8,'max_completion_tokens':16},{'functions':[],'tools':[{}]}]: + with self.assertRaises(CompatibilityError):normalize_chat(request) + + def test_metadata_is_explicitly_reported_not_forwarded(self): + body,info=normalize_chat({'store':False,'metadata':{'project':'synthetic'},'service_tier':'default'}) + self.assertEqual(body,{}) + self.assertEqual(set(info['ignored']),{'store','metadata','service_tier'}) + for request in [{'store':True},{'service_tier':'priority'},{'metadata':{'x':42}}]: + with self.assertRaises(CompatibilityError):normalize_chat(request) + + def test_llama_validation(self): + body,_=normalize_chat({'repeat_last_n':-1,'mirostat':2,'mirostat_eta':.1,'typical_p':1,'cache_prompt':True,'reasoning_format':'auto','logit_bias':{'123':-10}}) + self.assertEqual(body['logit_bias'],{'123':-10}) + for request in [{'min_p':float('nan')},{'cache_prompt':1},{'repeat_last_n':.5},{'logit_bias':{'not-token':1}},{'mirostat':3},{'reasoning_format':'invented'}]: + with self.assertRaises(CompatibilityError):normalize_chat(request) diff --git a/test_endpoint.py b/test_endpoint.py index 49b0046..c3adfdd 100644 --- a/test_endpoint.py +++ b/test_endpoint.py @@ -112,6 +112,20 @@ class EndpointTests(unittest.TestCase): self.assertEqual(self.request(route,token='bad')[0],401) self.rows[-1]['runnable']=False self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[]) + def test_extended_compatibility_forwarding(self): + self.enable('alpha') + data=dict(model='alpha',messages=[dict(role='user',content='synthetic')],max_completion_tokens=8,extra_body={'min_p':.05},metadata={'project':'synthetic'},store=False,functions=[{'name':'echo','parameters':{'type':'object'}}],function_call='auto',chat_template_kwargs={'enable_thinking':False}) + status,_=self.request('/v1/chat/completions',data) + self.assertEqual(status,200) + forwarded=self.worker.requests[-1] + self.assertEqual(forwarded['max_tokens'],8) + self.assertEqual(forwarded['min_p'],.05) + self.assertEqual(forwarded['tools'][0]['function']['name'],'echo') + self.assertNotIn('metadata',forwarded) + status,body=self.request('/v1/capabilities') + self.assertEqual(status,200) + self.assertIn('min_p',body['llama_extensions']) + def test_reasoning_effort_forwarded_and_validated(self): self.enable('alpha') for effort in ('none','minimal','low','medium','high','xhigh','max','ultra',None):