Isolate client reasoning normalization in API compatibility adapter
This commit is contained in:
1 parent
f337ade65a
commit
dad67acad8
8 files changed
+62
-12
No files matched your search
+7
-1
@@ -171,10 +171,16 @@ Profilfreigaben: mehrere Chatprofile, höchstens ein Bildprofil. Die Auswahl ein
|
||||
|
||||
`GET /v1/models` listet ausschließlich Chatprofile, damit Chatclients keine Bild-/Audio-Profile anbieten. Deck-Erweiterungen: `GET /v1/images/models` für Bildprofile, `GET /v1/audio/speech/models` für TTS und `GET /v1/audio/transcriptions/models` für STT. Alle Listen erfordern denselben Bearer-Token und enthalten nur freigegebene, ausführbare Profile. Die Inferenzrouten bleiben unverändert. Die Verwaltungs-API liefert weiterhin die Gesamtübersicht.
|
||||
|
||||
Chat akzeptiert `reasoning_effort`: `none`, `minimal`, `low`, `medium`, `high`, `xhigh` werden unverändert an llama.cpp weitergereicht; `null` wird wie ein fehlendes Feld behandelt. Der installierte Build unterstützt das Feld; die konkrete Wirkung hängt vom Modell-Chattemplate ab. `none` deaktiviert dort das Reasoning. Deck erfindet keine Tokenbudgets und ignoriert gesetzte Werte nicht stillschweigend.
|
||||
Chat akzeptiert `reasoning_effort`: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max` werden unverändert an llama.cpp weitergereicht; `null` wird wie ein fehlendes Feld behandelt. Der installierte Build unterstützt das Feld; die konkrete Wirkung hängt vom Modell-Chattemplate ab. `none` deaktiviert dort das Reasoning. Deck erfindet keine Tokenbudgets und ignoriert gesetzte Werte nicht stillschweigend.
|
||||
|
||||
### Fester Bildname
|
||||
|
||||
`athena-image` bezeichnet immer das aktuell freigegebene Bildprofil. `/v1/images/models` veröffentlicht ausschließlich diesen Alias, sofern ein ausführbares Bildprofil freigegeben ist. Direkte Namen aktivierter Bildprofile bleiben aus Kompatibilitätsgründen aufrufbar. Ohne Freigabe folgt `model_not_found`. Laufende Generierungen verwenden ihr bereits übernommenes Profil; wartende Anfragen werden bei Profiländerungen abgelehnt und können erneut gesendet werden.
|
||||
|
||||
Die Bildgröße wird aus dem Profil übernommen. `size` (z. B. `1536x1024` oder `auto`) ist ein Wunsch und verändert weder Profil noch Speicherbedarf. Die JSON-Antwort enthält zusätzlich `athena_deck.size`, `requested_size` und `size_policy: profile`. Es erfolgt keine automatische Skalierung oder Änderung des Seitenverhältnisses. In Hermes einmalig `image_gen.openai.model: athena-image` setzen.
|
||||
|
||||
### API-Kompatibilität
|
||||
|
||||
`api_compat.py` normalisiert Chat-Anfragen unabhängig von Client, Modellprofilen und Prozesssteuerung. Standard-Effortwerte bis `max` werden als Template-Hinweis weitergereicht. Die Erweiterung `ultra` wird auf `max` abgebildet, unabhängig davon, welcher Client sie sendet. Fehlend oder null bleibt ungesetzt; `none` wird unverändert weitergereicht. Es werden keine Tokenbudgets erfunden und keine Modellprofile verändert.
|
||||
|
||||
Bei gesetztem Effort melden JSON- und SSE-Antworten die Header `X-Athena-Reasoning-Requested`, `X-Athena-Reasoning-Effective` und `X-Athena-Reasoning-Semantics: model-template-hint`. Der installierte llama.cpp-Build reicht positive Werte an das Chattemplate weiter; das garantiert keine unterschiedlichen Denkstufen bei Qwen. Die Wirksamkeit wird nicht aus dem Profilnamen abgeleitet. Weitere Protokollübersetzungen gehören in diesen Adapter.
|
||||
@@ -0,0 +1,23 @@
|
||||
"""Client-neutral API normalization, separate from profiles and inference workers.
|
||||
|
||||
Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards
|
||||
positive levels to the model template; it handles 'none' as thinking disabled.
|
||||
"""
|
||||
CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'})
|
||||
LLAMA_EFFORTS=frozenset({'none','minimal','low','medium','high','xhigh','max'})
|
||||
EFFORT_ALIASES={'ultra':'max'}
|
||||
|
||||
class CompatibilityError(ValueError):pass
|
||||
|
||||
def normalize_chat(data):
|
||||
"""Return a new request and non-sensitive translation metadata; never mutate input."""
|
||||
unknown=set(data)-CHAT_FIELDS
|
||||
if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown)))
|
||||
body=dict(data);requested=body.get('reasoning_effort')
|
||||
if requested is None:
|
||||
body.pop('reasoning_effort',None)
|
||||
return body,{}
|
||||
if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys():
|
||||
raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.')
|
||||
effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective
|
||||
return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'}
|
||||
+1
-1
@@ -13,7 +13,7 @@ RUN git clone https://github.com/Comfy-Org/ComfyUI.git /opt/deck-comfy \
|
||||
WORKDIR /app
|
||||
COPY deploy/image-requirements.lock /app/deploy/image-requirements.lock
|
||||
COPY deploy/tts-requirements.lock /app/deploy/tts-requirements.lock
|
||||
COPY stt.py execution_setup.py tts_runtime.py tts_test.py tts_worker.py auto_test.py chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/
|
||||
COPY api_compat.py stt.py execution_setup.py tts_runtime.py tts_test.py tts_worker.py auto_test.py chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/
|
||||
COPY stt-ui.js tts-ui.js auto-test-ui.js chat-test-ui.js endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/
|
||||
COPY network/__init__.py network/client.py network/config.py network/rpc.py /app/network/
|
||||
ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 HOME=/tmp \
|
||||
|
||||
+1
-1
@@ -14,7 +14,7 @@ import urllib.request
|
||||
|
||||
ROOT = Path(__file__).resolve().parent.parent
|
||||
LABEL = 'de.casaderoll.athena-deck.standalone'
|
||||
FILES = ['deploy/tts-requirements.lock','stt.py','stt-ui.js','execution_setup.py','tts_runtime.py','tts_test.py','tts_worker.py','tts-ui.js','auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock']
|
||||
FILES = ['deploy/tts-requirements.lock','stt.py','stt-ui.js','api_compat.py','execution_setup.py','tts_runtime.py','tts_test.py','tts_worker.py','tts-ui.js','auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock']
|
||||
|
||||
|
||||
def run(*args, check=True, interactive=False):
|
||||
|
||||
+11
-6
@@ -10,6 +10,7 @@ import socket
|
||||
import threading
|
||||
import time
|
||||
from http.server import BaseHTTPRequestHandler,ThreadingHTTPServer
|
||||
from api_compat import normalize_chat,CompatibilityError
|
||||
from inference import InferenceError
|
||||
from stt import read_upload
|
||||
|
||||
@@ -136,9 +137,15 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
protocol_version='HTTP/1.1'
|
||||
def setup(self):super().setup();self.connection.settimeout(15);self.sent=False
|
||||
def log_message(self,*args):pass
|
||||
def compatibility_headers(self):
|
||||
info=getattr(self,'compatibility',{})
|
||||
if info:
|
||||
self.send_header('X-Athena-Reasoning-Requested',info['requested'])
|
||||
self.send_header('X-Athena-Reasoning-Effective',info['effective'])
|
||||
self.send_header('X-Athena-Reasoning-Semantics',info['semantics'])
|
||||
def send(self,payload,status=200):
|
||||
body=json.dumps(payload).encode();self.sent=True
|
||||
self.send_response(status);self.send_header('Content-Type','application/json');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);self.close_connection=True
|
||||
self.send_response(status);self.compatibility_headers();self.send_header('Content-Type','application/json');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);self.close_connection=True
|
||||
def failure(self,exc):
|
||||
if self.sent:return
|
||||
self.send({'error':{'message':str(exc),'type':getattr(exc,'code','server_error'),'param':None,'code':getattr(exc,'code','worker_unavailable')}},getattr(exc,'status',503))
|
||||
@@ -211,9 +218,8 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
self.sent=True;self.send_response(200);self.send_header('Content-Type','audio/wav');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);return
|
||||
raise APIError('TTS-Zeitlimit überschritten.',504)
|
||||
def chat(self,ep,data):
|
||||
supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'}
|
||||
if set(data)-supported:raise APIError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(set(data)-supported)))
|
||||
if 'reasoning_effort' in data and data['reasoning_effort'] is not None and (not isinstance(data['reasoning_effort'],str) or data['reasoning_effort'] not in ('none','minimal','low','medium','high','xhigh')):raise APIError('Ungültiger reasoning_effort-Wert.')
|
||||
try:data,self.compatibility=normalize_chat(data)
|
||||
except CompatibilityError as exc:raise APIError(str(exc)) from None
|
||||
if type(data.get('stream',False)) is not bool:raise APIError('stream muss true oder false sein.')
|
||||
if data.get('n',1)!=1:raise APIError('Zunächst wird n=1 unterstützt.')
|
||||
messages=data.get('messages')
|
||||
@@ -241,7 +247,6 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
def allowed():return ep.allowed() and any(p['id']==profile['id'] and p['revision']==profile['revision'] and p['enabled'] for p in ep.rows())
|
||||
with ep.scheduler.lease(key,profile['parameters']['slots'],lambda:ep.worker.ensure(profile),allowed=allowed):
|
||||
body=dict(data)
|
||||
if body.get('reasoning_effort') is None:body.pop('reasoning_effort',None)
|
||||
for field in ('temperature','top_p','top_k'):body.setdefault(field,profile['parameters'][field])
|
||||
conn,key=ep.worker.connect()
|
||||
try:
|
||||
@@ -254,7 +259,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
if len(raw)>16*1024*1024:raise InferenceError('Modellantwort überschreitet 16 MiB.')
|
||||
return self.send(json.loads(raw))
|
||||
self.sent=True;self.connection.settimeout(30)
|
||||
self.send_response(200);self.send_header('Content-Type','text/event-stream');self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers()
|
||||
self.send_response(200);self.compatibility_headers();self.send_header('Content-Type','text/event-stream');self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers()
|
||||
deadline=time.monotonic()+600
|
||||
while time.monotonic()<deadline:
|
||||
chunk=response.read1(8192)
|
||||
|
||||
@@ -12,7 +12,7 @@ import sys
|
||||
|
||||
NAME = 'athena-deck-network'
|
||||
BASE = Path('/opt/athena-deck')
|
||||
FILES = {'deploy/tts-requirements.lock','stt.py','stt-ui.js','execution_setup.py','tts_runtime.py','tts_test.py','tts_worker.py','tts-ui.js','auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'}
|
||||
FILES = {'deploy/tts-requirements.lock','stt.py','stt-ui.js','api_compat.py','execution_setup.py','tts_runtime.py','tts_test.py','tts_worker.py','tts-ui.js','auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'}
|
||||
|
||||
def run(*args, **kwargs):
|
||||
return subprocess.run(args, capture_output=True, timeout=600, **kwargs)
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
import unittest
|
||||
from api_compat import normalize_chat,CompatibilityError
|
||||
class CompatibilityTests(unittest.TestCase):
|
||||
def test_levels_and_alias_without_mutation(self):
|
||||
for value in ['none','minimal','low','medium','high','xhigh','max','ultra']:
|
||||
request={'model':'any','reasoning_effort':value};body,info=normalize_chat(request)
|
||||
self.assertEqual(request['reasoning_effort'],value)
|
||||
self.assertEqual(body['reasoning_effort'],'max' if value=='ultra' else value)
|
||||
self.assertEqual(info['requested'],value)
|
||||
def test_unset_and_null_do_not_invent_defaults(self):
|
||||
for request in [{},{'reasoning_effort':None}]:
|
||||
self.assertEqual(normalize_chat(request),({},{}))
|
||||
def test_invalid_inputs(self):
|
||||
for value in [True,1,{},[], '', 'MAX', 'x\r\ny']:
|
||||
with self.assertRaises(CompatibilityError):normalize_chat({'reasoning_effort':value})
|
||||
with self.assertRaises(CompatibilityError):normalize_chat({'unknown':1})
|
||||
+2
-2
@@ -66,10 +66,10 @@ class EndpointTests(unittest.TestCase):
|
||||
self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[])
|
||||
def test_reasoning_effort_forwarded_and_validated(self):
|
||||
self.enable('alpha')
|
||||
for effort in ('none','minimal','low','medium','high','xhigh',None):
|
||||
for effort in ('none','minimal','low','medium','high','xhigh','max','ultra',None):
|
||||
status,_=self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='synthetic')],reasoning_effort=effort))
|
||||
self.assertEqual(status,200)
|
||||
self.assertEqual(self.worker.requests[-1].get('reasoning_effort'),effort)
|
||||
self.assertEqual(self.worker.requests[-1].get('reasoning_effort'),'max' if effort=='ultra' else effort)
|
||||
for effort in ([],True,3,'invalid'):
|
||||
self.assertEqual(self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='synthetic')],reasoning_effort=effort))[0],400)
|
||||
def test_stt_endpoint_multipart_and_publication(self):
|
||||
|
||||
Reference in new issue
Block a user