diff --git a/ENDPOINT.md b/ENDPOINT.md index 25ae542..8417392 100644 --- a/ENDPOINT.md +++ b/ENDPOINT.md @@ -171,10 +171,16 @@ Profilfreigaben: mehrere Chatprofile, höchstens ein Bildprofil. Die Auswahl ein `GET /v1/models` listet ausschließlich Chatprofile, damit Chatclients keine Bild-/Audio-Profile anbieten. Deck-Erweiterungen: `GET /v1/images/models` für Bildprofile, `GET /v1/audio/speech/models` für TTS und `GET /v1/audio/transcriptions/models` für STT. Alle Listen erfordern denselben Bearer-Token und enthalten nur freigegebene, ausführbare Profile. Die Inferenzrouten bleiben unverändert. Die Verwaltungs-API liefert weiterhin die Gesamtübersicht. -Chat akzeptiert `reasoning_effort`: `none`, `minimal`, `low`, `medium`, `high`, `xhigh` werden unverändert an llama.cpp weitergereicht; `null` wird wie ein fehlendes Feld behandelt. Der installierte Build unterstützt das Feld; die konkrete Wirkung hängt vom Modell-Chattemplate ab. `none` deaktiviert dort das Reasoning. Deck erfindet keine Tokenbudgets und ignoriert gesetzte Werte nicht stillschweigend. +Chat akzeptiert `reasoning_effort`: `none`, `minimal`, `low`, `medium`, `high`, `xhigh`, `max` werden unverändert an llama.cpp weitergereicht; `null` wird wie ein fehlendes Feld behandelt. Der installierte Build unterstützt das Feld; die konkrete Wirkung hängt vom Modell-Chattemplate ab. `none` deaktiviert dort das Reasoning. Deck erfindet keine Tokenbudgets und ignoriert gesetzte Werte nicht stillschweigend. ### Fester Bildname `athena-image` bezeichnet immer das aktuell freigegebene Bildprofil. `/v1/images/models` veröffentlicht ausschließlich diesen Alias, sofern ein ausführbares Bildprofil freigegeben ist. Direkte Namen aktivierter Bildprofile bleiben aus Kompatibilitätsgründen aufrufbar. Ohne Freigabe folgt `model_not_found`. Laufende Generierungen verwenden ihr bereits übernommenes Profil; wartende Anfragen werden bei Profiländerungen abgelehnt und können erneut gesendet werden. Die Bildgröße wird aus dem Profil übernommen. `size` (z. B. `1536x1024` oder `auto`) ist ein Wunsch und verändert weder Profil noch Speicherbedarf. Die JSON-Antwort enthält zusätzlich `athena_deck.size`, `requested_size` und `size_policy: profile`. Es erfolgt keine automatische Skalierung oder Änderung des Seitenverhältnisses. In Hermes einmalig `image_gen.openai.model: athena-image` setzen. + +### API-Kompatibilität + +`api_compat.py` normalisiert Chat-Anfragen unabhängig von Client, Modellprofilen und Prozesssteuerung. Standard-Effortwerte bis `max` werden als Template-Hinweis weitergereicht. Die Erweiterung `ultra` wird auf `max` abgebildet, unabhängig davon, welcher Client sie sendet. Fehlend oder null bleibt ungesetzt; `none` wird unverändert weitergereicht. Es werden keine Tokenbudgets erfunden und keine Modellprofile verändert. + +Bei gesetztem Effort melden JSON- und SSE-Antworten die Header `X-Athena-Reasoning-Requested`, `X-Athena-Reasoning-Effective` und `X-Athena-Reasoning-Semantics: model-template-hint`. Der installierte llama.cpp-Build reicht positive Werte an das Chattemplate weiter; das garantiert keine unterschiedlichen Denkstufen bei Qwen. Die Wirksamkeit wird nicht aus dem Profilnamen abgeleitet. Weitere Protokollübersetzungen gehören in diesen Adapter. diff --git a/api_compat.py b/api_compat.py new file mode 100644 index 0000000..9fde386 --- /dev/null +++ b/api_compat.py @@ -0,0 +1,23 @@ +"""Client-neutral API normalization, separate from profiles and inference workers. + +Effort is a template hint, not a guaranteed compute budget. llama.cpp forwards +positive levels to the model template; it handles 'none' as thinking disabled. +""" +CHAT_FIELDS=frozenset({'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'}) +LLAMA_EFFORTS=frozenset({'none','minimal','low','medium','high','xhigh','max'}) +EFFORT_ALIASES={'ultra':'max'} + +class CompatibilityError(ValueError):pass + +def normalize_chat(data): + """Return a new request and non-sensitive translation metadata; never mutate input.""" + unknown=set(data)-CHAT_FIELDS + if unknown:raise CompatibilityError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(unknown))) + body=dict(data);requested=body.get('reasoning_effort') + if requested is None: + body.pop('reasoning_effort',None) + return body,{} + if not isinstance(requested,str) or requested not in LLAMA_EFFORTS|EFFORT_ALIASES.keys(): + raise CompatibilityError('Ungültiger reasoning_effort-Wert. Erlaubt: none, minimal, low, medium, high, xhigh, max, ultra.') + effective=EFFORT_ALIASES.get(requested,requested);body['reasoning_effort']=effective + return body,{'requested':requested,'effective':effective,'semantics':'model-template-hint'} diff --git a/deploy/Dockerfile b/deploy/Dockerfile index 92fc387..f97781f 100644 --- a/deploy/Dockerfile +++ b/deploy/Dockerfile @@ -13,7 +13,7 @@ RUN git clone https://github.com/Comfy-Org/ComfyUI.git /opt/deck-comfy \ WORKDIR /app COPY deploy/image-requirements.lock /app/deploy/image-requirements.lock COPY deploy/tts-requirements.lock /app/deploy/tts-requirements.lock -COPY stt.py execution_setup.py tts_runtime.py tts_test.py tts_worker.py auto_test.py chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/ +COPY api_compat.py stt.py execution_setup.py tts_runtime.py tts_test.py tts_worker.py auto_test.py chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/ COPY stt-ui.js tts-ui.js auto-test-ui.js chat-test-ui.js endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/ COPY network/__init__.py network/client.py network/config.py network/rpc.py /app/network/ ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 HOME=/tmp \ diff --git a/deploy/install.py b/deploy/install.py index 0985efe..6746fa3 100644 --- a/deploy/install.py +++ b/deploy/install.py @@ -14,7 +14,7 @@ import urllib.request ROOT = Path(__file__).resolve().parent.parent LABEL = 'de.casaderoll.athena-deck.standalone' -FILES = ['deploy/tts-requirements.lock','stt.py','stt-ui.js','execution_setup.py','tts_runtime.py','tts_test.py','tts_worker.py','tts-ui.js','auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock'] +FILES = ['deploy/tts-requirements.lock','stt.py','stt-ui.js','api_compat.py','execution_setup.py','tts_runtime.py','tts_test.py','tts_worker.py','tts-ui.js','auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock'] def run(*args, check=True, interactive=False): diff --git a/endpoint.py b/endpoint.py index bb7ef4f..6a9c118 100644 --- a/endpoint.py +++ b/endpoint.py @@ -10,6 +10,7 @@ import socket import threading import time from http.server import BaseHTTPRequestHandler,ThreadingHTTPServer +from api_compat import normalize_chat,CompatibilityError from inference import InferenceError from stt import read_upload @@ -136,9 +137,15 @@ class APIHandler(BaseHTTPRequestHandler): protocol_version='HTTP/1.1' def setup(self):super().setup();self.connection.settimeout(15);self.sent=False def log_message(self,*args):pass + def compatibility_headers(self): + info=getattr(self,'compatibility',{}) + if info: + self.send_header('X-Athena-Reasoning-Requested',info['requested']) + self.send_header('X-Athena-Reasoning-Effective',info['effective']) + self.send_header('X-Athena-Reasoning-Semantics',info['semantics']) def send(self,payload,status=200): body=json.dumps(payload).encode();self.sent=True - self.send_response(status);self.send_header('Content-Type','application/json');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);self.close_connection=True + self.send_response(status);self.compatibility_headers();self.send_header('Content-Type','application/json');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);self.close_connection=True def failure(self,exc): if self.sent:return self.send({'error':{'message':str(exc),'type':getattr(exc,'code','server_error'),'param':None,'code':getattr(exc,'code','worker_unavailable')}},getattr(exc,'status',503)) @@ -211,9 +218,8 @@ class APIHandler(BaseHTTPRequestHandler): self.sent=True;self.send_response(200);self.send_header('Content-Type','audio/wav');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);return raise APIError('TTS-Zeitlimit überschritten.',504) def chat(self,ep,data): - supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'} - if set(data)-supported:raise APIError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(set(data)-supported))) - if 'reasoning_effort' in data and data['reasoning_effort'] is not None and (not isinstance(data['reasoning_effort'],str) or data['reasoning_effort'] not in ('none','minimal','low','medium','high','xhigh')):raise APIError('Ungültiger reasoning_effort-Wert.') + try:data,self.compatibility=normalize_chat(data) + except CompatibilityError as exc:raise APIError(str(exc)) from None if type(data.get('stream',False)) is not bool:raise APIError('stream muss true oder false sein.') if data.get('n',1)!=1:raise APIError('Zunächst wird n=1 unterstützt.') messages=data.get('messages') @@ -241,7 +247,6 @@ class APIHandler(BaseHTTPRequestHandler): def allowed():return ep.allowed() and any(p['id']==profile['id'] and p['revision']==profile['revision'] and p['enabled'] for p in ep.rows()) with ep.scheduler.lease(key,profile['parameters']['slots'],lambda:ep.worker.ensure(profile),allowed=allowed): body=dict(data) - if body.get('reasoning_effort') is None:body.pop('reasoning_effort',None) for field in ('temperature','top_p','top_k'):body.setdefault(field,profile['parameters'][field]) conn,key=ep.worker.connect() try: @@ -254,7 +259,7 @@ class APIHandler(BaseHTTPRequestHandler): if len(raw)>16*1024*1024:raise InferenceError('Modellantwort überschreitet 16 MiB.') return self.send(json.loads(raw)) self.sent=True;self.connection.settimeout(30) - self.send_response(200);self.send_header('Content-Type','text/event-stream');self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers() + self.send_response(200);self.compatibility_headers();self.send_header('Content-Type','text/event-stream');self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers() deadline=time.monotonic()+600 while time.monotonic()