Separate modality model lists and forward chat reasoning effort

This commit is contained in:
Mikei386
2026-09-29 09:09:59 +02:00
parent 27012165eb
commit baad2b11a5
3 changed files with 32 additions and 6 deletions
+7 -1
View File
@@ -42,7 +42,7 @@ Alle Inferenzrouten benötigen `Authorization: Bearer <Deck-API-Token>`.
| Methode | Pfad | Verhalten |
|---|---|---|
| GET | `/v1/models` | Nur freigegebene, ausführbare Profilnamen; kein Modellstart |
| GET | `/v1/models` | Nur freigegebene, ausführbare Chatprofile; kein Modellstart |
| GET | `/health` | Listener-Health, ebenfalls authentifiziert |
| POST | `/v1/chat/completions` | Textchat, JSON und SSE-Streaming, Sampling-Defaults aus dem Profil |
| POST | `/v1/images/generations` | Qwen-Image-2.1-Rezept, n=1, `b64_json` |
@@ -166,3 +166,9 @@ Profilfreigaben: mehrere Chatprofile, höchstens ein Bildprofil. Die Auswahl ein
### Spracherkennung
`POST /v1/audio/transcriptions` benötigt Bearer-Token und ein explizit freigegebenes STT-Profil. Multipart-Felder: `model` (API-Profilname), `file` (PCM16-WAV, mono, 16 kHz, maximal 120 Sekunden / 8 MiB), optional `language` (`de`, `en`, `auto`, Standard `de`), `response_format` (`json`). Antwort: `{"text":"…"}`. Andere Formate, Chunked-Uploads, Zeitstempel und Streaming werden abgelehnt. Ein STT-Auftrag zur Zeit; CPU-Worker wird danach beendet. Keine Nutzung des alten Routers.
### Modelllisten für Clients
`GET /v1/models` listet ausschließlich Chatprofile, damit Chatclients keine Bild-/Audio-Profile anbieten. Deck-Erweiterungen: `GET /v1/images/models` für Bildprofile, `GET /v1/audio/speech/models` für TTS und `GET /v1/audio/transcriptions/models` für STT. Alle Listen erfordern denselben Bearer-Token und enthalten nur freigegebene, ausführbare Profile. Die Inferenzrouten bleiben unverändert. Die Verwaltungs-API liefert weiterhin die Gesamtübersicht.
Chat akzeptiert `reasoning_effort`: `none`, `minimal`, `low`, `medium`, `high`, `xhigh` werden unverändert an llama.cpp weitergereicht; `null` wird wie ein fehlendes Feld behandelt. Der installierte Build unterstützt das Feld; die konkrete Wirkung hängt vom Modell-Chattemplate ab. `none` deaktiviert dort das Reasoning. Deck erfindet keine Tokenbudgets und ignoriert gesetzte Werte nicht stillschweigend.
+7 -4
View File
@@ -115,8 +115,8 @@ class Endpoint:
if not row:raise APIError('Modellprofil nicht aktiviert oder unbekannt.',404,'model_not_found')
if not row['runnable']:raise APIError('Profil derzeit nicht ausführbar: '+' '.join(row['blockers']),503,'model_unavailable')
return row
def model_list(self):
return dict(object='list',data=[dict(id=p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable']])
def model_list(self,kind="chat"):
return dict(object='list',data=[dict(id=p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable'] and p['kind']==kind])
class APIHTTPServer(ThreadingHTTPServer):
daemon_threads=True
@@ -151,7 +151,8 @@ class APIHandler(BaseHTTPRequestHandler):
with ep.lock:
if not ep.allowed():raise APIError('Endpunkt wird gestoppt.',503,'endpoint_stopping')
ep.inflight+=1;admitted=True
if self.command=='GET' and self.path=='/v1/models':return self.send(ep.model_list())
model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt'}
if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path]))
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
if self.command!='POST':raise APIError('Route nicht gefunden.',404)
if self.path=='/v1/audio/transcriptions':return self.transcription(ep)
@@ -209,8 +210,9 @@ class APIHandler(BaseHTTPRequestHandler):
self.sent=True;self.send_response(200);self.send_header('Content-Type','audio/wav');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);return
raise APIError('TTS-Zeitlimit überschritten.',504)
def chat(self,ep,data):
supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n'}
supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'}
if set(data)-supported:raise APIError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(set(data)-supported)))
if 'reasoning_effort' in data and data['reasoning_effort'] is not None and (not isinstance(data['reasoning_effort'],str) or data['reasoning_effort'] not in ('none','minimal','low','medium','high','xhigh')):raise APIError('Ungültiger reasoning_effort-Wert.')
if type(data.get('stream',False)) is not bool:raise APIError('stream muss true oder false sein.')
if data.get('n',1)!=1:raise APIError('Zunächst wird n=1 unterstützt.')
messages=data.get('messages')
@@ -238,6 +240,7 @@ class APIHandler(BaseHTTPRequestHandler):
def allowed():return ep.allowed() and any(p['id']==profile['id'] and p['revision']==profile['revision'] and p['enabled'] for p in ep.rows())
with ep.scheduler.lease(key,profile['parameters']['slots'],lambda:ep.worker.ensure(profile),allowed=allowed):
body=dict(data)
if body.get('reasoning_effort') is None:body.pop('reasoning_effort',None)
for field in ('temperature','top_p','top_k'):body.setdefault(field,profile['parameters'][field])
conn,key=ep.worker.connect()
try:
+18 -1
View File
@@ -55,6 +55,23 @@ class EndpointTests(unittest.TestCase):
with self.assertRaises(ValueError):self.enable('audio')
self.assertEqual(self.request('/v1/audio/speech',{})[0],501)
self.record['api_token_hash']=hashlib.sha256(b'B'*32).hexdigest();self.assertEqual(self.request()[0],401);self.assertEqual(self.request(token='B'*32)[0],200)
def test_model_lists_separate_modalities(self):
self.rows[-1].update(runnable=True,blockers=[])
self.rows.append(dict(id='stt',name='stt',kind='stt',runnable=True,blockers=[],parameters={},updated_at=1))
for name in ('alpha','image','audio','stt'):self.enable(name)
for route,expected in [('/v1/models','alpha'),('/v1/images/models','image'),('/v1/audio/speech/models','audio'),('/v1/audio/transcriptions/models','stt')]:
self.assertEqual([p['id'] for p in self.request(route)[1]['data']],[expected])
self.assertEqual(self.request(route,token='bad')[0],401)
self.rows[-1]['runnable']=False
self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[])
def test_reasoning_effort_forwarded_and_validated(self):
self.enable('alpha')
for effort in ('none','minimal','low','medium','high','xhigh',None):
status,_=self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='synthetic')],reasoning_effort=effort))
self.assertEqual(status,200)
self.assertEqual(self.worker.requests[-1].get('reasoning_effort'),effort)
for effort in ([],True,3,'invalid'):
self.assertEqual(self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='synthetic')],reasoning_effort=effort))[0],400)
def test_stt_endpoint_multipart_and_publication(self):
from test_stt import audio
self.rows.append(dict(id='stt',name='stt',kind='stt',runnable=True,blockers=[],parameters={}))
@@ -78,7 +95,7 @@ class EndpointTests(unittest.TestCase):
self.rows.append(dict(self.rows[2],id='image2',name='image2'))
self.enable('alpha');self.enable('beta');self.enable('image');self.enable('image2')
self.assertEqual(set(self.ep.config['enabled_profiles']),{'alpha','beta','image2'})
self.assertEqual({r['id'] for r in self.request()[1]['data']},{'alpha','beta','image2'})
self.assertEqual({r['id'] for r in self.request()[1]['data']},{'alpha','beta'})
self.ep.enable(dict(id='image2',enabled=False));self.assertEqual(set(self.ep.config['enabled_profiles']),{'alpha','beta'})
def test_invalid_image_selection_keeps_previous(self):
self.rows.append(dict(self.rows[2],id='badimage',name='badimage',runnable=False,blockers=['missing']))