Separate modality model lists and forward chat reasoning effort
This commit is contained in:
+7
-1
@@ -42,7 +42,7 @@ Alle Inferenzrouten benötigen `Authorization: Bearer <Deck-API-Token>`.
|
|||||||
|
|
||||||
| Methode | Pfad | Verhalten |
|
| Methode | Pfad | Verhalten |
|
||||||
|---|---|---|
|
|---|---|---|
|
||||||
| GET | `/v1/models` | Nur freigegebene, ausführbare Profilnamen; kein Modellstart |
|
| GET | `/v1/models` | Nur freigegebene, ausführbare Chatprofile; kein Modellstart |
|
||||||
| GET | `/health` | Listener-Health, ebenfalls authentifiziert |
|
| GET | `/health` | Listener-Health, ebenfalls authentifiziert |
|
||||||
| POST | `/v1/chat/completions` | Textchat, JSON und SSE-Streaming, Sampling-Defaults aus dem Profil |
|
| POST | `/v1/chat/completions` | Textchat, JSON und SSE-Streaming, Sampling-Defaults aus dem Profil |
|
||||||
| POST | `/v1/images/generations` | Qwen-Image-2.1-Rezept, n=1, `b64_json` |
|
| POST | `/v1/images/generations` | Qwen-Image-2.1-Rezept, n=1, `b64_json` |
|
||||||
@@ -166,3 +166,9 @@ Profilfreigaben: mehrere Chatprofile, höchstens ein Bildprofil. Die Auswahl ein
|
|||||||
### Spracherkennung
|
### Spracherkennung
|
||||||
|
|
||||||
`POST /v1/audio/transcriptions` benötigt Bearer-Token und ein explizit freigegebenes STT-Profil. Multipart-Felder: `model` (API-Profilname), `file` (PCM16-WAV, mono, 16 kHz, maximal 120 Sekunden / 8 MiB), optional `language` (`de`, `en`, `auto`, Standard `de`), `response_format` (`json`). Antwort: `{"text":"…"}`. Andere Formate, Chunked-Uploads, Zeitstempel und Streaming werden abgelehnt. Ein STT-Auftrag zur Zeit; CPU-Worker wird danach beendet. Keine Nutzung des alten Routers.
|
`POST /v1/audio/transcriptions` benötigt Bearer-Token und ein explizit freigegebenes STT-Profil. Multipart-Felder: `model` (API-Profilname), `file` (PCM16-WAV, mono, 16 kHz, maximal 120 Sekunden / 8 MiB), optional `language` (`de`, `en`, `auto`, Standard `de`), `response_format` (`json`). Antwort: `{"text":"…"}`. Andere Formate, Chunked-Uploads, Zeitstempel und Streaming werden abgelehnt. Ein STT-Auftrag zur Zeit; CPU-Worker wird danach beendet. Keine Nutzung des alten Routers.
|
||||||
|
|
||||||
|
### Modelllisten für Clients
|
||||||
|
|
||||||
|
`GET /v1/models` listet ausschließlich Chatprofile, damit Chatclients keine Bild-/Audio-Profile anbieten. Deck-Erweiterungen: `GET /v1/images/models` für Bildprofile, `GET /v1/audio/speech/models` für TTS und `GET /v1/audio/transcriptions/models` für STT. Alle Listen erfordern denselben Bearer-Token und enthalten nur freigegebene, ausführbare Profile. Die Inferenzrouten bleiben unverändert. Die Verwaltungs-API liefert weiterhin die Gesamtübersicht.
|
||||||
|
|
||||||
|
Chat akzeptiert `reasoning_effort`: `none`, `minimal`, `low`, `medium`, `high`, `xhigh` werden unverändert an llama.cpp weitergereicht; `null` wird wie ein fehlendes Feld behandelt. Der installierte Build unterstützt das Feld; die konkrete Wirkung hängt vom Modell-Chattemplate ab. `none` deaktiviert dort das Reasoning. Deck erfindet keine Tokenbudgets und ignoriert gesetzte Werte nicht stillschweigend.
|
||||||
|
|||||||
+7
-4
@@ -115,8 +115,8 @@ class Endpoint:
|
|||||||
if not row:raise APIError('Modellprofil nicht aktiviert oder unbekannt.',404,'model_not_found')
|
if not row:raise APIError('Modellprofil nicht aktiviert oder unbekannt.',404,'model_not_found')
|
||||||
if not row['runnable']:raise APIError('Profil derzeit nicht ausführbar: '+' '.join(row['blockers']),503,'model_unavailable')
|
if not row['runnable']:raise APIError('Profil derzeit nicht ausführbar: '+' '.join(row['blockers']),503,'model_unavailable')
|
||||||
return row
|
return row
|
||||||
def model_list(self):
|
def model_list(self,kind="chat"):
|
||||||
return dict(object='list',data=[dict(id=p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable']])
|
return dict(object='list',data=[dict(id=p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable'] and p['kind']==kind])
|
||||||
|
|
||||||
class APIHTTPServer(ThreadingHTTPServer):
|
class APIHTTPServer(ThreadingHTTPServer):
|
||||||
daemon_threads=True
|
daemon_threads=True
|
||||||
@@ -151,7 +151,8 @@ class APIHandler(BaseHTTPRequestHandler):
|
|||||||
with ep.lock:
|
with ep.lock:
|
||||||
if not ep.allowed():raise APIError('Endpunkt wird gestoppt.',503,'endpoint_stopping')
|
if not ep.allowed():raise APIError('Endpunkt wird gestoppt.',503,'endpoint_stopping')
|
||||||
ep.inflight+=1;admitted=True
|
ep.inflight+=1;admitted=True
|
||||||
if self.command=='GET' and self.path=='/v1/models':return self.send(ep.model_list())
|
model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt'}
|
||||||
|
if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path]))
|
||||||
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
|
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
|
||||||
if self.command!='POST':raise APIError('Route nicht gefunden.',404)
|
if self.command!='POST':raise APIError('Route nicht gefunden.',404)
|
||||||
if self.path=='/v1/audio/transcriptions':return self.transcription(ep)
|
if self.path=='/v1/audio/transcriptions':return self.transcription(ep)
|
||||||
@@ -209,8 +210,9 @@ class APIHandler(BaseHTTPRequestHandler):
|
|||||||
self.sent=True;self.send_response(200);self.send_header('Content-Type','audio/wav');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);return
|
self.sent=True;self.send_response(200);self.send_header('Content-Type','audio/wav');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);return
|
||||||
raise APIError('TTS-Zeitlimit überschritten.',504)
|
raise APIError('TTS-Zeitlimit überschritten.',504)
|
||||||
def chat(self,ep,data):
|
def chat(self,ep,data):
|
||||||
supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n'}
|
supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'}
|
||||||
if set(data)-supported:raise APIError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(set(data)-supported)))
|
if set(data)-supported:raise APIError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(set(data)-supported)))
|
||||||
|
if 'reasoning_effort' in data and data['reasoning_effort'] is not None and (not isinstance(data['reasoning_effort'],str) or data['reasoning_effort'] not in ('none','minimal','low','medium','high','xhigh')):raise APIError('Ungültiger reasoning_effort-Wert.')
|
||||||
if type(data.get('stream',False)) is not bool:raise APIError('stream muss true oder false sein.')
|
if type(data.get('stream',False)) is not bool:raise APIError('stream muss true oder false sein.')
|
||||||
if data.get('n',1)!=1:raise APIError('Zunächst wird n=1 unterstützt.')
|
if data.get('n',1)!=1:raise APIError('Zunächst wird n=1 unterstützt.')
|
||||||
messages=data.get('messages')
|
messages=data.get('messages')
|
||||||
@@ -238,6 +240,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
|||||||
def allowed():return ep.allowed() and any(p['id']==profile['id'] and p['revision']==profile['revision'] and p['enabled'] for p in ep.rows())
|
def allowed():return ep.allowed() and any(p['id']==profile['id'] and p['revision']==profile['revision'] and p['enabled'] for p in ep.rows())
|
||||||
with ep.scheduler.lease(key,profile['parameters']['slots'],lambda:ep.worker.ensure(profile),allowed=allowed):
|
with ep.scheduler.lease(key,profile['parameters']['slots'],lambda:ep.worker.ensure(profile),allowed=allowed):
|
||||||
body=dict(data)
|
body=dict(data)
|
||||||
|
if body.get('reasoning_effort') is None:body.pop('reasoning_effort',None)
|
||||||
for field in ('temperature','top_p','top_k'):body.setdefault(field,profile['parameters'][field])
|
for field in ('temperature','top_p','top_k'):body.setdefault(field,profile['parameters'][field])
|
||||||
conn,key=ep.worker.connect()
|
conn,key=ep.worker.connect()
|
||||||
try:
|
try:
|
||||||
|
|||||||
+18
-1
@@ -55,6 +55,23 @@ class EndpointTests(unittest.TestCase):
|
|||||||
with self.assertRaises(ValueError):self.enable('audio')
|
with self.assertRaises(ValueError):self.enable('audio')
|
||||||
self.assertEqual(self.request('/v1/audio/speech',{})[0],501)
|
self.assertEqual(self.request('/v1/audio/speech',{})[0],501)
|
||||||
self.record['api_token_hash']=hashlib.sha256(b'B'*32).hexdigest();self.assertEqual(self.request()[0],401);self.assertEqual(self.request(token='B'*32)[0],200)
|
self.record['api_token_hash']=hashlib.sha256(b'B'*32).hexdigest();self.assertEqual(self.request()[0],401);self.assertEqual(self.request(token='B'*32)[0],200)
|
||||||
|
def test_model_lists_separate_modalities(self):
|
||||||
|
self.rows[-1].update(runnable=True,blockers=[])
|
||||||
|
self.rows.append(dict(id='stt',name='stt',kind='stt',runnable=True,blockers=[],parameters={},updated_at=1))
|
||||||
|
for name in ('alpha','image','audio','stt'):self.enable(name)
|
||||||
|
for route,expected in [('/v1/models','alpha'),('/v1/images/models','image'),('/v1/audio/speech/models','audio'),('/v1/audio/transcriptions/models','stt')]:
|
||||||
|
self.assertEqual([p['id'] for p in self.request(route)[1]['data']],[expected])
|
||||||
|
self.assertEqual(self.request(route,token='bad')[0],401)
|
||||||
|
self.rows[-1]['runnable']=False
|
||||||
|
self.assertEqual(self.request('/v1/audio/transcriptions/models')[1]['data'],[])
|
||||||
|
def test_reasoning_effort_forwarded_and_validated(self):
|
||||||
|
self.enable('alpha')
|
||||||
|
for effort in ('none','minimal','low','medium','high','xhigh',None):
|
||||||
|
status,_=self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='synthetic')],reasoning_effort=effort))
|
||||||
|
self.assertEqual(status,200)
|
||||||
|
self.assertEqual(self.worker.requests[-1].get('reasoning_effort'),effort)
|
||||||
|
for effort in ([],True,3,'invalid'):
|
||||||
|
self.assertEqual(self.request('/v1/chat/completions',dict(model='alpha',messages=[dict(role='user',content='synthetic')],reasoning_effort=effort))[0],400)
|
||||||
def test_stt_endpoint_multipart_and_publication(self):
|
def test_stt_endpoint_multipart_and_publication(self):
|
||||||
from test_stt import audio
|
from test_stt import audio
|
||||||
self.rows.append(dict(id='stt',name='stt',kind='stt',runnable=True,blockers=[],parameters={}))
|
self.rows.append(dict(id='stt',name='stt',kind='stt',runnable=True,blockers=[],parameters={}))
|
||||||
@@ -78,7 +95,7 @@ class EndpointTests(unittest.TestCase):
|
|||||||
self.rows.append(dict(self.rows[2],id='image2',name='image2'))
|
self.rows.append(dict(self.rows[2],id='image2',name='image2'))
|
||||||
self.enable('alpha');self.enable('beta');self.enable('image');self.enable('image2')
|
self.enable('alpha');self.enable('beta');self.enable('image');self.enable('image2')
|
||||||
self.assertEqual(set(self.ep.config['enabled_profiles']),{'alpha','beta','image2'})
|
self.assertEqual(set(self.ep.config['enabled_profiles']),{'alpha','beta','image2'})
|
||||||
self.assertEqual({r['id'] for r in self.request()[1]['data']},{'alpha','beta','image2'})
|
self.assertEqual({r['id'] for r in self.request()[1]['data']},{'alpha','beta'})
|
||||||
self.ep.enable(dict(id='image2',enabled=False));self.assertEqual(set(self.ep.config['enabled_profiles']),{'alpha','beta'})
|
self.ep.enable(dict(id='image2',enabled=False));self.assertEqual(set(self.ep.config['enabled_profiles']),{'alpha','beta'})
|
||||||
def test_invalid_image_selection_keeps_previous(self):
|
def test_invalid_image_selection_keeps_previous(self):
|
||||||
self.rows.append(dict(self.rows[2],id='badimage',name='badimage',runnable=False,blockers=['missing']))
|
self.rows.append(dict(self.rows[2],id='badimage',name='badimage',runnable=False,blockers=['missing']))
|
||||||
|
|||||||
Reference in New Issue
Block a user