Separate modality model lists and forward chat reasoning effort
This commit is contained in:
+7
-4
@@ -115,8 +115,8 @@ class Endpoint:
|
||||
if not row:raise APIError('Modellprofil nicht aktiviert oder unbekannt.',404,'model_not_found')
|
||||
if not row['runnable']:raise APIError('Profil derzeit nicht ausführbar: '+' '.join(row['blockers']),503,'model_unavailable')
|
||||
return row
|
||||
def model_list(self):
|
||||
return dict(object='list',data=[dict(id=p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable']])
|
||||
def model_list(self,kind="chat"):
|
||||
return dict(object='list',data=[dict(id=p['name'],object='model',created=int(p['updated_at']),owned_by='athena-deck') for p in self.rows() if p['enabled'] and p['runnable'] and p['kind']==kind])
|
||||
|
||||
class APIHTTPServer(ThreadingHTTPServer):
|
||||
daemon_threads=True
|
||||
@@ -151,7 +151,8 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
with ep.lock:
|
||||
if not ep.allowed():raise APIError('Endpunkt wird gestoppt.',503,'endpoint_stopping')
|
||||
ep.inflight+=1;admitted=True
|
||||
if self.command=='GET' and self.path=='/v1/models':return self.send(ep.model_list())
|
||||
model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt'}
|
||||
if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path]))
|
||||
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
|
||||
if self.command!='POST':raise APIError('Route nicht gefunden.',404)
|
||||
if self.path=='/v1/audio/transcriptions':return self.transcription(ep)
|
||||
@@ -209,8 +210,9 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
self.sent=True;self.send_response(200);self.send_header('Content-Type','audio/wav');self.send_header('Content-Length',str(len(body)));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers();self.wfile.write(body);return
|
||||
raise APIError('TTS-Zeitlimit überschritten.',504)
|
||||
def chat(self,ep,data):
|
||||
supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n'}
|
||||
supported={'model','messages','stream','stream_options','temperature','top_p','top_k','max_tokens','max_completion_tokens','stop','seed','tools','tool_choice','parallel_tool_calls','response_format','presence_penalty','frequency_penalty','logprobs','top_logprobs','user','n','reasoning_effort'}
|
||||
if set(data)-supported:raise APIError('Nicht unterstützte Chat-Felder: '+', '.join(sorted(set(data)-supported)))
|
||||
if 'reasoning_effort' in data and data['reasoning_effort'] is not None and (not isinstance(data['reasoning_effort'],str) or data['reasoning_effort'] not in ('none','minimal','low','medium','high','xhigh')):raise APIError('Ungültiger reasoning_effort-Wert.')
|
||||
if type(data.get('stream',False)) is not bool:raise APIError('stream muss true oder false sein.')
|
||||
if data.get('n',1)!=1:raise APIError('Zunächst wird n=1 unterstützt.')
|
||||
messages=data.get('messages')
|
||||
@@ -238,6 +240,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
def allowed():return ep.allowed() and any(p['id']==profile['id'] and p['revision']==profile['revision'] and p['enabled'] for p in ep.rows())
|
||||
with ep.scheduler.lease(key,profile['parameters']['slots'],lambda:ep.worker.ensure(profile),allowed=allowed):
|
||||
body=dict(data)
|
||||
if body.get('reasoning_effort') is None:body.pop('reasoning_effort',None)
|
||||
for field in ('temperature','top_p','top_k'):body.setdefault(field,profile['parameters'][field])
|
||||
conn,key=ep.worker.connect()
|
||||
try:
|
||||
|
||||
Reference in New Issue
Block a user