Add isolated Qwen3 ASR setup test UI and transcription API

This commit is contained in:
Mikei386
2026-09-29 08:54:06 +02:00
parent 7bd9096e67
commit 27012165eb
20 changed files with 294 additions and 24 deletions
+22 -3
View File
@@ -10,6 +10,7 @@ import threading
import time
from http.server import BaseHTTPRequestHandler,ThreadingHTTPServer
from inference import InferenceError
from stt import read_upload
class APIError(ValueError):
def __init__(self,message,status=400,code='invalid_request_error'):
@@ -17,7 +18,7 @@ class APIError(ValueError):
class Endpoint:
def __init__(self,root,profiles,worker,scheduler,images,credentials,management_port):
self.root=Path(root);self.profiles=profiles;self.worker=worker;self.scheduler=scheduler;self.images=images;self.tts=None;self.credentials=credentials
self.root=Path(root);self.profiles=profiles;self.worker=worker;self.scheduler=scheduler;self.images=images;self.tts=None;self.stt=None;self.credentials=credentials
self.management_port=management_port;self.lock=threading.RLock();self.http=None;self.thread=None;self.state='stopped';self.error=None;self.inflight=0
self.allowed_ports=[int(p) for p in os.environ.get('DECK_API_PORTS','').split(',') if p]
self.config=dict(port=self.allowed_ports[0] if self.allowed_ports else 8120,enabled_profiles=[],autostart=False)
@@ -38,7 +39,7 @@ class Endpoint:
rows=self.rows();worker=self.worker.status();job=self.images.status()['job'];counts={}
for key,kind in [('llm','chat'),('image','image'),('tts','audio'),('stt','stt')]:
subset=[p for p in rows if p['kind']==kind]
counts[key]=dict(enabled=sum(p['enabled'] for p in subset),available=sum(p['enabled'] and p['runnable'] for p in subset),loaded=bool(worker['state']=='ready' and kind=='chat') if kind=='chat' else bool(self.tts and self.tts.status()['job'] and self.tts.status()['job']['state']=='running') if kind=='audio' else bool(kind=='image' and job and job['state']=='running'),supported=kind in ('chat','image') or (kind=='audio' and self.tts is not None))
counts[key]=dict(enabled=sum(p['enabled'] for p in subset),available=sum(p['enabled'] and p['runnable'] for p in subset),loaded=bool(worker['state']=='ready' and kind=='chat') if kind=='chat' else bool(self.tts and self.tts.status()['job'] and self.tts.status()['job']['state']=='running') if kind=='audio' else bool(self.stt and self.stt.status()['job'] and self.stt.status()['job']['state']=='running') if kind=='stt' else bool(kind=='image' and job and job['state']=='running'),supported=kind in ('chat','image') or (kind=='audio' and self.tts is not None) or (kind=='stt' and self.stt is not None))
scheduler_status=self.scheduler.status()
with self.lock:return dict(state=self.state,reachable=bool(self.thread and self.thread.is_alive() and self.state=='running'),port=self.config['port'],bind=os.environ.get('DECK_API_BIND','127.0.0.1'),base_url=f"http://127.0.0.1:{self.config['port']}/v1",allowed_ports=self.allowed_ports,error=self.error,counts=counts,worker=worker,scheduler=scheduler_status,profiles=[dict(id=p['id'],name=p['name'],kind=p['kind'],enabled=p['enabled'],runnable=p['runnable'],blockers=p['blockers']) for p in rows],active_requests=self.inflight)
def configure(self,data):
@@ -100,6 +101,7 @@ class Endpoint:
with self.lock:self.state='stopping';http=self.http
if http:http.shutdown();http.server_close()
if self.tts:self.tts.stop()
if self.stt:self.stt.stop()
self.images.stop();self.worker.stop()
def allowed(self):
with self.lock:return self.state=='running'
@@ -152,7 +154,7 @@ class APIHandler(BaseHTTPRequestHandler):
if self.command=='GET' and self.path=='/v1/models':return self.send(ep.model_list())
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
if self.command!='POST':raise APIError('Route nicht gefunden.',404)
if self.path=='/v1/audio/transcriptions':raise APIError('Für TTS/STT ist noch keine Deck-Laufzeit eingerichtet.',501,'not_implemented')
if self.path=='/v1/audio/transcriptions':return self.transcription(ep)
if self.path not in ('/v1/chat/completions','/v1/images/generations','/v1/audio/speech'):raise APIError('Route nicht implementiert.',404)
if self.headers.get('Transfer-Encoding'):raise APIError('Chunked Upload wird nicht unterstützt.')
try:length=int(self.headers.get('Content-Length','0'))
@@ -174,6 +176,23 @@ class APIHandler(BaseHTTPRequestHandler):
self.close_connection=True
if admitted:
with ep.lock:ep.inflight-=1
def transcription(self,ep):
if not ep.stt:raise APIError('STT-Laufzeit nicht eingerichtet.',501,'not_implemented')
try:
fields,audio=read_upload(self)
if 'profile_id' in fields:raise ValueError('API-Profilname als model erforderlich.')
profile=ep.find_profile(fields.get('model'),'stt')
job=ep.stt.start(profile['id'],audio,fields.get('language','de'))
except APIError:raise
except ValueError as exc:raise APIError(str(exc)) from None
deadline=time.monotonic()+280
while time.monotonic()<deadline:
current=ep.stt.status()['job']
if not current or current['id']!=job['id']:raise APIError('STT-Ergebnis nicht mehr verfügbar.',409)
if current['state']=='complete':return self.send({'text':current['text']})
if current['state']!='running':raise APIError(current['phase'],503)
time.sleep(.2)
ep.stt.stop();raise APIError('STT-Zeitlimit überschritten.',504)
def speech(self,ep,data):
if not ep.tts:raise APIError('TTS-Laufzeit nicht eingerichtet.',501)
if set(data)-{'model','input','voice','response_format','speed','language'}:raise APIError('Nicht unterstützte TTS-Felder.')