Add isolated Qwen3 ASR setup test UI and transcription API
This commit is contained in:
+22
-3
@@ -10,6 +10,7 @@ import threading
|
||||
import time
|
||||
from http.server import BaseHTTPRequestHandler,ThreadingHTTPServer
|
||||
from inference import InferenceError
|
||||
from stt import read_upload
|
||||
|
||||
class APIError(ValueError):
|
||||
def __init__(self,message,status=400,code='invalid_request_error'):
|
||||
@@ -17,7 +18,7 @@ class APIError(ValueError):
|
||||
|
||||
class Endpoint:
|
||||
def __init__(self,root,profiles,worker,scheduler,images,credentials,management_port):
|
||||
self.root=Path(root);self.profiles=profiles;self.worker=worker;self.scheduler=scheduler;self.images=images;self.tts=None;self.credentials=credentials
|
||||
self.root=Path(root);self.profiles=profiles;self.worker=worker;self.scheduler=scheduler;self.images=images;self.tts=None;self.stt=None;self.credentials=credentials
|
||||
self.management_port=management_port;self.lock=threading.RLock();self.http=None;self.thread=None;self.state='stopped';self.error=None;self.inflight=0
|
||||
self.allowed_ports=[int(p) for p in os.environ.get('DECK_API_PORTS','').split(',') if p]
|
||||
self.config=dict(port=self.allowed_ports[0] if self.allowed_ports else 8120,enabled_profiles=[],autostart=False)
|
||||
@@ -38,7 +39,7 @@ class Endpoint:
|
||||
rows=self.rows();worker=self.worker.status();job=self.images.status()['job'];counts={}
|
||||
for key,kind in [('llm','chat'),('image','image'),('tts','audio'),('stt','stt')]:
|
||||
subset=[p for p in rows if p['kind']==kind]
|
||||
counts[key]=dict(enabled=sum(p['enabled'] for p in subset),available=sum(p['enabled'] and p['runnable'] for p in subset),loaded=bool(worker['state']=='ready' and kind=='chat') if kind=='chat' else bool(self.tts and self.tts.status()['job'] and self.tts.status()['job']['state']=='running') if kind=='audio' else bool(kind=='image' and job and job['state']=='running'),supported=kind in ('chat','image') or (kind=='audio' and self.tts is not None))
|
||||
counts[key]=dict(enabled=sum(p['enabled'] for p in subset),available=sum(p['enabled'] and p['runnable'] for p in subset),loaded=bool(worker['state']=='ready' and kind=='chat') if kind=='chat' else bool(self.tts and self.tts.status()['job'] and self.tts.status()['job']['state']=='running') if kind=='audio' else bool(self.stt and self.stt.status()['job'] and self.stt.status()['job']['state']=='running') if kind=='stt' else bool(kind=='image' and job and job['state']=='running'),supported=kind in ('chat','image') or (kind=='audio' and self.tts is not None) or (kind=='stt' and self.stt is not None))
|
||||
scheduler_status=self.scheduler.status()
|
||||
with self.lock:return dict(state=self.state,reachable=bool(self.thread and self.thread.is_alive() and self.state=='running'),port=self.config['port'],bind=os.environ.get('DECK_API_BIND','127.0.0.1'),base_url=f"http://127.0.0.1:{self.config['port']}/v1",allowed_ports=self.allowed_ports,error=self.error,counts=counts,worker=worker,scheduler=scheduler_status,profiles=[dict(id=p['id'],name=p['name'],kind=p['kind'],enabled=p['enabled'],runnable=p['runnable'],blockers=p['blockers']) for p in rows],active_requests=self.inflight)
|
||||
def configure(self,data):
|
||||
@@ -100,6 +101,7 @@ class Endpoint:
|
||||
with self.lock:self.state='stopping';http=self.http
|
||||
if http:http.shutdown();http.server_close()
|
||||
if self.tts:self.tts.stop()
|
||||
if self.stt:self.stt.stop()
|
||||
self.images.stop();self.worker.stop()
|
||||
def allowed(self):
|
||||
with self.lock:return self.state=='running'
|
||||
@@ -152,7 +154,7 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
if self.command=='GET' and self.path=='/v1/models':return self.send(ep.model_list())
|
||||
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
|
||||
if self.command!='POST':raise APIError('Route nicht gefunden.',404)
|
||||
if self.path=='/v1/audio/transcriptions':raise APIError('Für TTS/STT ist noch keine Deck-Laufzeit eingerichtet.',501,'not_implemented')
|
||||
if self.path=='/v1/audio/transcriptions':return self.transcription(ep)
|
||||
if self.path not in ('/v1/chat/completions','/v1/images/generations','/v1/audio/speech'):raise APIError('Route nicht implementiert.',404)
|
||||
if self.headers.get('Transfer-Encoding'):raise APIError('Chunked Upload wird nicht unterstützt.')
|
||||
try:length=int(self.headers.get('Content-Length','0'))
|
||||
@@ -174,6 +176,23 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
self.close_connection=True
|
||||
if admitted:
|
||||
with ep.lock:ep.inflight-=1
|
||||
def transcription(self,ep):
|
||||
if not ep.stt:raise APIError('STT-Laufzeit nicht eingerichtet.',501,'not_implemented')
|
||||
try:
|
||||
fields,audio=read_upload(self)
|
||||
if 'profile_id' in fields:raise ValueError('API-Profilname als model erforderlich.')
|
||||
profile=ep.find_profile(fields.get('model'),'stt')
|
||||
job=ep.stt.start(profile['id'],audio,fields.get('language','de'))
|
||||
except APIError:raise
|
||||
except ValueError as exc:raise APIError(str(exc)) from None
|
||||
deadline=time.monotonic()+280
|
||||
while time.monotonic()<deadline:
|
||||
current=ep.stt.status()['job']
|
||||
if not current or current['id']!=job['id']:raise APIError('STT-Ergebnis nicht mehr verfügbar.',409)
|
||||
if current['state']=='complete':return self.send({'text':current['text']})
|
||||
if current['state']!='running':raise APIError(current['phase'],503)
|
||||
time.sleep(.2)
|
||||
ep.stt.stop();raise APIError('STT-Zeitlimit überschritten.',504)
|
||||
def speech(self,ep,data):
|
||||
if not ep.tts:raise APIError('TTS-Laufzeit nicht eingerichtet.',501)
|
||||
if set(data)-{'model','input','voice','response_format','speed','language'}:raise APIError('Nicht unterstützte TTS-Felder.')
|
||||
|
||||
Reference in New Issue
Block a user