Add exclusive LLM and video modes with isolated LTX worker and job API
This commit is contained in:
+39
-2
@@ -20,7 +20,7 @@ class APIError(ValueError):
|
||||
|
||||
class Endpoint:
|
||||
def __init__(self,root,profiles,worker,scheduler,images,credentials,management_port):
|
||||
self.root=Path(root);self.profiles=profiles;self.worker=worker;self.scheduler=scheduler;self.images=images;self.tts=None;self.stt=None;self.credentials=credentials
|
||||
self.root=Path(root);self.profiles=profiles;self.worker=worker;self.scheduler=scheduler;self.images=images;self.tts=None;self.stt=None;self.video=None;self.credentials=credentials
|
||||
self.management_port=management_port;self.lock=threading.RLock();self.http=None;self.thread=None;self.state='stopped';self.error=None;self.inflight=0
|
||||
self.allowed_ports=[int(p) for p in os.environ.get('DECK_API_PORTS','').split(',') if p]
|
||||
self.config=dict(port=self.allowed_ports[0] if self.allowed_ports else 8120,enabled_profiles=[],autostart=False)
|
||||
@@ -43,7 +43,7 @@ class Endpoint:
|
||||
subset=[p for p in rows if p['kind']==kind]
|
||||
counts[key]=dict(enabled=sum(p['enabled'] for p in subset),available=sum(p['enabled'] and p['runnable'] for p in subset),loaded=bool(worker['state']=='ready' and kind=='chat') if kind=='chat' else bool(self.tts and self.tts.status()['job'] and self.tts.status()['job']['state']=='running') if kind=='audio' else bool(self.stt and self.stt.status()['job'] and self.stt.status()['job']['state']=='running') if kind=='stt' else bool(kind=='image' and job and job['state']=='running'),supported=kind in ('chat','image') or (kind=='audio' and self.tts is not None) or (kind=='stt' and self.stt is not None))
|
||||
scheduler_status=self.scheduler.status()
|
||||
with self.lock:return dict(state=self.state,reachable=bool(self.thread and self.thread.is_alive() and self.state=='running'),port=self.config['port'],bind=os.environ.get('DECK_API_BIND','127.0.0.1'),base_url=f"http://127.0.0.1:{self.config['port']}/v1",allowed_ports=self.allowed_ports,error=self.error,counts=counts,worker=worker,scheduler=scheduler_status,profiles=[dict(id=p['id'],name=p['name'],kind=p['kind'],enabled=p['enabled'],runnable=p['runnable'],blockers=p['blockers']) for p in rows],active_requests=self.inflight)
|
||||
with self.lock:return dict(video=self.video.status() if self.video else None,state=self.state,reachable=bool(self.thread and self.thread.is_alive() and self.state=='running'),port=self.config['port'],bind=os.environ.get('DECK_API_BIND','127.0.0.1'),base_url=f"http://127.0.0.1:{self.config['port']}/v1",allowed_ports=self.allowed_ports,error=self.error,counts=counts,worker=worker,scheduler=scheduler_status,profiles=[dict(id=p['id'],name=p['name'],kind=p['kind'],enabled=p['enabled'],runnable=p['runnable'],blockers=p['blockers']) for p in rows],active_requests=self.inflight)
|
||||
def configure(self,data):
|
||||
if set(data)!={'port'} or type(data['port']) is not int or not 1024<=data['port']<=65535:raise ValueError('Port zwischen 1024 und 65535 erforderlich.')
|
||||
port=data['port']
|
||||
@@ -61,6 +61,7 @@ class Endpoint:
|
||||
rows=self.rows()
|
||||
row=next((p for p in rows if p['id']==data['id']),None)
|
||||
if not row:raise ValueError('Profil nicht gefunden.')
|
||||
if row['kind']=='video':raise ValueError('Videoprofil in der Übersicht auswählen; Video nutzt eine exklusive Auswahl.')
|
||||
if data['enabled'] and not row['runnable']:raise ValueError('Profil nicht ausführbar: '+' '.join(row['blockers']))
|
||||
with self.lock:
|
||||
enabled=set(self.config['enabled_profiles'])
|
||||
@@ -94,6 +95,10 @@ class Endpoint:
|
||||
http=self.http
|
||||
if http:http.shutdown();http.server_close()
|
||||
while True:
|
||||
if self.video:
|
||||
vs=self.video.status()
|
||||
if vs['state']=='switching':time.sleep(.1);continue
|
||||
if vs['mode']=='video':self.video.switch('llm');time.sleep(.1);continue
|
||||
with self.lock:pending=self.inflight
|
||||
if not pending and self.scheduler.unload_idle():break
|
||||
time.sleep(.1)
|
||||
@@ -149,6 +154,36 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
def failure(self,exc):
|
||||
if self.sent:return
|
||||
self.send({'error':{'message':str(exc),'type':getattr(exc,'code','server_error'),'param':None,'code':getattr(exc,'code','worker_unavailable')}},getattr(exc,'status',503))
|
||||
def video_route(self,ep):
|
||||
if not self.path.startswith('/v1/videos'):return False
|
||||
if not ep.video:raise APIError('Video-Worker nicht eingerichtet.',503)
|
||||
if self.command=='GET' and self.path=='/v1/videos/models':
|
||||
ident=ep.video.selected
|
||||
self.send({'object':'list','data':[{'id':'athena-video','object':'model','owned_by':'athena-deck'}] if any(p['id']==ident and p['runnable'] for p in ep.rows()) else []});return True
|
||||
if self.command=='POST' and self.path=='/v1/videos':
|
||||
if self.headers.get('Transfer-Encoding'):raise APIError('Chunked Upload nicht unterstützt.')
|
||||
try:length=int(self.headers.get('Content-Length','0'))
|
||||
except ValueError:raise APIError('Ungültige Länge.')
|
||||
if not 0<length<=65536 or self.headers.get('Content-Type','').split(';')[0]!='application/json':raise APIError('JSON bis 64 KiB erforderlich.',413)
|
||||
try:
|
||||
data=json.loads(self.rfile.read(length))
|
||||
if not isinstance(data,dict):raise ValueError('JSON-Objekt erforderlich.')
|
||||
job=ep.video.generate(data)
|
||||
except ValueError as exc:raise APIError(str(exc),409,'video_not_ready') from None
|
||||
self.send(job,202);return True
|
||||
parts=self.path.split('/')
|
||||
if self.command=='GET' and len(parts) in (4,5) and re.fullmatch('[a-f0-9]{32}',parts[3]):
|
||||
job=ep.video.status()['job']
|
||||
if not job or job['id']!=parts[3]:raise APIError('Video-Auftrag nicht gefunden.',404)
|
||||
if len(parts)==4:self.send(job);return True
|
||||
if parts[4]=='content':
|
||||
try:path=ep.video.result(parts[3])
|
||||
except ValueError as exc:raise APIError(str(exc),409) from None
|
||||
self.sent=True;self.send_response(200);self.send_header('Content-Type','video/mp4');self.send_header('Content-Length',str(path.stat().st_size));self.send_header('Cache-Control','no-store');self.send_header('Connection','close');self.end_headers()
|
||||
with path.open('rb') as source:
|
||||
while chunk:=source.read(1024*1024):self.wfile.write(chunk)
|
||||
self.close_connection=True;return True
|
||||
raise APIError('Video-Route nicht gefunden.',404)
|
||||
def do_GET(self):self.route()
|
||||
def do_POST(self):self.route()
|
||||
def route(self):
|
||||
@@ -162,7 +197,9 @@ class APIHandler(BaseHTTPRequestHandler):
|
||||
model_routes={'/v1/models':'chat','/v1/images/models':'image','/v1/audio/speech/models':'audio','/v1/audio/transcriptions/models':'stt'}
|
||||
if self.command=='GET' and self.path in model_routes:return self.send(ep.model_list(model_routes[self.path]))
|
||||
if self.command=='GET' and self.path=='/health':return self.send({'status':'ok','service':'athena-deck-api'})
|
||||
if self.video_route(ep):return
|
||||
if self.command!='POST':raise APIError('Route nicht gefunden.',404)
|
||||
if ep.video and ep.scheduler.gpu_mode!='llm' and self.path!='/v1/audio/transcriptions':raise APIError('Video-Modus aktiv; Chat-, Bild- und TTS-Aufträge sind gesperrt. Auf LLM zurückschalten.',503,'video_mode_active')
|
||||
if self.path=='/v1/audio/transcriptions':return self.transcription(ep)
|
||||
if self.path not in ('/v1/chat/completions','/v1/images/generations','/v1/audio/speech'):raise APIError('Route nicht implementiert.',404)
|
||||
if self.headers.get('Transfer-Encoding'):raise APIError('Chunked Upload wird nicht unterstützt.')
|
||||
|
||||
Reference in New Issue
Block a user