diff --git a/ENDPOINT.md b/ENDPOINT.md
index 44dc3ac..156171d 100644
--- a/ENDPOINT.md
+++ b/ENDPOINT.md
@@ -129,3 +129,28 @@ Listener gestoppt und Modell entladen. Keine vorhandenen Nutzerprompts oder
Anwendungslogs gelesen. Testprofile und Testzugang gehören nicht zur normalen
Deck-Konfiguration. Der Test weist keine vollständige OpenAI-Kompatibilität oder
Langzeitstabilität nach.
+
+
+## Sprachmodelle → Testen
+
+Der interne Testchat verwendet denselben Scheduler und llama.cpp-Worker wie der
+API-Endpunkt, braucht aber keine Veröffentlichung des Profils und keinen
+API-Token im Browser. Profilauswahl, fortlaufender Textchat, Antwortlimit (bis
+4096 Token), Abbrechen und explizites Entladen stehen bereit. Der Modellprozess
+bleibt nach einer Antwort geladen; andere Profile/Bildaufträge wechseln regulär.
+Entladen wird bei aktiven Anfragen abgewiesen. Abbrechen schließt nur die eigene
+Chatverbindung; während des eigenen Modellstarts bricht es diesen Start ab.
+
+Der Verlauf liegt nur im Browser-Arbeitsspeicher. Beim Wechsel der Ansicht oder
+des Profils beginnt ein neuer Verlauf. Der Server hält nur den aktuellen Testjob
+mit gestreamter Antwort vorübergehend im RAM; weder Prompt noch Antwort werden
+als Chatdatei oder Log gespeichert. Reguläre Admin-Sitzung/CSRF-Schutz gelten für
+GET `/api/v1/chat-tests` und POST `/api/v1/chat-tests/start`, `/cancel`, `/unload`.
+Start: `profile_id`, Textnachrichten `messages`, `max_tokens`. Kein Bearer-Zugriff.
+
+Diagnose zeigt Ladephase, Wartezeit/Reservierungen, reale GPU-Belegung und die
+Speicherprognose mit Reserve je GPU, RAM-Budget und GPU-Layerlimit. Eine Prognose
+ist kein Nachweis für fehlerfreien Betrieb. Bestätigte Cgroup-OOM-Kills werden
+als RAM-OOM gemeldet; ein GPU-OOM wird ohne eindeutigen Nachweis nicht behauptet.
+Ein nicht eindeutig diagnostizierter Prozessabbruch nennt Speicher, Modell/Build
+oder Zeitlimit als mögliche Ursachen. Bestehende Anwendungslogs werden nicht gelesen.
diff --git a/README.md b/README.md
index 611b284..a7b71e3 100644
--- a/README.md
+++ b/README.md
@@ -17,7 +17,7 @@ Katalog, Datei-Downloads, Bibliothek, serverseitig gespeicherte Profile und
llama.cpp-Buildverwaltung sind live. Der Download-Reiter erlaubt das Ausblenden
abgeschlossener Einträge ohne Dateiverlust. Entdecken zeigt Größen und eine
konservative Gewichts-Speicherprüfung; noch keine vollständige Laufzeitprognose.
-Bildprofile mit vollständigem Qwen-Image-2.1-GGUF-Rezept lassen sich unter „Testen“ ausführen. Textprofile sind am eigenen [OpenAI-kompatiblen Endpunkt](ENDPOINT.md) ausführbar. Audio und Video sind noch nicht angebunden. Details: [Modellverwaltung](STUDIO.md).
+Bildprofile mit vollständigem Qwen-Image-2.1-GGUF-Rezept lassen sich unter „Testen“ ausführen. Textprofile sind am eigenen [OpenAI-kompatiblen Endpunkt](ENDPOINT.md) ausführbar. Audio und Video sind noch nicht angebunden. Sprachmodelle besitzen außerdem einen internen Testchat mit Abbruch und Speicherdiagnose. Details: [Modellverwaltung](STUDIO.md).
## Zugang und API-Token
diff --git a/STUDIO.md b/STUDIO.md
index ba1e37b..8aedd36 100644
--- a/STUDIO.md
+++ b/STUDIO.md
@@ -155,3 +155,5 @@ Leere Geräte-/Gewichtelisten bedeuten automatische Auswahl/Verteilung.
Alte Profile bleiben lesbar; fehlende neue Felder erhalten bei Anzeige/Speichern
Standardwerte (automatische Geräte, keine Aufteilung, Temperature 0.8, Top-p 0.95,
Top-k 40). Bestehende Dateien werden beim Lesen nicht umgeschrieben.
+
+Sprachmodelle besitzen jetzt den Reiter **Testen**: interner Textchat mit Streaming, Profilauswahl und Speicherdiagnose. Öffentliche Profilfreigabe ist dafür nicht erforderlich. Details: [ENDPOINT.md](ENDPOINT.md#sprachmodelle--testen).
diff --git a/chat-test-ui.js b/chat-test-ui.js
new file mode 100644
index 0000000..c5c250d
--- /dev/null
+++ b/chat-test-ui.js
@@ -0,0 +1,27 @@
+window.ChatTestUI=(()=>{
+ const e=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c]));
+ const gib=v=>v==null?'unbekannt':(v/1024).toFixed(2)+' GiB';
+ async function api(path,data){const r=await fetch('/api/v1/'+path,data===undefined?{}:{method:'POST',headers:{'Content-Type':'application/json','X-Athena-Deck':'1'},body:JSON.stringify(data)});const v=await r.json();if(!r.ok)throw Error(v.error||'Anfrage fehlgeschlagen');return v;}
+ function put(el,html){if(el._chatHtml!==html){el.innerHTML=html;el._chatHtml=html;}}
+ function html(){return `Sprachmodell testen Ein einfacher Textchat mit deinem gespeicherten Profil. Nutzt denselben Worker wie API-Anfragen; eine öffentliche Profilfreigabe ist nicht erforderlich.
Laufzeit & Diagnose
Test abbrechen Modell entladen
Entladen ist nur ohne laufende Anfragen möglich. Abbrechen beendet nur diesen Test; andere Antworten werden nicht gestoppt.
Speicherprüfung
Prognose mit Reserve, keine Garantie für jede Anfrage. GPU-Layer können automatisch auf die CPU ausgelagert werden. Bei Fehlern helfen häufig kleinerer Kontext oder Microbatch. Hardware-Messwerte →
`;}
+ function bind(){
+ const root=document.querySelector('#chat-test'),el=id=>root.querySelector('#'+id);let profiles=[],messages=[],turns=[],ownJob=null,finished=null,status=null,pending=false;
+ const notice=t=>{if(root.isConnected)el('chat-message').textContent=t;};
+ function transcript(){put(el('chat-transcript'),turns.map(t=>`
${t.role==='user'?'Du':'Modell'} ${e(t.content)}
${t.reasoning?`
Reasoning-Ausgabe ${e(t.reasoning)} `:''}
`).join(''));}
+ function selection(){const p=profiles.find(p=>p.id===el('chat-profile').value),running=status?.job?.state==='running';el('chat-profile-info').textContent=p?`${p.parameters.context.toLocaleString('de-DE')} Token gesamt · ${p.parameters.slots} Slots · Batch ${p.parameters.batch} / Microbatch ${p.parameters.ubatch}${p.blockers.length?' · '+p.blockers.join(' '):''}`:'Zuerst ein Sprachmodellprofil anlegen.';el('chat-send').disabled=pending||running||!p?.runnable;el('chat-profile').disabled=pending||running;el('chat-clear').disabled=pending||running;el('chat-cancel').disabled=!running;el('chat-unload').disabled=!!status?.scheduler.active_requests;}
+ async function refresh(){try{const [s,h]=await Promise.all([api('chat-tests'),api('hardware').catch(()=>({gpus:[]}))]);if(!root.isConnected)return;status=s;const j=s.job,w=s.worker,plan=w.memory_plan;
+ put(el('chat-state'),`${e(j?.phase||'Noch kein Chat-Test gestartet.')}
Worker: ${e(w.phase||w.state)} · ${e(w.profile_name||'kein Modell geladen')} ${s.scheduler.active_requests} laufende / ${s.scheduler.waiting_requests} wartende Anfragen
${j?.error?`${e(j.error)}
`:''}${w.error?`${e(w.error)}
`:''}${j?`Dauer: ${Math.max(0,Math.round((j.finished_at||Date.now()/1000)-j.started_at))} s
`:''}`);
+ put(el('chat-memory'),(plan?`Letzte Prognose für ${e(plan.profile_name||w.profile_name||'das Modell')}
${plan.fits?'Prognose passt einschließlich Reserve':'Prognose passt nicht in den verfügbaren Speicher'} · ${plan.context.toLocaleString('de-DE')} Token / ${plan.slots} Slots · GPU-Layerlimit ${plan.gpu_layers===999?'alle':plan.gpu_layers}
${plan.gpus.map(g=>`${e(g.name)}: geschätzt ${gib(g.required_mib)} + ${gib(g.reserve_mib)} Reserve / ${gib(g.free_mib)} beim Prüfen frei
`).join('')}RAM-Budget: ${gib(plan.host_required_mib)} benötigt / ${gib(plan.host_available_mib)} verfügbar
`:'Noch keine Speicherprognose für einen Modellstart vorhanden.
')+`Aktuelle GPU-Belegung ${(h.gpus||[]).map(g=>`${e(g.name)}: ${gib(g.used_mib)} / ${gib(g.total_mib)} belegt · ${g.percent??'—'} % Auslastung
`).join('')||'GPU-Messwerte nicht verfügbar.
'}`);
+ if(j&&j.id===ownJob){let response=turns[turns.length-1];if(response?.role!=='assistant'){response={role:'assistant',content:'',reasoning:''};turns.push(response);}response.content=j.answer;response.reasoning=j.reasoning;transcript();
+ if(j.state!=='running'&&finished!==j.id){finished=j.id;if(j.state==='complete'&&j.answer)messages.push({role:'assistant',content:j.answer});else{messages=[];notice('Der nächste Prompt beginnt einen neuen Kontext; der letzte Test wurde nicht vollständig beantwortet.');}if(j.finish_reason==='length')notice('Antwort am Tokenlimit beendet. Bei Bedarf das Antwortlimit erhöhen.');}
+ }selection();
+ }catch(error){notice(error.message);}finally{if(root.isConnected)setTimeout(refresh,1500);}}
+ api('profiles').then(s=>{if(!root.isConnected)return;profiles=s.profiles.filter(p=>p.kind==='chat');el('chat-profile').innerHTML=profiles.map(p=>`${e(p.name)}${p.runnable?'':' · nicht ausführbar'} `).join('')||'Zuerst ein Chatprofil anlegen ';selection();}).catch(err=>notice(err.message));
+ const clear=()=>{messages=[];turns=[];ownJob=null;finished=null;transcript();notice('Neuer Chat.');};el('chat-clear').onclick=clear;el('chat-profile').onchange=()=>{clear();selection();};
+ el('chat-test-form').onsubmit=async event=>{event.preventDefault();pending=true;selection();const text=el('chat-prompt').value.trim();try{const next=[...messages,{role:'user',content:text}];const job=await api('chat-tests/start',{profile_id:el('chat-profile').value,messages:next,max_tokens:Number(el('chat-limit').value)});messages=next;turns.push({role:'user',content:text});ownJob=job.id;finished=null;status={...status,job};el('chat-prompt').value='';transcript();notice('Test gestartet. Speicherprüfung, Modellstart und Antwort erscheinen unter Diagnose.');}catch(error){notice(error.message);}finally{pending=false;selection();}};
+ el('chat-cancel').onclick=async()=>{try{await api('chat-tests/cancel',{});notice('Abbruch angefordert.');}catch(error){notice(error.message);}};
+ el('chat-unload').onclick=async()=>{try{await api('chat-tests/unload',{});notice('Eigener Modellprozess entladen.');}catch(error){notice(error.message);}};
+ refresh();
+ }
+ return {html,bind};
+})();
diff --git a/chat_test.py b/chat_test.py
new file mode 100644
index 0000000..af14e51
--- /dev/null
+++ b/chat_test.py
@@ -0,0 +1,91 @@
+"""Ephemeral admin chat tests using the same worker and leases as the API."""
+import json
+import socket
+import threading
+import time
+import uuid
+from inference import InferenceError
+
+class ChatTests:
+ def __init__(self,profiles,worker,scheduler):
+ self.profiles=profiles;self.worker=worker;self.scheduler=scheduler
+ self.lock=threading.RLock();self.job=None;self.cancel=threading.Event();self.socket=None
+ def status(self):
+ with self.lock:job=dict(self.job) if self.job else None
+ return dict(job=job,worker=self.worker.status(),scheduler=self.scheduler.status())
+ def start(self,data):
+ if set(data)!={'profile_id','messages','max_tokens'}:raise ValueError('Profil, Nachrichten und Antwortlimit erforderlich.')
+ messages=data['messages'];limit=data['max_tokens']
+ if type(limit)!=int or not 1<=limit<=4096:raise ValueError('Antwortlimit: 1–4096 Token.')
+ if not isinstance(messages,list) or not 1<=len(messages)<=32:raise ValueError('1–32 Textnachrichten erforderlich.')
+ for m in messages:
+ if not isinstance(m,dict) or set(m)!={'role','content'} or m['role'] not in ('user','assistant','system') or not isinstance(m['content'],str) or not m['content'].strip():raise ValueError('Nur Textnachrichten mit Rolle user, assistant oder system unterstützt.')
+ if sum(len(m['content']) for m in messages)>24000:raise ValueError('Testverlauf zu lang (maximal 24.000 Zeichen). Neuen Chat beginnen.')
+ profile=next((p for p in self.profiles.status()['profiles'] if p['id']==data['profile_id'] and p['kind']=='chat'),None)
+ if not profile or not profile['runnable']:raise ValueError('Kein ausführbares Chatprofil. Modell und CUDA-Build prüfen.')
+ with self.lock:
+ if self.job and self.job['state']=='running':raise ValueError('Ein Chat-Test läuft bereits.')
+ self.cancel.clear();self.job=dict(id=uuid.uuid4().hex,state='running',phase='Wartet auf freie Modellreservierung',profile_id=profile['id'],profile_name=profile['name'],started_at=time.time(),answer='',reasoning='',error=None)
+ threading.Thread(target=self._run,args=(profile,messages,limit),daemon=True).start()
+ return dict(self.job)
+ def phase(self,text):
+ with self.lock:self.job['phase']=text
+ def stop(self):
+ self.cancel.set()
+ with self.lock:sock=self.socket
+ if sock:
+ try:sock.shutdown(socket.SHUT_RDWR)
+ except OSError:pass
+ return {'cancellation_requested':True}
+ def unload(self):
+ if not self.scheduler.unload_idle():raise ValueError('Es laufen noch Anfragen. Erst deren Ende abwarten; andere Antworten werden nicht abgebrochen.')
+ return self.status()
+ def _run(self,profile,messages,limit):
+ conn=None;outcome='failed';message='Chat-Test fehlgeschlagen.';error=None
+ try:
+ key=('chat',profile['id'],profile['revision'])
+ def allowed():return not self.cancel.is_set() and any(p['id']==profile['id'] and p['revision']==profile['revision'] for p in self.profiles.status()['profiles'])
+ def prepare():
+ self.phase('Speicher wird geprüft · Profil wird geladen')
+ self.worker.ensure(profile,cancel=self.cancel.is_set)
+ with self.scheduler.lease(key,profile['parameters']['slots'],prepare,allowed=allowed):
+ if self.cancel.is_set():raise InterruptedError()
+ self.phase('Antwort wird erzeugt')
+ body=dict(model=profile['name'],messages=messages,max_tokens=limit,stream=True,**{k:profile['parameters'][k] for k in ('temperature','top_p','top_k')})
+ conn,token=self.worker.connect()
+ conn.request('POST','/v1/chat/completions',json.dumps(body).encode(),headers={'Content-Type':'application/json','Authorization':'Bearer '+token})
+ with self.lock:self.socket=conn.sock
+ if self.cancel.is_set():raise InterruptedError()
+ response=conn.getresponse()
+ if response.status!=200:
+ raise InferenceError('llama.cpp hat den Test abgelehnt. Kontextbudget, Nachrichtenlänge und Profilparameter prüfen.')
+ buffer=b'';total=0;done=False;deadline=time.monotonic()+600
+ while time.monotonic()4*1024**2:raise InferenceError('Testantwort überschreitet 4 MiB.')
+ while b'\n' in buffer:
+ line,buffer=buffer.split(b'\n',1);line=line.strip()
+ if not line.startswith(b'data:'):continue
+ payload=line[5:].strip()
+ if payload==b'[DONE]':done=True;break
+ event=json.loads(payload)
+ if 'error' in event:raise InferenceError('Der Modellworker meldet einen Fehler während der Antwort.')
+ for choice in event.get('choices',[]):
+ delta=choice.get('delta',{})
+ with self.lock:
+ for source,target in [('content','answer'),('reasoning_content','reasoning')]:
+ if isinstance(delta.get(source),str):self.job[target]+=delta[source]
+ if choice.get('finish_reason'):self.job['finish_reason']=choice['finish_reason']
+ if done:break
+ if not done:raise InferenceError('Antwort unterbrochen oder Zeitlimit erreicht. '+(self.worker.status().get('error') or 'Ein GPU-OOM ist ohne eindeutigen Nachweis nicht bestätigt.'))
+ outcome='complete';message='Antwort fertig · Modell bleibt für weitere Anfragen geladen'
+ except Exception as exc:
+ cancelled=self.cancel.is_set()
+ message='Chat-Test abgebrochen.' if cancelled else (str(exc) if isinstance(exc,InferenceError) else 'Verbindung zum Modell unterbrochen. '+(self.worker.status().get('error') or 'Speicherfehler, Modellabsturz oder Zeitlimit möglich; Ursache nicht eindeutig.'))
+ outcome='cancelled' if cancelled else 'failed';error=None if cancelled else message
+ finally:
+ if conn:conn.close()
+ with self.lock:self.socket=None;self.job.update(state=outcome,error=error,phase=message,finished_at=time.time())
diff --git a/deploy/Dockerfile b/deploy/Dockerfile
index 6ec2f73..3ac6541 100644
--- a/deploy/Dockerfile
+++ b/deploy/Dockerfile
@@ -12,8 +12,8 @@ RUN git clone https://github.com/Comfy-Org/ComfyUI.git /opt/deck-comfy \
&& /opt/deck-image-python/bin/pip freeze > /opt/deck-comfy/deck-requirements.lock
WORKDIR /app
COPY deploy/image-requirements.lock /app/deploy/image-requirements.lock
-COPY endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/
-COPY endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/
+COPY chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/
+COPY chat-test-ui.js endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/
COPY network/__init__.py network/client.py network/config.py network/rpc.py /app/network/
ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 HOME=/tmp \
DECK_BIND_HOST=0.0.0.0 DECK_STATE_DIR=/var/lib/deck \
diff --git a/deploy/install.py b/deploy/install.py
index 414110c..d8adc4c 100644
--- a/deploy/install.py
+++ b/deploy/install.py
@@ -14,7 +14,7 @@ import urllib.request
ROOT = Path(__file__).resolve().parent.parent
LABEL = 'de.casaderoll.athena-deck.standalone'
-FILES = ['endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock']
+FILES = ['chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock']
def run(*args, check=True, interactive=False):
diff --git a/index.html b/index.html
index 85eb1cd..4f53b67 100644
--- a/index.html
+++ b/index.html
@@ -1 +1 @@
-Athena Deck ATHENA CONTROL SURFACE v0.7 · Router
+Athena Deck ATHENA CONTROL SURFACE v0.7 · Router
diff --git a/inference.py b/inference.py
index 7ff2819..dbb0600 100644
--- a/inference.py
+++ b/inference.py
@@ -72,7 +72,7 @@ class Scheduler:
class LlamaWorker:
def __init__(self,root,catalog,runtime):
self.root=Path(root);self.catalog=catalog;self.runtime=runtime;self.lock=threading.RLock()
- self.process=None;self.fit_process=None;self.generation=0;self.port=None;self.profile=None;self.state='stopped';self.error=None;self.key=None;self.devices=[]
+ self.process=None;self.fit_process=None;self.generation=0;self.port=None;self.profile=None;self.state='stopped';self.error=None;self.key=None;self.devices=[];self.memory_plan=None;self.phase='Gestoppt';self.oom_before=0
def build(self):
state=self.runtime.status()
build=next((b for b in state['builds'] if b['id']==state['active']),None)
@@ -88,10 +88,17 @@ class LlamaWorker:
if not p.get('model') or not p['model']['file'].endswith('.gguf'):raise InferenceError('Ein lokales GGUF-Modell wird benötigt.')
return []
except ValueError as exc:return [str(exc)]
+ @staticmethod
+ def oom_count():
+ try:return int(dict(line.split() for line in Path('/sys/fs/cgroup/memory.events').read_text().splitlines()).get('oom_kill',0))
+ except (OSError,ValueError):return 0
+ def crash_message(self):
+ if self.oom_count()>self.oom_before:return 'OOM-Kill im Deck-RAM-Bereich erkannt. Das RAM-Limit wurde überschritten; Kontext oder Modellgröße reduzieren.'
+ return 'Modellprozess beendet. GPU-OOM oder Modell-/Buildfehler möglich; Ursache nicht eindeutig bestätigt. Kontext oder Microbatch reduzieren und erneut testen.'
def status(self):
with self.lock:
- if self.process and self.process.poll() is not None and self.state=='ready':self.state='failed';self.error='llama.cpp wurde unerwartet beendet.'
- return dict(state=self.state,profile_id=self.profile['id'] if self.profile else None,profile_name=self.profile['name'] if self.profile else None,port=self.port,error=self.error,gpus=list(self.devices))
+ if self.process and self.process.poll() is not None and self.state=='ready':self.state='failed';self.error=self.crash_message()
+ return dict(state=self.state,profile_id=self.profile['id'] if self.profile else None,profile_name=self.profile['name'] if self.profile else None,port=self.port,error=self.error,gpus=list(self.devices),memory_plan=self.memory_plan,phase=self.phase)
def stop(self):
with self.lock:
self.generation+=1
@@ -106,18 +113,18 @@ class LlamaWorker:
try:p.wait(timeout=10)
except subprocess.TimeoutExpired:os.killpg(p.pid,signal.SIGKILL);p.wait(timeout=5)
except ProcessLookupError:pass
- self.process=None;self.profile=None;self.port=None;self.state='stopped';self.devices=[];self.key=None
+ self.process=None;self.profile=None;self.port=None;self.state='stopped';self.phase='Gestoppt';self.devices=[];self.key=None
(self.root/'worker.key').unlink(missing_ok=True)
- def ensure(self,profile):
+ def ensure(self,profile,cancel=lambda:False):
with self.lock:
if self.process and self.process.poll() is None and self.profile and (self.profile['id'],self.profile['revision'])==(profile['id'],profile['revision']):return
- self.stop();self.state='loading';self.error=None;generation=self.generation
- try:self._start(profile,generation)
+ self.stop();self.state='loading';self.error=None;self.memory_plan=None;self.phase='Speicherprüfung';self.oom_before=self.oom_count();generation=self.generation
+ try:self._start(profile,generation,cancel)
except Exception as exc:
self.stop()
- with self.lock:self.state='failed';self.error=str(exc) if isinstance(exc,ValueError) else 'llama.cpp konnte nicht gestartet werden.'
+ with self.lock:self.state='failed';self.phase='Modellstart fehlgeschlagen';self.error=str(exc) if isinstance(exc,ValueError) else 'llama.cpp konnte nicht gestartet werden.'
raise InferenceError(self.error) from None
- def _start(self,profile,generation):
+ def _start(self,profile,generation,cancel=lambda:False):
directory=self.build();params=profile['parameters'];entry=self.catalog.entry(profile['model_id'])
model=(self.catalog.root/entry['id']/('model'+Path(entry['file']).suffix)).resolve()
devices=probe();ids=params['gpu_devices']
@@ -141,7 +148,8 @@ class LlamaWorker:
tool=str(directory/'build/bin/llama-fit-params')
margins=[max(1024,round(g['total_mib']*.05)) for g in selected]
def estimate(layers,extra=()):
- output=self._fit_command([tool]+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation)
+ if cancel():raise InferenceError('Modellstart abgebrochen.')
+ output=self._fit_command([tool]+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation,cancel)
rows={}
for line in output.splitlines():
parts=line.split()
@@ -150,6 +158,7 @@ class LlamaWorker:
gpu_fits=all(rows['CUDA'+str(i)]+margins[i]<=g['free_mib'] for i,g in enumerate(selected))
need=max(staging,rows['Host']*1024**2+4*GIB)
host_fits=mem.get('MemAvailable',0)>=need+4*GIB and (headroom is None or headroom>=need)
+ with self.lock:self.memory_plan=dict(profile_name=profile['name'],estimated=True,gpu_layers=layers,context=params['context'],slots=params['slots'],fits=gpu_fits and host_fits,host_required_mib=round(need/1024**2),host_available_mib=round(min(mem.get('MemAvailable',0)-4*GIB,headroom if headroom is not None else mem.get('MemAvailable',0))/1024**2),gpus=[dict(name=g.get('name',g['uuid']),uuid=g['uuid'],free_mib=g['free_mib'],required_mib=rows['CUDA'+str(i)],reserve_mib=margins[i]) for i,g in enumerate(selected)])
return gpu_fits and host_fits,rows
extra=[]
if params['tensor_split']:
@@ -165,8 +174,9 @@ class LlamaWorker:
else:high=middle-1
if best is None:raise InferenceError('Kontext und fester GPU-Split passen nicht sicher in GPU/RAM.')
layers,memory=best
+ estimate(layers)
else:
- output=self._fit_command([tool]+common,env,generation)
+ output=self._fit_command([tool]+common,env,generation,cancel)
flags=shlex.split(output.strip());fit={}
if len(flags)%2:raise InferenceError('Fit-Werkzeug lieferte ungültige Parameter.')
for i in range(0,len(flags),2):
@@ -180,6 +190,8 @@ class LlamaWorker:
extra=['--tensor-split',fit['-ts']]
fits,memory=estimate(layers,extra)
if not fits:raise InferenceError('Speicherprognose überschreitet die GPU-/RAM-Reserve.')
+ if cancel():raise InferenceError('Modellstart abgebrochen.')
+ with self.lock:self.phase='Modell wird geladen'
launch=common+extra+['--gpu-layers',str(layers),'--fit','off','--kv-unified','--threads',str(params['threads']),'--load-mode','none','--host','127.0.0.1','--alias',profile['name'],'--no-webui','--log-disable']
self.root.mkdir(parents=True,exist_ok=True,mode=0o700)
with socket.socket() as sock:sock.bind(('127.0.0.1',0));port=sock.getsockname()[1]
@@ -193,27 +205,35 @@ class LlamaWorker:
self.profile=profile;self.port=port;self.key=key;self.devices=[g['uuid'] for g in selected]
deadline=time.monotonic()+300
while time.monotonic()1 for g in selected):raise InferenceError('Ein weiterer Prozess verwendet die reservierte GPU. Deck bricht seinen Start ab.')
conn=http.client.HTTPConnection('127.0.0.1',port,timeout=2)
try:
conn.request('GET','/health',headers={'Authorization':'Bearer '+key});r=conn.getresponse()
if r.status==200:
- with self.lock:self.state='ready'
+ with self.lock:self.state='ready';self.phase='Modell bereit'
threading.Thread(target=self._monitor,args=(generation,),daemon=True).start()
return
except OSError:pass
finally:conn.close()
time.sleep(1)
raise InferenceError('llama.cpp wurde nicht rechtzeitig bereit.')
- def _fit_command(self,args,env,generation):
+ def _fit_command(self,args,env,generation,cancel=lambda:False):
with self.lock:
if self.generation!=generation:raise InferenceError('Modellstart abgebrochen.')
process=subprocess.Popen(args,env=env,stdout=subprocess.PIPE,stderr=subprocess.DEVNULL,text=True)
self.fit_process=process
- try:output,_=process.communicate(timeout=180)
+ try:
+ deadline=time.monotonic()+180
+ while True:
+ if cancel():
+ process.terminate();process.communicate(timeout=3);raise InferenceError('Modellstart abgebrochen.')
+ try:output,_=process.communicate(timeout=1);break
+ except subprocess.TimeoutExpired:
+ if time.monotonic()>=deadline:raise
except subprocess.TimeoutExpired:
process.kill();process.communicate();raise InferenceError('Zeitlimit der Speicher-Einpassung überschritten.') from None
finally:
diff --git a/network/install_remote.py b/network/install_remote.py
index fa67377..12ed217 100644
--- a/network/install_remote.py
+++ b/network/install_remote.py
@@ -12,7 +12,7 @@ import sys
NAME = 'athena-deck-network'
BASE = Path('/opt/athena-deck')
-FILES = {'endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'}
+FILES = {'chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'}
def run(*args, **kwargs):
return subprocess.run(args, capture_output=True, timeout=600, **kwargs)
diff --git a/server.py b/server.py
index 2bb3f2f..40566f5 100644
--- a/server.py
+++ b/server.py
@@ -15,6 +15,7 @@ from image_test import ImageTests
from docker_support import DockerSupport
from inference import LlamaWorker,Scheduler
from endpoint import Endpoint
+from chat_test import ChatTests
from urllib.parse import urlsplit, parse_qs
from network.client import NetworkClient
from network.config import parse_config, ConfigError
@@ -88,6 +89,7 @@ class Server(ThreadingHTTPServer):
if self.endpoint.config['autostart']:
try:self.endpoint.start()
except ValueError as exc:self.endpoint.error=str(exc)
+ self.chat_tests=ChatTests(self.profiles,self.worker,self.scheduler)
self.sessions = {}
self.login_attempts = []
self.auth_lock = threading.Lock()
@@ -239,7 +241,7 @@ class Handler(BaseHTTPRequestHandler):
return self.respond({'error':'Anmeldung erforderlich.'},401)
if not self.authenticated() and self.path == '/':
return self.respond((ROOT/'login.html').read_bytes(), mime='text/html; charset=utf-8')
- routes = {'/endpoint-ui.js':('endpoint-ui.js','text/javascript'),'/docker-ui.js': ('docker-ui.js','text/javascript'), '/': ('index.html', 'text/html; charset=utf-8'), '/app.js': ('app.js', 'text/javascript'), '/style.css': ('style.css', 'text/css'), '/network-ui.js': ('network-ui.js', 'text/javascript'), '/access-ui.js': ('access-ui.js', 'text/javascript'), '/studio.js': ('studio.js', 'text/javascript'), '/catalog-ui.js': ('catalog-ui.js','text/javascript'), '/runtime-ui.js': ('runtime-ui.js','text/javascript'), '/profiles-ui.js': ('profiles-ui.js','text/javascript'), '/image-test-ui.js': ('image-test-ui.js','text/javascript')}
+ routes = {'/chat-test-ui.js':('chat-test-ui.js','text/javascript'),'/endpoint-ui.js':('endpoint-ui.js','text/javascript'),'/docker-ui.js': ('docker-ui.js','text/javascript'), '/': ('index.html', 'text/html; charset=utf-8'), '/app.js': ('app.js', 'text/javascript'), '/style.css': ('style.css', 'text/css'), '/network-ui.js': ('network-ui.js', 'text/javascript'), '/access-ui.js': ('access-ui.js', 'text/javascript'), '/studio.js': ('studio.js', 'text/javascript'), '/catalog-ui.js': ('catalog-ui.js','text/javascript'), '/runtime-ui.js': ('runtime-ui.js','text/javascript'), '/profiles-ui.js': ('profiles-ui.js','text/javascript'), '/image-test-ui.js': ('image-test-ui.js','text/javascript')}
if self.path in routes:
name, mime = routes[self.path]
return self.respond((ROOT/name).read_bytes(), mime=mime)
@@ -247,6 +249,7 @@ class Handler(BaseHTTPRequestHandler):
endpoint=self.server.endpoint.status()
public_endpoint={key:endpoint[key] for key in ('state','port','counts','active_requests')}
return self.respond(dict(name='Athena Deck', version='0.7.0', state='ready', uptime_seconds=round(time.time()-self.server.started), mode='isolated', location=os.environ.get('DECK_LOCATION', 'Athena · Debian-Server'), endpoint=public_endpoint))
+ if self.path == '/api/v1/chat-tests':return self.respond(self.server.chat_tests.status())
if self.path == '/api/v1/endpoint':return self.respond(self.server.endpoint.status())
if self.path == '/api/v1/docker':return self.respond(self.server.docker.status())
if self.path == '/api/v1/image-runtime':return self.respond(self.server.image_runtime.status())
@@ -334,6 +337,13 @@ class Handler(BaseHTTPRequestHandler):
raise ValueError('Ungültige Laufzeitaktion.')
except ValueError as exc:return self.respond({'error':str(exc)},400)
except (OSError, subprocess.SubprocessError):return self.respond({'error':'Laufzeitaktion fehlgeschlagen; Speicher und Werkzeuge prüfen.'},503)
+ if self.path in ('/api/v1/chat-tests/start','/api/v1/chat-tests/cancel','/api/v1/chat-tests/unload'):
+ try:
+ data=self.read_json()
+ if self.path.endswith('/start'):return self.respond(self.server.chat_tests.start(data))
+ if data:raise ValueError('Keine Parameter erwartet.')
+ return self.respond(self.server.chat_tests.stop() if self.path.endswith('/cancel') else self.server.chat_tests.unload())
+ except ValueError as exc:return self.respond({'error':str(exc)},400)
if self.path in ('/api/v1/endpoint/start','/api/v1/endpoint/stop','/api/v1/endpoint/config','/api/v1/endpoint/profile'):
try:
data=self.read_json();ep=self.server.endpoint
@@ -423,6 +433,7 @@ def main():
try:
server.serve_forever()
finally:
+ server.chat_tests.stop()
server.endpoint.close()
server.image_runtime.stop()
server.image_tests.stop()
diff --git a/studio.js b/studio.js
index 4fac666..353f62b 100644
--- a/studio.js
+++ b/studio.js
@@ -6,13 +6,14 @@ const Studio=(()=>{
if(!force&&document.querySelector('#studio')?.dataset.page===page)return;
if(category!==page){category=page;section='discover';}
if(modelId)section='profiles';
- const body=['docker-runtime','services'].includes(page)?DockerUI.html(page==='services'):page==='runtimes'?runtimeOverview():page==='image-runtime'?imageRuntime():page==='runtime'?RuntimeUI.html():`ATHENA / MODELLVERWALTUNG
${labels[page]} Modelle entdecken, herunterladen und mit gespeicherten Profilen konfigurieren.
${[['discover','Entdecken'],['library','Bibliothek'],['downloads','Downloads'],['profiles','Profile'],...(page==='image'?[['test','Testen']]:[]),['running','Laufend']].map(([id,label])=>`${label} `).join('')}
${section==='test'?ImageTestUI.html():section==='running'&&page==='image'?ImageTestUI.html(true):section==='profiles'?ProfilesUI.html():section==='running'&&page==='chat'?EndpointUI.runningHTML():section==='running'?'
Keine von Deck gestarteten Modelle Profile werden auf Athena gespeichert. Für diesen Bereich ist noch kein Modellworker angebunden. Fehlende Komponenten stehen beim jeweiligen Profil.
llama.cpp-Builds verwalten → ':CatalogUI.html(section==='library',section==='downloads')}
`;
+ const body=['docker-runtime','services'].includes(page)?DockerUI.html(page==='services'):page==='runtimes'?runtimeOverview():page==='image-runtime'?imageRuntime():page==='runtime'?RuntimeUI.html():`ATHENA / MODELLVERWALTUNG
${labels[page]} Modelle entdecken, herunterladen und mit gespeicherten Profilen konfigurieren.
${[['discover','Entdecken'],['library','Bibliothek'],['downloads','Downloads'],['profiles','Profile'],...(['image','chat'].includes(page)?[['test','Testen']]:[]),['running','Laufend']].map(([id,label])=>`${label} `).join('')}
${section==='test'?(page==='chat'?ChatTestUI.html():ImageTestUI.html()):section==='running'&&page==='image'?ImageTestUI.html(true):section==='profiles'?ProfilesUI.html():section==='running'&&page==='chat'?EndpointUI.runningHTML():section==='running'?'
Keine von Deck gestarteten Modelle Profile werden auf Athena gespeichert. Für diesen Bereich ist noch kein Modellworker angebunden. Fehlende Komponenten stehen beim jeweiligen Profil.
llama.cpp-Builds verwalten → ':CatalogUI.html(section==='library',section==='downloads')}
`;
document.querySelector('#view').innerHTML=`${body}
`;
if(['docker-runtime','services'].includes(page)){DockerUI.bind(page==='services');return;}
if(page==='runtime'){RuntimeUI.bind();return;}
if(page==='image-runtime'){ImageTestUI.bind(true);ImageTestUI.bindInstaller();return;}
if(page==='runtimes')return;
- if(section==='test'||(section==='running'&&page==='image'))ImageTestUI.bind(section==='running');
+ if(section==='test'&&page==='chat')ChatTestUI.bind();
+ else if(section==='test'||(section==='running'&&page==='image'))ImageTestUI.bind(section==='running');
else if(section==='running'&&page==='chat')EndpointUI.bindRunning();
else if(section==='profiles')ProfilesUI.bind(page,modelId);
else if(section!=='running')CatalogUI.bind(page,section==='library',section==='downloads');
diff --git a/style.css b/style.css
index f86e607..8e15d30 100644
--- a/style.css
+++ b/style.css
@@ -18,3 +18,5 @@ nav a.nav-sub{padding-left:28px;font-size:12px;color:#a4bda8}.component-card{mar
#image-test textarea{width:100%;background:#101718;color:#e4eee5;border:1px solid #465348;border-radius:8px;padding:14px;font:inherit;resize:vertical;margin:10px 0 20px}.generated-image{max-width:100%;height:auto;display:block;border-radius:8px}#image-test-result figure{margin:20px 0}#image-test-state{margin-top:20px}
.hub-base-filter{display:flex;align-items:center;gap:.6rem;margin-top:1rem}.hub-base-filter input{width:auto;margin:0}
+
+.chat-text,#chat-transcript pre{white-space:pre-wrap;overflow-wrap:anywhere;}
diff --git a/test_chat_test.py b/test_chat_test.py
new file mode 100644
index 0000000..5b02e26
--- /dev/null
+++ b/test_chat_test.py
@@ -0,0 +1,50 @@
+import json
+import threading
+import time
+import unittest
+from types import SimpleNamespace
+from unittest.mock import Mock
+from chat_test import ChatTests
+from inference import Scheduler,InferenceError,LlamaWorker
+
+class Reply:
+ status=200
+ def __init__(self,events):self.events=iter(events)
+ def read1(self,_):return next(self.events,b'')
+
+class Worker:
+ def __init__(self):self.error=None;self.loaded=False;self.conn=Mock();self.conn.getresponse.return_value=Reply([b'data: {"choices":[{"delta":{"content":"OK"}}]}\n\n',b'data: [DONE]\n\n'])
+ def stop(self):self.loaded=False
+ def ensure(self,p,cancel=lambda:False):
+ if self.error:raise InferenceError(self.error)
+ self.loaded=True
+ def status(self):return dict(state='ready' if self.loaded else 'stopped',error=self.error)
+ def connect(self):return self.conn,'synthetic-only'
+
+class ChatTestsTests(unittest.TestCase):
+ def setUp(self):
+ self.profile=dict(id='p',name='test',revision=1,kind='chat',runnable=True,parameters=dict(slots=2,temperature=.2,top_p=.8,top_k=20))
+ self.worker=Worker();self.manager=ChatTests(SimpleNamespace(status=lambda:dict(profiles=[self.profile])),self.worker,Scheduler(self.worker))
+ def request(self):return dict(profile_id='p',messages=[dict(role='user',content='synthetic test')],max_tokens=8)
+ def finish(self):
+ deadline=time.monotonic()+3
+ while self.manager.status()['job']['state']=='running' and time.monotonic()