From 7ada7f6f20a5b32105f71606126e36acd976bb1a Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Mon, 28 Sep 2026 20:43:40 +0200 Subject: [PATCH] Add internal chat testing with memory diagnostics and cancellation --- ENDPOINT.md | 25 +++++++++++ README.md | 2 +- STUDIO.md | 2 + chat-test-ui.js | 27 ++++++++++++ chat_test.py | 91 +++++++++++++++++++++++++++++++++++++++ deploy/Dockerfile | 4 +- deploy/install.py | 2 +- index.html | 2 +- inference.py | 50 ++++++++++++++------- network/install_remote.py | 2 +- server.py | 13 +++++- studio.js | 5 ++- style.css | 2 + test_chat_test.py | 50 +++++++++++++++++++++ test_server.py | 12 ++++++ 15 files changed, 265 insertions(+), 24 deletions(-) create mode 100644 chat-test-ui.js create mode 100644 chat_test.py create mode 100644 test_chat_test.py diff --git a/ENDPOINT.md b/ENDPOINT.md index 44dc3ac..156171d 100644 --- a/ENDPOINT.md +++ b/ENDPOINT.md @@ -129,3 +129,28 @@ Listener gestoppt und Modell entladen. Keine vorhandenen Nutzerprompts oder Anwendungslogs gelesen. Testprofile und Testzugang gehören nicht zur normalen Deck-Konfiguration. Der Test weist keine vollständige OpenAI-Kompatibilität oder Langzeitstabilität nach. + + +## Sprachmodelle → Testen + +Der interne Testchat verwendet denselben Scheduler und llama.cpp-Worker wie der +API-Endpunkt, braucht aber keine Veröffentlichung des Profils und keinen +API-Token im Browser. Profilauswahl, fortlaufender Textchat, Antwortlimit (bis +4096 Token), Abbrechen und explizites Entladen stehen bereit. Der Modellprozess +bleibt nach einer Antwort geladen; andere Profile/Bildaufträge wechseln regulär. +Entladen wird bei aktiven Anfragen abgewiesen. Abbrechen schließt nur die eigene +Chatverbindung; während des eigenen Modellstarts bricht es diesen Start ab. + +Der Verlauf liegt nur im Browser-Arbeitsspeicher. Beim Wechsel der Ansicht oder +des Profils beginnt ein neuer Verlauf. Der Server hält nur den aktuellen Testjob +mit gestreamter Antwort vorübergehend im RAM; weder Prompt noch Antwort werden +als Chatdatei oder Log gespeichert. Reguläre Admin-Sitzung/CSRF-Schutz gelten für +GET `/api/v1/chat-tests` und POST `/api/v1/chat-tests/start`, `/cancel`, `/unload`. +Start: `profile_id`, Textnachrichten `messages`, `max_tokens`. Kein Bearer-Zugriff. + +Diagnose zeigt Ladephase, Wartezeit/Reservierungen, reale GPU-Belegung und die +Speicherprognose mit Reserve je GPU, RAM-Budget und GPU-Layerlimit. Eine Prognose +ist kein Nachweis für fehlerfreien Betrieb. Bestätigte Cgroup-OOM-Kills werden +als RAM-OOM gemeldet; ein GPU-OOM wird ohne eindeutigen Nachweis nicht behauptet. +Ein nicht eindeutig diagnostizierter Prozessabbruch nennt Speicher, Modell/Build +oder Zeitlimit als mögliche Ursachen. Bestehende Anwendungslogs werden nicht gelesen. diff --git a/README.md b/README.md index 611b284..a7b71e3 100644 --- a/README.md +++ b/README.md @@ -17,7 +17,7 @@ Katalog, Datei-Downloads, Bibliothek, serverseitig gespeicherte Profile und llama.cpp-Buildverwaltung sind live. Der Download-Reiter erlaubt das Ausblenden abgeschlossener Einträge ohne Dateiverlust. Entdecken zeigt Größen und eine konservative Gewichts-Speicherprüfung; noch keine vollständige Laufzeitprognose. -Bildprofile mit vollständigem Qwen-Image-2.1-GGUF-Rezept lassen sich unter „Testen“ ausführen. Textprofile sind am eigenen [OpenAI-kompatiblen Endpunkt](ENDPOINT.md) ausführbar. Audio und Video sind noch nicht angebunden. Details: [Modellverwaltung](STUDIO.md). +Bildprofile mit vollständigem Qwen-Image-2.1-GGUF-Rezept lassen sich unter „Testen“ ausführen. Textprofile sind am eigenen [OpenAI-kompatiblen Endpunkt](ENDPOINT.md) ausführbar. Audio und Video sind noch nicht angebunden. Sprachmodelle besitzen außerdem einen internen Testchat mit Abbruch und Speicherdiagnose. Details: [Modellverwaltung](STUDIO.md). ## Zugang und API-Token diff --git a/STUDIO.md b/STUDIO.md index ba1e37b..8aedd36 100644 --- a/STUDIO.md +++ b/STUDIO.md @@ -155,3 +155,5 @@ Leere Geräte-/Gewichtelisten bedeuten automatische Auswahl/Verteilung. Alte Profile bleiben lesbar; fehlende neue Felder erhalten bei Anzeige/Speichern Standardwerte (automatische Geräte, keine Aufteilung, Temperature 0.8, Top-p 0.95, Top-k 40). Bestehende Dateien werden beim Lesen nicht umgeschrieben. + +Sprachmodelle besitzen jetzt den Reiter **Testen**: interner Textchat mit Streaming, Profilauswahl und Speicherdiagnose. Öffentliche Profilfreigabe ist dafür nicht erforderlich. Details: [ENDPOINT.md](ENDPOINT.md#sprachmodelle--testen). diff --git a/chat-test-ui.js b/chat-test-ui.js new file mode 100644 index 0000000..c5c250d --- /dev/null +++ b/chat-test-ui.js @@ -0,0 +1,27 @@ +window.ChatTestUI=(()=>{ + const e=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); + const gib=v=>v==null?'unbekannt':(v/1024).toFixed(2)+' GiB'; + async function api(path,data){const r=await fetch('/api/v1/'+path,data===undefined?{}:{method:'POST',headers:{'Content-Type':'application/json','X-Athena-Deck':'1'},body:JSON.stringify(data)});const v=await r.json();if(!r.ok)throw Error(v.error||'Anfrage fehlgeschlagen');return v;} + function put(el,html){if(el._chatHtml!==html){el.innerHTML=html;el._chatHtml=html;}} + function html(){return `

Sprachmodell testen

Ein einfacher Textchat mit deinem gespeicherten Profil. Nutzt denselben Worker wie API-Anfragen; eine öffentliche Profilfreigabe ist nicht erforderlich.

Der Verlauf bleibt nur in dieser Ansicht und wird bei Profilwechsel oder Neuladen verworfen. Prompts und Antworten werden nicht als Chatdatei gespeichert.

Laufzeit & Diagnose

Entladen ist nur ohne laufende Anfragen möglich. Abbrechen beendet nur diesen Test; andere Antworten werden nicht gestoppt.

Speicherprüfung

Prognose mit Reserve, keine Garantie für jede Anfrage. GPU-Layer können automatisch auf die CPU ausgelagert werden. Bei Fehlern helfen häufig kleinerer Kontext oder Microbatch. Hardware-Messwerte →

`;} + function bind(){ + const root=document.querySelector('#chat-test'),el=id=>root.querySelector('#'+id);let profiles=[],messages=[],turns=[],ownJob=null,finished=null,status=null,pending=false; + const notice=t=>{if(root.isConnected)el('chat-message').textContent=t;}; + function transcript(){put(el('chat-transcript'),turns.map(t=>`
${t.role==='user'?'Du':'Modell'}

${e(t.content)}

${t.reasoning?`
Reasoning-Ausgabe
${e(t.reasoning)}
`:''}
`).join(''));} + function selection(){const p=profiles.find(p=>p.id===el('chat-profile').value),running=status?.job?.state==='running';el('chat-profile-info').textContent=p?`${p.parameters.context.toLocaleString('de-DE')} Token gesamt · ${p.parameters.slots} Slots · Batch ${p.parameters.batch} / Microbatch ${p.parameters.ubatch}${p.blockers.length?' · '+p.blockers.join(' '):''}`:'Zuerst ein Sprachmodellprofil anlegen.';el('chat-send').disabled=pending||running||!p?.runnable;el('chat-profile').disabled=pending||running;el('chat-clear').disabled=pending||running;el('chat-cancel').disabled=!running;el('chat-unload').disabled=!!status?.scheduler.active_requests;} + async function refresh(){try{const [s,h]=await Promise.all([api('chat-tests'),api('hardware').catch(()=>({gpus:[]}))]);if(!root.isConnected)return;status=s;const j=s.job,w=s.worker,plan=w.memory_plan; + put(el('chat-state'),`

${e(j?.phase||'Noch kein Chat-Test gestartet.')}

Worker: ${e(w.phase||w.state)} · ${e(w.profile_name||'kein Modell geladen')}
${s.scheduler.active_requests} laufende / ${s.scheduler.waiting_requests} wartende Anfragen

${j?.error?`

${e(j.error)}

`:''}${w.error?`

${e(w.error)}

`:''}${j?`

Dauer: ${Math.max(0,Math.round((j.finished_at||Date.now()/1000)-j.started_at))} s

`:''}`); + put(el('chat-memory'),(plan?`

Letzte Prognose für ${e(plan.profile_name||w.profile_name||'das Modell')}

${plan.fits?'Prognose passt einschließlich Reserve':'Prognose passt nicht in den verfügbaren Speicher'} · ${plan.context.toLocaleString('de-DE')} Token / ${plan.slots} Slots · GPU-Layerlimit ${plan.gpu_layers===999?'alle':plan.gpu_layers}

${plan.gpus.map(g=>`

${e(g.name)}: geschätzt ${gib(g.required_mib)} + ${gib(g.reserve_mib)} Reserve / ${gib(g.free_mib)} beim Prüfen frei

`).join('')}

RAM-Budget: ${gib(plan.host_required_mib)} benötigt / ${gib(plan.host_available_mib)} verfügbar

`:'

Noch keine Speicherprognose für einen Modellstart vorhanden.

')+`

Aktuelle GPU-Belegung

${(h.gpus||[]).map(g=>`

${e(g.name)}: ${gib(g.used_mib)} / ${gib(g.total_mib)} belegt · ${g.percent??'—'} % Auslastung

`).join('')||'

GPU-Messwerte nicht verfügbar.

'}`); + if(j&&j.id===ownJob){let response=turns[turns.length-1];if(response?.role!=='assistant'){response={role:'assistant',content:'',reasoning:''};turns.push(response);}response.content=j.answer;response.reasoning=j.reasoning;transcript(); + if(j.state!=='running'&&finished!==j.id){finished=j.id;if(j.state==='complete'&&j.answer)messages.push({role:'assistant',content:j.answer});else{messages=[];notice('Der nächste Prompt beginnt einen neuen Kontext; der letzte Test wurde nicht vollständig beantwortet.');}if(j.finish_reason==='length')notice('Antwort am Tokenlimit beendet. Bei Bedarf das Antwortlimit erhöhen.');} + }selection(); + }catch(error){notice(error.message);}finally{if(root.isConnected)setTimeout(refresh,1500);}} + api('profiles').then(s=>{if(!root.isConnected)return;profiles=s.profiles.filter(p=>p.kind==='chat');el('chat-profile').innerHTML=profiles.map(p=>``).join('')||'';selection();}).catch(err=>notice(err.message)); + const clear=()=>{messages=[];turns=[];ownJob=null;finished=null;transcript();notice('Neuer Chat.');};el('chat-clear').onclick=clear;el('chat-profile').onchange=()=>{clear();selection();}; + el('chat-test-form').onsubmit=async event=>{event.preventDefault();pending=true;selection();const text=el('chat-prompt').value.trim();try{const next=[...messages,{role:'user',content:text}];const job=await api('chat-tests/start',{profile_id:el('chat-profile').value,messages:next,max_tokens:Number(el('chat-limit').value)});messages=next;turns.push({role:'user',content:text});ownJob=job.id;finished=null;status={...status,job};el('chat-prompt').value='';transcript();notice('Test gestartet. Speicherprüfung, Modellstart und Antwort erscheinen unter Diagnose.');}catch(error){notice(error.message);}finally{pending=false;selection();}}; + el('chat-cancel').onclick=async()=>{try{await api('chat-tests/cancel',{});notice('Abbruch angefordert.');}catch(error){notice(error.message);}}; + el('chat-unload').onclick=async()=>{try{await api('chat-tests/unload',{});notice('Eigener Modellprozess entladen.');}catch(error){notice(error.message);}}; + refresh(); + } + return {html,bind}; +})(); diff --git a/chat_test.py b/chat_test.py new file mode 100644 index 0000000..af14e51 --- /dev/null +++ b/chat_test.py @@ -0,0 +1,91 @@ +"""Ephemeral admin chat tests using the same worker and leases as the API.""" +import json +import socket +import threading +import time +import uuid +from inference import InferenceError + +class ChatTests: + def __init__(self,profiles,worker,scheduler): + self.profiles=profiles;self.worker=worker;self.scheduler=scheduler + self.lock=threading.RLock();self.job=None;self.cancel=threading.Event();self.socket=None + def status(self): + with self.lock:job=dict(self.job) if self.job else None + return dict(job=job,worker=self.worker.status(),scheduler=self.scheduler.status()) + def start(self,data): + if set(data)!={'profile_id','messages','max_tokens'}:raise ValueError('Profil, Nachrichten und Antwortlimit erforderlich.') + messages=data['messages'];limit=data['max_tokens'] + if type(limit)!=int or not 1<=limit<=4096:raise ValueError('Antwortlimit: 1–4096 Token.') + if not isinstance(messages,list) or not 1<=len(messages)<=32:raise ValueError('1–32 Textnachrichten erforderlich.') + for m in messages: + if not isinstance(m,dict) or set(m)!={'role','content'} or m['role'] not in ('user','assistant','system') or not isinstance(m['content'],str) or not m['content'].strip():raise ValueError('Nur Textnachrichten mit Rolle user, assistant oder system unterstützt.') + if sum(len(m['content']) for m in messages)>24000:raise ValueError('Testverlauf zu lang (maximal 24.000 Zeichen). Neuen Chat beginnen.') + profile=next((p for p in self.profiles.status()['profiles'] if p['id']==data['profile_id'] and p['kind']=='chat'),None) + if not profile or not profile['runnable']:raise ValueError('Kein ausführbares Chatprofil. Modell und CUDA-Build prüfen.') + with self.lock: + if self.job and self.job['state']=='running':raise ValueError('Ein Chat-Test läuft bereits.') + self.cancel.clear();self.job=dict(id=uuid.uuid4().hex,state='running',phase='Wartet auf freie Modellreservierung',profile_id=profile['id'],profile_name=profile['name'],started_at=time.time(),answer='',reasoning='',error=None) + threading.Thread(target=self._run,args=(profile,messages,limit),daemon=True).start() + return dict(self.job) + def phase(self,text): + with self.lock:self.job['phase']=text + def stop(self): + self.cancel.set() + with self.lock:sock=self.socket + if sock: + try:sock.shutdown(socket.SHUT_RDWR) + except OSError:pass + return {'cancellation_requested':True} + def unload(self): + if not self.scheduler.unload_idle():raise ValueError('Es laufen noch Anfragen. Erst deren Ende abwarten; andere Antworten werden nicht abgebrochen.') + return self.status() + def _run(self,profile,messages,limit): + conn=None;outcome='failed';message='Chat-Test fehlgeschlagen.';error=None + try: + key=('chat',profile['id'],profile['revision']) + def allowed():return not self.cancel.is_set() and any(p['id']==profile['id'] and p['revision']==profile['revision'] for p in self.profiles.status()['profiles']) + def prepare(): + self.phase('Speicher wird geprüft · Profil wird geladen') + self.worker.ensure(profile,cancel=self.cancel.is_set) + with self.scheduler.lease(key,profile['parameters']['slots'],prepare,allowed=allowed): + if self.cancel.is_set():raise InterruptedError() + self.phase('Antwort wird erzeugt') + body=dict(model=profile['name'],messages=messages,max_tokens=limit,stream=True,**{k:profile['parameters'][k] for k in ('temperature','top_p','top_k')}) + conn,token=self.worker.connect() + conn.request('POST','/v1/chat/completions',json.dumps(body).encode(),headers={'Content-Type':'application/json','Authorization':'Bearer '+token}) + with self.lock:self.socket=conn.sock + if self.cancel.is_set():raise InterruptedError() + response=conn.getresponse() + if response.status!=200: + raise InferenceError('llama.cpp hat den Test abgelehnt. Kontextbudget, Nachrichtenlänge und Profilparameter prüfen.') + buffer=b'';total=0;done=False;deadline=time.monotonic()+600 + while time.monotonic()4*1024**2:raise InferenceError('Testantwort überschreitet 4 MiB.') + while b'\n' in buffer: + line,buffer=buffer.split(b'\n',1);line=line.strip() + if not line.startswith(b'data:'):continue + payload=line[5:].strip() + if payload==b'[DONE]':done=True;break + event=json.loads(payload) + if 'error' in event:raise InferenceError('Der Modellworker meldet einen Fehler während der Antwort.') + for choice in event.get('choices',[]): + delta=choice.get('delta',{}) + with self.lock: + for source,target in [('content','answer'),('reasoning_content','reasoning')]: + if isinstance(delta.get(source),str):self.job[target]+=delta[source] + if choice.get('finish_reason'):self.job['finish_reason']=choice['finish_reason'] + if done:break + if not done:raise InferenceError('Antwort unterbrochen oder Zeitlimit erreicht. '+(self.worker.status().get('error') or 'Ein GPU-OOM ist ohne eindeutigen Nachweis nicht bestätigt.')) + outcome='complete';message='Antwort fertig · Modell bleibt für weitere Anfragen geladen' + except Exception as exc: + cancelled=self.cancel.is_set() + message='Chat-Test abgebrochen.' if cancelled else (str(exc) if isinstance(exc,InferenceError) else 'Verbindung zum Modell unterbrochen. '+(self.worker.status().get('error') or 'Speicherfehler, Modellabsturz oder Zeitlimit möglich; Ursache nicht eindeutig.')) + outcome='cancelled' if cancelled else 'failed';error=None if cancelled else message + finally: + if conn:conn.close() + with self.lock:self.socket=None;self.job.update(state=outcome,error=error,phase=message,finished_at=time.time()) diff --git a/deploy/Dockerfile b/deploy/Dockerfile index 6ec2f73..3ac6541 100644 --- a/deploy/Dockerfile +++ b/deploy/Dockerfile @@ -12,8 +12,8 @@ RUN git clone https://github.com/Comfy-Org/ComfyUI.git /opt/deck-comfy \ && /opt/deck-image-python/bin/pip freeze > /opt/deck-comfy/deck-requirements.lock WORKDIR /app COPY deploy/image-requirements.lock /app/deploy/image-requirements.lock -COPY endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/ -COPY endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/ +COPY chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/ +COPY chat-test-ui.js endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/ COPY network/__init__.py network/client.py network/config.py network/rpc.py /app/network/ ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 HOME=/tmp \ DECK_BIND_HOST=0.0.0.0 DECK_STATE_DIR=/var/lib/deck \ diff --git a/deploy/install.py b/deploy/install.py index 414110c..d8adc4c 100644 --- a/deploy/install.py +++ b/deploy/install.py @@ -14,7 +14,7 @@ import urllib.request ROOT = Path(__file__).resolve().parent.parent LABEL = 'de.casaderoll.athena-deck.standalone' -FILES = ['endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock'] +FILES = ['chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock'] def run(*args, check=True, interactive=False): diff --git a/index.html b/index.html index 85eb1cd..4f53b67 100644 --- a/index.html +++ b/index.html @@ -1 +1 @@ -Athena Deck
ATHENA CONTROL SURFACEv0.7 · Router
+Athena Deck
ATHENA CONTROL SURFACEv0.7 · Router
diff --git a/inference.py b/inference.py index 7ff2819..dbb0600 100644 --- a/inference.py +++ b/inference.py @@ -72,7 +72,7 @@ class Scheduler: class LlamaWorker: def __init__(self,root,catalog,runtime): self.root=Path(root);self.catalog=catalog;self.runtime=runtime;self.lock=threading.RLock() - self.process=None;self.fit_process=None;self.generation=0;self.port=None;self.profile=None;self.state='stopped';self.error=None;self.key=None;self.devices=[] + self.process=None;self.fit_process=None;self.generation=0;self.port=None;self.profile=None;self.state='stopped';self.error=None;self.key=None;self.devices=[];self.memory_plan=None;self.phase='Gestoppt';self.oom_before=0 def build(self): state=self.runtime.status() build=next((b for b in state['builds'] if b['id']==state['active']),None) @@ -88,10 +88,17 @@ class LlamaWorker: if not p.get('model') or not p['model']['file'].endswith('.gguf'):raise InferenceError('Ein lokales GGUF-Modell wird benötigt.') return [] except ValueError as exc:return [str(exc)] + @staticmethod + def oom_count(): + try:return int(dict(line.split() for line in Path('/sys/fs/cgroup/memory.events').read_text().splitlines()).get('oom_kill',0)) + except (OSError,ValueError):return 0 + def crash_message(self): + if self.oom_count()>self.oom_before:return 'OOM-Kill im Deck-RAM-Bereich erkannt. Das RAM-Limit wurde überschritten; Kontext oder Modellgröße reduzieren.' + return 'Modellprozess beendet. GPU-OOM oder Modell-/Buildfehler möglich; Ursache nicht eindeutig bestätigt. Kontext oder Microbatch reduzieren und erneut testen.' def status(self): with self.lock: - if self.process and self.process.poll() is not None and self.state=='ready':self.state='failed';self.error='llama.cpp wurde unerwartet beendet.' - return dict(state=self.state,profile_id=self.profile['id'] if self.profile else None,profile_name=self.profile['name'] if self.profile else None,port=self.port,error=self.error,gpus=list(self.devices)) + if self.process and self.process.poll() is not None and self.state=='ready':self.state='failed';self.error=self.crash_message() + return dict(state=self.state,profile_id=self.profile['id'] if self.profile else None,profile_name=self.profile['name'] if self.profile else None,port=self.port,error=self.error,gpus=list(self.devices),memory_plan=self.memory_plan,phase=self.phase) def stop(self): with self.lock: self.generation+=1 @@ -106,18 +113,18 @@ class LlamaWorker: try:p.wait(timeout=10) except subprocess.TimeoutExpired:os.killpg(p.pid,signal.SIGKILL);p.wait(timeout=5) except ProcessLookupError:pass - self.process=None;self.profile=None;self.port=None;self.state='stopped';self.devices=[];self.key=None + self.process=None;self.profile=None;self.port=None;self.state='stopped';self.phase='Gestoppt';self.devices=[];self.key=None (self.root/'worker.key').unlink(missing_ok=True) - def ensure(self,profile): + def ensure(self,profile,cancel=lambda:False): with self.lock: if self.process and self.process.poll() is None and self.profile and (self.profile['id'],self.profile['revision'])==(profile['id'],profile['revision']):return - self.stop();self.state='loading';self.error=None;generation=self.generation - try:self._start(profile,generation) + self.stop();self.state='loading';self.error=None;self.memory_plan=None;self.phase='Speicherprüfung';self.oom_before=self.oom_count();generation=self.generation + try:self._start(profile,generation,cancel) except Exception as exc: self.stop() - with self.lock:self.state='failed';self.error=str(exc) if isinstance(exc,ValueError) else 'llama.cpp konnte nicht gestartet werden.' + with self.lock:self.state='failed';self.phase='Modellstart fehlgeschlagen';self.error=str(exc) if isinstance(exc,ValueError) else 'llama.cpp konnte nicht gestartet werden.' raise InferenceError(self.error) from None - def _start(self,profile,generation): + def _start(self,profile,generation,cancel=lambda:False): directory=self.build();params=profile['parameters'];entry=self.catalog.entry(profile['model_id']) model=(self.catalog.root/entry['id']/('model'+Path(entry['file']).suffix)).resolve() devices=probe();ids=params['gpu_devices'] @@ -141,7 +148,8 @@ class LlamaWorker: tool=str(directory/'build/bin/llama-fit-params') margins=[max(1024,round(g['total_mib']*.05)) for g in selected] def estimate(layers,extra=()): - output=self._fit_command([tool]+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation) + if cancel():raise InferenceError('Modellstart abgebrochen.') + output=self._fit_command([tool]+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation,cancel) rows={} for line in output.splitlines(): parts=line.split() @@ -150,6 +158,7 @@ class LlamaWorker: gpu_fits=all(rows['CUDA'+str(i)]+margins[i]<=g['free_mib'] for i,g in enumerate(selected)) need=max(staging,rows['Host']*1024**2+4*GIB) host_fits=mem.get('MemAvailable',0)>=need+4*GIB and (headroom is None or headroom>=need) + with self.lock:self.memory_plan=dict(profile_name=profile['name'],estimated=True,gpu_layers=layers,context=params['context'],slots=params['slots'],fits=gpu_fits and host_fits,host_required_mib=round(need/1024**2),host_available_mib=round(min(mem.get('MemAvailable',0)-4*GIB,headroom if headroom is not None else mem.get('MemAvailable',0))/1024**2),gpus=[dict(name=g.get('name',g['uuid']),uuid=g['uuid'],free_mib=g['free_mib'],required_mib=rows['CUDA'+str(i)],reserve_mib=margins[i]) for i,g in enumerate(selected)]) return gpu_fits and host_fits,rows extra=[] if params['tensor_split']: @@ -165,8 +174,9 @@ class LlamaWorker: else:high=middle-1 if best is None:raise InferenceError('Kontext und fester GPU-Split passen nicht sicher in GPU/RAM.') layers,memory=best + estimate(layers) else: - output=self._fit_command([tool]+common,env,generation) + output=self._fit_command([tool]+common,env,generation,cancel) flags=shlex.split(output.strip());fit={} if len(flags)%2:raise InferenceError('Fit-Werkzeug lieferte ungültige Parameter.') for i in range(0,len(flags),2): @@ -180,6 +190,8 @@ class LlamaWorker: extra=['--tensor-split',fit['-ts']] fits,memory=estimate(layers,extra) if not fits:raise InferenceError('Speicherprognose überschreitet die GPU-/RAM-Reserve.') + if cancel():raise InferenceError('Modellstart abgebrochen.') + with self.lock:self.phase='Modell wird geladen' launch=common+extra+['--gpu-layers',str(layers),'--fit','off','--kv-unified','--threads',str(params['threads']),'--load-mode','none','--host','127.0.0.1','--alias',profile['name'],'--no-webui','--log-disable'] self.root.mkdir(parents=True,exist_ok=True,mode=0o700) with socket.socket() as sock:sock.bind(('127.0.0.1',0));port=sock.getsockname()[1] @@ -193,27 +205,35 @@ class LlamaWorker: self.profile=profile;self.port=port;self.key=key;self.devices=[g['uuid'] for g in selected] deadline=time.monotonic()+300 while time.monotonic()1 for g in selected):raise InferenceError('Ein weiterer Prozess verwendet die reservierte GPU. Deck bricht seinen Start ab.') conn=http.client.HTTPConnection('127.0.0.1',port,timeout=2) try: conn.request('GET','/health',headers={'Authorization':'Bearer '+key});r=conn.getresponse() if r.status==200: - with self.lock:self.state='ready' + with self.lock:self.state='ready';self.phase='Modell bereit' threading.Thread(target=self._monitor,args=(generation,),daemon=True).start() return except OSError:pass finally:conn.close() time.sleep(1) raise InferenceError('llama.cpp wurde nicht rechtzeitig bereit.') - def _fit_command(self,args,env,generation): + def _fit_command(self,args,env,generation,cancel=lambda:False): with self.lock: if self.generation!=generation:raise InferenceError('Modellstart abgebrochen.') process=subprocess.Popen(args,env=env,stdout=subprocess.PIPE,stderr=subprocess.DEVNULL,text=True) self.fit_process=process - try:output,_=process.communicate(timeout=180) + try: + deadline=time.monotonic()+180 + while True: + if cancel(): + process.terminate();process.communicate(timeout=3);raise InferenceError('Modellstart abgebrochen.') + try:output,_=process.communicate(timeout=1);break + except subprocess.TimeoutExpired: + if time.monotonic()>=deadline:raise except subprocess.TimeoutExpired: process.kill();process.communicate();raise InferenceError('Zeitlimit der Speicher-Einpassung überschritten.') from None finally: diff --git a/network/install_remote.py b/network/install_remote.py index fa67377..12ed217 100644 --- a/network/install_remote.py +++ b/network/install_remote.py @@ -12,7 +12,7 @@ import sys NAME = 'athena-deck-network' BASE = Path('/opt/athena-deck') -FILES = {'endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'} +FILES = {'chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'} def run(*args, **kwargs): return subprocess.run(args, capture_output=True, timeout=600, **kwargs) diff --git a/server.py b/server.py index 2bb3f2f..40566f5 100644 --- a/server.py +++ b/server.py @@ -15,6 +15,7 @@ from image_test import ImageTests from docker_support import DockerSupport from inference import LlamaWorker,Scheduler from endpoint import Endpoint +from chat_test import ChatTests from urllib.parse import urlsplit, parse_qs from network.client import NetworkClient from network.config import parse_config, ConfigError @@ -88,6 +89,7 @@ class Server(ThreadingHTTPServer): if self.endpoint.config['autostart']: try:self.endpoint.start() except ValueError as exc:self.endpoint.error=str(exc) + self.chat_tests=ChatTests(self.profiles,self.worker,self.scheduler) self.sessions = {} self.login_attempts = [] self.auth_lock = threading.Lock() @@ -239,7 +241,7 @@ class Handler(BaseHTTPRequestHandler): return self.respond({'error':'Anmeldung erforderlich.'},401) if not self.authenticated() and self.path == '/': return self.respond((ROOT/'login.html').read_bytes(), mime='text/html; charset=utf-8') - routes = {'/endpoint-ui.js':('endpoint-ui.js','text/javascript'),'/docker-ui.js': ('docker-ui.js','text/javascript'), '/': ('index.html', 'text/html; charset=utf-8'), '/app.js': ('app.js', 'text/javascript'), '/style.css': ('style.css', 'text/css'), '/network-ui.js': ('network-ui.js', 'text/javascript'), '/access-ui.js': ('access-ui.js', 'text/javascript'), '/studio.js': ('studio.js', 'text/javascript'), '/catalog-ui.js': ('catalog-ui.js','text/javascript'), '/runtime-ui.js': ('runtime-ui.js','text/javascript'), '/profiles-ui.js': ('profiles-ui.js','text/javascript'), '/image-test-ui.js': ('image-test-ui.js','text/javascript')} + routes = {'/chat-test-ui.js':('chat-test-ui.js','text/javascript'),'/endpoint-ui.js':('endpoint-ui.js','text/javascript'),'/docker-ui.js': ('docker-ui.js','text/javascript'), '/': ('index.html', 'text/html; charset=utf-8'), '/app.js': ('app.js', 'text/javascript'), '/style.css': ('style.css', 'text/css'), '/network-ui.js': ('network-ui.js', 'text/javascript'), '/access-ui.js': ('access-ui.js', 'text/javascript'), '/studio.js': ('studio.js', 'text/javascript'), '/catalog-ui.js': ('catalog-ui.js','text/javascript'), '/runtime-ui.js': ('runtime-ui.js','text/javascript'), '/profiles-ui.js': ('profiles-ui.js','text/javascript'), '/image-test-ui.js': ('image-test-ui.js','text/javascript')} if self.path in routes: name, mime = routes[self.path] return self.respond((ROOT/name).read_bytes(), mime=mime) @@ -247,6 +249,7 @@ class Handler(BaseHTTPRequestHandler): endpoint=self.server.endpoint.status() public_endpoint={key:endpoint[key] for key in ('state','port','counts','active_requests')} return self.respond(dict(name='Athena Deck', version='0.7.0', state='ready', uptime_seconds=round(time.time()-self.server.started), mode='isolated', location=os.environ.get('DECK_LOCATION', 'Athena · Debian-Server'), endpoint=public_endpoint)) + if self.path == '/api/v1/chat-tests':return self.respond(self.server.chat_tests.status()) if self.path == '/api/v1/endpoint':return self.respond(self.server.endpoint.status()) if self.path == '/api/v1/docker':return self.respond(self.server.docker.status()) if self.path == '/api/v1/image-runtime':return self.respond(self.server.image_runtime.status()) @@ -334,6 +337,13 @@ class Handler(BaseHTTPRequestHandler): raise ValueError('Ungültige Laufzeitaktion.') except ValueError as exc:return self.respond({'error':str(exc)},400) except (OSError, subprocess.SubprocessError):return self.respond({'error':'Laufzeitaktion fehlgeschlagen; Speicher und Werkzeuge prüfen.'},503) + if self.path in ('/api/v1/chat-tests/start','/api/v1/chat-tests/cancel','/api/v1/chat-tests/unload'): + try: + data=self.read_json() + if self.path.endswith('/start'):return self.respond(self.server.chat_tests.start(data)) + if data:raise ValueError('Keine Parameter erwartet.') + return self.respond(self.server.chat_tests.stop() if self.path.endswith('/cancel') else self.server.chat_tests.unload()) + except ValueError as exc:return self.respond({'error':str(exc)},400) if self.path in ('/api/v1/endpoint/start','/api/v1/endpoint/stop','/api/v1/endpoint/config','/api/v1/endpoint/profile'): try: data=self.read_json();ep=self.server.endpoint @@ -423,6 +433,7 @@ def main(): try: server.serve_forever() finally: + server.chat_tests.stop() server.endpoint.close() server.image_runtime.stop() server.image_tests.stop() diff --git a/studio.js b/studio.js index 4fac666..353f62b 100644 --- a/studio.js +++ b/studio.js @@ -6,13 +6,14 @@ const Studio=(()=>{ if(!force&&document.querySelector('#studio')?.dataset.page===page)return; if(category!==page){category=page;section='discover';} if(modelId)section='profiles'; - const body=['docker-runtime','services'].includes(page)?DockerUI.html(page==='services'):page==='runtimes'?runtimeOverview():page==='image-runtime'?imageRuntime():page==='runtime'?RuntimeUI.html():`
ATHENA / MODELLVERWALTUNG

${labels[page]}

Modelle entdecken, herunterladen und mit gespeicherten Profilen konfigurieren.

${[['discover','Entdecken'],['library','Bibliothek'],['downloads','Downloads'],['profiles','Profile'],...(page==='image'?[['test','Testen']]:[]),['running','Laufend']].map(([id,label])=>``).join('')}
${section==='test'?ImageTestUI.html():section==='running'&&page==='image'?ImageTestUI.html(true):section==='profiles'?ProfilesUI.html():section==='running'&&page==='chat'?EndpointUI.runningHTML():section==='running'?'

Keine von Deck gestarteten Modelle

Profile werden auf Athena gespeichert. Für diesen Bereich ist noch kein Modellworker angebunden. Fehlende Komponenten stehen beim jeweiligen Profil.

llama.cpp-Builds verwalten →
':CatalogUI.html(section==='library',section==='downloads')}
`; + const body=['docker-runtime','services'].includes(page)?DockerUI.html(page==='services'):page==='runtimes'?runtimeOverview():page==='image-runtime'?imageRuntime():page==='runtime'?RuntimeUI.html():`
ATHENA / MODELLVERWALTUNG

${labels[page]}

Modelle entdecken, herunterladen und mit gespeicherten Profilen konfigurieren.

${[['discover','Entdecken'],['library','Bibliothek'],['downloads','Downloads'],['profiles','Profile'],...(['image','chat'].includes(page)?[['test','Testen']]:[]),['running','Laufend']].map(([id,label])=>``).join('')}
${section==='test'?(page==='chat'?ChatTestUI.html():ImageTestUI.html()):section==='running'&&page==='image'?ImageTestUI.html(true):section==='profiles'?ProfilesUI.html():section==='running'&&page==='chat'?EndpointUI.runningHTML():section==='running'?'

Keine von Deck gestarteten Modelle

Profile werden auf Athena gespeichert. Für diesen Bereich ist noch kein Modellworker angebunden. Fehlende Komponenten stehen beim jeweiligen Profil.

llama.cpp-Builds verwalten →
':CatalogUI.html(section==='library',section==='downloads')}
`; document.querySelector('#view').innerHTML=`
${body}
`; if(['docker-runtime','services'].includes(page)){DockerUI.bind(page==='services');return;} if(page==='runtime'){RuntimeUI.bind();return;} if(page==='image-runtime'){ImageTestUI.bind(true);ImageTestUI.bindInstaller();return;} if(page==='runtimes')return; - if(section==='test'||(section==='running'&&page==='image'))ImageTestUI.bind(section==='running'); + if(section==='test'&&page==='chat')ChatTestUI.bind(); + else if(section==='test'||(section==='running'&&page==='image'))ImageTestUI.bind(section==='running'); else if(section==='running'&&page==='chat')EndpointUI.bindRunning(); else if(section==='profiles')ProfilesUI.bind(page,modelId); else if(section!=='running')CatalogUI.bind(page,section==='library',section==='downloads'); diff --git a/style.css b/style.css index f86e607..8e15d30 100644 --- a/style.css +++ b/style.css @@ -18,3 +18,5 @@ nav a.nav-sub{padding-left:28px;font-size:12px;color:#a4bda8}.component-card{mar #image-test textarea{width:100%;background:#101718;color:#e4eee5;border:1px solid #465348;border-radius:8px;padding:14px;font:inherit;resize:vertical;margin:10px 0 20px}.generated-image{max-width:100%;height:auto;display:block;border-radius:8px}#image-test-result figure{margin:20px 0}#image-test-state{margin-top:20px} .hub-base-filter{display:flex;align-items:center;gap:.6rem;margin-top:1rem}.hub-base-filter input{width:auto;margin:0} + +.chat-text,#chat-transcript pre{white-space:pre-wrap;overflow-wrap:anywhere;} diff --git a/test_chat_test.py b/test_chat_test.py new file mode 100644 index 0000000..5b02e26 --- /dev/null +++ b/test_chat_test.py @@ -0,0 +1,50 @@ +import json +import threading +import time +import unittest +from types import SimpleNamespace +from unittest.mock import Mock +from chat_test import ChatTests +from inference import Scheduler,InferenceError,LlamaWorker + +class Reply: + status=200 + def __init__(self,events):self.events=iter(events) + def read1(self,_):return next(self.events,b'') + +class Worker: + def __init__(self):self.error=None;self.loaded=False;self.conn=Mock();self.conn.getresponse.return_value=Reply([b'data: {"choices":[{"delta":{"content":"OK"}}]}\n\n',b'data: [DONE]\n\n']) + def stop(self):self.loaded=False + def ensure(self,p,cancel=lambda:False): + if self.error:raise InferenceError(self.error) + self.loaded=True + def status(self):return dict(state='ready' if self.loaded else 'stopped',error=self.error) + def connect(self):return self.conn,'synthetic-only' + +class ChatTestsTests(unittest.TestCase): + def setUp(self): + self.profile=dict(id='p',name='test',revision=1,kind='chat',runnable=True,parameters=dict(slots=2,temperature=.2,top_p=.8,top_k=20)) + self.worker=Worker();self.manager=ChatTests(SimpleNamespace(status=lambda:dict(profiles=[self.profile])),self.worker,Scheduler(self.worker)) + def request(self):return dict(profile_id='p',messages=[dict(role='user',content='synthetic test')],max_tokens=8) + def finish(self): + deadline=time.monotonic()+3 + while self.manager.status()['job']['state']=='running' and time.monotonic()