From a77266c7bd6d70b6ee7a372d4ebe19e00a126aa5 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Mon, 28 Sep 2026 21:29:48 +0200 Subject: [PATCH] Add bounded GPU-priority automatic model testing and profile export --- STUDIO.md | 10 ++++ auto-test-ui.js | 11 +++++ auto_test.py | 100 ++++++++++++++++++++++++++++++++++++++ deploy/Dockerfile | 4 +- deploy/install.py | 2 +- index.html | 2 +- inference.py | 6 ++- network/install_remote.py | 2 +- server.py | 14 +++++- studio.js | 5 +- test_auto_test.py | 39 +++++++++++++++ test_server.py | 2 +- 12 files changed, 187 insertions(+), 10 deletions(-) create mode 100644 auto-test-ui.js create mode 100644 auto_test.py create mode 100644 test_auto_test.py diff --git a/STUDIO.md b/STUDIO.md index fabea86..1cd5bd8 100644 --- a/STUDIO.md +++ b/STUDIO.md @@ -163,3 +163,13 @@ Sprachmodelle besitzen jetzt den Reiter **Testen**: interner Textchat mit Stream Im Profileditor kann MTP aktiviert werden, mit 1–8 Draft-Token und Mindestwahrscheinlichkeit 0–1. Medium-Referenz: 2 und 0,05; Draft-KV ist f16. Bestehende Profile bleiben standardmäßig ohne MTP. Das Modell muss eingebettete MTP-Gewichte enthalten. Der CUDA-Build benötigt den Deck-MTP-Fit-Adapter; neue GUI-Builds installieren ihn automatisch. Die Prognose zählt die gemeinsamen Gewichte einmal sowie Haupt- und MTP-Kontext und deren Compute-Puffer. Ein inkompatibles Modell oder ein alter Build wird nicht still ohne MTP gestartet. MTP garantiert keinen Geschwindigkeitsgewinn. GPU-Ausführung: „Automatisch“ darf bei Platzmangel Schichten auf die CPU verlagern. „Vollständig auf GPU“ verlangt alle Schichten einschließlich Ausgabe und bricht sonst vor dem Laden ab. Die Prognose lässt je GPU mindestens 512 MiB bzw. 2,5 % Gesamtspeicher frei (der größere Wert gilt); keine Garantie für beliebige Last. Die Chat-Diagnose kennzeichnet eine begrenzte Layerzahl. + +## Auto-Test für heruntergeladene Sprachmodelle + +Unter Sprachmodelle → Auto-Test ein GGUF und eine obere Kontextgrenze auswählen. Feste Gerätepriorität RTX 5080, dann RTX 3060; andere Hardware wird ausdrücklich abgelehnt. Kontextstufen: 2048, 4096, 8192, 16384, 32768, 65536, 131072, 160000, 192000, 262144 bis zur gewählten Grenze. Pro Kontext zuerst nur 5080, dann Layer-Splits 95:5, 90:10, 85:15, 75:25, 50:50. Nur wenn kein GPU-Kandidat besteht, wird automatische CPU-Auslagerung mit 85:15 getestet. Die erste erfolgreiche Gerätestufe hat Priorität; keine Behauptung eines globalen Geschwindigkeitsoptimums. + +Microbatch 128, 64 und 256, Batch 2048, ein Slot; optional MTP2/p-min0,05. Erst Speicherprognose, dann echter Start, Aufwärmrunde und synthetischer, per Modelltokenizer auf 75 Prozent des Kontexts gefüllter Prompt mit bis zu 64 Ausgabetoken. Speicherung nur von Parametern, Prognose, Messwerten und Fehlerstatus; keine Antworten. Keine absichtliche Überschreitung der Speicherreserve. Höchstens 48 Kandidaten bzw. zwei Stunden zwischen Kandidaten; laufende Einzeloperationen haben eigene Zeitlimits. Ein kurzer erfolgreicher Lauf ist kein Vollkontext-/Parallelitäts-Stabilitätsnachweis. + +Der gemeinsame Scheduler sperrt die Deck-Modelle während der Testreihe exklusiv. Bereits laufende Anfragen verhindern den Start; fremde GPU-Prozesse werden nicht beendet. Abbrechen schließt die eigene Anfrage, beendet nur die eigenen Testprozesse und gibt die Reservierung frei. Beim App-Neustart wird eine offene Reihe als unterbrochen markiert. Ergebnisse bleiben in auto-tests.json; ein neuer Test ersetzt diese Ergebnisliste. Erfolgreiche Kandidaten lassen sich explizit als neues Profil speichern, ohne Veröffentlichung am Endpunkt. + +Admin-API: GET /api/v1/auto-tests; POST /start mit model_id, max_context, mtp; POST /cancel mit {}; POST /save mit job_id, index, name (jeweils unter /api/v1/auto-tests). Session und CSRF-Schutz wie Profilverwaltung; API-Bearer gewährt keinen Zugriff. diff --git a/auto-test-ui.js b/auto-test-ui.js new file mode 100644 index 0000000..cce4095 --- /dev/null +++ b/auto-test-ui.js @@ -0,0 +1,11 @@ +window.AutoTestUI=(()=>{ + const e=x=>String(x??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c])); + async function api(path,data){const r=await fetch('/api/v1/'+path,data===undefined?{}:{method:'POST',headers:{'Content-Type':'application/json','X-Athena-Deck':'1'},body:JSON.stringify(data)});const v=await r.json();if(!r.ok)throw Error(v.error||'Anfrage fehlgeschlagen');return v;} + function html(){return `

Automatische Einpassung & Leistungstest

Priorität: RTX 5080 allein → möglichst kleiner Anteil auf RTX 3060 → erst zuletzt CPU-Auslagerung. Synthetische Tests belegen die Deck-Laufzeit exklusiv; API-Anfragen warten. Fremde GPU-Dienste werden nicht beendet.

Ein Slot, Batch 2048, Microbatch 128 / 64 / 256. Kontextstufen ab 2048; pro Kandidat werden 75 % des Kontextes tatsächlich gefüllt und 64 Token erzeugt. Höchstens 48 Kandidaten bzw. zwei Stunden zwischen Kandidaten. Kein absichtliches Übergehen der Speicherreserve.

Dies ist eine begrenzte Suche, kein Beweis des absoluten Optimums oder der Stabilität bei vollem Kontext und paralleler Last. Erfolgreiche Profile werden nur auf Wunsch gespeichert, nicht am Endpunkt freigegeben.

`;} + function bind(){const root=document.querySelector('#auto-test'),el=x=>root.querySelector('#'+x);let job=null,last='';const msg=x=>{if(root.isConnected)el('auto-message').textContent=x;}; + api('catalog').then(s=>{if(root.isConnected)el('auto-model').innerHTML=s.entries.filter(x=>x.kind==='chat'&&x.profile_eligible&&x.file.endsWith('.gguf')).map(x=>``).join('');}).catch(x=>msg(x.message)); + async function refresh(){try{const s=await api('auto-tests');if(!root.isConnected)return;job=s.job;const running=job?.state==='running';el('auto-start').disabled=running;el('auto-stop').disabled=!running;el('auto-phase').textContent=job?`${job.state} · ${job.phase} · ${job.attempts} Kandidaten${job.error?' · '+job.error:''}`:'Noch kein Auto-Test.';const rows=job?.results||[],signature=JSON.stringify(rows);if(last!==signature){last=signature;const best=rows.reduce((a,r,i)=>r.success&&(!a||r.timings.predicted_per_second>a.speed)?{i,speed:r.timings.predicted_per_second}:a,null);el('auto-results').innerHTML=rows.map((r,i)=>`

${r.context.toLocaleString('de-DE')} Kontext · ${e(r.tier)} ${best?.i===i?'· schnellster gemessener Kandidat':''}

Verteilung ${e(r.parameters.tensor_split.join(':')||'nur 5080')} · Microbatch ${r.parameters.ubatch}

${r.success?`

Prefill ${Number(r.timings.prompt_per_second).toFixed(1)} Token/s · Ausgabe ${Number(r.timings.predicted_per_second).toFixed(1)} Token/s · ${r.timings.prompt_n} Eingabe-Token geprüft

GPU-Layerlimit: ${r.memory_plan.gpu_layers===999?'alle':r.memory_plan.gpu_layers}

`:`

${e(r.error)}

`}
`).join('');root.querySelectorAll('[data-save]').forEach(b=>b.onclick=async()=>{try{await api('auto-tests/save',{job_id:job.id,index:Number(b.dataset.save),name:el('auto-name').value});msg('Profil gespeichert.');}catch(x){msg(x.message);}});}}catch(x){msg(x.message);}finally{if(root.isConnected)setTimeout(refresh,2000);}} + el('auto-form').onsubmit=async ev=>{ev.preventDefault();try{await api('auto-tests/start',{model_id:el('auto-model').value,max_context:Number(el('auto-context').value),mtp:el('auto-mtp').checked});msg('Auto-Test gestartet.');}catch(x){msg(x.message);}}; + el('auto-stop').onclick=()=>api('auto-tests/cancel',{}).then(()=>msg('Abbruch angefordert.')).catch(x=>msg(x.message));refresh(); + }return {html,bind}; +})(); diff --git a/auto_test.py b/auto_test.py new file mode 100644 index 0000000..932e8fb --- /dev/null +++ b/auto_test.py @@ -0,0 +1,100 @@ +"""Bounded, synthetic model tuning; exclusive shared scheduler, no foreign process control.""" +import copy,json,threading,time,uuid,socket +from pathlib import Path +from profiles import SCHEMAS,CHAT_GPU_DEFAULTS +from image_test import probe +from inference import InferenceError + +class AutoTests: + def __init__(self,path,profiles,worker,scheduler): + self.path=Path(path);self.profiles=profiles;self.worker=worker;self.scheduler=scheduler;self.lock=threading.RLock();self.cancel=threading.Event();self.conn=None;self.thread=None + self.job=json.loads(self.path.read_text()) if self.path.exists() else None + if self.job and self.job['state']=='running':self.job.update(state='interrupted',phase='Durch Neustart unterbrochen');self.persist() + def persist(self): + self.path.parent.mkdir(parents=True,exist_ok=True);p=self.path.with_suffix('.tmp');p.write_text(json.dumps(self.job));p.replace(self.path) + def status(self): + with self.lock:return {'job':copy.deepcopy(self.job)} + def update(self,**kw): + with self.lock:self.job.update(kw);self.persist() + def start(self,data): + if set(data)!={'model_id','max_context','mtp'} or type(data['max_context'])!=int or data['max_context'] not in (4096,8192,16384,32768,65536,131072,160000,192000,262144) or type(data['mtp'])!=bool:raise ValueError('Modell, Kontextgrenze und MTP auswählen.') + model=self.worker.catalog.entry(data['model_id']) + if model['kind']!='chat' or not model['file'].endswith('.gguf') or not model['profile_eligible']:raise ValueError('Ein heruntergeladenes Chat-GGUF auswählen.') + devices=probe();primary=next((g for g in devices if '5080' in g['name']),None);secondary=next((g for g in devices if '3060' in g['name']),None) + if not primary or not secondary:raise ValueError('Dieser Auto-Test benötigt RTX 5080 und RTX 3060. Keine stillschweigende andere GPU-Zuordnung.') + with self.lock: + if self.thread and self.thread.is_alive():raise ValueError('Ein Auto-Test läuft bereits.') + self.cancel.clear();self.job=dict(id=uuid.uuid4().hex,state='running',phase='Wartet auf exklusive Modellreservierung',model_id=model['id'],model_file=model['file'],max_context=data['max_context'],mtp=data['mtp'],results=[],attempts=0,started_at=time.time(),error=None);self.persist() + self.thread=threading.Thread(target=self.run,args=(primary['uuid'],secondary['uuid']),daemon=True);self.thread.start() + return self.status() + def stop(self): + self.cancel.set() + with self.lock:conn=self.conn + if conn and conn.sock: + try:conn.sock.shutdown(socket.SHUT_RDWR) + except OSError:pass + return {'cancellation_requested':True} + def request(self,path,data): + if self.cancel.is_set():raise InterruptedError() + conn,key=self.worker.connect();conn.timeout=180 + with self.lock:self.conn=conn + try: + conn.request('POST',path,json.dumps(data).encode(),{'Content-Type':'application/json','Authorization':'Bearer '+key}) + response=conn.getresponse();raw=response.read(8*1024*1024) + if response.status!=200:raise InferenceError('Synthetischer Test abgelehnt: Kontext oder Modell prüfen.') + return json.loads(raw) + finally: + conn.close() + with self.lock:self.conn=None + def benchmark(self,context): + # Tokenize synthetic text with this model; do not estimate token count from characters. + target=int(context*.75);unit='alpha beta gamma delta epsilon zeta eta theta 0123456789. ' + text=unit*(target//4+1);tokens=self.request('/tokenize',{'content':text,'add_special':True})['tokens'] + if len(tokens)=48 or time.time()-self.job['started_at']>7200: + self.update(state='complete',phase='Testbudget erreicht · Teilresultate verfügbar',finished_at=time.time());return + params={**{k:v[2] for k,v in SCHEMAS['chat'].items()},**CHAT_GPU_DEFAULTS,'context':context,'slots':1,'batch':2048,'ubatch':micro,'gpu_devices':devices,'split_mode':split,'tensor_split':ratio,'gpu_offload':offload,'mtp':self.job['mtp']} + p=dict(id='auto-'+self.job['id'],revision=self.job['attempts']+1,name='deck-auto-test',kind='chat',model_id=self.job['model_id'],parameters=params) + self.update(attempts=p['revision'],phase=f'{context} Kontext · {tier} · Microbatch {micro}') + row=dict(context=context,tier=tier,parameters=params,success=False) + self.worker.stop() + try: + self.worker.plan(p,cancel=self.cancel.is_set) + self.worker.ensure(p,cancel=self.cancel.is_set) + # Small warm-up, then populated-context measurement, bounded output. + self.request('/completion',dict(prompt='Synthetic warmup.',n_predict=16,temperature=0,cache_prompt=False)) + row.update(success=True,timings=self.benchmark(context),memory_plan=copy.deepcopy(self.worker.memory_plan),tested_context_fraction=.75) + successes.append(row) + except Exception as exc: + if self.cancel.is_set():raise InterruptedError() + row['error']=str(exc) if isinstance(exc,ValueError) else 'Testprozess oder Verbindung fehlgeschlagen; kein eindeutiger OOM-Nachweis.' + finally:self.worker.stop() + with self.lock:self.job['results'].append(row);self.persist() + # Smaller microbatch is useful after rejection; do not skip it. + if successes:found=True;break + if not found:break + self.update(state='complete',phase='Testreihe abgeschlossen · Ergebnisse sind keine Garantie für beliebige Last',finished_at=time.time()) + finally:self.worker.stop() + except Exception as exc:self.update(state='cancelled' if self.cancel.is_set() else 'failed',phase='Abgebrochen' if self.cancel.is_set() else 'Test beendet',error=None if self.cancel.is_set() else str(exc),finished_at=time.time()) + def save(self,data): + with self.lock: + if set(data)!={'job_id','index','name'} or not self.job or data['job_id']!=self.job['id'] or type(data['index'])!=int or not 0<=data['index'] /opt/deck-comfy/deck-requirements.lock WORKDIR /app COPY deploy/image-requirements.lock /app/deploy/image-requirements.lock -COPY chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/ -COPY chat-test-ui.js endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/ +COPY auto_test.py chat_test.py endpoint.py inference.py docker_support.py image_encoder_node.py image_runtime.py image_test.py profiles.py capacity.py server.py runtime.py catalog.py auth.py collect_hardware.py /app/ +COPY auto-test-ui.js chat-test-ui.js endpoint-ui.js docker-ui.js image-test-ui.js profiles-ui.js index.html app.js studio.js runtime-ui.js catalog-ui.js style.css login.html login.js access-ui.js network-ui.js /app/ COPY network/__init__.py network/client.py network/config.py network/rpc.py /app/network/ ENV PYTHONDONTWRITEBYTECODE=1 PYTHONUNBUFFERED=1 HOME=/tmp \ DECK_BIND_HOST=0.0.0.0 DECK_STATE_DIR=/var/lib/deck \ diff --git a/deploy/install.py b/deploy/install.py index d8adc4c..58be804 100644 --- a/deploy/install.py +++ b/deploy/install.py @@ -14,7 +14,7 @@ import urllib.request ROOT = Path(__file__).resolve().parent.parent LABEL = 'de.casaderoll.athena-deck.standalone' -FILES = ['chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock'] +FILES = ['auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','docker_support.py','docker-ui.js','deploy/docker_helper.py','deploy/setup_docker_helper.py','image_encoder_node.py','image_runtime.py','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py','runtime-ui.js','catalog.py','catalog-ui.js','server.py','auth.py','collect_hardware.py','index.html','app.js','studio.js','style.css','login.html','login.js','access-ui.js','network-ui.js','network/__init__.py','network/client.py','network/config.py','network/rpc.py','deploy/Dockerfile','deploy/image-requirements.lock'] def run(*args, check=True, interactive=False): diff --git a/index.html b/index.html index 4f53b67..619ee64 100644 --- a/index.html +++ b/index.html @@ -1 +1 @@ -Athena Deck
ATHENA CONTROL SURFACEv0.7 · Router
+Athena Deck
ATHENA CONTROL SURFACEv0.7 · Router
diff --git a/inference.py b/inference.py index 0534530..9da25b3 100644 --- a/inference.py +++ b/inference.py @@ -124,7 +124,10 @@ class LlamaWorker: self.stop() with self.lock:self.state='failed';self.phase='Modellstart fehlgeschlagen';self.error=str(exc) if isinstance(exc,ValueError) else 'llama.cpp konnte nicht gestartet werden.' raise InferenceError(self.error) from None - def _start(self,profile,generation,cancel=lambda:False): + def plan(self,profile,cancel=lambda:False): + self._start(profile,self.generation,cancel,plan_only=True) + return self.memory_plan + def _start(self,profile,generation,cancel=lambda:False,plan_only=False): directory=self.build();params=profile['parameters'];entry=self.catalog.entry(profile['model_id']) model=(self.catalog.root/entry['id']/('model'+Path(entry['file']).suffix)).resolve() mtp=params.get('mtp',False) @@ -197,6 +200,7 @@ class LlamaWorker: fits,memory=estimate(layers,extra) if not fits:raise InferenceError('Speicherprognose überschreitet die GPU-/RAM-Reserve.') if cancel():raise InferenceError('Modellstart abgebrochen.') + if plan_only:return with self.lock:self.phase='Modell wird geladen' launch=common+extra+['--gpu-layers',str(layers),'--fit','off','--kv-unified','--threads',str(params['threads']),'--load-mode','none','--host','127.0.0.1','--alias',profile['name'],'--no-webui','--log-disable'] if mtp:launch+=['--spec-type','draft-mtp','--spec-draft-n-max',str(params.get('mtp_tokens',2)),'--spec-draft-p-min',str(params.get('mtp_min_p',.05)),'--spec-draft-type-k','f16','--spec-draft-type-v','f16'] diff --git a/network/install_remote.py b/network/install_remote.py index 12ed217..2b45bbb 100644 --- a/network/install_remote.py +++ b/network/install_remote.py @@ -12,7 +12,7 @@ import sys NAME = 'athena-deck-network' BASE = Path('/opt/athena-deck') -FILES = {'chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'} +FILES = {'auto_test.py','auto-test-ui.js','chat_test.py','chat-test-ui.js','endpoint.py','inference.py','endpoint-ui.js','image_runtime.py','image_encoder_node.py','docker_support.py','docker-ui.js','deploy/image-requirements.lock','image_test.py','image-test-ui.js','profiles.py','profiles-ui.js','capacity.py','runtime.py', 'runtime-ui.js', 'catalog.py', 'catalog-ui.js', 'studio.js', 'auth.py', 'access-ui.js', 'server.py', 'collect_hardware.py', 'index.html', 'app.js', 'style.css', 'network-ui.js', 'login.html', 'login.js', 'network/__init__.py', 'network/config.py', 'network/policy.py', 'network/rpc.py', 'network/agent.py', 'network/client.py', 'network/Dockerfile', '.dockerignore'} def run(*args, **kwargs): return subprocess.run(args, capture_output=True, timeout=600, **kwargs) diff --git a/server.py b/server.py index 40566f5..b146d67 100644 --- a/server.py +++ b/server.py @@ -16,6 +16,7 @@ from docker_support import DockerSupport from inference import LlamaWorker,Scheduler from endpoint import Endpoint from chat_test import ChatTests +from auto_test import AutoTests from urllib.parse import urlsplit, parse_qs from network.client import NetworkClient from network.config import parse_config, ConfigError @@ -90,6 +91,7 @@ class Server(ThreadingHTTPServer): try:self.endpoint.start() except ValueError as exc:self.endpoint.error=str(exc) self.chat_tests=ChatTests(self.profiles,self.worker,self.scheduler) + self.auto_tests=AutoTests(self.catalog.root.parent/'auto-tests.json',self.profiles,self.worker,self.scheduler) self.sessions = {} self.login_attempts = [] self.auth_lock = threading.Lock() @@ -241,7 +243,7 @@ class Handler(BaseHTTPRequestHandler): return self.respond({'error':'Anmeldung erforderlich.'},401) if not self.authenticated() and self.path == '/': return self.respond((ROOT/'login.html').read_bytes(), mime='text/html; charset=utf-8') - routes = {'/chat-test-ui.js':('chat-test-ui.js','text/javascript'),'/endpoint-ui.js':('endpoint-ui.js','text/javascript'),'/docker-ui.js': ('docker-ui.js','text/javascript'), '/': ('index.html', 'text/html; charset=utf-8'), '/app.js': ('app.js', 'text/javascript'), '/style.css': ('style.css', 'text/css'), '/network-ui.js': ('network-ui.js', 'text/javascript'), '/access-ui.js': ('access-ui.js', 'text/javascript'), '/studio.js': ('studio.js', 'text/javascript'), '/catalog-ui.js': ('catalog-ui.js','text/javascript'), '/runtime-ui.js': ('runtime-ui.js','text/javascript'), '/profiles-ui.js': ('profiles-ui.js','text/javascript'), '/image-test-ui.js': ('image-test-ui.js','text/javascript')} + routes = {'/auto-test-ui.js':('auto-test-ui.js','text/javascript'),'/chat-test-ui.js':('chat-test-ui.js','text/javascript'),'/endpoint-ui.js':('endpoint-ui.js','text/javascript'),'/docker-ui.js': ('docker-ui.js','text/javascript'), '/': ('index.html', 'text/html; charset=utf-8'), '/app.js': ('app.js', 'text/javascript'), '/style.css': ('style.css', 'text/css'), '/network-ui.js': ('network-ui.js', 'text/javascript'), '/access-ui.js': ('access-ui.js', 'text/javascript'), '/studio.js': ('studio.js', 'text/javascript'), '/catalog-ui.js': ('catalog-ui.js','text/javascript'), '/runtime-ui.js': ('runtime-ui.js','text/javascript'), '/profiles-ui.js': ('profiles-ui.js','text/javascript'), '/image-test-ui.js': ('image-test-ui.js','text/javascript')} if self.path in routes: name, mime = routes[self.path] return self.respond((ROOT/name).read_bytes(), mime=mime) @@ -249,6 +251,7 @@ class Handler(BaseHTTPRequestHandler): endpoint=self.server.endpoint.status() public_endpoint={key:endpoint[key] for key in ('state','port','counts','active_requests')} return self.respond(dict(name='Athena Deck', version='0.7.0', state='ready', uptime_seconds=round(time.time()-self.server.started), mode='isolated', location=os.environ.get('DECK_LOCATION', 'Athena · Debian-Server'), endpoint=public_endpoint)) + if self.path == '/api/v1/auto-tests':return self.respond(self.server.auto_tests.status()) if self.path == '/api/v1/chat-tests':return self.respond(self.server.chat_tests.status()) if self.path == '/api/v1/endpoint':return self.respond(self.server.endpoint.status()) if self.path == '/api/v1/docker':return self.respond(self.server.docker.status()) @@ -337,6 +340,14 @@ class Handler(BaseHTTPRequestHandler): raise ValueError('Ungültige Laufzeitaktion.') except ValueError as exc:return self.respond({'error':str(exc)},400) except (OSError, subprocess.SubprocessError):return self.respond({'error':'Laufzeitaktion fehlgeschlagen; Speicher und Werkzeuge prüfen.'},503) + if self.path in ('/api/v1/auto-tests/start','/api/v1/auto-tests/cancel','/api/v1/auto-tests/save'): + try: + data=self.read_json() + if self.path.endswith('/start'):return self.respond(self.server.auto_tests.start(data)) + if self.path.endswith('/save'):return self.respond(self.server.auto_tests.save(data)) + if data:raise ValueError('Keine Parameter erwartet.') + return self.respond(self.server.auto_tests.stop()) + except ValueError as exc:return self.respond({'error':str(exc)},400) if self.path in ('/api/v1/chat-tests/start','/api/v1/chat-tests/cancel','/api/v1/chat-tests/unload'): try: data=self.read_json() @@ -433,6 +444,7 @@ def main(): try: server.serve_forever() finally: + server.auto_tests.stop() server.chat_tests.stop() server.endpoint.close() server.image_runtime.stop() diff --git a/studio.js b/studio.js index 353f62b..1fb8651 100644 --- a/studio.js +++ b/studio.js @@ -6,13 +6,14 @@ const Studio=(()=>{ if(!force&&document.querySelector('#studio')?.dataset.page===page)return; if(category!==page){category=page;section='discover';} if(modelId)section='profiles'; - const body=['docker-runtime','services'].includes(page)?DockerUI.html(page==='services'):page==='runtimes'?runtimeOverview():page==='image-runtime'?imageRuntime():page==='runtime'?RuntimeUI.html():`
ATHENA / MODELLVERWALTUNG

${labels[page]}

Modelle entdecken, herunterladen und mit gespeicherten Profilen konfigurieren.

${[['discover','Entdecken'],['library','Bibliothek'],['downloads','Downloads'],['profiles','Profile'],...(['image','chat'].includes(page)?[['test','Testen']]:[]),['running','Laufend']].map(([id,label])=>``).join('')}
${section==='test'?(page==='chat'?ChatTestUI.html():ImageTestUI.html()):section==='running'&&page==='image'?ImageTestUI.html(true):section==='profiles'?ProfilesUI.html():section==='running'&&page==='chat'?EndpointUI.runningHTML():section==='running'?'

Keine von Deck gestarteten Modelle

Profile werden auf Athena gespeichert. Für diesen Bereich ist noch kein Modellworker angebunden. Fehlende Komponenten stehen beim jeweiligen Profil.

llama.cpp-Builds verwalten →
':CatalogUI.html(section==='library',section==='downloads')}
`; + const body=['docker-runtime','services'].includes(page)?DockerUI.html(page==='services'):page==='runtimes'?runtimeOverview():page==='image-runtime'?imageRuntime():page==='runtime'?RuntimeUI.html():`
ATHENA / MODELLVERWALTUNG

${labels[page]}

Modelle entdecken, herunterladen und mit gespeicherten Profilen konfigurieren.

${[['discover','Entdecken'],['library','Bibliothek'],['downloads','Downloads'],['profiles','Profile'],...(['image','chat'].includes(page)?[['test','Testen'],...(page==='chat'?[['auto','Auto-Test']]:[])]:[]),['running','Laufend']].map(([id,label])=>``).join('')}
${section==='auto'?AutoTestUI.html():section==='test'?(page==='chat'?ChatTestUI.html():ImageTestUI.html()):section==='running'&&page==='image'?ImageTestUI.html(true):section==='profiles'?ProfilesUI.html():section==='running'&&page==='chat'?EndpointUI.runningHTML():section==='running'?'

Keine von Deck gestarteten Modelle

Profile werden auf Athena gespeichert. Für diesen Bereich ist noch kein Modellworker angebunden. Fehlende Komponenten stehen beim jeweiligen Profil.

llama.cpp-Builds verwalten →
':CatalogUI.html(section==='library',section==='downloads')}
`; document.querySelector('#view').innerHTML=`
${body}
`; if(['docker-runtime','services'].includes(page)){DockerUI.bind(page==='services');return;} if(page==='runtime'){RuntimeUI.bind();return;} if(page==='image-runtime'){ImageTestUI.bind(true);ImageTestUI.bindInstaller();return;} if(page==='runtimes')return; - if(section==='test'&&page==='chat')ChatTestUI.bind(); + if(section==='auto')AutoTestUI.bind(); + else if(section==='test'&&page==='chat')ChatTestUI.bind(); else if(section==='test'||(section==='running'&&page==='image'))ImageTestUI.bind(section==='running'); else if(section==='running'&&page==='chat')EndpointUI.bindRunning(); else if(section==='profiles')ProfilesUI.bind(page,modelId); diff --git a/test_auto_test.py b/test_auto_test.py new file mode 100644 index 0000000..3c5167a --- /dev/null +++ b/test_auto_test.py @@ -0,0 +1,39 @@ +import tempfile,unittest,json +from pathlib import Path +from types import SimpleNamespace +from unittest.mock import Mock,patch +from auto_test import AutoTests +from inference import Scheduler,InferenceError + +class AutoTestsTests(unittest.TestCase): + def setUp(self): + self.tmp=tempfile.TemporaryDirectory();self.worker=Mock();self.worker.memory_plan={'gpu_layers':999};self.worker.catalog.entry.return_value=dict(id='m',kind='chat',file='model.gguf',profile_eligible=True) + self.profiles=Mock();self.scheduler=Scheduler(self.worker);self.auto=AutoTests(Path(self.tmp.name)/'auto.json',self.profiles,self.worker,self.scheduler) + self.gpus=[dict(name='RTX 5080',uuid='first'),dict(name='RTX 3060',uuid='second')] + def tearDown(self):self.tmp.cleanup() + def start(self): + with patch('auto_test.probe',return_value=self.gpus):self.auto.start(dict(model_id='m',max_context=4096,mtp=False)) + self.auto.thread.join(3);self.assertFalse(self.auto.thread.is_alive()) + def test_primary_first_results_persist_and_save_explicitly(self): + self.auto.request=Mock(return_value={});self.auto.benchmark=Mock(return_value={'predicted_per_second':40}) + self.start();job=self.auto.status()['job'];self.assertEqual(job['state'],'complete');self.assertEqual(len(job['results']),6) + self.assertTrue(all(r['parameters']['gpu_devices']==['first'] for r in job['results']));self.profiles.save.assert_not_called();self.assertEqual(self.scheduler.active,0) + self.auto.save(dict(job_id=job['id'],index=0,name='saved'));self.profiles.save.assert_called_once() + self.assertEqual(AutoTests(self.auto.path,self.profiles,self.worker,self.scheduler).job['state'],'complete') + def test_secondary_before_cpu(self): + def plan(p,cancel): + if p['parameters']['gpu_offload']=='full':raise InferenceError('does not fit') + self.worker.plan.side_effect=plan;self.auto.request=Mock(return_value={});self.auto.benchmark=Mock(return_value={'predicted_per_second':10});self.start() + rows=self.auto.job['results'];first_success=next(i for i,r in enumerate(rows) if r['success']);self.assertEqual(first_success,18);self.assertEqual(rows[first_success]['parameters']['gpu_offload'],'auto') + def test_busy_lease_does_not_stop_another_worker(self): + with self.scheduler.lease(('other',)): + self.worker.reset_mock();self.start();self.assertEqual(self.auto.job['state'],'failed');self.worker.stop.assert_not_called() + def test_cancel_releases_lease_and_does_not_save_profile(self): + self.worker.plan.side_effect=lambda *a,**k:self.auto.cancel.set();self.auto.request=Mock(side_effect=InterruptedError());self.start();self.assertEqual(self.auto.job['state'],'cancelled');self.assertEqual(self.scheduler.active,0);self.profiles.save.assert_not_called() + def test_validation_and_failed_result_save(self): + with self.assertRaises(ValueError):self.auto.start(dict(model_id='m',max_context=True,mtp=False)) + self.auto.job={'id':'a','model_id':'m','results':[{'success':False}]} + with self.assertRaises(ValueError):self.auto.save(dict(job_id='a',index=0,name='x')) + def test_prompt_filled_by_token_count(self): + self.auto.request=Mock(side_effect=[{'tokens':list(range(4000))},{'timings':{'prompt_n':3072,'predicted_n':64,'predicted_per_second':12}}]);r=self.auto.benchmark(4096);self.assertEqual(r['prompt_n'],3072);self.assertEqual(len(self.auto.request.call_args.args[1]['prompt']),3072) +if __name__=='__main__':unittest.main() diff --git a/test_server.py b/test_server.py index 03181c5..322640d 100644 --- a/test_server.py +++ b/test_server.py @@ -94,7 +94,7 @@ class Tests(unittest.TestCase): def test_chat_test_routes_are_admin_only(self): self.assertIsNone(self.request('chat-tests')['job']) - for path,method,body in [('chat-tests','GET',None),('chat-tests/start','POST',b'{}')]: + for path,method,body in [('chat-tests','GET',None),('chat-tests/start','POST',b'{}'),('auto-tests','GET',None),('auto-tests/start','POST',b'{}'),('auto-tests/save','POST',b'{}')]: req=urllib.request.Request(self.url+'/api/v1/'+path,method=method,data=body,headers={'Authorization':'Bearer '+self.api_token,'X-Athena-Deck':'1','Content-Type':'application/json'}) with self.assertRaises(urllib.error.HTTPError) as exc:urllib.request.urlopen(req) self.assertEqual(exc.exception.code,401);exc.exception.close()