Include Deck chat profiles in runtime memory fit
This commit is contained in:
+2
-2
@@ -1,7 +1,7 @@
|
||||
const RuntimeUI=(()=>{
|
||||
const e=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c]));
|
||||
async function api(path='',data){const r=await fetch('/api/v1/runtime'+path,data?{method:'POST',headers:{'Content-Type':'application/json','X-Athena-Deck':'1'},body:JSON.stringify(data)}:{});const v=await r.json();if(!r.ok)throw Error(v.error||'Anfrage fehlgeschlagen');return v;}
|
||||
function html(){return `<div id="runtime-live"><div class="kicker">ATHENA / LLAMA.CPP</div><h1>llama.cpp & Updates</h1><p>Offizielle Versionen getrennt bauen und prüfen. Ein Build lädt kein Modell und ändert keine produktiven Laufzeiten.</p><section class="card"><h2>Voraussetzungen</h2><button id="rt-check">Werkzeuge & GPUs prüfen</button><div id="rt-prereq"></div></section><section class="card"><h2>Installation & Build</h2><form id="rt-build"><label>Release oder vollständiger Commit<input name="revision" required pattern="b[0-9]{3,8}|v[0-9]{1,4}[.][0-9]{1,4}[.][0-9]{1,4}|[a-f0-9]{40}" placeholder="Unten Releases abfragen und auswählen"></label><label>Backend<select name="backend"><option>CUDA</option><option>CPU</option></select></label><label>Parallele Build-Jobs<select name="jobs"><option>1</option><option>2</option></select></label><p>GPU-Architekturen werden automatisch aus beiden Karten übernommen. Begrenzte Build-Last für den Parallelbetrieb.</p><button>llama.cpp herunterladen & bauen</button></form><button id="rt-cancel" class="secondary">Laufenden Build abbrechen</button><p id="rt-message" role="status"></p><div id="rt-job"></div><details><summary>Build-Ausgabe (eigener Build)</summary><pre id="rt-log" style="max-height:320px;overflow:auto;white-space:pre-wrap"></pre></details></section><section class="card"><h2>Versionen & Änderungen</h2><p>Das aktuelle reguläre Release steht zuerst. Nightlies enthalten neuere Änderungen, sind aber keine pauschale Empfehlung. Modell-, MTP- und Projektor-Kompatibilität vor einem Standardwechsel testen.</p><button id="rt-releases">Nach Updates suchen</button><div id="rt-releases-list"></div></section><section class="card"><h2>Geprüfte Builds</h2><div id="rt-builds"></div><button id="rt-rollback" class="secondary">Vorherigen Standard wiederherstellen</button><p>Als Standard auswählen startet keinen Prozess. Bestehende Router bleiben unverändert.</p></section><section class="card"><h2>Automatische Speicheranpassung</h2><p>Kein pauschales Reservefeld: Die Laufzeit soll Modell, gewünschten Kontext, Slots und den aktuell freien Speicher gemeinsam berücksichtigen. Unterstützte Builds verwenden llama.cpp <code>--fit on</code>; der Kontext wird ausdrücklich vorgegeben. Keine absichtlichen OOM-Tests auf den produktiv genutzten GPUs.</p><form id="rt-fit"><label>Vorhandenes Routerprofil<select name="profile" id="rt-profiles"></select></label><label>Gewünschter Gesamtkontext<input name="context" type="number" min="512" max="2097152" value="160000" required></label><label>Slots<input name="slots" type="number" min="1" max="16" value="2" required></label><button>Auto-Einpassung berechnen</button></form><pre id="rt-fit-result" style="white-space:pre-wrap"></pre><p>Ein Build allein kann keine Modell-Layerzahl bestimmen. Die Berechnung verwendet llama-fit-params ohne Gewichtsallokation. Sie ist eine Prognose für das Textmodell und kein Lasttest. Vision-Projektor und MTP sind noch nicht eingerechnet. Ein Modellstart erfolgt nicht.</p></section></div>`;}
|
||||
function html(){return `<div id="runtime-live"><div class="kicker">ATHENA / LLAMA.CPP</div><h1>llama.cpp & Updates</h1><p>Offizielle Versionen getrennt bauen und prüfen. Ein Build lädt kein Modell und ändert keine produktiven Laufzeiten.</p><section class="card"><h2>Voraussetzungen</h2><button id="rt-check">Werkzeuge & GPUs prüfen</button><div id="rt-prereq"></div></section><section class="card"><h2>Installation & Build</h2><form id="rt-build"><label>Release oder vollständiger Commit<input name="revision" required pattern="b[0-9]{3,8}|v[0-9]{1,4}[.][0-9]{1,4}[.][0-9]{1,4}|[a-f0-9]{40}" placeholder="Unten Releases abfragen und auswählen"></label><label>Backend<select name="backend"><option>CUDA</option><option>CPU</option></select></label><label>Parallele Build-Jobs<select name="jobs"><option>1</option><option>2</option></select></label><p>GPU-Architekturen werden automatisch aus beiden Karten übernommen. Begrenzte Build-Last für den Parallelbetrieb.</p><button>llama.cpp herunterladen & bauen</button></form><button id="rt-cancel" class="secondary">Laufenden Build abbrechen</button><p id="rt-message" role="status"></p><div id="rt-job"></div><details><summary>Build-Ausgabe (eigener Build)</summary><pre id="rt-log" style="max-height:320px;overflow:auto;white-space:pre-wrap"></pre></details></section><section class="card"><h2>Versionen & Änderungen</h2><p>Das aktuelle reguläre Release steht zuerst. Nightlies enthalten neuere Änderungen, sind aber keine pauschale Empfehlung. Modell-, MTP- und Projektor-Kompatibilität vor einem Standardwechsel testen.</p><button id="rt-releases">Nach Updates suchen</button><div id="rt-releases-list"></div></section><section class="card"><h2>Geprüfte Builds</h2><div id="rt-builds"></div><button id="rt-rollback" class="secondary">Vorherigen Standard wiederherstellen</button><p>Als Standard auswählen startet keinen Prozess. Bestehende Router bleiben unverändert.</p></section><section class="card"><h2>Automatische Speicheranpassung</h2><p>Kein pauschales Reservefeld: Die Laufzeit soll Modell, gewünschten Kontext, Slots und den aktuell freien Speicher gemeinsam berücksichtigen. Unterstützte Builds verwenden llama.cpp <code>--fit on</code>; der Kontext wird ausdrücklich vorgegeben. Keine absichtlichen OOM-Tests auf den produktiv genutzten GPUs.</p><form id="rt-fit"><label>Chatprofil auswählen<select name="profile" id="rt-profiles"></select></label><label>Gewünschter Gesamtkontext<input name="context" type="number" min="512" max="2097152" value="160000" required></label><label>Slots<input name="slots" type="number" min="1" max="16" value="2" required></label><button>Auto-Einpassung berechnen</button></form><pre id="rt-fit-result" style="white-space:pre-wrap"></pre><p>Ein Build allein kann keine Modell-Layerzahl bestimmen. Die Berechnung verwendet llama-fit-params ohne Gewichtsallokation. Sie ist eine Prognose für das Textmodell und kein Lasttest. Vision-Projektor und MTP sind noch nicht eingerechnet. Ein Modellstart erfolgt nicht. Die Prognose verwendet aktuell freien VRAM; falls das gewählte Modell bereits läuft, belegt es diesen Speicher weiterhin. Vor einer realistischen Neuplanung das Deck-Modell bei Bedarf bewusst entladen.</p></section></div>`;}
|
||||
function bind(){const root=document.querySelector('#runtime-live');if(!root)return;let running=false;
|
||||
const msg=t=>{if(root.isConnected)root.querySelector('#rt-message').textContent=t;};
|
||||
async function status(){try{const s=await api();if(!root.isConnected)return;running=s.job?.state==='running';root.querySelector('#rt-job').textContent=s.job?`${s.job.revision} · ${s.job.state} · ${s.job.phase} ${s.job.error||''}`:'Noch kein Build gestartet.';root.querySelector('#rt-cancel').disabled=!running;root.querySelector('#rt-build button').disabled=running;root.querySelector('#rt-rollback').disabled=!s.previous;root.querySelector('#rt-builds').innerHTML=s.builds.map(b=>`<article><h3>${e(b.revision)} · ${e(b.backend)} ${s.active===b.id?'· STANDARD':''}</h3><p>Commit ${e(b.commit)} · GPU-Ziele ${e(b.architectures.join(', ')||'CPU')} · Auto-Fit ${b.fit_supported?'unterstützt':'nicht unterstützt'}</p><button data-build="${e(b.id)}" ${s.active===b.id?'disabled':''}>Als Standard auswählen</button></article>`).join('')||'<p>Noch kein geprüfter Build vorhanden.</p>';root.querySelectorAll('[data-build]').forEach(b=>b.onclick=()=>action('/activate',{build_id:b.dataset.build}));if(s.job){const log=await api('/log');if(root.isConnected)root.querySelector('#rt-log').textContent=log.text;}if(running)setTimeout(()=>{if(root.isConnected)status();},3000);}catch(err){msg(err.message);}}
|
||||
@@ -11,7 +11,7 @@ const RuntimeUI=(()=>{
|
||||
root.querySelector('#rt-build').onsubmit=event=>{event.preventDefault();const d=Object.fromEntries(new FormData(event.target));d.jobs=Number(d.jobs);action('/build',d);};
|
||||
root.querySelector('#rt-cancel').onclick=()=>action('/cancel',{});root.querySelector('#rt-rollback').onclick=()=>action('/rollback',{});
|
||||
root.querySelector('#rt-releases').onclick=async()=>{msg('Offizielle Releases werden abgefragt …');try{const v=await api('/releases');if(!root.isConnected)return;root.querySelector('#rt-releases-list').innerHTML=v.releases.map(r=>`<article><h3>${e(r.tag)} ${r.prerelease?'· Nightly / Vorabversion':'· Reguläres Release'} · ${e(r.published)}</h3><button data-release="${e(r.tag)}">Für Build auswählen</button><a href="${e(r.url)}" target="_blank" rel="noopener noreferrer">Original auf GitHub</a><details><summary>Änderungen / Release-Notes</summary><pre style="white-space:pre-wrap">${e(r.notes)}</pre></details></article>`).join('');root.querySelectorAll('[data-release]').forEach(b=>b.onclick=()=>{root.querySelector('[name=revision]').value=b.dataset.release;msg('Version ausgewählt. Build kann gestartet werden.');});msg(v.releases.length+' offizielle Releases geladen.');}catch(err){msg(err.message);}};
|
||||
api('/references').then(v=>{if(!root.isConnected)return;const select=root.querySelector('#rt-profiles');select.innerHTML=v.profiles.map(p=>`<option value="${e(p.id)}" ${p.id==='medium'?'selected':''}>${e(p.id)} · ${e(p.file)} ${p.available?'':'(nicht eingebunden)'}</option>`).join('');select.onchange=()=>{const p=v.profiles.find(p=>p.id===select.value);root.querySelector('#rt-fit [name=context]').value=p.context;root.querySelector('#rt-fit [name=slots]').value=p.slots;};}).catch(err=>msg(err.message));
|
||||
api('/references').then(v=>{if(!root.isConnected)return;const select=root.querySelector('#rt-profiles');select.innerHTML=['deck','router'].map(source=>`<optgroup label="${source==='deck'?'Athena Deck':'Alter Router · Referenz'}">${v.profiles.filter(p=>p.source===source).map(p=>`<option value="${e(p.id)}" ${p.available?'':'disabled'}>${e(p.name||p.id)} · ${e(p.file)} ${p.available?'':'(nicht eingebunden)'}</option>`).join('')}</optgroup>`).join('');select.onchange=()=>{const p=v.profiles.find(p=>p.id===select.value);if(!p)return;root.querySelector('#rt-fit [name=context]').value=p.context;root.querySelector('#rt-fit [name=slots]').value=p.slots;};select.onchange();}).catch(err=>msg(err.message));
|
||||
root.querySelector('#rt-fit').onsubmit=async event=>{event.preventDefault();const b=event.target.querySelector('button');b.disabled=true;msg('Modellmetadaten und Speicherbedarf werden berechnet …');const d=Object.fromEntries(new FormData(event.target));d.context=Number(d.context);d.slots=Number(d.slots);try{const v=await api('/fit',d);if(root.isConnected)root.querySelector('#rt-fit-result').textContent=(v.success?'Prognose erfolgreich. Kein Modell geladen.':'Aktuell keine passende Einpassung ermittelt.')+(v.within_ram_limit===false?'\nNICHT STARTBAR in der Testinstanz: berechneter Host-RAM überschreitet deren Limit von '+v.ram_limit_mib+' MiB.':'')+'\nGPU-Layer (Prognose): '+(v.fitted?.gpu_layers??'nicht ermittelt')+'\nGPU-Verteilung: '+(v.fitted?.tensor_split??'ein Gerät / nicht ermittelt')+'\nAusgelassene belegte GPUs: '+v.excluded_gpus.join(', ')+'\nAutomatische Reserve (MiB): '+v.reserve_mib.join(', ')+'\n'+v.memory.map(m=>`${m.device}: Modell ${m.model_mib} MiB · Kontext ${m.context_mib} MiB · Rechenspeicher ${m.compute_mib} MiB`).join('\n')+'\n'+v.arguments+'\n'+v.details;msg('Berechnung abgeschlossen. Freier Speicher kann sich durch andere Dienste ändern.');}catch(err){msg(err.message);}finally{b.disabled=false;}};
|
||||
prerequisites();status();
|
||||
}
|
||||
|
||||
+16
-4
@@ -201,11 +201,23 @@ class Runtime:
|
||||
def references(self):
|
||||
base=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))
|
||||
specs=[('fast','Qwen3.8-27B-IQ4-MIX.gguf',76800,1,64),('medium','qwen3.8-27b-IQ4_XS-pure.gguf',160000,2,256),('large','qwen3.8-27b-IQ4_XS-pure.gguf',192000,1,256),('ultra','qwen3.8-27b-IQ4_XS-pure.gguf',262144,1,128),('uncensored','Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf',80000,1,256)]
|
||||
return {'profiles':[dict(id=i,file=f,context=c,slots=n,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs]}
|
||||
references=[]
|
||||
profiles=getattr(self,'profiles',None);catalog=getattr(self,'catalog',None)
|
||||
if profiles is not None and catalog is not None:
|
||||
with profiles.lock: saved=list(profiles.rows)
|
||||
for profile in saved:
|
||||
if profile.get('kind')!='chat':continue
|
||||
try: model=catalog.entry(profile['model_id'])
|
||||
except (KeyError,ValueError,OSError):continue
|
||||
if not model['file'].lower().endswith('.gguf') or not model['profile_eligible']:continue
|
||||
params=profile['parameters']
|
||||
references.append(dict(id='deck:'+profile['id'],source='deck',name=profile['name'],file=model['file'],model_id=model['id'],context=params['context'],slots=params['slots'],batch=params['batch'],ubatch=params['ubatch'],cache='q4_0',available=True))
|
||||
references.extend(dict(id=i,source='router',name=i,file=f,context=c,slots=n,batch=2048,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs)
|
||||
return {'profiles':references}
|
||||
def fit(self,profile,context,slots):
|
||||
if type(context)!=int or not 512<=context<=2097152 or type(slots)!=int or not 1<=slots<=16:raise ValueError('Kontext oder Slots außerhalb des erlaubten Bereichs.')
|
||||
ref=next((p for p in self.references()['profiles'] if p['id']==profile),None)
|
||||
if not ref or not ref['available']:raise ValueError('Referenzmodell nicht lesend eingebunden.')
|
||||
if not ref or not ref['available']:raise ValueError('Modelldatei für die Einpassung nicht verfügbar.')
|
||||
with self.lock:
|
||||
build=next((b for b in self.state['builds'] if b['id']==self.state['active']),None)
|
||||
if not build or not build.get('fit_tool'):raise ValueError('Zuerst einen CUDA-Build mit Fit-Werkzeug erstellen und auswählen.')
|
||||
@@ -215,8 +227,8 @@ class Runtime:
|
||||
self.fitting=True
|
||||
try:
|
||||
binary=self.root/build['id']/'build/bin/llama-fit-params'
|
||||
model=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file']
|
||||
args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--batch-size','2048','--ubatch-size',str(ref['ubatch'])]
|
||||
model=(self.catalog.root/ref['model_id']/('model'+Path(ref['file']).suffix)) if ref.get('source')=='deck' else Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file']
|
||||
args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k',ref.get('cache','q4_0'),'--cache-type-v',ref.get('cache','q4_0'),'--flash-attn','on','--batch-size',str(ref.get('batch',2048)),'--ubatch-size',str(ref['ubatch'])]
|
||||
snapshot=self.prerequisites()['gpus']
|
||||
usable=[g for g in snapshot if g['free_mib']>=max(1024,g['total_mib']*.10)]
|
||||
if not usable:raise ValueError('GPUs derzeit belegt. Keine sichere Auto-Prognose; produktive Dienste bleiben unverändert.')
|
||||
|
||||
@@ -73,6 +73,8 @@ class Server(ThreadingHTTPServer):
|
||||
self.catalog = Catalog(Path(state_dir or os.environ.get("DECK_STATE_DIR", ROOT/".state"))/"models")
|
||||
self.runtime = Runtime(Path(state_dir or os.environ.get("DECK_STATE_DIR", ROOT/".state"))/"runtime")
|
||||
self.profiles = Profiles(self.catalog.root.parent/"profiles.json",self.catalog)
|
||||
self.runtime.catalog=self.catalog
|
||||
self.runtime.profiles=self.profiles
|
||||
self.image_tests = ImageTests(self.catalog.root.parent/"image-tests",self.profiles)
|
||||
self.image_runtime = ImageRuntime(self.catalog.root.parent/"image-runtime")
|
||||
self.image_tests.runtime = self.image_runtime
|
||||
|
||||
@@ -1,10 +1,31 @@
|
||||
import tempfile
|
||||
import threading
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import patch
|
||||
from runtime import Runtime
|
||||
|
||||
class RuntimeTests(unittest.TestCase):
|
||||
def test_deck_chat_profile_is_available_for_fit(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
r=Runtime(d);model_id='a'*64;model_root=Path(d)/'models';model_dir=model_root/model_id
|
||||
model_dir.mkdir(parents=True);(model_dir/'model.gguf').write_bytes(b'gguf')
|
||||
entry=dict(id=model_id,file='gemma-4-31B-it-Q4_0.gguf',kind='chat',profile_eligible=True)
|
||||
r.catalog=SimpleNamespace(root=model_root,entry=lambda ident:entry if ident==model_id else None)
|
||||
r.profiles=SimpleNamespace(lock=threading.RLock(),rows=[dict(id='gemma-profile',name='Gemma4',kind='chat',model_id=model_id,parameters=dict(context=32768,slots=1,batch=512,ubatch=128))])
|
||||
refs=r.references()['profiles'];gemma=next(p for p in refs if p['id']=='deck:gemma-profile')
|
||||
self.assertEqual((gemma['name'],gemma['context'],gemma['slots']),("Gemma4",32768,1))
|
||||
self.assertTrue(gemma['available'])
|
||||
r.state.update(active='build',builds=[dict(id='build',backend='CUDA',fit_tool=True)])
|
||||
result=type('Result',(),dict(returncode=0,stdout='-c 32768 -ngl 4',stderr=''))()
|
||||
gpus=[dict(uuid='GPU-free',name='5080',total_mib=16000,free_mib=8000)]
|
||||
with patch.object(r,'prerequisites',return_value={'gpus':gpus}),patch('runtime.subprocess.run',return_value=result) as run:
|
||||
r.fit('deck:gemma-profile',32768,1)
|
||||
args=run.call_args.args[0]
|
||||
self.assertEqual(args[args.index('--model')+1],str(model_dir/'model.gguf'))
|
||||
self.assertEqual(args[args.index('--batch-size')+1],'512')
|
||||
self.assertEqual(args[args.index('--ubatch-size')+1],'128')
|
||||
def test_untrusted_build_arguments(self):
|
||||
with tempfile.TemporaryDirectory() as d:
|
||||
r=Runtime(d)
|
||||
|
||||
Reference in New Issue
Block a user