Include Deck chat profiles in runtime memory fit

This commit is contained in:
Mikei386
2026-09-30 08:03:46 +02:00
parent dfe3c4398c
commit 638202ee86
4 changed files with 41 additions and 6 deletions
+2 -2
View File
@@ -1,7 +1,7 @@
const RuntimeUI=(()=>{ const RuntimeUI=(()=>{
const e=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;',"'":'&#39;'}[c])); const e=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&amp;','<':'&lt;','>':'&gt;','"':'&quot;',"'":'&#39;'}[c]));
async function api(path='',data){const r=await fetch('/api/v1/runtime'+path,data?{method:'POST',headers:{'Content-Type':'application/json','X-Athena-Deck':'1'},body:JSON.stringify(data)}:{});const v=await r.json();if(!r.ok)throw Error(v.error||'Anfrage fehlgeschlagen');return v;} async function api(path='',data){const r=await fetch('/api/v1/runtime'+path,data?{method:'POST',headers:{'Content-Type':'application/json','X-Athena-Deck':'1'},body:JSON.stringify(data)}:{});const v=await r.json();if(!r.ok)throw Error(v.error||'Anfrage fehlgeschlagen');return v;}
function html(){return `<div id="runtime-live"><div class="kicker">ATHENA / LLAMA.CPP</div><h1>llama.cpp &amp; Updates</h1><p>Offizielle Versionen getrennt bauen und prüfen. Ein Build lädt kein Modell und ändert keine produktiven Laufzeiten.</p><section class="card"><h2>Voraussetzungen</h2><button id="rt-check">Werkzeuge &amp; GPUs prüfen</button><div id="rt-prereq"></div></section><section class="card"><h2>Installation &amp; Build</h2><form id="rt-build"><label>Release oder vollständiger Commit<input name="revision" required pattern="b[0-9]{3,8}|v[0-9]{1,4}[.][0-9]{1,4}[.][0-9]{1,4}|[a-f0-9]{40}" placeholder="Unten Releases abfragen und auswählen"></label><label>Backend<select name="backend"><option>CUDA</option><option>CPU</option></select></label><label>Parallele Build-Jobs<select name="jobs"><option>1</option><option>2</option></select></label><p>GPU-Architekturen werden automatisch aus beiden Karten übernommen. Begrenzte Build-Last für den Parallelbetrieb.</p><button>llama.cpp herunterladen &amp; bauen</button></form><button id="rt-cancel" class="secondary">Laufenden Build abbrechen</button><p id="rt-message" role="status"></p><div id="rt-job"></div><details><summary>Build-Ausgabe (eigener Build)</summary><pre id="rt-log" style="max-height:320px;overflow:auto;white-space:pre-wrap"></pre></details></section><section class="card"><h2>Versionen &amp; Änderungen</h2><p>Das aktuelle reguläre Release steht zuerst. Nightlies enthalten neuere Änderungen, sind aber keine pauschale Empfehlung. Modell-, MTP- und Projektor-Kompatibilität vor einem Standardwechsel testen.</p><button id="rt-releases">Nach Updates suchen</button><div id="rt-releases-list"></div></section><section class="card"><h2>Geprüfte Builds</h2><div id="rt-builds"></div><button id="rt-rollback" class="secondary">Vorherigen Standard wiederherstellen</button><p>Als Standard auswählen startet keinen Prozess. Bestehende Router bleiben unverändert.</p></section><section class="card"><h2>Automatische Speicheranpassung</h2><p>Kein pauschales Reservefeld: Die Laufzeit soll Modell, gewünschten Kontext, Slots und den aktuell freien Speicher gemeinsam berücksichtigen. Unterstützte Builds verwenden llama.cpp <code>--fit on</code>; der Kontext wird ausdrücklich vorgegeben. Keine absichtlichen OOM-Tests auf den produktiv genutzten GPUs.</p><form id="rt-fit"><label>Vorhandenes Routerprofil<select name="profile" id="rt-profiles"></select></label><label>Gewünschter Gesamtkontext<input name="context" type="number" min="512" max="2097152" value="160000" required></label><label>Slots<input name="slots" type="number" min="1" max="16" value="2" required></label><button>Auto-Einpassung berechnen</button></form><pre id="rt-fit-result" style="white-space:pre-wrap"></pre><p>Ein Build allein kann keine Modell-Layerzahl bestimmen. Die Berechnung verwendet llama-fit-params ohne Gewichtsallokation. Sie ist eine Prognose für das Textmodell und kein Lasttest. Vision-Projektor und MTP sind noch nicht eingerechnet. Ein Modellstart erfolgt nicht.</p></section></div>`;} function html(){return `<div id="runtime-live"><div class="kicker">ATHENA / LLAMA.CPP</div><h1>llama.cpp &amp; Updates</h1><p>Offizielle Versionen getrennt bauen und prüfen. Ein Build lädt kein Modell und ändert keine produktiven Laufzeiten.</p><section class="card"><h2>Voraussetzungen</h2><button id="rt-check">Werkzeuge &amp; GPUs prüfen</button><div id="rt-prereq"></div></section><section class="card"><h2>Installation &amp; Build</h2><form id="rt-build"><label>Release oder vollständiger Commit<input name="revision" required pattern="b[0-9]{3,8}|v[0-9]{1,4}[.][0-9]{1,4}[.][0-9]{1,4}|[a-f0-9]{40}" placeholder="Unten Releases abfragen und auswählen"></label><label>Backend<select name="backend"><option>CUDA</option><option>CPU</option></select></label><label>Parallele Build-Jobs<select name="jobs"><option>1</option><option>2</option></select></label><p>GPU-Architekturen werden automatisch aus beiden Karten übernommen. Begrenzte Build-Last für den Parallelbetrieb.</p><button>llama.cpp herunterladen &amp; bauen</button></form><button id="rt-cancel" class="secondary">Laufenden Build abbrechen</button><p id="rt-message" role="status"></p><div id="rt-job"></div><details><summary>Build-Ausgabe (eigener Build)</summary><pre id="rt-log" style="max-height:320px;overflow:auto;white-space:pre-wrap"></pre></details></section><section class="card"><h2>Versionen &amp; Änderungen</h2><p>Das aktuelle reguläre Release steht zuerst. Nightlies enthalten neuere Änderungen, sind aber keine pauschale Empfehlung. Modell-, MTP- und Projektor-Kompatibilität vor einem Standardwechsel testen.</p><button id="rt-releases">Nach Updates suchen</button><div id="rt-releases-list"></div></section><section class="card"><h2>Geprüfte Builds</h2><div id="rt-builds"></div><button id="rt-rollback" class="secondary">Vorherigen Standard wiederherstellen</button><p>Als Standard auswählen startet keinen Prozess. Bestehende Router bleiben unverändert.</p></section><section class="card"><h2>Automatische Speicheranpassung</h2><p>Kein pauschales Reservefeld: Die Laufzeit soll Modell, gewünschten Kontext, Slots und den aktuell freien Speicher gemeinsam berücksichtigen. Unterstützte Builds verwenden llama.cpp <code>--fit on</code>; der Kontext wird ausdrücklich vorgegeben. Keine absichtlichen OOM-Tests auf den produktiv genutzten GPUs.</p><form id="rt-fit"><label>Chatprofil auswählen<select name="profile" id="rt-profiles"></select></label><label>Gewünschter Gesamtkontext<input name="context" type="number" min="512" max="2097152" value="160000" required></label><label>Slots<input name="slots" type="number" min="1" max="16" value="2" required></label><button>Auto-Einpassung berechnen</button></form><pre id="rt-fit-result" style="white-space:pre-wrap"></pre><p>Ein Build allein kann keine Modell-Layerzahl bestimmen. Die Berechnung verwendet llama-fit-params ohne Gewichtsallokation. Sie ist eine Prognose für das Textmodell und kein Lasttest. Vision-Projektor und MTP sind noch nicht eingerechnet. Ein Modellstart erfolgt nicht. Die Prognose verwendet aktuell freien VRAM; falls das gewählte Modell bereits läuft, belegt es diesen Speicher weiterhin. Vor einer realistischen Neuplanung das Deck-Modell bei Bedarf bewusst entladen.</p></section></div>`;}
function bind(){const root=document.querySelector('#runtime-live');if(!root)return;let running=false; function bind(){const root=document.querySelector('#runtime-live');if(!root)return;let running=false;
const msg=t=>{if(root.isConnected)root.querySelector('#rt-message').textContent=t;}; const msg=t=>{if(root.isConnected)root.querySelector('#rt-message').textContent=t;};
async function status(){try{const s=await api();if(!root.isConnected)return;running=s.job?.state==='running';root.querySelector('#rt-job').textContent=s.job?`${s.job.revision} · ${s.job.state} · ${s.job.phase} ${s.job.error||''}`:'Noch kein Build gestartet.';root.querySelector('#rt-cancel').disabled=!running;root.querySelector('#rt-build button').disabled=running;root.querySelector('#rt-rollback').disabled=!s.previous;root.querySelector('#rt-builds').innerHTML=s.builds.map(b=>`<article><h3>${e(b.revision)} · ${e(b.backend)} ${s.active===b.id?'· STANDARD':''}</h3><p>Commit ${e(b.commit)} · GPU-Ziele ${e(b.architectures.join(', ')||'CPU')} · Auto-Fit ${b.fit_supported?'unterstützt':'nicht unterstützt'}</p><button data-build="${e(b.id)}" ${s.active===b.id?'disabled':''}>Als Standard auswählen</button></article>`).join('')||'<p>Noch kein geprüfter Build vorhanden.</p>';root.querySelectorAll('[data-build]').forEach(b=>b.onclick=()=>action('/activate',{build_id:b.dataset.build}));if(s.job){const log=await api('/log');if(root.isConnected)root.querySelector('#rt-log').textContent=log.text;}if(running)setTimeout(()=>{if(root.isConnected)status();},3000);}catch(err){msg(err.message);}} async function status(){try{const s=await api();if(!root.isConnected)return;running=s.job?.state==='running';root.querySelector('#rt-job').textContent=s.job?`${s.job.revision} · ${s.job.state} · ${s.job.phase} ${s.job.error||''}`:'Noch kein Build gestartet.';root.querySelector('#rt-cancel').disabled=!running;root.querySelector('#rt-build button').disabled=running;root.querySelector('#rt-rollback').disabled=!s.previous;root.querySelector('#rt-builds').innerHTML=s.builds.map(b=>`<article><h3>${e(b.revision)} · ${e(b.backend)} ${s.active===b.id?'· STANDARD':''}</h3><p>Commit ${e(b.commit)} · GPU-Ziele ${e(b.architectures.join(', ')||'CPU')} · Auto-Fit ${b.fit_supported?'unterstützt':'nicht unterstützt'}</p><button data-build="${e(b.id)}" ${s.active===b.id?'disabled':''}>Als Standard auswählen</button></article>`).join('')||'<p>Noch kein geprüfter Build vorhanden.</p>';root.querySelectorAll('[data-build]').forEach(b=>b.onclick=()=>action('/activate',{build_id:b.dataset.build}));if(s.job){const log=await api('/log');if(root.isConnected)root.querySelector('#rt-log').textContent=log.text;}if(running)setTimeout(()=>{if(root.isConnected)status();},3000);}catch(err){msg(err.message);}}
@@ -11,7 +11,7 @@ const RuntimeUI=(()=>{
root.querySelector('#rt-build').onsubmit=event=>{event.preventDefault();const d=Object.fromEntries(new FormData(event.target));d.jobs=Number(d.jobs);action('/build',d);}; root.querySelector('#rt-build').onsubmit=event=>{event.preventDefault();const d=Object.fromEntries(new FormData(event.target));d.jobs=Number(d.jobs);action('/build',d);};
root.querySelector('#rt-cancel').onclick=()=>action('/cancel',{});root.querySelector('#rt-rollback').onclick=()=>action('/rollback',{}); root.querySelector('#rt-cancel').onclick=()=>action('/cancel',{});root.querySelector('#rt-rollback').onclick=()=>action('/rollback',{});
root.querySelector('#rt-releases').onclick=async()=>{msg('Offizielle Releases werden abgefragt …');try{const v=await api('/releases');if(!root.isConnected)return;root.querySelector('#rt-releases-list').innerHTML=v.releases.map(r=>`<article><h3>${e(r.tag)} ${r.prerelease?'· Nightly / Vorabversion':'· Reguläres Release'} · ${e(r.published)}</h3><button data-release="${e(r.tag)}">Für Build auswählen</button><a href="${e(r.url)}" target="_blank" rel="noopener noreferrer">Original auf GitHub</a><details><summary>Änderungen / Release-Notes</summary><pre style="white-space:pre-wrap">${e(r.notes)}</pre></details></article>`).join('');root.querySelectorAll('[data-release]').forEach(b=>b.onclick=()=>{root.querySelector('[name=revision]').value=b.dataset.release;msg('Version ausgewählt. Build kann gestartet werden.');});msg(v.releases.length+' offizielle Releases geladen.');}catch(err){msg(err.message);}}; root.querySelector('#rt-releases').onclick=async()=>{msg('Offizielle Releases werden abgefragt …');try{const v=await api('/releases');if(!root.isConnected)return;root.querySelector('#rt-releases-list').innerHTML=v.releases.map(r=>`<article><h3>${e(r.tag)} ${r.prerelease?'· Nightly / Vorabversion':'· Reguläres Release'} · ${e(r.published)}</h3><button data-release="${e(r.tag)}">Für Build auswählen</button><a href="${e(r.url)}" target="_blank" rel="noopener noreferrer">Original auf GitHub</a><details><summary>Änderungen / Release-Notes</summary><pre style="white-space:pre-wrap">${e(r.notes)}</pre></details></article>`).join('');root.querySelectorAll('[data-release]').forEach(b=>b.onclick=()=>{root.querySelector('[name=revision]').value=b.dataset.release;msg('Version ausgewählt. Build kann gestartet werden.');});msg(v.releases.length+' offizielle Releases geladen.');}catch(err){msg(err.message);}};
api('/references').then(v=>{if(!root.isConnected)return;const select=root.querySelector('#rt-profiles');select.innerHTML=v.profiles.map(p=>`<option value="${e(p.id)}" ${p.id==='medium'?'selected':''}>${e(p.id)} · ${e(p.file)} ${p.available?'':'(nicht eingebunden)'}</option>`).join('');select.onchange=()=>{const p=v.profiles.find(p=>p.id===select.value);root.querySelector('#rt-fit [name=context]').value=p.context;root.querySelector('#rt-fit [name=slots]').value=p.slots;};}).catch(err=>msg(err.message)); api('/references').then(v=>{if(!root.isConnected)return;const select=root.querySelector('#rt-profiles');select.innerHTML=['deck','router'].map(source=>`<optgroup label="${source==='deck'?'Athena Deck':'Alter Router · Referenz'}">${v.profiles.filter(p=>p.source===source).map(p=>`<option value="${e(p.id)}" ${p.available?'':'disabled'}>${e(p.name||p.id)} · ${e(p.file)} ${p.available?'':'(nicht eingebunden)'}</option>`).join('')}</optgroup>`).join('');select.onchange=()=>{const p=v.profiles.find(p=>p.id===select.value);if(!p)return;root.querySelector('#rt-fit [name=context]').value=p.context;root.querySelector('#rt-fit [name=slots]').value=p.slots;};select.onchange();}).catch(err=>msg(err.message));
root.querySelector('#rt-fit').onsubmit=async event=>{event.preventDefault();const b=event.target.querySelector('button');b.disabled=true;msg('Modellmetadaten und Speicherbedarf werden berechnet …');const d=Object.fromEntries(new FormData(event.target));d.context=Number(d.context);d.slots=Number(d.slots);try{const v=await api('/fit',d);if(root.isConnected)root.querySelector('#rt-fit-result').textContent=(v.success?'Prognose erfolgreich. Kein Modell geladen.':'Aktuell keine passende Einpassung ermittelt.')+(v.within_ram_limit===false?'\nNICHT STARTBAR in der Testinstanz: berechneter Host-RAM überschreitet deren Limit von '+v.ram_limit_mib+' MiB.':'')+'\nGPU-Layer (Prognose): '+(v.fitted?.gpu_layers??'nicht ermittelt')+'\nGPU-Verteilung: '+(v.fitted?.tensor_split??'ein Gerät / nicht ermittelt')+'\nAusgelassene belegte GPUs: '+v.excluded_gpus.join(', ')+'\nAutomatische Reserve (MiB): '+v.reserve_mib.join(', ')+'\n'+v.memory.map(m=>`${m.device}: Modell ${m.model_mib} MiB · Kontext ${m.context_mib} MiB · Rechenspeicher ${m.compute_mib} MiB`).join('\n')+'\n'+v.arguments+'\n'+v.details;msg('Berechnung abgeschlossen. Freier Speicher kann sich durch andere Dienste ändern.');}catch(err){msg(err.message);}finally{b.disabled=false;}}; root.querySelector('#rt-fit').onsubmit=async event=>{event.preventDefault();const b=event.target.querySelector('button');b.disabled=true;msg('Modellmetadaten und Speicherbedarf werden berechnet …');const d=Object.fromEntries(new FormData(event.target));d.context=Number(d.context);d.slots=Number(d.slots);try{const v=await api('/fit',d);if(root.isConnected)root.querySelector('#rt-fit-result').textContent=(v.success?'Prognose erfolgreich. Kein Modell geladen.':'Aktuell keine passende Einpassung ermittelt.')+(v.within_ram_limit===false?'\nNICHT STARTBAR in der Testinstanz: berechneter Host-RAM überschreitet deren Limit von '+v.ram_limit_mib+' MiB.':'')+'\nGPU-Layer (Prognose): '+(v.fitted?.gpu_layers??'nicht ermittelt')+'\nGPU-Verteilung: '+(v.fitted?.tensor_split??'ein Gerät / nicht ermittelt')+'\nAusgelassene belegte GPUs: '+v.excluded_gpus.join(', ')+'\nAutomatische Reserve (MiB): '+v.reserve_mib.join(', ')+'\n'+v.memory.map(m=>`${m.device}: Modell ${m.model_mib} MiB · Kontext ${m.context_mib} MiB · Rechenspeicher ${m.compute_mib} MiB`).join('\n')+'\n'+v.arguments+'\n'+v.details;msg('Berechnung abgeschlossen. Freier Speicher kann sich durch andere Dienste ändern.');}catch(err){msg(err.message);}finally{b.disabled=false;}};
prerequisites();status(); prerequisites();status();
} }
+16 -4
View File
@@ -201,11 +201,23 @@ class Runtime:
def references(self): def references(self):
base=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models')) base=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))
specs=[('fast','Qwen3.8-27B-IQ4-MIX.gguf',76800,1,64),('medium','qwen3.8-27b-IQ4_XS-pure.gguf',160000,2,256),('large','qwen3.8-27b-IQ4_XS-pure.gguf',192000,1,256),('ultra','qwen3.8-27b-IQ4_XS-pure.gguf',262144,1,128),('uncensored','Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf',80000,1,256)] specs=[('fast','Qwen3.8-27B-IQ4-MIX.gguf',76800,1,64),('medium','qwen3.8-27b-IQ4_XS-pure.gguf',160000,2,256),('large','qwen3.8-27b-IQ4_XS-pure.gguf',192000,1,256),('ultra','qwen3.8-27b-IQ4_XS-pure.gguf',262144,1,128),('uncensored','Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf',80000,1,256)]
return {'profiles':[dict(id=i,file=f,context=c,slots=n,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs]} references=[]
profiles=getattr(self,'profiles',None);catalog=getattr(self,'catalog',None)
if profiles is not None and catalog is not None:
with profiles.lock: saved=list(profiles.rows)
for profile in saved:
if profile.get('kind')!='chat':continue
try: model=catalog.entry(profile['model_id'])
except (KeyError,ValueError,OSError):continue
if not model['file'].lower().endswith('.gguf') or not model['profile_eligible']:continue
params=profile['parameters']
references.append(dict(id='deck:'+profile['id'],source='deck',name=profile['name'],file=model['file'],model_id=model['id'],context=params['context'],slots=params['slots'],batch=params['batch'],ubatch=params['ubatch'],cache='q4_0',available=True))
references.extend(dict(id=i,source='router',name=i,file=f,context=c,slots=n,batch=2048,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs)
return {'profiles':references}
def fit(self,profile,context,slots): def fit(self,profile,context,slots):
if type(context)!=int or not 512<=context<=2097152 or type(slots)!=int or not 1<=slots<=16:raise ValueError('Kontext oder Slots außerhalb des erlaubten Bereichs.') if type(context)!=int or not 512<=context<=2097152 or type(slots)!=int or not 1<=slots<=16:raise ValueError('Kontext oder Slots außerhalb des erlaubten Bereichs.')
ref=next((p for p in self.references()['profiles'] if p['id']==profile),None) ref=next((p for p in self.references()['profiles'] if p['id']==profile),None)
if not ref or not ref['available']:raise ValueError('Referenzmodell nicht lesend eingebunden.') if not ref or not ref['available']:raise ValueError('Modelldatei für die Einpassung nicht verfügbar.')
with self.lock: with self.lock:
build=next((b for b in self.state['builds'] if b['id']==self.state['active']),None) build=next((b for b in self.state['builds'] if b['id']==self.state['active']),None)
if not build or not build.get('fit_tool'):raise ValueError('Zuerst einen CUDA-Build mit Fit-Werkzeug erstellen und auswählen.') if not build or not build.get('fit_tool'):raise ValueError('Zuerst einen CUDA-Build mit Fit-Werkzeug erstellen und auswählen.')
@@ -215,8 +227,8 @@ class Runtime:
self.fitting=True self.fitting=True
try: try:
binary=self.root/build['id']/'build/bin/llama-fit-params' binary=self.root/build['id']/'build/bin/llama-fit-params'
model=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file'] model=(self.catalog.root/ref['model_id']/('model'+Path(ref['file']).suffix)) if ref.get('source')=='deck' else Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file']
args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--batch-size','2048','--ubatch-size',str(ref['ubatch'])] args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k',ref.get('cache','q4_0'),'--cache-type-v',ref.get('cache','q4_0'),'--flash-attn','on','--batch-size',str(ref.get('batch',2048)),'--ubatch-size',str(ref['ubatch'])]
snapshot=self.prerequisites()['gpus'] snapshot=self.prerequisites()['gpus']
usable=[g for g in snapshot if g['free_mib']>=max(1024,g['total_mib']*.10)] usable=[g for g in snapshot if g['free_mib']>=max(1024,g['total_mib']*.10)]
if not usable:raise ValueError('GPUs derzeit belegt. Keine sichere Auto-Prognose; produktive Dienste bleiben unverändert.') if not usable:raise ValueError('GPUs derzeit belegt. Keine sichere Auto-Prognose; produktive Dienste bleiben unverändert.')
+2
View File
@@ -73,6 +73,8 @@ class Server(ThreadingHTTPServer):
self.catalog = Catalog(Path(state_dir or os.environ.get("DECK_STATE_DIR", ROOT/".state"))/"models") self.catalog = Catalog(Path(state_dir or os.environ.get("DECK_STATE_DIR", ROOT/".state"))/"models")
self.runtime = Runtime(Path(state_dir or os.environ.get("DECK_STATE_DIR", ROOT/".state"))/"runtime") self.runtime = Runtime(Path(state_dir or os.environ.get("DECK_STATE_DIR", ROOT/".state"))/"runtime")
self.profiles = Profiles(self.catalog.root.parent/"profiles.json",self.catalog) self.profiles = Profiles(self.catalog.root.parent/"profiles.json",self.catalog)
self.runtime.catalog=self.catalog
self.runtime.profiles=self.profiles
self.image_tests = ImageTests(self.catalog.root.parent/"image-tests",self.profiles) self.image_tests = ImageTests(self.catalog.root.parent/"image-tests",self.profiles)
self.image_runtime = ImageRuntime(self.catalog.root.parent/"image-runtime") self.image_runtime = ImageRuntime(self.catalog.root.parent/"image-runtime")
self.image_tests.runtime = self.image_runtime self.image_tests.runtime = self.image_runtime
+21
View File
@@ -1,10 +1,31 @@
import tempfile import tempfile
import threading
import unittest import unittest
from pathlib import Path from pathlib import Path
from types import SimpleNamespace
from unittest.mock import patch from unittest.mock import patch
from runtime import Runtime from runtime import Runtime
class RuntimeTests(unittest.TestCase): class RuntimeTests(unittest.TestCase):
def test_deck_chat_profile_is_available_for_fit(self):
with tempfile.TemporaryDirectory() as d:
r=Runtime(d);model_id='a'*64;model_root=Path(d)/'models';model_dir=model_root/model_id
model_dir.mkdir(parents=True);(model_dir/'model.gguf').write_bytes(b'gguf')
entry=dict(id=model_id,file='gemma-4-31B-it-Q4_0.gguf',kind='chat',profile_eligible=True)
r.catalog=SimpleNamespace(root=model_root,entry=lambda ident:entry if ident==model_id else None)
r.profiles=SimpleNamespace(lock=threading.RLock(),rows=[dict(id='gemma-profile',name='Gemma4',kind='chat',model_id=model_id,parameters=dict(context=32768,slots=1,batch=512,ubatch=128))])
refs=r.references()['profiles'];gemma=next(p for p in refs if p['id']=='deck:gemma-profile')
self.assertEqual((gemma['name'],gemma['context'],gemma['slots']),("Gemma4",32768,1))
self.assertTrue(gemma['available'])
r.state.update(active='build',builds=[dict(id='build',backend='CUDA',fit_tool=True)])
result=type('Result',(),dict(returncode=0,stdout='-c 32768 -ngl 4',stderr=''))()
gpus=[dict(uuid='GPU-free',name='5080',total_mib=16000,free_mib=8000)]
with patch.object(r,'prerequisites',return_value={'gpus':gpus}),patch('runtime.subprocess.run',return_value=result) as run:
r.fit('deck:gemma-profile',32768,1)
args=run.call_args.args[0]
self.assertEqual(args[args.index('--model')+1],str(model_dir/'model.gguf'))
self.assertEqual(args[args.index('--batch-size')+1],'512')
self.assertEqual(args[args.index('--ubatch-size')+1],'128')
def test_untrusted_build_arguments(self): def test_untrusted_build_arguments(self):
with tempfile.TemporaryDirectory() as d: with tempfile.TemporaryDirectory() as d:
r=Runtime(d) r=Runtime(d)