Add bounded GPU-priority automatic model testing and profile export

This commit is contained in:
Mikei386 committed 2026-09-28 21:29:48 +02:00
1 parent f99cd26b1d
commit a77266c7bd
12 files changed
+187 -10

No files matched your search

+5 -1
View File
@@ -124,7 +124,10 @@ class LlamaWorker:
self.stop()
with self.lock:self.state='failed';self.phase='Modellstart fehlgeschlagen';self.error=str(exc) if isinstance(exc,ValueError) else 'llama.cpp konnte nicht gestartet werden.'
raise InferenceError(self.error) from None
def _start(self,profile,generation,cancel=lambda:False):
def plan(self,profile,cancel=lambda:False):
self._start(profile,self.generation,cancel,plan_only=True)
return self.memory_plan
def _start(self,profile,generation,cancel=lambda:False,plan_only=False):
directory=self.build();params=profile['parameters'];entry=self.catalog.entry(profile['model_id'])
model=(self.catalog.root/entry['id']/('model'+Path(entry['file']).suffix)).resolve()
mtp=params.get('mtp',False)
@@ -197,6 +200,7 @@ class LlamaWorker:
fits,memory=estimate(layers,extra)
if not fits:raise InferenceError('Speicherprognose überschreitet die GPU-/RAM-Reserve.')
if cancel():raise InferenceError('Modellstart abgebrochen.')
if plan_only:return
with self.lock:self.phase='Modell wird geladen'
launch=common+extra+['--gpu-layers',str(layers),'--fit','off','--kv-unified','--threads',str(params['threads']),'--load-mode','none','--host','127.0.0.1','--alias',profile['name'],'--no-webui','--log-disable']
if mtp:launch+=['--spec-type','draft-mtp','--spec-draft-n-max',str(params.get('mtp_tokens',2)),'--spec-draft-p-min',str(params.get('mtp_min_p',.05)),'--spec-draft-type-k','f16','--spec-draft-type-v','f16']