Add bounded GPU-priority automatic model testing and profile export
This commit is contained in:
1 parent
f99cd26b1d
commit
a77266c7bd
12 files changed
+187
-10
No files matched your search
+5
-1
@@ -124,7 +124,10 @@ class LlamaWorker:
|
||||
self.stop()
|
||||
with self.lock:self.state='failed';self.phase='Modellstart fehlgeschlagen';self.error=str(exc) if isinstance(exc,ValueError) else 'llama.cpp konnte nicht gestartet werden.'
|
||||
raise InferenceError(self.error) from None
|
||||
def _start(self,profile,generation,cancel=lambda:False):
|
||||
def plan(self,profile,cancel=lambda:False):
|
||||
self._start(profile,self.generation,cancel,plan_only=True)
|
||||
return self.memory_plan
|
||||
def _start(self,profile,generation,cancel=lambda:False,plan_only=False):
|
||||
directory=self.build();params=profile['parameters'];entry=self.catalog.entry(profile['model_id'])
|
||||
model=(self.catalog.root/entry['id']/('model'+Path(entry['file']).suffix)).resolve()
|
||||
mtp=params.get('mtp',False)
|
||||
@@ -197,6 +200,7 @@ class LlamaWorker:
|
||||
fits,memory=estimate(layers,extra)
|
||||
if not fits:raise InferenceError('Speicherprognose überschreitet die GPU-/RAM-Reserve.')
|
||||
if cancel():raise InferenceError('Modellstart abgebrochen.')
|
||||
if plan_only:return
|
||||
with self.lock:self.phase='Modell wird geladen'
|
||||
launch=common+extra+['--gpu-layers',str(layers),'--fit','off','--kv-unified','--threads',str(params['threads']),'--load-mode','none','--host','127.0.0.1','--alias',profile['name'],'--no-webui','--log-disable']
|
||||
if mtp:launch+=['--spec-type','draft-mtp','--spec-draft-n-max',str(params.get('mtp_tokens',2)),'--spec-draft-p-min',str(params.get('mtp_min_p',.05)),'--spec-draft-type-k','f16','--spec-draft-type-v','f16']
|
||||
|
||||
Reference in new issue
Block a user