Include Deck chat profiles in runtime memory fit
This commit is contained in:
+16
-4
@@ -201,11 +201,23 @@ class Runtime:
|
||||
def references(self):
|
||||
base=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))
|
||||
specs=[('fast','Qwen3.8-27B-IQ4-MIX.gguf',76800,1,64),('medium','qwen3.8-27b-IQ4_XS-pure.gguf',160000,2,256),('large','qwen3.8-27b-IQ4_XS-pure.gguf',192000,1,256),('ultra','qwen3.8-27b-IQ4_XS-pure.gguf',262144,1,128),('uncensored','Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf',80000,1,256)]
|
||||
return {'profiles':[dict(id=i,file=f,context=c,slots=n,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs]}
|
||||
references=[]
|
||||
profiles=getattr(self,'profiles',None);catalog=getattr(self,'catalog',None)
|
||||
if profiles is not None and catalog is not None:
|
||||
with profiles.lock: saved=list(profiles.rows)
|
||||
for profile in saved:
|
||||
if profile.get('kind')!='chat':continue
|
||||
try: model=catalog.entry(profile['model_id'])
|
||||
except (KeyError,ValueError,OSError):continue
|
||||
if not model['file'].lower().endswith('.gguf') or not model['profile_eligible']:continue
|
||||
params=profile['parameters']
|
||||
references.append(dict(id='deck:'+profile['id'],source='deck',name=profile['name'],file=model['file'],model_id=model['id'],context=params['context'],slots=params['slots'],batch=params['batch'],ubatch=params['ubatch'],cache='q4_0',available=True))
|
||||
references.extend(dict(id=i,source='router',name=i,file=f,context=c,slots=n,batch=2048,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs)
|
||||
return {'profiles':references}
|
||||
def fit(self,profile,context,slots):
|
||||
if type(context)!=int or not 512<=context<=2097152 or type(slots)!=int or not 1<=slots<=16:raise ValueError('Kontext oder Slots außerhalb des erlaubten Bereichs.')
|
||||
ref=next((p for p in self.references()['profiles'] if p['id']==profile),None)
|
||||
if not ref or not ref['available']:raise ValueError('Referenzmodell nicht lesend eingebunden.')
|
||||
if not ref or not ref['available']:raise ValueError('Modelldatei für die Einpassung nicht verfügbar.')
|
||||
with self.lock:
|
||||
build=next((b for b in self.state['builds'] if b['id']==self.state['active']),None)
|
||||
if not build or not build.get('fit_tool'):raise ValueError('Zuerst einen CUDA-Build mit Fit-Werkzeug erstellen und auswählen.')
|
||||
@@ -215,8 +227,8 @@ class Runtime:
|
||||
self.fitting=True
|
||||
try:
|
||||
binary=self.root/build['id']/'build/bin/llama-fit-params'
|
||||
model=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file']
|
||||
args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--batch-size','2048','--ubatch-size',str(ref['ubatch'])]
|
||||
model=(self.catalog.root/ref['model_id']/('model'+Path(ref['file']).suffix)) if ref.get('source')=='deck' else Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file']
|
||||
args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k',ref.get('cache','q4_0'),'--cache-type-v',ref.get('cache','q4_0'),'--flash-attn','on','--batch-size',str(ref.get('batch',2048)),'--ubatch-size',str(ref['ubatch'])]
|
||||
snapshot=self.prerequisites()['gpus']
|
||||
usable=[g for g in snapshot if g['free_mib']>=max(1024,g['total_mib']*.10)]
|
||||
if not usable:raise ValueError('GPUs derzeit belegt. Keine sichere Auto-Prognose; produktive Dienste bleiben unverändert.')
|
||||
|
||||
Reference in New Issue
Block a user