Make per-profile GPU reserve configurable without changing existing defaults

This commit is contained in:
Mikei386
2026-09-28 22:29:24 +02:00
parent 682211e6a9
commit 18a5183719
7 changed files with 36 additions and 10 deletions
+4 -3
View File
@@ -161,11 +161,12 @@ class LlamaWorker:
# Never inherit LLAMA_ARG_* or user HF credentials into the model server.
env={k:v for k,v in os.environ.items() if not k.startswith(('LLAMA_','HF_','DECK_'))}
env.update(CUDA_VISIBLE_DEVICES=','.join(g['uuid'] for g in selected),HF_HUB_OFFLINE='1',OMP_NUM_THREADS=str(params['threads']))
common=['--model',str(model),'--ctx-size',str(params['context']),'--parallel',str(params['slots']),'--batch-size',str(params['batch']),'--ubatch-size',str(params['ubatch']),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--split-mode',params['split_mode'],'--fit-target',','.join(str(max(512,round(g['total_mib']*.025))) for g in selected)]
reserve_mode=params.get('gpu_reserve_mode','auto')
margins=[0 if reserve_mode=='none' else params.get('gpu_reserve_mib',{}).get(g['uuid'],512) if reserve_mode=='manual' else max(512,round(g['total_mib']*.025)) for g in selected]
common=['--model',str(model),'--ctx-size',str(params['context']),'--parallel',str(params['slots']),'--batch-size',str(params['batch']),'--ubatch-size',str(params['ubatch']),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--split-mode',params['split_mode'],'--fit-target',','.join(map(str,margins))]
if params['tensor_split']:common+=['--tensor-split',','.join(map(str,params['tensor_split']))]
tool=str(directory/'build/bin/llama-fit-params')
mtp_fit=['--deck-mtp'] if mtp else []
margins=[max(512,round(g['total_mib']*.025)) for g in selected]
def estimate(layers,extra=()):
if cancel():raise InferenceError('Modellstart abgebrochen.')
output=self._fit_command([tool]+mtp_fit+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation,cancel)
@@ -180,7 +181,7 @@ class LlamaWorker:
gpu_fits=all(rows['CUDA'+str(i)]+margins[i]<=g['free_mib'] for i,g in enumerate(selected))
need=max(staging,rows['Host']*1024**2+4*GIB)
host_fits=mem.get('MemAvailable',0)>=need+4*GIB and (headroom is None or headroom>=need)
with self.lock:self.memory_plan=dict(profile_name=profile['name'],estimated=True,gpu_layers=layers,full_gpu=layers==999,context=params['context'],slots=params['slots'],fits=gpu_fits and host_fits,host_required_mib=round(need/1024**2),host_available_mib=round(min(mem.get('MemAvailable',0)-4*GIB,headroom if headroom is not None else mem.get('MemAvailable',0))/1024**2),gpus=[dict(name=g.get('name',g['uuid']),uuid=g['uuid'],free_mib=g['free_mib'],required_mib=rows['CUDA'+str(i)],reserve_mib=margins[i]) for i,g in enumerate(selected)])
with self.lock:self.memory_plan=dict(profile_name=profile['name'],estimated=True,reserve_mode=reserve_mode,gpu_layers=layers,full_gpu=layers==999,context=params['context'],slots=params['slots'],fits=gpu_fits and host_fits,host_required_mib=round(need/1024**2),host_available_mib=round(min(mem.get('MemAvailable',0)-4*GIB,headroom if headroom is not None else mem.get('MemAvailable',0))/1024**2),gpus=[dict(name=g.get('name',g['uuid']),uuid=g['uuid'],free_mib=g['free_mib'],required_mib=rows['CUDA'+str(i)],reserve_mib=margins[i]) for i,g in enumerate(selected)])
return gpu_fits and host_fits,rows
extra=[]
if params.get('gpu_offload')=='full':