Avoid unnecessary CPU offload and expose full GPU profile mode

This commit is contained in:
Mikei386
2026-09-28 21:18:54 +02:00
parent 13ed059aaf
commit f99cd26b1d
8 changed files with 39 additions and 11 deletions
+7 -4
View File
@@ -145,11 +145,11 @@ class LlamaWorker:
# Never inherit LLAMA_ARG_* or user HF credentials into the model server.
env={k:v for k,v in os.environ.items() if not k.startswith(('LLAMA_','HF_','DECK_'))}
env.update(CUDA_VISIBLE_DEVICES=','.join(g['uuid'] for g in selected),HF_HUB_OFFLINE='1',OMP_NUM_THREADS=str(params['threads']))
common=['--model',str(model),'--ctx-size',str(params['context']),'--parallel',str(params['slots']),'--batch-size',str(params['batch']),'--ubatch-size',str(params['ubatch']),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--split-mode',params['split_mode'],'--fit-target',','.join(str(max(1024,round(g['total_mib']*.05))) for g in selected)]
common=['--model',str(model),'--ctx-size',str(params['context']),'--parallel',str(params['slots']),'--batch-size',str(params['batch']),'--ubatch-size',str(params['ubatch']),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--split-mode',params['split_mode'],'--fit-target',','.join(str(max(512,round(g['total_mib']*.025))) for g in selected)]
if params['tensor_split']:common+=['--tensor-split',','.join(map(str,params['tensor_split']))]
tool=str(directory/'build/bin/llama-fit-params')
mtp_fit=['--deck-mtp'] if mtp else []
margins=[max(1024,round(g['total_mib']*.05)) for g in selected]
margins=[max(512,round(g['total_mib']*.025)) for g in selected]
def estimate(layers,extra=()):
if cancel():raise InferenceError('Modellstart abgebrochen.')
output=self._fit_command([tool]+mtp_fit+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation,cancel)
@@ -161,10 +161,13 @@ class LlamaWorker:
gpu_fits=all(rows['CUDA'+str(i)]+margins[i]<=g['free_mib'] for i,g in enumerate(selected))
need=max(staging,rows['Host']*1024**2+4*GIB)
host_fits=mem.get('MemAvailable',0)>=need+4*GIB and (headroom is None or headroom>=need)
with self.lock:self.memory_plan=dict(profile_name=profile['name'],estimated=True,gpu_layers=layers,context=params['context'],slots=params['slots'],fits=gpu_fits and host_fits,host_required_mib=round(need/1024**2),host_available_mib=round(min(mem.get('MemAvailable',0)-4*GIB,headroom if headroom is not None else mem.get('MemAvailable',0))/1024**2),gpus=[dict(name=g.get('name',g['uuid']),uuid=g['uuid'],free_mib=g['free_mib'],required_mib=rows['CUDA'+str(i)],reserve_mib=margins[i]) for i,g in enumerate(selected)])
with self.lock:self.memory_plan=dict(profile_name=profile['name'],estimated=True,gpu_layers=layers,full_gpu=layers==999,context=params['context'],slots=params['slots'],fits=gpu_fits and host_fits,host_required_mib=round(need/1024**2),host_available_mib=round(min(mem.get('MemAvailable',0)-4*GIB,headroom if headroom is not None else mem.get('MemAvailable',0))/1024**2),gpus=[dict(name=g.get('name',g['uuid']),uuid=g['uuid'],free_mib=g['free_mib'],required_mib=rows['CUDA'+str(i)],reserve_mib=margins[i]) for i,g in enumerate(selected)])
return gpu_fits and host_fits,rows
extra=[]
if params['tensor_split']:
if params.get('gpu_offload')=='full':
layers=999;fits,memory=estimate(layers)
if not fits:raise InferenceError('Vollständige GPU-Ausführung passt einschließlich Reserve nicht. Kontext reduzieren oder automatische CPU-Auslagerung wählen.')
elif params['tensor_split']:
# Upstream automatic fitting refuses user tensor_split. Predict the
# fixed split instead; bounded layer search preserves ratio/context.
layers=999;fits,memory=estimate(layers)