Allow readable profile names and configurable KV cache

This commit is contained in:
Mikei386
2026-09-30 09:52:36 +02:00
parent fac4a12d52
commit cbcd5a0131
7 changed files with 32 additions and 14 deletions
+5 -3
View File
@@ -73,16 +73,18 @@ class AutoTests:
def update(self,**kw):
with self.lock:self.job.update(kw);self.persist()
def start(self,data):
if set(data) not in ({'model_id','max_context','mtp'},{'model_id','start_context','max_context','mtp'}):raise ValueError('Modell, Start- und Endkontext sowie MTP auswählen.')
fields=set(data);base={'model_id','max_context','mtp'}
if fields-base-{'start_context','cache_type_k','cache_type_v'} or not base<=fields or (('cache_type_k' in fields)!=('cache_type_v' in fields)):raise ValueError('Modell, Start- und Endkontext, MTP und beide KV-Cache-Typen auswählen.')
start_context=data.get('start_context',2048)
if type(start_context)!=int or start_context not in CONTEXT_STEPS or type(data['max_context'])!=int or data['max_context'] not in CONTEXT_STEPS or start_context>data['max_context'] or type(data['mtp'])!=bool:raise ValueError('Start- und Endkontext müssen gültige Stufen in aufsteigender Reihenfolge sein.')
if data.get('cache_type_k','q4_0') not in ('f16','q8_0','q4_0') or data.get('cache_type_v','q4_0') not in ('f16','q8_0','q4_0'):raise ValueError('KV-Cache: nur f16, q8_0 oder q4_0 unterstützt.')
model=self.worker.catalog.entry(data['model_id'])
if model['kind']!='chat' or not model['file'].endswith('.gguf') or not model['profile_eligible']:raise ValueError('Ein heruntergeladenes Chat-GGUF auswählen.')
devices=probe();primary=next((g for g in devices if '5080' in g['name']),None);secondary=next((g for g in devices if '3060' in g['name']),None)
if not primary or not secondary:raise ValueError('Dieser Auto-Test benötigt RTX 5080 und RTX 3060. Keine stillschweigende andere GPU-Zuordnung.')
with self.lock:
if self.thread and self.thread.is_alive():raise ValueError('Ein Auto-Test läuft bereits.')
self.cancel.clear();self.job=dict(id=uuid.uuid4().hex,state='running',phase='Wartet auf exklusive Modellreservierung',model_id=model['id'],model_file=model['file'],start_context=start_context,max_context=data['max_context'],mtp=data['mtp'],results=[],attempts=0,max_candidates=MAX_CANDIDATES,completed_contexts=[],started_at=time.time(),error=None);self.persist()
self.cancel.clear();self.job=dict(id=uuid.uuid4().hex,state='running',phase='Wartet auf exklusive Modellreservierung',model_id=model['id'],model_file=model['file'],start_context=start_context,max_context=data['max_context'],mtp=data['mtp'],cache_type_k=data.get('cache_type_k','q4_0'),cache_type_v=data.get('cache_type_v','q4_0'),results=[],attempts=0,max_candidates=MAX_CANDIDATES,completed_contexts=[],started_at=time.time(),error=None);self.persist()
self.thread=threading.Thread(target=self.run,args=(primary['uuid'],secondary['uuid']),daemon=True);self.thread.start()
return self.status()
def stop(self):
@@ -120,7 +122,7 @@ class AutoTests:
if self.cancel.is_set():raise InterruptedError()
if self.job['attempts']>=self.job.get('max_candidates',MAX_CANDIDATES):raise TestBudget('Kandidatenlimit erreicht')
if time.time()-self.job['started_at']>MAX_SECONDS:raise TestBudget('Zeitlimit erreicht')
params={**{k:v[2] for k,v in SCHEMAS['chat'].items()},**CHAT_GPU_DEFAULTS,'context':context,'slots':1,'batch':2048,'ubatch':micro,'gpu_devices':devices,'split_mode':split,'tensor_split':ratio,'gpu_offload':offload,'mtp':self.job['mtp']}
params={**{k:v[2] for k,v in SCHEMAS['chat'].items()},**CHAT_GPU_DEFAULTS,'context':context,'slots':1,'batch':2048,'ubatch':micro,'gpu_devices':devices,'split_mode':split,'tensor_split':ratio,'gpu_offload':offload,'mtp':self.job['mtp'],'cache_type_k':self.job.get('cache_type_k','q4_0'),'cache_type_v':self.job.get('cache_type_v','q4_0')}
p=dict(id='auto-'+self.job['id'],revision=self.job['attempts']+1,name='deck-auto-test',kind='chat',model_id=self.job['model_id'],parameters=params)
self.update(attempts=p['revision'],phase=f'{context} Kontext · {tier} · Microbatch {micro}')
row=dict(context=context,tier=tier,parameters=params,success=False,primary_gpu_layers=layer_count)