Persist library profiles and download history with weight capacity checks

This commit is contained in:
Mikei386
2026-09-28 17:26:21 +02:00
parent f326de18d1
commit 458047a1d3
15 changed files with 360 additions and 180 deletions
+55
View File
@@ -0,0 +1,55 @@
"""Conservative file-weight checks, never a claim that a pipeline can execute."""
from pathlib import Path
import re
GIB=1024**3
def assess(size, hardware, filename=''):
if not isinstance(size,int) or size<1:return dict(label='Größe unbekannt',scope='unknown',gpus=[])
if filename.lower().endswith('.json'):return dict(label='Konfigurationsdatei · kein Modell',scope='configuration',gpus=[])
gpus=[]
for gpu in hardware.get('gpus',[]):
total=gpu.get('total_mib');used=gpu.get('used_mib')
reserve=max(512,(total or 0)*.05)
capacity=(total-reserve)*1024**2 if total is not None else None
free=max(0,total-used-reserve)*1024**2 if total is not None and used is not None else None
gpus.append(dict(name=gpu['name'],uuid=gpu.get('uuid'),capacity_bytes=capacity,free_bytes=free,
weights_fit_total=size<=capacity if capacity is not None else None,
weights_fit_now=size<=free if free is not None else None))
fits=any(x['weights_fit_total'] for x in gpus);now=any(x['weights_fit_now'] for x in gpus)
label='Gewichte passen auf eine GPU; Gesamtbedarf offen' if fits else 'Gewichte benötigen Aufteilung/Auslagerung; Gesamtbedarf offen' if gpus else 'GPU-Speicher nicht ermittelbar'
ram=hardware.get('ram',{});total=ram.get('total_bytes');used=ram.get('used_bytes')
limit=None
try:
raw=Path('/sys/fs/cgroup/memory.max').read_text().strip()
if raw!='max':limit=int(raw)
except (OSError,ValueError):pass
return dict(label=label,scope='weights_only',weight_bytes=size,gpus=gpus,weights_fit_now=now if gpus else None,
ram_total_bytes=total,ram_available_bytes=total-used if total is not None and used is not None else None,
deck_ram_limit_bytes=limit,sampled_at=hardware.get('sampled_at'),
explanation='Untergrenze: nur diese Datei, 5 % GPU-Reserve (mindestens 512 MiB). Kontext/KV-Cache, Aktivierungen, Textencoder und VAE kommen hinzu. Kein Startversprechen; GPUs werden nicht addiert.')
def overview(files,hardware):
# Do not rank a tiny LoRA, VAE or encoder as the base model. Prefer GGUF
# variants when available; otherwise inspect main safetensors weights.
weights=[f for f in files if f['name'].lower().endswith(('.gguf','.safetensors')) and not
re.search(r'(?:vae|text.encoder|tokenizer|mmproj|lora|adapter|clip|safety.checker|mlx)',f['name'],re.I)]
gguf=[f for f in weights if f['name'].lower().endswith('.gguf')]
if gguf:weights=gguf
groups={};variants=[]
for f in weights:
match=re.match(r'(.*)-(\d{5})-of-(\d{5})(\.(?:gguf|safetensors))$',f['name'],re.I)
if match:
prefix,idx,count,suffix=match.groups();key=(prefix,int(count),suffix)
groups.setdefault(key,[]).append((int(idx),f))
else:variants.append(f)
incomplete=False
for (prefix,count,suffix),parts in groups.items():
if {i for i,_ in parts}==set(range(1,count+1)):
variants.append(dict(name=prefix+' (alle '+str(count)+' Teile)',size=sum(f['size'] for _,f in parts)))
else:incomplete=True
if not variants:return dict(min_bytes=None,max_bytes=None,label='Modellgröße nicht eindeutig ermittelbar; Dateiauswahl öffnen',scope='unknown')
smallest=min(variants,key=lambda x:x['size']);largest=max(variants,key=lambda x:x['size'])
return dict(min_bytes=smallest['size'],max_bytes=largest['size'],file_count=len(variants),
**assess(smallest['size'],hardware,smallest['name']),basis=smallest['name'],incomplete_shards=incomplete,
size_note='Hauptgewichte (GGUF bevorzugt), vollständige Shards zusammengefasst; nicht Gesamtpaket. Zusatzmodelle und Laufzeitbedarf fehlen.')