Persist library profiles and download history with weight capacity checks
This commit is contained in:
+55
@@ -0,0 +1,55 @@
|
||||
"""Conservative file-weight checks, never a claim that a pipeline can execute."""
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
GIB=1024**3
|
||||
|
||||
def assess(size, hardware, filename=''):
|
||||
if not isinstance(size,int) or size<1:return dict(label='Größe unbekannt',scope='unknown',gpus=[])
|
||||
if filename.lower().endswith('.json'):return dict(label='Konfigurationsdatei · kein Modell',scope='configuration',gpus=[])
|
||||
gpus=[]
|
||||
for gpu in hardware.get('gpus',[]):
|
||||
total=gpu.get('total_mib');used=gpu.get('used_mib')
|
||||
reserve=max(512,(total or 0)*.05)
|
||||
capacity=(total-reserve)*1024**2 if total is not None else None
|
||||
free=max(0,total-used-reserve)*1024**2 if total is not None and used is not None else None
|
||||
gpus.append(dict(name=gpu['name'],uuid=gpu.get('uuid'),capacity_bytes=capacity,free_bytes=free,
|
||||
weights_fit_total=size<=capacity if capacity is not None else None,
|
||||
weights_fit_now=size<=free if free is not None else None))
|
||||
fits=any(x['weights_fit_total'] for x in gpus);now=any(x['weights_fit_now'] for x in gpus)
|
||||
label='Gewichte passen auf eine GPU; Gesamtbedarf offen' if fits else 'Gewichte benötigen Aufteilung/Auslagerung; Gesamtbedarf offen' if gpus else 'GPU-Speicher nicht ermittelbar'
|
||||
ram=hardware.get('ram',{});total=ram.get('total_bytes');used=ram.get('used_bytes')
|
||||
limit=None
|
||||
try:
|
||||
raw=Path('/sys/fs/cgroup/memory.max').read_text().strip()
|
||||
if raw!='max':limit=int(raw)
|
||||
except (OSError,ValueError):pass
|
||||
return dict(label=label,scope='weights_only',weight_bytes=size,gpus=gpus,weights_fit_now=now if gpus else None,
|
||||
ram_total_bytes=total,ram_available_bytes=total-used if total is not None and used is not None else None,
|
||||
deck_ram_limit_bytes=limit,sampled_at=hardware.get('sampled_at'),
|
||||
explanation='Untergrenze: nur diese Datei, 5 % GPU-Reserve (mindestens 512 MiB). Kontext/KV-Cache, Aktivierungen, Textencoder und VAE kommen hinzu. Kein Startversprechen; GPUs werden nicht addiert.')
|
||||
|
||||
def overview(files,hardware):
|
||||
# Do not rank a tiny LoRA, VAE or encoder as the base model. Prefer GGUF
|
||||
# variants when available; otherwise inspect main safetensors weights.
|
||||
weights=[f for f in files if f['name'].lower().endswith(('.gguf','.safetensors')) and not
|
||||
re.search(r'(?:vae|text.encoder|tokenizer|mmproj|lora|adapter|clip|safety.checker|mlx)',f['name'],re.I)]
|
||||
gguf=[f for f in weights if f['name'].lower().endswith('.gguf')]
|
||||
if gguf:weights=gguf
|
||||
groups={};variants=[]
|
||||
for f in weights:
|
||||
match=re.match(r'(.*)-(\d{5})-of-(\d{5})(\.(?:gguf|safetensors))$',f['name'],re.I)
|
||||
if match:
|
||||
prefix,idx,count,suffix=match.groups();key=(prefix,int(count),suffix)
|
||||
groups.setdefault(key,[]).append((int(idx),f))
|
||||
else:variants.append(f)
|
||||
incomplete=False
|
||||
for (prefix,count,suffix),parts in groups.items():
|
||||
if {i for i,_ in parts}==set(range(1,count+1)):
|
||||
variants.append(dict(name=prefix+' (alle '+str(count)+' Teile)',size=sum(f['size'] for _,f in parts)))
|
||||
else:incomplete=True
|
||||
if not variants:return dict(min_bytes=None,max_bytes=None,label='Modellgröße nicht eindeutig ermittelbar; Dateiauswahl öffnen',scope='unknown')
|
||||
smallest=min(variants,key=lambda x:x['size']);largest=max(variants,key=lambda x:x['size'])
|
||||
return dict(min_bytes=smallest['size'],max_bytes=largest['size'],file_count=len(variants),
|
||||
**assess(smallest['size'],hardware,smallest['name']),basis=smallest['name'],incomplete_shards=incomplete,
|
||||
size_note='Hauptgewichte (GGUF bevorzugt), vollständige Shards zusammengefasst; nicht Gesamtpaket. Zusatzmodelle und Laufzeitbedarf fehlen.')
|
||||
Reference in New Issue
Block a user