"""Public Hub catalogue and bounded, serial downloads; independent of Docker/systemd.""" import hashlib import json import os from pathlib import Path, PurePosixPath import re import shutil import threading import time import uuid import urllib.parse import urllib.request KINDS={'chat':'text-generation','image':'text-to-image','audio':'text-to-speech','video':'text-to-video'} def safe_url(url): p=urllib.parse.urlsplit(url) if p.scheme!='https' or p.username or p.password or p.port not in (None,443) or not any(p.hostname==d or (p.hostname or '').endswith('.'+d) for d in ('huggingface.co','hf.co','xethub.hf.co')): raise ValueError('Downloadziel außerhalb der erlaubten Hugging-Face-Domains.') return url class Redirect(urllib.request.HTTPRedirectHandler): def redirect_request(self, req, fp, code, msg, headers, newurl): return super().redirect_request(req,fp,code,msg,headers,safe_url(newurl)) def remote(url): return urllib.request.build_opener(Redirect()).open(urllib.request.Request(safe_url(url),headers={'User-Agent':'Athena-Deck/0.5'}),timeout=20) def metadata(path): with remote('https://huggingface.co'+path) as r: raw=r.read(8*1024*1024+1) if len(raw)>8*1024*1024:raise ValueError('Metadaten zu groß; Repository wird noch nicht unterstützt.') return json.loads(raw) def repo_id(value): if not isinstance(value,str) or not re.fullmatch(r'[A-Za-z0-9_-][A-Za-z0-9_.-]{0,95}/[A-Za-z0-9_-][A-Za-z0-9_.-]{0,95}',value):raise ValueError('Ungültige Repository-ID.') return value def file_role(filename): """One classification shared by library UI and profile validation.""" name=filename.lower();parts=PurePosixPath(name).parts if any(p in ('text_encoder','text_encoders') for p in parts) or re.search(r'(?:^|[/_-])(?:text.encoder|qwen3vl|t5xxl|clip_l|clip_g)(?:[/_.-]|$)',name): role,label='text_encoder','Textencoder' elif 'vae' in parts or re.search(r'(?:^|[/_-])vae(?:[/_.-]|$)',name): role,label='vae','VAE' elif re.search(r'(?:^|[/_-])(?:mmproj|lora|adapter)(?:[/_.-]|$)',name): role,label='auxiliary','Zusatzkomponente' elif name.endswith(('.gguf','.safetensors')): role,label='model','Modelldatei' else: role,label='configuration','Konfiguration' return dict(role=role,role_label=label,profile_eligible=role=='model') def is_derived_model(model): """Use publisher metadata, never infer lineage from a repository name.""" card=model.get('cardData') or {} if isinstance(card,dict) and (card.get('base_model') or card.get('base_model_relation')):return True tags=model.get('tags') or [] return any(isinstance(tag,str) and (tag.startswith('base_model:') or tag.lower() in {'lora','peft','adapter','merge','mergekit','finetune','fine-tuned','quantized','gguf','gptq','awq'}) for tag in tags) class Catalog: def __init__(self,root): self.root=Path(root);self.lock=threading.RLock();self.job=None;self.cancel=threading.Event() self.cache={};self.cache_lock=threading.Lock();self.history=[] path=self.root/'downloads.json' if path.exists(): self.history=json.loads(path.read_text()) for job in self.history: if job['state']=='downloading':job.update(state='interrupted',error='Deck wurde neu gestartet. Datei erneut auswählen und herunterladen.') else: for entry in self.root.glob('*/entry.json'): try: x=json.loads(entry.read_text()) self.history.append(dict(id='import-'+entry.parent.name,repo=x['repo'],file=x['file'],kind=x['kind'],state='complete',bytes=x['size'],total=x['size'],created_at=x.get('downloaded_at'),error=None)) except (OSError,ValueError,KeyError):pass if self.history:self._save_history() def _save_history(self): self.root.mkdir(parents=True,exist_ok=True,mode=0o700) temp=self.root/'downloads.tmp';temp.write_text(json.dumps(self.history));temp.replace(self.root/'downloads.json') def dismiss(self,job_id): with self.lock: job=next((x for x in self.history if x['id']==job_id),None) if not job:raise ValueError('Download nicht gefunden.') if job['state']=='downloading':raise ValueError('Laufenden Download zuerst abbrechen.') job['dismissed']=True if self.job and self.job.get('id')==job_id:self.job=None self._save_history() return {'dismissed':True,'model_deleted':False} def entry(self,ident): if not isinstance(ident,str) or not re.fullmatch('[a-f0-9]{64}',ident):raise ValueError('Ungültige Modelldatei-ID.') entry=next((x for x in self.status()['entries'] if x['id']==ident),None) if not entry:raise ValueError('Modelldatei nicht in der Bibliothek.') path=self.root/ident/('model'+PurePosixPath(entry['file']).suffix) if not path.is_file() or path.stat().st_size!=entry['size']:raise ValueError('Modelldatei fehlt oder ist unvollständig.') return entry def search(self,q,kind,sort="downloads",base_only=False): if kind not in KINDS or not isinstance(q,str) or len(q)>120:raise ValueError('Ungültige Suche.') if sort not in ('downloads','name','newest'):raise ValueError('Ungültige Sortierung.') if not isinstance(base_only,bool):raise ValueError('Ungültiger Basismodellfilter.') query=dict(search=q,filter=KINDS[kind],limit=100 if base_only else 20,sort='createdAt' if sort=='newest' else 'downloads',direction=-1) if base_only:query['expand']=['downloads','createdAt','gated','cardData','tags'] rows=metadata('/api/models?'+urllib.parse.urlencode(query,doseq=True)) scanned=len(rows) if base_only:rows=[x for x in rows if not is_derived_model(x)] return {'models':[dict(repo=x['id'],downloads=x.get('downloads'),gated=x.get('gated',False),created_at=x.get('createdAt')) for x in rows[:20]],'base_only':base_only,'scanned':scanned} def files(self,repo): repo=repo_id(repo) with self.cache_lock: cached=self.cache.get(repo) if cached and time.monotonic()-cached[0]<300:return cached[1] data=metadata('/api/models/'+repo+'?blobs=true') revision=data.get('sha','') if not re.fullmatch('[a-f0-9]{40}',revision):raise ValueError('Keine feste Repository-Version verfügbar.') files=[] for f in data.get('siblings',[]): name=f['rfilename'];p=PurePosixPath(name) if p.is_absolute() or '..' in p.parts or p.suffix.lower() not in ('.gguf','.safetensors','.json'):continue size=f.get('size',f.get('lfs',{}).get('size')) if not isinstance(size,int) or size<1:continue files.append(dict(name=name,size=size,sha256=f.get('lfs',{}).get('sha256'))) result=dict(repo=repo,revision=revision,files=files,gated=data.get('gated',False),license=(data.get('cardData') or {}).get('license'),url='https://huggingface.co/'+repo) with self.cache_lock: if len(self.cache)>=64:self.cache.pop(next(iter(self.cache))) self.cache[repo]=(time.monotonic(),result) return result def status(self): with self.lock: entries=[] if self.root.exists(): for p in self.root.glob('*/entry.json'): try: item=json.loads(p.read_text());item['id']=p.parent.name;item.update(file_role(item['file']));entries.append(item) except (OSError,ValueError):pass return dict(entries=entries,downloads=[dict(x) for x in reversed(self.history) if not x.get("dismissed")],job=dict(self.job) if self.job else None,free_bytes=shutil.disk_usage(self.root if self.root.exists() else self.root.parent).free) def start(self,repo,filename,revision,kind): if kind not in KINDS:raise ValueError('Ungültiger Bereich.') with self.lock: if self.job and self.job['state']=='downloading':raise ValueError('Ein Download läuft bereits.') with self.cache_lock:self.cache.pop(repo,None) data=self.files(repo) if data['gated']:raise ValueError('Zugangsbeschränkte Modelle werden noch nicht unterstützt.') if revision!=data['revision']:raise ValueError('Repository wurde geändert. Dateiliste neu laden.') item=next((x for x in data['files'] if x['name']==filename),None) if not item:raise ValueError('Datei nicht verfügbar.') self.root.mkdir(parents=True,exist_ok=True,mode=0o700) if item['size']>shutil.disk_usage(self.root).free-10*1024**3:raise ValueError('Nicht genug Platz mit 10 GiB freier Reserve.') ident=hashlib.sha256((repo+revision+filename).encode()).hexdigest() target=self.root/ident if (target/'entry.json').exists():raise ValueError('Datei bereits in der Bibliothek.') target.mkdir(exist_ok=True,mode=0o700) self.cancel.clear() self.job=dict(id=uuid.uuid4().hex,entry_id=ident,state='downloading',repo=repo,file=filename,kind=kind,revision=revision,created_at=time.time(),bytes=0,total=item['size'],error=None) self.history.append(self.job);self._save_history() threading.Thread(target=self._download,args=(data,item,target,kind),daemon=True).start() return dict(self.job) def stop(self): self.cancel.set() return {'cancellation_requested':True} def _download(self,data,item,target,kind): partial=target/'download.part';dest=target/('model'+PurePosixPath(item['name']).suffix) try: digest=hashlib.sha256();received=0 url='https://huggingface.co/'+data['repo']+'/resolve/'+data['revision']+'/'+urllib.parse.quote(item['name'],safe='/') with remote(url) as r, partial.open('wb') as out: while True: if self.cancel.is_set():raise InterruptedError() chunk=r.read(1024*1024) if not chunk:break received+=len(chunk) if received>item['size'] or shutil.disk_usage(target).free<10*1024**3:raise ValueError('Größe oder Speicherreserve überschritten.') out.write(chunk);digest.update(chunk) with self.lock:self.job['bytes']=received if received!=item['size']:raise ValueError('Unvollständiger Download.') if item['sha256'] and digest.hexdigest()!=item['sha256']:raise ValueError('SHA-256-Prüfung fehlgeschlagen.') partial.replace(dest) entry=dict(repo=data['repo'],revision=data['revision'],file=item['name'],size=received,sha256=digest.hexdigest(),upstream_hash_verified=bool(item['sha256']),kind=kind,downloaded_at=time.time(),state='downloaded',runtime_ready=False) temp=target/'entry.tmp';temp.write_text(json.dumps(entry));temp.replace(target/'entry.json') with self.lock:self.job['state']='complete' except Exception as exc: with self.lock: self.job['state']='cancelled' if isinstance(exc,InterruptedError) else 'failed' self.job['error']=str(exc) if isinstance(exc,ValueError) else ('Abgebrochen.' if isinstance(exc,InterruptedError) else 'Download fehlgeschlagen; Verbindung oder Anbieter prüfen.') partial.unlink(missing_ok=True) finally: with self.lock:self._save_history()