Files

262 lines
18 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
"""Unprivileged native build manager. No Docker calls, shell commands or host changes."""
import json
import os
from pathlib import Path
import re
import shutil
import shlex
import signal
import subprocess
import threading
import time
import urllib.request
import uuid
ORIGIN='https://github.com/ggml-org/llama.cpp.git'
def github(path):
req=urllib.request.Request('https://api.github.com/repos/ggml-org/llama.cpp/'+path,headers={'Accept':'application/vnd.github+json','User-Agent':'Athena-Deck'})
with urllib.request.urlopen(req,timeout=20) as r:return json.load(r)
def command(args,timeout=10):
return subprocess.run(args,capture_output=True,text=True,timeout=timeout,check=True).stdout.strip()
def prepare_fit_source(directory):
# The reference profiles share a KV pool; upstream fit CLI omits its switch.
path=Path(directory)/'tools/fit-params/fit-params.cpp'
if not path.is_file():raise ValueError('Diese Version enthält kein unterstütztes Fit-Werkzeug.')
source=path.read_text();marker=' params.kv_unified = true; // Athena Deck reference profiles'
if marker in source:return
anchor=' llama_backend_init();'
if source.count(anchor)!=1:raise ValueError('Fit-Adapter passt nicht zu dieser Version; Build abgebrochen.')
path.write_text(source.replace(anchor,marker+'\n'+anchor))
def prepare_mtp_source(directory):
"""Extend the no-allocation fit tool with a shared-weight MTP context."""
path=Path(directory)/'tools/fit-params/fit-params.cpp'
source=path.read_text()
if '// Deck MTP fit v1' in source:
if '#include "speculative.h"' not in source:path.write_text(source.replace('#include "fit.h"','#include "fit.h"\n#include "speculative.h"'))
return
changes={
'#include "fit.h"':'#include "fit.h"\n#include "speculative.h"',
' common_init();':''' // Deck MTP fit v1
bool deck_mtp = false;
for (int i = 1; i < argc; ++i) {
if (std::string(argv[i]) == "--deck-mtp") {
deck_mtp = true;
for (int j = i; j + 1 < argc; ++j) argv[j] = argv[j + 1];
--argc; --i;
}
}
common_init();''',
' auto mparams = common_model_params_to_llama(params);':''' if (deck_mtp) params.speculative.types = {COMMON_SPECULATIVE_TYPE_DRAFT_MTP};
auto mparams = common_model_params_to_llama(params);''',
' auto cparams = common_context_params_to_llama(params);':''' auto cparams = common_context_params_to_llama(params);
auto draft_params = common_base_params_to_speculative(params);
auto draft_mparams = common_model_params_to_llama(draft_params);
auto draft_cparams = common_context_params_to_llama(draft_params);
draft_cparams.ctx_type = LLAMA_CONTEXT_TYPE_MTP;
draft_cparams.n_rs_seq = 0;
const common_fit_extra_model draft_extra = {
params.model.path.c_str(), &draft_mparams, &draft_cparams, true
};''',
' nullptr,':' deck_mtp ? &draft_extra : nullptr,',
' common_fit_print(params.model.path.c_str(), &mparams, &cparams);':''' if (!deck_mtp) {
common_fit_print(params.model.path.c_str(), &mparams, &cparams);
} else {
std::vector<ggml_backend_dev_t> devs, draft_devs;
uint32_t ngl, ctx, expert;
auto main_mem = common_get_device_memory_data(params.model.path.c_str(), &mparams, &cparams, devs, ngl, ctx, expert, GGML_LOG_LEVEL_ERROR);
auto draft_mem = common_get_device_memory_data(params.model.path.c_str(), &draft_mparams, &draft_cparams, draft_devs, ngl, ctx, expert, GGML_LOG_LEVEL_ERROR);
if (main_mem.size() != draft_mem.size() || devs != draft_devs) return 2;
for (size_t i = 0; i < main_mem.size(); ++i) {
const auto & a = main_mem[i]; const auto & b = draft_mem[i];
auto mib = [](size_t x) { return (x + 1048575) / 1048576; };
printf("%s %zu %zu %zu\\n", i < devs.size() ? ggml_backend_dev_name(devs[i]) : "Host",
mib(a.model), mib(a.context + b.context), mib(a.compute + b.compute));
}
}'''
}
for old,new in changes.items():
if source.count(old)!=1:raise ValueError('MTP-Fit-Adapter passt nicht zu dieser Version; Build abgebrochen.')
source=source.replace(old,new)
path.write_text(source)
class Runtime:
def __init__(self,root):
self.root=Path(root);self.lock=threading.RLock();self.process=None;self.cancelled=threading.Event();self.busy=False
self.state={'job':None,'builds':[],'active':None,'previous':None}
if (self.root/'state.json').exists():
self.state=json.loads((self.root/'state.json').read_text())
if (self.state.get('job') or {}).get('state')=='running':
self.state['job'].update(state='interrupted',phase='Deck wurde neu gestartet; Build erneut starten.')
def save(self):
self.root.mkdir(parents=True,exist_ok=True,mode=0o700)
p=self.root/'state.tmp';p.write_text(json.dumps(self.state));p.replace(self.root/'state.json')
def prerequisites(self):
result={x:shutil.which(x) for x in ('git','cmake','g++','nvcc')};gpus=[]
try:
for line in command(['nvidia-smi','--query-gpu=index,uuid,name,compute_cap,memory.total,memory.free','--format=csv,noheader,nounits']).splitlines():
idx,gpu_uuid,name,cap,total,free=[x.strip() for x in line.split(',')]
gpus.append(dict(index=int(idx),uuid=gpu_uuid,name=name,architecture=cap.replace('.',''),total_mib=float(total),free_mib=float(free)))
except (OSError,ValueError,subprocess.SubprocessError):pass
result.update(gpus=gpus,cpu_ready=all(result[x] for x in ('git','cmake','g++')),cuda_ready=all(result[x] for x in ('git','cmake','g++','nvcc')) and bool(gpus))
return result
def status(self):
with self.lock:return json.loads(json.dumps(self.state))
def releases(self):
rows=github('releases?per_page=10')
latest=github('releases/latest')
rows=[latest]+[r for r in rows if r['tag_name']!=latest['tag_name']]
return {'releases':[dict(tag=r['tag_name'],name=r['name'],published=r['published_at'],notes=r.get('body',''),url=r['html_url'],prerelease=r['prerelease']) for r in rows if not r['draft']]}
def run(self,args,cwd,phase,timeout=7200):
with self.lock:
if self.cancelled.is_set():raise InterruptedError()
self.state['job']['phase']=phase;self.save()
log=self.root/'build.log'
stream=log.open('ab')
try:
self.process=subprocess.Popen(args,cwd=cwd,stdout=stream,stderr=subprocess.STDOUT,start_new_session=True)
except Exception:
stream.close();raise
try:
code=self.process.wait(timeout=timeout)
if self.cancelled.is_set():raise InterruptedError()
if code:raise ValueError('Build-Schritt fehlgeschlagen: '+phase)
except subprocess.TimeoutExpired:
self.stop();raise ValueError('Build-Zeitlimit erreicht.') from None
finally:
stream.close()
with self.lock:self.process=None
def start(self,revision,backend='CUDA',jobs=1):
if not isinstance(revision,str) or not re.fullmatch(r'(?:b[0-9]{3,8}|v[0-9]{1,4}\.[0-9]{1,4}\.[0-9]{1,4}|[a-f0-9]{40})',revision):raise ValueError('Offizielles v-Release, b-Nightly oder vollständigen Commit-SHA wählen.')
if backend not in ('CUDA','CPU') or type(jobs)!=int or not 1<=jobs<=2:raise ValueError('CPU/CUDA und 1–2 Build-Jobs unterstützt.')
with self.lock:
if self.busy:raise ValueError('Ein Build läuft bereits.')
p=self.prerequisites()
if not p['cuda_ready' if backend=='CUDA' else 'cpu_ready']:raise ValueError('Build-Werkzeuge fehlen. Voraussetzungen prüfen.')
self.root.mkdir(parents=True,exist_ok=True)
if shutil.disk_usage(self.root).free<15*1024**3:raise ValueError('Mindestens 15 GiB freier Speicher erforderlich.')
ident=uuid.uuid4().hex
self.state['job']=dict(id=ident,state='running',phase='Vorbereitung',revision=revision,backend=backend,started=time.time(),error=None)
self.cancelled.clear();self.busy=True;self.save()
threading.Thread(target=self.build,args=(ident,revision,backend,jobs,p),daemon=True).start()
return self.status()
def build(self,ident,revision,backend,jobs,prereq):
directory=self.root/ident
try:
directory.mkdir();(self.root/'build.log').write_text('')
self.run(['git','init',str(directory)],self.root,'Quellverzeichnis anlegen',30)
self.run(['git','-C',str(directory),'fetch','--depth','1',ORIGIN,revision],self.root,'Offizielle Quellen herunterladen',300)
self.run(['git','checkout','--detach','FETCH_HEAD'],directory,'Feste Version auswählen',30)
commit=command(['git','-C',str(directory),'rev-parse','HEAD'])
prepare_fit_source(directory)
prepare_mtp_source(directory)
arch=sorted(set(g['architecture'] for g in prereq['gpus']))
args=['cmake','-S','.', '-B','build','-DCMAKE_BUILD_TYPE=Release','-DGGML_NATIVE=OFF','-DLLAMA_BUILD_TESTS=OFF','-DLLAMA_BUILD_EXAMPLES=OFF','-DLLAMA_BUILD_SERVER=ON','-DGGML_CUDA='+('ON' if backend=='CUDA' else 'OFF')]
args+=['-DLLAMA_BUILD_IS_DEV='+('OFF' if revision.startswith('v') else 'ON')]
if backend=='CUDA':args+=['-DCMAKE_CUDA_ARCHITECTURES='+';'.join(arch)]
self.run(args,directory,'Build konfigurieren',300)
self.run(['cmake','--build','build','--target','llama-server','llama-fit-params','-j',str(jobs)],directory,'llama.cpp-Werkzeuge kompilieren')
(directory/'build/bin/deck-mtp-fit-v1').write_text('shared weights + main and MTP contexts, f16 draft KV\n')
binary=directory/'build/bin/llama-server'
self.run([str(binary),'--version'],directory,'Binärdatei prüfen',30)
help_text=command([str(binary),'--help'],30)
item=dict(id=ident,revision=revision,commit=commit,backend=backend,architectures=arch if backend=='CUDA' else [],created=time.time(),fit_supported='--fit ' in help_text,fit_tool=(directory/'build/bin/llama-fit-params').exists(),fit_adapter='shared-kv-mtp-v1')
with self.lock:
self.state['builds'].append(item);self.state['job'].update(state='complete',phase='Build geprüft; kann als Standard ausgewählt werden.');self.save()
except Exception as exc:
with self.lock:
self.state['job'].update(state='cancelled' if isinstance(exc,InterruptedError) else 'failed',error=str(exc) if isinstance(exc,ValueError) else 'Build unterbrochen oder Werkzeug nicht erreichbar.');self.save()
finally:
with self.lock:self.busy=False
def stop(self):
with self.lock:
self.cancelled.set()
if self.process and self.process.poll() is None:
try:os.killpg(self.process.pid,signal.SIGTERM)
except ProcessLookupError:pass
try:self.process.wait(timeout=3)
except subprocess.TimeoutExpired:
try:os.killpg(self.process.pid,signal.SIGKILL)
except ProcessLookupError:pass
return {'cancellation_requested':True}
def activate(self,build_id):
with self.lock:
if not any(b['id']==build_id for b in self.state['builds']):raise ValueError('Kein erfolgreich geprüfter Build.')
if self.state['active']!=build_id:
self.state['previous']=self.state['active'];self.state['active']=build_id;self.save()
return self.status()
def rollback(self):
with self.lock:
if not self.state['previous']:raise ValueError('Kein vorheriger Build vorhanden.')
return self.activate(self.state['previous'])
def log(self):
p=self.root/'build.log'
if not p.exists():return {'text':''}
with p.open('rb') as f:
f.seek(max(0,p.stat().st_size-24000));return {'text':f.read().decode(errors='replace')}
def references(self):
base=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))
specs=[('fast','Qwen3.8-27B-IQ4-MIX.gguf',76800,1,64),('medium','qwen3.8-27b-IQ4_XS-pure.gguf',160000,2,256),('large','qwen3.8-27b-IQ4_XS-pure.gguf',192000,1,256),('ultra','qwen3.8-27b-IQ4_XS-pure.gguf',262144,1,128),('uncensored','Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf',80000,1,256)]
references=[]
profiles=getattr(self,'profiles',None);catalog=getattr(self,'catalog',None)
if profiles is not None and catalog is not None:
with profiles.lock: saved=list(profiles.rows)
for profile in saved:
if profile.get('kind')!='chat':continue
try: model=catalog.entry(profile['model_id'])
except (KeyError,ValueError,OSError):continue
if not model['file'].lower().endswith('.gguf') or not model['profile_eligible']:continue
params=profile['parameters']
references.append(dict(id='deck:'+profile['id'],source='deck',name=profile['name'],file=model['file'],model_id=model['id'],context=params['context'],slots=params['slots'],batch=params['batch'],ubatch=params['ubatch'],cache='q4_0',available=True))
references.extend(dict(id=i,source='router',name=i,file=f,context=c,slots=n,batch=2048,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs)
return {'profiles':references}
def fit(self,profile,context,slots):
if type(context)!=int or not 512<=context<=2097152 or type(slots)!=int or not 1<=slots<=16:raise ValueError('Kontext oder Slots außerhalb des erlaubten Bereichs.')
ref=next((p for p in self.references()['profiles'] if p['id']==profile),None)
if not ref or not ref['available']:raise ValueError('Modelldatei für die Einpassung nicht verfügbar.')
with self.lock:
build=next((b for b in self.state['builds'] if b['id']==self.state['active']),None)
if not build or not build.get('fit_tool'):raise ValueError('Zuerst einen CUDA-Build mit Fit-Werkzeug erstellen und auswählen.')
if build['backend']!='CUDA':raise ValueError('Für GPU-Einpassung einen CUDA-Build auswählen.')
if self.busy:raise ValueError('Bitte Ende des Builds abwarten.')
if getattr(self,'fitting',False):raise ValueError('Eine Einpassung läuft bereits.')
self.fitting=True
try:
binary=self.root/build['id']/'build/bin/llama-fit-params'
model=(self.catalog.root/ref['model_id']/('model'+Path(ref['file']).suffix)) if ref.get('source')=='deck' else Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file']
args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k',ref.get('cache','q4_0'),'--cache-type-v',ref.get('cache','q4_0'),'--flash-attn','on','--batch-size',str(ref.get('batch',2048)),'--ubatch-size',str(ref['ubatch'])]
snapshot=self.prerequisites()['gpus']
usable=[g for g in snapshot if g['free_mib']>=max(1024,g['total_mib']*.10)]
if not usable:raise ValueError('GPUs derzeit belegt. Keine sichere Auto-Prognose; produktive Dienste bleiben unverändert.')
margins=[max(512,round(g['total_mib']*.05)) for g in usable]
args+=['--fit-target',','.join(map(str,margins))]
env=dict(os.environ,CUDA_VISIBLE_DEVICES=','.join(g['uuid'] for g in usable))
result=subprocess.run(args,capture_output=True,text=True,timeout=120,env=env)
text=result.stdout.strip()
fitted={};memory=[]
if result.returncode==0:
values=shlex.split(text)
for flag,key in [('-c','context'),('-ngl','gpu_layers'),('-ts','tensor_split')]:
if flag in values and values.index(flag)+1<len(values):fitted[key]=values[values.index(flag)+1]
if fitted.get('context')!=str(context):raise ValueError('Fit-Werkzeug hat den gewünschten Kontext nicht beibehalten; Ergebnis wird nicht übernommen.')
if len(values)%2==0 and all(values[i] in ('-c','-ngl','-ts') for i in range(0,len(values),2)):
probe=subprocess.run(args+['--fit-print','on']+values,capture_output=True,text=True,timeout=120,env=env)
if probe.returncode==0:
for line in probe.stdout.splitlines():
match=re.fullmatch(r'(\S+) +(\d+) +(\d+) +(\d+) *',line)
if match:memory.append(dict(device=match[1],model_mib=int(match[2]),context_mib=int(match[3]),compute_mib=int(match[4])))
ram_limit=None
try:
raw=Path('/sys/fs/cgroup/memory.max').read_text().strip()
if raw!='max':ram_limit=int(raw)//(1024*1024)
except (OSError,ValueError):pass
host=next((m for m in memory if m['device']=='Host'),None)
within_limit=None if host is None or ram_limit is None else sum(host[k] for k in ('model_mib','context_mib','compute_mib'))<=ram_limit
return dict(success=result.returncode==0,ram_limit_mib=ram_limit,within_ram_limit=within_limit,memory=memory,fitted=fitted,profile=profile,context=context,slots=slots,arguments=text[-16000:],details=result.stderr[-16000:],estimated=True,scope='text_model_without_vision_or_mtp',model_loaded=False,gpus=snapshot,used_gpu_uuids=[g['uuid'] for g in usable],reserve_mib=margins,excluded_gpus=[g['name'] for g in snapshot if g not in usable])
finally:
with self.lock:self.fitting=False