262 lines
18 KiB
Python
262 lines
18 KiB
Python
"""Unprivileged native build manager. No Docker calls, shell commands or host changes."""
|
||
import json
|
||
import os
|
||
from pathlib import Path
|
||
import re
|
||
import shutil
|
||
import shlex
|
||
import signal
|
||
import subprocess
|
||
import threading
|
||
import time
|
||
import urllib.request
|
||
import uuid
|
||
|
||
ORIGIN='https://github.com/ggml-org/llama.cpp.git'
|
||
|
||
def github(path):
|
||
req=urllib.request.Request('https://api.github.com/repos/ggml-org/llama.cpp/'+path,headers={'Accept':'application/vnd.github+json','User-Agent':'Athena-Deck'})
|
||
with urllib.request.urlopen(req,timeout=20) as r:return json.load(r)
|
||
|
||
def command(args,timeout=10):
|
||
return subprocess.run(args,capture_output=True,text=True,timeout=timeout,check=True).stdout.strip()
|
||
|
||
def prepare_fit_source(directory):
|
||
# The reference profiles share a KV pool; upstream fit CLI omits its switch.
|
||
path=Path(directory)/'tools/fit-params/fit-params.cpp'
|
||
if not path.is_file():raise ValueError('Diese Version enthält kein unterstütztes Fit-Werkzeug.')
|
||
source=path.read_text();marker=' params.kv_unified = true; // Athena Deck reference profiles'
|
||
if marker in source:return
|
||
anchor=' llama_backend_init();'
|
||
if source.count(anchor)!=1:raise ValueError('Fit-Adapter passt nicht zu dieser Version; Build abgebrochen.')
|
||
path.write_text(source.replace(anchor,marker+'\n'+anchor))
|
||
|
||
def prepare_mtp_source(directory):
|
||
"""Extend the no-allocation fit tool with a shared-weight MTP context."""
|
||
path=Path(directory)/'tools/fit-params/fit-params.cpp'
|
||
source=path.read_text()
|
||
if '// Deck MTP fit v1' in source:
|
||
if '#include "speculative.h"' not in source:path.write_text(source.replace('#include "fit.h"','#include "fit.h"\n#include "speculative.h"'))
|
||
return
|
||
changes={
|
||
'#include "fit.h"':'#include "fit.h"\n#include "speculative.h"',
|
||
' common_init();':''' // Deck MTP fit v1
|
||
bool deck_mtp = false;
|
||
for (int i = 1; i < argc; ++i) {
|
||
if (std::string(argv[i]) == "--deck-mtp") {
|
||
deck_mtp = true;
|
||
for (int j = i; j + 1 < argc; ++j) argv[j] = argv[j + 1];
|
||
--argc; --i;
|
||
}
|
||
}
|
||
common_init();''',
|
||
' auto mparams = common_model_params_to_llama(params);':''' if (deck_mtp) params.speculative.types = {COMMON_SPECULATIVE_TYPE_DRAFT_MTP};
|
||
auto mparams = common_model_params_to_llama(params);''',
|
||
' auto cparams = common_context_params_to_llama(params);':''' auto cparams = common_context_params_to_llama(params);
|
||
auto draft_params = common_base_params_to_speculative(params);
|
||
auto draft_mparams = common_model_params_to_llama(draft_params);
|
||
auto draft_cparams = common_context_params_to_llama(draft_params);
|
||
draft_cparams.ctx_type = LLAMA_CONTEXT_TYPE_MTP;
|
||
draft_cparams.n_rs_seq = 0;
|
||
const common_fit_extra_model draft_extra = {
|
||
params.model.path.c_str(), &draft_mparams, &draft_cparams, true
|
||
};''',
|
||
' nullptr,':' deck_mtp ? &draft_extra : nullptr,',
|
||
' common_fit_print(params.model.path.c_str(), &mparams, &cparams);':''' if (!deck_mtp) {
|
||
common_fit_print(params.model.path.c_str(), &mparams, &cparams);
|
||
} else {
|
||
std::vector<ggml_backend_dev_t> devs, draft_devs;
|
||
uint32_t ngl, ctx, expert;
|
||
auto main_mem = common_get_device_memory_data(params.model.path.c_str(), &mparams, &cparams, devs, ngl, ctx, expert, GGML_LOG_LEVEL_ERROR);
|
||
auto draft_mem = common_get_device_memory_data(params.model.path.c_str(), &draft_mparams, &draft_cparams, draft_devs, ngl, ctx, expert, GGML_LOG_LEVEL_ERROR);
|
||
if (main_mem.size() != draft_mem.size() || devs != draft_devs) return 2;
|
||
for (size_t i = 0; i < main_mem.size(); ++i) {
|
||
const auto & a = main_mem[i]; const auto & b = draft_mem[i];
|
||
auto mib = [](size_t x) { return (x + 1048575) / 1048576; };
|
||
printf("%s %zu %zu %zu\\n", i < devs.size() ? ggml_backend_dev_name(devs[i]) : "Host",
|
||
mib(a.model), mib(a.context + b.context), mib(a.compute + b.compute));
|
||
}
|
||
}'''
|
||
}
|
||
for old,new in changes.items():
|
||
if source.count(old)!=1:raise ValueError('MTP-Fit-Adapter passt nicht zu dieser Version; Build abgebrochen.')
|
||
source=source.replace(old,new)
|
||
path.write_text(source)
|
||
|
||
class Runtime:
|
||
def __init__(self,root):
|
||
self.root=Path(root);self.lock=threading.RLock();self.process=None;self.cancelled=threading.Event();self.busy=False
|
||
self.state={'job':None,'builds':[],'active':None,'previous':None}
|
||
if (self.root/'state.json').exists():
|
||
self.state=json.loads((self.root/'state.json').read_text())
|
||
if (self.state.get('job') or {}).get('state')=='running':
|
||
self.state['job'].update(state='interrupted',phase='Deck wurde neu gestartet; Build erneut starten.')
|
||
def save(self):
|
||
self.root.mkdir(parents=True,exist_ok=True,mode=0o700)
|
||
p=self.root/'state.tmp';p.write_text(json.dumps(self.state));p.replace(self.root/'state.json')
|
||
def prerequisites(self):
|
||
result={x:shutil.which(x) for x in ('git','cmake','g++','nvcc')};gpus=[]
|
||
try:
|
||
for line in command(['nvidia-smi','--query-gpu=index,uuid,name,compute_cap,memory.total,memory.free','--format=csv,noheader,nounits']).splitlines():
|
||
idx,gpu_uuid,name,cap,total,free=[x.strip() for x in line.split(',')]
|
||
gpus.append(dict(index=int(idx),uuid=gpu_uuid,name=name,architecture=cap.replace('.',''),total_mib=float(total),free_mib=float(free)))
|
||
except (OSError,ValueError,subprocess.SubprocessError):pass
|
||
result.update(gpus=gpus,cpu_ready=all(result[x] for x in ('git','cmake','g++')),cuda_ready=all(result[x] for x in ('git','cmake','g++','nvcc')) and bool(gpus))
|
||
return result
|
||
def status(self):
|
||
with self.lock:return json.loads(json.dumps(self.state))
|
||
def releases(self):
|
||
rows=github('releases?per_page=10')
|
||
latest=github('releases/latest')
|
||
rows=[latest]+[r for r in rows if r['tag_name']!=latest['tag_name']]
|
||
return {'releases':[dict(tag=r['tag_name'],name=r['name'],published=r['published_at'],notes=r.get('body',''),url=r['html_url'],prerelease=r['prerelease']) for r in rows if not r['draft']]}
|
||
def run(self,args,cwd,phase,timeout=7200):
|
||
with self.lock:
|
||
if self.cancelled.is_set():raise InterruptedError()
|
||
self.state['job']['phase']=phase;self.save()
|
||
log=self.root/'build.log'
|
||
stream=log.open('ab')
|
||
try:
|
||
self.process=subprocess.Popen(args,cwd=cwd,stdout=stream,stderr=subprocess.STDOUT,start_new_session=True)
|
||
except Exception:
|
||
stream.close();raise
|
||
try:
|
||
code=self.process.wait(timeout=timeout)
|
||
if self.cancelled.is_set():raise InterruptedError()
|
||
if code:raise ValueError('Build-Schritt fehlgeschlagen: '+phase)
|
||
except subprocess.TimeoutExpired:
|
||
self.stop();raise ValueError('Build-Zeitlimit erreicht.') from None
|
||
finally:
|
||
stream.close()
|
||
with self.lock:self.process=None
|
||
def start(self,revision,backend='CUDA',jobs=1):
|
||
if not isinstance(revision,str) or not re.fullmatch(r'(?:b[0-9]{3,8}|v[0-9]{1,4}\.[0-9]{1,4}\.[0-9]{1,4}|[a-f0-9]{40})',revision):raise ValueError('Offizielles v-Release, b-Nightly oder vollständigen Commit-SHA wählen.')
|
||
if backend not in ('CUDA','CPU') or type(jobs)!=int or not 1<=jobs<=2:raise ValueError('CPU/CUDA und 1–2 Build-Jobs unterstützt.')
|
||
with self.lock:
|
||
if self.busy:raise ValueError('Ein Build läuft bereits.')
|
||
p=self.prerequisites()
|
||
if not p['cuda_ready' if backend=='CUDA' else 'cpu_ready']:raise ValueError('Build-Werkzeuge fehlen. Voraussetzungen prüfen.')
|
||
self.root.mkdir(parents=True,exist_ok=True)
|
||
if shutil.disk_usage(self.root).free<15*1024**3:raise ValueError('Mindestens 15 GiB freier Speicher erforderlich.')
|
||
ident=uuid.uuid4().hex
|
||
self.state['job']=dict(id=ident,state='running',phase='Vorbereitung',revision=revision,backend=backend,started=time.time(),error=None)
|
||
self.cancelled.clear();self.busy=True;self.save()
|
||
threading.Thread(target=self.build,args=(ident,revision,backend,jobs,p),daemon=True).start()
|
||
return self.status()
|
||
def build(self,ident,revision,backend,jobs,prereq):
|
||
directory=self.root/ident
|
||
try:
|
||
directory.mkdir();(self.root/'build.log').write_text('')
|
||
self.run(['git','init',str(directory)],self.root,'Quellverzeichnis anlegen',30)
|
||
self.run(['git','-C',str(directory),'fetch','--depth','1',ORIGIN,revision],self.root,'Offizielle Quellen herunterladen',300)
|
||
self.run(['git','checkout','--detach','FETCH_HEAD'],directory,'Feste Version auswählen',30)
|
||
commit=command(['git','-C',str(directory),'rev-parse','HEAD'])
|
||
prepare_fit_source(directory)
|
||
prepare_mtp_source(directory)
|
||
arch=sorted(set(g['architecture'] for g in prereq['gpus']))
|
||
args=['cmake','-S','.', '-B','build','-DCMAKE_BUILD_TYPE=Release','-DGGML_NATIVE=OFF','-DLLAMA_BUILD_TESTS=OFF','-DLLAMA_BUILD_EXAMPLES=OFF','-DLLAMA_BUILD_SERVER=ON','-DGGML_CUDA='+('ON' if backend=='CUDA' else 'OFF')]
|
||
args+=['-DLLAMA_BUILD_IS_DEV='+('OFF' if revision.startswith('v') else 'ON')]
|
||
if backend=='CUDA':args+=['-DCMAKE_CUDA_ARCHITECTURES='+';'.join(arch)]
|
||
self.run(args,directory,'Build konfigurieren',300)
|
||
self.run(['cmake','--build','build','--target','llama-server','llama-fit-params','-j',str(jobs)],directory,'llama.cpp-Werkzeuge kompilieren')
|
||
(directory/'build/bin/deck-mtp-fit-v1').write_text('shared weights + main and MTP contexts, f16 draft KV\n')
|
||
binary=directory/'build/bin/llama-server'
|
||
self.run([str(binary),'--version'],directory,'Binärdatei prüfen',30)
|
||
help_text=command([str(binary),'--help'],30)
|
||
item=dict(id=ident,revision=revision,commit=commit,backend=backend,architectures=arch if backend=='CUDA' else [],created=time.time(),fit_supported='--fit ' in help_text,fit_tool=(directory/'build/bin/llama-fit-params').exists(),fit_adapter='shared-kv-mtp-v1')
|
||
with self.lock:
|
||
self.state['builds'].append(item);self.state['job'].update(state='complete',phase='Build geprüft; kann als Standard ausgewählt werden.');self.save()
|
||
except Exception as exc:
|
||
with self.lock:
|
||
self.state['job'].update(state='cancelled' if isinstance(exc,InterruptedError) else 'failed',error=str(exc) if isinstance(exc,ValueError) else 'Build unterbrochen oder Werkzeug nicht erreichbar.');self.save()
|
||
finally:
|
||
with self.lock:self.busy=False
|
||
def stop(self):
|
||
with self.lock:
|
||
self.cancelled.set()
|
||
if self.process and self.process.poll() is None:
|
||
try:os.killpg(self.process.pid,signal.SIGTERM)
|
||
except ProcessLookupError:pass
|
||
try:self.process.wait(timeout=3)
|
||
except subprocess.TimeoutExpired:
|
||
try:os.killpg(self.process.pid,signal.SIGKILL)
|
||
except ProcessLookupError:pass
|
||
return {'cancellation_requested':True}
|
||
def activate(self,build_id):
|
||
with self.lock:
|
||
if not any(b['id']==build_id for b in self.state['builds']):raise ValueError('Kein erfolgreich geprüfter Build.')
|
||
if self.state['active']!=build_id:
|
||
self.state['previous']=self.state['active'];self.state['active']=build_id;self.save()
|
||
return self.status()
|
||
def rollback(self):
|
||
with self.lock:
|
||
if not self.state['previous']:raise ValueError('Kein vorheriger Build vorhanden.')
|
||
return self.activate(self.state['previous'])
|
||
def log(self):
|
||
p=self.root/'build.log'
|
||
if not p.exists():return {'text':''}
|
||
with p.open('rb') as f:
|
||
f.seek(max(0,p.stat().st_size-24000));return {'text':f.read().decode(errors='replace')}
|
||
|
||
def references(self):
|
||
base=Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))
|
||
specs=[('fast','Qwen3.8-27B-IQ4-MIX.gguf',76800,1,64),('medium','qwen3.8-27b-IQ4_XS-pure.gguf',160000,2,256),('large','qwen3.8-27b-IQ4_XS-pure.gguf',192000,1,256),('ultra','qwen3.8-27b-IQ4_XS-pure.gguf',262144,1,128),('uncensored','Qwen3.8-27B-ABLITERATED-Q4_K_M.gguf',80000,1,256)]
|
||
references=[]
|
||
profiles=getattr(self,'profiles',None);catalog=getattr(self,'catalog',None)
|
||
if profiles is not None and catalog is not None:
|
||
with profiles.lock: saved=list(profiles.rows)
|
||
for profile in saved:
|
||
if profile.get('kind')!='chat':continue
|
||
try: model=catalog.entry(profile['model_id'])
|
||
except (KeyError,ValueError,OSError):continue
|
||
if not model['file'].lower().endswith('.gguf') or not model['profile_eligible']:continue
|
||
params=profile['parameters']
|
||
references.append(dict(id='deck:'+profile['id'],source='deck',name=profile['name'],file=model['file'],model_id=model['id'],context=params['context'],slots=params['slots'],batch=params['batch'],ubatch=params['ubatch'],cache='q4_0',available=True))
|
||
references.extend(dict(id=i,source='router',name=i,file=f,context=c,slots=n,batch=2048,ubatch=u,cache='q4_0',available=(base/f).is_file()) for i,f,c,n,u in specs)
|
||
return {'profiles':references}
|
||
def fit(self,profile,context,slots):
|
||
if type(context)!=int or not 512<=context<=2097152 or type(slots)!=int or not 1<=slots<=16:raise ValueError('Kontext oder Slots außerhalb des erlaubten Bereichs.')
|
||
ref=next((p for p in self.references()['profiles'] if p['id']==profile),None)
|
||
if not ref or not ref['available']:raise ValueError('Modelldatei für die Einpassung nicht verfügbar.')
|
||
with self.lock:
|
||
build=next((b for b in self.state['builds'] if b['id']==self.state['active']),None)
|
||
if not build or not build.get('fit_tool'):raise ValueError('Zuerst einen CUDA-Build mit Fit-Werkzeug erstellen und auswählen.')
|
||
if build['backend']!='CUDA':raise ValueError('Für GPU-Einpassung einen CUDA-Build auswählen.')
|
||
if self.busy:raise ValueError('Bitte Ende des Builds abwarten.')
|
||
if getattr(self,'fitting',False):raise ValueError('Eine Einpassung läuft bereits.')
|
||
self.fitting=True
|
||
try:
|
||
binary=self.root/build['id']/'build/bin/llama-fit-params'
|
||
model=(self.catalog.root/ref['model_id']/('model'+Path(ref['file']).suffix)) if ref.get('source')=='deck' else Path(os.environ.get('DECK_REFERENCE_MODELS','/reference-models'))/ref['file']
|
||
args=[str(binary),'--model',str(model),'--ctx-size',str(context),'--parallel',str(slots),'--cache-type-k',ref.get('cache','q4_0'),'--cache-type-v',ref.get('cache','q4_0'),'--flash-attn','on','--batch-size',str(ref.get('batch',2048)),'--ubatch-size',str(ref['ubatch'])]
|
||
snapshot=self.prerequisites()['gpus']
|
||
usable=[g for g in snapshot if g['free_mib']>=max(1024,g['total_mib']*.10)]
|
||
if not usable:raise ValueError('GPUs derzeit belegt. Keine sichere Auto-Prognose; produktive Dienste bleiben unverändert.')
|
||
margins=[max(512,round(g['total_mib']*.05)) for g in usable]
|
||
args+=['--fit-target',','.join(map(str,margins))]
|
||
env=dict(os.environ,CUDA_VISIBLE_DEVICES=','.join(g['uuid'] for g in usable))
|
||
result=subprocess.run(args,capture_output=True,text=True,timeout=120,env=env)
|
||
text=result.stdout.strip()
|
||
fitted={};memory=[]
|
||
if result.returncode==0:
|
||
values=shlex.split(text)
|
||
for flag,key in [('-c','context'),('-ngl','gpu_layers'),('-ts','tensor_split')]:
|
||
if flag in values and values.index(flag)+1<len(values):fitted[key]=values[values.index(flag)+1]
|
||
if fitted.get('context')!=str(context):raise ValueError('Fit-Werkzeug hat den gewünschten Kontext nicht beibehalten; Ergebnis wird nicht übernommen.')
|
||
if len(values)%2==0 and all(values[i] in ('-c','-ngl','-ts') for i in range(0,len(values),2)):
|
||
probe=subprocess.run(args+['--fit-print','on']+values,capture_output=True,text=True,timeout=120,env=env)
|
||
if probe.returncode==0:
|
||
for line in probe.stdout.splitlines():
|
||
match=re.fullmatch(r'(\S+) +(\d+) +(\d+) +(\d+) *',line)
|
||
if match:memory.append(dict(device=match[1],model_mib=int(match[2]),context_mib=int(match[3]),compute_mib=int(match[4])))
|
||
ram_limit=None
|
||
try:
|
||
raw=Path('/sys/fs/cgroup/memory.max').read_text().strip()
|
||
if raw!='max':ram_limit=int(raw)//(1024*1024)
|
||
except (OSError,ValueError):pass
|
||
host=next((m for m in memory if m['device']=='Host'),None)
|
||
within_limit=None if host is None or ram_limit is None else sum(host[k] for k in ('model_mib','context_mib','compute_mib'))<=ram_limit
|
||
return dict(success=result.returncode==0,ram_limit_mib=ram_limit,within_ram_limit=within_limit,memory=memory,fitted=fitted,profile=profile,context=context,slots=slots,arguments=text[-16000:],details=result.stderr[-16000:],estimated=True,scope='text_model_without_vision_or_mtp',model_loaded=False,gpus=snapshot,used_gpu_uuids=[g['uuid'] for g in usable],reserve_mib=margins,excluded_gpus=[g['name'] for g in snapshot if g not in usable])
|
||
finally:
|
||
with self.lock:self.fitting=False
|