diff --git a/STUDIO.md b/STUDIO.md
index 8aedd36..68f8ea9 100644
--- a/STUDIO.md
+++ b/STUDIO.md
@@ -157,3 +157,7 @@ Standardwerte (automatische Geräte, keine Aufteilung, Temperature 0.8, Top-p 0.
Top-k 40). Bestehende Dateien werden beim Lesen nicht umgeschrieben.
Sprachmodelle besitzen jetzt den Reiter **Testen**: interner Textchat mit Streaming, Profilauswahl und Speicherdiagnose. Öffentliche Profilfreigabe ist dafür nicht erforderlich. Details: [ENDPOINT.md](ENDPOINT.md#sprachmodelle--testen).
+
+### MTP bei Sprachmodellprofilen
+
+Im Profileditor kann MTP aktiviert werden, mit 1–8 Draft-Token und Mindestwahrscheinlichkeit 0–1. Medium-Referenz: 2 und 0,05; Draft-KV ist f16. Bestehende Profile bleiben standardmäßig ohne MTP. Das Modell muss eingebettete MTP-Gewichte enthalten. Der CUDA-Build benötigt den Deck-MTP-Fit-Adapter; neue GUI-Builds installieren ihn automatisch. Die Prognose zählt die gemeinsamen Gewichte einmal sowie Haupt- und MTP-Kontext und deren Compute-Puffer. Ein inkompatibles Modell oder ein alter Build wird nicht still ohne MTP gestartet. MTP garantiert keinen Geschwindigkeitsgewinn.
diff --git a/inference.py b/inference.py
index dbb0600..aec17f4 100644
--- a/inference.py
+++ b/inference.py
@@ -127,6 +127,8 @@ class LlamaWorker:
def _start(self,profile,generation,cancel=lambda:False):
directory=self.build();params=profile['parameters'];entry=self.catalog.entry(profile['model_id'])
model=(self.catalog.root/entry['id']/('model'+Path(entry['file']).suffix)).resolve()
+ mtp=params.get('mtp',False)
+ if mtp and not (directory/'build/bin/deck-mtp-fit-v1').is_file():raise InferenceError('MTP benötigt einen neu gebauten llama.cpp-Build mit Deck-MTP-Speicherprüfung. Unter Laufzeiten erneut bauen.')
devices=probe();ids=params['gpu_devices']
if ids:
selected=[next((g for g in devices if g['uuid']==ident),None) for ident in ids]
@@ -146,10 +148,11 @@ class LlamaWorker:
common=['--model',str(model),'--ctx-size',str(params['context']),'--parallel',str(params['slots']),'--batch-size',str(params['batch']),'--ubatch-size',str(params['ubatch']),'--cache-type-k','q4_0','--cache-type-v','q4_0','--flash-attn','on','--split-mode',params['split_mode'],'--fit-target',','.join(str(max(1024,round(g['total_mib']*.05))) for g in selected)]
if params['tensor_split']:common+=['--tensor-split',','.join(map(str,params['tensor_split']))]
tool=str(directory/'build/bin/llama-fit-params')
+ mtp_fit=['--deck-mtp'] if mtp else []
margins=[max(1024,round(g['total_mib']*.05)) for g in selected]
def estimate(layers,extra=()):
if cancel():raise InferenceError('Modellstart abgebrochen.')
- output=self._fit_command([tool]+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation,cancel)
+ output=self._fit_command([tool]+mtp_fit+common+['--gpu-layers',str(layers),'--fit-print','on']+list(extra),env,generation,cancel)
rows={}
for line in output.splitlines():
parts=line.split()
@@ -176,7 +179,7 @@ class LlamaWorker:
layers,memory=best
estimate(layers)
else:
- output=self._fit_command([tool]+common,env,generation,cancel)
+ output=self._fit_command([tool]+mtp_fit+common,env,generation,cancel)
flags=shlex.split(output.strip());fit={}
if len(flags)%2:raise InferenceError('Fit-Werkzeug lieferte ungültige Parameter.')
for i in range(0,len(flags),2):
@@ -193,6 +196,7 @@ class LlamaWorker:
if cancel():raise InferenceError('Modellstart abgebrochen.')
with self.lock:self.phase='Modell wird geladen'
launch=common+extra+['--gpu-layers',str(layers),'--fit','off','--kv-unified','--threads',str(params['threads']),'--load-mode','none','--host','127.0.0.1','--alias',profile['name'],'--no-webui','--log-disable']
+ if mtp:launch+=['--spec-type','draft-mtp','--spec-draft-n-max',str(params.get('mtp_tokens',2)),'--spec-draft-p-min',str(params.get('mtp_min_p',.05)),'--spec-draft-type-k','f16','--spec-draft-type-v','f16']
self.root.mkdir(parents=True,exist_ok=True,mode=0o700)
with socket.socket() as sock:sock.bind(('127.0.0.1',0));port=sock.getsockname()[1]
key=secrets.token_urlsafe(32);keypath=self.root/'worker.key'
diff --git a/profiles-ui.js b/profiles-ui.js
index ec9dabd..a6188a7 100644
--- a/profiles-ui.js
+++ b/profiles-ui.js
@@ -1,6 +1,6 @@
window.ProfilesUI=(()=>{
const e=v=>String(v??'').replace(/[&<>"']/g,c=>({'&':'&','<':'<','>':'>','"':'"',"'":'''}[c]));
- const labels={temperature:'Temperature',top_p:'Top-p',top_k:'Top-k',gpu_devices:'GPUs (in Reihenfolge)',split_mode:'GPU-Split',tensor_split:'GPU-Verteilung',context:'Gesamtes Kontextbudget (Token)',slots:'Parallele Slots',threads:'CPU-Threads',batch:'Batch-Größe',ubatch:'Microbatch-Größe',width:'Breite (Pixel)',height:'Höhe (Pixel)',steps:'Schritte',seed:'Seed (−1 = zufällig)',guidance:'Guidance / CFG',speed:'Sprechgeschwindigkeit',frames:'Bildanzahl',fps:'Bilder pro Sekunde'};
+ const labels={mtp:'MTP',mtp_tokens:'MTP: maximale Draft-Token',mtp_min_p:'MTP: Mindestwahrscheinlichkeit',temperature:'Temperature',top_p:'Top-p',top_k:'Top-k',gpu_devices:'GPUs (in Reihenfolge)',split_mode:'GPU-Split',tensor_split:'GPU-Verteilung',context:'Gesamtes Kontextbudget (Token)',slots:'Parallele Slots',threads:'CPU-Threads',batch:'Batch-Größe',ubatch:'Microbatch-Größe',width:'Breite (Pixel)',height:'Höhe (Pixel)',steps:'Schritte',seed:'Seed (−1 = zufällig)',guidance:'Guidance / CFG',speed:'Sprechgeschwindigkeit',frames:'Bildanzahl',fps:'Bilder pro Sekunde'};
const html=()=>' Auf Athena gespeichert. Eine Modelldatei kann mehrere Profile mit unterschiedlichen Parametern haben.Deine Profile
${data.id?'Profil bearbeiten':'Profil anlegen'}
${data.id?'Profil bearbeiten':'Profil anlegen'}
CPU-Threads steuern CPU-Arbeit und Offloading, nicht die Anzahl der CUDA-Rechenkerne.
';advanced.append(threads);el('profile-form').insertBefore(advanced,el('profile-form').querySelector('.note'));} el('profile-close').onclick=()=>el('profile-editor').replaceChildren(); el('profile-form').onsubmit=async event=>{event.preventDefault();const form=event.target,values=new FormData(form),button=form.querySelector('button');button.disabled=true; - try{const parameters=Object.fromEntries(Object.keys(schema).map(k=>[k,Number(values.get(k))]));if(kind==='chat'){if(values.get('gpu_second')&&!values.get('gpu_first'))throw Error('Bitte zuerst die erste GPU auswählen.');if((data.parameters.gpu_devices||[]).length>2)throw Error('Dieses Profil enthält mehr als zwei GPUs; der Editor unterstützt derzeit zwei.');parameters.gpu_devices=[values.get('gpu_first'),values.get('gpu_second')].filter(Boolean);parameters.split_mode=values.get('split_mode');const raw=String(values.get('tensor_split')).trim();parameters.tensor_split=raw?raw.split(',').map(x=>x.trim()?Number(x):NaN):[];}await api('profiles/save',{id:data.id,revision:data.revision,name:values.get('name'),kind,model_id:values.get('model_id'),parameters});if(!root.isConnected)return;el('profile-editor').replaceChildren();await load();message('Profil auf Athena gespeichert. Kein Modell gestartet.');} + try{const parameters=Object.fromEntries(Object.keys(schema).map(k=>[k,Number(values.get(k))]));if(kind==='chat'){parameters.mtp=values.has('mtp');if(values.get('gpu_second')&&!values.get('gpu_first'))throw Error('Bitte zuerst die erste GPU auswählen.');if((data.parameters.gpu_devices||[]).length>2)throw Error('Dieses Profil enthält mehr als zwei GPUs; der Editor unterstützt derzeit zwei.');parameters.gpu_devices=[values.get('gpu_first'),values.get('gpu_second')].filter(Boolean);parameters.split_mode=values.get('split_mode');const raw=String(values.get('tensor_split')).trim();parameters.tensor_split=raw?raw.split(',').map(x=>x.trim()?Number(x):NaN):[];}await api('profiles/save',{id:data.id,revision:data.revision,name:values.get('name'),kind,model_id:values.get('model_id'),parameters});if(!root.isConnected)return;el('profile-editor').replaceChildren();await load();message('Profil auf Athena gespeichert. Kein Modell gestartet.');} catch(error){if(root.isConnected)el('profile-error').textContent=error.message;}finally{button.disabled=false;} }; el('profile-editor').scrollIntoView({behavior:'smooth',block:'start'}); diff --git a/profiles.py b/profiles.py index 9576f8d..a009d7f 100644 --- a/profiles.py +++ b/profiles.py @@ -7,16 +7,17 @@ import uuid from pathlib import Path SCHEMAS={ - 'chat':{'context':(512,2097152,8192),'slots':(1,16,1),'threads':(1,256,6),'batch':(1,8192,512),'ubatch':(1,8192,128),'temperature':(0,5,.8),'top_p':(0,1,.95),'top_k':(0,1000,40)}, + 'chat':{'context':(512,2097152,8192),'slots':(1,16,1),'threads':(1,256,6),'batch':(1,8192,512),'ubatch':(1,8192,128),'temperature':(0,5,.8),'top_p':(0,1,.95),'top_k':(0,1000,40),'mtp_tokens':(1,8,2),'mtp_min_p':(0,1,.05)}, 'image':{'width':(256,2048,1024),'height':(256,2048,1024),'steps':(1,100,25),'seed':(-1,2147483647,-1),'guidance':(0,30,1)}, 'audio':{'speed':(.25,4,1)}, 'video':{'width':(256,1920,768),'height':(256,1088,512),'frames':(1,241,33),'fps':(1,60,24),'steps':(1,100,20),'seed':(-1,2147483647,-1)} } -CHAT_GPU_DEFAULTS={'gpu_devices':[], 'split_mode':'none', 'tensor_split':[]} -CHAT_SAMPLING={'temperature':.8,'top_p':.95,'top_k':40} +CHAT_GPU_DEFAULTS={'gpu_devices':[], 'split_mode':'none', 'tensor_split':[], 'mtp':False} +CHAT_SAMPLING={'temperature':.8,'top_p':.95,'top_k':40,'mtp_tokens':2,'mtp_min_p':.05} def chat_parameters(params): params={**CHAT_GPU_DEFAULTS,**CHAT_SAMPLING,**params} + if type(params['mtp']) is not bool:raise ValueError('MTP muss an oder aus sein.') devices=params['gpu_devices'];split=params['tensor_split'] if not isinstance(devices,list) or len(devices)>16 or any(not isinstance(v,str) or not re.fullmatch(r'GPU-[0-9a-fA-F-]{36}',v) for v in devices) or len(set(devices))!=len(devices):raise ValueError('GPUs müssen als eindeutige, geordnete GPU-UUIDs angegeben werden.') if params['split_mode'] not in ('none','layer','row'):raise ValueError('Ungültiger GPU-Split-Modus.') @@ -68,7 +69,7 @@ class Profiles: if not isinstance(params,dict) or set(params)!=(set(SCHEMAS[kind]) | (set(CHAT_GPU_DEFAULTS) if kind=='chat' else set())):raise ValueError('Unvollständige oder unbekannte Profilparameter.') for key,(lo,hi,_) in SCHEMAS[kind].items(): value=params[key] - floating=key in ('guidance','speed','temperature','top_p') + floating=key in ('guidance','speed','temperature','top_p','mtp_min_p') if isinstance(value,bool) or not isinstance(value,(float,int) if floating else int) or not lo<=value<=hi:raise ValueError('Ungültiger Parameter: '+key) if kind in ('image','video') and (params['width']%64 or params['height']%64):raise ValueError('Breite und Höhe müssen durch 64 teilbar sein.') if kind=='chat' and params['ubatch']>params['batch']:raise ValueError('Microbatch darf nicht größer als Batch sein.') diff --git a/runtime.py b/runtime.py index b363d02..0e5db62 100644 --- a/runtime.py +++ b/runtime.py @@ -31,6 +31,58 @@ def prepare_fit_source(directory): if source.count(anchor)!=1:raise ValueError('Fit-Adapter passt nicht zu dieser Version; Build abgebrochen.') path.write_text(source.replace(anchor,marker+'\n'+anchor)) +def prepare_mtp_source(directory): + """Extend the no-allocation fit tool with a shared-weight MTP context.""" + path=Path(directory)/'tools/fit-params/fit-params.cpp' + source=path.read_text() + if '// Deck MTP fit v1' in source: + if '#include "speculative.h"' not in source:path.write_text(source.replace('#include "fit.h"','#include "fit.h"\n#include "speculative.h"')) + return + changes={ + '#include "fit.h"':'#include "fit.h"\n#include "speculative.h"', + ' common_init();':''' // Deck MTP fit v1 + bool deck_mtp = false; + for (int i = 1; i < argc; ++i) { + if (std::string(argv[i]) == "--deck-mtp") { + deck_mtp = true; + for (int j = i; j + 1 < argc; ++j) argv[j] = argv[j + 1]; + --argc; --i; + } + } + common_init();''', + ' auto mparams = common_model_params_to_llama(params);':''' if (deck_mtp) params.speculative.types = {COMMON_SPECULATIVE_TYPE_DRAFT_MTP}; + auto mparams = common_model_params_to_llama(params);''', + ' auto cparams = common_context_params_to_llama(params);':''' auto cparams = common_context_params_to_llama(params); + auto draft_params = common_base_params_to_speculative(params); + auto draft_mparams = common_model_params_to_llama(draft_params); + auto draft_cparams = common_context_params_to_llama(draft_params); + draft_cparams.ctx_type = LLAMA_CONTEXT_TYPE_MTP; + draft_cparams.n_rs_seq = 0; + const common_fit_extra_model draft_extra = { + params.model.path.c_str(), &draft_mparams, &draft_cparams, true + };''', + ' nullptr,':' deck_mtp ? &draft_extra : nullptr,', + ' common_fit_print(params.model.path.c_str(), &mparams, &cparams);':''' if (!deck_mtp) { + common_fit_print(params.model.path.c_str(), &mparams, &cparams); + } else { + std::vector