diff --git a/README.md b/README.md index 6ec2002..5980432 100644 --- a/README.md +++ b/README.md @@ -21,7 +21,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.** - FLUX.2 Klein 9B FP8 Beta: Transformer/VAE auf RTX 5080, Textencoder auf RTX 3060 - Qwen3-TTS 1.7B auf der RTX 3060 hinter dem TTS-Gateway; kein Piper-Fallback - Whisper.cpp `small` auf der CPU für lokale deutsche Spracherkennung -- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099 +- Live-Dashboard mit 21 Tagen Detailhistorie für GPUs und Slot-Kontextbelegung auf Port 8099 - Portainer CE als optionale Container-Ansicht auf Port 9443 - WireGuard-Gateway, Datenbackup und Athena-Operator - keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz diff --git a/dev/test_dashboard_modes.py b/dev/test_dashboard_modes.py index 08e92bd..948401c 100644 --- a/dev/test_dashboard_modes.py +++ b/dev/test_dashboard_modes.py @@ -118,6 +118,40 @@ class DashboardModeTests(unittest.TestCase): self.assertEqual(backups[0]["sha256"], "a" * 64) self.assertTrue(backups[0]["download_url"].startswith("/api/backups/download/")) + def test_slot_context_history_is_recorded_and_exposed(self): + with tempfile.TemporaryDirectory() as tempdir: + store = self.dashboard.HistoryStore(Path(tempdir) / "slots.sqlite3") + store.record({ + "timestamp": 1_789_000_000, + "router": { + "current_profile": "medium", + "upstream": {"model": "qwen"}, + "llama_telemetry": { + "slots": [{ + "id": 0, + "context_used": 40_000, + "n_ctx": 160_000, + "processing": True, + }], + }, + }, + "llama_runtime": {"pid": 123, "model_file": "qwen.gguf"}, + "gpus": [], + }) + + result = store.query("all") + store._db.close() + + self.assertEqual(len(result["slot_points"]), 1) + self.assertEqual(result["slot_points"][0]["slot_id"], 0) + self.assertEqual(result["slot_points"][0]["context_percent"], 25.0) + self.assertEqual(result["slot_points"][0]["busy_ratio"], 1.0) + + def test_dashboard_renders_slot_context_history_controls(self): + self.assertIn('id="slotHistoryChart"', self.dashboard.HTML) + self.assertIn('id="slotHistoryRanges"', self.dashboard.HTML) + self.assertIn("drawSlotHistory(d.slot_points||[])", self.dashboard.HISTORY_JS) + if __name__ == "__main__": unittest.main() diff --git a/docs/CONTAINER_INVENTORY.md b/docs/CONTAINER_INVENTORY.md index d939afb..79ed809 100644 --- a/docs/CONTAINER_INVENTORY.md +++ b/docs/CONTAINER_INVENTORY.md @@ -13,7 +13,7 @@ nicht automatisch ein ungenutzter Rest. | `mike-ai-backup` | kein Modell; Offen Docker Volume Backup `v2` (Digest vom 15. September 2026) | Sichert `/data`, `/etc/mike-ai`, den Stack und die persistenten Docker-Volumes im Fünf-Stunden-Takt. | | `mike-ai-applio-studio` | Applio/RVC; Stimmenmodelle werden nutzerseitig ergänzt | Vollständige RVC-Oberfläche für Inferenz, Modellverwaltung und Training auf der RTX 5080. Für eine Konvertierung ist ein importiertes oder trainiertes `.pth`-Modell nötig; eine Referenzaufnahme allein reicht nicht. | | `mike-ai-image-worker` | FLUX.2 Klein 9B FP8, Qwen3-8B NF4 Textencoder und VAE | Erzeugt und bearbeitet Bilder transaktional; nutzt während eines Auftrags RTX 5080 und RTX 3060. | -| `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung und Betriebsarten an und bietet die Modusumschaltung. | +| `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung, Slot-Kontextbelegung mit Verlauf und Betriebsarten an und bietet die Modusumschaltung. | | `mike-ai-llama-fast` | Qwen3.8-27B `IQ4-MIX`, Qwen-MMProj BF16 | Schnelles Q4-Text-/Vision-Profil mit 76.800 Token Kontext. | | `mike-ai-llama-large` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Q4-Text-/Vision-Profil mit 192.000 Token Kontext und Verteilung auf beide GPUs. | | `mike-ai-llama-medium` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Standard-Q4-Text-/Vision-Profil mit 160.000 Token Kontext und Verteilung auf beide GPUs. | diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py index da4d4b3..32a0be0 100644 --- a/platform/llama-dashboard/app.py +++ b/platform/llama-dashboard/app.py @@ -500,6 +500,19 @@ class HistoryStore: temperature_sum REAL, temperature_max REAL, power_sum REAL, samples INTEGER NOT NULL, PRIMARY KEY(hour_ts, gpu_index) ); + CREATE TABLE IF NOT EXISTS slot_samples_raw ( + ts INTEGER NOT NULL, slot_id INTEGER NOT NULL, + context_used REAL, context_total REAL, processing INTEGER NOT NULL DEFAULT 0, + profile TEXT, model TEXT + ); + CREATE INDEX IF NOT EXISTS idx_slot_samples_raw_ts ON slot_samples_raw(ts); + CREATE TABLE IF NOT EXISTS slot_samples_hourly ( + hour_ts INTEGER NOT NULL, slot_id INTEGER NOT NULL, + context_percent_sum REAL, context_percent_max REAL, + context_used_max REAL, context_total_max REAL, + busy_samples INTEGER NOT NULL, samples INTEGER NOT NULL, + PRIMARY KEY(hour_ts, slot_id) + ); CREATE TABLE IF NOT EXISTS token_totals ( id INTEGER PRIMARY KEY CHECK(id=1), prompt_tokens INTEGER NOT NULL DEFAULT 0, cached_tokens INTEGER NOT NULL DEFAULT 0, output_tokens INTEGER NOT NULL DEFAULT 0 @@ -552,6 +565,18 @@ class HistoryStore: gpu.get("temperature_c"), gpu.get("power_w"), profile, model), ) + slots = ((router.get("llama_telemetry") or {}).get("slots") or []) + for slot in slots: + if slot.get("id") is None: + continue + used = max(0, float(slot.get("context_used") or 0)) + total = max(0, float(slot.get("n_ctx") or 0)) + self._db.execute( + "INSERT INTO slot_samples_raw VALUES(?,?,?,?,?,?,?)", + (ts, int(slot["id"]), used, total, int(bool(slot.get("processing"))), + profile, model), + ) + previous_runtime = self._meta("counter_runtime") deltas: list[int] = [] for key in self.TOKEN_KEYS: @@ -605,6 +630,22 @@ class HistoryStore: samples=samples+excluded.samples """, (cutoff,)) self._db.execute("DELETE FROM samples_raw WHERE ts < ?", (cutoff,)) + self._db.execute(""" + INSERT INTO slot_samples_hourly + SELECT ts-ts%3600, slot_id, + SUM(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END), + MAX(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END), + MAX(context_used), MAX(context_total), SUM(processing), COUNT(*) + FROM slot_samples_raw WHERE ts < ? GROUP BY ts-ts%3600, slot_id + ON CONFLICT(hour_ts,slot_id) DO UPDATE SET + context_percent_sum=context_percent_sum+excluded.context_percent_sum, + context_percent_max=MAX(context_percent_max,excluded.context_percent_max), + context_used_max=MAX(context_used_max,excluded.context_used_max), + context_total_max=MAX(context_total_max,excluded.context_total_max), + busy_samples=busy_samples+excluded.busy_samples, + samples=samples+excluded.samples + """, (cutoff,)) + self._db.execute("DELETE FROM slot_samples_raw WHERE ts < ?", (cutoff,)) def query(self, range_name: str) -> dict[str, Any]: ranges = { @@ -617,6 +658,7 @@ class HistoryStore: detail_cutoff = now - DETAIL_RETENTION_DAYS * 86400 with self._lock: points: list[dict[str, Any]] = [] + slot_points: list[dict[str, Any]] = [] if start < detail_cutoff: for row in self._db.execute(""" SELECT hour_ts ts,gpu_index,gpu_util_sum/samples gpu_util,gpu_util_max, @@ -635,6 +677,25 @@ class HistoryStore: FROM samples_raw WHERE ts>=? GROUP BY (ts/{bucket}),gpu_index ORDER BY ts,gpu_index """, (raw_start,)): points.append(dict(row)) + if start < detail_cutoff: + for row in self._db.execute(""" + SELECT hour_ts ts,slot_id,context_percent_sum/samples context_percent, + context_percent_max,context_used_max context_used, + context_total_max context_total, + 1.0*busy_samples/samples busy_ratio + FROM slot_samples_hourly WHERE hour_ts>=? ORDER BY hour_ts,slot_id + """, (start,)): + slot_points.append(dict(row)) + for row in self._db.execute(f""" + SELECT (ts/{bucket})*{bucket} ts,slot_id, + AVG(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END) context_percent, + MAX(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END) context_percent_max, + MAX(context_used) context_used,MAX(context_total) context_total, + AVG(processing) busy_ratio + FROM slot_samples_raw WHERE ts>=? + GROUP BY (ts/{bucket}),slot_id ORDER BY ts,slot_id + """, (raw_start,)): + slot_points.append(dict(row)) totals = dict(self._db.execute("SELECT * FROM token_totals WHERE id=1").fetchone()) range_tokens = dict(self._db.execute( "SELECT COALESCE(SUM(prompt_tokens),0) prompt_tokens, COALESCE(SUM(cached_tokens),0) cached_tokens, COALESCE(SUM(output_tokens),0) output_tokens FROM token_hourly WHERE hour_ts>=?", @@ -652,17 +713,23 @@ class HistoryStore: raw_info = dict(self._db.execute( "SELECT COUNT(*) rows, MIN(ts) oldest, MAX(ts) newest FROM samples_raw" ).fetchone()) + slot_raw_info = dict(self._db.execute( + "SELECT COUNT(*) rows, MIN(ts) oldest, MAX(ts) newest FROM slot_samples_raw" + ).fetchone()) points.sort(key=lambda item: (item["ts"], item["gpu_index"])) + slot_points.sort(key=lambda item: (item["ts"], item["slot_id"])) return { "range": range_name if range_name in ranges else "24h", "detail_retention_days": DETAIL_RETENTION_DAYS, "sample_interval_seconds": HISTORY_INTERVAL, "points": points, + "slot_points": slot_points, "token_totals": totals, "range_tokens": range_tokens, "profile_usage": profile_usage, "model_events": events, "storage": raw_info, + "slot_storage": slot_raw_info, } @@ -709,7 +776,7 @@ HTML = r'''
Prompt-Einlesen
–
Token pro Sekunde
Prompt-Cache
–
–
Anfragen
–
–
-
Slots und Kontextfenster
Telemetrie wird geladen …
+
Slots und Kontextfenster
Aktuelle Belegung und Verlauf des genutzten Kontextfensters
Telemetrie wird geladen …
Historie wird geladen …
MTP / Speculative Decoding
–
–Draft-Token
–akzeptiert
–Prüfschritte
–Draft-Tiefe
Tokenzähler seit Modellstart
Modell-Eigenschaften
@@ -823,6 +890,7 @@ HISTORY_JS = r''' let historyRange='24h',usageMode='total',lastProfileUsage=[]; const historyColors=['#45d7ff','#ffb454','#66e3a4','#c39bff']; const hiddenHistorySeries=new Set();let lastHistoryPoints=[]; +const hiddenSlotSeries=new Set();let lastSlotPoints=[]; function drawHistory(points){ lastHistoryPoints=points; const canvas=$('gpuHistoryChart'),rect=canvas.getBoundingClientRect(),ratio=window.devicePixelRatio||1; @@ -850,12 +918,35 @@ function drawHistory(points){ ids.forEach((id,idx)=>{let rows=points.filter(p=>p.gpu_index===id),load=historyColors[idx*2%historyColors.length],temp=historyColors[(idx*2+1)%historyColors.length]; [[load,'gpu_util','load'],[temp,'temperature_c','temp']].forEach(([color,key,kind])=>{if(hiddenHistorySeries.has(`${id}:${kind}`))return;x.beginPath();x.strokeStyle=color;x.lineWidth=2;let first=true;rows.forEach(p=>{if(p[key]==null)return;let xx=px(p.ts),yy=py(Number(p[key]));first?(x.moveTo(xx,yy),first=false):x.lineTo(xx,yy)});x.stroke()})}); } -async function refreshHistory(){try{let r=await fetch(`/api/history?range=${historyRange}`,{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),t=d.token_totals||{},input=Number(t.prompt_tokens||0)+Number(t.cached_tokens||0);$('historyInput').textContent=deNum(input);$('historyInputSub').textContent=`${deNum(t.prompt_tokens)} neu · ${deNum(t.cached_tokens)} aus Cache`;$('historyOutput').textContent=deNum(t.output_tokens||0);$('historyStorage').textContent=`${deNum((d.storage||{}).rows||0)} Messpunkte`;drawHistory(d.points||[]);renderProfileUsage(d.profile_usage||[]);$('historyNote').textContent=`Bereich ${d.range} · Messung alle ${d.sample_interval_seconds}s · Detaildaten ${d.detail_retention_days} Tage`;$('historyEvents').innerHTML=(d.model_events||[]).slice(0,15).map(e=>`${new Date(e.ts*1000).toLocaleString('de-DE')}${e.previous_profile||'–'} → ${e.profile||'–'}${e.previous_model||'–'} → ${e.model||'–'}`).join('')||'Noch keine Wechsel aufgezeichnet'}catch(e){$('historyNote').textContent=`Historie nicht verfügbar: ${e.message}`}} +function drawSlotHistory(points){ + lastSlotPoints=points; + const canvas=$('slotHistoryChart'),rect=canvas.getBoundingClientRect(),ratio=window.devicePixelRatio||1; + canvas.width=Math.max(1,Math.floor(rect.width*ratio));canvas.height=Math.max(1,Math.floor(rect.height*ratio)); + const x=canvas.getContext('2d');x.scale(ratio,ratio);const w=rect.width,h=rect.height,pad={l:42,r:18,t:16,b:28}; + x.clearRect(0,0,w,h);x.strokeStyle='#213044';x.fillStyle='#8fa1b5';x.font='11px system-ui';x.lineWidth=1; + for(let i=0;i<=4;i++){let y=pad.t+(h-pad.t-pad.b)*i/4;x.beginPath();x.moveTo(pad.l,y);x.lineTo(w-pad.r,y);x.stroke();x.fillText(`${100-i*25}%`,4,y+4)} + if(!points.length){x.fillText('Noch keine historischen Slot-Messwerte',pad.l+10,h/2);$('slotLegend').innerHTML='';return} + const min=Math.min(...points.map(p=>p.ts)),max=Math.max(...points.map(p=>p.ts)); + const px=t=>pad.l+(t-min)/Math.max(1,max-min)*(w-pad.l-pad.r),py=v=>pad.t+(100-Math.max(0,Math.min(100,v)))/100*(h-pad.t-pad.b); + const span=max-min,ticks=5; + for(let i=0;ip.slot_id))].sort((a,b)=>a-b); + $('slotLegend').innerHTML=ids.map((id,idx)=>{let color=historyColors[idx%historyColors.length],key=String(id);return ``}).join(''); + ids.forEach((id,idx)=>{let key=String(id);if(hiddenSlotSeries.has(key))return;let rows=points.filter(p=>p.slot_id===id);x.beginPath();x.strokeStyle=historyColors[idx%historyColors.length];x.lineWidth=2;let first=true;rows.forEach(p=>{if(p.context_percent==null)return;let xx=px(p.ts),yy=py(Number(p.context_percent));first?(x.moveTo(xx,yy),first=false):x.lineTo(xx,yy)});x.stroke()}); +} +async function refreshHistory(){try{let r=await fetch(`/api/history?range=${historyRange}`,{cache:'no-store'});if(!r.ok)throw Error(`HTTP ${r.status}`);let d=await r.json(),t=d.token_totals||{},input=Number(t.prompt_tokens||0)+Number(t.cached_tokens||0);$('historyInput').textContent=deNum(input);$('historyInputSub').textContent=`${deNum(t.prompt_tokens)} neu · ${deNum(t.cached_tokens)} aus Cache`;$('historyOutput').textContent=deNum(t.output_tokens||0);$('historyStorage').textContent=`${deNum((d.storage||{}).rows||0)} Messpunkte`;drawHistory(d.points||[]);drawSlotHistory(d.slot_points||[]);renderProfileUsage(d.profile_usage||[]);let note=`Bereich ${d.range} · Messung alle ${d.sample_interval_seconds}s · Detaildaten ${d.detail_retention_days} Tage`;$('historyNote').textContent=note;$('slotHistoryNote').textContent=`${note} · ${deNum((d.slot_storage||{}).rows||0)} Slot-Messpunkte`;$('historyEvents').innerHTML=(d.model_events||[]).slice(0,15).map(e=>`${new Date(e.ts*1000).toLocaleString('de-DE')}${e.previous_profile||'–'} → ${e.profile||'–'}${e.previous_model||'–'} → ${e.model||'–'}`).join('')||'Noch keine Wechsel aufgezeichnet'}catch(e){$('historyNote').textContent=`Historie nicht verfügbar: ${e.message}`;$('slotHistoryNote').textContent=`Historie nicht verfügbar: ${e.message}`}} function renderProfileUsage(rows){lastProfileUsage=rows;let value=r=>usageMode==='output'?Number(r.output_tokens||0):Number(r.prompt_tokens||0)+Number(r.cached_tokens||0)+Number(r.output_tokens||0),sum=rows.reduce((n,r)=>n+value(r),0);$('profileUsage').innerHTML=rows.map((r,i)=>{let v=value(r),p=sum?v/sum*100:0,input=Number(r.prompt_tokens||0)+Number(r.cached_tokens||0);return `
${r.profile||'unbekannt'} · ${r.model||'–'}${p.toFixed(1)} %
${deNum(input)} Eingabe · ${deNum(r.output_tokens||0)} Ausgabe${deNum(v)} gewertet
`}).join('')||'
In diesem Zeitraum wurden noch keine Token aufgezeichnet.
'} $('usageModes').addEventListener('click',e=>{let b=e.target.closest('button[data-mode]');if(!b)return;usageMode=b.dataset.mode;document.querySelectorAll('#usageModes button').forEach(x=>x.classList.toggle('active',x===b));renderProfileUsage(lastProfileUsage)}); -$('historyRanges').addEventListener('click',e=>{let b=e.target.closest('button[data-range]');if(!b)return;historyRange=b.dataset.range;document.querySelectorAll('#historyRanges button').forEach(x=>x.classList.toggle('active',x===b));refreshHistory()}); +$('historyRanges').addEventListener('click',e=>setHistoryRange(e)); +$('slotHistoryRanges').addEventListener('click',e=>setHistoryRange(e)); +function setHistoryRange(e){let b=e.target.closest('button[data-range]');if(!b)return;historyRange=b.dataset.range;document.querySelectorAll('#historyRanges button,#slotHistoryRanges button').forEach(x=>x.classList.toggle('active',x.dataset.range===historyRange));refreshHistory()} $('gpuLegend').addEventListener('click',e=>{let b=e.target.closest('button[data-series]');if(!b)return;let key=b.dataset.series;hiddenHistorySeries.has(key)?hiddenHistorySeries.delete(key):hiddenHistorySeries.add(key);drawHistory(lastHistoryPoints)}); -window.addEventListener('resize',()=>refreshHistory());refreshHistory();setInterval(refreshHistory,15000); +$('slotLegend').addEventListener('click',e=>{let b=e.target.closest('button[data-slot-series]');if(!b)return;let key=b.dataset.slotSeries;hiddenSlotSeries.has(key)?hiddenSlotSeries.delete(key):hiddenSlotSeries.add(key);drawSlotHistory(lastSlotPoints)}); +window.addEventListener('resize',()=>{drawHistory(lastHistoryPoints);drawSlotHistory(lastSlotPoints)});refreshHistory();setInterval(refreshHistory,15000); '''