From 54b9daae91bb670cedd83e0282faee9830bfe1b6 Mon Sep 17 00:00:00 2001
From: Mikei386 <44135113+Mikei386@users.noreply.github.com>
Date: Tue, 15 Sep 2026 23:05:00 +0200
Subject: [PATCH] Add slot context history to dashboard
---
README.md | 2 +-
dev/test_dashboard_modes.py | 34 +++++++++++
docs/CONTAINER_INVENTORY.md | 2 +-
platform/llama-dashboard/app.py | 99 +++++++++++++++++++++++++++++++--
4 files changed, 131 insertions(+), 6 deletions(-)
diff --git a/README.md b/README.md
index 6ec2002..5980432 100644
--- a/README.md
+++ b/README.md
@@ -21,7 +21,7 @@ Bild- und Sprachausgabe. **Hermes und die Fach-MCPs laufen auf Unraid.**
- FLUX.2 Klein 9B FP8 Beta: Transformer/VAE auf RTX 5080, Textencoder auf RTX 3060
- Qwen3-TTS 1.7B auf der RTX 3060 hinter dem TTS-Gateway; kein Piper-Fallback
- Whisper.cpp `small` auf der CPU für lokale deutsche Spracherkennung
-- Live-Dashboard mit 21 Tagen Detailhistorie auf Port 8099
+- Live-Dashboard mit 21 Tagen Detailhistorie für GPUs und Slot-Kontextbelegung auf Port 8099
- Portainer CE als optionale Container-Ansicht auf Port 9443
- WireGuard-Gateway, Datenbackup und Athena-Operator
- keine produktive Hermes-, OpenWebUI- oder portable Fach-MCP-Instanz
diff --git a/dev/test_dashboard_modes.py b/dev/test_dashboard_modes.py
index 08e92bd..948401c 100644
--- a/dev/test_dashboard_modes.py
+++ b/dev/test_dashboard_modes.py
@@ -118,6 +118,40 @@ class DashboardModeTests(unittest.TestCase):
self.assertEqual(backups[0]["sha256"], "a" * 64)
self.assertTrue(backups[0]["download_url"].startswith("/api/backups/download/"))
+ def test_slot_context_history_is_recorded_and_exposed(self):
+ with tempfile.TemporaryDirectory() as tempdir:
+ store = self.dashboard.HistoryStore(Path(tempdir) / "slots.sqlite3")
+ store.record({
+ "timestamp": 1_789_000_000,
+ "router": {
+ "current_profile": "medium",
+ "upstream": {"model": "qwen"},
+ "llama_telemetry": {
+ "slots": [{
+ "id": 0,
+ "context_used": 40_000,
+ "n_ctx": 160_000,
+ "processing": True,
+ }],
+ },
+ },
+ "llama_runtime": {"pid": 123, "model_file": "qwen.gguf"},
+ "gpus": [],
+ })
+
+ result = store.query("all")
+ store._db.close()
+
+ self.assertEqual(len(result["slot_points"]), 1)
+ self.assertEqual(result["slot_points"][0]["slot_id"], 0)
+ self.assertEqual(result["slot_points"][0]["context_percent"], 25.0)
+ self.assertEqual(result["slot_points"][0]["busy_ratio"], 1.0)
+
+ def test_dashboard_renders_slot_context_history_controls(self):
+ self.assertIn('id="slotHistoryChart"', self.dashboard.HTML)
+ self.assertIn('id="slotHistoryRanges"', self.dashboard.HTML)
+ self.assertIn("drawSlotHistory(d.slot_points||[])", self.dashboard.HISTORY_JS)
+
if __name__ == "__main__":
unittest.main()
diff --git a/docs/CONTAINER_INVENTORY.md b/docs/CONTAINER_INVENTORY.md
index d939afb..79ed809 100644
--- a/docs/CONTAINER_INVENTORY.md
+++ b/docs/CONTAINER_INVENTORY.md
@@ -13,7 +13,7 @@ nicht automatisch ein ungenutzter Rest.
| `mike-ai-backup` | kein Modell; Offen Docker Volume Backup `v2` (Digest vom 15. September 2026) | Sichert `/data`, `/etc/mike-ai`, den Stack und die persistenten Docker-Volumes im Fünf-Stunden-Takt. |
| `mike-ai-applio-studio` | Applio/RVC; Stimmenmodelle werden nutzerseitig ergänzt | Vollständige RVC-Oberfläche für Inferenz, Modellverwaltung und Training auf der RTX 5080. Für eine Konvertierung ist ein importiertes oder trainiertes `.pth`-Modell nötig; eine Referenzaufnahme allein reicht nicht. |
| `mike-ai-image-worker` | FLUX.2 Klein 9B FP8, Qwen3-8B NF4 Textencoder und VAE | Erzeugt und bearbeitet Bilder transaktional; nutzt während eines Auftrags RTX 5080 und RTX 3060. |
-| `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung und Betriebsarten an und bietet die Modusumschaltung. |
+| `mike-ai-llama-dashboard` | kein Modell | Zeigt Telemetrie, Profile, GPU-Nutzung, Slot-Kontextbelegung mit Verlauf und Betriebsarten an und bietet die Modusumschaltung. |
| `mike-ai-llama-fast` | Qwen3.8-27B `IQ4-MIX`, Qwen-MMProj BF16 | Schnelles Q4-Text-/Vision-Profil mit 76.800 Token Kontext. |
| `mike-ai-llama-large` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Q4-Text-/Vision-Profil mit 192.000 Token Kontext und Verteilung auf beide GPUs. |
| `mike-ai-llama-medium` | Qwen3.8-27B `IQ4_XS-pure`, Qwen-MMProj BF16 | Standard-Q4-Text-/Vision-Profil mit 160.000 Token Kontext und Verteilung auf beide GPUs. |
diff --git a/platform/llama-dashboard/app.py b/platform/llama-dashboard/app.py
index da4d4b3..32a0be0 100644
--- a/platform/llama-dashboard/app.py
+++ b/platform/llama-dashboard/app.py
@@ -500,6 +500,19 @@ class HistoryStore:
temperature_sum REAL, temperature_max REAL, power_sum REAL, samples INTEGER NOT NULL,
PRIMARY KEY(hour_ts, gpu_index)
);
+ CREATE TABLE IF NOT EXISTS slot_samples_raw (
+ ts INTEGER NOT NULL, slot_id INTEGER NOT NULL,
+ context_used REAL, context_total REAL, processing INTEGER NOT NULL DEFAULT 0,
+ profile TEXT, model TEXT
+ );
+ CREATE INDEX IF NOT EXISTS idx_slot_samples_raw_ts ON slot_samples_raw(ts);
+ CREATE TABLE IF NOT EXISTS slot_samples_hourly (
+ hour_ts INTEGER NOT NULL, slot_id INTEGER NOT NULL,
+ context_percent_sum REAL, context_percent_max REAL,
+ context_used_max REAL, context_total_max REAL,
+ busy_samples INTEGER NOT NULL, samples INTEGER NOT NULL,
+ PRIMARY KEY(hour_ts, slot_id)
+ );
CREATE TABLE IF NOT EXISTS token_totals (
id INTEGER PRIMARY KEY CHECK(id=1), prompt_tokens INTEGER NOT NULL DEFAULT 0,
cached_tokens INTEGER NOT NULL DEFAULT 0, output_tokens INTEGER NOT NULL DEFAULT 0
@@ -552,6 +565,18 @@ class HistoryStore:
gpu.get("temperature_c"), gpu.get("power_w"), profile, model),
)
+ slots = ((router.get("llama_telemetry") or {}).get("slots") or [])
+ for slot in slots:
+ if slot.get("id") is None:
+ continue
+ used = max(0, float(slot.get("context_used") or 0))
+ total = max(0, float(slot.get("n_ctx") or 0))
+ self._db.execute(
+ "INSERT INTO slot_samples_raw VALUES(?,?,?,?,?,?,?)",
+ (ts, int(slot["id"]), used, total, int(bool(slot.get("processing"))),
+ profile, model),
+ )
+
previous_runtime = self._meta("counter_runtime")
deltas: list[int] = []
for key in self.TOKEN_KEYS:
@@ -605,6 +630,22 @@ class HistoryStore:
samples=samples+excluded.samples
""", (cutoff,))
self._db.execute("DELETE FROM samples_raw WHERE ts < ?", (cutoff,))
+ self._db.execute("""
+ INSERT INTO slot_samples_hourly
+ SELECT ts-ts%3600, slot_id,
+ SUM(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END),
+ MAX(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END),
+ MAX(context_used), MAX(context_total), SUM(processing), COUNT(*)
+ FROM slot_samples_raw WHERE ts < ? GROUP BY ts-ts%3600, slot_id
+ ON CONFLICT(hour_ts,slot_id) DO UPDATE SET
+ context_percent_sum=context_percent_sum+excluded.context_percent_sum,
+ context_percent_max=MAX(context_percent_max,excluded.context_percent_max),
+ context_used_max=MAX(context_used_max,excluded.context_used_max),
+ context_total_max=MAX(context_total_max,excluded.context_total_max),
+ busy_samples=busy_samples+excluded.busy_samples,
+ samples=samples+excluded.samples
+ """, (cutoff,))
+ self._db.execute("DELETE FROM slot_samples_raw WHERE ts < ?", (cutoff,))
def query(self, range_name: str) -> dict[str, Any]:
ranges = {
@@ -617,6 +658,7 @@ class HistoryStore:
detail_cutoff = now - DETAIL_RETENTION_DAYS * 86400
with self._lock:
points: list[dict[str, Any]] = []
+ slot_points: list[dict[str, Any]] = []
if start < detail_cutoff:
for row in self._db.execute("""
SELECT hour_ts ts,gpu_index,gpu_util_sum/samples gpu_util,gpu_util_max,
@@ -635,6 +677,25 @@ class HistoryStore:
FROM samples_raw WHERE ts>=? GROUP BY (ts/{bucket}),gpu_index ORDER BY ts,gpu_index
""", (raw_start,)):
points.append(dict(row))
+ if start < detail_cutoff:
+ for row in self._db.execute("""
+ SELECT hour_ts ts,slot_id,context_percent_sum/samples context_percent,
+ context_percent_max,context_used_max context_used,
+ context_total_max context_total,
+ 1.0*busy_samples/samples busy_ratio
+ FROM slot_samples_hourly WHERE hour_ts>=? ORDER BY hour_ts,slot_id
+ """, (start,)):
+ slot_points.append(dict(row))
+ for row in self._db.execute(f"""
+ SELECT (ts/{bucket})*{bucket} ts,slot_id,
+ AVG(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END) context_percent,
+ MAX(CASE WHEN context_total>0 THEN 100.0*context_used/context_total ELSE 0 END) context_percent_max,
+ MAX(context_used) context_used,MAX(context_total) context_total,
+ AVG(processing) busy_ratio
+ FROM slot_samples_raw WHERE ts>=?
+ GROUP BY (ts/{bucket}),slot_id ORDER BY ts,slot_id
+ """, (raw_start,)):
+ slot_points.append(dict(row))
totals = dict(self._db.execute("SELECT * FROM token_totals WHERE id=1").fetchone())
range_tokens = dict(self._db.execute(
"SELECT COALESCE(SUM(prompt_tokens),0) prompt_tokens, COALESCE(SUM(cached_tokens),0) cached_tokens, COALESCE(SUM(output_tokens),0) output_tokens FROM token_hourly WHERE hour_ts>=?",
@@ -652,17 +713,23 @@ class HistoryStore:
raw_info = dict(self._db.execute(
"SELECT COUNT(*) rows, MIN(ts) oldest, MAX(ts) newest FROM samples_raw"
).fetchone())
+ slot_raw_info = dict(self._db.execute(
+ "SELECT COUNT(*) rows, MIN(ts) oldest, MAX(ts) newest FROM slot_samples_raw"
+ ).fetchone())
points.sort(key=lambda item: (item["ts"], item["gpu_index"]))
+ slot_points.sort(key=lambda item: (item["ts"], item["slot_id"]))
return {
"range": range_name if range_name in ranges else "24h",
"detail_retention_days": DETAIL_RETENTION_DAYS,
"sample_interval_seconds": HISTORY_INTERVAL,
"points": points,
+ "slot_points": slot_points,
"token_totals": totals,
"range_tokens": range_tokens,
"profile_usage": profile_usage,
"model_events": events,
"storage": raw_info,
+ "slot_storage": slot_raw_info,
}
@@ -709,7 +776,7 @@ HTML = r'''