Add OpenWebUI stability and privacy guards
This commit is contained in:
@@ -0,0 +1,171 @@
|
|||||||
|
#!/usr/bin/env python3
|
||||||
|
"""Offline tests for the versioned OpenWebUI filters.
|
||||||
|
|
||||||
|
The production container provides pydantic. A tiny local stand-in keeps these
|
||||||
|
logic tests dependency-free and prevents test setup from reaching the network.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import importlib.util
|
||||||
|
import json
|
||||||
|
import sys
|
||||||
|
import tempfile
|
||||||
|
import types
|
||||||
|
import unittest
|
||||||
|
from pathlib import Path
|
||||||
|
|
||||||
|
|
||||||
|
class _BaseModel:
|
||||||
|
def __init__(self, **values):
|
||||||
|
annotations = {}
|
||||||
|
for base in reversed(type(self).__mro__):
|
||||||
|
annotations.update(getattr(base, "__annotations__", {}))
|
||||||
|
for name in annotations:
|
||||||
|
setattr(self, name, values.get(name, getattr(type(self), name, None)))
|
||||||
|
|
||||||
|
|
||||||
|
fake_pydantic = types.ModuleType("pydantic")
|
||||||
|
fake_pydantic.BaseModel = _BaseModel
|
||||||
|
fake_pydantic.Field = lambda default=None, **kwargs: default
|
||||||
|
sys.modules.setdefault("pydantic", fake_pydantic)
|
||||||
|
|
||||||
|
FILTER_DIR = Path(__file__).parents[1] / "platform" / "openwebui" / "filters"
|
||||||
|
|
||||||
|
|
||||||
|
def _load(name: str):
|
||||||
|
spec = importlib.util.spec_from_file_location(name, FILTER_DIR / f"{name}.py")
|
||||||
|
module = importlib.util.module_from_spec(spec)
|
||||||
|
assert spec.loader is not None
|
||||||
|
spec.loader.exec_module(module)
|
||||||
|
return module
|
||||||
|
|
||||||
|
|
||||||
|
class StabilityGuardTests(unittest.IsolatedAsyncioTestCase):
|
||||||
|
async def asyncSetUp(self):
|
||||||
|
self.module = _load("stability_guard")
|
||||||
|
self.guard = self.module.Filter()
|
||||||
|
|
||||||
|
async def test_large_tool_output_is_bounded(self):
|
||||||
|
body = {
|
||||||
|
"model": "qwen-fast",
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": "Prüfe das Log."},
|
||||||
|
{"role": "tool", "tool_call_id": "x", "content": "A" * 50000},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
result = await self.guard.inlet(body)
|
||||||
|
content = result["messages"][1]["content"]
|
||||||
|
self.assertLessEqual(len(content), self.guard.valves.max_single_tool_chars)
|
||||||
|
self.assertIn("Werkzeugausgabe gekürzt", content)
|
||||||
|
|
||||||
|
async def test_duplicate_calls_disable_tools(self):
|
||||||
|
call = {
|
||||||
|
"id": "call",
|
||||||
|
"type": "function",
|
||||||
|
"function": {"name": "search", "arguments": '{"q":"same"}'},
|
||||||
|
}
|
||||||
|
body = {
|
||||||
|
"model": "qwen-fast",
|
||||||
|
"tools": [{"type": "function", "function": {"name": "search"}}],
|
||||||
|
"tool_ids": ["server:mcp:web"],
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": "Suche genau einmal."},
|
||||||
|
{"role": "assistant", "tool_calls": [call]},
|
||||||
|
{"role": "tool", "tool_call_id": "1", "content": "nichts"},
|
||||||
|
{"role": "assistant", "tool_calls": [call]},
|
||||||
|
{"role": "tool", "tool_call_id": "2", "content": "nichts"},
|
||||||
|
{"role": "assistant", "tool_calls": [call]},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
result = await self.guard.inlet(body)
|
||||||
|
self.assertEqual(result["tools"], [])
|
||||||
|
self.assertEqual(result["tool_ids"], [])
|
||||||
|
self.assertIn("Weitere Werkzeugaufrufe", result["messages"][0]["content"])
|
||||||
|
|
||||||
|
async def test_old_context_is_compacted_before_current_turn(self):
|
||||||
|
self.guard.valves.default_context_tokens = 10000
|
||||||
|
self.guard.valves.hard_context_ratio = 0.8
|
||||||
|
self.guard.valves.reserved_output_tokens = 1000
|
||||||
|
body = {
|
||||||
|
"model": "unknown",
|
||||||
|
"messages": [
|
||||||
|
{"role": "system", "content": "Sicher arbeiten."},
|
||||||
|
{"role": "user", "content": "alt " * 15000},
|
||||||
|
{"role": "assistant", "content": "altantwort " * 8000},
|
||||||
|
{"role": "tool", "tool_call_id": "old", "content": "log " * 20000},
|
||||||
|
{"role": "user", "content": "Aktuelle wichtige Frage"},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
result = await self.guard.inlet(body)
|
||||||
|
self.assertEqual(result["messages"][-1]["content"], "Aktuelle wichtige Frage")
|
||||||
|
self.assertLess(len(result["messages"][1]["content"]), 3000)
|
||||||
|
|
||||||
|
|
||||||
|
class MetricsTests(unittest.IsolatedAsyncioTestCase):
|
||||||
|
async def test_metrics_file_contains_no_chat_content_or_ids(self):
|
||||||
|
module = _load("local_performance_metrics")
|
||||||
|
metrics = module.Filter()
|
||||||
|
with tempfile.TemporaryDirectory() as directory:
|
||||||
|
path = Path(directory) / "metrics.jsonl"
|
||||||
|
metrics.valves.metrics_path = str(path)
|
||||||
|
metadata = {"message_id": "secret-message-id", "chat_id": "secret-chat-id"}
|
||||||
|
await metrics.inlet(
|
||||||
|
{
|
||||||
|
"model": "qwen-fast",
|
||||||
|
"messages": [{"role": "user", "content": "private prompt"}],
|
||||||
|
"tools": [{"name": "tool"}],
|
||||||
|
},
|
||||||
|
__metadata__=metadata,
|
||||||
|
)
|
||||||
|
await metrics.outlet(
|
||||||
|
{
|
||||||
|
"model": "qwen-fast",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "assistant",
|
||||||
|
"content": "private answer",
|
||||||
|
"usage": {
|
||||||
|
"prompt_tokens": 10,
|
||||||
|
"completion_tokens": 4,
|
||||||
|
"total_tokens": 14,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
],
|
||||||
|
},
|
||||||
|
__metadata__=metadata,
|
||||||
|
)
|
||||||
|
raw = path.read_text()
|
||||||
|
record = json.loads(raw)
|
||||||
|
self.assertEqual(record["prompt_tokens"], 10)
|
||||||
|
self.assertNotIn("private", raw)
|
||||||
|
self.assertNotIn("secret", raw)
|
||||||
|
|
||||||
|
|
||||||
|
class SecretRedactionTests(unittest.IsolatedAsyncioTestCase):
|
||||||
|
async def test_only_tool_and_assistant_content_is_redacted(self):
|
||||||
|
module = _load("secret_redaction")
|
||||||
|
guard = module.Filter()
|
||||||
|
token = "eyJ" + "A" * 24 + "." + "B" * 24 + "." + "C" * 16
|
||||||
|
body = {
|
||||||
|
"messages": [
|
||||||
|
{"role": "user", "content": f"Absichtlich lokal nutzen: {token}"},
|
||||||
|
{"role": "tool", "content": f'{{"api_key":"1234567890abcdef"}} {token}'},
|
||||||
|
]
|
||||||
|
}
|
||||||
|
result = await guard.inlet(body)
|
||||||
|
self.assertIn(token, result["messages"][0]["content"])
|
||||||
|
self.assertNotIn(token, result["messages"][1]["content"])
|
||||||
|
self.assertNotIn("1234567890abcdef", result["messages"][1]["content"])
|
||||||
|
|
||||||
|
outlet = {
|
||||||
|
"messages": [
|
||||||
|
{"role": "assistant", "content": "Bearer " + "abcdefghijklmnopqrstuvwxyz"}
|
||||||
|
]
|
||||||
|
}
|
||||||
|
result = await guard.outlet(outlet)
|
||||||
|
self.assertNotIn("abcdefghijklmnopqrstuvwxyz", result["messages"][0]["content"])
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
unittest.main()
|
||||||
@@ -35,6 +35,8 @@ Chats oder privaten Nutzdaten.
|
|||||||
| TinySearch-Volume | Das vom Installer vorbereitete Modellvolume wurde von Compose gleichzeitig als Compose-eigen behandelt. | Das Volume besitzt einen festen Namen und ist in Compose ausdrücklich als extern vorbereitet markiert. |
|
| TinySearch-Volume | Das vom Installer vorbereitete Modellvolume wurde von Compose gleichzeitig als Compose-eigen behandelt. | Das Volume besitzt einen festen Namen und ist in Compose ausdrücklich als extern vorbereitet markiert. |
|
||||||
| Reasoning-Filter | Beide globalen Filter besaßen Priorität 0; bei gleicher Priorität entschied die ID-Sortierung statt der gewünschten Logik. | `Reasoning Default Off` läuft mit Priorität 10 sicher vor dem optionalen `Thinking`-Override mit Priorität 20. Filter und sicherer Installer liegen versioniert im Repository. |
|
| Reasoning-Filter | Beide globalen Filter besaßen Priorität 0; bei gleicher Priorität entschied die ID-Sortierung statt der gewünschten Logik. | `Reasoning Default Off` läuft mit Priorität 10 sicher vor dem optionalen `Thinking`-Override mit Priorität 20. Filter und sicherer Installer liegen versioniert im Repository. |
|
||||||
| Folgefragen | OpenWebUI erzeugte nach Antworten zusätzliche Vorschläge und verbrauchte dafür einen weiteren Modellaufruf. | Folgefragengenerierung ist in der persistenten OpenWebUI-Konfiguration und im Compose-Standard deaktiviert. |
|
| Folgefragen | OpenWebUI erzeugte nach Antworten zusätzliche Vorschläge und verbrauchte dafür einen weiteren Modellaufruf. | Folgefragengenerierung ist in der persistenten OpenWebUI-Konfiguration und im Compose-Standard deaktiviert. |
|
||||||
|
| Werkzeug-/Kontextschutz | Große MCP-Antworten und wiederholte identische Aufrufe konnten Kontextfenster sprengen beziehungsweise Tool-Schleifen erzeugen. | Globaler Stability Guard begrenzt Resultate, verdichtet alte Inhalte profilabhängig und stoppt Wiederholungen; JSON-Schemas und Bilder werden nicht beschädigt. |
|
||||||
|
| Leistungsdaten | Benchmarkdaten sollten sichtbar sein, ohne private Chat-Inhalte zu protokollieren. | Ein globaler Abschlussfilter schreibt ausschließlich technische Zahlen in eine lokal rotierende JSONL-Datei und zeigt eine knappe Statuszeile. |
|
||||||
|
|
||||||
## Abnahmezustand am 21. August 2026
|
## Abnahmezustand am 21. August 2026
|
||||||
|
|
||||||
|
|||||||
+24
-2
@@ -1,6 +1,6 @@
|
|||||||
# Betrieb
|
# Betrieb
|
||||||
|
|
||||||
## Reasoning-Filter in OpenWebUI
|
## OpenWebUI-Filter und Stabilitätsschutz
|
||||||
|
|
||||||
Die versionierten Filter liegen unter `platform/openwebui/filters/`. Ihre
|
Die versionierten Filter liegen unter `platform/openwebui/filters/`. Ihre
|
||||||
Reihenfolge ist absichtlich festgelegt:
|
Reihenfolge ist absichtlich festgelegt:
|
||||||
@@ -9,9 +9,20 @@ Reihenfolge ist absichtlich festgelegt:
|
|||||||
`reasoning_effort=none`.
|
`reasoning_effort=none`.
|
||||||
2. `Thinking`, Priorität 20: läuft nur bei aktiviertem Brain-Schalter und
|
2. `Thinking`, Priorität 20: läuft nur bei aktiviertem Brain-Schalter und
|
||||||
überschreibt den Standard mit Low, Medium oder High.
|
überschreibt den Standard mit Low, Medium oder High.
|
||||||
|
3. `MikeAI Stability Guard`, Priorität 30: begrenzt einzelne und gesamte
|
||||||
|
Werkzeugresultate, verdichtet bei Bedarf zuerst alte Tool-Ausgaben und
|
||||||
|
Dialogteile und stoppt identische beziehungsweise ausufernde Tool-Schleifen.
|
||||||
|
4. `MikeAI Secret Redaction`, Priorität 40: entfernt übliche API-Keys, Tokens,
|
||||||
|
Passwörter, JWTs und private Schlüssel aus Tool-Ergebnissen, bevor sie das
|
||||||
|
Modell erreichen, sowie aus fertigen Modellantworten. Nutzereingaben und
|
||||||
|
Authentifizierungswege werden nicht verändert.
|
||||||
|
5. `MikeAI Local Performance Metrics`, Priorität 90: erfasst nach Abschluss
|
||||||
|
ausschließlich technische Zahlen wie Laufzeit, Tokenzähler, Token/s und
|
||||||
|
Tool-Anzahl. Nutzer-, Chat- und Nachrichten-IDs sowie sämtliche Textinhalte
|
||||||
|
werden weder geschrieben noch gehasht gespeichert.
|
||||||
|
|
||||||
OpenWebUI sortiert kleinere Prioritäten zuerst. Nach dem ersten Anlegen eines
|
OpenWebUI sortiert kleinere Prioritäten zuerst. Nach dem ersten Anlegen eines
|
||||||
Admin-Benutzers oder nach einer Datenwiederherstellung werden beide Filter mit
|
Admin-Benutzers oder nach einer Datenwiederherstellung werden alle Filter mit
|
||||||
einer vorherigen Datenbanksicherung installiert beziehungsweise aktualisiert:
|
einer vorherigen Datenbanksicherung installiert beziehungsweise aktualisiert:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
@@ -25,6 +36,17 @@ Folgefragen (`task.follow_up.enable=false`). Dieselbe Vorgabe steht zusätzlich
|
|||||||
als Container-Umgebungswert im Compose-Stack, damit bereits eine frische
|
als Container-Umgebungswert im Compose-Stack, damit bereits eine frische
|
||||||
OpenWebUI-Datenbank ohne Folgefragen startet.
|
OpenWebUI-Datenbank ohne Folgefragen startet.
|
||||||
|
|
||||||
|
Der Stabilitätsschutz kennt die drei Profilgrenzen 76.800, 94.208 und 131.072
|
||||||
|
Token. Er reserviert Ausgabetoken und greift vor der harten llama.cpp-Grenze
|
||||||
|
ein. Bilder bleiben unangetastet; JSON-Werkzeugschemas werden niemals
|
||||||
|
abgeschnitten. Sind allein die ausgewählten Schemas zu groß, wird der
|
||||||
|
Werkzeugzugriff nur für diesen Schritt deaktiviert und das Modell erhält eine
|
||||||
|
eindeutige Abschlussanweisung.
|
||||||
|
|
||||||
|
Die rotierende, inhaltsfreie Metrikdatei liegt im persistenten
|
||||||
|
OpenWebUI-Volume unter `mike-ai-request-metrics.jsonl` (maximal 5 MiB plus eine
|
||||||
|
Rotation). Sie darf für Benchmarks ausgewertet werden, ohne Chats auszulesen.
|
||||||
|
|
||||||
## Profile
|
## Profile
|
||||||
|
|
||||||
| Profil | Virtuelles Modell | Kontext | Zweck |
|
| Profil | Virtuelles Modell | Kontext | Zweck |
|
||||||
|
|||||||
@@ -0,0 +1,168 @@
|
|||||||
|
"""
|
||||||
|
title: MikeAI Local Performance Metrics
|
||||||
|
author: MikeAI
|
||||||
|
version: 1.0.0
|
||||||
|
description: Records content-free request timing and token counters locally.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import threading
|
||||||
|
import time
|
||||||
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
|
||||||
|
class Filter:
|
||||||
|
class Valves(BaseModel):
|
||||||
|
priority: int = 90
|
||||||
|
enabled: bool = True
|
||||||
|
show_status: bool = True
|
||||||
|
metrics_path: str = "/app/backend/data/mike-ai-request-metrics.jsonl"
|
||||||
|
rotate_bytes: int = 5 * 1024 * 1024
|
||||||
|
|
||||||
|
_started: dict[str, dict] = {}
|
||||||
|
_lock = threading.Lock()
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.valves = self.Valves()
|
||||||
|
self.toggle = False
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _key(metadata: dict | None) -> str | None:
|
||||||
|
if not isinstance(metadata, dict):
|
||||||
|
return None
|
||||||
|
return metadata.get("message_id") or metadata.get("id")
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _count_tool_calls(messages) -> int:
|
||||||
|
count = 0
|
||||||
|
for message in messages or []:
|
||||||
|
calls = message.get("tool_calls") if isinstance(message, dict) else None
|
||||||
|
if isinstance(calls, list):
|
||||||
|
count += len(calls)
|
||||||
|
elif isinstance(calls, dict):
|
||||||
|
count += 1
|
||||||
|
return count
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _numeric_metrics(message: dict) -> dict:
|
||||||
|
allowed = {
|
||||||
|
"prompt_tokens",
|
||||||
|
"completion_tokens",
|
||||||
|
"input_tokens",
|
||||||
|
"output_tokens",
|
||||||
|
"total_tokens",
|
||||||
|
"prompt_eval_count",
|
||||||
|
"eval_count",
|
||||||
|
"prompt_eval_duration",
|
||||||
|
"eval_duration",
|
||||||
|
"total_duration",
|
||||||
|
"load_duration",
|
||||||
|
"prompt_n",
|
||||||
|
"predicted_n",
|
||||||
|
"prompt_per_second",
|
||||||
|
"predicted_per_second",
|
||||||
|
"draft_n",
|
||||||
|
"draft_n_accepted",
|
||||||
|
}
|
||||||
|
result = {}
|
||||||
|
for container_name in ("usage", "info"):
|
||||||
|
container = message.get(container_name)
|
||||||
|
if not isinstance(container, dict):
|
||||||
|
continue
|
||||||
|
for key, value in container.items():
|
||||||
|
if key in allowed and isinstance(value, (int, float)) and not isinstance(value, bool):
|
||||||
|
result[key] = value
|
||||||
|
return result
|
||||||
|
|
||||||
|
def _write(self, record: dict) -> None:
|
||||||
|
path = self.valves.metrics_path
|
||||||
|
try:
|
||||||
|
if os.path.exists(path) and os.path.getsize(path) >= self.valves.rotate_bytes:
|
||||||
|
rotated = path + ".1"
|
||||||
|
if os.path.exists(rotated):
|
||||||
|
os.remove(rotated)
|
||||||
|
os.replace(path, rotated)
|
||||||
|
with open(path, "a", encoding="utf-8") as handle:
|
||||||
|
handle.write(json.dumps(record, sort_keys=True, separators=(",", ":")) + "\n")
|
||||||
|
except Exception:
|
||||||
|
# Metrics must never break a chat request.
|
||||||
|
pass
|
||||||
|
|
||||||
|
async def inlet(self, body: dict, __metadata__: dict = None, **kwargs) -> dict:
|
||||||
|
if not self.valves.enabled:
|
||||||
|
return body
|
||||||
|
key = self._key(__metadata__)
|
||||||
|
if key:
|
||||||
|
with self._lock:
|
||||||
|
self._started[key] = {
|
||||||
|
"started": time.monotonic(),
|
||||||
|
"model": str(body.get("model", "unknown"))[:80],
|
||||||
|
"tool_calls_before": self._count_tool_calls(body.get("messages")),
|
||||||
|
"tools_available": len(body.get("tools") or []),
|
||||||
|
}
|
||||||
|
if len(self._started) > 512:
|
||||||
|
oldest = next(iter(self._started))
|
||||||
|
self._started.pop(oldest, None)
|
||||||
|
return body
|
||||||
|
|
||||||
|
async def outlet(
|
||||||
|
self,
|
||||||
|
body: dict,
|
||||||
|
__metadata__: dict = None,
|
||||||
|
__event_emitter__=None,
|
||||||
|
**kwargs,
|
||||||
|
) -> dict:
|
||||||
|
if not self.valves.enabled:
|
||||||
|
return body
|
||||||
|
key = self._key(__metadata__)
|
||||||
|
with self._lock:
|
||||||
|
start = self._started.pop(key, None) if key else None
|
||||||
|
|
||||||
|
messages = body.get("messages") or []
|
||||||
|
assistant = next(
|
||||||
|
(m for m in reversed(messages) if isinstance(m, dict) and m.get("role") == "assistant"),
|
||||||
|
{},
|
||||||
|
)
|
||||||
|
elapsed = round(time.monotonic() - start["started"], 3) if start else None
|
||||||
|
numeric = self._numeric_metrics(assistant)
|
||||||
|
record = {
|
||||||
|
"timestamp": int(time.time()),
|
||||||
|
"model": (start or {}).get("model", str(body.get("model", "unknown"))[:80]),
|
||||||
|
"elapsed_seconds": elapsed,
|
||||||
|
"tool_calls_before": (start or {}).get("tool_calls_before", 0),
|
||||||
|
"tools_available": (start or {}).get("tools_available", 0),
|
||||||
|
**numeric,
|
||||||
|
}
|
||||||
|
# Deliberately absent: user/chat/message IDs and all textual content.
|
||||||
|
self._write(record)
|
||||||
|
|
||||||
|
if self.valves.show_status and __event_emitter__ is not None:
|
||||||
|
parts = []
|
||||||
|
if elapsed is not None:
|
||||||
|
parts.append(f"{elapsed:.1f} s")
|
||||||
|
input_tokens = numeric.get("prompt_tokens", numeric.get("input_tokens"))
|
||||||
|
output_tokens = numeric.get("completion_tokens", numeric.get("output_tokens"))
|
||||||
|
if input_tokens is not None:
|
||||||
|
parts.append(f"{int(input_tokens):,} Eingabetoken")
|
||||||
|
if output_tokens is not None:
|
||||||
|
parts.append(f"{int(output_tokens):,} Ausgabetoken")
|
||||||
|
speed = numeric.get("predicted_per_second")
|
||||||
|
if speed is not None:
|
||||||
|
parts.append(f"{speed:.1f} Token/s")
|
||||||
|
if parts:
|
||||||
|
try:
|
||||||
|
await __event_emitter__(
|
||||||
|
{
|
||||||
|
"type": "status",
|
||||||
|
"data": {
|
||||||
|
"description": "Lokale Leistung: " + " · ".join(parts),
|
||||||
|
"done": True,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
return body
|
||||||
@@ -0,0 +1,105 @@
|
|||||||
|
"""
|
||||||
|
title: MikeAI Secret Redaction
|
||||||
|
author: MikeAI
|
||||||
|
version: 1.0.0
|
||||||
|
description: Redacts common credentials from tool results and final assistant output.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import re
|
||||||
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
|
||||||
|
class Filter:
|
||||||
|
class Valves(BaseModel):
|
||||||
|
priority: int = 40
|
||||||
|
replacement: str = "[REDACTED]"
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.valves = self.Valves()
|
||||||
|
self.toggle = False
|
||||||
|
|
||||||
|
def _redact(self, text: str) -> tuple[str, int]:
|
||||||
|
if not isinstance(text, str) or not text:
|
||||||
|
return text, 0
|
||||||
|
count = 0
|
||||||
|
|
||||||
|
def replace_full(match):
|
||||||
|
nonlocal count
|
||||||
|
count += 1
|
||||||
|
return self.valves.replacement
|
||||||
|
|
||||||
|
def replace_value(match):
|
||||||
|
nonlocal count
|
||||||
|
count += 1
|
||||||
|
return match.group(1) + self.valves.replacement
|
||||||
|
|
||||||
|
patterns_full = [
|
||||||
|
r"-----BEGIN (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----.*?-----END (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----",
|
||||||
|
r"\beyJ[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{10,}\b",
|
||||||
|
]
|
||||||
|
for pattern in patterns_full:
|
||||||
|
text = re.sub(pattern, replace_full, text, flags=re.IGNORECASE | re.DOTALL)
|
||||||
|
|
||||||
|
patterns_value = [
|
||||||
|
r"((?:authorization\s*[:=]\s*)?(?:bearer\s+))[A-Za-z0-9._~+/=-]{16,}",
|
||||||
|
r"((?:\"|')?(?:api[_-]?key|access[_-]?token|refresh[_-]?token|password|passwd|secret|token)(?:\"|')?\s*[:=]\s*(?:\"|')?)[^\s\"',;&}]{8,}",
|
||||||
|
r"([?&](?:api[_-]?key|access[_-]?token|token|key)=)[^&\s]+",
|
||||||
|
]
|
||||||
|
for pattern in patterns_value:
|
||||||
|
text = re.sub(pattern, replace_value, text, flags=re.IGNORECASE)
|
||||||
|
return text, count
|
||||||
|
|
||||||
|
def _redact_content(self, content) -> tuple[object, int]:
|
||||||
|
if isinstance(content, str):
|
||||||
|
return self._redact(content)
|
||||||
|
if not isinstance(content, list):
|
||||||
|
return content, 0
|
||||||
|
result = []
|
||||||
|
total = 0
|
||||||
|
for part in content:
|
||||||
|
if isinstance(part, dict) and isinstance(part.get("text"), str):
|
||||||
|
item = dict(part)
|
||||||
|
item["text"], found = self._redact(item["text"])
|
||||||
|
total += found
|
||||||
|
result.append(item)
|
||||||
|
else:
|
||||||
|
result.append(part)
|
||||||
|
return result, total
|
||||||
|
|
||||||
|
async def _notify(self, emitter, count: int) -> None:
|
||||||
|
if not count or emitter is None:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
await emitter(
|
||||||
|
{
|
||||||
|
"type": "status",
|
||||||
|
"data": {
|
||||||
|
"description": f"Secret-Schutz: {count} mögliche Zugangsdaten entfernt.",
|
||||||
|
"done": True,
|
||||||
|
},
|
||||||
|
}
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
async def inlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict:
|
||||||
|
total = 0
|
||||||
|
for message in body.get("messages") or []:
|
||||||
|
if message.get("role") != "tool":
|
||||||
|
continue
|
||||||
|
message["content"], count = self._redact_content(message.get("content", ""))
|
||||||
|
total += count
|
||||||
|
await self._notify(__event_emitter__, total)
|
||||||
|
return body
|
||||||
|
|
||||||
|
async def outlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict:
|
||||||
|
total = 0
|
||||||
|
for message in body.get("messages") or []:
|
||||||
|
if message.get("role") != "assistant":
|
||||||
|
continue
|
||||||
|
message["content"], count = self._redact_content(message.get("content", ""))
|
||||||
|
total += count
|
||||||
|
await self._notify(__event_emitter__, total)
|
||||||
|
return body
|
||||||
@@ -0,0 +1,302 @@
|
|||||||
|
"""
|
||||||
|
title: MikeAI Stability Guard
|
||||||
|
author: MikeAI
|
||||||
|
version: 1.0.0
|
||||||
|
description: Bounds tool output and context use and breaks repeated tool-call loops.
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import json
|
||||||
|
from collections import Counter
|
||||||
|
from pydantic import BaseModel
|
||||||
|
|
||||||
|
|
||||||
|
class Filter:
|
||||||
|
class Valves(BaseModel):
|
||||||
|
priority: int = 30
|
||||||
|
fast_context_tokens: int = 76800
|
||||||
|
medium_context_tokens: int = 94208
|
||||||
|
long_context_tokens: int = 131072
|
||||||
|
default_context_tokens: int = 76800
|
||||||
|
soft_context_ratio: float = 0.70
|
||||||
|
hard_context_ratio: float = 0.84
|
||||||
|
reserved_output_tokens: int = 8192
|
||||||
|
max_single_tool_chars: int = 18000
|
||||||
|
max_total_tool_chars: int = 60000
|
||||||
|
compacted_tool_chars: int = 3000
|
||||||
|
duplicate_tool_call_limit: int = 3
|
||||||
|
max_tool_calls_per_turn: int = 16
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.valves = self.Valves()
|
||||||
|
self.toggle = False
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _text(value) -> str:
|
||||||
|
if isinstance(value, str):
|
||||||
|
return value
|
||||||
|
try:
|
||||||
|
return json.dumps(value, ensure_ascii=False, separators=(",", ":"))
|
||||||
|
except Exception:
|
||||||
|
return str(value)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _compact_text(text: str, limit: int, reason: str) -> str:
|
||||||
|
if len(text) <= limit:
|
||||||
|
return text
|
||||||
|
marker = f"\n\n[MikeAI: {reason}; Originalumfang {len(text)} Zeichen]\n\n"
|
||||||
|
available = max(256, limit - len(marker))
|
||||||
|
head = int(available * 0.72)
|
||||||
|
tail = available - head
|
||||||
|
return text[:head] + marker + text[-tail:]
|
||||||
|
|
||||||
|
def _compact_content(self, content, limit: int, reason: str):
|
||||||
|
if isinstance(content, str):
|
||||||
|
return self._compact_text(content, limit, reason)
|
||||||
|
if not isinstance(content, list):
|
||||||
|
return content
|
||||||
|
|
||||||
|
result = []
|
||||||
|
text_budget = limit
|
||||||
|
for part in content:
|
||||||
|
if not isinstance(part, dict):
|
||||||
|
result.append(part)
|
||||||
|
continue
|
||||||
|
item = dict(part)
|
||||||
|
if isinstance(item.get("text"), str):
|
||||||
|
new_text = self._compact_text(
|
||||||
|
item["text"], max(256, text_budget), reason
|
||||||
|
)
|
||||||
|
item["text"] = new_text
|
||||||
|
text_budget = max(0, text_budget - len(new_text))
|
||||||
|
# Image URLs/data are deliberately preserved. Truncating base64
|
||||||
|
# would corrupt the request and character count is not a useful
|
||||||
|
# estimate of vision tokens.
|
||||||
|
result.append(item)
|
||||||
|
return result
|
||||||
|
|
||||||
|
def _content_chars(self, content) -> int:
|
||||||
|
if isinstance(content, str):
|
||||||
|
return len(content)
|
||||||
|
if not isinstance(content, list):
|
||||||
|
return len(self._text(content))
|
||||||
|
total = 0
|
||||||
|
for part in content:
|
||||||
|
if isinstance(part, dict):
|
||||||
|
if isinstance(part.get("text"), str):
|
||||||
|
total += len(part["text"])
|
||||||
|
if part.get("type") in {"image", "image_url", "input_image"}:
|
||||||
|
total += 6144 # conservative fixed vision-token proxy
|
||||||
|
else:
|
||||||
|
total += len(self._text(part))
|
||||||
|
return total
|
||||||
|
|
||||||
|
def _estimate_tokens(self, body: dict) -> int:
|
||||||
|
# Conservative local estimate for mixed German/English, JSON and code.
|
||||||
|
chars = 0
|
||||||
|
for message in body.get("messages") or []:
|
||||||
|
chars += 24
|
||||||
|
chars += self._content_chars(message.get("content", ""))
|
||||||
|
for key in ("tool_calls", "function_call", "output"):
|
||||||
|
if key in message:
|
||||||
|
chars += len(self._text(message[key]))
|
||||||
|
if body.get("tools"):
|
||||||
|
chars += len(self._text(body["tools"]))
|
||||||
|
return max(1, (chars + 2) // 3)
|
||||||
|
|
||||||
|
def _context_limit(self, model: str) -> int:
|
||||||
|
model = (model or "").lower()
|
||||||
|
if "long" in model or "large" in model:
|
||||||
|
return self.valves.long_context_tokens
|
||||||
|
if "medium" in model:
|
||||||
|
return self.valves.medium_context_tokens
|
||||||
|
if "fast" in model:
|
||||||
|
return self.valves.fast_context_tokens
|
||||||
|
return self.valves.default_context_tokens
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _tool_signatures(message: dict) -> list[str]:
|
||||||
|
calls = message.get("tool_calls") or []
|
||||||
|
if isinstance(calls, dict):
|
||||||
|
calls = [calls]
|
||||||
|
signatures = []
|
||||||
|
for call in calls:
|
||||||
|
if not isinstance(call, dict):
|
||||||
|
continue
|
||||||
|
function = call.get("function") or call
|
||||||
|
name = function.get("name") or call.get("name") or "unknown"
|
||||||
|
arguments = function.get("arguments") or call.get("arguments") or ""
|
||||||
|
if not isinstance(arguments, str):
|
||||||
|
try:
|
||||||
|
arguments = json.dumps(arguments, sort_keys=True, separators=(",", ":"))
|
||||||
|
except Exception:
|
||||||
|
arguments = str(arguments)
|
||||||
|
signatures.append(f"{name}:{arguments}")
|
||||||
|
return signatures
|
||||||
|
|
||||||
|
def _current_turn_tool_signatures(self, messages: list[dict]) -> list[str]:
|
||||||
|
last_user = -1
|
||||||
|
for index, message in enumerate(messages):
|
||||||
|
if message.get("role") == "user":
|
||||||
|
last_user = index
|
||||||
|
result = []
|
||||||
|
for message in messages[last_user + 1 :]:
|
||||||
|
if message.get("role") == "assistant":
|
||||||
|
result.extend(self._tool_signatures(message))
|
||||||
|
return result
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _add_guard_instruction(body: dict, reason: str) -> None:
|
||||||
|
instruction = (
|
||||||
|
"MikeAI-Sicherheitsgrenze: Weitere Werkzeugaufrufe sind in diesem "
|
||||||
|
f"Schritt gesperrt ({reason}). Fasse die bereits vorhandenen Ergebnisse "
|
||||||
|
"zusammen, benenne fehlende Belege ehrlich und frage nicht in einer "
|
||||||
|
"Werkzeugschleife weiter."
|
||||||
|
)
|
||||||
|
messages = body.setdefault("messages", [])
|
||||||
|
for message in messages:
|
||||||
|
if message.get("role") == "system" and isinstance(message.get("content"), str):
|
||||||
|
message["content"] += "\n\n" + instruction
|
||||||
|
return
|
||||||
|
messages.insert(0, {"role": "system", "content": instruction})
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _disable_tools(body: dict) -> None:
|
||||||
|
body["tools"] = []
|
||||||
|
body["tool_ids"] = []
|
||||||
|
metadata = body.get("metadata")
|
||||||
|
if isinstance(metadata, dict):
|
||||||
|
metadata["tool_ids"] = []
|
||||||
|
metadata["tool_servers"] = []
|
||||||
|
|
||||||
|
async def _notify(self, emitter, description: str) -> None:
|
||||||
|
if emitter is None:
|
||||||
|
return
|
||||||
|
try:
|
||||||
|
await emitter(
|
||||||
|
{
|
||||||
|
"type": "status",
|
||||||
|
"data": {"description": description, "done": True},
|
||||||
|
}
|
||||||
|
)
|
||||||
|
except Exception:
|
||||||
|
pass
|
||||||
|
|
||||||
|
def _bound_tool_outputs(self, body: dict) -> tuple[int, int]:
|
||||||
|
messages = body.get("messages") or []
|
||||||
|
changed = 0
|
||||||
|
for message in messages:
|
||||||
|
if message.get("role") != "tool":
|
||||||
|
continue
|
||||||
|
size = self._content_chars(message.get("content", ""))
|
||||||
|
if size > self.valves.max_single_tool_chars:
|
||||||
|
message["content"] = self._compact_content(
|
||||||
|
message.get("content", ""),
|
||||||
|
self.valves.max_single_tool_chars,
|
||||||
|
"Werkzeugausgabe gekürzt",
|
||||||
|
)
|
||||||
|
changed += 1
|
||||||
|
|
||||||
|
tool_messages = [m for m in messages if m.get("role") == "tool"]
|
||||||
|
total = sum(self._content_chars(m.get("content", "")) for m in tool_messages)
|
||||||
|
for message in tool_messages:
|
||||||
|
if total <= self.valves.max_total_tool_chars:
|
||||||
|
break
|
||||||
|
old = self._content_chars(message.get("content", ""))
|
||||||
|
message["content"] = self._compact_content(
|
||||||
|
message.get("content", ""),
|
||||||
|
self.valves.compacted_tool_chars,
|
||||||
|
"älteres Werkzeugresultat wegen Gesamtbudget verdichtet",
|
||||||
|
)
|
||||||
|
new = self._content_chars(message.get("content", ""))
|
||||||
|
total -= max(0, old - new)
|
||||||
|
changed += 1
|
||||||
|
return changed, total
|
||||||
|
|
||||||
|
def _fit_context(self, body: dict, hard_budget: int) -> int:
|
||||||
|
messages = body.get("messages") or []
|
||||||
|
# Old tool results are the safest material to compact. Preserve the
|
||||||
|
# message and tool_call_id so the OpenAI tool-call sequence stays valid.
|
||||||
|
for message in messages:
|
||||||
|
if self._estimate_tokens(body) <= hard_budget:
|
||||||
|
break
|
||||||
|
if message.get("role") == "tool":
|
||||||
|
message["content"] = self._compact_content(
|
||||||
|
message.get("content", ""),
|
||||||
|
768,
|
||||||
|
"älteres Werkzeugresultat wegen Kontextgrenze entfernt",
|
||||||
|
)
|
||||||
|
|
||||||
|
# If necessary, compact dialogue before the current user turn. Never
|
||||||
|
# touch system text, the current turn or image data.
|
||||||
|
last_user = max(
|
||||||
|
(index for index, message in enumerate(messages) if message.get("role") == "user"),
|
||||||
|
default=len(messages),
|
||||||
|
)
|
||||||
|
cutoff = last_user
|
||||||
|
for message in messages[:cutoff]:
|
||||||
|
if self._estimate_tokens(body) <= hard_budget:
|
||||||
|
break
|
||||||
|
if message.get("role") in {"user", "assistant"} and not message.get("tool_calls"):
|
||||||
|
message["content"] = self._compact_content(
|
||||||
|
message.get("content", ""),
|
||||||
|
1500,
|
||||||
|
"älterer Dialog wegen Kontextgrenze verdichtet",
|
||||||
|
)
|
||||||
|
return self._estimate_tokens(body)
|
||||||
|
|
||||||
|
async def inlet(
|
||||||
|
self,
|
||||||
|
body: dict,
|
||||||
|
__event_emitter__=None,
|
||||||
|
**kwargs,
|
||||||
|
) -> dict:
|
||||||
|
changed, _ = self._bound_tool_outputs(body)
|
||||||
|
messages = body.get("messages") or []
|
||||||
|
signatures = self._current_turn_tool_signatures(messages)
|
||||||
|
counts = Counter(signatures)
|
||||||
|
duplicate = max(counts.values(), default=0)
|
||||||
|
|
||||||
|
breaker_reason = None
|
||||||
|
if duplicate >= self.valves.duplicate_tool_call_limit:
|
||||||
|
breaker_reason = f"derselbe Aufruf wurde {duplicate}-mal wiederholt"
|
||||||
|
elif len(signatures) >= self.valves.max_tool_calls_per_turn:
|
||||||
|
breaker_reason = f"{len(signatures)} Werkzeugaufrufe in einem Schritt"
|
||||||
|
|
||||||
|
if breaker_reason:
|
||||||
|
self._disable_tools(body)
|
||||||
|
self._add_guard_instruction(body, breaker_reason)
|
||||||
|
await self._notify(
|
||||||
|
__event_emitter__, f"Werkzeugschleife gestoppt: {breaker_reason}."
|
||||||
|
)
|
||||||
|
|
||||||
|
context = self._context_limit(body.get("model", ""))
|
||||||
|
hard_budget = max(
|
||||||
|
4096,
|
||||||
|
int(context * self.valves.hard_context_ratio)
|
||||||
|
- self.valves.reserved_output_tokens,
|
||||||
|
)
|
||||||
|
estimate = self._estimate_tokens(body)
|
||||||
|
if estimate > hard_budget:
|
||||||
|
estimate = self._fit_context(body, hard_budget)
|
||||||
|
if estimate > hard_budget and body.get("tools"):
|
||||||
|
# Never truncate JSON schemas: a damaged schema can make the
|
||||||
|
# model emit invalid calls. If schemas themselves no longer
|
||||||
|
# fit, finish this turn without tools and explain why.
|
||||||
|
self._disable_tools(body)
|
||||||
|
self._add_guard_instruction(
|
||||||
|
body,
|
||||||
|
"die ausgewählten Werkzeugdefinitionen überschreiten das Kontextbudget",
|
||||||
|
)
|
||||||
|
estimate = self._estimate_tokens(body)
|
||||||
|
await self._notify(
|
||||||
|
__event_emitter__,
|
||||||
|
f"Kontextschutz aktiv: Eingabe auf ungefähr {estimate:,} Token verdichtet.",
|
||||||
|
)
|
||||||
|
elif changed:
|
||||||
|
await self._notify(
|
||||||
|
__event_emitter__,
|
||||||
|
f"{changed} große Werkzeugausgabe(n) platzsparend verdichtet.",
|
||||||
|
)
|
||||||
|
return body
|
||||||
@@ -8,7 +8,7 @@ FILTER_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")/filters" && pwd)
|
|||||||
|
|
||||||
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
|
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
|
||||||
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
|
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
|
||||||
for file in reasoning_default_off.py thinking.py; do
|
for file in reasoning_default_off.py thinking.py stability_guard.py secret_redaction.py local_performance_metrics.py; do
|
||||||
[[ -s $FILTER_DIR/$file ]] || die "Filterdatei fehlt: $file"
|
[[ -s $FILTER_DIR/$file ]] || die "Filterdatei fehlt: $file"
|
||||||
done
|
done
|
||||||
|
|
||||||
@@ -56,8 +56,11 @@ if not {"key", "value", "updated_at"} <= config_columns:
|
|||||||
owner = requested_owner
|
owner = requested_owner
|
||||||
if not owner:
|
if not owner:
|
||||||
existing = con.execute(
|
existing = con.execute(
|
||||||
"select user_id from function where id in (?, ?) order by id limit 1",
|
"select user_id from function where id in (?, ?, ?, ?, ?) order by id limit 1",
|
||||||
("reasoning_default_off", "thinking"),
|
(
|
||||||
|
"reasoning_default_off", "thinking", "stability_guard",
|
||||||
|
"secret_redaction", "local_performance_metrics",
|
||||||
|
),
|
||||||
).fetchone()
|
).fetchone()
|
||||||
if existing:
|
if existing:
|
||||||
owner = existing[0]
|
owner = existing[0]
|
||||||
@@ -73,6 +76,9 @@ now = int(time.time())
|
|||||||
filters = [
|
filters = [
|
||||||
("reasoning_default_off", "Reasoning Default Off", 10),
|
("reasoning_default_off", "Reasoning Default Off", 10),
|
||||||
("thinking", "Thinking", 20),
|
("thinking", "Thinking", 20),
|
||||||
|
("stability_guard", "MikeAI Stability Guard", 30),
|
||||||
|
("secret_redaction", "MikeAI Secret Redaction", 40),
|
||||||
|
("local_performance_metrics", "MikeAI Local Performance Metrics", 90),
|
||||||
]
|
]
|
||||||
with con:
|
with con:
|
||||||
for function_id, name, priority in filters:
|
for function_id, name, priority in filters:
|
||||||
@@ -105,7 +111,10 @@ with con:
|
|||||||
""",
|
""",
|
||||||
("task.follow_up.enable", "false", now),
|
("task.follow_up.enable", "false", now),
|
||||||
)
|
)
|
||||||
print("OpenWebUI konfiguriert: Default Off=10, Thinking=20, Folgefragen=aus")
|
print(
|
||||||
|
"OpenWebUI konfiguriert: Default Off=10, Thinking=20, "
|
||||||
|
"Stability Guard=30, Secret Redaction=40, Local Metrics=90, Folgefragen=aus"
|
||||||
|
)
|
||||||
PY
|
PY
|
||||||
|
|
||||||
if [[ $was_running == true ]]; then
|
if [[ $was_running == true ]]; then
|
||||||
|
|||||||
Reference in New Issue
Block a user