Add OpenWebUI stability and privacy guards
This commit is contained in:
@@ -0,0 +1,171 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Offline tests for the versioned OpenWebUI filters.
|
||||
|
||||
The production container provides pydantic. A tiny local stand-in keeps these
|
||||
logic tests dependency-free and prevents test setup from reaching the network.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import importlib.util
|
||||
import json
|
||||
import sys
|
||||
import tempfile
|
||||
import types
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
class _BaseModel:
|
||||
def __init__(self, **values):
|
||||
annotations = {}
|
||||
for base in reversed(type(self).__mro__):
|
||||
annotations.update(getattr(base, "__annotations__", {}))
|
||||
for name in annotations:
|
||||
setattr(self, name, values.get(name, getattr(type(self), name, None)))
|
||||
|
||||
|
||||
fake_pydantic = types.ModuleType("pydantic")
|
||||
fake_pydantic.BaseModel = _BaseModel
|
||||
fake_pydantic.Field = lambda default=None, **kwargs: default
|
||||
sys.modules.setdefault("pydantic", fake_pydantic)
|
||||
|
||||
FILTER_DIR = Path(__file__).parents[1] / "platform" / "openwebui" / "filters"
|
||||
|
||||
|
||||
def _load(name: str):
|
||||
spec = importlib.util.spec_from_file_location(name, FILTER_DIR / f"{name}.py")
|
||||
module = importlib.util.module_from_spec(spec)
|
||||
assert spec.loader is not None
|
||||
spec.loader.exec_module(module)
|
||||
return module
|
||||
|
||||
|
||||
class StabilityGuardTests(unittest.IsolatedAsyncioTestCase):
|
||||
async def asyncSetUp(self):
|
||||
self.module = _load("stability_guard")
|
||||
self.guard = self.module.Filter()
|
||||
|
||||
async def test_large_tool_output_is_bounded(self):
|
||||
body = {
|
||||
"model": "qwen-fast",
|
||||
"messages": [
|
||||
{"role": "user", "content": "Prüfe das Log."},
|
||||
{"role": "tool", "tool_call_id": "x", "content": "A" * 50000},
|
||||
],
|
||||
}
|
||||
result = await self.guard.inlet(body)
|
||||
content = result["messages"][1]["content"]
|
||||
self.assertLessEqual(len(content), self.guard.valves.max_single_tool_chars)
|
||||
self.assertIn("Werkzeugausgabe gekürzt", content)
|
||||
|
||||
async def test_duplicate_calls_disable_tools(self):
|
||||
call = {
|
||||
"id": "call",
|
||||
"type": "function",
|
||||
"function": {"name": "search", "arguments": '{"q":"same"}'},
|
||||
}
|
||||
body = {
|
||||
"model": "qwen-fast",
|
||||
"tools": [{"type": "function", "function": {"name": "search"}}],
|
||||
"tool_ids": ["server:mcp:web"],
|
||||
"messages": [
|
||||
{"role": "user", "content": "Suche genau einmal."},
|
||||
{"role": "assistant", "tool_calls": [call]},
|
||||
{"role": "tool", "tool_call_id": "1", "content": "nichts"},
|
||||
{"role": "assistant", "tool_calls": [call]},
|
||||
{"role": "tool", "tool_call_id": "2", "content": "nichts"},
|
||||
{"role": "assistant", "tool_calls": [call]},
|
||||
],
|
||||
}
|
||||
result = await self.guard.inlet(body)
|
||||
self.assertEqual(result["tools"], [])
|
||||
self.assertEqual(result["tool_ids"], [])
|
||||
self.assertIn("Weitere Werkzeugaufrufe", result["messages"][0]["content"])
|
||||
|
||||
async def test_old_context_is_compacted_before_current_turn(self):
|
||||
self.guard.valves.default_context_tokens = 10000
|
||||
self.guard.valves.hard_context_ratio = 0.8
|
||||
self.guard.valves.reserved_output_tokens = 1000
|
||||
body = {
|
||||
"model": "unknown",
|
||||
"messages": [
|
||||
{"role": "system", "content": "Sicher arbeiten."},
|
||||
{"role": "user", "content": "alt " * 15000},
|
||||
{"role": "assistant", "content": "altantwort " * 8000},
|
||||
{"role": "tool", "tool_call_id": "old", "content": "log " * 20000},
|
||||
{"role": "user", "content": "Aktuelle wichtige Frage"},
|
||||
],
|
||||
}
|
||||
result = await self.guard.inlet(body)
|
||||
self.assertEqual(result["messages"][-1]["content"], "Aktuelle wichtige Frage")
|
||||
self.assertLess(len(result["messages"][1]["content"]), 3000)
|
||||
|
||||
|
||||
class MetricsTests(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_metrics_file_contains_no_chat_content_or_ids(self):
|
||||
module = _load("local_performance_metrics")
|
||||
metrics = module.Filter()
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
path = Path(directory) / "metrics.jsonl"
|
||||
metrics.valves.metrics_path = str(path)
|
||||
metadata = {"message_id": "secret-message-id", "chat_id": "secret-chat-id"}
|
||||
await metrics.inlet(
|
||||
{
|
||||
"model": "qwen-fast",
|
||||
"messages": [{"role": "user", "content": "private prompt"}],
|
||||
"tools": [{"name": "tool"}],
|
||||
},
|
||||
__metadata__=metadata,
|
||||
)
|
||||
await metrics.outlet(
|
||||
{
|
||||
"model": "qwen-fast",
|
||||
"messages": [
|
||||
{
|
||||
"role": "assistant",
|
||||
"content": "private answer",
|
||||
"usage": {
|
||||
"prompt_tokens": 10,
|
||||
"completion_tokens": 4,
|
||||
"total_tokens": 14,
|
||||
},
|
||||
}
|
||||
],
|
||||
},
|
||||
__metadata__=metadata,
|
||||
)
|
||||
raw = path.read_text()
|
||||
record = json.loads(raw)
|
||||
self.assertEqual(record["prompt_tokens"], 10)
|
||||
self.assertNotIn("private", raw)
|
||||
self.assertNotIn("secret", raw)
|
||||
|
||||
|
||||
class SecretRedactionTests(unittest.IsolatedAsyncioTestCase):
|
||||
async def test_only_tool_and_assistant_content_is_redacted(self):
|
||||
module = _load("secret_redaction")
|
||||
guard = module.Filter()
|
||||
token = "eyJ" + "A" * 24 + "." + "B" * 24 + "." + "C" * 16
|
||||
body = {
|
||||
"messages": [
|
||||
{"role": "user", "content": f"Absichtlich lokal nutzen: {token}"},
|
||||
{"role": "tool", "content": f'{{"api_key":"1234567890abcdef"}} {token}'},
|
||||
]
|
||||
}
|
||||
result = await guard.inlet(body)
|
||||
self.assertIn(token, result["messages"][0]["content"])
|
||||
self.assertNotIn(token, result["messages"][1]["content"])
|
||||
self.assertNotIn("1234567890abcdef", result["messages"][1]["content"])
|
||||
|
||||
outlet = {
|
||||
"messages": [
|
||||
{"role": "assistant", "content": "Bearer " + "abcdefghijklmnopqrstuvwxyz"}
|
||||
]
|
||||
}
|
||||
result = await guard.outlet(outlet)
|
||||
self.assertNotIn("abcdefghijklmnopqrstuvwxyz", result["messages"][0]["content"])
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
@@ -35,6 +35,8 @@ Chats oder privaten Nutzdaten.
|
||||
| TinySearch-Volume | Das vom Installer vorbereitete Modellvolume wurde von Compose gleichzeitig als Compose-eigen behandelt. | Das Volume besitzt einen festen Namen und ist in Compose ausdrücklich als extern vorbereitet markiert. |
|
||||
| Reasoning-Filter | Beide globalen Filter besaßen Priorität 0; bei gleicher Priorität entschied die ID-Sortierung statt der gewünschten Logik. | `Reasoning Default Off` läuft mit Priorität 10 sicher vor dem optionalen `Thinking`-Override mit Priorität 20. Filter und sicherer Installer liegen versioniert im Repository. |
|
||||
| Folgefragen | OpenWebUI erzeugte nach Antworten zusätzliche Vorschläge und verbrauchte dafür einen weiteren Modellaufruf. | Folgefragengenerierung ist in der persistenten OpenWebUI-Konfiguration und im Compose-Standard deaktiviert. |
|
||||
| Werkzeug-/Kontextschutz | Große MCP-Antworten und wiederholte identische Aufrufe konnten Kontextfenster sprengen beziehungsweise Tool-Schleifen erzeugen. | Globaler Stability Guard begrenzt Resultate, verdichtet alte Inhalte profilabhängig und stoppt Wiederholungen; JSON-Schemas und Bilder werden nicht beschädigt. |
|
||||
| Leistungsdaten | Benchmarkdaten sollten sichtbar sein, ohne private Chat-Inhalte zu protokollieren. | Ein globaler Abschlussfilter schreibt ausschließlich technische Zahlen in eine lokal rotierende JSONL-Datei und zeigt eine knappe Statuszeile. |
|
||||
|
||||
## Abnahmezustand am 21. August 2026
|
||||
|
||||
|
||||
+24
-2
@@ -1,6 +1,6 @@
|
||||
# Betrieb
|
||||
|
||||
## Reasoning-Filter in OpenWebUI
|
||||
## OpenWebUI-Filter und Stabilitätsschutz
|
||||
|
||||
Die versionierten Filter liegen unter `platform/openwebui/filters/`. Ihre
|
||||
Reihenfolge ist absichtlich festgelegt:
|
||||
@@ -9,9 +9,20 @@ Reihenfolge ist absichtlich festgelegt:
|
||||
`reasoning_effort=none`.
|
||||
2. `Thinking`, Priorität 20: läuft nur bei aktiviertem Brain-Schalter und
|
||||
überschreibt den Standard mit Low, Medium oder High.
|
||||
3. `MikeAI Stability Guard`, Priorität 30: begrenzt einzelne und gesamte
|
||||
Werkzeugresultate, verdichtet bei Bedarf zuerst alte Tool-Ausgaben und
|
||||
Dialogteile und stoppt identische beziehungsweise ausufernde Tool-Schleifen.
|
||||
4. `MikeAI Secret Redaction`, Priorität 40: entfernt übliche API-Keys, Tokens,
|
||||
Passwörter, JWTs und private Schlüssel aus Tool-Ergebnissen, bevor sie das
|
||||
Modell erreichen, sowie aus fertigen Modellantworten. Nutzereingaben und
|
||||
Authentifizierungswege werden nicht verändert.
|
||||
5. `MikeAI Local Performance Metrics`, Priorität 90: erfasst nach Abschluss
|
||||
ausschließlich technische Zahlen wie Laufzeit, Tokenzähler, Token/s und
|
||||
Tool-Anzahl. Nutzer-, Chat- und Nachrichten-IDs sowie sämtliche Textinhalte
|
||||
werden weder geschrieben noch gehasht gespeichert.
|
||||
|
||||
OpenWebUI sortiert kleinere Prioritäten zuerst. Nach dem ersten Anlegen eines
|
||||
Admin-Benutzers oder nach einer Datenwiederherstellung werden beide Filter mit
|
||||
Admin-Benutzers oder nach einer Datenwiederherstellung werden alle Filter mit
|
||||
einer vorherigen Datenbanksicherung installiert beziehungsweise aktualisiert:
|
||||
|
||||
```bash
|
||||
@@ -25,6 +36,17 @@ Folgefragen (`task.follow_up.enable=false`). Dieselbe Vorgabe steht zusätzlich
|
||||
als Container-Umgebungswert im Compose-Stack, damit bereits eine frische
|
||||
OpenWebUI-Datenbank ohne Folgefragen startet.
|
||||
|
||||
Der Stabilitätsschutz kennt die drei Profilgrenzen 76.800, 94.208 und 131.072
|
||||
Token. Er reserviert Ausgabetoken und greift vor der harten llama.cpp-Grenze
|
||||
ein. Bilder bleiben unangetastet; JSON-Werkzeugschemas werden niemals
|
||||
abgeschnitten. Sind allein die ausgewählten Schemas zu groß, wird der
|
||||
Werkzeugzugriff nur für diesen Schritt deaktiviert und das Modell erhält eine
|
||||
eindeutige Abschlussanweisung.
|
||||
|
||||
Die rotierende, inhaltsfreie Metrikdatei liegt im persistenten
|
||||
OpenWebUI-Volume unter `mike-ai-request-metrics.jsonl` (maximal 5 MiB plus eine
|
||||
Rotation). Sie darf für Benchmarks ausgewertet werden, ohne Chats auszulesen.
|
||||
|
||||
## Profile
|
||||
|
||||
| Profil | Virtuelles Modell | Kontext | Zweck |
|
||||
|
||||
@@ -0,0 +1,168 @@
|
||||
"""
|
||||
title: MikeAI Local Performance Metrics
|
||||
author: MikeAI
|
||||
version: 1.0.0
|
||||
description: Records content-free request timing and token counters locally.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import threading
|
||||
import time
|
||||
from pydantic import BaseModel
|
||||
|
||||
|
||||
class Filter:
|
||||
class Valves(BaseModel):
|
||||
priority: int = 90
|
||||
enabled: bool = True
|
||||
show_status: bool = True
|
||||
metrics_path: str = "/app/backend/data/mike-ai-request-metrics.jsonl"
|
||||
rotate_bytes: int = 5 * 1024 * 1024
|
||||
|
||||
_started: dict[str, dict] = {}
|
||||
_lock = threading.Lock()
|
||||
|
||||
def __init__(self):
|
||||
self.valves = self.Valves()
|
||||
self.toggle = False
|
||||
|
||||
@staticmethod
|
||||
def _key(metadata: dict | None) -> str | None:
|
||||
if not isinstance(metadata, dict):
|
||||
return None
|
||||
return metadata.get("message_id") or metadata.get("id")
|
||||
|
||||
@staticmethod
|
||||
def _count_tool_calls(messages) -> int:
|
||||
count = 0
|
||||
for message in messages or []:
|
||||
calls = message.get("tool_calls") if isinstance(message, dict) else None
|
||||
if isinstance(calls, list):
|
||||
count += len(calls)
|
||||
elif isinstance(calls, dict):
|
||||
count += 1
|
||||
return count
|
||||
|
||||
@staticmethod
|
||||
def _numeric_metrics(message: dict) -> dict:
|
||||
allowed = {
|
||||
"prompt_tokens",
|
||||
"completion_tokens",
|
||||
"input_tokens",
|
||||
"output_tokens",
|
||||
"total_tokens",
|
||||
"prompt_eval_count",
|
||||
"eval_count",
|
||||
"prompt_eval_duration",
|
||||
"eval_duration",
|
||||
"total_duration",
|
||||
"load_duration",
|
||||
"prompt_n",
|
||||
"predicted_n",
|
||||
"prompt_per_second",
|
||||
"predicted_per_second",
|
||||
"draft_n",
|
||||
"draft_n_accepted",
|
||||
}
|
||||
result = {}
|
||||
for container_name in ("usage", "info"):
|
||||
container = message.get(container_name)
|
||||
if not isinstance(container, dict):
|
||||
continue
|
||||
for key, value in container.items():
|
||||
if key in allowed and isinstance(value, (int, float)) and not isinstance(value, bool):
|
||||
result[key] = value
|
||||
return result
|
||||
|
||||
def _write(self, record: dict) -> None:
|
||||
path = self.valves.metrics_path
|
||||
try:
|
||||
if os.path.exists(path) and os.path.getsize(path) >= self.valves.rotate_bytes:
|
||||
rotated = path + ".1"
|
||||
if os.path.exists(rotated):
|
||||
os.remove(rotated)
|
||||
os.replace(path, rotated)
|
||||
with open(path, "a", encoding="utf-8") as handle:
|
||||
handle.write(json.dumps(record, sort_keys=True, separators=(",", ":")) + "\n")
|
||||
except Exception:
|
||||
# Metrics must never break a chat request.
|
||||
pass
|
||||
|
||||
async def inlet(self, body: dict, __metadata__: dict = None, **kwargs) -> dict:
|
||||
if not self.valves.enabled:
|
||||
return body
|
||||
key = self._key(__metadata__)
|
||||
if key:
|
||||
with self._lock:
|
||||
self._started[key] = {
|
||||
"started": time.monotonic(),
|
||||
"model": str(body.get("model", "unknown"))[:80],
|
||||
"tool_calls_before": self._count_tool_calls(body.get("messages")),
|
||||
"tools_available": len(body.get("tools") or []),
|
||||
}
|
||||
if len(self._started) > 512:
|
||||
oldest = next(iter(self._started))
|
||||
self._started.pop(oldest, None)
|
||||
return body
|
||||
|
||||
async def outlet(
|
||||
self,
|
||||
body: dict,
|
||||
__metadata__: dict = None,
|
||||
__event_emitter__=None,
|
||||
**kwargs,
|
||||
) -> dict:
|
||||
if not self.valves.enabled:
|
||||
return body
|
||||
key = self._key(__metadata__)
|
||||
with self._lock:
|
||||
start = self._started.pop(key, None) if key else None
|
||||
|
||||
messages = body.get("messages") or []
|
||||
assistant = next(
|
||||
(m for m in reversed(messages) if isinstance(m, dict) and m.get("role") == "assistant"),
|
||||
{},
|
||||
)
|
||||
elapsed = round(time.monotonic() - start["started"], 3) if start else None
|
||||
numeric = self._numeric_metrics(assistant)
|
||||
record = {
|
||||
"timestamp": int(time.time()),
|
||||
"model": (start or {}).get("model", str(body.get("model", "unknown"))[:80]),
|
||||
"elapsed_seconds": elapsed,
|
||||
"tool_calls_before": (start or {}).get("tool_calls_before", 0),
|
||||
"tools_available": (start or {}).get("tools_available", 0),
|
||||
**numeric,
|
||||
}
|
||||
# Deliberately absent: user/chat/message IDs and all textual content.
|
||||
self._write(record)
|
||||
|
||||
if self.valves.show_status and __event_emitter__ is not None:
|
||||
parts = []
|
||||
if elapsed is not None:
|
||||
parts.append(f"{elapsed:.1f} s")
|
||||
input_tokens = numeric.get("prompt_tokens", numeric.get("input_tokens"))
|
||||
output_tokens = numeric.get("completion_tokens", numeric.get("output_tokens"))
|
||||
if input_tokens is not None:
|
||||
parts.append(f"{int(input_tokens):,} Eingabetoken")
|
||||
if output_tokens is not None:
|
||||
parts.append(f"{int(output_tokens):,} Ausgabetoken")
|
||||
speed = numeric.get("predicted_per_second")
|
||||
if speed is not None:
|
||||
parts.append(f"{speed:.1f} Token/s")
|
||||
if parts:
|
||||
try:
|
||||
await __event_emitter__(
|
||||
{
|
||||
"type": "status",
|
||||
"data": {
|
||||
"description": "Lokale Leistung: " + " · ".join(parts),
|
||||
"done": True,
|
||||
},
|
||||
}
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
return body
|
||||
@@ -0,0 +1,105 @@
|
||||
"""
|
||||
title: MikeAI Secret Redaction
|
||||
author: MikeAI
|
||||
version: 1.0.0
|
||||
description: Redacts common credentials from tool results and final assistant output.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pydantic import BaseModel
|
||||
|
||||
|
||||
class Filter:
|
||||
class Valves(BaseModel):
|
||||
priority: int = 40
|
||||
replacement: str = "[REDACTED]"
|
||||
|
||||
def __init__(self):
|
||||
self.valves = self.Valves()
|
||||
self.toggle = False
|
||||
|
||||
def _redact(self, text: str) -> tuple[str, int]:
|
||||
if not isinstance(text, str) or not text:
|
||||
return text, 0
|
||||
count = 0
|
||||
|
||||
def replace_full(match):
|
||||
nonlocal count
|
||||
count += 1
|
||||
return self.valves.replacement
|
||||
|
||||
def replace_value(match):
|
||||
nonlocal count
|
||||
count += 1
|
||||
return match.group(1) + self.valves.replacement
|
||||
|
||||
patterns_full = [
|
||||
r"-----BEGIN (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----.*?-----END (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----",
|
||||
r"\beyJ[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{10,}\b",
|
||||
]
|
||||
for pattern in patterns_full:
|
||||
text = re.sub(pattern, replace_full, text, flags=re.IGNORECASE | re.DOTALL)
|
||||
|
||||
patterns_value = [
|
||||
r"((?:authorization\s*[:=]\s*)?(?:bearer\s+))[A-Za-z0-9._~+/=-]{16,}",
|
||||
r"((?:\"|')?(?:api[_-]?key|access[_-]?token|refresh[_-]?token|password|passwd|secret|token)(?:\"|')?\s*[:=]\s*(?:\"|')?)[^\s\"',;&}]{8,}",
|
||||
r"([?&](?:api[_-]?key|access[_-]?token|token|key)=)[^&\s]+",
|
||||
]
|
||||
for pattern in patterns_value:
|
||||
text = re.sub(pattern, replace_value, text, flags=re.IGNORECASE)
|
||||
return text, count
|
||||
|
||||
def _redact_content(self, content) -> tuple[object, int]:
|
||||
if isinstance(content, str):
|
||||
return self._redact(content)
|
||||
if not isinstance(content, list):
|
||||
return content, 0
|
||||
result = []
|
||||
total = 0
|
||||
for part in content:
|
||||
if isinstance(part, dict) and isinstance(part.get("text"), str):
|
||||
item = dict(part)
|
||||
item["text"], found = self._redact(item["text"])
|
||||
total += found
|
||||
result.append(item)
|
||||
else:
|
||||
result.append(part)
|
||||
return result, total
|
||||
|
||||
async def _notify(self, emitter, count: int) -> None:
|
||||
if not count or emitter is None:
|
||||
return
|
||||
try:
|
||||
await emitter(
|
||||
{
|
||||
"type": "status",
|
||||
"data": {
|
||||
"description": f"Secret-Schutz: {count} mögliche Zugangsdaten entfernt.",
|
||||
"done": True,
|
||||
},
|
||||
}
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
async def inlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict:
|
||||
total = 0
|
||||
for message in body.get("messages") or []:
|
||||
if message.get("role") != "tool":
|
||||
continue
|
||||
message["content"], count = self._redact_content(message.get("content", ""))
|
||||
total += count
|
||||
await self._notify(__event_emitter__, total)
|
||||
return body
|
||||
|
||||
async def outlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict:
|
||||
total = 0
|
||||
for message in body.get("messages") or []:
|
||||
if message.get("role") != "assistant":
|
||||
continue
|
||||
message["content"], count = self._redact_content(message.get("content", ""))
|
||||
total += count
|
||||
await self._notify(__event_emitter__, total)
|
||||
return body
|
||||
@@ -0,0 +1,302 @@
|
||||
"""
|
||||
title: MikeAI Stability Guard
|
||||
author: MikeAI
|
||||
version: 1.0.0
|
||||
description: Bounds tool output and context use and breaks repeated tool-call loops.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
from collections import Counter
|
||||
from pydantic import BaseModel
|
||||
|
||||
|
||||
class Filter:
|
||||
class Valves(BaseModel):
|
||||
priority: int = 30
|
||||
fast_context_tokens: int = 76800
|
||||
medium_context_tokens: int = 94208
|
||||
long_context_tokens: int = 131072
|
||||
default_context_tokens: int = 76800
|
||||
soft_context_ratio: float = 0.70
|
||||
hard_context_ratio: float = 0.84
|
||||
reserved_output_tokens: int = 8192
|
||||
max_single_tool_chars: int = 18000
|
||||
max_total_tool_chars: int = 60000
|
||||
compacted_tool_chars: int = 3000
|
||||
duplicate_tool_call_limit: int = 3
|
||||
max_tool_calls_per_turn: int = 16
|
||||
|
||||
def __init__(self):
|
||||
self.valves = self.Valves()
|
||||
self.toggle = False
|
||||
|
||||
@staticmethod
|
||||
def _text(value) -> str:
|
||||
if isinstance(value, str):
|
||||
return value
|
||||
try:
|
||||
return json.dumps(value, ensure_ascii=False, separators=(",", ":"))
|
||||
except Exception:
|
||||
return str(value)
|
||||
|
||||
@staticmethod
|
||||
def _compact_text(text: str, limit: int, reason: str) -> str:
|
||||
if len(text) <= limit:
|
||||
return text
|
||||
marker = f"\n\n[MikeAI: {reason}; Originalumfang {len(text)} Zeichen]\n\n"
|
||||
available = max(256, limit - len(marker))
|
||||
head = int(available * 0.72)
|
||||
tail = available - head
|
||||
return text[:head] + marker + text[-tail:]
|
||||
|
||||
def _compact_content(self, content, limit: int, reason: str):
|
||||
if isinstance(content, str):
|
||||
return self._compact_text(content, limit, reason)
|
||||
if not isinstance(content, list):
|
||||
return content
|
||||
|
||||
result = []
|
||||
text_budget = limit
|
||||
for part in content:
|
||||
if not isinstance(part, dict):
|
||||
result.append(part)
|
||||
continue
|
||||
item = dict(part)
|
||||
if isinstance(item.get("text"), str):
|
||||
new_text = self._compact_text(
|
||||
item["text"], max(256, text_budget), reason
|
||||
)
|
||||
item["text"] = new_text
|
||||
text_budget = max(0, text_budget - len(new_text))
|
||||
# Image URLs/data are deliberately preserved. Truncating base64
|
||||
# would corrupt the request and character count is not a useful
|
||||
# estimate of vision tokens.
|
||||
result.append(item)
|
||||
return result
|
||||
|
||||
def _content_chars(self, content) -> int:
|
||||
if isinstance(content, str):
|
||||
return len(content)
|
||||
if not isinstance(content, list):
|
||||
return len(self._text(content))
|
||||
total = 0
|
||||
for part in content:
|
||||
if isinstance(part, dict):
|
||||
if isinstance(part.get("text"), str):
|
||||
total += len(part["text"])
|
||||
if part.get("type") in {"image", "image_url", "input_image"}:
|
||||
total += 6144 # conservative fixed vision-token proxy
|
||||
else:
|
||||
total += len(self._text(part))
|
||||
return total
|
||||
|
||||
def _estimate_tokens(self, body: dict) -> int:
|
||||
# Conservative local estimate for mixed German/English, JSON and code.
|
||||
chars = 0
|
||||
for message in body.get("messages") or []:
|
||||
chars += 24
|
||||
chars += self._content_chars(message.get("content", ""))
|
||||
for key in ("tool_calls", "function_call", "output"):
|
||||
if key in message:
|
||||
chars += len(self._text(message[key]))
|
||||
if body.get("tools"):
|
||||
chars += len(self._text(body["tools"]))
|
||||
return max(1, (chars + 2) // 3)
|
||||
|
||||
def _context_limit(self, model: str) -> int:
|
||||
model = (model or "").lower()
|
||||
if "long" in model or "large" in model:
|
||||
return self.valves.long_context_tokens
|
||||
if "medium" in model:
|
||||
return self.valves.medium_context_tokens
|
||||
if "fast" in model:
|
||||
return self.valves.fast_context_tokens
|
||||
return self.valves.default_context_tokens
|
||||
|
||||
@staticmethod
|
||||
def _tool_signatures(message: dict) -> list[str]:
|
||||
calls = message.get("tool_calls") or []
|
||||
if isinstance(calls, dict):
|
||||
calls = [calls]
|
||||
signatures = []
|
||||
for call in calls:
|
||||
if not isinstance(call, dict):
|
||||
continue
|
||||
function = call.get("function") or call
|
||||
name = function.get("name") or call.get("name") or "unknown"
|
||||
arguments = function.get("arguments") or call.get("arguments") or ""
|
||||
if not isinstance(arguments, str):
|
||||
try:
|
||||
arguments = json.dumps(arguments, sort_keys=True, separators=(",", ":"))
|
||||
except Exception:
|
||||
arguments = str(arguments)
|
||||
signatures.append(f"{name}:{arguments}")
|
||||
return signatures
|
||||
|
||||
def _current_turn_tool_signatures(self, messages: list[dict]) -> list[str]:
|
||||
last_user = -1
|
||||
for index, message in enumerate(messages):
|
||||
if message.get("role") == "user":
|
||||
last_user = index
|
||||
result = []
|
||||
for message in messages[last_user + 1 :]:
|
||||
if message.get("role") == "assistant":
|
||||
result.extend(self._tool_signatures(message))
|
||||
return result
|
||||
|
||||
@staticmethod
|
||||
def _add_guard_instruction(body: dict, reason: str) -> None:
|
||||
instruction = (
|
||||
"MikeAI-Sicherheitsgrenze: Weitere Werkzeugaufrufe sind in diesem "
|
||||
f"Schritt gesperrt ({reason}). Fasse die bereits vorhandenen Ergebnisse "
|
||||
"zusammen, benenne fehlende Belege ehrlich und frage nicht in einer "
|
||||
"Werkzeugschleife weiter."
|
||||
)
|
||||
messages = body.setdefault("messages", [])
|
||||
for message in messages:
|
||||
if message.get("role") == "system" and isinstance(message.get("content"), str):
|
||||
message["content"] += "\n\n" + instruction
|
||||
return
|
||||
messages.insert(0, {"role": "system", "content": instruction})
|
||||
|
||||
@staticmethod
|
||||
def _disable_tools(body: dict) -> None:
|
||||
body["tools"] = []
|
||||
body["tool_ids"] = []
|
||||
metadata = body.get("metadata")
|
||||
if isinstance(metadata, dict):
|
||||
metadata["tool_ids"] = []
|
||||
metadata["tool_servers"] = []
|
||||
|
||||
async def _notify(self, emitter, description: str) -> None:
|
||||
if emitter is None:
|
||||
return
|
||||
try:
|
||||
await emitter(
|
||||
{
|
||||
"type": "status",
|
||||
"data": {"description": description, "done": True},
|
||||
}
|
||||
)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
def _bound_tool_outputs(self, body: dict) -> tuple[int, int]:
|
||||
messages = body.get("messages") or []
|
||||
changed = 0
|
||||
for message in messages:
|
||||
if message.get("role") != "tool":
|
||||
continue
|
||||
size = self._content_chars(message.get("content", ""))
|
||||
if size > self.valves.max_single_tool_chars:
|
||||
message["content"] = self._compact_content(
|
||||
message.get("content", ""),
|
||||
self.valves.max_single_tool_chars,
|
||||
"Werkzeugausgabe gekürzt",
|
||||
)
|
||||
changed += 1
|
||||
|
||||
tool_messages = [m for m in messages if m.get("role") == "tool"]
|
||||
total = sum(self._content_chars(m.get("content", "")) for m in tool_messages)
|
||||
for message in tool_messages:
|
||||
if total <= self.valves.max_total_tool_chars:
|
||||
break
|
||||
old = self._content_chars(message.get("content", ""))
|
||||
message["content"] = self._compact_content(
|
||||
message.get("content", ""),
|
||||
self.valves.compacted_tool_chars,
|
||||
"älteres Werkzeugresultat wegen Gesamtbudget verdichtet",
|
||||
)
|
||||
new = self._content_chars(message.get("content", ""))
|
||||
total -= max(0, old - new)
|
||||
changed += 1
|
||||
return changed, total
|
||||
|
||||
def _fit_context(self, body: dict, hard_budget: int) -> int:
|
||||
messages = body.get("messages") or []
|
||||
# Old tool results are the safest material to compact. Preserve the
|
||||
# message and tool_call_id so the OpenAI tool-call sequence stays valid.
|
||||
for message in messages:
|
||||
if self._estimate_tokens(body) <= hard_budget:
|
||||
break
|
||||
if message.get("role") == "tool":
|
||||
message["content"] = self._compact_content(
|
||||
message.get("content", ""),
|
||||
768,
|
||||
"älteres Werkzeugresultat wegen Kontextgrenze entfernt",
|
||||
)
|
||||
|
||||
# If necessary, compact dialogue before the current user turn. Never
|
||||
# touch system text, the current turn or image data.
|
||||
last_user = max(
|
||||
(index for index, message in enumerate(messages) if message.get("role") == "user"),
|
||||
default=len(messages),
|
||||
)
|
||||
cutoff = last_user
|
||||
for message in messages[:cutoff]:
|
||||
if self._estimate_tokens(body) <= hard_budget:
|
||||
break
|
||||
if message.get("role") in {"user", "assistant"} and not message.get("tool_calls"):
|
||||
message["content"] = self._compact_content(
|
||||
message.get("content", ""),
|
||||
1500,
|
||||
"älterer Dialog wegen Kontextgrenze verdichtet",
|
||||
)
|
||||
return self._estimate_tokens(body)
|
||||
|
||||
async def inlet(
|
||||
self,
|
||||
body: dict,
|
||||
__event_emitter__=None,
|
||||
**kwargs,
|
||||
) -> dict:
|
||||
changed, _ = self._bound_tool_outputs(body)
|
||||
messages = body.get("messages") or []
|
||||
signatures = self._current_turn_tool_signatures(messages)
|
||||
counts = Counter(signatures)
|
||||
duplicate = max(counts.values(), default=0)
|
||||
|
||||
breaker_reason = None
|
||||
if duplicate >= self.valves.duplicate_tool_call_limit:
|
||||
breaker_reason = f"derselbe Aufruf wurde {duplicate}-mal wiederholt"
|
||||
elif len(signatures) >= self.valves.max_tool_calls_per_turn:
|
||||
breaker_reason = f"{len(signatures)} Werkzeugaufrufe in einem Schritt"
|
||||
|
||||
if breaker_reason:
|
||||
self._disable_tools(body)
|
||||
self._add_guard_instruction(body, breaker_reason)
|
||||
await self._notify(
|
||||
__event_emitter__, f"Werkzeugschleife gestoppt: {breaker_reason}."
|
||||
)
|
||||
|
||||
context = self._context_limit(body.get("model", ""))
|
||||
hard_budget = max(
|
||||
4096,
|
||||
int(context * self.valves.hard_context_ratio)
|
||||
- self.valves.reserved_output_tokens,
|
||||
)
|
||||
estimate = self._estimate_tokens(body)
|
||||
if estimate > hard_budget:
|
||||
estimate = self._fit_context(body, hard_budget)
|
||||
if estimate > hard_budget and body.get("tools"):
|
||||
# Never truncate JSON schemas: a damaged schema can make the
|
||||
# model emit invalid calls. If schemas themselves no longer
|
||||
# fit, finish this turn without tools and explain why.
|
||||
self._disable_tools(body)
|
||||
self._add_guard_instruction(
|
||||
body,
|
||||
"die ausgewählten Werkzeugdefinitionen überschreiten das Kontextbudget",
|
||||
)
|
||||
estimate = self._estimate_tokens(body)
|
||||
await self._notify(
|
||||
__event_emitter__,
|
||||
f"Kontextschutz aktiv: Eingabe auf ungefähr {estimate:,} Token verdichtet.",
|
||||
)
|
||||
elif changed:
|
||||
await self._notify(
|
||||
__event_emitter__,
|
||||
f"{changed} große Werkzeugausgabe(n) platzsparend verdichtet.",
|
||||
)
|
||||
return body
|
||||
@@ -8,7 +8,7 @@ FILTER_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")/filters" && pwd)
|
||||
|
||||
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
|
||||
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
|
||||
for file in reasoning_default_off.py thinking.py; do
|
||||
for file in reasoning_default_off.py thinking.py stability_guard.py secret_redaction.py local_performance_metrics.py; do
|
||||
[[ -s $FILTER_DIR/$file ]] || die "Filterdatei fehlt: $file"
|
||||
done
|
||||
|
||||
@@ -56,8 +56,11 @@ if not {"key", "value", "updated_at"} <= config_columns:
|
||||
owner = requested_owner
|
||||
if not owner:
|
||||
existing = con.execute(
|
||||
"select user_id from function where id in (?, ?) order by id limit 1",
|
||||
("reasoning_default_off", "thinking"),
|
||||
"select user_id from function where id in (?, ?, ?, ?, ?) order by id limit 1",
|
||||
(
|
||||
"reasoning_default_off", "thinking", "stability_guard",
|
||||
"secret_redaction", "local_performance_metrics",
|
||||
),
|
||||
).fetchone()
|
||||
if existing:
|
||||
owner = existing[0]
|
||||
@@ -73,6 +76,9 @@ now = int(time.time())
|
||||
filters = [
|
||||
("reasoning_default_off", "Reasoning Default Off", 10),
|
||||
("thinking", "Thinking", 20),
|
||||
("stability_guard", "MikeAI Stability Guard", 30),
|
||||
("secret_redaction", "MikeAI Secret Redaction", 40),
|
||||
("local_performance_metrics", "MikeAI Local Performance Metrics", 90),
|
||||
]
|
||||
with con:
|
||||
for function_id, name, priority in filters:
|
||||
@@ -105,7 +111,10 @@ with con:
|
||||
""",
|
||||
("task.follow_up.enable", "false", now),
|
||||
)
|
||||
print("OpenWebUI konfiguriert: Default Off=10, Thinking=20, Folgefragen=aus")
|
||||
print(
|
||||
"OpenWebUI konfiguriert: Default Off=10, Thinking=20, "
|
||||
"Stability Guard=30, Secret Redaction=40, Local Metrics=90, Folgefragen=aus"
|
||||
)
|
||||
PY
|
||||
|
||||
if [[ $was_running == true ]]; then
|
||||
|
||||
Reference in New Issue
Block a user