diff --git a/dev/test_openwebui_filters.py b/dev/test_openwebui_filters.py new file mode 100644 index 0000000..ae1cb5b --- /dev/null +++ b/dev/test_openwebui_filters.py @@ -0,0 +1,171 @@ +#!/usr/bin/env python3 +"""Offline tests for the versioned OpenWebUI filters. + +The production container provides pydantic. A tiny local stand-in keeps these +logic tests dependency-free and prevents test setup from reaching the network. +""" + +from __future__ import annotations + +import importlib.util +import json +import sys +import tempfile +import types +import unittest +from pathlib import Path + + +class _BaseModel: + def __init__(self, **values): + annotations = {} + for base in reversed(type(self).__mro__): + annotations.update(getattr(base, "__annotations__", {})) + for name in annotations: + setattr(self, name, values.get(name, getattr(type(self), name, None))) + + +fake_pydantic = types.ModuleType("pydantic") +fake_pydantic.BaseModel = _BaseModel +fake_pydantic.Field = lambda default=None, **kwargs: default +sys.modules.setdefault("pydantic", fake_pydantic) + +FILTER_DIR = Path(__file__).parents[1] / "platform" / "openwebui" / "filters" + + +def _load(name: str): + spec = importlib.util.spec_from_file_location(name, FILTER_DIR / f"{name}.py") + module = importlib.util.module_from_spec(spec) + assert spec.loader is not None + spec.loader.exec_module(module) + return module + + +class StabilityGuardTests(unittest.IsolatedAsyncioTestCase): + async def asyncSetUp(self): + self.module = _load("stability_guard") + self.guard = self.module.Filter() + + async def test_large_tool_output_is_bounded(self): + body = { + "model": "qwen-fast", + "messages": [ + {"role": "user", "content": "Prüfe das Log."}, + {"role": "tool", "tool_call_id": "x", "content": "A" * 50000}, + ], + } + result = await self.guard.inlet(body) + content = result["messages"][1]["content"] + self.assertLessEqual(len(content), self.guard.valves.max_single_tool_chars) + self.assertIn("Werkzeugausgabe gekürzt", content) + + async def test_duplicate_calls_disable_tools(self): + call = { + "id": "call", + "type": "function", + "function": {"name": "search", "arguments": '{"q":"same"}'}, + } + body = { + "model": "qwen-fast", + "tools": [{"type": "function", "function": {"name": "search"}}], + "tool_ids": ["server:mcp:web"], + "messages": [ + {"role": "user", "content": "Suche genau einmal."}, + {"role": "assistant", "tool_calls": [call]}, + {"role": "tool", "tool_call_id": "1", "content": "nichts"}, + {"role": "assistant", "tool_calls": [call]}, + {"role": "tool", "tool_call_id": "2", "content": "nichts"}, + {"role": "assistant", "tool_calls": [call]}, + ], + } + result = await self.guard.inlet(body) + self.assertEqual(result["tools"], []) + self.assertEqual(result["tool_ids"], []) + self.assertIn("Weitere Werkzeugaufrufe", result["messages"][0]["content"]) + + async def test_old_context_is_compacted_before_current_turn(self): + self.guard.valves.default_context_tokens = 10000 + self.guard.valves.hard_context_ratio = 0.8 + self.guard.valves.reserved_output_tokens = 1000 + body = { + "model": "unknown", + "messages": [ + {"role": "system", "content": "Sicher arbeiten."}, + {"role": "user", "content": "alt " * 15000}, + {"role": "assistant", "content": "altantwort " * 8000}, + {"role": "tool", "tool_call_id": "old", "content": "log " * 20000}, + {"role": "user", "content": "Aktuelle wichtige Frage"}, + ], + } + result = await self.guard.inlet(body) + self.assertEqual(result["messages"][-1]["content"], "Aktuelle wichtige Frage") + self.assertLess(len(result["messages"][1]["content"]), 3000) + + +class MetricsTests(unittest.IsolatedAsyncioTestCase): + async def test_metrics_file_contains_no_chat_content_or_ids(self): + module = _load("local_performance_metrics") + metrics = module.Filter() + with tempfile.TemporaryDirectory() as directory: + path = Path(directory) / "metrics.jsonl" + metrics.valves.metrics_path = str(path) + metadata = {"message_id": "secret-message-id", "chat_id": "secret-chat-id"} + await metrics.inlet( + { + "model": "qwen-fast", + "messages": [{"role": "user", "content": "private prompt"}], + "tools": [{"name": "tool"}], + }, + __metadata__=metadata, + ) + await metrics.outlet( + { + "model": "qwen-fast", + "messages": [ + { + "role": "assistant", + "content": "private answer", + "usage": { + "prompt_tokens": 10, + "completion_tokens": 4, + "total_tokens": 14, + }, + } + ], + }, + __metadata__=metadata, + ) + raw = path.read_text() + record = json.loads(raw) + self.assertEqual(record["prompt_tokens"], 10) + self.assertNotIn("private", raw) + self.assertNotIn("secret", raw) + + +class SecretRedactionTests(unittest.IsolatedAsyncioTestCase): + async def test_only_tool_and_assistant_content_is_redacted(self): + module = _load("secret_redaction") + guard = module.Filter() + token = "eyJ" + "A" * 24 + "." + "B" * 24 + "." + "C" * 16 + body = { + "messages": [ + {"role": "user", "content": f"Absichtlich lokal nutzen: {token}"}, + {"role": "tool", "content": f'{{"api_key":"1234567890abcdef"}} {token}'}, + ] + } + result = await guard.inlet(body) + self.assertIn(token, result["messages"][0]["content"]) + self.assertNotIn(token, result["messages"][1]["content"]) + self.assertNotIn("1234567890abcdef", result["messages"][1]["content"]) + + outlet = { + "messages": [ + {"role": "assistant", "content": "Bearer " + "abcdefghijklmnopqrstuvwxyz"} + ] + } + result = await guard.outlet(outlet) + self.assertNotIn("abcdefghijklmnopqrstuvwxyz", result["messages"][0]["content"]) + + +if __name__ == "__main__": + unittest.main() diff --git a/docs/ATHENA_REBUILD_LOG.md b/docs/ATHENA_REBUILD_LOG.md index 3327a1f..0a58068 100644 --- a/docs/ATHENA_REBUILD_LOG.md +++ b/docs/ATHENA_REBUILD_LOG.md @@ -35,6 +35,8 @@ Chats oder privaten Nutzdaten. | TinySearch-Volume | Das vom Installer vorbereitete Modellvolume wurde von Compose gleichzeitig als Compose-eigen behandelt. | Das Volume besitzt einen festen Namen und ist in Compose ausdrücklich als extern vorbereitet markiert. | | Reasoning-Filter | Beide globalen Filter besaßen Priorität 0; bei gleicher Priorität entschied die ID-Sortierung statt der gewünschten Logik. | `Reasoning Default Off` läuft mit Priorität 10 sicher vor dem optionalen `Thinking`-Override mit Priorität 20. Filter und sicherer Installer liegen versioniert im Repository. | | Folgefragen | OpenWebUI erzeugte nach Antworten zusätzliche Vorschläge und verbrauchte dafür einen weiteren Modellaufruf. | Folgefragengenerierung ist in der persistenten OpenWebUI-Konfiguration und im Compose-Standard deaktiviert. | +| Werkzeug-/Kontextschutz | Große MCP-Antworten und wiederholte identische Aufrufe konnten Kontextfenster sprengen beziehungsweise Tool-Schleifen erzeugen. | Globaler Stability Guard begrenzt Resultate, verdichtet alte Inhalte profilabhängig und stoppt Wiederholungen; JSON-Schemas und Bilder werden nicht beschädigt. | +| Leistungsdaten | Benchmarkdaten sollten sichtbar sein, ohne private Chat-Inhalte zu protokollieren. | Ein globaler Abschlussfilter schreibt ausschließlich technische Zahlen in eine lokal rotierende JSONL-Datei und zeigt eine knappe Statuszeile. | ## Abnahmezustand am 21. August 2026 diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md index c4be333..7ef203e 100644 --- a/docs/OPERATIONS.md +++ b/docs/OPERATIONS.md @@ -1,6 +1,6 @@ # Betrieb -## Reasoning-Filter in OpenWebUI +## OpenWebUI-Filter und Stabilitätsschutz Die versionierten Filter liegen unter `platform/openwebui/filters/`. Ihre Reihenfolge ist absichtlich festgelegt: @@ -9,9 +9,20 @@ Reihenfolge ist absichtlich festgelegt: `reasoning_effort=none`. 2. `Thinking`, Priorität 20: läuft nur bei aktiviertem Brain-Schalter und überschreibt den Standard mit Low, Medium oder High. +3. `MikeAI Stability Guard`, Priorität 30: begrenzt einzelne und gesamte + Werkzeugresultate, verdichtet bei Bedarf zuerst alte Tool-Ausgaben und + Dialogteile und stoppt identische beziehungsweise ausufernde Tool-Schleifen. +4. `MikeAI Secret Redaction`, Priorität 40: entfernt übliche API-Keys, Tokens, + Passwörter, JWTs und private Schlüssel aus Tool-Ergebnissen, bevor sie das + Modell erreichen, sowie aus fertigen Modellantworten. Nutzereingaben und + Authentifizierungswege werden nicht verändert. +5. `MikeAI Local Performance Metrics`, Priorität 90: erfasst nach Abschluss + ausschließlich technische Zahlen wie Laufzeit, Tokenzähler, Token/s und + Tool-Anzahl. Nutzer-, Chat- und Nachrichten-IDs sowie sämtliche Textinhalte + werden weder geschrieben noch gehasht gespeichert. OpenWebUI sortiert kleinere Prioritäten zuerst. Nach dem ersten Anlegen eines -Admin-Benutzers oder nach einer Datenwiederherstellung werden beide Filter mit +Admin-Benutzers oder nach einer Datenwiederherstellung werden alle Filter mit einer vorherigen Datenbanksicherung installiert beziehungsweise aktualisiert: ```bash @@ -25,6 +36,17 @@ Folgefragen (`task.follow_up.enable=false`). Dieselbe Vorgabe steht zusätzlich als Container-Umgebungswert im Compose-Stack, damit bereits eine frische OpenWebUI-Datenbank ohne Folgefragen startet. +Der Stabilitätsschutz kennt die drei Profilgrenzen 76.800, 94.208 und 131.072 +Token. Er reserviert Ausgabetoken und greift vor der harten llama.cpp-Grenze +ein. Bilder bleiben unangetastet; JSON-Werkzeugschemas werden niemals +abgeschnitten. Sind allein die ausgewählten Schemas zu groß, wird der +Werkzeugzugriff nur für diesen Schritt deaktiviert und das Modell erhält eine +eindeutige Abschlussanweisung. + +Die rotierende, inhaltsfreie Metrikdatei liegt im persistenten +OpenWebUI-Volume unter `mike-ai-request-metrics.jsonl` (maximal 5 MiB plus eine +Rotation). Sie darf für Benchmarks ausgewertet werden, ohne Chats auszulesen. + ## Profile | Profil | Virtuelles Modell | Kontext | Zweck | diff --git a/platform/openwebui/filters/local_performance_metrics.py b/platform/openwebui/filters/local_performance_metrics.py new file mode 100644 index 0000000..a0cd5ac --- /dev/null +++ b/platform/openwebui/filters/local_performance_metrics.py @@ -0,0 +1,168 @@ +""" +title: MikeAI Local Performance Metrics +author: MikeAI +version: 1.0.0 +description: Records content-free request timing and token counters locally. +""" + +from __future__ import annotations + +import json +import os +import threading +import time +from pydantic import BaseModel + + +class Filter: + class Valves(BaseModel): + priority: int = 90 + enabled: bool = True + show_status: bool = True + metrics_path: str = "/app/backend/data/mike-ai-request-metrics.jsonl" + rotate_bytes: int = 5 * 1024 * 1024 + + _started: dict[str, dict] = {} + _lock = threading.Lock() + + def __init__(self): + self.valves = self.Valves() + self.toggle = False + + @staticmethod + def _key(metadata: dict | None) -> str | None: + if not isinstance(metadata, dict): + return None + return metadata.get("message_id") or metadata.get("id") + + @staticmethod + def _count_tool_calls(messages) -> int: + count = 0 + for message in messages or []: + calls = message.get("tool_calls") if isinstance(message, dict) else None + if isinstance(calls, list): + count += len(calls) + elif isinstance(calls, dict): + count += 1 + return count + + @staticmethod + def _numeric_metrics(message: dict) -> dict: + allowed = { + "prompt_tokens", + "completion_tokens", + "input_tokens", + "output_tokens", + "total_tokens", + "prompt_eval_count", + "eval_count", + "prompt_eval_duration", + "eval_duration", + "total_duration", + "load_duration", + "prompt_n", + "predicted_n", + "prompt_per_second", + "predicted_per_second", + "draft_n", + "draft_n_accepted", + } + result = {} + for container_name in ("usage", "info"): + container = message.get(container_name) + if not isinstance(container, dict): + continue + for key, value in container.items(): + if key in allowed and isinstance(value, (int, float)) and not isinstance(value, bool): + result[key] = value + return result + + def _write(self, record: dict) -> None: + path = self.valves.metrics_path + try: + if os.path.exists(path) and os.path.getsize(path) >= self.valves.rotate_bytes: + rotated = path + ".1" + if os.path.exists(rotated): + os.remove(rotated) + os.replace(path, rotated) + with open(path, "a", encoding="utf-8") as handle: + handle.write(json.dumps(record, sort_keys=True, separators=(",", ":")) + "\n") + except Exception: + # Metrics must never break a chat request. + pass + + async def inlet(self, body: dict, __metadata__: dict = None, **kwargs) -> dict: + if not self.valves.enabled: + return body + key = self._key(__metadata__) + if key: + with self._lock: + self._started[key] = { + "started": time.monotonic(), + "model": str(body.get("model", "unknown"))[:80], + "tool_calls_before": self._count_tool_calls(body.get("messages")), + "tools_available": len(body.get("tools") or []), + } + if len(self._started) > 512: + oldest = next(iter(self._started)) + self._started.pop(oldest, None) + return body + + async def outlet( + self, + body: dict, + __metadata__: dict = None, + __event_emitter__=None, + **kwargs, + ) -> dict: + if not self.valves.enabled: + return body + key = self._key(__metadata__) + with self._lock: + start = self._started.pop(key, None) if key else None + + messages = body.get("messages") or [] + assistant = next( + (m for m in reversed(messages) if isinstance(m, dict) and m.get("role") == "assistant"), + {}, + ) + elapsed = round(time.monotonic() - start["started"], 3) if start else None + numeric = self._numeric_metrics(assistant) + record = { + "timestamp": int(time.time()), + "model": (start or {}).get("model", str(body.get("model", "unknown"))[:80]), + "elapsed_seconds": elapsed, + "tool_calls_before": (start or {}).get("tool_calls_before", 0), + "tools_available": (start or {}).get("tools_available", 0), + **numeric, + } + # Deliberately absent: user/chat/message IDs and all textual content. + self._write(record) + + if self.valves.show_status and __event_emitter__ is not None: + parts = [] + if elapsed is not None: + parts.append(f"{elapsed:.1f} s") + input_tokens = numeric.get("prompt_tokens", numeric.get("input_tokens")) + output_tokens = numeric.get("completion_tokens", numeric.get("output_tokens")) + if input_tokens is not None: + parts.append(f"{int(input_tokens):,} Eingabetoken") + if output_tokens is not None: + parts.append(f"{int(output_tokens):,} Ausgabetoken") + speed = numeric.get("predicted_per_second") + if speed is not None: + parts.append(f"{speed:.1f} Token/s") + if parts: + try: + await __event_emitter__( + { + "type": "status", + "data": { + "description": "Lokale Leistung: " + " · ".join(parts), + "done": True, + }, + } + ) + except Exception: + pass + return body diff --git a/platform/openwebui/filters/secret_redaction.py b/platform/openwebui/filters/secret_redaction.py new file mode 100644 index 0000000..0fc7f13 --- /dev/null +++ b/platform/openwebui/filters/secret_redaction.py @@ -0,0 +1,105 @@ +""" +title: MikeAI Secret Redaction +author: MikeAI +version: 1.0.0 +description: Redacts common credentials from tool results and final assistant output. +""" + +from __future__ import annotations + +import re +from pydantic import BaseModel + + +class Filter: + class Valves(BaseModel): + priority: int = 40 + replacement: str = "[REDACTED]" + + def __init__(self): + self.valves = self.Valves() + self.toggle = False + + def _redact(self, text: str) -> tuple[str, int]: + if not isinstance(text, str) or not text: + return text, 0 + count = 0 + + def replace_full(match): + nonlocal count + count += 1 + return self.valves.replacement + + def replace_value(match): + nonlocal count + count += 1 + return match.group(1) + self.valves.replacement + + patterns_full = [ + r"-----BEGIN (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----.*?-----END (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----", + r"\beyJ[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{10,}\b", + ] + for pattern in patterns_full: + text = re.sub(pattern, replace_full, text, flags=re.IGNORECASE | re.DOTALL) + + patterns_value = [ + r"((?:authorization\s*[:=]\s*)?(?:bearer\s+))[A-Za-z0-9._~+/=-]{16,}", + r"((?:\"|')?(?:api[_-]?key|access[_-]?token|refresh[_-]?token|password|passwd|secret|token)(?:\"|')?\s*[:=]\s*(?:\"|')?)[^\s\"',;&}]{8,}", + r"([?&](?:api[_-]?key|access[_-]?token|token|key)=)[^&\s]+", + ] + for pattern in patterns_value: + text = re.sub(pattern, replace_value, text, flags=re.IGNORECASE) + return text, count + + def _redact_content(self, content) -> tuple[object, int]: + if isinstance(content, str): + return self._redact(content) + if not isinstance(content, list): + return content, 0 + result = [] + total = 0 + for part in content: + if isinstance(part, dict) and isinstance(part.get("text"), str): + item = dict(part) + item["text"], found = self._redact(item["text"]) + total += found + result.append(item) + else: + result.append(part) + return result, total + + async def _notify(self, emitter, count: int) -> None: + if not count or emitter is None: + return + try: + await emitter( + { + "type": "status", + "data": { + "description": f"Secret-Schutz: {count} mögliche Zugangsdaten entfernt.", + "done": True, + }, + } + ) + except Exception: + pass + + async def inlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict: + total = 0 + for message in body.get("messages") or []: + if message.get("role") != "tool": + continue + message["content"], count = self._redact_content(message.get("content", "")) + total += count + await self._notify(__event_emitter__, total) + return body + + async def outlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict: + total = 0 + for message in body.get("messages") or []: + if message.get("role") != "assistant": + continue + message["content"], count = self._redact_content(message.get("content", "")) + total += count + await self._notify(__event_emitter__, total) + return body diff --git a/platform/openwebui/filters/stability_guard.py b/platform/openwebui/filters/stability_guard.py new file mode 100644 index 0000000..e1910e9 --- /dev/null +++ b/platform/openwebui/filters/stability_guard.py @@ -0,0 +1,302 @@ +""" +title: MikeAI Stability Guard +author: MikeAI +version: 1.0.0 +description: Bounds tool output and context use and breaks repeated tool-call loops. +""" + +from __future__ import annotations + +import json +from collections import Counter +from pydantic import BaseModel + + +class Filter: + class Valves(BaseModel): + priority: int = 30 + fast_context_tokens: int = 76800 + medium_context_tokens: int = 94208 + long_context_tokens: int = 131072 + default_context_tokens: int = 76800 + soft_context_ratio: float = 0.70 + hard_context_ratio: float = 0.84 + reserved_output_tokens: int = 8192 + max_single_tool_chars: int = 18000 + max_total_tool_chars: int = 60000 + compacted_tool_chars: int = 3000 + duplicate_tool_call_limit: int = 3 + max_tool_calls_per_turn: int = 16 + + def __init__(self): + self.valves = self.Valves() + self.toggle = False + + @staticmethod + def _text(value) -> str: + if isinstance(value, str): + return value + try: + return json.dumps(value, ensure_ascii=False, separators=(",", ":")) + except Exception: + return str(value) + + @staticmethod + def _compact_text(text: str, limit: int, reason: str) -> str: + if len(text) <= limit: + return text + marker = f"\n\n[MikeAI: {reason}; Originalumfang {len(text)} Zeichen]\n\n" + available = max(256, limit - len(marker)) + head = int(available * 0.72) + tail = available - head + return text[:head] + marker + text[-tail:] + + def _compact_content(self, content, limit: int, reason: str): + if isinstance(content, str): + return self._compact_text(content, limit, reason) + if not isinstance(content, list): + return content + + result = [] + text_budget = limit + for part in content: + if not isinstance(part, dict): + result.append(part) + continue + item = dict(part) + if isinstance(item.get("text"), str): + new_text = self._compact_text( + item["text"], max(256, text_budget), reason + ) + item["text"] = new_text + text_budget = max(0, text_budget - len(new_text)) + # Image URLs/data are deliberately preserved. Truncating base64 + # would corrupt the request and character count is not a useful + # estimate of vision tokens. + result.append(item) + return result + + def _content_chars(self, content) -> int: + if isinstance(content, str): + return len(content) + if not isinstance(content, list): + return len(self._text(content)) + total = 0 + for part in content: + if isinstance(part, dict): + if isinstance(part.get("text"), str): + total += len(part["text"]) + if part.get("type") in {"image", "image_url", "input_image"}: + total += 6144 # conservative fixed vision-token proxy + else: + total += len(self._text(part)) + return total + + def _estimate_tokens(self, body: dict) -> int: + # Conservative local estimate for mixed German/English, JSON and code. + chars = 0 + for message in body.get("messages") or []: + chars += 24 + chars += self._content_chars(message.get("content", "")) + for key in ("tool_calls", "function_call", "output"): + if key in message: + chars += len(self._text(message[key])) + if body.get("tools"): + chars += len(self._text(body["tools"])) + return max(1, (chars + 2) // 3) + + def _context_limit(self, model: str) -> int: + model = (model or "").lower() + if "long" in model or "large" in model: + return self.valves.long_context_tokens + if "medium" in model: + return self.valves.medium_context_tokens + if "fast" in model: + return self.valves.fast_context_tokens + return self.valves.default_context_tokens + + @staticmethod + def _tool_signatures(message: dict) -> list[str]: + calls = message.get("tool_calls") or [] + if isinstance(calls, dict): + calls = [calls] + signatures = [] + for call in calls: + if not isinstance(call, dict): + continue + function = call.get("function") or call + name = function.get("name") or call.get("name") or "unknown" + arguments = function.get("arguments") or call.get("arguments") or "" + if not isinstance(arguments, str): + try: + arguments = json.dumps(arguments, sort_keys=True, separators=(",", ":")) + except Exception: + arguments = str(arguments) + signatures.append(f"{name}:{arguments}") + return signatures + + def _current_turn_tool_signatures(self, messages: list[dict]) -> list[str]: + last_user = -1 + for index, message in enumerate(messages): + if message.get("role") == "user": + last_user = index + result = [] + for message in messages[last_user + 1 :]: + if message.get("role") == "assistant": + result.extend(self._tool_signatures(message)) + return result + + @staticmethod + def _add_guard_instruction(body: dict, reason: str) -> None: + instruction = ( + "MikeAI-Sicherheitsgrenze: Weitere Werkzeugaufrufe sind in diesem " + f"Schritt gesperrt ({reason}). Fasse die bereits vorhandenen Ergebnisse " + "zusammen, benenne fehlende Belege ehrlich und frage nicht in einer " + "Werkzeugschleife weiter." + ) + messages = body.setdefault("messages", []) + for message in messages: + if message.get("role") == "system" and isinstance(message.get("content"), str): + message["content"] += "\n\n" + instruction + return + messages.insert(0, {"role": "system", "content": instruction}) + + @staticmethod + def _disable_tools(body: dict) -> None: + body["tools"] = [] + body["tool_ids"] = [] + metadata = body.get("metadata") + if isinstance(metadata, dict): + metadata["tool_ids"] = [] + metadata["tool_servers"] = [] + + async def _notify(self, emitter, description: str) -> None: + if emitter is None: + return + try: + await emitter( + { + "type": "status", + "data": {"description": description, "done": True}, + } + ) + except Exception: + pass + + def _bound_tool_outputs(self, body: dict) -> tuple[int, int]: + messages = body.get("messages") or [] + changed = 0 + for message in messages: + if message.get("role") != "tool": + continue + size = self._content_chars(message.get("content", "")) + if size > self.valves.max_single_tool_chars: + message["content"] = self._compact_content( + message.get("content", ""), + self.valves.max_single_tool_chars, + "Werkzeugausgabe gekürzt", + ) + changed += 1 + + tool_messages = [m for m in messages if m.get("role") == "tool"] + total = sum(self._content_chars(m.get("content", "")) for m in tool_messages) + for message in tool_messages: + if total <= self.valves.max_total_tool_chars: + break + old = self._content_chars(message.get("content", "")) + message["content"] = self._compact_content( + message.get("content", ""), + self.valves.compacted_tool_chars, + "älteres Werkzeugresultat wegen Gesamtbudget verdichtet", + ) + new = self._content_chars(message.get("content", "")) + total -= max(0, old - new) + changed += 1 + return changed, total + + def _fit_context(self, body: dict, hard_budget: int) -> int: + messages = body.get("messages") or [] + # Old tool results are the safest material to compact. Preserve the + # message and tool_call_id so the OpenAI tool-call sequence stays valid. + for message in messages: + if self._estimate_tokens(body) <= hard_budget: + break + if message.get("role") == "tool": + message["content"] = self._compact_content( + message.get("content", ""), + 768, + "älteres Werkzeugresultat wegen Kontextgrenze entfernt", + ) + + # If necessary, compact dialogue before the current user turn. Never + # touch system text, the current turn or image data. + last_user = max( + (index for index, message in enumerate(messages) if message.get("role") == "user"), + default=len(messages), + ) + cutoff = last_user + for message in messages[:cutoff]: + if self._estimate_tokens(body) <= hard_budget: + break + if message.get("role") in {"user", "assistant"} and not message.get("tool_calls"): + message["content"] = self._compact_content( + message.get("content", ""), + 1500, + "älterer Dialog wegen Kontextgrenze verdichtet", + ) + return self._estimate_tokens(body) + + async def inlet( + self, + body: dict, + __event_emitter__=None, + **kwargs, + ) -> dict: + changed, _ = self._bound_tool_outputs(body) + messages = body.get("messages") or [] + signatures = self._current_turn_tool_signatures(messages) + counts = Counter(signatures) + duplicate = max(counts.values(), default=0) + + breaker_reason = None + if duplicate >= self.valves.duplicate_tool_call_limit: + breaker_reason = f"derselbe Aufruf wurde {duplicate}-mal wiederholt" + elif len(signatures) >= self.valves.max_tool_calls_per_turn: + breaker_reason = f"{len(signatures)} Werkzeugaufrufe in einem Schritt" + + if breaker_reason: + self._disable_tools(body) + self._add_guard_instruction(body, breaker_reason) + await self._notify( + __event_emitter__, f"Werkzeugschleife gestoppt: {breaker_reason}." + ) + + context = self._context_limit(body.get("model", "")) + hard_budget = max( + 4096, + int(context * self.valves.hard_context_ratio) + - self.valves.reserved_output_tokens, + ) + estimate = self._estimate_tokens(body) + if estimate > hard_budget: + estimate = self._fit_context(body, hard_budget) + if estimate > hard_budget and body.get("tools"): + # Never truncate JSON schemas: a damaged schema can make the + # model emit invalid calls. If schemas themselves no longer + # fit, finish this turn without tools and explain why. + self._disable_tools(body) + self._add_guard_instruction( + body, + "die ausgewählten Werkzeugdefinitionen überschreiten das Kontextbudget", + ) + estimate = self._estimate_tokens(body) + await self._notify( + __event_emitter__, + f"Kontextschutz aktiv: Eingabe auf ungefähr {estimate:,} Token verdichtet.", + ) + elif changed: + await self._notify( + __event_emitter__, + f"{changed} große Werkzeugausgabe(n) platzsparend verdichtet.", + ) + return body diff --git a/platform/openwebui/install-filters.sh b/platform/openwebui/install-filters.sh index 38d137c..0518285 100755 --- a/platform/openwebui/install-filters.sh +++ b/platform/openwebui/install-filters.sh @@ -8,7 +8,7 @@ FILTER_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")/filters" && pwd) die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; } [[ $EUID -eq 0 ]] || die "Bitte als root ausführen." -for file in reasoning_default_off.py thinking.py; do +for file in reasoning_default_off.py thinking.py stability_guard.py secret_redaction.py local_performance_metrics.py; do [[ -s $FILTER_DIR/$file ]] || die "Filterdatei fehlt: $file" done @@ -56,8 +56,11 @@ if not {"key", "value", "updated_at"} <= config_columns: owner = requested_owner if not owner: existing = con.execute( - "select user_id from function where id in (?, ?) order by id limit 1", - ("reasoning_default_off", "thinking"), + "select user_id from function where id in (?, ?, ?, ?, ?) order by id limit 1", + ( + "reasoning_default_off", "thinking", "stability_guard", + "secret_redaction", "local_performance_metrics", + ), ).fetchone() if existing: owner = existing[0] @@ -73,6 +76,9 @@ now = int(time.time()) filters = [ ("reasoning_default_off", "Reasoning Default Off", 10), ("thinking", "Thinking", 20), + ("stability_guard", "MikeAI Stability Guard", 30), + ("secret_redaction", "MikeAI Secret Redaction", 40), + ("local_performance_metrics", "MikeAI Local Performance Metrics", 90), ] with con: for function_id, name, priority in filters: @@ -105,7 +111,10 @@ with con: """, ("task.follow_up.enable", "false", now), ) -print("OpenWebUI konfiguriert: Default Off=10, Thinking=20, Folgefragen=aus") +print( + "OpenWebUI konfiguriert: Default Off=10, Thinking=20, " + "Stability Guard=30, Secret Redaction=40, Local Metrics=90, Folgefragen=aus" +) PY if [[ $was_running == true ]]; then