Add OpenWebUI stability and privacy guards

This commit is contained in:
Mikei386
2026-08-21 15:12:00 +02:00
parent 422c4fac9e
commit 16c83ca7d6
7 changed files with 785 additions and 6 deletions
@@ -0,0 +1,168 @@
"""
title: MikeAI Local Performance Metrics
author: MikeAI
version: 1.0.0
description: Records content-free request timing and token counters locally.
"""
from __future__ import annotations
import json
import os
import threading
import time
from pydantic import BaseModel
class Filter:
class Valves(BaseModel):
priority: int = 90
enabled: bool = True
show_status: bool = True
metrics_path: str = "/app/backend/data/mike-ai-request-metrics.jsonl"
rotate_bytes: int = 5 * 1024 * 1024
_started: dict[str, dict] = {}
_lock = threading.Lock()
def __init__(self):
self.valves = self.Valves()
self.toggle = False
@staticmethod
def _key(metadata: dict | None) -> str | None:
if not isinstance(metadata, dict):
return None
return metadata.get("message_id") or metadata.get("id")
@staticmethod
def _count_tool_calls(messages) -> int:
count = 0
for message in messages or []:
calls = message.get("tool_calls") if isinstance(message, dict) else None
if isinstance(calls, list):
count += len(calls)
elif isinstance(calls, dict):
count += 1
return count
@staticmethod
def _numeric_metrics(message: dict) -> dict:
allowed = {
"prompt_tokens",
"completion_tokens",
"input_tokens",
"output_tokens",
"total_tokens",
"prompt_eval_count",
"eval_count",
"prompt_eval_duration",
"eval_duration",
"total_duration",
"load_duration",
"prompt_n",
"predicted_n",
"prompt_per_second",
"predicted_per_second",
"draft_n",
"draft_n_accepted",
}
result = {}
for container_name in ("usage", "info"):
container = message.get(container_name)
if not isinstance(container, dict):
continue
for key, value in container.items():
if key in allowed and isinstance(value, (int, float)) and not isinstance(value, bool):
result[key] = value
return result
def _write(self, record: dict) -> None:
path = self.valves.metrics_path
try:
if os.path.exists(path) and os.path.getsize(path) >= self.valves.rotate_bytes:
rotated = path + ".1"
if os.path.exists(rotated):
os.remove(rotated)
os.replace(path, rotated)
with open(path, "a", encoding="utf-8") as handle:
handle.write(json.dumps(record, sort_keys=True, separators=(",", ":")) + "\n")
except Exception:
# Metrics must never break a chat request.
pass
async def inlet(self, body: dict, __metadata__: dict = None, **kwargs) -> dict:
if not self.valves.enabled:
return body
key = self._key(__metadata__)
if key:
with self._lock:
self._started[key] = {
"started": time.monotonic(),
"model": str(body.get("model", "unknown"))[:80],
"tool_calls_before": self._count_tool_calls(body.get("messages")),
"tools_available": len(body.get("tools") or []),
}
if len(self._started) > 512:
oldest = next(iter(self._started))
self._started.pop(oldest, None)
return body
async def outlet(
self,
body: dict,
__metadata__: dict = None,
__event_emitter__=None,
**kwargs,
) -> dict:
if not self.valves.enabled:
return body
key = self._key(__metadata__)
with self._lock:
start = self._started.pop(key, None) if key else None
messages = body.get("messages") or []
assistant = next(
(m for m in reversed(messages) if isinstance(m, dict) and m.get("role") == "assistant"),
{},
)
elapsed = round(time.monotonic() - start["started"], 3) if start else None
numeric = self._numeric_metrics(assistant)
record = {
"timestamp": int(time.time()),
"model": (start or {}).get("model", str(body.get("model", "unknown"))[:80]),
"elapsed_seconds": elapsed,
"tool_calls_before": (start or {}).get("tool_calls_before", 0),
"tools_available": (start or {}).get("tools_available", 0),
**numeric,
}
# Deliberately absent: user/chat/message IDs and all textual content.
self._write(record)
if self.valves.show_status and __event_emitter__ is not None:
parts = []
if elapsed is not None:
parts.append(f"{elapsed:.1f} s")
input_tokens = numeric.get("prompt_tokens", numeric.get("input_tokens"))
output_tokens = numeric.get("completion_tokens", numeric.get("output_tokens"))
if input_tokens is not None:
parts.append(f"{int(input_tokens):,} Eingabetoken")
if output_tokens is not None:
parts.append(f"{int(output_tokens):,} Ausgabetoken")
speed = numeric.get("predicted_per_second")
if speed is not None:
parts.append(f"{speed:.1f} Token/s")
if parts:
try:
await __event_emitter__(
{
"type": "status",
"data": {
"description": "Lokale Leistung: " + " · ".join(parts),
"done": True,
},
}
)
except Exception:
pass
return body
@@ -0,0 +1,105 @@
"""
title: MikeAI Secret Redaction
author: MikeAI
version: 1.0.0
description: Redacts common credentials from tool results and final assistant output.
"""
from __future__ import annotations
import re
from pydantic import BaseModel
class Filter:
class Valves(BaseModel):
priority: int = 40
replacement: str = "[REDACTED]"
def __init__(self):
self.valves = self.Valves()
self.toggle = False
def _redact(self, text: str) -> tuple[str, int]:
if not isinstance(text, str) or not text:
return text, 0
count = 0
def replace_full(match):
nonlocal count
count += 1
return self.valves.replacement
def replace_value(match):
nonlocal count
count += 1
return match.group(1) + self.valves.replacement
patterns_full = [
r"-----BEGIN (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----.*?-----END (?:OPENSSH |RSA |EC |DSA )?PRIVATE KEY-----",
r"\beyJ[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{20,}\.[A-Za-z0-9_-]{10,}\b",
]
for pattern in patterns_full:
text = re.sub(pattern, replace_full, text, flags=re.IGNORECASE | re.DOTALL)
patterns_value = [
r"((?:authorization\s*[:=]\s*)?(?:bearer\s+))[A-Za-z0-9._~+/=-]{16,}",
r"((?:\"|')?(?:api[_-]?key|access[_-]?token|refresh[_-]?token|password|passwd|secret|token)(?:\"|')?\s*[:=]\s*(?:\"|')?)[^\s\"',;&}]{8,}",
r"([?&](?:api[_-]?key|access[_-]?token|token|key)=)[^&\s]+",
]
for pattern in patterns_value:
text = re.sub(pattern, replace_value, text, flags=re.IGNORECASE)
return text, count
def _redact_content(self, content) -> tuple[object, int]:
if isinstance(content, str):
return self._redact(content)
if not isinstance(content, list):
return content, 0
result = []
total = 0
for part in content:
if isinstance(part, dict) and isinstance(part.get("text"), str):
item = dict(part)
item["text"], found = self._redact(item["text"])
total += found
result.append(item)
else:
result.append(part)
return result, total
async def _notify(self, emitter, count: int) -> None:
if not count or emitter is None:
return
try:
await emitter(
{
"type": "status",
"data": {
"description": f"Secret-Schutz: {count} mögliche Zugangsdaten entfernt.",
"done": True,
},
}
)
except Exception:
pass
async def inlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict:
total = 0
for message in body.get("messages") or []:
if message.get("role") != "tool":
continue
message["content"], count = self._redact_content(message.get("content", ""))
total += count
await self._notify(__event_emitter__, total)
return body
async def outlet(self, body: dict, __event_emitter__=None, **kwargs) -> dict:
total = 0
for message in body.get("messages") or []:
if message.get("role") != "assistant":
continue
message["content"], count = self._redact_content(message.get("content", ""))
total += count
await self._notify(__event_emitter__, total)
return body
@@ -0,0 +1,302 @@
"""
title: MikeAI Stability Guard
author: MikeAI
version: 1.0.0
description: Bounds tool output and context use and breaks repeated tool-call loops.
"""
from __future__ import annotations
import json
from collections import Counter
from pydantic import BaseModel
class Filter:
class Valves(BaseModel):
priority: int = 30
fast_context_tokens: int = 76800
medium_context_tokens: int = 94208
long_context_tokens: int = 131072
default_context_tokens: int = 76800
soft_context_ratio: float = 0.70
hard_context_ratio: float = 0.84
reserved_output_tokens: int = 8192
max_single_tool_chars: int = 18000
max_total_tool_chars: int = 60000
compacted_tool_chars: int = 3000
duplicate_tool_call_limit: int = 3
max_tool_calls_per_turn: int = 16
def __init__(self):
self.valves = self.Valves()
self.toggle = False
@staticmethod
def _text(value) -> str:
if isinstance(value, str):
return value
try:
return json.dumps(value, ensure_ascii=False, separators=(",", ":"))
except Exception:
return str(value)
@staticmethod
def _compact_text(text: str, limit: int, reason: str) -> str:
if len(text) <= limit:
return text
marker = f"\n\n[MikeAI: {reason}; Originalumfang {len(text)} Zeichen]\n\n"
available = max(256, limit - len(marker))
head = int(available * 0.72)
tail = available - head
return text[:head] + marker + text[-tail:]
def _compact_content(self, content, limit: int, reason: str):
if isinstance(content, str):
return self._compact_text(content, limit, reason)
if not isinstance(content, list):
return content
result = []
text_budget = limit
for part in content:
if not isinstance(part, dict):
result.append(part)
continue
item = dict(part)
if isinstance(item.get("text"), str):
new_text = self._compact_text(
item["text"], max(256, text_budget), reason
)
item["text"] = new_text
text_budget = max(0, text_budget - len(new_text))
# Image URLs/data are deliberately preserved. Truncating base64
# would corrupt the request and character count is not a useful
# estimate of vision tokens.
result.append(item)
return result
def _content_chars(self, content) -> int:
if isinstance(content, str):
return len(content)
if not isinstance(content, list):
return len(self._text(content))
total = 0
for part in content:
if isinstance(part, dict):
if isinstance(part.get("text"), str):
total += len(part["text"])
if part.get("type") in {"image", "image_url", "input_image"}:
total += 6144 # conservative fixed vision-token proxy
else:
total += len(self._text(part))
return total
def _estimate_tokens(self, body: dict) -> int:
# Conservative local estimate for mixed German/English, JSON and code.
chars = 0
for message in body.get("messages") or []:
chars += 24
chars += self._content_chars(message.get("content", ""))
for key in ("tool_calls", "function_call", "output"):
if key in message:
chars += len(self._text(message[key]))
if body.get("tools"):
chars += len(self._text(body["tools"]))
return max(1, (chars + 2) // 3)
def _context_limit(self, model: str) -> int:
model = (model or "").lower()
if "long" in model or "large" in model:
return self.valves.long_context_tokens
if "medium" in model:
return self.valves.medium_context_tokens
if "fast" in model:
return self.valves.fast_context_tokens
return self.valves.default_context_tokens
@staticmethod
def _tool_signatures(message: dict) -> list[str]:
calls = message.get("tool_calls") or []
if isinstance(calls, dict):
calls = [calls]
signatures = []
for call in calls:
if not isinstance(call, dict):
continue
function = call.get("function") or call
name = function.get("name") or call.get("name") or "unknown"
arguments = function.get("arguments") or call.get("arguments") or ""
if not isinstance(arguments, str):
try:
arguments = json.dumps(arguments, sort_keys=True, separators=(",", ":"))
except Exception:
arguments = str(arguments)
signatures.append(f"{name}:{arguments}")
return signatures
def _current_turn_tool_signatures(self, messages: list[dict]) -> list[str]:
last_user = -1
for index, message in enumerate(messages):
if message.get("role") == "user":
last_user = index
result = []
for message in messages[last_user + 1 :]:
if message.get("role") == "assistant":
result.extend(self._tool_signatures(message))
return result
@staticmethod
def _add_guard_instruction(body: dict, reason: str) -> None:
instruction = (
"MikeAI-Sicherheitsgrenze: Weitere Werkzeugaufrufe sind in diesem "
f"Schritt gesperrt ({reason}). Fasse die bereits vorhandenen Ergebnisse "
"zusammen, benenne fehlende Belege ehrlich und frage nicht in einer "
"Werkzeugschleife weiter."
)
messages = body.setdefault("messages", [])
for message in messages:
if message.get("role") == "system" and isinstance(message.get("content"), str):
message["content"] += "\n\n" + instruction
return
messages.insert(0, {"role": "system", "content": instruction})
@staticmethod
def _disable_tools(body: dict) -> None:
body["tools"] = []
body["tool_ids"] = []
metadata = body.get("metadata")
if isinstance(metadata, dict):
metadata["tool_ids"] = []
metadata["tool_servers"] = []
async def _notify(self, emitter, description: str) -> None:
if emitter is None:
return
try:
await emitter(
{
"type": "status",
"data": {"description": description, "done": True},
}
)
except Exception:
pass
def _bound_tool_outputs(self, body: dict) -> tuple[int, int]:
messages = body.get("messages") or []
changed = 0
for message in messages:
if message.get("role") != "tool":
continue
size = self._content_chars(message.get("content", ""))
if size > self.valves.max_single_tool_chars:
message["content"] = self._compact_content(
message.get("content", ""),
self.valves.max_single_tool_chars,
"Werkzeugausgabe gekürzt",
)
changed += 1
tool_messages = [m for m in messages if m.get("role") == "tool"]
total = sum(self._content_chars(m.get("content", "")) for m in tool_messages)
for message in tool_messages:
if total <= self.valves.max_total_tool_chars:
break
old = self._content_chars(message.get("content", ""))
message["content"] = self._compact_content(
message.get("content", ""),
self.valves.compacted_tool_chars,
"älteres Werkzeugresultat wegen Gesamtbudget verdichtet",
)
new = self._content_chars(message.get("content", ""))
total -= max(0, old - new)
changed += 1
return changed, total
def _fit_context(self, body: dict, hard_budget: int) -> int:
messages = body.get("messages") or []
# Old tool results are the safest material to compact. Preserve the
# message and tool_call_id so the OpenAI tool-call sequence stays valid.
for message in messages:
if self._estimate_tokens(body) <= hard_budget:
break
if message.get("role") == "tool":
message["content"] = self._compact_content(
message.get("content", ""),
768,
"älteres Werkzeugresultat wegen Kontextgrenze entfernt",
)
# If necessary, compact dialogue before the current user turn. Never
# touch system text, the current turn or image data.
last_user = max(
(index for index, message in enumerate(messages) if message.get("role") == "user"),
default=len(messages),
)
cutoff = last_user
for message in messages[:cutoff]:
if self._estimate_tokens(body) <= hard_budget:
break
if message.get("role") in {"user", "assistant"} and not message.get("tool_calls"):
message["content"] = self._compact_content(
message.get("content", ""),
1500,
"älterer Dialog wegen Kontextgrenze verdichtet",
)
return self._estimate_tokens(body)
async def inlet(
self,
body: dict,
__event_emitter__=None,
**kwargs,
) -> dict:
changed, _ = self._bound_tool_outputs(body)
messages = body.get("messages") or []
signatures = self._current_turn_tool_signatures(messages)
counts = Counter(signatures)
duplicate = max(counts.values(), default=0)
breaker_reason = None
if duplicate >= self.valves.duplicate_tool_call_limit:
breaker_reason = f"derselbe Aufruf wurde {duplicate}-mal wiederholt"
elif len(signatures) >= self.valves.max_tool_calls_per_turn:
breaker_reason = f"{len(signatures)} Werkzeugaufrufe in einem Schritt"
if breaker_reason:
self._disable_tools(body)
self._add_guard_instruction(body, breaker_reason)
await self._notify(
__event_emitter__, f"Werkzeugschleife gestoppt: {breaker_reason}."
)
context = self._context_limit(body.get("model", ""))
hard_budget = max(
4096,
int(context * self.valves.hard_context_ratio)
- self.valves.reserved_output_tokens,
)
estimate = self._estimate_tokens(body)
if estimate > hard_budget:
estimate = self._fit_context(body, hard_budget)
if estimate > hard_budget and body.get("tools"):
# Never truncate JSON schemas: a damaged schema can make the
# model emit invalid calls. If schemas themselves no longer
# fit, finish this turn without tools and explain why.
self._disable_tools(body)
self._add_guard_instruction(
body,
"die ausgewählten Werkzeugdefinitionen überschreiten das Kontextbudget",
)
estimate = self._estimate_tokens(body)
await self._notify(
__event_emitter__,
f"Kontextschutz aktiv: Eingabe auf ungefähr {estimate:,} Token verdichtet.",
)
elif changed:
await self._notify(
__event_emitter__,
f"{changed} große Werkzeugausgabe(n) platzsparend verdichtet.",
)
return body
+13 -4
View File
@@ -8,7 +8,7 @@ FILTER_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")/filters" && pwd)
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
for file in reasoning_default_off.py thinking.py; do
for file in reasoning_default_off.py thinking.py stability_guard.py secret_redaction.py local_performance_metrics.py; do
[[ -s $FILTER_DIR/$file ]] || die "Filterdatei fehlt: $file"
done
@@ -56,8 +56,11 @@ if not {"key", "value", "updated_at"} <= config_columns:
owner = requested_owner
if not owner:
existing = con.execute(
"select user_id from function where id in (?, ?) order by id limit 1",
("reasoning_default_off", "thinking"),
"select user_id from function where id in (?, ?, ?, ?, ?) order by id limit 1",
(
"reasoning_default_off", "thinking", "stability_guard",
"secret_redaction", "local_performance_metrics",
),
).fetchone()
if existing:
owner = existing[0]
@@ -73,6 +76,9 @@ now = int(time.time())
filters = [
("reasoning_default_off", "Reasoning Default Off", 10),
("thinking", "Thinking", 20),
("stability_guard", "MikeAI Stability Guard", 30),
("secret_redaction", "MikeAI Secret Redaction", 40),
("local_performance_metrics", "MikeAI Local Performance Metrics", 90),
]
with con:
for function_id, name, priority in filters:
@@ -105,7 +111,10 @@ with con:
""",
("task.follow_up.enable", "false", now),
)
print("OpenWebUI konfiguriert: Default Off=10, Thinking=20, Folgefragen=aus")
print(
"OpenWebUI konfiguriert: Default Off=10, Thinking=20, "
"Stability Guard=30, Secret Redaction=40, Local Metrics=90, Folgefragen=aus"
)
PY
if [[ $was_running == true ]]; then