diff --git a/compose.yaml b/compose.yaml index 296104d..dccaac3 100644 --- a/compose.yaml +++ b/compose.yaml @@ -103,6 +103,7 @@ services: - --jinja - --reasoning - auto + - --reasoning-preserve - --host - 0.0.0.0 - --port @@ -175,6 +176,7 @@ services: - --jinja - --reasoning - auto + - --reasoning-preserve - --host - 0.0.0.0 - --port @@ -251,6 +253,7 @@ services: - --jinja - --reasoning - auto + - --reasoning-preserve - --host - 0.0.0.0 - --port @@ -322,6 +325,7 @@ services: - --jinja - --reasoning - auto + - --reasoning-preserve - --host - 0.0.0.0 - --port @@ -399,6 +403,7 @@ services: - --jinja - --reasoning - auto + - --reasoning-preserve - --host - 0.0.0.0 - --port @@ -462,6 +467,7 @@ services: - --jinja - --reasoning - auto + - --reasoning-preserve - --host - 0.0.0.0 - --port @@ -708,7 +714,7 @@ services: build: context: . dockerfile: platform/openwebui/Dockerfile - image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v3} + image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-agent-loop-v5} container_name: mike-ai-open-webui restart: unless-stopped volumes: @@ -750,9 +756,11 @@ services: ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false} ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false} # The derived image reserves the last round for a tool-free synthesis. - # Eight rounds leave room for real operator work without allowing a - # simple repository lookup to fan out into dozens of calls. - CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "8" + # Twelve rounds permit real multi-domain agent work. Exact-repeat, + # per-tool and total-execution limits in the derived image stop loops. + # Leave continuation headroom after the 12-execution middleware budget: + # one additional model turn is required to synthesize the visible answer. + CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "16" USER_AGENT: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36" # Seed native MCP connections on a fresh Open WebUI database. Secrets # stay inside the tool containers, so these internal URLs need no keys. diff --git a/docs/CURRENT_REFERENCE.md b/docs/CURRENT_REFERENCE.md index c51d9f9..5a5baaa 100644 --- a/docs/CURRENT_REFERENCE.md +++ b/docs/CURRENT_REFERENCE.md @@ -1,6 +1,6 @@ # Aktueller produktiver Referenzstand -Stand: 23. August 2026. Dieses Dokument beschreibt die auf Athena installierte +Stand: 24. August 2026. Dieses Dokument beschreibt die auf Athena installierte und geprüfte Docker-Referenz. Die verbindlichen Profilparameter stehen in `STANDARD_PROFILE_MATRIX.md`. @@ -49,6 +49,7 @@ Zielplattform. - Batch 64, Micro-Batch 32 - ein paralleler Slot - Jinja und automatisches Reasoning +- erhaltener Reasoning-Zustand über Werkzeugrunden (`--reasoning-preserve`) - Temperatur 0,2, Top-p 0,8, Top-k 20 ### Profile @@ -174,7 +175,9 @@ Aktuell existieren funktionale Adapter für: OpenWebUI bindet diese Kataloge nicht pauschal an jedes Modellprofil. Der lokale `MikeAI Auto Tool Selector` ergänzt anhand der jüngsten Nutzernachricht -höchstens zwei passende Fach-MCP-Verbindungen pro Anfrage. Allgemeine +höchstens drei passende Fach-MCP-Verbindungen pro Anfrage. Eine echte +Mehrdomänen-Aufgabe erhält automatisch ein begrenztes mittleres Reasoning- +Budget; einfache Aufgaben bleiben schnell. Allgemeine Webrecherche erfolgt über Open WebUIs native `search_web`/`fetch_url`-Werkzeuge; `web-local` ist nur noch manuell für Spezialfälle verfügbar. Dadurch bleiben Fachkataloge klein und kurze Profile verlieren keinen unnötigen Kontext. MUA @@ -194,6 +197,13 @@ Die Transportbrücke verwendet den OpenWebUI-kompatiblen `mcp-proxy` 0.12.0 im stateless Betrieb. Supergateway wurde nach reproduzierbaren HTTP-400-Fehlern bei `notifications/initialized` aus diesem Pfad entfernt. +Die agentische OpenWebUI-Schleife führt höchstens zwölf einzelne Werkzeuge aus, +höchstens vier Varianten desselben Werkzeugs und niemals zweimal exakt dieselbe +Signatur. Nach Ende des Budgets stehen zusätzliche interne Runden ausschließlich +für eine sichtbare werkzeugfreie Schlussantwort bereit. Das produktive +OpenWebUI-Derivat trägt den Tag +`mike-ai/openwebui:main-01f4282-agent-loop-v5`. + Der Platform Context MCP läuft ohne Docker-Socket, Shell, Egress oder Secrets. Ein root-eigener Minutentimer erzeugt nur einen begrenzten Laufzeitsnapshot. Der Schreibpfad ist auf `docs/*.md`, Vorschau, ausdrückliche Freigabe, atomare diff --git a/docs/OPERATIONS.md b/docs/OPERATIONS.md index 52e6bf6..be418af 100644 --- a/docs/OPERATIONS.md +++ b/docs/OPERATIONS.md @@ -10,7 +10,7 @@ Reihenfolge ist absichtlich festgelegt: 2. `Thinking`, Priorität 20: läuft nur bei aktiviertem Brain-Schalter und überschreibt den Standard mit Low, Medium oder High. 3. `MikeAI Auto Tool Selector`, Priorität 25: betrachtet ausschließlich die - jüngste Nutzernachricht und stellt pro Anfrage höchstens zwei passende MCPs + jüngste Nutzernachricht und stellt pro Anfrage höchstens drei passende MCPs bereit. Er erkennt GitHub, Home Assistant, Sonarr/Radarr, Navidrome, Unraid-Diagnose und Athena-Plattformwissen. Manuell gewählte Werkzeuge bleiben erhalten. MUA mit erweiterten Verwaltungsrechten wird nie diff --git a/docs/QWEN38_AGENTIC_BENCHMARK_2026-08-24.md b/docs/QWEN38_AGENTIC_BENCHMARK_2026-08-24.md new file mode 100644 index 0000000..3347561 --- /dev/null +++ b/docs/QWEN38_AGENTIC_BENCHMARK_2026-08-24.md @@ -0,0 +1,90 @@ +# Qwen3.8-27B – agentischer Werkzeugtest vom 24. August 2026 + +## Ergebnis in einem Satz + +Qwen3.8-27B ist auf Athena für längere agentische Aufgaben brauchbar, wenn die +Werkzeugschicht ihm gebündelte Fachoperationen, erhaltenes Reasoning und eine +garantierte Schlussrunde anbietet. Die früheren Ausfälle waren überwiegend +Orchestrierungs- und MCP-Probleme, nicht ein grundsätzliches Unvermögen des +Modells. + +## Warum die Community-Erfahrungen besser wirkten + +Community-Demos verwenden meist einen spezialisierten Agent-Harness, große +Kontexte, erhaltenes Reasoning und kompakte Werkzeuge. Athena kombinierte zuvor +76K Kontext, standardmäßig abgeschaltetes Thinking, sechs Einzelaufrufe, +teilweise sehr kleinteilige MCP-Operationen und OpenWebUIs hartes Ende ohne +Syntheserunde. Zusätzlich existieren aktuelle llama.cpp-Randfälle bei +Qwen3.8-Systemnachrichten, verschachtelten Werkzeugschemata und gestreamten +Parallelaufrufen. Die Differenz war daher kein fairer Modellvergleich. + +## Produktive Änderungen + +- `--reasoning-preserve` in allen Qwen-Profilen. +- Automatische Auswahl von bis zu drei Fach-MCPs. +- Automatisch mittleres, auf 3.072 Token begrenztes Reasoning nur bei echten + Mehrdomänen-Aufgaben; explizite Benutzereinstellungen werden nicht ersetzt. +- Zwölf tatsächliche Aufrufe, höchstens vier pro Werkzeug, keine identische + Signatur zweimal; 16 interne Runden lassen Raum für die Schlussantwort. +- Genau ein zusätzlicher werkzeugloser Syntheseversuch, falls Qwen nach Ende + der Recherche trotzdem noch einen Funktionsaufruf formuliert. +- Home Assistant `find_commented_blocks`: vollständige auskommentierte + Automationseinträge einschließlich ID, Alias und Zeilen in einem Aufruf. +- MUA r017: gebündelte CA-Suche mit bis zu fünf Namensvarianten. +- MUA r018: fokussierte Loganalyse mit mehreren `focus_terms` in einem Aufruf. +- Evidenzplan im Systemprompt: zuerst ein breiter Aufruf je Domäne, danach nur + gezielte Lücken schließen, Pflichtbedingungen früh prüfen und bei deren + Scheitern sofort den Kandidaten wechseln. + +## Browser-Benchmarks + +| Test | Vorher | Nachher | Bewertung | +|---|---:|---:|---| +| Auskommentierte HA-Automationen inventarisieren | 9 Aufrufe, 90,9 s | 1 Aufruf, 26,8 s | IDs, Aliase und Zeilen korrekt; sehr gut | +| Deemix: GitHub + laufender Unraid-Container + MCP-Entwurf | zuvor 37 GitHub-Aufrufe und Abbruch | 7 Aufrufe, sichtbare Antwort | klare Verbesserung; einzelne Betriebsannahmen noch zu optimistisch | +| Web + GitHub + Unraid: CA-Negativprüfung und Template-Entwurf | Kandidat zu spät verworfen, Budgetende | 11 Aufrufe, vollständiger Entwurf | Schlussantwort vorhanden; XML und Architektur müssen weiterhin fachlich geprüft werden | +| Drei Domänen: HA-YAML + Unraid-Logs + GitHub-Quelle | vorher kein finaler Text am Budgetende | 12 Aufrufe, vollständige Evidenzmatrix | Finalizer v5 bestanden; Quellenverwechslung wurde transparent als Unsicherheit markiert | +| Fokussierte Home-Assistant-Logprüfung | mehrere Shell-/grep-Aufrufe | 1 Inventar + 1 fokussierte Loganalyse, 44,7 s | MUA r018 korrekt gewählt; klare Beleggrenzen | + +## Qualitätsbefund + +### Stark + +- wählt nach der Anpassung die drei korrekten Fachdomänen automatisch; +- beginnt parallel/breit und liefert belastbare Livewerte; +- kann aus Werkzeugresultaten strukturierte Evidenzmatrizen und sichere Pläne + bauen; +- verschweigt verbleibende Unsicherheit überwiegend nicht; +- die HA-Spezialoperation reduziert Laufzeit und Kontextverbrauch drastisch. + +### Noch nicht auf Frontier-Agent-Niveau + +- bei ähnlichen GitHub-Repositories kann Qwen den falschen Treffer vertiefen, + statt zuerst den exakten installierten Upstream zu bestimmen; +- bei komplexen Containerstacks erzeugt es gelegentlich formal plausible, + aber fachlich fragwürdige Unraid-XMLs; +- ohne gebündelte Logoperation fällt es auf mehrere Shell-/grep-Aufrufe zurück; +- zwölf Werkzeugaufrufe sind kein Qualitätsbeweis: Die Auswahl und Form der + Werkzeuge sind wichtiger als eine möglichst große Zahl. + +## Empfehlung + +Fast bleibt für normale Aufgaben geeignet. Für Änderungen, längere Recherche +oder mehrere Systeme gleichzeitig soll das automatische Mehrdomänen-Reasoning +greifen; bei besonders kritischer Arbeit kann Thinking manuell auf Hoch gesetzt +werden. Ergebnisse, die Installationen, Sicherheit, Geld oder Datenänderungen +betreffen, benötigen weiterhin Vorschau, Belegprüfung und Freigabe. Weitere +Verbesserungen sollten bevorzugt gebündelte Fachoperationen ergänzen und nicht +das globale Aufruflimit erhöhen. + +## Versionierte Quellen + +- Qwen3.8-27B Modellkarte: +- OpenWebUI native tool calling: + +- llama.cpp Systemnachrichten-Randfall: + +- llama.cpp verschachtelte Schemas: + +- llama.cpp Streaming/Parallel-Toolcalls: + diff --git a/docs/TOOLING_RELIABILITY_2026-08-24.md b/docs/TOOLING_RELIABILITY_2026-08-24.md index 6b78d6f..b99dda8 100644 --- a/docs/TOOLING_RELIABILITY_2026-08-24.md +++ b/docs/TOOLING_RELIABILITY_2026-08-24.md @@ -27,10 +27,15 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen. damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem ParserError wegen einer falschen Spaltenzahl. Tabellenanalysen sollen im Regelfall mit einer Erkennungs- und einer Auswertungsrunde auskommen. -4. Pro Antwort sind höchstens acht Werkzeugrunden und sechs tatsächlich - ausgeführte Einzelaufrufe erlaubt. Das abgeleitete, +4. Pro Antwort sind höchstens 16 interne Werkzeugrunden und zwölf tatsächlich + ausgeführte Einzelaufrufe erlaubt. Pro konkretem Werkzeug sind höchstens + vier unterschiedliche Aufrufe zulässig; identische Argumente werden kein + zweites Mal ausgeführt. Die zusätzlichen vier internen Runden sind nur + Synthesepuffer und erhöhen nicht das Ausführungsbudget. Das abgeleitete, reproduzierbar gebaute OpenWebUI-Image verwendet die letzte Runde zwingend - als werkzeugfreie Synthese. Statt `Tool-call limit reached` ohne Ergebnis + als werkzeugfreie Synthese. Erzeugt das Modell trotz entfernter Schemata + noch einmal werkzeugförmige Ausgabe, folgt genau ein zweiter, ebenfalls + werkzeugloser Syntheseversuch. Statt `Tool-call limit reached` ohne Ergebnis erhält der Benutzer deshalb eine sichtbare Antwort aus den vorhandenen Befunden samt ehrlicher Angabe fehlender Belege. Inlet-Filter allein können dies nicht erzwingen, weil sie zwischen OpenWebUIs internen Werkzeugrunden @@ -50,6 +55,14 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen. er auch vom Außenstandort über WireGuard. 7. Task-Management ist keine Faktenquelle und wird nicht für einzelne Fragen, Nachschlageaufgaben oder Dateianalysen verwendet. +8. Mehrdomänen-Aufgaben erhalten automatisch höchstens drei passende + Fachkataloge und ein begrenztes Qwen-Reasoning-Budget. Einfache Ein-Domänen- + Aufgaben bleiben im schnellen Non-Thinking-Modus. Alle llama.cpp-Profile + bewahren Reasoning-Zustand zwischen Werkzeugrunden (`--reasoning-preserve`). +9. Wiederkehrende Fachsuchen werden serverseitig gebündelt: Home Assistant + inventarisiert auskommentierte YAML-Blöcke in einem Aufruf; MUA durchsucht + Community Applications mit mehreren Namensvarianten in einem Feed-Durchlauf + und filtert Containerlogs mit mehreren `focus_terms` in einem Aufruf. ## Abnahme diff --git a/platform/mcp/patches/hass_mcp/yaml_config.py b/platform/mcp/patches/hass_mcp/yaml_config.py index c8d7080..9bca6db 100644 --- a/platform/mcp/patches/hass_mcp/yaml_config.py +++ b/platform/mcp/patches/hass_mcp/yaml_config.py @@ -41,6 +41,7 @@ _OPS = ( "get", "read_source", "find_source", + "find_commented_blocks", "list_backups", "create", "update", @@ -66,8 +67,9 @@ _SECRET_REFERENCE = re.compile(r"!secret\s+[^\s#]+", re.IGNORECASE) "Safely inspect and edit Home Assistant YAML. Structured CRUD is limited to " "automations.yaml, scripts.yaml and scenes.yaml. Raw source operations also " "allow configuration.yaml so commented-out blocks can be found and reviewed. " - "secrets.yaml and arbitrary paths are impossible. Use read_source/find_source " - "for comments or exact YAML text. Every mutation first returns a diff/preview " + "secrets.yaml and arbitrary paths are impossible. Use find_commented_blocks once " + "to inventory fully commented YAML entries; use read_source/find_source only for " + "other comments or exact YAML text. Every mutation first returns a diff/preview " "and one-time approval_ticket; only repeat the exact unchanged call with " "confirm=true after explicit user approval. Writes create a backup, are atomic, " "run Home Assistant config validation, roll back on failure, and reload the " @@ -139,7 +141,7 @@ _SECRET_REFERENCE = re.compile(r"!secret\s+[^\s#]+", re.IGNORECASE) requires_admin=True, write_ops=["create", "update", "replace_source_text", "reload"], destructive_ops=["delete", "restore_backup"], - admin_ops=["list", "get", "read_source", "find_source", "list_backups"], + admin_ops=["list", "get", "read_source", "find_source", "find_commented_blocks", "list_backups"], ) async def ha_yaml_config( hass: HomeAssistant, @@ -178,6 +180,10 @@ async def ha_yaml_config( if not query: raise ToolError("op=find_source requires query") return await _find_source(hass, path, query, max_lines) + if op == "find_commented_blocks": + if kind not in {"automation", "scene"}: + raise ToolError("find_commented_blocks supports automation and scene list files") + return await _find_commented_blocks(hass, path, max_lines) if op == "list_backups": return await _list_backups(hass, filename, limit, offset) if op == "replace_source_text": @@ -387,6 +393,86 @@ async def _find_source(hass: HomeAssistant, path: Path, query: str, max_lines: i } +def _extract_commented_blocks(text: str, max_lines: int) -> tuple[list[dict[str, Any]], bool]: + """Return top-level YAML list entries whose every source line is commented. + + This intentionally recognizes only the conservative ``# - id:`` form used + by Home Assistant's automations/scenes editor. Ordinary prose comments, + partially disabled entries and nested comments are not treated as entries. + """ + lines = text.splitlines() + start_pattern = re.compile(r"^\s*#\s*-\s+id\s*:\s*(.*?)\s*$", re.IGNORECASE) + alias_pattern = re.compile(r"^\s*#\s+alias\s*:\s*(.*?)\s*$", re.IGNORECASE) + blocks: list[dict[str, Any]] = [] + consumed = 0 + index = 0 + truncated = False + + def clean_scalar(value: str) -> str: + value = value.strip() + if len(value) >= 2 and value[0] == value[-1] and value[0] in {"'", '"'}: + return value[1:-1] + return value + + while index < len(lines): + match = start_pattern.match(lines[index]) + if not match: + index += 1 + continue + start = index + block_lines = [lines[index]] + index += 1 + while index < len(lines): + if start_pattern.match(lines[index]): + break + if not lines[index].strip() or not re.match(r"^\s*#", lines[index]): + break + block_lines.append(lines[index]) + index += 1 + if consumed + len(block_lines) > max_lines: + truncated = True + break + alias = None + for line in block_lines: + alias_match = alias_pattern.match(line) + if alias_match: + alias = clean_scalar(alias_match.group(1)) + break + blocks.append( + { + "start_line": start + 1, + "end_line": start + len(block_lines), + "id": clean_scalar(match.group(1)), + "alias": alias, + "source": [ + {"line": start + offset + 1, "text": _redact_line(line)} + for offset, line in enumerate(block_lines) + ], + } + ) + consumed += len(block_lines) + return blocks, truncated + + +async def _find_commented_blocks( + hass: HomeAssistant, path: Path, max_lines: int +) -> dict[str, Any]: + text = await _read_text(hass, path) + cap = max(1, min(max_lines, 400)) + blocks, truncated = _extract_commented_blocks(text, cap) + return { + "file": path.name, + "sha256": _fingerprint(text), + "authoritative_block_count": len(blocks) if not truncated else None, + "returned_block_count": len(blocks), + "returned_source_lines": sum(len(block["source"]) for block in blocks), + "has_more": truncated, + "blocks": blocks, + "recognition_rule": "Only fully commented top-level '# - id:' YAML list entries are returned.", + "redaction_note": "Credential-like values and !secret reference names are redacted.", + } + + def _source_diff(filename: str, before: str, after: str) -> list[str]: return list( difflib.unified_diff( diff --git a/platform/openwebui/filters/auto_tool_selector.py b/platform/openwebui/filters/auto_tool_selector.py index dbd5fe5..396b9f6 100644 --- a/platform/openwebui/filters/auto_tool_selector.py +++ b/platform/openwebui/filters/auto_tool_selector.py @@ -1,7 +1,7 @@ """ title: MikeAI Auto Tool Selector author: MikeAI -version: 3.0.0 +version: 3.2.0 description: Selects a small, relevant set of MCP servers for each user request. """ @@ -16,8 +16,11 @@ class Filter: class Valves(BaseModel): priority: int = 25 enabled: bool = True - max_automatic_tools: int = 2 + max_automatic_tools: int = 3 show_selection_status: bool = True + enable_multidomain_reasoning: bool = True + multidomain_reasoning_effort: str = "medium" + multidomain_reasoning_budget: int = 3072 TOOL_IDS = { "github": "server:mcp:github-local", @@ -121,8 +124,9 @@ class Filter: homeassistant = self._matches( text, ( - r"\bhome\s*assist(?:ant|ent)\b", r"\bhomeassistant\b", r"\bhass\b", + r"\bhome\s*assist(?:ant|ent)s?\b", r"\bhomeassistants?\b", r"\bhass\b", r"\bautomatisierung(?:en)?\b", r"\bentit[aä]t(?:en)?\b", + r"\bautomations?\.ya?ml\b", r"\bconfiguration\.ya?ml\b", r"\b(?:sensor|light|switch|climate|automation)\.[\w.-]+", r"\b(?:temperatur|luftfeuchtigkeit|wie warm)\b.*\b" r"(?:k[uü]che|wohnzimmer|schlafzimmer|bad|toilette|keller|haus)\b", @@ -166,35 +170,30 @@ class Filter: ), ) - # Existing infrastructure must be inspected before a new integration - # is designed. For a GitHub-backed MCP that targets a service already - # running on Unraid, source plus Unraid inventory are the two most - # useful bounded capabilities; the Operator follows at implementation. - if github and unraid: - selected.extend(("github", "unraid")) - elif operator: + # A real integration or incident task can legitimately span three + # trust domains (for example GitHub source + Unraid runtime + Home + # Assistant configuration). Attach each relevant specialist once; + # the evidence-plan rule keeps the model breadth-first and bounded. + if operator: selected.append("operator") - elif platform: - selected.append("platform") - elif homeassistant: - selected.append("homeassistant") - elif arr: - selected.append("arr") - elif navidrome: - selected.append("navidrome") - elif unraid: - selected.append("unraid") - elif github: + if github: selected.append("github") + if unraid: + selected.append("unraid") + if homeassistant: + selected.append("homeassistant") + if arr: + selected.append("arr") + if navidrome: + selected.append("navidrome") + if platform and not operator: + selected.append("platform") # A direct GitHub reference should still use GitHub even if broader web # research is also requested. General public web research is provided # by Open WebUI's native search_web/fetch_url tools and therefore must # not auto-attach the older specialist web MCP. - if github and "github" not in selected: - selected.insert(0, "github") - - return selected[: max(0, self.valves.max_automatic_tools)] + return list(dict.fromkeys(selected))[: max(0, self.valves.max_automatic_tools)] async def _notify(self, emitter, labels: list[str]) -> None: if emitter is None or not self.valves.show_selection_status: @@ -230,6 +229,20 @@ class Filter: merged.append(tool_id) body["tool_ids"] = merged + # Qwen3.8 is markedly more reliable on long, cross-domain tool tasks + # when its reasoning channel is preserved. Keep ordinary one-tool + # requests fast, but give requests that genuinely span two specialist + # MCPs a bounded reasoning budget. An explicit user-selected effort is + # never overwritten. + if ( + self.valves.enable_multidomain_reasoning + and len(selected) >= 2 + and body.get("reasoning_effort") in (None, "", "none") + ): + body["reasoning_effort"] = self.valves.multidomain_reasoning_effort + if not body.get("reasoning_budget"): + body["reasoning_budget"] = self.valves.multidomain_reasoning_budget + labels = [self.LABELS[category] for category in selected] self._add_system_rule(body, labels) await self._notify(__event_emitter__, labels) diff --git a/platform/openwebui/filters/stability_guard.py b/platform/openwebui/filters/stability_guard.py index 55150b0..53621d2 100644 --- a/platform/openwebui/filters/stability_guard.py +++ b/platform/openwebui/filters/stability_guard.py @@ -1,7 +1,7 @@ """ title: MikeAI Stability Guard author: MikeAI -version: 2.2.0 +version: 2.3.0 description: Bounds tool output and context use and breaks repeated tool-call loops. """ @@ -32,7 +32,7 @@ class Filter: # Secondary protection for histories that re-enter the filter. The # live internal tool loop is bounded and finalized by the derived # OpenWebUI image because inlet filters do not run between its rounds. - max_tool_calls_per_turn: int = 10 + max_tool_calls_per_turn: int = 12 max_private_table_tool_calls: int = 6 def __init__(self): diff --git a/platform/openwebui/install-filters.sh b/platform/openwebui/install-filters.sh index 8f25e03..81c4aa1 100755 --- a/platform/openwebui/install-filters.sh +++ b/platform/openwebui/install-filters.sh @@ -95,7 +95,7 @@ functions = [ ("thinking", "Thinking", "filter", 20, filter_dir, ""), ( "auto_tool_selector", "MikeAI Auto Tool Selector", "filter", 25, filter_dir, - "Stellt pro Anfrage höchstens zwei passende MCP-Werkzeuge bereit. " + "Stellt pro Anfrage höchstens drei passende MCP-Werkzeuge bereit und aktiviert bei echten Mehrdomänen-Aufgaben begrenztes Reasoning. " "Die Auswahl ist keine Freigabe für schreibende Aktionen.", ), ("stability_guard", "MikeAI Stability Guard", "filter", 30, filter_dir, ""), diff --git a/platform/openwebui/install-models.sh b/platform/openwebui/install-models.sh index 9e39f44..1423ce3 100755 --- a/platform/openwebui/install-models.sh +++ b/platform/openwebui/install-models.sh @@ -225,6 +225,18 @@ params = { "public information you may make at most one focused fallback attempt with " "the general web tool, then synthesize the available evidence or stop clearly; " "never enter a fallback or synonym-search loop. " + "For a multi-step or multi-domain request, make a compact evidence plan before " + "the first tool call. Reserve at least one call for every requested domain instead " + "of exhausting the budget in the first system. Work breadth-first: obtain one " + "bounded inventory or overview from each relevant domain, then make only targeted " + "follow-up calls for facts still missing. Every call must answer a specific unresolved " + "question. Check cheap pass/fail constraints that can invalidate a candidate before " + "spending calls on deep research. If a candidate fails a mandatory constraint, switch " + "immediately; do not produce the explicitly forbidden candidate as the main result. " + "Never invoke the same operation with identical arguments twice, and normally " + "use one operation no more than four times. If the user asks for a simulation or plan, " + "perform read-only discovery only and do not execute the proposed state changes. Stop " + "research as soon as the evidence is sufficient and synthesize the complete answer. " "Never invent tool results, system state, files, measurements, or actions. " "For claims about current external or system state, you must successfully " "use the relevant domain tool during the current request before saying " diff --git a/platform/openwebui/patch_tool_finalization.py b/platform/openwebui/patch_tool_finalization.py index ea6353c..d71dd84 100644 --- a/platform/openwebui/patch_tool_finalization.py +++ b/platform/openwebui/patch_tool_finalization.py @@ -26,10 +26,15 @@ needle = """ tool_call_iterations = 0 """ replacement = """ tool_call_iterations = 0 # Open WebUI counts batches, while one model turn may request many - # functions in parallel. Bound actual executions as well so a small - # local model cannot expand eight rounds into dozens of API calls. + # functions in parallel. Keep an execution budget as a second bound, + # but make it large enough for genuinely agentic multi-domain work. + # Per-tool and exact-repeat limits below prevent one low-level MCP + # operation from consuming the whole turn. tool_call_executions = 0 - max_tool_call_executions = 6 + max_tool_call_executions = 12 + max_executions_per_tool = 4 + tool_execution_counts = {} + seen_tool_signatures = set() max_tool_call_iterations = getattr( """ if source.count(needle) != 1: @@ -44,11 +49,30 @@ needle = """ response_tool_calls = tool_calls.pop(0) """ replacement = """ response_tool_calls = tool_calls.pop(0) - remaining_tool_calls = max( - 0, max_tool_call_executions - tool_call_executions - ) - skipped_tool_calls = response_tool_calls[remaining_tool_calls:] - response_tool_calls = response_tool_calls[:remaining_tool_calls] + accepted_tool_calls = [] + skipped_tool_calls = [] + skip_reasons = [] + for candidate in response_tool_calls: + function = candidate.get('function') or candidate + tool_name = function.get('name') or candidate.get('name') or 'unknown' + arguments = function.get('arguments') or candidate.get('arguments') or '' + signature = f'{tool_name}:{arguments}' + if tool_call_executions + len(accepted_tool_calls) >= max_tool_call_executions: + skipped_tool_calls.append(candidate) + skip_reasons.append('total execution budget reached') + continue + if tool_execution_counts.get(tool_name, 0) >= max_executions_per_tool: + skipped_tool_calls.append(candidate) + skip_reasons.append(f'per-tool budget reached for {tool_name}') + continue + if signature in seen_tool_signatures: + skipped_tool_calls.append(candidate) + skip_reasons.append(f'exact duplicate suppressed for {tool_name}') + continue + accepted_tool_calls.append(candidate) + seen_tool_signatures.add(signature) + tool_execution_counts[tool_name] = tool_execution_counts.get(tool_name, 0) + 1 + response_tool_calls = accepted_tool_calls if skipped_tool_calls: skipped_ids = {call.get('id', '') for call in skipped_tool_calls} # Responses API streaming may already have exposed all calls @@ -91,6 +115,7 @@ replacement = """ # The upstream loop otherwise stops wit and tool_call_iterations >= max_tool_call_iterations ) or tool_call_executions >= max_tool_call_executions + or bool(skipped_tool_calls) ) if force_final_response: new_form_data.pop('tools', None) @@ -101,16 +126,32 @@ replacement = """ # The upstream loop otherwise stops wit final_metadata['tool_ids'] = [] final_metadata['tool_servers'] = [] new_form_data['metadata'] = final_metadata - new_form_data['messages'] = add_or_update_system_message( - 'The tool budget for this turn is exhausted. Do not call ' - 'or imitate any more tools. Give the user a concise final ' - 'answer now using only the tool results already present. ' - 'Explicitly distinguish verified findings from inference, ' - 'and state what could not be verified. Never end without a ' - 'visible answer.', - new_form_data['messages'], - append=True, + final_reason = '; '.join(dict.fromkeys(skip_reasons)) or 'execution budget reached' + final_instruction = ( + 'The tool research phase is finished (' + final_reason + '). ' + 'Do not call or imitate any more tools. Give the user a concise ' + 'final answer now using only the tool results already present. ' + 'Explicitly distinguish verified findings from inference, state ' + 'what could not be verified, and never end without a visible answer.' ) + # Qwen3.8 requires system messages to precede the conversation. + # Merge into the first system message (or create it at index 0) + # instead of appending a mid-conversation system message. + system_message = next( + ( + message + for message in new_form_data['messages'] + if message.get('role') == 'system' + and isinstance(message.get('content'), str) + ), + None, + ) + if system_message is None: + new_form_data['messages'].insert( + 0, {'role': 'system', 'content': final_instruction} + ) + else: + system_message['content'] += '\\n\\n' + final_instruction # A final user-role instruction is deliberately stronger # than another system suffix after a long tool-call pattern. # It is request-local and is not added to the saved chat. @@ -145,10 +186,46 @@ needle = """ await stream_body_handler(res, new_form_ output[:0] = prior_output """ replacement = """ await stream_body_handler(res, new_form_data) - if force_final_response: - # A model may still print tool-shaped output after the - # schemas were removed. It must not trigger the upstream - # hard-error path or another execution attempt. + if force_final_response and tool_calls: + # Some local models still emit one more function-shaped + # request after schemas were removed. Discard that + # invisible attempt and grant exactly one clean, + # tool-less synthesis retry. This is bounded and cannot + # execute another function. + tool_calls.clear() + output = [] + retry_form_data = { + **new_form_data, + 'messages': [ + *new_form_data['messages'], + { + 'role': 'user', + 'content': ( + 'Your previous synthesis attempt was not ' + 'visible because it looked like another ' + 'function call. Tools are no longer available. ' + 'Write the final human-readable answer now, ' + 'starting immediately with the conclusion.' + ), + }, + ], + } + retry_res = await generate_chat_completion( + request, + retry_form_data, + user, + bypass_system_prompt=True, + ) + if isinstance(retry_res, StreamingResponse): + await stream_body_handler(retry_res, retry_form_data) + # Never let tool-shaped retry output re-enter execution. + tool_calls.clear() + output[:] = [ + item + for item in output + if item.get('type') != 'function_call' + ] + elif force_final_response: tool_calls.clear() output[:0] = prior_output """