fix(openwebui): guarantee answer after tool budget
This commit is contained in:
+8
-4
@@ -705,7 +705,10 @@ services:
|
||||
start_period: 10s
|
||||
|
||||
open-webui:
|
||||
image: ${OPENWEBUI_IMAGE:-ghcr.io/open-webui/open-webui:v0.9.5}
|
||||
build:
|
||||
context: .
|
||||
dockerfile: platform/openwebui/Dockerfile
|
||||
image: ${OPENWEBUI_IMAGE:-mike-ai/open-webui:v0.9.5-tool-final-v1}
|
||||
container_name: mike-ai-open-webui
|
||||
restart: unless-stopped
|
||||
volumes:
|
||||
@@ -746,9 +749,10 @@ services:
|
||||
AUDIO_TTS_VOICE: alloy
|
||||
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
|
||||
ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
|
||||
# A model must synthesize an answer instead of spending hundreds of
|
||||
# iterations retrying an empty/dynamic page or traversing a repository.
|
||||
CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "12"
|
||||
# The derived image reserves the last round for a tool-free synthesis.
|
||||
# Eight rounds leave room for real operator work without allowing a
|
||||
# simple repository lookup to fan out into dozens of calls.
|
||||
CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "8"
|
||||
USER_AGENT: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||||
# Seed native MCP connections on a fresh Open WebUI database. Secrets
|
||||
# stay inside the tool containers, so these internal URLs need no keys.
|
||||
|
||||
@@ -26,17 +26,23 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen.
|
||||
Trennzeichen, Kopfzeile, Metadatenzeilen, Dezimal- und Datumsformat erkannt;
|
||||
damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem
|
||||
ParserError wegen einer falschen Spaltenzahl. Tabellenanalysen sollen im
|
||||
Regelfall mit einer Erkennungs- und einer Auswertungsrunde auskommen. Nach
|
||||
sechs lokalen Dateiaufrufen erzwingt der Guard eine werkzeugfreie
|
||||
Abschlussantwort, bevor OpenWebUI bei seiner harten Zwölf-Runden-Grenze
|
||||
ohne sichtbares Ergebnis abbricht.
|
||||
4. Pro Antwort sind höchstens zwölf Werkzeugrunden erlaubt. Der zweite
|
||||
Regelfall mit einer Erkennungs- und einer Auswertungsrunde auskommen.
|
||||
4. Pro Antwort sind höchstens acht Werkzeugrunden erlaubt. Das abgeleitete,
|
||||
reproduzierbar gebaute OpenWebUI-Image verwendet die letzte Runde zwingend
|
||||
als werkzeugfreie Synthese. Statt `Tool-call limit reached` ohne Ergebnis
|
||||
erhält der Benutzer deshalb eine sichtbare Antwort aus den vorhandenen
|
||||
Befunden samt ehrlicher Angabe fehlender Belege. Inlet-Filter allein können
|
||||
dies nicht erzwingen, weil sie zwischen OpenWebUIs internen Werkzeugrunden
|
||||
nicht erneut ausgeführt werden. Der zweite
|
||||
identische Aufruf wird gestoppt. Ein einzelnes Resultat ist auf 10.000, alle
|
||||
Resultate zusammen auf 36.000 Zeichen begrenzt.
|
||||
5. Der Home-Assistant-MCP behält den TLS-Namen `ha.casaderoll.de`, routet ihn
|
||||
5. Repository-Prüfungen beginnen mit README/Wurzel, verwenden anschließend
|
||||
höchstens drei gezielte Code-Suchen und öffnen nur relevante Treffer. Eine
|
||||
konkrete Laufzeitinstanz wird genau einmal über ihr Fachwerkzeug geprüft.
|
||||
6. Der Home-Assistant-MCP behält den TLS-Namen `ha.casaderoll.de`, routet ihn
|
||||
im Container aber auf `HOME_LAN_PROXY_IP` im Heimnetz. Dadurch funktioniert
|
||||
er auch vom Außenstandort über WireGuard.
|
||||
6. Task-Management ist keine Faktenquelle und wird nicht für einzelne Fragen,
|
||||
7. Task-Management ist keine Faktenquelle und wird nicht für einzelne Fragen,
|
||||
Nachschlageaufgaben oder Dateianalysen verwendet.
|
||||
|
||||
## Abnahme
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
FROM ghcr.io/open-webui/open-webui:v0.9.5
|
||||
|
||||
COPY platform/openwebui/patch_tool_finalization.py /tmp/patch_tool_finalization.py
|
||||
RUN python /tmp/patch_tool_finalization.py \
|
||||
/app/backend/open_webui/utils/middleware.py \
|
||||
&& rm /tmp/patch_tool_finalization.py
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
"""
|
||||
title: MikeAI Stability Guard
|
||||
author: MikeAI
|
||||
version: 2.1.0
|
||||
version: 2.2.0
|
||||
description: Bounds tool output and context use and breaks repeated tool-call loops.
|
||||
"""
|
||||
|
||||
@@ -29,9 +29,9 @@ class Filter:
|
||||
max_total_tool_chars: int = 36000
|
||||
compacted_tool_chars: int = 2000
|
||||
duplicate_tool_call_limit: int = 2
|
||||
# OpenWebUI itself stops after twelve rounds without giving the model
|
||||
# another completion. Stay below that ceiling so the model still gets
|
||||
# a final, tool-free turn.
|
||||
# Secondary protection for histories that re-enter the filter. The
|
||||
# live internal tool loop is bounded and finalized by the derived
|
||||
# OpenWebUI image because inlet filters do not run between its rounds.
|
||||
max_tool_calls_per_turn: int = 10
|
||||
max_private_table_tool_calls: int = 6
|
||||
|
||||
|
||||
@@ -237,7 +237,9 @@ with con:
|
||||
"Für Repository-Suche, echte Datei-Inhalte und gezielte "
|
||||
"Code-Suche auf GitHub. Bei Fragen zu Implementierung, README, API-Routen oder "
|
||||
"Quellcode dieses Werkzeug statt allgemeiner Websuche verwenden. Keine Issues, "
|
||||
"Pull Requests, Actions, rekursiven Komplettbäume oder Schreibzugriffe.",
|
||||
"Pull Requests, Actions, rekursiven Komplettbäume oder Schreibzugriffe. Bei einem "
|
||||
"konkreten Repository zuerst README beziehungsweise Wurzel einmal lesen, danach "
|
||||
"höchstens drei gezielte Code-Suchen und nur relevante Trefferdateien öffnen.",
|
||||
),
|
||||
"athena-operator-local": (
|
||||
"Athena Operator",
|
||||
|
||||
@@ -213,7 +213,13 @@ params = {
|
||||
"For GitHub repository implementation details, README files, source trees, "
|
||||
"API routes, or code search, use the dedicated official GitHub repository "
|
||||
"tool instead of guessing from ordinary web results. Use general web search "
|
||||
"for wider public discussion and non-repository sources. "
|
||||
"for wider public discussion and non-repository sources. For one named repository, "
|
||||
"do not enumerate files recursively. Read the root README or one root listing once, "
|
||||
"then use one to three targeted code searches for terms such as route, API, CLI, "
|
||||
"endpoint, command or the relevant framework, and open only the few matching files "
|
||||
"needed for evidence. If a live deployment is also mentioned, inspect that service "
|
||||
"once with its domain tool. Normally finish within six GitHub calls and always "
|
||||
"synthesize an answer from the evidence already obtained. "
|
||||
"If a specialist tool returns an authentication, authorization, connection, "
|
||||
"or configuration error, do not repeat the same call. Report the error. For "
|
||||
"public information you may make at most one focused fallback attempt with "
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Make Open WebUI finish with an answer when its internal tool budget is used.
|
||||
|
||||
Open WebUI 0.9.5 otherwise ends the request with only
|
||||
``Tool-call limit reached``. The patch is deliberately assertion-based: an
|
||||
upstream source change makes the image build fail instead of silently applying
|
||||
the modification at the wrong location.
|
||||
"""
|
||||
|
||||
from pathlib import Path
|
||||
import sys
|
||||
|
||||
|
||||
if len(sys.argv) != 2:
|
||||
raise SystemExit("usage: patch_tool_finalization.py MIDDLEWARE_PY")
|
||||
|
||||
path = Path(sys.argv[1])
|
||||
source = path.read_text()
|
||||
|
||||
patch_marker = "The tool budget for this turn is exhausted. Do not call"
|
||||
if patch_marker in source:
|
||||
raise SystemExit("Open WebUI tool-finalization patch is already present")
|
||||
|
||||
needle = """ res = await generate_chat_completion(
|
||||
request,
|
||||
new_form_data,
|
||||
user,
|
||||
bypass_system_prompt=True,
|
||||
)
|
||||
"""
|
||||
|
||||
replacement = """ # The upstream loop otherwise stops with only
|
||||
# `Tool-call limit reached` after executing this batch. Reserve
|
||||
# the final completion for synthesis: keep every accumulated
|
||||
# result, but expose no tools to the model and explicitly require
|
||||
# a useful, evidence-bounded answer.
|
||||
force_final_response = (
|
||||
max_tool_call_iterations is not None
|
||||
and tool_call_iterations >= max_tool_call_iterations
|
||||
)
|
||||
if force_final_response:
|
||||
new_form_data.pop('tools', None)
|
||||
new_form_data.pop('tool_ids', None)
|
||||
final_metadata = dict(metadata)
|
||||
final_metadata['tools'] = {}
|
||||
final_metadata['tool_ids'] = []
|
||||
final_metadata['tool_servers'] = []
|
||||
new_form_data['metadata'] = final_metadata
|
||||
new_form_data['messages'] = add_or_update_system_message(
|
||||
'The tool budget for this turn is exhausted. Do not call '
|
||||
'or imitate any more tools. Give the user a concise final '
|
||||
'answer now using only the tool results already present. '
|
||||
'Explicitly distinguish verified findings from inference, '
|
||||
'and state what could not be verified. Never end without a '
|
||||
'visible answer.',
|
||||
new_form_data['messages'],
|
||||
append=True,
|
||||
)
|
||||
|
||||
res = await generate_chat_completion(
|
||||
request,
|
||||
new_form_data,
|
||||
user,
|
||||
bypass_system_prompt=True,
|
||||
)
|
||||
"""
|
||||
|
||||
if source.count(needle) != 1:
|
||||
raise SystemExit(
|
||||
f"expected exactly one Open WebUI continuation anchor, found {source.count(needle)}"
|
||||
)
|
||||
source = source.replace(needle, replacement)
|
||||
|
||||
needle = """ await stream_body_handler(res, new_form_data)
|
||||
output[:0] = prior_output
|
||||
"""
|
||||
replacement = """ await stream_body_handler(res, new_form_data)
|
||||
if force_final_response:
|
||||
# A model may still print tool-shaped output after the
|
||||
# schemas were removed. It must not trigger the upstream
|
||||
# hard-error path or another execution attempt.
|
||||
tool_calls.clear()
|
||||
output[:0] = prior_output
|
||||
"""
|
||||
if source.count(needle) != 1:
|
||||
raise SystemExit(
|
||||
f"expected exactly one Open WebUI stream anchor, found {source.count(needle)}"
|
||||
)
|
||||
source = source.replace(needle, replacement)
|
||||
|
||||
path.write_text(source)
|
||||
Reference in New Issue
Block a user