fix(openwebui): bound calls and force prose synthesis

This commit is contained in:
Mikei386
2026-08-24 08:33:44 +02:00
parent 4d4eb0b8d8
commit 50c37b09bb
6 changed files with 79 additions and 8 deletions
+1 -1
View File
@@ -4,7 +4,7 @@ MODEL_DIR=/srv/mike-ai/models
ROUTER_API_KEY=GENERATED_BY_INSTALLER
CONTROLLER_TOKEN=GENERATED_BY_INSTALLER
WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v1
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v2
PIPER_TTS_VERSION=1.6.0
PIPER_VOICE=de_DE-thorsten-high
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
+1 -1
View File
@@ -708,7 +708,7 @@ services:
build:
context: .
dockerfile: platform/openwebui/Dockerfile
image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v1}
image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v2}
container_name: mike-ai-open-webui
restart: unless-stopped
volumes:
+1 -1
View File
@@ -88,7 +88,7 @@ UNCENSORED_MTP_MAX=2
EXPERIMENTAL_CONTEXT=76800
LLAMA_THREADS=6
LLAMA_THREADS_BATCH=6
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v1
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v2
OPENWEBUI_ENABLE_SIGNUP=false
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false
PIPER_TTS_VERSION=1.6.0
+2 -1
View File
@@ -27,7 +27,8 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen.
damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem
ParserError wegen einer falschen Spaltenzahl. Tabellenanalysen sollen im
Regelfall mit einer Erkennungs- und einer Auswertungsrunde auskommen.
4. Pro Antwort sind höchstens acht Werkzeugrunden erlaubt. Das abgeleitete,
4. Pro Antwort sind höchstens acht Werkzeugrunden und zwölf tatsächlich
ausgeführte Einzelaufrufe erlaubt. Das abgeleitete,
reproduzierbar gebaute OpenWebUI-Image verwendet die letzte Runde zwingend
als werkzeugfreie Synthese. Statt `Tool-call limit reached` ohne Ergebnis
erhält der Benutzer deshalb eine sichtbare Antwort aus den vorhandenen
+1 -1
View File
@@ -312,7 +312,7 @@ WIREGUARD_CONFIG_FILE=${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athe
ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key)
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
WEBUI_SECRET_KEY=$(<$SECRETS_DIR/webui-secret)
OPENWEBUI_IMAGE=${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v1}
OPENWEBUI_IMAGE=${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v2}
OPENWEBUI_ENABLE_SIGNUP=${OPENWEBUI_ENABLE_SIGNUP:-false}
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
+73 -3
View File
@@ -1,7 +1,7 @@
#!/usr/bin/env python3
"""Make Open WebUI finish with an answer when its internal tool budget is used.
Open WebUI 0.9.5 otherwise ends the request with only
The pinned Open WebUI revision otherwise ends the request with only
``Tool-call limit reached``. The patch is deliberately assertion-based: an
upstream source change makes the image build fail instead of silently applying
the modification at the wrong location.
@@ -21,6 +21,57 @@ patch_marker = "The tool budget for this turn is exhausted. Do not call"
if patch_marker in source:
raise SystemExit("Open WebUI tool-finalization patch is already present")
needle = """ tool_call_iterations = 0
max_tool_call_iterations = getattr(
"""
replacement = """ tool_call_iterations = 0
# Open WebUI counts batches, while one model turn may request many
# functions in parallel. Bound actual executions as well so a small
# local model cannot expand eight rounds into dozens of API calls.
tool_call_executions = 0
max_tool_call_executions = 12
max_tool_call_iterations = getattr(
"""
if source.count(needle) != 1:
raise SystemExit(
f"expected exactly one Open WebUI tool counter anchor, found {source.count(needle)}"
)
source = source.replace(needle, replacement)
needle = """ response_tool_calls = tool_calls.pop(0)
# Append function_call items for each tool call
"""
replacement = """ response_tool_calls = tool_calls.pop(0)
remaining_tool_calls = max(
0, max_tool_call_executions - tool_call_executions
)
skipped_tool_calls = response_tool_calls[remaining_tool_calls:]
response_tool_calls = response_tool_calls[:remaining_tool_calls]
if skipped_tool_calls:
skipped_ids = {call.get('id', '') for call in skipped_tool_calls}
# Responses API streaming may already have exposed all calls
# in `output`. Remove deliberately skipped calls so the next
# completion never receives an orphan function call.
output[:] = [
item
for item in output
if not (
item.get('type') == 'function_call'
and item.get('call_id', '') in skipped_ids
)
]
tool_call_executions += len(response_tool_calls)
# Append function_call items for each tool call
"""
if source.count(needle) != 1:
raise SystemExit(
f"expected exactly one Open WebUI tool batch anchor, found {source.count(needle)}"
)
source = source.replace(needle, replacement)
needle = """ res = await generate_chat_completion(
request,
new_form_data,
@@ -35,12 +86,16 @@ replacement = """ # The upstream loop otherwise stops wit
# result, but expose no tools to the model and explicitly require
# a useful, evidence-bounded answer.
force_final_response = (
max_tool_call_iterations is not None
and tool_call_iterations >= max_tool_call_iterations
(
max_tool_call_iterations is not None
and tool_call_iterations >= max_tool_call_iterations
)
or tool_call_executions >= max_tool_call_executions
)
if force_final_response:
new_form_data.pop('tools', None)
new_form_data.pop('tool_ids', None)
new_form_data.pop('tool_choice', None)
final_metadata = dict(metadata)
final_metadata['tools'] = {}
final_metadata['tool_ids'] = []
@@ -56,6 +111,21 @@ replacement = """ # The upstream loop otherwise stops wit
new_form_data['messages'],
append=True,
)
# A final user-role instruction is deliberately stronger
# than another system suffix after a long tool-call pattern.
# It is request-local and is not added to the saved chat.
new_form_data['messages'].append(
{
'role': 'user',
'content': (
'The research phase is finished. Answer my original '
'question now in normal prose. Do not output XML, JSON, '
'function names, tool_call blocks, or requests for more '
'files. Use the available evidence, mention uncertainty, '
'and provide a useful final conclusion.'
),
}
)
res = await generate_chat_completion(
request,