diff --git a/.env.example b/.env.example index 5afb7af..da5e535 100644 --- a/.env.example +++ b/.env.example @@ -4,7 +4,7 @@ MODEL_DIR=/srv/mike-ai/models ROUTER_API_KEY=GENERATED_BY_INSTALLER CONTROLLER_TOKEN=GENERATED_BY_INSTALLER WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER -OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v1 +OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v2 PIPER_TTS_VERSION=1.6.0 PIPER_VOICE=de_DE-thorsten-high XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90 diff --git a/compose.yaml b/compose.yaml index d2e74ec..19f3e13 100644 --- a/compose.yaml +++ b/compose.yaml @@ -708,7 +708,7 @@ services: build: context: . dockerfile: platform/openwebui/Dockerfile - image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v1} + image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v2} container_name: mike-ai-open-webui restart: unless-stopped volumes: diff --git a/config/install.env.example b/config/install.env.example index 296ce24..7ac3f32 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -88,7 +88,7 @@ UNCENSORED_MTP_MAX=2 EXPERIMENTAL_CONTEXT=76800 LLAMA_THREADS=6 LLAMA_THREADS_BATCH=6 -OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v1 +OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v2 OPENWEBUI_ENABLE_SIGNUP=false OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false PIPER_TTS_VERSION=1.6.0 diff --git a/docs/TOOLING_RELIABILITY_2026-08-24.md b/docs/TOOLING_RELIABILITY_2026-08-24.md index 715ea6e..03c82f8 100644 --- a/docs/TOOLING_RELIABILITY_2026-08-24.md +++ b/docs/TOOLING_RELIABILITY_2026-08-24.md @@ -27,7 +27,8 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen. damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem ParserError wegen einer falschen Spaltenzahl. Tabellenanalysen sollen im Regelfall mit einer Erkennungs- und einer Auswertungsrunde auskommen. -4. Pro Antwort sind höchstens acht Werkzeugrunden erlaubt. Das abgeleitete, +4. Pro Antwort sind höchstens acht Werkzeugrunden und zwölf tatsächlich + ausgeführte Einzelaufrufe erlaubt. Das abgeleitete, reproduzierbar gebaute OpenWebUI-Image verwendet die letzte Runde zwingend als werkzeugfreie Synthese. Statt `Tool-call limit reached` ohne Ergebnis erhält der Benutzer deshalb eine sichtbare Antwort aus den vorhandenen diff --git a/install.sh b/install.sh index 58fa4b9..3b9c5e4 100755 --- a/install.sh +++ b/install.sh @@ -312,7 +312,7 @@ WIREGUARD_CONFIG_FILE=${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athe ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key) CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token) WEBUI_SECRET_KEY=$(<$SECRETS_DIR/webui-secret) -OPENWEBUI_IMAGE=${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v1} +OPENWEBUI_IMAGE=${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v2} OPENWEBUI_ENABLE_SIGNUP=${OPENWEBUI_ENABLE_SIGNUP:-false} OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false} PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0} diff --git a/platform/openwebui/patch_tool_finalization.py b/platform/openwebui/patch_tool_finalization.py index 4ff9c7d..58bd89d 100644 --- a/platform/openwebui/patch_tool_finalization.py +++ b/platform/openwebui/patch_tool_finalization.py @@ -1,7 +1,7 @@ #!/usr/bin/env python3 """Make Open WebUI finish with an answer when its internal tool budget is used. -Open WebUI 0.9.5 otherwise ends the request with only +The pinned Open WebUI revision otherwise ends the request with only ``Tool-call limit reached``. The patch is deliberately assertion-based: an upstream source change makes the image build fail instead of silently applying the modification at the wrong location. @@ -21,6 +21,57 @@ patch_marker = "The tool budget for this turn is exhausted. Do not call" if patch_marker in source: raise SystemExit("Open WebUI tool-finalization patch is already present") +needle = """ tool_call_iterations = 0 + max_tool_call_iterations = getattr( +""" +replacement = """ tool_call_iterations = 0 + # Open WebUI counts batches, while one model turn may request many + # functions in parallel. Bound actual executions as well so a small + # local model cannot expand eight rounds into dozens of API calls. + tool_call_executions = 0 + max_tool_call_executions = 12 + max_tool_call_iterations = getattr( +""" +if source.count(needle) != 1: + raise SystemExit( + f"expected exactly one Open WebUI tool counter anchor, found {source.count(needle)}" + ) +source = source.replace(needle, replacement) + +needle = """ response_tool_calls = tool_calls.pop(0) + + # Append function_call items for each tool call +""" +replacement = """ response_tool_calls = tool_calls.pop(0) + + remaining_tool_calls = max( + 0, max_tool_call_executions - tool_call_executions + ) + skipped_tool_calls = response_tool_calls[remaining_tool_calls:] + response_tool_calls = response_tool_calls[:remaining_tool_calls] + if skipped_tool_calls: + skipped_ids = {call.get('id', '') for call in skipped_tool_calls} + # Responses API streaming may already have exposed all calls + # in `output`. Remove deliberately skipped calls so the next + # completion never receives an orphan function call. + output[:] = [ + item + for item in output + if not ( + item.get('type') == 'function_call' + and item.get('call_id', '') in skipped_ids + ) + ] + tool_call_executions += len(response_tool_calls) + + # Append function_call items for each tool call +""" +if source.count(needle) != 1: + raise SystemExit( + f"expected exactly one Open WebUI tool batch anchor, found {source.count(needle)}" + ) +source = source.replace(needle, replacement) + needle = """ res = await generate_chat_completion( request, new_form_data, @@ -35,12 +86,16 @@ replacement = """ # The upstream loop otherwise stops wit # result, but expose no tools to the model and explicitly require # a useful, evidence-bounded answer. force_final_response = ( - max_tool_call_iterations is not None - and tool_call_iterations >= max_tool_call_iterations + ( + max_tool_call_iterations is not None + and tool_call_iterations >= max_tool_call_iterations + ) + or tool_call_executions >= max_tool_call_executions ) if force_final_response: new_form_data.pop('tools', None) new_form_data.pop('tool_ids', None) + new_form_data.pop('tool_choice', None) final_metadata = dict(metadata) final_metadata['tools'] = {} final_metadata['tool_ids'] = [] @@ -56,6 +111,21 @@ replacement = """ # The upstream loop otherwise stops wit new_form_data['messages'], append=True, ) + # A final user-role instruction is deliberately stronger + # than another system suffix after a long tool-call pattern. + # It is request-local and is not added to the saved chat. + new_form_data['messages'].append( + { + 'role': 'user', + 'content': ( + 'The research phase is finished. Answer my original ' + 'question now in normal prose. Do not output XML, JSON, ' + 'function names, tool_call blocks, or requests for more ' + 'files. Use the available evidence, mention uncertainty, ' + 'and provide a useful final conclusion.' + ), + } + ) res = await generate_chat_completion( request,