fix(openwebui): bound calls and force prose synthesis
This commit is contained in:
+1
-1
@@ -4,7 +4,7 @@ MODEL_DIR=/srv/mike-ai/models
|
|||||||
ROUTER_API_KEY=GENERATED_BY_INSTALLER
|
ROUTER_API_KEY=GENERATED_BY_INSTALLER
|
||||||
CONTROLLER_TOKEN=GENERATED_BY_INSTALLER
|
CONTROLLER_TOKEN=GENERATED_BY_INSTALLER
|
||||||
WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER
|
WEBUI_SECRET_KEY=GENERATED_BY_INSTALLER
|
||||||
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v1
|
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v2
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
PIPER_VOICE=de_DE-thorsten-high
|
PIPER_VOICE=de_DE-thorsten-high
|
||||||
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f9d4bc88af94da1b5959d8002f1e0b003c97557164034eb8a29f01b90
|
||||||
|
|||||||
+1
-1
@@ -708,7 +708,7 @@ services:
|
|||||||
build:
|
build:
|
||||||
context: .
|
context: .
|
||||||
dockerfile: platform/openwebui/Dockerfile
|
dockerfile: platform/openwebui/Dockerfile
|
||||||
image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v1}
|
image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v2}
|
||||||
container_name: mike-ai-open-webui
|
container_name: mike-ai-open-webui
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
volumes:
|
volumes:
|
||||||
|
|||||||
@@ -88,7 +88,7 @@ UNCENSORED_MTP_MAX=2
|
|||||||
EXPERIMENTAL_CONTEXT=76800
|
EXPERIMENTAL_CONTEXT=76800
|
||||||
LLAMA_THREADS=6
|
LLAMA_THREADS=6
|
||||||
LLAMA_THREADS_BATCH=6
|
LLAMA_THREADS_BATCH=6
|
||||||
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v1
|
OPENWEBUI_IMAGE=mike-ai/openwebui:main-01f4282-tool-final-v2
|
||||||
OPENWEBUI_ENABLE_SIGNUP=false
|
OPENWEBUI_ENABLE_SIGNUP=false
|
||||||
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false
|
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=false
|
||||||
PIPER_TTS_VERSION=1.6.0
|
PIPER_TTS_VERSION=1.6.0
|
||||||
|
|||||||
@@ -27,7 +27,8 @@ benötigte deshalb kleinere, klarere Werkzeuge und harte Abbruchgrenzen.
|
|||||||
damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem
|
damit führen deutsche Bankexporte nicht mehr unnötig zuerst zu einem
|
||||||
ParserError wegen einer falschen Spaltenzahl. Tabellenanalysen sollen im
|
ParserError wegen einer falschen Spaltenzahl. Tabellenanalysen sollen im
|
||||||
Regelfall mit einer Erkennungs- und einer Auswertungsrunde auskommen.
|
Regelfall mit einer Erkennungs- und einer Auswertungsrunde auskommen.
|
||||||
4. Pro Antwort sind höchstens acht Werkzeugrunden erlaubt. Das abgeleitete,
|
4. Pro Antwort sind höchstens acht Werkzeugrunden und zwölf tatsächlich
|
||||||
|
ausgeführte Einzelaufrufe erlaubt. Das abgeleitete,
|
||||||
reproduzierbar gebaute OpenWebUI-Image verwendet die letzte Runde zwingend
|
reproduzierbar gebaute OpenWebUI-Image verwendet die letzte Runde zwingend
|
||||||
als werkzeugfreie Synthese. Statt `Tool-call limit reached` ohne Ergebnis
|
als werkzeugfreie Synthese. Statt `Tool-call limit reached` ohne Ergebnis
|
||||||
erhält der Benutzer deshalb eine sichtbare Antwort aus den vorhandenen
|
erhält der Benutzer deshalb eine sichtbare Antwort aus den vorhandenen
|
||||||
|
|||||||
+1
-1
@@ -312,7 +312,7 @@ WIREGUARD_CONFIG_FILE=${WIREGUARD_CONFIG_FILE:-/etc/mike-ai/wireguard/fritz-athe
|
|||||||
ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key)
|
ROUTER_API_KEY=$(<$SECRETS_DIR/router-api-key)
|
||||||
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
|
CONTROLLER_TOKEN=$(<$SECRETS_DIR/controller-token)
|
||||||
WEBUI_SECRET_KEY=$(<$SECRETS_DIR/webui-secret)
|
WEBUI_SECRET_KEY=$(<$SECRETS_DIR/webui-secret)
|
||||||
OPENWEBUI_IMAGE=${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v1}
|
OPENWEBUI_IMAGE=${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-tool-final-v2}
|
||||||
OPENWEBUI_ENABLE_SIGNUP=${OPENWEBUI_ENABLE_SIGNUP:-false}
|
OPENWEBUI_ENABLE_SIGNUP=${OPENWEBUI_ENABLE_SIGNUP:-false}
|
||||||
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
|
OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION=${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
|
||||||
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
|
PIPER_TTS_VERSION=${PIPER_TTS_VERSION:-1.6.0}
|
||||||
|
|||||||
@@ -1,7 +1,7 @@
|
|||||||
#!/usr/bin/env python3
|
#!/usr/bin/env python3
|
||||||
"""Make Open WebUI finish with an answer when its internal tool budget is used.
|
"""Make Open WebUI finish with an answer when its internal tool budget is used.
|
||||||
|
|
||||||
Open WebUI 0.9.5 otherwise ends the request with only
|
The pinned Open WebUI revision otherwise ends the request with only
|
||||||
``Tool-call limit reached``. The patch is deliberately assertion-based: an
|
``Tool-call limit reached``. The patch is deliberately assertion-based: an
|
||||||
upstream source change makes the image build fail instead of silently applying
|
upstream source change makes the image build fail instead of silently applying
|
||||||
the modification at the wrong location.
|
the modification at the wrong location.
|
||||||
@@ -21,6 +21,57 @@ patch_marker = "The tool budget for this turn is exhausted. Do not call"
|
|||||||
if patch_marker in source:
|
if patch_marker in source:
|
||||||
raise SystemExit("Open WebUI tool-finalization patch is already present")
|
raise SystemExit("Open WebUI tool-finalization patch is already present")
|
||||||
|
|
||||||
|
needle = """ tool_call_iterations = 0
|
||||||
|
max_tool_call_iterations = getattr(
|
||||||
|
"""
|
||||||
|
replacement = """ tool_call_iterations = 0
|
||||||
|
# Open WebUI counts batches, while one model turn may request many
|
||||||
|
# functions in parallel. Bound actual executions as well so a small
|
||||||
|
# local model cannot expand eight rounds into dozens of API calls.
|
||||||
|
tool_call_executions = 0
|
||||||
|
max_tool_call_executions = 12
|
||||||
|
max_tool_call_iterations = getattr(
|
||||||
|
"""
|
||||||
|
if source.count(needle) != 1:
|
||||||
|
raise SystemExit(
|
||||||
|
f"expected exactly one Open WebUI tool counter anchor, found {source.count(needle)}"
|
||||||
|
)
|
||||||
|
source = source.replace(needle, replacement)
|
||||||
|
|
||||||
|
needle = """ response_tool_calls = tool_calls.pop(0)
|
||||||
|
|
||||||
|
# Append function_call items for each tool call
|
||||||
|
"""
|
||||||
|
replacement = """ response_tool_calls = tool_calls.pop(0)
|
||||||
|
|
||||||
|
remaining_tool_calls = max(
|
||||||
|
0, max_tool_call_executions - tool_call_executions
|
||||||
|
)
|
||||||
|
skipped_tool_calls = response_tool_calls[remaining_tool_calls:]
|
||||||
|
response_tool_calls = response_tool_calls[:remaining_tool_calls]
|
||||||
|
if skipped_tool_calls:
|
||||||
|
skipped_ids = {call.get('id', '') for call in skipped_tool_calls}
|
||||||
|
# Responses API streaming may already have exposed all calls
|
||||||
|
# in `output`. Remove deliberately skipped calls so the next
|
||||||
|
# completion never receives an orphan function call.
|
||||||
|
output[:] = [
|
||||||
|
item
|
||||||
|
for item in output
|
||||||
|
if not (
|
||||||
|
item.get('type') == 'function_call'
|
||||||
|
and item.get('call_id', '') in skipped_ids
|
||||||
|
)
|
||||||
|
]
|
||||||
|
tool_call_executions += len(response_tool_calls)
|
||||||
|
|
||||||
|
# Append function_call items for each tool call
|
||||||
|
"""
|
||||||
|
if source.count(needle) != 1:
|
||||||
|
raise SystemExit(
|
||||||
|
f"expected exactly one Open WebUI tool batch anchor, found {source.count(needle)}"
|
||||||
|
)
|
||||||
|
source = source.replace(needle, replacement)
|
||||||
|
|
||||||
needle = """ res = await generate_chat_completion(
|
needle = """ res = await generate_chat_completion(
|
||||||
request,
|
request,
|
||||||
new_form_data,
|
new_form_data,
|
||||||
@@ -35,12 +86,16 @@ replacement = """ # The upstream loop otherwise stops wit
|
|||||||
# result, but expose no tools to the model and explicitly require
|
# result, but expose no tools to the model and explicitly require
|
||||||
# a useful, evidence-bounded answer.
|
# a useful, evidence-bounded answer.
|
||||||
force_final_response = (
|
force_final_response = (
|
||||||
|
(
|
||||||
max_tool_call_iterations is not None
|
max_tool_call_iterations is not None
|
||||||
and tool_call_iterations >= max_tool_call_iterations
|
and tool_call_iterations >= max_tool_call_iterations
|
||||||
)
|
)
|
||||||
|
or tool_call_executions >= max_tool_call_executions
|
||||||
|
)
|
||||||
if force_final_response:
|
if force_final_response:
|
||||||
new_form_data.pop('tools', None)
|
new_form_data.pop('tools', None)
|
||||||
new_form_data.pop('tool_ids', None)
|
new_form_data.pop('tool_ids', None)
|
||||||
|
new_form_data.pop('tool_choice', None)
|
||||||
final_metadata = dict(metadata)
|
final_metadata = dict(metadata)
|
||||||
final_metadata['tools'] = {}
|
final_metadata['tools'] = {}
|
||||||
final_metadata['tool_ids'] = []
|
final_metadata['tool_ids'] = []
|
||||||
@@ -56,6 +111,21 @@ replacement = """ # The upstream loop otherwise stops wit
|
|||||||
new_form_data['messages'],
|
new_form_data['messages'],
|
||||||
append=True,
|
append=True,
|
||||||
)
|
)
|
||||||
|
# A final user-role instruction is deliberately stronger
|
||||||
|
# than another system suffix after a long tool-call pattern.
|
||||||
|
# It is request-local and is not added to the saved chat.
|
||||||
|
new_form_data['messages'].append(
|
||||||
|
{
|
||||||
|
'role': 'user',
|
||||||
|
'content': (
|
||||||
|
'The research phase is finished. Answer my original '
|
||||||
|
'question now in normal prose. Do not output XML, JSON, '
|
||||||
|
'function names, tool_call blocks, or requests for more '
|
||||||
|
'files. Use the available evidence, mention uncertainty, '
|
||||||
|
'and provide a useful final conclusion.'
|
||||||
|
),
|
||||||
|
}
|
||||||
|
)
|
||||||
|
|
||||||
res = await generate_chat_completion(
|
res = await generate_chat_completion(
|
||||||
request,
|
request,
|
||||||
|
|||||||
Reference in New Issue
Block a user