239 lines
12 KiB
Python
239 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""Make Open WebUI finish with an answer when its internal tool budget is used.
|
|
|
|
The pinned Open WebUI revision otherwise ends the request with only
|
|
``Tool-call limit reached``. The patch is deliberately assertion-based: an
|
|
upstream source change makes the image build fail instead of silently applying
|
|
the modification at the wrong location.
|
|
"""
|
|
|
|
from pathlib import Path
|
|
import sys
|
|
|
|
|
|
if len(sys.argv) != 2:
|
|
raise SystemExit("usage: patch_tool_finalization.py MIDDLEWARE_PY")
|
|
|
|
path = Path(sys.argv[1])
|
|
source = path.read_text()
|
|
|
|
patch_marker = "The tool budget for this turn is exhausted. Do not call"
|
|
if patch_marker in source:
|
|
raise SystemExit("Open WebUI tool-finalization patch is already present")
|
|
|
|
needle = """ tool_call_iterations = 0
|
|
max_tool_call_iterations = getattr(
|
|
"""
|
|
replacement = """ tool_call_iterations = 0
|
|
# Open WebUI counts batches, while one model turn may request many
|
|
# functions in parallel. Keep an execution budget as a second bound,
|
|
# but make it large enough for genuinely agentic multi-domain work.
|
|
# Per-tool and exact-repeat limits below prevent one low-level MCP
|
|
# operation from consuming the whole turn.
|
|
tool_call_executions = 0
|
|
max_tool_call_executions = 40
|
|
max_executions_per_tool = 12
|
|
tool_execution_counts = {}
|
|
tool_signature_counts = {}
|
|
max_tool_call_iterations = getattr(
|
|
"""
|
|
if source.count(needle) != 1:
|
|
raise SystemExit(
|
|
f"expected exactly one Open WebUI tool counter anchor, found {source.count(needle)}"
|
|
)
|
|
source = source.replace(needle, replacement)
|
|
|
|
needle = """ response_tool_calls = tool_calls.pop(0)
|
|
|
|
# Append function_call items for each tool call
|
|
"""
|
|
replacement = """ response_tool_calls = tool_calls.pop(0)
|
|
|
|
accepted_tool_calls = []
|
|
skipped_tool_calls = []
|
|
skip_reasons = []
|
|
for candidate in response_tool_calls:
|
|
function = candidate.get('function') or candidate
|
|
tool_name = function.get('name') or candidate.get('name') or 'unknown'
|
|
arguments = function.get('arguments') or candidate.get('arguments') or ''
|
|
signature = f'{tool_name}:{arguments}'
|
|
if tool_call_executions + len(accepted_tool_calls) >= max_tool_call_executions:
|
|
skipped_tool_calls.append(candidate)
|
|
skip_reasons.append('total execution budget reached')
|
|
continue
|
|
if tool_execution_counts.get(tool_name, 0) >= max_executions_per_tool:
|
|
skipped_tool_calls.append(candidate)
|
|
skip_reasons.append(f'per-tool budget reached for {tool_name}')
|
|
continue
|
|
if tool_signature_counts.get(signature, 0) >= 2:
|
|
skipped_tool_calls.append(candidate)
|
|
skip_reasons.append(f'repeated identical call suppressed for {tool_name}')
|
|
continue
|
|
accepted_tool_calls.append(candidate)
|
|
tool_signature_counts[signature] = tool_signature_counts.get(signature, 0) + 1
|
|
tool_execution_counts[tool_name] = tool_execution_counts.get(tool_name, 0) + 1
|
|
response_tool_calls = accepted_tool_calls
|
|
if skipped_tool_calls:
|
|
skipped_ids = {call.get('id', '') for call in skipped_tool_calls}
|
|
# Responses API streaming may already have exposed all calls
|
|
# in `output`. Remove deliberately skipped calls so the next
|
|
# completion never receives an orphan function call.
|
|
output[:] = [
|
|
item
|
|
for item in output
|
|
if not (
|
|
item.get('type') == 'function_call'
|
|
and item.get('call_id', '') in skipped_ids
|
|
)
|
|
]
|
|
tool_call_executions += len(response_tool_calls)
|
|
|
|
# Append function_call items for each tool call
|
|
"""
|
|
if source.count(needle) != 1:
|
|
raise SystemExit(
|
|
f"expected exactly one Open WebUI tool batch anchor, found {source.count(needle)}"
|
|
)
|
|
source = source.replace(needle, replacement)
|
|
|
|
needle = """ res = await generate_chat_completion(
|
|
request,
|
|
new_form_data,
|
|
user,
|
|
bypass_system_prompt=True,
|
|
)
|
|
"""
|
|
|
|
replacement = """ # The upstream loop otherwise stops with only
|
|
# `Tool-call limit reached` after executing this batch. Reserve
|
|
# the final completion for synthesis: keep every accumulated
|
|
# result, but expose no tools to the model and explicitly require
|
|
# a useful, evidence-bounded answer.
|
|
force_final_response = (
|
|
(
|
|
max_tool_call_iterations is not None
|
|
and tool_call_iterations >= max_tool_call_iterations
|
|
)
|
|
or tool_call_executions >= max_tool_call_executions
|
|
or bool(skipped_tool_calls and not response_tool_calls)
|
|
)
|
|
if force_final_response:
|
|
new_form_data.pop('tools', None)
|
|
new_form_data.pop('tool_ids', None)
|
|
new_form_data.pop('tool_choice', None)
|
|
final_metadata = dict(metadata)
|
|
final_metadata['tools'] = {}
|
|
final_metadata['tool_ids'] = []
|
|
final_metadata['tool_servers'] = []
|
|
new_form_data['metadata'] = final_metadata
|
|
final_reason = '; '.join(dict.fromkeys(skip_reasons)) or 'execution budget reached'
|
|
final_instruction = (
|
|
'The tool research phase is finished (' + final_reason + '). '
|
|
'Do not call or imitate any more tools. Give the user a concise '
|
|
'final answer now using only the tool results already present. '
|
|
'Explicitly distinguish verified findings from inference, state '
|
|
'what could not be verified, and never end without a visible answer.'
|
|
)
|
|
# Qwen3.8 requires system messages to precede the conversation.
|
|
# Merge into the first system message (or create it at index 0)
|
|
# instead of appending a mid-conversation system message.
|
|
system_message = next(
|
|
(
|
|
message
|
|
for message in new_form_data['messages']
|
|
if message.get('role') == 'system'
|
|
and isinstance(message.get('content'), str)
|
|
),
|
|
None,
|
|
)
|
|
if system_message is None:
|
|
new_form_data['messages'].insert(
|
|
0, {'role': 'system', 'content': final_instruction}
|
|
)
|
|
else:
|
|
system_message['content'] += '\\n\\n' + final_instruction
|
|
# A final user-role instruction is deliberately stronger
|
|
# than another system suffix after a long tool-call pattern.
|
|
# It is request-local and is not added to the saved chat.
|
|
new_form_data['messages'].append(
|
|
{
|
|
'role': 'user',
|
|
'content': (
|
|
'The research phase is finished. Answer my original '
|
|
'question now in normal prose. Do not output XML, JSON, '
|
|
'function names, tool_call blocks, or requests for more '
|
|
'files. Use the available evidence, mention uncertainty, '
|
|
'and provide a useful final conclusion.'
|
|
),
|
|
}
|
|
)
|
|
|
|
res = await generate_chat_completion(
|
|
request,
|
|
new_form_data,
|
|
user,
|
|
bypass_system_prompt=True,
|
|
)
|
|
"""
|
|
|
|
if source.count(needle) != 1:
|
|
raise SystemExit(
|
|
f"expected exactly one Open WebUI continuation anchor, found {source.count(needle)}"
|
|
)
|
|
source = source.replace(needle, replacement)
|
|
|
|
needle = """ await stream_body_handler(res, new_form_data)
|
|
output[:0] = prior_output
|
|
"""
|
|
replacement = """ await stream_body_handler(res, new_form_data)
|
|
if force_final_response and tool_calls:
|
|
# Some local models still emit one more function-shaped
|
|
# request after schemas were removed. Discard that
|
|
# invisible attempt and grant exactly one clean,
|
|
# tool-less synthesis retry. This is bounded and cannot
|
|
# execute another function.
|
|
tool_calls.clear()
|
|
output = []
|
|
retry_form_data = {
|
|
**new_form_data,
|
|
'messages': [
|
|
*new_form_data['messages'],
|
|
{
|
|
'role': 'user',
|
|
'content': (
|
|
'Your previous synthesis attempt was not '
|
|
'visible because it looked like another '
|
|
'function call. Tools are no longer available. '
|
|
'Write the final human-readable answer now, '
|
|
'starting immediately with the conclusion.'
|
|
),
|
|
},
|
|
],
|
|
}
|
|
retry_res = await generate_chat_completion(
|
|
request,
|
|
retry_form_data,
|
|
user,
|
|
bypass_system_prompt=True,
|
|
)
|
|
if isinstance(retry_res, StreamingResponse):
|
|
await stream_body_handler(retry_res, retry_form_data)
|
|
# Never let tool-shaped retry output re-enter execution.
|
|
tool_calls.clear()
|
|
output[:] = [
|
|
item
|
|
for item in output
|
|
if item.get('type') != 'function_call'
|
|
]
|
|
elif force_final_response:
|
|
tool_calls.clear()
|
|
output[:0] = prior_output
|
|
"""
|
|
if source.count(needle) != 1:
|
|
raise SystemExit(
|
|
f"expected exactly one Open WebUI stream anchor, found {source.count(needle)}"
|
|
)
|
|
source = source.replace(needle, replacement)
|
|
|
|
path.write_text(source)
|