feat: harden Qwen agentic tool orchestration

This commit is contained in:
Mikei386
2026-08-24 10:47:13 +02:00
parent 4ac0c479c4
commit 5cb6079918
11 changed files with 371 additions and 62 deletions
+98 -21
View File
@@ -26,10 +26,15 @@ needle = """ tool_call_iterations = 0
"""
replacement = """ tool_call_iterations = 0
# Open WebUI counts batches, while one model turn may request many
# functions in parallel. Bound actual executions as well so a small
# local model cannot expand eight rounds into dozens of API calls.
# functions in parallel. Keep an execution budget as a second bound,
# but make it large enough for genuinely agentic multi-domain work.
# Per-tool and exact-repeat limits below prevent one low-level MCP
# operation from consuming the whole turn.
tool_call_executions = 0
max_tool_call_executions = 6
max_tool_call_executions = 12
max_executions_per_tool = 4
tool_execution_counts = {}
seen_tool_signatures = set()
max_tool_call_iterations = getattr(
"""
if source.count(needle) != 1:
@@ -44,11 +49,30 @@ needle = """ response_tool_calls = tool_calls.pop(0)
"""
replacement = """ response_tool_calls = tool_calls.pop(0)
remaining_tool_calls = max(
0, max_tool_call_executions - tool_call_executions
)
skipped_tool_calls = response_tool_calls[remaining_tool_calls:]
response_tool_calls = response_tool_calls[:remaining_tool_calls]
accepted_tool_calls = []
skipped_tool_calls = []
skip_reasons = []
for candidate in response_tool_calls:
function = candidate.get('function') or candidate
tool_name = function.get('name') or candidate.get('name') or 'unknown'
arguments = function.get('arguments') or candidate.get('arguments') or ''
signature = f'{tool_name}:{arguments}'
if tool_call_executions + len(accepted_tool_calls) >= max_tool_call_executions:
skipped_tool_calls.append(candidate)
skip_reasons.append('total execution budget reached')
continue
if tool_execution_counts.get(tool_name, 0) >= max_executions_per_tool:
skipped_tool_calls.append(candidate)
skip_reasons.append(f'per-tool budget reached for {tool_name}')
continue
if signature in seen_tool_signatures:
skipped_tool_calls.append(candidate)
skip_reasons.append(f'exact duplicate suppressed for {tool_name}')
continue
accepted_tool_calls.append(candidate)
seen_tool_signatures.add(signature)
tool_execution_counts[tool_name] = tool_execution_counts.get(tool_name, 0) + 1
response_tool_calls = accepted_tool_calls
if skipped_tool_calls:
skipped_ids = {call.get('id', '') for call in skipped_tool_calls}
# Responses API streaming may already have exposed all calls
@@ -91,6 +115,7 @@ replacement = """ # The upstream loop otherwise stops wit
and tool_call_iterations >= max_tool_call_iterations
)
or tool_call_executions >= max_tool_call_executions
or bool(skipped_tool_calls)
)
if force_final_response:
new_form_data.pop('tools', None)
@@ -101,16 +126,32 @@ replacement = """ # The upstream loop otherwise stops wit
final_metadata['tool_ids'] = []
final_metadata['tool_servers'] = []
new_form_data['metadata'] = final_metadata
new_form_data['messages'] = add_or_update_system_message(
'The tool budget for this turn is exhausted. Do not call '
'or imitate any more tools. Give the user a concise final '
'answer now using only the tool results already present. '
'Explicitly distinguish verified findings from inference, '
'and state what could not be verified. Never end without a '
'visible answer.',
new_form_data['messages'],
append=True,
final_reason = '; '.join(dict.fromkeys(skip_reasons)) or 'execution budget reached'
final_instruction = (
'The tool research phase is finished (' + final_reason + '). '
'Do not call or imitate any more tools. Give the user a concise '
'final answer now using only the tool results already present. '
'Explicitly distinguish verified findings from inference, state '
'what could not be verified, and never end without a visible answer.'
)
# Qwen3.8 requires system messages to precede the conversation.
# Merge into the first system message (or create it at index 0)
# instead of appending a mid-conversation system message.
system_message = next(
(
message
for message in new_form_data['messages']
if message.get('role') == 'system'
and isinstance(message.get('content'), str)
),
None,
)
if system_message is None:
new_form_data['messages'].insert(
0, {'role': 'system', 'content': final_instruction}
)
else:
system_message['content'] += '\\n\\n' + final_instruction
# A final user-role instruction is deliberately stronger
# than another system suffix after a long tool-call pattern.
# It is request-local and is not added to the saved chat.
@@ -145,10 +186,46 @@ needle = """ await stream_body_handler(res, new_form_
output[:0] = prior_output
"""
replacement = """ await stream_body_handler(res, new_form_data)
if force_final_response:
# A model may still print tool-shaped output after the
# schemas were removed. It must not trigger the upstream
# hard-error path or another execution attempt.
if force_final_response and tool_calls:
# Some local models still emit one more function-shaped
# request after schemas were removed. Discard that
# invisible attempt and grant exactly one clean,
# tool-less synthesis retry. This is bounded and cannot
# execute another function.
tool_calls.clear()
output = []
retry_form_data = {
**new_form_data,
'messages': [
*new_form_data['messages'],
{
'role': 'user',
'content': (
'Your previous synthesis attempt was not '
'visible because it looked like another '
'function call. Tools are no longer available. '
'Write the final human-readable answer now, '
'starting immediately with the conclusion.'
),
},
],
}
retry_res = await generate_chat_completion(
request,
retry_form_data,
user,
bypass_system_prompt=True,
)
if isinstance(retry_res, StreamingResponse):
await stream_body_handler(retry_res, retry_form_data)
# Never let tool-shaped retry output re-enter execution.
tool_calls.clear()
output[:] = [
item
for item in output
if item.get('type') != 'function_call'
]
elif force_final_response:
tool_calls.clear()
output[:0] = prior_output
"""