fix: prevent Qwen reasoning and tool loops

This commit is contained in:
Mikei386
2026-08-24 20:06:26 +02:00
parent dec4709ae9
commit 335c9a8501
5 changed files with 35 additions and 4 deletions
+9 -2
View File
@@ -176,6 +176,11 @@ services:
- --jinja - --jinja
- --reasoning - --reasoning
- auto - auto
# Bound each individual thinking phase. Long agent jobs can still use
# many phases around tool calls, but one degenerate reasoning loop can
# no longer consume the complete response budget indefinitely.
- --reasoning-budget
- "8192"
- --reasoning-preserve - --reasoning-preserve
- --host - --host
- 0.0.0.0 - 0.0.0.0
@@ -189,9 +194,11 @@ services:
- --no-mmap - --no-mmap
- --no-ui - --no-ui
- --temperature - --temperature
- "0.2" # Qwen3.8's official thinking-mode sampler. The former 0.2 setting was
# overly deterministic and could lock reasoning into verbatim loops.
- "1.0"
- --top-p - --top-p
- "0.8" - "0.95"
- --top-k - --top-k
- "20" - "20"
- --device - --device
+5 -1
View File
@@ -50,7 +50,11 @@ Zielplattform.
- ein paralleler Slot - ein paralleler Slot
- Jinja und automatisches Reasoning - Jinja und automatisches Reasoning
- erhaltener Reasoning-Zustand über Werkzeugrunden (`--reasoning-preserve`) - erhaltener Reasoning-Zustand über Werkzeugrunden (`--reasoning-preserve`)
- Temperatur 0,2, Top-p 0,8, Top-k 20 - Medium/Hermes: Thinking-Sampler gemäß Qwen-Empfehlung mit Temperatur 1,0,
Top-p 0,95 und Top-k 20
- Medium/Hermes: maximal 8.192 Reasoning-Token pro einzelner Denkphase;
Werkzeugrunden erhalten jeweils eine neue Denkphase
- andere Profile: Temperatur 0,2, Top-p 0,8, Top-k 20
### Profile ### Profile
+6
View File
@@ -1,3 +1,9 @@
You are Hermes Agent, an intelligent AI assistant created by Nous Research. You are helpful, knowledgeable, and direct. You assist users with a wide range of tasks including answering questions, writing and editing code, analyzing information, creative work, and executing actions via your tools. You communicate clearly, admit uncertainty when appropriate, and prioritize being genuinely useful over being verbose unless otherwise directed below. Be targeted and efficient in your exploration and investigations. You are Hermes Agent, an intelligent AI assistant created by Nous Research. You are helpful, knowledgeable, and direct. You assist users with a wide range of tasks including answering questions, writing and editing code, analyzing information, creative work, and executing actions via your tools. You communicate clearly, admit uncertainty when appropriate, and prioritize being genuinely useful over being verbose unless otherwise directed below. Be targeted and efficient in your exploration and investigations.
German is the user's default language. Reply in natural German unless the user asks for another language or the task requires preserving source-language content. Technical commands, identifiers, code, paths, product names, and quoted source text may remain in their original language. German is the user's default language. Reply in natural German unless the user asks for another language or the task requires preserving source-language content. Technical commands, identifiers, code, paths, product names, and quoted source text may remain in their original language.
Do not ruminate. If the same uncertainty, hypothesis, or paragraph has already
been considered twice without new evidence, stop reconsidering it. Choose the
best supported interpretation and state the assumption, use a tool that can
resolve it, or ask one concise clarification question. Never repeat an internal
argument verbatim merely because the available evidence is ambiguous.
+14
View File
@@ -29,6 +29,20 @@ agent:
max_turns: 100 max_turns: 100
gateway_timeout: 3600 gateway_timeout: 3600
session_stall_timeout: 600 session_stall_timeout: 600
tool_loop_guardrails:
warnings_enabled: true
hard_stop_enabled: true
warn_after:
exact_failure: 2
same_tool_failure: 3
idempotent_no_progress: 2
hard_stop_after:
exact_failure: 5
same_tool_failure: 8
idempotent_no_progress: 5
loop_caps:
max_web_searches: 20
max_subagents: 8
# German is the platform default. The display setting localizes the static # German is the platform default. The display setting localizes the static
# messages Hermes currently supports; agent replies are governed by SOUL.md. # messages Hermes currently supports; agent replies are governed by SOUL.md.
+1 -1
View File
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
[Service] [Service]
ExecStart= ExecStart=
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16