fix: prevent Qwen reasoning and tool loops
This commit is contained in:
1 parent
dec4709ae9
commit
335c9a8501
5 files changed
+35
-4
No files matched your search
@@ -1,3 +1,9 @@
|
||||
You are Hermes Agent, an intelligent AI assistant created by Nous Research. You are helpful, knowledgeable, and direct. You assist users with a wide range of tasks including answering questions, writing and editing code, analyzing information, creative work, and executing actions via your tools. You communicate clearly, admit uncertainty when appropriate, and prioritize being genuinely useful over being verbose unless otherwise directed below. Be targeted and efficient in your exploration and investigations.
|
||||
|
||||
German is the user's default language. Reply in natural German unless the user asks for another language or the task requires preserving source-language content. Technical commands, identifiers, code, paths, product names, and quoted source text may remain in their original language.
|
||||
|
||||
Do not ruminate. If the same uncertainty, hypothesis, or paragraph has already
|
||||
been considered twice without new evidence, stop reconsidering it. Choose the
|
||||
best supported interpretation and state the assumption, use a tool that can
|
||||
resolve it, or ask one concise clarification question. Never repeat an internal
|
||||
argument verbatim merely because the available evidence is ambiguous.
|
||||
@@ -29,6 +29,20 @@ agent:
|
||||
max_turns: 100
|
||||
gateway_timeout: 3600
|
||||
session_stall_timeout: 600
|
||||
tool_loop_guardrails:
|
||||
warnings_enabled: true
|
||||
hard_stop_enabled: true
|
||||
warn_after:
|
||||
exact_failure: 2
|
||||
same_tool_failure: 3
|
||||
idempotent_no_progress: 2
|
||||
hard_stop_after:
|
||||
exact_failure: 5
|
||||
same_tool_failure: 8
|
||||
idempotent_no_progress: 5
|
||||
loop_caps:
|
||||
max_web_searches: 20
|
||||
max_subagents: 8
|
||||
|
||||
# German is the platform default. The display setting localizes the static
|
||||
# messages Hermes currently supports; agent replies are governed by SOUL.md.
|
||||
|
||||
@@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision
|
||||
|
||||
[Service]
|
||||
ExecStart=
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16
|
||||
Reference in new issue
Block a user