diff --git a/compose.yaml b/compose.yaml index ddab671..4060d4c 100644 --- a/compose.yaml +++ b/compose.yaml @@ -176,6 +176,11 @@ services: - --jinja - --reasoning - auto + # Bound each individual thinking phase. Long agent jobs can still use + # many phases around tool calls, but one degenerate reasoning loop can + # no longer consume the complete response budget indefinitely. + - --reasoning-budget + - "8192" - --reasoning-preserve - --host - 0.0.0.0 @@ -189,9 +194,11 @@ services: - --no-mmap - --no-ui - --temperature - - "0.2" + # Qwen3.8's official thinking-mode sampler. The former 0.2 setting was + # overly deterministic and could lock reasoning into verbatim loops. + - "1.0" - --top-p - - "0.8" + - "0.95" - --top-k - "20" - --device diff --git a/docs/CURRENT_REFERENCE.md b/docs/CURRENT_REFERENCE.md index 2e40c49..6e6eecd 100644 --- a/docs/CURRENT_REFERENCE.md +++ b/docs/CURRENT_REFERENCE.md @@ -50,7 +50,11 @@ Zielplattform. - ein paralleler Slot - Jinja und automatisches Reasoning - erhaltener Reasoning-Zustand über Werkzeugrunden (`--reasoning-preserve`) -- Temperatur 0,2, Top-p 0,8, Top-k 20 +- Medium/Hermes: Thinking-Sampler gemäß Qwen-Empfehlung mit Temperatur 1,0, + Top-p 0,95 und Top-k 20 +- Medium/Hermes: maximal 8.192 Reasoning-Token pro einzelner Denkphase; + Werkzeugrunden erhalten jeweils eine neue Denkphase +- andere Profile: Temperatur 0,2, Top-p 0,8, Top-k 20 ### Profile diff --git a/platform/hermes/SOUL.md b/platform/hermes/SOUL.md index 94cc3ce..4f8cafd 100644 --- a/platform/hermes/SOUL.md +++ b/platform/hermes/SOUL.md @@ -1,3 +1,9 @@ You are Hermes Agent, an intelligent AI assistant created by Nous Research. You are helpful, knowledgeable, and direct. You assist users with a wide range of tasks including answering questions, writing and editing code, analyzing information, creative work, and executing actions via your tools. You communicate clearly, admit uncertainty when appropriate, and prioritize being genuinely useful over being verbose unless otherwise directed below. Be targeted and efficient in your exploration and investigations. German is the user's default language. Reply in natural German unless the user asks for another language or the task requires preserving source-language content. Technical commands, identifiers, code, paths, product names, and quoted source text may remain in their original language. + +Do not ruminate. If the same uncertainty, hypothesis, or paragraph has already +been considered twice without new evidence, stop reconsidering it. Choose the +best supported interpretation and state the assumption, use a tool that can +resolve it, or ask one concise clarification question. Never repeat an internal +argument verbatim merely because the available evidence is ambiguous. diff --git a/platform/hermes/config.yaml b/platform/hermes/config.yaml index 3477605..4e1d953 100644 --- a/platform/hermes/config.yaml +++ b/platform/hermes/config.yaml @@ -29,6 +29,20 @@ agent: max_turns: 100 gateway_timeout: 3600 session_stall_timeout: 600 + tool_loop_guardrails: + warnings_enabled: true + hard_stop_enabled: true + warn_after: + exact_failure: 2 + same_tool_failure: 3 + idempotent_no_progress: 2 + hard_stop_after: + exact_failure: 5 + same_tool_failure: 8 + idempotent_no_progress: 5 + loop_caps: + max_web_searches: 20 + max_subagents: 8 # German is the platform default. The display setting localizes the static # messages Hermes currently supports; agent replies are governed by SOUL.md. diff --git a/platform/profiles/profile-medium.conf b/platform/profiles/profile-medium.conf index 7f6c078..c1987bf 100644 --- a/platform/profiles/profile-medium.conf +++ b/platform/profiles/profile-medium.conf @@ -3,4 +3,4 @@ Description=Local AI llama.cpp - Qwen Medium 160K IQ4_XS Pure with CPU Vision [Service] ExecStart= -ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 0.2 --top-p 0.8 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16 +ExecStart=/opt/mike-ai/llama.cpp/build/bin/llama-server --model /opt/mike-ai/models/qwen3.8-27b-iq4-xs-pure/qwen3.8-27b-IQ4_XS-pure.gguf --mmproj /opt/mike-ai/models/qwen3.8-27b-nvfp4/mmproj-BF16.gguf --no-mmproj-offload --alias qwen-medium --ctx-size 160000 --flash-attn on --cache-type-k q4_0 --cache-type-v q4_0 --threads 6 --threads-batch 6 --batch-size 64 --ubatch-size 32 --parallel 1 --jinja --reasoning auto --reasoning-budget 8192 --host 127.0.0.1 --port 8080 --metrics --fit off --n-gpu-layers all --no-mmap --temperature 1.0 --top-p 0.95 --top-k 20 --device CUDA0,CUDA1 --main-gpu 0 --split-mode layer --tensor-split 90,10 --spec-type draft-mtp --spec-draft-n-max 3 --spec-draft-type-k f16 --spec-draft-type-v f16