fix: prevent Qwen reasoning and tool loops

This commit is contained in:
Mikei386
2026-08-24 20:06:26 +02:00
parent dec4709ae9
commit 335c9a8501
5 changed files with 35 additions and 4 deletions
+9 -2
View File
@@ -176,6 +176,11 @@ services:
- --jinja
- --reasoning
- auto
# Bound each individual thinking phase. Long agent jobs can still use
# many phases around tool calls, but one degenerate reasoning loop can
# no longer consume the complete response budget indefinitely.
- --reasoning-budget
- "8192"
- --reasoning-preserve
- --host
- 0.0.0.0
@@ -189,9 +194,11 @@ services:
- --no-mmap
- --no-ui
- --temperature
- "0.2"
# Qwen3.8's official thinking-mode sampler. The former 0.2 setting was
# overly deterministic and could lock reasoning into verbatim loops.
- "1.0"
- --top-p
- "0.8"
- "0.95"
- --top-k
- "20"
- --device