Use minimal reasoning for responsive Hermes chats
This commit is contained in:
@@ -140,7 +140,7 @@ Der Router übernimmt:
|
||||
80K; alle verwenden dieselben MCPs, Skills, Sprach- und Sicherheitsvorgaben
|
||||
- Profilübergreifende Latenzgrenzen: höchstens 8.192 Ausgabetoken je
|
||||
Modellaufruf (sichtbare Antwort, Tool-Aufruf und verborgenes Denken
|
||||
zusammen), Standard-Reasoning `low`; höhere Denkstufen bleiben pro Sitzung
|
||||
zusammen), Standard-Reasoning `minimal`; höhere Denkstufen bleiben pro Sitzung
|
||||
über `/reasoning` wählbar.
|
||||
- Automatische Sitzungstitel sind deaktiviert. Sie sind kosmetisch, erzeugten
|
||||
aber auf dem einzelnen llama.cpp-Slot nach dem ersten Turn konkurrierende
|
||||
|
||||
@@ -59,7 +59,7 @@ agent:
|
||||
# Keep enough deliberation for tool choice while avoiding the provider's
|
||||
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
||||
# per session with /reasoning.
|
||||
reasoning_effort: "low"
|
||||
reasoning_effort: "minimal"
|
||||
gateway_timeout: 3600
|
||||
session_stall_timeout: 600
|
||||
tool_loop_guardrails:
|
||||
|
||||
@@ -28,7 +28,7 @@ create_profile() {
|
||||
# tuning. Enforce them on old profiles as well as newly cloned profiles.
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens 8192
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.max_turns 64
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort low
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort minimal
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold_tokens 60000
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 8
|
||||
|
||||
Reference in New Issue
Block a user