Use minimal reasoning for responsive Hermes chats
This commit is contained in:
@@ -140,7 +140,7 @@ Der Router übernimmt:
|
|||||||
80K; alle verwenden dieselben MCPs, Skills, Sprach- und Sicherheitsvorgaben
|
80K; alle verwenden dieselben MCPs, Skills, Sprach- und Sicherheitsvorgaben
|
||||||
- Profilübergreifende Latenzgrenzen: höchstens 8.192 Ausgabetoken je
|
- Profilübergreifende Latenzgrenzen: höchstens 8.192 Ausgabetoken je
|
||||||
Modellaufruf (sichtbare Antwort, Tool-Aufruf und verborgenes Denken
|
Modellaufruf (sichtbare Antwort, Tool-Aufruf und verborgenes Denken
|
||||||
zusammen), Standard-Reasoning `low`; höhere Denkstufen bleiben pro Sitzung
|
zusammen), Standard-Reasoning `minimal`; höhere Denkstufen bleiben pro Sitzung
|
||||||
über `/reasoning` wählbar.
|
über `/reasoning` wählbar.
|
||||||
- Automatische Sitzungstitel sind deaktiviert. Sie sind kosmetisch, erzeugten
|
- Automatische Sitzungstitel sind deaktiviert. Sie sind kosmetisch, erzeugten
|
||||||
aber auf dem einzelnen llama.cpp-Slot nach dem ersten Turn konkurrierende
|
aber auf dem einzelnen llama.cpp-Slot nach dem ersten Turn konkurrierende
|
||||||
|
|||||||
@@ -59,7 +59,7 @@ agent:
|
|||||||
# Keep enough deliberation for tool choice while avoiding the provider's
|
# Keep enough deliberation for tool choice while avoiding the provider's
|
||||||
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
||||||
# per session with /reasoning.
|
# per session with /reasoning.
|
||||||
reasoning_effort: "low"
|
reasoning_effort: "minimal"
|
||||||
gateway_timeout: 3600
|
gateway_timeout: 3600
|
||||||
session_stall_timeout: 600
|
session_stall_timeout: 600
|
||||||
tool_loop_guardrails:
|
tool_loop_guardrails:
|
||||||
|
|||||||
@@ -28,7 +28,7 @@ create_profile() {
|
|||||||
# tuning. Enforce them on old profiles as well as newly cloned profiles.
|
# tuning. Enforce them on old profiles as well as newly cloned profiles.
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens 8192
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens 8192
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.max_turns 64
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.max_turns 64
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort low
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort minimal
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold_tokens 60000
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold_tokens 60000
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 8
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 8
|
||||||
|
|||||||
Reference in New Issue
Block a user