Bound Hermes prompt and reasoning latency
This commit is contained in:
1 parent
37f744eaa2
commit
8dfc879f26
2 files changed
+38
-1
No files matched your search
@@ -6,6 +6,10 @@ model:
|
|||||||
base_url: "http://router:8081/v1"
|
base_url: "http://router:8081/v1"
|
||||||
api_key: "${ROUTER_API_KEY}"
|
api_key: "${ROUTER_API_KEY}"
|
||||||
context_length: 160000
|
context_length: 160000
|
||||||
|
# Applies to visible text, tool calls and hidden reasoning together. Without
|
||||||
|
# a cap a local reasoning model can spend tens of thousands of tokens before
|
||||||
|
# producing its first useful action.
|
||||||
|
max_tokens: 8192
|
||||||
api_mode: "chat_completions"
|
api_mode: "chat_completions"
|
||||||
|
|
||||||
# A named provider is required by current Hermes releases. A bare `custom`
|
# A named provider is required by current Hermes releases. A bare `custom`
|
||||||
@@ -52,6 +56,10 @@ agent:
|
|||||||
# checkpoint in a fresh turn instead of receiving an effectively unlimited
|
# checkpoint in a fresh turn instead of receiving an effectively unlimited
|
||||||
# 500-step budget.
|
# 500-step budget.
|
||||||
max_turns: 64
|
max_turns: 64
|
||||||
|
# Keep enough deliberation for tool choice while avoiding the provider's
|
||||||
|
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
||||||
|
# per session with /reasoning.
|
||||||
|
reasoning_effort: "low"
|
||||||
gateway_timeout: 3600
|
gateway_timeout: 3600
|
||||||
session_stall_timeout: 600
|
session_stall_timeout: 600
|
||||||
tool_loop_guardrails:
|
tool_loop_guardrails:
|
||||||
@@ -76,20 +84,38 @@ compression:
|
|||||||
enabled: true
|
enabled: true
|
||||||
progress_notices: true
|
progress_notices: true
|
||||||
threshold: 0.65
|
threshold: 0.65
|
||||||
|
# A common absolute ceiling keeps all router profiles responsive. It also
|
||||||
|
# prevents a long Medium/Large/Ultra chat from becoming impossible to move
|
||||||
|
# back to Fast later.
|
||||||
|
threshold_tokens: 60000
|
||||||
target_ratio: 0.15
|
target_ratio: 0.15
|
||||||
tail_mode: "lean"
|
tail_mode: "lean"
|
||||||
protect_last_n: 20
|
protect_last_n: 8
|
||||||
|
protect_first_n: 0
|
||||||
proactive_prune_tokens: 50000
|
proactive_prune_tokens: 50000
|
||||||
proactive_prune_min_result_chars: 4000
|
proactive_prune_min_result_chars: 4000
|
||||||
proactive_prune_min_reclaim_tokens: 4096
|
proactive_prune_min_reclaim_tokens: 4096
|
||||||
context_total_ceiling_seconds: 600
|
context_total_ceiling_seconds: 600
|
||||||
|
|
||||||
auxiliary:
|
auxiliary:
|
||||||
|
# Session names are cosmetic and used to create a second concurrent LLM
|
||||||
|
# request after every first reply. On a single inference slot this blocks the
|
||||||
|
# actual chat, so keep the original timestamp/session id instead.
|
||||||
|
title_generation:
|
||||||
|
enabled: false
|
||||||
compression:
|
compression:
|
||||||
provider: "main"
|
provider: "main"
|
||||||
model: ""
|
model: ""
|
||||||
reasoning_effort: "none"
|
reasoning_effort: "none"
|
||||||
|
|
||||||
|
# The full hermes-cli preset injects several large, overlapping schemas on
|
||||||
|
# every turn. Athena already exposes browsing, orchestration and host services
|
||||||
|
# through its MCPs; retain the generally useful local primitives and load MCP
|
||||||
|
# tools through Hermes' deferred discovery. This setting is shared by every
|
||||||
|
# model profile.
|
||||||
|
platform_toolsets:
|
||||||
|
cli: [web, terminal, file, skills, todo, memory, vision, tts]
|
||||||
|
|
||||||
# German is the platform default. The display setting localizes the static
|
# German is the platform default. The display setting localizes the static
|
||||||
# messages Hermes currently supports; agent replies are governed by SOUL.md.
|
# messages Hermes currently supports; agent replies are governed by SOUL.md.
|
||||||
display:
|
display:
|
||||||
|
|||||||
@@ -24,6 +24,17 @@ create_profile() {
|
|||||||
fi
|
fi
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.default "$model"
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.default "$model"
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.context_length "$context"
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.context_length "$context"
|
||||||
|
# These are platform-wide latency and loop safeguards, not model-specific
|
||||||
|
# tuning. Enforce them on old profiles as well as newly cloned profiles.
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens 8192
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.max_turns 64
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort low
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold_tokens 60000
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 8
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \
|
||||||
|
'["web","terminal","file","skills","todo","memory","vision","tts"]'
|
||||||
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.default) == "$model" ]] || \
|
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.default) == "$model" ]] || \
|
||||||
die "Modellalias von Profil $name konnte nicht verifiziert werden."
|
die "Modellalias von Profil $name konnte nicht verifiziert werden."
|
||||||
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.context_length) == "$context" ]] || \
|
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.context_length) == "$context" ]] || \
|
||||||
|
|||||||
Reference in new issue
Block a user