Bound Hermes prompt and reasoning latency

This commit is contained in:
Mikei386
2026-08-25 18:01:52 +02:00
parent 37f744eaa2
commit 8dfc879f26
2 changed files with 38 additions and 1 deletions
+27 -1
View File
@@ -6,6 +6,10 @@ model:
base_url: "http://router:8081/v1" base_url: "http://router:8081/v1"
api_key: "${ROUTER_API_KEY}" api_key: "${ROUTER_API_KEY}"
context_length: 160000 context_length: 160000
# Applies to visible text, tool calls and hidden reasoning together. Without
# a cap a local reasoning model can spend tens of thousands of tokens before
# producing its first useful action.
max_tokens: 8192
api_mode: "chat_completions" api_mode: "chat_completions"
# A named provider is required by current Hermes releases. A bare `custom` # A named provider is required by current Hermes releases. A bare `custom`
@@ -52,6 +56,10 @@ agent:
# checkpoint in a fresh turn instead of receiving an effectively unlimited # checkpoint in a fresh turn instead of receiving an effectively unlimited
# 500-step budget. # 500-step budget.
max_turns: 64 max_turns: 64
# Keep enough deliberation for tool choice while avoiding the provider's
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
# per session with /reasoning.
reasoning_effort: "low"
gateway_timeout: 3600 gateway_timeout: 3600
session_stall_timeout: 600 session_stall_timeout: 600
tool_loop_guardrails: tool_loop_guardrails:
@@ -76,20 +84,38 @@ compression:
enabled: true enabled: true
progress_notices: true progress_notices: true
threshold: 0.65 threshold: 0.65
# A common absolute ceiling keeps all router profiles responsive. It also
# prevents a long Medium/Large/Ultra chat from becoming impossible to move
# back to Fast later.
threshold_tokens: 60000
target_ratio: 0.15 target_ratio: 0.15
tail_mode: "lean" tail_mode: "lean"
protect_last_n: 20 protect_last_n: 8
protect_first_n: 0
proactive_prune_tokens: 50000 proactive_prune_tokens: 50000
proactive_prune_min_result_chars: 4000 proactive_prune_min_result_chars: 4000
proactive_prune_min_reclaim_tokens: 4096 proactive_prune_min_reclaim_tokens: 4096
context_total_ceiling_seconds: 600 context_total_ceiling_seconds: 600
auxiliary: auxiliary:
# Session names are cosmetic and used to create a second concurrent LLM
# request after every first reply. On a single inference slot this blocks the
# actual chat, so keep the original timestamp/session id instead.
title_generation:
enabled: false
compression: compression:
provider: "main" provider: "main"
model: "" model: ""
reasoning_effort: "none" reasoning_effort: "none"
# The full hermes-cli preset injects several large, overlapping schemas on
# every turn. Athena already exposes browsing, orchestration and host services
# through its MCPs; retain the generally useful local primitives and load MCP
# tools through Hermes' deferred discovery. This setting is shared by every
# model profile.
platform_toolsets:
cli: [web, terminal, file, skills, todo, memory, vision, tts]
# German is the platform default. The display setting localizes the static # German is the platform default. The display setting localizes the static
# messages Hermes currently supports; agent replies are governed by SOUL.md. # messages Hermes currently supports; agent replies are governed by SOUL.md.
display: display:
+11
View File
@@ -24,6 +24,17 @@ create_profile() {
fi fi
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.default "$model" docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.default "$model"
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.context_length "$context" docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.context_length "$context"
# These are platform-wide latency and loop safeguards, not model-specific
# tuning. Enforce them on old profiles as well as newly cloned profiles.
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens 8192
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.max_turns 64
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort low
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold_tokens 60000
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 8
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \
'["web","terminal","file","skills","todo","memory","vision","tts"]'
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.default) == "$model" ]] || \ [[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.default) == "$model" ]] || \
die "Modellalias von Profil $name konnte nicht verifiziert werden." die "Modellalias von Profil $name konnte nicht verifiziert werden."
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.context_length) == "$context" ]] || \ [[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.context_length) == "$context" ]] || \