Bound Hermes prompt and reasoning latency
This commit is contained in:
1 parent
37f744eaa2
commit
8dfc879f26
2 files changed
+38
-1
No files matched your search
@@ -6,6 +6,10 @@ model:
|
||||
base_url: "http://router:8081/v1"
|
||||
api_key: "${ROUTER_API_KEY}"
|
||||
context_length: 160000
|
||||
# Applies to visible text, tool calls and hidden reasoning together. Without
|
||||
# a cap a local reasoning model can spend tens of thousands of tokens before
|
||||
# producing its first useful action.
|
||||
max_tokens: 8192
|
||||
api_mode: "chat_completions"
|
||||
|
||||
# A named provider is required by current Hermes releases. A bare `custom`
|
||||
@@ -52,6 +56,10 @@ agent:
|
||||
# checkpoint in a fresh turn instead of receiving an effectively unlimited
|
||||
# 500-step budget.
|
||||
max_turns: 64
|
||||
# Keep enough deliberation for tool choice while avoiding the provider's
|
||||
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
||||
# per session with /reasoning.
|
||||
reasoning_effort: "low"
|
||||
gateway_timeout: 3600
|
||||
session_stall_timeout: 600
|
||||
tool_loop_guardrails:
|
||||
@@ -76,20 +84,38 @@ compression:
|
||||
enabled: true
|
||||
progress_notices: true
|
||||
threshold: 0.65
|
||||
# A common absolute ceiling keeps all router profiles responsive. It also
|
||||
# prevents a long Medium/Large/Ultra chat from becoming impossible to move
|
||||
# back to Fast later.
|
||||
threshold_tokens: 60000
|
||||
target_ratio: 0.15
|
||||
tail_mode: "lean"
|
||||
protect_last_n: 20
|
||||
protect_last_n: 8
|
||||
protect_first_n: 0
|
||||
proactive_prune_tokens: 50000
|
||||
proactive_prune_min_result_chars: 4000
|
||||
proactive_prune_min_reclaim_tokens: 4096
|
||||
context_total_ceiling_seconds: 600
|
||||
|
||||
auxiliary:
|
||||
# Session names are cosmetic and used to create a second concurrent LLM
|
||||
# request after every first reply. On a single inference slot this blocks the
|
||||
# actual chat, so keep the original timestamp/session id instead.
|
||||
title_generation:
|
||||
enabled: false
|
||||
compression:
|
||||
provider: "main"
|
||||
model: ""
|
||||
reasoning_effort: "none"
|
||||
|
||||
# The full hermes-cli preset injects several large, overlapping schemas on
|
||||
# every turn. Athena already exposes browsing, orchestration and host services
|
||||
# through its MCPs; retain the generally useful local primitives and load MCP
|
||||
# tools through Hermes' deferred discovery. This setting is shared by every
|
||||
# model profile.
|
||||
platform_toolsets:
|
||||
cli: [web, terminal, file, skills, todo, memory, vision, tts]
|
||||
|
||||
# German is the platform default. The display setting localizes the static
|
||||
# messages Hermes currently supports; agent replies are governed by SOUL.md.
|
||||
display:
|
||||
|
||||
@@ -24,6 +24,17 @@ create_profile() {
|
||||
fi
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.default "$model"
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.context_length "$context"
|
||||
# These are platform-wide latency and loop safeguards, not model-specific
|
||||
# tuning. Enforce them on old profiles as well as newly cloned profiles.
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens 8192
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.max_turns 64
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort low
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold_tokens 60000
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 8
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \
|
||||
'["web","terminal","file","skills","todo","memory","vision","tts"]'
|
||||
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.default) == "$model" ]] || \
|
||||
die "Modellalias von Profil $name konnte nicht verifiziert werden."
|
||||
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.context_length) == "$context" ]] || \
|
||||
|
||||
Reference in new issue
Block a user