Bound Hermes prompt and reasoning latency

This commit is contained in:
Mikei386
2026-08-25 18:01:52 +02:00
parent 37f744eaa2
commit 8dfc879f26
2 changed files with 38 additions and 1 deletions
+27 -1
View File
@@ -6,6 +6,10 @@ model:
base_url: "http://router:8081/v1"
api_key: "${ROUTER_API_KEY}"
context_length: 160000
# Applies to visible text, tool calls and hidden reasoning together. Without
# a cap a local reasoning model can spend tens of thousands of tokens before
# producing its first useful action.
max_tokens: 8192
api_mode: "chat_completions"
# A named provider is required by current Hermes releases. A bare `custom`
@@ -52,6 +56,10 @@ agent:
# checkpoint in a fresh turn instead of receiving an effectively unlimited
# 500-step budget.
max_turns: 64
# Keep enough deliberation for tool choice while avoiding the provider's
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
# per session with /reasoning.
reasoning_effort: "low"
gateway_timeout: 3600
session_stall_timeout: 600
tool_loop_guardrails:
@@ -76,20 +84,38 @@ compression:
enabled: true
progress_notices: true
threshold: 0.65
# A common absolute ceiling keeps all router profiles responsive. It also
# prevents a long Medium/Large/Ultra chat from becoming impossible to move
# back to Fast later.
threshold_tokens: 60000
target_ratio: 0.15
tail_mode: "lean"
protect_last_n: 20
protect_last_n: 8
protect_first_n: 0
proactive_prune_tokens: 50000
proactive_prune_min_result_chars: 4000
proactive_prune_min_reclaim_tokens: 4096
context_total_ceiling_seconds: 600
auxiliary:
# Session names are cosmetic and used to create a second concurrent LLM
# request after every first reply. On a single inference slot this blocks the
# actual chat, so keep the original timestamp/session id instead.
title_generation:
enabled: false
compression:
provider: "main"
model: ""
reasoning_effort: "none"
# The full hermes-cli preset injects several large, overlapping schemas on
# every turn. Athena already exposes browsing, orchestration and host services
# through its MCPs; retain the generally useful local primitives and load MCP
# tools through Hermes' deferred discovery. This setting is shared by every
# model profile.
platform_toolsets:
cli: [web, terminal, file, skills, todo, memory, vision, tts]
# German is the platform default. The display setting localizes the static
# messages Hermes currently supports; agent replies are governed by SOUL.md.
display: