Files
AI-Profile-Router/platform/hermes/config.yaml
T

156 lines
4.9 KiB
YAML

_config_version: 38
model:
default: "qwen-medium"
provider: "custom:router"
base_url: "http://router:8081/v1"
api_key: "${ROUTER_API_KEY}"
context_length: 160000
# Applies to visible text, tool calls and hidden reasoning together. Without
# a cap a local reasoning model can spend tens of thousands of tokens before
# producing its first useful action.
max_tokens: 8192
api_mode: "chat_completions"
# A named provider is required by current Hermes releases. A bare `custom`
# endpoint is canonicalized to a host-derived identity and can lose its key
# association in resumed sessions. The stable identity below always resolves
# its credential from the container environment.
providers:
router:
name: "Athena Profile Router"
api: "http://router:8081/v1"
key_env: "ROUTER_API_KEY"
transport: "chat_completions"
default_model: "qwen-medium"
discover_models: true
# Commands run in an isolated, persistent workspace. Host and Unraid changes
# use the audited operator/MUA MCPs instead of a Docker socket or host mount.
terminal:
backend: "local"
cwd: "/workspace"
timeout: 600
home_mode: "profile"
persistent_shell: true
lifetime_seconds: 1800
web:
search_backend: "searxng"
extract_backend: "native"
extract_char_limit: 15000
keyless_fallback: true
keyless_rescue: true
# Hermes discovers MCP servers in the background. Athena has several remote
# servers and the Home Assistant relay can finish just after the short upstream
# default, leaving its tools absent from the first agent turn until a manual
# reload. Wait long enough to build the initial tool snapshot completely;
# completed discovery returns immediately, so this does not add a fixed delay.
mcp_discovery_timeout: 15.0
mcp_single_query_discovery_timeout: 30.0
agent:
# A single turn may be substantial, but must not silently consume an entire
# context window in a no-progress loop. Long builds continue from a compact
# checkpoint in a fresh turn instead of receiving an effectively unlimited
# 500-step budget.
max_turns: 64
# Keep enough deliberation for tool choice while avoiding the provider's
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
# per session with /reasoning.
reasoning_effort: "minimal"
gateway_timeout: 3600
session_stall_timeout: 600
tool_loop_guardrails:
warnings_enabled: true
hard_stop_enabled: true
warn_after:
exact_failure: 2
same_tool_failure: 3
idempotent_no_progress: 2
hard_stop_after:
exact_failure: 5
same_tool_failure: 8
idempotent_no_progress: 5
loop_caps:
max_web_searches: 8
max_subagents: 8
# Keep ample room for long agent work. Compression starts at 82% of whichever
# router profile is selected and retains a useful 35% instead of collapsing a
# large conversation to a tiny summary.
compression:
enabled: true
progress_notices: true
threshold: 0.82
target_ratio: 0.35
tail_mode: "lean"
protect_last_n: 20
protect_first_n: 0
proactive_prune_tokens: 50000
proactive_prune_min_result_chars: 4000
proactive_prune_min_reclaim_tokens: 4096
context_total_ceiling_seconds: 600
auxiliary:
# Session names are cosmetic and used to create a second concurrent LLM
# request after every first reply. On a single inference slot this blocks the
# actual chat, so keep the original timestamp/session id instead.
title_generation:
enabled: false
# The full hermes-cli preset injects several large, overlapping schemas on
# every turn. Athena already exposes browsing, orchestration and host services
# through its MCPs; retain the generally useful local primitives and load MCP
# tools through Hermes' deferred discovery. This setting is shared by every
# model profile.
platform_toolsets:
cli: [web, terminal, file, skills, todo, memory, vision, tts]
# German is the platform default. The display setting localizes the static
# messages Hermes currently supports; agent replies are governed by SOUL.md.
display:
language: "de"
# Speech input stays local and private. A fixed German hint avoids Whisper
# interpreting short utterances as English while retaining the fast base model.
stt:
enabled: true
provider: "local"
language: "de"
prompt: "Hermes, Athena, Qwen, OpenWebUI, Unraid, Home Assistant, Sonarr, Radarr, Navidrome"
local:
model: "base"
language: "de"
# Reuse Athena's OpenAI-compatible TTS route. It currently serves XTTS v2 with
# Annmarie Nele and transparently falls back to Piper when XTTS is unavailable.
tts:
provider: "openai"
speed: 1.0
openai:
model: "piper"
voice: "alloy"
speed: 1.0
base_url: "http://router:8081/v1"
voice:
auto_tts: true
client_direct: false
skills:
creation_nudge_interval: 20
plugins:
enabled: ["web-searxng"]
timeouts:
tools:
concurrent_batch: 900
sequential_call: 900
# BEGIN MANAGED MCP SERVERS
mcp_servers: {}
# END MANAGED MCP SERVERS