_config_version: 38 model: default: "qwen-medium" provider: "custom:router" base_url: "http://router:8081/v1" api_key: "${ROUTER_API_KEY}" context_length: 160000 # Applies to visible text, tool calls and hidden reasoning together. Without # a cap a local reasoning model can spend tens of thousands of tokens before # producing its first useful action. max_tokens: 8192 api_mode: "chat_completions" # A named provider is required by current Hermes releases. A bare `custom` # endpoint is canonicalized to a host-derived identity and can lose its key # association in resumed sessions. The stable identity below always resolves # its credential from the container environment. providers: router: name: "Athena Profile Router" api: "http://router:8081/v1" key_env: "ROUTER_API_KEY" transport: "chat_completions" default_model: "qwen-medium" discover_models: true # Commands run in an isolated, persistent workspace. Host and Unraid changes # use the audited operator/MUA MCPs instead of a Docker socket or host mount. terminal: backend: "local" cwd: "/workspace" timeout: 600 home_mode: "profile" persistent_shell: true lifetime_seconds: 1800 web: search_backend: "searxng" extract_backend: "native" extract_char_limit: 15000 keyless_fallback: true keyless_rescue: true # Hermes discovers MCP servers in the background. Athena has several remote # servers and the Home Assistant relay can finish just after the short upstream # default, leaving its tools absent from the first agent turn until a manual # reload. Wait long enough to build the initial tool snapshot completely; # completed discovery returns immediately, so this does not add a fixed delay. mcp_discovery_timeout: 15.0 mcp_single_query_discovery_timeout: 30.0 agent: # A single turn may be substantial, but must not silently consume an entire # context window in a no-progress loop. Long builds continue from a compact # checkpoint in a fresh turn instead of receiving an effectively unlimited # 500-step budget. max_turns: 64 # Keep enough deliberation for tool choice while avoiding the provider's # unbounded `auto` reasoning mode on ordinary turns. Users can still raise it # per session with /reasoning. reasoning_effort: "minimal" gateway_timeout: 3600 session_stall_timeout: 600 tool_loop_guardrails: warnings_enabled: true hard_stop_enabled: true warn_after: exact_failure: 2 same_tool_failure: 3 idempotent_no_progress: 2 hard_stop_after: exact_failure: 5 same_tool_failure: 8 idempotent_no_progress: 5 loop_caps: max_web_searches: 8 max_subagents: 8 # Keep ample room for long agent work. Compression starts at 82% of whichever # router profile is selected and retains a useful 35% instead of collapsing a # large conversation to a tiny summary. compression: enabled: true progress_notices: true threshold: 0.82 target_ratio: 0.35 tail_mode: "lean" protect_last_n: 12 protect_first_n: 0 proactive_prune_tokens: 50000 proactive_prune_min_result_chars: 4000 proactive_prune_min_reclaim_tokens: 4096 context_total_ceiling_seconds: 600 auxiliary: # Session names are cosmetic and used to create a second concurrent LLM # request after every first reply. On a single inference slot this blocks the # actual chat, so keep the original timestamp/session id instead. title_generation: enabled: false compression: provider: "openai-api" model: "qwen-compression" base_url: "http://192.168.1.212:8099/v1" api_key: "local" reasoning_effort: "none" extra_body: chat_template_kwargs: enable_thinking: false timeout: 600 # The full hermes-cli preset injects several large, overlapping schemas on # every turn. Athena already exposes browsing, orchestration and host services # through its MCPs; retain the generally useful local primitives and load MCP # tools through Hermes' deferred discovery. This setting is shared by every # model profile. platform_toolsets: cli: [web, terminal, file, skills, todo, memory, vision, tts] # German is the platform default. The display setting localizes the static # messages Hermes currently supports; agent replies are governed by SOUL.md. display: language: "de" # Speech input stays local and private. A fixed German hint avoids Whisper # interpreting short utterances as English while retaining the fast base model. stt: enabled: true provider: "local" language: "de" prompt: "Hermes, Athena, Qwen, OpenWebUI, Unraid, Home Assistant, Sonarr, Radarr, Navidrome" local: model: "base" language: "de" # Reuse Athena's OpenAI-compatible TTS route. It currently serves XTTS v2 with # Annmarie Nele and transparently falls back to Piper when XTTS is unavailable. tts: provider: "openai" speed: 1.0 openai: model: "piper" voice: "alloy" speed: 1.0 base_url: "http://router:8081/v1" voice: auto_tts: true client_direct: false skills: creation_nudge_interval: 20 plugins: enabled: ["web-searxng"] timeouts: tools: concurrent_batch: 900 sequential_call: 900 # BEGIN MANAGED MCP SERVERS mcp_servers: {} # END MANAGED MCP SERVERS