Files

175 lines
5.7 KiB
YAML

_config_version: 38
model:
default: "qwen-medium"
provider: "custom:router"
base_url: "http://router:8081/v1"
api_key: "${ROUTER_API_KEY}"
context_length: 160000
# Applies to visible text, tool calls and hidden reasoning together. Without
# a cap a local reasoning model can spend tens of thousands of tokens before
# producing its first useful action.
max_tokens: 8192
api_mode: "chat_completions"
# A named provider is required by current Hermes releases. A bare `custom`
# endpoint is canonicalized to a host-derived identity and can lose its key
# association in resumed sessions. The stable identity below always resolves
# its credential from the container environment.
providers:
router:
name: "Athena Profile Router"
api: "http://router:8081/v1"
key_env: "ROUTER_API_KEY"
transport: "chat_completions"
default_model: "qwen-medium"
discover_models: true
# Commands run in an isolated, persistent workspace. Host and Unraid changes
# use the audited operator/MUA MCPs instead of a Docker socket or host mount.
terminal:
backend: "local"
cwd: "/workspace"
timeout: 600
home_mode: "profile"
persistent_shell: true
lifetime_seconds: 1800
web:
# Keenable is bundled with Hermes and provides both search and extraction
# through its keyless tier. This remains available after Hermes moved to
# Unraid, where the former Athena-local SearXNG no longer exists.
search_backend: "keenable"
extract_backend: "keenable"
extract_char_limit: 15000
keyless_fallback: true
keyless_rescue: true
# Hermes discovers MCP servers in the background. Athena has several remote
# servers and the Home Assistant relay can finish just after the short upstream
# default, leaving its tools absent from the first agent turn until a manual
# reload. Wait long enough to build the initial tool snapshot completely;
# completed discovery returns immediately, so this does not add a fixed delay.
mcp_discovery_timeout: 15.0
mcp_single_query_discovery_timeout: 30.0
agent:
# A single turn may be substantial, but must not silently consume an entire
# context window in a no-progress loop. Long builds continue from a compact
# checkpoint in a fresh turn instead of receiving an effectively unlimited
# 500-step budget.
max_turns: 64
# Ordinary turns run without hidden reasoning. Users can enable it explicitly
# per session with /reasoning medium or /reasoning high when a task needs it.
reasoning_effort: false
gateway_timeout: 3600
session_stall_timeout: 600
tool_loop_guardrails:
warnings_enabled: true
hard_stop_enabled: true
warn_after:
exact_failure: 2
same_tool_failure: 3
idempotent_no_progress: 2
hard_stop_after:
exact_failure: 5
same_tool_failure: 8
idempotent_no_progress: 5
loop_caps:
max_web_searches: 8
max_subagents: 8
# Keep ample room for long agent work. Compression starts late; deterministic
# tool-result pruning and rolling micro-compaction keep it from getting there
# during normal jobs.
# router profile is selected. Keep the result compact enough that a local
# model does not spend many minutes generating the handoff.
compression:
enabled: true
# Fold one older assistant/tool exchange into the rolling summary every five
# completed turns. The selected 27B profile remains the summarizer; there is
# no separate low-fidelity compression model.
micro_compact: true
micro_compact_every_n_turns: 5
micro_compact_defrag_threshold_tokens: 2000
progress_notices: true
threshold: 0.95
target_ratio: 0.15
max_attempts: 1
tail_mode: "lean"
protect_last_n: 20
protect_first_n: 0
proactive_prune_tokens: 32000
proactive_prune_min_result_chars: 2000
proactive_prune_min_reclaim_tokens: 2048
context_timeout_seconds: 45
# A failed local summarizer must not block an interactive client for ten
# minutes. Continue without dropping messages after two minutes.
context_total_ceiling_seconds: 120
auxiliary:
compression:
provider: "main"
reasoning_effort: false
timeout: 120
max_concurrency: 1
# Session names are cosmetic and used to create a second concurrent LLM
# request after every first reply. On a single inference slot this blocks the
# actual chat, so keep the original timestamp/session id instead.
title_generation:
enabled: false
# The full hermes-cli preset injects several large, overlapping schemas on
# every turn. Athena already exposes browsing, orchestration and host services
# through its MCPs; retain the generally useful local primitives and load MCP
# tools through Hermes' deferred discovery. This setting is shared by every
# model profile.
platform_toolsets:
cli: [web, terminal, file, skills, todo, memory, vision, tts]
# German is the platform default. The display setting localizes the static
# messages Hermes currently supports; agent replies are governed by SOUL.md.
display:
language: "de"
show_reasoning: false
# Speech input stays local and private. The fixed German hint helps with
# short utterances and product names.
stt:
enabled: true
provider: "local"
language: "de"
prompt: "Hermes, Athena, Qwen, OpenWebUI, Unraid, Home Assistant, Sonarr, Radarr, Navidrome"
local:
model: "base"
language: "de"
# Reuse Athena's OpenAI-compatible Qwen3-TTS route.
tts:
provider: "openai"
speed: 1.0
openai:
model: "qwen3-tts"
voice: "alloy"
speed: 1.0
base_url: "http://router:8081/v1"
voice:
auto_tts: true
client_direct: false
skills:
creation_nudge_interval: 20
plugins:
enabled: ["browser-browser-use", "web-keenable"]
timeouts:
tools:
concurrent_batch: 900
sequential_call: 900
# BEGIN MANAGED MCP SERVERS
mcp_servers: {}
# END MANAGED MCP SERVERS