166 lines
5.1 KiB
YAML
166 lines
5.1 KiB
YAML
_config_version: 38
|
|
|
|
model:
|
|
default: "qwen-medium"
|
|
provider: "custom:router"
|
|
base_url: "http://router:8081/v1"
|
|
api_key: "${ROUTER_API_KEY}"
|
|
context_length: 160000
|
|
# Applies to visible text, tool calls and hidden reasoning together. Without
|
|
# a cap a local reasoning model can spend tens of thousands of tokens before
|
|
# producing its first useful action.
|
|
max_tokens: 8192
|
|
api_mode: "chat_completions"
|
|
|
|
# A named provider is required by current Hermes releases. A bare `custom`
|
|
# endpoint is canonicalized to a host-derived identity and can lose its key
|
|
# association in resumed sessions. The stable identity below always resolves
|
|
# its credential from the container environment.
|
|
providers:
|
|
router:
|
|
name: "Athena Profile Router"
|
|
api: "http://router:8081/v1"
|
|
key_env: "ROUTER_API_KEY"
|
|
transport: "chat_completions"
|
|
default_model: "qwen-medium"
|
|
discover_models: true
|
|
|
|
# Commands run in an isolated, persistent workspace. Host and Unraid changes
|
|
# use the audited operator/MUA MCPs instead of a Docker socket or host mount.
|
|
terminal:
|
|
backend: "local"
|
|
cwd: "/workspace"
|
|
timeout: 600
|
|
home_mode: "profile"
|
|
persistent_shell: true
|
|
lifetime_seconds: 1800
|
|
|
|
web:
|
|
search_backend: "searxng"
|
|
extract_backend: "native"
|
|
extract_char_limit: 15000
|
|
keyless_fallback: true
|
|
keyless_rescue: true
|
|
|
|
# Hermes discovers MCP servers in the background. Athena has several remote
|
|
# servers and the Home Assistant relay can finish just after the short upstream
|
|
# default, leaving its tools absent from the first agent turn until a manual
|
|
# reload. Wait long enough to build the initial tool snapshot completely;
|
|
# completed discovery returns immediately, so this does not add a fixed delay.
|
|
mcp_discovery_timeout: 15.0
|
|
mcp_single_query_discovery_timeout: 30.0
|
|
|
|
agent:
|
|
# A single turn may be substantial, but must not silently consume an entire
|
|
# context window in a no-progress loop. Long builds continue from a compact
|
|
# checkpoint in a fresh turn instead of receiving an effectively unlimited
|
|
# 500-step budget.
|
|
max_turns: 64
|
|
# Keep enough deliberation for tool choice while avoiding the provider's
|
|
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
|
# per session with /reasoning.
|
|
reasoning_effort: "minimal"
|
|
gateway_timeout: 3600
|
|
session_stall_timeout: 600
|
|
tool_loop_guardrails:
|
|
warnings_enabled: true
|
|
hard_stop_enabled: true
|
|
warn_after:
|
|
exact_failure: 2
|
|
same_tool_failure: 3
|
|
idempotent_no_progress: 2
|
|
hard_stop_after:
|
|
exact_failure: 5
|
|
same_tool_failure: 8
|
|
idempotent_no_progress: 5
|
|
loop_caps:
|
|
max_web_searches: 8
|
|
max_subagents: 8
|
|
|
|
# Keep ample room for long agent work. Compression starts at 82% of whichever
|
|
# router profile is selected and retains a useful 35% instead of collapsing a
|
|
# large conversation to a tiny summary.
|
|
compression:
|
|
enabled: true
|
|
progress_notices: true
|
|
threshold: 0.82
|
|
target_ratio: 0.35
|
|
tail_mode: "lean"
|
|
protect_last_n: 12
|
|
protect_first_n: 0
|
|
proactive_prune_tokens: 50000
|
|
proactive_prune_min_result_chars: 4000
|
|
proactive_prune_min_reclaim_tokens: 4096
|
|
context_total_ceiling_seconds: 600
|
|
|
|
auxiliary:
|
|
# Session names are cosmetic and used to create a second concurrent LLM
|
|
# request after every first reply. On a single inference slot this blocks the
|
|
# actual chat, so keep the original timestamp/session id instead.
|
|
title_generation:
|
|
enabled: false
|
|
compression:
|
|
provider: "openai-api"
|
|
model: "qwen-compression"
|
|
base_url: "http://192.168.1.212:8099/v1"
|
|
api_key: "local"
|
|
reasoning_effort: "none"
|
|
extra_body:
|
|
chat_template_kwargs:
|
|
enable_thinking: false
|
|
timeout: 600
|
|
|
|
# The full hermes-cli preset injects several large, overlapping schemas on
|
|
# every turn. Athena already exposes browsing, orchestration and host services
|
|
# through its MCPs; retain the generally useful local primitives and load MCP
|
|
# tools through Hermes' deferred discovery. This setting is shared by every
|
|
# model profile.
|
|
platform_toolsets:
|
|
cli: [web, terminal, file, skills, todo, memory, vision, tts]
|
|
|
|
# German is the platform default. The display setting localizes the static
|
|
# messages Hermes currently supports; agent replies are governed by SOUL.md.
|
|
display:
|
|
language: "de"
|
|
|
|
# Speech input stays local and private. A fixed German hint avoids Whisper
|
|
# interpreting short utterances as English while retaining the fast base model.
|
|
stt:
|
|
enabled: true
|
|
provider: "local"
|
|
language: "de"
|
|
prompt: "Hermes, Athena, Qwen, OpenWebUI, Unraid, Home Assistant, Sonarr, Radarr, Navidrome"
|
|
local:
|
|
model: "base"
|
|
language: "de"
|
|
|
|
# Reuse Athena's OpenAI-compatible TTS route. It currently serves XTTS v2 with
|
|
# Annmarie Nele and transparently falls back to Piper when XTTS is unavailable.
|
|
tts:
|
|
provider: "openai"
|
|
speed: 1.0
|
|
openai:
|
|
model: "piper"
|
|
voice: "alloy"
|
|
speed: 1.0
|
|
base_url: "http://router:8081/v1"
|
|
|
|
voice:
|
|
auto_tts: true
|
|
client_direct: false
|
|
|
|
skills:
|
|
creation_nudge_interval: 20
|
|
|
|
plugins:
|
|
enabled: ["web-searxng"]
|
|
|
|
timeouts:
|
|
tools:
|
|
concurrent_batch: 900
|
|
sequential_call: 900
|
|
|
|
# BEGIN MANAGED MCP SERVERS
|
|
mcp_servers: {}
|
|
# END MANAGED MCP SERVERS
|