simplify Athena runtime and centralize MCP management
This commit is contained in:
@@ -59,7 +59,7 @@ agent:
|
||||
# Keep enough deliberation for tool choice while avoiding the provider's
|
||||
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
||||
# per session with /reasoning.
|
||||
reasoning_effort: "minimal"
|
||||
reasoning_effort: "low"
|
||||
gateway_timeout: 3600
|
||||
session_stall_timeout: 600
|
||||
tool_loop_guardrails:
|
||||
@@ -77,7 +77,9 @@ agent:
|
||||
max_web_searches: 8
|
||||
max_subagents: 8
|
||||
|
||||
# Keep ample room for long agent work. Compression starts at 82% of whichever
|
||||
# Keep ample room for long agent work. Compression starts late; deterministic
|
||||
# tool-result pruning and rolling micro-compaction keep it from getting there
|
||||
# during normal jobs.
|
||||
# router profile is selected. Keep the result compact enough that a local
|
||||
# model does not spend many minutes generating the handoff.
|
||||
compression:
|
||||
@@ -89,19 +91,26 @@ compression:
|
||||
micro_compact_every_n_turns: 5
|
||||
micro_compact_defrag_threshold_tokens: 2000
|
||||
progress_notices: true
|
||||
threshold: 0.82
|
||||
threshold: 0.95
|
||||
target_ratio: 0.15
|
||||
max_attempts: 1
|
||||
tail_mode: "lean"
|
||||
protect_last_n: 20
|
||||
protect_first_n: 0
|
||||
proactive_prune_tokens: 50000
|
||||
proactive_prune_min_result_chars: 4000
|
||||
proactive_prune_min_reclaim_tokens: 4096
|
||||
proactive_prune_tokens: 32000
|
||||
proactive_prune_min_result_chars: 2000
|
||||
proactive_prune_min_reclaim_tokens: 2048
|
||||
context_timeout_seconds: 45
|
||||
# A failed local summarizer must not block an interactive client for ten
|
||||
# minutes. Continue without dropping messages after two minutes.
|
||||
context_total_ceiling_seconds: 120
|
||||
|
||||
auxiliary:
|
||||
compression:
|
||||
provider: "main"
|
||||
reasoning_effort: "none"
|
||||
timeout: 120
|
||||
max_concurrency: 1
|
||||
# Session names are cosmetic and used to create a second concurrent LLM
|
||||
# request after every first reply. On a single inference slot this blocks the
|
||||
# actual chat, so keep the original timestamp/session id instead.
|
||||
|
||||
Reference in New Issue
Block a user