From 400de54ff9afa0db5b0d93d4caa0d7c7eb37401c Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Wed, 26 Aug 2026 06:54:57 +0200 Subject: [PATCH] Bound Hermes context compression latency --- platform/hermes/config.yaml | 10 ++++++---- platform/hermes/install-profiles.sh | 3 ++- 2 files changed, 8 insertions(+), 5 deletions(-) diff --git a/platform/hermes/config.yaml b/platform/hermes/config.yaml index 77b51bd..0dd468c 100644 --- a/platform/hermes/config.yaml +++ b/platform/hermes/config.yaml @@ -78,8 +78,8 @@ agent: max_subagents: 8 # Keep ample room for long agent work. Compression starts at 82% of whichever -# router profile is selected and retains a useful 35% instead of collapsing a -# large conversation to a tiny summary. +# router profile is selected. Keep the result compact enough that a local +# model does not spend many minutes generating the handoff. compression: enabled: true # Fold one older assistant/tool exchange into the rolling summary every five @@ -90,14 +90,16 @@ compression: micro_compact_defrag_threshold_tokens: 2000 progress_notices: true threshold: 0.82 - target_ratio: 0.35 + target_ratio: 0.15 tail_mode: "lean" protect_last_n: 20 protect_first_n: 0 proactive_prune_tokens: 50000 proactive_prune_min_result_chars: 4000 proactive_prune_min_reclaim_tokens: 4096 - context_total_ceiling_seconds: 600 + # A failed local summarizer must not block an interactive client for ten + # minutes. Continue without dropping messages after two minutes. + context_total_ceiling_seconds: 120 auxiliary: # Session names are cosmetic and used to create a second concurrent LLM diff --git a/platform/hermes/install-profiles.sh b/platform/hermes/install-profiles.sh index aadbc32..af883e3 100755 --- a/platform/hermes/install-profiles.sh +++ b/platform/hermes/install-profiles.sh @@ -33,7 +33,8 @@ create_profile() { docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset compression.threshold_tokens || true docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold 0.82 - docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.35 + docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.15 + docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.context_total_ceiling_seconds 120 docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 20 docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0 docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact true