Bound Hermes context compression latency
This commit is contained in:
1 parent
a702353134
commit
400de54ff9
2 files changed
+8
-5
No files matched your search
@@ -78,8 +78,8 @@ agent:
|
|||||||
max_subagents: 8
|
max_subagents: 8
|
||||||
|
|
||||||
# Keep ample room for long agent work. Compression starts at 82% of whichever
|
# Keep ample room for long agent work. Compression starts at 82% of whichever
|
||||||
# router profile is selected and retains a useful 35% instead of collapsing a
|
# router profile is selected. Keep the result compact enough that a local
|
||||||
# large conversation to a tiny summary.
|
# model does not spend many minutes generating the handoff.
|
||||||
compression:
|
compression:
|
||||||
enabled: true
|
enabled: true
|
||||||
# Fold one older assistant/tool exchange into the rolling summary every five
|
# Fold one older assistant/tool exchange into the rolling summary every five
|
||||||
@@ -90,14 +90,16 @@ compression:
|
|||||||
micro_compact_defrag_threshold_tokens: 2000
|
micro_compact_defrag_threshold_tokens: 2000
|
||||||
progress_notices: true
|
progress_notices: true
|
||||||
threshold: 0.82
|
threshold: 0.82
|
||||||
target_ratio: 0.35
|
target_ratio: 0.15
|
||||||
tail_mode: "lean"
|
tail_mode: "lean"
|
||||||
protect_last_n: 20
|
protect_last_n: 20
|
||||||
protect_first_n: 0
|
protect_first_n: 0
|
||||||
proactive_prune_tokens: 50000
|
proactive_prune_tokens: 50000
|
||||||
proactive_prune_min_result_chars: 4000
|
proactive_prune_min_result_chars: 4000
|
||||||
proactive_prune_min_reclaim_tokens: 4096
|
proactive_prune_min_reclaim_tokens: 4096
|
||||||
context_total_ceiling_seconds: 600
|
# A failed local summarizer must not block an interactive client for ten
|
||||||
|
# minutes. Continue without dropping messages after two minutes.
|
||||||
|
context_total_ceiling_seconds: 120
|
||||||
|
|
||||||
auxiliary:
|
auxiliary:
|
||||||
# Session names are cosmetic and used to create a second concurrent LLM
|
# Session names are cosmetic and used to create a second concurrent LLM
|
||||||
|
|||||||
@@ -33,7 +33,8 @@ create_profile() {
|
|||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset compression.threshold_tokens || true
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset compression.threshold_tokens || true
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold 0.82
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold 0.82
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.35
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.15
|
||||||
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.context_total_ceiling_seconds 120
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 20
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 20
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact true
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact true
|
||||||
|
|||||||
Reference in new issue
Block a user