Add dedicated Hermes compression model

This commit is contained in:
Mikei386
2026-08-26 00:45:32 +02:00
parent e872829b11
commit 43d9903d13
8 changed files with 120 additions and 3 deletions
+8 -2
View File
@@ -100,9 +100,15 @@ auxiliary:
title_generation:
enabled: false
compression:
provider: "main"
model: ""
provider: "openai-api"
model: "qwen-compression"
base_url: "http://192.168.1.212:8099/v1"
api_key: "local"
reasoning_effort: "none"
extra_body:
chat_template_kwargs:
enable_thinking: false
timeout: 600
# The full hermes-cli preset injects several large, overlapping schemas on
# every turn. Athena already exposes browsing, orchestration and host services
+8
View File
@@ -35,6 +35,14 @@ create_profile() {
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.35
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 12
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.provider openai-api
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.model qwen-compression
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.base_url http://192.168.1.212:8099/v1
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.api_key local
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.reasoning_effort none
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.extra_body \
'{"chat_template_kwargs":{"enable_thinking":false}}'
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.timeout 600
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \
'["web","terminal","file","skills","todo","memory","vision","tts"]'
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.default) == "$model" ]] || \