Replace compression sidecar with Hermes micro-compaction
This commit is contained in:
1 parent
438bd34b72
commit
896c12193e
8 files changed
+23
-140
No files matched your search
@@ -12,11 +12,6 @@ XTTS_CACHE_DIR=/data/models/xtts-v2-cache
|
|||||||
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
||||||
AI_DNS=192.168.1.1
|
AI_DNS=192.168.1.1
|
||||||
|
|
||||||
# Experimental context-compression benchmark service on the RTX 3060.
|
|
||||||
COMPRESSION_MODEL_FILE=qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf
|
|
||||||
COMPRESSION_CONTEXT=65536
|
|
||||||
COMPRESSION_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
|
||||||
|
|
||||||
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||||
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf
|
||||||
|
|||||||
@@ -10,13 +10,10 @@ zusammen mit dem übrigen Appdata gesichert.
|
|||||||
- `compose.yaml` ist der einzige Einstieg für die KI-Dienste auf Athena.
|
- `compose.yaml` ist der einzige Einstieg für die KI-Dienste auf Athena.
|
||||||
- Genau ein llama.cpp-Profil ist aktiv. Der Router schaltet zwischen Fast,
|
- Genau ein llama.cpp-Profil ist aktiv. Der Router schaltet zwischen Fast,
|
||||||
Medium, Large, Ultra und Uncensored.
|
Medium, Large, Ultra und Uncensored.
|
||||||
- Ein unabhängiges Qwen3.5-4B-Testmodell kann auf der RTX 3060 über
|
- Hermes verdichtet ältere Assistenten- und Werkzeug-Turns fortlaufend per
|
||||||
`http://192.168.1.212:8099/v1` gezielt gestartet werden. Es ist ein
|
Micro-Compaction (alle fünf abgeschlossenen Turns). Das jeweils gewählte
|
||||||
standardmäßig gestoppter, optionaler Benchmark-Endpunkt und
|
27B-Hauptprofil erstellt die Zusammenfassung; ein separates, weniger
|
||||||
ausdrücklich **nicht** global für Hermes-Kompression aktiviert: technische
|
zuverlässiges Kompressionsmodell wird nicht betrieben.
|
||||||
End-to-End-Tests zeigten trotz hoher Gesamttreue ausgelassene ältere
|
|
||||||
`KEY=value`-Zustände. Produktive Kompression übernimmt das jeweils gewählte
|
|
||||||
27B-Hauptprofil.
|
|
||||||
- Der **Athena Operator** bleibt als einziger hostgebundener administrativer
|
- Der **Athena Operator** bleibt als einziger hostgebundener administrativer
|
||||||
MCP direkt auf Athena. MCPHub veröffentlicht seinen vorhandenen
|
MCP direkt auf Athena. MCPHub veröffentlicht seinen vorhandenen
|
||||||
WireGuard-HTTP-Endpunkt zentral unter `/mcp/athena-operator`; es gibt keinen
|
WireGuard-HTTP-Endpunkt zentral unter `/mcp/athena-operator`; es gibt keinen
|
||||||
|
|||||||
@@ -65,96 +65,6 @@ services:
|
|||||||
retries: 12
|
retries: 12
|
||||||
start_period: 10s
|
start_period: 10s
|
||||||
|
|
||||||
# Small, non-thinking side model used only for Hermes context compression.
|
|
||||||
# It shares the WireGuard namespace, so Unraid can reach port 8099 while the
|
|
||||||
# university LAN receives no published host port. The stable GPU UUID keeps
|
|
||||||
# it on the RTX 3060 even if PCI enumeration changes again.
|
|
||||||
llama-compression:
|
|
||||||
image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local}
|
|
||||||
container_name: mike-ai-llama-compression
|
|
||||||
restart: unless-stopped
|
|
||||||
profiles: [compression]
|
|
||||||
deploy:
|
|
||||||
resources:
|
|
||||||
reservations:
|
|
||||||
devices:
|
|
||||||
- driver: nvidia
|
|
||||||
device_ids: ["${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}"]
|
|
||||||
capabilities: [gpu]
|
|
||||||
network_mode: "service:wireguard-gateway"
|
|
||||||
depends_on:
|
|
||||||
wireguard-gateway:
|
|
||||||
condition: service_healthy
|
|
||||||
read_only: true
|
|
||||||
tmpfs:
|
|
||||||
- /tmp:size=256m,mode=1777
|
|
||||||
volumes:
|
|
||||||
- "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro"
|
|
||||||
environment:
|
|
||||||
NVIDIA_VISIBLE_DEVICES: ${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
|
||||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
|
||||||
security_opt: ["no-new-privileges:true"]
|
|
||||||
cap_drop: [ALL]
|
|
||||||
command:
|
|
||||||
- --model
|
|
||||||
- "/models/${COMPRESSION_MODEL_FILE:-qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf}"
|
|
||||||
- --alias
|
|
||||||
- qwen-compression
|
|
||||||
- --ctx-size
|
|
||||||
- "${COMPRESSION_CONTEXT:-65536}"
|
|
||||||
- --flash-attn
|
|
||||||
- "on"
|
|
||||||
- --cache-type-k
|
|
||||||
- q4_0
|
|
||||||
- --cache-type-v
|
|
||||||
- q4_0
|
|
||||||
- --cache-prompt
|
|
||||||
- --parallel
|
|
||||||
- "1"
|
|
||||||
- --jinja
|
|
||||||
- --host
|
|
||||||
- 0.0.0.0
|
|
||||||
- --port
|
|
||||||
- "8099"
|
|
||||||
- --metrics
|
|
||||||
# Official Qwen3.5 non-thinking sampler. The presence penalty is
|
|
||||||
# essential here: without it one technical YAML benchmark repeated the
|
|
||||||
# summary structure for more than 40K tokens.
|
|
||||||
- --temperature
|
|
||||||
- "1.0"
|
|
||||||
- --top-p
|
|
||||||
- "1.0"
|
|
||||||
- --top-k
|
|
||||||
- "20"
|
|
||||||
- --presence-penalty
|
|
||||||
- "2.0"
|
|
||||||
# Hermes' own summary ceiling is 10K. Keep a little completion margin,
|
|
||||||
# but never allow an accidental repetition loop to fill the 65K slot.
|
|
||||||
- --n-predict
|
|
||||||
- "12000"
|
|
||||||
- --n-gpu-layers
|
|
||||||
- all
|
|
||||||
- --device
|
|
||||||
- CUDA0
|
|
||||||
- --split-mode
|
|
||||||
- none
|
|
||||||
- --no-mmap
|
|
||||||
- --no-ui
|
|
||||||
- --batch-size
|
|
||||||
- "512"
|
|
||||||
- --ubatch-size
|
|
||||||
- "256"
|
|
||||||
- --threads
|
|
||||||
- "${LLAMA_THREADS:-6}"
|
|
||||||
- --threads-batch
|
|
||||||
- "${LLAMA_THREADS_BATCH:-6}"
|
|
||||||
healthcheck:
|
|
||||||
test: [CMD, curl, -fsS, "http://127.0.0.1:8099/health"]
|
|
||||||
interval: 10s
|
|
||||||
timeout: 5s
|
|
||||||
retries: 60
|
|
||||||
start_period: 30s
|
|
||||||
|
|
||||||
llama-fast:
|
llama-fast:
|
||||||
<<: *llama-common
|
<<: *llama-common
|
||||||
container_name: mike-ai-llama-fast
|
container_name: mike-ai-llama-fast
|
||||||
|
|||||||
@@ -72,11 +72,6 @@ EXPERIMENTAL_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd74
|
|||||||
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf
|
||||||
VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf
|
VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf
|
||||||
VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53
|
||||||
COMPRESSION_MODEL_FILE=qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf
|
|
||||||
COMPRESSION_MODEL_URL=https://huggingface.co/bartowski/Qwen_Qwen3.5-4B-GGUF/resolve/main/Qwen_Qwen3.5-4B-Q4_K_M.gguf
|
|
||||||
COMPRESSION_MODEL_SHA256=13c16f426047e2de38cd075bdade4a7bcbc8c774384876f677740cda65f8a983
|
|
||||||
COMPRESSION_CONTEXT=65536
|
|
||||||
COMPRESSION_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b
|
|
||||||
# All standard profiles use the MTP tensor embedded in their GGUF. A separate
|
# All standard profiles use the MTP tensor embedded in their GGUF. A separate
|
||||||
# draft-model artifact is neither downloaded nor passed to llama-server.
|
# draft-model artifact is neither downloaded nor passed to llama-server.
|
||||||
|
|
||||||
|
|||||||
@@ -30,27 +30,14 @@ und einer bewussten Aktualisierung dieser Datei.
|
|||||||
- Das gesonderte Experimentalprofil gehört nicht zur Benutzer-Matrix und wird
|
- Das gesonderte Experimentalprofil gehört nicht zur Benutzer-Matrix und wird
|
||||||
in Open WebUI nicht als reguläres Modell angeboten.
|
in Open WebUI nicht als reguläres Modell angeboten.
|
||||||
|
|
||||||
## Experimentelles Hermes-Kompressionsmodell
|
## Hermes-Kontextpflege
|
||||||
|
|
||||||
Der Alias `qwen-compression` ist kein Chatprofil. Qwen3.5-4B Q4_K_M kann
|
Hermes verwendet das jeweils ausgewählte 27B-Profil auch für die
|
||||||
unabhängig mit 65.536 Token Kontext auf der RTX 3060 gestartet werden und ist
|
Kontextzusammenfassung. Micro-Compaction übernimmt alle fünf abgeschlossenen
|
||||||
dann nur über WireGuard auf Port 8099 erreichbar. Non-Thinking und ein 12K-Notausgangslimit
|
Turns einen älteren Assistenten-/Werkzeugabschnitt in die laufende
|
||||||
verhindern bekannte Wiederholungsschleifen. Das Modell ist absichtlich **nicht
|
Zusammenfassung. Die normale schwellenbasierte Kompression bleibt als
|
||||||
global in Hermes aktiviert**; produktive Kompression nutzt das gewählte
|
Rückfallebene aktiv. Ein separates kleines Kompressionsmodell wird nicht
|
||||||
27B-Hauptprofil.
|
installiert, weil 2B/4B-Tests exakte technische Zustände verlieren konnten.
|
||||||
|
|
||||||
- 12 technische Fälle: 79,27 % exakte Anker im reinen Modellteil und 98,78 %
|
|
||||||
im vollständigen Hermes-Lean-Ergebnis; 10 von 12 formal bestanden.
|
|
||||||
- Echter 205K-Token-End-to-End-Lauf: Kompression auf 22,8K in 199 Sekunden,
|
|
||||||
aber ältere `BUILD_ID`- und kurze Git-HEAD-Zustände gingen verloren.
|
|
||||||
- Lange technische Einzeltests benötigten 120 bis 185 Sekunden; Ausgabe bei
|
|
||||||
großem Kontext etwa 50 Tok/s.
|
|
||||||
- Speicherbelegung der RTX 3060 im gemessenen Gesamtzustand: 9.450 MiB,
|
|
||||||
verbleibend rund 2.460 MiB.
|
|
||||||
- Sicherheitsentscheidung: als reproduzierbarer Testdienst behalten, aber
|
|
||||||
erst nach zuverlässig vollständiger technischer Zustandsbewahrung als
|
|
||||||
globalen Kompressor freigeben. Der Container bleibt außerhalb gezielter
|
|
||||||
Benchmarks gestoppt, damit die RTX 3060 für Medium/Large/Ultra frei ist.
|
|
||||||
|
|
||||||
## Nachweise
|
## Nachweise
|
||||||
|
|
||||||
|
|||||||
+1
-9
@@ -43,8 +43,7 @@ required=(AI_HOSTNAME ADMIN_USER MODEL_DIR FAST_MODEL_FILE
|
|||||||
MEDIUM_MODEL_SHA256 LARGE_MODEL_FILE LARGE_MODEL_URL LARGE_MODEL_SHA256
|
MEDIUM_MODEL_SHA256 LARGE_MODEL_FILE LARGE_MODEL_URL LARGE_MODEL_SHA256
|
||||||
ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256
|
ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256
|
||||||
EXPERIMENTAL_MODEL_FILE EXPERIMENTAL_MODEL_URL EXPERIMENTAL_MODEL_SHA256
|
EXPERIMENTAL_MODEL_FILE EXPERIMENTAL_MODEL_URL EXPERIMENTAL_MODEL_SHA256
|
||||||
VISION_PROJECTOR_FILE VISION_PROJECTOR_URL VISION_PROJECTOR_SHA256
|
VISION_PROJECTOR_FILE VISION_PROJECTOR_URL VISION_PROJECTOR_SHA256)
|
||||||
COMPRESSION_MODEL_FILE COMPRESSION_MODEL_URL COMPRESSION_MODEL_SHA256)
|
|
||||||
for name in "${required[@]}"; do
|
for name in "${required[@]}"; do
|
||||||
[[ -n "${!name:-}" ]] || die "Pflichtwert $name fehlt."
|
[[ -n "${!name:-}" ]] || die "Pflichtwert $name fehlt."
|
||||||
done
|
done
|
||||||
@@ -340,9 +339,6 @@ UNCENSORED_MODEL_FILE=$UNCENSORED_MODEL_FILE
|
|||||||
UNCENSORED_PROJECTOR_FILE=$UNCENSORED_PROJECTOR_FILE
|
UNCENSORED_PROJECTOR_FILE=$UNCENSORED_PROJECTOR_FILE
|
||||||
EXPERIMENTAL_MODEL_FILE=$EXPERIMENTAL_MODEL_FILE
|
EXPERIMENTAL_MODEL_FILE=$EXPERIMENTAL_MODEL_FILE
|
||||||
VISION_PROJECTOR_FILE=$VISION_PROJECTOR_FILE
|
VISION_PROJECTOR_FILE=$VISION_PROJECTOR_FILE
|
||||||
COMPRESSION_MODEL_FILE=$COMPRESSION_MODEL_FILE
|
|
||||||
COMPRESSION_CONTEXT=${COMPRESSION_CONTEXT:-65536}
|
|
||||||
COMPRESSION_GPU_DEVICE=${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}
|
|
||||||
FAST_CONTEXT=${FAST_CONTEXT:-76800}
|
FAST_CONTEXT=${FAST_CONTEXT:-76800}
|
||||||
MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000}
|
MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000}
|
||||||
LARGE_CONTEXT=${LARGE_CONTEXT:-192000}
|
LARGE_CONTEXT=${LARGE_CONTEXT:-192000}
|
||||||
@@ -404,7 +400,6 @@ $UNCENSORED_MODEL_FILE|$UNCENSORED_MODEL_URL|$UNCENSORED_MODEL_SHA256
|
|||||||
$UNCENSORED_PROJECTOR_FILE|$UNCENSORED_PROJECTOR_URL|$UNCENSORED_PROJECTOR_SHA256
|
$UNCENSORED_PROJECTOR_FILE|$UNCENSORED_PROJECTOR_URL|$UNCENSORED_PROJECTOR_SHA256
|
||||||
$EXPERIMENTAL_MODEL_FILE|$EXPERIMENTAL_MODEL_URL|$EXPERIMENTAL_MODEL_SHA256
|
$EXPERIMENTAL_MODEL_FILE|$EXPERIMENTAL_MODEL_URL|$EXPERIMENTAL_MODEL_SHA256
|
||||||
$VISION_PROJECTOR_FILE|$VISION_PROJECTOR_URL|$VISION_PROJECTOR_SHA256
|
$VISION_PROJECTOR_FILE|$VISION_PROJECTOR_URL|$VISION_PROJECTOR_SHA256
|
||||||
$COMPRESSION_MODEL_FILE|$COMPRESSION_MODEL_URL|$COMPRESSION_MODEL_SHA256
|
|
||||||
EOF
|
EOF
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -500,9 +495,6 @@ build_and_start() {
|
|||||||
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker
|
docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \
|
||||||
xtts piper tts-gateway profile-controller router open-webui hermes
|
xtts piper tts-gateway profile-controller router open-webui hermes
|
||||||
docker compose --env-file "$SECRETS_DIR/stack.env" --profile compression up -d \
|
|
||||||
llama-compression
|
|
||||||
|
|
||||||
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
|
if [[ ${WIREGUARD_MODE:-container} == container ]]; then
|
||||||
systemctl restart mike-ai-container-vpn-guard.service
|
systemctl restart mike-ai-container-vpn-guard.service
|
||||||
fi
|
fi
|
||||||
|
|||||||
@@ -82,6 +82,12 @@ agent:
|
|||||||
# large conversation to a tiny summary.
|
# large conversation to a tiny summary.
|
||||||
compression:
|
compression:
|
||||||
enabled: true
|
enabled: true
|
||||||
|
# Fold one older assistant/tool exchange into the rolling summary every five
|
||||||
|
# completed turns. The selected 27B profile remains the summarizer; there is
|
||||||
|
# no separate low-fidelity compression model.
|
||||||
|
micro_compact: true
|
||||||
|
micro_compact_every_n_turns: 5
|
||||||
|
micro_compact_defrag_threshold_tokens: 2000
|
||||||
progress_notices: true
|
progress_notices: true
|
||||||
threshold: 0.82
|
threshold: 0.82
|
||||||
target_ratio: 0.35
|
target_ratio: 0.35
|
||||||
|
|||||||
@@ -35,10 +35,11 @@ create_profile() {
|
|||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.35
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.35
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 20
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 20
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
|
||||||
# The small compressor remains an opt-in benchmark endpoint. It is not safe
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact true
|
||||||
# as a global technical-session compressor: end-to-end tests showed that it
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact_every_n_turns 5
|
||||||
# can omit older exact KEY=value state. Remove stale opt-in settings so the
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact_defrag_threshold_tokens 2000
|
||||||
# selected 27B profile performs production compaction.
|
# The selected 27B profile performs production compaction. A separate small
|
||||||
|
# compressor was removed after it lost exact technical state in benchmarks.
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset auxiliary.compression || true
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset auxiliary.compression || true
|
||||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \
|
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \
|
||||||
'["web","terminal","file","skills","todo","memory","vision","tts"]'
|
'["web","terminal","file","skills","todo","memory","vision","tts"]'
|
||||||
|
|||||||
Reference in new issue
Block a user