diff --git a/.env.example b/.env.example index 2c61a59..d9ffd75 100644 --- a/.env.example +++ b/.env.example @@ -12,11 +12,6 @@ XTTS_CACHE_DIR=/data/models/xtts-v2-cache XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b AI_DNS=192.168.1.1 -# Experimental context-compression benchmark service on the RTX 3060. -COMPRESSION_MODEL_FILE=qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf -COMPRESSION_CONTEXT=65536 -COMPRESSION_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b - FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf LARGE_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf diff --git a/README.md b/README.md index 73c7c4f..50497e0 100644 --- a/README.md +++ b/README.md @@ -10,13 +10,10 @@ zusammen mit dem übrigen Appdata gesichert. - `compose.yaml` ist der einzige Einstieg für die KI-Dienste auf Athena. - Genau ein llama.cpp-Profil ist aktiv. Der Router schaltet zwischen Fast, Medium, Large, Ultra und Uncensored. -- Ein unabhängiges Qwen3.5-4B-Testmodell kann auf der RTX 3060 über - `http://192.168.1.212:8099/v1` gezielt gestartet werden. Es ist ein - standardmäßig gestoppter, optionaler Benchmark-Endpunkt und - ausdrücklich **nicht** global für Hermes-Kompression aktiviert: technische - End-to-End-Tests zeigten trotz hoher Gesamttreue ausgelassene ältere - `KEY=value`-Zustände. Produktive Kompression übernimmt das jeweils gewählte - 27B-Hauptprofil. +- Hermes verdichtet ältere Assistenten- und Werkzeug-Turns fortlaufend per + Micro-Compaction (alle fünf abgeschlossenen Turns). Das jeweils gewählte + 27B-Hauptprofil erstellt die Zusammenfassung; ein separates, weniger + zuverlässiges Kompressionsmodell wird nicht betrieben. - Der **Athena Operator** bleibt als einziger hostgebundener administrativer MCP direkt auf Athena. MCPHub veröffentlicht seinen vorhandenen WireGuard-HTTP-Endpunkt zentral unter `/mcp/athena-operator`; es gibt keinen diff --git a/compose.yaml b/compose.yaml index 138b166..c782763 100644 --- a/compose.yaml +++ b/compose.yaml @@ -65,96 +65,6 @@ services: retries: 12 start_period: 10s - # Small, non-thinking side model used only for Hermes context compression. - # It shares the WireGuard namespace, so Unraid can reach port 8099 while the - # university LAN receives no published host port. The stable GPU UUID keeps - # it on the RTX 3060 even if PCI enumeration changes again. - llama-compression: - image: ${LLAMA_IMAGE:-mike-ai/llama.cpp:local} - container_name: mike-ai-llama-compression - restart: unless-stopped - profiles: [compression] - deploy: - resources: - reservations: - devices: - - driver: nvidia - device_ids: ["${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b}"] - capabilities: [gpu] - network_mode: "service:wireguard-gateway" - depends_on: - wireguard-gateway: - condition: service_healthy - read_only: true - tmpfs: - - /tmp:size=256m,mode=1777 - volumes: - - "${MODEL_DIR:-/srv/mike-ai/models}:/models:ro" - environment: - NVIDIA_VISIBLE_DEVICES: ${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} - NVIDIA_DRIVER_CAPABILITIES: compute,utility - security_opt: ["no-new-privileges:true"] - cap_drop: [ALL] - command: - - --model - - "/models/${COMPRESSION_MODEL_FILE:-qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf}" - - --alias - - qwen-compression - - --ctx-size - - "${COMPRESSION_CONTEXT:-65536}" - - --flash-attn - - "on" - - --cache-type-k - - q4_0 - - --cache-type-v - - q4_0 - - --cache-prompt - - --parallel - - "1" - - --jinja - - --host - - 0.0.0.0 - - --port - - "8099" - - --metrics - # Official Qwen3.5 non-thinking sampler. The presence penalty is - # essential here: without it one technical YAML benchmark repeated the - # summary structure for more than 40K tokens. - - --temperature - - "1.0" - - --top-p - - "1.0" - - --top-k - - "20" - - --presence-penalty - - "2.0" - # Hermes' own summary ceiling is 10K. Keep a little completion margin, - # but never allow an accidental repetition loop to fill the 65K slot. - - --n-predict - - "12000" - - --n-gpu-layers - - all - - --device - - CUDA0 - - --split-mode - - none - - --no-mmap - - --no-ui - - --batch-size - - "512" - - --ubatch-size - - "256" - - --threads - - "${LLAMA_THREADS:-6}" - - --threads-batch - - "${LLAMA_THREADS_BATCH:-6}" - healthcheck: - test: [CMD, curl, -fsS, "http://127.0.0.1:8099/health"] - interval: 10s - timeout: 5s - retries: 60 - start_period: 30s - llama-fast: <<: *llama-common container_name: mike-ai-llama-fast diff --git a/config/install.env.example b/config/install.env.example index 48325a9..71ae05b 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -72,11 +72,6 @@ EXPERIMENTAL_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd74 VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 -COMPRESSION_MODEL_FILE=qwen3.5-4b-compression/Qwen_Qwen3.5-4B-Q4_K_M.gguf -COMPRESSION_MODEL_URL=https://huggingface.co/bartowski/Qwen_Qwen3.5-4B-GGUF/resolve/main/Qwen_Qwen3.5-4B-Q4_K_M.gguf -COMPRESSION_MODEL_SHA256=13c16f426047e2de38cd075bdade4a7bcbc8c774384876f677740cda65f8a983 -COMPRESSION_CONTEXT=65536 -COMPRESSION_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b # All standard profiles use the MTP tensor embedded in their GGUF. A separate # draft-model artifact is neither downloaded nor passed to llama-server. diff --git a/docs/STANDARD_PROFILE_MATRIX.md b/docs/STANDARD_PROFILE_MATRIX.md index 755c840..6aa7ffc 100644 --- a/docs/STANDARD_PROFILE_MATRIX.md +++ b/docs/STANDARD_PROFILE_MATRIX.md @@ -30,27 +30,14 @@ und einer bewussten Aktualisierung dieser Datei. - Das gesonderte Experimentalprofil gehört nicht zur Benutzer-Matrix und wird in Open WebUI nicht als reguläres Modell angeboten. -## Experimentelles Hermes-Kompressionsmodell +## Hermes-Kontextpflege -Der Alias `qwen-compression` ist kein Chatprofil. Qwen3.5-4B Q4_K_M kann -unabhängig mit 65.536 Token Kontext auf der RTX 3060 gestartet werden und ist -dann nur über WireGuard auf Port 8099 erreichbar. Non-Thinking und ein 12K-Notausgangslimit -verhindern bekannte Wiederholungsschleifen. Das Modell ist absichtlich **nicht -global in Hermes aktiviert**; produktive Kompression nutzt das gewählte -27B-Hauptprofil. - -- 12 technische Fälle: 79,27 % exakte Anker im reinen Modellteil und 98,78 % - im vollständigen Hermes-Lean-Ergebnis; 10 von 12 formal bestanden. -- Echter 205K-Token-End-to-End-Lauf: Kompression auf 22,8K in 199 Sekunden, - aber ältere `BUILD_ID`- und kurze Git-HEAD-Zustände gingen verloren. -- Lange technische Einzeltests benötigten 120 bis 185 Sekunden; Ausgabe bei - großem Kontext etwa 50 Tok/s. -- Speicherbelegung der RTX 3060 im gemessenen Gesamtzustand: 9.450 MiB, - verbleibend rund 2.460 MiB. -- Sicherheitsentscheidung: als reproduzierbarer Testdienst behalten, aber - erst nach zuverlässig vollständiger technischer Zustandsbewahrung als - globalen Kompressor freigeben. Der Container bleibt außerhalb gezielter - Benchmarks gestoppt, damit die RTX 3060 für Medium/Large/Ultra frei ist. +Hermes verwendet das jeweils ausgewählte 27B-Profil auch für die +Kontextzusammenfassung. Micro-Compaction übernimmt alle fünf abgeschlossenen +Turns einen älteren Assistenten-/Werkzeugabschnitt in die laufende +Zusammenfassung. Die normale schwellenbasierte Kompression bleibt als +Rückfallebene aktiv. Ein separates kleines Kompressionsmodell wird nicht +installiert, weil 2B/4B-Tests exakte technische Zustände verlieren konnten. ## Nachweise diff --git a/install.sh b/install.sh index 3c6c920..8f81b43 100755 --- a/install.sh +++ b/install.sh @@ -43,8 +43,7 @@ required=(AI_HOSTNAME ADMIN_USER MODEL_DIR FAST_MODEL_FILE MEDIUM_MODEL_SHA256 LARGE_MODEL_FILE LARGE_MODEL_URL LARGE_MODEL_SHA256 ULTRA_MODEL_FILE ULTRA_MODEL_URL ULTRA_MODEL_SHA256 EXPERIMENTAL_MODEL_FILE EXPERIMENTAL_MODEL_URL EXPERIMENTAL_MODEL_SHA256 - VISION_PROJECTOR_FILE VISION_PROJECTOR_URL VISION_PROJECTOR_SHA256 - COMPRESSION_MODEL_FILE COMPRESSION_MODEL_URL COMPRESSION_MODEL_SHA256) + VISION_PROJECTOR_FILE VISION_PROJECTOR_URL VISION_PROJECTOR_SHA256) for name in "${required[@]}"; do [[ -n "${!name:-}" ]] || die "Pflichtwert $name fehlt." done @@ -340,9 +339,6 @@ UNCENSORED_MODEL_FILE=$UNCENSORED_MODEL_FILE UNCENSORED_PROJECTOR_FILE=$UNCENSORED_PROJECTOR_FILE EXPERIMENTAL_MODEL_FILE=$EXPERIMENTAL_MODEL_FILE VISION_PROJECTOR_FILE=$VISION_PROJECTOR_FILE -COMPRESSION_MODEL_FILE=$COMPRESSION_MODEL_FILE -COMPRESSION_CONTEXT=${COMPRESSION_CONTEXT:-65536} -COMPRESSION_GPU_DEVICE=${COMPRESSION_GPU_DEVICE:-GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b} FAST_CONTEXT=${FAST_CONTEXT:-76800} MEDIUM_CONTEXT=${MEDIUM_CONTEXT:-160000} LARGE_CONTEXT=${LARGE_CONTEXT:-192000} @@ -404,7 +400,6 @@ $UNCENSORED_MODEL_FILE|$UNCENSORED_MODEL_URL|$UNCENSORED_MODEL_SHA256 $UNCENSORED_PROJECTOR_FILE|$UNCENSORED_PROJECTOR_URL|$UNCENSORED_PROJECTOR_SHA256 $EXPERIMENTAL_MODEL_FILE|$EXPERIMENTAL_MODEL_URL|$EXPERIMENTAL_MODEL_SHA256 $VISION_PROJECTOR_FILE|$VISION_PROJECTOR_URL|$VISION_PROJECTOR_SHA256 -$COMPRESSION_MODEL_FILE|$COMPRESSION_MODEL_URL|$COMPRESSION_MODEL_SHA256 EOF } @@ -500,9 +495,6 @@ build_and_start() { docker compose --env-file "$SECRETS_DIR/stack.env" --profile image create flux-worker docker compose --env-file "$SECRETS_DIR/stack.env" up -d --build \ xtts piper tts-gateway profile-controller router open-webui hermes - docker compose --env-file "$SECRETS_DIR/stack.env" --profile compression up -d \ - llama-compression - if [[ ${WIREGUARD_MODE:-container} == container ]]; then systemctl restart mike-ai-container-vpn-guard.service fi diff --git a/platform/hermes/config.yaml b/platform/hermes/config.yaml index 2a25c72..7899930 100644 --- a/platform/hermes/config.yaml +++ b/platform/hermes/config.yaml @@ -82,6 +82,12 @@ agent: # large conversation to a tiny summary. compression: enabled: true + # Fold one older assistant/tool exchange into the rolling summary every five + # completed turns. The selected 27B profile remains the summarizer; there is + # no separate low-fidelity compression model. + micro_compact: true + micro_compact_every_n_turns: 5 + micro_compact_defrag_threshold_tokens: 2000 progress_notices: true threshold: 0.82 target_ratio: 0.35 diff --git a/platform/hermes/install-profiles.sh b/platform/hermes/install-profiles.sh index b688da2..8ec6372 100755 --- a/platform/hermes/install-profiles.sh +++ b/platform/hermes/install-profiles.sh @@ -35,10 +35,11 @@ create_profile() { docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.35 docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 20 docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0 - # The small compressor remains an opt-in benchmark endpoint. It is not safe - # as a global technical-session compressor: end-to-end tests showed that it - # can omit older exact KEY=value state. Remove stale opt-in settings so the - # selected 27B profile performs production compaction. + docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact true + docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact_every_n_turns 5 + docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact_defrag_threshold_tokens 2000 + # The selected 27B profile performs production compaction. A separate small + # compressor was removed after it lost exact technical state in benchmarks. docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset auxiliary.compression || true docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \ '["web","terminal","file","skills","todo","memory","vision","tts"]'