simplify Athena runtime and centralize MCP management
This commit is contained in:
@@ -59,7 +59,7 @@ agent:
|
||||
# Keep enough deliberation for tool choice while avoiding the provider's
|
||||
# unbounded `auto` reasoning mode on ordinary turns. Users can still raise it
|
||||
# per session with /reasoning.
|
||||
reasoning_effort: "minimal"
|
||||
reasoning_effort: "low"
|
||||
gateway_timeout: 3600
|
||||
session_stall_timeout: 600
|
||||
tool_loop_guardrails:
|
||||
@@ -77,7 +77,9 @@ agent:
|
||||
max_web_searches: 8
|
||||
max_subagents: 8
|
||||
|
||||
# Keep ample room for long agent work. Compression starts at 82% of whichever
|
||||
# Keep ample room for long agent work. Compression starts late; deterministic
|
||||
# tool-result pruning and rolling micro-compaction keep it from getting there
|
||||
# during normal jobs.
|
||||
# router profile is selected. Keep the result compact enough that a local
|
||||
# model does not spend many minutes generating the handoff.
|
||||
compression:
|
||||
@@ -89,19 +91,26 @@ compression:
|
||||
micro_compact_every_n_turns: 5
|
||||
micro_compact_defrag_threshold_tokens: 2000
|
||||
progress_notices: true
|
||||
threshold: 0.82
|
||||
threshold: 0.95
|
||||
target_ratio: 0.15
|
||||
max_attempts: 1
|
||||
tail_mode: "lean"
|
||||
protect_last_n: 20
|
||||
protect_first_n: 0
|
||||
proactive_prune_tokens: 50000
|
||||
proactive_prune_min_result_chars: 4000
|
||||
proactive_prune_min_reclaim_tokens: 4096
|
||||
proactive_prune_tokens: 32000
|
||||
proactive_prune_min_result_chars: 2000
|
||||
proactive_prune_min_reclaim_tokens: 2048
|
||||
context_timeout_seconds: 45
|
||||
# A failed local summarizer must not block an interactive client for ten
|
||||
# minutes. Continue without dropping messages after two minutes.
|
||||
context_total_ceiling_seconds: 120
|
||||
|
||||
auxiliary:
|
||||
compression:
|
||||
provider: "main"
|
||||
reasoning_effort: "none"
|
||||
timeout: 120
|
||||
max_concurrency: 1
|
||||
# Session names are cosmetic and used to create a second concurrent LLM
|
||||
# request after every first reply. On a single inference slot this blocks the
|
||||
# actual chat, so keep the original timestamp/session id instead.
|
||||
|
||||
@@ -1,23 +1,26 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
HERMES_CONTAINER=${HERMES_CONTAINER:-mike-ai-hermes}
|
||||
HERMES_CONTAINER=${HERMES_CONTAINER:-Hermes-Agent}
|
||||
SECRETS_DIR=${SECRETS_DIR:-/etc/mike-ai}
|
||||
HERMES_DATA_DIR=${HERMES_DATA_DIR:-/data/hermes}
|
||||
HERMES_DATA_DIR=${HERMES_DATA_DIR:-/mnt/nvme-storage/appdata/Hermes-Agent}
|
||||
STACK_DIR=${STACK_DIR:-$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)}
|
||||
PROFILE_MATRIX=${PROFILE_MATRIX:-$STACK_DIR/config/profile-matrix.json}
|
||||
|
||||
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
|
||||
docker inspect "$HERMES_CONTAINER" >/dev/null 2>&1 || \
|
||||
die "Hermes-Container fehlt: $HERMES_CONTAINER"
|
||||
[[ -s $PROFILE_MATRIX ]] || die "Profilmatrix fehlt: $PROFILE_MATRIX"
|
||||
|
||||
deadline=$((SECONDS + 180))
|
||||
until [[ $(docker inspect --format '{{if .State.Health}}{{.State.Health.Status}}{{else}}{{.State.Status}}{{end}}' \
|
||||
"$HERMES_CONTAINER" 2>/dev/null || true) == healthy ]]; do
|
||||
"$HERMES_CONTAINER" 2>/dev/null || true) =~ ^(healthy|running)$ ]]; do
|
||||
(( SECONDS < deadline )) || die "Hermes wurde nicht rechtzeitig gesund."
|
||||
sleep 2
|
||||
done
|
||||
|
||||
create_profile() {
|
||||
local name=$1 model=$2 context=$3 description=$4
|
||||
local name=$1 model=$2 context=$3 description=$4 max_tokens=$5
|
||||
if ! docker exec "$HERMES_CONTAINER" hermes profile show "$name" >/dev/null 2>&1; then
|
||||
docker exec "$HERMES_CONTAINER" hermes profile create "$name" \
|
||||
--clone-from default --description "$description"
|
||||
@@ -26,23 +29,32 @@ create_profile() {
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.context_length "$context"
|
||||
# These are platform-wide latency and loop safeguards, not model-specific
|
||||
# tuning. Enforce them on old profiles as well as newly cloned profiles.
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens 8192
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set model.max_tokens "$max_tokens"
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.max_turns 64
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort minimal
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set agent.reasoning_effort low
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set display.show_reasoning false
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.title_generation.enabled false
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset compression.threshold_tokens || true
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold 0.82
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.threshold 0.95
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.target_ratio 0.15
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.max_attempts 1
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.context_timeout_seconds 45
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.context_total_ceiling_seconds 120
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_last_n 20
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.protect_first_n 0
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact true
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact_every_n_turns 5
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.micro_compact_defrag_threshold_tokens 2000
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.proactive_prune_tokens 32000
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.proactive_prune_min_result_chars 2000
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set compression.proactive_prune_min_reclaim_tokens 2048
|
||||
# The selected 27B profile performs production compaction. A separate small
|
||||
# compressor was removed after it lost exact technical state in benchmarks.
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config unset auxiliary.compression || true
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.provider main
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.reasoning_effort none
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.timeout 120
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set auxiliary.compression.max_concurrency 1
|
||||
docker exec "$HERMES_CONTAINER" hermes -p "$name" config set platform_toolsets.cli \
|
||||
'["web","terminal","file","skills","todo","memory","vision","tts"]'
|
||||
[[ $(docker exec "$HERMES_CONTAINER" hermes -p "$name" config get model.default) == "$model" ]] || \
|
||||
@@ -51,51 +63,32 @@ create_profile() {
|
||||
die "Kontext von Profil $name konnte nicht verifiziert werden."
|
||||
}
|
||||
|
||||
create_profile fast qwen-fast 76800 \
|
||||
"Schnelles Qwen3.8-27B-Profil mit 76,8K Kontext fuer kurze Chats und schnelle Aufgaben."
|
||||
create_profile medium qwen-medium 160000 \
|
||||
"Ausgewogenes Qwen3.8-27B-Standardprofil mit 160K Kontext fuer Alltag und agentische Aufgaben."
|
||||
create_profile large qwen-large 192000 \
|
||||
"Grosses Qwen3.8-27B-Profil mit 192K Kontext fuer umfangreiche Dokumente und lange Aufgaben."
|
||||
create_profile ultra qwen-ultra 262144 \
|
||||
"Maximales Qwen3.8-27B-Profil mit 262K Kontext fuer sehr grosse Kontexte; langsamer als die Standardprofile."
|
||||
create_profile uncensored qwen-uncensored 80000 \
|
||||
"Unzensiertes Qwen3.8-27B-Profil mit 80K Kontext fuer spezielle Anfragen."
|
||||
while IFS=$'\t' read -r name model context description max_tokens; do
|
||||
create_profile "$name" "$model" "$context" "$description" "$max_tokens"
|
||||
done < <(jq -r --argjson max "$(jq '.max_output_tokens' "$PROFILE_MATRIX")" \
|
||||
'.profiles[] | [.id,.alias,(.context|tostring),(.description|gsub("[\\t\\n]";" ")),($max|tostring)] | @tsv' \
|
||||
"$PROFILE_MATRIX")
|
||||
|
||||
# Existing profiles may predate managed secret rendering and therefore contain
|
||||
# the literal ${ROUTER_API_KEY}. Repair only that exact placeholder; never log
|
||||
# or commit the secret itself.
|
||||
[[ -s $SECRETS_DIR/router-api-key ]] || die "Router-API-Key fehlt."
|
||||
router_key=$(<"$SECRETS_DIR/router-api-key")
|
||||
ROUTER_API_KEY="$router_key" HERMES_DATA_DIR="$HERMES_DATA_DIR" python3 <<'PY'
|
||||
import os
|
||||
import pathlib
|
||||
|
||||
root = pathlib.Path(os.environ["HERMES_DATA_DIR"]) / "profiles"
|
||||
placeholder = "${ROUTER_API_KEY}"
|
||||
paths = [pathlib.Path(os.environ["HERMES_DATA_DIR"]) / "config.yaml"]
|
||||
paths.extend(sorted(root.glob("*/config.yaml")))
|
||||
for path in paths:
|
||||
if not path.exists():
|
||||
continue
|
||||
text = path.read_text()
|
||||
if placeholder in text:
|
||||
text = text.replace(placeholder, os.environ["ROUTER_API_KEY"], 1)
|
||||
# Avoid collision with Hermes' disabled built-in `homeassistant` toolset.
|
||||
# The collision filters the healthy external MCP out of agent snapshots.
|
||||
text = text.replace(
|
||||
"\n homeassistant:\n url: http://mcp-homeassistant:8000/mcp\n",
|
||||
"\n homeassistant-admin:\n url: http://mcp-homeassistant:8000/mcp\n",
|
||||
)
|
||||
path.write_text(text)
|
||||
PY
|
||||
|
||||
sync_args=(--registry "${STACK_DIR:-/opt/mike-ai/stack}/config/mcp-registry.json")
|
||||
# The Unraid host deliberately has no system Python. Run the small declarative
|
||||
# client renderer in the already version-pinned MCPHub image instead of adding
|
||||
# host dependencies.
|
||||
sync_args=(--registry /stack/config/mcp-registry.json)
|
||||
mcphub_token=${MCPHUB_TOKEN_FILE:-/mnt/nvme-storage/appdata/MCPHub/client-token}
|
||||
token_mount=()
|
||||
if [[ -s $mcphub_token ]]; then
|
||||
token_mount=(-v "$mcphub_token:/run/input/mcphub-token:ro")
|
||||
sync_args+=(--mcphub-token-file /run/input/mcphub-token)
|
||||
fi
|
||||
while IFS= read -r config; do
|
||||
sync_args+=(--hermes "$config")
|
||||
sync_args+=(--hermes "/hermes/${config#"$HERMES_DATA_DIR"/}")
|
||||
done < <(find "$HERMES_DATA_DIR" -name config.yaml -type f -print)
|
||||
python3 "${STACK_DIR:-/opt/mike-ai/stack}/platform/mcp/sync-clients.py" "${sync_args[@]}"
|
||||
docker run --rm --entrypoint python \
|
||||
-v "$STACK_DIR:/stack:ro" \
|
||||
-v "$HERMES_DATA_DIR:/hermes:rw" \
|
||||
"${token_mount[@]}" \
|
||||
casaderoll/mcphub:1.1.0 \
|
||||
/stack/platform/mcp/sync-clients.py "${sync_args[@]}"
|
||||
|
||||
"${STACK_DIR:-/opt/mike-ai/stack}/platform/hermes/install-skills.sh"
|
||||
"$STACK_DIR/platform/hermes/install-skills.sh"
|
||||
docker exec "$HERMES_CONTAINER" hermes profile list
|
||||
printf 'HERMES_PROFILES_OK\n'
|
||||
|
||||
@@ -1,31 +1,39 @@
|
||||
#!/usr/bin/env bash
|
||||
set -Eeuo pipefail
|
||||
|
||||
STACK_DIR=${STACK_DIR:-/opt/mike-ai/stack}
|
||||
HERMES_DATA_DIR=${HERMES_DATA_DIR:-/data/hermes}
|
||||
SKILL_SOURCE=$STACK_DIR/platform/hermes/skills/athena-operator/SKILL.md
|
||||
PROFILES=(fast medium large ultra uncensored)
|
||||
STACK_DIR=${STACK_DIR:-$(cd "$(dirname "${BASH_SOURCE[0]}")/../.." && pwd)}
|
||||
HERMES_DATA_DIR=${HERMES_DATA_DIR:-/mnt/nvme-storage/appdata/Hermes-Agent}
|
||||
PROFILE_MATRIX=${PROFILE_MATRIX:-$STACK_DIR/config/profile-matrix.json}
|
||||
SKILL_ROOT=$STACK_DIR/platform/hermes/skills
|
||||
|
||||
die() { printf 'FEHLER: %s\n' "$*" >&2; exit 1; }
|
||||
[[ $EUID -eq 0 ]] || die "Bitte als root ausführen."
|
||||
[[ -s $SKILL_SOURCE ]] || die "Skill-Quelle fehlt: $SKILL_SOURCE"
|
||||
grep -Fxq -- 'name: athena-operator' "$SKILL_SOURCE" || \
|
||||
die "Skill-Quelle hat kein gültiges Athena-Operator-Frontmatter."
|
||||
[[ -s $PROFILE_MATRIX ]] || die "Profilmatrix fehlt: $PROFILE_MATRIX"
|
||||
|
||||
install_skill() {
|
||||
local root=$1 target=$1/platform/athena-operator/SKILL.md
|
||||
local source=$1 root=$2 name target
|
||||
name=${source%/SKILL.md}
|
||||
name=${name##*/}
|
||||
target=$root/platform/$name/SKILL.md
|
||||
grep -Fxq -- "name: $name" "$source" || \
|
||||
die "Skill-Quelle hat kein gültiges Frontmatter: $source"
|
||||
install -d -o 10000 -g 10000 -m 0750 "${target%/*}"
|
||||
if [[ -s $target ]] && ! cmp -s "$SKILL_SOURCE" "$target"; then
|
||||
if [[ -s $target ]] && ! cmp -s "$source" "$target"; then
|
||||
cp -a "$target" "$target.before-managed-update-$(date +%Y%m%d-%H%M%S)"
|
||||
fi
|
||||
install -o 10000 -g 10000 -m 0640 "$SKILL_SOURCE" "$target"
|
||||
cmp -s "$SKILL_SOURCE" "$target" || die "Skill-Synchronisierung fehlgeschlagen: $target"
|
||||
install -o 10000 -g 10000 -m 0640 "$source" "$target"
|
||||
cmp -s "$source" "$target" || die "Skill-Synchronisierung fehlgeschlagen: $target"
|
||||
}
|
||||
|
||||
install_skill "$HERMES_DATA_DIR/skills"
|
||||
for profile in "${PROFILES[@]}"; do
|
||||
[[ -d $HERMES_DATA_DIR/profiles/$profile ]] || continue
|
||||
install_skill "$HERMES_DATA_DIR/profiles/$profile/skills"
|
||||
mapfile -t profiles < <(jq -r '.profiles[].id' "$PROFILE_MATRIX")
|
||||
|
||||
for source in "$SKILL_ROOT"/*/SKILL.md; do
|
||||
[[ -s $source ]] || continue
|
||||
install_skill "$source" "$HERMES_DATA_DIR/skills"
|
||||
for profile in "${profiles[@]}"; do
|
||||
[[ -d $HERMES_DATA_DIR/profiles/$profile ]] || continue
|
||||
install_skill "$source" "$HERMES_DATA_DIR/profiles/$profile/skills"
|
||||
done
|
||||
done
|
||||
|
||||
printf 'HERMES_ATHENA_OPERATOR_SKILL_OK\n'
|
||||
printf 'HERMES_SKILLS_OK\n'
|
||||
|
||||
@@ -16,10 +16,10 @@ Use these paths directly. Do not search the filesystem for alternatives.
|
||||
- Host: Unraid `192.168.1.2`
|
||||
- Container: `MCPHub`
|
||||
- UI/base URL: `http://192.168.1.2:8787`
|
||||
- Operational source/build tree: `/mnt/nvme-storage/appdata/MCPHub/build`
|
||||
- Operational source/build tree: `/mnt/nvme-storage/appdata/MCPHub/build/repo`
|
||||
- Dockerfile: `platform/mcphub/Dockerfile` below that tree
|
||||
- Server declaration: `platform/mcphub/configure-settings.py`
|
||||
- Client registry: `config/mcp-registry.json`
|
||||
- Single server and client registry: `config/mcp-registry.json`
|
||||
- Registry renderer: `platform/mcphub/configure-settings.py` (normally unchanged)
|
||||
- Unraid template: `config/unraid-templates/my-MCPHub.xml`
|
||||
- Persistent state: `/mnt/nvme-storage/appdata/MCPHub`
|
||||
- Secrets: `/mnt/nvme-storage/appdata/MCPHub/secrets/<server>.env`, mode `0600`
|
||||
@@ -91,8 +91,8 @@ Modify only the necessary fixed production files. Rules:
|
||||
- Put credentials only in the matching secret file with mode `0600`.
|
||||
- Never print, log, commit, summarize, or return a secret.
|
||||
- Use `/usr/local/bin/run-with-env` for secret-backed stdio servers.
|
||||
- Declare servers in `configure-settings.py`; do not manually treat
|
||||
`mcp_settings.json` as the source of truth.
|
||||
- Declare servers only in `config/mcp-registry.json`; do not hard-code a server
|
||||
in `configure-settings.py` and do not edit `mcp_settings.json` manually.
|
||||
- Preserve existing users, bearer keys, prompts, resources, enabled states,
|
||||
and per-tool toggles.
|
||||
- Build a new image tag. Never overwrite the tag currently running.
|
||||
@@ -104,7 +104,7 @@ unfiltered large server to clients.
|
||||
|
||||
### 5. Deploy without collateral changes
|
||||
|
||||
Build from `/mnt/nvme-storage/appdata/MCPHub/build`, then recreate only
|
||||
Build from `/mnt/nvme-storage/appdata/MCPHub/build/repo`, then recreate only
|
||||
`MCPHub` through Unraid DockerMan so it stays a managed Unraid container.
|
||||
Preserve all Appdata and mounts.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user