Simplify Athena runtime and document current architecture

This commit is contained in:
Mikei386 committed 2026-08-30 22:11:26 +02:00
1 parent 9bc7d9803a
commit 0b927d47b9
56 files changed
+1276 -4712

No files matched your search

+43 -165
View File
@@ -477,7 +477,6 @@ services:
labels:
com.mike-ai.llama-profile: experimental
environment:
SEARXNG_URL: http://searxng:8080
NVIDIA_VISIBLE_DEVICES: ${EXPERIMENTAL_GPU_DEVICES:-0}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
command:
@@ -532,7 +531,7 @@ services:
environment:
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
ALLOWED_PROFILES: fast,medium,large,ultra,uncensored,experimental
IMAGE_WORKER: flux
IMAGE_WORKER: image
networks: [control]
security_opt: ["no-new-privileges:true"]
healthcheck:
@@ -572,8 +571,9 @@ services:
# consume the complete context before yielding visible output.
MAX_GENERATION_TOKENS: "8192"
IMAGE_DIR: /data/images
IMAGE_WORKER_URL: http://flux-worker:8086
IMAGE_WORKER_URL: http://image-worker:8086
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
IMAGE_MODEL_NAME: Z-Image-Turbo
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
ENABLE_IMAGE_GENERATION: "true"
ENABLE_TTS: "true"
@@ -610,31 +610,31 @@ services:
tts-gateway:
condition: service_healthy
flux-worker:
image-worker:
build:
context: platform/docker/flux-worker
context: platform/docker/image-worker
args:
DIFFUSERS_VERSION: ${DIFFUSERS_VERSION:-0.40.0}
TRANSFORMERS_VERSION: ${TRANSFORMERS_VERSION:-5.15.1}
ACCELERATE_VERSION: ${ACCELERATE_VERSION:-1.14.0}
HF_HUB_VERSION: ${HF_HUB_VERSION:-1.28.0}
image: mike-ai/flux-worker:local
container_name: mike-ai-flux-worker
image: mike-ai/image-worker:local
container_name: mike-ai-image-worker
restart: "no"
profiles: [image]
labels:
com.mike-ai.image-worker: flux
com.mike-ai.image-worker: image
gpus: all
read_only: true
tmpfs: ["/tmp:size=1g,mode=1777"]
volumes:
- "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro"
- "${Z_IMAGE_MODEL_DIR:-/data/models/Z-Image-Turbo}:/models/Z-Image-Turbo:ro"
- router-images:/data/images
environment:
NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1}
NVIDIA_DRIVER_CAPABILITIES: compute,utility
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
FLUX_MODEL_DIR: /models/FLUX.2-klein-4B
Z_IMAGE_MODEL_DIR: /models/Z-Image-Turbo
IMAGE_DIR: /data/images
networks: [inference]
security_opt: ["no-new-privileges:true"]
@@ -750,168 +750,47 @@ services:
retries: 12
start_period: 10s
open-webui:
build:
context: .
dockerfile: platform/openwebui/Dockerfile
image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-agent-loop-v9}
container_name: mike-ai-open-webui
llama-dashboard:
build: ./platform/llama-dashboard
image: mike-ai/llama-dashboard:local
container_name: mike-ai-llama-dashboard
restart: unless-stopped
labels:
# SQLite is quiesced briefly while the scheduled data backup is created.
docker-volume-backup.stop-during-backup: "true"
volumes:
- open-webui-data:/app/backend/data
# Upstream-supported static customization hooks. Keeping these files in
# the repository makes the global dark theme reproducible and update-safe.
- ./platform/openwebui/theme/custom.css:/app/build/static/custom.css:ro
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
- ./platform/openwebui/theme/midnight-aurora.svg:/app/build/static/midnight-aurora.svg:ro
- ./platform/openwebui/theme/tool-status:/app/build/static/tool-status:ro
environment:
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
DEFAULT_MODELS: mikeai-medium
OLLAMA_BASE_URL: ""
OPENAI_API_BASE_URLS: http://router:8081/v1
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
# Open WebUI uses the router's OpenAI-compatible image endpoint. The
# router performs the exclusive RTX-5080 hot swap and restores the
# previously active Qwen profile after every image.
ENABLE_IMAGE_GENERATION: "true"
IMAGE_GENERATION_ENGINE: openai
# OpenWebUI v0.9.x exposes a fixed OpenAI image-model dropdown. The
# router accepts this compatibility alias and still executes local
# FLUX.2 Klein; no request is sent to OpenAI.
IMAGE_GENERATION_MODEL: gpt-image-1
# The local router can return embedded image data. Force that mode so
# OpenWebUI does not reject the router's private Docker/LAN URL through
# its correct SSRF protection.
IMAGE_URL_RESPONSE_MODELS_REGEX_PATTERN: "^$"
IMAGES_OPENAI_API_BASE_URL: http://router:8081/v1
IMAGES_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
IMAGE_SIZE: 1024x1024
IMAGE_STEPS: "4"
AUDIO_TTS_ENGINE: openai
AUDIO_TTS_OPENAI_API_BASE_URL: http://router:8081/v1
AUDIO_TTS_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
AUDIO_TTS_MODEL: piper
AUDIO_TTS_VOICE: alloy
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
# The derived image reserves the last round for a tool-free synthesis.
# Forty executions permit real multi-domain agent work. Exact-repeat,
# per-tool and total-execution limits in the derived image stop loops.
# Leave continuation headroom after the execution middleware budget:
# one additional model turn is required to synthesize the visible answer.
CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "48"
USER_AGENT: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
DO_NOT_TRACK: "true"
SCARF_NO_ANALYTICS: "true"
dns: ["${AI_DNS:-1.1.1.1}"]
networks: [frontend, tools]
depends_on:
wireguard-gateway:
condition: service_healthy
router:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
hermes:
build:
context: .
dockerfile: platform/hermes/Dockerfile
image: ${HERMES_IMAGE:-mike-ai/hermes-agent:0.20.5-mcpfix1}
container_name: mike-ai-hermes
restart: unless-stopped
command: [/usr/local/bin/start-hermes-managed]
env_file:
- /data/hermes/.env
volumes:
- /data/hermes:/opt/data
- /data/hermes/workspace:/workspace
- ./platform/hermes/start-hermes-managed.sh:/usr/local/bin/start-hermes-managed:ro
- ./platform/hermes/patch-api-mcp-refresh.py:/usr/local/lib/mike-ai/patch-api-mcp-refresh.py:ro
environment:
HERMES_HOME: /opt/data
dns: ["${AI_DNS:-1.1.1.1}"]
networks: [frontend, tools, tools-egress]
depends_on:
wireguard-gateway:
condition: service_healthy
router:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
healthcheck:
test: [CMD, curl, -fsS, "http://127.0.0.1:8642/health"]
interval: 15s
timeout: 5s
retries: 20
start_period: 45s
# Optional, fully removable community chat surface. Chat execution goes
# through the existing Hermes gateway. Upstream's container entrypoint
# requires a writable Hermes home for its ownership/init checks; UI-only
# state still remains on a separate bind mount for easy removal.
hermes-webui:
image: ${HERMES_WEBUI_IMAGE:-mike-ai/hermes-webui:0.52.113-hermes-source-v1}
container_name: mike-ai-hermes-webui
restart: unless-stopped
profiles: [hermes-webui]
env_file:
- /data/hermes-webui/.env
volumes:
- /data/hermes:/home/hermeswebui/.hermes
- /data/hermes-webui/state:/state
- /data/hermes-webui/hermes-agent:/home/hermeswebui/.hermes/hermes-agent:ro
- /data/hermes/workspace:/workspace
environment:
HERMES_HOME: /home/hermeswebui/.hermes
HERMES_WEBUI_STATE_DIR: /state
HERMES_WEBUI_HOST: 0.0.0.0
HERMES_WEBUI_PORT: "8787"
HERMES_WEBUI_CHAT_BACKEND: gateway
HERMES_WEBUI_GATEWAY_BASE_URL: http://hermes:8642
HERMES_API_URL: http://hermes:8642
HERMES_WEBUI_AGENT_DIR: /home/hermeswebui/.hermes/hermes-agent
HERMES_WEBUI_GATEWAY_USE_RUNS_API: "true"
HERMES_SKIP_CHMOD: "1"
WANTED_UID: "10000"
WANTED_GID: "10000"
networks: [frontend]
depends_on:
hermes:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8787/health', timeout=3)"]
interval: 15s
timeout: 5s
retries: 20
start_period: 45s
# Persistent VPN listener for the optional WebUI. Sharing the existing
# WireGuard network namespace avoids recreating the remote-access gateway
# merely to add one listener.
hermes-webui-vpn-proxy:
image: mike-ai/wireguard-gateway:local
container_name: mike-ai-hermes-webui-vpn-proxy
restart: unless-stopped
profiles: [hermes-webui]
network_mode: "service:wireguard-gateway"
entrypoint: [socat]
command:
- TCP-LISTEN:8787,bind=192.168.1.212,reuseaddr,fork
- TCP:hermes-webui:8787
gpus: all
read_only: true
tmpfs:
- /tmp:size=4m,mode=1777
cap_drop: [ALL]
security_opt: ["no-new-privileges:true"]
- /tmp:size=16m,mode=1777
volumes:
- /proc:/host/proc:ro
- /data:/host/data:ro
- /data/models:/host/models:ro
- /data/llama-dashboard:/var/lib/llama-dashboard
environment:
DASHBOARD_HOST: 0.0.0.0
DASHBOARD_PORT: "8099"
ROUTER_URL: http://router:8081
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
HOST_PROC: /host/proc
HOST_DATA: /host/data
HOST_MODELS: /host/models
DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3
DASHBOARD_HISTORY_INTERVAL: "15"
DASHBOARD_DETAIL_RETENTION_DAYS: "21"
NVIDIA_VISIBLE_DEVICES: all
NVIDIA_DRIVER_CAPABILITIES: compute,utility
depends_on:
wireguard-gateway:
condition: service_healthy
hermes-webui:
router:
condition: service_healthy
security_opt: ["no-new-privileges:true"]
cap_drop: [ALL]
healthcheck:
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8099/health', timeout=2)"]
interval: 10s
timeout: 3s
retries: 12
start_period: 10s
backup:
image: ${BACKUP_IMAGE:-offen/docker-volume-backup@sha256:19102d8e59eb1d598cf8c647c2b21100abaadc5a1c808ac643fa612e323c3013}
@@ -954,7 +833,6 @@ networks:
name: mike-ai-tools-egress
volumes:
open-webui-data:
piper-data:
router-state:
router-images: