Simplify Athena runtime and document current architecture
This commit is contained in:
+43
-165
@@ -477,7 +477,6 @@ services:
|
||||
labels:
|
||||
com.mike-ai.llama-profile: experimental
|
||||
environment:
|
||||
SEARXNG_URL: http://searxng:8080
|
||||
NVIDIA_VISIBLE_DEVICES: ${EXPERIMENTAL_GPU_DEVICES:-0}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
command:
|
||||
@@ -532,7 +531,7 @@ services:
|
||||
environment:
|
||||
CONTROLLER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||
ALLOWED_PROFILES: fast,medium,large,ultra,uncensored,experimental
|
||||
IMAGE_WORKER: flux
|
||||
IMAGE_WORKER: image
|
||||
networks: [control]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
healthcheck:
|
||||
@@ -572,8 +571,9 @@ services:
|
||||
# consume the complete context before yielding visible output.
|
||||
MAX_GENERATION_TOKENS: "8192"
|
||||
IMAGE_DIR: /data/images
|
||||
IMAGE_WORKER_URL: http://flux-worker:8086
|
||||
IMAGE_WORKER_URL: http://image-worker:8086
|
||||
IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||
IMAGE_MODEL_NAME: Z-Image-Turbo
|
||||
CHAT_IMAGE_ALLOW_REMOTE_URLS: "false"
|
||||
ENABLE_IMAGE_GENERATION: "true"
|
||||
ENABLE_TTS: "true"
|
||||
@@ -610,31 +610,31 @@ services:
|
||||
tts-gateway:
|
||||
condition: service_healthy
|
||||
|
||||
flux-worker:
|
||||
image-worker:
|
||||
build:
|
||||
context: platform/docker/flux-worker
|
||||
context: platform/docker/image-worker
|
||||
args:
|
||||
DIFFUSERS_VERSION: ${DIFFUSERS_VERSION:-0.40.0}
|
||||
TRANSFORMERS_VERSION: ${TRANSFORMERS_VERSION:-5.15.1}
|
||||
ACCELERATE_VERSION: ${ACCELERATE_VERSION:-1.14.0}
|
||||
HF_HUB_VERSION: ${HF_HUB_VERSION:-1.28.0}
|
||||
image: mike-ai/flux-worker:local
|
||||
container_name: mike-ai-flux-worker
|
||||
image: mike-ai/image-worker:local
|
||||
container_name: mike-ai-image-worker
|
||||
restart: "no"
|
||||
profiles: [image]
|
||||
labels:
|
||||
com.mike-ai.image-worker: flux
|
||||
com.mike-ai.image-worker: image
|
||||
gpus: all
|
||||
read_only: true
|
||||
tmpfs: ["/tmp:size=1g,mode=1777"]
|
||||
volumes:
|
||||
- "${FLUX_MODEL_DIR:-/data/models/FLUX.2-klein-4B}:/models/FLUX.2-klein-4B:ro"
|
||||
- "${Z_IMAGE_MODEL_DIR:-/data/models/Z-Image-Turbo}:/models/Z-Image-Turbo:ro"
|
||||
- router-images:/data/images
|
||||
environment:
|
||||
NVIDIA_VISIBLE_DEVICES: ${IMAGE_GPU_DEVICES:-1}
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}"
|
||||
FLUX_MODEL_DIR: /models/FLUX.2-klein-4B
|
||||
Z_IMAGE_MODEL_DIR: /models/Z-Image-Turbo
|
||||
IMAGE_DIR: /data/images
|
||||
networks: [inference]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
@@ -750,168 +750,47 @@ services:
|
||||
retries: 12
|
||||
start_period: 10s
|
||||
|
||||
open-webui:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: platform/openwebui/Dockerfile
|
||||
image: ${OPENWEBUI_IMAGE:-mike-ai/openwebui:main-01f4282-agent-loop-v9}
|
||||
container_name: mike-ai-open-webui
|
||||
llama-dashboard:
|
||||
build: ./platform/llama-dashboard
|
||||
image: mike-ai/llama-dashboard:local
|
||||
container_name: mike-ai-llama-dashboard
|
||||
restart: unless-stopped
|
||||
labels:
|
||||
# SQLite is quiesced briefly while the scheduled data backup is created.
|
||||
docker-volume-backup.stop-during-backup: "true"
|
||||
volumes:
|
||||
- open-webui-data:/app/backend/data
|
||||
# Upstream-supported static customization hooks. Keeping these files in
|
||||
# the repository makes the global dark theme reproducible and update-safe.
|
||||
- ./platform/openwebui/theme/custom.css:/app/build/static/custom.css:ro
|
||||
- ./platform/openwebui/theme/loader.js:/app/build/static/loader.js:ro
|
||||
- ./platform/openwebui/theme/midnight-aurora.svg:/app/build/static/midnight-aurora.svg:ro
|
||||
- ./platform/openwebui/theme/tool-status:/app/build/static/tool-status:ro
|
||||
environment:
|
||||
WEBUI_SECRET_KEY: "${WEBUI_SECRET_KEY:?WEBUI_SECRET_KEY is required}"
|
||||
DEFAULT_MODELS: mikeai-medium
|
||||
OLLAMA_BASE_URL: ""
|
||||
OPENAI_API_BASE_URLS: http://router:8081/v1
|
||||
OPENAI_API_KEYS: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||
# Open WebUI uses the router's OpenAI-compatible image endpoint. The
|
||||
# router performs the exclusive RTX-5080 hot swap and restores the
|
||||
# previously active Qwen profile after every image.
|
||||
ENABLE_IMAGE_GENERATION: "true"
|
||||
IMAGE_GENERATION_ENGINE: openai
|
||||
# OpenWebUI v0.9.x exposes a fixed OpenAI image-model dropdown. The
|
||||
# router accepts this compatibility alias and still executes local
|
||||
# FLUX.2 Klein; no request is sent to OpenAI.
|
||||
IMAGE_GENERATION_MODEL: gpt-image-1
|
||||
# The local router can return embedded image data. Force that mode so
|
||||
# OpenWebUI does not reject the router's private Docker/LAN URL through
|
||||
# its correct SSRF protection.
|
||||
IMAGE_URL_RESPONSE_MODELS_REGEX_PATTERN: "^$"
|
||||
IMAGES_OPENAI_API_BASE_URL: http://router:8081/v1
|
||||
IMAGES_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||
IMAGE_SIZE: 1024x1024
|
||||
IMAGE_STEPS: "4"
|
||||
AUDIO_TTS_ENGINE: openai
|
||||
AUDIO_TTS_OPENAI_API_BASE_URL: http://router:8081/v1
|
||||
AUDIO_TTS_OPENAI_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||
AUDIO_TTS_MODEL: piper
|
||||
AUDIO_TTS_VOICE: alloy
|
||||
ENABLE_SIGNUP: ${OPENWEBUI_ENABLE_SIGNUP:-false}
|
||||
ENABLE_FOLLOW_UP_GENERATION: ${OPENWEBUI_ENABLE_FOLLOW_UP_GENERATION:-false}
|
||||
# The derived image reserves the last round for a tool-free synthesis.
|
||||
# Forty executions permit real multi-domain agent work. Exact-repeat,
|
||||
# per-tool and total-execution limits in the derived image stop loops.
|
||||
# Leave continuation headroom after the execution middleware budget:
|
||||
# one additional model turn is required to synthesize the visible answer.
|
||||
CHAT_RESPONSE_MAX_TOOL_CALL_ITERATIONS: "48"
|
||||
USER_AGENT: "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36"
|
||||
DO_NOT_TRACK: "true"
|
||||
SCARF_NO_ANALYTICS: "true"
|
||||
dns: ["${AI_DNS:-1.1.1.1}"]
|
||||
networks: [frontend, tools]
|
||||
depends_on:
|
||||
wireguard-gateway:
|
||||
condition: service_healthy
|
||||
router:
|
||||
condition: service_healthy
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
|
||||
hermes:
|
||||
build:
|
||||
context: .
|
||||
dockerfile: platform/hermes/Dockerfile
|
||||
image: ${HERMES_IMAGE:-mike-ai/hermes-agent:0.20.5-mcpfix1}
|
||||
container_name: mike-ai-hermes
|
||||
restart: unless-stopped
|
||||
command: [/usr/local/bin/start-hermes-managed]
|
||||
env_file:
|
||||
- /data/hermes/.env
|
||||
volumes:
|
||||
- /data/hermes:/opt/data
|
||||
- /data/hermes/workspace:/workspace
|
||||
- ./platform/hermes/start-hermes-managed.sh:/usr/local/bin/start-hermes-managed:ro
|
||||
- ./platform/hermes/patch-api-mcp-refresh.py:/usr/local/lib/mike-ai/patch-api-mcp-refresh.py:ro
|
||||
environment:
|
||||
HERMES_HOME: /opt/data
|
||||
dns: ["${AI_DNS:-1.1.1.1}"]
|
||||
networks: [frontend, tools, tools-egress]
|
||||
depends_on:
|
||||
wireguard-gateway:
|
||||
condition: service_healthy
|
||||
router:
|
||||
condition: service_healthy
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
healthcheck:
|
||||
test: [CMD, curl, -fsS, "http://127.0.0.1:8642/health"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
retries: 20
|
||||
start_period: 45s
|
||||
|
||||
# Optional, fully removable community chat surface. Chat execution goes
|
||||
# through the existing Hermes gateway. Upstream's container entrypoint
|
||||
# requires a writable Hermes home for its ownership/init checks; UI-only
|
||||
# state still remains on a separate bind mount for easy removal.
|
||||
hermes-webui:
|
||||
image: ${HERMES_WEBUI_IMAGE:-mike-ai/hermes-webui:0.52.113-hermes-source-v1}
|
||||
container_name: mike-ai-hermes-webui
|
||||
restart: unless-stopped
|
||||
profiles: [hermes-webui]
|
||||
env_file:
|
||||
- /data/hermes-webui/.env
|
||||
volumes:
|
||||
- /data/hermes:/home/hermeswebui/.hermes
|
||||
- /data/hermes-webui/state:/state
|
||||
- /data/hermes-webui/hermes-agent:/home/hermeswebui/.hermes/hermes-agent:ro
|
||||
- /data/hermes/workspace:/workspace
|
||||
environment:
|
||||
HERMES_HOME: /home/hermeswebui/.hermes
|
||||
HERMES_WEBUI_STATE_DIR: /state
|
||||
HERMES_WEBUI_HOST: 0.0.0.0
|
||||
HERMES_WEBUI_PORT: "8787"
|
||||
HERMES_WEBUI_CHAT_BACKEND: gateway
|
||||
HERMES_WEBUI_GATEWAY_BASE_URL: http://hermes:8642
|
||||
HERMES_API_URL: http://hermes:8642
|
||||
HERMES_WEBUI_AGENT_DIR: /home/hermeswebui/.hermes/hermes-agent
|
||||
HERMES_WEBUI_GATEWAY_USE_RUNS_API: "true"
|
||||
HERMES_SKIP_CHMOD: "1"
|
||||
WANTED_UID: "10000"
|
||||
WANTED_GID: "10000"
|
||||
networks: [frontend]
|
||||
depends_on:
|
||||
hermes:
|
||||
condition: service_healthy
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
healthcheck:
|
||||
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8787/health', timeout=3)"]
|
||||
interval: 15s
|
||||
timeout: 5s
|
||||
retries: 20
|
||||
start_period: 45s
|
||||
|
||||
# Persistent VPN listener for the optional WebUI. Sharing the existing
|
||||
# WireGuard network namespace avoids recreating the remote-access gateway
|
||||
# merely to add one listener.
|
||||
hermes-webui-vpn-proxy:
|
||||
image: mike-ai/wireguard-gateway:local
|
||||
container_name: mike-ai-hermes-webui-vpn-proxy
|
||||
restart: unless-stopped
|
||||
profiles: [hermes-webui]
|
||||
network_mode: "service:wireguard-gateway"
|
||||
entrypoint: [socat]
|
||||
command:
|
||||
- TCP-LISTEN:8787,bind=192.168.1.212,reuseaddr,fork
|
||||
- TCP:hermes-webui:8787
|
||||
gpus: all
|
||||
read_only: true
|
||||
tmpfs:
|
||||
- /tmp:size=4m,mode=1777
|
||||
cap_drop: [ALL]
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
- /tmp:size=16m,mode=1777
|
||||
volumes:
|
||||
- /proc:/host/proc:ro
|
||||
- /data:/host/data:ro
|
||||
- /data/models:/host/models:ro
|
||||
- /data/llama-dashboard:/var/lib/llama-dashboard
|
||||
environment:
|
||||
DASHBOARD_HOST: 0.0.0.0
|
||||
DASHBOARD_PORT: "8099"
|
||||
ROUTER_URL: http://router:8081
|
||||
ROUTER_API_KEY: "${ROUTER_API_KEY:?ROUTER_API_KEY is required}"
|
||||
HOST_PROC: /host/proc
|
||||
HOST_DATA: /host/data
|
||||
HOST_MODELS: /host/models
|
||||
DASHBOARD_HISTORY_DB: /var/lib/llama-dashboard/history.sqlite3
|
||||
DASHBOARD_HISTORY_INTERVAL: "15"
|
||||
DASHBOARD_DETAIL_RETENTION_DAYS: "21"
|
||||
NVIDIA_VISIBLE_DEVICES: all
|
||||
NVIDIA_DRIVER_CAPABILITIES: compute,utility
|
||||
depends_on:
|
||||
wireguard-gateway:
|
||||
condition: service_healthy
|
||||
hermes-webui:
|
||||
router:
|
||||
condition: service_healthy
|
||||
security_opt: ["no-new-privileges:true"]
|
||||
cap_drop: [ALL]
|
||||
healthcheck:
|
||||
test: [CMD, python, -c, "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8099/health', timeout=2)"]
|
||||
interval: 10s
|
||||
timeout: 3s
|
||||
retries: 12
|
||||
start_period: 10s
|
||||
|
||||
backup:
|
||||
image: ${BACKUP_IMAGE:-offen/docker-volume-backup@sha256:19102d8e59eb1d598cf8c647c2b21100abaadc5a1c808ac643fa612e323c3013}
|
||||
@@ -954,7 +833,6 @@ networks:
|
||||
name: mike-ai-tools-egress
|
||||
|
||||
volumes:
|
||||
open-webui-data:
|
||||
piper-data:
|
||||
router-state:
|
||||
router-images:
|
||||
|
||||
Reference in New Issue
Block a user