Add reproducible Docker and WireGuard host bootstrap
This commit is contained in:
@@ -0,0 +1,34 @@
|
||||
ARG CUDA_VERSION=12.8.1
|
||||
ARG UBUNTU_VERSION=24.04
|
||||
|
||||
FROM nvidia/cuda:${CUDA_VERSION}-devel-ubuntu${UBUNTU_VERSION} AS build
|
||||
ARG DEBIAN_FRONTEND=noninteractive
|
||||
ARG LLAMA_CPP_COMMIT
|
||||
ARG CUDA_ARCHITECTURES="86;120"
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates cmake git libcurl4-openssl-dev ninja-build pkg-config && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
RUN test -n "$LLAMA_CPP_COMMIT"
|
||||
RUN git clone --filter=blob:none https://github.com/ggml-org/llama.cpp /src/llama.cpp && \
|
||||
git -C /src/llama.cpp checkout "$LLAMA_CPP_COMMIT"
|
||||
RUN cmake -S /src/llama.cpp -B /src/llama.cpp/build -G Ninja \
|
||||
-DCMAKE_BUILD_TYPE=Release \
|
||||
-DGGML_CUDA=ON \
|
||||
-DGGML_NATIVE=OFF \
|
||||
-DCMAKE_CUDA_ARCHITECTURES="$CUDA_ARCHITECTURES" && \
|
||||
cmake --build /src/llama.cpp/build --target llama-server -j "$(nproc)"
|
||||
|
||||
FROM nvidia/cuda:${CUDA_VERSION}-runtime-ubuntu${UBUNTU_VERSION}
|
||||
ARG DEBIAN_FRONTEND=noninteractive
|
||||
ARG LLAMA_CPP_COMMIT
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
ca-certificates curl libcurl4 libgomp1 python3 && \
|
||||
rm -rf /var/lib/apt/lists/* && \
|
||||
useradd --system --uid 10003 --home /nonexistent --shell /usr/sbin/nologin llama
|
||||
COPY --from=build /src/llama.cpp/build/bin/ /opt/llama/bin/
|
||||
COPY platform/web-search/web_search_mcp.py /opt/mike-ai/mcp/web_search_mcp.py
|
||||
LABEL org.opencontainers.image.source="https://github.com/ggml-org/llama.cpp" \
|
||||
com.mike-ai.llama-cpp-commit="$LLAMA_CPP_COMMIT"
|
||||
ENV LD_LIBRARY_PATH=/opt/llama/bin
|
||||
USER 10003:10003
|
||||
ENTRYPOINT ["/opt/llama/bin/llama-server"]
|
||||
@@ -0,0 +1,12 @@
|
||||
{
|
||||
"mcpServers": {
|
||||
"web": {
|
||||
"command": "/usr/bin/python3",
|
||||
"args": ["/opt/mike-ai/mcp/web_search_mcp.py"],
|
||||
"env": {
|
||||
"SEARXNG_URL": "http://searxng:8080"
|
||||
},
|
||||
"timeout_ms": 120000
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,3 @@
|
||||
FROM python:3.13.7-slim
|
||||
COPY profile_controller.py /app/profile_controller.py
|
||||
ENTRYPOINT ["python", "/app/profile_controller.py"]
|
||||
@@ -0,0 +1,164 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Strict Docker profile switcher for the local AI stack.
|
||||
|
||||
Only status and activation of a fixed set of labelled llama.cpp containers are
|
||||
exposed. Callers cannot provide images, commands, mounts or Docker API paths.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import http.client
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import threading
|
||||
import urllib.parse
|
||||
from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
|
||||
|
||||
HOST = os.environ.get("CONTROLLER_HOST", "0.0.0.0")
|
||||
PORT = int(os.environ.get("CONTROLLER_PORT", "8090"))
|
||||
SOCKET_PATH = os.environ.get("DOCKER_SOCKET", "/var/run/docker.sock")
|
||||
TOKEN_FILE = os.environ.get("CONTROLLER_TOKEN_FILE", "/run/secrets/controller-token")
|
||||
ALLOWED = tuple(x.strip() for x in os.environ.get(
|
||||
"ALLOWED_PROFILES", "fast,medium,long,experimental").split(",") if x.strip())
|
||||
LABEL_KEY = "com.mike-ai.llama-profile"
|
||||
LOCK = threading.Lock()
|
||||
log = logging.getLogger("profile-controller")
|
||||
|
||||
|
||||
class UnixConnection(http.client.HTTPConnection):
|
||||
def connect(self) -> None:
|
||||
self.sock = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
|
||||
self.sock.connect(SOCKET_PATH)
|
||||
|
||||
|
||||
def docker_request(method: str, path: str) -> tuple[int, bytes]:
|
||||
conn = UnixConnection("localhost", timeout=30)
|
||||
try:
|
||||
conn.request(method, path, headers={"Content-Type": "application/json"})
|
||||
response = conn.getresponse()
|
||||
return response.status, response.read()
|
||||
finally:
|
||||
conn.close()
|
||||
|
||||
|
||||
def containers() -> dict[str, dict]:
|
||||
filters = urllib.parse.quote(json.dumps({"label": [LABEL_KEY]}))
|
||||
status, body = docker_request("GET", f"/containers/json?all=1&filters={filters}")
|
||||
if status != 200:
|
||||
raise RuntimeError(f"Docker list failed with HTTP {status}")
|
||||
result: dict[str, dict] = {}
|
||||
for item in json.loads(body):
|
||||
profile = item.get("Labels", {}).get(LABEL_KEY)
|
||||
if profile in ALLOWED:
|
||||
if profile in result:
|
||||
raise RuntimeError(f"duplicate container for profile {profile}")
|
||||
result[profile] = item
|
||||
return result
|
||||
|
||||
|
||||
def active_profile(items: dict[str, dict] | None = None) -> str | None:
|
||||
items = items or containers()
|
||||
active = [name for name, item in items.items() if item.get("State") == "running"]
|
||||
if len(active) > 1:
|
||||
raise RuntimeError(f"multiple llama profiles active: {', '.join(active)}")
|
||||
return active[0] if active else None
|
||||
|
||||
|
||||
def activate(profile: str) -> dict:
|
||||
if profile not in ALLOWED:
|
||||
raise ValueError("profile is not allowlisted")
|
||||
with LOCK:
|
||||
items = containers()
|
||||
missing = [name for name in ALLOWED if name not in items]
|
||||
if missing:
|
||||
raise RuntimeError("profile containers missing: " + ", ".join(missing))
|
||||
current = active_profile(items)
|
||||
if current == profile:
|
||||
return {"active_profile": current, "changed": False}
|
||||
for name, item in items.items():
|
||||
if name == profile or item.get("State") != "running":
|
||||
continue
|
||||
status, _ = docker_request("POST", f"/containers/{item['Id']}/stop?t=120")
|
||||
if status not in (204, 304):
|
||||
raise RuntimeError(f"failed to stop profile {name}: HTTP {status}")
|
||||
target = items[profile]
|
||||
status, _ = docker_request("POST", f"/containers/{target['Id']}/start")
|
||||
if status not in (204, 304):
|
||||
raise RuntimeError(f"failed to start profile {profile}: HTTP {status}")
|
||||
log.info("activated profile %s (previous=%s)", profile, current)
|
||||
return {"active_profile": profile, "changed": True}
|
||||
|
||||
|
||||
def load_token() -> str:
|
||||
token = os.environ.get("CONTROLLER_TOKEN", "").strip()
|
||||
if not token:
|
||||
with open(TOKEN_FILE, encoding="utf-8") as handle:
|
||||
token = handle.read().strip()
|
||||
if len(token) < 32:
|
||||
raise RuntimeError("controller token is missing or too short")
|
||||
return token
|
||||
|
||||
|
||||
TOKEN = load_token()
|
||||
|
||||
|
||||
class Handler(BaseHTTPRequestHandler):
|
||||
server_version = "mike-ai-profile-controller/1"
|
||||
|
||||
def log_message(self, fmt: str, *args: object) -> None:
|
||||
log.info("%s - %s", self.client_address[0], fmt % args)
|
||||
|
||||
def reply(self, status: int, payload: dict) -> None:
|
||||
body = json.dumps(payload, separators=(",", ":")).encode()
|
||||
self.send_response(status)
|
||||
self.send_header("Content-Type", "application/json")
|
||||
self.send_header("Content-Length", str(len(body)))
|
||||
self.end_headers()
|
||||
self.wfile.write(body)
|
||||
|
||||
def authenticated(self) -> bool:
|
||||
return self.headers.get("Authorization", "") == f"Bearer {TOKEN}"
|
||||
|
||||
def do_GET(self) -> None: # noqa: N802
|
||||
if self.path == "/health":
|
||||
self.reply(200, {"status": "ok"})
|
||||
return
|
||||
if self.path != "/status":
|
||||
self.reply(404, {"error": "not found"})
|
||||
return
|
||||
if not self.authenticated():
|
||||
self.reply(401, {"error": "unauthorized"})
|
||||
return
|
||||
try:
|
||||
items = containers()
|
||||
self.reply(200, {"active_profile": active_profile(items),
|
||||
"profiles": {name: items.get(name, {}).get(
|
||||
"State", "missing") for name in ALLOWED}})
|
||||
except Exception as exc:
|
||||
log.exception("status failed")
|
||||
self.reply(503, {"error": str(exc)})
|
||||
|
||||
def do_POST(self) -> None: # noqa: N802
|
||||
if not self.authenticated():
|
||||
self.reply(401, {"error": "unauthorized"})
|
||||
return
|
||||
prefix, suffix = "/profiles/", "/activate"
|
||||
if not self.path.startswith(prefix) or not self.path.endswith(suffix):
|
||||
self.reply(404, {"error": "not found"})
|
||||
return
|
||||
profile = self.path[len(prefix):-len(suffix)]
|
||||
if profile not in ALLOWED:
|
||||
self.reply(400, {"error": "profile is not allowlisted"})
|
||||
return
|
||||
try:
|
||||
self.reply(200, activate(profile))
|
||||
except Exception as exc:
|
||||
log.exception("activation failed for %s", profile)
|
||||
self.reply(503, {"error": str(exc)})
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
logging.basicConfig(level=logging.INFO, format="%(asctime)s %(levelname)s %(message)s")
|
||||
ThreadingHTTPServer((HOST, PORT), Handler).serve_forever()
|
||||
@@ -0,0 +1,9 @@
|
||||
FROM python:3.13.7-slim
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends gosu && \
|
||||
rm -rf /var/lib/apt/lists/* && \
|
||||
useradd --system --uid 10002 --home /nonexistent --shell /usr/sbin/nologin router
|
||||
WORKDIR /app
|
||||
COPY router/ai_profile_router.py router/router_support.py /app/
|
||||
COPY platform/docker/router/entrypoint.sh /usr/local/bin/router-entrypoint
|
||||
RUN chmod 0755 /usr/local/bin/router-entrypoint
|
||||
ENTRYPOINT ["router-entrypoint"]
|
||||
Executable
+4
@@ -0,0 +1,4 @@
|
||||
#!/bin/sh
|
||||
set -eu
|
||||
chown 10002:10002 /var/lib/mike-ai-profile-router /data/images
|
||||
exec gosu 10002:10002 python /app/ai_profile_router.py
|
||||
@@ -2,10 +2,11 @@ schema: 1
|
||||
models:
|
||||
qwen_fast_long:
|
||||
role: primary-text-fast-and-long
|
||||
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||
source: "local migration from the reference host; public origin still to document"
|
||||
file: Qwen3.8-27B-IQ4-MIX.gguf
|
||||
target: /opt/mike-ai/models/qwen3.8-27b-iq4-mix/Qwen3.8-27B-IQ4-MIX.gguf
|
||||
sha256: "REPLACE_AFTER_VERIFICATION"
|
||||
sha256: "54879ae8738d5938f46cb3b8cbf16bf42b8c85b7d68d7c73f062b612ec183e36"
|
||||
size_bytes: 14111614400
|
||||
qwen_medium:
|
||||
role: primary-text-medium
|
||||
source: "REPLACE_WITH_MODEL_REPOSITORY"
|
||||
|
||||
@@ -31,6 +31,7 @@ SERVER_VERSION = "2.1.0"
|
||||
TINYSEARCH_CONTAINER = os.environ.get(
|
||||
"TINYSEARCH_CONTAINER", "mike-ai-web-search-tinysearch-1"
|
||||
)
|
||||
SEARXNG_URL = os.environ.get("SEARXNG_URL", "").rstrip("/")
|
||||
CHILD_TIMEOUT_SECONDS = float(os.environ.get("TINYSEARCH_CHILD_TIMEOUT", "110"))
|
||||
HTTP_TIMEOUT_SECONDS = float(os.environ.get("WEB_API_TIMEOUT", "18"))
|
||||
GITHUB_TOKEN = os.environ.get("GITHUB_TOKEN", "").strip()
|
||||
@@ -349,6 +350,31 @@ def api_json(url: str, service: str) -> Any:
|
||||
return decoded
|
||||
|
||||
|
||||
def searxng_json(query: str) -> Any:
|
||||
"""Query only the administrator-configured internal SearXNG endpoint.
|
||||
|
||||
Public API fetches intentionally reject private addresses. SearXNG is the
|
||||
one explicit internal exception; callers cannot influence its scheme,
|
||||
authority or path, only the encoded search term.
|
||||
"""
|
||||
base = urlparse(SEARXNG_URL)
|
||||
if base.scheme not in {"http", "https"} or not base.hostname:
|
||||
raise RuntimeError("invalid configured SearXNG URL")
|
||||
url = f"{SEARXNG_URL}/search?" + urlencode({
|
||||
"q": query, "format": "json", "language": "auto"})
|
||||
try:
|
||||
with urlopen(Request(url, headers={
|
||||
"Accept": "application/json",
|
||||
"User-Agent": f"mike-ai-web/{SERVER_VERSION}",
|
||||
}), timeout=HTTP_TIMEOUT_SECONDS) as response:
|
||||
payload = response.read(2_000_000)
|
||||
except HTTPError as exc:
|
||||
raise RuntimeError(f"searxng returned HTTP {exc.code}") from exc
|
||||
except (URLError, TimeoutError) as exc:
|
||||
raise RuntimeError("searxng unavailable") from exc
|
||||
return json.loads(payload.decode("utf-8", errors="replace"))
|
||||
|
||||
|
||||
def infer_backend(query: str, requested: str = "auto") -> str:
|
||||
if requested not in {"auto", "web", "github", "huggingface"}:
|
||||
raise ValueError("backend must be auto, web, github or huggingface")
|
||||
@@ -746,7 +772,19 @@ def general_discovery(query: str, limit: int) -> tuple[list[dict[str, Any]], lis
|
||||
results: list[dict[str, Any]] = []
|
||||
warnings: list[str] = []
|
||||
try:
|
||||
results.extend(parse_search_xml(client().call("search", {"query": query}), limit))
|
||||
if SEARXNG_URL:
|
||||
data = searxng_json(query)
|
||||
for row in (data.get("results") or [])[:limit]:
|
||||
results.append({
|
||||
"title": clean_text(str(row.get("title", "")), 240),
|
||||
"url": str(row.get("url", "")),
|
||||
"preview": clean_text(str(row.get("content", "")), 500),
|
||||
"source": "searxng",
|
||||
"published_at": row.get("publishedDate") or row.get("published_date"),
|
||||
})
|
||||
else:
|
||||
results.extend(parse_search_xml(
|
||||
client().call("search", {"query": query}), limit))
|
||||
except Exception as exc:
|
||||
warnings.append(clean_text(str(exc), 240))
|
||||
if len(results) < limit and BRAVE_SEARCH_API_KEY:
|
||||
|
||||
Reference in New Issue
Block a user