Add reproducible Docker and WireGuard host bootstrap

This commit is contained in:
Mikei386
2026-08-20 21:23:16 +02:00
parent cb07779f5a
commit e83c0e2c70
25 changed files with 1584 additions and 835 deletions
+77 -20
View File
@@ -92,6 +92,19 @@ UPSTREAM_URL = os.environ.get("UPSTREAM_URL", "http://127.0.0.1:8080").rstrip("/
PROFILE_SCRIPT = os.environ.get("PROFILE_SCRIPT", "/usr/local/bin/llama-profile")
PROFILE_DIR = os.environ.get(
"PROFILE_DIR", "/etc/systemd/system/mike-ai-llama-ui.service.d")
PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/")
PROFILE_CONTROL_TOKEN_FILE = os.environ.get(
"PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token")
# Optional worker APIs. The clean Docker baseline deliberately ships only
# text/multimodal chat; absent workers must fail explicitly instead of trying
# legacy systemd paths inside the container.
ENABLE_IMAGE_GENERATION = os.environ.get(
"ENABLE_IMAGE_GENERATION", "true").lower() in {"1", "true", "yes"}
ENABLE_TTS = os.environ.get(
"ENABLE_TTS", "true").lower() in {"1", "true", "yes"}
ENABLE_STT = os.environ.get(
"ENABLE_STT", "true").lower() in {"1", "true", "yes"}
SWITCH_TIMEOUT = float(os.environ.get("SWITCH_TIMEOUT", "600")) # s, Warten auf llama.cpp
REQUEST_TIMEOUT = float(os.environ.get("REQUEST_TIMEOUT", "600")) # s, Read-Timeout Upstream
@@ -420,12 +433,40 @@ def _read(path: str) -> str:
return f.read().strip()
def _profile_controller_request(method: str, path: str) -> dict:
token = os.environ.get("PROFILE_CONTROL_TOKEN", "").strip()
if not token:
token = _read(PROFILE_CONTROL_TOKEN_FILE)
if len(token) < 32:
raise RuntimeError("Profil-Controller-Token fehlt oder ist zu kurz")
request = urllib.request.Request(
PROFILE_CONTROL_URL + path,
method=method,
headers={"Authorization": f"Bearer {token}"},
)
try:
with urllib.request.urlopen(request, timeout=120) as response:
return json.load(response)
except urllib.error.HTTPError as exc:
body = exc.read(500).decode(errors="replace")
raise RuntimeError(
f"Profil-Controller HTTP {exc.code}: {body}") from exc
def current_profile() -> str | None:
"""Aktives Profil anhand semantischer Werte der override.conf.
Kommentare, Leerraum oder die Reihenfolge anderer llama.cpp-Optionen
beeinflussen die Erkennung nicht mehr.
"""
if PROFILE_CONTROL_URL:
try:
profile = _profile_controller_request("GET", "/status").get(
"active_profile")
return profile if profile in PROFILES else None
except Exception as exc:
log.warning("Profil-Controller-Status nicht verfügbar: %s", exc)
return None
try:
override = _read(os.path.join(PROFILE_DIR, "override.conf"))
except OSError:
@@ -512,23 +553,27 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
f"llama.cpp nicht erreichbar (Profil {profile} ist "
f"bereits aktiv; Neustart über /{profile})")
log.info("Profilwechsel: %s -> %s", cur, profile)
try:
proc = subprocess.run(
[PROFILE_SCRIPT, profile],
stdin=subprocess.DEVNULL,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=120,
)
out = proc.stdout.decode(errors="replace").strip()
if out:
log.info("llama-profile: %s", out[-500:])
if proc.returncode != 0:
raise RuntimeError(
f"llama-profile fehlgeschlagen (Exit-Code "
f"{proc.returncode}): {out[-500:]}")
except subprocess.TimeoutExpired:
raise RuntimeError("llama-profile hat 120 s überschritten")
if PROFILE_CONTROL_URL:
_profile_controller_request(
"POST", f"/profiles/{profile}/activate")
else:
try:
proc = subprocess.run(
[PROFILE_SCRIPT, profile],
stdin=subprocess.DEVNULL,
stdout=subprocess.PIPE,
stderr=subprocess.STDOUT,
timeout=120,
)
out = proc.stdout.decode(errors="replace").strip()
if out:
log.info("llama-profile: %s", out[-500:])
if proc.returncode != 0:
raise RuntimeError(
f"llama-profile fehlgeschlagen (Exit-Code "
f"{proc.returncode}): {out[-500:]}")
except subprocess.TimeoutExpired:
raise RuntimeError("llama-profile hat 120 s überschritten")
if current_profile() != profile:
raise RuntimeError(
f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
@@ -1041,11 +1086,23 @@ class Handler(BaseHTTPRequestHandler):
elif path == "/v1/audio/voices" and self.command == "GET":
self._send_json(200, self._audio_voices_payload())
elif path == "/v1/images/generations" and self.command == "POST":
self._image_generate()
if ENABLE_IMAGE_GENERATION:
self._image_generate()
else:
self._send_error(503, "Bildgenerierung ist nicht installiert",
"server_error", "feature_disabled")
elif path == "/v1/audio/speech" and self.command == "POST":
self._speech()
if ENABLE_TTS:
self._speech()
else:
self._send_error(503, "Sprachausgabe ist nicht installiert",
"server_error", "feature_disabled")
elif path == "/v1/audio/transcriptions" and self.command == "POST":
self._transcribe()
if ENABLE_STT:
self._transcribe()
else:
self._send_error(503, "Spracherkennung ist nicht installiert",
"server_error", "feature_disabled")
elif path == "/images" and self.command == "GET":
self._images_list()
elif path.startswith("/images/") and self.command == "GET":
+3 -3
View File
@@ -2,15 +2,15 @@
"profiles": {
"fast": {
"context": 76800,
"model_alias": "qwen38-27b-iq4mix-76k-mtp2-vision"
"model_alias": "qwen-fast"
},
"medium": {
"context": 94208,
"model_alias": "qwen38-27b-iq4xs-pure-92k"
"model_alias": "qwen-medium"
},
"long": {
"context": 131072,
"model_alias": "qwen38-27b-iq4mix-128k-mtp2-ffn12"
"model_alias": "qwen-long"
}
}
}