Add reproducible Docker and WireGuard host bootstrap
This commit is contained in:
+77
-20
@@ -92,6 +92,19 @@ UPSTREAM_URL = os.environ.get("UPSTREAM_URL", "http://127.0.0.1:8080").rstrip("/
|
||||
PROFILE_SCRIPT = os.environ.get("PROFILE_SCRIPT", "/usr/local/bin/llama-profile")
|
||||
PROFILE_DIR = os.environ.get(
|
||||
"PROFILE_DIR", "/etc/systemd/system/mike-ai-llama-ui.service.d")
|
||||
PROFILE_CONTROL_URL = os.environ.get("PROFILE_CONTROL_URL", "").rstrip("/")
|
||||
PROFILE_CONTROL_TOKEN_FILE = os.environ.get(
|
||||
"PROFILE_CONTROL_TOKEN_FILE", "/run/secrets/controller-token")
|
||||
|
||||
# Optional worker APIs. The clean Docker baseline deliberately ships only
|
||||
# text/multimodal chat; absent workers must fail explicitly instead of trying
|
||||
# legacy systemd paths inside the container.
|
||||
ENABLE_IMAGE_GENERATION = os.environ.get(
|
||||
"ENABLE_IMAGE_GENERATION", "true").lower() in {"1", "true", "yes"}
|
||||
ENABLE_TTS = os.environ.get(
|
||||
"ENABLE_TTS", "true").lower() in {"1", "true", "yes"}
|
||||
ENABLE_STT = os.environ.get(
|
||||
"ENABLE_STT", "true").lower() in {"1", "true", "yes"}
|
||||
|
||||
SWITCH_TIMEOUT = float(os.environ.get("SWITCH_TIMEOUT", "600")) # s, Warten auf llama.cpp
|
||||
REQUEST_TIMEOUT = float(os.environ.get("REQUEST_TIMEOUT", "600")) # s, Read-Timeout Upstream
|
||||
@@ -420,12 +433,40 @@ def _read(path: str) -> str:
|
||||
return f.read().strip()
|
||||
|
||||
|
||||
def _profile_controller_request(method: str, path: str) -> dict:
|
||||
token = os.environ.get("PROFILE_CONTROL_TOKEN", "").strip()
|
||||
if not token:
|
||||
token = _read(PROFILE_CONTROL_TOKEN_FILE)
|
||||
if len(token) < 32:
|
||||
raise RuntimeError("Profil-Controller-Token fehlt oder ist zu kurz")
|
||||
request = urllib.request.Request(
|
||||
PROFILE_CONTROL_URL + path,
|
||||
method=method,
|
||||
headers={"Authorization": f"Bearer {token}"},
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(request, timeout=120) as response:
|
||||
return json.load(response)
|
||||
except urllib.error.HTTPError as exc:
|
||||
body = exc.read(500).decode(errors="replace")
|
||||
raise RuntimeError(
|
||||
f"Profil-Controller HTTP {exc.code}: {body}") from exc
|
||||
|
||||
|
||||
def current_profile() -> str | None:
|
||||
"""Aktives Profil anhand semantischer Werte der override.conf.
|
||||
|
||||
Kommentare, Leerraum oder die Reihenfolge anderer llama.cpp-Optionen
|
||||
beeinflussen die Erkennung nicht mehr.
|
||||
"""
|
||||
if PROFILE_CONTROL_URL:
|
||||
try:
|
||||
profile = _profile_controller_request("GET", "/status").get(
|
||||
"active_profile")
|
||||
return profile if profile in PROFILES else None
|
||||
except Exception as exc:
|
||||
log.warning("Profil-Controller-Status nicht verfügbar: %s", exc)
|
||||
return None
|
||||
try:
|
||||
override = _read(os.path.join(PROFILE_DIR, "override.conf"))
|
||||
except OSError:
|
||||
@@ -512,23 +553,27 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
f"llama.cpp nicht erreichbar (Profil {profile} ist "
|
||||
f"bereits aktiv; Neustart über /{profile})")
|
||||
log.info("Profilwechsel: %s -> %s", cur, profile)
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[PROFILE_SCRIPT, profile],
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
timeout=120,
|
||||
)
|
||||
out = proc.stdout.decode(errors="replace").strip()
|
||||
if out:
|
||||
log.info("llama-profile: %s", out[-500:])
|
||||
if proc.returncode != 0:
|
||||
raise RuntimeError(
|
||||
f"llama-profile fehlgeschlagen (Exit-Code "
|
||||
f"{proc.returncode}): {out[-500:]}")
|
||||
except subprocess.TimeoutExpired:
|
||||
raise RuntimeError("llama-profile hat 120 s überschritten")
|
||||
if PROFILE_CONTROL_URL:
|
||||
_profile_controller_request(
|
||||
"POST", f"/profiles/{profile}/activate")
|
||||
else:
|
||||
try:
|
||||
proc = subprocess.run(
|
||||
[PROFILE_SCRIPT, profile],
|
||||
stdin=subprocess.DEVNULL,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
timeout=120,
|
||||
)
|
||||
out = proc.stdout.decode(errors="replace").strip()
|
||||
if out:
|
||||
log.info("llama-profile: %s", out[-500:])
|
||||
if proc.returncode != 0:
|
||||
raise RuntimeError(
|
||||
f"llama-profile fehlgeschlagen (Exit-Code "
|
||||
f"{proc.returncode}): {out[-500:]}")
|
||||
except subprocess.TimeoutExpired:
|
||||
raise RuntimeError("llama-profile hat 120 s überschritten")
|
||||
if current_profile() != profile:
|
||||
raise RuntimeError(
|
||||
f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
|
||||
@@ -1041,11 +1086,23 @@ class Handler(BaseHTTPRequestHandler):
|
||||
elif path == "/v1/audio/voices" and self.command == "GET":
|
||||
self._send_json(200, self._audio_voices_payload())
|
||||
elif path == "/v1/images/generations" and self.command == "POST":
|
||||
self._image_generate()
|
||||
if ENABLE_IMAGE_GENERATION:
|
||||
self._image_generate()
|
||||
else:
|
||||
self._send_error(503, "Bildgenerierung ist nicht installiert",
|
||||
"server_error", "feature_disabled")
|
||||
elif path == "/v1/audio/speech" and self.command == "POST":
|
||||
self._speech()
|
||||
if ENABLE_TTS:
|
||||
self._speech()
|
||||
else:
|
||||
self._send_error(503, "Sprachausgabe ist nicht installiert",
|
||||
"server_error", "feature_disabled")
|
||||
elif path == "/v1/audio/transcriptions" and self.command == "POST":
|
||||
self._transcribe()
|
||||
if ENABLE_STT:
|
||||
self._transcribe()
|
||||
else:
|
||||
self._send_error(503, "Spracherkennung ist nicht installiert",
|
||||
"server_error", "feature_disabled")
|
||||
elif path == "/images" and self.command == "GET":
|
||||
self._images_list()
|
||||
elif path.startswith("/images/") and self.command == "GET":
|
||||
|
||||
Reference in New Issue
Block a user