From c5d92acd9392d0ad4fa115a5278c329a6ea6ea98 Mon Sep 17 00:00:00 2001 From: Mikei386 <44135113+Mikei386@users.noreply.github.com> Date: Tue, 18 Aug 2026 22:58:50 +0200 Subject: [PATCH] router: Readiness-Check wartet auf geladenes Modell (ctx-match) --- router/ai_profile_router.py | 46 ++++++++++++++++++++++++------------- 1 file changed, 30 insertions(+), 16 deletions(-) diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index d5241f3..526c135 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -137,10 +137,27 @@ def current_profile() -> str | None: return None -def switch_profile(profile: str, implicit: bool = False) -> None: - """Führt einen Profilwechsel aus und wartet, bis llama.cpp bereit ist. +def _wait_ready(profile: str, deadline: float) -> None: + """Wartet, bis llama.cpp das Profil geladen hat (Modell + ctx).""" + expected_ctx = PROFILES[profile] + while True: + status = upstream_status() + if (status["reachable"] and status.get("model") + and status.get("ctx") == expected_ctx): + log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s", + profile, status.get("model"), status.get("ctx")) + return + if time.monotonic() > deadline: + raise RuntimeError( + f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht bereit " + f"(erwartet ctx {expected_ctx}, aktuell: {status.get('ctx')})") + time.sleep(POLL_INTERVAL) - Wirft RuntimeError, wenn der Wechsel nicht erfolgreich war. + +def switch_profile(profile: str, implicit: bool = False) -> None: + """Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist. + + Wirft RuntimeError, wenn das Profil nicht aktiviert werden konnte. implicit=True (ausgelöst durch ein virtuelles Modell in einem Chat-Request): Wenn das Profil bereits aktiv ist, aber llama.cpp down ist, wird sofort @@ -155,9 +172,16 @@ def switch_profile(profile: str, implicit: bool = False) -> None: try: cur = current_profile() up = upstream_status() - if cur == profile and up["reachable"]: + ready = (up["reachable"] and up.get("model") + and up.get("ctx") == PROFILES[profile]) + if cur == profile and ready: log.info("Profil %s ist bereits aktiv", profile) return + if cur == profile and up["reachable"] and not ready: + # Modell wird gerade geladen (z.B. nach einem Wechsel) + log.info("Warte, bis Profil %s geladen ist ...", profile) + _wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT) + return if cur == profile and not up["reachable"] and implicit: raise RuntimeError( f"llama.cpp nicht erreichbar (Profil {profile} ist bereits " @@ -184,18 +208,8 @@ def switch_profile(profile: str, implicit: bool = False) -> None: if current_profile() != profile: raise RuntimeError( f"Profildatei wurde nicht gesetzt (erwartet: {profile})") - log.info("Warte, bis llama.cpp wieder erreichbar ist ...") - deadline = time.monotonic() + SWITCH_TIMEOUT - while True: - status = upstream_status() - if status["reachable"]: - log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s", - profile, status.get("model"), status.get("ctx")) - return - if time.monotonic() > deadline: - raise RuntimeError( - f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht erreichbar") - time.sleep(POLL_INTERVAL) + log.info("Warte, bis llama.cpp das Profil geladen hat ...") + _wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT) finally: STATE.switching = None