router: Readiness-Check wartet auf geladenes Modell (ctx-match)
This commit is contained in:
+30
-16
@@ -137,10 +137,27 @@ def current_profile() -> str | None:
|
||||
return None
|
||||
|
||||
|
||||
def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
"""Führt einen Profilwechsel aus und wartet, bis llama.cpp bereit ist.
|
||||
def _wait_ready(profile: str, deadline: float) -> None:
|
||||
"""Wartet, bis llama.cpp das Profil geladen hat (Modell + ctx)."""
|
||||
expected_ctx = PROFILES[profile]
|
||||
while True:
|
||||
status = upstream_status()
|
||||
if (status["reachable"] and status.get("model")
|
||||
and status.get("ctx") == expected_ctx):
|
||||
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
|
||||
profile, status.get("model"), status.get("ctx"))
|
||||
return
|
||||
if time.monotonic() > deadline:
|
||||
raise RuntimeError(
|
||||
f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht bereit "
|
||||
f"(erwartet ctx {expected_ctx}, aktuell: {status.get('ctx')})")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
|
||||
Wirft RuntimeError, wenn der Wechsel nicht erfolgreich war.
|
||||
|
||||
def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
"""Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist.
|
||||
|
||||
Wirft RuntimeError, wenn das Profil nicht aktiviert werden konnte.
|
||||
|
||||
implicit=True (ausgelöst durch ein virtuelles Modell in einem Chat-Request):
|
||||
Wenn das Profil bereits aktiv ist, aber llama.cpp down ist, wird sofort
|
||||
@@ -155,9 +172,16 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
try:
|
||||
cur = current_profile()
|
||||
up = upstream_status()
|
||||
if cur == profile and up["reachable"]:
|
||||
ready = (up["reachable"] and up.get("model")
|
||||
and up.get("ctx") == PROFILES[profile])
|
||||
if cur == profile and ready:
|
||||
log.info("Profil %s ist bereits aktiv", profile)
|
||||
return
|
||||
if cur == profile and up["reachable"] and not ready:
|
||||
# Modell wird gerade geladen (z.B. nach einem Wechsel)
|
||||
log.info("Warte, bis Profil %s geladen ist ...", profile)
|
||||
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
|
||||
return
|
||||
if cur == profile and not up["reachable"] and implicit:
|
||||
raise RuntimeError(
|
||||
f"llama.cpp nicht erreichbar (Profil {profile} ist bereits "
|
||||
@@ -184,18 +208,8 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
if current_profile() != profile:
|
||||
raise RuntimeError(
|
||||
f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
|
||||
log.info("Warte, bis llama.cpp wieder erreichbar ist ...")
|
||||
deadline = time.monotonic() + SWITCH_TIMEOUT
|
||||
while True:
|
||||
status = upstream_status()
|
||||
if status["reachable"]:
|
||||
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
|
||||
profile, status.get("model"), status.get("ctx"))
|
||||
return
|
||||
if time.monotonic() > deadline:
|
||||
raise RuntimeError(
|
||||
f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht erreichbar")
|
||||
time.sleep(POLL_INTERVAL)
|
||||
log.info("Warte, bis llama.cpp das Profil geladen hat ...")
|
||||
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
|
||||
finally:
|
||||
STATE.switching = None
|
||||
|
||||
|
||||
Reference in New Issue
Block a user