router: Readiness-Check wartet auf geladenes Modell (ctx-match)
This commit is contained in:
+30
-16
@@ -137,10 +137,27 @@ def current_profile() -> str | None:
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def switch_profile(profile: str, implicit: bool = False) -> None:
|
def _wait_ready(profile: str, deadline: float) -> None:
|
||||||
"""Führt einen Profilwechsel aus und wartet, bis llama.cpp bereit ist.
|
"""Wartet, bis llama.cpp das Profil geladen hat (Modell + ctx)."""
|
||||||
|
expected_ctx = PROFILES[profile]
|
||||||
|
while True:
|
||||||
|
status = upstream_status()
|
||||||
|
if (status["reachable"] and status.get("model")
|
||||||
|
and status.get("ctx") == expected_ctx):
|
||||||
|
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
|
||||||
|
profile, status.get("model"), status.get("ctx"))
|
||||||
|
return
|
||||||
|
if time.monotonic() > deadline:
|
||||||
|
raise RuntimeError(
|
||||||
|
f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht bereit "
|
||||||
|
f"(erwartet ctx {expected_ctx}, aktuell: {status.get('ctx')})")
|
||||||
|
time.sleep(POLL_INTERVAL)
|
||||||
|
|
||||||
Wirft RuntimeError, wenn der Wechsel nicht erfolgreich war.
|
|
||||||
|
def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||||
|
"""Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist.
|
||||||
|
|
||||||
|
Wirft RuntimeError, wenn das Profil nicht aktiviert werden konnte.
|
||||||
|
|
||||||
implicit=True (ausgelöst durch ein virtuelles Modell in einem Chat-Request):
|
implicit=True (ausgelöst durch ein virtuelles Modell in einem Chat-Request):
|
||||||
Wenn das Profil bereits aktiv ist, aber llama.cpp down ist, wird sofort
|
Wenn das Profil bereits aktiv ist, aber llama.cpp down ist, wird sofort
|
||||||
@@ -155,9 +172,16 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
|||||||
try:
|
try:
|
||||||
cur = current_profile()
|
cur = current_profile()
|
||||||
up = upstream_status()
|
up = upstream_status()
|
||||||
if cur == profile and up["reachable"]:
|
ready = (up["reachable"] and up.get("model")
|
||||||
|
and up.get("ctx") == PROFILES[profile])
|
||||||
|
if cur == profile and ready:
|
||||||
log.info("Profil %s ist bereits aktiv", profile)
|
log.info("Profil %s ist bereits aktiv", profile)
|
||||||
return
|
return
|
||||||
|
if cur == profile and up["reachable"] and not ready:
|
||||||
|
# Modell wird gerade geladen (z.B. nach einem Wechsel)
|
||||||
|
log.info("Warte, bis Profil %s geladen ist ...", profile)
|
||||||
|
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
|
||||||
|
return
|
||||||
if cur == profile and not up["reachable"] and implicit:
|
if cur == profile and not up["reachable"] and implicit:
|
||||||
raise RuntimeError(
|
raise RuntimeError(
|
||||||
f"llama.cpp nicht erreichbar (Profil {profile} ist bereits "
|
f"llama.cpp nicht erreichbar (Profil {profile} ist bereits "
|
||||||
@@ -184,18 +208,8 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
|||||||
if current_profile() != profile:
|
if current_profile() != profile:
|
||||||
raise RuntimeError(
|
raise RuntimeError(
|
||||||
f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
|
f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
|
||||||
log.info("Warte, bis llama.cpp wieder erreichbar ist ...")
|
log.info("Warte, bis llama.cpp das Profil geladen hat ...")
|
||||||
deadline = time.monotonic() + SWITCH_TIMEOUT
|
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
|
||||||
while True:
|
|
||||||
status = upstream_status()
|
|
||||||
if status["reachable"]:
|
|
||||||
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
|
|
||||||
profile, status.get("model"), status.get("ctx"))
|
|
||||||
return
|
|
||||||
if time.monotonic() > deadline:
|
|
||||||
raise RuntimeError(
|
|
||||||
f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht erreichbar")
|
|
||||||
time.sleep(POLL_INTERVAL)
|
|
||||||
finally:
|
finally:
|
||||||
STATE.switching = None
|
STATE.switching = None
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user