router: Readiness-Check wartet auf geladenes Modell (ctx-match)

This commit is contained in:
Mikei386
2026-08-18 22:58:50 +02:00
parent bddbc35476
commit c5d92acd93
+30 -16
View File
@@ -137,10 +137,27 @@ def current_profile() -> str | None:
return None return None
def switch_profile(profile: str, implicit: bool = False) -> None: def _wait_ready(profile: str, deadline: float) -> None:
"""Führt einen Profilwechsel aus und wartet, bis llama.cpp bereit ist. """Wartet, bis llama.cpp das Profil geladen hat (Modell + ctx)."""
expected_ctx = PROFILES[profile]
while True:
status = upstream_status()
if (status["reachable"] and status.get("model")
and status.get("ctx") == expected_ctx):
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
profile, status.get("model"), status.get("ctx"))
return
if time.monotonic() > deadline:
raise RuntimeError(
f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht bereit "
f"(erwartet ctx {expected_ctx}, aktuell: {status.get('ctx')})")
time.sleep(POLL_INTERVAL)
Wirft RuntimeError, wenn der Wechsel nicht erfolgreich war.
def switch_profile(profile: str, implicit: bool = False) -> None:
"""Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist.
Wirft RuntimeError, wenn das Profil nicht aktiviert werden konnte.
implicit=True (ausgelöst durch ein virtuelles Modell in einem Chat-Request): implicit=True (ausgelöst durch ein virtuelles Modell in einem Chat-Request):
Wenn das Profil bereits aktiv ist, aber llama.cpp down ist, wird sofort Wenn das Profil bereits aktiv ist, aber llama.cpp down ist, wird sofort
@@ -155,9 +172,16 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
try: try:
cur = current_profile() cur = current_profile()
up = upstream_status() up = upstream_status()
if cur == profile and up["reachable"]: ready = (up["reachable"] and up.get("model")
and up.get("ctx") == PROFILES[profile])
if cur == profile and ready:
log.info("Profil %s ist bereits aktiv", profile) log.info("Profil %s ist bereits aktiv", profile)
return return
if cur == profile and up["reachable"] and not ready:
# Modell wird gerade geladen (z.B. nach einem Wechsel)
log.info("Warte, bis Profil %s geladen ist ...", profile)
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
return
if cur == profile and not up["reachable"] and implicit: if cur == profile and not up["reachable"] and implicit:
raise RuntimeError( raise RuntimeError(
f"llama.cpp nicht erreichbar (Profil {profile} ist bereits " f"llama.cpp nicht erreichbar (Profil {profile} ist bereits "
@@ -184,18 +208,8 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
if current_profile() != profile: if current_profile() != profile:
raise RuntimeError( raise RuntimeError(
f"Profildatei wurde nicht gesetzt (erwartet: {profile})") f"Profildatei wurde nicht gesetzt (erwartet: {profile})")
log.info("Warte, bis llama.cpp wieder erreichbar ist ...") log.info("Warte, bis llama.cpp das Profil geladen hat ...")
deadline = time.monotonic() + SWITCH_TIMEOUT _wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
while True:
status = upstream_status()
if status["reachable"]:
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
profile, status.get("model"), status.get("ctx"))
return
if time.monotonic() > deadline:
raise RuntimeError(
f"llama.cpp nach {SWITCH_TIMEOUT:.0f} s nicht erreichbar")
time.sleep(POLL_INTERVAL)
finally: finally:
STATE.switching = None STATE.switching = None