Reduce controller polling and bound chat admission waits

This commit is contained in:
Mikei386
2026-09-20 19:10:59 +02:00
parent 53320fa6c3
commit 6071b1a2cd
10 changed files with 303 additions and 134 deletions
+63 -41
View File
@@ -73,6 +73,7 @@ import threading
import time
import uuid
import http.client
from contextlib import contextmanager
import urllib.error
import urllib.parse
import urllib.request
@@ -299,6 +300,21 @@ class _State:
STATE = _State()
class ModelWaitTimeout(RuntimeError):
"""The request did not obtain the GPU coordinator within its wait budget."""
@contextmanager
def _model_lock():
# Socket timeouts do not bound a Python lock acquisition.
if not STATE.lock.acquire(timeout=max(0, CHAT_WAIT_TIMEOUT)):
raise ModelWaitTimeout("Wartezeit auf das Modell überschritten; bitte erneut versuchen")
try:
yield
finally:
STATE.lock.release()
def _wait_chats_drained(timeout: float | None = None) -> None:
"""Wartet, bis keine aktiven Chat-Requests mehr laufen.
@@ -965,7 +981,7 @@ def current_profile() -> str | None:
"""
if PROFILE_CONTROL_URL:
try:
profile = _profile_controller_request("GET", "/status").get(
profile = _profile_controller_request("GET", "/profiles/status", timeout=3).get(
"active_profile")
return profile if profile in PROFILES else None
except Exception as exc:
@@ -1029,7 +1045,7 @@ def _context_matches(expected: int, reported: object) -> bool:
and expected <= reported <= expected + 1024)
def switch_profile(profile: str, implicit: bool = False) -> None:
def switch_profile(profile: str, implicit: bool = False) -> dict:
"""Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist.
Wirft RuntimeError, wenn das Profil nicht aktiviert werden konnte.
@@ -1058,7 +1074,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
# so clear the stale flag before returning.
_set_qwen_unavailable(False)
log.info("Profil %s ist bereits aktiv", profile)
return
return up
# Qwen wird neu geladen/gewechselt → für Chats nicht verfügbar.
_set_qwen_unavailable(True)
try:
@@ -1067,7 +1083,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
# Modell wird gerade geladen (z.B. nach einem Wechsel)
log.info("Warte, bis Profil %s geladen ist ...", profile)
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
return
return upstream_status()
if cur == profile and not up["reachable"] and implicit:
raise RuntimeError(
f"llama.cpp nicht erreichbar (Profil {profile} ist "
@@ -1114,6 +1130,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
_set_qwen_unavailable(not available)
finally:
STATE.switching = None
return upstream_status()
# ---------------------------------------------------------------------------
@@ -2102,7 +2119,8 @@ class Handler(BaseHTTPRequestHandler):
"value": "loaded" if name == active else "unloaded",
},
"architecture": {
"input_modalities": ["text", "image"],
"input_modalities": (["text", "image"]
if PROFILE_REGISTRY[name]["vision"] else ["text"]),
"output_modalities": ["text"],
},
})
@@ -2155,29 +2173,27 @@ class Handler(BaseHTTPRequestHandler):
@staticmethod
def _mode_payload() -> dict:
state = RUNTIME.load()
return {
snapshot = {}
if PROFILE_CONTROL_URL:
try:
snapshot = _profile_controller_request("GET", "/status", timeout=3)
except Exception as exc:
log.warning("Controller-Status nicht verfügbar: %s", exc)
fallback = "unknown" if PROFILE_CONTROL_URL else "unsupported"
result = {
"active": STATE.mode,
"phase": STATE.mode_phase,
"music_worker": _music_worker_state(),
"music_health": _music_worker_health(),
"yue2_worker": _worker_field("yue2_worker"),
"yue2_health": _worker_field("yue2_health"),
"separator_worker": _separator_worker_state(),
"separator_health": _separator_worker_health(),
"voice_worker": _voice_worker_state(),
"voice_health": _voice_worker_health(),
"voice_change_worker": _voice_change_worker_state(),
"voice_change_health": _voice_change_worker_health(),
"applio_worker": _worker_field("applio_worker"),
"applio_health": _worker_field("applio_health"),
"trellis_worker": _worker_field("trellis_worker"),
"trellis_health": _worker_field("trellis_health"),
"video_worker": _worker_field("video_worker"),
"video_health": _worker_field("video_health"),
"return_profile": state.get("return_profile"),
"last_error": STATE.mode_error,
"enabled": ENABLE_MUSIC_MODE,
"worker_errors": snapshot.get("worker_errors", {}),
}
for worker in ("music", "yue2", "separator", "voice", "voice_change",
"applio", "trellis", "video"):
for suffix in ("worker", "health"):
field = f"{worker}_{suffix}"
result[field] = snapshot.get(field, fallback)
return result
def _mode_change(self) -> None:
try:
@@ -2940,10 +2956,9 @@ class Handler(BaseHTTPRequestHandler):
def _acquire_model_lease(self, profile: str | None = None) -> dict:
"""Atomar Profil sicherstellen und einen aktiven Request registrieren."""
with STATE.lock:
if profile is not None:
switch_profile(profile, implicit=True)
up = upstream_status()
with _model_lock():
up = (switch_profile(profile, implicit=True)
if profile is not None else upstream_status())
if not up["reachable"] or not up.get("model"):
raise RuntimeError("llama.cpp nicht erreichbar")
with STATE.avail_lock:
@@ -2961,6 +2976,9 @@ class Handler(BaseHTTPRequestHandler):
profile: str) -> None:
try:
up = self._acquire_model_lease(profile)
except ModelWaitTimeout as e:
self._send_error(503, str(e), "server_error", "model_wait_timeout")
return
except (ValueError, RuntimeError) as e:
self._send_error(502, str(e), "server_error",
"upstream_unavailable")
@@ -2976,19 +2994,22 @@ class Handler(BaseHTTPRequestHandler):
"""Bildvalidierung, Profilwahl und Chat-Lease als eine Transaktion."""
lease_acquired = False
try:
with STATE.lock:
if profile is not None:
switch_profile(profile, implicit=True)
data = _normalize_llamacpp_reasoning(data)
data = _cap_chat_generation(data)
if _request_has_image(data):
data = _normalize_chat_images(data)
# Validation and image decoding need no GPU ownership.
data = _normalize_llamacpp_reasoning(data)
data = _cap_chat_generation(data)
has_image = _request_has_image(data)
if has_image:
data = _normalize_chat_images(data)
with _model_lock():
selected = profile or (current_profile() if has_image else None)
if has_image and selected in PROFILE_REGISTRY and not PROFILE_REGISTRY[selected]["vision"]:
raise ValueError(f"Profil {selected} unterstützt keine Bilder")
up = (switch_profile(profile, implicit=True)
if profile is not None else upstream_status())
if has_image:
log.info("Vision: Bild wird direkt an das aktive "
"multimodale Qwen-Profil weitergeleitet")
up = upstream_status()
if not up["reachable"] or not up.get("model"):
raise RuntimeError("llama.cpp nicht erreichbar")
if profile is not None:
@@ -3000,7 +3021,11 @@ class Handler(BaseHTTPRequestHandler):
STATE.active_chats += 1
lease_acquired = True
self._proxy(body)
except (ValueError, RuntimeError) as e:
except ModelWaitTimeout as e:
self._send_error(503, str(e), "server_error", "model_wait_timeout")
except ValueError as e:
self._send_error(400, str(e), "invalid_request_error", "invalid_chat_request")
except RuntimeError as e:
self._send_error(502, str(e), "server_error", "upstream_unavailable")
finally:
if lease_acquired:
@@ -3218,10 +3243,7 @@ def _startup_reconcile() -> None:
return
up = upstream_status()
if (up["reachable"] and up.get("model")
and up.get("ctx") == PROFILES[profile]
and (not EXPECTED_MODELS.get(profile)
or up.get("model") == EXPECTED_MODELS[profile])):
if _profile_is_ready(profile, up):
_set_qwen_unavailable(False)
STATE.mode = "llm"
STATE.mode_phase = "ready"