Reduce controller polling and bound chat admission waits
This commit is contained in:
+63
-41
@@ -73,6 +73,7 @@ import threading
|
||||
import time
|
||||
import uuid
|
||||
import http.client
|
||||
from contextlib import contextmanager
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
@@ -299,6 +300,21 @@ class _State:
|
||||
STATE = _State()
|
||||
|
||||
|
||||
class ModelWaitTimeout(RuntimeError):
|
||||
"""The request did not obtain the GPU coordinator within its wait budget."""
|
||||
|
||||
|
||||
@contextmanager
|
||||
def _model_lock():
|
||||
# Socket timeouts do not bound a Python lock acquisition.
|
||||
if not STATE.lock.acquire(timeout=max(0, CHAT_WAIT_TIMEOUT)):
|
||||
raise ModelWaitTimeout("Wartezeit auf das Modell überschritten; bitte erneut versuchen")
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
STATE.lock.release()
|
||||
|
||||
|
||||
def _wait_chats_drained(timeout: float | None = None) -> None:
|
||||
"""Wartet, bis keine aktiven Chat-Requests mehr laufen.
|
||||
|
||||
@@ -965,7 +981,7 @@ def current_profile() -> str | None:
|
||||
"""
|
||||
if PROFILE_CONTROL_URL:
|
||||
try:
|
||||
profile = _profile_controller_request("GET", "/status").get(
|
||||
profile = _profile_controller_request("GET", "/profiles/status", timeout=3).get(
|
||||
"active_profile")
|
||||
return profile if profile in PROFILES else None
|
||||
except Exception as exc:
|
||||
@@ -1029,7 +1045,7 @@ def _context_matches(expected: int, reported: object) -> bool:
|
||||
and expected <= reported <= expected + 1024)
|
||||
|
||||
|
||||
def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
def switch_profile(profile: str, implicit: bool = False) -> dict:
|
||||
"""Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist.
|
||||
|
||||
Wirft RuntimeError, wenn das Profil nicht aktiviert werden konnte.
|
||||
@@ -1058,7 +1074,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
# so clear the stale flag before returning.
|
||||
_set_qwen_unavailable(False)
|
||||
log.info("Profil %s ist bereits aktiv", profile)
|
||||
return
|
||||
return up
|
||||
# Qwen wird neu geladen/gewechselt → für Chats nicht verfügbar.
|
||||
_set_qwen_unavailable(True)
|
||||
try:
|
||||
@@ -1067,7 +1083,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
# Modell wird gerade geladen (z.B. nach einem Wechsel)
|
||||
log.info("Warte, bis Profil %s geladen ist ...", profile)
|
||||
_wait_ready(profile, time.monotonic() + SWITCH_TIMEOUT)
|
||||
return
|
||||
return upstream_status()
|
||||
if cur == profile and not up["reachable"] and implicit:
|
||||
raise RuntimeError(
|
||||
f"llama.cpp nicht erreichbar (Profil {profile} ist "
|
||||
@@ -1114,6 +1130,7 @@ def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
_set_qwen_unavailable(not available)
|
||||
finally:
|
||||
STATE.switching = None
|
||||
return upstream_status()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -2102,7 +2119,8 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"value": "loaded" if name == active else "unloaded",
|
||||
},
|
||||
"architecture": {
|
||||
"input_modalities": ["text", "image"],
|
||||
"input_modalities": (["text", "image"]
|
||||
if PROFILE_REGISTRY[name]["vision"] else ["text"]),
|
||||
"output_modalities": ["text"],
|
||||
},
|
||||
})
|
||||
@@ -2155,29 +2173,27 @@ class Handler(BaseHTTPRequestHandler):
|
||||
@staticmethod
|
||||
def _mode_payload() -> dict:
|
||||
state = RUNTIME.load()
|
||||
return {
|
||||
snapshot = {}
|
||||
if PROFILE_CONTROL_URL:
|
||||
try:
|
||||
snapshot = _profile_controller_request("GET", "/status", timeout=3)
|
||||
except Exception as exc:
|
||||
log.warning("Controller-Status nicht verfügbar: %s", exc)
|
||||
fallback = "unknown" if PROFILE_CONTROL_URL else "unsupported"
|
||||
result = {
|
||||
"active": STATE.mode,
|
||||
"phase": STATE.mode_phase,
|
||||
"music_worker": _music_worker_state(),
|
||||
"music_health": _music_worker_health(),
|
||||
"yue2_worker": _worker_field("yue2_worker"),
|
||||
"yue2_health": _worker_field("yue2_health"),
|
||||
"separator_worker": _separator_worker_state(),
|
||||
"separator_health": _separator_worker_health(),
|
||||
"voice_worker": _voice_worker_state(),
|
||||
"voice_health": _voice_worker_health(),
|
||||
"voice_change_worker": _voice_change_worker_state(),
|
||||
"voice_change_health": _voice_change_worker_health(),
|
||||
"applio_worker": _worker_field("applio_worker"),
|
||||
"applio_health": _worker_field("applio_health"),
|
||||
"trellis_worker": _worker_field("trellis_worker"),
|
||||
"trellis_health": _worker_field("trellis_health"),
|
||||
"video_worker": _worker_field("video_worker"),
|
||||
"video_health": _worker_field("video_health"),
|
||||
"return_profile": state.get("return_profile"),
|
||||
"last_error": STATE.mode_error,
|
||||
"enabled": ENABLE_MUSIC_MODE,
|
||||
"worker_errors": snapshot.get("worker_errors", {}),
|
||||
}
|
||||
for worker in ("music", "yue2", "separator", "voice", "voice_change",
|
||||
"applio", "trellis", "video"):
|
||||
for suffix in ("worker", "health"):
|
||||
field = f"{worker}_{suffix}"
|
||||
result[field] = snapshot.get(field, fallback)
|
||||
return result
|
||||
|
||||
def _mode_change(self) -> None:
|
||||
try:
|
||||
@@ -2940,10 +2956,9 @@ class Handler(BaseHTTPRequestHandler):
|
||||
|
||||
def _acquire_model_lease(self, profile: str | None = None) -> dict:
|
||||
"""Atomar Profil sicherstellen und einen aktiven Request registrieren."""
|
||||
with STATE.lock:
|
||||
if profile is not None:
|
||||
switch_profile(profile, implicit=True)
|
||||
up = upstream_status()
|
||||
with _model_lock():
|
||||
up = (switch_profile(profile, implicit=True)
|
||||
if profile is not None else upstream_status())
|
||||
if not up["reachable"] or not up.get("model"):
|
||||
raise RuntimeError("llama.cpp nicht erreichbar")
|
||||
with STATE.avail_lock:
|
||||
@@ -2961,6 +2976,9 @@ class Handler(BaseHTTPRequestHandler):
|
||||
profile: str) -> None:
|
||||
try:
|
||||
up = self._acquire_model_lease(profile)
|
||||
except ModelWaitTimeout as e:
|
||||
self._send_error(503, str(e), "server_error", "model_wait_timeout")
|
||||
return
|
||||
except (ValueError, RuntimeError) as e:
|
||||
self._send_error(502, str(e), "server_error",
|
||||
"upstream_unavailable")
|
||||
@@ -2976,19 +2994,22 @@ class Handler(BaseHTTPRequestHandler):
|
||||
"""Bildvalidierung, Profilwahl und Chat-Lease als eine Transaktion."""
|
||||
lease_acquired = False
|
||||
try:
|
||||
with STATE.lock:
|
||||
if profile is not None:
|
||||
switch_profile(profile, implicit=True)
|
||||
|
||||
data = _normalize_llamacpp_reasoning(data)
|
||||
data = _cap_chat_generation(data)
|
||||
|
||||
if _request_has_image(data):
|
||||
data = _normalize_chat_images(data)
|
||||
# Validation and image decoding need no GPU ownership.
|
||||
data = _normalize_llamacpp_reasoning(data)
|
||||
data = _cap_chat_generation(data)
|
||||
has_image = _request_has_image(data)
|
||||
if has_image:
|
||||
data = _normalize_chat_images(data)
|
||||
with _model_lock():
|
||||
selected = profile or (current_profile() if has_image else None)
|
||||
if has_image and selected in PROFILE_REGISTRY and not PROFILE_REGISTRY[selected]["vision"]:
|
||||
raise ValueError(f"Profil {selected} unterstützt keine Bilder")
|
||||
up = (switch_profile(profile, implicit=True)
|
||||
if profile is not None else upstream_status())
|
||||
if has_image:
|
||||
log.info("Vision: Bild wird direkt an das aktive "
|
||||
"multimodale Qwen-Profil weitergeleitet")
|
||||
|
||||
up = upstream_status()
|
||||
if not up["reachable"] or not up.get("model"):
|
||||
raise RuntimeError("llama.cpp nicht erreichbar")
|
||||
if profile is not None:
|
||||
@@ -3000,7 +3021,11 @@ class Handler(BaseHTTPRequestHandler):
|
||||
STATE.active_chats += 1
|
||||
lease_acquired = True
|
||||
self._proxy(body)
|
||||
except (ValueError, RuntimeError) as e:
|
||||
except ModelWaitTimeout as e:
|
||||
self._send_error(503, str(e), "server_error", "model_wait_timeout")
|
||||
except ValueError as e:
|
||||
self._send_error(400, str(e), "invalid_request_error", "invalid_chat_request")
|
||||
except RuntimeError as e:
|
||||
self._send_error(502, str(e), "server_error", "upstream_unavailable")
|
||||
finally:
|
||||
if lease_acquired:
|
||||
@@ -3218,10 +3243,7 @@ def _startup_reconcile() -> None:
|
||||
return
|
||||
|
||||
up = upstream_status()
|
||||
if (up["reachable"] and up.get("model")
|
||||
and up.get("ctx") == PROFILES[profile]
|
||||
and (not EXPECTED_MODELS.get(profile)
|
||||
or up.get("model") == EXPECTED_MODELS[profile])):
|
||||
if _profile_is_ready(profile, up):
|
||||
_set_qwen_unavailable(False)
|
||||
STATE.mode = "llm"
|
||||
STATE.mode_phase = "ready"
|
||||
|
||||
@@ -2,23 +2,28 @@
|
||||
"profiles": {
|
||||
"fast": {
|
||||
"context": 76800,
|
||||
"model_alias": "qwen-fast"
|
||||
"model_alias": "qwen-fast",
|
||||
"vision": true
|
||||
},
|
||||
"medium": {
|
||||
"context": 160000,
|
||||
"model_alias": "qwen-medium"
|
||||
"model_alias": "qwen-medium",
|
||||
"vision": true
|
||||
},
|
||||
"large": {
|
||||
"context": 192000,
|
||||
"model_alias": "qwen-large"
|
||||
"model_alias": "qwen-large",
|
||||
"vision": true
|
||||
},
|
||||
"ultra": {
|
||||
"context": 262144,
|
||||
"model_alias": "qwen-ultra"
|
||||
"model_alias": "qwen-ultra",
|
||||
"vision": false
|
||||
},
|
||||
"uncensored": {
|
||||
"context": 80000,
|
||||
"model_alias": "qwen-uncensored"
|
||||
"model_alias": "qwen-uncensored",
|
||||
"vision": true
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -39,6 +39,8 @@ def load_profile_registry(path: str | None) -> dict[str, dict]:
|
||||
"ultra": {"context": 262144, "model_alias": None},
|
||||
"uncensored": {"context": 80000, "model_alias": None},
|
||||
}
|
||||
for name, definition in fallback.items():
|
||||
definition["vision"] = name != "ultra"
|
||||
if not path:
|
||||
return fallback
|
||||
try:
|
||||
@@ -56,12 +58,16 @@ def load_profile_registry(path: str | None) -> dict[str, dict]:
|
||||
raise ConfigurationError(f"Profil {name!r} ist kein Objekt")
|
||||
context = definition.get("context")
|
||||
alias = definition.get("model_alias")
|
||||
vision = definition.get("vision", name != "ultra")
|
||||
if not isinstance(vision, bool):
|
||||
raise ConfigurationError(f"ungültige Vision-Fähigkeit für Profil {name!r}")
|
||||
if not isinstance(context, int) or context < 1024:
|
||||
raise ConfigurationError(f"ungültiger Kontext für Profil {name!r}")
|
||||
if alias is not None and (not isinstance(alias, str) or not alias.strip()):
|
||||
raise ConfigurationError(f"ungültiger Modellalias für Profil {name!r}")
|
||||
validated[name] = {"context": context,
|
||||
"model_alias": alias.strip() if alias else None}
|
||||
"model_alias": alias.strip() if alias else None,
|
||||
"vision": vision}
|
||||
return validated
|
||||
|
||||
|
||||
|
||||
Reference in New Issue
Block a user