Update llama runtime and isolate background reviews
This commit is contained in:
+65
-14
@@ -94,6 +94,9 @@ from router_support import (
|
||||
HOST = os.environ.get("ROUTER_HOST", "0.0.0.0")
|
||||
PORT = int(os.environ.get("ROUTER_PORT", "8081"))
|
||||
UPSTREAM_URL = os.environ.get("UPSTREAM_URL", "http://127.0.0.1:8080").rstrip("/")
|
||||
REVIEW_UPSTREAM_URL = os.environ.get("REVIEW_UPSTREAM_URL", "").rstrip("/")
|
||||
REVIEW_MODEL_NAME = os.environ.get("REVIEW_MODEL_NAME", "qwen-review").strip()
|
||||
REVIEW_CONTEXT_LENGTH = int(os.environ.get("REVIEW_CONTEXT_LENGTH", "32768"))
|
||||
PROFILE_SCRIPT = os.environ.get("PROFILE_SCRIPT", "/usr/local/bin/llama-profile")
|
||||
PROFILE_DIR = os.environ.get(
|
||||
"PROFILE_DIR", "/etc/systemd/system/mike-ai-llama-ui.service.d")
|
||||
@@ -225,6 +228,11 @@ def _parse_upstream(url: str) -> tuple[str, int]:
|
||||
|
||||
|
||||
UPSTREAM_HOST, UPSTREAM_PORT = _parse_upstream(UPSTREAM_URL)
|
||||
if REVIEW_UPSTREAM_URL:
|
||||
REVIEW_UPSTREAM_HOST, REVIEW_UPSTREAM_PORT = _parse_upstream(
|
||||
REVIEW_UPSTREAM_URL)
|
||||
else:
|
||||
REVIEW_UPSTREAM_HOST, REVIEW_UPSTREAM_PORT = "", 0
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -1590,19 +1598,29 @@ class Handler(BaseHTTPRequestHandler):
|
||||
|
||||
@staticmethod
|
||||
def _models_payload() -> dict:
|
||||
models = [
|
||||
{
|
||||
"id": f"qwen-{name}",
|
||||
"object": "model",
|
||||
"created": 0,
|
||||
"owned_by": "ai-profile-router",
|
||||
"context_length": ctx,
|
||||
"context_window": ctx,
|
||||
}
|
||||
for name, ctx in PROFILES.items()
|
||||
]
|
||||
if REVIEW_UPSTREAM_URL:
|
||||
models.append({
|
||||
"id": REVIEW_MODEL_NAME,
|
||||
"object": "model",
|
||||
"created": 0,
|
||||
"owned_by": "ai-profile-router",
|
||||
"context_length": REVIEW_CONTEXT_LENGTH,
|
||||
"context_window": REVIEW_CONTEXT_LENGTH,
|
||||
})
|
||||
return {
|
||||
"object": "list",
|
||||
"data": [
|
||||
{
|
||||
"id": f"qwen-{name}",
|
||||
"object": "model",
|
||||
"created": 0,
|
||||
"owned_by": "ai-profile-router",
|
||||
"context_length": ctx,
|
||||
"context_window": ctx,
|
||||
}
|
||||
for name, ctx in PROFILES.items()
|
||||
],
|
||||
"data": models,
|
||||
}
|
||||
|
||||
def _status_payload(self) -> dict:
|
||||
@@ -2165,6 +2183,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
|
||||
data = None
|
||||
requested_profile: str | None = None
|
||||
requested_review = False
|
||||
# Virtuelles Modell erkennen. Umschalten und Chat-Lease werden weiter
|
||||
# unten atomar unter dem zentralen Orchestrierungs-Lock ausgeführt.
|
||||
if body is not None and self.path.startswith("/v1/"):
|
||||
@@ -2173,7 +2192,10 @@ class Handler(BaseHTTPRequestHandler):
|
||||
except ValueError:
|
||||
data = None
|
||||
model = data.get("model") if isinstance(data, dict) else None
|
||||
if isinstance(model, str) and model in VIRTUAL_MODELS:
|
||||
if (isinstance(model, str) and REVIEW_UPSTREAM_URL
|
||||
and model == REVIEW_MODEL_NAME):
|
||||
requested_review = True
|
||||
elif isinstance(model, str) and model in VIRTUAL_MODELS:
|
||||
requested_profile = VIRTUAL_MODELS[model]
|
||||
elif isinstance(model, str) and model.startswith("qwen-"):
|
||||
# qwen-* ist der Namensraum des Routers
|
||||
@@ -2185,6 +2207,9 @@ class Handler(BaseHTTPRequestHandler):
|
||||
# kann kein zweiter Client zwischen Profilwahl und Upstream-Request das
|
||||
# Modell austauschen. Vision-Vorbereitung gehört zur selben Transaktion.
|
||||
if isinstance(data, dict) and path == "/v1/chat/completions":
|
||||
if requested_review:
|
||||
self._review_chat_proxy(data)
|
||||
return
|
||||
self._chat_proxy(body, data, requested_profile)
|
||||
return
|
||||
if requested_profile is not None:
|
||||
@@ -2262,6 +2287,28 @@ class Handler(BaseHTTPRequestHandler):
|
||||
if lease_acquired:
|
||||
self._release_model_lease()
|
||||
|
||||
def _review_chat_proxy(self, data: dict) -> None:
|
||||
"""Leitet kompakte Hermes-Hintergrundreviews an das Hilfsmodell.
|
||||
|
||||
Absichtlich ohne Hauptmodell-Lease und Profilwechsel: Der Review darf
|
||||
den aktiven Qwen-Chat weder anhalten noch dessen Prompt-Cache ersetzen.
|
||||
"""
|
||||
try:
|
||||
data = _normalize_llamacpp_reasoning(data)
|
||||
data = _cap_chat_generation(data)
|
||||
if _request_has_image(data):
|
||||
raise ValueError("qwen-review unterstützt keine Bilder")
|
||||
data["model"] = REVIEW_MODEL_NAME
|
||||
self._proxy_to(
|
||||
json.dumps(data).encode(),
|
||||
REVIEW_UPSTREAM_HOST,
|
||||
REVIEW_UPSTREAM_PORT,
|
||||
"Review-Modell",
|
||||
)
|
||||
except ValueError as exc:
|
||||
self._send_error(400, str(exc), "invalid_request_error",
|
||||
"invalid_review_request")
|
||||
|
||||
def _proxy_with_wait(self, body: bytes | None) -> None:
|
||||
"""Leitet an llama.cpp weiter, wartet aber erst, bis Qwen verfügbar ist.
|
||||
|
||||
@@ -2295,9 +2342,13 @@ class Handler(BaseHTTPRequestHandler):
|
||||
STATE.active_chats -= 1
|
||||
|
||||
def _proxy(self, body: bytes | None) -> None:
|
||||
self._proxy_to(body, UPSTREAM_HOST, UPSTREAM_PORT, "llama.cpp")
|
||||
|
||||
def _proxy_to(self, body: bytes | None, host: str, port: int,
|
||||
upstream_name: str) -> None:
|
||||
# An llama.cpp weiterleiten (Streaming bleibt erhalten).
|
||||
try:
|
||||
conn = http.client.HTTPConnection(UPSTREAM_HOST, UPSTREAM_PORT,
|
||||
conn = http.client.HTTPConnection(host, port,
|
||||
timeout=CONNECT_TIMEOUT)
|
||||
conn.connect()
|
||||
conn.sock.settimeout(REQUEST_TIMEOUT)
|
||||
@@ -2306,7 +2357,7 @@ class Handler(BaseHTTPRequestHandler):
|
||||
conn.request(self.command, self.path, body=body, headers=headers)
|
||||
resp = conn.getresponse()
|
||||
except (OSError, http.client.HTTPException) as e:
|
||||
self._send_error(502, f"llama.cpp nicht erreichbar: {e}",
|
||||
self._send_error(502, f"{upstream_name} nicht erreichbar: {e}",
|
||||
"server_error", "upstream_unavailable")
|
||||
return
|
||||
|
||||
|
||||
Reference in New Issue
Block a user