Add final Qwen3.8 runtime and uncensored profile
This commit is contained in:
@@ -2,7 +2,7 @@
|
||||
"""AI Profile Router – OpenAI-kompatibler Proxy vor llama.cpp.
|
||||
|
||||
Leitet OpenAI-kompatible Requests transparent an den lokalen llama.cpp-Server
|
||||
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen vier festen
|
||||
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen fünf festen
|
||||
Profilen um:
|
||||
|
||||
Profil Kontext
|
||||
@@ -11,9 +11,11 @@ Profilen um:
|
||||
medium 160000
|
||||
large 192000
|
||||
ultra 262144
|
||||
uncensored 80000
|
||||
|
||||
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra
|
||||
Kommandos: POST /fast, /medium, /large, /ultra (Profilwechsel)
|
||||
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra,
|
||||
qwen-uncensored
|
||||
Kommandos: POST /fast, /medium, /large, /ultra, /uncensored
|
||||
GET /status (Zustand)
|
||||
|
||||
Bildgenerierung (FLUX.2 [klein] 4B Base):
|
||||
@@ -496,7 +498,7 @@ def _wait_ready(profile: str, deadline: float) -> None:
|
||||
while True:
|
||||
status = upstream_status()
|
||||
if (status["reachable"] and status.get("model")
|
||||
and status.get("ctx") == expected_ctx
|
||||
and _context_matches(expected_ctx, status.get("ctx"))
|
||||
and (not expected_model or status.get("model") == expected_model)):
|
||||
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
|
||||
profile, status.get("model"), status.get("ctx"))
|
||||
@@ -514,11 +516,22 @@ def _profile_is_ready(profile: str, status: dict | None = None) -> bool:
|
||||
status = status or upstream_status()
|
||||
expected_model = EXPECTED_MODELS.get(profile)
|
||||
return bool(status.get("reachable") and status.get("model")
|
||||
and status.get("ctx") == PROFILES[profile]
|
||||
and _context_matches(PROFILES[profile], status.get("ctx"))
|
||||
and (not expected_model
|
||||
or status.get("model") == expected_model))
|
||||
|
||||
|
||||
def _context_matches(expected: int, reported: object) -> bool:
|
||||
"""Allow llama.cpp's small MTP/speculative context overhead.
|
||||
|
||||
Recent builds can report a slot context slightly above the requested
|
||||
``--ctx-size`` (currently 128 tokens with the tested Qwen MTP setup).
|
||||
The alias check and this narrow bound still reject neighbouring profiles.
|
||||
"""
|
||||
return (isinstance(reported, int)
|
||||
and expected <= reported <= expected + 1024)
|
||||
|
||||
|
||||
def switch_profile(profile: str, implicit: bool = False) -> None:
|
||||
"""Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist.
|
||||
|
||||
@@ -1184,11 +1197,11 @@ class Handler(BaseHTTPRequestHandler):
|
||||
self._images_list()
|
||||
elif path.startswith("/images/") and self.command == "GET":
|
||||
self._image_serve(path[len("/images/"):])
|
||||
elif (path in ("/fast", "/medium", "/large", "/ultra")
|
||||
elif (path in ("/fast", "/medium", "/large", "/ultra", "/uncensored")
|
||||
and (self.command == "POST"
|
||||
or (self.command == "GET" and ALLOW_LEGACY_GET_SWITCH))):
|
||||
self._switch(path[1:])
|
||||
elif (path in ("/fast", "/medium", "/large", "/ultra")
|
||||
elif (path in ("/fast", "/medium", "/large", "/ultra", "/uncensored")
|
||||
and self.command == "GET"):
|
||||
self._send_error(405, "Profilwechsel erfordert POST",
|
||||
"invalid_request_error", "method_not_allowed")
|
||||
|
||||
Reference in New Issue
Block a user