Add final Qwen3.8 runtime and uncensored profile

This commit is contained in:
Mikei386
2026-08-23 00:10:53 +02:00
parent 1a9d94d22f
commit 2ef894160a
57 changed files with 10089 additions and 54 deletions
+20 -7
View File
@@ -2,7 +2,7 @@
"""AI Profile Router – OpenAI-kompatibler Proxy vor llama.cpp.
Leitet OpenAI-kompatible Requests transparent an den lokalen llama.cpp-Server
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen vier festen
weiter (Streaming, Tool Calls, JSON) und schaltet zwischen fünf festen
Profilen um:
Profil Kontext
@@ -11,9 +11,11 @@ Profilen um:
medium 160000
large 192000
ultra 262144
uncensored 80000
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra
Kommandos: POST /fast, /medium, /large, /ultra (Profilwechsel)
Virtuelle Modelle: qwen-fast, qwen-medium, qwen-large, qwen-ultra,
qwen-uncensored
Kommandos: POST /fast, /medium, /large, /ultra, /uncensored
GET /status (Zustand)
Bildgenerierung (FLUX.2 [klein] 4B Base):
@@ -496,7 +498,7 @@ def _wait_ready(profile: str, deadline: float) -> None:
while True:
status = upstream_status()
if (status["reachable"] and status.get("model")
and status.get("ctx") == expected_ctx
and _context_matches(expected_ctx, status.get("ctx"))
and (not expected_model or status.get("model") == expected_model)):
log.info("llama.cpp bereit: Profil=%s Modell=%s ctx=%s",
profile, status.get("model"), status.get("ctx"))
@@ -514,11 +516,22 @@ def _profile_is_ready(profile: str, status: dict | None = None) -> bool:
status = status or upstream_status()
expected_model = EXPECTED_MODELS.get(profile)
return bool(status.get("reachable") and status.get("model")
and status.get("ctx") == PROFILES[profile]
and _context_matches(PROFILES[profile], status.get("ctx"))
and (not expected_model
or status.get("model") == expected_model))
def _context_matches(expected: int, reported: object) -> bool:
"""Allow llama.cpp's small MTP/speculative context overhead.
Recent builds can report a slot context slightly above the requested
``--ctx-size`` (currently 128 tokens with the tested Qwen MTP setup).
The alias check and this narrow bound still reject neighbouring profiles.
"""
return (isinstance(reported, int)
and expected <= reported <= expected + 1024)
def switch_profile(profile: str, implicit: bool = False) -> None:
"""Stellt sicher, dass das Profil aktiv ist, und wartet bis es geladen ist.
@@ -1184,11 +1197,11 @@ class Handler(BaseHTTPRequestHandler):
self._images_list()
elif path.startswith("/images/") and self.command == "GET":
self._image_serve(path[len("/images/"):])
elif (path in ("/fast", "/medium", "/large", "/ultra")
elif (path in ("/fast", "/medium", "/large", "/ultra", "/uncensored")
and (self.command == "POST"
or (self.command == "GET" and ALLOW_LEGACY_GET_SWITCH))):
self._switch(path[1:])
elif (path in ("/fast", "/medium", "/large", "/ultra")
elif (path in ("/fast", "/medium", "/large", "/ultra", "/uncensored")
and self.command == "GET"):
self._send_error(405, "Profilwechsel erfordert POST",
"invalid_request_error", "method_not_allowed")