Fix Thinking Off handling
This commit is contained in:
@@ -117,7 +117,7 @@ CONNECT_TIMEOUT = float(os.environ.get("CONNECT_TIMEOUT", "10")) # s, Connect
|
||||
POLL_INTERVAL = float(os.environ.get("POLL_INTERVAL", "2")) # s, Polling-Intervall
|
||||
MAX_GENERATION_TOKENS = int(os.environ.get("MAX_GENERATION_TOKENS", "8192"))
|
||||
DEFAULT_REASONING_EFFORT = os.environ.get(
|
||||
"DEFAULT_REASONING_EFFORT", "medium").strip().lower()
|
||||
"DEFAULT_REASONING_EFFORT", "off").strip().lower()
|
||||
|
||||
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 4B) ---
|
||||
LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service")
|
||||
@@ -1280,7 +1280,9 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
else:
|
||||
# A client may already speak llama.cpp's native template dialect.
|
||||
# Preserve that explicit choice; otherwise apply the platform-wide
|
||||
# default so every OpenAI-compatible client behaves consistently.
|
||||
# default. Hermes deliberately omits reasoning_effort when its UI is
|
||||
# set to Off, so the safe default must remain Off; enabled levels are
|
||||
# sent explicitly by Hermes and other capable clients.
|
||||
existing_kwargs = data.get("chat_template_kwargs")
|
||||
if isinstance(existing_kwargs, dict) and (
|
||||
"reasoning_effort" in existing_kwargs
|
||||
|
||||
Reference in New Issue
Block a user