Default reasoning effort to medium
This commit is contained in:
1 parent
2661553e20
commit
db3719af7a
4 files changed
+18
-4
No files matched your search
@@ -116,6 +116,8 @@ REQUEST_TIMEOUT = float(os.environ.get("REQUEST_TIMEOUT", "600")) # s, Read-Ti
|
||||
CONNECT_TIMEOUT = float(os.environ.get("CONNECT_TIMEOUT", "10")) # s, Connect-Timeout
|
||||
POLL_INTERVAL = float(os.environ.get("POLL_INTERVAL", "2")) # s, Polling-Intervall
|
||||
MAX_GENERATION_TOKENS = int(os.environ.get("MAX_GENERATION_TOKENS", "8192"))
|
||||
DEFAULT_REASONING_EFFORT = os.environ.get(
|
||||
"DEFAULT_REASONING_EFFORT", "medium").strip().lower()
|
||||
|
||||
# --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 4B) ---
|
||||
LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service")
|
||||
@@ -1272,10 +1274,19 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict:
|
||||
entfernt das wirkungslose Top-Level-Feld. Andere Template-Argumente des
|
||||
Clients bleiben erhalten.
|
||||
"""
|
||||
if "reasoning_effort" not in data:
|
||||
return data
|
||||
|
||||
raw_effort = data.pop("reasoning_effort")
|
||||
explicit_effort = "reasoning_effort" in data
|
||||
if explicit_effort:
|
||||
raw_effort = data.pop("reasoning_effort")
|
||||
else:
|
||||
# A client may already speak llama.cpp's native template dialect.
|
||||
# Preserve that explicit choice; otherwise apply the platform-wide
|
||||
# default so every OpenAI-compatible client behaves consistently.
|
||||
existing_kwargs = data.get("chat_template_kwargs")
|
||||
if isinstance(existing_kwargs, dict) and (
|
||||
"reasoning_effort" in existing_kwargs
|
||||
or "enable_thinking" in existing_kwargs):
|
||||
return data
|
||||
raw_effort = DEFAULT_REASONING_EFFORT
|
||||
effort = str(raw_effort).strip().lower() if raw_effort is not None else ""
|
||||
template_kwargs = data.get("chat_template_kwargs")
|
||||
if not isinstance(template_kwargs, dict):
|
||||
|
||||
Reference in new issue
Block a user