diff --git a/.env.example b/.env.example index d6c9b28..38f57d1 100644 --- a/.env.example +++ b/.env.example @@ -11,7 +11,7 @@ XTTS_IMAGE=ghcr.io/coqui-ai/xtts-streaming-server:latest-cuda121@sha256:f7fb3b1f XTTS_CACHE_DIR=/data/models/xtts-v2-cache XTTS_GPU_DEVICE=GPU-4834d9d7-5b61-3004-1fb3-4ae49d482d4b AI_DNS=192.168.1.1 -DEFAULT_REASONING_EFFORT=medium +DEFAULT_REASONING_EFFORT=off FAST_MODEL_FILE=qwen-mix/Qwen3.8-27B-IQ4-MIX.gguf MEDIUM_MODEL_FILE=qwen-pure/qwen3.8-27b-IQ4_XS-pure.gguf diff --git a/compose.yaml b/compose.yaml index b633fb4..d412108 100644 --- a/compose.yaml +++ b/compose.yaml @@ -576,7 +576,7 @@ services: # request limit llama.cpp uses n_predict=-1 and a reasoning loop can # consume the complete context before yielding visible output. MAX_GENERATION_TOKENS: "8192" - DEFAULT_REASONING_EFFORT: "${DEFAULT_REASONING_EFFORT:-medium}" + DEFAULT_REASONING_EFFORT: "${DEFAULT_REASONING_EFFORT:-off}" IMAGE_DIR: /data/images IMAGE_WORKER_URL: http://image-worker:8086 IMAGE_WORKER_TOKEN: "${CONTROLLER_TOKEN:?CONTROLLER_TOKEN is required}" diff --git a/config/install.env.example b/config/install.env.example index da3221f..9054cc9 100644 --- a/config/install.env.example +++ b/config/install.env.example @@ -72,7 +72,7 @@ EXPERIMENTAL_MODEL_SHA256=40fac4050e940397dbf13087afd50f4734a11805bf9d65ef8ddd74 VISION_PROJECTOR_FILE=qwen/mmproj-BF16.gguf VISION_PROJECTOR_URL=https://huggingface.co/unsloth/Qwen3.8-27B-GGUF/resolve/main/mmproj-BF16.gguf VISION_PROJECTOR_SHA256=83ee4f4f205fa514161778c41df1ea14144faa0f713510893b63c2395f5c2d53 -DEFAULT_REASONING_EFFORT=medium +DEFAULT_REASONING_EFFORT=off # All standard profiles use the MTP tensor embedded in their GGUF. A separate # draft-model artifact is neither downloaded nor passed to llama-server. diff --git a/dev/test_router_support.py b/dev/test_router_support.py index 4074626..64f1c3a 100644 --- a/dev/test_router_support.py +++ b/dev/test_router_support.py @@ -129,14 +129,16 @@ class ChatImageInputTests(unittest.TestCase): class LlamaCppReasoningTests(unittest.TestCase): - def test_none_really_disables_thinking(self) -> None: - request = {"reasoning_effort": "none", "messages": []} - normalized = _normalize_llamacpp_reasoning(request) - self.assertNotIn("reasoning_effort", normalized) - self.assertEqual( - normalized["chat_template_kwargs"], - {"enable_thinking": False}, - ) + def test_disabled_values_really_disable_thinking(self) -> None: + for effort in (None, "none", "off", "disabled", False): + with self.subTest(effort=effort): + request = {"reasoning_effort": effort, "messages": []} + normalized = _normalize_llamacpp_reasoning(request) + self.assertNotIn("reasoning_effort", normalized) + self.assertEqual( + normalized["chat_template_kwargs"], + {"enable_thinking": False}, + ) def test_low_and_medium_reach_chat_template(self) -> None: for effort in ("low", "medium"): @@ -171,10 +173,13 @@ class LlamaCppReasoningTests(unittest.TestCase): "reasoning_effort": "low", }) - def test_request_without_effort_is_unchanged(self) -> None: + def test_request_without_effort_uses_safe_off_default(self) -> None: request = {"messages": []} self.assertIs(_normalize_llamacpp_reasoning(request), request) - self.assertNotIn("chat_template_kwargs", request) + self.assertEqual( + request["chat_template_kwargs"], + {"enable_thinking": False}, + ) class ChatGenerationLimitTests(unittest.TestCase): diff --git a/router/ai_profile_router.py b/router/ai_profile_router.py index 983ed3a..caece2d 100755 --- a/router/ai_profile_router.py +++ b/router/ai_profile_router.py @@ -117,7 +117,7 @@ CONNECT_TIMEOUT = float(os.environ.get("CONNECT_TIMEOUT", "10")) # s, Connect POLL_INTERVAL = float(os.environ.get("POLL_INTERVAL", "2")) # s, Polling-Intervall MAX_GENERATION_TOKENS = int(os.environ.get("MAX_GENERATION_TOKENS", "8192")) DEFAULT_REASONING_EFFORT = os.environ.get( - "DEFAULT_REASONING_EFFORT", "medium").strip().lower() + "DEFAULT_REASONING_EFFORT", "off").strip().lower() # --- Bildgenerierung und Referenzbild-Bearbeitung (FLUX.2 Klein 4B) --- LLAMA_SERVICE = os.environ.get("LLAMA_SERVICE", "mike-ai-llama-ui.service") @@ -1280,7 +1280,9 @@ def _normalize_llamacpp_reasoning(data: dict) -> dict: else: # A client may already speak llama.cpp's native template dialect. # Preserve that explicit choice; otherwise apply the platform-wide - # default so every OpenAI-compatible client behaves consistently. + # default. Hermes deliberately omits reasoning_effort when its UI is + # set to Off, so the safe default must remain Off; enabled levels are + # sent explicitly by Hermes and other capable clients. existing_kwargs = data.get("chat_template_kwargs") if isinstance(existing_kwargs, dict) and ( "reasoning_effort" in existing_kwargs