diff --git a/plugins/model-providers/custom/__init__.py b/plugins/model-providers/custom/__init__.py index 2847d161adf0..de744d50914e 100644 --- a/plugins/model-providers/custom/__init__.py +++ b/plugins/model-providers/custom/__init__.py @@ -4,7 +4,9 @@ Covers any endpoint registered as provider="custom", including local Ollama instances and OpenAI-compatible reasoning endpoints (GLM-5.2 on Volcengine ARK, vLLM, llama.cpp). Key quirks: - ollama_num_ctx → extra_body.options.num_ctx (local context window) - - reasoning_config disabled → extra_body.think = False + - reasoning_config disabled → top-level reasoning_effort="none" + (Ollama /v1/chat/completions ignores think=False — ollama#14820) + + extra_body.think = False for /api/chat and proxies - reasoning_config enabled + effort → top-level reasoning_effort (the native OpenAI-compatible format GLM/ARK expect; unset omits it so the endpoint's server default applies) @@ -53,6 +55,13 @@ class CustomProfile(ProviderProfile): _effort = (reasoning_config.get("effort") or "").strip().lower() _enabled = reasoning_config.get("enabled", True) if _effort == "none" or _enabled is False: + # Ollama's /v1/chat/completions silently ignores + # extra_body.think (only /api/chat honours it — ollama#14820) + # but respects the top-level reasoning_effort field, so both + # are needed to actually stop a thinking-capable model from + # reasoning (#25758). Endpoints that recognize neither simply + # ignore them. + top_level["reasoning_effort"] = "none" extra_body["think"] = False elif _effort: top_level["reasoning_effort"] = _effort diff --git a/tests/plugins/model_providers/test_custom_profile.py b/tests/plugins/model_providers/test_custom_profile.py index e20ee57b8ceb..152359ed2f30 100644 --- a/tests/plugins/model_providers/test_custom_profile.py +++ b/tests/plugins/model_providers/test_custom_profile.py @@ -49,20 +49,26 @@ class TestCustomReasoningWireShape: assert tl == {} def test_disabled_sends_think_false(self, custom_profile): - """enabled=False → extra_body.think = False (Ollama thinking-off flag).""" + """enabled=False → reasoning_effort='none' top-level + think=False. + + Both fields are required: Ollama's /v1/chat/completions silently + ignores extra_body.think (only /api/chat honours it — ollama#14820) + but respects top-level reasoning_effort (#25758). think=False stays + for proxies and the native /api/chat path. + """ eb, tl = custom_profile.build_api_kwargs_extras( reasoning_config={"enabled": False}, model="glm-5.2" ) assert eb == {"think": False} - assert tl == {} + assert tl == {"reasoning_effort": "none"} def test_effort_none_sends_think_false(self, custom_profile): - """effort='none' is the disable alias → think=False, no effort.""" + """effort='none' is the disable alias → same dual emission.""" eb, tl = custom_profile.build_api_kwargs_extras( reasoning_config={"enabled": True, "effort": "none"}, model="glm-5.2" ) assert eb == {"think": False} - assert tl == {} + assert tl == {"reasoning_effort": "none"} @pytest.mark.parametrize( "effort", ["minimal", "low", "medium", "high", "xhigh", "max"]