From 31f0336da7136dc2c4c3394ae1ae23eeab51941c Mon Sep 17 00:00:00 2001 From: xxxigm Date: Sat, 29 Aug 2026 22:30:36 +0700 Subject: [PATCH] fix(custom): omit Ollama-only think=false on strict OpenAI-compat endpoints reasoning_effort: none was injecting extra_body.think=false for every custom provider. Mistral (and other extra=forbid hosts) reject that field with HTTP 422. Keep think=false on Ollama URLs only; still send top-level reasoning_effort=none so /v1 thinking-off keeps working. --- plugins/model-providers/custom/__init__.py | 43 ++++++++++++++++++---- 1 file changed, 35 insertions(+), 8 deletions(-) diff --git a/plugins/model-providers/custom/__init__.py b/plugins/model-providers/custom/__init__.py index e3f02bc822..42b974d308 100644 --- a/plugins/model-providers/custom/__init__.py +++ b/plugins/model-providers/custom/__init__.py @@ -6,18 +6,41 @@ Volcengine ARK, vLLM, llama.cpp). Key quirks: - ollama_num_ctx → extra_body.options.num_ctx (local context window) - reasoning_config disabled → top-level reasoning_effort="none" (Ollama /v1/chat/completions ignores think=False — ollama#14820) - + extra_body.think = False for /api/chat and proxies + + extra_body.think = False only on Ollama URLs (/api/chat and proxies) - reasoning_config enabled + effort → top-level reasoning_effort (the native OpenAI-compatible format GLM/ARK expect; unset omits it so the endpoint's server default applies) """ from typing import Any +from urllib.parse import urlparse from providers import register_provider from providers.base import ProviderProfile +def _looks_like_ollama_endpoint(base_url: str | None) -> bool: + """True when ``base_url`` is an Ollama host, not a generic OpenAI-compat relay. + + ``think`` is an Ollama-native extra_body field. Strict hosts (Mistral + ``extra=forbid``, Groq, …) reject it with HTTP 422. Match only explicit + Ollama signatures — default port 11434, or ``ollama`` as a hostname + label — not arbitrary localhost (llama.cpp / vLLM / LM Studio). + """ + raw = (base_url or "").strip() + if not raw: + return False + parsed = urlparse(raw if "://" in raw else f"//{raw}") + if parsed.port == 11434: + return True + host = (parsed.hostname or "").lower().rstrip(".") + if not host: + return False + if host == "ollama.com" or host.endswith(".ollama.com"): + return True + return "ollama" in host.split(".") + + class CustomProfile(ProviderProfile): """Custom/Ollama local provider — think=false and num_ctx support.""" @@ -40,7 +63,8 @@ class CustomProfile(ProviderProfile): # Reasoning / thinking control for custom OpenAI-compatible endpoints # (GLM-5.2 on Volcengine ARK, vLLM, Ollama, llama.cpp, …). # - # - disabled → extra_body.think = False (Ollama's thinking-off flag) + # - disabled → top-level reasoning_effort="none"; extra_body.think + # = False only on Ollama URLs (Ollama's thinking-off flag) # - enabled + effort set → TOP-LEVEL reasoning_effort string, the # format GLM-5.2/ARK and other OpenAI-compatible reasoning APIs # expect (GLM documents "high" and "max"; "max" is its default). @@ -50,19 +74,22 @@ class CustomProfile(ProviderProfile): # We deliberately do NOT emit ``think=True`` on enable: it is an # Ollama-only flag and thinking is already server-default-on for these # backends, so forcing it risks a 400 on GLM/vLLM endpoints that don't - # recognize it. Mirrors the DeepSeek/Zai profile precedent. + # recognize it. Mirrors the DeepSeek/Zai profile precedent. The same + # constraint applies to ``think=False`` on disable — Mistral/Groq + # reject unknown fields (HTTP 422 extra_forbidden) rather than ignoring + # them, so that flag stays Ollama-URL-gated. if reasoning_config and isinstance(reasoning_config, dict): _effort = (reasoning_config.get("effort") or "").strip().lower() _enabled = reasoning_config.get("enabled", True) if _effort == "none" or _enabled is False: # Ollama's /v1/chat/completions silently ignores # extra_body.think (only /api/chat honours it — ollama#14820) - # but respects the top-level reasoning_effort field, so both - # are needed to actually stop a thinking-capable model from - # reasoning (#25758). Endpoints that recognize neither simply - # ignore them. + # but respects the top-level reasoning_effort field (#25758). + # Always emit reasoning_effort="none"; only add think=False + # when the URL is actually Ollama. top_level["reasoning_effort"] = "none" - extra_body["think"] = False + if _looks_like_ollama_endpoint(ctx.get("base_url")): + extra_body["think"] = False elif _effort: # Clamp the internal ladder onto the widest OpenAI-compatible # wire vocabulary (shared policy in agent.reasoning_effort) —