refactor(model-providers): thinking-toggle XOR effort translation lives in agent.reasoning_effort; opencode-free imports it instead of borrowing via sys.modules
Four chat_completions profiles (kimi-coding, deepseek, opencode-go's Kimi K2 and DeepSeek branches, actual) each hand-rolled the same extra_body.thinking / top-level reasoning_effort translation, and the copies had already drifted in small ways (kimi's `.get("enabled", True)`, deepseek's separate effort parsing). agent.reasoning_effort.thinking_toggle_extras is now the single implementation: the Moonshot default emits effort XOR toggle (both is an HTTP 400), and always_emit_toggle=True covers DeepSeek's contract where the toggle must ride on every request to dodge the reasoning_content echo trap. actual keeps its two contract-specific lines (reasoning_config None -> nothing; effort "none" -> disabled toggle plus reasoning_effort="none", which the relay accepts as a real level) and delegates the rest. ox_alpha_reasoning_extras moves alongside so opencode-free imports it like any other helper instead of reaching into the zen plugin's module through sys.modules and swallowing every exception into ({}, {}) - a failure there previously silently dropped the user's effort setting. No wire behavior changes; tests/plugins/model_providers/test_thinking_toggle_parity.py pins the XOR invariant across the matrix and zen/free parity.
This commit is contained in:
@@ -156,6 +156,41 @@ def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]:
|
||||
return str(reasoning_config.get("effort") or "").strip().lower() or None
|
||||
|
||||
|
||||
def thinking_toggle_extras(
|
||||
reasoning_config: Optional[dict],
|
||||
efforts: Sequence[str],
|
||||
overrides: Optional[dict[str, str]] = None,
|
||||
*,
|
||||
always_emit_toggle: bool = False,
|
||||
) -> tuple[dict, dict]:
|
||||
"""Translate a reasoning config onto the Moonshot/DeepSeek chat_completions wire:
|
||||
``extra_body.thinking`` toggle and top-level ``reasoning_effort``.
|
||||
|
||||
Moonshot 400s when both are sent, so by default the effort (when it lands in
|
||||
``efforts``) replaces the toggle. DeepSeek instead requires the toggle on every
|
||||
request (an omitted toggle defaults thinking on and then demands
|
||||
``reasoning_content`` echoes), hence ``always_emit_toggle``. A requested effort of
|
||||
``none`` is not a level on these wires; it falls back to the plain toggle.
|
||||
"""
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
|
||||
return {"thinking": {"type": "disabled"}}, {}
|
||||
effort = requested_effort(reasoning_config)
|
||||
clamped = clamp_effort(None if effort == "none" else effort, efforts, overrides)
|
||||
if clamped in efforts:
|
||||
return ({"thinking": {"type": "enabled"}} if always_emit_toggle else {}), {"reasoning_effort": clamped}
|
||||
return {"thinking": {"type": "enabled"}}, {}
|
||||
|
||||
|
||||
def ox_alpha_reasoning_extras(reasoning_config: Optional[dict], model: Optional[str]) -> tuple[dict, dict]:
|
||||
"""Ox Alpha (``x-preview-f-free``) ``reasoning_effort`` translation, shared by the
|
||||
opencode-zen and opencode-free profiles (low/high/max only; anything else 400s)."""
|
||||
if (model or "").strip().rsplit("/", 1)[-1].lower() != "x-preview-f-free":
|
||||
return {}, {}
|
||||
effort = requested_effort(reasoning_config)
|
||||
clamped = clamp_effort(None if effort == "none" else effort, OX_ALPHA_EFFORTS, OX_ALPHA_OVERRIDES)
|
||||
return ({}, {"reasoning_effort": clamped}) if clamped in OX_ALPHA_EFFORTS else ({}, {})
|
||||
|
||||
|
||||
# ---- BEGIN PLUGIN-COMPAT (revert-scheduled; see COMPAT_MANIFEST.md) ----
|
||||
# Names external plugins imported from this module before the Sep 2026 decomposition.
|
||||
# Internal code MUST NOT use these (scripts/check_compat_pointers.py fails CI if it does).
|
||||
|
||||
@@ -47,20 +47,15 @@ class ActualProfile(ProviderProfile):
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return {}, {}
|
||||
from agent.reasoning_effort import clamp_effort, requested_effort
|
||||
from agent.reasoning_effort import thinking_toggle_extras
|
||||
|
||||
enabled = reasoning_config.get("enabled") is not False
|
||||
if str(reasoning_config.get("effort") or "").strip().lower() == "none":
|
||||
enabled = False
|
||||
extra_body = {"thinking": {"type": "enabled" if enabled else "disabled"}}
|
||||
top_level: dict[str, Any] = {}
|
||||
effort = requested_effort(reasoning_config)
|
||||
if effort is not None:
|
||||
supported = self.supported_reasoning_efforts(context.get("model"))
|
||||
clamped = clamp_effort(effort, supported)
|
||||
if clamped in (supported or ()):
|
||||
top_level["reasoning_effort"] = clamped
|
||||
return extra_body, top_level
|
||||
# The relay accepts ``none`` as a real effort level: it switches thinking off AND
|
||||
# is echoed as reasoning_effort, unlike the Moonshot/DeepSeek wires.
|
||||
effort_none = str(reasoning_config.get("effort") or "").strip().lower() == "none"
|
||||
if reasoning_config.get("enabled") is not False and effort_none:
|
||||
return {"thinking": {"type": "disabled"}}, {"reasoning_effort": "none"}
|
||||
supported = self.supported_reasoning_efforts(context.get("model")) or ()
|
||||
return thinking_toggle_extras(reasoning_config, supported, always_emit_toggle=True)
|
||||
|
||||
def fetch_models(
|
||||
self,
|
||||
|
||||
@@ -10,7 +10,7 @@ Retired ``deepseek-chat``/``deepseek-reasoner`` IDs are remapped in
|
||||
|
||||
from typing import Any
|
||||
|
||||
from agent.reasoning_effort import DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, clamp_effort
|
||||
from agent.reasoning_effort import DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, thinking_toggle_extras
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
|
||||
@@ -36,19 +36,11 @@ class DeepSeekProfile(ProviderProfile):
|
||||
versioned_v4_plus = m.startswith("deepseek-v") and not m.startswith("deepseek-v3")
|
||||
if not versioned_v4_plus and m not in _THINKING_CAPABLE_IDS:
|
||||
return {}, {}
|
||||
rc = reasoning_config if isinstance(reasoning_config, dict) else None
|
||||
# Always set thinking explicitly (default enabled, matching the API default)
|
||||
# to avoid the reasoning_content echo trap on subsequent turns.
|
||||
if rc is not None and rc.get("enabled") is False:
|
||||
return {"thinking": {"type": "disabled"}}, {}
|
||||
top_level: dict[str, Any] = {}
|
||||
# No effort -> omit reasoning_effort so DeepSeek applies its server default.
|
||||
effort = (rc.get("effort") or "").strip().lower() if rc is not None else ""
|
||||
if effort and effort != "none":
|
||||
clamped = clamp_effort(effort, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES)
|
||||
if clamped in DEEPSEEK_V4_EFFORTS:
|
||||
top_level["reasoning_effort"] = clamped
|
||||
return {"thinking": {"type": "enabled"}}, top_level
|
||||
return thinking_toggle_extras(
|
||||
reasoning_config, DEEPSEEK_V4_EFFORTS, DEEPSEEK_V4_OVERRIDES, always_emit_toggle=True
|
||||
)
|
||||
|
||||
|
||||
deepseek = DeepSeekProfile(
|
||||
|
||||
@@ -4,7 +4,7 @@ redirected to api.kimi.com/coding by core)."""
|
||||
from typing import Any
|
||||
from urllib.parse import urlparse
|
||||
|
||||
from agent.reasoning_effort import KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, clamp_effort, requested_effort
|
||||
from agent.reasoning_effort import KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, thinking_toggle_extras
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
from providers.base import OMIT_TEMPERATURE, ProviderProfile
|
||||
@@ -52,13 +52,7 @@ class KimiProfile(ProviderProfile):
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Moonshot treats extra_body.thinking and reasoning_effort as mutually
|
||||
exclusive (400 on both): send effort when requested, else the toggle."""
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled", True) is False:
|
||||
return {"thinking": {"type": "disabled"}}, {}
|
||||
effort = requested_effort(reasoning_config)
|
||||
k3_effort = clamp_effort(effort, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES) if effort != "none" else None
|
||||
if k3_effort in KIMI_K3_EFFORTS:
|
||||
return {}, {"reasoning_effort": k3_effort}
|
||||
return {"thinking": {"type": "enabled"}}, {}
|
||||
return thinking_toggle_extras(reasoning_config, KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES)
|
||||
|
||||
|
||||
def _kimi(name: str, aliases: tuple, env_vars: tuple, base_url: str) -> KimiProfile:
|
||||
|
||||
@@ -8,6 +8,7 @@ hermes_cli.models.opencode_zen_free_runtime). Select via ``/model free``.
|
||||
|
||||
from typing import Any
|
||||
|
||||
from agent.reasoning_effort import ox_alpha_reasoning_extras
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from providers import register_provider
|
||||
from providers.base import ProviderProfile
|
||||
@@ -17,22 +18,13 @@ class OpenCodeFreeProfile(ProviderProfile):
|
||||
"""OpenCode Free — keyless, with Ox Alpha reasoning controls.
|
||||
|
||||
Ox Alpha (x-preview-f-free) is also reachable via opencode-zen with the same wire
|
||||
contract; the translation lives in the zen plugin and is resolved through the
|
||||
registered zen profile's module so the two providers can never drift.
|
||||
contract; both profiles call ``agent.reasoning_effort.ox_alpha_reasoning_extras``.
|
||||
"""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
try:
|
||||
import sys
|
||||
|
||||
from providers import get_provider_profile
|
||||
|
||||
zen_module = sys.modules[type(get_provider_profile("opencode-zen")).__module__]
|
||||
return zen_module._build_ox_alpha_reasoning_extras(reasoning_config, model)
|
||||
except Exception:
|
||||
return {}, {}
|
||||
return ox_alpha_reasoning_extras(reasoning_config, model)
|
||||
|
||||
|
||||
opencode_free = OpenCodeFreeProfile(
|
||||
|
||||
@@ -42,25 +42,6 @@ def _is_glm_5_2_model(model: str | None) -> bool:
|
||||
return any(token in m for token in ("glm-5.2", "glm-5-2", "glm-5p2"))
|
||||
|
||||
|
||||
def _requested_effort(reasoning_config: dict | None) -> str | None:
|
||||
"""Normalized effort when reasoning is enabled and an effort is set, else None."""
|
||||
effort = re_.requested_effort(reasoning_config)
|
||||
return None if effort == "none" else effort
|
||||
|
||||
|
||||
def _thinking_toggle_extras(
|
||||
reasoning_config: dict | None, efforts, overrides=None
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Moonshot/DeepSeek wire shape: extra_body.thinking XOR top-level reasoning_effort
|
||||
(sending both is an HTTP 400)."""
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
|
||||
return {"thinking": {"type": "disabled"}}, {}
|
||||
clamped = re_.clamp_effort(_requested_effort(reasoning_config), efforts, overrides)
|
||||
if clamped in efforts:
|
||||
return {}, {"reasoning_effort": clamped}
|
||||
return {"thinking": {"type": "enabled"}}, {}
|
||||
|
||||
|
||||
class OpenCodeGoProfile(ProviderProfile):
|
||||
"""OpenCode Go - model-specific reasoning controls."""
|
||||
|
||||
@@ -77,38 +58,27 @@ class OpenCodeGoProfile(ProviderProfile):
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
if _is_glm_5_2_model(model):
|
||||
# Native reasoning_effort knob (high/max); server default when unset/disabled.
|
||||
effort = _requested_effort(reasoning_config)
|
||||
if effort is None:
|
||||
effort = re_.requested_effort(reasoning_config)
|
||||
if effort is None or effort == "none":
|
||||
return {}, {}
|
||||
clamped = re_.clamp_effort(effort, re_.GLM52_EFFORTS, re_.GLM52_OVERRIDES)
|
||||
return {}, {"reasoning_effort": clamped if clamped in re_.GLM52_EFFORTS else "high"}
|
||||
if _flat_model_name(model).startswith("kimi-k2"):
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return {}, {}
|
||||
return _thinking_toggle_extras(reasoning_config, re_.KIMI_K2_EFFORTS)
|
||||
return re_.thinking_toggle_extras(reasoning_config, re_.KIMI_K2_EFFORTS)
|
||||
if _is_deepseek_thinking_model(model):
|
||||
return _thinking_toggle_extras(reasoning_config, re_.DEEPSEEK_V4_EFFORTS, re_.DEEPSEEK_V4_OVERRIDES)
|
||||
return re_.thinking_toggle_extras(reasoning_config, re_.DEEPSEEK_V4_EFFORTS, re_.DEEPSEEK_V4_OVERRIDES)
|
||||
return {}, {}
|
||||
|
||||
|
||||
def _build_ox_alpha_reasoning_extras(
|
||||
reasoning_config: dict | None, model: str | None
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
"""Ox Alpha (x-preview-f-free) reasoning_effort translation, shared with the
|
||||
opencode-free profile (low/high/max only; anything else 400s)."""
|
||||
if _flat_model_name(model) != "x-preview-f-free":
|
||||
return {}, {}
|
||||
clamped = re_.clamp_effort(_requested_effort(reasoning_config), re_.OX_ALPHA_EFFORTS, re_.OX_ALPHA_OVERRIDES)
|
||||
return ({}, {"reasoning_effort": clamped}) if clamped in re_.OX_ALPHA_EFFORTS else ({}, {})
|
||||
|
||||
|
||||
class OpenCodeZenProfile(ProviderProfile):
|
||||
"""OpenCode Zen - model-specific reasoning controls."""
|
||||
|
||||
def build_api_kwargs_extras(
|
||||
self, *, reasoning_config: dict | None = None, model: str | None = None, **context
|
||||
) -> tuple[dict[str, Any], dict[str, Any]]:
|
||||
return _build_ox_alpha_reasoning_extras(reasoning_config, model)
|
||||
return re_.ox_alpha_reasoning_extras(reasoning_config, model)
|
||||
|
||||
|
||||
opencode_zen = OpenCodeZenProfile(
|
||||
|
||||
@@ -0,0 +1,45 @@
|
||||
"""Thinking-toggle / reasoning_effort wire invariants shared by the Moonshot- and
|
||||
DeepSeek-style chat_completions profiles (all route through
|
||||
``agent.reasoning_effort.thinking_toggle_extras``)."""
|
||||
|
||||
import pytest
|
||||
|
||||
from agent.reasoning_effort import DEEPSEEK_V4_EFFORTS
|
||||
from providers import get_provider_profile
|
||||
|
||||
REASONING_MATRIX = (
|
||||
None,
|
||||
{"enabled": False},
|
||||
{"enabled": True},
|
||||
*({"enabled": True, "effort": e} for e in ("low", "medium", "high", "xhigh", "max", "none")),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize("reasoning_config", REASONING_MATRIX, ids=str)
|
||||
def test_thinking_toggle_and_effort_never_both_on_moonshot_wire(reasoning_config):
|
||||
for provider, model in (("kimi-coding", "kimi-k3"), ("opencode-go", "kimi-k2.6"), ("opencode-go", "deepseek-v4-pro")):
|
||||
extra_body, top_level = get_provider_profile(provider).build_api_kwargs_extras(
|
||||
reasoning_config=reasoning_config, model=model
|
||||
)
|
||||
assert not ("thinking" in extra_body and "reasoning_effort" in top_level), (provider, model, reasoning_config)
|
||||
|
||||
# DeepSeek's own API wants the toggle on every request (omitting it defaults thinking on
|
||||
# and then demands reasoning_content echoes); effort rides alongside only when supported.
|
||||
extra_body, top_level = get_provider_profile("deepseek").build_api_kwargs_extras(
|
||||
reasoning_config=reasoning_config, model="deepseek-v4-pro"
|
||||
)
|
||||
assert extra_body["thinking"]["type"] in ("enabled", "disabled")
|
||||
assert top_level.get("reasoning_effort", DEEPSEEK_V4_EFFORTS[0]) in DEEPSEEK_V4_EFFORTS
|
||||
if isinstance(reasoning_config, dict) and reasoning_config.get("enabled") is False:
|
||||
assert (extra_body, top_level) == ({"thinking": {"type": "disabled"}}, {})
|
||||
|
||||
|
||||
@pytest.mark.parametrize("reasoning_config", REASONING_MATRIX, ids=str)
|
||||
def test_ox_alpha_translation_identical_on_zen_and_free(reasoning_config):
|
||||
zen = get_provider_profile("opencode-zen").build_api_kwargs_extras(
|
||||
reasoning_config=reasoning_config, model="x-preview-f-free"
|
||||
)
|
||||
free = get_provider_profile("opencode-free").build_api_kwargs_extras(
|
||||
reasoning_config=reasoning_config, model="x-preview-f-free"
|
||||
)
|
||||
assert zen == free
|
||||
Reference in New Issue
Block a user