fix: clamp aux reasoning effort once before profile projection
Follow-up to the cherry-picked #112019 (@KoNit-K): the clamp in the generic ``extra_body.reasoning`` fallback only covered providers WITHOUT a reasoning-aware profile. On the profile path (OpenRouter/Nous slots used as MoA aggregator or aux model) ``_project_provider_profile`` received the raw config and the OpenRouter profile passes ``ultra`` through whenever the catalog vocabulary is cold, so the 400 from #112010 survived there. Move the clamp up to ``_build_call_kwargs`` so both the profile projection and the fallback see a wire-level effort — the same entry clamp the main transport applies in ``_reasoning_config_for_model`` (#89503). The shared policy lives once in ``agent.reasoning_effort.clamp_reasoning_config``; the transport delegates to it instead of carrying its own copy. Offline kwargs probe (issue's exact call): before ``extra_body.reasoning == {'enabled': True, 'effort': 'ultra'}`` on nous and openrouter aux/MoA routes; after ``'effort': 'max'`` on every route, ``high`` verbatim and ``{'enabled': False}`` unchanged.
This commit is contained in:
@@ -6157,14 +6157,8 @@ def _merge_aux_extra_body(
|
||||
if reasoning_config.get("enabled") is False:
|
||||
merged_extra["reasoning"] = {"enabled": False}
|
||||
else:
|
||||
# This fallback uses the OpenAI-compatible chat-completions wire. Hermes'
|
||||
# internal ``ultra`` tier is not accepted there, including for MoA slots.
|
||||
from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort
|
||||
effort = reasoning_config.get("effort") or "medium"
|
||||
merged_extra["reasoning"] = {
|
||||
"enabled": True,
|
||||
"effort": clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS),
|
||||
}
|
||||
# ``reasoning_config`` is already clamped to the OpenAI-compat wire by _build_call_kwargs.
|
||||
merged_extra["reasoning"] = {"enabled": True, "effort": reasoning_config.get("effort") or "medium"}
|
||||
# Portal tags + sticky session_id fallback when the profile didn't supply them; session_id
|
||||
# keeps aux calls on the main turn's upstream instance (cache warmth) — tags alone are not
|
||||
# enough on /v1/messages.
|
||||
@@ -6209,7 +6203,11 @@ def _build_call_kwargs(
|
||||
kwargs["tools"] = _dedupe_tool_names(tools, provider, model)
|
||||
# Provider profiles are the source of truth for reasoning wire shapes (top-level, nested body,
|
||||
# or extra_body.reasoning); providers without a reasoning-aware profile keep the generic
|
||||
# ``extra_body.reasoning`` fallback.
|
||||
# ``extra_body.reasoning`` fallback. Clamp Hermes-internal levels (``ultra``) to the
|
||||
# OpenAI-compat wire ONCE here, before either path sees the config — the same entry clamp the
|
||||
# main transport applies (#89503); MoA aggregator/reference and aux calls 400'd without it (#112010).
|
||||
from agent.reasoning_effort import clamp_reasoning_config
|
||||
reasoning_config = clamp_reasoning_config(reasoning_config)
|
||||
projection = _project_provider_profile(provider, provider_norm, model, effective_base, reasoning_config)
|
||||
kwargs.update(projection.top_level)
|
||||
if merged_extra := _merge_aux_extra_body(extra_body, projection, reasoning_config, provider_norm):
|
||||
|
||||
@@ -156,6 +156,21 @@ def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]:
|
||||
return str(reasoning_config.get("effort") or "").strip().lower() or None
|
||||
|
||||
|
||||
def clamp_reasoning_config(reasoning_config: Optional[dict], supported: Sequence[str] = OPENAI_COMPAT_WIRE_EFFORTS) -> Optional[dict]:
|
||||
"""Return ``reasoning_config`` with its ``effort`` clamped onto ``supported`` (non-dicts and
|
||||
configs without an effort pass through untouched).
|
||||
|
||||
The entry clamp for an OpenAI-compatible chat-completions request builder: Hermes-internal
|
||||
``ultra`` never reaches a wire (#89503 main transport, #112010 aux/MoA), while provider
|
||||
profiles with narrower vocabularies clamp again downstream. Unset stays unset.
|
||||
"""
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return reasoning_config
|
||||
effort = str(reasoning_config.get("effort") or "").strip().lower()
|
||||
clamped = clamp_effort(effort, supported) if effort else effort
|
||||
return {**reasoning_config, "effort": clamped} if clamped != effort else reasoning_config
|
||||
|
||||
|
||||
def thinking_toggle_extras(
|
||||
reasoning_config: Optional[dict],
|
||||
efforts: Sequence[str],
|
||||
|
||||
@@ -11,7 +11,7 @@ from urllib.parse import urlparse
|
||||
from agent.lmstudio_reasoning import resolve_lmstudio_effort
|
||||
from agent.reasoning_effort import (
|
||||
KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, OPENAI_COMPAT_WIRE_EFFORTS, TOKENHUB_EFFORTS, clamp_effort,
|
||||
kimi_supported_efforts, requested_effort,
|
||||
clamp_reasoning_config, kimi_supported_efforts, requested_effort,
|
||||
)
|
||||
from agent.message_sanitization import normalize_finish_reason as _normalize_finish_reason
|
||||
from agent.moonshot_schema import is_moonshot_model, sanitize_moonshot_tools
|
||||
@@ -119,11 +119,7 @@ def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> di
|
||||
the declared wire vocabulary via the shared policy in ``agent.reasoning_effort``; provider profiles with
|
||||
narrower sets clamp again downstream.
|
||||
"""
|
||||
if not isinstance(reasoning_config, dict):
|
||||
return reasoning_config
|
||||
effort = str(reasoning_config.get("effort") or "").strip().lower()
|
||||
clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS) if effort else effort
|
||||
return {**reasoning_config, "effort": clamped} if clamped != effort else reasoning_config
|
||||
return clamp_reasoning_config(reasoning_config, OPENAI_COMPAT_WIRE_EFFORTS)
|
||||
|
||||
|
||||
def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> dict | None:
|
||||
|
||||
@@ -2519,6 +2519,31 @@ class TestAuxiliaryTaskExtraBody:
|
||||
|
||||
assert kwargs["extra_body"]["reasoning"] == {"enabled": True, "effort": "max"}
|
||||
|
||||
def test_profile_projection_receives_wire_clamped_effort(self, monkeypatch):
|
||||
"""Profiles clamp only against their own narrower sets (or a catalog that may be cold), so
|
||||
``ultra`` must already be a wire level when the projection sees it — the MoA aggregator on
|
||||
an OpenRouter/Nous slot 400'd otherwise (#112010)."""
|
||||
import agent.auxiliary_client as aux
|
||||
|
||||
seen = {}
|
||||
real = aux._project_provider_profile
|
||||
|
||||
def spy(provider, provider_norm, model, effective_base, reasoning_config):
|
||||
seen["config"] = reasoning_config
|
||||
return real(provider, provider_norm, model, effective_base, reasoning_config)
|
||||
|
||||
monkeypatch.setattr(aux, "_project_provider_profile", spy)
|
||||
kwargs = aux._build_call_kwargs(
|
||||
provider="openrouter",
|
||||
model="deepseek/deepseek-v4.1-flash",
|
||||
messages=[{"role": "user", "content": "hello"}],
|
||||
reasoning_config={"enabled": True, "effort": "ultra"},
|
||||
task="moa_aggregator",
|
||||
)
|
||||
|
||||
assert seen["config"] == {"enabled": True, "effort": "max"}
|
||||
assert "ultra" not in json.dumps(kwargs.get("extra_body")) and kwargs.get("reasoning_effort") != "ultra"
|
||||
|
||||
def test_sync_call_merges_task_extra_body_from_config(self):
|
||||
client = MagicMock()
|
||||
client.base_url = "https://api.example.com/v1"
|
||||
|
||||
Reference in New Issue
Block a user