From 9e45a90488a3bc3905dba9310fa4c1e72b3f3894 Mon Sep 17 00:00:00 2001 From: teknium1 <127238744+teknium1@users.noreply.github.com> Date: Tue, 15 Sep 2026 12:04:40 -0700 Subject: [PATCH] fix: clamp aux reasoning effort once before profile projection MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Follow-up to the cherry-picked #112019 (@KoNit-K): the clamp in the generic ``extra_body.reasoning`` fallback only covered providers WITHOUT a reasoning-aware profile. On the profile path (OpenRouter/Nous slots used as MoA aggregator or aux model) ``_project_provider_profile`` received the raw config and the OpenRouter profile passes ``ultra`` through whenever the catalog vocabulary is cold, so the 400 from #112010 survived there. Move the clamp up to ``_build_call_kwargs`` so both the profile projection and the fallback see a wire-level effort — the same entry clamp the main transport applies in ``_reasoning_config_for_model`` (#89503). The shared policy lives once in ``agent.reasoning_effort.clamp_reasoning_config``; the transport delegates to it instead of carrying its own copy. Offline kwargs probe (issue's exact call): before ``extra_body.reasoning == {'enabled': True, 'effort': 'ultra'}`` on nous and openrouter aux/MoA routes; after ``'effort': 'max'`` on every route, ``high`` verbatim and ``{'enabled': False}`` unchanged. --- agent/auxiliary_client.py | 16 +++++++--------- agent/reasoning_effort.py | 15 +++++++++++++++ agent/transports/chat_completions.py | 8 ++------ tests/agent/test_auxiliary_client.py | 25 +++++++++++++++++++++++++ 4 files changed, 49 insertions(+), 15 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 34966ae0fa..d483e3f4dd 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -6157,14 +6157,8 @@ def _merge_aux_extra_body( if reasoning_config.get("enabled") is False: merged_extra["reasoning"] = {"enabled": False} else: - # This fallback uses the OpenAI-compatible chat-completions wire. Hermes' - # internal ``ultra`` tier is not accepted there, including for MoA slots. - from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort - effort = reasoning_config.get("effort") or "medium" - merged_extra["reasoning"] = { - "enabled": True, - "effort": clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS), - } + # ``reasoning_config`` is already clamped to the OpenAI-compat wire by _build_call_kwargs. + merged_extra["reasoning"] = {"enabled": True, "effort": reasoning_config.get("effort") or "medium"} # Portal tags + sticky session_id fallback when the profile didn't supply them; session_id # keeps aux calls on the main turn's upstream instance (cache warmth) — tags alone are not # enough on /v1/messages. @@ -6209,7 +6203,11 @@ def _build_call_kwargs( kwargs["tools"] = _dedupe_tool_names(tools, provider, model) # Provider profiles are the source of truth for reasoning wire shapes (top-level, nested body, # or extra_body.reasoning); providers without a reasoning-aware profile keep the generic - # ``extra_body.reasoning`` fallback. + # ``extra_body.reasoning`` fallback. Clamp Hermes-internal levels (``ultra``) to the + # OpenAI-compat wire ONCE here, before either path sees the config — the same entry clamp the + # main transport applies (#89503); MoA aggregator/reference and aux calls 400'd without it (#112010). + from agent.reasoning_effort import clamp_reasoning_config + reasoning_config = clamp_reasoning_config(reasoning_config) projection = _project_provider_profile(provider, provider_norm, model, effective_base, reasoning_config) kwargs.update(projection.top_level) if merged_extra := _merge_aux_extra_body(extra_body, projection, reasoning_config, provider_norm): diff --git a/agent/reasoning_effort.py b/agent/reasoning_effort.py index 81472daa19..53e5393ce8 100644 --- a/agent/reasoning_effort.py +++ b/agent/reasoning_effort.py @@ -156,6 +156,21 @@ def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]: return str(reasoning_config.get("effort") or "").strip().lower() or None +def clamp_reasoning_config(reasoning_config: Optional[dict], supported: Sequence[str] = OPENAI_COMPAT_WIRE_EFFORTS) -> Optional[dict]: + """Return ``reasoning_config`` with its ``effort`` clamped onto ``supported`` (non-dicts and + configs without an effort pass through untouched). + + The entry clamp for an OpenAI-compatible chat-completions request builder: Hermes-internal + ``ultra`` never reaches a wire (#89503 main transport, #112010 aux/MoA), while provider + profiles with narrower vocabularies clamp again downstream. Unset stays unset. + """ + if not isinstance(reasoning_config, dict): + return reasoning_config + effort = str(reasoning_config.get("effort") or "").strip().lower() + clamped = clamp_effort(effort, supported) if effort else effort + return {**reasoning_config, "effort": clamped} if clamped != effort else reasoning_config + + def thinking_toggle_extras( reasoning_config: Optional[dict], efforts: Sequence[str], diff --git a/agent/transports/chat_completions.py b/agent/transports/chat_completions.py index de56b9fa8d..ec77a52587 100644 --- a/agent/transports/chat_completions.py +++ b/agent/transports/chat_completions.py @@ -11,7 +11,7 @@ from urllib.parse import urlparse from agent.lmstudio_reasoning import resolve_lmstudio_effort from agent.reasoning_effort import ( KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, OPENAI_COMPAT_WIRE_EFFORTS, TOKENHUB_EFFORTS, clamp_effort, - kimi_supported_efforts, requested_effort, + clamp_reasoning_config, kimi_supported_efforts, requested_effort, ) from agent.message_sanitization import normalize_finish_reason as _normalize_finish_reason from agent.moonshot_schema import is_moonshot_model, sanitize_moonshot_tools @@ -119,11 +119,7 @@ def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> di the declared wire vocabulary via the shared policy in ``agent.reasoning_effort``; provider profiles with narrower sets clamp again downstream. """ - if not isinstance(reasoning_config, dict): - return reasoning_config - effort = str(reasoning_config.get("effort") or "").strip().lower() - clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS) if effort else effort - return {**reasoning_config, "effort": clamped} if clamped != effort else reasoning_config + return clamp_reasoning_config(reasoning_config, OPENAI_COMPAT_WIRE_EFFORTS) def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> dict | None: diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index 3e51216365..cba10878d1 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -2519,6 +2519,31 @@ class TestAuxiliaryTaskExtraBody: assert kwargs["extra_body"]["reasoning"] == {"enabled": True, "effort": "max"} + def test_profile_projection_receives_wire_clamped_effort(self, monkeypatch): + """Profiles clamp only against their own narrower sets (or a catalog that may be cold), so + ``ultra`` must already be a wire level when the projection sees it — the MoA aggregator on + an OpenRouter/Nous slot 400'd otherwise (#112010).""" + import agent.auxiliary_client as aux + + seen = {} + real = aux._project_provider_profile + + def spy(provider, provider_norm, model, effective_base, reasoning_config): + seen["config"] = reasoning_config + return real(provider, provider_norm, model, effective_base, reasoning_config) + + monkeypatch.setattr(aux, "_project_provider_profile", spy) + kwargs = aux._build_call_kwargs( + provider="openrouter", + model="deepseek/deepseek-v4.1-flash", + messages=[{"role": "user", "content": "hello"}], + reasoning_config={"enabled": True, "effort": "ultra"}, + task="moa_aggregator", + ) + + assert seen["config"] == {"enabled": True, "effort": "max"} + assert "ultra" not in json.dumps(kwargs.get("extra_body")) and kwargs.get("reasoning_effort") != "ultra" + def test_sync_call_merges_task_extra_body_from_config(self): client = MagicMock() client.base_url = "https://api.example.com/v1"