fix: clamp aux reasoning effort once before profile projection

Follow-up to the cherry-picked #112019 (@KoNit-K): the clamp in the generic
``extra_body.reasoning`` fallback only covered providers WITHOUT a
reasoning-aware profile. On the profile path (OpenRouter/Nous slots used as
MoA aggregator or aux model) ``_project_provider_profile`` received the raw
config and the OpenRouter profile passes ``ultra`` through whenever the
catalog vocabulary is cold, so the 400 from #112010 survived there.

Move the clamp up to ``_build_call_kwargs`` so both the profile projection
and the fallback see a wire-level effort — the same entry clamp the main
transport applies in ``_reasoning_config_for_model`` (#89503). The shared
policy lives once in ``agent.reasoning_effort.clamp_reasoning_config``; the
transport delegates to it instead of carrying its own copy.

Offline kwargs probe (issue's exact call): before
``extra_body.reasoning == {'enabled': True, 'effort': 'ultra'}`` on nous and
openrouter aux/MoA routes; after ``'effort': 'max'`` on every route,
``high`` verbatim and ``{'enabled': False}`` unchanged.
This commit is contained in:
teknium1
2026-09-15 12:04:40 -07:00
committed by Teknium
parent c02db64077
commit 9e45a90488
4 changed files with 49 additions and 15 deletions
+7 -9
View File
@@ -6157,14 +6157,8 @@ def _merge_aux_extra_body(
if reasoning_config.get("enabled") is False:
merged_extra["reasoning"] = {"enabled": False}
else:
# This fallback uses the OpenAI-compatible chat-completions wire. Hermes'
# internal ``ultra`` tier is not accepted there, including for MoA slots.
from agent.reasoning_effort import OPENAI_COMPAT_WIRE_EFFORTS, clamp_effort
effort = reasoning_config.get("effort") or "medium"
merged_extra["reasoning"] = {
"enabled": True,
"effort": clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS),
}
# ``reasoning_config`` is already clamped to the OpenAI-compat wire by _build_call_kwargs.
merged_extra["reasoning"] = {"enabled": True, "effort": reasoning_config.get("effort") or "medium"}
# Portal tags + sticky session_id fallback when the profile didn't supply them; session_id
# keeps aux calls on the main turn's upstream instance (cache warmth) — tags alone are not
# enough on /v1/messages.
@@ -6209,7 +6203,11 @@ def _build_call_kwargs(
kwargs["tools"] = _dedupe_tool_names(tools, provider, model)
# Provider profiles are the source of truth for reasoning wire shapes (top-level, nested body,
# or extra_body.reasoning); providers without a reasoning-aware profile keep the generic
# ``extra_body.reasoning`` fallback.
# ``extra_body.reasoning`` fallback. Clamp Hermes-internal levels (``ultra``) to the
# OpenAI-compat wire ONCE here, before either path sees the config — the same entry clamp the
# main transport applies (#89503); MoA aggregator/reference and aux calls 400'd without it (#112010).
from agent.reasoning_effort import clamp_reasoning_config
reasoning_config = clamp_reasoning_config(reasoning_config)
projection = _project_provider_profile(provider, provider_norm, model, effective_base, reasoning_config)
kwargs.update(projection.top_level)
if merged_extra := _merge_aux_extra_body(extra_body, projection, reasoning_config, provider_norm):
+15
View File
@@ -156,6 +156,21 @@ def requested_effort(reasoning_config: Optional[dict]) -> Optional[str]:
return str(reasoning_config.get("effort") or "").strip().lower() or None
def clamp_reasoning_config(reasoning_config: Optional[dict], supported: Sequence[str] = OPENAI_COMPAT_WIRE_EFFORTS) -> Optional[dict]:
"""Return ``reasoning_config`` with its ``effort`` clamped onto ``supported`` (non-dicts and
configs without an effort pass through untouched).
The entry clamp for an OpenAI-compatible chat-completions request builder: Hermes-internal
``ultra`` never reaches a wire (#89503 main transport, #112010 aux/MoA), while provider
profiles with narrower vocabularies clamp again downstream. Unset stays unset.
"""
if not isinstance(reasoning_config, dict):
return reasoning_config
effort = str(reasoning_config.get("effort") or "").strip().lower()
clamped = clamp_effort(effort, supported) if effort else effort
return {**reasoning_config, "effort": clamped} if clamped != effort else reasoning_config
def thinking_toggle_extras(
reasoning_config: Optional[dict],
efforts: Sequence[str],
+2 -6
View File
@@ -11,7 +11,7 @@ from urllib.parse import urlparse
from agent.lmstudio_reasoning import resolve_lmstudio_effort
from agent.reasoning_effort import (
KIMI_K3_EFFORTS, KIMI_K3_OVERRIDES, OPENAI_COMPAT_WIRE_EFFORTS, TOKENHUB_EFFORTS, clamp_effort,
kimi_supported_efforts, requested_effort,
clamp_reasoning_config, kimi_supported_efforts, requested_effort,
)
from agent.message_sanitization import normalize_finish_reason as _normalize_finish_reason
from agent.moonshot_schema import is_moonshot_model, sanitize_moonshot_tools
@@ -119,11 +119,7 @@ def _reasoning_config_for_model(model: str, reasoning_config: dict | None) -> di
the declared wire vocabulary via the shared policy in ``agent.reasoning_effort``; provider profiles with
narrower sets clamp again downstream.
"""
if not isinstance(reasoning_config, dict):
return reasoning_config
effort = str(reasoning_config.get("effort") or "").strip().lower()
clamped = clamp_effort(effort, OPENAI_COMPAT_WIRE_EFFORTS) if effort else effort
return {**reasoning_config, "effort": clamped} if clamped != effort else reasoning_config
return clamp_reasoning_config(reasoning_config, OPENAI_COMPAT_WIRE_EFFORTS)
def _build_gemini_thinking_config(model: str, reasoning_config: dict | None) -> dict | None:
+25
View File
@@ -2519,6 +2519,31 @@ class TestAuxiliaryTaskExtraBody:
assert kwargs["extra_body"]["reasoning"] == {"enabled": True, "effort": "max"}
def test_profile_projection_receives_wire_clamped_effort(self, monkeypatch):
"""Profiles clamp only against their own narrower sets (or a catalog that may be cold), so
``ultra`` must already be a wire level when the projection sees it — the MoA aggregator on
an OpenRouter/Nous slot 400'd otherwise (#112010)."""
import agent.auxiliary_client as aux
seen = {}
real = aux._project_provider_profile
def spy(provider, provider_norm, model, effective_base, reasoning_config):
seen["config"] = reasoning_config
return real(provider, provider_norm, model, effective_base, reasoning_config)
monkeypatch.setattr(aux, "_project_provider_profile", spy)
kwargs = aux._build_call_kwargs(
provider="openrouter",
model="deepseek/deepseek-v4.1-flash",
messages=[{"role": "user", "content": "hello"}],
reasoning_config={"enabled": True, "effort": "ultra"},
task="moa_aggregator",
)
assert seen["config"] == {"enabled": True, "effort": "max"}
assert "ultra" not in json.dumps(kwargs.get("extra_body")) and kwargs.get("reasoning_effort") != "ultra"
def test_sync_call_merges_task_extra_body_from_config(self):
client = MagicMock()
client.base_url = "https://api.example.com/v1"