Files
hermes-agent/tests/agent/transports/test_router_codex_efforts.py
T
Neel Patel 804f8b4732 feat(providers): add Ramp Router (router.com) provider plugin
Ramp Router is an OpenAI Responses-compatible LLM gateway at
https://api.router.com/v1 that routes each request across upstream
providers (OpenAI, Anthropic, xAI, Fireworks, ...) with server-side
fallbacks and spend controls. Nous asked for a PR adding it as a
provider, so:

- plugins/model-providers/router/: RouterProfile plugin —
  api_mode=codex_responses, RAMP_ROUTER_API_KEY auth,
  RAMP_ROUTER_BASE_URL override, live account-scoped catalog via
  GET /v1/models (no hardcoded fallback_models: IDs are key-scoped and
  Router's docs mandate runtime catalog reads).
- hermes_cli/providers.host_mandated_api_mode +
  runtime_provider._detect_api_mode_for_url: api.router.com ->
  codex_responses. The host is Responses-only — POST /v1/chat/completions
  does not exist and 404s — so this is a genuine host mandate (exact
  hostname match per #32243, mirroring the api.meta.ai precedent).
- providers/base.py: new overrideable supported_reasoning_efforts(model)
  hook (tri-state: None=defer, ()=model takes no reasoning params,
  tuple=clamp set). Router validates reasoning.effort per model and
  returns HTTP 400 invalid-argument on levels outside the model's
  published vocabulary, and 400 unsupported_parameter when a
  non-reasoning model receives any reasoning field (both verified live).
  The profile answers from a cached copy of the catalog's
  router.capabilities.reasoning block: cache-only on the hot path,
  seeded for free by fetch_models(), disk-mirrored across processes
  (/cache/router_catalog.json), background-warmed when cold
  — same design as the OpenRouter reasoning-caps clamp on the chat path.
- agent/transports/codex.py: consult the profile-declared vocabulary in
  the generic effort-clamp branch (xai/actual/github branches untouched;
  profiles that do not override the hook see no behavior change).
- cli-config.yaml.example + adding-providers.md + providers/README.md:
  document the provider, the host mandate, and the new hook.
- tests: behavior contracts for the host mandate/URL detection/spoof
  rejection, profile registration + auth auto-registry wiring, catalog
  parsing, and transport clamp/suppression/fallback paths.

Verified live against api.router.com (Aug 2026): one-shot chat,
streaming SSE, tool calls + parallel_tool_calls, encrypted-reasoning
replay on OpenAI-served models, function_call_output follow-up turns on
OpenAI- and Fireworks-served models; store:false / prompt_cache_key /
include:[reasoning.encrypted_content] / reasoning.summary accepted
across backends; effort clamp confirmed to convert a would-be 400
(xhigh on o3) into a successful request via the disk mirror.
2026-08-29 20:29:04 +05:30

157 lines
6.3 KiB
Python

"""Router catalog-declared reasoning-effort clamping on the codex transport.
Ramp Router (api.router.com) validates ``reasoning.effort`` against each
model's published vocabulary — HTTP 400 ``invalid-argument`` on an
unsupported level, and 400 ``unsupported_parameter`` when a non-reasoning
model receives any reasoning field (both verified live, Aug 2026). The
router profile declares each model's vocabulary from its cached catalog via
``ProviderProfile.supported_reasoning_efforts``; these tests pin how the
codex transport consumes that declaration.
All tests seed the plugin's in-memory cache directly — no network.
"""
import sys
import pytest
from agent.transports import get_transport
def _router_plugin_module():
from providers import get_provider_profile
profile = get_provider_profile("router")
assert profile is not None, "router profile must be registered"
return profile, sys.modules[type(profile).__module__]
@pytest.fixture
def transport():
import agent.transports.codex # noqa: F401
return get_transport("codex_responses")
@pytest.fixture
def seeded_catalog(monkeypatch):
"""Seed the router efforts cache with catalog-shaped verdicts."""
profile, mod = _router_plugin_module()
monkeypatch.setattr(mod, "_efforts_cache", {
# grok via Router: no "none", no "max" (live catalog shape)
"grok-4.6": ["minimal", "low", "medium", "high", "xhigh"],
# non-reasoning model: any reasoning field 400s
"gpt-4.1-mini": [],
# full ladder including max
"accounts/fireworks/models/kimi-k3": [
"minimal", "low", "medium", "high", "xhigh", "max",
],
})
monkeypatch.setattr(mod, "_disk_checked", True)
return profile
class TestProfileContract:
def test_declared_vocabulary(self, seeded_catalog):
assert seeded_catalog.supported_reasoning_efforts("grok-4.6") == (
"minimal", "low", "medium", "high", "xhigh",
)
def test_non_reasoning_model_is_definitive_empty(self, seeded_catalog):
assert seeded_catalog.supported_reasoning_efforts("gpt-4.1-mini") == ()
def test_unknown_model_is_none(self, seeded_catalog):
assert seeded_catalog.supported_reasoning_efforts("some-byok-route") is None
def test_cold_cache_is_none_and_never_blocks(self, monkeypatch):
profile, mod = _router_plugin_module()
monkeypatch.setattr(mod, "_efforts_cache", None)
monkeypatch.setattr(mod, "_disk_checked", True)
monkeypatch.setattr(mod, "_warm_efforts_async", lambda: None)
assert profile.supported_reasoning_efforts("grok-4.6") is None
def test_parse_efforts_catalog_shapes(self):
_, mod = _router_plugin_module()
parsed = mod._parse_efforts([
{
"id": "grok-4.6",
"router": {"capabilities": {"reasoning": {
"supported": True,
"efforts": [{"value": "low"}, {"value": "high"}],
}}},
},
{
"id": "gpt-4.1",
"router": {"capabilities": {"reasoning": {"supported": False, "efforts": []}}},
},
# reasoning supported but vocabulary unpublished -> omitted (unknown)
{
"id": "mystery-model",
"router": {"capabilities": {"reasoning": {"supported": True, "efforts": []}}},
},
# no router metadata at all -> omitted
{"id": "bare-model"},
])
assert parsed == {"grok-4.6": ["low", "high"], "gpt-4.1": []}
class TestTransportClamp:
def _kwargs(self, transport, model, reasoning_config=None):
return transport.build_kwargs(
model=model,
messages=[{"role": "user", "content": "Hi"}],
tools=[],
base_url="https://api.router.com/v1",
session_id="sid",
provider="router",
reasoning_config=reasoning_config,
)
def test_clamps_to_catalog_vocabulary(self, transport, seeded_catalog):
# grok-4.6 via Router has no "max" — nearest weaker supported is xhigh.
kw = self._kwargs(transport, "grok-4.6", {"effort": "max"})
assert kw["reasoning"]["effort"] == "xhigh"
def test_supported_effort_passes_through(self, transport, seeded_catalog):
kw = self._kwargs(
transport, "accounts/fireworks/models/kimi-k3", {"effort": "max"}
)
assert kw["reasoning"]["effort"] == "max"
def test_non_reasoning_model_suppresses_reasoning(self, transport, seeded_catalog):
# Default reasoning_config is enabled — the () verdict must strip the
# reasoning field entirely (Router 400s rather than ignoring it).
kw = self._kwargs(transport, "gpt-4.1-mini")
assert "reasoning" not in kw
assert kw.get("include") == []
def test_unknown_model_falls_back_to_codex_default(self, transport, seeded_catalog):
# Not in the catalog -> default codex vocabulary applies (legacy has
# xhigh but no max: max clamps to xhigh, medium is untouched).
kw = self._kwargs(transport, "some-byok-route", {"effort": "max"})
assert kw["reasoning"]["effort"] == "xhigh"
kw = self._kwargs(transport, "some-byok-route", {"effort": "medium"})
assert kw["reasoning"]["effort"] == "medium"
def test_cold_cache_keeps_default_behavior(self, transport, monkeypatch):
_, mod = _router_plugin_module()
monkeypatch.setattr(mod, "_efforts_cache", None)
monkeypatch.setattr(mod, "_disk_checked", True)
monkeypatch.setattr(mod, "_warm_efforts_async", lambda: None)
kw = self._kwargs(transport, "grok-4.6", {"effort": "xhigh"})
# Cold cache -> no declaration -> default codex vocabulary (xhigh ok).
assert kw["reasoning"]["effort"] == "xhigh"
def test_other_providers_unaffected(self, transport, seeded_catalog):
kw = transport.build_kwargs(
model="gpt-4.1-mini",
messages=[{"role": "user", "content": "Hi"}],
tools=[],
base_url="https://generic.example.com/v1",
session_id="sid",
provider="some-other-provider",
reasoning_config={"effort": "medium"},
)
# The router catalog's () verdict for gpt-4.1-mini must not leak
# into other providers' requests.
assert kw["reasoning"]["effort"] == "medium"