diff --git a/cli-config.yaml.example b/cli-config.yaml.example index 8bb49abda8..5b3a73acd9 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -152,10 +152,11 @@ model: # router: # base_url: https://api.router.com/v1 # api_key: ${RAMP_ROUTER_API_KEY} -# # api_mode auto-detected as codex_responses for api.router.com — the host -# # is Responses-only (/v1/chat/completions 404s). The bundled router -# # provider covers this; a named custom provider is only needed for a -# # non-default Router-compatible endpoint. +# # api_mode auto-detected as codex_responses for api.router.com — the +# # Responses API is Router's native wire (chat/completions is only a +# # compatibility shim). The bundled router provider covers this; a named +# # custom provider is only needed for a non-default Router-compatible +# # endpoint. # Command-minted credentials (optional): key_cmd diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py index ca41414821..2a02909ff6 100644 --- a/hermes_cli/providers.py +++ b/hermes_cli/providers.py @@ -663,8 +663,10 @@ def host_mandated_api_mode(base_url: str = "") -> Optional[str]: - api.meta.ai only achieves KV-cache hits on /v1/responses with prompt_cache_retention; /v1/chat/completions returns 0 cached tokens (measured 0% vs 93-99% on /responses with retention). - - api.router.com (Ramp Router) implements ONLY the Responses API — - POST /v1/chat/completions does not exist on the host and 404s. + - api.router.com (Ramp Router) is Responses-native: per-model + reasoning-effort validation, reasoning summaries, and prompt + caching live on /v1/responses; /v1/chat/completions is only a + minimal compatibility shim translated onto it. - api.anthropic.com / ``…/anthropic`` suffixes speak native Messages. - Kimi's ``/coding`` endpoint speaks native Messages. - AWS Bedrock runtime hosts speak Converse. @@ -697,9 +699,11 @@ def host_mandated_api_mode(base_url: str = "") -> Optional[str]: # cache-cold (0% vs 93-99% measured). Exact-hostname match per #32243. if hostname == "api.meta.ai": return "codex_responses" - # Ramp Router (api.router.com) is Responses-only: the host serves - # GET /v1/models and POST /v1/responses, and /v1/chat/completions 404s - # (docs.router.com/api/endpoint). Exact-hostname match per #32243. + # Ramp Router (api.router.com) is Responses-native: reasoning-effort + # validation, reasoning summaries, and prompt caching live on + # /v1/responses, and /v1/chat/completions is only a minimal + # compatibility shim (docs.router.com/api/endpoint). Exact-hostname + # match per #32243. if hostname == "api.router.com": return "codex_responses" if hostname.startswith("bedrock-runtime.") and base_url_host_matches(base_url, "amazonaws.com"): diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py index 4d7b546262..beb8f13a78 100644 --- a/hermes_cli/runtime_provider.py +++ b/hermes_cli/runtime_provider.py @@ -163,8 +163,9 @@ def _detect_api_mode_for_url(base_url: str) -> Optional[str]: return "codex_responses" if hostname == "api.actual.inc": return "codex_responses" - # Ramp Router: Responses-only host — /v1/chat/completions does not - # exist and 404s (docs.router.com/api/endpoint). Mirrors the + # Ramp Router: Responses-native host — /v1/chat/completions is only a + # minimal compatibility shim, while reasoning and caching support live + # on /v1/responses (docs.router.com/api/endpoint). Mirrors the # host_mandated_api_mode clause in hermes_cli/providers.py so the # runtime resolver stays in lockstep. Exact hostname per #32243. if hostname == "api.router.com": diff --git a/plugins/model-providers/router/__init__.py b/plugins/model-providers/router/__init__.py index a8f938656f..d1806d5883 100644 --- a/plugins/model-providers/router/__init__.py +++ b/plugins/model-providers/router/__init__.py @@ -8,10 +8,14 @@ spend controls server-side. Wire notes (verified live against api.router.com, Aug 2026): -* **Responses API only.** Router implements ``GET /v1/models`` and - ``POST /v1/responses``; ``POST /v1/chat/completions`` does not exist and - 404s. ``api_mode="codex_responses"`` plus the ``api.router.com`` host - mandate in ``hermes_cli/providers.py`` keep every path off the chat wire. +* **Responses API is the native wire.** Router serves ``GET /v1/models`` + and ``POST /v1/responses``; ``POST /v1/chat/completions`` is only a + minimal compatibility shim (added Aug 2026) that translates onto + Responses. Per-model reasoning-effort validation, reasoning summaries, + and prompt caching are Responses-surface features, so + ``api_mode="codex_responses"`` plus the ``api.router.com`` host mandate + in ``hermes_cli/providers.py`` keep every path on the native wire — + the same shape as the ``api.openai.com`` mandate. * **Account-scoped catalog.** Valid model IDs are whatever the key's ``GET /v1/models`` returns (BYOK accounts see extra entries), so this profile ships **no** ``fallback_models`` — the picker relies on the live diff --git a/tests/hermes_cli/test_router_provider.py b/tests/hermes_cli/test_router_provider.py index e7ca90403a..3d5c6a2c08 100644 --- a/tests/hermes_cli/test_router_provider.py +++ b/tests/hermes_cli/test_router_provider.py @@ -1,7 +1,8 @@ """Behavior contracts for the Ramp Router (api.router.com) provider. -Router is Responses-only: the host implements GET /v1/models and -POST /v1/responses, and POST /v1/chat/completions does not exist (404). +Router is Responses-native: the host implements GET /v1/models and +POST /v1/responses, and /v1/chat/completions is only a minimal +compatibility shim translated onto Responses. These tests pin the host mandate, the runtime URL detection that mirrors it, and the profile/auth registry wiring — same contract suite shape as tests/hermes_cli/test_meta_prompt_cache.py. diff --git a/website/docs/developer-guide/adding-providers.md b/website/docs/developer-guide/adding-providers.md index c8f448e0b9..ede1f622fd 100644 --- a/website/docs/developer-guide/adding-providers.md +++ b/website/docs/developer-guide/adding-providers.md @@ -35,7 +35,7 @@ The important abstraction is `api_mode`. - Most providers use `chat_completions`. - Codex and Meta Model API (`api.meta.ai` — Muse Spark) use `codex_responses` (auto-sends `prompt_cache_retention: 24h` for prompt caching; `api.meta.ai` achieves 93–99% cache hits only on `/v1/responses`). -- Ramp Router (`api.router.com`) also uses `codex_responses` — the host is Responses-only (`/v1/chat/completions` 404s) and validates `reasoning.effort` per model, which the router profile handles by declaring each model's vocabulary from the live catalog (`ProviderProfile.supported_reasoning_efforts`). +- Ramp Router (`api.router.com`) also uses `codex_responses` — Responses is Router's native wire (`/v1/chat/completions` is only a minimal compatibility shim), and it validates `reasoning.effort` per model, which the router profile handles by declaring each model's vocabulary from the live catalog (`ProviderProfile.supported_reasoning_efforts`). - Anthropic uses `anthropic_messages`. - A new non-OpenAI protocol usually means adding a new adapter and a new `api_mode` branch.