feat(openrouter): per-model provider_routing.models.<id> overrides
`provider_routing.models.<model-id>` now takes the same only/ignore/order/sort/
require_parameters/data_collection keys and overlays the flat provider_routing
values whenever the agent is on that model. Resolution lives in the one
chokepoint every request path already uses (_provider_preferences_for_agent),
so CLI, gateway, TUI/Desktop, cron, /model switches, fallback activation and
delegated children on another model all honour it with no per-surface plumbing.
Matching is spelling-tolerant, sharing _canonical_model_variants with
agent.reasoning_overrides.
The OpenRouter profile's speed-tier pin no longer overwrites an explicit user
`only` on the BASE gpt-6-astra slug: the pin exists to keep default routing off
flex/fast, and a user pin is the stronger intent (only: [openai] stays [openai]
instead of becoming [openai, azure, azure/us]). Tier slugs (-fast/-flex) keep
owning `only`.
Live A/B (config only: {gpt-6-astra: [openai], claude-fable-5.1: [anthropic]}):
main sent {"sort":"price"} for fable and OpenRouter served it from Azure; with
this change it sends {"only":["anthropic"],"sort":"price"} and Anthropic serves it.
Schema proposed in #24495 (samplesabotage) and #100711 (Artemonim); this is a
slim chokepoint implementation of that design.
Co-authored-by: samplesabotage <samplesabotage@users.noreply.github.com>
This commit is contained in:
@@ -378,15 +378,24 @@ def _validated_openrouter_provider_sort(raw_sort: Any) -> Optional[str]:
|
||||
|
||||
|
||||
def _provider_preferences_for_agent(agent) -> Dict[str, Any]:
|
||||
"""Build the validated provider-routing object shared by request paths."""
|
||||
preferences: Dict[str, Any] = {}
|
||||
for key, value in (("only", agent.providers_allowed), ("ignore", agent.providers_ignored),
|
||||
("order", agent.providers_order), ("sort", _validated_openrouter_provider_sort(agent.provider_sort)),
|
||||
("require_parameters", True if agent.provider_require_parameters else None),
|
||||
("data_collection", agent.provider_data_collection)):
|
||||
if value:
|
||||
preferences[key] = value
|
||||
return preferences
|
||||
"""Build the validated provider-routing object shared by request paths.
|
||||
|
||||
``provider_routing.models.<id>`` overlays the flat constructor values for the CURRENT
|
||||
``agent.model`` (so ``/model`` switches, fallbacks, and delegated children on another
|
||||
model each get their own pins without any surface re-plumbing the kwargs)."""
|
||||
flat = {"only": agent.providers_allowed, "ignore": agent.providers_ignored, "order": agent.providers_order,
|
||||
"sort": agent.provider_sort, "require_parameters": agent.provider_require_parameters,
|
||||
"data_collection": agent.provider_data_collection}
|
||||
per_model = {}
|
||||
with contextlib.suppress(Exception):
|
||||
from hermes_cli.config import load_config_readonly
|
||||
from hermes_constants import resolve_per_model_provider_routing
|
||||
_pr = load_config_readonly().get("provider_routing")
|
||||
per_model = resolve_per_model_provider_routing(agent.model, (_pr or {}).get("models") if isinstance(_pr, dict) else None)
|
||||
merged = {**flat, **{k: v for k, v in per_model.items() if k in flat}}
|
||||
merged["sort"] = _validated_openrouter_provider_sort(merged["sort"])
|
||||
merged["require_parameters"] = True if merged["require_parameters"] else None
|
||||
return {key: value for key, value in merged.items() if value}
|
||||
|
||||
|
||||
def _prompt_cache_scope_for_agent(agent) -> "str | None":
|
||||
|
||||
@@ -892,7 +892,8 @@ def log_api_error_attempt(
|
||||
|
||||
if agent._is_openrouter_url() and "support tool use" in error_msg:
|
||||
_blines(agent, f" 💡 No OpenRouter providers for {_model} support tool calling with your current settings.")
|
||||
if agent.providers_allowed:
|
||||
from agent.chat_completion_helpers import _provider_preferences_for_agent
|
||||
if _provider_preferences_for_agent(agent).get("only"):
|
||||
_blines(
|
||||
agent,
|
||||
" Your provider_routing.only restriction is filtering out tool-capable providers.",
|
||||
|
||||
@@ -301,6 +301,14 @@ model:
|
||||
#
|
||||
# # Data policy: "allow" (default) or "deny" to exclude providers that may store data
|
||||
# # data_collection: "deny"
|
||||
#
|
||||
# # Per-model overrides: same keys, applied only when the agent is on that model
|
||||
# # (spelling-tolerant match; unset keys fall through to the flat values above).
|
||||
# # models:
|
||||
# # "openai/gpt-6-astra":
|
||||
# # only: ["openai"]
|
||||
# # "anthropic/claude-fable-5.1":
|
||||
# # only: ["anthropic"]
|
||||
|
||||
# =============================================================================
|
||||
# OpenRouter Response Caching (only applies when using OpenRouter)
|
||||
|
||||
@@ -941,6 +941,19 @@ def resolve_per_model_reasoning_effort(model: str, overrides: dict | None) -> di
|
||||
return None
|
||||
|
||||
|
||||
def resolve_per_model_provider_routing(model: str, models: dict | None) -> dict:
|
||||
"""``provider_routing.models.<id>`` entry for *model*, spelling-tolerant like
|
||||
``reasoning_overrides``; ``{}`` when none matches. Only the keys a user sets per model
|
||||
are returned so unset ones fall through to the flat ``provider_routing`` values."""
|
||||
if not model or not isinstance(models, dict):
|
||||
return {}
|
||||
for variant in _canonical_model_variants(model):
|
||||
entry = models.get(variant)
|
||||
if isinstance(entry, dict):
|
||||
return entry
|
||||
return {}
|
||||
|
||||
|
||||
def resolve_reasoning_config(cfg: dict | None, model: str = "") -> dict | None:
|
||||
"""Effective reasoning config for *model*: per-model override, then global ``agent.reasoning_effort``.
|
||||
|
||||
|
||||
@@ -127,8 +127,9 @@ class OpenRouterProfile(ProviderProfile):
|
||||
body["session_id"] = sticky_key
|
||||
prefs = context.get("provider_preferences")
|
||||
pin = OPENROUTER_ENDPOINT_PINS.get(context.get("model") or "")
|
||||
if pin:
|
||||
# The tier pin owns ``only``; the user's other routing prefs (ignore/sort/...) still apply.
|
||||
# The tier pin owns ``only`` (ignore/sort/... still apply) — except on the BASE slug, where the pin
|
||||
# merely keeps default routing off flex/fast and an explicit user ``only`` is the stronger intent.
|
||||
if pin and not (pin[0] == context.get("model") and (prefs or {}).get("only")):
|
||||
prefs = {**(prefs or {}), "only": list(pin[1])}
|
||||
if prefs:
|
||||
body["provider"] = prefs
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
"""``provider_routing.models.<id>`` overlays the flat OpenRouter routing for the CURRENT agent.model."""
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from agent import chat_completion_helpers as cch
|
||||
|
||||
|
||||
def _agent(model, **flat):
|
||||
base = dict(providers_allowed=None, providers_ignored=None, providers_order=None, provider_sort="price",
|
||||
provider_require_parameters=False, provider_data_collection=None)
|
||||
base.update(flat)
|
||||
return SimpleNamespace(model=model, **base)
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def routing_cfg(monkeypatch):
|
||||
cfg = {"provider_routing": {"sort": "price", "models": {
|
||||
"openai/gpt-6-astra": {"only": ["openai"]},
|
||||
"anthropic/claude-fable-5.1": {"only": ["anthropic"], "sort": "throughput"},
|
||||
}}}
|
||||
import hermes_cli.config as config_mod
|
||||
monkeypatch.setattr(config_mod, "load_config_readonly", lambda: cfg)
|
||||
return cfg
|
||||
|
||||
|
||||
def test_per_model_entry_overlays_flat_routing_for_that_model_only(routing_cfg):
|
||||
assert cch._provider_preferences_for_agent(_agent("openai/gpt-6-astra")) == {"only": ["openai"], "sort": "price"}
|
||||
# A per-model key wins over the flat one; unset keys fall through.
|
||||
assert cch._provider_preferences_for_agent(_agent("anthropic/claude-fable-5.1")) == {
|
||||
"only": ["anthropic"], "sort": "throughput"}
|
||||
# Unlisted model keeps the flat behaviour; no pin leaks across models.
|
||||
assert cch._provider_preferences_for_agent(_agent("moonshotai/kimi-k2.6")) == {"sort": "price"}
|
||||
|
||||
|
||||
def test_per_model_match_is_spelling_tolerant_and_follows_model_switch(routing_cfg):
|
||||
agent = _agent("openrouter/openai/gpt-6-astra", providers_allowed=["together"])
|
||||
assert cch._provider_preferences_for_agent(agent)["only"] == ["openai"]
|
||||
agent.model = "claude-fable-5-1"
|
||||
assert cch._provider_preferences_for_agent(agent)["only"] == ["anthropic"]
|
||||
@@ -1558,9 +1558,12 @@ provider_routing:
|
||||
# order: ["anthropic", "google"] # Try providers in this order
|
||||
# require_parameters: true # Only use providers that support all request params
|
||||
# data_collection: "deny" # Exclude providers that may store/train on data
|
||||
# models: # Per-model pins (same keys; unset keys fall through)
|
||||
# "openai/gpt-6-astra": {only: ["openai"]}
|
||||
# "anthropic/claude-fable-5.1": {only: ["anthropic"]}
|
||||
```
|
||||
|
||||
**Shortcuts:** Append `:nitro` to any model name for throughput sorting (e.g., `anthropic/claude-sonnet-4:nitro`), or `:floor` for price sorting.
|
||||
**Shortcuts:** Append `:nitro` to any model name for throughput sorting (e.g., `anthropic/claude-sonnet-4:nitro`), or `:floor` for price sorting. Per-model details: [Provider Routing](/user-guide/features/provider-routing#per-model-overrides-models).
|
||||
|
||||
## OpenRouter Pareto Code Router
|
||||
|
||||
|
||||
@@ -102,6 +102,31 @@ provider_routing:
|
||||
data_collection: "deny"
|
||||
```
|
||||
|
||||
### Per-model overrides (`models`)
|
||||
|
||||
Pin a different provider set per model. Keys under `models` are model ids; each entry takes the same
|
||||
`sort` / `only` / `ignore` / `order` / `require_parameters` / `data_collection` keys and overrides the
|
||||
flat value for that model only. Anything you don't set per model falls through to the flat defaults.
|
||||
|
||||
```yaml
|
||||
provider_routing:
|
||||
sort: "price" # applies to every model
|
||||
models:
|
||||
"openai/gpt-6-astra":
|
||||
only: ["openai"] # never let a reseller serve this one
|
||||
"anthropic/claude-fable-5.1":
|
||||
only: ["anthropic"]
|
||||
"moonshotai/kimi-k2.6":
|
||||
order: ["moonshotai", "together"]
|
||||
sort: "throughput"
|
||||
```
|
||||
|
||||
Matching is spelling-tolerant like `agent.reasoning_overrides` (`claude-fable-5.1` / `claude-fable-5-1`,
|
||||
with or without the `openrouter/` prefix). The override follows the model the agent is *currently* on, so
|
||||
`/model` switches, fallback activation, cron jobs, and delegated subagents on another model each get their
|
||||
own pins. Edit `config.yaml` directly for these keys: model ids contain dots, which `hermes config set`
|
||||
reads as path separators.
|
||||
|
||||
## Practical Examples
|
||||
|
||||
### Optimize for Cost
|
||||
|
||||
Reference in New Issue
Block a user