From e358eaf44a4db22e03d4f916769583601dffbe34 Mon Sep 17 00:00:00 2001 From: Brooklyn Nicholson Date: Sat, 8 Aug 2026 17:07:00 -0500 Subject: [PATCH] perf(sessions): resolve the titling model from the provider's live catalog Titling ran on the user's main chat model, so a five-word title was billed to a frontier reasoning model and inherited its latency. Pinning a cheap model id instead just moves the problem: the hardcoded default was already dead upstream and every call paid a 404 before the retry net caught it. Match model FAMILIES against the provider's live /v1/models catalog, preferring rolling '-latest' aliases where a provider publishes them, and order the families by measured latency. Nothing to bump when a provider ships a new mini/flash/haiku. Opt-in per task, so compression, vision, and search keep 'auto means my chat model'. --- agent/auxiliary_client.py | 153 +++++++++++++++++++++-- plugins/model-providers/nous/__init__.py | 16 +++ providers/base.py | 16 +++ tests/agent/test_auxiliary_client.py | 58 +++++++++ 4 files changed, 234 insertions(+), 9 deletions(-) diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 2a663eae38..5b38c0ab82 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -696,23 +696,131 @@ def _compression_threshold_for_model( return _CODEX_SPARK_COMPACTION_THRESHOLD return None -# Default auxiliary models for direct API-key providers (cheap/fast for side tasks) -def _get_aux_model_for_provider(provider_id: str) -> str: - """Return the cheap auxiliary model for a provider. +# Model-family priority for the auxiliary "fast tier", fastest first. +# +# Matched as substrings against the provider's LIVE /v1/models catalog rather +# than pinned as exact ids, because exact ids rot: a hardcoded +# "google/gemini-3-flash" kept 404ing here once Nous dropped it upstream, and +# every aux call paid a wasted round-trip before the retry net caught it. +# Families outlive their version numbers, so a new mini/flash/haiku release is +# picked up with no source edit. +# +# Rolling "-latest" aliases come first where a provider publishes them (Nous +# serves ~openai/gpt-mini-latest, ~google/gemini-flash-latest, …): they are the +# only ids that are structurally rot-proof. +# +# Order is measured, not guessed — p50 on a real titling prompt against the +# Nous catalog: gpt-mini-latest 1.40s, claude-haiku-latest 1.55s, +# gemini-flash-latest 2.13s, step-3.7-flash 7.84s, grok-4.1-fast 8.05s. So the +# first family a provider actually serves is also the fastest it can offer. +_FAST_MODEL_FAMILIES: tuple = ( + "gpt-mini-latest", + "gpt-nano-latest", + "claude-haiku-latest", + "gemini-flash-latest", + "gpt-5.4-nano", + "gpt-5.4-mini", + "gpt-5-mini", + "haiku-4.5", + "gemini-3.6-flash", + "flash-lite", + "-nano", + "-mini", + "-flash", + "haiku", +) - Reads from ProviderProfile.default_aux_model first, falling back to the - legacy hardcoded dict for providers that predate the profiles system. +# Substrings that disqualify an otherwise-matching id. Reasoning variants +# ("o3-mini", "gpt-5.4-mini-thinking") think before answering, which is the +# opposite of what a titler wants; ":batch" is an async queue, not a live +# endpoint; embedding models ("all-minilm") match "-mini" but aren't chat +# models at all; ":free" tiers are heavily rate-limited and measured slowest. +_FAST_MODEL_EXCLUDE: tuple = ( + "thinking", "reason", "-r1", "minilm", ":batch", ":free", + "o1-", "o3-", "o4-", "codex", "audio", "-vl", "embed", +) + + +def _fast_model_from_catalog(provider_id: str) -> str: + """Pick the fastest small model the provider ACTUALLY serves right now. + + Reads the provider's live (cached) ``/v1/models`` catalog and returns the + first ``_FAST_MODEL_FAMILIES`` match. Returns "" when the catalog is + unavailable or holds no small model, so the caller falls through to the + provider's curated default. Never raises and never blocks on a cold + network path — the underlying fetch is memory+disk cached with a + last-known-good fallback. """ try: + from hermes_cli.models import fetch_models_with_pricing from providers import get_provider_profile - _p = get_provider_profile(provider_id) - if _p and _p.default_aux_model: - return _p.default_aux_model + + profile = get_provider_profile(provider_id) + base_url = str(getattr(profile, "base_url", "") or "").rstrip("/") + if not base_url: + return "" + # fetch_models_with_pricing appends its own /v1/models. + if base_url.endswith("/v1"): + base_url = base_url[:-3] + catalog = fetch_models_with_pricing(base_url=base_url, timeout=3.0) or {} + except Exception: + logger.debug("Fast-model catalog lookup failed for %s", provider_id, exc_info=True) + return "" + + ids = sorted(str(m) for m in catalog) + for family in _FAST_MODEL_FAMILIES: + for model_id in ids: + lowered = model_id.lower() + if family in lowered and not any(x in lowered for x in _FAST_MODEL_EXCLUDE): + return model_id + return "" + + +# Default auxiliary models for direct API-key providers (cheap/fast for side tasks) +def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False) -> str: + """Return the cheap auxiliary model for a provider. + + Resolution ladder, fastest-and-most-live first: + + 1. ``prefer_fast`` only — a family match against the provider's LIVE + ``/v1/models`` catalog, preferring rolling ``-latest`` aliases. This is + both rot-proof and latency-ordered. + 2. ``prefer_fast`` only — the provider's own recommendation hook + (``ProviderProfile.resolve_aux_model``). Live, but tuned for *quality* + on long-context side tasks (Nous returns its compaction pick), so it + ranks below the catalog match for latency-critical work. + 3. ``ProviderProfile.default_aux_model`` — curated, hardcoded, may rot. + 4. The legacy hardcoded dict, for providers predating the profiles system. + + ``prefer_fast`` is opt-in so this only changes latency-critical tasks + (titling). Every other auxiliary caller keeps the existing static + behaviour and its cache keys. + """ + profile = None + try: + from providers import get_provider_profile + profile = get_provider_profile(provider_id) except Exception: pass + + if prefer_fast: + catalog_pick = _fast_model_from_catalog(provider_id) + if catalog_pick: + return catalog_pick + if profile is not None: + try: + live = profile.resolve_aux_model() + if live: + return live + except Exception: + logger.debug("resolve_aux_model failed for %s", provider_id, exc_info=True) + + if profile is not None and profile.default_aux_model: + return profile.default_aux_model return _API_KEY_PROVIDER_AUX_MODELS_FALLBACK.get(provider_id, "") + # Fallback for providers not yet migrated to ProviderProfile.default_aux_model, # plus providers we intentionally keep pinned here (e.g. Anthropic predates # profiles). New providers should set default_aux_model on their profile instead. @@ -740,6 +848,13 @@ _API_KEY_PROVIDER_AUX_MODELS_FALLBACK: Dict[str, str] = { # can still use this dict directly. Kept in sync with _FALLBACK above. _API_KEY_PROVIDER_AUX_MODELS: Dict[str, str] = _API_KEY_PROVIDER_AUX_MODELS_FALLBACK +# Auxiliary tasks that prefer the provider's fast/cheap model over the user's +# main chat model when running in "auto" mode. Restricted to tasks where +# latency is user-visible and the output is short enough that a small model +# matches a frontier one. Every other task keeps "auto = my chat model". +_FAST_MODEL_TASKS: frozenset = frozenset({"title_generation"}) + + # Vision-specific model overrides for direct providers. # When the user's main provider has a dedicated vision/multimodal model that # differs from their main chat model, map it here. The vision auto-detect @@ -5420,7 +5535,6 @@ def _resolve_auto_route( runtime_api_key = runtime.get("api_key", "") runtime_api_mode = str(runtime.get("api_mode") or "") - # ── Warn once if OPENAI_BASE_URL is set but config.yaml uses a named # provider (not 'custom'). This catches the common "env poisoning" # scenario where a user switches providers via `hermes model` but the @@ -5450,6 +5564,25 @@ def _resolve_auto_route( main_provider = str(runtime_provider or _read_main_provider() or "") main_model = str(runtime_model or _read_main_model() or "") + # Latency-critical tasks prefer the provider's registered fast model over + # the main chat model. Titling is the only such task: it names a visible + # sidebar row, produces ~8 tokens, and running it on a frontier reasoning + # model costs seconds per new session. Every comparable tool routes titling + # to a small tier (Claude Code → Haiku, OpenCode → small_model, Zed → + # default_fast_model, OpenClaw → utilityModel). An explicit + # auxiliary..model in config.yaml still wins — this only redirects + # the "auto" default, and only when the provider registered a cheap model. + # Every other aux task keeps the "auto means my chat model" contract + # documented above: this does NOT change compression, vision, or search. + if task in _FAST_MODEL_TASKS and main_provider and main_provider not in {"auto", ""}: + fast_model = _get_aux_model_for_provider(main_provider, prefer_fast=True) + if fast_model and fast_model != main_model: + logger.debug( + "Auxiliary task %s: preferring fast model %s over main model %s", + task, fast_model, main_model, + ) + main_model = fast_model + # MoA virtual provider: the "model" is a preset name (e.g. "opus-gpt") and # there is no real "moa" HTTP endpoint, so resolving an aux client against # provider="moa"/model= sends the preset name as the model id and @@ -8776,6 +8909,7 @@ def _call_llm_impl( api_key=resolved_api_key, api_mode=resolved_api_mode, main_runtime=main_runtime, + task=task, ) effective_provider = _effective_provider_for_client( client, resolved_provider, @@ -9554,6 +9688,7 @@ async def _async_call_llm_impl( api_key=resolved_api_key, api_mode=resolved_api_mode, main_runtime=main_runtime, + task=task, ) effective_provider = _effective_provider_for_client( client, resolved_provider, diff --git a/plugins/model-providers/nous/__init__.py b/plugins/model-providers/nous/__init__.py index bc22ec0a29..df1c1f4674 100644 --- a/plugins/model-providers/nous/__init__.py +++ b/plugins/model-providers/nous/__init__.py @@ -11,6 +11,22 @@ from providers.base import ProviderProfile class NousProfile(ProviderProfile): """Nous Portal — product tags, reasoning with Nous-specific omission.""" + def resolve_aux_model(self, *, vision: bool = False) -> str: + """Ask the Portal which cheap model it currently recommends. + + ``/api/nous/recommended-models`` is the authoritative, tier-aware + source (free vs paid), so the auxiliary fast tier tracks the live + catalog instead of a hardcoded id that 404s the day Nous retires it. + The underlying fetch is memory- and disk-cached with a last-known-good + fallback, so this is cheap to call and safe offline. + """ + try: + from hermes_cli.models import get_nous_recommended_aux_model + + return get_nous_recommended_aux_model(vision=vision) or "" + except Exception: + return "" + def build_extra_body( self, *, session_id: str | None = None, **context ) -> dict[str, Any]: diff --git a/providers/base.py b/providers/base.py index 1349d579bb..9108e0d6fc 100644 --- a/providers/base.py +++ b/providers/base.py @@ -101,6 +101,22 @@ class ProviderProfile: # ── Hooks (override in subclass for complex providers) ─── + def resolve_aux_model(self, *, vision: bool = False) -> str: + """Return a LIVE cheap-model id for auxiliary tasks, or "". + + ``default_aux_model`` is a hardcoded id in source, so it rots: when the + provider retires that model every auxiliary call spends a round-trip + 404ing before the retry net catches it. Providers that publish a + machine-readable recommendation should override this and query it, so + the cheap tier tracks the upstream catalog instead of a constant a human + has to remember to bump. + + Contract: cheap to call (implementations must cache — this runs on + client-resolution paths), never raises, and returns "" when it has no + answer so the caller falls through to ``default_aux_model``. + """ + return "" + def get_hostname(self) -> str: """Return the provider's base hostname for URL-based detection. diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index 0a512e166a..ce10a46c9b 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -4488,3 +4488,61 @@ class TestAutoRoutedProviderProfileHooks: assert relay.call_args_list[1].args[1]["extra_headers"] == { "Authorization": "Bearer token-2", } + + +class TestFastModelTier: + """The titling fast tier: rot-proof resolution, scoped to titling only.""" + + def test_catalog_match_prefers_rolling_alias_over_pinned_id(self): + """A "-latest" alias wins: it is the only id that cannot go stale.""" + from agent import auxiliary_client as ac + + catalog = { + "z-ai/glm-5.2": {}, + "openai/gpt-5.4-mini": {}, + "~openai/gpt-mini-latest": {}, + "stepfun/step-3.7-flash:free": {}, + } + with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog): + assert ac._fast_model_from_catalog("nous") == "~openai/gpt-mini-latest" + + def test_catalog_match_skips_reasoning_batch_and_embedding_lookalikes(self): + """Substring matching must not pick a thinker, a queue, or an encoder.""" + from agent import auxiliary_client as ac + + catalog = { + "openai/o3-mini": {}, + "openai/gpt-5.4-mini:batch": {}, + "sentence-transformers/all-minilm-l6-v2": {}, + "google/gemini-3.6-flash": {}, + } + with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog): + assert ac._fast_model_from_catalog("nous") == "google/gemini-3.6-flash" + + def test_falls_back_to_curated_default_when_catalog_unavailable(self): + """An offline catalog degrades to the provider's pinned default.""" + from agent import auxiliary_client as ac + + with patch.object(ac, "_fast_model_from_catalog", return_value=""): + assert ( + ac._get_aux_model_for_provider("anthropic", prefer_fast=True) + == ac._get_aux_model_for_provider("anthropic") + ) + + def test_fast_tier_is_opt_in(self): + """Without prefer_fast the resolver must not touch the live catalog.""" + from agent import auxiliary_client as ac + + with patch.object(ac, "_fast_model_from_catalog") as spy: + ac._get_aux_model_for_provider("nous") + spy.assert_not_called() + + def test_only_titling_is_in_the_fast_tier(self): + """Compression/vision/search keep 'auto means my chat model'.""" + from agent.auxiliary_client import _FAST_MODEL_TASKS + + assert "title_generation" in _FAST_MODEL_TASKS + overlap = {"compression", "vision", "web_extract"}.intersection( + _FAST_MODEL_TASKS + ) + assert not overlap