perf(sessions): resolve the titling model from the provider's live catalog

Titling ran on the user's main chat model, so a five-word title was billed
to a frontier reasoning model and inherited its latency. Pinning a cheap
model id instead just moves the problem: the hardcoded default was already
dead upstream and every call paid a 404 before the retry net caught it.

Match model FAMILIES against the provider's live /v1/models catalog,
preferring rolling '-latest' aliases where a provider publishes them, and
order the families by measured latency. Nothing to bump when a provider
ships a new mini/flash/haiku. Opt-in per task, so compression, vision, and
search keep 'auto means my chat model'.
This commit is contained in:
Brooklyn Nicholson
2026-08-08 17:07:00 -05:00
parent 5566379f57
commit e358eaf44a
4 changed files with 234 additions and 9 deletions
+144 -9
View File
@@ -696,23 +696,131 @@ def _compression_threshold_for_model(
return _CODEX_SPARK_COMPACTION_THRESHOLD
return None
# Default auxiliary models for direct API-key providers (cheap/fast for side tasks)
def _get_aux_model_for_provider(provider_id: str) -> str:
"""Return the cheap auxiliary model for a provider.
# Model-family priority for the auxiliary "fast tier", fastest first.
#
# Matched as substrings against the provider's LIVE /v1/models catalog rather
# than pinned as exact ids, because exact ids rot: a hardcoded
# "google/gemini-3-flash" kept 404ing here once Nous dropped it upstream, and
# every aux call paid a wasted round-trip before the retry net caught it.
# Families outlive their version numbers, so a new mini/flash/haiku release is
# picked up with no source edit.
#
# Rolling "-latest" aliases come first where a provider publishes them (Nous
# serves ~openai/gpt-mini-latest, ~google/gemini-flash-latest, …): they are the
# only ids that are structurally rot-proof.
#
# Order is measured, not guessed — p50 on a real titling prompt against the
# Nous catalog: gpt-mini-latest 1.40s, claude-haiku-latest 1.55s,
# gemini-flash-latest 2.13s, step-3.7-flash 7.84s, grok-4.1-fast 8.05s. So the
# first family a provider actually serves is also the fastest it can offer.
_FAST_MODEL_FAMILIES: tuple = (
"gpt-mini-latest",
"gpt-nano-latest",
"claude-haiku-latest",
"gemini-flash-latest",
"gpt-5.4-nano",
"gpt-5.4-mini",
"gpt-5-mini",
"haiku-4.5",
"gemini-3.6-flash",
"flash-lite",
"-nano",
"-mini",
"-flash",
"haiku",
)
Reads from ProviderProfile.default_aux_model first, falling back to the
legacy hardcoded dict for providers that predate the profiles system.
# Substrings that disqualify an otherwise-matching id. Reasoning variants
# ("o3-mini", "gpt-5.4-mini-thinking") think before answering, which is the
# opposite of what a titler wants; ":batch" is an async queue, not a live
# endpoint; embedding models ("all-minilm") match "-mini" but aren't chat
# models at all; ":free" tiers are heavily rate-limited and measured slowest.
_FAST_MODEL_EXCLUDE: tuple = (
"thinking", "reason", "-r1", "minilm", ":batch", ":free",
"o1-", "o3-", "o4-", "codex", "audio", "-vl", "embed",
)
def _fast_model_from_catalog(provider_id: str) -> str:
"""Pick the fastest small model the provider ACTUALLY serves right now.
Reads the provider's live (cached) ``/v1/models`` catalog and returns the
first ``_FAST_MODEL_FAMILIES`` match. Returns "" when the catalog is
unavailable or holds no small model, so the caller falls through to the
provider's curated default. Never raises and never blocks on a cold
network path — the underlying fetch is memory+disk cached with a
last-known-good fallback.
"""
try:
from hermes_cli.models import fetch_models_with_pricing
from providers import get_provider_profile
_p = get_provider_profile(provider_id)
if _p and _p.default_aux_model:
return _p.default_aux_model
profile = get_provider_profile(provider_id)
base_url = str(getattr(profile, "base_url", "") or "").rstrip("/")
if not base_url:
return ""
# fetch_models_with_pricing appends its own /v1/models.
if base_url.endswith("/v1"):
base_url = base_url[:-3]
catalog = fetch_models_with_pricing(base_url=base_url, timeout=3.0) or {}
except Exception:
logger.debug("Fast-model catalog lookup failed for %s", provider_id, exc_info=True)
return ""
ids = sorted(str(m) for m in catalog)
for family in _FAST_MODEL_FAMILIES:
for model_id in ids:
lowered = model_id.lower()
if family in lowered and not any(x in lowered for x in _FAST_MODEL_EXCLUDE):
return model_id
return ""
# Default auxiliary models for direct API-key providers (cheap/fast for side tasks)
def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False) -> str:
"""Return the cheap auxiliary model for a provider.
Resolution ladder, fastest-and-most-live first:
1. ``prefer_fast`` only — a family match against the provider's LIVE
``/v1/models`` catalog, preferring rolling ``-latest`` aliases. This is
both rot-proof and latency-ordered.
2. ``prefer_fast`` only — the provider's own recommendation hook
(``ProviderProfile.resolve_aux_model``). Live, but tuned for *quality*
on long-context side tasks (Nous returns its compaction pick), so it
ranks below the catalog match for latency-critical work.
3. ``ProviderProfile.default_aux_model`` — curated, hardcoded, may rot.
4. The legacy hardcoded dict, for providers predating the profiles system.
``prefer_fast`` is opt-in so this only changes latency-critical tasks
(titling). Every other auxiliary caller keeps the existing static
behaviour and its cache keys.
"""
profile = None
try:
from providers import get_provider_profile
profile = get_provider_profile(provider_id)
except Exception:
pass
if prefer_fast:
catalog_pick = _fast_model_from_catalog(provider_id)
if catalog_pick:
return catalog_pick
if profile is not None:
try:
live = profile.resolve_aux_model()
if live:
return live
except Exception:
logger.debug("resolve_aux_model failed for %s", provider_id, exc_info=True)
if profile is not None and profile.default_aux_model:
return profile.default_aux_model
return _API_KEY_PROVIDER_AUX_MODELS_FALLBACK.get(provider_id, "")
# Fallback for providers not yet migrated to ProviderProfile.default_aux_model,
# plus providers we intentionally keep pinned here (e.g. Anthropic predates
# profiles). New providers should set default_aux_model on their profile instead.
@@ -740,6 +848,13 @@ _API_KEY_PROVIDER_AUX_MODELS_FALLBACK: Dict[str, str] = {
# can still use this dict directly. Kept in sync with _FALLBACK above.
_API_KEY_PROVIDER_AUX_MODELS: Dict[str, str] = _API_KEY_PROVIDER_AUX_MODELS_FALLBACK
# Auxiliary tasks that prefer the provider's fast/cheap model over the user's
# main chat model when running in "auto" mode. Restricted to tasks where
# latency is user-visible and the output is short enough that a small model
# matches a frontier one. Every other task keeps "auto = my chat model".
_FAST_MODEL_TASKS: frozenset = frozenset({"title_generation"})
# Vision-specific model overrides for direct providers.
# When the user's main provider has a dedicated vision/multimodal model that
# differs from their main chat model, map it here. The vision auto-detect
@@ -5420,7 +5535,6 @@ def _resolve_auto_route(
runtime_api_key = runtime.get("api_key", "")
runtime_api_mode = str(runtime.get("api_mode") or "")
# ── Warn once if OPENAI_BASE_URL is set but config.yaml uses a named
# provider (not 'custom'). This catches the common "env poisoning"
# scenario where a user switches providers via `hermes model` but the
@@ -5450,6 +5564,25 @@ def _resolve_auto_route(
main_provider = str(runtime_provider or _read_main_provider() or "")
main_model = str(runtime_model or _read_main_model() or "")
# Latency-critical tasks prefer the provider's registered fast model over
# the main chat model. Titling is the only such task: it names a visible
# sidebar row, produces ~8 tokens, and running it on a frontier reasoning
# model costs seconds per new session. Every comparable tool routes titling
# to a small tier (Claude Code → Haiku, OpenCode → small_model, Zed →
# default_fast_model, OpenClaw → utilityModel). An explicit
# auxiliary.<task>.model in config.yaml still wins — this only redirects
# the "auto" default, and only when the provider registered a cheap model.
# Every other aux task keeps the "auto means my chat model" contract
# documented above: this does NOT change compression, vision, or search.
if task in _FAST_MODEL_TASKS and main_provider and main_provider not in {"auto", ""}:
fast_model = _get_aux_model_for_provider(main_provider, prefer_fast=True)
if fast_model and fast_model != main_model:
logger.debug(
"Auxiliary task %s: preferring fast model %s over main model %s",
task, fast_model, main_model,
)
main_model = fast_model
# MoA virtual provider: the "model" is a preset name (e.g. "opus-gpt") and
# there is no real "moa" HTTP endpoint, so resolving an aux client against
# provider="moa"/model=<preset> sends the preset name as the model id and
@@ -8776,6 +8909,7 @@ def _call_llm_impl(
api_key=resolved_api_key,
api_mode=resolved_api_mode,
main_runtime=main_runtime,
task=task,
)
effective_provider = _effective_provider_for_client(
client, resolved_provider,
@@ -9554,6 +9688,7 @@ async def _async_call_llm_impl(
api_key=resolved_api_key,
api_mode=resolved_api_mode,
main_runtime=main_runtime,
task=task,
)
effective_provider = _effective_provider_for_client(
client, resolved_provider,
+16
View File
@@ -11,6 +11,22 @@ from providers.base import ProviderProfile
class NousProfile(ProviderProfile):
"""Nous Portal — product tags, reasoning with Nous-specific omission."""
def resolve_aux_model(self, *, vision: bool = False) -> str:
"""Ask the Portal which cheap model it currently recommends.
``/api/nous/recommended-models`` is the authoritative, tier-aware
source (free vs paid), so the auxiliary fast tier tracks the live
catalog instead of a hardcoded id that 404s the day Nous retires it.
The underlying fetch is memory- and disk-cached with a last-known-good
fallback, so this is cheap to call and safe offline.
"""
try:
from hermes_cli.models import get_nous_recommended_aux_model
return get_nous_recommended_aux_model(vision=vision) or ""
except Exception:
return ""
def build_extra_body(
self, *, session_id: str | None = None, **context
) -> dict[str, Any]:
+16
View File
@@ -101,6 +101,22 @@ class ProviderProfile:
# ── Hooks (override in subclass for complex providers) ───
def resolve_aux_model(self, *, vision: bool = False) -> str:
"""Return a LIVE cheap-model id for auxiliary tasks, or "".
``default_aux_model`` is a hardcoded id in source, so it rots: when the
provider retires that model every auxiliary call spends a round-trip
404ing before the retry net catches it. Providers that publish a
machine-readable recommendation should override this and query it, so
the cheap tier tracks the upstream catalog instead of a constant a human
has to remember to bump.
Contract: cheap to call (implementations must cache — this runs on
client-resolution paths), never raises, and returns "" when it has no
answer so the caller falls through to ``default_aux_model``.
"""
return ""
def get_hostname(self) -> str:
"""Return the provider's base hostname for URL-based detection.
+58
View File
@@ -4488,3 +4488,61 @@ class TestAutoRoutedProviderProfileHooks:
assert relay.call_args_list[1].args[1]["extra_headers"] == {
"Authorization": "Bearer token-2",
}
class TestFastModelTier:
"""The titling fast tier: rot-proof resolution, scoped to titling only."""
def test_catalog_match_prefers_rolling_alias_over_pinned_id(self):
"""A "-latest" alias wins: it is the only id that cannot go stale."""
from agent import auxiliary_client as ac
catalog = {
"z-ai/glm-5.2": {},
"openai/gpt-5.4-mini": {},
"~openai/gpt-mini-latest": {},
"stepfun/step-3.7-flash:free": {},
}
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
assert ac._fast_model_from_catalog("nous") == "~openai/gpt-mini-latest"
def test_catalog_match_skips_reasoning_batch_and_embedding_lookalikes(self):
"""Substring matching must not pick a thinker, a queue, or an encoder."""
from agent import auxiliary_client as ac
catalog = {
"openai/o3-mini": {},
"openai/gpt-5.4-mini:batch": {},
"sentence-transformers/all-minilm-l6-v2": {},
"google/gemini-3.6-flash": {},
}
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
assert ac._fast_model_from_catalog("nous") == "google/gemini-3.6-flash"
def test_falls_back_to_curated_default_when_catalog_unavailable(self):
"""An offline catalog degrades to the provider's pinned default."""
from agent import auxiliary_client as ac
with patch.object(ac, "_fast_model_from_catalog", return_value=""):
assert (
ac._get_aux_model_for_provider("anthropic", prefer_fast=True)
== ac._get_aux_model_for_provider("anthropic")
)
def test_fast_tier_is_opt_in(self):
"""Without prefer_fast the resolver must not touch the live catalog."""
from agent import auxiliary_client as ac
with patch.object(ac, "_fast_model_from_catalog") as spy:
ac._get_aux_model_for_provider("nous")
spy.assert_not_called()
def test_only_titling_is_in_the_fast_tier(self):
"""Compression/vision/search keep 'auto means my chat model'."""
from agent.auxiliary_client import _FAST_MODEL_TASKS
assert "title_generation" in _FAST_MODEL_TASKS
overlap = {"compression", "vision", "web_extract"}.intersection(
_FAST_MODEL_TASKS
)
assert not overlap