perf(sessions): resolve the titling model from the provider's live catalog
Titling ran on the user's main chat model, so a five-word title was billed to a frontier reasoning model and inherited its latency. Pinning a cheap model id instead just moves the problem: the hardcoded default was already dead upstream and every call paid a 404 before the retry net caught it. Match model FAMILIES against the provider's live /v1/models catalog, preferring rolling '-latest' aliases where a provider publishes them, and order the families by measured latency. Nothing to bump when a provider ships a new mini/flash/haiku. Opt-in per task, so compression, vision, and search keep 'auto means my chat model'.
This commit is contained in:
+144
-9
@@ -696,23 +696,131 @@ def _compression_threshold_for_model(
|
||||
return _CODEX_SPARK_COMPACTION_THRESHOLD
|
||||
return None
|
||||
|
||||
# Default auxiliary models for direct API-key providers (cheap/fast for side tasks)
|
||||
def _get_aux_model_for_provider(provider_id: str) -> str:
|
||||
"""Return the cheap auxiliary model for a provider.
|
||||
# Model-family priority for the auxiliary "fast tier", fastest first.
|
||||
#
|
||||
# Matched as substrings against the provider's LIVE /v1/models catalog rather
|
||||
# than pinned as exact ids, because exact ids rot: a hardcoded
|
||||
# "google/gemini-3-flash" kept 404ing here once Nous dropped it upstream, and
|
||||
# every aux call paid a wasted round-trip before the retry net caught it.
|
||||
# Families outlive their version numbers, so a new mini/flash/haiku release is
|
||||
# picked up with no source edit.
|
||||
#
|
||||
# Rolling "-latest" aliases come first where a provider publishes them (Nous
|
||||
# serves ~openai/gpt-mini-latest, ~google/gemini-flash-latest, …): they are the
|
||||
# only ids that are structurally rot-proof.
|
||||
#
|
||||
# Order is measured, not guessed — p50 on a real titling prompt against the
|
||||
# Nous catalog: gpt-mini-latest 1.40s, claude-haiku-latest 1.55s,
|
||||
# gemini-flash-latest 2.13s, step-3.7-flash 7.84s, grok-4.1-fast 8.05s. So the
|
||||
# first family a provider actually serves is also the fastest it can offer.
|
||||
_FAST_MODEL_FAMILIES: tuple = (
|
||||
"gpt-mini-latest",
|
||||
"gpt-nano-latest",
|
||||
"claude-haiku-latest",
|
||||
"gemini-flash-latest",
|
||||
"gpt-5.4-nano",
|
||||
"gpt-5.4-mini",
|
||||
"gpt-5-mini",
|
||||
"haiku-4.5",
|
||||
"gemini-3.6-flash",
|
||||
"flash-lite",
|
||||
"-nano",
|
||||
"-mini",
|
||||
"-flash",
|
||||
"haiku",
|
||||
)
|
||||
|
||||
Reads from ProviderProfile.default_aux_model first, falling back to the
|
||||
legacy hardcoded dict for providers that predate the profiles system.
|
||||
# Substrings that disqualify an otherwise-matching id. Reasoning variants
|
||||
# ("o3-mini", "gpt-5.4-mini-thinking") think before answering, which is the
|
||||
# opposite of what a titler wants; ":batch" is an async queue, not a live
|
||||
# endpoint; embedding models ("all-minilm") match "-mini" but aren't chat
|
||||
# models at all; ":free" tiers are heavily rate-limited and measured slowest.
|
||||
_FAST_MODEL_EXCLUDE: tuple = (
|
||||
"thinking", "reason", "-r1", "minilm", ":batch", ":free",
|
||||
"o1-", "o3-", "o4-", "codex", "audio", "-vl", "embed",
|
||||
)
|
||||
|
||||
|
||||
def _fast_model_from_catalog(provider_id: str) -> str:
|
||||
"""Pick the fastest small model the provider ACTUALLY serves right now.
|
||||
|
||||
Reads the provider's live (cached) ``/v1/models`` catalog and returns the
|
||||
first ``_FAST_MODEL_FAMILIES`` match. Returns "" when the catalog is
|
||||
unavailable or holds no small model, so the caller falls through to the
|
||||
provider's curated default. Never raises and never blocks on a cold
|
||||
network path — the underlying fetch is memory+disk cached with a
|
||||
last-known-good fallback.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.models import fetch_models_with_pricing
|
||||
from providers import get_provider_profile
|
||||
_p = get_provider_profile(provider_id)
|
||||
if _p and _p.default_aux_model:
|
||||
return _p.default_aux_model
|
||||
|
||||
profile = get_provider_profile(provider_id)
|
||||
base_url = str(getattr(profile, "base_url", "") or "").rstrip("/")
|
||||
if not base_url:
|
||||
return ""
|
||||
# fetch_models_with_pricing appends its own /v1/models.
|
||||
if base_url.endswith("/v1"):
|
||||
base_url = base_url[:-3]
|
||||
catalog = fetch_models_with_pricing(base_url=base_url, timeout=3.0) or {}
|
||||
except Exception:
|
||||
logger.debug("Fast-model catalog lookup failed for %s", provider_id, exc_info=True)
|
||||
return ""
|
||||
|
||||
ids = sorted(str(m) for m in catalog)
|
||||
for family in _FAST_MODEL_FAMILIES:
|
||||
for model_id in ids:
|
||||
lowered = model_id.lower()
|
||||
if family in lowered and not any(x in lowered for x in _FAST_MODEL_EXCLUDE):
|
||||
return model_id
|
||||
return ""
|
||||
|
||||
|
||||
# Default auxiliary models for direct API-key providers (cheap/fast for side tasks)
|
||||
def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False) -> str:
|
||||
"""Return the cheap auxiliary model for a provider.
|
||||
|
||||
Resolution ladder, fastest-and-most-live first:
|
||||
|
||||
1. ``prefer_fast`` only — a family match against the provider's LIVE
|
||||
``/v1/models`` catalog, preferring rolling ``-latest`` aliases. This is
|
||||
both rot-proof and latency-ordered.
|
||||
2. ``prefer_fast`` only — the provider's own recommendation hook
|
||||
(``ProviderProfile.resolve_aux_model``). Live, but tuned for *quality*
|
||||
on long-context side tasks (Nous returns its compaction pick), so it
|
||||
ranks below the catalog match for latency-critical work.
|
||||
3. ``ProviderProfile.default_aux_model`` — curated, hardcoded, may rot.
|
||||
4. The legacy hardcoded dict, for providers predating the profiles system.
|
||||
|
||||
``prefer_fast`` is opt-in so this only changes latency-critical tasks
|
||||
(titling). Every other auxiliary caller keeps the existing static
|
||||
behaviour and its cache keys.
|
||||
"""
|
||||
profile = None
|
||||
try:
|
||||
from providers import get_provider_profile
|
||||
profile = get_provider_profile(provider_id)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
if prefer_fast:
|
||||
catalog_pick = _fast_model_from_catalog(provider_id)
|
||||
if catalog_pick:
|
||||
return catalog_pick
|
||||
if profile is not None:
|
||||
try:
|
||||
live = profile.resolve_aux_model()
|
||||
if live:
|
||||
return live
|
||||
except Exception:
|
||||
logger.debug("resolve_aux_model failed for %s", provider_id, exc_info=True)
|
||||
|
||||
if profile is not None and profile.default_aux_model:
|
||||
return profile.default_aux_model
|
||||
return _API_KEY_PROVIDER_AUX_MODELS_FALLBACK.get(provider_id, "")
|
||||
|
||||
|
||||
|
||||
# Fallback for providers not yet migrated to ProviderProfile.default_aux_model,
|
||||
# plus providers we intentionally keep pinned here (e.g. Anthropic predates
|
||||
# profiles). New providers should set default_aux_model on their profile instead.
|
||||
@@ -740,6 +848,13 @@ _API_KEY_PROVIDER_AUX_MODELS_FALLBACK: Dict[str, str] = {
|
||||
# can still use this dict directly. Kept in sync with _FALLBACK above.
|
||||
_API_KEY_PROVIDER_AUX_MODELS: Dict[str, str] = _API_KEY_PROVIDER_AUX_MODELS_FALLBACK
|
||||
|
||||
# Auxiliary tasks that prefer the provider's fast/cheap model over the user's
|
||||
# main chat model when running in "auto" mode. Restricted to tasks where
|
||||
# latency is user-visible and the output is short enough that a small model
|
||||
# matches a frontier one. Every other task keeps "auto = my chat model".
|
||||
_FAST_MODEL_TASKS: frozenset = frozenset({"title_generation"})
|
||||
|
||||
|
||||
# Vision-specific model overrides for direct providers.
|
||||
# When the user's main provider has a dedicated vision/multimodal model that
|
||||
# differs from their main chat model, map it here. The vision auto-detect
|
||||
@@ -5420,7 +5535,6 @@ def _resolve_auto_route(
|
||||
runtime_api_key = runtime.get("api_key", "")
|
||||
runtime_api_mode = str(runtime.get("api_mode") or "")
|
||||
|
||||
|
||||
# ── Warn once if OPENAI_BASE_URL is set but config.yaml uses a named
|
||||
# provider (not 'custom'). This catches the common "env poisoning"
|
||||
# scenario where a user switches providers via `hermes model` but the
|
||||
@@ -5450,6 +5564,25 @@ def _resolve_auto_route(
|
||||
main_provider = str(runtime_provider or _read_main_provider() or "")
|
||||
main_model = str(runtime_model or _read_main_model() or "")
|
||||
|
||||
# Latency-critical tasks prefer the provider's registered fast model over
|
||||
# the main chat model. Titling is the only such task: it names a visible
|
||||
# sidebar row, produces ~8 tokens, and running it on a frontier reasoning
|
||||
# model costs seconds per new session. Every comparable tool routes titling
|
||||
# to a small tier (Claude Code → Haiku, OpenCode → small_model, Zed →
|
||||
# default_fast_model, OpenClaw → utilityModel). An explicit
|
||||
# auxiliary.<task>.model in config.yaml still wins — this only redirects
|
||||
# the "auto" default, and only when the provider registered a cheap model.
|
||||
# Every other aux task keeps the "auto means my chat model" contract
|
||||
# documented above: this does NOT change compression, vision, or search.
|
||||
if task in _FAST_MODEL_TASKS and main_provider and main_provider not in {"auto", ""}:
|
||||
fast_model = _get_aux_model_for_provider(main_provider, prefer_fast=True)
|
||||
if fast_model and fast_model != main_model:
|
||||
logger.debug(
|
||||
"Auxiliary task %s: preferring fast model %s over main model %s",
|
||||
task, fast_model, main_model,
|
||||
)
|
||||
main_model = fast_model
|
||||
|
||||
# MoA virtual provider: the "model" is a preset name (e.g. "opus-gpt") and
|
||||
# there is no real "moa" HTTP endpoint, so resolving an aux client against
|
||||
# provider="moa"/model=<preset> sends the preset name as the model id and
|
||||
@@ -8776,6 +8909,7 @@ def _call_llm_impl(
|
||||
api_key=resolved_api_key,
|
||||
api_mode=resolved_api_mode,
|
||||
main_runtime=main_runtime,
|
||||
task=task,
|
||||
)
|
||||
effective_provider = _effective_provider_for_client(
|
||||
client, resolved_provider,
|
||||
@@ -9554,6 +9688,7 @@ async def _async_call_llm_impl(
|
||||
api_key=resolved_api_key,
|
||||
api_mode=resolved_api_mode,
|
||||
main_runtime=main_runtime,
|
||||
task=task,
|
||||
)
|
||||
effective_provider = _effective_provider_for_client(
|
||||
client, resolved_provider,
|
||||
|
||||
@@ -11,6 +11,22 @@ from providers.base import ProviderProfile
|
||||
class NousProfile(ProviderProfile):
|
||||
"""Nous Portal — product tags, reasoning with Nous-specific omission."""
|
||||
|
||||
def resolve_aux_model(self, *, vision: bool = False) -> str:
|
||||
"""Ask the Portal which cheap model it currently recommends.
|
||||
|
||||
``/api/nous/recommended-models`` is the authoritative, tier-aware
|
||||
source (free vs paid), so the auxiliary fast tier tracks the live
|
||||
catalog instead of a hardcoded id that 404s the day Nous retires it.
|
||||
The underlying fetch is memory- and disk-cached with a last-known-good
|
||||
fallback, so this is cheap to call and safe offline.
|
||||
"""
|
||||
try:
|
||||
from hermes_cli.models import get_nous_recommended_aux_model
|
||||
|
||||
return get_nous_recommended_aux_model(vision=vision) or ""
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
def build_extra_body(
|
||||
self, *, session_id: str | None = None, **context
|
||||
) -> dict[str, Any]:
|
||||
|
||||
@@ -101,6 +101,22 @@ class ProviderProfile:
|
||||
|
||||
# ── Hooks (override in subclass for complex providers) ───
|
||||
|
||||
def resolve_aux_model(self, *, vision: bool = False) -> str:
|
||||
"""Return a LIVE cheap-model id for auxiliary tasks, or "".
|
||||
|
||||
``default_aux_model`` is a hardcoded id in source, so it rots: when the
|
||||
provider retires that model every auxiliary call spends a round-trip
|
||||
404ing before the retry net catches it. Providers that publish a
|
||||
machine-readable recommendation should override this and query it, so
|
||||
the cheap tier tracks the upstream catalog instead of a constant a human
|
||||
has to remember to bump.
|
||||
|
||||
Contract: cheap to call (implementations must cache — this runs on
|
||||
client-resolution paths), never raises, and returns "" when it has no
|
||||
answer so the caller falls through to ``default_aux_model``.
|
||||
"""
|
||||
return ""
|
||||
|
||||
def get_hostname(self) -> str:
|
||||
"""Return the provider's base hostname for URL-based detection.
|
||||
|
||||
|
||||
@@ -4488,3 +4488,61 @@ class TestAutoRoutedProviderProfileHooks:
|
||||
assert relay.call_args_list[1].args[1]["extra_headers"] == {
|
||||
"Authorization": "Bearer token-2",
|
||||
}
|
||||
|
||||
|
||||
class TestFastModelTier:
|
||||
"""The titling fast tier: rot-proof resolution, scoped to titling only."""
|
||||
|
||||
def test_catalog_match_prefers_rolling_alias_over_pinned_id(self):
|
||||
"""A "-latest" alias wins: it is the only id that cannot go stale."""
|
||||
from agent import auxiliary_client as ac
|
||||
|
||||
catalog = {
|
||||
"z-ai/glm-5.2": {},
|
||||
"openai/gpt-5.4-mini": {},
|
||||
"~openai/gpt-mini-latest": {},
|
||||
"stepfun/step-3.7-flash:free": {},
|
||||
}
|
||||
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
|
||||
assert ac._fast_model_from_catalog("nous") == "~openai/gpt-mini-latest"
|
||||
|
||||
def test_catalog_match_skips_reasoning_batch_and_embedding_lookalikes(self):
|
||||
"""Substring matching must not pick a thinker, a queue, or an encoder."""
|
||||
from agent import auxiliary_client as ac
|
||||
|
||||
catalog = {
|
||||
"openai/o3-mini": {},
|
||||
"openai/gpt-5.4-mini:batch": {},
|
||||
"sentence-transformers/all-minilm-l6-v2": {},
|
||||
"google/gemini-3.6-flash": {},
|
||||
}
|
||||
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
|
||||
assert ac._fast_model_from_catalog("nous") == "google/gemini-3.6-flash"
|
||||
|
||||
def test_falls_back_to_curated_default_when_catalog_unavailable(self):
|
||||
"""An offline catalog degrades to the provider's pinned default."""
|
||||
from agent import auxiliary_client as ac
|
||||
|
||||
with patch.object(ac, "_fast_model_from_catalog", return_value=""):
|
||||
assert (
|
||||
ac._get_aux_model_for_provider("anthropic", prefer_fast=True)
|
||||
== ac._get_aux_model_for_provider("anthropic")
|
||||
)
|
||||
|
||||
def test_fast_tier_is_opt_in(self):
|
||||
"""Without prefer_fast the resolver must not touch the live catalog."""
|
||||
from agent import auxiliary_client as ac
|
||||
|
||||
with patch.object(ac, "_fast_model_from_catalog") as spy:
|
||||
ac._get_aux_model_for_provider("nous")
|
||||
spy.assert_not_called()
|
||||
|
||||
def test_only_titling_is_in_the_fast_tier(self):
|
||||
"""Compression/vision/search keep 'auto means my chat model'."""
|
||||
from agent.auxiliary_client import _FAST_MODEL_TASKS
|
||||
|
||||
assert "title_generation" in _FAST_MODEL_TASKS
|
||||
overlap = {"compression", "vision", "web_extract"}.intersection(
|
||||
_FAST_MODEL_TASKS
|
||||
)
|
||||
assert not overlap
|
||||
|
||||
Reference in New Issue
Block a user