simplify(compat): models — drop 52 re-exports from hermes_cli.models, repoint 16 callers + 41 test files

This commit is contained in:
Teknium
2026-09-03 13:48:49 -07:00
parent e3ab65fe80
commit 7b8c11bcf7
63 changed files with 396 additions and 458 deletions
+1 -1
View File
@@ -29,7 +29,7 @@ def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, st
from hermes_cli.model_switch import _declared_model_ids, _entry_models_discovered, _models_config_is_allowlist
from hermes_cli.model_switch_providers import _NativePickerModelList, _fetch_picker_live_models
from hermes_cli.model_switch_providers import _discover_flag
from hermes_cli.models import should_use_ollama_native_catalog
from hermes_cli.models_local import should_use_ollama_native_catalog
from hermes_cli.providers import custom_provider_slug
except ImportError:
return []
+5 -5
View File
@@ -689,7 +689,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
is_nous = provider_id.strip().lower() == "nous"
try:
from hermes_cli.auth import resolve_api_key_provider_credentials
from hermes_cli.models import fetch_models_with_pricing
from hermes_cli.models_pricing import fetch_models_with_pricing
from providers import get_provider_profile
# Most /v1/models endpoints are authenticated; an anonymous 401 would read as "no small
# model" and pin the curated default forever.
@@ -704,7 +704,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
if not api_key and is_nous:
# Nous is OAuth (resolver raises); anonymous reads return the full catalog.
try:
from hermes_cli.models import _resolve_nous_pricing_credentials
from hermes_cli.models_pricing import _resolve_nous_pricing_credentials
api_key, base_url = _resolve_nous_pricing_credentials()
except Exception:
logger.debug("No Nous credentials for catalog", exc_info=True)
@@ -719,7 +719,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
# policy-catalog expiry.
_nous_kwargs = {}
if is_nous:
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS
_nous_kwargs = {"include_sale_original": True, "cache_ttl_seconds": _NOUS_CATALOG_TTL_SECONDS}
catalog = fetch_models_with_pricing(
api_key=api_key or None, base_url=base_url, timeout=3.0, **_nous_kwargs) or {}
@@ -730,7 +730,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
if is_nous:
# Narrow catalog ids by org policy, as the pickers do.
try:
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy
ids = restrict_to_nous_policy(ids, nous_policy_allowed_ids())
except Exception:
logger.debug("Nous policy filter unavailable", exc_info=True)
@@ -770,7 +770,7 @@ def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False)
# let the caller keep the main model.
if picked and provider_id.strip().lower() == "nous":
try:
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy
allowed = nous_policy_allowed_ids()
if allowed and not restrict_to_nous_policy([picked], allowed):
return ""
+2 -1
View File
@@ -131,7 +131,8 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
if not base_url:
return False
try:
from hermes_cli.models import _is_model_free, peek_cached_pricing
from hermes_cli.models import _is_model_free
from hermes_cli.models_pricing import peek_cached_pricing
pricing = peek_cached_pricing(base_url) # owns the /v1-suffix and auth-state key details
return bool(pricing) and _is_model_free(model, pricing)
+3 -3
View File
@@ -71,7 +71,7 @@ class ReasoningParamsMixin:
# Live-catalog metadata first (OpenRouter /v1/models supported_parameters) — the static prefix
# allowlist repeatedly went stale one vendor at a time. Unknown falls back to the static list.
try:
from hermes_cli.models import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async
caps = openrouter_model_reasoning_capabilities(self.model)
if caps is None:
warm_openrouter_reasoning_caps_async() # cache cold — warm in the background, never block
@@ -85,7 +85,7 @@ class ReasoningParamsMixin:
def _lmstudio_reasoning_options_cached(self) -> list[str]:
"""LM Studio's published reasoning ``allowed_options`` (gate + clamp so toggle models don't 400 on ``high``)."""
try:
from hermes_cli.models import lmstudio_model_reasoning_options
from hermes_cli.models_local import lmstudio_model_reasoning_options
except Exception:
return []
return _cached_probe(self, "_lm_reasoning_opts_cache", lmstudio_model_reasoning_options, [], bool)
@@ -93,7 +93,7 @@ class ReasoningParamsMixin:
def _ollama_supports_thinking_cached(self) -> bool:
"""True only if Ollama's ``/api/show`` declares the ``thinking`` capability."""
try:
from hermes_cli.models import ollama_model_supports_thinking
from hermes_cli.models_local import ollama_model_supports_thinking
except Exception:
return False
return bool(_cached_probe(self, "_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None))
+1 -1
View File
@@ -59,7 +59,7 @@ class _ModelPickerRows:
self, all_models: List[str], pricing: Optional[Dict[str, Dict[str, str]]], *,
current_model: str, sale_chrome: bool,
) -> None:
from hermes_cli.models import _format_price_per_mtok, compute_sale_discount
from hermes_cli.models_pricing import _format_price_per_mtok, compute_sale_discount
self.current_model = current_model
self.has_pricing = bool(pricing and any(pricing.get(m) for m in all_models))
# Leave room for a leading "★ " on sale rows (Nous only).
+11 -3
View File
@@ -1328,9 +1328,17 @@ def _pick_nous_model_after_login(
if not isinstance(runtime_key, str) or not runtime_key:
raise _nous_err("No runtime API key available to fetch models", "invalid_token")
from hermes_cli.models import (
get_curated_nous_model_ids, get_pricing_for_provider, check_nous_free_tier,
partition_nous_models_by_tier, nous_policy_allowed_ids, restrict_to_nous_policy,
union_with_portal_free_recommendations, union_with_portal_paid_recommendations)
get_curated_nous_model_ids,
check_nous_free_tier,
partition_nous_models_by_tier,
union_with_portal_free_recommendations,
union_with_portal_paid_recommendations,
)
from hermes_cli.models_pricing import (
get_pricing_for_provider,
nous_policy_allowed_ids,
restrict_to_nous_policy,
)
model_ids = get_curated_nous_model_ids()
_portal = auth_state.get("portal_base_url", "")
print()
+14 -6
View File
@@ -249,9 +249,11 @@ def _reasoning_catalog_reader(slug: str):
"""Per-model reasoning-capability reader for aggregators that publish one. Cache-only — the picker
must never block on HTTP; a cold cache warms in the background and reports no restriction until then."""
try:
from hermes_cli.models import (
nous_model_reasoning_capabilities, openrouter_model_reasoning_capabilities,
warm_nous_reasoning_caps_async, warm_openrouter_reasoning_caps_async,
from hermes_cli.models_reasoning_caps import (
nous_model_reasoning_capabilities,
openrouter_model_reasoning_capabilities,
warm_nous_reasoning_caps_async,
warm_openrouter_reasoning_caps_async,
)
except Exception:
return None
@@ -573,9 +575,15 @@ def _apply_pricing(rows: list[dict], *, force_fresh_nous_tier: bool = False, cac
``free_tier`` (account is free-tier) and ``unavailable_models`` (paid models a free user can't pick).
``cached_only`` never hits the network: unknown Nous entitlement fails closed (``free_tier_pending``,
all models locked) and missing pricing is marked ``pricing_pending``."""
from hermes_cli.models_pricing import (
_format_price_per_mtok,
compute_sale_discount,
get_pricing_for_provider,
)
from hermes_cli.models import (
_format_price_per_mtok, check_nous_free_tier, compute_sale_discount, get_cached_nous_free_tier,
get_pricing_for_provider, partition_nous_models_by_tier,
check_nous_free_tier,
get_cached_nous_free_tier,
partition_nous_models_by_tier,
)
nous_free_tier: Optional[bool] = None # resolved once (cached in models.py for the TTL window)
@@ -686,7 +694,7 @@ def _prewarm_pricing_async(
"""Warm picker pricing caches without delaying the current payload (one worker per
profile + endpoint scope; a live worker is reused)."""
from hermes_constants import hermes_home_key
from hermes_cli.models import pricing_cache_scope
from hermes_cli.models_pricing import pricing_cache_scope
slugs = {str(row.get("slug") or "").lower() for row in rows if row.get("slug")}
endpoint_scope = tuple(sorted(
+3 -2
View File
@@ -260,7 +260,7 @@ def _aux_flow_provider_model(task: str, provider_slug: str, curated_models: list
current_model: str = "") -> None:
"""Prompt for a model under an already-authenticated provider, save to aux."""
from hermes_cli.auth import _prompt_model_selection
from hermes_cli.models import get_pricing_for_provider
from hermes_cli.models_pricing import get_pricing_for_provider
display_name = _aux_task_display_name(task)
try:
pricing = get_pricing_for_provider(provider_slug) or {}
@@ -780,7 +780,8 @@ def _build_provider_picker_rows(config: dict, active: str, provider_labels: dict
fold into display groups (PROVIDER_GROUPS): a group row's ``members`` drive a sub-picker, leaf
rows have ``members == []``; saved custom providers and trailing actions stay flat. Honors
``model_catalog.excluded_providers`` (slug or alias, case-insensitive) like the gateway/TUI."""
from hermes_cli.models import CANONICAL_PROVIDERS, _PROVIDER_ALIASES, group_providers, provider_group_for_slug
from hermes_cli.models import CANONICAL_PROVIDERS, _PROVIDER_ALIASES
from hermes_cli.models_catalog_static import group_providers, provider_group_for_slug
canonical_descs = {p.slug: p.tui_desc for p in CANONICAL_PROVIDERS}
_cli_excluded = {
str(p).strip().lower()
+6 -3
View File
@@ -207,9 +207,12 @@ def _discover_named_custom_models(provider_info: dict, api_key: str, configured_
"""Live catalog probe for a named custom endpoint (native ``/api/tags`` for Ollama).
Returns ``(models, native_catalog_empty)``; persists the live catalog as a side effect."""
from hermes_cli.config import normalize_extra_headers
from hermes_cli.models import (
fetch_api_models, fetch_ollama_local_models, _get_ollama_native_headers, _normalize_openai_base_url,
should_use_ollama_native_catalog)
from hermes_cli.models import fetch_api_models, _get_ollama_native_headers
from hermes_cli.models_local import (
fetch_ollama_local_models,
_normalize_openai_base_url,
should_use_ollama_native_catalog,
)
name, base_url = provider_info["name"], provider_info["base_url"]
api_mode = provider_info.get("api_mode", "")
+5 -4
View File
@@ -976,7 +976,7 @@ def _apply_direct_alias_endpoint(st: "_Switch", da: DirectAlias) -> None:
own endpoint decides: its declared key; else the session key only for the SAME ORIGIN; else a
fresh resolution against the alias base_url (env-key fallbacks are host-gated: OLLAMA_API_KEY
resolves for ollama.com, OPENROUTER_API_KEY never reaches an unrelated host)."""
from hermes_cli.models import _same_ollama_native_root
from hermes_cli.models_local import _same_ollama_native_root
from hermes_cli.runtime_provider import resolve_runtime_provider
alias_key = direct_alias_api_key(da)
same_host = _may_reuse_session_credential(st.base_url, da.base_url)
@@ -1282,12 +1282,12 @@ def _creds_for_current_provider(st: _Switch) -> None:
"""Credentials when staying on the current provider. Mid-session ``/model <name>`` on a local
Ollama-compatible endpoint keeps the endpoint in use; re-resolving bare ``custom`` from config
can fall through to an unrelated default provider."""
from hermes_cli.models import _get_ollama_request_headers, _same_ollama_native_root
from hermes_cli.models_local import _get_ollama_request_headers, _same_ollama_native_root
keep_current_ollama_endpoint = False
ollama_headers: dict[str, str] = {}
if st.current_provider == "custom" and st.current_base_url:
try:
from hermes_cli.models import should_use_ollama_native_catalog
from hermes_cli.models_local import should_use_ollama_native_catalog
ollama_headers = _get_ollama_request_headers()
_, configured_ollama_base = _ollama_configured_base()
# Provider-level Ollama headers only belong to the configured native root; without
@@ -1343,7 +1343,8 @@ def _resolve_switch_credentials(st: _Switch) -> Optional[ModelSwitchResult]:
def _validate_switch(st: _Switch) -> Optional[ModelSwitchResult]:
"""COMMON PATH part 2: normalize the model name for the target provider, validate it, and
accept config-declared models the remote catalog lacks."""
from hermes_cli.models import _get_ollama_request_headers, validate_requested_model
from hermes_cli.models_local import _get_ollama_request_headers
from hermes_cli.models_validate import validate_requested_model
st.new_model = _resolve_named_custom_model_id(st.new_model, st.target_provider, st.custom_providers)
st.new_model = normalize_model_for_provider(st.new_model, st.target_provider)
+13 -7
View File
@@ -89,9 +89,12 @@ def _fetch_picker_live_models(
headers: dict[str, str] | None = None, timeout: float = 5.0,
api_mode: str | None = None) -> list[str] | None:
"""Fetch picker models with native Ollama and cached generic discovery."""
from hermes_cli.models import (
_get_ollama_native_headers, _normalize_openai_base_url, cached_fetch_api_models,
fetch_ollama_local_models, should_use_ollama_native_catalog)
from hermes_cli.models import _get_ollama_native_headers, cached_fetch_api_models
from hermes_cli.models_local import (
_normalize_openai_base_url,
fetch_ollama_local_models,
should_use_ollama_native_catalog,
)
candidate_headers = _get_ollama_native_headers(api_url, api_key=api_key)
@@ -395,9 +398,12 @@ def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list:
recommendation fetch still yields a policy-filtered curated list."""
model_ids = curated.get("nous", [])
try:
from hermes_cli.models_pricing import get_pricing_for_provider
from hermes_cli.models import (
get_pricing_for_provider, check_nous_free_tier, union_with_portal_free_recommendations,
union_with_portal_paid_recommendations)
check_nous_free_tier,
union_with_portal_free_recommendations,
union_with_portal_paid_recommendations,
)
from hermes_cli.auth import get_provider_auth_state
pricing = get_pricing_for_provider("nous") or {}
try:
@@ -411,7 +417,7 @@ def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list:
except Exception:
pass
try:
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy
model_ids = restrict_to_nous_policy(model_ids, nous_policy_allowed_ids(), rescue_empty=True)
except Exception:
pass
@@ -996,7 +1002,7 @@ def _build_curated_lists(current_provider: str, current_base_url: str, current_m
# unreachable, fall back to the current model so the picker still shows something offline.
is_current_lmstudio = current_provider.strip().lower() == "lmstudio"
if "lmstudio" not in curated and (os.environ.get("LM_API_KEY") or os.environ.get("LM_BASE_URL") or is_current_lmstudio):
from hermes_cli.models import fetch_lmstudio_models
from hermes_cli.models_local import fetch_lmstudio_models
from hermes_cli.auth import AuthError
lm_base = (
os.environ.get("LM_BASE_URL")
+6 -65
View File
@@ -26,12 +26,10 @@ if TYPE_CHECKING:
from hermes_cli import __version__ as _HERMES_VERSION
from hermes_cli.urllib_security import open_credentialed_url
from hermes_cli.models_catalog_static import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
from hermes_cli.models_catalog_static import (
CANONICAL_PROVIDERS,
OPENROUTER_MODELS,
PREFERRED_SILENT_DEFAULT_MODEL,
PROVIDER_GROUPS,
ProviderEntry,
VERCEL_AI_GATEWAY_MODELS,
_AGGREGATOR_PROVIDERS,
_AZURE_FOUNDRY_RESPONSES_PREFIXES,
@@ -47,78 +45,21 @@ from hermes_cli.models_catalog_static import ( # noqa: F401 (re-exported; test
_PROVIDER_MODELS,
_PROVIDER_RETIRED_ALIASES,
_SILENT_DEFAULT_PROVIDERS,
_xai_finalize_catalog,
group_providers,
provider_group_for_slug)
from hermes_cli.models_reasoning_caps import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
_xai_finalize_catalog)
from hermes_cli.models_reasoning_caps import (
_OPENROUTER_CATALOG_URL,
_read_reasoning_caps_disk,
_reasoning_caps_disk_path,
_save_reasoning_caps_disk,
_seed_reasoning_caps,
nous_catalog_url,
nous_model_reasoning_capabilities,
# Live-catalog metadata first (ported from PrimeIntellect-ai/prime-agent#1258): OpenRouter's /v1/models
# entries advertise reasoning support via supported_parameters + a reasoning object, which covers every
# routed vendor without a hand-maintained prefix list. The static prefix allowlist below repeatedly went
# stale one vendor at a time (nvidia/ missing → #75386; same class as tencent/, xiaomi/ additions before
# it) — metadata makes new vendors work without a code change. One catalog fetch per process, cached;
# unknown (catalog unreachable / unlisted model) falls back to the static list.
openrouter_model_reasoning_capabilities,
parse_openrouter_reasoning_capabilities,
refresh_reasoning_caps_async,
warm_nous_reasoning_caps_async,
warm_openrouter_reasoning_caps_async)
from hermes_cli.models_local import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
LMStudioLoadResult,
_seed_reasoning_caps)
from hermes_cli.models_local import (
_OLLAMA_LOCAL_MODELS_CACHE,
_OLLAMA_LOCAL_MODELS_CACHE_TTL,
_OLLAMA_LOCAL_PROBE_FAILURE_CACHE,
_OLLAMA_LOCAL_PROBE_REACHABLE,
_get_ollama_base_url,
_get_ollama_native_headers,
_get_ollama_request_headers,
_lmstudio_fetch_raw_models,
_lmstudio_server_root,
_normalize_openai_base_url,
_ollama_local_catalog,
_ollama_probe_cache_key,
_root_for_ollama_native_api,
_same_ollama_native_root,
_strip_ollama_cloud_suffix,
ensure_lmstudio_model_loaded,
fetch_lmstudio_models,
fetch_ollama_cloud_models,
fetch_ollama_local_models,
lmstudio_model_reasoning_options,
ollama_model_supports_thinking,
probe_lmstudio_models,
probe_ollama_local_models,
should_use_ollama_native_catalog)
from hermes_cli.models_pricing import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
_FAILED_CATALOG_TTL_SECONDS,
_NOUS_CATALOG_TTL_SECONDS,
_NOUS_POLICY_APPEND_MAX,
_cache_catalog,
_cached_catalog,
_fetch_deepinfra_pricing,
_fetch_novita_pricing,
_format_price_per_mtok,
_pricing_auth_fingerprint,
_pricing_cache,
_pricing_cache_retry_after,
_resolve_nous_pricing_credentials,
compute_sale_discount,
fetch_ai_gateway_pricing,
fetch_models_with_pricing,
_pricing_provider_cache_keys,
get_cached_nous_inference_base_url,
get_pricing_for_provider,
nous_policy_allowed_ids,
peek_cached_pricing,
pricing_cache_scope,
restrict_to_nous_policy)
from hermes_cli.models_validate import validate_requested_model # noqa: F401 (re-exported)
fetch_ollama_cloud_models)
logger = logging.getLogger(__name__)
+1 -2
View File
@@ -1,8 +1,7 @@
"""Static provider/model catalog tables (data only — no network).
Curated per-provider model lists, the canonical provider registry, display groups and alias
maps. Split out of ``hermes_cli.models``, which re-imports every name so
``hermes_cli.models.<name>`` keeps resolving (and monkeypatching) as before.
maps. Split out of ``hermes_cli.models``.
"""
from __future__ import annotations
+4 -11
View File
@@ -2,9 +2,8 @@
Ollama (native ``/api/tags`` probe, request headers, base-url resolution), LM Studio
(``/api/v1/models``, load-on-demand) and Ollama Cloud (live + models.dev merged catalog with a
disk cache). Split out of ``hermes_cli.models``, which re-imports every name; origin helpers are
looked up on ``hermes_cli.models`` at call time so ``patch("hermes_cli.models.<name>")`` mocks
keep intercepting.
disk cache). Split out of ``hermes_cli.models``; helpers still defined there are looked up on
``hermes_cli.models`` at call time so ``patch("hermes_cli.models.<name>")`` mocks keep intercepting.
"""
from __future__ import annotations
@@ -96,7 +95,7 @@ def _get_ollama_base_url() -> str:
the active provider is ollama, or custom AND the endpoint actually serves ``/api/tags``
(otherwise the picker would probe an unrelated endpoint and hide the local catalog) →
``OLLAMA_HOST`` → Ollama's local default."""
from hermes_cli.models import _get_model_config_dict, should_use_ollama_native_catalog
from hermes_cli.models import _get_model_config_dict
configured = _configured_ollama_base_url()
if configured:
return configured
@@ -152,7 +151,6 @@ def _get_ollama_native_headers(base_url: Optional[str], *, api_key: Optional[str
"""Ollama credentials and headers for one endpoint origin. Configured headers apply only when
*base_url* shares the configured Ollama root; an explicit *api_key* replaces any configured
Authorization variant rather than inheriting it."""
from hermes_cli.models import _get_ollama_request_headers
configured_base = _configured_ollama_base_url()
explicit_key = str(api_key or "").strip()
configured_matches = bool(configured_base and base_url and _same_ollama_native_root(base_url, configured_base))
@@ -260,7 +258,6 @@ def fetch_ollama_local_models(
headers: Optional[dict[str, str]] = None,
) -> Optional[list[str]]:
"""Fetch local Ollama-compatible models, preserving probe failure as ``None``."""
from hermes_cli.models import probe_ollama_local_models
return probe_ollama_local_models(base_url, timeout, headers=headers)
@@ -295,7 +292,6 @@ def should_use_ollama_native_catalog(
Ollama's default port actually serves ``/api/tags``. (Bare ``ollama`` is normalized to
``custom`` elsewhere so runtime paths share the OpenAI client, but ``/api/tags`` is the
authoritative local list; other custom endpoints keep the ``/models`` probe.)"""
from hermes_cli.models import probe_ollama_local_models
requested = str(provider or "").strip().lower()
root = _root_for_ollama_native_api(base_url or "")
if root:
@@ -342,7 +338,7 @@ def _ollama_local_catalog(force_refresh: bool) -> list[str]:
"""Catalog for the raw ``ollama`` provider: native ``/api/tags`` when the endpoint is a real
Ollama server, else the OpenAI-style ``/v1/models`` of the configured gateway (incl. Ollama
Cloud)."""
from hermes_cli.models import _get_ollama_base_url, _get_ollama_native_headers, _get_provider_config_dict, fetch_api_models, fetch_ollama_local_models, should_use_ollama_native_catalog
from hermes_cli.models import _get_provider_config_dict, fetch_api_models
if force_refresh:
_OLLAMA_LOCAL_MODELS_CACHE.clear()
_OLLAMA_LOCAL_PROBE_FAILURE_CACHE.clear()
@@ -413,7 +409,6 @@ def _lmstudio_fetch_raw_models(
def _lmstudio_raw_models_or_none(api_key, base_url, timeout) -> Optional[list[dict]]:
"""``_lmstudio_fetch_raw_models`` with every failure (incl. AuthError) collapsed to None."""
from hermes_cli.models import _lmstudio_fetch_raw_models
try:
return _lmstudio_fetch_raw_models(api_key=api_key, base_url=base_url, timeout=timeout)
except Exception:
@@ -435,7 +430,6 @@ def probe_lmstudio_models(
"""Chat-capable LM Studio model keys — a valid empty list when the server is reachable but has
no non-embedding models; ``None`` on network errors, malformed responses, or bad base URLs.
Raises ``AuthError`` on HTTP 401/403 so token issues surface separately from reachability."""
from hermes_cli.models import _lmstudio_fetch_raw_models
raw_models = _lmstudio_fetch_raw_models(api_key=api_key, base_url=base_url, timeout=timeout)
if raw_models is None:
return None
@@ -457,7 +451,6 @@ def fetch_lmstudio_models(
) -> list[str]:
"""LM Studio chat-capable model keys; ``[]`` when unreachable/malformed. Raises ``AuthError`` on
HTTP 401/403 so callers can tell a wrong ``LM_API_KEY`` from an unreachable server."""
from hermes_cli.models import probe_lmstudio_models
return probe_lmstudio_models(api_key=api_key, base_url=base_url, timeout=timeout) or []
+7 -26
View File
@@ -2,9 +2,9 @@
OpenRouter-compatible ``/v1/models`` pricing fetch with a per-endpoint/per-credential cache,
Nous Portal sale chrome and org-policy filtering, and the Vercel AI Gateway / Novita / Fireworks /
DeepInfra pricing adapters. Split out of ``hermes_cli.models``, which re-imports every name;
origin helpers and the cache dicts are looked up on ``hermes_cli.models`` at call time so
``patch("hermes_cli.models.<name>")`` mocks keep intercepting.
DeepInfra pricing adapters. Split out of ``hermes_cli.models``; helpers still defined there are
looked up on ``hermes_cli.models`` at call time so ``patch("hermes_cli.models.<name>")`` mocks keep
intercepting.
"""
from __future__ import annotations
@@ -33,7 +33,6 @@ _pricing_cache_retry_after: dict[str, float] = {}
def _cached_catalog(cache_key: str) -> Optional[dict[str, dict[str, Any]]]:
"""The cached catalog for *cache_key*, or None to go fetch it."""
from hermes_cli.models import _pricing_cache, _pricing_cache_retry_after
cached = _pricing_cache.get(cache_key)
if cached is None:
return None
@@ -53,7 +52,6 @@ def _cache_catalog(
"""Cache a catalog result, giving an empty one an expiry. *ttl_seconds* expires a non-empty
result too — only for a catalog whose contents depend on server-side state the client cannot
observe (an org's model policy can change while a long-lived process holds the entry)."""
from hermes_cli.models import _pricing_cache, _pricing_cache_retry_after
_pricing_cache[cache_key] = result
if not result:
_pricing_cache_retry_after[cache_key] = time.monotonic() + _FAILED_CATALOG_TTL_SECONDS
@@ -84,7 +82,6 @@ def peek_cached_pricing(base_url: str) -> dict[str, dict[str, Any]]:
"""Pricing already cached for *base_url* (with or without ``/v1``), or ``{}``; never fetches.
Prefers an authenticated catalog, scanning newest first (callers hold no credential) and
skipping expired entries so a rotated credential does not answer from its predecessor's."""
from hermes_cli.models import _pricing_cache
root = _strip_v1((base_url or "").rstrip("/"))
authed_prefix = root + _PRICING_AUTH_KEY_PREFIX
for key in reversed(list(_pricing_cache)):
@@ -317,7 +314,6 @@ _NOUS_CATALOG_TTL_SECONDS = 300.0
def _fetch_nous_pricing(api_key: str, base_url: str, *, force_refresh: bool) -> dict[str, dict[str, Any]]:
"""Shared by pricing and policy lookups so both read one cache entry."""
from hermes_cli.models import fetch_models_with_pricing
return fetch_models_with_pricing(
api_key=api_key,
base_url=base_url,
@@ -332,7 +328,6 @@ def nous_policy_allowed_ids(*, force_refresh: bool = False) -> Optional[set[str]
which omits policy-blocked rows), or ``None`` to not filter: no policy (or a token too old to
say), an anonymous read (unfiltered catalog), or an empty read (a fetch failure, not an org
that may reach nothing)."""
from hermes_cli.models import _resolve_nous_pricing_credentials
try:
from hermes_cli.nous_account import nous_policy_present
@@ -373,12 +368,11 @@ def restrict_to_nous_policy(
def _remember_provider_cache_key(provider: str, cache_key: str) -> None:
from hermes_cli.models import _pricing_profile_key, _pricing_provider_cache_keys
from hermes_cli.models import _pricing_profile_key
_pricing_provider_cache_keys[(_pricing_profile_key(), provider)] = cache_key
def _fetch_openrouter_pricing(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
from hermes_cli.models import fetch_models_with_pricing
_remember_provider_cache_key("openrouter", _OPENROUTER_PRICING_BASE)
return fetch_models_with_pricing(
api_key=_resolve_openrouter_api_key(),
@@ -388,13 +382,11 @@ def _fetch_openrouter_pricing(*, force_refresh: bool = False) -> dict[str, dict[
def _fetch_ai_gateway_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
from hermes_cli.models import fetch_ai_gateway_pricing
_remember_provider_cache_key("ai-gateway", _ai_gateway_pricing_scope())
return fetch_ai_gateway_pricing(force_refresh=force_refresh)
def _fetch_novita_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
from hermes_cli.models import _fetch_novita_pricing
_remember_provider_cache_key("novita", _novita_pricing_scope())
return _fetch_novita_pricing(force_refresh=force_refresh)
@@ -405,7 +397,6 @@ def _fetch_fireworks_pricing_for_provider(*, force_refresh: bool = False) -> dic
def _fetch_nous_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
from hermes_cli.models import _resolve_nous_pricing_credentials
api_key, base_url = _resolve_nous_pricing_credentials()
if not base_url:
return {}
@@ -453,9 +444,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba
"""The current endpoint identity a provider's pricing cache is keyed on. Resolves local configuration
only, never fetches: picker prewarm single-flight uses it so an endpoint rotation can start a new
worker while the previous endpoint is still slow or unreachable."""
from hermes_cli.models import (
_deepinfra_catalog_url, _pricing_profile_key, _pricing_provider_cache_keys, normalize_provider,
)
from hermes_cli.models import _deepinfra_catalog_url, _pricing_profile_key, normalize_provider
normalized = normalize_provider(provider)
static = _STATIC_PRICING_SCOPES.get(normalized)
if static:
@@ -473,12 +462,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba
return env_base.rstrip("/").removesuffix("/v1")
if normalize_provider(current_provider) == "nous" and current_base_url:
return current_base_url.rstrip("/").removesuffix("/v1")
# Call-time lookup through the facade: tests (and plugins) patch
# ``hermes_cli.models.get_cached_nous_inference_base_url``; a bare module-global read here would
# silently bypass the patch and fall back to the default endpoint.
from hermes_cli.models import get_cached_nous_inference_base_url as _persisted_nous_base
persisted_base = _persisted_nous_base()
persisted_base = get_cached_nous_inference_base_url()
if persisted_base:
return persisted_base
return _pricing_provider_cache_keys.get((_pricing_profile_key(), normalized), _DEFAULT_NOUS_INFERENCE_BASE)
@@ -487,10 +471,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba
def _cached_only_pricing(normalized: str) -> dict[str, dict[str, str]]:
"""Process-resident pricing for *normalized* without any provider I/O."""
from hermes_cli.models import (
_deepinfra_catalog_cache, _deepinfra_catalog_url, _fetch_deepinfra_pricing, _pricing_profile_key,
_pricing_provider_cache_keys,
)
from hermes_cli.models import _deepinfra_catalog_cache, _deepinfra_catalog_url, _pricing_profile_key
if normalized == "deepinfra":
cache_key, _url = _deepinfra_catalog_url()
return _fetch_deepinfra_pricing() if cache_key in _deepinfra_catalog_cache else {}
+12 -5
View File
@@ -1,6 +1,6 @@
"""Per-model reasoning capabilities from OpenRouter-schema ``/v1/models`` catalogs.
Split out of ``hermes_cli.models``; every public/patched name is re-imported there. OpenRouter and
Split out of ``hermes_cli.models``. OpenRouter and
Nous Portal share one implementation parametrized by :class:`_CapsSource`; the per-source module
globals (``_openrouter_reasoning_caps_cache``, ``_nous_caps_disk_checked``, ...) stay defined on
``hermes_cli.models`` — tests reset them there — and are read/written by attribute name.
@@ -81,7 +81,7 @@ def _read_reasoning_caps_disk() -> dict[str, Any]:
def _load_reasoning_caps_disk(url: str) -> tuple[Optional[Caps], float]:
"""Return ``(caps, age_seconds)`` for *url*, or ``(None, 0.0)``."""
entry = _origin()._read_reasoning_caps_disk().get(url)
entry = _read_reasoning_caps_disk().get(url)
caps = entry.get("caps") if isinstance(entry, dict) else None
if not isinstance(caps, dict) or not caps:
return None, 0.0
@@ -96,7 +96,7 @@ def _save_reasoning_caps_disk(url: str, caps: Caps) -> None:
"""Merge *url*'s catalog into the shared disk mirror, atomically."""
from hermes_cli.models import _write_json_cache
try:
data = _origin()._read_reasoning_caps_disk()
data = _read_reasoning_caps_disk()
data[url] = {"ts": time.time(), "caps": caps}
_write_json_cache(_reasoning_caps_disk_path(), data, indent=0, separators=(",", ":"))
except Exception as exc:
@@ -250,16 +250,23 @@ _OPENROUTER_CAPS = _CapsSource(
_NOUS_CAPS = _CapsSource(
"_nous_reasoning_caps_cache", "_nous_reasoning_caps_failed_at",
"_nous_caps_disk_checked", "_nous_caps_warm_started",
lambda: _origin().nous_catalog_url(),
lambda: nous_catalog_url(),
)
def nous_catalog_url() -> str:
"""The Portal ``/v1/models`` URL for the endpoint we actually talk to (``NOUS_INFERENCE_BASE_URL``
→ resolved credential base → prod), so a staging profile reads staging's capabilities."""
return f"{_origin()._resolve_nous_pricing_credentials()[1]}/v1/models"
from hermes_cli.models_pricing import _resolve_nous_pricing_credentials
return f"{_resolve_nous_pricing_credentials()[1]}/v1/models"
# Live-catalog metadata first (ported from PrimeIntellect-ai/prime-agent#1258): OpenRouter's /v1/models
# entries advertise reasoning support via supported_parameters + a reasoning object, which covers every
# routed vendor without a hand-maintained prefix list. The static prefix allowlist repeatedly went
# stale one vendor at a time (nvidia/ missing → #75386; same class as tencent/, xiaomi/ additions before
# it) — metadata makes new vendors work without a code change. One catalog fetch per process, cached;
# unknown (catalog unreachable / unlisted model) falls back to the static list.
def openrouter_model_reasoning_capabilities(
model_id: Optional[str], *, timeout: float = 6.0, allow_fetch: bool = False,
) -> Optional[dict[str, Any]]:
+11 -9
View File
@@ -1,8 +1,8 @@
"""Validate a requested ``/model`` value against the active provider's catalog.
Split out of ``hermes_cli.models``; :func:`validate_requested_model` is re-imported there. Every
catalog fetcher this module calls is looked up on ``hermes_cli.models`` at call time (``_m.<name>``)
so existing ``patch("hermes_cli.models.<name>")`` mocks keep intercepting.
Split out of ``hermes_cli.models``. Catalog fetchers defined in ``hermes_cli.models`` are looked up
there at call time (``_m.<name>``) so ``patch("hermes_cli.models.<name>")`` mocks keep intercepting;
local-server probes are looked up on ``hermes_cli.models_local`` (``_ml.<name>``) the same way.
Every provider branch returns a verdict dict (see :func:`_verdict`) or ``None`` for "not decided
here — keep walking the ladder". The ladder ORDER is behavior (see ``_LADDER``).
@@ -170,13 +170,13 @@ def _parse_openrouter_preset(req: _Request) -> Optional[dict[str, Any]]:
def _validate_lmstudio(req: _Request) -> dict[str, Any]:
from hermes_cli import models as _m
from hermes_cli import models_local as _ml
from hermes_cli.auth import AuthError
# probe_lmstudio_models distinguishes None (unreachable / malformed) from [] (reachable,
# nothing chat-capable loaded); fetch_lmstudio_models collapses both to [].
try:
models = _m.probe_lmstudio_models(api_key=req.api_key, base_url=req.base_url)
models = _ml.probe_lmstudio_models(api_key=req.api_key, base_url=req.base_url)
except AuthError as exc:
return _reject(f"{exc} Set `LM_API_KEY` (or update it) to match the server's bearer token.")
if models is None:
@@ -194,10 +194,11 @@ def _ollama_probe_headers(req: _Request) -> dict[str, str]:
when the probed endpoint is the configured one (never leak them to a different host). Caller
headers win; a caller ``api_key`` becomes the Authorization header unless the caller sent one."""
from hermes_cli import models as _m
from hermes_cli import models_local as _ml
from hermes_cli.models_local import _configured_ollama_base_url, _drop_authorization
configured_base = _configured_ollama_base_url()
configured_allowed = not configured_base or _m._same_ollama_native_root(req.base_url or "", configured_base)
configured_allowed = not configured_base or _ml._same_ollama_native_root(req.base_url or "", configured_base)
configured = _m._get_ollama_native_headers(req.base_url, api_key=req.api_key) if configured_allowed else {}
if req.headers is None:
return configured
@@ -215,17 +216,18 @@ def _validate_ollama_native(req: _Request) -> Optional[dict[str, Any]]:
looks like a local Ollama server. Also resolves ``base_url`` for the raw ``ollama`` provider,
which later branches (custom) rely on."""
from hermes_cli import models as _m
from hermes_cli import models_local as _ml
if str(req.provider or "").strip().lower() == "ollama" and not req.base_url:
req.base_url = _m._get_ollama_base_url()
headers = _ollama_probe_headers(req)
if not _m.should_use_ollama_native_catalog(req.provider, req.base_url, headers=headers):
if not _ml.should_use_ollama_native_catalog(req.provider, req.base_url, headers=headers):
return None
models = _m.probe_ollama_local_models(req.base_url, headers=headers)
models = _ml.probe_ollama_local_models(req.base_url, headers=headers)
if models is None:
# A failed native probe is not authoritative; fall back to the OpenAI-compatible catalog.
models = _m.probe_api_models(
req.api_key, _m._normalize_openai_base_url(req.base_url), request_headers=headers,
req.api_key, _ml._normalize_openai_base_url(req.base_url), request_headers=headers,
).get("models")
if models is None:
return _soft_accept(
+4 -3
View File
@@ -131,10 +131,11 @@ async def get_model_options(
def _nous_recommended_default() -> dict:
from hermes_cli import models as m
from hermes_cli import models_pricing as mp
from hermes_cli.auth import get_provider_auth_state
model_ids = m.get_curated_nous_model_ids()
pricing = m.get_pricing_for_provider("nous") or {}
pricing = mp.get_pricing_for_provider("nous") or {}
free_tier = m.check_nous_free_tier(force_fresh=True)
try:
@@ -145,10 +146,10 @@ def _nous_recommended_default() -> dict:
# This endpoint picks the model a user lands on without choosing it, so an unreachable
# one is worse than in a picker. Narrow to policy BEFORE the tier split, so a rescued
# id still has to pass the free/paid predicate.
policy_allowed = m.nous_policy_allowed_ids()
policy_allowed = mp.nous_policy_allowed_ids()
union = m.union_with_portal_free_recommendations if free_tier else m.union_with_portal_paid_recommendations
model_ids, pricing = union(model_ids, pricing, portal_url)
model_ids = m.restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True)
model_ids = mp.restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True)
if free_tier:
model_ids, _unavailable = m.partition_nous_models_by_tier(model_ids, pricing, free_tier=True)
+1 -1
View File
@@ -39,7 +39,7 @@ class NousProfile(ProviderProfile):
"""True when ``reasoning: {enabled: false}`` would 400 on *model*. Cache-only catalog
lookup; unknown/cold (warmer kicked) and no-reasoning routes both answer True (omit > 400)."""
try:
from hermes_cli.models import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async
caps = nous_model_reasoning_capabilities(model)
if caps is None:
@@ -64,7 +64,8 @@ class OpenRouterProfile(ProviderProfile):
if not effort and not disabled:
return cfg
try:
from hermes_cli.models import clamp_reasoning_effort_to_supported, openrouter_model_reasoning_capabilities
from hermes_cli.models import clamp_reasoning_effort_to_supported
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
caps = openrouter_model_reasoning_capabilities(model)
if not caps or not caps.get("supports_reasoning"):
+2 -2
View File
@@ -3958,7 +3958,7 @@ class TelegramAdapter(BasePlatformAdapter):
"""Paginated top-level provider keyboard folding provider families (Kimi/Moonshot, MiniMax, xAI…)
into one ``mpg:<gid>`` button via the shared ``group_providers`` fold; singles are ``mp:<slug>``."""
try:
from hermes_cli.models import group_providers
from hermes_cli.models_catalog_static import group_providers
except Exception:
group_providers = None
by_slug = {p.get("slug"): p for p in providers}
@@ -4117,7 +4117,7 @@ class TelegramAdapter(BasePlatformAdapter):
elif data.startswith("mpg:"): # provider group selected: show member providers
group_id = data[4:]
try:
from hermes_cli.models import PROVIDER_GROUPS
from hermes_cli.models_catalog_static import PROVIDER_GROUPS
_label, _desc, member_slugs = PROVIDER_GROUPS.get(group_id, ("", "", []))
except Exception:
_label, member_slugs = "", []
+1 -1
View File
@@ -447,7 +447,7 @@ class AIAgent(
if (getattr(self, "lmstudio_load_mode", "explicit") or "explicit").strip().lower() == "jit":
logger.debug("LM Studio explicit preload skipped: lmstudio_load_mode=jit")
return None
from hermes_cli.models import ensure_lmstudio_model_loaded
from hermes_cli.models_local import ensure_lmstudio_model_loaded
if config_context_length is None:
config_context_length = getattr(self, "_config_context_length", None)
+3 -3
View File
@@ -120,7 +120,7 @@ class TestNamedCustomProviderCatalogs:
}
)
with patch("hermes_cli.config.load_config", return_value=cfg), patch(
"hermes_cli.models.should_use_ollama_native_catalog",
"hermes_cli.models_local.should_use_ollama_native_catalog",
return_value=True,
), patch(
"hermes_cli.model_switch_providers._fetch_picker_live_models",
@@ -148,7 +148,7 @@ class TestNamedCustomProviderCatalogs:
]
)
with patch("hermes_cli.config.load_config", return_value=cfg), patch(
"hermes_cli.models.should_use_ollama_native_catalog",
"hermes_cli.models_local.should_use_ollama_native_catalog",
return_value=True,
), patch(
"hermes_cli.model_switch_providers._fetch_picker_live_models",
@@ -172,7 +172,7 @@ class TestNamedCustomProviderCatalogs:
from hermes_cli.model_switch_providers import _NativePickerModelList
with patch("hermes_cli.config.load_config", return_value=cfg), patch(
"hermes_cli.models.should_use_ollama_native_catalog",
"hermes_cli.models_local.should_use_ollama_native_catalog",
return_value=True,
), patch(
"hermes_cli.model_switch_providers._fetch_picker_live_models",
+5 -5
View File
@@ -4772,7 +4772,7 @@ class TestFastModelTier:
"~openai/gpt-mini-latest": {},
"stepfun/step-3.7-flash:free": {},
}
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
assert ac._fast_model_from_catalog("nous") == "~openai/gpt-mini-latest"
def test_catalog_match_skips_reasoning_batch_and_embedding_lookalikes(self):
@@ -4785,7 +4785,7 @@ class TestFastModelTier:
"sentence-transformers/all-minilm-l6-v2": {},
"google/gemini-3.6-flash": {},
}
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
assert ac._fast_model_from_catalog("nous") == "google/gemini-3.6-flash"
def test_catalog_match_skips_the_non_chat_siblings_of_a_chat_model(self):
@@ -4799,7 +4799,7 @@ class TestFastModelTier:
"openai/gpt-4o-mini-search-preview": {},
"openai/gpt-4o-mini": {},
}
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
assert ac._fast_model_from_catalog("nous") == "openai/gpt-4o-mini"
def test_catalog_match_takes_the_newest_of_a_family(self):
@@ -4816,7 +4816,7 @@ class TestFastModelTier:
"openai/gpt-9-mini": {},
"openai/gpt-10-mini": {},
}
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
assert ac._fast_model_from_catalog("nous") == "openai/gpt-10-mini"
def test_catalog_fetch_is_authenticated(self):
@@ -4831,7 +4831,7 @@ class TestFastModelTier:
"hermes_cli.auth.resolve_api_key_provider_credentials",
return_value={"api_key": "sk-test", "base_url": "https://api.example.com/v1"},
), patch(
"hermes_cli.models.fetch_models_with_pricing", return_value={}
"hermes_cli.models_pricing.fetch_models_with_pricing", return_value={}
) as fetch:
ac._fast_model_from_catalog("openai")
+4 -4
View File
@@ -317,12 +317,11 @@ class TestIsFreeTierModel:
def test_pricing_cache_peek_zero_priced_model(self, monkeypatch):
from agent.credits_tracker import is_free_tier_model
import hermes_cli.models as models_mod
from hermes_cli import models_pricing
# The picker keys the cache on the pre-/v1 root (get_pricing_for_provider
# strips a trailing /v1 before fetch_models_with_pricing).
monkeypatch.setattr(
models_mod,
"_pricing_cache",
monkeypatch.setattr(models_pricing, "_pricing_cache",
{
"https://inference-api.nousresearch.com": {
"some/zero-priced": {"prompt": "0", "completion": "0"},
@@ -343,12 +342,13 @@ class TestIsFreeTierModel:
def test_exception_fails_open_to_false(self, monkeypatch):
from agent.credits_tracker import is_free_tier_model
import hermes_cli.models as models_mod
from hermes_cli import models_pricing
class _Exploding:
def get(self, *_a, **_kw):
raise RuntimeError("boom")
monkeypatch.setattr(models_mod, "_pricing_cache", _Exploding())
monkeypatch.setattr(models_pricing, "_pricing_cache", _Exploding())
assert is_free_tier_model("some/model", "https://inference-api.nousresearch.com") is False
def test_stealth_prefix_detected_as_free(self):
+2 -1
View File
@@ -8,6 +8,7 @@ import pytest
from hermes_cli.auth import AuthError
from hermes_cli import main as hermes_main
from hermes_cli import model_setup_flows
# ---------------------------------------------------------------------------
@@ -329,7 +330,7 @@ def test_model_flow_nous_does_not_restore_stale_custom_api_key(tmp_path, monkeyp
"hermes_cli.models.get_curated_nous_model_ids",
lambda: [selected_model],
)
monkeypatch.setattr("hermes_cli.models.get_pricing_for_provider", lambda provider: {})
monkeypatch.setattr("hermes_cli.models_pricing.get_pricing_for_provider", lambda provider: {})
monkeypatch.setattr("hermes_cli.models.check_nous_free_tier", lambda **kwargs: False)
monkeypatch.setattr(
"hermes_cli.models.union_with_portal_paid_recommendations",
+4 -7
View File
@@ -9,12 +9,9 @@ import json
from unittest.mock import patch, MagicMock
from hermes_cli import models as models_module
from hermes_cli.models import (
VERCEL_AI_GATEWAY_MODELS,
_ai_gateway_model_is_free,
fetch_ai_gateway_models,
fetch_ai_gateway_pricing,
)
from hermes_cli import models_pricing
from hermes_cli.models import VERCEL_AI_GATEWAY_MODELS, _ai_gateway_model_is_free, fetch_ai_gateway_models
from hermes_cli.models_pricing import fetch_ai_gateway_pricing
def _mock_urlopen(payload):
@@ -29,7 +26,7 @@ def _mock_urlopen(payload):
def _reset_caches():
models_module._ai_gateway_catalog_cache = None
models_module._pricing_cache.clear()
models_pricing._pricing_cache.clear()
def test_ai_gateway_pricing_translates_input_output_to_prompt_completion():
+11 -5
View File
@@ -902,9 +902,10 @@ class TestNovitaProvider:
def test_novita_pricing_cache(self, monkeypatch):
"""_fetch_novita_pricing should cache results in _pricing_cache."""
from hermes_cli import models as models_mod
from hermes_cli import models_pricing
monkeypatch.setenv("NOVITA_API_KEY", "sk-test-key")
monkeypatch.setenv("NOVITA_BASE_URL", "https://api.novita.ai/openai/v1")
models_mod._pricing_cache.pop("https://api.novita.ai/openai/v1", None)
models_pricing._pricing_cache.pop("https://api.novita.ai/openai/v1", None)
call_count = {"n": 0}
fake_payload = {
@@ -937,17 +938,17 @@ class TestNovitaProvider:
)
# First call hits the network.
first = models_mod._fetch_novita_pricing()
first = models_pricing._fetch_novita_pricing()
assert "x/y" in first
assert call_count["n"] == 1
# Second call returns cached result without re-hitting the network.
second = models_mod._fetch_novita_pricing()
second = models_pricing._fetch_novita_pricing()
assert second == first
assert call_count["n"] == 1
# force_refresh bypasses the cache.
models_mod._fetch_novita_pricing(force_refresh=True)
models_pricing._fetch_novita_pricing(force_refresh=True)
assert call_count["n"] == 2
@@ -1004,6 +1005,7 @@ def _deepinfra_cache_isolation(monkeypatch):
a later test's fetch within the failure TTL.
"""
import hermes_cli.models as _models_mod
from hermes_cli import models_pricing
monkeypatch.setattr(_models_mod, "_deepinfra_catalog_cache", {})
monkeypatch.setattr(_models_mod, "_deepinfra_catalog_neg_cache", {})
yield
@@ -1030,6 +1032,7 @@ class TestFetchDeepInfraModels:
]}).encode()
import hermes_cli.models as models
from hermes_cli import models_pricing
monkeypatch.setattr(
models, "_urlopen_model_catalog_request", lambda *a, **kw: _Resp()
)
@@ -1047,6 +1050,7 @@ class TestFetchDeepInfraModels:
def test_catalog_uses_credential_safe_opener(self, monkeypatch):
import hermes_cli.models as models
from hermes_cli import models_pricing
seen = {}
@@ -1119,6 +1123,7 @@ class TestDeepInfraTagFiltering:
]}
from hermes_cli.models import _fetch_deepinfra_models_by_tag
import hermes_cli.models as _m
from hermes_cli import models_pricing
for surface in ("chat", "image-gen", "tts", "stt", "embed"):
monkeypatch.setattr(
@@ -1171,12 +1176,13 @@ class TestDeepInfraPricingFetcher:
{"id": "vendor/model-image", "metadata": {"tags": ["image-gen"], "pricing": {"per_image_unit": 0.05}}},
]}
import hermes_cli.models as models
from hermes_cli import models_pricing
monkeypatch.setattr(
models,
"_urlopen_model_catalog_request",
_make_urlopen_returning(payload),
)
from hermes_cli.models import get_pricing_for_provider
from hermes_cli.models_pricing import get_pricing_for_provider
# get_pricing_for_provider → _fetch_deepinfra_pricing dispatch path
result = get_pricing_for_provider("deepinfra")
+2 -1
View File
@@ -500,6 +500,7 @@ class TestLoginNousSkipKeepsCurrent:
"""Patch OAuth + model-list + prompt so _login_nous doesn't hit network."""
import hermes_cli.auth as auth_mod
import hermes_cli.models as models_mod
from hermes_cli import models_pricing
import hermes_cli.nous_subscription as ns
fake_auth_state = {
@@ -518,7 +519,7 @@ class TestLoginNousSkipKeepsCurrent:
auth_mod, "_prompt_model_selection",
lambda *a, **kw: prompt_returns,
)
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda p: {})
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda p: {})
free_tier_calls = []
def _check_nous_free_tier(**kwargs):
@@ -459,7 +459,7 @@ class TestCustomProviderDiscoverModels:
}
with patch("hermes_cli.models.fetch_api_models") as mock_fetch, \
patch("hermes_cli.models.fetch_ollama_local_models") as mock_ollama, \
patch("hermes_cli.models_local.fetch_ollama_local_models") as mock_ollama, \
patch("hermes_cli.curses_ui.curses_radiolist", side_effect=ImportError), \
patch("builtins.input", return_value="1"), \
patch("builtins.print"):
+33 -47
View File
@@ -10,10 +10,11 @@ from time import monotonic
import hermes_cli.inventory as inv
import hermes_cli.models as models_mod
from hermes_cli import models_pricing
def _patch_pricing(monkeypatch, *, free_tier, pricing, unavailable=None):
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda slug, **kw: pricing.get(slug, {}))
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda slug, **kw: pricing.get(slug, {}))
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda *, force_fresh=False: free_tier)
monkeypatch.setattr(
models_mod, "partition_nous_models_by_tier",
@@ -125,7 +126,7 @@ def test_model_options_cold_pricing_fetch_runs_off_the_request_path(monkeypatch)
"is_user_defined": False,
"source": "built-in",
}
monkeypatch.setattr(models_mod, "get_pricing_for_provider", fake_pricing)
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", fake_pricing)
monkeypatch.setattr(
"hermes_cli.model_switch.list_authenticated_providers",
lambda **_kwargs: [row],
@@ -160,8 +161,7 @@ def test_model_options_cold_pricing_fetch_runs_off_the_request_path(monkeypatch)
def test_cold_nous_entitlement_keeps_models_unselectable(monkeypatch):
"""A cold nonblocking response must not expose paid models fail-open."""
monkeypatch.setattr(
models_mod, "get_pricing_for_provider", lambda *_args, **_kwargs: {}
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda *_args, **_kwargs: {}
)
monkeypatch.setattr(models_mod, "get_cached_nous_free_tier", lambda: None)
rows = [{"slug": "nous", "models": ["free/model", "paid/model"]}]
@@ -288,12 +288,10 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch):
endpoint_b: {"b/model": {"prompt": "3", "completion": "4"}},
}
monkeypatch.setattr(inv, "_pricing_prewarm_threads", {})
monkeypatch.setattr(models_mod, "_pricing_cache", {})
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
monkeypatch.setattr(
models_mod,
"_resolve_nous_pricing_credentials",
monkeypatch.setattr(models_pricing, "_pricing_cache", {})
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
lambda: ("", active_endpoint["value"]),
)
@@ -301,13 +299,13 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch):
started[base_url].set()
if base_url == endpoint_a:
release_a.wait(timeout=5)
return models_mod._cache_catalog(base_url, expected[base_url])
return models_pricing._cache_catalog(base_url, expected[base_url])
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", fetch_pricing)
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", fetch_pricing)
monkeypatch.setattr(
inv,
"_apply_pricing",
lambda _rows: models_mod.get_pricing_for_provider("nous"),
lambda _rows: models_pricing.get_pricing_for_provider("nous"),
)
token = set_hermes_home_override(str(tmp_path / "profile"))
@@ -335,7 +333,7 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch):
assert started[endpoint_b].wait(timeout=1)
threads[1].join(timeout=2)
assert not threads[1].is_alive()
assert models_mod.get_pricing_for_provider(
assert models_pricing.get_pricing_for_provider(
"nous", cached_only=True
) == expected[endpoint_b]
finally:
@@ -363,17 +361,13 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey
endpoint_b: {"b/model": {"prompt": "3", "completion": "4"}},
}
monkeypatch.setattr(inv, "_pricing_prewarm_threads", {})
monkeypatch.setattr(models_mod, "_pricing_cache", {})
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
monkeypatch.setattr(
models_mod,
"get_cached_nous_inference_base_url",
monkeypatch.setattr(models_pricing, "_pricing_cache", {})
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
monkeypatch.setattr(models_pricing, "get_cached_nous_inference_base_url",
lambda: active_endpoint["value"],
)
monkeypatch.setattr(
models_mod,
"_resolve_nous_pricing_credentials",
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
lambda: ("", active_endpoint["value"]),
)
@@ -381,13 +375,13 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey
started[base_url].set()
if base_url == endpoint_a:
release_a.wait(timeout=5)
return models_mod._cache_catalog(base_url, expected[base_url])
return models_pricing._cache_catalog(base_url, expected[base_url])
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", fetch_pricing)
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", fetch_pricing)
monkeypatch.setattr(
inv,
"_apply_pricing",
lambda _rows: models_mod.get_pricing_for_provider("nous"),
lambda _rows: models_pricing.get_pricing_for_provider("nous"),
)
token = set_hermes_home_override(str(tmp_path / "profile"))
@@ -415,7 +409,7 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey
assert started[endpoint_b].wait(timeout=1)
threads[1].join(timeout=2)
assert not threads[1].is_alive()
assert models_mod.get_pricing_for_provider(
assert models_pricing.get_pricing_for_provider(
"nous", cached_only=True
) == expected[endpoint_b]
finally:
@@ -430,16 +424,14 @@ def test_cached_only_pricing_returns_a_warm_value_without_fetching(monkeypatch):
"""Cache-only picker reads preserve pricing once the prewarm completes."""
cache_key = "https://openrouter.ai/api"
expected = {"vendor/model": {"prompt": "0.000001", "completion": "0.000002"}}
monkeypatch.setattr(models_mod, "_pricing_cache", {cache_key: expected})
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
monkeypatch.setattr(
models_mod,
"fetch_models_with_pricing",
monkeypatch.setattr(models_pricing, "_pricing_cache", {cache_key: expected})
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing",
lambda **_kwargs: (_ for _ in ()).throw(AssertionError("network fetch started")),
)
assert models_mod.get_pricing_for_provider(
assert models_pricing.get_pricing_for_provider(
"openrouter", cached_only=True
) == expected
@@ -455,30 +447,24 @@ def test_cached_only_dynamic_pricing_is_profile_scoped(tmp_path, monkeypatch):
endpoint_b = "https://profile-b.example"
expected_a = {"a/model": {"prompt": "1", "completion": "2"}}
expected_b = {"b/model": {"prompt": "3", "completion": "4"}}
monkeypatch.setattr(
models_mod,
"_pricing_cache",
monkeypatch.setattr(models_pricing, "_pricing_cache",
{endpoint_a: expected_a, endpoint_b: expected_b},
)
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
active_endpoint = {"value": endpoint_a}
monkeypatch.setattr(
models_mod,
"_resolve_nous_pricing_credentials",
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
lambda: ("", active_endpoint["value"]),
)
monkeypatch.setattr(
models_mod,
"fetch_models_with_pricing",
lambda **kwargs: models_mod._pricing_cache[kwargs["base_url"]],
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing",
lambda **kwargs: models_pricing._pricing_cache[kwargs["base_url"]],
)
def in_profile(home, endpoint, *, cached_only):
token = set_hermes_home_override(str(home))
active_endpoint["value"] = endpoint
try:
return models_mod.get_pricing_for_provider(
return models_pricing.get_pricing_for_provider(
"nous", cached_only=cached_only
)
finally:
@@ -11,15 +11,16 @@ invite a picker filter that hides working levels.
import hermes_cli.inventory as inv
import hermes_cli.models as models_mod
from hermes_cli import models_reasoning_caps
def _patch_catalog(monkeypatch, caps_by_model, *, provider="nous"):
"""Point the Nous/OpenRouter catalog readers at a fixed capability map."""
monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda model: False)
monkeypatch.setattr(models_mod, "warm_nous_reasoning_caps_async", lambda: None)
monkeypatch.setattr(models_mod, "warm_openrouter_reasoning_caps_async", lambda: None)
monkeypatch.setattr(models_reasoning_caps, "warm_nous_reasoning_caps_async", lambda: None)
monkeypatch.setattr(models_reasoning_caps, "warm_openrouter_reasoning_caps_async", lambda: None)
monkeypatch.setattr(
models_mod,
models_reasoning_caps,
f"{provider}_model_reasoning_capabilities",
lambda model, **kw: caps_by_model.get(model),
)
@@ -143,12 +144,12 @@ def test_openrouter_uses_its_own_catalog(monkeypatch):
def test_catalog_failure_never_breaks_the_picker(monkeypatch):
"""A raising catalog reader degrades to "unknown", not to a broken payload."""
monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda model: False)
monkeypatch.setattr(models_mod, "warm_nous_reasoning_caps_async", lambda: None)
monkeypatch.setattr(models_reasoning_caps, "warm_nous_reasoning_caps_async", lambda: None)
def _boom(model, **kw):
raise RuntimeError("catalog exploded")
monkeypatch.setattr(models_mod, "nous_model_reasoning_capabilities", _boom)
monkeypatch.setattr(models_reasoning_caps, "nous_model_reasoning_capabilities", _boom)
rows = [{"slug": "nous", "models": ["deepseek/deepseek-v4-pro"]}]
inv._apply_capabilities(rows)
@@ -159,6 +159,7 @@ def _stub_kimi_discovery(monkeypatch, *, canonical):
"""
import agent.models_dev as md
import hermes_cli.models as hm
from hermes_cli import models_catalog_static
kimi_map = {
"kimi": "kimi-for-coding",
@@ -188,10 +189,11 @@ def _stub_kimi_discovery(monkeypatch, *, canonical):
def test_single_kimi_credential_yields_one_canonical_row(monkeypatch):
"""One Kimi key yields a single row under the canonical 'kimi-coding' slug."""
import hermes_cli.models as hm
from hermes_cli import models_catalog_static
_stub_kimi_discovery(
monkeypatch,
canonical=[hm.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc")],
canonical=[models_catalog_static.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc")],
)
monkeypatch.setenv("KIMI_API_KEY", "sk-test-kimi")
@@ -215,12 +217,13 @@ def test_distinct_kimi_china_credential_still_listed(monkeypatch):
pair that share a credential, not legitimately distinct providers.
"""
import hermes_cli.models as hm
from hermes_cli import models_catalog_static
_stub_kimi_discovery(
monkeypatch,
canonical=[
hm.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc"),
hm.ProviderEntry("kimi-coding-cn", "Kimi / Moonshot (China)", "desc"),
models_catalog_static.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc"),
models_catalog_static.ProviderEntry("kimi-coding-cn", "Kimi / Moonshot (China)", "desc"),
],
)
monkeypatch.setenv("KIMI_API_KEY", "sk-test-kimi")
@@ -3,6 +3,7 @@ import json
import pytest
from hermes_cli import models
from hermes_cli import models_local
MODEL = "publisher/model"
@@ -54,12 +55,11 @@ def _capture_load(monkeypatch, response_payload):
def test_missing_echo_refreshes_loaded_state(monkeypatch):
catalogs = iter([_catalog(), _catalog(loaded_context=88_000)])
monkeypatch.setattr(
models, "_lmstudio_fetch_raw_models", lambda **_kwargs: next(catalogs)
monkeypatch.setattr(models_local, "_lmstudio_fetch_raw_models", lambda **_kwargs: next(catalogs)
)
_capture_load(monkeypatch, {"status": "loaded"})
result = models.ensure_lmstudio_model_loaded(
result = models_local.ensure_lmstudio_model_loaded(
MODEL, BASE_URL, api_key="", target_context_length=100_000
)
@@ -67,9 +67,7 @@ def test_missing_echo_refreshes_loaded_state(monkeypatch):
def test_explicit_override_above_known_maximum_rejects_even_when_loaded(monkeypatch):
monkeypatch.setattr(
models,
"_lmstudio_fetch_raw_models",
monkeypatch.setattr(models_local, "_lmstudio_fetch_raw_models",
lambda **_kwargs: _catalog(loaded_context=64_000, maximum=128_000),
)
monkeypatch.setattr(
@@ -78,7 +76,7 @@ def test_explicit_override_above_known_maximum_rejects_even_when_loaded(monkeypa
lambda *_args, **_kwargs: pytest.fail("invalid override must not be posted"),
)
result = models.ensure_lmstudio_model_loaded(
result = models_local.ensure_lmstudio_model_loaded(
MODEL,
BASE_URL,
api_key="",
@@ -47,7 +47,7 @@ def _switch_to_alias(monkeypatch, alias_entry):
return {"accepted": True, "persist": True, "recognized": True, "message": ""}
monkeypatch.setattr(
"hermes_cli.models.validate_requested_model", _fake_validate
"hermes_cli.models_validate.validate_requested_model", _fake_validate
)
import hermes_cli.model_switch as ms
@@ -208,7 +208,7 @@ class TestSessionKeyIsHostScoped:
"hermes_cli.runtime_provider.load_config", lambda *a, **k: cfg
)
monkeypatch.setattr(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
lambda *a, **k: {
"accepted": True,
"persist": True,
@@ -278,7 +278,7 @@ class TestBuiltinProviderKeysDoNotLeak:
return {"accepted": True, "persist": True, "recognized": True, "message": ""}
monkeypatch.setattr(
"hermes_cli.models.validate_requested_model", _fake_validate
"hermes_cli.models_validate.validate_requested_model", _fake_validate
)
import hermes_cli.model_switch as ms
@@ -320,7 +320,7 @@ class TestProviderLabelCannotSelectAKeyForAnArbitraryHost:
probed["api_key"] = api_key
return {"accepted": True, "persist": True, "recognized": True, "message": ""}
monkeypatch.setattr("hermes_cli.models.validate_requested_model", _fake_validate)
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", _fake_validate)
import hermes_cli.model_switch as ms
monkeypatch.setattr(ms, "DIRECT_ALIASES", {})
@@ -378,7 +378,7 @@ class TestSessionCredentialIsScopedToTheOrigin:
monkeypatch.setattr("hermes_cli.config.load_config", lambda *a, **k: cfg)
monkeypatch.setattr("hermes_cli.runtime_provider.load_config", lambda *a, **k: cfg)
monkeypatch.setattr(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
lambda *a, **k: {"accepted": True, "persist": True,
"recognized": True, "message": ""},
)
@@ -60,7 +60,7 @@ def _run_switch(
with patch("hermes_cli.model_switch.resolve_alias", return_value=None), \
patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \
patch("hermes_cli.model_switch.normalize_model_for_provider", side_effect=lambda model, provider: model), \
patch("hermes_cli.models.validate_requested_model", return_value=validation), \
patch("hermes_cli.models_validate.validate_requested_model", return_value=validation), \
patch("hermes_cli.models.detect_provider_for_model", return_value=None), \
patch("hermes_cli.model_switch.get_model_info", return_value=None), \
patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \
@@ -40,7 +40,7 @@ def _run_copilot_switch(
},
),
patch(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
return_value=_MOCK_VALIDATION,
),
patch("hermes_cli.model_switch.get_model_info", return_value=None),
@@ -39,19 +39,19 @@ def _disable_live_custom_provider_model_probe(monkeypatch):
"hermes_cli.models.provider_model_ids", lambda *_a, **_kw: []
)
monkeypatch.setattr(
"hermes_cli.models.fetch_ollama_local_models", lambda *_a, **_kw: None
"hermes_cli.models_local.fetch_ollama_local_models", lambda *_a, **_kw: None
)
def test_picker_native_probe_failure_falls_back_to_openai_catalog(monkeypatch):
monkeypatch.setattr(
"hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: True
"hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: True
)
monkeypatch.setattr(
"hermes_cli.models._get_ollama_native_headers", lambda *a, **k: {}
)
monkeypatch.setattr(
"hermes_cli.models.fetch_ollama_local_models", lambda *a, **k: None
"hermes_cli.models_local.fetch_ollama_local_models", lambda *a, **k: None
)
monkeypatch.setattr(
"hermes_cli.models.fetch_api_models", lambda *a, **k: ["fallback-model"]
@@ -70,7 +70,7 @@ def test_picker_generic_discovery_preserves_api_mode(monkeypatch):
return ["model-a"]
monkeypatch.setattr(
"hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: False
"hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: False
)
monkeypatch.setattr("hermes_cli.models.cached_fetch_api_models", cached)
@@ -172,7 +172,7 @@ def test_providers_singular_model_does_not_suppress_ollama_native_discovery(monk
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
monkeypatch.setattr(
"hermes_cli.models.fetch_ollama_local_models",
"hermes_cli.models_local.fetch_ollama_local_models",
lambda *a, **k: ["qwen3:latest", "llama3.2:latest"],
)
@@ -362,7 +362,7 @@ def test_list_authenticated_providers_can_probe_active_bare_custom_endpoint(monk
def test_switch_model_accepts_explicit_bare_custom_current_endpoint(monkeypatch):
"""Picker selections for bare custom endpoints should route to current base_url."""
monkeypatch.setattr("hermes_cli.models.validate_requested_model", lambda *a, **k: _MOCK_VALIDATION)
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", lambda *a, **k: _MOCK_VALIDATION)
monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None)
monkeypatch.setattr("hermes_cli.model_switch.get_model_capabilities", lambda *a, **k: None)
@@ -408,11 +408,11 @@ def test_switch_model_does_not_send_ollama_headers_to_unrelated_custom_endpoint(
return _MOCK_VALIDATION
monkeypatch.setattr(
"hermes_cli.models.should_use_ollama_native_catalog",
"hermes_cli.models_local.should_use_ollama_native_catalog",
fake_native_detection,
)
monkeypatch.setattr(
"hermes_cli.models._get_ollama_request_headers",
"hermes_cli.models_local._get_ollama_request_headers",
lambda: {"Authorization": "Bearer configured-ollama-secret"},
)
monkeypatch.setattr(
@@ -431,7 +431,7 @@ def test_switch_model_does_not_send_ollama_headers_to_unrelated_custom_endpoint(
"api_mode": "chat_completions",
},
)
monkeypatch.setattr("hermes_cli.models.validate_requested_model", fake_validation)
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", fake_validation)
monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None)
monkeypatch.setattr("hermes_cli.model_switch.get_model_capabilities", lambda *a, **k: None)
@@ -480,7 +480,7 @@ def test_picker_selection_resolves_named_custom_provider_model_id(monkeypatch):
},
)
monkeypatch.setattr(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
lambda *a, **k: _MOCK_VALIDATION,
)
monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None)
@@ -980,8 +980,8 @@ def test_picker_endpoint_authorization_overrides_inferred_bearer(monkeypatch):
captured.update(headers or {})
return ["model-a"]
monkeypatch.setattr("hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: True)
monkeypatch.setattr("hermes_cli.models.fetch_ollama_local_models", fake_native)
monkeypatch.setattr("hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: True)
monkeypatch.setattr("hermes_cli.models_local.fetch_ollama_local_models", fake_native)
result = _fetch_picker_live_models(
"endpoint-key",
"http://127.0.0.1:11434/v1",
@@ -1217,7 +1217,7 @@ def test_lmstudio_picker_probes_active_config_base_url(monkeypatch):
captured["api_key"] = api_key
return ["qwen/qwen3-coder-30b"]
monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch)
monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch)
list_authenticated_providers(
current_provider="lmstudio",
@@ -1244,7 +1244,7 @@ def test_lmstudio_picker_lm_base_url_env_wins_over_active_config(monkeypatch):
captured["base_url"] = base_url
return []
monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch)
monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch)
list_authenticated_providers(
current_provider="lmstudio",
@@ -1270,7 +1270,7 @@ def test_lmstudio_picker_skips_probe_when_not_configured(monkeypatch):
captured["base_url"] = base_url
return []
monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch)
monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch)
list_authenticated_providers(
current_provider="openrouter",
@@ -49,7 +49,7 @@ def _run_openai_switch(
},
),
patch(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
return_value=_MOCK_VALIDATION,
),
patch("hermes_cli.model_switch.get_model_info", return_value=None),
@@ -57,7 +57,7 @@ def _run_opencode_switch(
},
),
patch(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
return_value=_MOCK_VALIDATION,
),
patch("hermes_cli.model_switch.get_model_info", return_value=None),
@@ -24,7 +24,7 @@ def _run_switch(raw_input: str, current_provider: str = "openrouter") -> str:
patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \
patch("hermes_cli.runtime_provider.resolve_runtime_provider",
return_value={"api_key": "test", "base_url": "", "api_mode": "chat_completions"}), \
patch("hermes_cli.models.validate_requested_model", return_value=_MOCK_VALIDATION), \
patch("hermes_cli.models_validate.validate_requested_model", return_value=_MOCK_VALIDATION), \
patch("hermes_cli.model_switch.get_model_info", return_value=None), \
patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \
patch("hermes_cli.models.detect_provider_for_model", return_value=None):
+3 -18
View File
@@ -3,24 +3,9 @@
import pytest
from unittest.mock import MagicMock, patch
from hermes_cli.models import (
azure_foundry_model_api_mode,
copilot_model_api_mode,
fetch_github_model_catalog,
curated_models_for_provider,
fetch_api_models,
fetch_lmstudio_models,
github_model_reasoning_efforts,
normalize_copilot_model_id,
normalize_opencode_model_id,
normalize_provider,
opencode_model_api_mode,
parse_model_input,
probe_api_models,
provider_label,
provider_model_ids,
validate_requested_model,
)
from hermes_cli.models import azure_foundry_model_api_mode, copilot_model_api_mode, fetch_github_model_catalog, curated_models_for_provider, fetch_api_models, github_model_reasoning_efforts, normalize_copilot_model_id, normalize_opencode_model_id, normalize_provider, opencode_model_api_mode, parse_model_input, probe_api_models, provider_label, provider_model_ids
from hermes_cli.models_local import fetch_lmstudio_models
from hermes_cli.models_validate import validate_requested_model
# -- helpers -----------------------------------------------------------------
+66 -48
View File
@@ -14,6 +14,8 @@ from hermes_cli.models import (
union_with_portal_paid_recommendations,
)
import hermes_cli.models as _models_mod
from hermes_cli import models_local
from hermes_cli import models_validate
LIVE_OPENROUTER_MODELS = [
("anthropic/claude-opus-4.6", "recommended"),
@@ -448,7 +450,7 @@ class TestCodexSoftAcceptPlausibilityGate:
and mislabel the provider as 'OpenAI Codex')."""
def test_unrelated_name_rejected_on_openai_codex(self):
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
r = validate_requested_model("qwen3.5-4b", "openai-codex")
assert r["accepted"] is False
assert r["persist"] is False
@@ -456,7 +458,7 @@ class TestCodexSoftAcceptPlausibilityGate:
def test_real_catalog_model_unaffected(self):
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
r = validate_requested_model("gpt-5.5", "openai-codex")
assert r["accepted"] is True
assert r["recognized"] is True
@@ -474,24 +476,24 @@ class TestFormatPricePerMtok:
"""_format_price_per_mtok: sub-cent prices must not collapse to 'free'/'$0.00'."""
def test_standard_prices_keep_two_decimals(self):
from hermes_cli.models import _format_price_per_mtok
from hermes_cli.models_pricing import _format_price_per_mtok
assert _format_price_per_mtok("0.000003") == "$3.00"
assert _format_price_per_mtok("0.00003") == "$30.00"
assert _format_price_per_mtok("0.00000015") == "$0.15"
assert _format_price_per_mtok("0.00018") == "$180.00"
def test_zero_is_free(self):
from hermes_cli.models import _format_price_per_mtok
from hermes_cli.models_pricing import _format_price_per_mtok
assert _format_price_per_mtok("0") == "free"
assert _format_price_per_mtok("0.0") == "free"
def test_invalid_is_question_mark(self):
from hermes_cli.models import _format_price_per_mtok
from hermes_cli.models_pricing import _format_price_per_mtok
assert _format_price_per_mtok("garbage") == "?"
assert _format_price_per_mtok(None) == "?"
def test_sub_cent_price_extends_precision(self):
from hermes_cli.models import _format_price_per_mtok
from hermes_cli.models_pricing import _format_price_per_mtok
# DeepSeek V4 Flash 0731 promo cache-hit rate: $0.0018/Mtok.
assert _format_price_per_mtok("0.0000000018") == "$0.0018"
assert _format_price_per_mtok("0.000000001") == "$0.001"
@@ -501,7 +503,7 @@ class TestFormatPricePerMtok:
assert _format_price_per_mtok("0.00000000001") == "$0.00001"
def test_one_cent_boundary_stays_two_decimals(self):
from hermes_cli.models import _format_price_per_mtok
from hermes_cli.models_pricing import _format_price_per_mtok
assert _format_price_per_mtok("0.00000001") == "$0.01"
@@ -590,7 +592,7 @@ class TestLocalOllamaModelDiscovery:
server.shutdown()
def test_native_tags_cache_expires(self, monkeypatch):
from hermes_cli.models import fetch_ollama_local_models
from hermes_cli.models_local import fetch_ollama_local_models
server, port = _start_fake_ollama_server(models=[{"name": "old-model"}])
try:
@@ -612,7 +614,7 @@ class TestLocalOllamaModelDiscovery:
def test_fetch_ollama_models_accepts_base_url_without_scheme(self):
"""OLLAMA_HOST commonly omits http://; discovery should normalize it."""
from hermes_cli.models import fetch_ollama_local_models
from hermes_cli.models_local import fetch_ollama_local_models
server, port = _start_fake_ollama_server(models=[{"name": "qwen2.5:1.5b"}])
try:
@@ -622,7 +624,7 @@ class TestLocalOllamaModelDiscovery:
def test_fetch_ollama_models_accepts_full_models_url(self):
"""Pasted OpenAI-style /v1/models URLs should normalize to the native root."""
from hermes_cli.models import fetch_ollama_local_models
from hermes_cli.models_local import fetch_ollama_local_models
server, port = _start_fake_ollama_server(models=[{"name": "qwen2.5:1.5b"}])
try:
@@ -636,10 +638,11 @@ class TestLocalOllamaModelDiscovery:
def test_runtime_error_from_config_load_does_not_escape_ollama_helpers(self):
"""Managed-mode config failures should degrade to defaults, not crash pickers."""
from hermes_cli.models import _get_ollama_base_url, should_use_ollama_native_catalog
from hermes_cli.models import _get_ollama_base_url
from hermes_cli.models_local import should_use_ollama_native_catalog
with patch("hermes_cli.config.load_config", side_effect=RuntimeError("bad home")), patch(
"hermes_cli.models.probe_ollama_local_models",
"hermes_cli.models_local.probe_ollama_local_models",
return_value=None,
):
assert _get_ollama_base_url() == "http://localhost:11434"
@@ -647,22 +650,22 @@ class TestLocalOllamaModelDiscovery:
def test_probe_ollama_models_malformed_base_url_returns_none(self):
"""Malformed user-configured URLs should behave like probe failures, not crashes."""
from hermes_cli.models import probe_ollama_local_models
from hermes_cli.models_local import probe_ollama_local_models
assert probe_ollama_local_models("http://127.0.0.1:bad-port/v1") is None
def test_fetch_ollama_models_preserves_probe_failure(self):
from hermes_cli.models import fetch_ollama_local_models
from hermes_cli.models_local import fetch_ollama_local_models
with patch("hermes_cli.models.probe_ollama_local_models", return_value=None):
with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=None):
assert fetch_ollama_local_models("http://127.0.0.1:11434") is None
def test_ollama_port_detection_requires_working_api_tags(self):
from hermes_cli.models import should_use_ollama_native_catalog
from hermes_cli.models_local import should_use_ollama_native_catalog
with patch("hermes_cli.models.probe_ollama_local_models", return_value=["qwen3:1.7b"]):
with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=["qwen3:1.7b"]):
assert should_use_ollama_native_catalog("custom", "192.168.1.5:11434/v1") is True
with patch("hermes_cli.models.probe_ollama_local_models", return_value=None):
with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=None):
assert should_use_ollama_native_catalog("custom", "192.168.1.5:11434/v1") is False
def test_provider_model_ids_ollama_cloud_config_uses_generic_catalog(self):
@@ -678,7 +681,7 @@ class TestLocalOllamaModelDiscovery:
}
}
},
), patch("hermes_cli.models.fetch_ollama_local_models") as fetch_local, patch(
), patch("hermes_cli.models_local.fetch_ollama_local_models") as fetch_local, patch(
"hermes_cli.models.fetch_api_models",
return_value=["qwen3:1.7b"],
) as fetch_generic:
@@ -691,7 +694,7 @@ class TestLocalOllamaModelDiscovery:
)
def test_native_ollama_catalog_uses_configured_key_env(self, monkeypatch):
from hermes_cli.models import _get_ollama_request_headers
from hermes_cli.models_local import _get_ollama_request_headers
monkeypatch.setenv("TEST_OLLAMA_API_KEY", "env-key")
with patch(
@@ -710,7 +713,7 @@ class TestLocalOllamaModelDiscovery:
}
def test_native_ollama_catalog_uses_api_key_env_alias(self, monkeypatch):
from hermes_cli.models import _get_ollama_request_headers
from hermes_cli.models_local import _get_ollama_request_headers
monkeypatch.setenv("TEST_OLLAMA_API_KEY_ALIAS", "alias-key")
with patch(
@@ -740,7 +743,7 @@ class TestLocalOllamaModelDiscovery:
}
},
), patch(
"hermes_cli.models.fetch_ollama_local_models",
"hermes_cli.models_local.fetch_ollama_local_models",
return_value=["qwen3:1.7b"],
) as fetch_local:
assert provider_model_ids("ollama", force_refresh=True) == ["qwen3:1.7b"]
@@ -757,7 +760,7 @@ class TestLocalOllamaModelDiscovery:
"base_url": "http://127.0.0.1:11434/v1",
}
},
), patch("hermes_cli.models.probe_ollama_local_models") as probe_ollama:
), patch("hermes_cli.models_local.probe_ollama_local_models") as probe_ollama:
assert _credential_fingerprint("ollama")
probe_ollama.assert_not_called()
@@ -801,6 +804,8 @@ class TestLocalOllamaModelDiscovery:
def test_clear_provider_models_cache_clears_ollama_native_tags_cache(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
cache = getattr(models, "_OLLAMA_LOCAL_MODELS_CACHE")
cache["http://127.0.0.1:11434"] = ("old-model",)
@@ -809,6 +814,8 @@ class TestLocalOllamaModelDiscovery:
def test_clear_provider_models_cache_custom_clears_native_tags_cache(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
cache = getattr(models, "_OLLAMA_LOCAL_MODELS_CACHE")
cache["http://127.0.0.1:11434"] = ("old-model",)
@@ -817,6 +824,8 @@ class TestLocalOllamaModelDiscovery:
def test_clear_provider_models_cache_does_not_remove_custom_disk_cache(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
disk_cache = {
"custom": {"models": ["custom-model"]},
@@ -830,13 +839,13 @@ class TestLocalOllamaModelDiscovery:
def test_ollama_cloud_urls_do_not_use_native_local_catalog(self):
from hermes_cli.models import should_use_ollama_native_catalog
from hermes_cli.models_local import should_use_ollama_native_catalog
assert should_use_ollama_native_catalog("ollama-cloud", "https://ollama.com/v1") is False
assert should_use_ollama_native_catalog("ollama", "https://ollama.com/v1") is False
def test_non_ollama_custom_endpoint_uses_generic_catalog_path(self):
from hermes_cli.models import should_use_ollama_native_catalog
from hermes_cli.models_local import should_use_ollama_native_catalog
assert should_use_ollama_native_catalog("custom", "https://example.test/v1") is False
assert should_use_ollama_native_catalog("openrouter", "http://localhost:11434/v1") is False
@@ -888,10 +897,10 @@ class TestLocalOllamaModelDiscovery:
from hermes_cli.model_switch import list_authenticated_providers
with patch("hermes_cli.config.load_config", return_value={"providers": {}}), patch(
"hermes_cli.models.should_use_ollama_native_catalog",
"hermes_cli.models_local.should_use_ollama_native_catalog",
return_value=True,
), patch(
"hermes_cli.models.fetch_ollama_local_models",
"hermes_cli.models_local.fetch_ollama_local_models",
return_value=["qwen3:1.7b"],
), patch("hermes_cli.models.fetch_api_models", return_value=[]) as fetch_api:
rows = list_authenticated_providers(
@@ -1090,7 +1099,7 @@ class TestLocalOllamaModelDiscovery:
from hermes_cli.model_switch import list_authenticated_providers
with patch("hermes_cli.config.load_config", return_value={"providers": {}}), patch(
"hermes_cli.models.probe_ollama_local_models",
"hermes_cli.models_local.probe_ollama_local_models",
return_value=["qwen3:1.7b"],
), patch("hermes_cli.models.fetch_api_models", return_value=[]) as fetch_api:
rows = list_authenticated_providers(
@@ -1105,7 +1114,7 @@ class TestLocalOllamaModelDiscovery:
def test_model_validation_uses_ollama_api_tags_for_ollama_provider(self):
"""`/model` validation for provider=ollama should not probe `/models`."""
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
server, port = _start_fake_ollama_server()
try:
@@ -1128,12 +1137,12 @@ class TestLocalOllamaModelDiscovery:
def test_model_validation_ollama_cloud_config_does_not_use_local_tags(self):
"""provider=ollama with a cloud base URL should not fall into local /api/tags."""
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
with patch(
"hermes_cli.config.load_config",
return_value={"providers": {"ollama": {"base_url": "https://ollama.com/v1"}}},
), patch("hermes_cli.models.probe_ollama_local_models") as probe_ollama, patch(
), patch("hermes_cli.models_local.probe_ollama_local_models") as probe_ollama, patch(
"hermes_cli.models.probe_api_models",
return_value={
"models": ["qwen3:1.7b"],
@@ -1152,7 +1161,7 @@ class TestLocalOllamaModelDiscovery:
def test_model_validation_uses_ollama_api_tags_for_matching_custom_endpoint(self):
"""Current-provider `custom` on the configured Ollama URL should use `/api/tags`."""
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
server, port = _start_fake_ollama_server()
base_url = f"http://127.0.0.1:{port}"
@@ -1178,7 +1187,7 @@ class TestLocalOllamaModelDiscovery:
def test_model_validation_empty_ollama_tags_does_not_fall_back_to_models(self):
"""Reachable but empty /api/tags should not produce a misleading /models warning."""
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
server, port = _start_fake_ollama_server(models=[])
base_url = f"http://127.0.0.1:{port}"
@@ -1247,7 +1256,7 @@ class TestLocalOllamaModelDiscovery:
"api_mode": "chat_completions",
},
), patch(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
return_value={
"accepted": True,
"persist": True,
@@ -1277,7 +1286,7 @@ class TestLocalOllamaModelDiscovery:
from hermes_cli.model_switch import switch_model
with patch(
"hermes_cli.models.should_use_ollama_native_catalog",
"hermes_cli.models_local.should_use_ollama_native_catalog",
side_effect=RuntimeError("config unavailable"),
), patch(
"hermes_cli.runtime_provider.resolve_runtime_provider",
@@ -1287,7 +1296,7 @@ class TestLocalOllamaModelDiscovery:
"api_mode": "chat_completions",
},
), patch(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
return_value={
"accepted": True,
"persist": True,
@@ -1313,7 +1322,7 @@ class TestLocalOllamaModelDiscovery:
assert result.api_key == "new-key"
def test_ollama_root_matching_is_case_insensitive_for_hostnames(self):
from hermes_cli.models import _same_ollama_native_root
from hermes_cli.models_local import _same_ollama_native_root
assert _same_ollama_native_root(
"HTTP://OLLAMA.EXAMPLE:11434/v1",
@@ -1339,6 +1348,8 @@ class TestLocalOllamaModelDiscovery:
def test_ollama_failed_probe_is_cached_briefly(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
models._OLLAMA_LOCAL_MODELS_CACHE.clear()
models._OLLAMA_LOCAL_PROBE_FAILURE_CACHE.clear()
@@ -1346,12 +1357,14 @@ class TestLocalOllamaModelDiscovery:
"hermes_cli.models._urlopen_model_catalog_request",
side_effect=OSError("offline"),
) as request:
assert models.probe_ollama_local_models("http://127.0.0.1:19999") is None
assert models.probe_ollama_local_models("http://127.0.0.1:19999") is None
assert models_local.probe_ollama_local_models("http://127.0.0.1:19999") is None
assert models_local.probe_ollama_local_models("http://127.0.0.1:19999") is None
request.assert_called_once()
def test_empty_ollama_catalog_does_not_resurrect_stale_disk_models(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
base_url = "http://127.0.0.1:11434"
probe_key = models._ollama_probe_cache_key(base_url, None)
@@ -1365,13 +1378,15 @@ class TestLocalOllamaModelDiscovery:
models, "_credential_fingerprint", return_value="same"
), patch.object(models, "provider_model_ids", return_value=[]), patch.object(
models, "_get_ollama_base_url", return_value=base_url
), patch.object(models, "_get_ollama_request_headers", return_value={}):
), patch.object(models_local, "_get_ollama_request_headers", return_value={}):
assert models.cached_provider_model_ids("ollama") == []
finally:
models._OLLAMA_LOCAL_PROBE_REACHABLE.pop(probe_key, None)
def test_failed_ollama_catalog_preserves_stale_disk_models(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
base_url = "http://127.0.0.1:11434"
probe_key = models._ollama_probe_cache_key(base_url, None)
@@ -1385,13 +1400,15 @@ class TestLocalOllamaModelDiscovery:
models, "_credential_fingerprint", return_value="same"
), patch.object(models, "provider_model_ids", return_value=[]), patch.object(
models, "_get_ollama_base_url", return_value=base_url
), patch.object(models, "_get_ollama_request_headers", return_value={}):
), patch.object(models_local, "_get_ollama_request_headers", return_value={}):
assert models.cached_provider_model_ids("ollama") == ["stale:model"]
finally:
models._OLLAMA_LOCAL_PROBE_REACHABLE.pop(probe_key, None)
def test_ollama_native_request_uses_redirect_safe_catalog_helper(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
response = MagicMock()
response.read.return_value = b'{"models": [{"name": "qwen3:1.7b"}]}'
@@ -1399,13 +1416,15 @@ class TestLocalOllamaModelDiscovery:
with patch.object(
models, "_urlopen_model_catalog_request", return_value=response
) as request:
assert models.fetch_ollama_local_models("http://127.0.0.1:11434") == [
assert models_local.fetch_ollama_local_models("http://127.0.0.1:11434") == [
"qwen3:1.7b"
]
request.assert_called_once()
def test_validation_with_nonmatching_ollama_root_does_not_forward_config_headers(self):
import hermes_cli.models as models
from hermes_cli import models_local
from hermes_cli import models_validate
with patch(
"hermes_cli.config.load_config",
@@ -1417,16 +1436,15 @@ class TestLocalOllamaModelDiscovery:
}
}
},
), patch.object(models, "should_use_ollama_native_catalog", return_value=True), patch.object(
models, "probe_ollama_local_models", return_value=[]
), patch.object(models_local, "should_use_ollama_native_catalog", return_value=True), patch.object(models_local, "probe_ollama_local_models", return_value=[]
) as probe:
models.validate_requested_model(
models_validate.validate_requested_model(
"qwen3:1.7b",
"ollama",
base_url="https://other.internal/v1",
)
assert probe.call_args.kwargs["headers"] == {}
models.validate_requested_model(
models_validate.validate_requested_model(
"qwen3:1.7b",
"ollama",
base_url="https://other.internal/v1",
@@ -1450,7 +1468,7 @@ class TestLocalOllamaModelDiscovery:
"hermes_cli.models._get_provider_config_dict",
return_value={"base_url": base_url, "api_key": "secret"},
) as config_provider, patch(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
return_value={"accepted": True, "persist": True, "recognized": True, "message": ""},
):
result = model_switch.switch_model(
@@ -1480,7 +1498,7 @@ class TestLocalOllamaModelDiscovery:
)
try:
with patch.object(model_switch, "get_model_info", return_value=None), patch(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
return_value={"accepted": True, "persist": True, "recognized": True, "message": ""},
):
result = model_switch.switch_model(
+8 -13
View File
@@ -12,12 +12,9 @@ import json
import pytest
import hermes_cli.models as models_mod
from hermes_cli import models_pricing
import hermes_cli.nous_account as account_mod
from hermes_cli.models import (
_NOUS_POLICY_APPEND_MAX,
nous_policy_allowed_ids,
restrict_to_nous_policy,
)
from hermes_cli.models_pricing import _NOUS_POLICY_APPEND_MAX, nous_policy_allowed_ids, restrict_to_nous_policy
from hermes_cli.nous_account import nous_policy_present
@@ -67,20 +64,18 @@ class TestRestrictToNousPolicy:
class TestNousPolicyAllowedIds:
@pytest.fixture(autouse=True)
def _clear_cache(self):
models_mod._pricing_cache.clear()
models_mod._pricing_cache_retry_after.clear()
models_pricing._pricing_cache.clear()
models_pricing._pricing_cache_retry_after.clear()
yield
models_mod._pricing_cache.clear()
models_mod._pricing_cache_retry_after.clear()
models_pricing._pricing_cache.clear()
models_pricing._pricing_cache_retry_after.clear()
def _patch(self, monkeypatch, *, policy_present, api_key="sk-test", pricing=None):
calls = []
monkeypatch.setattr(
account_mod, "nous_policy_present", lambda: policy_present
)
monkeypatch.setattr(
models_mod,
"_resolve_nous_pricing_credentials",
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
lambda: (api_key, "https://inference.example.com"),
)
@@ -88,7 +83,7 @@ class TestNousPolicyAllowedIds:
calls.append(kwargs)
return pricing if pricing is not None else {}
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch)
return calls
def test_returns_the_authenticated_catalog_keys(self, monkeypatch):
+16 -16
View File
@@ -11,7 +11,7 @@ import argparse
import pytest
import hermes_cli.models as models_mod
from hermes_cli import model_switch_providers
from hermes_cli import models_pricing
CURATED = ["vendor/allowed", "vendor/blocked"]
ALLOWED = {"vendor/allowed"}
@@ -20,14 +20,14 @@ ALLOWED = {"vendor/allowed"}
@pytest.fixture
def policy(monkeypatch):
"""An org whose policy admits only ``vendor/allowed``."""
monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: ALLOWED)
monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: ALLOWED)
return ALLOWED
@pytest.fixture
def no_policy(monkeypatch):
"""An unrestricted org — lists must come through untouched."""
monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: None)
monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: None)
class TestLoginNous:
@@ -51,7 +51,7 @@ class TestLoginNous:
},
)
monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED))
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {})
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
monkeypatch.setattr(
models_mod,
@@ -95,7 +95,7 @@ class TestModelSwitchPicker:
lambda *a, **k: {"providers": {"nous": {"access_token": "tok"}}},
)
monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED))
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {})
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
monkeypatch.setattr(
models_mod,
@@ -122,7 +122,7 @@ class TestModelSwitchPicker:
def _boom(_p):
raise RuntimeError("portal down")
monkeypatch.setattr(models_mod, "get_pricing_for_provider", _boom)
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", _boom)
row = self._rows(monkeypatch)
assert row is not None
assert "vendor/blocked" not in row["models"]
@@ -141,7 +141,7 @@ class TestRecommendedDefaultEndpoint:
models_mod, "get_curated_nous_model_ids",
lambda: ["vendor/blocked", "vendor/allowed"],
)
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {})
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
monkeypatch.setattr(
models_mod,
@@ -171,10 +171,10 @@ class TestAuxiliaryFastModel:
return {mid: {} for mid in catalog}
monkeypatch.setattr(
models_mod, "_resolve_nous_pricing_credentials",
models_pricing, "_resolve_nous_pricing_credentials",
lambda: ("sk-nous", "https://inference.example.com"),
)
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch)
picked = aux._fast_model_from_catalog("nous")
return picked, seen
@@ -187,7 +187,7 @@ class TestAuxiliaryFastModel:
import agent.auxiliary_client as aux
monkeypatch.setattr(
models_mod, "nous_policy_allowed_ids", lambda **_k: {"vendor/allowed"}
models_pricing, "nous_policy_allowed_ids", lambda **_k: {"vendor/allowed"}
)
monkeypatch.setattr(aux, "_FAST_MODEL_FAMILIES", ("vendor/",))
monkeypatch.setattr(aux, "_FAST_MODEL_EXCLUDE", ())
@@ -240,14 +240,14 @@ class TestAuxFallbackRespectsPolicy:
import agent.auxiliary_client as aux
import providers
monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: allowed)
monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: allowed)
monkeypatch.setattr(
models_mod, "_resolve_nous_pricing_credentials",
models_pricing, "_resolve_nous_pricing_credentials",
lambda: ("sk", "https://inference.example.com"),
)
# No fast-family match, so the catalog step yields nothing.
monkeypatch.setattr(
models_mod, "fetch_models_with_pricing",
models_pricing, "fetch_models_with_pricing",
lambda **_k: {"vendor/allowed-large": {}},
)
@@ -294,7 +294,7 @@ def test_titling_seeds_the_shared_catalog_entry_like_the_pickers(monkeypatch):
import agent.auxiliary_client as aux
monkeypatch.setattr(
models_mod, "_resolve_nous_pricing_credentials",
models_pricing, "_resolve_nous_pricing_credentials",
lambda: ("tok", "https://inference.example.com"),
)
seen: dict = {}
@@ -303,8 +303,8 @@ def test_titling_seeds_the_shared_catalog_entry_like_the_pickers(monkeypatch):
seen.update(kwargs)
return {"vendor/haiku": {}}
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch)
aux._fast_model_from_catalog("nous")
assert seen.get("include_sale_original") is True
assert seen.get("cache_ttl_seconds") == models_mod._NOUS_CATALOG_TTL_SECONDS
assert seen.get("cache_ttl_seconds") == models_pricing._NOUS_CATALOG_TTL_SECONDS
@@ -38,18 +38,19 @@ _CATALOG = (
def cold_cache(monkeypatch):
"""A freshly started process that has never mirrored a catalog to disk."""
import hermes_cli.models as models_mod
from hermes_cli import models_reasoning_caps
monkeypatch.setattr(models_mod, "_nous_reasoning_caps_cache", None)
monkeypatch.setattr(models_mod, "_nous_reasoning_caps_failed_at", None)
monkeypatch.setattr(models_mod, "_nous_caps_disk_checked", False)
monkeypatch.setattr(models_mod, "_nous_caps_warm_started", False)
models_mod._reasoning_caps_disk_path().unlink(missing_ok=True)
models_reasoning_caps._reasoning_caps_disk_path().unlink(missing_ok=True)
return models_mod
class TestNousModelReasoningCapabilities:
def test_fetch_parses_mandatory_flag(self, cold_cache, monkeypatch):
from hermes_cli.models import nous_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
monkeypatch.setattr(
cold_cache, "_urlopen_model_catalog_request",
@@ -67,7 +68,7 @@ class TestNousModelReasoningCapabilities:
def test_catalog_read_sends_user_agent(self, cold_cache, monkeypatch):
"""The Portal 403s an anonymous catalog read."""
from hermes_cli.models import nous_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
seen = []
@@ -89,7 +90,7 @@ class TestNousModelReasoningCapabilities:
Pinned to production, a staging profile would take its
reasoning-mandatory verdicts from a different deployment's catalog.
"""
from hermes_cli.models import nous_catalog_url
from hermes_cli.models_reasoning_caps import nous_catalog_url
monkeypatch.setenv(
"NOUS_INFERENCE_BASE_URL", "https://staging.nousresearch.com/v1"
@@ -100,7 +101,7 @@ class TestNousModelReasoningCapabilities:
assert nous_catalog_url().endswith("/v1/models")
def test_unlisted_and_empty_models_return_none(self, cold_cache, monkeypatch):
from hermes_cli.models import nous_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
monkeypatch.setattr(
cold_cache, "_urlopen_model_catalog_request",
@@ -111,7 +112,7 @@ class TestNousModelReasoningCapabilities:
assert nous_model_reasoning_capabilities(None) is None
def test_cache_only_by_default_never_fetches(self, cold_cache, monkeypatch):
from hermes_cli.models import nous_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
def _boom(req, *, timeout):
raise AssertionError("hot path must not fetch")
@@ -120,7 +121,7 @@ class TestNousModelReasoningCapabilities:
assert nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None
def test_unreachable_catalog_rate_limits_refetch(self, cold_cache, monkeypatch):
from hermes_cli.models import nous_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
calls = {"n": 0}
@@ -136,10 +137,7 @@ class TestNousModelReasoningCapabilities:
def test_openrouter_cache_is_independent(self, cold_cache, monkeypatch):
"""Two catalogs, two caches — a Portal fetch must not answer for OpenRouter."""
from hermes_cli.models import (
nous_model_reasoning_capabilities,
openrouter_model_reasoning_capabilities,
)
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities, openrouter_model_reasoning_capabilities
monkeypatch.setattr(cold_cache, "_openrouter_reasoning_caps_cache", None)
monkeypatch.setattr(cold_cache, "_openrouter_reasoning_caps_failed_at", None)
+2 -2
View File
@@ -386,7 +386,7 @@ class TestSwitchModelDirectAliasOverride:
lambda **kwargs: {"api_key": "", "base_url": "", "api_mode": "openai_compat", "provider": "custom"},
)
monkeypatch.setattr("hermes_cli.models.validate_requested_model",
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model",
lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None})
monkeypatch.setattr("hermes_cli.models.opencode_model_api_mode",
lambda *a, **kw: "openai_compat")
@@ -411,7 +411,7 @@ class TestSwitchModelDirectAliasOverride:
"hermes_cli.runtime_provider.resolve_runtime_provider",
lambda **kwargs: {"api_key": "", "base_url": "", "api_mode": "openai_compat", "provider": "custom"},
)
monkeypatch.setattr("hermes_cli.models.validate_requested_model",
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model",
lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None})
monkeypatch.setattr("hermes_cli.models.opencode_model_api_mode",
lambda *a, **kw: "openai_compat")
@@ -340,7 +340,7 @@ class TestOllamaCloudSuffixStripping:
def test_strip_suffix_helper(self):
"""Unit test for the _strip_ollama_cloud_suffix helper."""
from hermes_cli.models import _strip_ollama_cloud_suffix
from hermes_cli.models_local import _strip_ollama_cloud_suffix
assert _strip_ollama_cloud_suffix("kimi-k2.6:cloud") == "kimi-k2.6"
assert _strip_ollama_cloud_suffix("glm-5.1:cloud") == "glm-5.1"
@@ -18,7 +18,7 @@ it.
from unittest.mock import patch
from hermes_cli.model_switch import switch_model
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
def test_openai_codex_unknown_but_plausible_model_is_accepted_with_warning():
@@ -21,11 +21,12 @@ from unittest.mock import patch as mock_patch
import pytest
from hermes_cli import models as models_mod
from hermes_cli import models_validate
def _validate(requested: str, base_url: str, live: list[str], provider: str = "openai-api"):
with mock_patch.object(models_mod, "fetch_api_models", return_value=live):
return models_mod.validate_requested_model(
return models_validate.validate_requested_model(
requested,
provider=provider,
api_key="sk-test",
@@ -14,7 +14,7 @@ These tests cover the catalog-fallback path: when ``fetch_api_models`` returns
from unittest.mock import patch
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
_UNREACHABLE_PROBE = {
@@ -9,7 +9,7 @@ from unittest.mock import patch
import pytest
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
@pytest.mark.parametrize(
@@ -10,10 +10,8 @@ Covers:
import pytest
from hermes_cli.models import (
clamp_reasoning_effort_to_supported,
parse_openrouter_reasoning_capabilities,
)
from hermes_cli.models import clamp_reasoning_effort_to_supported
from hermes_cli.models_reasoning_caps import parse_openrouter_reasoning_capabilities
class TestParseReasoningCapabilities:
@@ -121,7 +119,7 @@ class TestOpenRouterModelReasoningCapabilities:
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None)
def test_known_model(self, monkeypatch):
from hermes_cli.models import openrouter_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
self._prime_cache(monkeypatch, {
"nvidia/nemotron-3-ultra": {
"supports_reasoning": True,
@@ -133,19 +131,19 @@ class TestOpenRouterModelReasoningCapabilities:
assert caps["supports_reasoning"] is True
def test_unlisted_model_returns_none(self, monkeypatch):
from hermes_cli.models import openrouter_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
self._prime_cache(monkeypatch, {"a/b": {"supports_reasoning": True}})
assert openrouter_model_reasoning_capabilities("private/custom") is None
def test_empty_model_returns_none(self, monkeypatch):
from hermes_cli.models import openrouter_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
self._prime_cache(monkeypatch, {"a/b": {"supports_reasoning": True}})
assert openrouter_model_reasoning_capabilities("") is None
assert openrouter_model_reasoning_capabilities(None) is None
def test_catalog_unreachable_returns_none_and_rate_limits(self, monkeypatch):
import hermes_cli.models as models_mod
from hermes_cli.models import openrouter_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_cache", None)
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None)
@@ -163,7 +161,7 @@ class TestOpenRouterModelReasoningCapabilities:
def test_cache_only_by_default_never_fetches(self, monkeypatch):
import hermes_cli.models as models_mod
from hermes_cli.models import openrouter_model_reasoning_capabilities
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_cache", None)
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None)
@@ -12,7 +12,8 @@ from unittest.mock import MagicMock
import pytest
import hermes_cli.models as models_mod
from hermes_cli.models import fetch_models_with_pricing, peek_cached_pricing
from hermes_cli import models_pricing
from hermes_cli.models_pricing import fetch_models_with_pricing, peek_cached_pricing
BASE = "https://inference-api.example.com"
@@ -23,11 +24,11 @@ _FILTERED = ["vendor/allowed"]
@pytest.fixture(autouse=True)
def _clear_pricing_cache():
models_mod._pricing_cache.clear()
models_mod._pricing_cache_retry_after.clear()
models_pricing._pricing_cache.clear()
models_pricing._pricing_cache_retry_after.clear()
yield
models_mod._pricing_cache.clear()
models_mod._pricing_cache_retry_after.clear()
models_pricing._pricing_cache.clear()
models_pricing._pricing_cache_retry_after.clear()
@pytest.fixture
@@ -96,7 +97,7 @@ def test_one_token_does_not_receive_another_tokens_catalog(per_org_catalog):
def test_credential_value_does_not_appear_in_the_cache_key():
"""Guards against keying on the raw token."""
assert "sk-super-secret" not in models_mod._pricing_auth_fingerprint("sk-super-secret")
assert "sk-super-secret" not in models_pricing._pricing_auth_fingerprint("sk-super-secret")
def test_anonymous_and_authenticated_reads_are_separate(catalog):
@@ -159,7 +160,7 @@ class TestNousCatalogExpiry:
a long-lived process holds the entry."""
def test_entry_expires_so_a_policy_change_is_picked_up(self, catalog, monkeypatch):
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS
fetch_models_with_pricing(
api_key="sk-test", base_url=BASE,
@@ -196,7 +197,7 @@ class TestNousCatalogExpiry:
def test_peek_skips_an_expired_entry(self, catalog, monkeypatch):
"""Reading _pricing_cache directly walked straight past the TTL."""
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS
fetch_models_with_pricing(
api_key="sk-test", base_url=BASE,
+2 -6
View File
@@ -6,12 +6,8 @@ These are invariant tests, not catalog snapshots: they assert how
vendors, which is expected to change over time.
"""
from hermes_cli.models import (
CANONICAL_PROVIDERS,
PROVIDER_GROUPS,
group_providers,
provider_group_for_slug,
)
from hermes_cli.models import CANONICAL_PROVIDERS
from hermes_cli.models_catalog_static import PROVIDER_GROUPS, group_providers, provider_group_for_slug
def _slugs(rows):
@@ -17,6 +17,8 @@ import json
import pytest
import hermes_cli.models as models_mod
from hermes_cli import models_pricing
from hermes_cli import models_reasoning_caps
_CATALOG = json.dumps({
@@ -95,16 +97,16 @@ def test_fetched_catalog_answers_a_later_process_offline(
models_mod, "_urlopen_model_catalog_request",
lambda req, *, timeout: _response(_CATALOG),
)
assert models_mod.nous_model_reasoning_capabilities(
assert models_reasoning_caps.nous_model_reasoning_capabilities(
"deepseek/deepseek-v4-pro", allow_fetch=True
) is not None
cold_process()
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
assert caps["mandatory"] is False
mandatory = models_mod.nous_model_reasoning_capabilities(
mandatory = models_reasoning_caps.nous_model_reasoning_capabilities(
"arcee-ai/trinity-large-thinking"
)
assert mandatory["mandatory"] is True
@@ -121,14 +123,14 @@ def test_mirror_is_keyed_by_catalog_url(cold_process, offline, monkeypatch):
models_mod, "_urlopen_model_catalog_request",
lambda req, *, timeout: _response(_CATALOG),
)
models_mod.nous_model_reasoning_capabilities(
models_reasoning_caps.nous_model_reasoning_capabilities(
"deepseek/deepseek-v4-pro", allow_fetch=True
)
cold_process()
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
assert models_mod.openrouter_model_reasoning_capabilities(
assert models_reasoning_caps.openrouter_model_reasoning_capabilities(
"deepseek/deepseek-v4-pro"
) is None
@@ -140,7 +142,7 @@ def test_staging_portal_does_not_read_productions_mirror(
models_mod, "_urlopen_model_catalog_request",
lambda req, *, timeout: _response(_CATALOG),
)
models_mod.nous_model_reasoning_capabilities(
models_reasoning_caps.nous_model_reasoning_capabilities(
"deepseek/deepseek-v4-pro", allow_fetch=True
)
@@ -148,7 +150,7 @@ def test_staging_portal_does_not_read_productions_mirror(
monkeypatch.setenv("NOUS_INFERENCE_BASE_URL", "https://staging.nousresearch.com")
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
assert models_mod.nous_model_reasoning_capabilities(
assert models_reasoning_caps.nous_model_reasoning_capabilities(
"deepseek/deepseek-v4-pro"
) is None
@@ -159,29 +161,29 @@ def test_stale_copy_is_still_served(cold_process, offline, monkeypatch):
Refusing to read an aged mirror would put every long-idle install back on
the cold-start fallback it exists to prevent.
"""
url = models_mod.nous_catalog_url()
models_mod._save_reasoning_caps_disk(
url = models_reasoning_caps.nous_catalog_url()
models_reasoning_caps._save_reasoning_caps_disk(
url, {"deepseek/deepseek-v4-pro": {"supports_reasoning": True, "mandatory": False}}
)
raw = json.loads(models_mod._reasoning_caps_disk_path().read_text())
raw = json.loads(models_reasoning_caps._reasoning_caps_disk_path().read_text())
raw[url]["ts"] = 0 # epoch — far past any TTL
models_mod._reasoning_caps_disk_path().write_text(json.dumps(raw))
models_reasoning_caps._reasoning_caps_disk_path().write_text(json.dumps(raw))
cold_process()
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
assert caps["mandatory"] is False
def test_unreadable_mirror_degrades_to_unknown(cold_process, offline, monkeypatch):
"""A corrupt file answers "unknown", never raises into the request path."""
path = models_mod._reasoning_caps_disk_path()
path = models_reasoning_caps._reasoning_caps_disk_path()
path.parent.mkdir(parents=True, exist_ok=True)
path.write_text("{ this is not json")
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
assert models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None
assert models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None
def test_pricing_fetch_seeds_the_mirror(cold_process, offline, monkeypatch):
@@ -190,20 +192,20 @@ def test_pricing_fetch_seeds_the_mirror(cold_process, offline, monkeypatch):
Every surface that renders prices goes through here, so the common case
never pays a second round-trip to learn the same thing.
"""
monkeypatch.setattr(models_mod, "_pricing_cache", {})
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
monkeypatch.setattr(models_pricing, "_pricing_cache", {})
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
monkeypatch.setattr(
models_mod, "_urlopen_model_catalog_request",
lambda req, *, timeout: _response(_CATALOG),
)
models_mod.fetch_models_with_pricing(
models_pricing.fetch_models_with_pricing(
base_url="https://inference-api.nousresearch.com"
)
cold_process()
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
assert caps is not None
@@ -215,8 +217,8 @@ def test_a_missing_mirror_is_looked_for_once_per_process(cold_process, monkeypat
token. Both halves have to be paid at most once.
"""
counts = {"url": 0, "read": 0}
real_url = models_mod.nous_catalog_url
real_read = models_mod._read_reasoning_caps_disk
real_url = models_reasoning_caps.nous_catalog_url
real_read = models_reasoning_caps._read_reasoning_caps_disk
def _counting_url():
counts["url"] += 1
@@ -226,9 +228,9 @@ def test_a_missing_mirror_is_looked_for_once_per_process(cold_process, monkeypat
counts["read"] += 1
return real_read()
monkeypatch.setattr(models_mod, "nous_catalog_url", _counting_url)
monkeypatch.setattr(models_mod, "_read_reasoning_caps_disk", _counting_read)
monkeypatch.setattr(models_reasoning_caps, "nous_catalog_url", _counting_url)
monkeypatch.setattr(models_reasoning_caps, "_read_reasoning_caps_disk", _counting_read)
for _ in range(5):
models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
assert counts == {"url": 1, "read": 1}
+11 -13
View File
@@ -6,10 +6,8 @@ import json
from unittest.mock import MagicMock
import hermes_cli.models as models_mod
from hermes_cli.models import (
compute_sale_discount,
fetch_models_with_pricing,
)
from hermes_cli import models_pricing
from hermes_cli.models_pricing import compute_sale_discount, fetch_models_with_pricing
def test_free_model_gets_flat_100_percent_discount():
@@ -31,7 +29,7 @@ def test_paid_model_without_original_shows_no_sale():
def test_fetch_models_with_pricing_copies_nested_original(monkeypatch):
models_mod._pricing_cache.clear()
models_pricing._pricing_cache.clear()
payload = {
"data": [
{
@@ -100,7 +98,7 @@ def test_resolve_nous_pricing_credentials_honors_inference_env_override(monkeypa
"hermes_cli.auth.resolve_nous_runtime_credentials",
lambda: None,
)
api_key, base_url = models_mod._resolve_nous_pricing_credentials()
api_key, base_url = models_pricing._resolve_nous_pricing_credentials()
assert api_key == ""
# The bare origin, whichever form the override was written in: callers
# append their own path (``/v1/models``), so a suffix here would double up.
@@ -119,7 +117,7 @@ def test_resolve_nous_pricing_credentials_normalizes_either_suffix(monkeypatch):
"https://stg-inference-api.nousresearch.com/v1/",
):
monkeypatch.setenv("NOUS_INFERENCE_BASE_URL", override)
assert models_mod._resolve_nous_pricing_credentials()[1] == (
assert models_pricing._resolve_nous_pricing_credentials()[1] == (
"https://stg-inference-api.nousresearch.com"
)
@@ -131,8 +129,8 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch):
call, but it expires — the processes that read this run for weeks, and
every caller silently falls back to a curated list meanwhile.
"""
models_mod._pricing_cache.clear()
models_mod._pricing_cache_retry_after.clear()
models_pricing._pricing_cache.clear()
models_pricing._pricing_cache_retry_after.clear()
calls = []
@@ -151,7 +149,7 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch):
monkeypatch.setattr(
models_mod.time,
"monotonic",
lambda: now + models_mod._FAILED_CATALOG_TTL_SECONDS + 1,
lambda: now + models_pricing._FAILED_CATALOG_TTL_SECONDS + 1,
)
assert fetch_models_with_pricing(base_url="https://example.test") == {}
assert len(calls) == 2
@@ -159,8 +157,8 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch):
def test_a_successful_catalog_fetch_stays_cached(monkeypatch):
"""Only the failures expire; a real catalog is still fetched once."""
models_mod._pricing_cache.clear()
models_mod._pricing_cache_retry_after.clear()
models_pricing._pricing_cache.clear()
models_pricing._pricing_cache_retry_after.clear()
calls = []
body = json.dumps(
@@ -182,7 +180,7 @@ def test_a_successful_catalog_fetch_stays_cached(monkeypatch):
monkeypatch.setattr(
models_mod.time,
"monotonic",
lambda: now + models_mod._FAILED_CATALOG_TTL_SECONDS + 1,
lambda: now + models_pricing._FAILED_CATALOG_TTL_SECONDS + 1,
)
assert "a/b" in fetch_models_with_pricing(base_url="https://example.test")
assert len(calls) == 1
@@ -72,7 +72,7 @@ def test_show_status_reports_empty_lmstudio_listing_as_reachable(monkeypatch, ca
monkeypatch.setattr(status_mod, "resolve_provider", lambda requested=None, **kwargs: "lmstudio", raising=False)
monkeypatch.setattr(status_mod, "provider_label", lambda provider: "LM Studio", raising=False)
monkeypatch.setattr(
"hermes_cli.models.probe_lmstudio_models",
"hermes_cli.models_local.probe_lmstudio_models",
lambda api_key=None, base_url=None, timeout=5.0: [],
)
@@ -378,7 +378,7 @@ def test_switch_model_resolves_user_provider_credentials(monkeypatch, tmp_path):
# Mock validation to pass
monkeypatch.setattr(
"hermes_cli.models.validate_requested_model",
"hermes_cli.models_validate.validate_requested_model",
lambda *a, **k: {"accepted": True, "persist": True, "recognized": True, "message": None}
)
@@ -443,7 +443,7 @@ def _run_user_provider_override_case(
with patch("hermes_cli.model_switch.resolve_alias", return_value=None), \
patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \
patch("hermes_cli.model_switch.normalize_model_for_provider", side_effect=lambda model, provider: model), \
patch("hermes_cli.models.validate_requested_model", return_value=_REJECTED_VALIDATION), \
patch("hermes_cli.models_validate.validate_requested_model", return_value=_REJECTED_VALIDATION), \
patch("hermes_cli.models.detect_provider_for_model", return_value=None), \
patch("hermes_cli.model_switch.get_model_info", return_value=None), \
patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \
@@ -199,7 +199,7 @@ class TestOllamaModelSupportsThinking:
monkeypatch.setattr(httpx, "Client", _Client)
def test_thinking_capability_true(self, monkeypatch):
from hermes_cli.models import ollama_model_supports_thinking
from hermes_cli.models_local import ollama_model_supports_thinking
self._patch_show(monkeypatch, capabilities=["completion", "tools", "thinking"])
assert (
@@ -211,7 +211,7 @@ class TestOllamaModelSupportsThinking:
def test_probe_failure_returns_none(self, monkeypatch):
from hermes_cli.models import ollama_model_supports_thinking
from hermes_cli.models_local import ollama_model_supports_thinking
self._patch_show(monkeypatch, status=404)
assert (
@@ -219,7 +219,7 @@ class TestOllamaModelSupportsThinking:
)
def test_exception_returns_none(self, monkeypatch):
from hermes_cli.models import ollama_model_supports_thinking
from hermes_cli.models_local import ollama_model_supports_thinking
self._patch_show(monkeypatch, raise_exc=RuntimeError("boom"))
assert (
+2 -2
View File
@@ -1,7 +1,7 @@
from types import SimpleNamespace
from typing import Any, cast
from hermes_cli.models import LMStudioLoadResult
from hermes_cli.models_local import LMStudioLoadResult
from run_agent import AIAgent
@@ -25,7 +25,7 @@ def test_lmstudio_jit_load_mode_skips_explicit_preload(monkeypatch):
calls.append((args, kwargs))
return LMStudioLoadResult(64_000)
monkeypatch.setattr("hermes_cli.models.ensure_lmstudio_model_loaded", fake_ensure)
monkeypatch.setattr("hermes_cli.models_local.ensure_lmstudio_model_loaded", fake_ensure)
result = AIAgent._ensure_lmstudio_runtime_loaded(cast(Any, _agent("jit")))
+1 -1
View File
@@ -4,7 +4,7 @@ from unittest.mock import MagicMock, patch
import pytest
from hermes_cli.models import LMStudioLoadResult
from hermes_cli.models_local import LMStudioLoadResult
from run_agent import AIAgent
from hermes_cli.route_identity import normalize_route_base_url
from agent.context_compressor import ContextCompressor
+1 -1
View File
@@ -8,7 +8,7 @@ from unittest.mock import patch
import pytest
from hermes_cli.models import validate_requested_model
from hermes_cli.models_validate import validate_requested_model
class TestMiniMaxModelValidation: