diff --git a/acp_adapter/model_catalog.py b/acp_adapter/model_catalog.py index 120060cb25..c6ac6045d4 100644 --- a/acp_adapter/model_catalog.py +++ b/acp_adapter/model_catalog.py @@ -29,7 +29,7 @@ def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, st from hermes_cli.model_switch import _declared_model_ids, _entry_models_discovered, _models_config_is_allowlist from hermes_cli.model_switch_providers import _NativePickerModelList, _fetch_picker_live_models from hermes_cli.model_switch_providers import _discover_flag - from hermes_cli.models import should_use_ollama_native_catalog + from hermes_cli.models_local import should_use_ollama_native_catalog from hermes_cli.providers import custom_provider_slug except ImportError: return [] diff --git a/agent/auxiliary_client.py b/agent/auxiliary_client.py index 4fc976f7e3..4172e4f515 100644 --- a/agent/auxiliary_client.py +++ b/agent/auxiliary_client.py @@ -689,7 +689,7 @@ def _fast_model_from_catalog(provider_id: str) -> str: is_nous = provider_id.strip().lower() == "nous" try: from hermes_cli.auth import resolve_api_key_provider_credentials - from hermes_cli.models import fetch_models_with_pricing + from hermes_cli.models_pricing import fetch_models_with_pricing from providers import get_provider_profile # Most /v1/models endpoints are authenticated; an anonymous 401 would read as "no small # model" and pin the curated default forever. @@ -704,7 +704,7 @@ def _fast_model_from_catalog(provider_id: str) -> str: if not api_key and is_nous: # Nous is OAuth (resolver raises); anonymous reads return the full catalog. try: - from hermes_cli.models import _resolve_nous_pricing_credentials + from hermes_cli.models_pricing import _resolve_nous_pricing_credentials api_key, base_url = _resolve_nous_pricing_credentials() except Exception: logger.debug("No Nous credentials for catalog", exc_info=True) @@ -719,7 +719,7 @@ def _fast_model_from_catalog(provider_id: str) -> str: # policy-catalog expiry. _nous_kwargs = {} if is_nous: - from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS + from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS _nous_kwargs = {"include_sale_original": True, "cache_ttl_seconds": _NOUS_CATALOG_TTL_SECONDS} catalog = fetch_models_with_pricing( api_key=api_key or None, base_url=base_url, timeout=3.0, **_nous_kwargs) or {} @@ -730,7 +730,7 @@ def _fast_model_from_catalog(provider_id: str) -> str: if is_nous: # Narrow catalog ids by org policy, as the pickers do. try: - from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy + from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy ids = restrict_to_nous_policy(ids, nous_policy_allowed_ids()) except Exception: logger.debug("Nous policy filter unavailable", exc_info=True) @@ -770,7 +770,7 @@ def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False) # let the caller keep the main model. if picked and provider_id.strip().lower() == "nous": try: - from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy + from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy allowed = nous_policy_allowed_ids() if allowed and not restrict_to_nous_policy([picked], allowed): return "" diff --git a/agent/credits_tracker.py b/agent/credits_tracker.py index 0b54133550..cd6005cbd8 100644 --- a/agent/credits_tracker.py +++ b/agent/credits_tracker.py @@ -131,7 +131,8 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool: if not base_url: return False try: - from hermes_cli.models import _is_model_free, peek_cached_pricing + from hermes_cli.models import _is_model_free + from hermes_cli.models_pricing import peek_cached_pricing pricing = peek_cached_pricing(base_url) # owns the /v1-suffix and auth-state key details return bool(pricing) and _is_model_free(model, pricing) diff --git a/agent/reasoning_params.py b/agent/reasoning_params.py index 973fb64ab1..17711e73d5 100644 --- a/agent/reasoning_params.py +++ b/agent/reasoning_params.py @@ -71,7 +71,7 @@ class ReasoningParamsMixin: # Live-catalog metadata first (OpenRouter /v1/models supported_parameters) — the static prefix # allowlist repeatedly went stale one vendor at a time. Unknown falls back to the static list. try: - from hermes_cli.models import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async + from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async caps = openrouter_model_reasoning_capabilities(self.model) if caps is None: warm_openrouter_reasoning_caps_async() # cache cold — warm in the background, never block @@ -85,7 +85,7 @@ class ReasoningParamsMixin: def _lmstudio_reasoning_options_cached(self) -> list[str]: """LM Studio's published reasoning ``allowed_options`` (gate + clamp so toggle models don't 400 on ``high``).""" try: - from hermes_cli.models import lmstudio_model_reasoning_options + from hermes_cli.models_local import lmstudio_model_reasoning_options except Exception: return [] return _cached_probe(self, "_lm_reasoning_opts_cache", lmstudio_model_reasoning_options, [], bool) @@ -93,7 +93,7 @@ class ReasoningParamsMixin: def _ollama_supports_thinking_cached(self) -> bool: """True only if Ollama's ``/api/show`` declares the ``thinking`` capability.""" try: - from hermes_cli.models import ollama_model_supports_thinking + from hermes_cli.models_local import ollama_model_supports_thinking except Exception: return False return bool(_cached_probe(self, "_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None)) diff --git a/hermes_cli/auth_model_picker.py b/hermes_cli/auth_model_picker.py index 181d7c5497..1eda39aad0 100644 --- a/hermes_cli/auth_model_picker.py +++ b/hermes_cli/auth_model_picker.py @@ -59,7 +59,7 @@ class _ModelPickerRows: self, all_models: List[str], pricing: Optional[Dict[str, Dict[str, str]]], *, current_model: str, sale_chrome: bool, ) -> None: - from hermes_cli.models import _format_price_per_mtok, compute_sale_discount + from hermes_cli.models_pricing import _format_price_per_mtok, compute_sale_discount self.current_model = current_model self.has_pricing = bool(pricing and any(pricing.get(m) for m in all_models)) # Leave room for a leading "★ " on sale rows (Nous only). diff --git a/hermes_cli/auth_nous.py b/hermes_cli/auth_nous.py index 7ec409f7ab..7289291d13 100644 --- a/hermes_cli/auth_nous.py +++ b/hermes_cli/auth_nous.py @@ -1328,9 +1328,17 @@ def _pick_nous_model_after_login( if not isinstance(runtime_key, str) or not runtime_key: raise _nous_err("No runtime API key available to fetch models", "invalid_token") from hermes_cli.models import ( - get_curated_nous_model_ids, get_pricing_for_provider, check_nous_free_tier, - partition_nous_models_by_tier, nous_policy_allowed_ids, restrict_to_nous_policy, - union_with_portal_free_recommendations, union_with_portal_paid_recommendations) + get_curated_nous_model_ids, + check_nous_free_tier, + partition_nous_models_by_tier, + union_with_portal_free_recommendations, + union_with_portal_paid_recommendations, + ) + from hermes_cli.models_pricing import ( + get_pricing_for_provider, + nous_policy_allowed_ids, + restrict_to_nous_policy, + ) model_ids = get_curated_nous_model_ids() _portal = auth_state.get("portal_base_url", "") print() diff --git a/hermes_cli/inventory.py b/hermes_cli/inventory.py index 4102115891..5cd90fb988 100644 --- a/hermes_cli/inventory.py +++ b/hermes_cli/inventory.py @@ -249,9 +249,11 @@ def _reasoning_catalog_reader(slug: str): """Per-model reasoning-capability reader for aggregators that publish one. Cache-only — the picker must never block on HTTP; a cold cache warms in the background and reports no restriction until then.""" try: - from hermes_cli.models import ( - nous_model_reasoning_capabilities, openrouter_model_reasoning_capabilities, - warm_nous_reasoning_caps_async, warm_openrouter_reasoning_caps_async, + from hermes_cli.models_reasoning_caps import ( + nous_model_reasoning_capabilities, + openrouter_model_reasoning_capabilities, + warm_nous_reasoning_caps_async, + warm_openrouter_reasoning_caps_async, ) except Exception: return None @@ -573,9 +575,15 @@ def _apply_pricing(rows: list[dict], *, force_fresh_nous_tier: bool = False, cac ``free_tier`` (account is free-tier) and ``unavailable_models`` (paid models a free user can't pick). ``cached_only`` never hits the network: unknown Nous entitlement fails closed (``free_tier_pending``, all models locked) and missing pricing is marked ``pricing_pending``.""" + from hermes_cli.models_pricing import ( + _format_price_per_mtok, + compute_sale_discount, + get_pricing_for_provider, + ) from hermes_cli.models import ( - _format_price_per_mtok, check_nous_free_tier, compute_sale_discount, get_cached_nous_free_tier, - get_pricing_for_provider, partition_nous_models_by_tier, + check_nous_free_tier, + get_cached_nous_free_tier, + partition_nous_models_by_tier, ) nous_free_tier: Optional[bool] = None # resolved once (cached in models.py for the TTL window) @@ -686,7 +694,7 @@ def _prewarm_pricing_async( """Warm picker pricing caches without delaying the current payload (one worker per profile + endpoint scope; a live worker is reused).""" from hermes_constants import hermes_home_key - from hermes_cli.models import pricing_cache_scope + from hermes_cli.models_pricing import pricing_cache_scope slugs = {str(row.get("slug") or "").lower() for row in rows if row.get("slug")} endpoint_scope = tuple(sorted( diff --git a/hermes_cli/main_provider_setup.py b/hermes_cli/main_provider_setup.py index c35bcb2f51..a8a89fb974 100644 --- a/hermes_cli/main_provider_setup.py +++ b/hermes_cli/main_provider_setup.py @@ -260,7 +260,7 @@ def _aux_flow_provider_model(task: str, provider_slug: str, curated_models: list current_model: str = "") -> None: """Prompt for a model under an already-authenticated provider, save to aux.""" from hermes_cli.auth import _prompt_model_selection - from hermes_cli.models import get_pricing_for_provider + from hermes_cli.models_pricing import get_pricing_for_provider display_name = _aux_task_display_name(task) try: pricing = get_pricing_for_provider(provider_slug) or {} @@ -780,7 +780,8 @@ def _build_provider_picker_rows(config: dict, active: str, provider_labels: dict fold into display groups (PROVIDER_GROUPS): a group row's ``members`` drive a sub-picker, leaf rows have ``members == []``; saved custom providers and trailing actions stay flat. Honors ``model_catalog.excluded_providers`` (slug or alias, case-insensitive) like the gateway/TUI.""" - from hermes_cli.models import CANONICAL_PROVIDERS, _PROVIDER_ALIASES, group_providers, provider_group_for_slug + from hermes_cli.models import CANONICAL_PROVIDERS, _PROVIDER_ALIASES + from hermes_cli.models_catalog_static import group_providers, provider_group_for_slug canonical_descs = {p.slug: p.tui_desc for p in CANONICAL_PROVIDERS} _cli_excluded = { str(p).strip().lower() diff --git a/hermes_cli/model_setup_flows_custom.py b/hermes_cli/model_setup_flows_custom.py index 7c0711c46e..91d101b906 100644 --- a/hermes_cli/model_setup_flows_custom.py +++ b/hermes_cli/model_setup_flows_custom.py @@ -207,9 +207,12 @@ def _discover_named_custom_models(provider_info: dict, api_key: str, configured_ """Live catalog probe for a named custom endpoint (native ``/api/tags`` for Ollama). Returns ``(models, native_catalog_empty)``; persists the live catalog as a side effect.""" from hermes_cli.config import normalize_extra_headers - from hermes_cli.models import ( - fetch_api_models, fetch_ollama_local_models, _get_ollama_native_headers, _normalize_openai_base_url, - should_use_ollama_native_catalog) + from hermes_cli.models import fetch_api_models, _get_ollama_native_headers + from hermes_cli.models_local import ( + fetch_ollama_local_models, + _normalize_openai_base_url, + should_use_ollama_native_catalog, + ) name, base_url = provider_info["name"], provider_info["base_url"] api_mode = provider_info.get("api_mode", "") diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 45c2dd1397..6649653f60 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -976,7 +976,7 @@ def _apply_direct_alias_endpoint(st: "_Switch", da: DirectAlias) -> None: own endpoint decides: its declared key; else the session key only for the SAME ORIGIN; else a fresh resolution against the alias base_url (env-key fallbacks are host-gated: OLLAMA_API_KEY resolves for ollama.com, OPENROUTER_API_KEY never reaches an unrelated host).""" - from hermes_cli.models import _same_ollama_native_root + from hermes_cli.models_local import _same_ollama_native_root from hermes_cli.runtime_provider import resolve_runtime_provider alias_key = direct_alias_api_key(da) same_host = _may_reuse_session_credential(st.base_url, da.base_url) @@ -1282,12 +1282,12 @@ def _creds_for_current_provider(st: _Switch) -> None: """Credentials when staying on the current provider. Mid-session ``/model `` on a local Ollama-compatible endpoint keeps the endpoint in use; re-resolving bare ``custom`` from config can fall through to an unrelated default provider.""" - from hermes_cli.models import _get_ollama_request_headers, _same_ollama_native_root + from hermes_cli.models_local import _get_ollama_request_headers, _same_ollama_native_root keep_current_ollama_endpoint = False ollama_headers: dict[str, str] = {} if st.current_provider == "custom" and st.current_base_url: try: - from hermes_cli.models import should_use_ollama_native_catalog + from hermes_cli.models_local import should_use_ollama_native_catalog ollama_headers = _get_ollama_request_headers() _, configured_ollama_base = _ollama_configured_base() # Provider-level Ollama headers only belong to the configured native root; without @@ -1343,7 +1343,8 @@ def _resolve_switch_credentials(st: _Switch) -> Optional[ModelSwitchResult]: def _validate_switch(st: _Switch) -> Optional[ModelSwitchResult]: """COMMON PATH part 2: normalize the model name for the target provider, validate it, and accept config-declared models the remote catalog lacks.""" - from hermes_cli.models import _get_ollama_request_headers, validate_requested_model + from hermes_cli.models_local import _get_ollama_request_headers + from hermes_cli.models_validate import validate_requested_model st.new_model = _resolve_named_custom_model_id(st.new_model, st.target_provider, st.custom_providers) st.new_model = normalize_model_for_provider(st.new_model, st.target_provider) diff --git a/hermes_cli/model_switch_providers.py b/hermes_cli/model_switch_providers.py index 8580599643..548460e063 100644 --- a/hermes_cli/model_switch_providers.py +++ b/hermes_cli/model_switch_providers.py @@ -89,9 +89,12 @@ def _fetch_picker_live_models( headers: dict[str, str] | None = None, timeout: float = 5.0, api_mode: str | None = None) -> list[str] | None: """Fetch picker models with native Ollama and cached generic discovery.""" - from hermes_cli.models import ( - _get_ollama_native_headers, _normalize_openai_base_url, cached_fetch_api_models, - fetch_ollama_local_models, should_use_ollama_native_catalog) + from hermes_cli.models import _get_ollama_native_headers, cached_fetch_api_models + from hermes_cli.models_local import ( + _normalize_openai_base_url, + fetch_ollama_local_models, + should_use_ollama_native_catalog, + ) candidate_headers = _get_ollama_native_headers(api_url, api_key=api_key) @@ -395,9 +398,12 @@ def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list: recommendation fetch still yields a policy-filtered curated list.""" model_ids = curated.get("nous", []) try: + from hermes_cli.models_pricing import get_pricing_for_provider from hermes_cli.models import ( - get_pricing_for_provider, check_nous_free_tier, union_with_portal_free_recommendations, - union_with_portal_paid_recommendations) + check_nous_free_tier, + union_with_portal_free_recommendations, + union_with_portal_paid_recommendations, + ) from hermes_cli.auth import get_provider_auth_state pricing = get_pricing_for_provider("nous") or {} try: @@ -411,7 +417,7 @@ def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list: except Exception: pass try: - from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy + from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy model_ids = restrict_to_nous_policy(model_ids, nous_policy_allowed_ids(), rescue_empty=True) except Exception: pass @@ -996,7 +1002,7 @@ def _build_curated_lists(current_provider: str, current_base_url: str, current_m # unreachable, fall back to the current model so the picker still shows something offline. is_current_lmstudio = current_provider.strip().lower() == "lmstudio" if "lmstudio" not in curated and (os.environ.get("LM_API_KEY") or os.environ.get("LM_BASE_URL") or is_current_lmstudio): - from hermes_cli.models import fetch_lmstudio_models + from hermes_cli.models_local import fetch_lmstudio_models from hermes_cli.auth import AuthError lm_base = ( os.environ.get("LM_BASE_URL") diff --git a/hermes_cli/models.py b/hermes_cli/models.py index dac18836e3..541d23f1e8 100644 --- a/hermes_cli/models.py +++ b/hermes_cli/models.py @@ -26,12 +26,10 @@ if TYPE_CHECKING: from hermes_cli import __version__ as _HERMES_VERSION from hermes_cli.urllib_security import open_credentialed_url -from hermes_cli.models_catalog_static import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.) +from hermes_cli.models_catalog_static import ( CANONICAL_PROVIDERS, OPENROUTER_MODELS, PREFERRED_SILENT_DEFAULT_MODEL, - PROVIDER_GROUPS, - ProviderEntry, VERCEL_AI_GATEWAY_MODELS, _AGGREGATOR_PROVIDERS, _AZURE_FOUNDRY_RESPONSES_PREFIXES, @@ -47,78 +45,21 @@ from hermes_cli.models_catalog_static import ( # noqa: F401 (re-exported; test _PROVIDER_MODELS, _PROVIDER_RETIRED_ALIASES, _SILENT_DEFAULT_PROVIDERS, - _xai_finalize_catalog, - group_providers, - provider_group_for_slug) -from hermes_cli.models_reasoning_caps import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.) + _xai_finalize_catalog) +from hermes_cli.models_reasoning_caps import ( _OPENROUTER_CATALOG_URL, - _read_reasoning_caps_disk, - _reasoning_caps_disk_path, - _save_reasoning_caps_disk, - _seed_reasoning_caps, - nous_catalog_url, - nous_model_reasoning_capabilities, - # Live-catalog metadata first (ported from PrimeIntellect-ai/prime-agent#1258): OpenRouter's /v1/models - # entries advertise reasoning support via supported_parameters + a reasoning object, which covers every - # routed vendor without a hand-maintained prefix list. The static prefix allowlist below repeatedly went - # stale one vendor at a time (nvidia/ missing → #75386; same class as tencent/, xiaomi/ additions before - # it) — metadata makes new vendors work without a code change. One catalog fetch per process, cached; - # unknown (catalog unreachable / unlisted model) falls back to the static list. - openrouter_model_reasoning_capabilities, - parse_openrouter_reasoning_capabilities, - refresh_reasoning_caps_async, - warm_nous_reasoning_caps_async, - warm_openrouter_reasoning_caps_async) -from hermes_cli.models_local import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.) - LMStudioLoadResult, + _seed_reasoning_caps) +from hermes_cli.models_local import ( _OLLAMA_LOCAL_MODELS_CACHE, _OLLAMA_LOCAL_MODELS_CACHE_TTL, _OLLAMA_LOCAL_PROBE_FAILURE_CACHE, _OLLAMA_LOCAL_PROBE_REACHABLE, _get_ollama_base_url, _get_ollama_native_headers, - _get_ollama_request_headers, - _lmstudio_fetch_raw_models, - _lmstudio_server_root, - _normalize_openai_base_url, _ollama_local_catalog, _ollama_probe_cache_key, _root_for_ollama_native_api, - _same_ollama_native_root, - _strip_ollama_cloud_suffix, - ensure_lmstudio_model_loaded, - fetch_lmstudio_models, - fetch_ollama_cloud_models, - fetch_ollama_local_models, - lmstudio_model_reasoning_options, - ollama_model_supports_thinking, - probe_lmstudio_models, - probe_ollama_local_models, - should_use_ollama_native_catalog) -from hermes_cli.models_pricing import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.) - _FAILED_CATALOG_TTL_SECONDS, - _NOUS_CATALOG_TTL_SECONDS, - _NOUS_POLICY_APPEND_MAX, - _cache_catalog, - _cached_catalog, - _fetch_deepinfra_pricing, - _fetch_novita_pricing, - _format_price_per_mtok, - _pricing_auth_fingerprint, - _pricing_cache, - _pricing_cache_retry_after, - _resolve_nous_pricing_credentials, - compute_sale_discount, - fetch_ai_gateway_pricing, - fetch_models_with_pricing, - _pricing_provider_cache_keys, - get_cached_nous_inference_base_url, - get_pricing_for_provider, - nous_policy_allowed_ids, - peek_cached_pricing, - pricing_cache_scope, - restrict_to_nous_policy) -from hermes_cli.models_validate import validate_requested_model # noqa: F401 (re-exported) + fetch_ollama_cloud_models) logger = logging.getLogger(__name__) diff --git a/hermes_cli/models_catalog_static.py b/hermes_cli/models_catalog_static.py index 83d52ef965..c986e60db5 100644 --- a/hermes_cli/models_catalog_static.py +++ b/hermes_cli/models_catalog_static.py @@ -1,8 +1,7 @@ """Static provider/model catalog tables (data only — no network). Curated per-provider model lists, the canonical provider registry, display groups and alias -maps. Split out of ``hermes_cli.models``, which re-imports every name so -``hermes_cli.models.`` keeps resolving (and monkeypatching) as before. +maps. Split out of ``hermes_cli.models``. """ from __future__ import annotations diff --git a/hermes_cli/models_local.py b/hermes_cli/models_local.py index 858df73b18..2ffc968e20 100644 --- a/hermes_cli/models_local.py +++ b/hermes_cli/models_local.py @@ -2,9 +2,8 @@ Ollama (native ``/api/tags`` probe, request headers, base-url resolution), LM Studio (``/api/v1/models``, load-on-demand) and Ollama Cloud (live + models.dev merged catalog with a -disk cache). Split out of ``hermes_cli.models``, which re-imports every name; origin helpers are -looked up on ``hermes_cli.models`` at call time so ``patch("hermes_cli.models.")`` mocks -keep intercepting. +disk cache). Split out of ``hermes_cli.models``; helpers still defined there are looked up on +``hermes_cli.models`` at call time so ``patch("hermes_cli.models.")`` mocks keep intercepting. """ from __future__ import annotations @@ -96,7 +95,7 @@ def _get_ollama_base_url() -> str: the active provider is ollama, or custom AND the endpoint actually serves ``/api/tags`` (otherwise the picker would probe an unrelated endpoint and hide the local catalog) → ``OLLAMA_HOST`` → Ollama's local default.""" - from hermes_cli.models import _get_model_config_dict, should_use_ollama_native_catalog + from hermes_cli.models import _get_model_config_dict configured = _configured_ollama_base_url() if configured: return configured @@ -152,7 +151,6 @@ def _get_ollama_native_headers(base_url: Optional[str], *, api_key: Optional[str """Ollama credentials and headers for one endpoint origin. Configured headers apply only when *base_url* shares the configured Ollama root; an explicit *api_key* replaces any configured Authorization variant rather than inheriting it.""" - from hermes_cli.models import _get_ollama_request_headers configured_base = _configured_ollama_base_url() explicit_key = str(api_key or "").strip() configured_matches = bool(configured_base and base_url and _same_ollama_native_root(base_url, configured_base)) @@ -260,7 +258,6 @@ def fetch_ollama_local_models( headers: Optional[dict[str, str]] = None, ) -> Optional[list[str]]: """Fetch local Ollama-compatible models, preserving probe failure as ``None``.""" - from hermes_cli.models import probe_ollama_local_models return probe_ollama_local_models(base_url, timeout, headers=headers) @@ -295,7 +292,6 @@ def should_use_ollama_native_catalog( Ollama's default port actually serves ``/api/tags``. (Bare ``ollama`` is normalized to ``custom`` elsewhere so runtime paths share the OpenAI client, but ``/api/tags`` is the authoritative local list; other custom endpoints keep the ``/models`` probe.)""" - from hermes_cli.models import probe_ollama_local_models requested = str(provider or "").strip().lower() root = _root_for_ollama_native_api(base_url or "") if root: @@ -342,7 +338,7 @@ def _ollama_local_catalog(force_refresh: bool) -> list[str]: """Catalog for the raw ``ollama`` provider: native ``/api/tags`` when the endpoint is a real Ollama server, else the OpenAI-style ``/v1/models`` of the configured gateway (incl. Ollama Cloud).""" - from hermes_cli.models import _get_ollama_base_url, _get_ollama_native_headers, _get_provider_config_dict, fetch_api_models, fetch_ollama_local_models, should_use_ollama_native_catalog + from hermes_cli.models import _get_provider_config_dict, fetch_api_models if force_refresh: _OLLAMA_LOCAL_MODELS_CACHE.clear() _OLLAMA_LOCAL_PROBE_FAILURE_CACHE.clear() @@ -413,7 +409,6 @@ def _lmstudio_fetch_raw_models( def _lmstudio_raw_models_or_none(api_key, base_url, timeout) -> Optional[list[dict]]: """``_lmstudio_fetch_raw_models`` with every failure (incl. AuthError) collapsed to None.""" - from hermes_cli.models import _lmstudio_fetch_raw_models try: return _lmstudio_fetch_raw_models(api_key=api_key, base_url=base_url, timeout=timeout) except Exception: @@ -435,7 +430,6 @@ def probe_lmstudio_models( """Chat-capable LM Studio model keys — a valid empty list when the server is reachable but has no non-embedding models; ``None`` on network errors, malformed responses, or bad base URLs. Raises ``AuthError`` on HTTP 401/403 so token issues surface separately from reachability.""" - from hermes_cli.models import _lmstudio_fetch_raw_models raw_models = _lmstudio_fetch_raw_models(api_key=api_key, base_url=base_url, timeout=timeout) if raw_models is None: return None @@ -457,7 +451,6 @@ def fetch_lmstudio_models( ) -> list[str]: """LM Studio chat-capable model keys; ``[]`` when unreachable/malformed. Raises ``AuthError`` on HTTP 401/403 so callers can tell a wrong ``LM_API_KEY`` from an unreachable server.""" - from hermes_cli.models import probe_lmstudio_models return probe_lmstudio_models(api_key=api_key, base_url=base_url, timeout=timeout) or [] diff --git a/hermes_cli/models_pricing.py b/hermes_cli/models_pricing.py index 0a8d17e51a..a87652fea9 100644 --- a/hermes_cli/models_pricing.py +++ b/hermes_cli/models_pricing.py @@ -2,9 +2,9 @@ OpenRouter-compatible ``/v1/models`` pricing fetch with a per-endpoint/per-credential cache, Nous Portal sale chrome and org-policy filtering, and the Vercel AI Gateway / Novita / Fireworks / -DeepInfra pricing adapters. Split out of ``hermes_cli.models``, which re-imports every name; -origin helpers and the cache dicts are looked up on ``hermes_cli.models`` at call time so -``patch("hermes_cli.models.")`` mocks keep intercepting. +DeepInfra pricing adapters. Split out of ``hermes_cli.models``; helpers still defined there are +looked up on ``hermes_cli.models`` at call time so ``patch("hermes_cli.models.")`` mocks keep +intercepting. """ from __future__ import annotations @@ -33,7 +33,6 @@ _pricing_cache_retry_after: dict[str, float] = {} def _cached_catalog(cache_key: str) -> Optional[dict[str, dict[str, Any]]]: """The cached catalog for *cache_key*, or None to go fetch it.""" - from hermes_cli.models import _pricing_cache, _pricing_cache_retry_after cached = _pricing_cache.get(cache_key) if cached is None: return None @@ -53,7 +52,6 @@ def _cache_catalog( """Cache a catalog result, giving an empty one an expiry. *ttl_seconds* expires a non-empty result too — only for a catalog whose contents depend on server-side state the client cannot observe (an org's model policy can change while a long-lived process holds the entry).""" - from hermes_cli.models import _pricing_cache, _pricing_cache_retry_after _pricing_cache[cache_key] = result if not result: _pricing_cache_retry_after[cache_key] = time.monotonic() + _FAILED_CATALOG_TTL_SECONDS @@ -84,7 +82,6 @@ def peek_cached_pricing(base_url: str) -> dict[str, dict[str, Any]]: """Pricing already cached for *base_url* (with or without ``/v1``), or ``{}``; never fetches. Prefers an authenticated catalog, scanning newest first (callers hold no credential) and skipping expired entries so a rotated credential does not answer from its predecessor's.""" - from hermes_cli.models import _pricing_cache root = _strip_v1((base_url or "").rstrip("/")) authed_prefix = root + _PRICING_AUTH_KEY_PREFIX for key in reversed(list(_pricing_cache)): @@ -317,7 +314,6 @@ _NOUS_CATALOG_TTL_SECONDS = 300.0 def _fetch_nous_pricing(api_key: str, base_url: str, *, force_refresh: bool) -> dict[str, dict[str, Any]]: """Shared by pricing and policy lookups so both read one cache entry.""" - from hermes_cli.models import fetch_models_with_pricing return fetch_models_with_pricing( api_key=api_key, base_url=base_url, @@ -332,7 +328,6 @@ def nous_policy_allowed_ids(*, force_refresh: bool = False) -> Optional[set[str] which omits policy-blocked rows), or ``None`` to not filter: no policy (or a token too old to say), an anonymous read (unfiltered catalog), or an empty read (a fetch failure, not an org that may reach nothing).""" - from hermes_cli.models import _resolve_nous_pricing_credentials try: from hermes_cli.nous_account import nous_policy_present @@ -373,12 +368,11 @@ def restrict_to_nous_policy( def _remember_provider_cache_key(provider: str, cache_key: str) -> None: - from hermes_cli.models import _pricing_profile_key, _pricing_provider_cache_keys + from hermes_cli.models import _pricing_profile_key _pricing_provider_cache_keys[(_pricing_profile_key(), provider)] = cache_key def _fetch_openrouter_pricing(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]: - from hermes_cli.models import fetch_models_with_pricing _remember_provider_cache_key("openrouter", _OPENROUTER_PRICING_BASE) return fetch_models_with_pricing( api_key=_resolve_openrouter_api_key(), @@ -388,13 +382,11 @@ def _fetch_openrouter_pricing(*, force_refresh: bool = False) -> dict[str, dict[ def _fetch_ai_gateway_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]: - from hermes_cli.models import fetch_ai_gateway_pricing _remember_provider_cache_key("ai-gateway", _ai_gateway_pricing_scope()) return fetch_ai_gateway_pricing(force_refresh=force_refresh) def _fetch_novita_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]: - from hermes_cli.models import _fetch_novita_pricing _remember_provider_cache_key("novita", _novita_pricing_scope()) return _fetch_novita_pricing(force_refresh=force_refresh) @@ -405,7 +397,6 @@ def _fetch_fireworks_pricing_for_provider(*, force_refresh: bool = False) -> dic def _fetch_nous_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]: - from hermes_cli.models import _resolve_nous_pricing_credentials api_key, base_url = _resolve_nous_pricing_credentials() if not base_url: return {} @@ -453,9 +444,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba """The current endpoint identity a provider's pricing cache is keyed on. Resolves local configuration only, never fetches: picker prewarm single-flight uses it so an endpoint rotation can start a new worker while the previous endpoint is still slow or unreachable.""" - from hermes_cli.models import ( - _deepinfra_catalog_url, _pricing_profile_key, _pricing_provider_cache_keys, normalize_provider, - ) + from hermes_cli.models import _deepinfra_catalog_url, _pricing_profile_key, normalize_provider normalized = normalize_provider(provider) static = _STATIC_PRICING_SCOPES.get(normalized) if static: @@ -473,12 +462,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba return env_base.rstrip("/").removesuffix("/v1") if normalize_provider(current_provider) == "nous" and current_base_url: return current_base_url.rstrip("/").removesuffix("/v1") - # Call-time lookup through the facade: tests (and plugins) patch - # ``hermes_cli.models.get_cached_nous_inference_base_url``; a bare module-global read here would - # silently bypass the patch and fall back to the default endpoint. - from hermes_cli.models import get_cached_nous_inference_base_url as _persisted_nous_base - - persisted_base = _persisted_nous_base() + persisted_base = get_cached_nous_inference_base_url() if persisted_base: return persisted_base return _pricing_provider_cache_keys.get((_pricing_profile_key(), normalized), _DEFAULT_NOUS_INFERENCE_BASE) @@ -487,10 +471,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba def _cached_only_pricing(normalized: str) -> dict[str, dict[str, str]]: """Process-resident pricing for *normalized* without any provider I/O.""" - from hermes_cli.models import ( - _deepinfra_catalog_cache, _deepinfra_catalog_url, _fetch_deepinfra_pricing, _pricing_profile_key, - _pricing_provider_cache_keys, - ) + from hermes_cli.models import _deepinfra_catalog_cache, _deepinfra_catalog_url, _pricing_profile_key if normalized == "deepinfra": cache_key, _url = _deepinfra_catalog_url() return _fetch_deepinfra_pricing() if cache_key in _deepinfra_catalog_cache else {} diff --git a/hermes_cli/models_reasoning_caps.py b/hermes_cli/models_reasoning_caps.py index 7df6de8924..459be653c5 100644 --- a/hermes_cli/models_reasoning_caps.py +++ b/hermes_cli/models_reasoning_caps.py @@ -1,6 +1,6 @@ """Per-model reasoning capabilities from OpenRouter-schema ``/v1/models`` catalogs. -Split out of ``hermes_cli.models``; every public/patched name is re-imported there. OpenRouter and +Split out of ``hermes_cli.models``. OpenRouter and Nous Portal share one implementation parametrized by :class:`_CapsSource`; the per-source module globals (``_openrouter_reasoning_caps_cache``, ``_nous_caps_disk_checked``, ...) stay defined on ``hermes_cli.models`` — tests reset them there — and are read/written by attribute name. @@ -81,7 +81,7 @@ def _read_reasoning_caps_disk() -> dict[str, Any]: def _load_reasoning_caps_disk(url: str) -> tuple[Optional[Caps], float]: """Return ``(caps, age_seconds)`` for *url*, or ``(None, 0.0)``.""" - entry = _origin()._read_reasoning_caps_disk().get(url) + entry = _read_reasoning_caps_disk().get(url) caps = entry.get("caps") if isinstance(entry, dict) else None if not isinstance(caps, dict) or not caps: return None, 0.0 @@ -96,7 +96,7 @@ def _save_reasoning_caps_disk(url: str, caps: Caps) -> None: """Merge *url*'s catalog into the shared disk mirror, atomically.""" from hermes_cli.models import _write_json_cache try: - data = _origin()._read_reasoning_caps_disk() + data = _read_reasoning_caps_disk() data[url] = {"ts": time.time(), "caps": caps} _write_json_cache(_reasoning_caps_disk_path(), data, indent=0, separators=(",", ":")) except Exception as exc: @@ -250,16 +250,23 @@ _OPENROUTER_CAPS = _CapsSource( _NOUS_CAPS = _CapsSource( "_nous_reasoning_caps_cache", "_nous_reasoning_caps_failed_at", "_nous_caps_disk_checked", "_nous_caps_warm_started", - lambda: _origin().nous_catalog_url(), + lambda: nous_catalog_url(), ) def nous_catalog_url() -> str: """The Portal ``/v1/models`` URL for the endpoint we actually talk to (``NOUS_INFERENCE_BASE_URL`` → resolved credential base → prod), so a staging profile reads staging's capabilities.""" - return f"{_origin()._resolve_nous_pricing_credentials()[1]}/v1/models" + from hermes_cli.models_pricing import _resolve_nous_pricing_credentials + return f"{_resolve_nous_pricing_credentials()[1]}/v1/models" +# Live-catalog metadata first (ported from PrimeIntellect-ai/prime-agent#1258): OpenRouter's /v1/models +# entries advertise reasoning support via supported_parameters + a reasoning object, which covers every +# routed vendor without a hand-maintained prefix list. The static prefix allowlist repeatedly went +# stale one vendor at a time (nvidia/ missing → #75386; same class as tencent/, xiaomi/ additions before +# it) — metadata makes new vendors work without a code change. One catalog fetch per process, cached; +# unknown (catalog unreachable / unlisted model) falls back to the static list. def openrouter_model_reasoning_capabilities( model_id: Optional[str], *, timeout: float = 6.0, allow_fetch: bool = False, ) -> Optional[dict[str, Any]]: diff --git a/hermes_cli/models_validate.py b/hermes_cli/models_validate.py index 31466cf59e..2adc43c0b2 100644 --- a/hermes_cli/models_validate.py +++ b/hermes_cli/models_validate.py @@ -1,8 +1,8 @@ """Validate a requested ``/model`` value against the active provider's catalog. -Split out of ``hermes_cli.models``; :func:`validate_requested_model` is re-imported there. Every -catalog fetcher this module calls is looked up on ``hermes_cli.models`` at call time (``_m.``) -so existing ``patch("hermes_cli.models.")`` mocks keep intercepting. +Split out of ``hermes_cli.models``. Catalog fetchers defined in ``hermes_cli.models`` are looked up +there at call time (``_m.``) so ``patch("hermes_cli.models.")`` mocks keep intercepting; +local-server probes are looked up on ``hermes_cli.models_local`` (``_ml.``) the same way. Every provider branch returns a verdict dict (see :func:`_verdict`) or ``None`` for "not decided here — keep walking the ladder". The ladder ORDER is behavior (see ``_LADDER``). @@ -170,13 +170,13 @@ def _parse_openrouter_preset(req: _Request) -> Optional[dict[str, Any]]: def _validate_lmstudio(req: _Request) -> dict[str, Any]: - from hermes_cli import models as _m + from hermes_cli import models_local as _ml from hermes_cli.auth import AuthError # probe_lmstudio_models distinguishes None (unreachable / malformed) from [] (reachable, # nothing chat-capable loaded); fetch_lmstudio_models collapses both to []. try: - models = _m.probe_lmstudio_models(api_key=req.api_key, base_url=req.base_url) + models = _ml.probe_lmstudio_models(api_key=req.api_key, base_url=req.base_url) except AuthError as exc: return _reject(f"{exc} Set `LM_API_KEY` (or update it) to match the server's bearer token.") if models is None: @@ -194,10 +194,11 @@ def _ollama_probe_headers(req: _Request) -> dict[str, str]: when the probed endpoint is the configured one (never leak them to a different host). Caller headers win; a caller ``api_key`` becomes the Authorization header unless the caller sent one.""" from hermes_cli import models as _m + from hermes_cli import models_local as _ml from hermes_cli.models_local import _configured_ollama_base_url, _drop_authorization configured_base = _configured_ollama_base_url() - configured_allowed = not configured_base or _m._same_ollama_native_root(req.base_url or "", configured_base) + configured_allowed = not configured_base or _ml._same_ollama_native_root(req.base_url or "", configured_base) configured = _m._get_ollama_native_headers(req.base_url, api_key=req.api_key) if configured_allowed else {} if req.headers is None: return configured @@ -215,17 +216,18 @@ def _validate_ollama_native(req: _Request) -> Optional[dict[str, Any]]: looks like a local Ollama server. Also resolves ``base_url`` for the raw ``ollama`` provider, which later branches (custom) rely on.""" from hermes_cli import models as _m + from hermes_cli import models_local as _ml if str(req.provider or "").strip().lower() == "ollama" and not req.base_url: req.base_url = _m._get_ollama_base_url() headers = _ollama_probe_headers(req) - if not _m.should_use_ollama_native_catalog(req.provider, req.base_url, headers=headers): + if not _ml.should_use_ollama_native_catalog(req.provider, req.base_url, headers=headers): return None - models = _m.probe_ollama_local_models(req.base_url, headers=headers) + models = _ml.probe_ollama_local_models(req.base_url, headers=headers) if models is None: # A failed native probe is not authoritative; fall back to the OpenAI-compatible catalog. models = _m.probe_api_models( - req.api_key, _m._normalize_openai_base_url(req.base_url), request_headers=headers, + req.api_key, _ml._normalize_openai_base_url(req.base_url), request_headers=headers, ).get("models") if models is None: return _soft_accept( diff --git a/hermes_cli/web_routers/models.py b/hermes_cli/web_routers/models.py index 1fa483eda9..c116e17619 100644 --- a/hermes_cli/web_routers/models.py +++ b/hermes_cli/web_routers/models.py @@ -131,10 +131,11 @@ async def get_model_options( def _nous_recommended_default() -> dict: from hermes_cli import models as m + from hermes_cli import models_pricing as mp from hermes_cli.auth import get_provider_auth_state model_ids = m.get_curated_nous_model_ids() - pricing = m.get_pricing_for_provider("nous") or {} + pricing = mp.get_pricing_for_provider("nous") or {} free_tier = m.check_nous_free_tier(force_fresh=True) try: @@ -145,10 +146,10 @@ def _nous_recommended_default() -> dict: # This endpoint picks the model a user lands on without choosing it, so an unreachable # one is worse than in a picker. Narrow to policy BEFORE the tier split, so a rescued # id still has to pass the free/paid predicate. - policy_allowed = m.nous_policy_allowed_ids() + policy_allowed = mp.nous_policy_allowed_ids() union = m.union_with_portal_free_recommendations if free_tier else m.union_with_portal_paid_recommendations model_ids, pricing = union(model_ids, pricing, portal_url) - model_ids = m.restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True) + model_ids = mp.restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True) if free_tier: model_ids, _unavailable = m.partition_nous_models_by_tier(model_ids, pricing, free_tier=True) diff --git a/plugins/model-providers/nous/__init__.py b/plugins/model-providers/nous/__init__.py index a6413d1f0b..1c28cef37f 100644 --- a/plugins/model-providers/nous/__init__.py +++ b/plugins/model-providers/nous/__init__.py @@ -39,7 +39,7 @@ class NousProfile(ProviderProfile): """True when ``reasoning: {enabled: false}`` would 400 on *model*. Cache-only catalog lookup; unknown/cold (warmer kicked) and no-reasoning routes both answer True (omit > 400).""" try: - from hermes_cli.models import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async + from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async caps = nous_model_reasoning_capabilities(model) if caps is None: diff --git a/plugins/model-providers/openrouter/__init__.py b/plugins/model-providers/openrouter/__init__.py index 51bf812fc9..7040eb30fd 100644 --- a/plugins/model-providers/openrouter/__init__.py +++ b/plugins/model-providers/openrouter/__init__.py @@ -64,7 +64,8 @@ class OpenRouterProfile(ProviderProfile): if not effort and not disabled: return cfg try: - from hermes_cli.models import clamp_reasoning_effort_to_supported, openrouter_model_reasoning_capabilities + from hermes_cli.models import clamp_reasoning_effort_to_supported + from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities caps = openrouter_model_reasoning_capabilities(model) if not caps or not caps.get("supports_reasoning"): diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py index 4c7efd5a9f..9d4737371e 100644 --- a/plugins/platforms/telegram/adapter.py +++ b/plugins/platforms/telegram/adapter.py @@ -3958,7 +3958,7 @@ class TelegramAdapter(BasePlatformAdapter): """Paginated top-level provider keyboard folding provider families (Kimi/Moonshot, MiniMax, xAI…) into one ``mpg:`` button via the shared ``group_providers`` fold; singles are ``mp:``.""" try: - from hermes_cli.models import group_providers + from hermes_cli.models_catalog_static import group_providers except Exception: group_providers = None by_slug = {p.get("slug"): p for p in providers} @@ -4117,7 +4117,7 @@ class TelegramAdapter(BasePlatformAdapter): elif data.startswith("mpg:"): # provider group selected: show member providers group_id = data[4:] try: - from hermes_cli.models import PROVIDER_GROUPS + from hermes_cli.models_catalog_static import PROVIDER_GROUPS _label, _desc, member_slugs = PROVIDER_GROUPS.get(group_id, ("", "", [])) except Exception: _label, member_slugs = "", [] diff --git a/run_agent.py b/run_agent.py index 63a76014cd..1cc624a452 100644 --- a/run_agent.py +++ b/run_agent.py @@ -447,7 +447,7 @@ class AIAgent( if (getattr(self, "lmstudio_load_mode", "explicit") or "explicit").strip().lower() == "jit": logger.debug("LM Studio explicit preload skipped: lmstudio_load_mode=jit") return None - from hermes_cli.models import ensure_lmstudio_model_loaded + from hermes_cli.models_local import ensure_lmstudio_model_loaded if config_context_length is None: config_context_length = getattr(self, "_config_context_length", None) diff --git a/tests/acp/test_named_provider_catalogs.py b/tests/acp/test_named_provider_catalogs.py index eb37a2d0ec..811b8214cd 100644 --- a/tests/acp/test_named_provider_catalogs.py +++ b/tests/acp/test_named_provider_catalogs.py @@ -120,7 +120,7 @@ class TestNamedCustomProviderCatalogs: } ) with patch("hermes_cli.config.load_config", return_value=cfg), patch( - "hermes_cli.models.should_use_ollama_native_catalog", + "hermes_cli.models_local.should_use_ollama_native_catalog", return_value=True, ), patch( "hermes_cli.model_switch_providers._fetch_picker_live_models", @@ -148,7 +148,7 @@ class TestNamedCustomProviderCatalogs: ] ) with patch("hermes_cli.config.load_config", return_value=cfg), patch( - "hermes_cli.models.should_use_ollama_native_catalog", + "hermes_cli.models_local.should_use_ollama_native_catalog", return_value=True, ), patch( "hermes_cli.model_switch_providers._fetch_picker_live_models", @@ -172,7 +172,7 @@ class TestNamedCustomProviderCatalogs: from hermes_cli.model_switch_providers import _NativePickerModelList with patch("hermes_cli.config.load_config", return_value=cfg), patch( - "hermes_cli.models.should_use_ollama_native_catalog", + "hermes_cli.models_local.should_use_ollama_native_catalog", return_value=True, ), patch( "hermes_cli.model_switch_providers._fetch_picker_live_models", diff --git a/tests/agent/test_auxiliary_client.py b/tests/agent/test_auxiliary_client.py index 7bcb5f3fba..18c7c531b1 100644 --- a/tests/agent/test_auxiliary_client.py +++ b/tests/agent/test_auxiliary_client.py @@ -4772,7 +4772,7 @@ class TestFastModelTier: "~openai/gpt-mini-latest": {}, "stepfun/step-3.7-flash:free": {}, } - with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog): + with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog): assert ac._fast_model_from_catalog("nous") == "~openai/gpt-mini-latest" def test_catalog_match_skips_reasoning_batch_and_embedding_lookalikes(self): @@ -4785,7 +4785,7 @@ class TestFastModelTier: "sentence-transformers/all-minilm-l6-v2": {}, "google/gemini-3.6-flash": {}, } - with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog): + with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog): assert ac._fast_model_from_catalog("nous") == "google/gemini-3.6-flash" def test_catalog_match_skips_the_non_chat_siblings_of_a_chat_model(self): @@ -4799,7 +4799,7 @@ class TestFastModelTier: "openai/gpt-4o-mini-search-preview": {}, "openai/gpt-4o-mini": {}, } - with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog): + with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog): assert ac._fast_model_from_catalog("nous") == "openai/gpt-4o-mini" def test_catalog_match_takes_the_newest_of_a_family(self): @@ -4816,7 +4816,7 @@ class TestFastModelTier: "openai/gpt-9-mini": {}, "openai/gpt-10-mini": {}, } - with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog): + with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog): assert ac._fast_model_from_catalog("nous") == "openai/gpt-10-mini" def test_catalog_fetch_is_authenticated(self): @@ -4831,7 +4831,7 @@ class TestFastModelTier: "hermes_cli.auth.resolve_api_key_provider_credentials", return_value={"api_key": "sk-test", "base_url": "https://api.example.com/v1"}, ), patch( - "hermes_cli.models.fetch_models_with_pricing", return_value={} + "hermes_cli.models_pricing.fetch_models_with_pricing", return_value={} ) as fetch: ac._fast_model_from_catalog("openai") diff --git a/tests/agent/test_credits_policy.py b/tests/agent/test_credits_policy.py index b03e9c28bb..231d900ec1 100644 --- a/tests/agent/test_credits_policy.py +++ b/tests/agent/test_credits_policy.py @@ -317,12 +317,11 @@ class TestIsFreeTierModel: def test_pricing_cache_peek_zero_priced_model(self, monkeypatch): from agent.credits_tracker import is_free_tier_model import hermes_cli.models as models_mod + from hermes_cli import models_pricing # The picker keys the cache on the pre-/v1 root (get_pricing_for_provider # strips a trailing /v1 before fetch_models_with_pricing). - monkeypatch.setattr( - models_mod, - "_pricing_cache", + monkeypatch.setattr(models_pricing, "_pricing_cache", { "https://inference-api.nousresearch.com": { "some/zero-priced": {"prompt": "0", "completion": "0"}, @@ -343,12 +342,13 @@ class TestIsFreeTierModel: def test_exception_fails_open_to_false(self, monkeypatch): from agent.credits_tracker import is_free_tier_model import hermes_cli.models as models_mod + from hermes_cli import models_pricing class _Exploding: def get(self, *_a, **_kw): raise RuntimeError("boom") - monkeypatch.setattr(models_mod, "_pricing_cache", _Exploding()) + monkeypatch.setattr(models_pricing, "_pricing_cache", _Exploding()) assert is_free_tier_model("some/model", "https://inference-api.nousresearch.com") is False def test_stealth_prefix_detected_as_free(self): diff --git a/tests/cli/test_cli_provider_resolution.py b/tests/cli/test_cli_provider_resolution.py index 959aed8b08..46336f95cc 100644 --- a/tests/cli/test_cli_provider_resolution.py +++ b/tests/cli/test_cli_provider_resolution.py @@ -8,6 +8,7 @@ import pytest from hermes_cli.auth import AuthError from hermes_cli import main as hermes_main +from hermes_cli import model_setup_flows # --------------------------------------------------------------------------- @@ -329,7 +330,7 @@ def test_model_flow_nous_does_not_restore_stale_custom_api_key(tmp_path, monkeyp "hermes_cli.models.get_curated_nous_model_ids", lambda: [selected_model], ) - monkeypatch.setattr("hermes_cli.models.get_pricing_for_provider", lambda provider: {}) + monkeypatch.setattr("hermes_cli.models_pricing.get_pricing_for_provider", lambda provider: {}) monkeypatch.setattr("hermes_cli.models.check_nous_free_tier", lambda **kwargs: False) monkeypatch.setattr( "hermes_cli.models.union_with_portal_paid_recommendations", diff --git a/tests/hermes_cli/test_ai_gateway_models.py b/tests/hermes_cli/test_ai_gateway_models.py index ba608fd08e..b08c98d192 100644 --- a/tests/hermes_cli/test_ai_gateway_models.py +++ b/tests/hermes_cli/test_ai_gateway_models.py @@ -9,12 +9,9 @@ import json from unittest.mock import patch, MagicMock from hermes_cli import models as models_module -from hermes_cli.models import ( - VERCEL_AI_GATEWAY_MODELS, - _ai_gateway_model_is_free, - fetch_ai_gateway_models, - fetch_ai_gateway_pricing, -) +from hermes_cli import models_pricing +from hermes_cli.models import VERCEL_AI_GATEWAY_MODELS, _ai_gateway_model_is_free, fetch_ai_gateway_models +from hermes_cli.models_pricing import fetch_ai_gateway_pricing def _mock_urlopen(payload): @@ -29,7 +26,7 @@ def _mock_urlopen(payload): def _reset_caches(): models_module._ai_gateway_catalog_cache = None - models_module._pricing_cache.clear() + models_pricing._pricing_cache.clear() def test_ai_gateway_pricing_translates_input_output_to_prompt_completion(): diff --git a/tests/hermes_cli/test_api_key_providers.py b/tests/hermes_cli/test_api_key_providers.py index 18fd8dd1fb..8b59e337f3 100644 --- a/tests/hermes_cli/test_api_key_providers.py +++ b/tests/hermes_cli/test_api_key_providers.py @@ -902,9 +902,10 @@ class TestNovitaProvider: def test_novita_pricing_cache(self, monkeypatch): """_fetch_novita_pricing should cache results in _pricing_cache.""" from hermes_cli import models as models_mod + from hermes_cli import models_pricing monkeypatch.setenv("NOVITA_API_KEY", "sk-test-key") monkeypatch.setenv("NOVITA_BASE_URL", "https://api.novita.ai/openai/v1") - models_mod._pricing_cache.pop("https://api.novita.ai/openai/v1", None) + models_pricing._pricing_cache.pop("https://api.novita.ai/openai/v1", None) call_count = {"n": 0} fake_payload = { @@ -937,17 +938,17 @@ class TestNovitaProvider: ) # First call hits the network. - first = models_mod._fetch_novita_pricing() + first = models_pricing._fetch_novita_pricing() assert "x/y" in first assert call_count["n"] == 1 # Second call returns cached result without re-hitting the network. - second = models_mod._fetch_novita_pricing() + second = models_pricing._fetch_novita_pricing() assert second == first assert call_count["n"] == 1 # force_refresh bypasses the cache. - models_mod._fetch_novita_pricing(force_refresh=True) + models_pricing._fetch_novita_pricing(force_refresh=True) assert call_count["n"] == 2 @@ -1004,6 +1005,7 @@ def _deepinfra_cache_isolation(monkeypatch): a later test's fetch within the failure TTL. """ import hermes_cli.models as _models_mod + from hermes_cli import models_pricing monkeypatch.setattr(_models_mod, "_deepinfra_catalog_cache", {}) monkeypatch.setattr(_models_mod, "_deepinfra_catalog_neg_cache", {}) yield @@ -1030,6 +1032,7 @@ class TestFetchDeepInfraModels: ]}).encode() import hermes_cli.models as models + from hermes_cli import models_pricing monkeypatch.setattr( models, "_urlopen_model_catalog_request", lambda *a, **kw: _Resp() ) @@ -1047,6 +1050,7 @@ class TestFetchDeepInfraModels: def test_catalog_uses_credential_safe_opener(self, monkeypatch): import hermes_cli.models as models + from hermes_cli import models_pricing seen = {} @@ -1119,6 +1123,7 @@ class TestDeepInfraTagFiltering: ]} from hermes_cli.models import _fetch_deepinfra_models_by_tag import hermes_cli.models as _m + from hermes_cli import models_pricing for surface in ("chat", "image-gen", "tts", "stt", "embed"): monkeypatch.setattr( @@ -1171,12 +1176,13 @@ class TestDeepInfraPricingFetcher: {"id": "vendor/model-image", "metadata": {"tags": ["image-gen"], "pricing": {"per_image_unit": 0.05}}}, ]} import hermes_cli.models as models + from hermes_cli import models_pricing monkeypatch.setattr( models, "_urlopen_model_catalog_request", _make_urlopen_returning(payload), ) - from hermes_cli.models import get_pricing_for_provider + from hermes_cli.models_pricing import get_pricing_for_provider # get_pricing_for_provider → _fetch_deepinfra_pricing dispatch path result = get_pricing_for_provider("deepinfra") diff --git a/tests/hermes_cli/test_auth_nous_provider.py b/tests/hermes_cli/test_auth_nous_provider.py index a1772fd326..b093e049b0 100644 --- a/tests/hermes_cli/test_auth_nous_provider.py +++ b/tests/hermes_cli/test_auth_nous_provider.py @@ -500,6 +500,7 @@ class TestLoginNousSkipKeepsCurrent: """Patch OAuth + model-list + prompt so _login_nous doesn't hit network.""" import hermes_cli.auth as auth_mod import hermes_cli.models as models_mod + from hermes_cli import models_pricing import hermes_cli.nous_subscription as ns fake_auth_state = { @@ -518,7 +519,7 @@ class TestLoginNousSkipKeepsCurrent: auth_mod, "_prompt_model_selection", lambda *a, **kw: prompt_returns, ) - monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda p: {}) + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda p: {}) free_tier_calls = [] def _check_nous_free_tier(**kwargs): diff --git a/tests/hermes_cli/test_custom_provider_model_switch.py b/tests/hermes_cli/test_custom_provider_model_switch.py index f7d0099543..acbfe0b89a 100644 --- a/tests/hermes_cli/test_custom_provider_model_switch.py +++ b/tests/hermes_cli/test_custom_provider_model_switch.py @@ -459,7 +459,7 @@ class TestCustomProviderDiscoverModels: } with patch("hermes_cli.models.fetch_api_models") as mock_fetch, \ - patch("hermes_cli.models.fetch_ollama_local_models") as mock_ollama, \ + patch("hermes_cli.models_local.fetch_ollama_local_models") as mock_ollama, \ patch("hermes_cli.curses_ui.curses_radiolist", side_effect=ImportError), \ patch("builtins.input", return_value="1"), \ patch("builtins.print"): diff --git a/tests/hermes_cli/test_inventory_pricing.py b/tests/hermes_cli/test_inventory_pricing.py index d626593f34..6168632547 100644 --- a/tests/hermes_cli/test_inventory_pricing.py +++ b/tests/hermes_cli/test_inventory_pricing.py @@ -10,10 +10,11 @@ from time import monotonic import hermes_cli.inventory as inv import hermes_cli.models as models_mod +from hermes_cli import models_pricing def _patch_pricing(monkeypatch, *, free_tier, pricing, unavailable=None): - monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda slug, **kw: pricing.get(slug, {})) + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda slug, **kw: pricing.get(slug, {})) monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda *, force_fresh=False: free_tier) monkeypatch.setattr( models_mod, "partition_nous_models_by_tier", @@ -125,7 +126,7 @@ def test_model_options_cold_pricing_fetch_runs_off_the_request_path(monkeypatch) "is_user_defined": False, "source": "built-in", } - monkeypatch.setattr(models_mod, "get_pricing_for_provider", fake_pricing) + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", fake_pricing) monkeypatch.setattr( "hermes_cli.model_switch.list_authenticated_providers", lambda **_kwargs: [row], @@ -160,8 +161,7 @@ def test_model_options_cold_pricing_fetch_runs_off_the_request_path(monkeypatch) def test_cold_nous_entitlement_keeps_models_unselectable(monkeypatch): """A cold nonblocking response must not expose paid models fail-open.""" - monkeypatch.setattr( - models_mod, "get_pricing_for_provider", lambda *_args, **_kwargs: {} + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda *_args, **_kwargs: {} ) monkeypatch.setattr(models_mod, "get_cached_nous_free_tier", lambda: None) rows = [{"slug": "nous", "models": ["free/model", "paid/model"]}] @@ -288,12 +288,10 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch): endpoint_b: {"b/model": {"prompt": "3", "completion": "4"}}, } monkeypatch.setattr(inv, "_pricing_prewarm_threads", {}) - monkeypatch.setattr(models_mod, "_pricing_cache", {}) - monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {}) - monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {}) - monkeypatch.setattr( - models_mod, - "_resolve_nous_pricing_credentials", + monkeypatch.setattr(models_pricing, "_pricing_cache", {}) + monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {}) + monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {}) + monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials", lambda: ("", active_endpoint["value"]), ) @@ -301,13 +299,13 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch): started[base_url].set() if base_url == endpoint_a: release_a.wait(timeout=5) - return models_mod._cache_catalog(base_url, expected[base_url]) + return models_pricing._cache_catalog(base_url, expected[base_url]) - monkeypatch.setattr(models_mod, "fetch_models_with_pricing", fetch_pricing) + monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", fetch_pricing) monkeypatch.setattr( inv, "_apply_pricing", - lambda _rows: models_mod.get_pricing_for_provider("nous"), + lambda _rows: models_pricing.get_pricing_for_provider("nous"), ) token = set_hermes_home_override(str(tmp_path / "profile")) @@ -335,7 +333,7 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch): assert started[endpoint_b].wait(timeout=1) threads[1].join(timeout=2) assert not threads[1].is_alive() - assert models_mod.get_pricing_for_provider( + assert models_pricing.get_pricing_for_provider( "nous", cached_only=True ) == expected[endpoint_b] finally: @@ -363,17 +361,13 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey endpoint_b: {"b/model": {"prompt": "3", "completion": "4"}}, } monkeypatch.setattr(inv, "_pricing_prewarm_threads", {}) - monkeypatch.setattr(models_mod, "_pricing_cache", {}) - monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {}) - monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {}) - monkeypatch.setattr( - models_mod, - "get_cached_nous_inference_base_url", + monkeypatch.setattr(models_pricing, "_pricing_cache", {}) + monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {}) + monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {}) + monkeypatch.setattr(models_pricing, "get_cached_nous_inference_base_url", lambda: active_endpoint["value"], ) - monkeypatch.setattr( - models_mod, - "_resolve_nous_pricing_credentials", + monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials", lambda: ("", active_endpoint["value"]), ) @@ -381,13 +375,13 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey started[base_url].set() if base_url == endpoint_a: release_a.wait(timeout=5) - return models_mod._cache_catalog(base_url, expected[base_url]) + return models_pricing._cache_catalog(base_url, expected[base_url]) - monkeypatch.setattr(models_mod, "fetch_models_with_pricing", fetch_pricing) + monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", fetch_pricing) monkeypatch.setattr( inv, "_apply_pricing", - lambda _rows: models_mod.get_pricing_for_provider("nous"), + lambda _rows: models_pricing.get_pricing_for_provider("nous"), ) token = set_hermes_home_override(str(tmp_path / "profile")) @@ -415,7 +409,7 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey assert started[endpoint_b].wait(timeout=1) threads[1].join(timeout=2) assert not threads[1].is_alive() - assert models_mod.get_pricing_for_provider( + assert models_pricing.get_pricing_for_provider( "nous", cached_only=True ) == expected[endpoint_b] finally: @@ -430,16 +424,14 @@ def test_cached_only_pricing_returns_a_warm_value_without_fetching(monkeypatch): """Cache-only picker reads preserve pricing once the prewarm completes.""" cache_key = "https://openrouter.ai/api" expected = {"vendor/model": {"prompt": "0.000001", "completion": "0.000002"}} - monkeypatch.setattr(models_mod, "_pricing_cache", {cache_key: expected}) - monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {}) - monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {}) - monkeypatch.setattr( - models_mod, - "fetch_models_with_pricing", + monkeypatch.setattr(models_pricing, "_pricing_cache", {cache_key: expected}) + monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {}) + monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {}) + monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", lambda **_kwargs: (_ for _ in ()).throw(AssertionError("network fetch started")), ) - assert models_mod.get_pricing_for_provider( + assert models_pricing.get_pricing_for_provider( "openrouter", cached_only=True ) == expected @@ -455,30 +447,24 @@ def test_cached_only_dynamic_pricing_is_profile_scoped(tmp_path, monkeypatch): endpoint_b = "https://profile-b.example" expected_a = {"a/model": {"prompt": "1", "completion": "2"}} expected_b = {"b/model": {"prompt": "3", "completion": "4"}} - monkeypatch.setattr( - models_mod, - "_pricing_cache", + monkeypatch.setattr(models_pricing, "_pricing_cache", {endpoint_a: expected_a, endpoint_b: expected_b}, ) - monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {}) - monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {}) + monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {}) + monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {}) active_endpoint = {"value": endpoint_a} - monkeypatch.setattr( - models_mod, - "_resolve_nous_pricing_credentials", + monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials", lambda: ("", active_endpoint["value"]), ) - monkeypatch.setattr( - models_mod, - "fetch_models_with_pricing", - lambda **kwargs: models_mod._pricing_cache[kwargs["base_url"]], + monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", + lambda **kwargs: models_pricing._pricing_cache[kwargs["base_url"]], ) def in_profile(home, endpoint, *, cached_only): token = set_hermes_home_override(str(home)) active_endpoint["value"] = endpoint try: - return models_mod.get_pricing_for_provider( + return models_pricing.get_pricing_for_provider( "nous", cached_only=cached_only ) finally: diff --git a/tests/hermes_cli/test_inventory_reasoning_caps.py b/tests/hermes_cli/test_inventory_reasoning_caps.py index 7fc40cc9c2..a44878d2a6 100644 --- a/tests/hermes_cli/test_inventory_reasoning_caps.py +++ b/tests/hermes_cli/test_inventory_reasoning_caps.py @@ -11,15 +11,16 @@ invite a picker filter that hides working levels. import hermes_cli.inventory as inv import hermes_cli.models as models_mod +from hermes_cli import models_reasoning_caps def _patch_catalog(monkeypatch, caps_by_model, *, provider="nous"): """Point the Nous/OpenRouter catalog readers at a fixed capability map.""" monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda model: False) - monkeypatch.setattr(models_mod, "warm_nous_reasoning_caps_async", lambda: None) - monkeypatch.setattr(models_mod, "warm_openrouter_reasoning_caps_async", lambda: None) + monkeypatch.setattr(models_reasoning_caps, "warm_nous_reasoning_caps_async", lambda: None) + monkeypatch.setattr(models_reasoning_caps, "warm_openrouter_reasoning_caps_async", lambda: None) monkeypatch.setattr( - models_mod, + models_reasoning_caps, f"{provider}_model_reasoning_capabilities", lambda model, **kw: caps_by_model.get(model), ) @@ -143,12 +144,12 @@ def test_openrouter_uses_its_own_catalog(monkeypatch): def test_catalog_failure_never_breaks_the_picker(monkeypatch): """A raising catalog reader degrades to "unknown", not to a broken payload.""" monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda model: False) - monkeypatch.setattr(models_mod, "warm_nous_reasoning_caps_async", lambda: None) + monkeypatch.setattr(models_reasoning_caps, "warm_nous_reasoning_caps_async", lambda: None) def _boom(model, **kw): raise RuntimeError("catalog exploded") - monkeypatch.setattr(models_mod, "nous_model_reasoning_capabilities", _boom) + monkeypatch.setattr(models_reasoning_caps, "nous_model_reasoning_capabilities", _boom) rows = [{"slug": "nous", "models": ["deepseek/deepseek-v4-pro"]}] inv._apply_capabilities(rows) diff --git a/tests/hermes_cli/test_list_picker_providers.py b/tests/hermes_cli/test_list_picker_providers.py index 0fdea335cd..ab15213cd4 100644 --- a/tests/hermes_cli/test_list_picker_providers.py +++ b/tests/hermes_cli/test_list_picker_providers.py @@ -159,6 +159,7 @@ def _stub_kimi_discovery(monkeypatch, *, canonical): """ import agent.models_dev as md import hermes_cli.models as hm + from hermes_cli import models_catalog_static kimi_map = { "kimi": "kimi-for-coding", @@ -188,10 +189,11 @@ def _stub_kimi_discovery(monkeypatch, *, canonical): def test_single_kimi_credential_yields_one_canonical_row(monkeypatch): """One Kimi key yields a single row under the canonical 'kimi-coding' slug.""" import hermes_cli.models as hm + from hermes_cli import models_catalog_static _stub_kimi_discovery( monkeypatch, - canonical=[hm.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc")], + canonical=[models_catalog_static.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc")], ) monkeypatch.setenv("KIMI_API_KEY", "sk-test-kimi") @@ -215,12 +217,13 @@ def test_distinct_kimi_china_credential_still_listed(monkeypatch): pair that share a credential, not legitimately distinct providers. """ import hermes_cli.models as hm + from hermes_cli import models_catalog_static _stub_kimi_discovery( monkeypatch, canonical=[ - hm.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc"), - hm.ProviderEntry("kimi-coding-cn", "Kimi / Moonshot (China)", "desc"), + models_catalog_static.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc"), + models_catalog_static.ProviderEntry("kimi-coding-cn", "Kimi / Moonshot (China)", "desc"), ], ) monkeypatch.setenv("KIMI_API_KEY", "sk-test-kimi") diff --git a/tests/hermes_cli/test_lmstudio_context_policy.py b/tests/hermes_cli/test_lmstudio_context_policy.py index 3510722cd4..15678fa9a8 100644 --- a/tests/hermes_cli/test_lmstudio_context_policy.py +++ b/tests/hermes_cli/test_lmstudio_context_policy.py @@ -3,6 +3,7 @@ import json import pytest from hermes_cli import models +from hermes_cli import models_local MODEL = "publisher/model" @@ -54,12 +55,11 @@ def _capture_load(monkeypatch, response_payload): def test_missing_echo_refreshes_loaded_state(monkeypatch): catalogs = iter([_catalog(), _catalog(loaded_context=88_000)]) - monkeypatch.setattr( - models, "_lmstudio_fetch_raw_models", lambda **_kwargs: next(catalogs) + monkeypatch.setattr(models_local, "_lmstudio_fetch_raw_models", lambda **_kwargs: next(catalogs) ) _capture_load(monkeypatch, {"status": "loaded"}) - result = models.ensure_lmstudio_model_loaded( + result = models_local.ensure_lmstudio_model_loaded( MODEL, BASE_URL, api_key="", target_context_length=100_000 ) @@ -67,9 +67,7 @@ def test_missing_echo_refreshes_loaded_state(monkeypatch): def test_explicit_override_above_known_maximum_rejects_even_when_loaded(monkeypatch): - monkeypatch.setattr( - models, - "_lmstudio_fetch_raw_models", + monkeypatch.setattr(models_local, "_lmstudio_fetch_raw_models", lambda **_kwargs: _catalog(loaded_context=64_000, maximum=128_000), ) monkeypatch.setattr( @@ -78,7 +76,7 @@ def test_explicit_override_above_known_maximum_rejects_even_when_loaded(monkeypa lambda *_args, **_kwargs: pytest.fail("invalid override must not be posted"), ) - result = models.ensure_lmstudio_model_loaded( + result = models_local.ensure_lmstudio_model_loaded( MODEL, BASE_URL, api_key="", diff --git a/tests/hermes_cli/test_model_alias_credentials_83612.py b/tests/hermes_cli/test_model_alias_credentials_83612.py index 8b23b93b87..e39eaec71e 100644 --- a/tests/hermes_cli/test_model_alias_credentials_83612.py +++ b/tests/hermes_cli/test_model_alias_credentials_83612.py @@ -47,7 +47,7 @@ def _switch_to_alias(monkeypatch, alias_entry): return {"accepted": True, "persist": True, "recognized": True, "message": ""} monkeypatch.setattr( - "hermes_cli.models.validate_requested_model", _fake_validate + "hermes_cli.models_validate.validate_requested_model", _fake_validate ) import hermes_cli.model_switch as ms @@ -208,7 +208,7 @@ class TestSessionKeyIsHostScoped: "hermes_cli.runtime_provider.load_config", lambda *a, **k: cfg ) monkeypatch.setattr( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", lambda *a, **k: { "accepted": True, "persist": True, @@ -278,7 +278,7 @@ class TestBuiltinProviderKeysDoNotLeak: return {"accepted": True, "persist": True, "recognized": True, "message": ""} monkeypatch.setattr( - "hermes_cli.models.validate_requested_model", _fake_validate + "hermes_cli.models_validate.validate_requested_model", _fake_validate ) import hermes_cli.model_switch as ms @@ -320,7 +320,7 @@ class TestProviderLabelCannotSelectAKeyForAnArbitraryHost: probed["api_key"] = api_key return {"accepted": True, "persist": True, "recognized": True, "message": ""} - monkeypatch.setattr("hermes_cli.models.validate_requested_model", _fake_validate) + monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", _fake_validate) import hermes_cli.model_switch as ms monkeypatch.setattr(ms, "DIRECT_ALIASES", {}) @@ -378,7 +378,7 @@ class TestSessionCredentialIsScopedToTheOrigin: monkeypatch.setattr("hermes_cli.config.load_config", lambda *a, **k: cfg) monkeypatch.setattr("hermes_cli.runtime_provider.load_config", lambda *a, **k: cfg) monkeypatch.setattr( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", lambda *a, **k: {"accepted": True, "persist": True, "recognized": True, "message": ""}, ) diff --git a/tests/hermes_cli/test_model_switch_configured_provider_routing.py b/tests/hermes_cli/test_model_switch_configured_provider_routing.py index e46a480c39..47e482f2cc 100644 --- a/tests/hermes_cli/test_model_switch_configured_provider_routing.py +++ b/tests/hermes_cli/test_model_switch_configured_provider_routing.py @@ -60,7 +60,7 @@ def _run_switch( with patch("hermes_cli.model_switch.resolve_alias", return_value=None), \ patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \ patch("hermes_cli.model_switch.normalize_model_for_provider", side_effect=lambda model, provider: model), \ - patch("hermes_cli.models.validate_requested_model", return_value=validation), \ + patch("hermes_cli.models_validate.validate_requested_model", return_value=validation), \ patch("hermes_cli.models.detect_provider_for_model", return_value=None), \ patch("hermes_cli.model_switch.get_model_info", return_value=None), \ patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \ diff --git a/tests/hermes_cli/test_model_switch_copilot_api_mode.py b/tests/hermes_cli/test_model_switch_copilot_api_mode.py index 21d3f3e20c..72ce9d6dba 100644 --- a/tests/hermes_cli/test_model_switch_copilot_api_mode.py +++ b/tests/hermes_cli/test_model_switch_copilot_api_mode.py @@ -40,7 +40,7 @@ def _run_copilot_switch( }, ), patch( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", return_value=_MOCK_VALIDATION, ), patch("hermes_cli.model_switch.get_model_info", return_value=None), diff --git a/tests/hermes_cli/test_model_switch_custom_providers.py b/tests/hermes_cli/test_model_switch_custom_providers.py index 62c1b6c374..7073ad4262 100644 --- a/tests/hermes_cli/test_model_switch_custom_providers.py +++ b/tests/hermes_cli/test_model_switch_custom_providers.py @@ -39,19 +39,19 @@ def _disable_live_custom_provider_model_probe(monkeypatch): "hermes_cli.models.provider_model_ids", lambda *_a, **_kw: [] ) monkeypatch.setattr( - "hermes_cli.models.fetch_ollama_local_models", lambda *_a, **_kw: None + "hermes_cli.models_local.fetch_ollama_local_models", lambda *_a, **_kw: None ) def test_picker_native_probe_failure_falls_back_to_openai_catalog(monkeypatch): monkeypatch.setattr( - "hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: True + "hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: True ) monkeypatch.setattr( "hermes_cli.models._get_ollama_native_headers", lambda *a, **k: {} ) monkeypatch.setattr( - "hermes_cli.models.fetch_ollama_local_models", lambda *a, **k: None + "hermes_cli.models_local.fetch_ollama_local_models", lambda *a, **k: None ) monkeypatch.setattr( "hermes_cli.models.fetch_api_models", lambda *a, **k: ["fallback-model"] @@ -70,7 +70,7 @@ def test_picker_generic_discovery_preserves_api_mode(monkeypatch): return ["model-a"] monkeypatch.setattr( - "hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: False + "hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: False ) monkeypatch.setattr("hermes_cli.models.cached_fetch_api_models", cached) @@ -172,7 +172,7 @@ def test_providers_singular_model_does_not_suppress_ollama_native_discovery(monk monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {}) monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {}) monkeypatch.setattr( - "hermes_cli.models.fetch_ollama_local_models", + "hermes_cli.models_local.fetch_ollama_local_models", lambda *a, **k: ["qwen3:latest", "llama3.2:latest"], ) @@ -362,7 +362,7 @@ def test_list_authenticated_providers_can_probe_active_bare_custom_endpoint(monk def test_switch_model_accepts_explicit_bare_custom_current_endpoint(monkeypatch): """Picker selections for bare custom endpoints should route to current base_url.""" - monkeypatch.setattr("hermes_cli.models.validate_requested_model", lambda *a, **k: _MOCK_VALIDATION) + monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", lambda *a, **k: _MOCK_VALIDATION) monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None) monkeypatch.setattr("hermes_cli.model_switch.get_model_capabilities", lambda *a, **k: None) @@ -408,11 +408,11 @@ def test_switch_model_does_not_send_ollama_headers_to_unrelated_custom_endpoint( return _MOCK_VALIDATION monkeypatch.setattr( - "hermes_cli.models.should_use_ollama_native_catalog", + "hermes_cli.models_local.should_use_ollama_native_catalog", fake_native_detection, ) monkeypatch.setattr( - "hermes_cli.models._get_ollama_request_headers", + "hermes_cli.models_local._get_ollama_request_headers", lambda: {"Authorization": "Bearer configured-ollama-secret"}, ) monkeypatch.setattr( @@ -431,7 +431,7 @@ def test_switch_model_does_not_send_ollama_headers_to_unrelated_custom_endpoint( "api_mode": "chat_completions", }, ) - monkeypatch.setattr("hermes_cli.models.validate_requested_model", fake_validation) + monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", fake_validation) monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None) monkeypatch.setattr("hermes_cli.model_switch.get_model_capabilities", lambda *a, **k: None) @@ -480,7 +480,7 @@ def test_picker_selection_resolves_named_custom_provider_model_id(monkeypatch): }, ) monkeypatch.setattr( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", lambda *a, **k: _MOCK_VALIDATION, ) monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None) @@ -980,8 +980,8 @@ def test_picker_endpoint_authorization_overrides_inferred_bearer(monkeypatch): captured.update(headers or {}) return ["model-a"] - monkeypatch.setattr("hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: True) - monkeypatch.setattr("hermes_cli.models.fetch_ollama_local_models", fake_native) + monkeypatch.setattr("hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: True) + monkeypatch.setattr("hermes_cli.models_local.fetch_ollama_local_models", fake_native) result = _fetch_picker_live_models( "endpoint-key", "http://127.0.0.1:11434/v1", @@ -1217,7 +1217,7 @@ def test_lmstudio_picker_probes_active_config_base_url(monkeypatch): captured["api_key"] = api_key return ["qwen/qwen3-coder-30b"] - monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch) + monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch) list_authenticated_providers( current_provider="lmstudio", @@ -1244,7 +1244,7 @@ def test_lmstudio_picker_lm_base_url_env_wins_over_active_config(monkeypatch): captured["base_url"] = base_url return [] - monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch) + monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch) list_authenticated_providers( current_provider="lmstudio", @@ -1270,7 +1270,7 @@ def test_lmstudio_picker_skips_probe_when_not_configured(monkeypatch): captured["base_url"] = base_url return [] - monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch) + monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch) list_authenticated_providers( current_provider="openrouter", diff --git a/tests/hermes_cli/test_model_switch_openai_api_mode.py b/tests/hermes_cli/test_model_switch_openai_api_mode.py index 8c7c1058d3..fd890a5157 100644 --- a/tests/hermes_cli/test_model_switch_openai_api_mode.py +++ b/tests/hermes_cli/test_model_switch_openai_api_mode.py @@ -49,7 +49,7 @@ def _run_openai_switch( }, ), patch( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", return_value=_MOCK_VALIDATION, ), patch("hermes_cli.model_switch.get_model_info", return_value=None), diff --git a/tests/hermes_cli/test_model_switch_opencode_anthropic.py b/tests/hermes_cli/test_model_switch_opencode_anthropic.py index faad9c4b30..44a0b73fe7 100644 --- a/tests/hermes_cli/test_model_switch_opencode_anthropic.py +++ b/tests/hermes_cli/test_model_switch_opencode_anthropic.py @@ -57,7 +57,7 @@ def _run_opencode_switch( }, ), patch( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", return_value=_MOCK_VALIDATION, ), patch("hermes_cli.model_switch.get_model_info", return_value=None), diff --git a/tests/hermes_cli/test_model_switch_variant_tags.py b/tests/hermes_cli/test_model_switch_variant_tags.py index f6d2b8db66..b3d4ee707b 100644 --- a/tests/hermes_cli/test_model_switch_variant_tags.py +++ b/tests/hermes_cli/test_model_switch_variant_tags.py @@ -24,7 +24,7 @@ def _run_switch(raw_input: str, current_provider: str = "openrouter") -> str: patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \ patch("hermes_cli.runtime_provider.resolve_runtime_provider", return_value={"api_key": "test", "base_url": "", "api_mode": "chat_completions"}), \ - patch("hermes_cli.models.validate_requested_model", return_value=_MOCK_VALIDATION), \ + patch("hermes_cli.models_validate.validate_requested_model", return_value=_MOCK_VALIDATION), \ patch("hermes_cli.model_switch.get_model_info", return_value=None), \ patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \ patch("hermes_cli.models.detect_provider_for_model", return_value=None): diff --git a/tests/hermes_cli/test_model_validation.py b/tests/hermes_cli/test_model_validation.py index 304492f87c..ec3e26cdfe 100644 --- a/tests/hermes_cli/test_model_validation.py +++ b/tests/hermes_cli/test_model_validation.py @@ -3,24 +3,9 @@ import pytest from unittest.mock import MagicMock, patch -from hermes_cli.models import ( - azure_foundry_model_api_mode, - copilot_model_api_mode, - fetch_github_model_catalog, - curated_models_for_provider, - fetch_api_models, - fetch_lmstudio_models, - github_model_reasoning_efforts, - normalize_copilot_model_id, - normalize_opencode_model_id, - normalize_provider, - opencode_model_api_mode, - parse_model_input, - probe_api_models, - provider_label, - provider_model_ids, - validate_requested_model, -) +from hermes_cli.models import azure_foundry_model_api_mode, copilot_model_api_mode, fetch_github_model_catalog, curated_models_for_provider, fetch_api_models, github_model_reasoning_efforts, normalize_copilot_model_id, normalize_opencode_model_id, normalize_provider, opencode_model_api_mode, parse_model_input, probe_api_models, provider_label, provider_model_ids +from hermes_cli.models_local import fetch_lmstudio_models +from hermes_cli.models_validate import validate_requested_model # -- helpers ----------------------------------------------------------------- diff --git a/tests/hermes_cli/test_models.py b/tests/hermes_cli/test_models.py index 0cb4213db0..717940271a 100644 --- a/tests/hermes_cli/test_models.py +++ b/tests/hermes_cli/test_models.py @@ -14,6 +14,8 @@ from hermes_cli.models import ( union_with_portal_paid_recommendations, ) import hermes_cli.models as _models_mod +from hermes_cli import models_local +from hermes_cli import models_validate LIVE_OPENROUTER_MODELS = [ ("anthropic/claude-opus-4.6", "recommended"), @@ -448,7 +450,7 @@ class TestCodexSoftAcceptPlausibilityGate: and mislabel the provider as 'OpenAI Codex').""" def test_unrelated_name_rejected_on_openai_codex(self): - from hermes_cli.models import validate_requested_model + from hermes_cli.models_validate import validate_requested_model r = validate_requested_model("qwen3.5-4b", "openai-codex") assert r["accepted"] is False assert r["persist"] is False @@ -456,7 +458,7 @@ class TestCodexSoftAcceptPlausibilityGate: def test_real_catalog_model_unaffected(self): - from hermes_cli.models import validate_requested_model + from hermes_cli.models_validate import validate_requested_model r = validate_requested_model("gpt-5.5", "openai-codex") assert r["accepted"] is True assert r["recognized"] is True @@ -474,24 +476,24 @@ class TestFormatPricePerMtok: """_format_price_per_mtok: sub-cent prices must not collapse to 'free'/'$0.00'.""" def test_standard_prices_keep_two_decimals(self): - from hermes_cli.models import _format_price_per_mtok + from hermes_cli.models_pricing import _format_price_per_mtok assert _format_price_per_mtok("0.000003") == "$3.00" assert _format_price_per_mtok("0.00003") == "$30.00" assert _format_price_per_mtok("0.00000015") == "$0.15" assert _format_price_per_mtok("0.00018") == "$180.00" def test_zero_is_free(self): - from hermes_cli.models import _format_price_per_mtok + from hermes_cli.models_pricing import _format_price_per_mtok assert _format_price_per_mtok("0") == "free" assert _format_price_per_mtok("0.0") == "free" def test_invalid_is_question_mark(self): - from hermes_cli.models import _format_price_per_mtok + from hermes_cli.models_pricing import _format_price_per_mtok assert _format_price_per_mtok("garbage") == "?" assert _format_price_per_mtok(None) == "?" def test_sub_cent_price_extends_precision(self): - from hermes_cli.models import _format_price_per_mtok + from hermes_cli.models_pricing import _format_price_per_mtok # DeepSeek V4 Flash 0731 promo cache-hit rate: $0.0018/Mtok. assert _format_price_per_mtok("0.0000000018") == "$0.0018" assert _format_price_per_mtok("0.000000001") == "$0.001" @@ -501,7 +503,7 @@ class TestFormatPricePerMtok: assert _format_price_per_mtok("0.00000000001") == "$0.00001" def test_one_cent_boundary_stays_two_decimals(self): - from hermes_cli.models import _format_price_per_mtok + from hermes_cli.models_pricing import _format_price_per_mtok assert _format_price_per_mtok("0.00000001") == "$0.01" @@ -590,7 +592,7 @@ class TestLocalOllamaModelDiscovery: server.shutdown() def test_native_tags_cache_expires(self, monkeypatch): - from hermes_cli.models import fetch_ollama_local_models + from hermes_cli.models_local import fetch_ollama_local_models server, port = _start_fake_ollama_server(models=[{"name": "old-model"}]) try: @@ -612,7 +614,7 @@ class TestLocalOllamaModelDiscovery: def test_fetch_ollama_models_accepts_base_url_without_scheme(self): """OLLAMA_HOST commonly omits http://; discovery should normalize it.""" - from hermes_cli.models import fetch_ollama_local_models + from hermes_cli.models_local import fetch_ollama_local_models server, port = _start_fake_ollama_server(models=[{"name": "qwen2.5:1.5b"}]) try: @@ -622,7 +624,7 @@ class TestLocalOllamaModelDiscovery: def test_fetch_ollama_models_accepts_full_models_url(self): """Pasted OpenAI-style /v1/models URLs should normalize to the native root.""" - from hermes_cli.models import fetch_ollama_local_models + from hermes_cli.models_local import fetch_ollama_local_models server, port = _start_fake_ollama_server(models=[{"name": "qwen2.5:1.5b"}]) try: @@ -636,10 +638,11 @@ class TestLocalOllamaModelDiscovery: def test_runtime_error_from_config_load_does_not_escape_ollama_helpers(self): """Managed-mode config failures should degrade to defaults, not crash pickers.""" - from hermes_cli.models import _get_ollama_base_url, should_use_ollama_native_catalog + from hermes_cli.models import _get_ollama_base_url + from hermes_cli.models_local import should_use_ollama_native_catalog with patch("hermes_cli.config.load_config", side_effect=RuntimeError("bad home")), patch( - "hermes_cli.models.probe_ollama_local_models", + "hermes_cli.models_local.probe_ollama_local_models", return_value=None, ): assert _get_ollama_base_url() == "http://localhost:11434" @@ -647,22 +650,22 @@ class TestLocalOllamaModelDiscovery: def test_probe_ollama_models_malformed_base_url_returns_none(self): """Malformed user-configured URLs should behave like probe failures, not crashes.""" - from hermes_cli.models import probe_ollama_local_models + from hermes_cli.models_local import probe_ollama_local_models assert probe_ollama_local_models("http://127.0.0.1:bad-port/v1") is None def test_fetch_ollama_models_preserves_probe_failure(self): - from hermes_cli.models import fetch_ollama_local_models + from hermes_cli.models_local import fetch_ollama_local_models - with patch("hermes_cli.models.probe_ollama_local_models", return_value=None): + with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=None): assert fetch_ollama_local_models("http://127.0.0.1:11434") is None def test_ollama_port_detection_requires_working_api_tags(self): - from hermes_cli.models import should_use_ollama_native_catalog + from hermes_cli.models_local import should_use_ollama_native_catalog - with patch("hermes_cli.models.probe_ollama_local_models", return_value=["qwen3:1.7b"]): + with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=["qwen3:1.7b"]): assert should_use_ollama_native_catalog("custom", "192.168.1.5:11434/v1") is True - with patch("hermes_cli.models.probe_ollama_local_models", return_value=None): + with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=None): assert should_use_ollama_native_catalog("custom", "192.168.1.5:11434/v1") is False def test_provider_model_ids_ollama_cloud_config_uses_generic_catalog(self): @@ -678,7 +681,7 @@ class TestLocalOllamaModelDiscovery: } } }, - ), patch("hermes_cli.models.fetch_ollama_local_models") as fetch_local, patch( + ), patch("hermes_cli.models_local.fetch_ollama_local_models") as fetch_local, patch( "hermes_cli.models.fetch_api_models", return_value=["qwen3:1.7b"], ) as fetch_generic: @@ -691,7 +694,7 @@ class TestLocalOllamaModelDiscovery: ) def test_native_ollama_catalog_uses_configured_key_env(self, monkeypatch): - from hermes_cli.models import _get_ollama_request_headers + from hermes_cli.models_local import _get_ollama_request_headers monkeypatch.setenv("TEST_OLLAMA_API_KEY", "env-key") with patch( @@ -710,7 +713,7 @@ class TestLocalOllamaModelDiscovery: } def test_native_ollama_catalog_uses_api_key_env_alias(self, monkeypatch): - from hermes_cli.models import _get_ollama_request_headers + from hermes_cli.models_local import _get_ollama_request_headers monkeypatch.setenv("TEST_OLLAMA_API_KEY_ALIAS", "alias-key") with patch( @@ -740,7 +743,7 @@ class TestLocalOllamaModelDiscovery: } }, ), patch( - "hermes_cli.models.fetch_ollama_local_models", + "hermes_cli.models_local.fetch_ollama_local_models", return_value=["qwen3:1.7b"], ) as fetch_local: assert provider_model_ids("ollama", force_refresh=True) == ["qwen3:1.7b"] @@ -757,7 +760,7 @@ class TestLocalOllamaModelDiscovery: "base_url": "http://127.0.0.1:11434/v1", } }, - ), patch("hermes_cli.models.probe_ollama_local_models") as probe_ollama: + ), patch("hermes_cli.models_local.probe_ollama_local_models") as probe_ollama: assert _credential_fingerprint("ollama") probe_ollama.assert_not_called() @@ -801,6 +804,8 @@ class TestLocalOllamaModelDiscovery: def test_clear_provider_models_cache_clears_ollama_native_tags_cache(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate cache = getattr(models, "_OLLAMA_LOCAL_MODELS_CACHE") cache["http://127.0.0.1:11434"] = ("old-model",) @@ -809,6 +814,8 @@ class TestLocalOllamaModelDiscovery: def test_clear_provider_models_cache_custom_clears_native_tags_cache(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate cache = getattr(models, "_OLLAMA_LOCAL_MODELS_CACHE") cache["http://127.0.0.1:11434"] = ("old-model",) @@ -817,6 +824,8 @@ class TestLocalOllamaModelDiscovery: def test_clear_provider_models_cache_does_not_remove_custom_disk_cache(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate disk_cache = { "custom": {"models": ["custom-model"]}, @@ -830,13 +839,13 @@ class TestLocalOllamaModelDiscovery: def test_ollama_cloud_urls_do_not_use_native_local_catalog(self): - from hermes_cli.models import should_use_ollama_native_catalog + from hermes_cli.models_local import should_use_ollama_native_catalog assert should_use_ollama_native_catalog("ollama-cloud", "https://ollama.com/v1") is False assert should_use_ollama_native_catalog("ollama", "https://ollama.com/v1") is False def test_non_ollama_custom_endpoint_uses_generic_catalog_path(self): - from hermes_cli.models import should_use_ollama_native_catalog + from hermes_cli.models_local import should_use_ollama_native_catalog assert should_use_ollama_native_catalog("custom", "https://example.test/v1") is False assert should_use_ollama_native_catalog("openrouter", "http://localhost:11434/v1") is False @@ -888,10 +897,10 @@ class TestLocalOllamaModelDiscovery: from hermes_cli.model_switch import list_authenticated_providers with patch("hermes_cli.config.load_config", return_value={"providers": {}}), patch( - "hermes_cli.models.should_use_ollama_native_catalog", + "hermes_cli.models_local.should_use_ollama_native_catalog", return_value=True, ), patch( - "hermes_cli.models.fetch_ollama_local_models", + "hermes_cli.models_local.fetch_ollama_local_models", return_value=["qwen3:1.7b"], ), patch("hermes_cli.models.fetch_api_models", return_value=[]) as fetch_api: rows = list_authenticated_providers( @@ -1090,7 +1099,7 @@ class TestLocalOllamaModelDiscovery: from hermes_cli.model_switch import list_authenticated_providers with patch("hermes_cli.config.load_config", return_value={"providers": {}}), patch( - "hermes_cli.models.probe_ollama_local_models", + "hermes_cli.models_local.probe_ollama_local_models", return_value=["qwen3:1.7b"], ), patch("hermes_cli.models.fetch_api_models", return_value=[]) as fetch_api: rows = list_authenticated_providers( @@ -1105,7 +1114,7 @@ class TestLocalOllamaModelDiscovery: def test_model_validation_uses_ollama_api_tags_for_ollama_provider(self): """`/model` validation for provider=ollama should not probe `/models`.""" - from hermes_cli.models import validate_requested_model + from hermes_cli.models_validate import validate_requested_model server, port = _start_fake_ollama_server() try: @@ -1128,12 +1137,12 @@ class TestLocalOllamaModelDiscovery: def test_model_validation_ollama_cloud_config_does_not_use_local_tags(self): """provider=ollama with a cloud base URL should not fall into local /api/tags.""" - from hermes_cli.models import validate_requested_model + from hermes_cli.models_validate import validate_requested_model with patch( "hermes_cli.config.load_config", return_value={"providers": {"ollama": {"base_url": "https://ollama.com/v1"}}}, - ), patch("hermes_cli.models.probe_ollama_local_models") as probe_ollama, patch( + ), patch("hermes_cli.models_local.probe_ollama_local_models") as probe_ollama, patch( "hermes_cli.models.probe_api_models", return_value={ "models": ["qwen3:1.7b"], @@ -1152,7 +1161,7 @@ class TestLocalOllamaModelDiscovery: def test_model_validation_uses_ollama_api_tags_for_matching_custom_endpoint(self): """Current-provider `custom` on the configured Ollama URL should use `/api/tags`.""" - from hermes_cli.models import validate_requested_model + from hermes_cli.models_validate import validate_requested_model server, port = _start_fake_ollama_server() base_url = f"http://127.0.0.1:{port}" @@ -1178,7 +1187,7 @@ class TestLocalOllamaModelDiscovery: def test_model_validation_empty_ollama_tags_does_not_fall_back_to_models(self): """Reachable but empty /api/tags should not produce a misleading /models warning.""" - from hermes_cli.models import validate_requested_model + from hermes_cli.models_validate import validate_requested_model server, port = _start_fake_ollama_server(models=[]) base_url = f"http://127.0.0.1:{port}" @@ -1247,7 +1256,7 @@ class TestLocalOllamaModelDiscovery: "api_mode": "chat_completions", }, ), patch( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", return_value={ "accepted": True, "persist": True, @@ -1277,7 +1286,7 @@ class TestLocalOllamaModelDiscovery: from hermes_cli.model_switch import switch_model with patch( - "hermes_cli.models.should_use_ollama_native_catalog", + "hermes_cli.models_local.should_use_ollama_native_catalog", side_effect=RuntimeError("config unavailable"), ), patch( "hermes_cli.runtime_provider.resolve_runtime_provider", @@ -1287,7 +1296,7 @@ class TestLocalOllamaModelDiscovery: "api_mode": "chat_completions", }, ), patch( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", return_value={ "accepted": True, "persist": True, @@ -1313,7 +1322,7 @@ class TestLocalOllamaModelDiscovery: assert result.api_key == "new-key" def test_ollama_root_matching_is_case_insensitive_for_hostnames(self): - from hermes_cli.models import _same_ollama_native_root + from hermes_cli.models_local import _same_ollama_native_root assert _same_ollama_native_root( "HTTP://OLLAMA.EXAMPLE:11434/v1", @@ -1339,6 +1348,8 @@ class TestLocalOllamaModelDiscovery: def test_ollama_failed_probe_is_cached_briefly(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate models._OLLAMA_LOCAL_MODELS_CACHE.clear() models._OLLAMA_LOCAL_PROBE_FAILURE_CACHE.clear() @@ -1346,12 +1357,14 @@ class TestLocalOllamaModelDiscovery: "hermes_cli.models._urlopen_model_catalog_request", side_effect=OSError("offline"), ) as request: - assert models.probe_ollama_local_models("http://127.0.0.1:19999") is None - assert models.probe_ollama_local_models("http://127.0.0.1:19999") is None + assert models_local.probe_ollama_local_models("http://127.0.0.1:19999") is None + assert models_local.probe_ollama_local_models("http://127.0.0.1:19999") is None request.assert_called_once() def test_empty_ollama_catalog_does_not_resurrect_stale_disk_models(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate base_url = "http://127.0.0.1:11434" probe_key = models._ollama_probe_cache_key(base_url, None) @@ -1365,13 +1378,15 @@ class TestLocalOllamaModelDiscovery: models, "_credential_fingerprint", return_value="same" ), patch.object(models, "provider_model_ids", return_value=[]), patch.object( models, "_get_ollama_base_url", return_value=base_url - ), patch.object(models, "_get_ollama_request_headers", return_value={}): + ), patch.object(models_local, "_get_ollama_request_headers", return_value={}): assert models.cached_provider_model_ids("ollama") == [] finally: models._OLLAMA_LOCAL_PROBE_REACHABLE.pop(probe_key, None) def test_failed_ollama_catalog_preserves_stale_disk_models(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate base_url = "http://127.0.0.1:11434" probe_key = models._ollama_probe_cache_key(base_url, None) @@ -1385,13 +1400,15 @@ class TestLocalOllamaModelDiscovery: models, "_credential_fingerprint", return_value="same" ), patch.object(models, "provider_model_ids", return_value=[]), patch.object( models, "_get_ollama_base_url", return_value=base_url - ), patch.object(models, "_get_ollama_request_headers", return_value={}): + ), patch.object(models_local, "_get_ollama_request_headers", return_value={}): assert models.cached_provider_model_ids("ollama") == ["stale:model"] finally: models._OLLAMA_LOCAL_PROBE_REACHABLE.pop(probe_key, None) def test_ollama_native_request_uses_redirect_safe_catalog_helper(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate response = MagicMock() response.read.return_value = b'{"models": [{"name": "qwen3:1.7b"}]}' @@ -1399,13 +1416,15 @@ class TestLocalOllamaModelDiscovery: with patch.object( models, "_urlopen_model_catalog_request", return_value=response ) as request: - assert models.fetch_ollama_local_models("http://127.0.0.1:11434") == [ + assert models_local.fetch_ollama_local_models("http://127.0.0.1:11434") == [ "qwen3:1.7b" ] request.assert_called_once() def test_validation_with_nonmatching_ollama_root_does_not_forward_config_headers(self): import hermes_cli.models as models + from hermes_cli import models_local + from hermes_cli import models_validate with patch( "hermes_cli.config.load_config", @@ -1417,16 +1436,15 @@ class TestLocalOllamaModelDiscovery: } } }, - ), patch.object(models, "should_use_ollama_native_catalog", return_value=True), patch.object( - models, "probe_ollama_local_models", return_value=[] + ), patch.object(models_local, "should_use_ollama_native_catalog", return_value=True), patch.object(models_local, "probe_ollama_local_models", return_value=[] ) as probe: - models.validate_requested_model( + models_validate.validate_requested_model( "qwen3:1.7b", "ollama", base_url="https://other.internal/v1", ) assert probe.call_args.kwargs["headers"] == {} - models.validate_requested_model( + models_validate.validate_requested_model( "qwen3:1.7b", "ollama", base_url="https://other.internal/v1", @@ -1450,7 +1468,7 @@ class TestLocalOllamaModelDiscovery: "hermes_cli.models._get_provider_config_dict", return_value={"base_url": base_url, "api_key": "secret"}, ) as config_provider, patch( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", return_value={"accepted": True, "persist": True, "recognized": True, "message": ""}, ): result = model_switch.switch_model( @@ -1480,7 +1498,7 @@ class TestLocalOllamaModelDiscovery: ) try: with patch.object(model_switch, "get_model_info", return_value=None), patch( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", return_value={"accepted": True, "persist": True, "recognized": True, "message": ""}, ): result = model_switch.switch_model( diff --git a/tests/hermes_cli/test_nous_policy_filter.py b/tests/hermes_cli/test_nous_policy_filter.py index 948885603f..9dd1f8a2d0 100644 --- a/tests/hermes_cli/test_nous_policy_filter.py +++ b/tests/hermes_cli/test_nous_policy_filter.py @@ -12,12 +12,9 @@ import json import pytest import hermes_cli.models as models_mod +from hermes_cli import models_pricing import hermes_cli.nous_account as account_mod -from hermes_cli.models import ( - _NOUS_POLICY_APPEND_MAX, - nous_policy_allowed_ids, - restrict_to_nous_policy, -) +from hermes_cli.models_pricing import _NOUS_POLICY_APPEND_MAX, nous_policy_allowed_ids, restrict_to_nous_policy from hermes_cli.nous_account import nous_policy_present @@ -67,20 +64,18 @@ class TestRestrictToNousPolicy: class TestNousPolicyAllowedIds: @pytest.fixture(autouse=True) def _clear_cache(self): - models_mod._pricing_cache.clear() - models_mod._pricing_cache_retry_after.clear() + models_pricing._pricing_cache.clear() + models_pricing._pricing_cache_retry_after.clear() yield - models_mod._pricing_cache.clear() - models_mod._pricing_cache_retry_after.clear() + models_pricing._pricing_cache.clear() + models_pricing._pricing_cache_retry_after.clear() def _patch(self, monkeypatch, *, policy_present, api_key="sk-test", pricing=None): calls = [] monkeypatch.setattr( account_mod, "nous_policy_present", lambda: policy_present ) - monkeypatch.setattr( - models_mod, - "_resolve_nous_pricing_credentials", + monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials", lambda: (api_key, "https://inference.example.com"), ) @@ -88,7 +83,7 @@ class TestNousPolicyAllowedIds: calls.append(kwargs) return pricing if pricing is not None else {} - monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch) + monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch) return calls def test_returns_the_authenticated_catalog_keys(self, monkeypatch): diff --git a/tests/hermes_cli/test_nous_policy_surfaces.py b/tests/hermes_cli/test_nous_policy_surfaces.py index 5a7bd75a41..f2bc84ea71 100644 --- a/tests/hermes_cli/test_nous_policy_surfaces.py +++ b/tests/hermes_cli/test_nous_policy_surfaces.py @@ -11,7 +11,7 @@ import argparse import pytest import hermes_cli.models as models_mod -from hermes_cli import model_switch_providers +from hermes_cli import models_pricing CURATED = ["vendor/allowed", "vendor/blocked"] ALLOWED = {"vendor/allowed"} @@ -20,14 +20,14 @@ ALLOWED = {"vendor/allowed"} @pytest.fixture def policy(monkeypatch): """An org whose policy admits only ``vendor/allowed``.""" - monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: ALLOWED) + monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: ALLOWED) return ALLOWED @pytest.fixture def no_policy(monkeypatch): """An unrestricted org — lists must come through untouched.""" - monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: None) + monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: None) class TestLoginNous: @@ -51,7 +51,7 @@ class TestLoginNous: }, ) monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED)) - monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {}) + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {}) monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None) monkeypatch.setattr( models_mod, @@ -95,7 +95,7 @@ class TestModelSwitchPicker: lambda *a, **k: {"providers": {"nous": {"access_token": "tok"}}}, ) monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED)) - monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {}) + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {}) monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None) monkeypatch.setattr( models_mod, @@ -122,7 +122,7 @@ class TestModelSwitchPicker: def _boom(_p): raise RuntimeError("portal down") - monkeypatch.setattr(models_mod, "get_pricing_for_provider", _boom) + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", _boom) row = self._rows(monkeypatch) assert row is not None assert "vendor/blocked" not in row["models"] @@ -141,7 +141,7 @@ class TestRecommendedDefaultEndpoint: models_mod, "get_curated_nous_model_ids", lambda: ["vendor/blocked", "vendor/allowed"], ) - monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {}) + monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {}) monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None) monkeypatch.setattr( models_mod, @@ -171,10 +171,10 @@ class TestAuxiliaryFastModel: return {mid: {} for mid in catalog} monkeypatch.setattr( - models_mod, "_resolve_nous_pricing_credentials", + models_pricing, "_resolve_nous_pricing_credentials", lambda: ("sk-nous", "https://inference.example.com"), ) - monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch) + monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch) picked = aux._fast_model_from_catalog("nous") return picked, seen @@ -187,7 +187,7 @@ class TestAuxiliaryFastModel: import agent.auxiliary_client as aux monkeypatch.setattr( - models_mod, "nous_policy_allowed_ids", lambda **_k: {"vendor/allowed"} + models_pricing, "nous_policy_allowed_ids", lambda **_k: {"vendor/allowed"} ) monkeypatch.setattr(aux, "_FAST_MODEL_FAMILIES", ("vendor/",)) monkeypatch.setattr(aux, "_FAST_MODEL_EXCLUDE", ()) @@ -240,14 +240,14 @@ class TestAuxFallbackRespectsPolicy: import agent.auxiliary_client as aux import providers - monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: allowed) + monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: allowed) monkeypatch.setattr( - models_mod, "_resolve_nous_pricing_credentials", + models_pricing, "_resolve_nous_pricing_credentials", lambda: ("sk", "https://inference.example.com"), ) # No fast-family match, so the catalog step yields nothing. monkeypatch.setattr( - models_mod, "fetch_models_with_pricing", + models_pricing, "fetch_models_with_pricing", lambda **_k: {"vendor/allowed-large": {}}, ) @@ -294,7 +294,7 @@ def test_titling_seeds_the_shared_catalog_entry_like_the_pickers(monkeypatch): import agent.auxiliary_client as aux monkeypatch.setattr( - models_mod, "_resolve_nous_pricing_credentials", + models_pricing, "_resolve_nous_pricing_credentials", lambda: ("tok", "https://inference.example.com"), ) seen: dict = {} @@ -303,8 +303,8 @@ def test_titling_seeds_the_shared_catalog_entry_like_the_pickers(monkeypatch): seen.update(kwargs) return {"vendor/haiku": {}} - monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch) + monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch) aux._fast_model_from_catalog("nous") assert seen.get("include_sale_original") is True - assert seen.get("cache_ttl_seconds") == models_mod._NOUS_CATALOG_TTL_SECONDS + assert seen.get("cache_ttl_seconds") == models_pricing._NOUS_CATALOG_TTL_SECONDS diff --git a/tests/hermes_cli/test_nous_reasoning_metadata.py b/tests/hermes_cli/test_nous_reasoning_metadata.py index 6c52f2cb61..55ee3316e0 100644 --- a/tests/hermes_cli/test_nous_reasoning_metadata.py +++ b/tests/hermes_cli/test_nous_reasoning_metadata.py @@ -38,18 +38,19 @@ _CATALOG = ( def cold_cache(monkeypatch): """A freshly started process that has never mirrored a catalog to disk.""" import hermes_cli.models as models_mod + from hermes_cli import models_reasoning_caps monkeypatch.setattr(models_mod, "_nous_reasoning_caps_cache", None) monkeypatch.setattr(models_mod, "_nous_reasoning_caps_failed_at", None) monkeypatch.setattr(models_mod, "_nous_caps_disk_checked", False) monkeypatch.setattr(models_mod, "_nous_caps_warm_started", False) - models_mod._reasoning_caps_disk_path().unlink(missing_ok=True) + models_reasoning_caps._reasoning_caps_disk_path().unlink(missing_ok=True) return models_mod class TestNousModelReasoningCapabilities: def test_fetch_parses_mandatory_flag(self, cold_cache, monkeypatch): - from hermes_cli.models import nous_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities monkeypatch.setattr( cold_cache, "_urlopen_model_catalog_request", @@ -67,7 +68,7 @@ class TestNousModelReasoningCapabilities: def test_catalog_read_sends_user_agent(self, cold_cache, monkeypatch): """The Portal 403s an anonymous catalog read.""" - from hermes_cli.models import nous_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities seen = [] @@ -89,7 +90,7 @@ class TestNousModelReasoningCapabilities: Pinned to production, a staging profile would take its reasoning-mandatory verdicts from a different deployment's catalog. """ - from hermes_cli.models import nous_catalog_url + from hermes_cli.models_reasoning_caps import nous_catalog_url monkeypatch.setenv( "NOUS_INFERENCE_BASE_URL", "https://staging.nousresearch.com/v1" @@ -100,7 +101,7 @@ class TestNousModelReasoningCapabilities: assert nous_catalog_url().endswith("/v1/models") def test_unlisted_and_empty_models_return_none(self, cold_cache, monkeypatch): - from hermes_cli.models import nous_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities monkeypatch.setattr( cold_cache, "_urlopen_model_catalog_request", @@ -111,7 +112,7 @@ class TestNousModelReasoningCapabilities: assert nous_model_reasoning_capabilities(None) is None def test_cache_only_by_default_never_fetches(self, cold_cache, monkeypatch): - from hermes_cli.models import nous_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities def _boom(req, *, timeout): raise AssertionError("hot path must not fetch") @@ -120,7 +121,7 @@ class TestNousModelReasoningCapabilities: assert nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None def test_unreachable_catalog_rate_limits_refetch(self, cold_cache, monkeypatch): - from hermes_cli.models import nous_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities calls = {"n": 0} @@ -136,10 +137,7 @@ class TestNousModelReasoningCapabilities: def test_openrouter_cache_is_independent(self, cold_cache, monkeypatch): """Two catalogs, two caches — a Portal fetch must not answer for OpenRouter.""" - from hermes_cli.models import ( - nous_model_reasoning_capabilities, - openrouter_model_reasoning_capabilities, - ) + from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities, openrouter_model_reasoning_capabilities monkeypatch.setattr(cold_cache, "_openrouter_reasoning_caps_cache", None) monkeypatch.setattr(cold_cache, "_openrouter_reasoning_caps_failed_at", None) diff --git a/tests/hermes_cli/test_ollama_cloud_auth.py b/tests/hermes_cli/test_ollama_cloud_auth.py index c9b88a66c5..93cb610067 100644 --- a/tests/hermes_cli/test_ollama_cloud_auth.py +++ b/tests/hermes_cli/test_ollama_cloud_auth.py @@ -386,7 +386,7 @@ class TestSwitchModelDirectAliasOverride: lambda **kwargs: {"api_key": "", "base_url": "", "api_mode": "openai_compat", "provider": "custom"}, ) - monkeypatch.setattr("hermes_cli.models.validate_requested_model", + monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None}) monkeypatch.setattr("hermes_cli.models.opencode_model_api_mode", lambda *a, **kw: "openai_compat") @@ -411,7 +411,7 @@ class TestSwitchModelDirectAliasOverride: "hermes_cli.runtime_provider.resolve_runtime_provider", lambda **kwargs: {"api_key": "", "base_url": "", "api_mode": "openai_compat", "provider": "custom"}, ) - monkeypatch.setattr("hermes_cli.models.validate_requested_model", + monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None}) monkeypatch.setattr("hermes_cli.models.opencode_model_api_mode", lambda *a, **kw: "openai_compat") diff --git a/tests/hermes_cli/test_ollama_cloud_provider.py b/tests/hermes_cli/test_ollama_cloud_provider.py index 19c0de347f..416e9bc9ac 100644 --- a/tests/hermes_cli/test_ollama_cloud_provider.py +++ b/tests/hermes_cli/test_ollama_cloud_provider.py @@ -340,7 +340,7 @@ class TestOllamaCloudSuffixStripping: def test_strip_suffix_helper(self): """Unit test for the _strip_ollama_cloud_suffix helper.""" - from hermes_cli.models import _strip_ollama_cloud_suffix + from hermes_cli.models_local import _strip_ollama_cloud_suffix assert _strip_ollama_cloud_suffix("kimi-k2.6:cloud") == "kimi-k2.6" assert _strip_ollama_cloud_suffix("glm-5.1:cloud") == "glm-5.1" diff --git a/tests/hermes_cli/test_openai_codex_model_validation_fallback.py b/tests/hermes_cli/test_openai_codex_model_validation_fallback.py index 2b742b058e..666ed4eaf1 100644 --- a/tests/hermes_cli/test_openai_codex_model_validation_fallback.py +++ b/tests/hermes_cli/test_openai_codex_model_validation_fallback.py @@ -18,7 +18,7 @@ it. from unittest.mock import patch from hermes_cli.model_switch import switch_model -from hermes_cli.models import validate_requested_model +from hermes_cli.models_validate import validate_requested_model def test_openai_codex_unknown_but_plausible_model_is_accepted_with_warning(): diff --git a/tests/hermes_cli/test_openai_listing_authority.py b/tests/hermes_cli/test_openai_listing_authority.py index ff34fdc1e7..e29156e17f 100644 --- a/tests/hermes_cli/test_openai_listing_authority.py +++ b/tests/hermes_cli/test_openai_listing_authority.py @@ -21,11 +21,12 @@ from unittest.mock import patch as mock_patch import pytest from hermes_cli import models as models_mod +from hermes_cli import models_validate def _validate(requested: str, base_url: str, live: list[str], provider: str = "openai-api"): with mock_patch.object(models_mod, "fetch_api_models", return_value=live): - return models_mod.validate_requested_model( + return models_validate.validate_requested_model( requested, provider=provider, api_key="sk-test", diff --git a/tests/hermes_cli/test_opencode_go_validation_fallback.py b/tests/hermes_cli/test_opencode_go_validation_fallback.py index 3003acbb19..b4168a7541 100644 --- a/tests/hermes_cli/test_opencode_go_validation_fallback.py +++ b/tests/hermes_cli/test_opencode_go_validation_fallback.py @@ -14,7 +14,7 @@ These tests cover the catalog-fallback path: when ``fetch_api_models`` returns from unittest.mock import patch -from hermes_cli.models import validate_requested_model +from hermes_cli.models_validate import validate_requested_model _UNREACHABLE_PROBE = { diff --git a/tests/hermes_cli/test_openrouter_preset_validation.py b/tests/hermes_cli/test_openrouter_preset_validation.py index 292df9d8c9..a7e280221b 100644 --- a/tests/hermes_cli/test_openrouter_preset_validation.py +++ b/tests/hermes_cli/test_openrouter_preset_validation.py @@ -9,7 +9,7 @@ from unittest.mock import patch import pytest -from hermes_cli.models import validate_requested_model +from hermes_cli.models_validate import validate_requested_model @pytest.mark.parametrize( diff --git a/tests/hermes_cli/test_openrouter_reasoning_metadata.py b/tests/hermes_cli/test_openrouter_reasoning_metadata.py index f4e6d5d620..96499ce6ff 100644 --- a/tests/hermes_cli/test_openrouter_reasoning_metadata.py +++ b/tests/hermes_cli/test_openrouter_reasoning_metadata.py @@ -10,10 +10,8 @@ Covers: import pytest -from hermes_cli.models import ( - clamp_reasoning_effort_to_supported, - parse_openrouter_reasoning_capabilities, -) +from hermes_cli.models import clamp_reasoning_effort_to_supported +from hermes_cli.models_reasoning_caps import parse_openrouter_reasoning_capabilities class TestParseReasoningCapabilities: @@ -121,7 +119,7 @@ class TestOpenRouterModelReasoningCapabilities: monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None) def test_known_model(self, monkeypatch): - from hermes_cli.models import openrouter_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities self._prime_cache(monkeypatch, { "nvidia/nemotron-3-ultra": { "supports_reasoning": True, @@ -133,19 +131,19 @@ class TestOpenRouterModelReasoningCapabilities: assert caps["supports_reasoning"] is True def test_unlisted_model_returns_none(self, monkeypatch): - from hermes_cli.models import openrouter_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities self._prime_cache(monkeypatch, {"a/b": {"supports_reasoning": True}}) assert openrouter_model_reasoning_capabilities("private/custom") is None def test_empty_model_returns_none(self, monkeypatch): - from hermes_cli.models import openrouter_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities self._prime_cache(monkeypatch, {"a/b": {"supports_reasoning": True}}) assert openrouter_model_reasoning_capabilities("") is None assert openrouter_model_reasoning_capabilities(None) is None def test_catalog_unreachable_returns_none_and_rate_limits(self, monkeypatch): import hermes_cli.models as models_mod - from hermes_cli.models import openrouter_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_cache", None) monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None) @@ -163,7 +161,7 @@ class TestOpenRouterModelReasoningCapabilities: def test_cache_only_by_default_never_fetches(self, monkeypatch): import hermes_cli.models as models_mod - from hermes_cli.models import openrouter_model_reasoning_capabilities + from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_cache", None) monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None) diff --git a/tests/hermes_cli/test_pricing_cache_auth_key.py b/tests/hermes_cli/test_pricing_cache_auth_key.py index d9846097fd..4aacc9db28 100644 --- a/tests/hermes_cli/test_pricing_cache_auth_key.py +++ b/tests/hermes_cli/test_pricing_cache_auth_key.py @@ -12,7 +12,8 @@ from unittest.mock import MagicMock import pytest import hermes_cli.models as models_mod -from hermes_cli.models import fetch_models_with_pricing, peek_cached_pricing +from hermes_cli import models_pricing +from hermes_cli.models_pricing import fetch_models_with_pricing, peek_cached_pricing BASE = "https://inference-api.example.com" @@ -23,11 +24,11 @@ _FILTERED = ["vendor/allowed"] @pytest.fixture(autouse=True) def _clear_pricing_cache(): - models_mod._pricing_cache.clear() - models_mod._pricing_cache_retry_after.clear() + models_pricing._pricing_cache.clear() + models_pricing._pricing_cache_retry_after.clear() yield - models_mod._pricing_cache.clear() - models_mod._pricing_cache_retry_after.clear() + models_pricing._pricing_cache.clear() + models_pricing._pricing_cache_retry_after.clear() @pytest.fixture @@ -96,7 +97,7 @@ def test_one_token_does_not_receive_another_tokens_catalog(per_org_catalog): def test_credential_value_does_not_appear_in_the_cache_key(): """Guards against keying on the raw token.""" - assert "sk-super-secret" not in models_mod._pricing_auth_fingerprint("sk-super-secret") + assert "sk-super-secret" not in models_pricing._pricing_auth_fingerprint("sk-super-secret") def test_anonymous_and_authenticated_reads_are_separate(catalog): @@ -159,7 +160,7 @@ class TestNousCatalogExpiry: a long-lived process holds the entry.""" def test_entry_expires_so_a_policy_change_is_picked_up(self, catalog, monkeypatch): - from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS + from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS fetch_models_with_pricing( api_key="sk-test", base_url=BASE, @@ -196,7 +197,7 @@ class TestNousCatalogExpiry: def test_peek_skips_an_expired_entry(self, catalog, monkeypatch): """Reading _pricing_cache directly walked straight past the TTL.""" - from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS + from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS fetch_models_with_pricing( api_key="sk-test", base_url=BASE, diff --git a/tests/hermes_cli/test_provider_groups.py b/tests/hermes_cli/test_provider_groups.py index d8a9be99d7..6be7051234 100644 --- a/tests/hermes_cli/test_provider_groups.py +++ b/tests/hermes_cli/test_provider_groups.py @@ -6,12 +6,8 @@ These are invariant tests, not catalog snapshots: they assert how vendors, which is expected to change over time. """ -from hermes_cli.models import ( - CANONICAL_PROVIDERS, - PROVIDER_GROUPS, - group_providers, - provider_group_for_slug, -) +from hermes_cli.models import CANONICAL_PROVIDERS +from hermes_cli.models_catalog_static import PROVIDER_GROUPS, group_providers, provider_group_for_slug def _slugs(rows): diff --git a/tests/hermes_cli/test_reasoning_caps_disk_cache.py b/tests/hermes_cli/test_reasoning_caps_disk_cache.py index 3b6dcd1759..4ded3cbb29 100644 --- a/tests/hermes_cli/test_reasoning_caps_disk_cache.py +++ b/tests/hermes_cli/test_reasoning_caps_disk_cache.py @@ -17,6 +17,8 @@ import json import pytest import hermes_cli.models as models_mod +from hermes_cli import models_pricing +from hermes_cli import models_reasoning_caps _CATALOG = json.dumps({ @@ -95,16 +97,16 @@ def test_fetched_catalog_answers_a_later_process_offline( models_mod, "_urlopen_model_catalog_request", lambda req, *, timeout: _response(_CATALOG), ) - assert models_mod.nous_model_reasoning_capabilities( + assert models_reasoning_caps.nous_model_reasoning_capabilities( "deepseek/deepseek-v4-pro", allow_fetch=True ) is not None cold_process() monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline) - caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") + caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") assert caps["mandatory"] is False - mandatory = models_mod.nous_model_reasoning_capabilities( + mandatory = models_reasoning_caps.nous_model_reasoning_capabilities( "arcee-ai/trinity-large-thinking" ) assert mandatory["mandatory"] is True @@ -121,14 +123,14 @@ def test_mirror_is_keyed_by_catalog_url(cold_process, offline, monkeypatch): models_mod, "_urlopen_model_catalog_request", lambda req, *, timeout: _response(_CATALOG), ) - models_mod.nous_model_reasoning_capabilities( + models_reasoning_caps.nous_model_reasoning_capabilities( "deepseek/deepseek-v4-pro", allow_fetch=True ) cold_process() monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline) - assert models_mod.openrouter_model_reasoning_capabilities( + assert models_reasoning_caps.openrouter_model_reasoning_capabilities( "deepseek/deepseek-v4-pro" ) is None @@ -140,7 +142,7 @@ def test_staging_portal_does_not_read_productions_mirror( models_mod, "_urlopen_model_catalog_request", lambda req, *, timeout: _response(_CATALOG), ) - models_mod.nous_model_reasoning_capabilities( + models_reasoning_caps.nous_model_reasoning_capabilities( "deepseek/deepseek-v4-pro", allow_fetch=True ) @@ -148,7 +150,7 @@ def test_staging_portal_does_not_read_productions_mirror( monkeypatch.setenv("NOUS_INFERENCE_BASE_URL", "https://staging.nousresearch.com") monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline) - assert models_mod.nous_model_reasoning_capabilities( + assert models_reasoning_caps.nous_model_reasoning_capabilities( "deepseek/deepseek-v4-pro" ) is None @@ -159,29 +161,29 @@ def test_stale_copy_is_still_served(cold_process, offline, monkeypatch): Refusing to read an aged mirror would put every long-idle install back on the cold-start fallback it exists to prevent. """ - url = models_mod.nous_catalog_url() - models_mod._save_reasoning_caps_disk( + url = models_reasoning_caps.nous_catalog_url() + models_reasoning_caps._save_reasoning_caps_disk( url, {"deepseek/deepseek-v4-pro": {"supports_reasoning": True, "mandatory": False}} ) - raw = json.loads(models_mod._reasoning_caps_disk_path().read_text()) + raw = json.loads(models_reasoning_caps._reasoning_caps_disk_path().read_text()) raw[url]["ts"] = 0 # epoch — far past any TTL - models_mod._reasoning_caps_disk_path().write_text(json.dumps(raw)) + models_reasoning_caps._reasoning_caps_disk_path().write_text(json.dumps(raw)) cold_process() monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline) - caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") + caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") assert caps["mandatory"] is False def test_unreadable_mirror_degrades_to_unknown(cold_process, offline, monkeypatch): """A corrupt file answers "unknown", never raises into the request path.""" - path = models_mod._reasoning_caps_disk_path() + path = models_reasoning_caps._reasoning_caps_disk_path() path.parent.mkdir(parents=True, exist_ok=True) path.write_text("{ this is not json") monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline) - assert models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None + assert models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None def test_pricing_fetch_seeds_the_mirror(cold_process, offline, monkeypatch): @@ -190,20 +192,20 @@ def test_pricing_fetch_seeds_the_mirror(cold_process, offline, monkeypatch): Every surface that renders prices goes through here, so the common case never pays a second round-trip to learn the same thing. """ - monkeypatch.setattr(models_mod, "_pricing_cache", {}) - monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {}) + monkeypatch.setattr(models_pricing, "_pricing_cache", {}) + monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {}) monkeypatch.setattr( models_mod, "_urlopen_model_catalog_request", lambda req, *, timeout: _response(_CATALOG), ) - models_mod.fetch_models_with_pricing( + models_pricing.fetch_models_with_pricing( base_url="https://inference-api.nousresearch.com" ) cold_process() monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline) - caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") + caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") assert caps is not None @@ -215,8 +217,8 @@ def test_a_missing_mirror_is_looked_for_once_per_process(cold_process, monkeypat token. Both halves have to be paid at most once. """ counts = {"url": 0, "read": 0} - real_url = models_mod.nous_catalog_url - real_read = models_mod._read_reasoning_caps_disk + real_url = models_reasoning_caps.nous_catalog_url + real_read = models_reasoning_caps._read_reasoning_caps_disk def _counting_url(): counts["url"] += 1 @@ -226,9 +228,9 @@ def test_a_missing_mirror_is_looked_for_once_per_process(cold_process, monkeypat counts["read"] += 1 return real_read() - monkeypatch.setattr(models_mod, "nous_catalog_url", _counting_url) - monkeypatch.setattr(models_mod, "_read_reasoning_caps_disk", _counting_read) + monkeypatch.setattr(models_reasoning_caps, "nous_catalog_url", _counting_url) + monkeypatch.setattr(models_reasoning_caps, "_read_reasoning_caps_disk", _counting_read) for _ in range(5): - models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") + models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") assert counts == {"url": 1, "read": 1} diff --git a/tests/hermes_cli/test_sale_pricing.py b/tests/hermes_cli/test_sale_pricing.py index 251a65af10..f4dc6cf255 100644 --- a/tests/hermes_cli/test_sale_pricing.py +++ b/tests/hermes_cli/test_sale_pricing.py @@ -6,10 +6,8 @@ import json from unittest.mock import MagicMock import hermes_cli.models as models_mod -from hermes_cli.models import ( - compute_sale_discount, - fetch_models_with_pricing, -) +from hermes_cli import models_pricing +from hermes_cli.models_pricing import compute_sale_discount, fetch_models_with_pricing def test_free_model_gets_flat_100_percent_discount(): @@ -31,7 +29,7 @@ def test_paid_model_without_original_shows_no_sale(): def test_fetch_models_with_pricing_copies_nested_original(monkeypatch): - models_mod._pricing_cache.clear() + models_pricing._pricing_cache.clear() payload = { "data": [ { @@ -100,7 +98,7 @@ def test_resolve_nous_pricing_credentials_honors_inference_env_override(monkeypa "hermes_cli.auth.resolve_nous_runtime_credentials", lambda: None, ) - api_key, base_url = models_mod._resolve_nous_pricing_credentials() + api_key, base_url = models_pricing._resolve_nous_pricing_credentials() assert api_key == "" # The bare origin, whichever form the override was written in: callers # append their own path (``/v1/models``), so a suffix here would double up. @@ -119,7 +117,7 @@ def test_resolve_nous_pricing_credentials_normalizes_either_suffix(monkeypatch): "https://stg-inference-api.nousresearch.com/v1/", ): monkeypatch.setenv("NOUS_INFERENCE_BASE_URL", override) - assert models_mod._resolve_nous_pricing_credentials()[1] == ( + assert models_pricing._resolve_nous_pricing_credentials()[1] == ( "https://stg-inference-api.nousresearch.com" ) @@ -131,8 +129,8 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch): call, but it expires — the processes that read this run for weeks, and every caller silently falls back to a curated list meanwhile. """ - models_mod._pricing_cache.clear() - models_mod._pricing_cache_retry_after.clear() + models_pricing._pricing_cache.clear() + models_pricing._pricing_cache_retry_after.clear() calls = [] @@ -151,7 +149,7 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch): monkeypatch.setattr( models_mod.time, "monotonic", - lambda: now + models_mod._FAILED_CATALOG_TTL_SECONDS + 1, + lambda: now + models_pricing._FAILED_CATALOG_TTL_SECONDS + 1, ) assert fetch_models_with_pricing(base_url="https://example.test") == {} assert len(calls) == 2 @@ -159,8 +157,8 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch): def test_a_successful_catalog_fetch_stays_cached(monkeypatch): """Only the failures expire; a real catalog is still fetched once.""" - models_mod._pricing_cache.clear() - models_mod._pricing_cache_retry_after.clear() + models_pricing._pricing_cache.clear() + models_pricing._pricing_cache_retry_after.clear() calls = [] body = json.dumps( @@ -182,7 +180,7 @@ def test_a_successful_catalog_fetch_stays_cached(monkeypatch): monkeypatch.setattr( models_mod.time, "monotonic", - lambda: now + models_mod._FAILED_CATALOG_TTL_SECONDS + 1, + lambda: now + models_pricing._FAILED_CATALOG_TTL_SECONDS + 1, ) assert "a/b" in fetch_models_with_pricing(base_url="https://example.test") assert len(calls) == 1 diff --git a/tests/hermes_cli/test_status_model_provider.py b/tests/hermes_cli/test_status_model_provider.py index ebed771cc4..e19bf7da8c 100644 --- a/tests/hermes_cli/test_status_model_provider.py +++ b/tests/hermes_cli/test_status_model_provider.py @@ -72,7 +72,7 @@ def test_show_status_reports_empty_lmstudio_listing_as_reachable(monkeypatch, ca monkeypatch.setattr(status_mod, "resolve_provider", lambda requested=None, **kwargs: "lmstudio", raising=False) monkeypatch.setattr(status_mod, "provider_label", lambda provider: "LM Studio", raising=False) monkeypatch.setattr( - "hermes_cli.models.probe_lmstudio_models", + "hermes_cli.models_local.probe_lmstudio_models", lambda api_key=None, base_url=None, timeout=5.0: [], ) diff --git a/tests/hermes_cli/test_user_providers_model_switch.py b/tests/hermes_cli/test_user_providers_model_switch.py index 5fe4a74178..963ab9c98f 100644 --- a/tests/hermes_cli/test_user_providers_model_switch.py +++ b/tests/hermes_cli/test_user_providers_model_switch.py @@ -378,7 +378,7 @@ def test_switch_model_resolves_user_provider_credentials(monkeypatch, tmp_path): # Mock validation to pass monkeypatch.setattr( - "hermes_cli.models.validate_requested_model", + "hermes_cli.models_validate.validate_requested_model", lambda *a, **k: {"accepted": True, "persist": True, "recognized": True, "message": None} ) @@ -443,7 +443,7 @@ def _run_user_provider_override_case( with patch("hermes_cli.model_switch.resolve_alias", return_value=None), \ patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \ patch("hermes_cli.model_switch.normalize_model_for_provider", side_effect=lambda model, provider: model), \ - patch("hermes_cli.models.validate_requested_model", return_value=_REJECTED_VALIDATION), \ + patch("hermes_cli.models_validate.validate_requested_model", return_value=_REJECTED_VALIDATION), \ patch("hermes_cli.models.detect_provider_for_model", return_value=None), \ patch("hermes_cli.model_switch.get_model_info", return_value=None), \ patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \ diff --git a/tests/plugins/model_providers/test_ollama_cloud_profile.py b/tests/plugins/model_providers/test_ollama_cloud_profile.py index 540a509aa4..e93a09ce7d 100644 --- a/tests/plugins/model_providers/test_ollama_cloud_profile.py +++ b/tests/plugins/model_providers/test_ollama_cloud_profile.py @@ -199,7 +199,7 @@ class TestOllamaModelSupportsThinking: monkeypatch.setattr(httpx, "Client", _Client) def test_thinking_capability_true(self, monkeypatch): - from hermes_cli.models import ollama_model_supports_thinking + from hermes_cli.models_local import ollama_model_supports_thinking self._patch_show(monkeypatch, capabilities=["completion", "tools", "thinking"]) assert ( @@ -211,7 +211,7 @@ class TestOllamaModelSupportsThinking: def test_probe_failure_returns_none(self, monkeypatch): - from hermes_cli.models import ollama_model_supports_thinking + from hermes_cli.models_local import ollama_model_supports_thinking self._patch_show(monkeypatch, status=404) assert ( @@ -219,7 +219,7 @@ class TestOllamaModelSupportsThinking: ) def test_exception_returns_none(self, monkeypatch): - from hermes_cli.models import ollama_model_supports_thinking + from hermes_cli.models_local import ollama_model_supports_thinking self._patch_show(monkeypatch, raise_exc=RuntimeError("boom")) assert ( diff --git a/tests/run_agent/test_lmstudio_load_mode.py b/tests/run_agent/test_lmstudio_load_mode.py index 8cf4860313..43f5391ac2 100644 --- a/tests/run_agent/test_lmstudio_load_mode.py +++ b/tests/run_agent/test_lmstudio_load_mode.py @@ -1,7 +1,7 @@ from types import SimpleNamespace from typing import Any, cast -from hermes_cli.models import LMStudioLoadResult +from hermes_cli.models_local import LMStudioLoadResult from run_agent import AIAgent @@ -25,7 +25,7 @@ def test_lmstudio_jit_load_mode_skips_explicit_preload(monkeypatch): calls.append((args, kwargs)) return LMStudioLoadResult(64_000) - monkeypatch.setattr("hermes_cli.models.ensure_lmstudio_model_loaded", fake_ensure) + monkeypatch.setattr("hermes_cli.models_local.ensure_lmstudio_model_loaded", fake_ensure) result = AIAgent._ensure_lmstudio_runtime_loaded(cast(Any, _agent("jit"))) diff --git a/tests/run_agent/test_switch_model_context.py b/tests/run_agent/test_switch_model_context.py index 79e7b0812f..f7dff8d633 100644 --- a/tests/run_agent/test_switch_model_context.py +++ b/tests/run_agent/test_switch_model_context.py @@ -4,7 +4,7 @@ from unittest.mock import MagicMock, patch import pytest -from hermes_cli.models import LMStudioLoadResult +from hermes_cli.models_local import LMStudioLoadResult from run_agent import AIAgent from hermes_cli.route_identity import normalize_route_base_url from agent.context_compressor import ContextCompressor diff --git a/tests/test_minimax_model_validation.py b/tests/test_minimax_model_validation.py index 76ef96c1ee..05fa4c447c 100644 --- a/tests/test_minimax_model_validation.py +++ b/tests/test_minimax_model_validation.py @@ -8,7 +8,7 @@ from unittest.mock import patch import pytest -from hermes_cli.models import validate_requested_model +from hermes_cli.models_validate import validate_requested_model class TestMiniMaxModelValidation: