simplify(compat): models — drop 52 re-exports from hermes_cli.models, repoint 16 callers + 41 test files
This commit is contained in:
@@ -29,7 +29,7 @@ def _named_custom_provider_catalogs() -> list[tuple[str, str, list[tuple[str, st
|
||||
from hermes_cli.model_switch import _declared_model_ids, _entry_models_discovered, _models_config_is_allowlist
|
||||
from hermes_cli.model_switch_providers import _NativePickerModelList, _fetch_picker_live_models
|
||||
from hermes_cli.model_switch_providers import _discover_flag
|
||||
from hermes_cli.models import should_use_ollama_native_catalog
|
||||
from hermes_cli.models_local import should_use_ollama_native_catalog
|
||||
from hermes_cli.providers import custom_provider_slug
|
||||
except ImportError:
|
||||
return []
|
||||
|
||||
@@ -689,7 +689,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
|
||||
is_nous = provider_id.strip().lower() == "nous"
|
||||
try:
|
||||
from hermes_cli.auth import resolve_api_key_provider_credentials
|
||||
from hermes_cli.models import fetch_models_with_pricing
|
||||
from hermes_cli.models_pricing import fetch_models_with_pricing
|
||||
from providers import get_provider_profile
|
||||
# Most /v1/models endpoints are authenticated; an anonymous 401 would read as "no small
|
||||
# model" and pin the curated default forever.
|
||||
@@ -704,7 +704,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
|
||||
if not api_key and is_nous:
|
||||
# Nous is OAuth (resolver raises); anonymous reads return the full catalog.
|
||||
try:
|
||||
from hermes_cli.models import _resolve_nous_pricing_credentials
|
||||
from hermes_cli.models_pricing import _resolve_nous_pricing_credentials
|
||||
api_key, base_url = _resolve_nous_pricing_credentials()
|
||||
except Exception:
|
||||
logger.debug("No Nous credentials for catalog", exc_info=True)
|
||||
@@ -719,7 +719,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
|
||||
# policy-catalog expiry.
|
||||
_nous_kwargs = {}
|
||||
if is_nous:
|
||||
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
|
||||
from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS
|
||||
_nous_kwargs = {"include_sale_original": True, "cache_ttl_seconds": _NOUS_CATALOG_TTL_SECONDS}
|
||||
catalog = fetch_models_with_pricing(
|
||||
api_key=api_key or None, base_url=base_url, timeout=3.0, **_nous_kwargs) or {}
|
||||
@@ -730,7 +730,7 @@ def _fast_model_from_catalog(provider_id: str) -> str:
|
||||
if is_nous:
|
||||
# Narrow catalog ids by org policy, as the pickers do.
|
||||
try:
|
||||
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
|
||||
from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy
|
||||
ids = restrict_to_nous_policy(ids, nous_policy_allowed_ids())
|
||||
except Exception:
|
||||
logger.debug("Nous policy filter unavailable", exc_info=True)
|
||||
@@ -770,7 +770,7 @@ def _get_aux_model_for_provider(provider_id: str, *, prefer_fast: bool = False)
|
||||
# let the caller keep the main model.
|
||||
if picked and provider_id.strip().lower() == "nous":
|
||||
try:
|
||||
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
|
||||
from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy
|
||||
allowed = nous_policy_allowed_ids()
|
||||
if allowed and not restrict_to_nous_policy([picked], allowed):
|
||||
return ""
|
||||
|
||||
@@ -131,7 +131,8 @@ def is_free_tier_model(model: str, base_url: str = "") -> bool:
|
||||
if not base_url:
|
||||
return False
|
||||
try:
|
||||
from hermes_cli.models import _is_model_free, peek_cached_pricing
|
||||
from hermes_cli.models import _is_model_free
|
||||
from hermes_cli.models_pricing import peek_cached_pricing
|
||||
|
||||
pricing = peek_cached_pricing(base_url) # owns the /v1-suffix and auth-state key details
|
||||
return bool(pricing) and _is_model_free(model, pricing)
|
||||
|
||||
@@ -71,7 +71,7 @@ class ReasoningParamsMixin:
|
||||
# Live-catalog metadata first (OpenRouter /v1/models supported_parameters) — the static prefix
|
||||
# allowlist repeatedly went stale one vendor at a time. Unknown falls back to the static list.
|
||||
try:
|
||||
from hermes_cli.models import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async
|
||||
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities, warm_openrouter_reasoning_caps_async
|
||||
caps = openrouter_model_reasoning_capabilities(self.model)
|
||||
if caps is None:
|
||||
warm_openrouter_reasoning_caps_async() # cache cold — warm in the background, never block
|
||||
@@ -85,7 +85,7 @@ class ReasoningParamsMixin:
|
||||
def _lmstudio_reasoning_options_cached(self) -> list[str]:
|
||||
"""LM Studio's published reasoning ``allowed_options`` (gate + clamp so toggle models don't 400 on ``high``)."""
|
||||
try:
|
||||
from hermes_cli.models import lmstudio_model_reasoning_options
|
||||
from hermes_cli.models_local import lmstudio_model_reasoning_options
|
||||
except Exception:
|
||||
return []
|
||||
return _cached_probe(self, "_lm_reasoning_opts_cache", lmstudio_model_reasoning_options, [], bool)
|
||||
@@ -93,7 +93,7 @@ class ReasoningParamsMixin:
|
||||
def _ollama_supports_thinking_cached(self) -> bool:
|
||||
"""True only if Ollama's ``/api/show`` declares the ``thinking`` capability."""
|
||||
try:
|
||||
from hermes_cli.models import ollama_model_supports_thinking
|
||||
from hermes_cli.models_local import ollama_model_supports_thinking
|
||||
except Exception:
|
||||
return False
|
||||
return bool(_cached_probe(self, "_ollama_thinking_cache", ollama_model_supports_thinking, None, lambda v: v is not None))
|
||||
|
||||
@@ -59,7 +59,7 @@ class _ModelPickerRows:
|
||||
self, all_models: List[str], pricing: Optional[Dict[str, Dict[str, str]]], *,
|
||||
current_model: str, sale_chrome: bool,
|
||||
) -> None:
|
||||
from hermes_cli.models import _format_price_per_mtok, compute_sale_discount
|
||||
from hermes_cli.models_pricing import _format_price_per_mtok, compute_sale_discount
|
||||
self.current_model = current_model
|
||||
self.has_pricing = bool(pricing and any(pricing.get(m) for m in all_models))
|
||||
# Leave room for a leading "★ " on sale rows (Nous only).
|
||||
|
||||
+11
-3
@@ -1328,9 +1328,17 @@ def _pick_nous_model_after_login(
|
||||
if not isinstance(runtime_key, str) or not runtime_key:
|
||||
raise _nous_err("No runtime API key available to fetch models", "invalid_token")
|
||||
from hermes_cli.models import (
|
||||
get_curated_nous_model_ids, get_pricing_for_provider, check_nous_free_tier,
|
||||
partition_nous_models_by_tier, nous_policy_allowed_ids, restrict_to_nous_policy,
|
||||
union_with_portal_free_recommendations, union_with_portal_paid_recommendations)
|
||||
get_curated_nous_model_ids,
|
||||
check_nous_free_tier,
|
||||
partition_nous_models_by_tier,
|
||||
union_with_portal_free_recommendations,
|
||||
union_with_portal_paid_recommendations,
|
||||
)
|
||||
from hermes_cli.models_pricing import (
|
||||
get_pricing_for_provider,
|
||||
nous_policy_allowed_ids,
|
||||
restrict_to_nous_policy,
|
||||
)
|
||||
model_ids = get_curated_nous_model_ids()
|
||||
_portal = auth_state.get("portal_base_url", "")
|
||||
print()
|
||||
|
||||
+14
-6
@@ -249,9 +249,11 @@ def _reasoning_catalog_reader(slug: str):
|
||||
"""Per-model reasoning-capability reader for aggregators that publish one. Cache-only — the picker
|
||||
must never block on HTTP; a cold cache warms in the background and reports no restriction until then."""
|
||||
try:
|
||||
from hermes_cli.models import (
|
||||
nous_model_reasoning_capabilities, openrouter_model_reasoning_capabilities,
|
||||
warm_nous_reasoning_caps_async, warm_openrouter_reasoning_caps_async,
|
||||
from hermes_cli.models_reasoning_caps import (
|
||||
nous_model_reasoning_capabilities,
|
||||
openrouter_model_reasoning_capabilities,
|
||||
warm_nous_reasoning_caps_async,
|
||||
warm_openrouter_reasoning_caps_async,
|
||||
)
|
||||
except Exception:
|
||||
return None
|
||||
@@ -573,9 +575,15 @@ def _apply_pricing(rows: list[dict], *, force_fresh_nous_tier: bool = False, cac
|
||||
``free_tier`` (account is free-tier) and ``unavailable_models`` (paid models a free user can't pick).
|
||||
``cached_only`` never hits the network: unknown Nous entitlement fails closed (``free_tier_pending``,
|
||||
all models locked) and missing pricing is marked ``pricing_pending``."""
|
||||
from hermes_cli.models_pricing import (
|
||||
_format_price_per_mtok,
|
||||
compute_sale_discount,
|
||||
get_pricing_for_provider,
|
||||
)
|
||||
from hermes_cli.models import (
|
||||
_format_price_per_mtok, check_nous_free_tier, compute_sale_discount, get_cached_nous_free_tier,
|
||||
get_pricing_for_provider, partition_nous_models_by_tier,
|
||||
check_nous_free_tier,
|
||||
get_cached_nous_free_tier,
|
||||
partition_nous_models_by_tier,
|
||||
)
|
||||
|
||||
nous_free_tier: Optional[bool] = None # resolved once (cached in models.py for the TTL window)
|
||||
@@ -686,7 +694,7 @@ def _prewarm_pricing_async(
|
||||
"""Warm picker pricing caches without delaying the current payload (one worker per
|
||||
profile + endpoint scope; a live worker is reused)."""
|
||||
from hermes_constants import hermes_home_key
|
||||
from hermes_cli.models import pricing_cache_scope
|
||||
from hermes_cli.models_pricing import pricing_cache_scope
|
||||
|
||||
slugs = {str(row.get("slug") or "").lower() for row in rows if row.get("slug")}
|
||||
endpoint_scope = tuple(sorted(
|
||||
|
||||
@@ -260,7 +260,7 @@ def _aux_flow_provider_model(task: str, provider_slug: str, curated_models: list
|
||||
current_model: str = "") -> None:
|
||||
"""Prompt for a model under an already-authenticated provider, save to aux."""
|
||||
from hermes_cli.auth import _prompt_model_selection
|
||||
from hermes_cli.models import get_pricing_for_provider
|
||||
from hermes_cli.models_pricing import get_pricing_for_provider
|
||||
display_name = _aux_task_display_name(task)
|
||||
try:
|
||||
pricing = get_pricing_for_provider(provider_slug) or {}
|
||||
@@ -780,7 +780,8 @@ def _build_provider_picker_rows(config: dict, active: str, provider_labels: dict
|
||||
fold into display groups (PROVIDER_GROUPS): a group row's ``members`` drive a sub-picker, leaf
|
||||
rows have ``members == []``; saved custom providers and trailing actions stay flat. Honors
|
||||
``model_catalog.excluded_providers`` (slug or alias, case-insensitive) like the gateway/TUI."""
|
||||
from hermes_cli.models import CANONICAL_PROVIDERS, _PROVIDER_ALIASES, group_providers, provider_group_for_slug
|
||||
from hermes_cli.models import CANONICAL_PROVIDERS, _PROVIDER_ALIASES
|
||||
from hermes_cli.models_catalog_static import group_providers, provider_group_for_slug
|
||||
canonical_descs = {p.slug: p.tui_desc for p in CANONICAL_PROVIDERS}
|
||||
_cli_excluded = {
|
||||
str(p).strip().lower()
|
||||
|
||||
@@ -207,9 +207,12 @@ def _discover_named_custom_models(provider_info: dict, api_key: str, configured_
|
||||
"""Live catalog probe for a named custom endpoint (native ``/api/tags`` for Ollama).
|
||||
Returns ``(models, native_catalog_empty)``; persists the live catalog as a side effect."""
|
||||
from hermes_cli.config import normalize_extra_headers
|
||||
from hermes_cli.models import (
|
||||
fetch_api_models, fetch_ollama_local_models, _get_ollama_native_headers, _normalize_openai_base_url,
|
||||
should_use_ollama_native_catalog)
|
||||
from hermes_cli.models import fetch_api_models, _get_ollama_native_headers
|
||||
from hermes_cli.models_local import (
|
||||
fetch_ollama_local_models,
|
||||
_normalize_openai_base_url,
|
||||
should_use_ollama_native_catalog,
|
||||
)
|
||||
|
||||
name, base_url = provider_info["name"], provider_info["base_url"]
|
||||
api_mode = provider_info.get("api_mode", "")
|
||||
|
||||
@@ -976,7 +976,7 @@ def _apply_direct_alias_endpoint(st: "_Switch", da: DirectAlias) -> None:
|
||||
own endpoint decides: its declared key; else the session key only for the SAME ORIGIN; else a
|
||||
fresh resolution against the alias base_url (env-key fallbacks are host-gated: OLLAMA_API_KEY
|
||||
resolves for ollama.com, OPENROUTER_API_KEY never reaches an unrelated host)."""
|
||||
from hermes_cli.models import _same_ollama_native_root
|
||||
from hermes_cli.models_local import _same_ollama_native_root
|
||||
from hermes_cli.runtime_provider import resolve_runtime_provider
|
||||
alias_key = direct_alias_api_key(da)
|
||||
same_host = _may_reuse_session_credential(st.base_url, da.base_url)
|
||||
@@ -1282,12 +1282,12 @@ def _creds_for_current_provider(st: _Switch) -> None:
|
||||
"""Credentials when staying on the current provider. Mid-session ``/model <name>`` on a local
|
||||
Ollama-compatible endpoint keeps the endpoint in use; re-resolving bare ``custom`` from config
|
||||
can fall through to an unrelated default provider."""
|
||||
from hermes_cli.models import _get_ollama_request_headers, _same_ollama_native_root
|
||||
from hermes_cli.models_local import _get_ollama_request_headers, _same_ollama_native_root
|
||||
keep_current_ollama_endpoint = False
|
||||
ollama_headers: dict[str, str] = {}
|
||||
if st.current_provider == "custom" and st.current_base_url:
|
||||
try:
|
||||
from hermes_cli.models import should_use_ollama_native_catalog
|
||||
from hermes_cli.models_local import should_use_ollama_native_catalog
|
||||
ollama_headers = _get_ollama_request_headers()
|
||||
_, configured_ollama_base = _ollama_configured_base()
|
||||
# Provider-level Ollama headers only belong to the configured native root; without
|
||||
@@ -1343,7 +1343,8 @@ def _resolve_switch_credentials(st: _Switch) -> Optional[ModelSwitchResult]:
|
||||
def _validate_switch(st: _Switch) -> Optional[ModelSwitchResult]:
|
||||
"""COMMON PATH part 2: normalize the model name for the target provider, validate it, and
|
||||
accept config-declared models the remote catalog lacks."""
|
||||
from hermes_cli.models import _get_ollama_request_headers, validate_requested_model
|
||||
from hermes_cli.models_local import _get_ollama_request_headers
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
st.new_model = _resolve_named_custom_model_id(st.new_model, st.target_provider, st.custom_providers)
|
||||
st.new_model = normalize_model_for_provider(st.new_model, st.target_provider)
|
||||
|
||||
|
||||
@@ -89,9 +89,12 @@ def _fetch_picker_live_models(
|
||||
headers: dict[str, str] | None = None, timeout: float = 5.0,
|
||||
api_mode: str | None = None) -> list[str] | None:
|
||||
"""Fetch picker models with native Ollama and cached generic discovery."""
|
||||
from hermes_cli.models import (
|
||||
_get_ollama_native_headers, _normalize_openai_base_url, cached_fetch_api_models,
|
||||
fetch_ollama_local_models, should_use_ollama_native_catalog)
|
||||
from hermes_cli.models import _get_ollama_native_headers, cached_fetch_api_models
|
||||
from hermes_cli.models_local import (
|
||||
_normalize_openai_base_url,
|
||||
fetch_ollama_local_models,
|
||||
should_use_ollama_native_catalog,
|
||||
)
|
||||
|
||||
candidate_headers = _get_ollama_native_headers(api_url, api_key=api_key)
|
||||
|
||||
@@ -395,9 +398,12 @@ def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list:
|
||||
recommendation fetch still yields a policy-filtered curated list."""
|
||||
model_ids = curated.get("nous", [])
|
||||
try:
|
||||
from hermes_cli.models_pricing import get_pricing_for_provider
|
||||
from hermes_cli.models import (
|
||||
get_pricing_for_provider, check_nous_free_tier, union_with_portal_free_recommendations,
|
||||
union_with_portal_paid_recommendations)
|
||||
check_nous_free_tier,
|
||||
union_with_portal_free_recommendations,
|
||||
union_with_portal_paid_recommendations,
|
||||
)
|
||||
from hermes_cli.auth import get_provider_auth_state
|
||||
pricing = get_pricing_for_provider("nous") or {}
|
||||
try:
|
||||
@@ -411,7 +417,7 @@ def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list:
|
||||
except Exception:
|
||||
pass
|
||||
try:
|
||||
from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
|
||||
from hermes_cli.models_pricing import nous_policy_allowed_ids, restrict_to_nous_policy
|
||||
model_ids = restrict_to_nous_policy(model_ids, nous_policy_allowed_ids(), rescue_empty=True)
|
||||
except Exception:
|
||||
pass
|
||||
@@ -996,7 +1002,7 @@ def _build_curated_lists(current_provider: str, current_base_url: str, current_m
|
||||
# unreachable, fall back to the current model so the picker still shows something offline.
|
||||
is_current_lmstudio = current_provider.strip().lower() == "lmstudio"
|
||||
if "lmstudio" not in curated and (os.environ.get("LM_API_KEY") or os.environ.get("LM_BASE_URL") or is_current_lmstudio):
|
||||
from hermes_cli.models import fetch_lmstudio_models
|
||||
from hermes_cli.models_local import fetch_lmstudio_models
|
||||
from hermes_cli.auth import AuthError
|
||||
lm_base = (
|
||||
os.environ.get("LM_BASE_URL")
|
||||
|
||||
+6
-65
@@ -26,12 +26,10 @@ if TYPE_CHECKING:
|
||||
|
||||
from hermes_cli import __version__ as _HERMES_VERSION
|
||||
from hermes_cli.urllib_security import open_credentialed_url
|
||||
from hermes_cli.models_catalog_static import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
|
||||
from hermes_cli.models_catalog_static import (
|
||||
CANONICAL_PROVIDERS,
|
||||
OPENROUTER_MODELS,
|
||||
PREFERRED_SILENT_DEFAULT_MODEL,
|
||||
PROVIDER_GROUPS,
|
||||
ProviderEntry,
|
||||
VERCEL_AI_GATEWAY_MODELS,
|
||||
_AGGREGATOR_PROVIDERS,
|
||||
_AZURE_FOUNDRY_RESPONSES_PREFIXES,
|
||||
@@ -47,78 +45,21 @@ from hermes_cli.models_catalog_static import ( # noqa: F401 (re-exported; test
|
||||
_PROVIDER_MODELS,
|
||||
_PROVIDER_RETIRED_ALIASES,
|
||||
_SILENT_DEFAULT_PROVIDERS,
|
||||
_xai_finalize_catalog,
|
||||
group_providers,
|
||||
provider_group_for_slug)
|
||||
from hermes_cli.models_reasoning_caps import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
|
||||
_xai_finalize_catalog)
|
||||
from hermes_cli.models_reasoning_caps import (
|
||||
_OPENROUTER_CATALOG_URL,
|
||||
_read_reasoning_caps_disk,
|
||||
_reasoning_caps_disk_path,
|
||||
_save_reasoning_caps_disk,
|
||||
_seed_reasoning_caps,
|
||||
nous_catalog_url,
|
||||
nous_model_reasoning_capabilities,
|
||||
# Live-catalog metadata first (ported from PrimeIntellect-ai/prime-agent#1258): OpenRouter's /v1/models
|
||||
# entries advertise reasoning support via supported_parameters + a reasoning object, which covers every
|
||||
# routed vendor without a hand-maintained prefix list. The static prefix allowlist below repeatedly went
|
||||
# stale one vendor at a time (nvidia/ missing → #75386; same class as tencent/, xiaomi/ additions before
|
||||
# it) — metadata makes new vendors work without a code change. One catalog fetch per process, cached;
|
||||
# unknown (catalog unreachable / unlisted model) falls back to the static list.
|
||||
openrouter_model_reasoning_capabilities,
|
||||
parse_openrouter_reasoning_capabilities,
|
||||
refresh_reasoning_caps_async,
|
||||
warm_nous_reasoning_caps_async,
|
||||
warm_openrouter_reasoning_caps_async)
|
||||
from hermes_cli.models_local import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
|
||||
LMStudioLoadResult,
|
||||
_seed_reasoning_caps)
|
||||
from hermes_cli.models_local import (
|
||||
_OLLAMA_LOCAL_MODELS_CACHE,
|
||||
_OLLAMA_LOCAL_MODELS_CACHE_TTL,
|
||||
_OLLAMA_LOCAL_PROBE_FAILURE_CACHE,
|
||||
_OLLAMA_LOCAL_PROBE_REACHABLE,
|
||||
_get_ollama_base_url,
|
||||
_get_ollama_native_headers,
|
||||
_get_ollama_request_headers,
|
||||
_lmstudio_fetch_raw_models,
|
||||
_lmstudio_server_root,
|
||||
_normalize_openai_base_url,
|
||||
_ollama_local_catalog,
|
||||
_ollama_probe_cache_key,
|
||||
_root_for_ollama_native_api,
|
||||
_same_ollama_native_root,
|
||||
_strip_ollama_cloud_suffix,
|
||||
ensure_lmstudio_model_loaded,
|
||||
fetch_lmstudio_models,
|
||||
fetch_ollama_cloud_models,
|
||||
fetch_ollama_local_models,
|
||||
lmstudio_model_reasoning_options,
|
||||
ollama_model_supports_thinking,
|
||||
probe_lmstudio_models,
|
||||
probe_ollama_local_models,
|
||||
should_use_ollama_native_catalog)
|
||||
from hermes_cli.models_pricing import ( # noqa: F401 (re-exported; tests patch hermes_cli.models.<name>)
|
||||
_FAILED_CATALOG_TTL_SECONDS,
|
||||
_NOUS_CATALOG_TTL_SECONDS,
|
||||
_NOUS_POLICY_APPEND_MAX,
|
||||
_cache_catalog,
|
||||
_cached_catalog,
|
||||
_fetch_deepinfra_pricing,
|
||||
_fetch_novita_pricing,
|
||||
_format_price_per_mtok,
|
||||
_pricing_auth_fingerprint,
|
||||
_pricing_cache,
|
||||
_pricing_cache_retry_after,
|
||||
_resolve_nous_pricing_credentials,
|
||||
compute_sale_discount,
|
||||
fetch_ai_gateway_pricing,
|
||||
fetch_models_with_pricing,
|
||||
_pricing_provider_cache_keys,
|
||||
get_cached_nous_inference_base_url,
|
||||
get_pricing_for_provider,
|
||||
nous_policy_allowed_ids,
|
||||
peek_cached_pricing,
|
||||
pricing_cache_scope,
|
||||
restrict_to_nous_policy)
|
||||
from hermes_cli.models_validate import validate_requested_model # noqa: F401 (re-exported)
|
||||
fetch_ollama_cloud_models)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@@ -1,8 +1,7 @@
|
||||
"""Static provider/model catalog tables (data only — no network).
|
||||
|
||||
Curated per-provider model lists, the canonical provider registry, display groups and alias
|
||||
maps. Split out of ``hermes_cli.models``, which re-imports every name so
|
||||
``hermes_cli.models.<name>`` keeps resolving (and monkeypatching) as before.
|
||||
maps. Split out of ``hermes_cli.models``.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
@@ -2,9 +2,8 @@
|
||||
|
||||
Ollama (native ``/api/tags`` probe, request headers, base-url resolution), LM Studio
|
||||
(``/api/v1/models``, load-on-demand) and Ollama Cloud (live + models.dev merged catalog with a
|
||||
disk cache). Split out of ``hermes_cli.models``, which re-imports every name; origin helpers are
|
||||
looked up on ``hermes_cli.models`` at call time so ``patch("hermes_cli.models.<name>")`` mocks
|
||||
keep intercepting.
|
||||
disk cache). Split out of ``hermes_cli.models``; helpers still defined there are looked up on
|
||||
``hermes_cli.models`` at call time so ``patch("hermes_cli.models.<name>")`` mocks keep intercepting.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -96,7 +95,7 @@ def _get_ollama_base_url() -> str:
|
||||
the active provider is ollama, or custom AND the endpoint actually serves ``/api/tags``
|
||||
(otherwise the picker would probe an unrelated endpoint and hide the local catalog) →
|
||||
``OLLAMA_HOST`` → Ollama's local default."""
|
||||
from hermes_cli.models import _get_model_config_dict, should_use_ollama_native_catalog
|
||||
from hermes_cli.models import _get_model_config_dict
|
||||
configured = _configured_ollama_base_url()
|
||||
if configured:
|
||||
return configured
|
||||
@@ -152,7 +151,6 @@ def _get_ollama_native_headers(base_url: Optional[str], *, api_key: Optional[str
|
||||
"""Ollama credentials and headers for one endpoint origin. Configured headers apply only when
|
||||
*base_url* shares the configured Ollama root; an explicit *api_key* replaces any configured
|
||||
Authorization variant rather than inheriting it."""
|
||||
from hermes_cli.models import _get_ollama_request_headers
|
||||
configured_base = _configured_ollama_base_url()
|
||||
explicit_key = str(api_key or "").strip()
|
||||
configured_matches = bool(configured_base and base_url and _same_ollama_native_root(base_url, configured_base))
|
||||
@@ -260,7 +258,6 @@ def fetch_ollama_local_models(
|
||||
headers: Optional[dict[str, str]] = None,
|
||||
) -> Optional[list[str]]:
|
||||
"""Fetch local Ollama-compatible models, preserving probe failure as ``None``."""
|
||||
from hermes_cli.models import probe_ollama_local_models
|
||||
return probe_ollama_local_models(base_url, timeout, headers=headers)
|
||||
|
||||
|
||||
@@ -295,7 +292,6 @@ def should_use_ollama_native_catalog(
|
||||
Ollama's default port actually serves ``/api/tags``. (Bare ``ollama`` is normalized to
|
||||
``custom`` elsewhere so runtime paths share the OpenAI client, but ``/api/tags`` is the
|
||||
authoritative local list; other custom endpoints keep the ``/models`` probe.)"""
|
||||
from hermes_cli.models import probe_ollama_local_models
|
||||
requested = str(provider or "").strip().lower()
|
||||
root = _root_for_ollama_native_api(base_url or "")
|
||||
if root:
|
||||
@@ -342,7 +338,7 @@ def _ollama_local_catalog(force_refresh: bool) -> list[str]:
|
||||
"""Catalog for the raw ``ollama`` provider: native ``/api/tags`` when the endpoint is a real
|
||||
Ollama server, else the OpenAI-style ``/v1/models`` of the configured gateway (incl. Ollama
|
||||
Cloud)."""
|
||||
from hermes_cli.models import _get_ollama_base_url, _get_ollama_native_headers, _get_provider_config_dict, fetch_api_models, fetch_ollama_local_models, should_use_ollama_native_catalog
|
||||
from hermes_cli.models import _get_provider_config_dict, fetch_api_models
|
||||
if force_refresh:
|
||||
_OLLAMA_LOCAL_MODELS_CACHE.clear()
|
||||
_OLLAMA_LOCAL_PROBE_FAILURE_CACHE.clear()
|
||||
@@ -413,7 +409,6 @@ def _lmstudio_fetch_raw_models(
|
||||
|
||||
def _lmstudio_raw_models_or_none(api_key, base_url, timeout) -> Optional[list[dict]]:
|
||||
"""``_lmstudio_fetch_raw_models`` with every failure (incl. AuthError) collapsed to None."""
|
||||
from hermes_cli.models import _lmstudio_fetch_raw_models
|
||||
try:
|
||||
return _lmstudio_fetch_raw_models(api_key=api_key, base_url=base_url, timeout=timeout)
|
||||
except Exception:
|
||||
@@ -435,7 +430,6 @@ def probe_lmstudio_models(
|
||||
"""Chat-capable LM Studio model keys — a valid empty list when the server is reachable but has
|
||||
no non-embedding models; ``None`` on network errors, malformed responses, or bad base URLs.
|
||||
Raises ``AuthError`` on HTTP 401/403 so token issues surface separately from reachability."""
|
||||
from hermes_cli.models import _lmstudio_fetch_raw_models
|
||||
raw_models = _lmstudio_fetch_raw_models(api_key=api_key, base_url=base_url, timeout=timeout)
|
||||
if raw_models is None:
|
||||
return None
|
||||
@@ -457,7 +451,6 @@ def fetch_lmstudio_models(
|
||||
) -> list[str]:
|
||||
"""LM Studio chat-capable model keys; ``[]`` when unreachable/malformed. Raises ``AuthError`` on
|
||||
HTTP 401/403 so callers can tell a wrong ``LM_API_KEY`` from an unreachable server."""
|
||||
from hermes_cli.models import probe_lmstudio_models
|
||||
return probe_lmstudio_models(api_key=api_key, base_url=base_url, timeout=timeout) or []
|
||||
|
||||
|
||||
|
||||
@@ -2,9 +2,9 @@
|
||||
|
||||
OpenRouter-compatible ``/v1/models`` pricing fetch with a per-endpoint/per-credential cache,
|
||||
Nous Portal sale chrome and org-policy filtering, and the Vercel AI Gateway / Novita / Fireworks /
|
||||
DeepInfra pricing adapters. Split out of ``hermes_cli.models``, which re-imports every name;
|
||||
origin helpers and the cache dicts are looked up on ``hermes_cli.models`` at call time so
|
||||
``patch("hermes_cli.models.<name>")`` mocks keep intercepting.
|
||||
DeepInfra pricing adapters. Split out of ``hermes_cli.models``; helpers still defined there are
|
||||
looked up on ``hermes_cli.models`` at call time so ``patch("hermes_cli.models.<name>")`` mocks keep
|
||||
intercepting.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -33,7 +33,6 @@ _pricing_cache_retry_after: dict[str, float] = {}
|
||||
|
||||
def _cached_catalog(cache_key: str) -> Optional[dict[str, dict[str, Any]]]:
|
||||
"""The cached catalog for *cache_key*, or None to go fetch it."""
|
||||
from hermes_cli.models import _pricing_cache, _pricing_cache_retry_after
|
||||
cached = _pricing_cache.get(cache_key)
|
||||
if cached is None:
|
||||
return None
|
||||
@@ -53,7 +52,6 @@ def _cache_catalog(
|
||||
"""Cache a catalog result, giving an empty one an expiry. *ttl_seconds* expires a non-empty
|
||||
result too — only for a catalog whose contents depend on server-side state the client cannot
|
||||
observe (an org's model policy can change while a long-lived process holds the entry)."""
|
||||
from hermes_cli.models import _pricing_cache, _pricing_cache_retry_after
|
||||
_pricing_cache[cache_key] = result
|
||||
if not result:
|
||||
_pricing_cache_retry_after[cache_key] = time.monotonic() + _FAILED_CATALOG_TTL_SECONDS
|
||||
@@ -84,7 +82,6 @@ def peek_cached_pricing(base_url: str) -> dict[str, dict[str, Any]]:
|
||||
"""Pricing already cached for *base_url* (with or without ``/v1``), or ``{}``; never fetches.
|
||||
Prefers an authenticated catalog, scanning newest first (callers hold no credential) and
|
||||
skipping expired entries so a rotated credential does not answer from its predecessor's."""
|
||||
from hermes_cli.models import _pricing_cache
|
||||
root = _strip_v1((base_url or "").rstrip("/"))
|
||||
authed_prefix = root + _PRICING_AUTH_KEY_PREFIX
|
||||
for key in reversed(list(_pricing_cache)):
|
||||
@@ -317,7 +314,6 @@ _NOUS_CATALOG_TTL_SECONDS = 300.0
|
||||
|
||||
def _fetch_nous_pricing(api_key: str, base_url: str, *, force_refresh: bool) -> dict[str, dict[str, Any]]:
|
||||
"""Shared by pricing and policy lookups so both read one cache entry."""
|
||||
from hermes_cli.models import fetch_models_with_pricing
|
||||
return fetch_models_with_pricing(
|
||||
api_key=api_key,
|
||||
base_url=base_url,
|
||||
@@ -332,7 +328,6 @@ def nous_policy_allowed_ids(*, force_refresh: bool = False) -> Optional[set[str]
|
||||
which omits policy-blocked rows), or ``None`` to not filter: no policy (or a token too old to
|
||||
say), an anonymous read (unfiltered catalog), or an empty read (a fetch failure, not an org
|
||||
that may reach nothing)."""
|
||||
from hermes_cli.models import _resolve_nous_pricing_credentials
|
||||
try:
|
||||
from hermes_cli.nous_account import nous_policy_present
|
||||
|
||||
@@ -373,12 +368,11 @@ def restrict_to_nous_policy(
|
||||
|
||||
|
||||
def _remember_provider_cache_key(provider: str, cache_key: str) -> None:
|
||||
from hermes_cli.models import _pricing_profile_key, _pricing_provider_cache_keys
|
||||
from hermes_cli.models import _pricing_profile_key
|
||||
_pricing_provider_cache_keys[(_pricing_profile_key(), provider)] = cache_key
|
||||
|
||||
|
||||
def _fetch_openrouter_pricing(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
|
||||
from hermes_cli.models import fetch_models_with_pricing
|
||||
_remember_provider_cache_key("openrouter", _OPENROUTER_PRICING_BASE)
|
||||
return fetch_models_with_pricing(
|
||||
api_key=_resolve_openrouter_api_key(),
|
||||
@@ -388,13 +382,11 @@ def _fetch_openrouter_pricing(*, force_refresh: bool = False) -> dict[str, dict[
|
||||
|
||||
|
||||
def _fetch_ai_gateway_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
|
||||
from hermes_cli.models import fetch_ai_gateway_pricing
|
||||
_remember_provider_cache_key("ai-gateway", _ai_gateway_pricing_scope())
|
||||
return fetch_ai_gateway_pricing(force_refresh=force_refresh)
|
||||
|
||||
|
||||
def _fetch_novita_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
|
||||
from hermes_cli.models import _fetch_novita_pricing
|
||||
_remember_provider_cache_key("novita", _novita_pricing_scope())
|
||||
return _fetch_novita_pricing(force_refresh=force_refresh)
|
||||
|
||||
@@ -405,7 +397,6 @@ def _fetch_fireworks_pricing_for_provider(*, force_refresh: bool = False) -> dic
|
||||
|
||||
|
||||
def _fetch_nous_pricing_for_provider(*, force_refresh: bool = False) -> dict[str, dict[str, Any]]:
|
||||
from hermes_cli.models import _resolve_nous_pricing_credentials
|
||||
api_key, base_url = _resolve_nous_pricing_credentials()
|
||||
if not base_url:
|
||||
return {}
|
||||
@@ -453,9 +444,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba
|
||||
"""The current endpoint identity a provider's pricing cache is keyed on. Resolves local configuration
|
||||
only, never fetches: picker prewarm single-flight uses it so an endpoint rotation can start a new
|
||||
worker while the previous endpoint is still slow or unreachable."""
|
||||
from hermes_cli.models import (
|
||||
_deepinfra_catalog_url, _pricing_profile_key, _pricing_provider_cache_keys, normalize_provider,
|
||||
)
|
||||
from hermes_cli.models import _deepinfra_catalog_url, _pricing_profile_key, normalize_provider
|
||||
normalized = normalize_provider(provider)
|
||||
static = _STATIC_PRICING_SCOPES.get(normalized)
|
||||
if static:
|
||||
@@ -473,12 +462,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba
|
||||
return env_base.rstrip("/").removesuffix("/v1")
|
||||
if normalize_provider(current_provider) == "nous" and current_base_url:
|
||||
return current_base_url.rstrip("/").removesuffix("/v1")
|
||||
# Call-time lookup through the facade: tests (and plugins) patch
|
||||
# ``hermes_cli.models.get_cached_nous_inference_base_url``; a bare module-global read here would
|
||||
# silently bypass the patch and fall back to the default endpoint.
|
||||
from hermes_cli.models import get_cached_nous_inference_base_url as _persisted_nous_base
|
||||
|
||||
persisted_base = _persisted_nous_base()
|
||||
persisted_base = get_cached_nous_inference_base_url()
|
||||
if persisted_base:
|
||||
return persisted_base
|
||||
return _pricing_provider_cache_keys.get((_pricing_profile_key(), normalized), _DEFAULT_NOUS_INFERENCE_BASE)
|
||||
@@ -487,10 +471,7 @@ def pricing_cache_scope(provider: str, *, current_provider: str = "", current_ba
|
||||
|
||||
def _cached_only_pricing(normalized: str) -> dict[str, dict[str, str]]:
|
||||
"""Process-resident pricing for *normalized* without any provider I/O."""
|
||||
from hermes_cli.models import (
|
||||
_deepinfra_catalog_cache, _deepinfra_catalog_url, _fetch_deepinfra_pricing, _pricing_profile_key,
|
||||
_pricing_provider_cache_keys,
|
||||
)
|
||||
from hermes_cli.models import _deepinfra_catalog_cache, _deepinfra_catalog_url, _pricing_profile_key
|
||||
if normalized == "deepinfra":
|
||||
cache_key, _url = _deepinfra_catalog_url()
|
||||
return _fetch_deepinfra_pricing() if cache_key in _deepinfra_catalog_cache else {}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
"""Per-model reasoning capabilities from OpenRouter-schema ``/v1/models`` catalogs.
|
||||
|
||||
Split out of ``hermes_cli.models``; every public/patched name is re-imported there. OpenRouter and
|
||||
Split out of ``hermes_cli.models``. OpenRouter and
|
||||
Nous Portal share one implementation parametrized by :class:`_CapsSource`; the per-source module
|
||||
globals (``_openrouter_reasoning_caps_cache``, ``_nous_caps_disk_checked``, ...) stay defined on
|
||||
``hermes_cli.models`` — tests reset them there — and are read/written by attribute name.
|
||||
@@ -81,7 +81,7 @@ def _read_reasoning_caps_disk() -> dict[str, Any]:
|
||||
|
||||
def _load_reasoning_caps_disk(url: str) -> tuple[Optional[Caps], float]:
|
||||
"""Return ``(caps, age_seconds)`` for *url*, or ``(None, 0.0)``."""
|
||||
entry = _origin()._read_reasoning_caps_disk().get(url)
|
||||
entry = _read_reasoning_caps_disk().get(url)
|
||||
caps = entry.get("caps") if isinstance(entry, dict) else None
|
||||
if not isinstance(caps, dict) or not caps:
|
||||
return None, 0.0
|
||||
@@ -96,7 +96,7 @@ def _save_reasoning_caps_disk(url: str, caps: Caps) -> None:
|
||||
"""Merge *url*'s catalog into the shared disk mirror, atomically."""
|
||||
from hermes_cli.models import _write_json_cache
|
||||
try:
|
||||
data = _origin()._read_reasoning_caps_disk()
|
||||
data = _read_reasoning_caps_disk()
|
||||
data[url] = {"ts": time.time(), "caps": caps}
|
||||
_write_json_cache(_reasoning_caps_disk_path(), data, indent=0, separators=(",", ":"))
|
||||
except Exception as exc:
|
||||
@@ -250,16 +250,23 @@ _OPENROUTER_CAPS = _CapsSource(
|
||||
_NOUS_CAPS = _CapsSource(
|
||||
"_nous_reasoning_caps_cache", "_nous_reasoning_caps_failed_at",
|
||||
"_nous_caps_disk_checked", "_nous_caps_warm_started",
|
||||
lambda: _origin().nous_catalog_url(),
|
||||
lambda: nous_catalog_url(),
|
||||
)
|
||||
|
||||
|
||||
def nous_catalog_url() -> str:
|
||||
"""The Portal ``/v1/models`` URL for the endpoint we actually talk to (``NOUS_INFERENCE_BASE_URL``
|
||||
→ resolved credential base → prod), so a staging profile reads staging's capabilities."""
|
||||
return f"{_origin()._resolve_nous_pricing_credentials()[1]}/v1/models"
|
||||
from hermes_cli.models_pricing import _resolve_nous_pricing_credentials
|
||||
return f"{_resolve_nous_pricing_credentials()[1]}/v1/models"
|
||||
|
||||
|
||||
# Live-catalog metadata first (ported from PrimeIntellect-ai/prime-agent#1258): OpenRouter's /v1/models
|
||||
# entries advertise reasoning support via supported_parameters + a reasoning object, which covers every
|
||||
# routed vendor without a hand-maintained prefix list. The static prefix allowlist repeatedly went
|
||||
# stale one vendor at a time (nvidia/ missing → #75386; same class as tencent/, xiaomi/ additions before
|
||||
# it) — metadata makes new vendors work without a code change. One catalog fetch per process, cached;
|
||||
# unknown (catalog unreachable / unlisted model) falls back to the static list.
|
||||
def openrouter_model_reasoning_capabilities(
|
||||
model_id: Optional[str], *, timeout: float = 6.0, allow_fetch: bool = False,
|
||||
) -> Optional[dict[str, Any]]:
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
"""Validate a requested ``/model`` value against the active provider's catalog.
|
||||
|
||||
Split out of ``hermes_cli.models``; :func:`validate_requested_model` is re-imported there. Every
|
||||
catalog fetcher this module calls is looked up on ``hermes_cli.models`` at call time (``_m.<name>``)
|
||||
so existing ``patch("hermes_cli.models.<name>")`` mocks keep intercepting.
|
||||
Split out of ``hermes_cli.models``. Catalog fetchers defined in ``hermes_cli.models`` are looked up
|
||||
there at call time (``_m.<name>``) so ``patch("hermes_cli.models.<name>")`` mocks keep intercepting;
|
||||
local-server probes are looked up on ``hermes_cli.models_local`` (``_ml.<name>``) the same way.
|
||||
|
||||
Every provider branch returns a verdict dict (see :func:`_verdict`) or ``None`` for "not decided
|
||||
here — keep walking the ladder". The ladder ORDER is behavior (see ``_LADDER``).
|
||||
@@ -170,13 +170,13 @@ def _parse_openrouter_preset(req: _Request) -> Optional[dict[str, Any]]:
|
||||
|
||||
|
||||
def _validate_lmstudio(req: _Request) -> dict[str, Any]:
|
||||
from hermes_cli import models as _m
|
||||
from hermes_cli import models_local as _ml
|
||||
from hermes_cli.auth import AuthError
|
||||
|
||||
# probe_lmstudio_models distinguishes None (unreachable / malformed) from [] (reachable,
|
||||
# nothing chat-capable loaded); fetch_lmstudio_models collapses both to [].
|
||||
try:
|
||||
models = _m.probe_lmstudio_models(api_key=req.api_key, base_url=req.base_url)
|
||||
models = _ml.probe_lmstudio_models(api_key=req.api_key, base_url=req.base_url)
|
||||
except AuthError as exc:
|
||||
return _reject(f"{exc} Set `LM_API_KEY` (or update it) to match the server's bearer token.")
|
||||
if models is None:
|
||||
@@ -194,10 +194,11 @@ def _ollama_probe_headers(req: _Request) -> dict[str, str]:
|
||||
when the probed endpoint is the configured one (never leak them to a different host). Caller
|
||||
headers win; a caller ``api_key`` becomes the Authorization header unless the caller sent one."""
|
||||
from hermes_cli import models as _m
|
||||
from hermes_cli import models_local as _ml
|
||||
from hermes_cli.models_local import _configured_ollama_base_url, _drop_authorization
|
||||
|
||||
configured_base = _configured_ollama_base_url()
|
||||
configured_allowed = not configured_base or _m._same_ollama_native_root(req.base_url or "", configured_base)
|
||||
configured_allowed = not configured_base or _ml._same_ollama_native_root(req.base_url or "", configured_base)
|
||||
configured = _m._get_ollama_native_headers(req.base_url, api_key=req.api_key) if configured_allowed else {}
|
||||
if req.headers is None:
|
||||
return configured
|
||||
@@ -215,17 +216,18 @@ def _validate_ollama_native(req: _Request) -> Optional[dict[str, Any]]:
|
||||
looks like a local Ollama server. Also resolves ``base_url`` for the raw ``ollama`` provider,
|
||||
which later branches (custom) rely on."""
|
||||
from hermes_cli import models as _m
|
||||
from hermes_cli import models_local as _ml
|
||||
|
||||
if str(req.provider or "").strip().lower() == "ollama" and not req.base_url:
|
||||
req.base_url = _m._get_ollama_base_url()
|
||||
headers = _ollama_probe_headers(req)
|
||||
if not _m.should_use_ollama_native_catalog(req.provider, req.base_url, headers=headers):
|
||||
if not _ml.should_use_ollama_native_catalog(req.provider, req.base_url, headers=headers):
|
||||
return None
|
||||
models = _m.probe_ollama_local_models(req.base_url, headers=headers)
|
||||
models = _ml.probe_ollama_local_models(req.base_url, headers=headers)
|
||||
if models is None:
|
||||
# A failed native probe is not authoritative; fall back to the OpenAI-compatible catalog.
|
||||
models = _m.probe_api_models(
|
||||
req.api_key, _m._normalize_openai_base_url(req.base_url), request_headers=headers,
|
||||
req.api_key, _ml._normalize_openai_base_url(req.base_url), request_headers=headers,
|
||||
).get("models")
|
||||
if models is None:
|
||||
return _soft_accept(
|
||||
|
||||
@@ -131,10 +131,11 @@ async def get_model_options(
|
||||
|
||||
def _nous_recommended_default() -> dict:
|
||||
from hermes_cli import models as m
|
||||
from hermes_cli import models_pricing as mp
|
||||
from hermes_cli.auth import get_provider_auth_state
|
||||
|
||||
model_ids = m.get_curated_nous_model_ids()
|
||||
pricing = m.get_pricing_for_provider("nous") or {}
|
||||
pricing = mp.get_pricing_for_provider("nous") or {}
|
||||
free_tier = m.check_nous_free_tier(force_fresh=True)
|
||||
|
||||
try:
|
||||
@@ -145,10 +146,10 @@ def _nous_recommended_default() -> dict:
|
||||
# This endpoint picks the model a user lands on without choosing it, so an unreachable
|
||||
# one is worse than in a picker. Narrow to policy BEFORE the tier split, so a rescued
|
||||
# id still has to pass the free/paid predicate.
|
||||
policy_allowed = m.nous_policy_allowed_ids()
|
||||
policy_allowed = mp.nous_policy_allowed_ids()
|
||||
union = m.union_with_portal_free_recommendations if free_tier else m.union_with_portal_paid_recommendations
|
||||
model_ids, pricing = union(model_ids, pricing, portal_url)
|
||||
model_ids = m.restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True)
|
||||
model_ids = mp.restrict_to_nous_policy(model_ids, policy_allowed, rescue_empty=True)
|
||||
if free_tier:
|
||||
model_ids, _unavailable = m.partition_nous_models_by_tier(model_ids, pricing, free_tier=True)
|
||||
|
||||
|
||||
@@ -39,7 +39,7 @@ class NousProfile(ProviderProfile):
|
||||
"""True when ``reasoning: {enabled: false}`` would 400 on *model*. Cache-only catalog
|
||||
lookup; unknown/cold (warmer kicked) and no-reasoning routes both answer True (omit > 400)."""
|
||||
try:
|
||||
from hermes_cli.models import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async
|
||||
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities, warm_nous_reasoning_caps_async
|
||||
|
||||
caps = nous_model_reasoning_capabilities(model)
|
||||
if caps is None:
|
||||
|
||||
@@ -64,7 +64,8 @@ class OpenRouterProfile(ProviderProfile):
|
||||
if not effort and not disabled:
|
||||
return cfg
|
||||
try:
|
||||
from hermes_cli.models import clamp_reasoning_effort_to_supported, openrouter_model_reasoning_capabilities
|
||||
from hermes_cli.models import clamp_reasoning_effort_to_supported
|
||||
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
|
||||
|
||||
caps = openrouter_model_reasoning_capabilities(model)
|
||||
if not caps or not caps.get("supports_reasoning"):
|
||||
|
||||
@@ -3958,7 +3958,7 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
"""Paginated top-level provider keyboard folding provider families (Kimi/Moonshot, MiniMax, xAI…)
|
||||
into one ``mpg:<gid>`` button via the shared ``group_providers`` fold; singles are ``mp:<slug>``."""
|
||||
try:
|
||||
from hermes_cli.models import group_providers
|
||||
from hermes_cli.models_catalog_static import group_providers
|
||||
except Exception:
|
||||
group_providers = None
|
||||
by_slug = {p.get("slug"): p for p in providers}
|
||||
@@ -4117,7 +4117,7 @@ class TelegramAdapter(BasePlatformAdapter):
|
||||
elif data.startswith("mpg:"): # provider group selected: show member providers
|
||||
group_id = data[4:]
|
||||
try:
|
||||
from hermes_cli.models import PROVIDER_GROUPS
|
||||
from hermes_cli.models_catalog_static import PROVIDER_GROUPS
|
||||
_label, _desc, member_slugs = PROVIDER_GROUPS.get(group_id, ("", "", []))
|
||||
except Exception:
|
||||
_label, member_slugs = "", []
|
||||
|
||||
+1
-1
@@ -447,7 +447,7 @@ class AIAgent(
|
||||
if (getattr(self, "lmstudio_load_mode", "explicit") or "explicit").strip().lower() == "jit":
|
||||
logger.debug("LM Studio explicit preload skipped: lmstudio_load_mode=jit")
|
||||
return None
|
||||
from hermes_cli.models import ensure_lmstudio_model_loaded
|
||||
from hermes_cli.models_local import ensure_lmstudio_model_loaded
|
||||
|
||||
if config_context_length is None:
|
||||
config_context_length = getattr(self, "_config_context_length", None)
|
||||
|
||||
@@ -120,7 +120,7 @@ class TestNamedCustomProviderCatalogs:
|
||||
}
|
||||
)
|
||||
with patch("hermes_cli.config.load_config", return_value=cfg), patch(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog",
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog",
|
||||
return_value=True,
|
||||
), patch(
|
||||
"hermes_cli.model_switch_providers._fetch_picker_live_models",
|
||||
@@ -148,7 +148,7 @@ class TestNamedCustomProviderCatalogs:
|
||||
]
|
||||
)
|
||||
with patch("hermes_cli.config.load_config", return_value=cfg), patch(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog",
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog",
|
||||
return_value=True,
|
||||
), patch(
|
||||
"hermes_cli.model_switch_providers._fetch_picker_live_models",
|
||||
@@ -172,7 +172,7 @@ class TestNamedCustomProviderCatalogs:
|
||||
from hermes_cli.model_switch_providers import _NativePickerModelList
|
||||
|
||||
with patch("hermes_cli.config.load_config", return_value=cfg), patch(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog",
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog",
|
||||
return_value=True,
|
||||
), patch(
|
||||
"hermes_cli.model_switch_providers._fetch_picker_live_models",
|
||||
|
||||
@@ -4772,7 +4772,7 @@ class TestFastModelTier:
|
||||
"~openai/gpt-mini-latest": {},
|
||||
"stepfun/step-3.7-flash:free": {},
|
||||
}
|
||||
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
|
||||
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
|
||||
assert ac._fast_model_from_catalog("nous") == "~openai/gpt-mini-latest"
|
||||
|
||||
def test_catalog_match_skips_reasoning_batch_and_embedding_lookalikes(self):
|
||||
@@ -4785,7 +4785,7 @@ class TestFastModelTier:
|
||||
"sentence-transformers/all-minilm-l6-v2": {},
|
||||
"google/gemini-3.6-flash": {},
|
||||
}
|
||||
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
|
||||
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
|
||||
assert ac._fast_model_from_catalog("nous") == "google/gemini-3.6-flash"
|
||||
|
||||
def test_catalog_match_skips_the_non_chat_siblings_of_a_chat_model(self):
|
||||
@@ -4799,7 +4799,7 @@ class TestFastModelTier:
|
||||
"openai/gpt-4o-mini-search-preview": {},
|
||||
"openai/gpt-4o-mini": {},
|
||||
}
|
||||
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
|
||||
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
|
||||
assert ac._fast_model_from_catalog("nous") == "openai/gpt-4o-mini"
|
||||
|
||||
def test_catalog_match_takes_the_newest_of_a_family(self):
|
||||
@@ -4816,7 +4816,7 @@ class TestFastModelTier:
|
||||
"openai/gpt-9-mini": {},
|
||||
"openai/gpt-10-mini": {},
|
||||
}
|
||||
with patch("hermes_cli.models.fetch_models_with_pricing", return_value=catalog):
|
||||
with patch("hermes_cli.models_pricing.fetch_models_with_pricing", return_value=catalog):
|
||||
assert ac._fast_model_from_catalog("nous") == "openai/gpt-10-mini"
|
||||
|
||||
def test_catalog_fetch_is_authenticated(self):
|
||||
@@ -4831,7 +4831,7 @@ class TestFastModelTier:
|
||||
"hermes_cli.auth.resolve_api_key_provider_credentials",
|
||||
return_value={"api_key": "sk-test", "base_url": "https://api.example.com/v1"},
|
||||
), patch(
|
||||
"hermes_cli.models.fetch_models_with_pricing", return_value={}
|
||||
"hermes_cli.models_pricing.fetch_models_with_pricing", return_value={}
|
||||
) as fetch:
|
||||
ac._fast_model_from_catalog("openai")
|
||||
|
||||
|
||||
@@ -317,12 +317,11 @@ class TestIsFreeTierModel:
|
||||
def test_pricing_cache_peek_zero_priced_model(self, monkeypatch):
|
||||
from agent.credits_tracker import is_free_tier_model
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_pricing
|
||||
|
||||
# The picker keys the cache on the pre-/v1 root (get_pricing_for_provider
|
||||
# strips a trailing /v1 before fetch_models_with_pricing).
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"_pricing_cache",
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache",
|
||||
{
|
||||
"https://inference-api.nousresearch.com": {
|
||||
"some/zero-priced": {"prompt": "0", "completion": "0"},
|
||||
@@ -343,12 +342,13 @@ class TestIsFreeTierModel:
|
||||
def test_exception_fails_open_to_false(self, monkeypatch):
|
||||
from agent.credits_tracker import is_free_tier_model
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_pricing
|
||||
|
||||
class _Exploding:
|
||||
def get(self, *_a, **_kw):
|
||||
raise RuntimeError("boom")
|
||||
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache", _Exploding())
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache", _Exploding())
|
||||
assert is_free_tier_model("some/model", "https://inference-api.nousresearch.com") is False
|
||||
|
||||
def test_stealth_prefix_detected_as_free(self):
|
||||
|
||||
@@ -8,6 +8,7 @@ import pytest
|
||||
|
||||
from hermes_cli.auth import AuthError
|
||||
from hermes_cli import main as hermes_main
|
||||
from hermes_cli import model_setup_flows
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -329,7 +330,7 @@ def test_model_flow_nous_does_not_restore_stale_custom_api_key(tmp_path, monkeyp
|
||||
"hermes_cli.models.get_curated_nous_model_ids",
|
||||
lambda: [selected_model],
|
||||
)
|
||||
monkeypatch.setattr("hermes_cli.models.get_pricing_for_provider", lambda provider: {})
|
||||
monkeypatch.setattr("hermes_cli.models_pricing.get_pricing_for_provider", lambda provider: {})
|
||||
monkeypatch.setattr("hermes_cli.models.check_nous_free_tier", lambda **kwargs: False)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.union_with_portal_paid_recommendations",
|
||||
|
||||
@@ -9,12 +9,9 @@ import json
|
||||
from unittest.mock import patch, MagicMock
|
||||
|
||||
from hermes_cli import models as models_module
|
||||
from hermes_cli.models import (
|
||||
VERCEL_AI_GATEWAY_MODELS,
|
||||
_ai_gateway_model_is_free,
|
||||
fetch_ai_gateway_models,
|
||||
fetch_ai_gateway_pricing,
|
||||
)
|
||||
from hermes_cli import models_pricing
|
||||
from hermes_cli.models import VERCEL_AI_GATEWAY_MODELS, _ai_gateway_model_is_free, fetch_ai_gateway_models
|
||||
from hermes_cli.models_pricing import fetch_ai_gateway_pricing
|
||||
|
||||
|
||||
def _mock_urlopen(payload):
|
||||
@@ -29,7 +26,7 @@ def _mock_urlopen(payload):
|
||||
|
||||
def _reset_caches():
|
||||
models_module._ai_gateway_catalog_cache = None
|
||||
models_module._pricing_cache.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
|
||||
|
||||
def test_ai_gateway_pricing_translates_input_output_to_prompt_completion():
|
||||
|
||||
@@ -902,9 +902,10 @@ class TestNovitaProvider:
|
||||
def test_novita_pricing_cache(self, monkeypatch):
|
||||
"""_fetch_novita_pricing should cache results in _pricing_cache."""
|
||||
from hermes_cli import models as models_mod
|
||||
from hermes_cli import models_pricing
|
||||
monkeypatch.setenv("NOVITA_API_KEY", "sk-test-key")
|
||||
monkeypatch.setenv("NOVITA_BASE_URL", "https://api.novita.ai/openai/v1")
|
||||
models_mod._pricing_cache.pop("https://api.novita.ai/openai/v1", None)
|
||||
models_pricing._pricing_cache.pop("https://api.novita.ai/openai/v1", None)
|
||||
|
||||
call_count = {"n": 0}
|
||||
fake_payload = {
|
||||
@@ -937,17 +938,17 @@ class TestNovitaProvider:
|
||||
)
|
||||
|
||||
# First call hits the network.
|
||||
first = models_mod._fetch_novita_pricing()
|
||||
first = models_pricing._fetch_novita_pricing()
|
||||
assert "x/y" in first
|
||||
assert call_count["n"] == 1
|
||||
|
||||
# Second call returns cached result without re-hitting the network.
|
||||
second = models_mod._fetch_novita_pricing()
|
||||
second = models_pricing._fetch_novita_pricing()
|
||||
assert second == first
|
||||
assert call_count["n"] == 1
|
||||
|
||||
# force_refresh bypasses the cache.
|
||||
models_mod._fetch_novita_pricing(force_refresh=True)
|
||||
models_pricing._fetch_novita_pricing(force_refresh=True)
|
||||
assert call_count["n"] == 2
|
||||
|
||||
|
||||
@@ -1004,6 +1005,7 @@ def _deepinfra_cache_isolation(monkeypatch):
|
||||
a later test's fetch within the failure TTL.
|
||||
"""
|
||||
import hermes_cli.models as _models_mod
|
||||
from hermes_cli import models_pricing
|
||||
monkeypatch.setattr(_models_mod, "_deepinfra_catalog_cache", {})
|
||||
monkeypatch.setattr(_models_mod, "_deepinfra_catalog_neg_cache", {})
|
||||
yield
|
||||
@@ -1030,6 +1032,7 @@ class TestFetchDeepInfraModels:
|
||||
]}).encode()
|
||||
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_pricing
|
||||
monkeypatch.setattr(
|
||||
models, "_urlopen_model_catalog_request", lambda *a, **kw: _Resp()
|
||||
)
|
||||
@@ -1047,6 +1050,7 @@ class TestFetchDeepInfraModels:
|
||||
|
||||
def test_catalog_uses_credential_safe_opener(self, monkeypatch):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_pricing
|
||||
|
||||
seen = {}
|
||||
|
||||
@@ -1119,6 +1123,7 @@ class TestDeepInfraTagFiltering:
|
||||
]}
|
||||
from hermes_cli.models import _fetch_deepinfra_models_by_tag
|
||||
import hermes_cli.models as _m
|
||||
from hermes_cli import models_pricing
|
||||
|
||||
for surface in ("chat", "image-gen", "tts", "stt", "embed"):
|
||||
monkeypatch.setattr(
|
||||
@@ -1171,12 +1176,13 @@ class TestDeepInfraPricingFetcher:
|
||||
{"id": "vendor/model-image", "metadata": {"tags": ["image-gen"], "pricing": {"per_image_unit": 0.05}}},
|
||||
]}
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_pricing
|
||||
monkeypatch.setattr(
|
||||
models,
|
||||
"_urlopen_model_catalog_request",
|
||||
_make_urlopen_returning(payload),
|
||||
)
|
||||
from hermes_cli.models import get_pricing_for_provider
|
||||
from hermes_cli.models_pricing import get_pricing_for_provider
|
||||
|
||||
# get_pricing_for_provider → _fetch_deepinfra_pricing dispatch path
|
||||
result = get_pricing_for_provider("deepinfra")
|
||||
|
||||
@@ -500,6 +500,7 @@ class TestLoginNousSkipKeepsCurrent:
|
||||
"""Patch OAuth + model-list + prompt so _login_nous doesn't hit network."""
|
||||
import hermes_cli.auth as auth_mod
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_pricing
|
||||
import hermes_cli.nous_subscription as ns
|
||||
|
||||
fake_auth_state = {
|
||||
@@ -518,7 +519,7 @@ class TestLoginNousSkipKeepsCurrent:
|
||||
auth_mod, "_prompt_model_selection",
|
||||
lambda *a, **kw: prompt_returns,
|
||||
)
|
||||
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda p: {})
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda p: {})
|
||||
free_tier_calls = []
|
||||
|
||||
def _check_nous_free_tier(**kwargs):
|
||||
|
||||
@@ -459,7 +459,7 @@ class TestCustomProviderDiscoverModels:
|
||||
}
|
||||
|
||||
with patch("hermes_cli.models.fetch_api_models") as mock_fetch, \
|
||||
patch("hermes_cli.models.fetch_ollama_local_models") as mock_ollama, \
|
||||
patch("hermes_cli.models_local.fetch_ollama_local_models") as mock_ollama, \
|
||||
patch("hermes_cli.curses_ui.curses_radiolist", side_effect=ImportError), \
|
||||
patch("builtins.input", return_value="1"), \
|
||||
patch("builtins.print"):
|
||||
|
||||
@@ -10,10 +10,11 @@ from time import monotonic
|
||||
|
||||
import hermes_cli.inventory as inv
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_pricing
|
||||
|
||||
|
||||
def _patch_pricing(monkeypatch, *, free_tier, pricing, unavailable=None):
|
||||
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda slug, **kw: pricing.get(slug, {}))
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda slug, **kw: pricing.get(slug, {}))
|
||||
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda *, force_fresh=False: free_tier)
|
||||
monkeypatch.setattr(
|
||||
models_mod, "partition_nous_models_by_tier",
|
||||
@@ -125,7 +126,7 @@ def test_model_options_cold_pricing_fetch_runs_off_the_request_path(monkeypatch)
|
||||
"is_user_defined": False,
|
||||
"source": "built-in",
|
||||
}
|
||||
monkeypatch.setattr(models_mod, "get_pricing_for_provider", fake_pricing)
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", fake_pricing)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.model_switch.list_authenticated_providers",
|
||||
lambda **_kwargs: [row],
|
||||
@@ -160,8 +161,7 @@ def test_model_options_cold_pricing_fetch_runs_off_the_request_path(monkeypatch)
|
||||
|
||||
def test_cold_nous_entitlement_keeps_models_unselectable(monkeypatch):
|
||||
"""A cold nonblocking response must not expose paid models fail-open."""
|
||||
monkeypatch.setattr(
|
||||
models_mod, "get_pricing_for_provider", lambda *_args, **_kwargs: {}
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda *_args, **_kwargs: {}
|
||||
)
|
||||
monkeypatch.setattr(models_mod, "get_cached_nous_free_tier", lambda: None)
|
||||
rows = [{"slug": "nous", "models": ["free/model", "paid/model"]}]
|
||||
@@ -288,12 +288,10 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch):
|
||||
endpoint_b: {"b/model": {"prompt": "3", "completion": "4"}},
|
||||
}
|
||||
monkeypatch.setattr(inv, "_pricing_prewarm_threads", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"_resolve_nous_pricing_credentials",
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
|
||||
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
|
||||
lambda: ("", active_endpoint["value"]),
|
||||
)
|
||||
|
||||
@@ -301,13 +299,13 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch):
|
||||
started[base_url].set()
|
||||
if base_url == endpoint_a:
|
||||
release_a.wait(timeout=5)
|
||||
return models_mod._cache_catalog(base_url, expected[base_url])
|
||||
return models_pricing._cache_catalog(base_url, expected[base_url])
|
||||
|
||||
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", fetch_pricing)
|
||||
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", fetch_pricing)
|
||||
monkeypatch.setattr(
|
||||
inv,
|
||||
"_apply_pricing",
|
||||
lambda _rows: models_mod.get_pricing_for_provider("nous"),
|
||||
lambda _rows: models_pricing.get_pricing_for_provider("nous"),
|
||||
)
|
||||
|
||||
token = set_hermes_home_override(str(tmp_path / "profile"))
|
||||
@@ -335,7 +333,7 @@ def test_prewarm_endpoint_rotation_starts_a_new_worker(tmp_path, monkeypatch):
|
||||
assert started[endpoint_b].wait(timeout=1)
|
||||
threads[1].join(timeout=2)
|
||||
assert not threads[1].is_alive()
|
||||
assert models_mod.get_pricing_for_provider(
|
||||
assert models_pricing.get_pricing_for_provider(
|
||||
"nous", cached_only=True
|
||||
) == expected[endpoint_b]
|
||||
finally:
|
||||
@@ -363,17 +361,13 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey
|
||||
endpoint_b: {"b/model": {"prompt": "3", "completion": "4"}},
|
||||
}
|
||||
monkeypatch.setattr(inv, "_pricing_prewarm_threads", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"get_cached_nous_inference_base_url",
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
|
||||
monkeypatch.setattr(models_pricing, "get_cached_nous_inference_base_url",
|
||||
lambda: active_endpoint["value"],
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"_resolve_nous_pricing_credentials",
|
||||
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
|
||||
lambda: ("", active_endpoint["value"]),
|
||||
)
|
||||
|
||||
@@ -381,13 +375,13 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey
|
||||
started[base_url].set()
|
||||
if base_url == endpoint_a:
|
||||
release_a.wait(timeout=5)
|
||||
return models_mod._cache_catalog(base_url, expected[base_url])
|
||||
return models_pricing._cache_catalog(base_url, expected[base_url])
|
||||
|
||||
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", fetch_pricing)
|
||||
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", fetch_pricing)
|
||||
monkeypatch.setattr(
|
||||
inv,
|
||||
"_apply_pricing",
|
||||
lambda _rows: models_mod.get_pricing_for_provider("nous"),
|
||||
lambda _rows: models_pricing.get_pricing_for_provider("nous"),
|
||||
)
|
||||
|
||||
token = set_hermes_home_override(str(tmp_path / "profile"))
|
||||
@@ -415,7 +409,7 @@ def test_prewarm_nous_rotation_when_another_provider_is_current(tmp_path, monkey
|
||||
assert started[endpoint_b].wait(timeout=1)
|
||||
threads[1].join(timeout=2)
|
||||
assert not threads[1].is_alive()
|
||||
assert models_mod.get_pricing_for_provider(
|
||||
assert models_pricing.get_pricing_for_provider(
|
||||
"nous", cached_only=True
|
||||
) == expected[endpoint_b]
|
||||
finally:
|
||||
@@ -430,16 +424,14 @@ def test_cached_only_pricing_returns_a_warm_value_without_fetching(monkeypatch):
|
||||
"""Cache-only picker reads preserve pricing once the prewarm completes."""
|
||||
cache_key = "https://openrouter.ai/api"
|
||||
expected = {"vendor/model": {"prompt": "0.000001", "completion": "0.000002"}}
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache", {cache_key: expected})
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"fetch_models_with_pricing",
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache", {cache_key: expected})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
|
||||
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing",
|
||||
lambda **_kwargs: (_ for _ in ()).throw(AssertionError("network fetch started")),
|
||||
)
|
||||
|
||||
assert models_mod.get_pricing_for_provider(
|
||||
assert models_pricing.get_pricing_for_provider(
|
||||
"openrouter", cached_only=True
|
||||
) == expected
|
||||
|
||||
@@ -455,30 +447,24 @@ def test_cached_only_dynamic_pricing_is_profile_scoped(tmp_path, monkeypatch):
|
||||
endpoint_b = "https://profile-b.example"
|
||||
expected_a = {"a/model": {"prompt": "1", "completion": "2"}}
|
||||
expected_b = {"b/model": {"prompt": "3", "completion": "4"}}
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"_pricing_cache",
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache",
|
||||
{endpoint_a: expected_a, endpoint_b: expected_b},
|
||||
)
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_provider_cache_keys", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_provider_cache_keys", {})
|
||||
active_endpoint = {"value": endpoint_a}
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"_resolve_nous_pricing_credentials",
|
||||
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
|
||||
lambda: ("", active_endpoint["value"]),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"fetch_models_with_pricing",
|
||||
lambda **kwargs: models_mod._pricing_cache[kwargs["base_url"]],
|
||||
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing",
|
||||
lambda **kwargs: models_pricing._pricing_cache[kwargs["base_url"]],
|
||||
)
|
||||
|
||||
def in_profile(home, endpoint, *, cached_only):
|
||||
token = set_hermes_home_override(str(home))
|
||||
active_endpoint["value"] = endpoint
|
||||
try:
|
||||
return models_mod.get_pricing_for_provider(
|
||||
return models_pricing.get_pricing_for_provider(
|
||||
"nous", cached_only=cached_only
|
||||
)
|
||||
finally:
|
||||
|
||||
@@ -11,15 +11,16 @@ invite a picker filter that hides working levels.
|
||||
|
||||
import hermes_cli.inventory as inv
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_reasoning_caps
|
||||
|
||||
|
||||
def _patch_catalog(monkeypatch, caps_by_model, *, provider="nous"):
|
||||
"""Point the Nous/OpenRouter catalog readers at a fixed capability map."""
|
||||
monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda model: False)
|
||||
monkeypatch.setattr(models_mod, "warm_nous_reasoning_caps_async", lambda: None)
|
||||
monkeypatch.setattr(models_mod, "warm_openrouter_reasoning_caps_async", lambda: None)
|
||||
monkeypatch.setattr(models_reasoning_caps, "warm_nous_reasoning_caps_async", lambda: None)
|
||||
monkeypatch.setattr(models_reasoning_caps, "warm_openrouter_reasoning_caps_async", lambda: None)
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
models_reasoning_caps,
|
||||
f"{provider}_model_reasoning_capabilities",
|
||||
lambda model, **kw: caps_by_model.get(model),
|
||||
)
|
||||
@@ -143,12 +144,12 @@ def test_openrouter_uses_its_own_catalog(monkeypatch):
|
||||
def test_catalog_failure_never_breaks_the_picker(monkeypatch):
|
||||
"""A raising catalog reader degrades to "unknown", not to a broken payload."""
|
||||
monkeypatch.setattr(models_mod, "model_supports_fast_mode", lambda model: False)
|
||||
monkeypatch.setattr(models_mod, "warm_nous_reasoning_caps_async", lambda: None)
|
||||
monkeypatch.setattr(models_reasoning_caps, "warm_nous_reasoning_caps_async", lambda: None)
|
||||
|
||||
def _boom(model, **kw):
|
||||
raise RuntimeError("catalog exploded")
|
||||
|
||||
monkeypatch.setattr(models_mod, "nous_model_reasoning_capabilities", _boom)
|
||||
monkeypatch.setattr(models_reasoning_caps, "nous_model_reasoning_capabilities", _boom)
|
||||
rows = [{"slug": "nous", "models": ["deepseek/deepseek-v4-pro"]}]
|
||||
inv._apply_capabilities(rows)
|
||||
|
||||
|
||||
@@ -159,6 +159,7 @@ def _stub_kimi_discovery(monkeypatch, *, canonical):
|
||||
"""
|
||||
import agent.models_dev as md
|
||||
import hermes_cli.models as hm
|
||||
from hermes_cli import models_catalog_static
|
||||
|
||||
kimi_map = {
|
||||
"kimi": "kimi-for-coding",
|
||||
@@ -188,10 +189,11 @@ def _stub_kimi_discovery(monkeypatch, *, canonical):
|
||||
def test_single_kimi_credential_yields_one_canonical_row(monkeypatch):
|
||||
"""One Kimi key yields a single row under the canonical 'kimi-coding' slug."""
|
||||
import hermes_cli.models as hm
|
||||
from hermes_cli import models_catalog_static
|
||||
|
||||
_stub_kimi_discovery(
|
||||
monkeypatch,
|
||||
canonical=[hm.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc")],
|
||||
canonical=[models_catalog_static.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc")],
|
||||
)
|
||||
monkeypatch.setenv("KIMI_API_KEY", "sk-test-kimi")
|
||||
|
||||
@@ -215,12 +217,13 @@ def test_distinct_kimi_china_credential_still_listed(monkeypatch):
|
||||
pair that share a credential, not legitimately distinct providers.
|
||||
"""
|
||||
import hermes_cli.models as hm
|
||||
from hermes_cli import models_catalog_static
|
||||
|
||||
_stub_kimi_discovery(
|
||||
monkeypatch,
|
||||
canonical=[
|
||||
hm.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc"),
|
||||
hm.ProviderEntry("kimi-coding-cn", "Kimi / Moonshot (China)", "desc"),
|
||||
models_catalog_static.ProviderEntry("kimi-coding", "Kimi / Kimi Coding Plan", "desc"),
|
||||
models_catalog_static.ProviderEntry("kimi-coding-cn", "Kimi / Moonshot (China)", "desc"),
|
||||
],
|
||||
)
|
||||
monkeypatch.setenv("KIMI_API_KEY", "sk-test-kimi")
|
||||
|
||||
@@ -3,6 +3,7 @@ import json
|
||||
import pytest
|
||||
|
||||
from hermes_cli import models
|
||||
from hermes_cli import models_local
|
||||
|
||||
|
||||
MODEL = "publisher/model"
|
||||
@@ -54,12 +55,11 @@ def _capture_load(monkeypatch, response_payload):
|
||||
|
||||
def test_missing_echo_refreshes_loaded_state(monkeypatch):
|
||||
catalogs = iter([_catalog(), _catalog(loaded_context=88_000)])
|
||||
monkeypatch.setattr(
|
||||
models, "_lmstudio_fetch_raw_models", lambda **_kwargs: next(catalogs)
|
||||
monkeypatch.setattr(models_local, "_lmstudio_fetch_raw_models", lambda **_kwargs: next(catalogs)
|
||||
)
|
||||
_capture_load(monkeypatch, {"status": "loaded"})
|
||||
|
||||
result = models.ensure_lmstudio_model_loaded(
|
||||
result = models_local.ensure_lmstudio_model_loaded(
|
||||
MODEL, BASE_URL, api_key="", target_context_length=100_000
|
||||
)
|
||||
|
||||
@@ -67,9 +67,7 @@ def test_missing_echo_refreshes_loaded_state(monkeypatch):
|
||||
|
||||
|
||||
def test_explicit_override_above_known_maximum_rejects_even_when_loaded(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
models,
|
||||
"_lmstudio_fetch_raw_models",
|
||||
monkeypatch.setattr(models_local, "_lmstudio_fetch_raw_models",
|
||||
lambda **_kwargs: _catalog(loaded_context=64_000, maximum=128_000),
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
@@ -78,7 +76,7 @@ def test_explicit_override_above_known_maximum_rejects_even_when_loaded(monkeypa
|
||||
lambda *_args, **_kwargs: pytest.fail("invalid override must not be posted"),
|
||||
)
|
||||
|
||||
result = models.ensure_lmstudio_model_loaded(
|
||||
result = models_local.ensure_lmstudio_model_loaded(
|
||||
MODEL,
|
||||
BASE_URL,
|
||||
api_key="",
|
||||
|
||||
@@ -47,7 +47,7 @@ def _switch_to_alias(monkeypatch, alias_entry):
|
||||
return {"accepted": True, "persist": True, "recognized": True, "message": ""}
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.validate_requested_model", _fake_validate
|
||||
"hermes_cli.models_validate.validate_requested_model", _fake_validate
|
||||
)
|
||||
|
||||
import hermes_cli.model_switch as ms
|
||||
@@ -208,7 +208,7 @@ class TestSessionKeyIsHostScoped:
|
||||
"hermes_cli.runtime_provider.load_config", lambda *a, **k: cfg
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
lambda *a, **k: {
|
||||
"accepted": True,
|
||||
"persist": True,
|
||||
@@ -278,7 +278,7 @@ class TestBuiltinProviderKeysDoNotLeak:
|
||||
return {"accepted": True, "persist": True, "recognized": True, "message": ""}
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.validate_requested_model", _fake_validate
|
||||
"hermes_cli.models_validate.validate_requested_model", _fake_validate
|
||||
)
|
||||
import hermes_cli.model_switch as ms
|
||||
|
||||
@@ -320,7 +320,7 @@ class TestProviderLabelCannotSelectAKeyForAnArbitraryHost:
|
||||
probed["api_key"] = api_key
|
||||
return {"accepted": True, "persist": True, "recognized": True, "message": ""}
|
||||
|
||||
monkeypatch.setattr("hermes_cli.models.validate_requested_model", _fake_validate)
|
||||
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", _fake_validate)
|
||||
import hermes_cli.model_switch as ms
|
||||
|
||||
monkeypatch.setattr(ms, "DIRECT_ALIASES", {})
|
||||
@@ -378,7 +378,7 @@ class TestSessionCredentialIsScopedToTheOrigin:
|
||||
monkeypatch.setattr("hermes_cli.config.load_config", lambda *a, **k: cfg)
|
||||
monkeypatch.setattr("hermes_cli.runtime_provider.load_config", lambda *a, **k: cfg)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
lambda *a, **k: {"accepted": True, "persist": True,
|
||||
"recognized": True, "message": ""},
|
||||
)
|
||||
|
||||
@@ -60,7 +60,7 @@ def _run_switch(
|
||||
with patch("hermes_cli.model_switch.resolve_alias", return_value=None), \
|
||||
patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \
|
||||
patch("hermes_cli.model_switch.normalize_model_for_provider", side_effect=lambda model, provider: model), \
|
||||
patch("hermes_cli.models.validate_requested_model", return_value=validation), \
|
||||
patch("hermes_cli.models_validate.validate_requested_model", return_value=validation), \
|
||||
patch("hermes_cli.models.detect_provider_for_model", return_value=None), \
|
||||
patch("hermes_cli.model_switch.get_model_info", return_value=None), \
|
||||
patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \
|
||||
|
||||
@@ -40,7 +40,7 @@ def _run_copilot_switch(
|
||||
},
|
||||
),
|
||||
patch(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
return_value=_MOCK_VALIDATION,
|
||||
),
|
||||
patch("hermes_cli.model_switch.get_model_info", return_value=None),
|
||||
|
||||
@@ -39,19 +39,19 @@ def _disable_live_custom_provider_model_probe(monkeypatch):
|
||||
"hermes_cli.models.provider_model_ids", lambda *_a, **_kw: []
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.fetch_ollama_local_models", lambda *_a, **_kw: None
|
||||
"hermes_cli.models_local.fetch_ollama_local_models", lambda *_a, **_kw: None
|
||||
)
|
||||
|
||||
|
||||
def test_picker_native_probe_failure_falls_back_to_openai_catalog(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: True
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: True
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models._get_ollama_native_headers", lambda *a, **k: {}
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.fetch_ollama_local_models", lambda *a, **k: None
|
||||
"hermes_cli.models_local.fetch_ollama_local_models", lambda *a, **k: None
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.fetch_api_models", lambda *a, **k: ["fallback-model"]
|
||||
@@ -70,7 +70,7 @@ def test_picker_generic_discovery_preserves_api_mode(monkeypatch):
|
||||
return ["model-a"]
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: False
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: False
|
||||
)
|
||||
monkeypatch.setattr("hermes_cli.models.cached_fetch_api_models", cached)
|
||||
|
||||
@@ -172,7 +172,7 @@ def test_providers_singular_model_does_not_suppress_ollama_native_discovery(monk
|
||||
monkeypatch.setattr("agent.models_dev.fetch_models_dev", lambda: {})
|
||||
monkeypatch.setattr(providers_mod, "HERMES_OVERLAYS", {})
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.fetch_ollama_local_models",
|
||||
"hermes_cli.models_local.fetch_ollama_local_models",
|
||||
lambda *a, **k: ["qwen3:latest", "llama3.2:latest"],
|
||||
)
|
||||
|
||||
@@ -362,7 +362,7 @@ def test_list_authenticated_providers_can_probe_active_bare_custom_endpoint(monk
|
||||
|
||||
def test_switch_model_accepts_explicit_bare_custom_current_endpoint(monkeypatch):
|
||||
"""Picker selections for bare custom endpoints should route to current base_url."""
|
||||
monkeypatch.setattr("hermes_cli.models.validate_requested_model", lambda *a, **k: _MOCK_VALIDATION)
|
||||
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", lambda *a, **k: _MOCK_VALIDATION)
|
||||
monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None)
|
||||
monkeypatch.setattr("hermes_cli.model_switch.get_model_capabilities", lambda *a, **k: None)
|
||||
|
||||
@@ -408,11 +408,11 @@ def test_switch_model_does_not_send_ollama_headers_to_unrelated_custom_endpoint(
|
||||
return _MOCK_VALIDATION
|
||||
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog",
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog",
|
||||
fake_native_detection,
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models._get_ollama_request_headers",
|
||||
"hermes_cli.models_local._get_ollama_request_headers",
|
||||
lambda: {"Authorization": "Bearer configured-ollama-secret"},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
@@ -431,7 +431,7 @@ def test_switch_model_does_not_send_ollama_headers_to_unrelated_custom_endpoint(
|
||||
"api_mode": "chat_completions",
|
||||
},
|
||||
)
|
||||
monkeypatch.setattr("hermes_cli.models.validate_requested_model", fake_validation)
|
||||
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model", fake_validation)
|
||||
monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None)
|
||||
monkeypatch.setattr("hermes_cli.model_switch.get_model_capabilities", lambda *a, **k: None)
|
||||
|
||||
@@ -480,7 +480,7 @@ def test_picker_selection_resolves_named_custom_provider_model_id(monkeypatch):
|
||||
},
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
lambda *a, **k: _MOCK_VALIDATION,
|
||||
)
|
||||
monkeypatch.setattr("hermes_cli.model_switch.get_model_info", lambda *a, **k: None)
|
||||
@@ -980,8 +980,8 @@ def test_picker_endpoint_authorization_overrides_inferred_bearer(monkeypatch):
|
||||
captured.update(headers or {})
|
||||
return ["model-a"]
|
||||
|
||||
monkeypatch.setattr("hermes_cli.models.should_use_ollama_native_catalog", lambda *a, **k: True)
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_ollama_local_models", fake_native)
|
||||
monkeypatch.setattr("hermes_cli.models_local.should_use_ollama_native_catalog", lambda *a, **k: True)
|
||||
monkeypatch.setattr("hermes_cli.models_local.fetch_ollama_local_models", fake_native)
|
||||
result = _fetch_picker_live_models(
|
||||
"endpoint-key",
|
||||
"http://127.0.0.1:11434/v1",
|
||||
@@ -1217,7 +1217,7 @@ def test_lmstudio_picker_probes_active_config_base_url(monkeypatch):
|
||||
captured["api_key"] = api_key
|
||||
return ["qwen/qwen3-coder-30b"]
|
||||
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch)
|
||||
monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch)
|
||||
|
||||
list_authenticated_providers(
|
||||
current_provider="lmstudio",
|
||||
@@ -1244,7 +1244,7 @@ def test_lmstudio_picker_lm_base_url_env_wins_over_active_config(monkeypatch):
|
||||
captured["base_url"] = base_url
|
||||
return []
|
||||
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch)
|
||||
monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch)
|
||||
|
||||
list_authenticated_providers(
|
||||
current_provider="lmstudio",
|
||||
@@ -1270,7 +1270,7 @@ def test_lmstudio_picker_skips_probe_when_not_configured(monkeypatch):
|
||||
captured["base_url"] = base_url
|
||||
return []
|
||||
|
||||
monkeypatch.setattr("hermes_cli.models.fetch_lmstudio_models", _fake_fetch)
|
||||
monkeypatch.setattr("hermes_cli.models_local.fetch_lmstudio_models", _fake_fetch)
|
||||
|
||||
list_authenticated_providers(
|
||||
current_provider="openrouter",
|
||||
|
||||
@@ -49,7 +49,7 @@ def _run_openai_switch(
|
||||
},
|
||||
),
|
||||
patch(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
return_value=_MOCK_VALIDATION,
|
||||
),
|
||||
patch("hermes_cli.model_switch.get_model_info", return_value=None),
|
||||
|
||||
@@ -57,7 +57,7 @@ def _run_opencode_switch(
|
||||
},
|
||||
),
|
||||
patch(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
return_value=_MOCK_VALIDATION,
|
||||
),
|
||||
patch("hermes_cli.model_switch.get_model_info", return_value=None),
|
||||
|
||||
@@ -24,7 +24,7 @@ def _run_switch(raw_input: str, current_provider: str = "openrouter") -> str:
|
||||
patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \
|
||||
patch("hermes_cli.runtime_provider.resolve_runtime_provider",
|
||||
return_value={"api_key": "test", "base_url": "", "api_mode": "chat_completions"}), \
|
||||
patch("hermes_cli.models.validate_requested_model", return_value=_MOCK_VALIDATION), \
|
||||
patch("hermes_cli.models_validate.validate_requested_model", return_value=_MOCK_VALIDATION), \
|
||||
patch("hermes_cli.model_switch.get_model_info", return_value=None), \
|
||||
patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \
|
||||
patch("hermes_cli.models.detect_provider_for_model", return_value=None):
|
||||
|
||||
@@ -3,24 +3,9 @@
|
||||
import pytest
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from hermes_cli.models import (
|
||||
azure_foundry_model_api_mode,
|
||||
copilot_model_api_mode,
|
||||
fetch_github_model_catalog,
|
||||
curated_models_for_provider,
|
||||
fetch_api_models,
|
||||
fetch_lmstudio_models,
|
||||
github_model_reasoning_efforts,
|
||||
normalize_copilot_model_id,
|
||||
normalize_opencode_model_id,
|
||||
normalize_provider,
|
||||
opencode_model_api_mode,
|
||||
parse_model_input,
|
||||
probe_api_models,
|
||||
provider_label,
|
||||
provider_model_ids,
|
||||
validate_requested_model,
|
||||
)
|
||||
from hermes_cli.models import azure_foundry_model_api_mode, copilot_model_api_mode, fetch_github_model_catalog, curated_models_for_provider, fetch_api_models, github_model_reasoning_efforts, normalize_copilot_model_id, normalize_opencode_model_id, normalize_provider, opencode_model_api_mode, parse_model_input, probe_api_models, provider_label, provider_model_ids
|
||||
from hermes_cli.models_local import fetch_lmstudio_models
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
|
||||
# -- helpers -----------------------------------------------------------------
|
||||
|
||||
@@ -14,6 +14,8 @@ from hermes_cli.models import (
|
||||
union_with_portal_paid_recommendations,
|
||||
)
|
||||
import hermes_cli.models as _models_mod
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
LIVE_OPENROUTER_MODELS = [
|
||||
("anthropic/claude-opus-4.6", "recommended"),
|
||||
@@ -448,7 +450,7 @@ class TestCodexSoftAcceptPlausibilityGate:
|
||||
and mislabel the provider as 'OpenAI Codex')."""
|
||||
|
||||
def test_unrelated_name_rejected_on_openai_codex(self):
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
r = validate_requested_model("qwen3.5-4b", "openai-codex")
|
||||
assert r["accepted"] is False
|
||||
assert r["persist"] is False
|
||||
@@ -456,7 +458,7 @@ class TestCodexSoftAcceptPlausibilityGate:
|
||||
|
||||
|
||||
def test_real_catalog_model_unaffected(self):
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
r = validate_requested_model("gpt-5.5", "openai-codex")
|
||||
assert r["accepted"] is True
|
||||
assert r["recognized"] is True
|
||||
@@ -474,24 +476,24 @@ class TestFormatPricePerMtok:
|
||||
"""_format_price_per_mtok: sub-cent prices must not collapse to 'free'/'$0.00'."""
|
||||
|
||||
def test_standard_prices_keep_two_decimals(self):
|
||||
from hermes_cli.models import _format_price_per_mtok
|
||||
from hermes_cli.models_pricing import _format_price_per_mtok
|
||||
assert _format_price_per_mtok("0.000003") == "$3.00"
|
||||
assert _format_price_per_mtok("0.00003") == "$30.00"
|
||||
assert _format_price_per_mtok("0.00000015") == "$0.15"
|
||||
assert _format_price_per_mtok("0.00018") == "$180.00"
|
||||
|
||||
def test_zero_is_free(self):
|
||||
from hermes_cli.models import _format_price_per_mtok
|
||||
from hermes_cli.models_pricing import _format_price_per_mtok
|
||||
assert _format_price_per_mtok("0") == "free"
|
||||
assert _format_price_per_mtok("0.0") == "free"
|
||||
|
||||
def test_invalid_is_question_mark(self):
|
||||
from hermes_cli.models import _format_price_per_mtok
|
||||
from hermes_cli.models_pricing import _format_price_per_mtok
|
||||
assert _format_price_per_mtok("garbage") == "?"
|
||||
assert _format_price_per_mtok(None) == "?"
|
||||
|
||||
def test_sub_cent_price_extends_precision(self):
|
||||
from hermes_cli.models import _format_price_per_mtok
|
||||
from hermes_cli.models_pricing import _format_price_per_mtok
|
||||
# DeepSeek V4 Flash 0731 promo cache-hit rate: $0.0018/Mtok.
|
||||
assert _format_price_per_mtok("0.0000000018") == "$0.0018"
|
||||
assert _format_price_per_mtok("0.000000001") == "$0.001"
|
||||
@@ -501,7 +503,7 @@ class TestFormatPricePerMtok:
|
||||
assert _format_price_per_mtok("0.00000000001") == "$0.00001"
|
||||
|
||||
def test_one_cent_boundary_stays_two_decimals(self):
|
||||
from hermes_cli.models import _format_price_per_mtok
|
||||
from hermes_cli.models_pricing import _format_price_per_mtok
|
||||
assert _format_price_per_mtok("0.00000001") == "$0.01"
|
||||
|
||||
|
||||
@@ -590,7 +592,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
server.shutdown()
|
||||
|
||||
def test_native_tags_cache_expires(self, monkeypatch):
|
||||
from hermes_cli.models import fetch_ollama_local_models
|
||||
from hermes_cli.models_local import fetch_ollama_local_models
|
||||
|
||||
server, port = _start_fake_ollama_server(models=[{"name": "old-model"}])
|
||||
try:
|
||||
@@ -612,7 +614,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_fetch_ollama_models_accepts_base_url_without_scheme(self):
|
||||
"""OLLAMA_HOST commonly omits http://; discovery should normalize it."""
|
||||
from hermes_cli.models import fetch_ollama_local_models
|
||||
from hermes_cli.models_local import fetch_ollama_local_models
|
||||
|
||||
server, port = _start_fake_ollama_server(models=[{"name": "qwen2.5:1.5b"}])
|
||||
try:
|
||||
@@ -622,7 +624,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_fetch_ollama_models_accepts_full_models_url(self):
|
||||
"""Pasted OpenAI-style /v1/models URLs should normalize to the native root."""
|
||||
from hermes_cli.models import fetch_ollama_local_models
|
||||
from hermes_cli.models_local import fetch_ollama_local_models
|
||||
|
||||
server, port = _start_fake_ollama_server(models=[{"name": "qwen2.5:1.5b"}])
|
||||
try:
|
||||
@@ -636,10 +638,11 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_runtime_error_from_config_load_does_not_escape_ollama_helpers(self):
|
||||
"""Managed-mode config failures should degrade to defaults, not crash pickers."""
|
||||
from hermes_cli.models import _get_ollama_base_url, should_use_ollama_native_catalog
|
||||
from hermes_cli.models import _get_ollama_base_url
|
||||
from hermes_cli.models_local import should_use_ollama_native_catalog
|
||||
|
||||
with patch("hermes_cli.config.load_config", side_effect=RuntimeError("bad home")), patch(
|
||||
"hermes_cli.models.probe_ollama_local_models",
|
||||
"hermes_cli.models_local.probe_ollama_local_models",
|
||||
return_value=None,
|
||||
):
|
||||
assert _get_ollama_base_url() == "http://localhost:11434"
|
||||
@@ -647,22 +650,22 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_probe_ollama_models_malformed_base_url_returns_none(self):
|
||||
"""Malformed user-configured URLs should behave like probe failures, not crashes."""
|
||||
from hermes_cli.models import probe_ollama_local_models
|
||||
from hermes_cli.models_local import probe_ollama_local_models
|
||||
|
||||
assert probe_ollama_local_models("http://127.0.0.1:bad-port/v1") is None
|
||||
|
||||
def test_fetch_ollama_models_preserves_probe_failure(self):
|
||||
from hermes_cli.models import fetch_ollama_local_models
|
||||
from hermes_cli.models_local import fetch_ollama_local_models
|
||||
|
||||
with patch("hermes_cli.models.probe_ollama_local_models", return_value=None):
|
||||
with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=None):
|
||||
assert fetch_ollama_local_models("http://127.0.0.1:11434") is None
|
||||
|
||||
def test_ollama_port_detection_requires_working_api_tags(self):
|
||||
from hermes_cli.models import should_use_ollama_native_catalog
|
||||
from hermes_cli.models_local import should_use_ollama_native_catalog
|
||||
|
||||
with patch("hermes_cli.models.probe_ollama_local_models", return_value=["qwen3:1.7b"]):
|
||||
with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=["qwen3:1.7b"]):
|
||||
assert should_use_ollama_native_catalog("custom", "192.168.1.5:11434/v1") is True
|
||||
with patch("hermes_cli.models.probe_ollama_local_models", return_value=None):
|
||||
with patch("hermes_cli.models_local.probe_ollama_local_models", return_value=None):
|
||||
assert should_use_ollama_native_catalog("custom", "192.168.1.5:11434/v1") is False
|
||||
|
||||
def test_provider_model_ids_ollama_cloud_config_uses_generic_catalog(self):
|
||||
@@ -678,7 +681,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
}
|
||||
}
|
||||
},
|
||||
), patch("hermes_cli.models.fetch_ollama_local_models") as fetch_local, patch(
|
||||
), patch("hermes_cli.models_local.fetch_ollama_local_models") as fetch_local, patch(
|
||||
"hermes_cli.models.fetch_api_models",
|
||||
return_value=["qwen3:1.7b"],
|
||||
) as fetch_generic:
|
||||
@@ -691,7 +694,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
)
|
||||
|
||||
def test_native_ollama_catalog_uses_configured_key_env(self, monkeypatch):
|
||||
from hermes_cli.models import _get_ollama_request_headers
|
||||
from hermes_cli.models_local import _get_ollama_request_headers
|
||||
|
||||
monkeypatch.setenv("TEST_OLLAMA_API_KEY", "env-key")
|
||||
with patch(
|
||||
@@ -710,7 +713,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
}
|
||||
|
||||
def test_native_ollama_catalog_uses_api_key_env_alias(self, monkeypatch):
|
||||
from hermes_cli.models import _get_ollama_request_headers
|
||||
from hermes_cli.models_local import _get_ollama_request_headers
|
||||
|
||||
monkeypatch.setenv("TEST_OLLAMA_API_KEY_ALIAS", "alias-key")
|
||||
with patch(
|
||||
@@ -740,7 +743,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
}
|
||||
},
|
||||
), patch(
|
||||
"hermes_cli.models.fetch_ollama_local_models",
|
||||
"hermes_cli.models_local.fetch_ollama_local_models",
|
||||
return_value=["qwen3:1.7b"],
|
||||
) as fetch_local:
|
||||
assert provider_model_ids("ollama", force_refresh=True) == ["qwen3:1.7b"]
|
||||
@@ -757,7 +760,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
"base_url": "http://127.0.0.1:11434/v1",
|
||||
}
|
||||
},
|
||||
), patch("hermes_cli.models.probe_ollama_local_models") as probe_ollama:
|
||||
), patch("hermes_cli.models_local.probe_ollama_local_models") as probe_ollama:
|
||||
assert _credential_fingerprint("ollama")
|
||||
probe_ollama.assert_not_called()
|
||||
|
||||
@@ -801,6 +804,8 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_clear_provider_models_cache_clears_ollama_native_tags_cache(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
cache = getattr(models, "_OLLAMA_LOCAL_MODELS_CACHE")
|
||||
cache["http://127.0.0.1:11434"] = ("old-model",)
|
||||
@@ -809,6 +814,8 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_clear_provider_models_cache_custom_clears_native_tags_cache(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
cache = getattr(models, "_OLLAMA_LOCAL_MODELS_CACHE")
|
||||
cache["http://127.0.0.1:11434"] = ("old-model",)
|
||||
@@ -817,6 +824,8 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_clear_provider_models_cache_does_not_remove_custom_disk_cache(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
disk_cache = {
|
||||
"custom": {"models": ["custom-model"]},
|
||||
@@ -830,13 +839,13 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
|
||||
def test_ollama_cloud_urls_do_not_use_native_local_catalog(self):
|
||||
from hermes_cli.models import should_use_ollama_native_catalog
|
||||
from hermes_cli.models_local import should_use_ollama_native_catalog
|
||||
|
||||
assert should_use_ollama_native_catalog("ollama-cloud", "https://ollama.com/v1") is False
|
||||
assert should_use_ollama_native_catalog("ollama", "https://ollama.com/v1") is False
|
||||
|
||||
def test_non_ollama_custom_endpoint_uses_generic_catalog_path(self):
|
||||
from hermes_cli.models import should_use_ollama_native_catalog
|
||||
from hermes_cli.models_local import should_use_ollama_native_catalog
|
||||
|
||||
assert should_use_ollama_native_catalog("custom", "https://example.test/v1") is False
|
||||
assert should_use_ollama_native_catalog("openrouter", "http://localhost:11434/v1") is False
|
||||
@@ -888,10 +897,10 @@ class TestLocalOllamaModelDiscovery:
|
||||
from hermes_cli.model_switch import list_authenticated_providers
|
||||
|
||||
with patch("hermes_cli.config.load_config", return_value={"providers": {}}), patch(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog",
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog",
|
||||
return_value=True,
|
||||
), patch(
|
||||
"hermes_cli.models.fetch_ollama_local_models",
|
||||
"hermes_cli.models_local.fetch_ollama_local_models",
|
||||
return_value=["qwen3:1.7b"],
|
||||
), patch("hermes_cli.models.fetch_api_models", return_value=[]) as fetch_api:
|
||||
rows = list_authenticated_providers(
|
||||
@@ -1090,7 +1099,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
from hermes_cli.model_switch import list_authenticated_providers
|
||||
|
||||
with patch("hermes_cli.config.load_config", return_value={"providers": {}}), patch(
|
||||
"hermes_cli.models.probe_ollama_local_models",
|
||||
"hermes_cli.models_local.probe_ollama_local_models",
|
||||
return_value=["qwen3:1.7b"],
|
||||
), patch("hermes_cli.models.fetch_api_models", return_value=[]) as fetch_api:
|
||||
rows = list_authenticated_providers(
|
||||
@@ -1105,7 +1114,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_model_validation_uses_ollama_api_tags_for_ollama_provider(self):
|
||||
"""`/model` validation for provider=ollama should not probe `/models`."""
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
server, port = _start_fake_ollama_server()
|
||||
try:
|
||||
@@ -1128,12 +1137,12 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_model_validation_ollama_cloud_config_does_not_use_local_tags(self):
|
||||
"""provider=ollama with a cloud base URL should not fall into local /api/tags."""
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
with patch(
|
||||
"hermes_cli.config.load_config",
|
||||
return_value={"providers": {"ollama": {"base_url": "https://ollama.com/v1"}}},
|
||||
), patch("hermes_cli.models.probe_ollama_local_models") as probe_ollama, patch(
|
||||
), patch("hermes_cli.models_local.probe_ollama_local_models") as probe_ollama, patch(
|
||||
"hermes_cli.models.probe_api_models",
|
||||
return_value={
|
||||
"models": ["qwen3:1.7b"],
|
||||
@@ -1152,7 +1161,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_model_validation_uses_ollama_api_tags_for_matching_custom_endpoint(self):
|
||||
"""Current-provider `custom` on the configured Ollama URL should use `/api/tags`."""
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
server, port = _start_fake_ollama_server()
|
||||
base_url = f"http://127.0.0.1:{port}"
|
||||
@@ -1178,7 +1187,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_model_validation_empty_ollama_tags_does_not_fall_back_to_models(self):
|
||||
"""Reachable but empty /api/tags should not produce a misleading /models warning."""
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
server, port = _start_fake_ollama_server(models=[])
|
||||
base_url = f"http://127.0.0.1:{port}"
|
||||
@@ -1247,7 +1256,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
"api_mode": "chat_completions",
|
||||
},
|
||||
), patch(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
return_value={
|
||||
"accepted": True,
|
||||
"persist": True,
|
||||
@@ -1277,7 +1286,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
from hermes_cli.model_switch import switch_model
|
||||
|
||||
with patch(
|
||||
"hermes_cli.models.should_use_ollama_native_catalog",
|
||||
"hermes_cli.models_local.should_use_ollama_native_catalog",
|
||||
side_effect=RuntimeError("config unavailable"),
|
||||
), patch(
|
||||
"hermes_cli.runtime_provider.resolve_runtime_provider",
|
||||
@@ -1287,7 +1296,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
"api_mode": "chat_completions",
|
||||
},
|
||||
), patch(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
return_value={
|
||||
"accepted": True,
|
||||
"persist": True,
|
||||
@@ -1313,7 +1322,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
assert result.api_key == "new-key"
|
||||
|
||||
def test_ollama_root_matching_is_case_insensitive_for_hostnames(self):
|
||||
from hermes_cli.models import _same_ollama_native_root
|
||||
from hermes_cli.models_local import _same_ollama_native_root
|
||||
|
||||
assert _same_ollama_native_root(
|
||||
"HTTP://OLLAMA.EXAMPLE:11434/v1",
|
||||
@@ -1339,6 +1348,8 @@ class TestLocalOllamaModelDiscovery:
|
||||
|
||||
def test_ollama_failed_probe_is_cached_briefly(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
models._OLLAMA_LOCAL_MODELS_CACHE.clear()
|
||||
models._OLLAMA_LOCAL_PROBE_FAILURE_CACHE.clear()
|
||||
@@ -1346,12 +1357,14 @@ class TestLocalOllamaModelDiscovery:
|
||||
"hermes_cli.models._urlopen_model_catalog_request",
|
||||
side_effect=OSError("offline"),
|
||||
) as request:
|
||||
assert models.probe_ollama_local_models("http://127.0.0.1:19999") is None
|
||||
assert models.probe_ollama_local_models("http://127.0.0.1:19999") is None
|
||||
assert models_local.probe_ollama_local_models("http://127.0.0.1:19999") is None
|
||||
assert models_local.probe_ollama_local_models("http://127.0.0.1:19999") is None
|
||||
request.assert_called_once()
|
||||
|
||||
def test_empty_ollama_catalog_does_not_resurrect_stale_disk_models(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
base_url = "http://127.0.0.1:11434"
|
||||
probe_key = models._ollama_probe_cache_key(base_url, None)
|
||||
@@ -1365,13 +1378,15 @@ class TestLocalOllamaModelDiscovery:
|
||||
models, "_credential_fingerprint", return_value="same"
|
||||
), patch.object(models, "provider_model_ids", return_value=[]), patch.object(
|
||||
models, "_get_ollama_base_url", return_value=base_url
|
||||
), patch.object(models, "_get_ollama_request_headers", return_value={}):
|
||||
), patch.object(models_local, "_get_ollama_request_headers", return_value={}):
|
||||
assert models.cached_provider_model_ids("ollama") == []
|
||||
finally:
|
||||
models._OLLAMA_LOCAL_PROBE_REACHABLE.pop(probe_key, None)
|
||||
|
||||
def test_failed_ollama_catalog_preserves_stale_disk_models(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
base_url = "http://127.0.0.1:11434"
|
||||
probe_key = models._ollama_probe_cache_key(base_url, None)
|
||||
@@ -1385,13 +1400,15 @@ class TestLocalOllamaModelDiscovery:
|
||||
models, "_credential_fingerprint", return_value="same"
|
||||
), patch.object(models, "provider_model_ids", return_value=[]), patch.object(
|
||||
models, "_get_ollama_base_url", return_value=base_url
|
||||
), patch.object(models, "_get_ollama_request_headers", return_value={}):
|
||||
), patch.object(models_local, "_get_ollama_request_headers", return_value={}):
|
||||
assert models.cached_provider_model_ids("ollama") == ["stale:model"]
|
||||
finally:
|
||||
models._OLLAMA_LOCAL_PROBE_REACHABLE.pop(probe_key, None)
|
||||
|
||||
def test_ollama_native_request_uses_redirect_safe_catalog_helper(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
response = MagicMock()
|
||||
response.read.return_value = b'{"models": [{"name": "qwen3:1.7b"}]}'
|
||||
@@ -1399,13 +1416,15 @@ class TestLocalOllamaModelDiscovery:
|
||||
with patch.object(
|
||||
models, "_urlopen_model_catalog_request", return_value=response
|
||||
) as request:
|
||||
assert models.fetch_ollama_local_models("http://127.0.0.1:11434") == [
|
||||
assert models_local.fetch_ollama_local_models("http://127.0.0.1:11434") == [
|
||||
"qwen3:1.7b"
|
||||
]
|
||||
request.assert_called_once()
|
||||
|
||||
def test_validation_with_nonmatching_ollama_root_does_not_forward_config_headers(self):
|
||||
import hermes_cli.models as models
|
||||
from hermes_cli import models_local
|
||||
from hermes_cli import models_validate
|
||||
|
||||
with patch(
|
||||
"hermes_cli.config.load_config",
|
||||
@@ -1417,16 +1436,15 @@ class TestLocalOllamaModelDiscovery:
|
||||
}
|
||||
}
|
||||
},
|
||||
), patch.object(models, "should_use_ollama_native_catalog", return_value=True), patch.object(
|
||||
models, "probe_ollama_local_models", return_value=[]
|
||||
), patch.object(models_local, "should_use_ollama_native_catalog", return_value=True), patch.object(models_local, "probe_ollama_local_models", return_value=[]
|
||||
) as probe:
|
||||
models.validate_requested_model(
|
||||
models_validate.validate_requested_model(
|
||||
"qwen3:1.7b",
|
||||
"ollama",
|
||||
base_url="https://other.internal/v1",
|
||||
)
|
||||
assert probe.call_args.kwargs["headers"] == {}
|
||||
models.validate_requested_model(
|
||||
models_validate.validate_requested_model(
|
||||
"qwen3:1.7b",
|
||||
"ollama",
|
||||
base_url="https://other.internal/v1",
|
||||
@@ -1450,7 +1468,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
"hermes_cli.models._get_provider_config_dict",
|
||||
return_value={"base_url": base_url, "api_key": "secret"},
|
||||
) as config_provider, patch(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
return_value={"accepted": True, "persist": True, "recognized": True, "message": ""},
|
||||
):
|
||||
result = model_switch.switch_model(
|
||||
@@ -1480,7 +1498,7 @@ class TestLocalOllamaModelDiscovery:
|
||||
)
|
||||
try:
|
||||
with patch.object(model_switch, "get_model_info", return_value=None), patch(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
return_value={"accepted": True, "persist": True, "recognized": True, "message": ""},
|
||||
):
|
||||
result = model_switch.switch_model(
|
||||
|
||||
@@ -12,12 +12,9 @@ import json
|
||||
import pytest
|
||||
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_pricing
|
||||
import hermes_cli.nous_account as account_mod
|
||||
from hermes_cli.models import (
|
||||
_NOUS_POLICY_APPEND_MAX,
|
||||
nous_policy_allowed_ids,
|
||||
restrict_to_nous_policy,
|
||||
)
|
||||
from hermes_cli.models_pricing import _NOUS_POLICY_APPEND_MAX, nous_policy_allowed_ids, restrict_to_nous_policy
|
||||
from hermes_cli.nous_account import nous_policy_present
|
||||
|
||||
|
||||
@@ -67,20 +64,18 @@ class TestRestrictToNousPolicy:
|
||||
class TestNousPolicyAllowedIds:
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_cache(self):
|
||||
models_mod._pricing_cache.clear()
|
||||
models_mod._pricing_cache_retry_after.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
models_pricing._pricing_cache_retry_after.clear()
|
||||
yield
|
||||
models_mod._pricing_cache.clear()
|
||||
models_mod._pricing_cache_retry_after.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
models_pricing._pricing_cache_retry_after.clear()
|
||||
|
||||
def _patch(self, monkeypatch, *, policy_present, api_key="sk-test", pricing=None):
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
account_mod, "nous_policy_present", lambda: policy_present
|
||||
)
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
"_resolve_nous_pricing_credentials",
|
||||
monkeypatch.setattr(models_pricing, "_resolve_nous_pricing_credentials",
|
||||
lambda: (api_key, "https://inference.example.com"),
|
||||
)
|
||||
|
||||
@@ -88,7 +83,7 @@ class TestNousPolicyAllowedIds:
|
||||
calls.append(kwargs)
|
||||
return pricing if pricing is not None else {}
|
||||
|
||||
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
|
||||
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch)
|
||||
return calls
|
||||
|
||||
def test_returns_the_authenticated_catalog_keys(self, monkeypatch):
|
||||
|
||||
@@ -11,7 +11,7 @@ import argparse
|
||||
import pytest
|
||||
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import model_switch_providers
|
||||
from hermes_cli import models_pricing
|
||||
|
||||
CURATED = ["vendor/allowed", "vendor/blocked"]
|
||||
ALLOWED = {"vendor/allowed"}
|
||||
@@ -20,14 +20,14 @@ ALLOWED = {"vendor/allowed"}
|
||||
@pytest.fixture
|
||||
def policy(monkeypatch):
|
||||
"""An org whose policy admits only ``vendor/allowed``."""
|
||||
monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: ALLOWED)
|
||||
monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: ALLOWED)
|
||||
return ALLOWED
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def no_policy(monkeypatch):
|
||||
"""An unrestricted org — lists must come through untouched."""
|
||||
monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: None)
|
||||
monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: None)
|
||||
|
||||
|
||||
class TestLoginNous:
|
||||
@@ -51,7 +51,7 @@ class TestLoginNous:
|
||||
},
|
||||
)
|
||||
monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED))
|
||||
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {})
|
||||
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
@@ -95,7 +95,7 @@ class TestModelSwitchPicker:
|
||||
lambda *a, **k: {"providers": {"nous": {"access_token": "tok"}}},
|
||||
)
|
||||
monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED))
|
||||
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {})
|
||||
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
@@ -122,7 +122,7 @@ class TestModelSwitchPicker:
|
||||
def _boom(_p):
|
||||
raise RuntimeError("portal down")
|
||||
|
||||
monkeypatch.setattr(models_mod, "get_pricing_for_provider", _boom)
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", _boom)
|
||||
row = self._rows(monkeypatch)
|
||||
assert row is not None
|
||||
assert "vendor/blocked" not in row["models"]
|
||||
@@ -141,7 +141,7 @@ class TestRecommendedDefaultEndpoint:
|
||||
models_mod, "get_curated_nous_model_ids",
|
||||
lambda: ["vendor/blocked", "vendor/allowed"],
|
||||
)
|
||||
monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
|
||||
monkeypatch.setattr(models_pricing, "get_pricing_for_provider", lambda _p: {})
|
||||
monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
|
||||
monkeypatch.setattr(
|
||||
models_mod,
|
||||
@@ -171,10 +171,10 @@ class TestAuxiliaryFastModel:
|
||||
return {mid: {} for mid in catalog}
|
||||
|
||||
monkeypatch.setattr(
|
||||
models_mod, "_resolve_nous_pricing_credentials",
|
||||
models_pricing, "_resolve_nous_pricing_credentials",
|
||||
lambda: ("sk-nous", "https://inference.example.com"),
|
||||
)
|
||||
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
|
||||
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch)
|
||||
picked = aux._fast_model_from_catalog("nous")
|
||||
return picked, seen
|
||||
|
||||
@@ -187,7 +187,7 @@ class TestAuxiliaryFastModel:
|
||||
import agent.auxiliary_client as aux
|
||||
|
||||
monkeypatch.setattr(
|
||||
models_mod, "nous_policy_allowed_ids", lambda **_k: {"vendor/allowed"}
|
||||
models_pricing, "nous_policy_allowed_ids", lambda **_k: {"vendor/allowed"}
|
||||
)
|
||||
monkeypatch.setattr(aux, "_FAST_MODEL_FAMILIES", ("vendor/",))
|
||||
monkeypatch.setattr(aux, "_FAST_MODEL_EXCLUDE", ())
|
||||
@@ -240,14 +240,14 @@ class TestAuxFallbackRespectsPolicy:
|
||||
import agent.auxiliary_client as aux
|
||||
import providers
|
||||
|
||||
monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: allowed)
|
||||
monkeypatch.setattr(models_pricing, "nous_policy_allowed_ids", lambda **_k: allowed)
|
||||
monkeypatch.setattr(
|
||||
models_mod, "_resolve_nous_pricing_credentials",
|
||||
models_pricing, "_resolve_nous_pricing_credentials",
|
||||
lambda: ("sk", "https://inference.example.com"),
|
||||
)
|
||||
# No fast-family match, so the catalog step yields nothing.
|
||||
monkeypatch.setattr(
|
||||
models_mod, "fetch_models_with_pricing",
|
||||
models_pricing, "fetch_models_with_pricing",
|
||||
lambda **_k: {"vendor/allowed-large": {}},
|
||||
)
|
||||
|
||||
@@ -294,7 +294,7 @@ def test_titling_seeds_the_shared_catalog_entry_like_the_pickers(monkeypatch):
|
||||
import agent.auxiliary_client as aux
|
||||
|
||||
monkeypatch.setattr(
|
||||
models_mod, "_resolve_nous_pricing_credentials",
|
||||
models_pricing, "_resolve_nous_pricing_credentials",
|
||||
lambda: ("tok", "https://inference.example.com"),
|
||||
)
|
||||
seen: dict = {}
|
||||
@@ -303,8 +303,8 @@ def test_titling_seeds_the_shared_catalog_entry_like_the_pickers(monkeypatch):
|
||||
seen.update(kwargs)
|
||||
return {"vendor/haiku": {}}
|
||||
|
||||
monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
|
||||
monkeypatch.setattr(models_pricing, "fetch_models_with_pricing", _fake_fetch)
|
||||
aux._fast_model_from_catalog("nous")
|
||||
|
||||
assert seen.get("include_sale_original") is True
|
||||
assert seen.get("cache_ttl_seconds") == models_mod._NOUS_CATALOG_TTL_SECONDS
|
||||
assert seen.get("cache_ttl_seconds") == models_pricing._NOUS_CATALOG_TTL_SECONDS
|
||||
|
||||
@@ -38,18 +38,19 @@ _CATALOG = (
|
||||
def cold_cache(monkeypatch):
|
||||
"""A freshly started process that has never mirrored a catalog to disk."""
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_reasoning_caps
|
||||
|
||||
monkeypatch.setattr(models_mod, "_nous_reasoning_caps_cache", None)
|
||||
monkeypatch.setattr(models_mod, "_nous_reasoning_caps_failed_at", None)
|
||||
monkeypatch.setattr(models_mod, "_nous_caps_disk_checked", False)
|
||||
monkeypatch.setattr(models_mod, "_nous_caps_warm_started", False)
|
||||
models_mod._reasoning_caps_disk_path().unlink(missing_ok=True)
|
||||
models_reasoning_caps._reasoning_caps_disk_path().unlink(missing_ok=True)
|
||||
return models_mod
|
||||
|
||||
|
||||
class TestNousModelReasoningCapabilities:
|
||||
def test_fetch_parses_mandatory_flag(self, cold_cache, monkeypatch):
|
||||
from hermes_cli.models import nous_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
|
||||
|
||||
monkeypatch.setattr(
|
||||
cold_cache, "_urlopen_model_catalog_request",
|
||||
@@ -67,7 +68,7 @@ class TestNousModelReasoningCapabilities:
|
||||
|
||||
def test_catalog_read_sends_user_agent(self, cold_cache, monkeypatch):
|
||||
"""The Portal 403s an anonymous catalog read."""
|
||||
from hermes_cli.models import nous_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
|
||||
|
||||
seen = []
|
||||
|
||||
@@ -89,7 +90,7 @@ class TestNousModelReasoningCapabilities:
|
||||
Pinned to production, a staging profile would take its
|
||||
reasoning-mandatory verdicts from a different deployment's catalog.
|
||||
"""
|
||||
from hermes_cli.models import nous_catalog_url
|
||||
from hermes_cli.models_reasoning_caps import nous_catalog_url
|
||||
|
||||
monkeypatch.setenv(
|
||||
"NOUS_INFERENCE_BASE_URL", "https://staging.nousresearch.com/v1"
|
||||
@@ -100,7 +101,7 @@ class TestNousModelReasoningCapabilities:
|
||||
assert nous_catalog_url().endswith("/v1/models")
|
||||
|
||||
def test_unlisted_and_empty_models_return_none(self, cold_cache, monkeypatch):
|
||||
from hermes_cli.models import nous_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
|
||||
|
||||
monkeypatch.setattr(
|
||||
cold_cache, "_urlopen_model_catalog_request",
|
||||
@@ -111,7 +112,7 @@ class TestNousModelReasoningCapabilities:
|
||||
assert nous_model_reasoning_capabilities(None) is None
|
||||
|
||||
def test_cache_only_by_default_never_fetches(self, cold_cache, monkeypatch):
|
||||
from hermes_cli.models import nous_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
|
||||
|
||||
def _boom(req, *, timeout):
|
||||
raise AssertionError("hot path must not fetch")
|
||||
@@ -120,7 +121,7 @@ class TestNousModelReasoningCapabilities:
|
||||
assert nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None
|
||||
|
||||
def test_unreachable_catalog_rate_limits_refetch(self, cold_cache, monkeypatch):
|
||||
from hermes_cli.models import nous_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities
|
||||
|
||||
calls = {"n": 0}
|
||||
|
||||
@@ -136,10 +137,7 @@ class TestNousModelReasoningCapabilities:
|
||||
|
||||
def test_openrouter_cache_is_independent(self, cold_cache, monkeypatch):
|
||||
"""Two catalogs, two caches — a Portal fetch must not answer for OpenRouter."""
|
||||
from hermes_cli.models import (
|
||||
nous_model_reasoning_capabilities,
|
||||
openrouter_model_reasoning_capabilities,
|
||||
)
|
||||
from hermes_cli.models_reasoning_caps import nous_model_reasoning_capabilities, openrouter_model_reasoning_capabilities
|
||||
|
||||
monkeypatch.setattr(cold_cache, "_openrouter_reasoning_caps_cache", None)
|
||||
monkeypatch.setattr(cold_cache, "_openrouter_reasoning_caps_failed_at", None)
|
||||
|
||||
@@ -386,7 +386,7 @@ class TestSwitchModelDirectAliasOverride:
|
||||
lambda **kwargs: {"api_key": "", "base_url": "", "api_mode": "openai_compat", "provider": "custom"},
|
||||
)
|
||||
|
||||
monkeypatch.setattr("hermes_cli.models.validate_requested_model",
|
||||
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model",
|
||||
lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None})
|
||||
monkeypatch.setattr("hermes_cli.models.opencode_model_api_mode",
|
||||
lambda *a, **kw: "openai_compat")
|
||||
@@ -411,7 +411,7 @@ class TestSwitchModelDirectAliasOverride:
|
||||
"hermes_cli.runtime_provider.resolve_runtime_provider",
|
||||
lambda **kwargs: {"api_key": "", "base_url": "", "api_mode": "openai_compat", "provider": "custom"},
|
||||
)
|
||||
monkeypatch.setattr("hermes_cli.models.validate_requested_model",
|
||||
monkeypatch.setattr("hermes_cli.models_validate.validate_requested_model",
|
||||
lambda *a, **kw: {"accepted": True, "persist": True, "recognized": True, "message": None})
|
||||
monkeypatch.setattr("hermes_cli.models.opencode_model_api_mode",
|
||||
lambda *a, **kw: "openai_compat")
|
||||
|
||||
@@ -340,7 +340,7 @@ class TestOllamaCloudSuffixStripping:
|
||||
|
||||
def test_strip_suffix_helper(self):
|
||||
"""Unit test for the _strip_ollama_cloud_suffix helper."""
|
||||
from hermes_cli.models import _strip_ollama_cloud_suffix
|
||||
from hermes_cli.models_local import _strip_ollama_cloud_suffix
|
||||
|
||||
assert _strip_ollama_cloud_suffix("kimi-k2.6:cloud") == "kimi-k2.6"
|
||||
assert _strip_ollama_cloud_suffix("glm-5.1:cloud") == "glm-5.1"
|
||||
|
||||
@@ -18,7 +18,7 @@ it.
|
||||
from unittest.mock import patch
|
||||
|
||||
from hermes_cli.model_switch import switch_model
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
|
||||
def test_openai_codex_unknown_but_plausible_model_is_accepted_with_warning():
|
||||
|
||||
@@ -21,11 +21,12 @@ from unittest.mock import patch as mock_patch
|
||||
import pytest
|
||||
|
||||
from hermes_cli import models as models_mod
|
||||
from hermes_cli import models_validate
|
||||
|
||||
|
||||
def _validate(requested: str, base_url: str, live: list[str], provider: str = "openai-api"):
|
||||
with mock_patch.object(models_mod, "fetch_api_models", return_value=live):
|
||||
return models_mod.validate_requested_model(
|
||||
return models_validate.validate_requested_model(
|
||||
requested,
|
||||
provider=provider,
|
||||
api_key="sk-test",
|
||||
|
||||
@@ -14,7 +14,7 @@ These tests cover the catalog-fallback path: when ``fetch_api_models`` returns
|
||||
|
||||
from unittest.mock import patch
|
||||
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
|
||||
_UNREACHABLE_PROBE = {
|
||||
|
||||
@@ -9,7 +9,7 @@ from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
|
||||
@@ -10,10 +10,8 @@ Covers:
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli.models import (
|
||||
clamp_reasoning_effort_to_supported,
|
||||
parse_openrouter_reasoning_capabilities,
|
||||
)
|
||||
from hermes_cli.models import clamp_reasoning_effort_to_supported
|
||||
from hermes_cli.models_reasoning_caps import parse_openrouter_reasoning_capabilities
|
||||
|
||||
|
||||
class TestParseReasoningCapabilities:
|
||||
@@ -121,7 +119,7 @@ class TestOpenRouterModelReasoningCapabilities:
|
||||
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None)
|
||||
|
||||
def test_known_model(self, monkeypatch):
|
||||
from hermes_cli.models import openrouter_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
|
||||
self._prime_cache(monkeypatch, {
|
||||
"nvidia/nemotron-3-ultra": {
|
||||
"supports_reasoning": True,
|
||||
@@ -133,19 +131,19 @@ class TestOpenRouterModelReasoningCapabilities:
|
||||
assert caps["supports_reasoning"] is True
|
||||
|
||||
def test_unlisted_model_returns_none(self, monkeypatch):
|
||||
from hermes_cli.models import openrouter_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
|
||||
self._prime_cache(monkeypatch, {"a/b": {"supports_reasoning": True}})
|
||||
assert openrouter_model_reasoning_capabilities("private/custom") is None
|
||||
|
||||
def test_empty_model_returns_none(self, monkeypatch):
|
||||
from hermes_cli.models import openrouter_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
|
||||
self._prime_cache(monkeypatch, {"a/b": {"supports_reasoning": True}})
|
||||
assert openrouter_model_reasoning_capabilities("") is None
|
||||
assert openrouter_model_reasoning_capabilities(None) is None
|
||||
|
||||
def test_catalog_unreachable_returns_none_and_rate_limits(self, monkeypatch):
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli.models import openrouter_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
|
||||
|
||||
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_cache", None)
|
||||
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None)
|
||||
@@ -163,7 +161,7 @@ class TestOpenRouterModelReasoningCapabilities:
|
||||
|
||||
def test_cache_only_by_default_never_fetches(self, monkeypatch):
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli.models import openrouter_model_reasoning_capabilities
|
||||
from hermes_cli.models_reasoning_caps import openrouter_model_reasoning_capabilities
|
||||
|
||||
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_cache", None)
|
||||
monkeypatch.setattr(models_mod, "_openrouter_reasoning_caps_failed_at", None)
|
||||
|
||||
@@ -12,7 +12,8 @@ from unittest.mock import MagicMock
|
||||
import pytest
|
||||
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli.models import fetch_models_with_pricing, peek_cached_pricing
|
||||
from hermes_cli import models_pricing
|
||||
from hermes_cli.models_pricing import fetch_models_with_pricing, peek_cached_pricing
|
||||
|
||||
BASE = "https://inference-api.example.com"
|
||||
|
||||
@@ -23,11 +24,11 @@ _FILTERED = ["vendor/allowed"]
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _clear_pricing_cache():
|
||||
models_mod._pricing_cache.clear()
|
||||
models_mod._pricing_cache_retry_after.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
models_pricing._pricing_cache_retry_after.clear()
|
||||
yield
|
||||
models_mod._pricing_cache.clear()
|
||||
models_mod._pricing_cache_retry_after.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
models_pricing._pricing_cache_retry_after.clear()
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
@@ -96,7 +97,7 @@ def test_one_token_does_not_receive_another_tokens_catalog(per_org_catalog):
|
||||
|
||||
def test_credential_value_does_not_appear_in_the_cache_key():
|
||||
"""Guards against keying on the raw token."""
|
||||
assert "sk-super-secret" not in models_mod._pricing_auth_fingerprint("sk-super-secret")
|
||||
assert "sk-super-secret" not in models_pricing._pricing_auth_fingerprint("sk-super-secret")
|
||||
|
||||
|
||||
def test_anonymous_and_authenticated_reads_are_separate(catalog):
|
||||
@@ -159,7 +160,7 @@ class TestNousCatalogExpiry:
|
||||
a long-lived process holds the entry."""
|
||||
|
||||
def test_entry_expires_so_a_policy_change_is_picked_up(self, catalog, monkeypatch):
|
||||
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
|
||||
from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS
|
||||
|
||||
fetch_models_with_pricing(
|
||||
api_key="sk-test", base_url=BASE,
|
||||
@@ -196,7 +197,7 @@ class TestNousCatalogExpiry:
|
||||
|
||||
def test_peek_skips_an_expired_entry(self, catalog, monkeypatch):
|
||||
"""Reading _pricing_cache directly walked straight past the TTL."""
|
||||
from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
|
||||
from hermes_cli.models_pricing import _NOUS_CATALOG_TTL_SECONDS
|
||||
|
||||
fetch_models_with_pricing(
|
||||
api_key="sk-test", base_url=BASE,
|
||||
|
||||
@@ -6,12 +6,8 @@ These are invariant tests, not catalog snapshots: they assert how
|
||||
vendors, which is expected to change over time.
|
||||
"""
|
||||
|
||||
from hermes_cli.models import (
|
||||
CANONICAL_PROVIDERS,
|
||||
PROVIDER_GROUPS,
|
||||
group_providers,
|
||||
provider_group_for_slug,
|
||||
)
|
||||
from hermes_cli.models import CANONICAL_PROVIDERS
|
||||
from hermes_cli.models_catalog_static import PROVIDER_GROUPS, group_providers, provider_group_for_slug
|
||||
|
||||
|
||||
def _slugs(rows):
|
||||
|
||||
@@ -17,6 +17,8 @@ import json
|
||||
import pytest
|
||||
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli import models_pricing
|
||||
from hermes_cli import models_reasoning_caps
|
||||
|
||||
|
||||
_CATALOG = json.dumps({
|
||||
@@ -95,16 +97,16 @@ def test_fetched_catalog_answers_a_later_process_offline(
|
||||
models_mod, "_urlopen_model_catalog_request",
|
||||
lambda req, *, timeout: _response(_CATALOG),
|
||||
)
|
||||
assert models_mod.nous_model_reasoning_capabilities(
|
||||
assert models_reasoning_caps.nous_model_reasoning_capabilities(
|
||||
"deepseek/deepseek-v4-pro", allow_fetch=True
|
||||
) is not None
|
||||
|
||||
cold_process()
|
||||
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
|
||||
|
||||
caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
assert caps["mandatory"] is False
|
||||
mandatory = models_mod.nous_model_reasoning_capabilities(
|
||||
mandatory = models_reasoning_caps.nous_model_reasoning_capabilities(
|
||||
"arcee-ai/trinity-large-thinking"
|
||||
)
|
||||
assert mandatory["mandatory"] is True
|
||||
@@ -121,14 +123,14 @@ def test_mirror_is_keyed_by_catalog_url(cold_process, offline, monkeypatch):
|
||||
models_mod, "_urlopen_model_catalog_request",
|
||||
lambda req, *, timeout: _response(_CATALOG),
|
||||
)
|
||||
models_mod.nous_model_reasoning_capabilities(
|
||||
models_reasoning_caps.nous_model_reasoning_capabilities(
|
||||
"deepseek/deepseek-v4-pro", allow_fetch=True
|
||||
)
|
||||
|
||||
cold_process()
|
||||
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
|
||||
|
||||
assert models_mod.openrouter_model_reasoning_capabilities(
|
||||
assert models_reasoning_caps.openrouter_model_reasoning_capabilities(
|
||||
"deepseek/deepseek-v4-pro"
|
||||
) is None
|
||||
|
||||
@@ -140,7 +142,7 @@ def test_staging_portal_does_not_read_productions_mirror(
|
||||
models_mod, "_urlopen_model_catalog_request",
|
||||
lambda req, *, timeout: _response(_CATALOG),
|
||||
)
|
||||
models_mod.nous_model_reasoning_capabilities(
|
||||
models_reasoning_caps.nous_model_reasoning_capabilities(
|
||||
"deepseek/deepseek-v4-pro", allow_fetch=True
|
||||
)
|
||||
|
||||
@@ -148,7 +150,7 @@ def test_staging_portal_does_not_read_productions_mirror(
|
||||
monkeypatch.setenv("NOUS_INFERENCE_BASE_URL", "https://staging.nousresearch.com")
|
||||
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
|
||||
|
||||
assert models_mod.nous_model_reasoning_capabilities(
|
||||
assert models_reasoning_caps.nous_model_reasoning_capabilities(
|
||||
"deepseek/deepseek-v4-pro"
|
||||
) is None
|
||||
|
||||
@@ -159,29 +161,29 @@ def test_stale_copy_is_still_served(cold_process, offline, monkeypatch):
|
||||
Refusing to read an aged mirror would put every long-idle install back on
|
||||
the cold-start fallback it exists to prevent.
|
||||
"""
|
||||
url = models_mod.nous_catalog_url()
|
||||
models_mod._save_reasoning_caps_disk(
|
||||
url = models_reasoning_caps.nous_catalog_url()
|
||||
models_reasoning_caps._save_reasoning_caps_disk(
|
||||
url, {"deepseek/deepseek-v4-pro": {"supports_reasoning": True, "mandatory": False}}
|
||||
)
|
||||
raw = json.loads(models_mod._reasoning_caps_disk_path().read_text())
|
||||
raw = json.loads(models_reasoning_caps._reasoning_caps_disk_path().read_text())
|
||||
raw[url]["ts"] = 0 # epoch — far past any TTL
|
||||
models_mod._reasoning_caps_disk_path().write_text(json.dumps(raw))
|
||||
models_reasoning_caps._reasoning_caps_disk_path().write_text(json.dumps(raw))
|
||||
|
||||
cold_process()
|
||||
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
|
||||
|
||||
caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
assert caps["mandatory"] is False
|
||||
|
||||
|
||||
def test_unreadable_mirror_degrades_to_unknown(cold_process, offline, monkeypatch):
|
||||
"""A corrupt file answers "unknown", never raises into the request path."""
|
||||
path = models_mod._reasoning_caps_disk_path()
|
||||
path = models_reasoning_caps._reasoning_caps_disk_path()
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text("{ this is not json")
|
||||
|
||||
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
|
||||
assert models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None
|
||||
assert models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro") is None
|
||||
|
||||
|
||||
def test_pricing_fetch_seeds_the_mirror(cold_process, offline, monkeypatch):
|
||||
@@ -190,20 +192,20 @@ def test_pricing_fetch_seeds_the_mirror(cold_process, offline, monkeypatch):
|
||||
Every surface that renders prices goes through here, so the common case
|
||||
never pays a second round-trip to learn the same thing.
|
||||
"""
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache", {})
|
||||
monkeypatch.setattr(models_mod, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache", {})
|
||||
monkeypatch.setattr(models_pricing, "_pricing_cache_retry_after", {})
|
||||
monkeypatch.setattr(
|
||||
models_mod, "_urlopen_model_catalog_request",
|
||||
lambda req, *, timeout: _response(_CATALOG),
|
||||
)
|
||||
models_mod.fetch_models_with_pricing(
|
||||
models_pricing.fetch_models_with_pricing(
|
||||
base_url="https://inference-api.nousresearch.com"
|
||||
)
|
||||
|
||||
cold_process()
|
||||
monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", offline)
|
||||
|
||||
caps = models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
caps = models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
assert caps is not None
|
||||
|
||||
|
||||
@@ -215,8 +217,8 @@ def test_a_missing_mirror_is_looked_for_once_per_process(cold_process, monkeypat
|
||||
token. Both halves have to be paid at most once.
|
||||
"""
|
||||
counts = {"url": 0, "read": 0}
|
||||
real_url = models_mod.nous_catalog_url
|
||||
real_read = models_mod._read_reasoning_caps_disk
|
||||
real_url = models_reasoning_caps.nous_catalog_url
|
||||
real_read = models_reasoning_caps._read_reasoning_caps_disk
|
||||
|
||||
def _counting_url():
|
||||
counts["url"] += 1
|
||||
@@ -226,9 +228,9 @@ def test_a_missing_mirror_is_looked_for_once_per_process(cold_process, monkeypat
|
||||
counts["read"] += 1
|
||||
return real_read()
|
||||
|
||||
monkeypatch.setattr(models_mod, "nous_catalog_url", _counting_url)
|
||||
monkeypatch.setattr(models_mod, "_read_reasoning_caps_disk", _counting_read)
|
||||
monkeypatch.setattr(models_reasoning_caps, "nous_catalog_url", _counting_url)
|
||||
monkeypatch.setattr(models_reasoning_caps, "_read_reasoning_caps_disk", _counting_read)
|
||||
for _ in range(5):
|
||||
models_mod.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
models_reasoning_caps.nous_model_reasoning_capabilities("deepseek/deepseek-v4-pro")
|
||||
|
||||
assert counts == {"url": 1, "read": 1}
|
||||
|
||||
@@ -6,10 +6,8 @@ import json
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import hermes_cli.models as models_mod
|
||||
from hermes_cli.models import (
|
||||
compute_sale_discount,
|
||||
fetch_models_with_pricing,
|
||||
)
|
||||
from hermes_cli import models_pricing
|
||||
from hermes_cli.models_pricing import compute_sale_discount, fetch_models_with_pricing
|
||||
|
||||
|
||||
def test_free_model_gets_flat_100_percent_discount():
|
||||
@@ -31,7 +29,7 @@ def test_paid_model_without_original_shows_no_sale():
|
||||
|
||||
|
||||
def test_fetch_models_with_pricing_copies_nested_original(monkeypatch):
|
||||
models_mod._pricing_cache.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
payload = {
|
||||
"data": [
|
||||
{
|
||||
@@ -100,7 +98,7 @@ def test_resolve_nous_pricing_credentials_honors_inference_env_override(monkeypa
|
||||
"hermes_cli.auth.resolve_nous_runtime_credentials",
|
||||
lambda: None,
|
||||
)
|
||||
api_key, base_url = models_mod._resolve_nous_pricing_credentials()
|
||||
api_key, base_url = models_pricing._resolve_nous_pricing_credentials()
|
||||
assert api_key == ""
|
||||
# The bare origin, whichever form the override was written in: callers
|
||||
# append their own path (``/v1/models``), so a suffix here would double up.
|
||||
@@ -119,7 +117,7 @@ def test_resolve_nous_pricing_credentials_normalizes_either_suffix(monkeypatch):
|
||||
"https://stg-inference-api.nousresearch.com/v1/",
|
||||
):
|
||||
monkeypatch.setenv("NOUS_INFERENCE_BASE_URL", override)
|
||||
assert models_mod._resolve_nous_pricing_credentials()[1] == (
|
||||
assert models_pricing._resolve_nous_pricing_credentials()[1] == (
|
||||
"https://stg-inference-api.nousresearch.com"
|
||||
)
|
||||
|
||||
@@ -131,8 +129,8 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch):
|
||||
call, but it expires — the processes that read this run for weeks, and
|
||||
every caller silently falls back to a curated list meanwhile.
|
||||
"""
|
||||
models_mod._pricing_cache.clear()
|
||||
models_mod._pricing_cache_retry_after.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
models_pricing._pricing_cache_retry_after.clear()
|
||||
|
||||
calls = []
|
||||
|
||||
@@ -151,7 +149,7 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
models_mod.time,
|
||||
"monotonic",
|
||||
lambda: now + models_mod._FAILED_CATALOG_TTL_SECONDS + 1,
|
||||
lambda: now + models_pricing._FAILED_CATALOG_TTL_SECONDS + 1,
|
||||
)
|
||||
assert fetch_models_with_pricing(base_url="https://example.test") == {}
|
||||
assert len(calls) == 2
|
||||
@@ -159,8 +157,8 @@ def test_a_failed_catalog_fetch_is_not_cached_forever(monkeypatch):
|
||||
|
||||
def test_a_successful_catalog_fetch_stays_cached(monkeypatch):
|
||||
"""Only the failures expire; a real catalog is still fetched once."""
|
||||
models_mod._pricing_cache.clear()
|
||||
models_mod._pricing_cache_retry_after.clear()
|
||||
models_pricing._pricing_cache.clear()
|
||||
models_pricing._pricing_cache_retry_after.clear()
|
||||
|
||||
calls = []
|
||||
body = json.dumps(
|
||||
@@ -182,7 +180,7 @@ def test_a_successful_catalog_fetch_stays_cached(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
models_mod.time,
|
||||
"monotonic",
|
||||
lambda: now + models_mod._FAILED_CATALOG_TTL_SECONDS + 1,
|
||||
lambda: now + models_pricing._FAILED_CATALOG_TTL_SECONDS + 1,
|
||||
)
|
||||
assert "a/b" in fetch_models_with_pricing(base_url="https://example.test")
|
||||
assert len(calls) == 1
|
||||
|
||||
@@ -72,7 +72,7 @@ def test_show_status_reports_empty_lmstudio_listing_as_reachable(monkeypatch, ca
|
||||
monkeypatch.setattr(status_mod, "resolve_provider", lambda requested=None, **kwargs: "lmstudio", raising=False)
|
||||
monkeypatch.setattr(status_mod, "provider_label", lambda provider: "LM Studio", raising=False)
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.probe_lmstudio_models",
|
||||
"hermes_cli.models_local.probe_lmstudio_models",
|
||||
lambda api_key=None, base_url=None, timeout=5.0: [],
|
||||
)
|
||||
|
||||
|
||||
@@ -378,7 +378,7 @@ def test_switch_model_resolves_user_provider_credentials(monkeypatch, tmp_path):
|
||||
|
||||
# Mock validation to pass
|
||||
monkeypatch.setattr(
|
||||
"hermes_cli.models.validate_requested_model",
|
||||
"hermes_cli.models_validate.validate_requested_model",
|
||||
lambda *a, **k: {"accepted": True, "persist": True, "recognized": True, "message": None}
|
||||
)
|
||||
|
||||
@@ -443,7 +443,7 @@ def _run_user_provider_override_case(
|
||||
with patch("hermes_cli.model_switch.resolve_alias", return_value=None), \
|
||||
patch("hermes_cli.model_switch.list_provider_models", return_value=[]), \
|
||||
patch("hermes_cli.model_switch.normalize_model_for_provider", side_effect=lambda model, provider: model), \
|
||||
patch("hermes_cli.models.validate_requested_model", return_value=_REJECTED_VALIDATION), \
|
||||
patch("hermes_cli.models_validate.validate_requested_model", return_value=_REJECTED_VALIDATION), \
|
||||
patch("hermes_cli.models.detect_provider_for_model", return_value=None), \
|
||||
patch("hermes_cli.model_switch.get_model_info", return_value=None), \
|
||||
patch("hermes_cli.model_switch.get_model_capabilities", return_value=None), \
|
||||
|
||||
@@ -199,7 +199,7 @@ class TestOllamaModelSupportsThinking:
|
||||
monkeypatch.setattr(httpx, "Client", _Client)
|
||||
|
||||
def test_thinking_capability_true(self, monkeypatch):
|
||||
from hermes_cli.models import ollama_model_supports_thinking
|
||||
from hermes_cli.models_local import ollama_model_supports_thinking
|
||||
|
||||
self._patch_show(monkeypatch, capabilities=["completion", "tools", "thinking"])
|
||||
assert (
|
||||
@@ -211,7 +211,7 @@ class TestOllamaModelSupportsThinking:
|
||||
|
||||
|
||||
def test_probe_failure_returns_none(self, monkeypatch):
|
||||
from hermes_cli.models import ollama_model_supports_thinking
|
||||
from hermes_cli.models_local import ollama_model_supports_thinking
|
||||
|
||||
self._patch_show(monkeypatch, status=404)
|
||||
assert (
|
||||
@@ -219,7 +219,7 @@ class TestOllamaModelSupportsThinking:
|
||||
)
|
||||
|
||||
def test_exception_returns_none(self, monkeypatch):
|
||||
from hermes_cli.models import ollama_model_supports_thinking
|
||||
from hermes_cli.models_local import ollama_model_supports_thinking
|
||||
|
||||
self._patch_show(monkeypatch, raise_exc=RuntimeError("boom"))
|
||||
assert (
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
from types import SimpleNamespace
|
||||
from typing import Any, cast
|
||||
|
||||
from hermes_cli.models import LMStudioLoadResult
|
||||
from hermes_cli.models_local import LMStudioLoadResult
|
||||
from run_agent import AIAgent
|
||||
|
||||
|
||||
@@ -25,7 +25,7 @@ def test_lmstudio_jit_load_mode_skips_explicit_preload(monkeypatch):
|
||||
calls.append((args, kwargs))
|
||||
return LMStudioLoadResult(64_000)
|
||||
|
||||
monkeypatch.setattr("hermes_cli.models.ensure_lmstudio_model_loaded", fake_ensure)
|
||||
monkeypatch.setattr("hermes_cli.models_local.ensure_lmstudio_model_loaded", fake_ensure)
|
||||
|
||||
result = AIAgent._ensure_lmstudio_runtime_loaded(cast(Any, _agent("jit")))
|
||||
|
||||
|
||||
@@ -4,7 +4,7 @@ from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli.models import LMStudioLoadResult
|
||||
from hermes_cli.models_local import LMStudioLoadResult
|
||||
from run_agent import AIAgent
|
||||
from hermes_cli.route_identity import normalize_route_base_url
|
||||
from agent.context_compressor import ContextCompressor
|
||||
|
||||
@@ -8,7 +8,7 @@ from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from hermes_cli.models import validate_requested_model
|
||||
from hermes_cli.models_validate import validate_requested_model
|
||||
|
||||
|
||||
class TestMiniMaxModelValidation:
|
||||
|
||||
Reference in New Issue
Block a user