From 0fd89c0ae56e310d76b171bbfb434ca94a3a04d8 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 16:10:32 -0700 Subject: [PATCH] wip(model_switch): snapshot of in-progress provider-listing extraction + status split (worker died mid-run) --- hermes_cli/model_switch.py | 2521 ++++++-------------------- hermes_cli/model_switch_providers.py | 1504 +++++++++++++++ hermes_cli/status.py | 467 +++-- 3 files changed, 2323 insertions(+), 2169 deletions(-) create mode 100644 hermes_cli/model_switch_providers.py diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py index 0c18e800f5..4f331445ec 100644 --- a/hermes_cli/model_switch.py +++ b/hermes_cli/model_switch.py @@ -20,18 +20,15 @@ OpenRouter variant suffixes (``:free``, ``:extended``, ``:fast``). from __future__ import annotations -import http.client import logging import os import re -import time from dataclasses import dataclass, field -from typing import Any, List, NamedTuple, Optional +from typing import Any, NamedTuple, Optional from hermes_cli.providers import ( ProviderDef, custom_provider_aliases, - custom_provider_slug, determine_api_mode, get_label, host_mandated_api_mode, @@ -48,12 +45,51 @@ from agent.models_dev import ( get_model_info, list_provider_models, ) -from utils import base_url_host_matches, base_url_hostname, base_url_origin +from utils import base_url_hostname, base_url_origin +from hermes_cli.model_switch_providers import ( # noqa: F401 (re-exported; tests patch hermes_cli.model_switch.) + _MODEL_DISCOVERY_ERRORS, + _NativePickerModelList, + _PARALLEL_PREFETCH_WORKERS, + _PickerBuild, + _UNCAPPED_PICKER_PROVIDERS, + _auth_store_has_provider, + _aws_live_or_curated_ids, + _build_curated_lists, + _cap_models, + _collect_authed_provider_slugs, + _credential_identity, + _credential_pool_is_usable, + _discover_endpoint_models, + _discover_flag, + _display_prefix, + _entry_api_mode, + _entry_base_url, + _fetch_picker_live_models, + _has_aws_sdk_creds_for_listing, + _has_fast_aws_sdk_signal, + _is_aws_sdk, + _iter_builtin_candidates, + _lap_bare_custom_row, + _lap_builtin_rows, + _lap_canonical_rows, + _lap_custom_provider_rows, + _lap_overlay_rows, + _lap_user_provider_rows, + _live_or_curated_ids, + _norm_url, + _nous_picker_model_ids, + _overlay_has_env_creds, + _picker_prewarm_done, + _pool_usable, + _prefetch_provider_models_parallel, + _prepend_moa_picker_provider, + _raw_pool_usable, + _save_discovered_models_to_config, + list_authenticated_providers, + list_picker_providers, + prewarm_picker_cache_async, +) -# Providers whose picker model list should NOT be capped by max_models. -# OpenCode Zen / Go are aggregators whose full catalogs (70+ models each) must -# be visible so users can pick any model they have access to. -_UNCAPPED_PICKER_PROVIDERS: frozenset[str] = frozenset({"opencode-zen", "opencode-go"}) logger = logging.getLogger(__name__) @@ -163,94 +199,6 @@ def _models_config_is_allowlist(value: Any, discovered: bool = False) -> bool: return False -def _save_discovered_models_to_config( - api_url: str, - model_ids: list[str], - *, - api_mode: Optional[str] = None, - headers: Optional[dict[str, str]] = None, -) -> None: - """Persist discovered models into ``custom_providers`` in config.yaml. - - Called after a successful ``/v1/models`` probe so that the next read - with ``discover_models: false`` uses the cached list instead of a stale - or minimal manually-configured subset. - - Matches entries by ``base_url`` (trailing-slash-normalised). A failed - config write is swallowed — the picker still shows the live models for - this session. - """ - if not api_url or not model_ids: - return - try: - from hermes_cli.config import load_config, save_config - - cfg = load_config() - providers = cfg.get("custom_providers") or [] - if not isinstance(providers, list): - return - - norm_url = api_url.strip().rstrip("/").lower() - changed = False - for entry in providers: - if not isinstance(entry, dict): - continue - entry_url = (entry.get("base_url", "") or entry.get("url", "")).strip() - if entry_url.rstrip("/").lower() != norm_url: - continue - entry_mode = str( - entry.get("api_mode") or entry.get("transport") or "" - ).strip().lower() or None - if entry_mode != api_mode: - continue - if headers is not None: - entry_headers = _extra_headers_from_config(entry) - if entry_headers != headers: - continue - existing = entry.get("models") - legacy_discovered = ( - isinstance(existing, dict) - and existing.get("__discovered_model_catalog__") is True - ) - entry_discovered = ( - entry.get("models_discovered") is True or legacy_discovered - ) - # Preserve per-model metadata: when ``models`` is a mapping - # (e.g. ``{"model-a": {"context_length": 8192}}``) or a list of - # dicts (e.g. ``[{"id": "model-a", "context_length": 8192}]``), - # the user has curated metadata per model — do not replace it. - # A mapping Hermes itself discovered (``models_discovered: true`` - # or the legacy in-mapping sentinel) is ours to refresh. - if isinstance(existing, dict) and not entry_discovered: - continue - if isinstance(existing, list) and any( - isinstance(m, dict) for m in existing - ): - continue - # Only update when models are stale — avoids unnecessary - # config writes on every picker open. A legacy-shape entry - # (sentinel inside ``models``) is always rewritten so the next - # save migrates it to the clean entry-level flag. - if isinstance(existing, list) and existing == model_ids: - continue - if ( - isinstance(existing, dict) - and entry_discovered - and not legacy_discovered - and list(existing) == model_ids - ): - continue - entry["models"] = {model_id: {} for model_id in model_ids} - entry["models_discovered"] = True - changed = True - - if changed: - cfg["custom_providers"] = providers - save_config(cfg) - except Exception: - pass - - def _bare_custom_provider_def(current_base_url: str) -> Optional[ProviderDef]: """ProviderDef for a direct ``model.provider: custom`` endpoint.""" base_url = str(current_base_url or "").strip() @@ -268,89 +216,6 @@ def _bare_custom_provider_def(current_base_url: str) -> Optional[ProviderDef]: ) -_MODEL_DISCOVERY_ERRORS = ( - ImportError, - OSError, - RuntimeError, - TimeoutError, - TypeError, - ValueError, - http.client.HTTPException, -) - - -class _NativePickerModelList(list[str]): - """A successful native catalog, including an authoritative empty one.""" - - -def _fetch_picker_live_models( - api_key: str, - api_url: str, - native_catalog_provider: str, - preserve_native_models: bool, - headers: dict[str, str] | None = None, - timeout: float = 5.0, - api_mode: str | None = None, -) -> list[str] | None: - """Fetch picker models with native Ollama and cached generic discovery.""" - from hermes_cli.models import ( - _get_ollama_native_headers, - _normalize_openai_base_url, - cached_fetch_api_models, - fetch_ollama_local_models, - should_use_ollama_native_catalog, - ) - - candidate_headers = _get_ollama_native_headers(api_url, api_key=api_key) - caller_has_authorization = any( - key.lower() == "authorization" for key in (headers or {}) - ) - if caller_has_authorization: - for key in tuple(candidate_headers): - if key.lower() == "authorization": - del candidate_headers[key] - if headers: - for key in tuple(candidate_headers): - if any(key.lower() == existing.lower() for existing in headers): - del candidate_headers[key] - candidate_headers.update(headers) - if api_key and not caller_has_authorization: - for key in tuple(candidate_headers): - if key.lower() == "authorization": - del candidate_headers[key] - candidate_headers["Authorization"] = f"Bearer {api_key}" - use_native = should_use_ollama_native_catalog( - native_catalog_provider, api_url, headers=candidate_headers or None - ) - resolved_headers = candidate_headers or None if use_native else headers - - if use_native: - if preserve_native_models: - return None - native_models = fetch_ollama_local_models( - api_url, timeout=timeout, headers=resolved_headers - ) - if native_models is not None: - return _NativePickerModelList(native_models) - # A failed native probe is not authoritative: retry the cached generic - # OpenAI-compatible catalog before reporting no models. - return cached_fetch_api_models( - api_key, - _normalize_openai_base_url(api_url), - timeout=timeout, - headers=resolved_headers, - api_mode=api_mode, - ) - generic_models = cached_fetch_api_models( - api_key, - api_url, - timeout=timeout, - headers=resolved_headers, - api_mode=api_mode, - ) - return generic_models if generic_models else None - - # --------------------------------------------------------------------------- # Non-agentic model warning # --------------------------------------------------------------------------- @@ -917,6 +782,7 @@ class ModelFlagParseResult: # Flag parsing # --------------------------------------------------------------------------- + def parse_model_flags_detailed(raw_args: str) -> ModelFlagParseResult: """Parse flags from /model command args. @@ -1702,9 +1568,9 @@ def _switch_fail(is_global: bool, message: str, **fields) -> ModelSwitchResult: return ModelSwitchResult(success=False, is_global=is_global, error_message=message, **fields) -def _runtime_creds(fallback_headers: dict, **kwargs) -> tuple[str, str, str, dict, dict]: +def _runtime_creds(fallback_headers: dict, **kwargs) -> tuple[str, str, str, dict]: """``resolve_runtime_provider`` unpacked as ``(api_key, base_url, api_mode, - capabilities, extra_headers)``; ``extra_headers`` falls back to *fallback_headers*.""" + extra_headers)``; ``extra_headers`` falls back to *fallback_headers*.""" from hermes_cli.runtime_provider import resolve_runtime_provider runtime = resolve_runtime_provider(**kwargs) @@ -1712,7 +1578,6 @@ def _runtime_creds(fallback_headers: dict, **kwargs) -> tuple[str, str, str, dic runtime.get("api_key", ""), runtime.get("base_url", ""), runtime.get("api_mode", ""), - runtime.get("capabilities") or {}, runtime.get("extra_headers") or fallback_headers, ) @@ -1890,6 +1755,529 @@ def _apply_direct_alias_endpoint( return api_key or "no-key-required", base_url, headers_override, suppress +def _moa_default_preset() -> str: + try: + from hermes_cli.config import load_config + from hermes_cli.moa_config import normalize_moa_config + + return normalize_moa_config(load_config().get("moa") or {})["default_preset"] + except Exception: + return "default" + + +@dataclass +class _Switch: + """Mutable state threaded through the ``switch_model`` steps. + + The routing steps settle ``target_provider`` / ``new_model`` / + ``resolved_alias`` (and may promote a config-routed ``providers.`` to + ``explicit_provider`` so the credential step resolves its block); the + credential step fills ``api_key`` / ``base_url`` / ``api_mode`` / + ``validation_headers``. + """ + + raw_input: str + current_provider: str + current_model: str + current_base_url: str + current_api_key: str + is_global: bool + explicit_provider: str + user_providers: Optional[dict] + custom_providers: Optional[list] + new_model: str = "" + target_provider: str = "" + resolved_alias: str = "" + provider_label: str = "" + api_key: str = "" + base_url: str = "" + api_mode: str = "" + validation_headers: dict = field(default_factory=dict) + suppress_ollama_headers: bool = False + validation: dict = field(default_factory=dict) + + def fail(self, message: str, **fields) -> ModelSwitchResult: + return _switch_fail(self.is_global, message, **fields) + + @property + def provider_changed(self) -> bool: + return self.target_provider != self.current_provider + + +def _route_explicit_provider(st: _Switch) -> Optional[ModelSwitchResult]: + """PATH A (``--provider`` given): resolve the provider, auto-detect a model + from a local endpoint when none was typed, then resolve the alias on the + TARGET provider.""" + pdef = resolve_provider_full(st.explicit_provider, st.user_providers, st.custom_providers) + if pdef is None and st.explicit_provider.strip().lower() == "custom": + pdef = _bare_custom_provider_def(st.current_base_url) + if pdef is None: + return st.fail(_unknown_provider_message(st.explicit_provider)) + + st.target_provider = pdef.id + if st.target_provider == "moa" and not st.new_model: + st.new_model = _moa_default_preset() + + agg_err = _aggregator_alias_error( + st.explicit_provider, st.target_provider, st.current_provider, st.user_providers, st.custom_providers, + ) + if agg_err: + return st.fail(agg_err, target_provider=st.target_provider, provider_label=pdef.name) + + if not st.new_model: + if not pdef.base_url: + return st.fail( + f"Provider '{pdef.name}' has no base URL configured. " + f"Specify a model: /model --provider {st.explicit_provider}", + target_provider=st.target_provider, provider_label=pdef.name, + ) + from hermes_cli.runtime_provider import _auto_detect_local_model + st.new_model = _auto_detect_local_model(pdef.base_url) + if not st.new_model: + return st.fail( + f"No model detected on {pdef.name} ({pdef.base_url}). " + f"Specify the model explicitly: /model --provider {st.explicit_provider}", + target_provider=st.target_provider, provider_label=pdef.name, + ) + + try: + alias_result = resolve_alias(st.new_model, st.target_provider) + except AmbiguousAliasError as err: + return st.fail(_ambiguous_alias_message(err), target_provider=st.target_provider) + if alias_result is not None: + _, st.new_model, st.resolved_alias = alias_result + return None + + +def _route_alias_fallback(st: _Switch, key: str) -> Optional[ModelSwitchResult]: + """Step b: the alias exists but not on the current provider -> try the + user's authenticated providers.""" + authed = get_authenticated_provider_slugs( + current_provider=st.current_provider, user_providers=st.user_providers, custom_providers=st.custom_providers, + ) + try: + fallback_result = _resolve_alias_fallback(st.raw_input, authed) + except AmbiguousAliasError as err: + return st.fail(_ambiguous_alias_message(err)) + if fallback_result is None: + identity = MODEL_ALIASES[key] + return st.fail( + f"Alias '{key}' maps to {identity.vendor}/{identity.family} " + f"but no matching model was found in any provider catalog. " + f"Try specifying the full model name.", + ) + st.target_provider, st.new_model, st.resolved_alias = fallback_result + logger.debug( + "Alias '%s' resolved via fallback to %s on %s", st.resolved_alias, st.new_model, st.target_provider, + ) + return None + + +def _convert_vendor_colon_slug(st: _Switch) -> None: + """Step c: on an aggregator, ``vendor:model`` -> ``vendor/model``. Only + without a slash: with one, the colon is a variant tag (:free, :extended, + :fast) that must be preserved.""" + raw_input = st.raw_input + colon_pos = raw_input.find(":") + cur_norm = str(st.current_provider).strip().lower() + if ( + colon_pos > 0 + and "/" not in raw_input + and is_aggregator(st.current_provider) + and not cur_norm.startswith("custom") + and cur_norm != "ollama" + ): + left = raw_input[:colon_pos].strip().lower() + right = raw_input[colon_pos + 1:].strip() + if left and right: + st.new_model = f"{left}/{right}" + logger.debug("Converted vendor:model '%s' to aggregator slug '%s'", raw_input, st.new_model) + + +def _route_configured_provider(st: _Switch) -> Optional[ModelSwitchResult] | bool: + """Step d.5: a model declared in user/custom provider config routes there + BEFORE detect_provider_for_model() guesses from static catalogs and before a + soft-accepting current provider (openai-codex) can swallow it as an unknown + hidden model. Returns a failure result, ``True`` when routed, else ``False``.""" + cfg_matches = _configured_provider_matches(st.new_model, st.user_providers, st.custom_providers) + if not cfg_matches: + return False + if st.current_provider in cfg_matches: + st.new_model = cfg_matches[st.current_provider] + return True + match_slugs = sorted(cfg_matches) + if len(match_slugs) > 1: + return st.fail( + f"'{st.new_model}' is declared by multiple configured " + f"providers ({', '.join(match_slugs)}). Re-run with " + f"--provider to choose which one to use.", + ) + st.target_provider = match_slugs[0] + st.new_model = cfg_matches[st.target_provider] + logger.debug("Configured-provider detection routed '%s' to %s", st.new_model, st.target_provider) + # providers. endpoints resolve in the credential block via + # resolve_user_provider(), which is gated on explicit_provider; custom:* + # slugs resolve at runtime directly. + if isinstance(st.user_providers, dict) and st.target_provider in st.user_providers: + st.explicit_provider = st.target_provider + return True + + +def _route_from_model_input(st: _Switch) -> Optional[ModelSwitchResult]: + """PATH B (no ``--provider``): MoA preset / alias on the current provider + (a) -> alias fallback (b) or ``vendor:model`` conversion (c) -> aggregator + catalog search (d) -> configured-provider match (d.5) -> + detect_provider_for_model() as last resort (e).""" + from hermes_cli.models import detect_provider_for_model + + raw_input, current_provider = st.raw_input, st.current_provider + resolved_moa_preset = False + try: + from hermes_cli.config import load_config + from hermes_cli.moa_config import exact_moa_preset_name, normalize_moa_config + + moa_match = exact_moa_preset_name(normalize_moa_config(load_config().get("moa") or {}), raw_input) + if moa_match: + st.target_provider, st.new_model, st.resolved_alias = "moa", moa_match, "" + resolved_moa_preset = True + alias_result = None + else: + alias_result = resolve_alias(raw_input, current_provider) + except AmbiguousAliasError as err: + return st.fail(_ambiguous_alias_message(err)) + except Exception: + try: + alias_result = resolve_alias(raw_input, current_provider) + except AmbiguousAliasError as err: + return st.fail(_ambiguous_alias_message(err)) + + if resolved_moa_preset: + pass + elif alias_result is not None: + st.target_provider, st.new_model, st.resolved_alias = alias_result + logger.debug("Alias '%s' resolved to %s on %s", st.resolved_alias, st.new_model, st.target_provider) + elif raw_input.strip().lower() in MODEL_ALIASES: + fail = _route_alias_fallback(st, raw_input.strip().lower()) + if fail is not None: + return fail + else: + _convert_vendor_colon_slug(st) + + # Step d: if the CURRENT provider's live catalog resolved the model, step e + # must not second-guess and switch providers — flat-namespace resellers + # (opencode-go/zen) return bare ids that coincidentally match native + # providers' static catalogs. + resolved_in_current_catalog = False + if is_aggregator(st.target_provider) and not st.resolved_alias: + catalog = list_provider_models(st.target_provider) + if catalog: + matched = _aggregator_catalog_match(st.new_model, catalog) + if matched is not None: + st.new_model, resolved_in_current_catalog = matched, True + + # Step d.5 — deliberately NOT gated on ``not is_custom``. + config_routed = False + if not st.resolved_alias and not resolved_in_current_catalog and st.target_provider == current_provider: + config_routed = _route_configured_provider(st) + if isinstance(config_routed, ModelSwitchResult): + return config_routed + + # Step e + is_custom = ( + current_provider in {"custom", "local"} + or current_provider.startswith("custom:") + or base_url_hostname(st.current_base_url or "") in ("localhost", "127.0.0.1") + ) + if ( + st.target_provider == current_provider + and not is_custom + and not st.resolved_alias + and not resolved_in_current_catalog + and not config_routed + ): + detected = detect_provider_for_model(st.new_model, current_provider) + if detected: + st.target_provider, st.new_model = detected + return None + + +def _switch_provider_label(st: _Switch) -> str: + label = get_label(st.target_provider) + if st.target_provider == "custom" and st.current_base_url: + label = "Custom endpoint" + if st.target_provider.startswith("custom:"): + custom_pdef = resolve_provider_full(st.target_provider, st.user_providers, st.custom_providers) + if custom_pdef is not None: + label = custom_pdef.name + return label + + +def _creds_for_switched_provider(st: _Switch) -> Optional[ModelSwitchResult]: + """Credentials when the provider changed or ``--provider`` was given. + + ``providers.`` blocks carry their own base_url + transport + key + reference; resolve_runtime_provider() resolves by provider NAME and would + re-resolve a block named "openai" from scratch (or hop to an aggregator), + so use the pdef's endpoint directly. + """ + user_pdef = None + if st.explicit_provider and st.user_providers: + from hermes_cli.providers import resolve_user_provider + user_pdef = resolve_user_provider(st.explicit_provider.strip().lower(), st.user_providers) + if user_pdef is None: + user_pdef = resolve_user_provider(st.target_provider, st.user_providers) + if user_pdef is not None and user_pdef.base_url: + ucfg = (st.user_providers or {}).get(st.explicit_provider.strip().lower()) \ + or (st.user_providers or {}).get(st.target_provider) or {} + # Key reads go through the per-profile secret scope: a raw os.environ + # read would hand this profile another profile's key under the + # multiplexed gateway. + ukey = _entry_configured_key(ucfg, _scoped_key_env) + st.validation_headers = _extra_headers_from_config(ucfg) + try: + api_key, base_url, st.api_mode, st.validation_headers = _runtime_creds( + st.validation_headers, + requested=st.target_provider, + explicit_api_key=ukey or None, + explicit_base_url=user_pdef.base_url, + target_model=st.new_model, + ) + st.api_key = api_key or ukey + st.base_url = base_url or user_pdef.base_url + except Exception: + st.api_key, st.base_url, st.api_mode = ukey, user_pdef.base_url, "" + elif st.target_provider == "custom" and st.current_base_url: + st.api_key, st.base_url = st.current_api_key, st.current_base_url + st.api_mode = determine_api_mode(st.target_provider, st.base_url) + else: + try: + st.api_key, st.base_url, st.api_mode, st.validation_headers = _runtime_creds( + st.validation_headers, requested=st.target_provider, target_model=st.new_model, + ) + except Exception as e: + return st.fail( + f"Could not resolve credentials for provider '{st.provider_label}': {e}", + target_provider=st.target_provider, provider_label=st.provider_label, + ) + return None + + +def _creds_for_current_provider(st: _Switch) -> None: + """Credentials when staying on the current provider. Mid-session + ``/model `` on a local Ollama-compatible endpoint keeps the endpoint + in use; re-resolving bare ``custom`` from config can fall through to an + unrelated default provider.""" + from hermes_cli.models import _get_ollama_request_headers, _same_ollama_native_root + + keep_current_ollama_endpoint = False + ollama_headers: dict[str, str] = {} + if st.current_provider == "custom" and st.current_base_url: + try: + from hermes_cli.models import should_use_ollama_native_catalog + ollama_headers = _get_ollama_request_headers() + _, configured_ollama_base = _ollama_configured_base() + # Provider-level Ollama headers only belong to the configured + # native root; without one there is no safe origin for them. + if not configured_ollama_base or not _same_ollama_native_root(st.current_base_url, configured_ollama_base): + ollama_headers = {} + st.suppress_ollama_headers = True + keep_current_ollama_endpoint = should_use_ollama_native_catalog( + st.current_provider, st.current_base_url, headers=ollama_headers, + ) + except (ImportError, OSError, RuntimeError, TypeError, ValueError): + keep_current_ollama_endpoint = False + if keep_current_ollama_endpoint: + st.api_key = st.current_api_key or "no-key-required" + st.base_url = st.current_base_url + st.api_mode = determine_api_mode(st.current_provider, st.base_url) + st.validation_headers = ollama_headers + else: + try: + st.api_key, st.base_url, st.api_mode, st.validation_headers = _runtime_creds( + st.validation_headers, requested=st.current_provider, target_model=st.new_model, + ) + except Exception: + pass + + +def _resolve_switch_credentials(st: _Switch) -> Optional[ModelSwitchResult]: + """COMMON PATH part 1: credentials, direct-alias endpoint override, and the + api_mode for the final (provider, base_url) before validation.""" + st.provider_label = _switch_provider_label(st) + st.api_key, st.base_url = st.current_api_key, st.current_base_url + if st.provider_changed or st.explicit_provider: + fail = _creds_for_switched_provider(st) + if fail is not None: + return fail + else: + _creds_for_current_provider(st) + + # Direct alias override: use the alias's exact base_url if set. + if st.resolved_alias: + _ensure_direct_aliases() + da = DIRECT_ALIASES.get(st.resolved_alias) + if da is not None and da.base_url: + st.api_key, st.base_url, headers_override, suppress = _apply_direct_alias_endpoint( + da, st.target_provider, st.new_model, st.api_key, st.base_url, + ) + st.api_mode = "" # clear so determine_api_mode re-detects from URL + if headers_override is not None: + st.validation_headers = headers_override + if suppress: + st.suppress_ollama_headers = True + + # Fills an empty mode (alias cleared it) and overrides a STALE mode carried + # from previous session state when the host mandates one wire protocol + # (e.g. gpt-5.x on api.openai.com would otherwise 400 on tools+reasoning). + mandated_mode = host_mandated_api_mode(st.base_url) + if mandated_mode is not None: + st.api_mode = mandated_mode + elif not st.api_mode: + st.api_mode = determine_api_mode(st.target_provider, st.base_url) + return None + + +def _validate_switch(st: _Switch) -> Optional[ModelSwitchResult]: + """COMMON PATH part 2: normalize the model name for the target provider, + validate it, and accept config-declared models the remote catalog lacks.""" + from hermes_cli.models import _get_ollama_request_headers, validate_requested_model + + st.new_model = _resolve_named_custom_model_id(st.new_model, st.target_provider, st.custom_providers) + st.new_model = normalize_model_for_provider(st.new_model, st.target_provider) + + if st.target_provider.strip().lower() == "ollama": + headers = {} if st.suppress_ollama_headers else (st.validation_headers or _get_ollama_request_headers()) + else: + headers = st.validation_headers or ( + _extra_headers_from_config(st.user_providers.get(st.target_provider)) + if st.user_providers and st.target_provider in st.user_providers + else None + ) + try: + validation = validate_requested_model( + st.new_model, st.target_provider, api_key=st.api_key, base_url=st.base_url, + api_mode=st.api_mode or None, headers=headers, + ) + except Exception as e: + validation = { + "accepted": False, + "persist": False, + "recognized": False, + "message": f"Could not validate `{st.new_model}`: {e}", + } + + if not validation.get("accepted"): + if _config_declares_model(st.new_model, st.target_provider, st.base_url, st.user_providers, st.custom_providers): + validation = {"accepted": True, "persist": True, "recognized": False, "message": validation.get("message", "")} + else: + return st.fail( + validation.get("message", "Invalid model"), + new_model=st.new_model, target_provider=st.target_provider, provider_label=st.provider_label, + ) + if validation.get("corrected_model"): + st.new_model = validation["corrected_model"] + st.validation = validation + return None + + +def _copilot_api_mode(provider: str, model: str, api_key: str) -> str: + from hermes_cli.models import copilot_model_api_mode + + return copilot_model_api_mode(model, api_key=api_key) + + +def _opencode_api_mode(provider: str, model: str, api_key: str) -> str: + from hermes_cli.models import opencode_model_api_mode + + return opencode_model_api_mode(provider, model) + + +def _nous_api_mode(provider: str, model: str, api_key: str) -> str: + # Portal serves anthropic/* on /v1/messages and everything else on + # /chat/completions; re-derive from the FINAL model so alias clears / + # empty fallbacks cannot leave Claude on the OpenAI wire. + from hermes_cli.providers import nous_api_mode + + return nous_api_mode(model) + + +# Per-provider api_mode overrides applied after validation, keyed on the final +# target provider (the key sets are disjoint, so exactly one — or none — fires). +_PROVIDER_API_MODE_OVERRIDES: dict[str, Any] = { + **dict.fromkeys(("copilot", "github-copilot"), _copilot_api_mode), + **dict.fromkeys(("opencode-zen", "opencode-go", "opencode"), _opencode_api_mode), + **dict.fromkeys(("nous", "nous-portal", "nousresearch"), _nous_api_mode), +} + + +def _build_switch_result(st: _Switch) -> ModelSwitchResult: + """COMMON PATH part 3: final api_mode / base_url shaping, metadata, warnings.""" + override = _PROVIDER_API_MODE_OVERRIDES.get(st.target_provider) + if override is not None: + st.api_mode = override(st.target_provider, st.new_model, st.api_key) + if not st.api_mode: + st.api_mode = determine_api_mode(st.target_provider, st.base_url, model=st.new_model) + + # OpenCode base URLs end with /v1 for OpenAI-compatible models but the + # Anthropic SDK prepends its own /v1/messages: strip for anthropic_messages, + # re-append for chat_completions/codex_responses (mirrors + # resolve_runtime_provider; either direction alone breaks the other family). + from hermes_cli.models import normalize_opencode_base_url, opencode_provider_family + if opencode_provider_family(st.target_provider) is not None and isinstance(st.base_url, str): + st.base_url = normalize_opencode_base_url(st.target_provider, st.api_mode, st.base_url) + + capabilities = get_model_capabilities(st.target_provider, st.new_model, allow_network=True) + from agent.native_compaction import resolve_native_compaction_capabilities + runtime_capabilities = resolve_native_compaction_capabilities( + model=st.new_model, + base_url=st.base_url, + provider=st.target_provider, + is_codex_backend=st.target_provider.strip().lower() == "openai-codex", + ) + model_info = get_model_info(st.target_provider, st.new_model, allow_network=True) + + warnings: list[str] = [] + if st.validation.get("message"): + warnings.append(st.validation["message"]) + hermes_warn = _check_hermes_model_warning(st.new_model) + if hermes_warn: + warnings.append(hermes_warn) + + # Carry the switched provider's request_overrides (custom_providers + # ``extra_body`` such as chat_template_kwargs) so the gateway applies them + # like the default-provider path does. + request_overrides = None + try: + from hermes_cli.runtime_provider import _get_named_custom_provider, _custom_provider_request_overrides + cp_for_ro = _get_named_custom_provider(st.target_provider) + if cp_for_ro: + request_overrides = _custom_provider_request_overrides(cp_for_ro) or None + except Exception: + request_overrides = None + + return ModelSwitchResult( + success=True, + new_model=st.new_model, + target_provider=st.target_provider, + provider_changed=st.provider_changed, + api_key=st.api_key, + base_url=st.base_url, + api_mode=st.api_mode, + request_overrides=dict(request_overrides or {}), + warning_message=" | ".join(warnings) if warnings else "", + provider_label=st.provider_label, + resolved_via_alias=st.resolved_alias, + capabilities=capabilities, + runtime_capabilities={ + key: value + for key, value in runtime_capabilities.items() + if isinstance(key, str) and isinstance(value, bool) + }, + model_info=model_info, + is_global=st.is_global, + ) + + def switch_model( raw_input: str, current_provider: str, @@ -1903,495 +2291,35 @@ def switch_model( ) -> ModelSwitchResult: """Core model-switching pipeline shared between CLI and gateway. - Resolution chain: - - If --provider given: - a. Resolve provider via resolve_provider_full() - b. Resolve credentials - c. If model given, resolve alias on target provider or use as-is - d. If no model, auto-detect from endpoint - - If no --provider: - a. Try alias resolution on current provider - b. If alias exists but not on current provider -> fallback - c. On aggregator, try vendor/model slug conversion - d. Aggregator catalog search - e. detect_provider_for_model() as last resort - f. Resolve credentials - g. Normalize model name for target provider - - Finally: - h. Get full model metadata from models.dev - i. Build result + Resolution chain: route the request (:func:`_route_explicit_provider` when + ``--provider`` was given, else :func:`_route_from_model_input`) -> + :func:`_resolve_switch_credentials` -> :func:`_validate_switch` -> + :func:`_build_switch_result`. Each step returns a failure + :class:`ModelSwitchResult` to stop the chain, or ``None`` to continue. ``explicit_provider`` comes from the --provider flag (empty = none); ``user_providers`` / ``custom_providers`` are the ``providers:`` dict and ``custom_providers:`` list from config.yaml. """ - from hermes_cli.models import ( - copilot_model_api_mode, - detect_provider_for_model, - validate_requested_model, - opencode_model_api_mode, - _get_ollama_request_headers, - _same_ollama_native_root, - ) - - resolved_alias = "" - new_model = raw_input.strip() - target_provider = current_provider - resolved_moa_preset = False - - # ================================================================= - # PATH A: Explicit --provider given - # ================================================================= - if explicit_provider: - pdef = resolve_provider_full(explicit_provider, user_providers, custom_providers) - if pdef is None and explicit_provider.strip().lower() == "custom": - pdef = _bare_custom_provider_def(current_base_url) - if pdef is None: - return _switch_fail(is_global, _unknown_provider_message(explicit_provider)) - - target_provider = pdef.id - if target_provider == "moa" and not new_model: - try: - from hermes_cli.config import load_config - from hermes_cli.moa_config import normalize_moa_config - - new_model = normalize_moa_config(load_config().get("moa") or {})["default_preset"] - except Exception: - new_model = "default" - - agg_err = _aggregator_alias_error( - explicit_provider, target_provider, current_provider, user_providers, custom_providers, - ) - if agg_err: - return _switch_fail(is_global, agg_err, target_provider=target_provider, provider_label=pdef.name) - - # No model specified: auto-detect from the endpoint - if not new_model: - if not pdef.base_url: - return _switch_fail( - is_global, - f"Provider '{pdef.name}' has no base URL configured. " - f"Specify a model: /model --provider {explicit_provider}", - target_provider=target_provider, provider_label=pdef.name, - ) - from hermes_cli.runtime_provider import _auto_detect_local_model - new_model = _auto_detect_local_model(pdef.base_url) - if not new_model: - return _switch_fail( - is_global, - f"No model detected on {pdef.name} ({pdef.base_url}). " - f"Specify the model explicitly: /model --provider {explicit_provider}", - target_provider=target_provider, provider_label=pdef.name, - ) - - # Resolve alias on the TARGET provider - try: - alias_result = resolve_alias(new_model, target_provider) - except AmbiguousAliasError as err: - return _switch_fail(is_global, _ambiguous_alias_message(err), target_provider=target_provider) - if alias_result is not None: - _, new_model, resolved_alias = alias_result - - # ================================================================= - # PATH B: No explicit provider — resolve from model input - # ================================================================= - else: - try: - from hermes_cli.config import load_config - from hermes_cli.moa_config import exact_moa_preset_name, normalize_moa_config - - moa_match = exact_moa_preset_name(normalize_moa_config(load_config().get("moa") or {}), raw_input) - if moa_match: - target_provider, new_model, resolved_alias = "moa", moa_match, "" - resolved_moa_preset = True - alias_result = None - else: - alias_result = resolve_alias(raw_input, current_provider) - except AmbiguousAliasError as err: - return _switch_fail(is_global, _ambiguous_alias_message(err)) - except Exception: - try: - alias_result = resolve_alias(raw_input, current_provider) - except AmbiguousAliasError as err: - return _switch_fail(is_global, _ambiguous_alias_message(err)) - - # --- Step a: alias on current provider --- - if resolved_moa_preset: - pass - elif alias_result is not None: - target_provider, new_model, resolved_alias = alias_result - logger.debug("Alias '%s' resolved to %s on %s", resolved_alias, new_model, target_provider) - else: - # --- Step b: alias exists but not on current provider -> fallback --- - key = raw_input.strip().lower() - if key in MODEL_ALIASES: - authed = get_authenticated_provider_slugs( - current_provider=current_provider, user_providers=user_providers, custom_providers=custom_providers, - ) - try: - fallback_result = _resolve_alias_fallback(raw_input, authed) - except AmbiguousAliasError as err: - return _switch_fail(is_global, _ambiguous_alias_message(err)) - if fallback_result is None: - identity = MODEL_ALIASES[key] - return _switch_fail( - is_global, - f"Alias '{key}' maps to {identity.vendor}/{identity.family} " - f"but no matching model was found in any provider catalog. " - f"Try specifying the full model name.", - ) - target_provider, new_model, resolved_alias = fallback_result - logger.debug( - "Alias '%s' resolved via fallback to %s on %s", resolved_alias, new_model, target_provider, - ) - else: - # --- Step c: on an aggregator, vendor:model -> vendor/model --- - # Only without a slash: with one, the colon is a variant tag - # (:free, :extended, :fast) that must be preserved. - colon_pos = raw_input.find(":") - cur_norm = str(current_provider).strip().lower() - if ( - colon_pos > 0 - and "/" not in raw_input - and is_aggregator(current_provider) - and not cur_norm.startswith("custom") - and cur_norm != "ollama" - ): - left = raw_input[:colon_pos].strip().lower() - right = raw_input[colon_pos + 1:].strip() - if left and right: - new_model = f"{left}/{right}" - logger.debug("Converted vendor:model '%s' to aggregator slug '%s'", raw_input, new_model) - - # --- Step d: aggregator catalog search --- - # If the CURRENT provider's live catalog resolved the model, step e must - # not second-guess and switch providers — flat-namespace resellers - # (opencode-go/zen) return bare ids that coincidentally match native - # providers' static catalogs. - resolved_in_current_catalog = False - if is_aggregator(target_provider) and not resolved_alias: - catalog = list_provider_models(target_provider) - if catalog: - matched = _aggregator_catalog_match(new_model, catalog) - if matched is not None: - new_model, resolved_in_current_catalog = matched, True - - # --- Step d.5: configured-provider exact match --- - # A model declared in user/custom provider config routes there BEFORE - # detect_provider_for_model() guesses from static catalogs and before a - # soft-accepting current provider (openai-codex) can swallow it as an - # unknown hidden model. Deliberately NOT gated on ``not is_custom``. - config_routed = False - if not resolved_alias and not resolved_in_current_catalog and target_provider == current_provider: - cfg_matches = _configured_provider_matches(new_model, user_providers, custom_providers) - if cfg_matches: - if current_provider in cfg_matches: - new_model = cfg_matches[current_provider] - config_routed = True - else: - match_slugs = sorted(cfg_matches) - if len(match_slugs) > 1: - return _switch_fail( - is_global, - f"'{new_model}' is declared by multiple configured " - f"providers ({', '.join(match_slugs)}). Re-run with " - f"--provider to choose which one to use.", - ) - target_provider = match_slugs[0] - new_model = cfg_matches[target_provider] - config_routed = True - logger.debug("Configured-provider detection routed '%s' to %s", new_model, target_provider) - # providers. endpoints resolve in the credential block - # via resolve_user_provider(), which is gated on - # explicit_provider; custom:* slugs resolve at runtime directly. - if isinstance(user_providers, dict) and target_provider in user_providers: - explicit_provider = target_provider - - # --- Step e: detect_provider_for_model() as last resort --- - is_custom = ( - current_provider in {"custom", "local"} - or current_provider.startswith("custom:") - or base_url_hostname(current_base_url or "") in ("localhost", "127.0.0.1") - ) - if ( - target_provider == current_provider - and not is_custom - and not resolved_alias - and not resolved_in_current_catalog - and not config_routed - ): - detected = detect_provider_for_model(new_model, current_provider) - if detected: - target_provider, new_model = detected - - # ================================================================= - # COMMON PATH: Resolve credentials, normalize, get metadata - # ================================================================= - provider_changed = target_provider != current_provider - provider_label = get_label(target_provider) - if target_provider == "custom" and current_base_url: - provider_label = "Custom endpoint" - if target_provider.startswith("custom:"): - custom_pdef = resolve_provider_full(target_provider, user_providers, custom_providers) - if custom_pdef is not None: - provider_label = custom_pdef.name - - # --- Resolve credentials --- - api_key = current_api_key - base_url = current_base_url - api_mode = "" - runtime_capabilities: dict[str, bool] = {} - ollama_headers: dict[str, str] = {} - validation_headers: dict[str, str] = {} - suppress_ollama_headers = False - - if provider_changed or explicit_provider: - # providers. blocks carry their own base_url + transport + key - # reference; resolve_runtime_provider() resolves by provider NAME and - # would re-resolve a block named "openai" from scratch (or hop to an - # aggregator), so use the pdef's endpoint directly. - user_pdef = None - if explicit_provider and user_providers: - from hermes_cli.providers import resolve_user_provider - user_pdef = resolve_user_provider(explicit_provider.strip().lower(), user_providers) - if user_pdef is None: - user_pdef = resolve_user_provider(target_provider, user_providers) - if user_pdef is not None and user_pdef.base_url: - ucfg = (user_providers or {}).get(explicit_provider.strip().lower()) \ - or (user_providers or {}).get(target_provider) or {} - # Key reads go through the per-profile secret scope: a raw - # os.environ read would hand this profile another profile's key - # under the multiplexed gateway. - ukey = _entry_configured_key(ucfg, _scoped_key_env) - validation_headers = _extra_headers_from_config(ucfg) - try: - api_key, base_url, api_mode, runtime_capabilities, validation_headers = _runtime_creds( - validation_headers, - requested=target_provider, - explicit_api_key=ukey or None, - explicit_base_url=user_pdef.base_url, - target_model=new_model, - ) - api_key = api_key or ukey - base_url = base_url or user_pdef.base_url - except Exception: - api_key, base_url, api_mode = ukey, user_pdef.base_url, "" - elif target_provider == "custom" and current_base_url: - api_key, base_url = current_api_key, current_base_url - api_mode = determine_api_mode(target_provider, base_url) - else: - try: - api_key, base_url, api_mode, runtime_capabilities, validation_headers = _runtime_creds( - validation_headers, requested=target_provider, target_model=new_model, - ) - except Exception as e: - return _switch_fail( - is_global, - f"Could not resolve credentials for provider '{provider_label}': {e}", - target_provider=target_provider, provider_label=provider_label, - ) - else: - keep_current_ollama_endpoint = False - if current_provider == "custom" and current_base_url: - try: - from hermes_cli.models import should_use_ollama_native_catalog - ollama_headers = _get_ollama_request_headers() - _, configured_ollama_base = _ollama_configured_base() - # Provider-level Ollama headers only belong to the configured - # native root; without one there is no safe origin for them. - if not configured_ollama_base or not _same_ollama_native_root(current_base_url, configured_ollama_base): - ollama_headers = {} - suppress_ollama_headers = True - keep_current_ollama_endpoint = should_use_ollama_native_catalog( - current_provider, current_base_url, headers=ollama_headers, - ) - except (ImportError, OSError, RuntimeError, TypeError, ValueError): - keep_current_ollama_endpoint = False - if keep_current_ollama_endpoint: - # Mid-session `/model ` on a local Ollama-compatible endpoint - # keeps the endpoint in use; re-resolving bare `custom` from config - # can fall through to an unrelated default provider. - api_key = current_api_key or "no-key-required" - base_url = current_base_url - api_mode = determine_api_mode(current_provider, base_url) - validation_headers = ollama_headers - else: - try: - api_key, base_url, api_mode, runtime_capabilities, validation_headers = _runtime_creds( - validation_headers, requested=current_provider, target_model=new_model, - ) - except Exception: - pass - - # --- Direct alias override: use the alias's exact base_url if set --- - if resolved_alias: - _ensure_direct_aliases() - da = DIRECT_ALIASES.get(resolved_alias) - if da is not None and da.base_url: - api_key, base_url, headers_override, suppress = _apply_direct_alias_endpoint( - da, target_provider, new_model, api_key, base_url, - ) - api_mode = "" # clear so determine_api_mode re-detects from URL - if headers_override is not None: - validation_headers = headers_override - if suppress: - suppress_ollama_headers = True - - # --- api_mode from the final (provider, base_url) before validation --- - # Fills an empty mode (alias cleared it) and overrides a STALE mode carried - # from previous session state when the host mandates one wire protocol - # (e.g. gpt-5.x on api.openai.com would otherwise 400 on tools+reasoning). - mandated_mode = host_mandated_api_mode(base_url) - if mandated_mode is not None: - api_mode = mandated_mode - elif not api_mode: - api_mode = determine_api_mode(target_provider, base_url) - - # --- Normalize model name for target provider --- - new_model = _resolve_named_custom_model_id(new_model, target_provider, custom_providers) - new_model = normalize_model_for_provider(new_model, target_provider) - - # --- Validate --- - if target_provider.strip().lower() == "ollama": - headers = {} if suppress_ollama_headers else (validation_headers or _get_ollama_request_headers()) - else: - headers = validation_headers or ( - _extra_headers_from_config(user_providers.get(target_provider)) - if user_providers and target_provider in user_providers - else None - ) - try: - validation = validate_requested_model( - new_model, target_provider, api_key=api_key, base_url=base_url, api_mode=api_mode or None, headers=headers, - ) - except Exception as e: - validation = { - "accepted": False, - "persist": False, - "recognized": False, - "message": f"Could not validate `{new_model}`: {e}", - } - - if not validation.get("accepted"): - if _config_declares_model(new_model, target_provider, base_url, user_providers, custom_providers): - validation = {"accepted": True, "persist": True, "recognized": False, "message": validation.get("message", "")} - else: - return _switch_fail( - is_global, validation.get("message", "Invalid model"), - new_model=new_model, target_provider=target_provider, provider_label=provider_label, - ) - - if validation.get("corrected_model"): - new_model = validation["corrected_model"] - - # --- Per-provider api_mode overrides --- - if target_provider in {"copilot", "github-copilot"}: - api_mode = copilot_model_api_mode(new_model, api_key=api_key) - if target_provider in {"opencode-zen", "opencode-go", "opencode"}: - api_mode = opencode_model_api_mode(target_provider, new_model) - if target_provider in {"nous", "nous-portal", "nousresearch"}: - # Portal serves anthropic/* on /v1/messages and everything else on - # /chat/completions; re-derive from the FINAL model so alias clears / - # empty fallbacks cannot leave Claude on the OpenAI wire. - from hermes_cli.providers import nous_api_mode - - api_mode = nous_api_mode(new_model) - if not api_mode: - api_mode = determine_api_mode(target_provider, base_url, model=new_model) - - # OpenCode base URLs end with /v1 for OpenAI-compatible models but the - # Anthropic SDK prepends its own /v1/messages: strip for anthropic_messages, - # re-append for chat_completions/codex_responses (mirrors - # resolve_runtime_provider; either direction alone breaks the other family). - from hermes_cli.models import opencode_provider_family - if opencode_provider_family(target_provider) is not None and isinstance(base_url, str): - from hermes_cli.models import normalize_opencode_base_url - base_url = normalize_opencode_base_url(target_provider, api_mode, base_url) - - capabilities = get_model_capabilities(target_provider, new_model, allow_network=True) - from agent.native_compaction import resolve_native_compaction_capabilities - runtime_capabilities = resolve_native_compaction_capabilities( - model=new_model, - base_url=base_url, - provider=target_provider, - is_codex_backend=target_provider.strip().lower() == "openai-codex", - ) - model_info = get_model_info(target_provider, new_model, allow_network=True) - - warnings: list[str] = [] - if validation.get("message"): - warnings.append(validation["message"]) - hermes_warn = _check_hermes_model_warning(new_model) - if hermes_warn: - warnings.append(hermes_warn) - - # Carry the switched provider's request_overrides (custom_providers - # ``extra_body`` such as chat_template_kwargs) so the gateway applies them - # like the default-provider path does. - request_overrides = None - try: - from hermes_cli.runtime_provider import _get_named_custom_provider, _custom_provider_request_overrides - cp_for_ro = _get_named_custom_provider(target_provider) - if cp_for_ro: - request_overrides = _custom_provider_request_overrides(cp_for_ro) or None - except Exception: - request_overrides = None - - return ModelSwitchResult( - success=True, - new_model=new_model, - target_provider=target_provider, - provider_changed=provider_changed, - api_key=api_key, - base_url=base_url, - api_mode=api_mode, - request_overrides=dict(request_overrides or {}), - warning_message=" | ".join(warnings) if warnings else "", - provider_label=provider_label, - resolved_via_alias=resolved_alias, - capabilities=capabilities, - runtime_capabilities={ - key: value - for key, value in runtime_capabilities.items() - if isinstance(key, str) and isinstance(value, bool) - }, - model_info=model_info, + st = _Switch( + raw_input=raw_input, + current_provider=current_provider, + current_model=current_model, + current_base_url=current_base_url, + current_api_key=current_api_key, is_global=is_global, + explicit_provider=explicit_provider, + user_providers=user_providers, + custom_providers=custom_providers, + new_model=raw_input.strip(), + target_provider=current_provider, ) - - -# --------------------------------------------------------------------------- -# Authenticated providers listing (for /model no-args display) -# --------------------------------------------------------------------------- - -# Process-level guard so the picker prewarm thread is spawned at most once per -# process — mirrors run_agent's _openrouter_prewarm_done. Without a guard a -# long-lived process (or repeated triggers) would leak one OS thread per call. -import threading as _threading # noqa: E402 - -_picker_prewarm_done = _threading.Event() - - -def _credential_pool_is_usable(provider: str, *, raw_pool_present: bool = False) -> bool: - """Return whether *provider* has a credential that can be selected now. - - ``auth.json`` historically allowed opaque token-style pool values that do - not deserialize into ``PooledCredential`` entries. Preserve visibility for - those legacy values, but when a real pool exists its availability state is - authoritative: an all-exhausted/dead pool is not authenticated. - """ - try: - from agent.credential_pool import load_pool - - pool = load_pool(provider) - if pool.has_credentials(): - return pool.has_available() - except Exception: - pass - return raw_pool_present + route = _route_explicit_provider if explicit_provider else _route_from_model_input + for step in (route, _resolve_switch_credentials, _validate_switch): + fail = step(st) + if fail is not None: + return fail + return _build_switch_result(st) def _extra_headers_from_config(entry: Any) -> dict[str, str]: @@ -2402,54 +2330,6 @@ def _extra_headers_from_config(entry: Any) -> dict[str, str]: return normalize_extra_headers(entry.get("extra_headers")) -def prewarm_picker_cache_async() -> Optional["_threading.Thread"]: - """Warm the provider-models disk cache in a background daemon thread. - - The no-args ``/model`` picker calls ``list_authenticated_providers()``, - which fetches each authenticated provider's live ``/v1/models`` list on a - cold/stale cache. Those fetches are independent HTTP round-trips but run - serially, so the first ``/model`` open in a session (or any open after the - 1h cache TTL expires) blocks ~1-2s on the user's critical path. - - This pre-warms that exact path off-thread during idle session time: it - runs ``list_authenticated_providers()`` once, which populates - ``provider_models_cache.json`` for every authed provider. By the time the - user types ``/model``, the picker hits the warm disk cache and renders in - ~100ms. - - Fire-and-forget. Process-level Event guard ensures it runs at most once. - Fully exception-isolated — a slow or offline provider can never affect the - session. Returns the spawned thread (for tests) or None if already warmed. - """ - if _picker_prewarm_done.is_set(): - return None - _picker_prewarm_done.set() - - def _warm() -> None: - try: - from hermes_cli.inventory import load_picker_context - - ctx = load_picker_context() - # Calling this is what populates cached_provider_model_ids() -> - # provider_models_cache.json for each authed provider. We discard - # the result; the side effect (warm disk cache) is the point. - list_authenticated_providers( - current_provider=ctx.current_provider, - current_base_url=ctx.current_base_url, - current_model=ctx.current_model, - user_providers=ctx.user_providers, - custom_providers=ctx.custom_providers, - excluded_providers=ctx.excluded_providers or [], - ) - except Exception: - # Best-effort warmup — never surface errors into the session. - logger.debug("picker cache prewarm failed", exc_info=True) - - t = _threading.Thread(target=_warm, daemon=True, name="picker-cache-prewarm") - t.start() - return t - - def _scoped_key_env(name: str) -> str: """Read a provider key env var through the per-profile secret scope. @@ -2492,79 +2372,6 @@ def _scoped_key_env(name: str) -> str: # Before: ~20s serial blocking (sum of all provider latencies) # After: ~8s parallel (max single provider latency), rest served from cache -_PARALLEL_PREFETCH_WORKERS = 8 - - -def _prefetch_provider_models_parallel(provider_slugs: list[str]) -> None: - """Fetch model catalogs for multiple providers in parallel. - - Only providers whose cache entry is stale or missing are fetched; fresh - entries are skipped to avoid unnecessary network calls. Each worker uses - :func:`update_provider_cache_entry` (thread-safe) to persist its result, - so concurrent writes to ``provider_models_cache.json`` don't clobber each - other. - - :param provider_slugs: Hermes provider IDs to prefetch (e.g. ``["openrouter", - "anthropic", "deepseek"]``). Unknown providers are silently skipped. - """ - from hermes_cli.models import cached_provider_model_ids - - # Quick-stale-check: skip providers whose cache is already fresh so we - # don't waste network calls on a warm cache. We check staleness the same - # way cached_provider_model_ids does internally: load the cache, compare - # age to TTL. This is a read-only check — if the cache file changes - # between this check and the actual fetch, cached_provider_model_ids will - # still do the right thing (it re-reads the cache internally). - from hermes_cli.models import ( - _load_provider_models_cache, - _credential_fingerprint, - _PROVIDER_MODELS_CACHE_TTL, - normalize_provider, - ) - - now = time.time() - stale_slugs: list[str] = [] - cache = _load_provider_models_cache() - for slug in provider_slugs: - normalized = normalize_provider(slug) or (slug or "") - if not normalized: - continue - entry = cache.get(normalized) - fp = _credential_fingerprint(normalized) - if ( - isinstance(entry, dict) - and entry.get("fp") == fp - and isinstance(entry.get("models"), list) - and entry["models"] - ): - age = now - float(entry.get("at", 0)) - if age < _PROVIDER_MODELS_CACHE_TTL: - continue # fresh, skip - stale_slugs.append(normalized) - - if not stale_slugs: - return - - import concurrent.futures - - def _fetch_one(slug: str) -> None: - try: - models = cached_provider_model_ids(slug, force_refresh=True) - # cached_provider_model_ids already persists the result, but in a - # non-locked read-modify-write. Re-persist via the thread-safe - # path to guarantee no lost writes under concurrency. - if models: - from hermes_cli.models import update_provider_cache_entry - update_provider_cache_entry(slug, models) - except Exception: - pass # best-effort; picker falls back to curated list - - with concurrent.futures.ThreadPoolExecutor( - max_workers=min(_PARALLEL_PREFETCH_WORKERS, len(stale_slugs)), - thread_name_prefix="model-cache-prefetch", - ) as executor: - list(executor.map(_fetch_one, stale_slugs)) - # --- Provider-row discovery shared by the picker and the prefetch scan ------- # @@ -2580,1145 +2387,3 @@ def _prefetch_provider_models_parallel(provider_slugs: list[str]) -> None: # patch those modules. -def _iter_builtin_candidates(models_dev_data: dict, excluded: set, seen: set): - """Yield ``(hermes_id, mdev_id, pconfig, env_vars)`` for section-1 rows. - - Skips vendor names that are aliases routing through an aggregator (bare - "openai" -> "openrouter": emitting them would silently switch a user onto an - endpoint they may have no key for), hermes_ids that are aliases of another - canonical profile ("kimi" -> "kimi-coding"), non-api_key auth types (section - 2 handles them with auth-store checks), and providers Hermes cannot route. - PROVIDER_REGISTRY env var names win over models.dev's (which can be wrong). - """ - from agent.models_dev import PROVIDER_TO_MODELS_DEV - from hermes_cli.auth import PROVIDER_REGISTRY, is_runtime_provider_routable - from hermes_cli.models import _AGGREGATOR_PROVIDERS - from hermes_cli.providers import ALIASES - - for hermes_id, mdev_id in PROVIDER_TO_MODELS_DEV.items(): - alias_target = ALIASES.get(hermes_id) - if alias_target and alias_target != hermes_id and alias_target in _AGGREGATOR_PROVIDERS: - continue - canonical = hermes_id - try: - from providers import get_provider_profile - prof = get_provider_profile(hermes_id) - if prof is not None: - canonical = prof.name - except Exception: - pass - if canonical != hermes_id or hermes_id.lower() in seen: - continue - if hermes_id.lower() in excluded or mdev_id.lower() in excluded: - continue - pdata = models_dev_data.get(mdev_id) - if not isinstance(pdata, dict): - continue - pconfig = PROVIDER_REGISTRY.get(hermes_id) - if pconfig and pconfig.auth_type != "api_key": - continue - if not is_runtime_provider_routable(hermes_id): - continue - if pconfig and pconfig.api_key_env_vars: - env_vars = list(pconfig.api_key_env_vars) - else: - env_vars = pdata.get("env", []) - if not isinstance(env_vars, list): - continue - yield hermes_id, mdev_id, pconfig, env_vars - - -def _auth_store_has_provider(*keys: str) -> bool: - """True when ``auth.json`` has a ``providers`` entry under any of *keys*.""" - try: - from hermes_cli.auth import _load_auth_store - store = _load_auth_store() - providers_store = store.get("providers", {}) - return bool(store and any(k in providers_store for k in keys)) - except Exception as exc: - logger.debug("Auth store check failed for %s: %s", keys[0] if keys else "", exc) - return False - - -def _raw_pool_usable(hermes_id: str) -> bool: - """Section-1 pool check: only consult the pool when auth.json lists a raw entry.""" - try: - from hermes_cli.auth import _load_auth_store - store = _load_auth_store() - if store and store.get("credential_pool", {}).get(hermes_id): - return _credential_pool_is_usable(hermes_id, raw_pool_present=True) - except Exception: - pass - return False - - -def _pool_usable(slug: str) -> bool: - try: - return _credential_pool_is_usable(slug) - except Exception as exc: - logger.debug("Credential pool check failed for %s: %s", slug, exc) - return False - - -def _overlay_has_env_creds(pid: str, hermes_slug: str, overlay, read_env) -> bool: - """Section-2 env/SDK credential check shared by the picker and the prefetch scan. - - Vertex authenticates via OAuth2 (service-account JSON / ADC), not an API - key, so it gets its own probe; otherwise the provider is hidden from the - picker even when fully configured. - """ - from hermes_cli.auth import PROVIDER_REGISTRY - - has_creds = False - if overlay.auth_type == "vertex": - try: - from agent.vertex_adapter import has_vertex_credentials - has_creds = has_vertex_credentials() - except Exception as exc: - logger.debug("Vertex credential check failed: %s", exc) - elif overlay.extra_env_vars: - has_creds = any(read_env(ev) for ev in overlay.extra_env_vars) - if not has_creds and overlay.auth_type == "api_key": - for key in (pid, hermes_slug): - pcfg = PROVIDER_REGISTRY.get(key) - if pcfg and pcfg.api_key_env_vars and any(read_env(ev) for ev in pcfg.api_key_env_vars): - return True - return has_creds - - -def _has_fast_aws_sdk_signal() -> bool: - """True when explicit AWS auth config is present in the environment. - - Deliberately avoids botocore's full credential chain: picker discovery runs - for non-Bedrock providers too, and botocore may probe EC2 IMDS - (169.254.169.254) on local machines before returning no credentials. - """ - env = os.environ - if env.get("AWS_BEARER_TOKEN_BEDROCK", "").strip(): - return True - if env.get("AWS_ACCESS_KEY_ID", "").strip() and env.get("AWS_SECRET_ACCESS_KEY", "").strip(): - return True - return any( - env.get(name, "").strip() - for name in ( - "AWS_PROFILE", - "AWS_CONTAINER_CREDENTIALS_RELATIVE_URI", - "AWS_CONTAINER_CREDENTIALS_FULL_URI", - "AWS_WEB_IDENTITY_TOKEN_FILE", - ) - ) - - -def _has_aws_sdk_creds_for_listing(slug: str, current_provider: str) -> bool: - """Credential check for AWS SDK providers in non-runtime discovery. - - The full boto3 chain is only consulted for the *current* provider. - """ - if _has_fast_aws_sdk_signal(): - return True - if str(slug or "").strip().lower() != str(current_provider or "").strip().lower(): - return False - try: - from agent.bedrock_adapter import has_aws_credentials - return bool(has_aws_credentials()) - except Exception: - return False - - -def _is_aws_sdk(pconfig) -> bool: - return bool(pconfig) and getattr(pconfig, "auth_type", "") == "aws_sdk" - - -def _live_or_curated_ids(slug: str, curated: dict, *fallback_keys: str, merge_models_dev: bool = True) -> list: - """Unified pathway: ``cached_provider_model_ids`` so the /model picker sees the - SAME list ``hermes model`` builds (disk-cached), falling back to the curated - static list (merged with models.dev for preferred providers) when live is empty. - """ - from hermes_cli.models import _MODELS_DEV_PREFERRED, _merge_with_models_dev, cached_provider_model_ids - - model_ids = cached_provider_model_ids(slug) - if not model_ids: - for key in fallback_keys or (slug,): - model_ids = curated.get(key, []) - if model_ids: - break - if merge_models_dev and slug in _MODELS_DEV_PREFERRED: - model_ids = _merge_with_models_dev(slug, model_ids) - return model_ids - - -def _aws_live_or_curated_ids(slug: str, curated: dict, *fallback_keys: str) -> list: - """Bedrock: live discovery reflects the active region (eu.*, ap.*) rather than - the static us.* list; any failure falls back to the curated list.""" - from hermes_cli.models import cached_provider_model_ids - - fallback_keys = fallback_keys or (slug,) - try: - ids = cached_provider_model_ids(slug) - if ids: - return ids - except Exception: - pass - for key in fallback_keys: - ids = curated.get(key, []) - if ids: - return ids - return [] - - -def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list: - """Nous serves a huge alphabetical live catalog; the picker shows ONLY the - curated agentic list, augmented with the Portal's free/paid recommendations - (so newly launched models surface without a CLI release) and narrowed by org - policy. Mirrors ``_model_flow_nous`` so GUI pickers match the CLI. A failed - recommendation fetch still yields a policy-filtered curated list. - """ - model_ids = curated.get("nous", []) - try: - from hermes_cli.models import ( - get_pricing_for_provider, - check_nous_free_tier, - union_with_portal_free_recommendations, - union_with_portal_paid_recommendations, - ) - from hermes_cli.auth import get_provider_auth_state - - pricing = get_pricing_for_provider("nous") or {} - try: - portal = (get_provider_auth_state("nous") or {}).get("portal_base_url", "") or "" - except Exception: - portal = "" - if check_nous_free_tier(force_fresh=force_fresh_nous_tier): - model_ids, _ = union_with_portal_free_recommendations(model_ids, pricing, portal) - else: - model_ids, _ = union_with_portal_paid_recommendations(model_ids, pricing, portal) - except Exception: - pass - try: - from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy - - model_ids = restrict_to_nous_policy(model_ids, nous_policy_allowed_ids(), rescue_empty=True) - except Exception: - pass - return model_ids - - -def _cap_models(model_ids: list, max_models: int | None, slug: str = "") -> list: - """Apply ``max_models``; aggregators in ``_UNCAPPED_PICKER_PROVIDERS`` show everything.""" - if slug in _UNCAPPED_PICKER_PROVIDERS or max_models is None: - return model_ids - return model_ids[:max_models] - - -def _norm_url(url: Any) -> str: - return str(url or "").strip().rstrip("/").lower() - - -def _entry_base_url(entry: dict, keys: tuple = ("base_url", "url", "api")) -> str: - for key in keys: - value = entry.get(key, "") - if value: - return value - return "" - - -def _entry_api_mode(entry: dict) -> str | None: - return str(entry.get("api_mode") or entry.get("transport") or "").strip().lower() or None - - -def _credential_identity(inline_api_key: str, key_env: str) -> str: - return inline_api_key if inline_api_key else (f"env:{key_env}" if key_env else "") - - -def _discover_flag(entry: dict): - """``discover_models`` (default True); ``"false"/"no"/"0"`` strings mean False.""" - discover = entry.get("discover_models", True) - if isinstance(discover, str): - discover = discover.lower() not in {"false", "no", "0"} - return discover - - -def _display_prefix(name: str) -> str: - """Text before the per-model separator Hermes's own writer uses ("—" / " - ").""" - for sep in ("—", " - "): - if sep in name: - return name.split(sep)[0].strip() - return name - - -def _discover_endpoint_models( - api_key: str, - api_url: str, - native_catalog_provider: str, - has_explicit_models: bool, - *, - headers: dict | None, - api_mode: str | None, - probe_live: bool, - discovery_allowed: bool, - for_picker: bool, -) -> tuple[list | None, bool]: - """Return ``(models, native_catalog_empty)`` for a custom endpoint row. - - ``probe_live`` runs the native-aware picker fetch; otherwise, when discovery - is allowed, a warm same-fingerprint cache entry still serves the full catalog - with no round-trip. ``has_explicit_models`` gates the *probe* (a network-cost - guard for keyless endpoints that declare a catalog), never the cache read — - applying it to the read re-pins the endpoint to its declared subset. Returns - ``(None, False)`` when nothing usable was found. - """ - timeout = 1.5 if for_picker else 5.0 - if probe_live: - try: - live_models = _fetch_picker_live_models( - api_key, api_url, native_catalog_provider, has_explicit_models, - headers=headers, timeout=timeout, api_mode=api_mode, - ) - is_native = isinstance(live_models, _NativePickerModelList) - if live_models is not None and (live_models or not has_explicit_models or is_native): - return live_models, (is_native and not live_models) - except Exception: - pass - elif discovery_allowed: - try: - from hermes_cli.models import cached_fetch_api_models - - cached_models = cached_fetch_api_models( - api_key, api_url, cache_only=True, timeout=timeout, headers=headers, api_mode=api_mode, - ) - if cached_models: - return cached_models, False - except _MODEL_DISCOVERY_ERRORS: - pass - return None, False - - -def _collect_authed_provider_slugs( - models_dev_data: dict, - curated: dict[str, list[str]], - excluded: list[str], -) -> list[str]: - """Quick-scan which providers have credentials, without fetching model lists. - - Mirrors the credential checks of sections 1, 2 and 2b of - :func:`list_authenticated_providers` but never calls - ``cached_provider_model_ids``; the result feeds - :func:`_prefetch_provider_models_parallel`. Env vars are read through the - per-profile secret scope. AWS SDK providers are skipped (heavier detection). - """ - from agent.models_dev import PROVIDER_TO_MODELS_DEV - from hermes_cli.auth import PROVIDER_REGISTRY - from hermes_cli.providers import HERMES_OVERLAYS - from hermes_cli.models import CANONICAL_PROVIDERS - - excluded_set = {str(p).strip().lower() for p in excluded if p} - slugs: list[str] = [] - seen: set[str] = set() - - for hermes_id, _mdev_id, _pconfig, env_vars in _iter_builtin_candidates(models_dev_data, excluded_set, seen): - if any(_scoped_key_env(ev) for ev in env_vars) or _raw_pool_usable(hermes_id): - slugs.append(hermes_id) - seen.add(hermes_id.lower()) - - mdev_to_hermes = {v: k for k, v in PROVIDER_TO_MODELS_DEV.items()} - for pid, overlay in HERMES_OVERLAYS.items(): - hermes_slug = mdev_to_hermes.get(pid, pid) - if pid.lower() in seen or hermes_slug.lower() in seen: - continue - if pid.lower() in excluded_set or hermes_slug.lower() in excluded_set: - continue - if overlay.auth_type == "aws_sdk": - continue - if ( - _overlay_has_env_creds(pid, hermes_slug, overlay, _scoped_key_env) - or _auth_store_has_provider(pid, hermes_slug) - or _pool_usable(hermes_slug) - ): - slugs.append(hermes_slug) - seen.add(pid.lower()) - seen.add(hermes_slug.lower()) - - for cp in CANONICAL_PROVIDERS: - if cp.slug.lower() in seen or cp.slug.lower() in excluded_set: - continue - cp_config = PROVIDER_REGISTRY.get(cp.slug) - has_creds = bool( - cp_config and cp_config.api_key_env_vars and any(_scoped_key_env(ev) for ev in cp_config.api_key_env_vars) - ) - if has_creds or _auth_store_has_provider(cp.slug) or _pool_usable(cp.slug): - slugs.append(cp.slug) - seen.add(cp.slug.lower()) - - # Nous excluded: its picker branch builds from the curated list and never - # reads the api_key-only cache entry a prefetch would write. - return [s for s in slugs if s != "nous"] - - -@dataclass -class _PickerBuild: - """Mutable state threaded through the ``list_authenticated_providers`` sections.""" - - current_provider: str - current_base_url: str - current_model: str - max_models: int | None - for_picker: bool - force_fresh_nous_tier: bool - probe_custom_providers: bool - probe_current_custom_provider: bool - refresh: bool - excluded: set - curated: dict - results: list = field(default_factory=list) - seen_slugs: set = field(default_factory=set) # lowercase-normalized to catch case variants - # Effective base URLs of every built-in row, so section 4 hides - # ``custom_providers`` entries that duplicate a built-in endpoint. - builtin_endpoints: set = field(default_factory=set) - # (display_name, base_url) pairs emitted by section 3 so section 4 skips - # overlapping ``custom_providers`` rows (callers often pass both). - section3_pairs: set = field(default_factory=set) - current_provider_norm: str = field(init=False) - current_base_url_norm: str = field(init=False) - - def __post_init__(self): - self.current_provider_norm = self.current_provider.lower() - self.current_base_url_norm = self.current_base_url.rstrip("/").lower() - - def can_probe_custom(self, *, row_is_current: bool) -> bool: - return bool(self.probe_custom_providers or (self.probe_current_custom_provider and row_is_current)) - - def record_builtin_endpoint(self, slug: str) -> None: - """Prefer the live env override (e.g. DASHSCOPE_BASE_URL) over the static - inference_base_url so dedup matches what a user typing that URL into - custom_providers would actually hit.""" - try: - from hermes_cli.auth import PROVIDER_REGISTRY - except Exception: - return - pcfg = PROVIDER_REGISTRY.get(slug) - if not pcfg: - return - url = os.environ.get(pcfg.base_url_env_var, "") if getattr(pcfg, "base_url_env_var", "") else "" - normed = _norm_url(url or getattr(pcfg, "inference_base_url", "") or "") - if normed: - self.builtin_endpoints.add(normed) - - def add_builtin_row(self, slug: str, name: str, is_current: bool, model_ids: list, source: str, *, uncapped_ok: bool = True) -> None: - self.results.append({ - "slug": slug, - "name": name, - "is_current": is_current, - "is_user_defined": False, - "models": _cap_models(model_ids, self.max_models, slug if uncapped_ok else ""), - "total_models": len(model_ids), - "source": source, - }) - self.seen_slugs.add(slug.lower()) - self.record_builtin_endpoint(slug) - - -def _lap_builtin_rows(b: _PickerBuild, data: dict, user_providers: dict) -> None: - """Section 1: models.dev-mapped providers with api_key auth.""" - from agent.models_dev import get_provider_info - - for hermes_id, mdev_id, pconfig, env_vars in _iter_builtin_candidates(data, b.excluded, b.seen_slugs): - if not (any(os.environ.get(ev) for ev in env_vars) or _raw_pool_usable(hermes_id)): - continue - model_ids = _live_or_curated_ids(hermes_id, b.curated) - # A providers..models block extends the discovered catalog; - # section 3 cannot emit it later because this row owns the slug. - configured = user_providers.get(hermes_id) if isinstance(user_providers, dict) else None - configured_models = _declared_model_ids(configured.get("models")) if isinstance(configured, dict) else [] - model_ids = list(dict.fromkeys([*configured_models, *model_ids])) - pinfo = get_provider_info(mdev_id) - display_name = pconfig.name if pconfig and pconfig.name else (pinfo.name if pinfo else mdev_id) - b.add_builtin_row( - hermes_id, display_name, b.current_provider in (hermes_id, mdev_id), model_ids, "built-in", - ) - - -def _lap_overlay_rows(b: _PickerBuild, data: dict) -> None: - """Section 2: Hermes-only providers (nous, openai-codex, copilot, opencode-go, ...).""" - from agent.models_dev import PROVIDER_TO_MODELS_DEV - from hermes_cli.providers import HERMES_OVERLAYS - - # HERMES_OVERLAYS keys may be models.dev IDs ("github-copilot") while - # config.yaml uses Hermes IDs ("copilot"). - mdev_to_hermes = {v: k for k, v in PROVIDER_TO_MODELS_DEV.items()} - for pid, overlay in HERMES_OVERLAYS.items(): - hermes_slug = mdev_to_hermes.get(pid, pid) - if pid.lower() in b.seen_slugs or hermes_slug.lower() in b.seen_slugs: - continue - if pid.lower() in b.excluded or hermes_slug.lower() in b.excluded: - continue - - if getattr(overlay, "keyless", False): - has_creds = True # served anonymously (opencode-free) - elif overlay.auth_type == "aws_sdk": - has_creds = _has_aws_sdk_creds_for_listing(hermes_slug, b.current_provider) - else: - has_creds = _overlay_has_env_creds(pid, hermes_slug, overlay, os.environ.get) - # External-process providers (copilot-acp) hold no key/token/pool entry by - # design — the spawned ACP subprocess brings its own auth. "Configured" - # means the executable resolves, which is what get_auth_status() reports; - # without this the has_creds filter hides the provider from every picker. - if not has_creds and overlay.auth_type == "external_process": - try: - from hermes_cli.auth import get_auth_status - _ext_status = get_auth_status(hermes_slug) or {} - has_creds = bool(_ext_status.get("logged_in") or _ext_status.get("configured")) - except Exception as exc: - logger.debug("External-process check failed for %s: %s", pid, exc) - # Auth store / credential pool cover OAuth providers AND api_key providers - # that also support OAuth (anthropic via Claude Code credential files). - if not has_creds: - has_creds = _auth_store_has_provider(pid, hermes_slug) - if not has_creds: - # Full auto-seeding pool check catches external stores (Codex CLI - # ~/.codex/auth.json) not yet in auth.json. - try: - if _credential_pool_is_usable(hermes_slug): - has_creds = True - elif b.for_picker: - # Show providers whose pool is entirely in cooldown: limits are - # per-model for many providers, so another model may work. - try: - from agent.credential_pool import load_pool - has_creds = load_pool(hermes_slug).has_credentials() - except Exception: - pass - except Exception as exc: - logger.debug("Credential pool check failed for %s: %s", hermes_slug, exc) - if not has_creds and hermes_slug == "anthropic": - # The pool gates anthropic behind is_provider_explicitly_configured() - # (aux tasks must not consume Claude Code tokens); the picker is - # discovery-oriented, so read the external credential files directly. - try: - from agent.anthropic_adapter import read_claude_code_credentials, read_hermes_oauth_credentials - hermes_creds = read_hermes_oauth_credentials() - cc_creds = read_claude_code_credentials() - if (hermes_creds and hermes_creds.get("accessToken")) or (cc_creds and cc_creds.get("accessToken")): - has_creds = True - except Exception as exc: - logger.debug("Anthropic external creds check failed: %s", exc) - if not has_creds: - continue - - if hermes_slug in {"openai-codex", "copilot", "copilot-acp"}: - # Live OAuth-backed discovery so Pro-only Codex slugs not in the static - # catalog appear; falls back to curated when unreachable. - from hermes_cli.models import cached_provider_model_ids - model_ids = cached_provider_model_ids(hermes_slug) - elif overlay.auth_type == "aws_sdk": - model_ids = _aws_live_or_curated_ids(hermes_slug, b.curated, hermes_slug, pid) - elif hermes_slug == "nous": - model_ids = _nous_picker_model_ids(b.curated, b.force_fresh_nous_tier) - else: - model_ids = _live_or_curated_ids(hermes_slug, b.curated, hermes_slug, pid) - b.add_builtin_row( - hermes_slug, get_label(hermes_slug), b.current_provider in (hermes_slug, pid), model_ids, "hermes", - ) - b.seen_slugs.add(pid.lower()) - - -def _lap_canonical_rows(b: _PickerBuild) -> None: - """Section 2b: CANONICAL_PROVIDERS missed by sections 1/2.""" - from hermes_cli.auth import PROVIDER_REGISTRY - try: - from hermes_cli.models import CANONICAL_PROVIDERS - except ImportError: - CANONICAL_PROVIDERS = [] - - for cp in CANONICAL_PROVIDERS: - if cp.slug.lower() in b.seen_slugs or cp.slug.lower() in b.excluded: - continue - cp_config = PROVIDER_REGISTRY.get(cp.slug) - has_creds = False - if cp_config and cp_config.api_key_env_vars: - lit = {ev for ev in cp_config.api_key_env_vars if os.environ.get(ev)} - has_creds = bool(lit) - # A regional "-cn" twin lit only by key vars shared with its non-CN - # sibling is a phantom row: hide it unless it is the current provider, - # and only when it has a dedicated var of its own the user could set. - sib = PROVIDER_REGISTRY.get(cp.slug[:-3]) if cp.slug.endswith("-cn") else None - sib_vars = set(sib.api_key_env_vars) if sib else set() - if lit and lit <= sib_vars < set(cp_config.api_key_env_vars) and cp.slug != b.current_provider: - continue - if not has_creds: - has_creds = _auth_store_has_provider(cp.slug) or _pool_usable(cp.slug) - if not has_creds and _is_aws_sdk(cp_config): - has_creds = _has_aws_sdk_creds_for_listing(cp.slug, b.current_provider) - if not has_creds: - continue - if _is_aws_sdk(cp_config): - model_ids = _aws_live_or_curated_ids(cp.slug, b.curated) - else: - model_ids = _live_or_curated_ids(cp.slug, b.curated, merge_models_dev=False) - b.add_builtin_row( - cp.slug, cp.label, cp.slug == b.current_provider, model_ids, "canonical", uncapped_ok=False, - ) - - -def _lap_user_provider_rows(b: _PickerBuild, user_providers: dict) -> None: - """Section 3: ``providers:`` dict entries, grouped by (api_url, credential, - api_mode, extra_headers) so keyed providers on one endpoint with the same - wire protocol collapse into one row (e.g. two Palantir Claude entries -> - one "Palantir Claude" row); a different key_env/api_mode/headers keeps - distinct rows since the wire protocol or tenant differs.""" - from collections import OrderedDict - from hermes_cli.config import coerce_provider_id, is_provider_enabled - - ep_groups: "OrderedDict[tuple, dict]" = OrderedDict() - for ep_name, ep_cfg in user_providers.items(): - if not isinstance(ep_cfg, dict) or not is_provider_enabled(ep_cfg): - continue - if ep_name.lower() in b.seen_slugs: - continue - display_name = coerce_provider_id(ep_cfg.get("name")) or ep_name - api_url = _entry_base_url(ep_cfg, ("base_url", "api", "url")) - key_env = str(ep_cfg.get("key_env") or ep_cfg.get("api_key_env") or "").strip() - inline_api_key = str(ep_cfg.get("api_key", "") or "").strip() - api_mode = _entry_api_mode(ep_cfg) - headers_identity = tuple(sorted(_extra_headers_from_config(ep_cfg).items())) - group_key = (_norm_url(api_url), _credential_identity(inline_api_key, key_env), api_mode, headers_identity) - - # ``default_model`` is the legacy key; ``model`` matches custom_providers. - default_model = ep_cfg.get("default_model", "") or ep_cfg.get("model", "") - entry_models = [default_model] if default_model else [] - for model_id in _declared_model_ids(ep_cfg.get("models", [])): - if model_id not in entry_models: - entry_models.append(model_id) - - if group_key not in ep_groups: - # Strip the per-model suffix and trailing version tokens ("Palantir - # Claude 4.7 Opus" -> "Palantir Claude"): cut at the first token with - # a digit, only when >=2 words remain (avoids over-trimming). - grp_display = _display_prefix(display_name) - toks = grp_display.split() - cut_at = next((i for i, t in enumerate(toks) if any(c.isdigit() for c in t.strip(".,()"))), None) - if cut_at is not None and cut_at >= 2: - grp_display = " ".join(toks[:cut_at]).strip() - ep_groups[group_key] = { - "slug": ep_name, # first ep_name encountered - "name": grp_display or display_name, - "api_url": api_url, - "models": [], - "has_explicit_models": False, - "ep_cfg": ep_cfg, - "raw_names": [], - "aliases": set(), - } - grp = ep_groups[group_key] - for m in entry_models: - if m and m not in grp["models"]: - grp["models"].append(m) - # A singular default_model/model is only the active selection and must - # not suppress discovery; dict-shaped ``models:`` is context_length - # metadata, not an allowlist — see ``_models_config_is_allowlist``. - if _models_config_is_allowlist(ep_cfg.get("models"), _entry_models_discovered(ep_cfg)): - grp["has_explicit_models"] = True - grp["raw_names"].append(display_name) - grp["aliases"].update(custom_provider_aliases(display_name, str(ep_name))) - - for grp in ep_groups.values(): - ep_cfg, ep_name, display_name, api_url = grp["ep_cfg"], grp["slug"], grp["name"], grp["api_url"] - models_list = list(grp["models"]) - # Official OpenAI rows often have base_url but no models: dict — avoid a - # misleading zero count. - if not models_list and base_url_host_matches(str(api_url).strip().lower(), "api.openai.com"): - models_list = list(b.curated.get("openai") or []) - - # Probe policy (mirrors section 4): with an api_key always probe; without - # one, skip only when an allowlist-shaped ``models:`` narrows the endpoint. - api_key = str(ep_cfg.get("api_key", "") or "").strip() - if not api_key: - key_env = str(ep_cfg.get("key_env") or ep_cfg.get("api_key_env") or "").strip() - api_key = _scoped_key_env(key_env) if key_env else "" - has_explicit_models = bool(grp.get("has_explicit_models")) - ep_url_norm = _norm_url(api_url) - ep_aliases = {str(alias).lower() for alias in grp.get("aliases", set())} - is_current = ( - str(ep_name).strip().lower() == b.current_provider_norm - or b.current_provider_norm in ep_aliases - or ( - b.current_provider_norm == "custom" - and bool(b.current_base_url_norm) - and ep_url_norm == b.current_base_url_norm - ) - ) - discovery_allowed = bool(api_url) and _discover_flag(ep_cfg) - discovered, native_catalog_empty = _discover_endpoint_models( - api_key, - api_url, - ep_name if str(ep_name).strip().lower() in {"ollama", "custom:ollama"} else "custom", - has_explicit_models, - headers=_extra_headers_from_config(ep_cfg) or None, - api_mode=ep_cfg.get("api_mode"), - probe_live=( - discovery_allowed - and (bool(api_key) or not has_explicit_models) - and b.can_probe_custom(row_is_current=is_current) - ), - discovery_allowed=discovery_allowed, - for_picker=b.for_picker, - ) - if discovered is not None: - models_list = discovered - - b.results.append({ - "slug": ep_name, - "name": display_name, - "is_current": is_current, - "is_user_defined": True, - "models": models_list, - "total_models": len(models_list) if models_list else 0, - "source": "user-config", - "api_url": api_url, - "native_catalog_empty": native_catalog_empty, - }) - b.seen_slugs.add(ep_name.lower()) - b.seen_slugs.update(ep_aliases) - # Record every raw member name so section 4 can match per-model - # custom_providers rows even though the group label was collapsed. - for raw_name in grp.get("raw_names") or [display_name]: - pair = (str(raw_name).strip().lower(), ep_url_norm) - if pair[0] and pair[1]: - b.section3_pairs.add(pair) - b.seen_slugs.add(custom_provider_slug(raw_name).lower()) - pair = (str(display_name).strip().lower(), ep_url_norm) - if pair[0] and pair[1]: - b.section3_pairs.add(pair) - - -def _lap_bare_custom_row(b: _PickerBuild, custom_providers: list | None) -> None: - """Section 3b: ``model.provider: custom`` + ``model.base_url`` with no named - providers:/custom_providers row — surface it so /model does not look like it - ignored config.yaml.""" - if not (b.current_provider_norm == "custom" and b.current_base_url and "custom" not in b.seen_slugs): - return - if any( - isinstance(cp, dict) and _norm_url(_entry_base_url(cp)) == _norm_url(b.current_base_url) - for cp in (custom_providers or []) - ): - return - api_url = str(b.current_base_url).strip().rstrip("/") - models = [b.current_model] if b.current_model else [] - native_catalog_empty = False - try: - discovered, native_catalog_empty = _discover_endpoint_models( - "", api_url, "custom", False, - headers=None, api_mode=None, - probe_live=bool(b.refresh or b.probe_current_custom_provider), - discovery_allowed=True, - for_picker=b.for_picker, - ) - if discovered is not None: - models = discovered - except Exception: - pass - b.results.append({ - "slug": "custom", - "name": "Custom endpoint", - "is_current": True, - "is_user_defined": True, - "models": _cap_models(models, b.max_models), - "total_models": len(models), - "source": "model-config", - "api_url": api_url, - "native_catalog_empty": native_catalog_empty, - }) - b.seen_slugs.add("custom") - - -def _lap_custom_provider_rows(b: _PickerBuild, custom_providers: list) -> None: - """Section 4: ``custom_providers:`` entries (one model each) grouped into one - row per (endpoint, credential identity, api_mode, extra_headers, display - prefix). Four "Ollama — X" entries on one host become one "Ollama" row; - distinct prefixes sharing a proxy URL keep their own rows.""" - from collections import OrderedDict - from hermes_cli.config import coerce_provider_id - - groups: "OrderedDict[tuple, dict]" = OrderedDict() - for entry in custom_providers: - if not isinstance(entry, dict): - continue - raw_name = coerce_provider_id(entry.get("name")) - api_url = str(_entry_base_url(entry) or "").strip().rstrip("/") - if not raw_name or not api_url: - continue - inline_api_key = str(entry.get("api_key") or "").strip() - key_env = str(entry.get("key_env") or "").strip() - api_key = inline_api_key or _scoped_key_env(key_env) - api_mode = _entry_api_mode(entry) - discover = _discover_flag(entry) - entry_extra_headers = _extra_headers_from_config(entry) - prefix = _display_prefix(raw_name) - group_key = ( - api_url, _credential_identity(inline_api_key, key_env), api_mode, - tuple(sorted(entry_extra_headers.items())), prefix.lower(), - ) - if group_key not in groups: - display_name = prefix or raw_name - groups[group_key] = { - "slug": custom_provider_slug(display_name, str(entry.get("provider_key") or "").strip()), - "name": display_name, - "api_url": api_url, - "api_key": api_key, - "models": [], - "has_explicit_models": False, - "discover_models": discover, - "api_mode": api_mode, - "extra_headers": entry_extra_headers, - "aliases": set(), - } - else: - if api_key and not groups[group_key].get("api_key"): - groups[group_key]["api_key"] = api_key - if not discover: # one opt-out pins the whole grouped row - groups[group_key]["discover_models"] = False - grp = groups[group_key] - grp["aliases"].update(custom_provider_aliases(raw_name, str(entry.get("provider_key") or ""))) - # ``model:`` is only the active selection; every configured model lives - # under ``models:`` (dict written by _save_custom_provider). - default_model = (entry.get("model") or "").strip() - if default_model and default_model not in grp["models"]: - grp["models"].append(default_model) - models_field = entry.get("models", {}) - if _models_config_is_allowlist(models_field, _entry_models_discovered(entry)): - grp["has_explicit_models"] = True - for model_id in _declared_model_ids(models_field): - if model_id not in grp["models"]: - grp["models"].append(model_id) - - section4_slugs: set = set() - current_url_group_count = sum( - 1 for grp in groups.values() - if b.current_base_url_norm and _norm_url(grp["api_url"]) == b.current_base_url_norm - ) - for grp in groups.values(): - api_url, api_key, slug = grp["api_url"], grp.get("api_key", ""), grp["slug"] - # Slug claimed by a built-in/overlay/providers: row -> skip (don't shadow). - if slug.lower() in b.seen_slugs and slug.lower() not in section4_slugs: - continue - # Two custom endpoints with the same cleaned name: suffix a counter so - # both stay visible. - if slug.lower() in section4_slugs: - base_slug, n = slug, 2 - while f"{base_slug}-{n}".lower() in b.seen_slugs: - n += 1 - slug = f"{base_slug}-{n}" - grp["slug"] = slug - grp_url_norm = _norm_url(api_url) - pair_key = (str(grp["name"]).strip().lower(), grp_url_norm) - if pair_key[0] and pair_key[1] and pair_key in b.section3_pairs: - continue - # A built-in row already represents this endpoint (e.g. "my-dashscope" - # vs the alibaba-coding-plan row): keep the built-in, hide the shadow. - if grp_url_norm and grp_url_norm in b.builtin_endpoints: - continue - is_current = ( - slug.lower() == b.current_provider_norm - or b.current_provider_norm in {str(alias).lower() for alias in grp.get("aliases", set())} - ) or ( - b.current_provider_norm == "custom" - and bool(b.current_base_url_norm) - and grp_url_norm == b.current_base_url_norm - and current_url_group_count == 1 - ) - # Probe policy: with an api_key live /models is the source of truth (replace - # the partial ``models:`` subset); without one, an allowlist-shaped - # ``models:`` narrows a public endpoint and skips the probe. A dict-shaped - # ``models:`` is metadata, so still probe; pin with discover_models: false. - has_explicit_models = bool(grp.get("has_explicit_models")) - discovery_allowed = bool(api_url) and grp.get("discover_models", True) - probe_live = ( - discovery_allowed - and (bool(api_key) or not has_explicit_models) - and b.can_probe_custom(row_is_current=is_current) - ) - discovered, native_catalog_empty = _discover_endpoint_models( - api_key, - api_url, - "ollama" if "ollama" in {str(slug).strip().lower(), str(grp.get("name") or "").strip().lower()} else "custom", - has_explicit_models, - headers=grp.get("extra_headers") or None, - api_mode=grp.get("api_mode"), - probe_live=probe_live, - discovery_allowed=discovery_allowed, - for_picker=b.for_picker, - ) - if discovered is not None: - grp["models"] = discovered - if probe_live: - # A successful live probe persists the catalog for no-probe surfaces. - try: - _save_discovered_models_to_config( - api_url, discovered, api_mode=grp.get("api_mode"), headers=grp.get("extra_headers") or None, - ) - except Exception: - pass - b.results.append({ - "slug": slug, - "name": grp["name"], - "is_current": is_current, - "is_user_defined": True, - "models": grp["models"], - "total_models": len(grp["models"]), - "source": "user-config", - "api_url": grp["api_url"], - "native_catalog_empty": native_catalog_empty, - }) - b.seen_slugs.add(slug.lower()) - section4_slugs.add(slug.lower()) - - -def _build_curated_lists(current_provider: str, current_base_url: str, current_model: str) -> dict[str, list[str]]: - """Curated model lists keyed by hermes provider id, plus the dynamic ones - (nous manifest, Ollama Cloud, LM Studio live probe).""" - from hermes_cli.models import OPENROUTER_MODELS, _PROVIDER_MODELS, get_curated_nous_model_ids - - curated: dict[str, list[str]] = dict(_PROVIDER_MODELS) - curated["openrouter"] = [mid for mid, _ in OPENROUTER_MODELS] - # Remote model-catalog manifest so new Portal models surface without a - # release; falls back to the in-repo snapshot when unreachable. - curated["nous"] = get_curated_nous_model_ids() - if "ollama-cloud" not in curated: - from hermes_cli.models import fetch_ollama_cloud_models - curated["ollama-cloud"] = fetch_ollama_cloud_models() - # LM Studio has no static catalog: probe its native endpoint live. Base URL - # precedence: LM_BASE_URL > active config base_url (when current) > default. - # On auth rejection / unreachable, fall back to the current model so the - # picker still shows something offline. - is_current_lmstudio = current_provider.strip().lower() == "lmstudio" - if "lmstudio" not in curated and (os.environ.get("LM_API_KEY") or os.environ.get("LM_BASE_URL") or is_current_lmstudio): - from hermes_cli.models import fetch_lmstudio_models - from hermes_cli.auth import AuthError - lm_base = ( - os.environ.get("LM_BASE_URL") - or (current_base_url if is_current_lmstudio and current_base_url else None) - or "http://127.0.0.1:1234/v1" - ) - try: - live = fetch_lmstudio_models(api_key=os.environ.get("LM_API_KEY", ""), base_url=lm_base, timeout=1.5) - except AuthError: - live = [] - if not live and is_current_lmstudio and current_model: - live = [current_model] - curated["lmstudio"] = live - return curated - - -def list_authenticated_providers( - current_provider: str = "", - current_base_url: str = "", - user_providers: dict = None, - custom_providers: list | None = None, - *, - force_fresh_nous_tier: bool = False, - max_models: int | None = None, - current_model: str = "", - refresh: bool = False, - probe_custom_providers: bool = True, - probe_current_custom_provider: bool = False, - for_picker: bool = False, - excluded_providers: list | None = None, -) -> List[dict]: - """Detect which providers have credentials and list their curated models. - - Uses the curated lists from hermes_cli/models.py (OPENROUTER_MODELS, - _PROVIDER_MODELS) — hand-picked agentic models, NOT the full models.dev - catalog. Only providers with API keys set or user-defined endpoints appear. - - Returns a list of dicts: ``slug`` (the --provider value), ``name``, - ``is_current``, ``is_user_defined``, ``models`` (up to max_models), - ``total_models``, ``source`` ("built-in", "hermes", "canonical", - "user-config", "model-config"). - - ``force_fresh_nous_tier`` bypasses the short Nous tier cache for explicit - account-sensitive flows; picker opens should leave it false. - ``refresh`` busts the per-provider model-id disk cache up front so every row - re-fetches live — for an explicit user "refresh models" action only. - ``probe_custom_providers`` controls live ``/models`` discovery for saved - custom endpoints (default true for CLI parity; GUI opens pass false). - ``probe_current_custom_provider`` probes only the currently-selected custom - endpoint so its list matches without blocking on offline ones. - """ - from agent.models_dev import fetch_models_dev - from hermes_cli.config import coerce_provider_id, stringify_provider_map - - # Explicit refresh: drop every cached model-id list so the calls below all - # re-fetch live. A stale cache can fall back to the curated static list when - # its live fetch fails, silently dropping live-only models the user had seen. - if refresh: - try: - from hermes_cli.models import clear_provider_models_cache - clear_provider_models_cache() - except Exception: - pass - - # PyYAML parses unquoted numeric names (`provider: 2070`) as int. - current_provider = coerce_provider_id(current_provider) - current_base_url = str(current_base_url or "").strip() - current_model = str(current_model or "").strip() - user_providers = stringify_provider_map(user_providers) - data = fetch_models_dev() - - b = _PickerBuild( - current_provider=current_provider, - current_base_url=current_base_url, - current_model=current_model, - max_models=max_models, - for_picker=for_picker, - force_fresh_nous_tier=force_fresh_nous_tier, - probe_custom_providers=probe_custom_providers, - probe_current_custom_provider=probe_current_custom_provider, - refresh=refresh, - # A single entry like ``copilot`` hides the provider under every key it - # surfaces as (hermes_id / mdev_id / canonical slug). - excluded={str(p).strip().lower() for p in (excluded_providers or []) if p}, - curated=_build_curated_lists(current_provider, current_base_url, current_model), - ) - - # Warm the disk cache in parallel before the serial section loops, which - # otherwise stack 15-30s of live /v1/models round-trips on a cold cache. - # Skipped when refresh=True (serial path force-refreshes) and for <=3 - # providers (serial is fast enough; avoids thread-pool overhead). - prefetch_slugs = [] if refresh else _collect_authed_provider_slugs(data, b.curated, excluded_providers or []) - if len(prefetch_slugs) > 3: - try: - _prefetch_provider_models_parallel(prefetch_slugs) - except Exception: - pass # best-effort; serial path still works - - _lap_builtin_rows(b, data, user_providers) - _lap_overlay_rows(b, data) - _lap_canonical_rows(b) - if user_providers and isinstance(user_providers, dict): - _lap_user_provider_rows(b, user_providers) - _lap_bare_custom_row(b, custom_providers) - if custom_providers and isinstance(custom_providers, list): - _lap_custom_provider_rows(b, custom_providers) - results = b.results - - # ``providers..enabled: false`` post-filter covers built-in rows - # (sections 1-2) that bypass the per-section gate; matched by slug and - # ``provider_id``. - try: - from hermes_cli.config import is_provider_enabled - if isinstance(user_providers, dict): - disabled = { - str(name).strip().lower() - for name, cfg in user_providers.items() - if isinstance(cfg, dict) and not is_provider_enabled(cfg) - } - if disabled: - results = [ - r for r in results - if str(r.get("provider_id", "")).strip().lower() not in disabled - and str(r.get("slug", "")).strip().lower() not in disabled - ] - except Exception: - pass - - # A custom/uncurated model set via `/model /` would be - # invisible in every picker (main and MoA slot pickers read these rows); - # inject it at the front of the current provider's row as a uniform post-pass. - if current_model: - for row in results: - if not row.get("is_current") or row.get("native_catalog_empty"): - continue - models = row.get("models") or [] - if current_model not in models: - row["models"] = [current_model, *models] - row["total_models"] = row.get("total_models", len(models)) + 1 - break - - # Current provider first, then by model count descending - results.sort(key=lambda r: (not r["is_current"], -r["total_models"])) - return results - - -def _prepend_moa_picker_provider(providers: List[dict], current_provider: str = "") -> List[dict]: - """Add the virtual MoA provider row used by interactive model pickers. - - ``list_authenticated_providers()`` only returns real/auth-backed providers. - The CLI model inventory adds MoA separately so named presets appear next to - normal providers; gateway pickers call ``list_picker_providers()`` directly, - so they need the same virtual row here. Reuse the inventory's single row - builder so the row shape stays defined in one place. - """ - try: - from hermes_cli.inventory import _moa_provider_row - - moa_row = _moa_provider_row(current_provider) - if moa_row is None: - return providers - return [moa_row] + [p for p in providers if str(p.get("slug", "")).lower() != "moa"] - except Exception: - return providers - - -def list_picker_providers( - current_provider: str = "", - current_base_url: str = "", - user_providers: dict = None, - custom_providers: list | None = None, - max_models: int | None = None, - current_model: str = "", - include_moa: bool = False, - excluded_providers: list | None = None, -) -> List[dict]: - """Interactive-picker variant of :func:`list_authenticated_providers`. - - Post-processes the base list so the ``/model`` picker (Telegram/Discord - inline keyboards) only surfaces models that are actually callable in the - current install: - - - OpenRouter's model list is replaced with the output of - :func:`hermes_cli.models.fetch_openrouter_models`, which filters the - curated ``OPENROUTER_MODELS`` snapshot against the live OpenRouter - catalog. IDs the live catalog no longer carries drop out, so the - picker never offers a model the user can't call. - - Provider rows whose model list ends up empty are dropped, except - custom endpoints (``is_user_defined=True`` with an ``api_url``) where - the user may supply their own model set through config. - - All other providers and metadata fields are passed through unchanged. - The typed ``/model `` path is unaffected -- only the interactive - picker payload is narrowed. - """ - from hermes_cli.models import fetch_openrouter_models - - providers = list_authenticated_providers( - current_provider=current_provider, - current_base_url=current_base_url, - user_providers=user_providers, - custom_providers=custom_providers, - max_models=max_models, - current_model=current_model, - for_picker=True, - excluded_providers=excluded_providers, - ) - if include_moa: - providers = _prepend_moa_picker_provider(providers, current_provider=current_provider) - - filtered: List[dict] = [] - for p in providers: - slug = str(p.get("slug", "")).lower() - if slug == "openrouter": - try: - live = fetch_openrouter_models() - live_ids = [mid for mid, _ in live] - except Exception: - live_ids = list(p.get("models", [])) - p = dict(p) - p["models"] = live_ids[:max_models] if max_models is not None else live_ids - p["total_models"] = len(live_ids) - - has_models = bool(p.get("models")) - is_custom_endpoint = bool(p.get("is_user_defined")) and bool(p.get("api_url")) - if not has_models and not is_custom_endpoint: - continue - filtered.append(p) - - return filtered diff --git a/hermes_cli/model_switch_providers.py b/hermes_cli/model_switch_providers.py new file mode 100644 index 0000000000..9d3874ac49 --- /dev/null +++ b/hermes_cli/model_switch_providers.py @@ -0,0 +1,1504 @@ +"""Picker provider listing: credential discovery, curated/live model lists, row builders for list_authenticated_providers / list_picker_providers, and the parallel cache prefetch. + +Split out of ``hermes_cli/model_switch.py``; every moved name is re-imported there so +``hermes_cli.model_switch.`` keeps resolving (and monkeypatching) as before. +""" + +from __future__ import annotations + +import logging +import http.client +import os +import time +import threading as _threading +from dataclasses import dataclass, field +from typing import Any, List, Optional +from hermes_cli.providers import ( + custom_provider_aliases, + custom_provider_slug, + get_label, +) +from utils import base_url_host_matches + +# Log-record parity with the origin module. +logger = logging.getLogger("hermes_cli.model_switch") + + +# Providers whose picker model list should NOT be capped by max_models. +# OpenCode Zen / Go are aggregators whose full catalogs (70+ models each) must +# be visible so users can pick any model they have access to. +_UNCAPPED_PICKER_PROVIDERS: frozenset[str] = frozenset({"opencode-zen", "opencode-go"}) + + +def _save_discovered_models_to_config( + api_url: str, + model_ids: list[str], + *, + api_mode: Optional[str] = None, + headers: Optional[dict[str, str]] = None, +) -> None: + """Persist discovered models into ``custom_providers`` in config.yaml. + + Called after a successful ``/v1/models`` probe so that the next read + with ``discover_models: false`` uses the cached list instead of a stale + or minimal manually-configured subset. + + Matches entries by ``base_url`` (trailing-slash-normalised). A failed + config write is swallowed — the picker still shows the live models for + this session. + """ + from hermes_cli.model_switch import _extra_headers_from_config + if not api_url or not model_ids: + return + try: + from hermes_cli.config import load_config, save_config + + cfg = load_config() + providers = cfg.get("custom_providers") or [] + if not isinstance(providers, list): + return + + norm_url = api_url.strip().rstrip("/").lower() + changed = False + for entry in providers: + if not isinstance(entry, dict): + continue + entry_url = (entry.get("base_url", "") or entry.get("url", "")).strip() + if entry_url.rstrip("/").lower() != norm_url: + continue + entry_mode = str( + entry.get("api_mode") or entry.get("transport") or "" + ).strip().lower() or None + if entry_mode != api_mode: + continue + if headers is not None: + entry_headers = _extra_headers_from_config(entry) + if entry_headers != headers: + continue + existing = entry.get("models") + legacy_discovered = ( + isinstance(existing, dict) + and existing.get("__discovered_model_catalog__") is True + ) + entry_discovered = ( + entry.get("models_discovered") is True or legacy_discovered + ) + # Preserve per-model metadata: when ``models`` is a mapping + # (e.g. ``{"model-a": {"context_length": 8192}}``) or a list of + # dicts (e.g. ``[{"id": "model-a", "context_length": 8192}]``), + # the user has curated metadata per model — do not replace it. + # A mapping Hermes itself discovered (``models_discovered: true`` + # or the legacy in-mapping sentinel) is ours to refresh. + if isinstance(existing, dict) and not entry_discovered: + continue + if isinstance(existing, list) and any( + isinstance(m, dict) for m in existing + ): + continue + # Only update when models are stale — avoids unnecessary + # config writes on every picker open. A legacy-shape entry + # (sentinel inside ``models``) is always rewritten so the next + # save migrates it to the clean entry-level flag. + if isinstance(existing, list) and existing == model_ids: + continue + if ( + isinstance(existing, dict) + and entry_discovered + and not legacy_discovered + and list(existing) == model_ids + ): + continue + entry["models"] = {model_id: {} for model_id in model_ids} + entry["models_discovered"] = True + changed = True + + if changed: + cfg["custom_providers"] = providers + save_config(cfg) + except Exception: + pass + + +_MODEL_DISCOVERY_ERRORS = ( + ImportError, + OSError, + RuntimeError, + TimeoutError, + TypeError, + ValueError, + http.client.HTTPException, +) + + +class _NativePickerModelList(list[str]): + """A successful native catalog, including an authoritative empty one.""" + + +def _fetch_picker_live_models( + api_key: str, + api_url: str, + native_catalog_provider: str, + preserve_native_models: bool, + headers: dict[str, str] | None = None, + timeout: float = 5.0, + api_mode: str | None = None, +) -> list[str] | None: + """Fetch picker models with native Ollama and cached generic discovery.""" + from hermes_cli.models import ( + _get_ollama_native_headers, + _normalize_openai_base_url, + cached_fetch_api_models, + fetch_ollama_local_models, + should_use_ollama_native_catalog, + ) + + candidate_headers = _get_ollama_native_headers(api_url, api_key=api_key) + caller_has_authorization = any( + key.lower() == "authorization" for key in (headers or {}) + ) + if caller_has_authorization: + for key in tuple(candidate_headers): + if key.lower() == "authorization": + del candidate_headers[key] + if headers: + for key in tuple(candidate_headers): + if any(key.lower() == existing.lower() for existing in headers): + del candidate_headers[key] + candidate_headers.update(headers) + if api_key and not caller_has_authorization: + for key in tuple(candidate_headers): + if key.lower() == "authorization": + del candidate_headers[key] + candidate_headers["Authorization"] = f"Bearer {api_key}" + use_native = should_use_ollama_native_catalog( + native_catalog_provider, api_url, headers=candidate_headers or None + ) + resolved_headers = candidate_headers or None if use_native else headers + + if use_native: + if preserve_native_models: + return None + native_models = fetch_ollama_local_models( + api_url, timeout=timeout, headers=resolved_headers + ) + if native_models is not None: + return _NativePickerModelList(native_models) + # A failed native probe is not authoritative: retry the cached generic + # OpenAI-compatible catalog before reporting no models. + return cached_fetch_api_models( + api_key, + _normalize_openai_base_url(api_url), + timeout=timeout, + headers=resolved_headers, + api_mode=api_mode, + ) + generic_models = cached_fetch_api_models( + api_key, + api_url, + timeout=timeout, + headers=resolved_headers, + api_mode=api_mode, + ) + return generic_models if generic_models else None + + +# Process-level guard so the picker prewarm thread is spawned at most once per +# process — mirrors run_agent's _openrouter_prewarm_done. Without a guard a +# long-lived process (or repeated triggers) would leak one OS thread per call. +_picker_prewarm_done = _threading.Event() + + +def _credential_pool_is_usable(provider: str, *, raw_pool_present: bool = False) -> bool: + """Return whether *provider* has a credential that can be selected now. + + ``auth.json`` historically allowed opaque token-style pool values that do + not deserialize into ``PooledCredential`` entries. Preserve visibility for + those legacy values, but when a real pool exists its availability state is + authoritative: an all-exhausted/dead pool is not authenticated. + """ + try: + from agent.credential_pool import load_pool + + pool = load_pool(provider) + if pool.has_credentials(): + return pool.has_available() + except Exception: + pass + return raw_pool_present + + +def prewarm_picker_cache_async() -> Optional["_threading.Thread"]: + """Warm the provider-models disk cache in a background daemon thread. + + The no-args ``/model`` picker calls ``list_authenticated_providers()``, + which fetches each authenticated provider's live ``/v1/models`` list on a + cold/stale cache. Those fetches are independent HTTP round-trips but run + serially, so the first ``/model`` open in a session (or any open after the + 1h cache TTL expires) blocks ~1-2s on the user's critical path. + + This pre-warms that exact path off-thread during idle session time: it + runs ``list_authenticated_providers()`` once, which populates + ``provider_models_cache.json`` for every authed provider. By the time the + user types ``/model``, the picker hits the warm disk cache and renders in + ~100ms. + + Fire-and-forget. Process-level Event guard ensures it runs at most once. + Fully exception-isolated — a slow or offline provider can never affect the + session. Returns the spawned thread (for tests) or None if already warmed. + """ + from hermes_cli.model_switch import list_authenticated_providers + if _picker_prewarm_done.is_set(): + return None + _picker_prewarm_done.set() + + def _warm() -> None: + try: + from hermes_cli.inventory import load_picker_context + + ctx = load_picker_context() + # Calling this is what populates cached_provider_model_ids() -> + # provider_models_cache.json for each authed provider. We discard + # the result; the side effect (warm disk cache) is the point. + list_authenticated_providers( + current_provider=ctx.current_provider, + current_base_url=ctx.current_base_url, + current_model=ctx.current_model, + user_providers=ctx.user_providers, + custom_providers=ctx.custom_providers, + excluded_providers=ctx.excluded_providers or [], + ) + except Exception: + # Best-effort warmup — never surface errors into the session. + logger.debug("picker cache prewarm failed", exc_info=True) + + t = _threading.Thread(target=_warm, daemon=True, name="picker-cache-prewarm") + t.start() + return t + + +_PARALLEL_PREFETCH_WORKERS = 8 + + +def _prefetch_provider_models_parallel(provider_slugs: list[str]) -> None: + """Fetch model catalogs for multiple providers in parallel. + + Only providers whose cache entry is stale or missing are fetched; fresh + entries are skipped to avoid unnecessary network calls. Each worker uses + :func:`update_provider_cache_entry` (thread-safe) to persist its result, + so concurrent writes to ``provider_models_cache.json`` don't clobber each + other. + + :param provider_slugs: Hermes provider IDs to prefetch (e.g. ``["openrouter", + "anthropic", "deepseek"]``). Unknown providers are silently skipped. + """ + from hermes_cli.models import cached_provider_model_ids + + # Quick-stale-check: skip providers whose cache is already fresh so we + # don't waste network calls on a warm cache. We check staleness the same + # way cached_provider_model_ids does internally: load the cache, compare + # age to TTL. This is a read-only check — if the cache file changes + # between this check and the actual fetch, cached_provider_model_ids will + # still do the right thing (it re-reads the cache internally). + from hermes_cli.models import ( + _load_provider_models_cache, + _credential_fingerprint, + _PROVIDER_MODELS_CACHE_TTL, + normalize_provider, + ) + + now = time.time() + stale_slugs: list[str] = [] + cache = _load_provider_models_cache() + for slug in provider_slugs: + normalized = normalize_provider(slug) or (slug or "") + if not normalized: + continue + entry = cache.get(normalized) + fp = _credential_fingerprint(normalized) + if ( + isinstance(entry, dict) + and entry.get("fp") == fp + and isinstance(entry.get("models"), list) + and entry["models"] + ): + age = now - float(entry.get("at", 0)) + if age < _PROVIDER_MODELS_CACHE_TTL: + continue # fresh, skip + stale_slugs.append(normalized) + + if not stale_slugs: + return + + import concurrent.futures + + def _fetch_one(slug: str) -> None: + try: + models = cached_provider_model_ids(slug, force_refresh=True) + # cached_provider_model_ids already persists the result, but in a + # non-locked read-modify-write. Re-persist via the thread-safe + # path to guarantee no lost writes under concurrency. + if models: + from hermes_cli.models import update_provider_cache_entry + update_provider_cache_entry(slug, models) + except Exception: + pass # best-effort; picker falls back to curated list + + with concurrent.futures.ThreadPoolExecutor( + max_workers=min(_PARALLEL_PREFETCH_WORKERS, len(stale_slugs)), + thread_name_prefix="model-cache-prefetch", + ) as executor: + list(executor.map(_fetch_one, stale_slugs)) + + +def _iter_builtin_candidates(models_dev_data: dict, excluded: set, seen: set): + """Yield ``(hermes_id, mdev_id, pconfig, env_vars)`` for section-1 rows. + + Skips vendor names that are aliases routing through an aggregator (bare + "openai" -> "openrouter": emitting them would silently switch a user onto an + endpoint they may have no key for), hermes_ids that are aliases of another + canonical profile ("kimi" -> "kimi-coding"), non-api_key auth types (section + 2 handles them with auth-store checks), and providers Hermes cannot route. + PROVIDER_REGISTRY env var names win over models.dev's (which can be wrong). + """ + from agent.models_dev import PROVIDER_TO_MODELS_DEV + from hermes_cli.auth import PROVIDER_REGISTRY, is_runtime_provider_routable + from hermes_cli.models import _AGGREGATOR_PROVIDERS + from hermes_cli.providers import ALIASES + + for hermes_id, mdev_id in PROVIDER_TO_MODELS_DEV.items(): + alias_target = ALIASES.get(hermes_id) + if alias_target and alias_target != hermes_id and alias_target in _AGGREGATOR_PROVIDERS: + continue + canonical = hermes_id + try: + from providers import get_provider_profile + prof = get_provider_profile(hermes_id) + if prof is not None: + canonical = prof.name + except Exception: + pass + if canonical != hermes_id or hermes_id.lower() in seen: + continue + if hermes_id.lower() in excluded or mdev_id.lower() in excluded: + continue + pdata = models_dev_data.get(mdev_id) + if not isinstance(pdata, dict): + continue + pconfig = PROVIDER_REGISTRY.get(hermes_id) + if pconfig and pconfig.auth_type != "api_key": + continue + if not is_runtime_provider_routable(hermes_id): + continue + if pconfig and pconfig.api_key_env_vars: + env_vars = list(pconfig.api_key_env_vars) + else: + env_vars = pdata.get("env", []) + if not isinstance(env_vars, list): + continue + yield hermes_id, mdev_id, pconfig, env_vars + + +def _auth_store_has_provider(*keys: str) -> bool: + """True when ``auth.json`` has a ``providers`` entry under any of *keys*.""" + try: + from hermes_cli.auth import _load_auth_store + store = _load_auth_store() + providers_store = store.get("providers", {}) + return bool(store and any(k in providers_store for k in keys)) + except Exception as exc: + logger.debug("Auth store check failed for %s: %s", keys[0] if keys else "", exc) + return False + + +def _raw_pool_usable(hermes_id: str) -> bool: + """Section-1 pool check: only consult the pool when auth.json lists a raw entry.""" + from hermes_cli.model_switch import _credential_pool_is_usable + try: + from hermes_cli.auth import _load_auth_store + store = _load_auth_store() + if store and store.get("credential_pool", {}).get(hermes_id): + return _credential_pool_is_usable(hermes_id, raw_pool_present=True) + except Exception: + pass + return False + + +def _pool_usable(slug: str) -> bool: + from hermes_cli.model_switch import _credential_pool_is_usable + try: + return _credential_pool_is_usable(slug) + except Exception as exc: + logger.debug("Credential pool check failed for %s: %s", slug, exc) + return False + + +def _overlay_has_env_creds(pid: str, hermes_slug: str, overlay, read_env) -> bool: + """Section-2 env/SDK credential check shared by the picker and the prefetch scan. + + Vertex authenticates via OAuth2 (service-account JSON / ADC), not an API + key, so it gets its own probe; otherwise the provider is hidden from the + picker even when fully configured. + """ + from hermes_cli.auth import PROVIDER_REGISTRY + + has_creds = False + if overlay.auth_type == "vertex": + try: + from agent.vertex_adapter import has_vertex_credentials + has_creds = has_vertex_credentials() + except Exception as exc: + logger.debug("Vertex credential check failed: %s", exc) + elif overlay.extra_env_vars: + has_creds = any(read_env(ev) for ev in overlay.extra_env_vars) + if not has_creds and overlay.auth_type == "api_key": + for key in (pid, hermes_slug): + pcfg = PROVIDER_REGISTRY.get(key) + if pcfg and pcfg.api_key_env_vars and any(read_env(ev) for ev in pcfg.api_key_env_vars): + return True + return has_creds + + +def _has_fast_aws_sdk_signal() -> bool: + """True when explicit AWS auth config is present in the environment. + + Deliberately avoids botocore's full credential chain: picker discovery runs + for non-Bedrock providers too, and botocore may probe EC2 IMDS + (169.254.169.254) on local machines before returning no credentials. + """ + env = os.environ + if env.get("AWS_BEARER_TOKEN_BEDROCK", "").strip(): + return True + if env.get("AWS_ACCESS_KEY_ID", "").strip() and env.get("AWS_SECRET_ACCESS_KEY", "").strip(): + return True + return any( + env.get(name, "").strip() + for name in ( + "AWS_PROFILE", + "AWS_CONTAINER_CREDENTIALS_RELATIVE_URI", + "AWS_CONTAINER_CREDENTIALS_FULL_URI", + "AWS_WEB_IDENTITY_TOKEN_FILE", + ) + ) + + +def _has_aws_sdk_creds_for_listing(slug: str, current_provider: str) -> bool: + """Credential check for AWS SDK providers in non-runtime discovery. + + The full boto3 chain is only consulted for the *current* provider. + """ + if _has_fast_aws_sdk_signal(): + return True + if str(slug or "").strip().lower() != str(current_provider or "").strip().lower(): + return False + try: + from agent.bedrock_adapter import has_aws_credentials + return bool(has_aws_credentials()) + except Exception: + return False + + +def _is_aws_sdk(pconfig) -> bool: + return bool(pconfig) and getattr(pconfig, "auth_type", "") == "aws_sdk" + + +def _live_or_curated_ids(slug: str, curated: dict, *fallback_keys: str, merge_models_dev: bool = True) -> list: + """Unified pathway: ``cached_provider_model_ids`` so the /model picker sees the + SAME list ``hermes model`` builds (disk-cached), falling back to the curated + static list (merged with models.dev for preferred providers) when live is empty. + """ + from hermes_cli.models import _MODELS_DEV_PREFERRED, _merge_with_models_dev, cached_provider_model_ids + + model_ids = cached_provider_model_ids(slug) + if not model_ids: + for key in fallback_keys or (slug,): + model_ids = curated.get(key, []) + if model_ids: + break + if merge_models_dev and slug in _MODELS_DEV_PREFERRED: + model_ids = _merge_with_models_dev(slug, model_ids) + return model_ids + + +def _aws_live_or_curated_ids(slug: str, curated: dict, *fallback_keys: str) -> list: + """Bedrock: live discovery reflects the active region (eu.*, ap.*) rather than + the static us.* list; any failure falls back to the curated list.""" + from hermes_cli.models import cached_provider_model_ids + + fallback_keys = fallback_keys or (slug,) + try: + ids = cached_provider_model_ids(slug) + if ids: + return ids + except Exception: + pass + for key in fallback_keys: + ids = curated.get(key, []) + if ids: + return ids + return [] + + +def _nous_picker_model_ids(curated: dict, force_fresh_nous_tier: bool) -> list: + """Nous serves a huge alphabetical live catalog; the picker shows ONLY the + curated agentic list, augmented with the Portal's free/paid recommendations + (so newly launched models surface without a CLI release) and narrowed by org + policy. Mirrors ``_model_flow_nous`` so GUI pickers match the CLI. A failed + recommendation fetch still yields a policy-filtered curated list. + """ + model_ids = curated.get("nous", []) + try: + from hermes_cli.models import ( + get_pricing_for_provider, + check_nous_free_tier, + union_with_portal_free_recommendations, + union_with_portal_paid_recommendations, + ) + from hermes_cli.auth import get_provider_auth_state + + pricing = get_pricing_for_provider("nous") or {} + try: + portal = (get_provider_auth_state("nous") or {}).get("portal_base_url", "") or "" + except Exception: + portal = "" + if check_nous_free_tier(force_fresh=force_fresh_nous_tier): + model_ids, _ = union_with_portal_free_recommendations(model_ids, pricing, portal) + else: + model_ids, _ = union_with_portal_paid_recommendations(model_ids, pricing, portal) + except Exception: + pass + try: + from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy + + model_ids = restrict_to_nous_policy(model_ids, nous_policy_allowed_ids(), rescue_empty=True) + except Exception: + pass + return model_ids + + +def _cap_models(model_ids: list, max_models: int | None, slug: str = "") -> list: + """Apply ``max_models``; aggregators in ``_UNCAPPED_PICKER_PROVIDERS`` show everything.""" + if slug in _UNCAPPED_PICKER_PROVIDERS or max_models is None: + return model_ids + return model_ids[:max_models] + + +def _norm_url(url: Any) -> str: + return str(url or "").strip().rstrip("/").lower() + + +def _entry_base_url(entry: dict, keys: tuple = ("base_url", "url", "api")) -> str: + for key in keys: + value = entry.get(key, "") + if value: + return value + return "" + + +def _entry_api_mode(entry: dict) -> str | None: + return str(entry.get("api_mode") or entry.get("transport") or "").strip().lower() or None + + +def _credential_identity(inline_api_key: str, key_env: str) -> str: + return inline_api_key if inline_api_key else (f"env:{key_env}" if key_env else "") + + +def _discover_flag(entry: dict): + """``discover_models`` (default True); ``"false"/"no"/"0"`` strings mean False.""" + discover = entry.get("discover_models", True) + if isinstance(discover, str): + discover = discover.lower() not in {"false", "no", "0"} + return discover + + +def _display_prefix(name: str) -> str: + """Text before the per-model separator Hermes's own writer uses ("—" / " - ").""" + for sep in ("—", " - "): + if sep in name: + return name.split(sep)[0].strip() + return name + + +def _discover_endpoint_models( + api_key: str, + api_url: str, + native_catalog_provider: str, + has_explicit_models: bool, + *, + headers: dict | None, + api_mode: str | None, + probe_live: bool, + discovery_allowed: bool, + for_picker: bool, +) -> tuple[list | None, bool]: + """Return ``(models, native_catalog_empty)`` for a custom endpoint row. + + ``probe_live`` runs the native-aware picker fetch; otherwise, when discovery + is allowed, a warm same-fingerprint cache entry still serves the full catalog + with no round-trip. ``has_explicit_models`` gates the *probe* (a network-cost + guard for keyless endpoints that declare a catalog), never the cache read — + applying it to the read re-pins the endpoint to its declared subset. Returns + ``(None, False)`` when nothing usable was found. + """ + from hermes_cli.model_switch import _fetch_picker_live_models + timeout = 1.5 if for_picker else 5.0 + if probe_live: + try: + live_models = _fetch_picker_live_models( + api_key, api_url, native_catalog_provider, has_explicit_models, + headers=headers, timeout=timeout, api_mode=api_mode, + ) + is_native = isinstance(live_models, _NativePickerModelList) + if live_models is not None and (live_models or not has_explicit_models or is_native): + return live_models, (is_native and not live_models) + except Exception: + pass + elif discovery_allowed: + try: + from hermes_cli.models import cached_fetch_api_models + + cached_models = cached_fetch_api_models( + api_key, api_url, cache_only=True, timeout=timeout, headers=headers, api_mode=api_mode, + ) + if cached_models: + return cached_models, False + except _MODEL_DISCOVERY_ERRORS: + pass + return None, False + + +def _collect_authed_provider_slugs( + models_dev_data: dict, + curated: dict[str, list[str]], + excluded: list[str], +) -> list[str]: + """Quick-scan which providers have credentials, without fetching model lists. + + Mirrors the credential checks of sections 1, 2 and 2b of + :func:`list_authenticated_providers` but never calls + ``cached_provider_model_ids``; the result feeds + :func:`_prefetch_provider_models_parallel`. Env vars are read through the + per-profile secret scope. AWS SDK providers are skipped (heavier detection). + """ + from hermes_cli.model_switch import _scoped_key_env + from agent.models_dev import PROVIDER_TO_MODELS_DEV + from hermes_cli.auth import PROVIDER_REGISTRY + from hermes_cli.providers import HERMES_OVERLAYS + from hermes_cli.models import CANONICAL_PROVIDERS + + excluded_set = {str(p).strip().lower() for p in excluded if p} + slugs: list[str] = [] + seen: set[str] = set() + + for hermes_id, _mdev_id, _pconfig, env_vars in _iter_builtin_candidates(models_dev_data, excluded_set, seen): + if any(_scoped_key_env(ev) for ev in env_vars) or _raw_pool_usable(hermes_id): + slugs.append(hermes_id) + seen.add(hermes_id.lower()) + + mdev_to_hermes = {v: k for k, v in PROVIDER_TO_MODELS_DEV.items()} + for pid, overlay in HERMES_OVERLAYS.items(): + hermes_slug = mdev_to_hermes.get(pid, pid) + if pid.lower() in seen or hermes_slug.lower() in seen: + continue + if pid.lower() in excluded_set or hermes_slug.lower() in excluded_set: + continue + if overlay.auth_type == "aws_sdk": + continue + if ( + _overlay_has_env_creds(pid, hermes_slug, overlay, _scoped_key_env) + or _auth_store_has_provider(pid, hermes_slug) + or _pool_usable(hermes_slug) + ): + slugs.append(hermes_slug) + seen.add(pid.lower()) + seen.add(hermes_slug.lower()) + + for cp in CANONICAL_PROVIDERS: + if cp.slug.lower() in seen or cp.slug.lower() in excluded_set: + continue + cp_config = PROVIDER_REGISTRY.get(cp.slug) + has_creds = bool( + cp_config and cp_config.api_key_env_vars and any(_scoped_key_env(ev) for ev in cp_config.api_key_env_vars) + ) + if has_creds or _auth_store_has_provider(cp.slug) or _pool_usable(cp.slug): + slugs.append(cp.slug) + seen.add(cp.slug.lower()) + + # Nous excluded: its picker branch builds from the curated list and never + # reads the api_key-only cache entry a prefetch would write. + return [s for s in slugs if s != "nous"] + + +@dataclass +class _PickerBuild: + """Mutable state threaded through the ``list_authenticated_providers`` sections.""" + + current_provider: str + current_base_url: str + current_model: str + max_models: int | None + for_picker: bool + force_fresh_nous_tier: bool + probe_custom_providers: bool + probe_current_custom_provider: bool + refresh: bool + excluded: set + curated: dict + results: list = field(default_factory=list) + seen_slugs: set = field(default_factory=set) # lowercase-normalized to catch case variants + # Effective base URLs of every built-in row, so section 4 hides + # ``custom_providers`` entries that duplicate a built-in endpoint. + builtin_endpoints: set = field(default_factory=set) + # (display_name, base_url) pairs emitted by section 3 so section 4 skips + # overlapping ``custom_providers`` rows (callers often pass both). + section3_pairs: set = field(default_factory=set) + current_provider_norm: str = field(init=False) + current_base_url_norm: str = field(init=False) + + def __post_init__(self): + self.current_provider_norm = self.current_provider.lower() + self.current_base_url_norm = self.current_base_url.rstrip("/").lower() + + def can_probe_custom(self, *, row_is_current: bool) -> bool: + return bool(self.probe_custom_providers or (self.probe_current_custom_provider and row_is_current)) + + def record_builtin_endpoint(self, slug: str) -> None: + """Prefer the live env override (e.g. DASHSCOPE_BASE_URL) over the static + inference_base_url so dedup matches what a user typing that URL into + custom_providers would actually hit.""" + try: + from hermes_cli.auth import PROVIDER_REGISTRY + except Exception: + return + pcfg = PROVIDER_REGISTRY.get(slug) + if not pcfg: + return + url = os.environ.get(pcfg.base_url_env_var, "") if getattr(pcfg, "base_url_env_var", "") else "" + normed = _norm_url(url or getattr(pcfg, "inference_base_url", "") or "") + if normed: + self.builtin_endpoints.add(normed) + + def add_builtin_row(self, slug: str, name: str, is_current: bool, model_ids: list, source: str, *, uncapped_ok: bool = True) -> None: + self.results.append({ + "slug": slug, + "name": name, + "is_current": is_current, + "is_user_defined": False, + "models": _cap_models(model_ids, self.max_models, slug if uncapped_ok else ""), + "total_models": len(model_ids), + "source": source, + }) + self.seen_slugs.add(slug.lower()) + self.record_builtin_endpoint(slug) + + +def _lap_builtin_rows(b: _PickerBuild, data: dict, user_providers: dict) -> None: + """Section 1: models.dev-mapped providers with api_key auth.""" + from hermes_cli.model_switch import _declared_model_ids + from agent.models_dev import get_provider_info + + for hermes_id, mdev_id, pconfig, env_vars in _iter_builtin_candidates(data, b.excluded, b.seen_slugs): + if not (any(os.environ.get(ev) for ev in env_vars) or _raw_pool_usable(hermes_id)): + continue + model_ids = _live_or_curated_ids(hermes_id, b.curated) + # A providers..models block extends the discovered catalog; + # section 3 cannot emit it later because this row owns the slug. + configured = user_providers.get(hermes_id) if isinstance(user_providers, dict) else None + configured_models = _declared_model_ids(configured.get("models")) if isinstance(configured, dict) else [] + model_ids = list(dict.fromkeys([*configured_models, *model_ids])) + pinfo = get_provider_info(mdev_id) + display_name = pconfig.name if pconfig and pconfig.name else (pinfo.name if pinfo else mdev_id) + b.add_builtin_row( + hermes_id, display_name, b.current_provider in (hermes_id, mdev_id), model_ids, "built-in", + ) + + +def _lap_overlay_rows(b: _PickerBuild, data: dict) -> None: + """Section 2: Hermes-only providers (nous, openai-codex, copilot, opencode-go, ...).""" + from hermes_cli.model_switch import _credential_pool_is_usable + from agent.models_dev import PROVIDER_TO_MODELS_DEV + from hermes_cli.providers import HERMES_OVERLAYS + + # HERMES_OVERLAYS keys may be models.dev IDs ("github-copilot") while + # config.yaml uses Hermes IDs ("copilot"). + mdev_to_hermes = {v: k for k, v in PROVIDER_TO_MODELS_DEV.items()} + for pid, overlay in HERMES_OVERLAYS.items(): + hermes_slug = mdev_to_hermes.get(pid, pid) + if pid.lower() in b.seen_slugs or hermes_slug.lower() in b.seen_slugs: + continue + if pid.lower() in b.excluded or hermes_slug.lower() in b.excluded: + continue + + if getattr(overlay, "keyless", False): + has_creds = True # served anonymously (opencode-free) + elif overlay.auth_type == "aws_sdk": + has_creds = _has_aws_sdk_creds_for_listing(hermes_slug, b.current_provider) + else: + has_creds = _overlay_has_env_creds(pid, hermes_slug, overlay, os.environ.get) + # External-process providers (copilot-acp) hold no key/token/pool entry by + # design — the spawned ACP subprocess brings its own auth. "Configured" + # means the executable resolves, which is what get_auth_status() reports; + # without this the has_creds filter hides the provider from every picker. + if not has_creds and overlay.auth_type == "external_process": + try: + from hermes_cli.auth import get_auth_status + _ext_status = get_auth_status(hermes_slug) or {} + has_creds = bool(_ext_status.get("logged_in") or _ext_status.get("configured")) + except Exception as exc: + logger.debug("External-process check failed for %s: %s", pid, exc) + # Auth store / credential pool cover OAuth providers AND api_key providers + # that also support OAuth (anthropic via Claude Code credential files). + if not has_creds: + has_creds = _auth_store_has_provider(pid, hermes_slug) + if not has_creds: + # Full auto-seeding pool check catches external stores (Codex CLI + # ~/.codex/auth.json) not yet in auth.json. + try: + if _credential_pool_is_usable(hermes_slug): + has_creds = True + elif b.for_picker: + # Show providers whose pool is entirely in cooldown: limits are + # per-model for many providers, so another model may work. + try: + from agent.credential_pool import load_pool + has_creds = load_pool(hermes_slug).has_credentials() + except Exception: + pass + except Exception as exc: + logger.debug("Credential pool check failed for %s: %s", hermes_slug, exc) + if not has_creds and hermes_slug == "anthropic": + # The pool gates anthropic behind is_provider_explicitly_configured() + # (aux tasks must not consume Claude Code tokens); the picker is + # discovery-oriented, so read the external credential files directly. + try: + from agent.anthropic_adapter import read_claude_code_credentials, read_hermes_oauth_credentials + hermes_creds = read_hermes_oauth_credentials() + cc_creds = read_claude_code_credentials() + if (hermes_creds and hermes_creds.get("accessToken")) or (cc_creds and cc_creds.get("accessToken")): + has_creds = True + except Exception as exc: + logger.debug("Anthropic external creds check failed: %s", exc) + if not has_creds: + continue + + if hermes_slug in {"openai-codex", "copilot", "copilot-acp"}: + # Live OAuth-backed discovery so Pro-only Codex slugs not in the static + # catalog appear; falls back to curated when unreachable. + from hermes_cli.models import cached_provider_model_ids + model_ids = cached_provider_model_ids(hermes_slug) + elif overlay.auth_type == "aws_sdk": + model_ids = _aws_live_or_curated_ids(hermes_slug, b.curated, hermes_slug, pid) + elif hermes_slug == "nous": + model_ids = _nous_picker_model_ids(b.curated, b.force_fresh_nous_tier) + else: + model_ids = _live_or_curated_ids(hermes_slug, b.curated, hermes_slug, pid) + b.add_builtin_row( + hermes_slug, get_label(hermes_slug), b.current_provider in (hermes_slug, pid), model_ids, "hermes", + ) + b.seen_slugs.add(pid.lower()) + + +def _lap_canonical_rows(b: _PickerBuild) -> None: + """Section 2b: CANONICAL_PROVIDERS missed by sections 1/2.""" + from hermes_cli.auth import PROVIDER_REGISTRY + try: + from hermes_cli.models import CANONICAL_PROVIDERS + except ImportError: + CANONICAL_PROVIDERS = [] + + for cp in CANONICAL_PROVIDERS: + if cp.slug.lower() in b.seen_slugs or cp.slug.lower() in b.excluded: + continue + cp_config = PROVIDER_REGISTRY.get(cp.slug) + has_creds = False + if cp_config and cp_config.api_key_env_vars: + lit = {ev for ev in cp_config.api_key_env_vars if os.environ.get(ev)} + has_creds = bool(lit) + # A regional "-cn" twin lit only by key vars shared with its non-CN + # sibling is a phantom row: hide it unless it is the current provider, + # and only when it has a dedicated var of its own the user could set. + sib = PROVIDER_REGISTRY.get(cp.slug[:-3]) if cp.slug.endswith("-cn") else None + sib_vars = set(sib.api_key_env_vars) if sib else set() + if lit and lit <= sib_vars < set(cp_config.api_key_env_vars) and cp.slug != b.current_provider: + continue + if not has_creds: + has_creds = _auth_store_has_provider(cp.slug) or _pool_usable(cp.slug) + if not has_creds and _is_aws_sdk(cp_config): + has_creds = _has_aws_sdk_creds_for_listing(cp.slug, b.current_provider) + if not has_creds: + continue + if _is_aws_sdk(cp_config): + model_ids = _aws_live_or_curated_ids(cp.slug, b.curated) + else: + model_ids = _live_or_curated_ids(cp.slug, b.curated, merge_models_dev=False) + b.add_builtin_row( + cp.slug, cp.label, cp.slug == b.current_provider, model_ids, "canonical", uncapped_ok=False, + ) + + +def _lap_user_provider_rows(b: _PickerBuild, user_providers: dict) -> None: + """Section 3: ``providers:`` dict entries, grouped by (api_url, credential, + api_mode, extra_headers) so keyed providers on one endpoint with the same + wire protocol collapse into one row (e.g. two Palantir Claude entries -> + one "Palantir Claude" row); a different key_env/api_mode/headers keeps + distinct rows since the wire protocol or tenant differs.""" + from hermes_cli.model_switch import _declared_model_ids, _entry_models_discovered, _extra_headers_from_config, _models_config_is_allowlist, _scoped_key_env + from collections import OrderedDict + from hermes_cli.config import coerce_provider_id, is_provider_enabled + + ep_groups: "OrderedDict[tuple, dict]" = OrderedDict() + for ep_name, ep_cfg in user_providers.items(): + if not isinstance(ep_cfg, dict) or not is_provider_enabled(ep_cfg): + continue + if ep_name.lower() in b.seen_slugs: + continue + display_name = coerce_provider_id(ep_cfg.get("name")) or ep_name + api_url = _entry_base_url(ep_cfg, ("base_url", "api", "url")) + key_env = str(ep_cfg.get("key_env") or ep_cfg.get("api_key_env") or "").strip() + inline_api_key = str(ep_cfg.get("api_key", "") or "").strip() + api_mode = _entry_api_mode(ep_cfg) + headers_identity = tuple(sorted(_extra_headers_from_config(ep_cfg).items())) + group_key = (_norm_url(api_url), _credential_identity(inline_api_key, key_env), api_mode, headers_identity) + + # ``default_model`` is the legacy key; ``model`` matches custom_providers. + default_model = ep_cfg.get("default_model", "") or ep_cfg.get("model", "") + entry_models = [default_model] if default_model else [] + for model_id in _declared_model_ids(ep_cfg.get("models", [])): + if model_id not in entry_models: + entry_models.append(model_id) + + if group_key not in ep_groups: + # Strip the per-model suffix and trailing version tokens ("Palantir + # Claude 4.7 Opus" -> "Palantir Claude"): cut at the first token with + # a digit, only when >=2 words remain (avoids over-trimming). + grp_display = _display_prefix(display_name) + toks = grp_display.split() + cut_at = next((i for i, t in enumerate(toks) if any(c.isdigit() for c in t.strip(".,()"))), None) + if cut_at is not None and cut_at >= 2: + grp_display = " ".join(toks[:cut_at]).strip() + ep_groups[group_key] = { + "slug": ep_name, # first ep_name encountered + "name": grp_display or display_name, + "api_url": api_url, + "models": [], + "has_explicit_models": False, + "ep_cfg": ep_cfg, + "raw_names": [], + "aliases": set(), + } + grp = ep_groups[group_key] + for m in entry_models: + if m and m not in grp["models"]: + grp["models"].append(m) + # A singular default_model/model is only the active selection and must + # not suppress discovery; dict-shaped ``models:`` is context_length + # metadata, not an allowlist — see ``_models_config_is_allowlist``. + if _models_config_is_allowlist(ep_cfg.get("models"), _entry_models_discovered(ep_cfg)): + grp["has_explicit_models"] = True + grp["raw_names"].append(display_name) + grp["aliases"].update(custom_provider_aliases(display_name, str(ep_name))) + + for grp in ep_groups.values(): + ep_cfg, ep_name, display_name, api_url = grp["ep_cfg"], grp["slug"], grp["name"], grp["api_url"] + models_list = list(grp["models"]) + # Official OpenAI rows often have base_url but no models: dict — avoid a + # misleading zero count. + if not models_list and base_url_host_matches(str(api_url).strip().lower(), "api.openai.com"): + models_list = list(b.curated.get("openai") or []) + + # Probe policy (mirrors section 4): with an api_key always probe; without + # one, skip only when an allowlist-shaped ``models:`` narrows the endpoint. + api_key = str(ep_cfg.get("api_key", "") or "").strip() + if not api_key: + key_env = str(ep_cfg.get("key_env") or ep_cfg.get("api_key_env") or "").strip() + api_key = _scoped_key_env(key_env) if key_env else "" + has_explicit_models = bool(grp.get("has_explicit_models")) + ep_url_norm = _norm_url(api_url) + ep_aliases = {str(alias).lower() for alias in grp.get("aliases", set())} + is_current = ( + str(ep_name).strip().lower() == b.current_provider_norm + or b.current_provider_norm in ep_aliases + or ( + b.current_provider_norm == "custom" + and bool(b.current_base_url_norm) + and ep_url_norm == b.current_base_url_norm + ) + ) + discovery_allowed = bool(api_url) and _discover_flag(ep_cfg) + discovered, native_catalog_empty = _discover_endpoint_models( + api_key, + api_url, + ep_name if str(ep_name).strip().lower() in {"ollama", "custom:ollama"} else "custom", + has_explicit_models, + headers=_extra_headers_from_config(ep_cfg) or None, + api_mode=ep_cfg.get("api_mode"), + probe_live=( + discovery_allowed + and (bool(api_key) or not has_explicit_models) + and b.can_probe_custom(row_is_current=is_current) + ), + discovery_allowed=discovery_allowed, + for_picker=b.for_picker, + ) + if discovered is not None: + models_list = discovered + + b.results.append({ + "slug": ep_name, + "name": display_name, + "is_current": is_current, + "is_user_defined": True, + "models": models_list, + "total_models": len(models_list) if models_list else 0, + "source": "user-config", + "api_url": api_url, + "native_catalog_empty": native_catalog_empty, + }) + b.seen_slugs.add(ep_name.lower()) + b.seen_slugs.update(ep_aliases) + # Record every raw member name so section 4 can match per-model + # custom_providers rows even though the group label was collapsed. + for raw_name in grp.get("raw_names") or [display_name]: + pair = (str(raw_name).strip().lower(), ep_url_norm) + if pair[0] and pair[1]: + b.section3_pairs.add(pair) + b.seen_slugs.add(custom_provider_slug(raw_name).lower()) + pair = (str(display_name).strip().lower(), ep_url_norm) + if pair[0] and pair[1]: + b.section3_pairs.add(pair) + + +def _lap_bare_custom_row(b: _PickerBuild, custom_providers: list | None) -> None: + """Section 3b: ``model.provider: custom`` + ``model.base_url`` with no named + providers:/custom_providers row — surface it so /model does not look like it + ignored config.yaml.""" + if not (b.current_provider_norm == "custom" and b.current_base_url and "custom" not in b.seen_slugs): + return + if any( + isinstance(cp, dict) and _norm_url(_entry_base_url(cp)) == _norm_url(b.current_base_url) + for cp in (custom_providers or []) + ): + return + api_url = str(b.current_base_url).strip().rstrip("/") + models = [b.current_model] if b.current_model else [] + native_catalog_empty = False + try: + discovered, native_catalog_empty = _discover_endpoint_models( + "", api_url, "custom", False, + headers=None, api_mode=None, + probe_live=bool(b.refresh or b.probe_current_custom_provider), + discovery_allowed=True, + for_picker=b.for_picker, + ) + if discovered is not None: + models = discovered + except Exception: + pass + b.results.append({ + "slug": "custom", + "name": "Custom endpoint", + "is_current": True, + "is_user_defined": True, + "models": _cap_models(models, b.max_models), + "total_models": len(models), + "source": "model-config", + "api_url": api_url, + "native_catalog_empty": native_catalog_empty, + }) + b.seen_slugs.add("custom") + + +def _lap_custom_provider_rows(b: _PickerBuild, custom_providers: list) -> None: + """Section 4: ``custom_providers:`` entries (one model each) grouped into one + row per (endpoint, credential identity, api_mode, extra_headers, display + prefix). Four "Ollama — X" entries on one host become one "Ollama" row; + distinct prefixes sharing a proxy URL keep their own rows.""" + from hermes_cli.model_switch import _declared_model_ids, _entry_models_discovered, _extra_headers_from_config, _models_config_is_allowlist, _save_discovered_models_to_config, _scoped_key_env + from collections import OrderedDict + from hermes_cli.config import coerce_provider_id + + groups: "OrderedDict[tuple, dict]" = OrderedDict() + for entry in custom_providers: + if not isinstance(entry, dict): + continue + raw_name = coerce_provider_id(entry.get("name")) + api_url = str(_entry_base_url(entry) or "").strip().rstrip("/") + if not raw_name or not api_url: + continue + inline_api_key = str(entry.get("api_key") or "").strip() + key_env = str(entry.get("key_env") or "").strip() + api_key = inline_api_key or _scoped_key_env(key_env) + api_mode = _entry_api_mode(entry) + discover = _discover_flag(entry) + entry_extra_headers = _extra_headers_from_config(entry) + prefix = _display_prefix(raw_name) + group_key = ( + api_url, _credential_identity(inline_api_key, key_env), api_mode, + tuple(sorted(entry_extra_headers.items())), prefix.lower(), + ) + if group_key not in groups: + display_name = prefix or raw_name + groups[group_key] = { + "slug": custom_provider_slug(display_name, str(entry.get("provider_key") or "").strip()), + "name": display_name, + "api_url": api_url, + "api_key": api_key, + "models": [], + "has_explicit_models": False, + "discover_models": discover, + "api_mode": api_mode, + "extra_headers": entry_extra_headers, + "aliases": set(), + } + else: + if api_key and not groups[group_key].get("api_key"): + groups[group_key]["api_key"] = api_key + if not discover: # one opt-out pins the whole grouped row + groups[group_key]["discover_models"] = False + grp = groups[group_key] + grp["aliases"].update(custom_provider_aliases(raw_name, str(entry.get("provider_key") or ""))) + # ``model:`` is only the active selection; every configured model lives + # under ``models:`` (dict written by _save_custom_provider). + default_model = (entry.get("model") or "").strip() + if default_model and default_model not in grp["models"]: + grp["models"].append(default_model) + models_field = entry.get("models", {}) + if _models_config_is_allowlist(models_field, _entry_models_discovered(entry)): + grp["has_explicit_models"] = True + for model_id in _declared_model_ids(models_field): + if model_id not in grp["models"]: + grp["models"].append(model_id) + + section4_slugs: set = set() + current_url_group_count = sum( + 1 for grp in groups.values() + if b.current_base_url_norm and _norm_url(grp["api_url"]) == b.current_base_url_norm + ) + for grp in groups.values(): + api_url, api_key, slug = grp["api_url"], grp.get("api_key", ""), grp["slug"] + # Slug claimed by a built-in/overlay/providers: row -> skip (don't shadow). + if slug.lower() in b.seen_slugs and slug.lower() not in section4_slugs: + continue + # Two custom endpoints with the same cleaned name: suffix a counter so + # both stay visible. + if slug.lower() in section4_slugs: + base_slug, n = slug, 2 + while f"{base_slug}-{n}".lower() in b.seen_slugs: + n += 1 + slug = f"{base_slug}-{n}" + grp["slug"] = slug + grp_url_norm = _norm_url(api_url) + pair_key = (str(grp["name"]).strip().lower(), grp_url_norm) + if pair_key[0] and pair_key[1] and pair_key in b.section3_pairs: + continue + # A built-in row already represents this endpoint (e.g. "my-dashscope" + # vs the alibaba-coding-plan row): keep the built-in, hide the shadow. + if grp_url_norm and grp_url_norm in b.builtin_endpoints: + continue + is_current = ( + slug.lower() == b.current_provider_norm + or b.current_provider_norm in {str(alias).lower() for alias in grp.get("aliases", set())} + ) or ( + b.current_provider_norm == "custom" + and bool(b.current_base_url_norm) + and grp_url_norm == b.current_base_url_norm + and current_url_group_count == 1 + ) + # Probe policy: with an api_key live /models is the source of truth (replace + # the partial ``models:`` subset); without one, an allowlist-shaped + # ``models:`` narrows a public endpoint and skips the probe. A dict-shaped + # ``models:`` is metadata, so still probe; pin with discover_models: false. + has_explicit_models = bool(grp.get("has_explicit_models")) + discovery_allowed = bool(api_url) and grp.get("discover_models", True) + probe_live = ( + discovery_allowed + and (bool(api_key) or not has_explicit_models) + and b.can_probe_custom(row_is_current=is_current) + ) + discovered, native_catalog_empty = _discover_endpoint_models( + api_key, + api_url, + "ollama" if "ollama" in {str(slug).strip().lower(), str(grp.get("name") or "").strip().lower()} else "custom", + has_explicit_models, + headers=grp.get("extra_headers") or None, + api_mode=grp.get("api_mode"), + probe_live=probe_live, + discovery_allowed=discovery_allowed, + for_picker=b.for_picker, + ) + if discovered is not None: + grp["models"] = discovered + if probe_live: + # A successful live probe persists the catalog for no-probe surfaces. + try: + _save_discovered_models_to_config( + api_url, discovered, api_mode=grp.get("api_mode"), headers=grp.get("extra_headers") or None, + ) + except Exception: + pass + b.results.append({ + "slug": slug, + "name": grp["name"], + "is_current": is_current, + "is_user_defined": True, + "models": grp["models"], + "total_models": len(grp["models"]), + "source": "user-config", + "api_url": grp["api_url"], + "native_catalog_empty": native_catalog_empty, + }) + b.seen_slugs.add(slug.lower()) + section4_slugs.add(slug.lower()) + + +def _build_curated_lists(current_provider: str, current_base_url: str, current_model: str) -> dict[str, list[str]]: + """Curated model lists keyed by hermes provider id, plus the dynamic ones + (nous manifest, Ollama Cloud, LM Studio live probe).""" + from hermes_cli.models import OPENROUTER_MODELS, _PROVIDER_MODELS, get_curated_nous_model_ids + + curated: dict[str, list[str]] = dict(_PROVIDER_MODELS) + curated["openrouter"] = [mid for mid, _ in OPENROUTER_MODELS] + # Remote model-catalog manifest so new Portal models surface without a + # release; falls back to the in-repo snapshot when unreachable. + curated["nous"] = get_curated_nous_model_ids() + if "ollama-cloud" not in curated: + from hermes_cli.models import fetch_ollama_cloud_models + curated["ollama-cloud"] = fetch_ollama_cloud_models() + # LM Studio has no static catalog: probe its native endpoint live. Base URL + # precedence: LM_BASE_URL > active config base_url (when current) > default. + # On auth rejection / unreachable, fall back to the current model so the + # picker still shows something offline. + is_current_lmstudio = current_provider.strip().lower() == "lmstudio" + if "lmstudio" not in curated and (os.environ.get("LM_API_KEY") or os.environ.get("LM_BASE_URL") or is_current_lmstudio): + from hermes_cli.models import fetch_lmstudio_models + from hermes_cli.auth import AuthError + lm_base = ( + os.environ.get("LM_BASE_URL") + or (current_base_url if is_current_lmstudio and current_base_url else None) + or "http://127.0.0.1:1234/v1" + ) + try: + live = fetch_lmstudio_models(api_key=os.environ.get("LM_API_KEY", ""), base_url=lm_base, timeout=1.5) + except AuthError: + live = [] + if not live and is_current_lmstudio and current_model: + live = [current_model] + curated["lmstudio"] = live + return curated + + +def list_authenticated_providers( + current_provider: str = "", + current_base_url: str = "", + user_providers: dict = None, + custom_providers: list | None = None, + *, + force_fresh_nous_tier: bool = False, + max_models: int | None = None, + current_model: str = "", + refresh: bool = False, + probe_custom_providers: bool = True, + probe_current_custom_provider: bool = False, + for_picker: bool = False, + excluded_providers: list | None = None, +) -> List[dict]: + """Detect which providers have credentials and list their curated models. + + Uses the curated lists from hermes_cli/models.py (OPENROUTER_MODELS, + _PROVIDER_MODELS) — hand-picked agentic models, NOT the full models.dev + catalog. Only providers with API keys set or user-defined endpoints appear. + + Returns a list of dicts: ``slug`` (the --provider value), ``name``, + ``is_current``, ``is_user_defined``, ``models`` (up to max_models), + ``total_models``, ``source`` ("built-in", "hermes", "canonical", + "user-config", "model-config"). + + ``force_fresh_nous_tier`` bypasses the short Nous tier cache for explicit + account-sensitive flows; picker opens should leave it false. + ``refresh`` busts the per-provider model-id disk cache up front so every row + re-fetches live — for an explicit user "refresh models" action only. + ``probe_custom_providers`` controls live ``/models`` discovery for saved + custom endpoints (default true for CLI parity; GUI opens pass false). + ``probe_current_custom_provider`` probes only the currently-selected custom + endpoint so its list matches without blocking on offline ones. + """ + from hermes_cli.model_switch import _collect_authed_provider_slugs, _prefetch_provider_models_parallel + from agent.models_dev import fetch_models_dev + from hermes_cli.config import coerce_provider_id, stringify_provider_map + + # Explicit refresh: drop every cached model-id list so the calls below all + # re-fetch live. A stale cache can fall back to the curated static list when + # its live fetch fails, silently dropping live-only models the user had seen. + if refresh: + try: + from hermes_cli.models import clear_provider_models_cache + clear_provider_models_cache() + except Exception: + pass + + # PyYAML parses unquoted numeric names (`provider: 2070`) as int. + current_provider = coerce_provider_id(current_provider) + current_base_url = str(current_base_url or "").strip() + current_model = str(current_model or "").strip() + user_providers = stringify_provider_map(user_providers) + data = fetch_models_dev() + + b = _PickerBuild( + current_provider=current_provider, + current_base_url=current_base_url, + current_model=current_model, + max_models=max_models, + for_picker=for_picker, + force_fresh_nous_tier=force_fresh_nous_tier, + probe_custom_providers=probe_custom_providers, + probe_current_custom_provider=probe_current_custom_provider, + refresh=refresh, + # A single entry like ``copilot`` hides the provider under every key it + # surfaces as (hermes_id / mdev_id / canonical slug). + excluded={str(p).strip().lower() for p in (excluded_providers or []) if p}, + curated=_build_curated_lists(current_provider, current_base_url, current_model), + ) + + # Warm the disk cache in parallel before the serial section loops, which + # otherwise stack 15-30s of live /v1/models round-trips on a cold cache. + # Skipped when refresh=True (serial path force-refreshes) and for <=3 + # providers (serial is fast enough; avoids thread-pool overhead). + prefetch_slugs = [] if refresh else _collect_authed_provider_slugs(data, b.curated, excluded_providers or []) + if len(prefetch_slugs) > 3: + try: + _prefetch_provider_models_parallel(prefetch_slugs) + except Exception: + pass # best-effort; serial path still works + + _lap_builtin_rows(b, data, user_providers) + _lap_overlay_rows(b, data) + _lap_canonical_rows(b) + if user_providers and isinstance(user_providers, dict): + _lap_user_provider_rows(b, user_providers) + _lap_bare_custom_row(b, custom_providers) + if custom_providers and isinstance(custom_providers, list): + _lap_custom_provider_rows(b, custom_providers) + results = b.results + + # ``providers..enabled: false`` post-filter covers built-in rows + # (sections 1-2) that bypass the per-section gate; matched by slug and + # ``provider_id``. + try: + from hermes_cli.config import is_provider_enabled + if isinstance(user_providers, dict): + disabled = { + str(name).strip().lower() + for name, cfg in user_providers.items() + if isinstance(cfg, dict) and not is_provider_enabled(cfg) + } + if disabled: + results = [ + r for r in results + if str(r.get("provider_id", "")).strip().lower() not in disabled + and str(r.get("slug", "")).strip().lower() not in disabled + ] + except Exception: + pass + + # A custom/uncurated model set via `/model /` would be + # invisible in every picker (main and MoA slot pickers read these rows); + # inject it at the front of the current provider's row as a uniform post-pass. + if current_model: + for row in results: + if not row.get("is_current") or row.get("native_catalog_empty"): + continue + models = row.get("models") or [] + if current_model not in models: + row["models"] = [current_model, *models] + row["total_models"] = row.get("total_models", len(models)) + 1 + break + + # Current provider first, then by model count descending + results.sort(key=lambda r: (not r["is_current"], -r["total_models"])) + return results + + +def _prepend_moa_picker_provider(providers: List[dict], current_provider: str = "") -> List[dict]: + """Add the virtual MoA provider row used by interactive model pickers. + + ``list_authenticated_providers()`` only returns real/auth-backed providers. + The CLI model inventory adds MoA separately so named presets appear next to + normal providers; gateway pickers call ``list_picker_providers()`` directly, + so they need the same virtual row here. Reuse the inventory's single row + builder so the row shape stays defined in one place. + """ + try: + from hermes_cli.inventory import _moa_provider_row + + moa_row = _moa_provider_row(current_provider) + if moa_row is None: + return providers + return [moa_row] + [p for p in providers if str(p.get("slug", "")).lower() != "moa"] + except Exception: + return providers + + +def list_picker_providers( + current_provider: str = "", + current_base_url: str = "", + user_providers: dict = None, + custom_providers: list | None = None, + max_models: int | None = None, + current_model: str = "", + include_moa: bool = False, + excluded_providers: list | None = None, +) -> List[dict]: + """Interactive-picker variant of :func:`list_authenticated_providers`. + + Post-processes the base list so the ``/model`` picker (Telegram/Discord + inline keyboards) only surfaces models that are actually callable in the + current install: + + - OpenRouter's model list is replaced with the output of + :func:`hermes_cli.models.fetch_openrouter_models`, which filters the + curated ``OPENROUTER_MODELS`` snapshot against the live OpenRouter + catalog. IDs the live catalog no longer carries drop out, so the + picker never offers a model the user can't call. + - Provider rows whose model list ends up empty are dropped, except + custom endpoints (``is_user_defined=True`` with an ``api_url``) where + the user may supply their own model set through config. + + All other providers and metadata fields are passed through unchanged. + The typed ``/model `` path is unaffected -- only the interactive + picker payload is narrowed. + """ + from hermes_cli.model_switch import list_authenticated_providers + from hermes_cli.models import fetch_openrouter_models + + providers = list_authenticated_providers( + current_provider=current_provider, + current_base_url=current_base_url, + user_providers=user_providers, + custom_providers=custom_providers, + max_models=max_models, + current_model=current_model, + for_picker=True, + excluded_providers=excluded_providers, + ) + if include_moa: + providers = _prepend_moa_picker_provider(providers, current_provider=current_provider) + + filtered: List[dict] = [] + for p in providers: + slug = str(p.get("slug", "")).lower() + if slug == "openrouter": + try: + live = fetch_openrouter_models() + live_ids = [mid for mid, _ in live] + except Exception: + live_ids = list(p.get("models", [])) + p = dict(p) + p["models"] = live_ids[:max_models] if max_models is not None else live_ids + p["total_models"] = len(live_ids) + + has_models = bool(p.get("models")) + is_custom_endpoint = bool(p.get("is_user_defined")) and bool(p.get("api_url")) + if not has_models and not is_custom_endpoint: + continue + filtered.append(p) + + return filtered diff --git a/hermes_cli/status.py b/hermes_cli/status.py index bc56e8719c..4c683aef70 100644 --- a/hermes_cli/status.py +++ b/hermes_cli/status.py @@ -22,8 +22,10 @@ from hermes_cli.nous_subscription import get_nous_subscription_features from hermes_cli.runtime_provider import resolve_requested_provider from hermes_cli.vercel_auth import describe_vercel_auth from hermes_constants import OPENROUTER_MODELS_URL +from hermes_constants import is_termux as _is_termux from tools.tool_backend_helpers import managed_nous_tools_enabled + def check_mark(ok: bool) -> str: return color("✓", Colors.GREEN) if ok else color("✗", Colors.RED) @@ -44,14 +46,18 @@ def _detail(label: str, value) -> None: print(f" {label:<12}{value}") -def _oauth_block(name: str, status: dict, hint: str, details) -> bool: - """Print an OAuth provider row plus its conditional detail lines; returns logged-in state.""" +def _oauth_block(name: str, status: dict, hint: str, rows) -> None: + """Print an OAuth provider row plus its conditional detail lines. + + ``rows`` are ``(label, status_key, formatter, gate)``: a detail prints when the raw value is + truthy and ``gate`` is None or equals the logged-in state (False = only while logged out). + """ logged_in = bool(status.get("logged_in")) _row(name, logged_in, "logged in" if logged_in else f"not logged in (run: {hint})") - for label, value, show in details(logged_in): - if show: - _detail(label, value) - return logged_in + for label, key, fmt, gate in rows: + raw = status.get(key) + if raw and (gate is None or gate == logged_in): + _detail(label, fmt(raw) if fmt else raw) def _first_env_value(names) -> str: @@ -64,6 +70,7 @@ def _first_env_value(names) -> str: return v return "" + def redact_key(key: str) -> str: """Redact an API key for display. @@ -91,6 +98,11 @@ def _format_iso_timestamp(value) -> str: return parsed.astimezone().strftime("%Y-%m-%d %H:%M:%S %Z") +def _qwen_expiry(expires_at_ms) -> str: + from datetime import datetime, timezone + return datetime.fromtimestamp(int(expires_at_ms) / 1000, tz=timezone.utc).isoformat() + + def _configured_model_label(config: dict) -> str: """Return the configured default model from config.yaml.""" model_cfg = config.get("model") @@ -126,9 +138,6 @@ def _effective_provider_label() -> str: return provider_label(effective) -from hermes_constants import is_termux as _is_termux - - def _estop_status_line(): """One-line pause banner for `hermes status`, or None when not paused.""" try: @@ -142,194 +151,208 @@ def _estop_status_line(): return f"⏸️ PAUSED (global emergency stop{f' — reason: {reason}' if reason else ''}; `hermes resume` to lift)" -def show_status(args): - """Show status of all Hermes Agent components.""" - deep = getattr(args, 'deep', False) +# --------------------------------------------------------------------------- +# Data tables driving the per-section renderers below. +# --------------------------------------------------------------------------- +# Values may be a single env var name (str) or a tuple of alternates (first found wins). +_API_KEYS: dict[str, str | tuple[str, ...]] = { + "OpenRouter": "OPENROUTER_API_KEY", + "OpenAI": "OPENAI_API_KEY", + "Google / Gemini": ("GOOGLE_API_KEY", "GEMINI_API_KEY"), + "DeepSeek": "DEEPSEEK_API_KEY", + "xAI / Grok": "XAI_API_KEY", + "NVIDIA NIM": "NVIDIA_API_KEY", + "Z.AI / GLM": "GLM_API_KEY", + "Kimi": "KIMI_API_KEY", + "StepFun Step Plan": "STEPFUN_API_KEY", + "MiniMax": "MINIMAX_API_KEY", + "MiniMax-CN": "MINIMAX_CN_API_KEY", + "DeepInfra": "DEEPINFRA_API_KEY", + "Firecrawl": "FIRECRAWL_API_KEY", + "Tavily": "TAVILY_API_KEY", + "Keenable": "KEENABLE_API_KEY", + "Browser Use": "BROWSER_USE_API_KEY", # Optional — local browser works without this + "Browserbase": "BROWSERBASE_API_KEY", # Optional — direct credentials only + "FAL": "FAL_KEY", + "ElevenLabs": "ELEVENLABS_API_KEY", + "GitHub": "GITHUB_TOKEN", +} + +# OAuth detail rows: (label, status key, formatter, gate) — see _oauth_block. +_FILE_REFRESH_ROWS = ( + ("Auth file:", "auth_store", None, None), + ("Refreshed:", "last_refresh", _format_iso_timestamp, None), + ("Error:", "error", None, False), +) +_OAUTH_BLOCKS = ( + # (row name, auth getter, login hint, detail rows) + ("OpenAI Codex", "get_codex_auth_status", "hermes model", _FILE_REFRESH_ROWS), + ("Qwen OAuth", "get_qwen_auth_status", "qwen auth qwen-oauth", ( + ("Auth file:", "auth_file", None, None), + ("Access exp:", "expires_at_ms", _qwen_expiry, None), + ("Error:", "error", None, False), + )), + ("MiniMax OAuth", "get_minimax_oauth_auth_status", "hermes auth add minimax-oauth", ( + ("Region:", "region", None, True), + ("Access exp:", "expires_at", None, None), + ("Error:", "error", None, False), + )), + ("xAI OAuth", "get_xai_oauth_auth_status", "hermes auth add xai-oauth", _FILE_REFRESH_ROWS), +) + +_APIKEY_PROVIDERS = { + "Z.AI / GLM": ("GLM_API_KEY", "ZAI_API_KEY", "Z_AI_API_KEY"), + "Kimi / Moonshot": ("KIMI_API_KEY",), + "StepFun Step Plan": ("STEPFUN_API_KEY",), + "MiniMax": ("MINIMAX_API_KEY",), + "MiniMax (China)": ("MINIMAX_CN_API_KEY",), + "DeepInfra": ("DEEPINFRA_API_KEY",), +} + +# Simple env-driven terminal backends: (label, env var, default, empty-counts-as-unset). +_TERMINAL_ENV_ROWS = { + "ssh": (("SSH Host:", "TERMINAL_SSH_HOST", "(not set)", True), ("SSH User:", "TERMINAL_SSH_USER", "(not set)", True)), + "docker": (("Docker Image:", "TERMINAL_DOCKER_IMAGE", "python:3.11-slim", False),), + "daytona": (("Daytona Image:", "TERMINAL_DAYTONA_IMAGE", "nikolaik/python-nodejs:python3.11-nodejs20", False),), +} + +_PLATFORMS = { # name -> (token env var, home-channel env var or None) + "Telegram": ("TELEGRAM_BOT_TOKEN", "TELEGRAM_HOME_CHANNEL"), + "Discord": ("DISCORD_BOT_TOKEN", "DISCORD_HOME_CHANNEL"), + "WhatsApp": ("WHATSAPP_ENABLED", None), + "Signal": ("SIGNAL_HTTP_URL", "SIGNAL_HOME_CHANNEL"), + "Slack": ("SLACK_BOT_TOKEN", None), + "Email": ("EMAIL_ADDRESS", "EMAIL_HOME_ADDRESS"), + "SMS": ("TWILIO_ACCOUNT_SID", "SMS_HOME_CHANNEL"), + "DingTalk": ("DINGTALK_CLIENT_ID", None), + "Feishu": ("FEISHU_APP_ID", "FEISHU_HOME_CHANNEL"), + "WeCom": ("WECOM_BOT_ID", "WECOM_HOME_CHANNEL"), + "WeCom Callback": ("WECOM_CALLBACK_CORP_ID", None), + "Weixin": ("WEIXIN_ACCOUNT_ID", "WEIXIN_HOME_CHANNEL"), + "BlueBubbles": ("BLUEBUBBLES_SERVER_URL", "BLUEBUBBLES_HOME_CHANNEL"), + "QQBot": ("QQ_APP_ID", "QQ_HOME_CHANNEL"), + "Yuanbao": ("YUANBAO_APP_ID", "YUANBAO_HOME_CHANNEL"), +} + +# Gateway manager label when the runtime snapshot is unavailable, keyed by platform. +_GATEWAY_FALLBACK = {"linux": ("unknown", "systemd/manual"), "darwin": ("unknown", "launchd")} + + +class _StatusContext: + """State shared across section renderers: config, --deep, and the Nous login facts + the Auth Providers section derives that the Nous Tool Gateway section needs later.""" + + def __init__(self, deep: bool): + self.deep = deep + self.config: dict = {} + self.nous_logged_in = False + self.nous_inference_present = False + self.nous_account_info = None + + +def _render_header(ctx): print() print(color("┌─────────────────────────────────────────────────────────┐", Colors.CYAN)) print(color("│ ⚕ Hermes Agent Status │", Colors.CYAN)) print(color("└─────────────────────────────────────────────────────────┘", Colors.CYAN)) - _paused_line = _estop_status_line() if _paused_line: print() print(color(_paused_line, Colors.YELLOW, Colors.BOLD)) - # ========================================================================= - # Environment - # ========================================================================= + +def _render_environment(ctx): _section("Environment") print(f" Project: {PROJECT_ROOT}") print(f" Python: {sys.version.split()[0]}") - env_exists = get_env_path().exists() print(f" .env file: {check_mark(env_exists)} {'exists' if env_exists else 'not found'}") - try: - config = load_config() + ctx.config = load_config() except Exception: - config = {} - - print(f" Model: {_configured_model_label(config)}") + ctx.config = {} + print(f" Model: {_configured_model_label(ctx.config)}") print(f" Provider: {_effective_provider_label()}") - # ========================================================================= - # API Keys - # ========================================================================= + +def _render_api_keys(ctx): _section("API Keys") - - # Values may be a single env var name (str) or a tuple of alternates (first found wins). - keys: dict[str, str | tuple[str, ...]] = { - "OpenRouter": "OPENROUTER_API_KEY", - "OpenAI": "OPENAI_API_KEY", - "Google / Gemini": ("GOOGLE_API_KEY", "GEMINI_API_KEY"), - "DeepSeek": "DEEPSEEK_API_KEY", - "xAI / Grok": "XAI_API_KEY", - "NVIDIA NIM": "NVIDIA_API_KEY", - "Z.AI / GLM": "GLM_API_KEY", - "Kimi": "KIMI_API_KEY", - "StepFun Step Plan": "STEPFUN_API_KEY", - "MiniMax": "MINIMAX_API_KEY", - "MiniMax-CN": "MINIMAX_CN_API_KEY", - "DeepInfra": "DEEPINFRA_API_KEY", - "Firecrawl": "FIRECRAWL_API_KEY", - "Tavily": "TAVILY_API_KEY", - "Keenable": "KEENABLE_API_KEY", - "Browser Use": "BROWSER_USE_API_KEY", # Optional — local browser works without this - "Browserbase": "BROWSERBASE_API_KEY", # Optional — direct credentials only - "FAL": "FAL_KEY", - "ElevenLabs": "ELEVENLABS_API_KEY", - "GitHub": "GITHUB_TOKEN", - } - - for name, env_ref in keys.items(): + for name, env_ref in _API_KEYS.items(): value = _first_env_value(env_ref) _row(name, bool(value), redact_key(value)) - # Anthropic uses the dedicated lookup (it also resolves OAuth tokens). from hermes_cli.auth import get_anthropic_key anthropic_value = get_anthropic_key() _row("Anthropic", bool(anthropic_value), redact_key(anthropic_value)) - # ========================================================================= - # Auth Providers (OAuth) - # ========================================================================= - _section("Auth Providers") +def _render_auth_providers(ctx): + _section("Auth Providers") + import hermes_cli.auth as auth try: - from hermes_cli.auth import ( - get_nous_auth_status_local, - get_codex_auth_status, - get_qwen_auth_status, - get_minimax_oauth_auth_status, - ) # Read-only display: use the refresh-free snapshot so `hermes status` # never performs an OAuth refresh or burns a single-use refresh token. - nous_status = get_nous_auth_status_local() - codex_status = get_codex_auth_status() - qwen_status = get_qwen_auth_status() - minimax_status = get_minimax_oauth_auth_status() + nous_status = auth.get_nous_auth_status_local() + statuses = {getter: getattr(auth, getter)() for _, getter, _, _ in _OAUTH_BLOCKS[:3]} except Exception: - nous_status = codex_status = qwen_status = minimax_status = {} + nous_status, statuses = {}, {} + # xAI OAuth — separate try/except so an import failure here cannot + # disrupt the Nous/Codex/Qwen/MiniMax rows. + try: + statuses["get_xai_oauth_auth_status"] = auth.get_xai_oauth_auth_status() or {} + except Exception: + statuses["get_xai_oauth_auth_status"] = {} - nous_account_info = None + info = None if any(nous_status.get(k) for k in ( "logged_in", "access_token", "portal_base_url", "inference_credential_present", "error_code" )): try: - nous_account_info = get_nous_portal_account_info() + info = get_nous_portal_account_info() except Exception: - nous_account_info = None - - nous_logged_in = bool(nous_status.get("logged_in") or (nous_account_info and nous_account_info.logged_in)) - nous_inference_present = bool( - nous_status.get("inference_credential_present") - or (nous_account_info and nous_account_info.inference_credential_present) + info = None + ctx.nous_account_info = info + ctx.nous_logged_in = logged_in = bool(nous_status.get("logged_in") or (info and info.logged_in)) + ctx.nous_inference_present = inference = bool( + nous_status.get("inference_credential_present") or (info and info.inference_credential_present) ) nous_error = nous_status.get("error") _row( - "Nous Portal", nous_logged_in, - "logged in" if nous_logged_in - else "not logged in (Nous inference key configured)" if nous_inference_present + "Nous Portal", logged_in, + "logged in" if logged_in + else "not logged in (Nous inference key configured)" if inference else "not logged in (run: hermes portal)", ) portal_url = nous_status.get("portal_base_url") or "(unknown)" - inference_url = nous_status.get("inference_base_url") or ( - nous_account_info.inference_base_url if nous_account_info else None - ) + inference_url = nous_status.get("inference_base_url") or (info.inference_base_url if info else None) for label, value, show in ( - ("Portal URL:", portal_url, nous_logged_in or portal_url != "(unknown)" or nous_error), - ("Inference:", inference_url, nous_inference_present and inference_url), + ("Portal URL:", portal_url, logged_in or portal_url != "(unknown)" or nous_error), + ("Inference:", inference_url, inference and inference_url), ("Access exp:", _format_iso_timestamp(nous_status.get("access_expires_at")), - nous_logged_in or nous_status.get("access_expires_at")), + logged_in or nous_status.get("access_expires_at")), ("Key exp:", _format_iso_timestamp(nous_status.get("agent_key_expires_at")), - nous_logged_in or nous_inference_present or nous_status.get("agent_key_expires_at")), + logged_in or inference or nous_status.get("agent_key_expires_at")), ("Refresh:", "yes" if nous_status.get("has_refresh_token") else "no", - nous_logged_in or nous_status.get("has_refresh_token")), + logged_in or nous_status.get("has_refresh_token")), ("Error:", nous_error, nous_error), ): if show: _detail(label, value) + for name, getter, hint, rows in _OAUTH_BLOCKS: + _oauth_block(name, statuses.get(getter, {}), hint, rows) - def _file_refresh_error(status, file_key): - return lambda logged_in: ( - ("Auth file:", status.get(file_key), status.get(file_key)), - ("Refreshed:", _format_iso_timestamp(status.get("last_refresh")), status.get("last_refresh")), - ("Error:", status.get("error"), status.get("error") and not logged_in), - ) - _oauth_block("OpenAI Codex", codex_status, "hermes model", _file_refresh_error(codex_status, "auth_store")) - - def _qwen_details(logged_in): - qwen_exp = qwen_status.get("expires_at_ms") - exp_text = "" - if qwen_exp: - from datetime import datetime, timezone - exp_text = datetime.fromtimestamp(int(qwen_exp) / 1000, tz=timezone.utc).isoformat() - return ( - ("Auth file:", qwen_status.get("auth_file"), qwen_status.get("auth_file")), - ("Access exp:", exp_text, qwen_exp), - ("Error:", qwen_status.get("error"), qwen_status.get("error") and not logged_in), - ) - - _oauth_block("Qwen OAuth", qwen_status, "qwen auth qwen-oauth", _qwen_details) - - _oauth_block( - "MiniMax OAuth", minimax_status, "hermes auth add minimax-oauth", - lambda logged_in: ( - ("Region:", minimax_status.get("region"), logged_in and minimax_status.get("region")), - ("Access exp:", minimax_status.get("expires_at"), minimax_status.get("expires_at")), - ("Error:", minimax_status.get("error"), minimax_status.get("error") and not logged_in), - ), - ) - - # xAI OAuth — separate try/except so an import failure here cannot - # disrupt the already-printed Nous/Codex/Qwen/MiniMax rows above. - try: - from hermes_cli.auth import get_xai_oauth_auth_status - xai_oauth_status = get_xai_oauth_auth_status() or {} - except Exception: - xai_oauth_status = {} - - _oauth_block( - "xAI OAuth", xai_oauth_status, "hermes auth add xai-oauth", - _file_refresh_error(xai_oauth_status, "auth_store"), - ) - - # ========================================================================= - # Nous Subscription Features - # ========================================================================= +def _render_nous_gateway(ctx): if managed_nous_tools_enabled(): - features = get_nous_subscription_features(config) + features = get_nous_subscription_features(ctx.config) _section("Nous Tool Gateway") print(" Nous Portal ✓ managed tools available" if features.nous_auth_present else " Nous Portal ✗ not logged in") for feature in features.items(): if feature.managed_by_nous: state = "active via Nous subscription" elif feature.active: - current = feature.current_provider or "configured provider" - state = f"active via {current}" + state = f"active via {feature.current_provider or 'configured provider'}" elif feature.included_by_default and features.nous_auth_present: state = "included by subscription, not currently selected" elif feature.key == "modal" and features.nous_auth_present: @@ -337,30 +360,20 @@ def show_status(args): else: state = "not configured" print(f" {feature.label:<15} {check_mark(feature.available or feature.active or feature.managed_by_nous)} {state}") - elif nous_logged_in or nous_inference_present: + elif ctx.nous_logged_in or ctx.nous_inference_present: # Nous OAuth without entitlement, or an opaque inference key without # Portal account information, cannot enable the Tool Gateway. _section("Nous Tool Gateway") message = format_nous_portal_entitlement_message( - nous_account_info, capability="managed web, image, TTS, STT, browser, and Modal tools" + ctx.nous_account_info, capability="managed web, image, TTS, STT, browser, and Modal tools" ) for line in (message or "").splitlines(): print(f" {line}") - # ========================================================================= - # API-Key Providers - # ========================================================================= - _section("API-Key Providers") - apikey_providers = { - "Z.AI / GLM": ("GLM_API_KEY", "ZAI_API_KEY", "Z_AI_API_KEY"), - "Kimi / Moonshot": ("KIMI_API_KEY",), - "StepFun Step Plan": ("STEPFUN_API_KEY",), - "MiniMax": ("MINIMAX_API_KEY",), - "MiniMax (China)": ("MINIMAX_CN_API_KEY",), - "DeepInfra": ("DEEPINFRA_API_KEY",), - } - for pname, env_vars in apikey_providers.items(): +def _render_apikey_providers(ctx): + _section("API-Key Providers") + for pname, env_vars in _APIKEY_PROVIDERS.items(): configured = bool(_first_env_value(env_vars)) label = "configured" if configured else "not configured (run: hermes model)" print(f" {pname:<16} {check_mark(configured)} {label}") @@ -370,7 +383,7 @@ def show_status(args): # empty list is the most common LM Studio support case. if _effective_provider_label() == "LM Studio": from hermes_cli.models import probe_lmstudio_models - model_cfg = config.get("model") + model_cfg = ctx.config.get("model") base = (model_cfg.get("base_url") if isinstance(model_cfg, dict) else None) or get_env_value("LM_BASE_URL") or "http://127.0.0.1:1234/v1" try: models = probe_lmstudio_models(api_key=get_env_value("LM_API_KEY") or "", base_url=base, timeout=1.5) @@ -380,22 +393,17 @@ def show_status(args): ok, msg = False, "auth rejected — set LM_API_KEY" print(f" {'LM Studio':<16} {check_mark(ok)} {msg}") - # ========================================================================= - # Terminal Configuration - # ========================================================================= - _section("Terminal Backend") - terminal_cfg = config.get("terminal", {}) if isinstance(config.get("terminal"), dict) else {} +def _render_terminal(ctx): + _section("Terminal Backend") + terminal_cfg = ctx.config.get("terminal", {}) if isinstance(ctx.config.get("terminal"), dict) else {} terminal_env = os.getenv("TERMINAL_ENV", "") or terminal_cfg.get("backend", "local") print(f" Backend: {terminal_env}") - if terminal_env == "ssh": - print(f" SSH Host: {os.getenv('TERMINAL_SSH_HOST', '') or '(not set)'}") - print(f" SSH User: {os.getenv('TERMINAL_SSH_USER', '') or '(not set)'}") - elif terminal_env == "docker": - print(f" Docker Image: {os.getenv('TERMINAL_DOCKER_IMAGE', 'python:3.11-slim')}") - elif terminal_env == "daytona": - print(f" Daytona Image: {os.getenv('TERMINAL_DAYTONA_IMAGE', 'nikolaik/python-nodejs:python3.11-nodejs20')}") + if terminal_env in _TERMINAL_ENV_ROWS: + for label, var, default, empty_is_unset in _TERMINAL_ENV_ROWS[terminal_env]: + value = (os.getenv(var, "") or default) if empty_is_unset else os.getenv(var, default) + print(f" {label:<13} {value}") elif terminal_env == "vercel_sandbox": runtime = os.getenv("TERMINAL_VERCEL_RUNTIME") or terminal_cfg.get("vercel_runtime") or "node24" persist = os.getenv("TERMINAL_CONTAINER_PERSISTENT") @@ -419,10 +427,8 @@ def show_status(args): # provider's doctor rows (fail-soft — never break `hermes status`). try: from hermes_cli.plugins import discover_plugins - discover_plugins() from agent.terminal_env_registry import get_provider - _provider = get_provider(terminal_env) if _provider is not None: for _ok, _label, _text in _provider.doctor_checks(): @@ -433,35 +439,12 @@ def show_status(args): sudo_password = os.getenv("SUDO_PASSWORD", "") print(f" Sudo: {check_mark(bool(sudo_password))} {'enabled' if sudo_password else 'disabled'}") - # ========================================================================= - # Messaging Platforms - # ========================================================================= + +def _render_platforms(ctx): _section("Messaging Platforms") - - platforms = { - "Telegram": ("TELEGRAM_BOT_TOKEN", "TELEGRAM_HOME_CHANNEL"), - "Discord": ("DISCORD_BOT_TOKEN", "DISCORD_HOME_CHANNEL"), - "WhatsApp": ("WHATSAPP_ENABLED", None), - "Signal": ("SIGNAL_HTTP_URL", "SIGNAL_HOME_CHANNEL"), - "Slack": ("SLACK_BOT_TOKEN", None), - "Email": ("EMAIL_ADDRESS", "EMAIL_HOME_ADDRESS"), - "SMS": ("TWILIO_ACCOUNT_SID", "SMS_HOME_CHANNEL"), - "DingTalk": ("DINGTALK_CLIENT_ID", None), - "Feishu": ("FEISHU_APP_ID", "FEISHU_HOME_CHANNEL"), - "WeCom": ("WECOM_BOT_ID", "WECOM_HOME_CHANNEL"), - "WeCom Callback": ("WECOM_CALLBACK_CORP_ID", None), - "Weixin": ("WEIXIN_ACCOUNT_ID", "WEIXIN_HOME_CHANNEL"), - "BlueBubbles": ("BLUEBUBBLES_SERVER_URL", "BLUEBUBBLES_HOME_CHANNEL"), - "QQBot": ("QQ_APP_ID", "QQ_HOME_CHANNEL"), - "Yuanbao": ("YUANBAO_APP_ID", "YUANBAO_HOME_CHANNEL"), - } - - for name, (token_var, home_var) in platforms.items(): + for name, (token_var, home_var) in _PLATFORMS.items(): has_token = bool(os.getenv(token_var, "")) home_channel = os.getenv(home_var, "") if home_var else "" - # Back-compat: QQBot home channel was renamed from QQ_HOME_CHANNEL to QQBOT_HOME_CHANNEL - if not home_channel and home_var == "QQBOT_HOME_CHANNEL": - home_channel = os.getenv("QQ_HOME_CHANNEL", "") status = "configured" if has_token else "not configured" if home_channel: status += f" (home: {home_channel})" @@ -482,11 +465,9 @@ def show_status(args): except Exception: pass - # ========================================================================= - # Gateway Status - # ========================================================================= - _section("Gateway Service") +def _render_gateway(ctx): + _section("Gateway Service") try: from hermes_cli.gateway import get_gateway_runtime_snapshot, _format_gateway_pids @@ -506,20 +487,15 @@ def show_status(args): except Exception: if _is_termux(): status_text, manager = "unknown", "Termux / manual process" - elif sys.platform.startswith('linux'): - status_text, manager = "unknown", "systemd/manual" - elif sys.platform == 'darwin': - status_text, manager = "unknown", "launchd" else: - status_text, manager = "N/A", "(not supported on this platform)" + platform = "linux" if sys.platform.startswith("linux") else sys.platform + status_text, manager = _GATEWAY_FALLBACK.get(platform, ("N/A", "(not supported on this platform)")) print(f" Status: {color(status_text, Colors.DIM)}") print(f" Manager: {manager}") - # ========================================================================= - # Cron Jobs - # ========================================================================= - _section("Scheduled Jobs") +def _render_cron(ctx): + _section("Scheduled Jobs") jobs_file = get_hermes_home() / "cron" / "jobs.json" if jobs_file.exists(): try: @@ -534,15 +510,12 @@ def show_status(args): else: print(" Jobs: 0") - # ========================================================================= - # Sessions - # ========================================================================= - _section("Sessions") +def _render_sessions(ctx): + _section("Sessions") # Gateway session count: state.db is the source of truth (#9006); # fall back to sessions.json for pre-migration installs. - _session_count = None - _gateway_rows = [] + _session_count, _gateway_rows = None, [] try: from hermes_state import SessionDB _db = SessionDB() @@ -554,15 +527,13 @@ def show_status(args): finally: _db.close() except Exception: - _session_count = None - _gateway_rows = [] + _session_count, _gateway_rows = None, [] if _session_count: print(f" Active: {_session_count} session(s)") freshest = max((float(r.get("last_active") or 0) for r in _gateway_rows), default=0.0) if freshest > 0: from hermes_cli.timefmt import relative_time - print(f" Last activity:{relative_time(freshest):>13}") else: sessions_file = get_hermes_home() / "sessions" / "sessions.json" @@ -585,8 +556,7 @@ def show_status(args): from hermes_cli.active_sessions import ( active_session_registry_snapshot, format_age, resolve_max_concurrent_sessions, ) - - _cap = resolve_max_concurrent_sessions(config) + _cap = resolve_max_concurrent_sessions(ctx.config) except Exception: _cap = None if _cap: @@ -604,36 +574,51 @@ def show_status(args): f"{_entry.get('session_id') or '?':<24} {_age}" ) - # ========================================================================= - # Deep checks - # ========================================================================= - if deep: - _section("Deep Checks") - # Check OpenRouter connectivity - openrouter_key = os.getenv("OPENROUTER_API_KEY", "") - if openrouter_key: - try: - import httpx - response = httpx.get(OPENROUTER_MODELS_URL, headers={"Authorization": f"Bearer {openrouter_key}"}, timeout=10) - ok = response.status_code == 200 - print(f" OpenRouter: {check_mark(ok)} {'reachable' if ok else f'error ({response.status_code})'}") - except Exception as e: - print(f" OpenRouter: {check_mark(False)} error: {e}") - - # Check gateway port +def _render_deep(ctx): + if not ctx.deep: + return + _section("Deep Checks") + # Check OpenRouter connectivity + openrouter_key = os.getenv("OPENROUTER_API_KEY", "") + if openrouter_key: try: - import socket - sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) - sock.settimeout(1) - port_in_use = sock.connect_ex(('127.0.0.1', 18789)) == 0 # informational: gateway likely running - sock.close() - print(f" Port 18789: {'in use' if port_in_use else 'available'}") - except OSError: - pass + import httpx + response = httpx.get(OPENROUTER_MODELS_URL, headers={"Authorization": f"Bearer {openrouter_key}"}, timeout=10) + ok = response.status_code == 200 + print(f" OpenRouter: {check_mark(ok)} {'reachable' if ok else f'error ({response.status_code})'}") + except Exception as e: + print(f" OpenRouter: {check_mark(False)} error: {e}") + # Check gateway port + try: + import socket + sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + sock.settimeout(1) + port_in_use = sock.connect_ex(('127.0.0.1', 18789)) == 0 # informational: gateway likely running + sock.close() + print(f" Port 18789: {'in use' if port_in_use else 'available'}") + except OSError: + pass + +def _render_footer(ctx): print() print(color("─" * 60, Colors.DIM)) print(color(" Run 'hermes doctor' for detailed diagnostics", Colors.DIM)) print(color(" Run 'hermes setup' to configure", Colors.DIM)) print() + + +# Print order of `hermes status`; each renderer takes the shared _StatusContext. +_SECTIONS = ( + _render_header, _render_environment, _render_api_keys, _render_auth_providers, _render_nous_gateway, + _render_apikey_providers, _render_terminal, _render_platforms, _render_gateway, _render_cron, + _render_sessions, _render_deep, _render_footer, +) + + +def show_status(args): + """Show status of all Hermes Agent components.""" + ctx = _StatusContext(deep=getattr(args, 'deep', False)) + for render in _SECTIONS: + render(ctx)