"""Shared runtime provider resolution for CLI, gateway, cron, and helpers. Layout: this module owns the resolution ORDER (:func:`resolve_runtime_provider`), the api_mode / base_url helpers and the pool / OAuth / explicit paths. Custom-provider lookup lives in :mod:`hermes_cli.runtime_provider_custom`; Azure Foundry, OpenRouter/bare-custom, Bedrock and external-process builders in :mod:`hermes_cli.runtime_provider_backends`. Both are re-exported here so ``hermes_cli.runtime_provider.`` imports and test patches keep working. """ from __future__ import annotations import logging import re from dataclasses import dataclass from typing import Any, Callable, Dict, Optional from urllib.parse import urlparse logger = logging.getLogger(__name__) from hermes_cli import auth as auth_mod from agent.credential_pool import ( CredentialPool, PooledCredential, credential_pool_matches_provider, custom_provider_pool_key_candidates, # noqa: F401 — read via origin by runtime_provider_custom (patchable) load_pool, ) from agent.secret_scope import get_secret as _get_secret from hermes_cli.auth import ( ACTUAL_LOCAL_NOAUTH_PLACEHOLDER, AuthError, DEFAULT_CODEX_BASE_URL, DEFAULT_QWEN_BASE_URL, DEFAULT_XAI_OAUTH_BASE_URL, PROVIDER_REGISTRY, _agent_key_is_usable, _nous_inference_env_override, format_auth_error, resolve_provider, resolve_nous_runtime_credentials, resolve_codex_runtime_credentials, resolve_xai_oauth_runtime_credentials, resolve_qwen_runtime_credentials, resolve_api_key_provider_credentials, resolve_external_process_provider_credentials, # noqa: F401 — read via origin by runtime_provider_backends has_usable_secret, is_actual_local_base_url, normalize_actual_base_url, ) from hermes_cli import config as _config_mod from hermes_constants import OPENROUTER_BASE_URL from hermes_cli.providers import is_official_openai_host from utils import base_url_host_matches, base_url_hostname, env_int def load_config(): """Late-bound delegate to :func:`hermes_cli.config.load_config`. Deliberately NOT a module-level from-import: this module is often imported lazily, so its first import can happen while a test has ``hermes_cli.config.load_config`` patched — a from-import would bind the MagicMock permanently and poison every later caller. """ return _config_mod.load_config() def get_compatible_custom_providers(config=None): """Late-bound delegate — see :func:`load_config` for why.""" return _config_mod.get_compatible_custom_providers(config) def normalize_extra_headers(value): """Late-bound delegate — see :func:`load_config` for why.""" return _config_mod.normalize_extra_headers(value) def _getenv(name: str, default: str = "") -> str: """Profile-scoped ``os.getenv`` for credential/provider reads. Identical to ``os.getenv`` when multiplexing is off; scope-aware (fail-closed on an unscoped read) when on. Genuinely-global vars are handled inside ``get_secret``. """ val = _get_secret(name, default) return val if val is not None else default def _loopback_hostname(host: str) -> bool: return (host or "").lower().rstrip(".") in {"localhost", "127.0.0.1", "::1", "0.0.0.0"} def _resolves_to_custom(name: str) -> bool: """True when a provider alias (ollama, vllm, llamacpp, …) resolves to ``custom``.""" try: return auth_mod.resolve_provider(name) == "custom" except Exception: return False def _config_base_url_trustworthy_for_bare_custom(cfg_base_url: str, cfg_provider: str) -> bool: """Whether ``model.base_url`` may back bare ``custom`` runtime resolution. The model picker can select Custom while ``model.provider`` still reflects a previous provider. Non-loopback URLs are rejected unless the YAML provider is already ``custom`` or a local-server alias (ollama/vllm/llamacpp — else a legit LAN ollama endpoint silently falls through to OpenRouter), so a stale OpenRouter/Z.ai base_url cannot hijack local sessions. """ cfg_provider_norm = (cfg_provider or "").strip().lower() bu = (cfg_base_url or "").strip() if not bu: return False if cfg_provider_norm == "custom" or _resolves_to_custom(cfg_provider_norm): return True if base_url_host_matches(bu, "openrouter.ai"): return False return _loopback_hostname(base_url_hostname(bu)) # ── api_mode detection ───────────────────────────────────────────────────────────────────── # Hosts that only speak one wire protocol. Mirrors host_mandated_api_mode in hermes_cli/providers.py # so the runtime resolver stays in lockstep: api.meta.ai — prompt caching only on Responses; # api.router.com — /v1/chat/completions is a minimal shim; api.anthropic.com — native Messages. _HOST_MANDATED_API_MODES = { "api.x.ai": "codex_responses", "api.meta.ai": "codex_responses", "api.actual.inc": "codex_responses", "api.router.com": "codex_responses", "api.anthropic.com": "anthropic_messages", } _VALID_API_MODES = { "chat_completions", "codex_responses", "anthropic_messages", "bedrock_converse", # Opt-in: hand the whole turn to a `codex app-server` subprocess (Codex's own tool runtime). # Gated on `model.openai_runtime == "codex_app_server"` AND provider in {openai, openai-codex}. "codex_app_server", } def _detect_api_mode_for_url(base_url: str) -> Optional[str]: """Auto-detect api_mode from the resolved base URL, or None. Exact-hostname matches reject lookalike subdomains (api.anthropic.com.attacker.test) and path-segment spoofing (proxy.test/api.anthropic.com/v1). Official OpenAI hosts (incl. the data-residency us./eu. regional hosts) need Responses for GPT-5.x tool calls with reasoning. """ normalized = (base_url or "").strip().lower().rstrip("/") hostname = base_url_hostname(base_url) mandated = _HOST_MANDATED_API_MODES.get(hostname) if mandated: return mandated if is_official_openai_host(base_url): return "codex_responses" path = urlparse(normalized).path.rstrip("/") if path.endswith("/anthropic") or path.endswith("/anthropic/v1"): return "anthropic_messages" if hostname == "api.kimi.com" and "/coding" in normalized: return "anthropic_messages" return None def _parse_api_mode(raw: Any) -> Optional[str]: """Validate an api_mode from config (None if invalid). Legacy/alias spellings (``openai``, ``anthropic``, ``responses``, …) are canonicalized first so old configs keep their transport instead of silently falling through to hostname-based detection.""" if isinstance(raw, str): from hermes_cli.config import _canonical_api_mode normalized = _canonical_api_mode(raw).lower() if normalized in _VALID_API_MODES: return normalized return None def _fallback_api_mode(provider: str, base_url: str, model: str = "") -> str: """api_mode when no explicit/persisted mode applies: URL detection (host-mandated wire shapes) first, then the transport the provider overlay declares via ``providers.determine_api_mode`` (which was never consulted before — ``openai-api`` pointed at us.api.openai.com 400'd on every tool call), then ``chat_completions``.""" detected = _detect_api_mode_for_url(base_url) if detected: return detected from hermes_cli.providers import determine_api_mode return determine_api_mode(provider, base_url, model) or "chat_completions" def _resolve_plain_custom_api_mode(model_cfg: Dict[str, Any], base_url: str) -> str: """api_mode for legacy/plain ``provider: custom`` endpoints — conservative by default: only direct OpenAI/xAI/Meta URLs imply Responses; named custom providers opt in via ``api_mode``.""" configured_mode = _parse_api_mode(model_cfg.get("api_mode")) detected_mode = _detect_api_mode_for_url(base_url) if configured_mode == "codex_responses" and detected_mode != "codex_responses": logger.info( "Ignoring persisted custom api_mode=codex_responses for non-OpenAI endpoint %s", base_url or "(unknown)", ) configured_mode = None return configured_mode or detected_mode or "chat_completions" def _provider_supports_explicit_api_mode(provider: Optional[str], configured_provider: Optional[str] = None) -> bool: """Whether a persisted api_mode may be honored for ``provider`` — only when the config's provider matches (or none is recorded), so a stale mode never leaks across a switch.""" normalized_provider = (provider or "").strip().lower() normalized_configured = (configured_provider or "").strip().lower() if not normalized_configured: return True if normalized_provider == "custom": return normalized_configured == "custom" or normalized_configured.startswith("custom:") return normalized_configured == normalized_provider def _copilot_runtime_api_mode(model_cfg: Dict[str, Any], api_key: str, *, target_model: Optional[str] = None) -> str: configured_mode = _parse_api_mode(model_cfg.get("api_mode")) if configured_mode and _provider_supports_explicit_api_mode("copilot", _cfg_provider(model_cfg)): return configured_mode # Use the model being resolved, not the persisted default: a Claude MoA slot inheriting # codex_responses from a GPT-5 default fails with "model ... does not support Responses API". model_name = str(target_model or model_cfg.get("default") or "").strip() if not model_name: return "chat_completions" try: from hermes_cli.models import copilot_model_api_mode return copilot_model_api_mode(model_name, api_key=api_key) except Exception: return "chat_completions" def _azure_inferred_api_mode(effective_model: str, api_mode: str) -> str: """Upgrade api_mode for GPT-5.x / codex / o1-o4 deployments on Azure Foundry (Azure 400s /chat/completions on these). Skipped when the user explicitly picked anthropic_messages.""" if not effective_model or api_mode == "anthropic_messages": return api_mode try: from hermes_cli.models import azure_foundry_model_api_mode inferred = azure_foundry_model_api_mode(effective_model) except Exception: inferred = None return inferred or api_mode def _configured_or_fallback_api_mode( provider: str, model_cfg: Dict[str, Any], base_url: str, effective_model: Any, *, opencode_by_model: bool ) -> str: """Persisted ``model.api_mode`` when it belongs to this provider, else URL/transport fallback. OpenCode Zen/Go serve both anthropic_messages and chat_completions models, so (when ``opencode_by_model``) their mode is always re-derived from the effective model. """ if opencode_by_model: from hermes_cli.models import opencode_provider_family if opencode_provider_family(provider) is not None: from hermes_cli.models import opencode_model_api_mode return opencode_model_api_mode(provider, effective_model) configured_mode = _parse_api_mode(model_cfg.get("api_mode")) if configured_mode and _provider_supports_explicit_api_mode(provider, _cfg_provider(model_cfg)): return configured_mode return _fallback_api_mode(provider, base_url, effective_model) def _api_key_provider_api_mode( provider: str, model_cfg: Dict[str, Any], api_key: str, base_url: str, effective_model: Any, *, opencode_by_model: bool ) -> str: """api_mode for a registry ``api_key`` provider (explicit and env/config paths).""" if provider == "copilot": return _copilot_runtime_api_mode(model_cfg, api_key, target_model=effective_model) if provider in ("xai", "actual"): return "codex_responses" return _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=opencode_by_model) def _maybe_apply_codex_app_server_runtime(*, provider: str, api_mode: str, model_cfg: Optional[Dict[str, Any]]) -> str: """Opt-in rewrite to "codex_app_server" via ``model.openai_runtime``; only ``openai`` / ``openai-codex`` are eligible. No-op when unset, "auto", or empty.""" if ( model_cfg and provider in {"openai", "openai-codex"} and str(model_cfg.get("openai_runtime") or "").strip().lower() == "codex_app_server" ): return "codex_app_server" return api_mode # ── base_url / credential helpers ────────────────────────────────────────────────────────── _ANTHROPIC_DEFAULT_BASE_URL = "https://api.anthropic.com" _NO_ANTHROPIC_CREDENTIALS_MSG = ( "No Anthropic credentials found. Set ANTHROPIC_TOKEN or ANTHROPIC_API_KEY, " "run 'claude setup-token', or authenticate with 'claude /login'." ) def _runtime(provider: str, api_mode: str, base_url: Any, api_key: Any, **extra: Any) -> Dict[str, Any]: """Build a resolved-runtime dict; ``extra`` carries source/requested_provider/provider-specific keys.""" return {"provider": provider, "api_mode": api_mode, "base_url": base_url, "api_key": api_key, **extra} def _cfg_provider(model_cfg: Dict[str, Any]) -> str: return str(model_cfg.get("provider") or "").strip().lower() def _config_base_url_for_provider(model_cfg: Dict[str, Any], provider: str) -> str: """``model.base_url`` (stripped, no trailing slash) only when ``model.provider`` is ``provider`` — a stale base_url must not leak into another provider.""" if _cfg_provider(model_cfg) != provider: return "" return str(model_cfg.get("base_url") or "").strip().rstrip("/") def _anthropic_base_url_override_ok(base_url: str) -> bool: """Whether a configured ``model.base_url`` plausibly speaks the Anthropic Messages protocol: official Anthropic/Claude hosts, Azure Foundry, or ``/anthropic`` / Kimi ``/coding`` proxies (the same signal :func:`_detect_api_mode_for_url` uses). Otherwise the caller falls back to ``https://api.anthropic.com`` so a stale non-Anthropic URL cannot hijack native Anthropic.""" candidate = (base_url or "").strip() hostname = (base_url_hostname(candidate) or "").lower() if candidate else "" if not hostname: return False if hostname == "api.anthropic.com" or hostname.endswith((".anthropic.com", ".claude.com", ".azure.com")): return True return _detect_api_mode_for_url(candidate) == "anthropic_messages" def _anthropic_cfg_base_url(model_cfg: Dict[str, Any]) -> str: """Config base_url for native Anthropic, or "" when absent/untrustworthy.""" cfg_base_url = _config_base_url_for_provider(model_cfg, "anthropic") return cfg_base_url if _anthropic_base_url_override_ok(cfg_base_url) else "" def _host_derived_api_key(base_url: str) -> str: """``_API_KEY`` from the env, vendor = registrable hostname label (``api.deepseek.com`` → ``deepseek``). Lookalike hosts pick the ATTACKER's label (api.deepseek.com.attacker.test → "attacker") so DEEPSEEK_API_KEY stays put. "" for IPs/loopback/single-label hosts and for OPENAI/OPENROUTER/OLLAMA, which have their own host-gated paths.""" hostname = base_url_hostname(base_url) if not hostname or any(ch.isdigit() for ch in hostname.split(".")[-1]) or hostname == "localhost" or ":" in hostname: return "" labels = [lbl for lbl in hostname.split(".") if lbl] while labels and labels[0] in ("api", "www"): labels.pop(0) if len(labels) < 2: return "" sanitized = "".join(ch if ch.isalnum() else "_" for ch in labels[-2]).upper() if not sanitized or not sanitized[0].isalpha() or sanitized in ("OPENAI", "OPENROUTER", "OLLAMA"): return "" return (_getenv(f"{sanitized}_API_KEY", "") or "").strip() def _host_gated_env_key_candidates(base_url: str, *, ollama: bool) -> list: """Env API keys gated on their authoritative hosts, then the host-derived ``_API_KEY``. Sending OPENAI/OPENROUTER/OLLAMA keys to an unrelated endpoint leaks credentials (GHSA-76xc-57q6-vm5m); match on HOST, not substring. ``_host_derived_api_key`` skips OLLAMA, so callers that want it opt in via ``ollama``. """ is_openai = base_url_host_matches(base_url, "openai.com") or base_url_host_matches(base_url, "openai.azure.com") candidates = [] if ollama: candidates.append(_getenv("OLLAMA_API_KEY", "").strip() if base_url_host_matches(base_url, "ollama.com") else "") candidates += [ _getenv("OPENAI_API_KEY", "").strip() if is_openai else "", _getenv("OPENROUTER_API_KEY", "").strip() if base_url_host_matches(base_url, "openrouter.ai") else "", _host_derived_api_key(base_url), ] return candidates def _pool_entry_api_key(entry: Any) -> str: return getattr(entry, "runtime_api_key", None) or getattr(entry, "access_token", "") def _pool_entry_base_url(entry: Any) -> str: return getattr(entry, "runtime_base_url", None) or getattr(entry, "base_url", None) or "" def _nous_pool_state(entry: Any) -> Dict[str, Any]: return {k: getattr(entry, k, None) for k in ("agent_key", "agent_key_expires_at", "scope")} def _registry_base_url(provider: str) -> str: pconfig = PROVIDER_REGISTRY.get(provider) return pconfig.inference_base_url if pconfig else "" def _nous_inference_base_url_override() -> str: """Trusted ``NOUS_INFERENCE_BASE_URL`` override (bypasses the network host allowlist); one normalization path via ``auth._nous_inference_env_override``.""" return _nous_inference_env_override() or "" def _nous_min_key_ttl() -> int: return max(60, env_int("HERMES_NOUS_MIN_KEY_TTL_SECONDS", 1800)) def _resolve_nous_creds() -> Dict[str, Any]: return resolve_nous_runtime_credentials(timeout_seconds=float(_getenv("HERMES_NOUS_TIMEOUT_SECONDS", "15"))) def _nous_api_mode(model: str) -> str: from hermes_cli.providers import nous_api_mode return nous_api_mode(model) def _normalize_opencode_runtime_base_url(provider: str, api_mode: str, base_url: str) -> str: """OpenCode base URLs end with /v1 for OpenAI-compatible models, but the Anthropic SDK prepends its own /v1/messages: strip /v1 for anthropic_messages, re-append otherwise.""" from hermes_cli.models import opencode_provider_family if opencode_provider_family(provider) is None: return base_url from hermes_cli.models import normalize_opencode_base_url return normalize_opencode_base_url(provider, api_mode, base_url) def _finalize_base_url(provider: str, api_mode: str, base_url: str) -> str: """Shared tail for pool-entry and api-key paths: OpenCode /v1 rule, then LM Studio normalization.""" base_url = _normalize_opencode_runtime_base_url(provider, api_mode, base_url) if provider == "lmstudio": base_url = auth_mod._normalize_lmstudio_runtime_base_url(base_url) return base_url # ── model config ─────────────────────────────────────────────────────────────────────────── def _auto_detect_local_model(base_url: str) -> str: """Query a local server for its model name when only one model is loaded.""" if not base_url: return "" try: import requests url = base_url.rstrip("/") if not url.endswith("/v1"): url += "/v1" resp = requests.get(url + "/models", timeout=(2, 3)) if resp.ok: models = resp.json().get("data", []) if len(models) == 1 and models[0].get("id", ""): return models[0]["id"] except Exception as exc: logger.debug("Auto-detect model from %s failed: %s", base_url, exc) return "" def _get_model_config() -> Dict[str, Any]: """``model`` config section with ``model`` accepted as an alias for ``default``, a dict ``default`` split into model/provider, and a local single-model server auto-detected.""" config = load_config() model_cfg = config.get("model") if isinstance(model_cfg, str) and model_cfg.strip(): return {"default": model_cfg.strip()} if not isinstance(model_cfg, dict): return {} cfg = dict(model_cfg) if not cfg.get("default") and cfg.get("model"): cfg["default"] = cfg["model"] _default = cfg.get("default") if isinstance(_default, dict): from hermes_cli.config import split_model_config_default cfg_model, cfg_provider = split_model_config_default(_default) cfg_provider = cfg_provider or str(model_cfg.get("provider") or "") cfg["default"] = cfg_model if cfg_provider and not cfg.get("provider"): cfg["provider"] = cfg_provider _default = cfg_model base_url = (cfg.get("base_url") or "").strip() if not str(_default or "").strip() and base_url and base_url_hostname(base_url) in ("localhost", "127.0.0.1"): detected = _auto_detect_local_model(base_url) if detected: cfg["default"] = detected return cfg def resolve_requested_provider(requested: Optional[str] = None) -> str: """Provider request from explicit arg, then config, then ``HERMES_INFERENCE_PROVIDER``, else "auto". Config beats the env so chat uses the endpoint the user last saved, not a stale shell/.env override.""" if requested and requested.strip(): return requested.strip().lower() cfg_provider = _get_model_config().get("provider") if isinstance(cfg_provider, str) and cfg_provider.strip(): return cfg_provider.strip().lower() return _getenv("HERMES_INFERENCE_PROVIDER", "").strip().lower() or "auto" # ── extracted collaborators (re-exported; see module docstring) ──────────────────────────── from hermes_cli.runtime_provider_custom import ( # noqa: E402,F401 _apply_custom_provider_extras, _custom_provider_request_overrides, _filter_capabilities, _find_custom_identity, _get_named_custom_provider, _lift_common_custom_fields, _lift_extra_headers, _lift_max_output_tokens, _lift_model_capabilities, _normalize_base_url_for_match, _normalize_custom_provider_name, _resolve_named_custom_runtime, _try_resolve_from_custom_pool, canonical_custom_identity, find_custom_provider_identity, find_custom_provider_identity_by_model, has_named_custom_provider, is_routable_provider, ) from hermes_cli.runtime_provider_backends import ( # noqa: E402,F401 _is_external_process_provider, _resolve_azure_foundry_runtime, _resolve_bedrock_runtime, _resolve_external_process_runtime, _resolve_openrouter_runtime, ) # ── credential-pool entries ──────────────────────────────────────────────────────────────── # Pool-entry providers whose api_mode is fixed: provider -> (api_mode, default base_url when the # pool entry carries none). Callables are evaluated lazily (registry lookups). MiniMax OAuth tokens # are valid only against the Anthropic Messages endpoint, so a stale model.api_mode from a prior # OpenAI-compatible provider is never honoured for it (it would 404 on /chat/completions). _POOL_ENTRY_SIMPLE_MODES: Dict[str, tuple] = { "openai-codex": ("codex_responses", DEFAULT_CODEX_BASE_URL), "xai-oauth": ("codex_responses", DEFAULT_XAI_OAUTH_BASE_URL), "qwen-oauth": ("chat_completions", DEFAULT_QWEN_BASE_URL), "minimax-oauth": ("anthropic_messages", lambda: _registry_base_url("minimax-oauth")), "openrouter": ("chat_completions", OPENROUTER_BASE_URL), "xai": ("codex_responses", ""), } def _resolve_runtime_from_pool_entry( *, provider: str, entry: PooledCredential, requested_provider: str, model_cfg: Optional[Dict[str, Any]] = None, pool: Optional[CredentialPool] = None, target_model: Optional[str] = None, ) -> Dict[str, Any]: model_cfg = model_cfg or _get_model_config() # The caller's target model (e.g. /model switch) beats the persisted default, else api_mode is # computed from a stale default. effective_model = target_model or model_cfg.get("default") or "" base_url = _pool_entry_base_url(entry).rstrip("/") api_key = _pool_entry_api_key(entry) if provider in _POOL_ENTRY_SIMPLE_MODES: api_mode, default_url = _POOL_ENTRY_SIMPLE_MODES[provider] base_url = base_url or (default_url() if callable(default_url) else default_url) elif provider == "anthropic": api_mode = "anthropic_messages" base_url = _anthropic_cfg_base_url(model_cfg) or base_url or _ANTHROPIC_DEFAULT_BASE_URL elif provider == "nous": api_mode = _nous_api_mode(effective_model) base_url = _nous_inference_base_url_override() or base_url elif provider == "copilot": api_mode = _copilot_runtime_api_mode(model_cfg, getattr(entry, "runtime_api_key", ""), target_model=effective_model) base_url = base_url or PROVIDER_REGISTRY["copilot"].inference_base_url elif provider == "azure-foundry": api_mode = "chat_completions" if _cfg_provider(model_cfg) == "azure-foundry": base_url = _config_base_url_for_provider(model_cfg, "azure-foundry") or base_url api_mode = _parse_api_mode(model_cfg.get("api_mode")) or api_mode api_mode = _azure_inferred_api_mode(effective_model, api_mode) if api_mode == "anthropic_messages": base_url = re.sub(r"/v1/?$", "", base_url) else: # Honour model.base_url only when the pool entry carries no explicit base_url (i.e. it # fell back to the registry default). Env var overrides win. pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and base_url.rstrip("/") == pconfig.inference_base_url.rstrip("/"): base_url = _config_base_url_for_provider(model_cfg, provider) or base_url api_mode = _configured_or_fallback_api_mode(provider, model_cfg, base_url, effective_model, opencode_by_model=True) base_url = _finalize_base_url(provider, api_mode, base_url) api_mode = _maybe_apply_codex_app_server_runtime(provider=provider, api_mode=api_mode, model_cfg=model_cfg) return _runtime( provider, api_mode, base_url, api_key, source=getattr(entry, "source", "pool"), credential_pool=pool, requested_provider=requested_provider, ) def _openrouter_should_use_pool(requested_provider, model_cfg, explicit_api_key, explicit_base_url) -> bool: """OpenRouter pool only for a plain openrouter/auto request with no custom endpoint or override.""" cfg_base_url = str(model_cfg.get("base_url") or "").strip() env_openai_base_url = _getenv("OPENAI_BASE_URL", "").strip() env_openrouter_base_url = _getenv("OPENROUTER_BASE_URL", "").strip() has_custom_endpoint = bool(explicit_base_url or env_openai_base_url or env_openrouter_base_url) or bool( cfg_base_url and _cfg_provider(model_cfg) in {"auto", "custom"} ) return requested_provider in {"openrouter", "auto"} and not has_custom_endpoint and not bool(explicit_api_key or explicit_base_url) def _refresh_nous_pool_entry(pool: CredentialPool, entry: Any, pool_api_key: str): """Nous pool entries carry the agent_key (an invoke JWT) which the pool does not refresh on selection (avoids network calls in `hermes auth list`); refresh here before falling back to singleton auth resolution. Returns (entry, pool_api_key) — key "" when still unusable.""" min_ttl = _nous_min_key_ttl() if _agent_key_is_usable(_nous_pool_state(entry), min_ttl): return entry, pool_api_key logger.debug("Nous pool entry agent_key expired/missing, refreshing selected pool entry") try: refreshed = pool.try_refresh_current() except Exception as exc: logger.debug("Nous pool entry refresh failed: %s", exc) refreshed = None if refreshed is not None: entry = refreshed pool_api_key = _pool_entry_api_key(entry) if not pool_api_key or not _agent_key_is_usable(_nous_pool_state(entry), min_ttl): logger.debug("Nous pool entry agent_key still unavailable, falling through to runtime resolution") pool_api_key = "" return entry, pool_api_key def _resolve_from_pool( provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key, explicit_base_url, target_model ) -> Optional[Dict[str, Any]]: """Runtime from the provider's credential pool, or None to continue down the ladder.""" should_use_pool = provider != "openrouter" or _openrouter_should_use_pool( requested_provider, model_cfg, explicit_api_key, explicit_base_url ) try: pool = load_pool(provider) if should_use_pool else None except Exception: pool = None if not (pool and pool.has_credentials()): return None entry = pool.select() pool_api_key = _pool_entry_api_key(entry) if entry is not None else "" if provider == "nous" and entry is not None: entry, pool_api_key = _refresh_nous_pool_entry(pool, entry, pool_api_key) if ( entry is not None and pool_api_key and credential_pool_matches_provider(pool, provider, base_url=_pool_entry_base_url(entry)) ): return _resolve_runtime_from_pool_entry( provider=provider, entry=entry, requested_provider=requested_provider, model_cfg=model_cfg, pool=pool, target_model=target_model, ) return None # ── explicit (--api-key / --base-url) path ───────────────────────────────────────────────── def _explicit_anthropic(requested_provider, model_cfg, api_key, base_url, target_model): base_url = base_url or _anthropic_cfg_base_url(model_cfg) or _ANTHROPIC_DEFAULT_BASE_URL if not api_key: from agent.anthropic_adapter import resolve_anthropic_token api_key = resolve_anthropic_token() if not api_key: raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) return _runtime("anthropic", "anthropic_messages", base_url, api_key, source="explicit", requested_provider=requested_provider) def _explicit_codex(requested_provider, model_cfg, api_key, explicit_base_url, target_model): base_url = explicit_base_url or DEFAULT_CODEX_BASE_URL last_refresh = None if not api_key: creds = resolve_codex_runtime_credentials() api_key = creds.get("api_key", "") last_refresh = creds.get("last_refresh") base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url return _runtime( "openai-codex", "codex_responses", base_url, api_key, source="explicit", last_refresh=last_refresh, requested_provider=requested_provider, ) def _explicit_nous(requested_provider, model_cfg, api_key, explicit_base_url, target_model): state = auth_mod.get_provider_auth_state("nous") or {} base_url = ( explicit_base_url or _nous_inference_base_url_override() or str(state.get("inference_base_url") or auth_mod.DEFAULT_NOUS_INFERENCE_URL).strip().rstrip("/") ) # The agent_key compatibility field is used for inference only when it holds a NAS invoke JWT; # raw OAuth access_token fallback is handled by resolve_nous_runtime_credentials(). api_key = api_key or ( str(state.get("agent_key") or "").strip() if _agent_key_is_usable(state, _nous_min_key_ttl()) else "" ) expires_at = state.get("agent_key_expires_at") or state.get("expires_at") if not api_key: creds = _resolve_nous_creds() api_key = creds.get("api_key", "") expires_at = creds.get("expires_at") base_url = explicit_base_url or creds.get("base_url", "").rstrip("/") or base_url return _runtime( "nous", _nous_api_mode(target_model or model_cfg.get("default") or ""), base_url, api_key, source="explicit", expires_at=expires_at, requested_provider=requested_provider, ) def _explicit_azure_foundry(requested_provider, model_cfg, api_key, base_url, target_model): return _resolve_azure_foundry_runtime( requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=api_key, explicit_base_url=base_url ) def _explicit_api_key_provider(provider, pconfig, requested_provider, model_cfg, api_key, base_url, target_model): if not base_url: if provider in {"kimi-coding", "kimi-coding-cn"}: base_url = resolve_api_key_provider_credentials(provider).get("base_url", "").rstrip("/") else: env_url = _getenv(pconfig.base_url_env_var, "").strip().rstrip("/") if pconfig.base_url_env_var else "" base_url = env_url or pconfig.inference_base_url if provider == "actual": base_url = normalize_actual_base_url(base_url) if not api_key: creds = resolve_api_key_provider_credentials(provider) api_key = creds.get("api_key", "") if not base_url: base_url = creds.get("base_url", "").rstrip("/") if provider == "actual": base_url = normalize_actual_base_url(base_url) api_mode = _api_key_provider_api_mode( provider, model_cfg, api_key, base_url, target_model or model_cfg.get("default", ""), opencode_by_model=False ) if provider == "actual" and not api_key and is_actual_local_base_url(base_url): api_key = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER return _runtime(provider, api_mode, base_url.rstrip("/"), api_key, source="explicit", requested_provider=requested_provider) # Providers with a dedicated explicit-credential builder; everything else goes through the # registry ``api_key`` path (or None when the provider takes no explicit creds). _EXPLICIT_RESOLVERS: Dict[str, Callable[..., Dict[str, Any]]] = { "anthropic": _explicit_anthropic, "openai-codex": _explicit_codex, "nous": _explicit_nous, "azure-foundry": _explicit_azure_foundry, } def _resolve_explicit_runtime( *, provider: str, requested_provider: str, model_cfg: Dict[str, Any], explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, ) -> Optional[Dict[str, Any]]: explicit_api_key = str(explicit_api_key or "").strip() explicit_base_url = str(explicit_base_url or "").strip().rstrip("/") if not explicit_api_key and not explicit_base_url: return None resolver = _EXPLICIT_RESOLVERS.get(provider) if resolver is not None: return resolver(requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and pconfig.auth_type == "api_key": return _explicit_api_key_provider( provider, pconfig, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model ) return None # ── OAuth / auth-store providers ─────────────────────────────────────────────────────────── @dataclass(frozen=True) class _OAuthRuntimeSpec: """Env/auth-store OAuth providers resolved by a single credential call.""" resolve: Callable[[], Dict[str, Any]] api_mode: Any # str, or callable(model) -> str default_source: str expiry_key: str failure_msg: str default_base_url: str = "" # ``resolve`` entries are late-bound lambdas so tests can monkeypatch the module-level # ``resolve_*_runtime_credentials`` names. _OAUTH_RUNTIME_PROVIDERS: Dict[str, _OAuthRuntimeSpec] = { "nous": _OAuthRuntimeSpec( _resolve_nous_creds, _nous_api_mode, "portal", "expires_at", "Auto-detected Nous provider but credentials failed", ), "openai-codex": _OAuthRuntimeSpec( lambda: resolve_codex_runtime_credentials(), "codex_responses", "hermes-auth-store", "last_refresh", "Auto-detected Codex provider but credentials failed", ), "xai-oauth": _OAuthRuntimeSpec( lambda: resolve_xai_oauth_runtime_credentials(), "codex_responses", "hermes-auth-store", "last_refresh", "Auto-detected xAI OAuth provider but credentials failed", default_base_url=DEFAULT_XAI_OAUTH_BASE_URL, ), "qwen-oauth": _OAuthRuntimeSpec( lambda: resolve_qwen_runtime_credentials(), "chat_completions", "qwen-cli", "expires_at_ms", "Qwen OAuth credentials failed", ), } def _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) -> Optional[Dict[str, Any]]: """Runtime from an ``_OAUTH_RUNTIME_PROVIDERS`` spec. On AuthError: re-raise for an explicit request; for "auto" (auto-detected but credentials stale/revoked) log and return None so the ladder falls through to env-var providers (e.g. OpenRouter).""" spec = _OAUTH_RUNTIME_PROVIDERS[provider] try: creds = spec.resolve() except AuthError: if requested_provider != "auto": raise logger.info("%s; falling through to next provider.", spec.failure_msg) return None api_mode = spec.api_mode if callable(api_mode): api_mode = api_mode(target_model or model_cfg.get("default") or "") return _runtime( provider, api_mode, (creds.get("base_url") or "").rstrip("/") or spec.default_base_url, creds.get("api_key", ""), source=creds.get("source", spec.default_source), **{spec.expiry_key: creds.get(spec.expiry_key)}, requested_provider=requested_provider, ) # ── env/config paths for anthropic and registry api_key providers ────────────────────────── def _anthropic_env_runtime(requested_provider: str, model_cfg: Dict[str, Any]) -> Dict[str, Any]: """Native Anthropic (Messages API) from env/auth store; ``model.base_url`` honoured only when the configured provider is anthropic (else a Codex endpoint would leak into Anthropic requests).""" cfg_base_url = _anthropic_cfg_base_url(model_cfg) base_url = cfg_base_url or _ANTHROPIC_DEFAULT_BASE_URL # Microsoft Foundry endpoints reject Claude Code OAuth tokens, which resolve_anthropic_token() # would return first — use the env key directly: `key_env` / `api_key_env` hints on the model # config, then an inline api_key (multi-profile setups), then the historical fixed names. if base_url_host_matches(base_url, "azure.com") or (cfg_base_url and base_url_host_matches(cfg_base_url, "azure.com")): token = "" for hint_key in ("key_env", "api_key_env"): env_var = str(model_cfg.get(hint_key) or "").strip() if env_var: token = _getenv(env_var, "").strip() if token: break token = ( token or str(model_cfg.get("api_key") or "").strip() or _getenv("AZURE_ANTHROPIC_KEY", "").strip() or _getenv("ANTHROPIC_API_KEY", "").strip() ) if not token: raise AuthError( "No Azure Anthropic API key found. Set AZURE_ANTHROPIC_KEY or " "ANTHROPIC_API_KEY, or point key_env/api_key_env in your " "config.yaml model section at a custom env var." ) else: from agent.anthropic_adapter import resolve_anthropic_token token = resolve_anthropic_token() if not token: raise AuthError(_NO_ANTHROPIC_CREDENTIALS_MSG) return _runtime("anthropic", "anthropic_messages", base_url, token, source="env", requested_provider=requested_provider) def _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) -> Dict[str, Any]: """Registry ``api_key`` providers (z.ai/GLM, Kimi, MiniMax, copilot, …) from env/config.""" creds = resolve_api_key_provider_credentials(provider) # Actual Computer: a loopback model_cfg base_url selects the daemon's no-auth local API; inject # the placeholder BEFORE the usable-secret gate (mirrors the env-driven path). if provider == "actual" and not has_usable_secret(creds.get("api_key")): cfg_url = _config_base_url_for_provider(model_cfg, provider) if is_actual_local_base_url(normalize_actual_base_url(cfg_url or creds.get("base_url", "").rstrip("/"))): creds = dict(creds) creds["api_key"] = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER creds["source"] = creds.get("source") or "local-offline" # An explicitly selected API-key provider is authoritative: an empty key would defer failure # to the first request and make a later fallback look like a silent provider switch. if not has_usable_secret(creds.get("api_key")): env_names = ", ".join(pconfig.api_key_env_vars) hint = f" Set {env_names}." if env_names else "" raise AuthError(f"No usable credentials found for provider '{provider}'.{hint}", provider=provider, code="missing_api_key") # Honour model.base_url when the configured provider matches (e.g. api.minimaxi.com China endpoint). base_url = _config_base_url_for_provider(model_cfg, provider) or creds.get("base_url", "").rstrip("/") if provider == "actual": base_url = normalize_actual_base_url(base_url) api_mode = _api_key_provider_api_mode( provider, model_cfg, creds.get("api_key", ""), base_url, target_model or model_cfg.get("default", ""), opencode_by_model=True ) base_url = _finalize_base_url(provider, api_mode, base_url) api_key = creds.get("api_key", "") if provider == "actual" and not api_key and is_actual_local_base_url(base_url): api_key = ACTUAL_LOCAL_NOAUTH_PLACEHOLDER return _runtime(provider, api_mode, base_url, api_key, source=creds.get("source", "env"), requested_provider=requested_provider) # ── the resolution ladder ────────────────────────────────────────────────────────────────── _VERTEX_NAMES = ("vertex", "google-vertex", "vertex-ai", "gcp-vertex", "vertexai") _LOCAL_BYPASS_CLOUD_HOSTS = ("openrouter.ai", "anthropic.com", "openai.com") def _raise_if_provider_disabled(requested_provider: str) -> None: """Honour ``providers..enabled: false`` for built-ins too (the custom lookup gate only covers custom blocks); a typed error lets the fallback chain advance.""" from hermes_cli.config import is_provider_enabled, load_config full_cfg = load_config() provs_cfg = full_cfg.get("providers") if isinstance(full_cfg, dict) else None block = provs_cfg.get(requested_provider) if isinstance(provs_cfg, dict) else None if isinstance(block, dict) and not is_provider_enabled(block): raise ValueError( f"provider {requested_provider!r} is disabled in config " f"(providers.{requested_provider}.enabled: false)" ) def _resolve_vertex_runtime(requested_provider: str) -> Dict[str, Any]: """Vertex AI (OAuth2). The credential *path* (GOOGLE_APPLICATION_CREDENTIALS) must never be treated as a static API key; a short-lived token is minted per call, and mid-session expiry is recovered on 401 by run_agent._try_refresh_vertex_client_credentials().""" from agent.vertex_adapter import get_vertex_config token, base_url = get_vertex_config() if not token or not base_url: raise AuthError( "Vertex AI credentials could not be resolved. Vertex uses " "OAuth2 (not a static API key): provide a service-account JSON " "via GOOGLE_APPLICATION_CREDENTIALS (or VERTEX_CREDENTIALS_PATH) " "in ~/.hermes/.env, or run 'gcloud auth application-default " "login' for ADC. Set the GCP project/region under vertex: in " "config.yaml if they aren't embedded in the credentials. " "Run `hermes setup` to install Vertex support." ) return _runtime("vertex", "chat_completions", base_url.rstrip("/"), token, source="vertex-oauth", requested_provider=requested_provider) def _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) -> Optional[Dict[str, Any]]: """Providers decided on the REQUESTED name alone, before custom / pool / generic paths.""" if requested_provider == "moa": return _runtime( "moa", "chat_completions", "moa://local", "moa-virtual-provider", source="moa-virtual-provider", requested_provider=requested_provider, ) # Azure Anthropic short-circuit: an explicit Azure endpoint with provider="anthropic" must # bypass _resolve_named_custom_runtime (which would yield custom/chat_completions/no key). eff_base = (explicit_base_url or "").strip() if requested_provider == "anthropic" and base_url_host_matches(eff_base, "azure.com"): azure_key = ( (explicit_api_key or "").strip() or _getenv("AZURE_ANTHROPIC_KEY", "").strip() or _getenv("ANTHROPIC_API_KEY", "").strip() ) return _runtime( "anthropic", "anthropic_messages", eff_base.rstrip("/"), azure_key, source="azure-explicit", requested_provider=requested_provider, ) # Azure Foundry resolves before the custom-runtime / pool / generic paths so its config is # always picked up from model.base_url + model.api_mode, with or without explicit_* args. if requested_provider == "azure-foundry": return _resolve_azure_foundry_runtime( requested_provider=requested_provider, model_cfg=_get_model_config(), explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, ) if requested_provider in _VERTEX_NAMES: return _resolve_vertex_runtime(requested_provider) return None def _local_endpoint_bypass(requested_provider: str, explicit_api_key, explicit_base_url) -> Optional[Dict[str, Any]]: """provider "auto"/unset with a config base_url at a custom/local endpoint routes through the OpenAI-compatible resolver, so resolve_provider() cannot pick up an env ANTHROPIC/OPENAI key and send the request to a cloud API. Only non-cloud roots take the bypass; match on HOST, not substring, so a look-alike (api.anthropic.com.attacker.test) cannot leak a cloud credential.""" model_cfg = _get_model_config() cfg_base_url = str(model_cfg.get("base_url") or "").strip() if not cfg_base_url or _cfg_provider(model_cfg) not in ("auto", ""): return None if any(base_url_host_matches(cfg_base_url, host) for host in _LOCAL_BYPASS_CLOUD_HOSTS): return None runtime = _resolve_openrouter_runtime( requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url ) runtime["requested_provider"] = requested_provider return runtime def _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) -> Optional[Dict[str, Any]]: """OpenCode Zen free tier (*-free slugs) is served ANONYMOUSLY on the Zen relay only: unknown bearers 401 and the Go relay rejects free models, so free slugs route through the keyless Zen runtime BEFORE the pool / explicit / api_key paths.""" from hermes_cli.models import opencode_provider_family, opencode_zen_free_runtime if opencode_provider_family(provider) is None: return None model = str(target_model or model_cfg.get("default") or model_cfg.get("model") or "").strip() free_runtime = opencode_zen_free_runtime(provider, model) if free_runtime is not None: free_runtime["requested_provider"] = requested_provider return free_runtime def resolve_runtime_provider( *, requested: Optional[str] = None, explicit_api_key: Optional[str] = None, explicit_base_url: Optional[str] = None, target_model: Optional[str] = None, ) -> Dict[str, Any]: """Resolve runtime provider credentials for agent execution. Ladder (order is behavior — each rung returns or raises, else falls to the next): 1. disabled-provider guard (``providers..enabled: false``) 2. requested-name shortcuts: moa, anthropic@azure, azure-foundry, vertex 3. named custom provider / llamacpp alias / bare-custom direct alias 4. local-endpoint bypass (no explicit creds, config base_url at a non-cloud host) 5. ``auth.resolve_provider`` → OpenCode free tier → explicit --api-key/--base-url path 6. credential pool (OpenRouter pool only without custom endpoint/override) 7. OAuth specs (nous/codex/xai/qwen; "auto" swallows AuthError and logs) → minimax-oauth → external-process → anthropic env → bedrock → registry api_key providers 8. OpenRouter / bare-custom fallback target_model: overrides model_cfg["default"] when computing provider-specific api_mode (e.g. OpenCode Zen/Go where different models route through different API surfaces). """ requested_provider = resolve_requested_provider(requested) _raise_if_provider_disabled(requested_provider) runtime = _resolve_requested_shortcuts(requested_provider, explicit_api_key, explicit_base_url, target_model) if runtime: return runtime runtime = _resolve_named_custom_runtime( requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, ) if runtime: runtime["requested_provider"] = requested_provider return runtime if not explicit_base_url and not explicit_api_key: runtime = _local_endpoint_bypass(requested_provider, explicit_api_key, explicit_base_url) if runtime: return runtime provider = resolve_provider(requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url) model_cfg = _get_model_config() runtime = _opencode_free_runtime(provider, requested_provider, model_cfg, target_model) if runtime is not None: return runtime runtime = _resolve_explicit_runtime( provider=provider, requested_provider=requested_provider, model_cfg=model_cfg, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url, target_model=target_model, ) if runtime: return runtime runtime = _resolve_from_pool(provider, requested_provider, model_cfg, explicit_api_key, explicit_base_url, target_model) if runtime: return runtime if provider in _OAUTH_RUNTIME_PROVIDERS: runtime = _resolve_oauth_runtime(provider, requested_provider, model_cfg, target_model) if runtime: return runtime if provider == "minimax-oauth": pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and pconfig.auth_type == "oauth_minimax": from hermes_cli.auth import resolve_minimax_oauth_runtime_credentials creds = resolve_minimax_oauth_runtime_credentials() return _runtime( provider, "anthropic_messages", creds["base_url"], creds["api_key"], source=creds.get("source", "oauth"), requested_provider=requested_provider, ) if _is_external_process_provider(provider): return _resolve_external_process_runtime(provider, requested_provider) if provider == "anthropic": return _anthropic_env_runtime(requested_provider, model_cfg) if provider == "bedrock": return _resolve_bedrock_runtime(requested_provider, model_cfg, target_model) pconfig = PROVIDER_REGISTRY.get(provider) if pconfig and pconfig.auth_type == "api_key": return _api_key_provider_runtime(provider, pconfig, requested_provider, model_cfg, target_model) runtime = _resolve_openrouter_runtime( requested_provider=requested_provider, explicit_api_key=explicit_api_key, explicit_base_url=explicit_base_url ) runtime["requested_provider"] = requested_provider return runtime def format_runtime_provider_error(error: Exception) -> str: if isinstance(error, AuthError): return format_auth_error(error) return str(error)