51e39af967
The free tier depends on the account service (NAS) and the welcome inference host, and Hermes had no honest answer for most of the ways either can refuse or fail: the NAS codes it matched were never sent, the tier-dark 403 carried no message to match, a single boot-time blip disabled minting for the whole process, and a structured rate-limit refusal never reached the cross-session guard, so the "sign in for a bigger allowance" prompt was dead code. Backend - anon_auth: classify what NAS actually sends (404 not_found, 503 temporarily_disabled, 429 + Retry-After, 428 pow_*, 403 account_locked) into one ANON_* code each, carrying retry_after / retryable on AuthError. - Replace the process-lifetime mint memo with a per-profile cooldown that honours the server's wait, climbs a short ladder when the service is unreachable, never retries terminal codes, and yields to the user's own retry (force=True). - Bootstrap record carries error_code / retryable / retry_after; a bounded background loop retries transient failures and re-announces setup.ready. setup.status and free_tier.status expose the block; free_tier.provision is the forced retry. - Inference: a generic 403 from a welcome host is the tier refusing (keyed on the route); model_not_free moves onto the gateway's alternate once; anon_on_paid_host re-reads the route once; a long rate_limited refusal trips the cross-session guard; a locked account is retired but never replaced; terminal copy on the free route is one plain sentence. - Sign-in: Failed keeps the service's code and wait; account_busy is retryable; the OAuth poll reports retryable / retry_after. - All user-facing copy rewritten for first-time users: never "the free service is off" (what is unavailable is using Hermes without signing in, and signing in is free), no jargon, spoken waits. Desktop - A setup-failure notice above the provider picker: one sentence per code, a retry when the backend says one can work, the sign-in pointer only when the account service answered at all. The overlay re-checks readiness on setup.ready so a background success dismisses it. - Sign-in dialog gains busy / unreachable / unavailable screens. Rehearsal - scripts/free_tier_fault_server.py stands in for both services with the real wire contract and a CORS-open scenario switch; HERMES_EXTRA_WELCOME_HOSTS (dev-only, env-only) lets the route rules treat it as the welcome host. Walkthrough in website/docs/developer-guide/free-tier-fault-rehearsal.md. Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
184 lines
8.9 KiB
Python
184 lines
8.9 KiB
Python
"""Shared constants, the lazy ``httpx`` proxy and :class:`AuthError` for the auth package.
|
|
|
|
Pure leaf: imports nothing from ``hermes_cli.auth`` so the per-provider modules
|
|
(``auth_nous``, ``auth_codex``, ...) can import it at module scope without cycles."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import base64
|
|
import json
|
|
from typing import Any, Callable, Dict, Optional
|
|
|
|
# httpx is imported lazily (~30ms) because hermes_cli.auth is on the interactive-CLI startup path
|
|
# (credential_pool -> auxiliary_client -> cli_commands_mixin). The proxy resolves on first attribute
|
|
# access; ``from __future__ import annotations`` keeps ``httpx.Client`` annotations unevaluated.
|
|
import importlib as _importlib
|
|
from typing import TYPE_CHECKING
|
|
|
|
if TYPE_CHECKING:
|
|
import httpx
|
|
else:
|
|
class _LazyHttpx:
|
|
__slots__ = ("_mod",)
|
|
|
|
def __init__(self) -> None:
|
|
object.__setattr__(self, "_mod", None)
|
|
|
|
def _resolve(self):
|
|
mod = object.__getattribute__(self, "_mod")
|
|
if mod is None:
|
|
mod = _importlib.import_module("httpx")
|
|
object.__setattr__(self, "_mod", mod)
|
|
return mod
|
|
|
|
def __getattr__(self, name):
|
|
return getattr(self._resolve(), name)
|
|
|
|
# set/del forward to the real module so monkeypatch.setattr("hermes_cli.auth.httpx.Client")
|
|
# keeps working in tests.
|
|
def __setattr__(self, name, value):
|
|
setattr(self._resolve(), name, value)
|
|
|
|
def __delattr__(self, name):
|
|
delattr(self._resolve(), name)
|
|
|
|
httpx = _LazyHttpx()
|
|
|
|
# ── Constants ───────────────────────────────────────────────────────────────────────────────────────
|
|
|
|
AUTH_STORE_VERSION = 1
|
|
AUTH_LOCK_TIMEOUT_SECONDS = 15.0
|
|
|
|
# Nous Portal defaults
|
|
DEFAULT_NOUS_PORTAL_URL = "https://portal.nousresearch.com"
|
|
DEFAULT_NOUS_INFERENCE_URL = "https://inference-api.nousresearch.com/v1"
|
|
# The free tier's (anonymous account) inference host. NAS hands it to the client on every token
|
|
# exchange (``inference_base_url``); this literal is the fallback when that field is absent or fails
|
|
# the host allowlist, because the paid host cross-refuses an anonymous JWT with a 400.
|
|
DEFAULT_NOUS_WELCOME_URL = "https://welcome-api.nousresearch.com/v1"
|
|
DEFAULT_NOUS_CLIENT_ID = "hermes-cli"
|
|
NOUS_INFERENCE_INVOKE_SCOPE = "inference:invoke"
|
|
NOUS_BILLING_MANAGE_SCOPE = "billing:manage"
|
|
DEFAULT_NOUS_SCOPE = NOUS_INFERENCE_INVOKE_SCOPE
|
|
NOUS_DEVICE_CODE_SOURCE = "device_code"
|
|
NOUS_AUTH_PATH_INVOKE_JWT = "invoke_jwt"
|
|
ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120 # refresh 2 min before expiry
|
|
NOUS_INVOKE_JWT_MIN_TTL_SECONDS = ACCESS_TOKEN_REFRESH_SKEW_SECONDS
|
|
DEVICE_AUTH_POLL_INTERVAL_CAP_SECONDS = 1 # poll at most every 1s
|
|
DEVICE_CODE_GRANT_TYPE = "urn:ietf:params:oauth:grant-type:device_code"
|
|
_FORM_JSON_HEADERS = {"Content-Type": "application/x-www-form-urlencoded", "Accept": "application/json"}
|
|
DEFAULT_CODEX_BASE_URL = "https://chatgpt.com/backend-api/codex"
|
|
DEFAULT_XAI_OAUTH_BASE_URL = "https://api.x.ai/v1"
|
|
MINIMAX_OAUTH_CLIENT_ID = "78257093-7e40-4613-99e0-527b14b39113"
|
|
MINIMAX_OAUTH_SCOPE = "group_id profile model.completion"
|
|
MINIMAX_OAUTH_GRANT_TYPE = "urn:ietf:params:oauth:grant-type:user_code"
|
|
MINIMAX_OAUTH_GLOBAL_BASE = "https://api.minimax.io"
|
|
MINIMAX_OAUTH_CN_BASE = "https://api.minimaxi.com"
|
|
MINIMAX_OAUTH_GLOBAL_INFERENCE = "https://api.minimax.io/anthropic"
|
|
MINIMAX_OAUTH_CN_INFERENCE = "https://api.minimaxi.com/anthropic"
|
|
MINIMAX_OAUTH_REFRESH_SKEW_SECONDS = 60
|
|
DEFAULT_QWEN_BASE_URL = "https://portal.qwen.ai/v1"
|
|
DEFAULT_GITHUB_MODELS_BASE_URL = "https://api.githubcopilot.com"
|
|
DEFAULT_COPILOT_ACP_BASE_URL = "acp://copilot"
|
|
DEFAULT_OLLAMA_CLOUD_BASE_URL = "https://ollama.com/v1"
|
|
DEFAULT_ACTUAL_BASE_URL = "https://api.actual.inc/v1"
|
|
DEFAULT_ACTUAL_LOCAL_BASE_URL = "http://127.0.0.1:8080/v1"
|
|
STEPFUN_STEP_PLAN_INTL_BASE_URL = "https://api.stepfun.ai/step_plan/v1"
|
|
STEPFUN_STEP_PLAN_CN_BASE_URL = "https://api.stepfun.com/step_plan/v1"
|
|
CODEX_OAUTH_CLIENT_ID = "app_EMoamEEZ73f0CkXaXp7hrann"
|
|
CODEX_OAUTH_TOKEN_URL = "https://auth.openai.com/oauth/token"
|
|
try: # Version tag for the Codex token-endpoint User-Agent; fall back if unavailable.
|
|
from hermes_cli import __version__ as _HERMES_CLI_VERSION
|
|
except Exception: # pragma: no cover - version import should always succeed
|
|
_HERMES_CLI_VERSION = "unknown"
|
|
CODEX_OAUTH_USER_AGENT = f"hermes-cli/{_HERMES_CLI_VERSION}"
|
|
CODEX_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
|
|
XAI_OAUTH_ISSUER = "https://auth.x.ai"
|
|
XAI_OAUTH_DISCOVERY_URL = f"{XAI_OAUTH_ISSUER}/.well-known/openid-configuration"
|
|
XAI_OAUTH_CLIENT_ID = "b1a00492-073a-47ea-816f-4c329264a828"
|
|
XAI_OAUTH_SCOPE = "openid profile email offline_access grok-cli:access api:access"
|
|
XAI_OAUTH_DEVICE_CODE_URL = f"{XAI_OAUTH_ISSUER}/oauth2/device/code"
|
|
# xAI/Grok OAuth access tokens are short-lived (~6h). A two-minute refresh window leaves noisy
|
|
# credential-expiry gaps for gateway/cron workloads that touch the provider every ~30 min, so refresh
|
|
# up to an hour early.
|
|
XAI_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 3600
|
|
QWEN_OAUTH_CLIENT_ID = "f0304373b74a44d2b584a3fb70ca9e56"
|
|
QWEN_OAUTH_TOKEN_URL = "https://chat.qwen.ai/api/v1/oauth2/token"
|
|
QWEN_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
|
|
DEFAULT_SPOTIFY_ACCOUNTS_BASE_URL = "https://accounts.spotify.com"
|
|
DEFAULT_SPOTIFY_API_BASE_URL = "https://api.spotify.com/v1"
|
|
DEFAULT_SPOTIFY_REDIRECT_URI = "http://127.0.0.1:43827/spotify/callback"
|
|
SPOTIFY_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/user-guide/features/spotify"
|
|
SPOTIFY_DASHBOARD_URL = "https://developer.spotify.com/dashboard"
|
|
SPOTIFY_ACCESS_TOKEN_REFRESH_SKEW_SECONDS = 120
|
|
# OpenRouter PKCE (https://openrouter.ai/docs/guides/overview/auth/oauth): the "token" endpoint
|
|
# mints a plain user-controlled API key; there is no refresh token.
|
|
OPENROUTER_AUTH_URL = "https://openrouter.ai/auth"
|
|
OPENROUTER_AUTH_KEYS_URL = "https://openrouter.ai/api/v1/auth/keys"
|
|
OPENROUTER_OAUTH_DOCS_URL = "https://openrouter.ai/docs/guides/overview/auth/oauth"
|
|
|
|
OAUTH_OVER_SSH_DOCS_URL = "https://hermes-agent.nousresearch.com/docs/guides/oauth-over-ssh"
|
|
DEFAULT_SPOTIFY_SCOPE = " ".join((
|
|
"user-modify-playback-state", "user-read-playback-state", "user-read-currently-playing",
|
|
"user-read-recently-played", "playlist-read-private", "playlist-read-collaborative",
|
|
"playlist-modify-public", "playlist-modify-private", "user-library-read", "user-library-modify",
|
|
))
|
|
SERVICE_PROVIDER_NAMES: Dict[str, str] = {"spotify": "Spotify"}
|
|
|
|
# LM Studio's default no-auth mode still needs *some* non-empty bearer for the API-key code paths to
|
|
# treat the provider as configured. Sent only to LM Studio, never to a remote service.
|
|
LMSTUDIO_NOAUTH_PLACEHOLDER = "dummy-lm-api-key"
|
|
ACTUAL_LOCAL_NOAUTH_PLACEHOLDER = "dummy-actual-local-api-key"
|
|
|
|
# Upstream rate-limit / usage-quota exhaustion (HTTP 429): transient, re-authenticating cannot resolve
|
|
# it, so it must stay distinct from missing/expired-credential errors.
|
|
CODEX_RATE_LIMITED_CODE = "codex_rate_limited"
|
|
|
|
|
|
class AuthError(RuntimeError):
|
|
"""Structured auth error with UX mapping hints."""
|
|
|
|
def __init__(
|
|
self, message: str, *, provider: str = "", code: Optional[str] = None, relogin_required: bool = False,
|
|
retry_after: Optional[float] = None, retryable: Optional[bool] = None,
|
|
) -> None:
|
|
super().__init__(message)
|
|
self.provider = provider
|
|
self.code = code
|
|
self.relogin_required = relogin_required
|
|
# Optional wait hint in seconds (a server ``Retry-After`` or a client cooldown) and whether a
|
|
# later attempt can succeed at all. None = the raiser did not say; callers treat None as
|
|
# "retryable, no hint" for transport-shaped errors and as terminal for auth refusals.
|
|
self.retry_after = retry_after
|
|
self.retryable = retryable
|
|
|
|
|
|
def _provider_error_factory(provider: str) -> Callable[..., AuthError]:
|
|
def factory(message: str, code: Optional[str] = None, *, relogin: bool = False) -> AuthError:
|
|
return AuthError(message, provider=provider, code=code, relogin_required=relogin)
|
|
|
|
return factory
|
|
|
|
|
|
# Per-provider AuthError constructors: ``_xai_err(message, code, relogin=True)``.
|
|
_nous_err = _provider_error_factory("nous")
|
|
_xai_err = _provider_error_factory("xai-oauth")
|
|
_codex_err = _provider_error_factory("openai-codex")
|
|
_spotify_err = _provider_error_factory("spotify")
|
|
_qwen_err = _provider_error_factory("qwen-oauth")
|
|
_minimax_err = _provider_error_factory("minimax-oauth")
|
|
_openrouter_err = _provider_error_factory("openrouter")
|
|
|
|
|
|
def _decode_jwt_claims(token: Any) -> Dict[str, Any]:
|
|
if not isinstance(token, str) or token.count(".") != 2:
|
|
return {}
|
|
payload = token.split(".")[1]
|
|
payload += "=" * ((4 - len(payload) % 4) % 4)
|
|
try:
|
|
raw = base64.urlsafe_b64decode(payload.encode("utf-8"))
|
|
claims = json.loads(raw.decode("utf-8"))
|
|
except Exception:
|
|
return {}
|
|
return claims if isinstance(claims, dict) else {}
|