From 16579014cb6da54b8ee793cd16549ddf3b396474 Mon Sep 17 00:00:00 2001 From: Teknium <127238744+teknium1@users.noreply.github.com> Date: Wed, 2 Sep 2026 21:59:38 -0700 Subject: [PATCH] =?UTF-8?q?refactor(hermes=5Fcli):=20group=20C=20auth=20mo?= =?UTF-8?q?dules=20=E2=80=94=20compact=20WHY-preserving=20comments=20and?= =?UTF-8?q?=20docstrings?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- hermes_cli/auth_minimax.py | 28 ++++------ hermes_cli/auth_model_picker.py | 28 ++++------ hermes_cli/auth_qwen.py | 18 ++----- hermes_cli/auth_spotify.py | 13 ++--- hermes_cli/auth_xai.py | 92 +++++++++++++-------------------- hermes_cli/auth_zai_kimi.py | 32 ++++-------- 6 files changed, 73 insertions(+), 138 deletions(-) diff --git a/hermes_cli/auth_minimax.py b/hermes_cli/auth_minimax.py index b3aa9e6281..65ac3a2cd8 100644 --- a/hermes_cli/auth_minimax.py +++ b/hermes_cli/auth_minimax.py @@ -1,9 +1,7 @@ """MiniMax OAuth (user-code grant) login, refresh and runtime credentials. -Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so -``hermes_cli.auth.`` keeps resolving (and monkeypatching) as before. Origin-internal -helpers are imported lazily inside each function (no import cycle; patches on -``hermes_cli.auth.`` still intercept). +Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported +lazily per function so ``hermes_cli.auth.`` patches still intercept and no cycle forms. """ from __future__ import annotations @@ -23,7 +21,6 @@ from hermes_cli.auth_constants import ( if TYPE_CHECKING: # annotation-only; the runtime import would be a cycle from hermes_cli.auth import ProviderConfig -# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth"). logger = logging.getLogger("hermes_cli.auth") _MINIMAX_OAUTH_ERROR_BODY_LIMIT = 16 * 1024 @@ -121,8 +118,7 @@ def _minimax_poll_token( client: httpx.Client, *, portal_base_url: str, client_id: str, user_code: str, code_verifier: str, expired_in: int, interval_ms: Optional[int], ) -> Dict[str, Any]: - # OpenClaw treats expired_in as a unix-ms timestamp; if it's small enough to be a duration, - # treat it as seconds. + # expired_in is a unix-ms timestamp upstream (OpenClaw) but small values are TTL seconds. deadline = _minimax_resolve_token_expiry_unix(expired_in, now=datetime.now(timezone.utc)) interval = max(2.0, (interval_ms or 2000) / 1000.0) @@ -247,9 +243,8 @@ def _refresh_minimax_oauth_state(state: Dict[str, Any], *, timeout_seconds: floa data={"grant_type": "refresh_token", "client_id": state["client_id"], "refresh_token": state["refresh_token"]}, headers=_FORM_JSON_HEADERS, ) - # The non-200 branch reads a STREAMED body, so it must run while the client is still open - # (iter_bytes() after close raises StreamClosed). The 200 path was already read by - # _minimax_post_form, so response.json() below is safe outside. + # Non-200 reads a STREAMED body, so it must run inside the client context (iter_bytes() + # after close raises StreamClosed); the 200 body was already read by _minimax_post_form. if response.status_code != 200: body = _minimax_response_error_text(response) body_lower = body.lower() @@ -298,11 +293,10 @@ def _minimax_fresh_state() -> Dict[str, Any]: def build_minimax_oauth_token_provider() -> Callable[[], str]: - """Return a zero-arg callable that yields a fresh MiniMax access token. + """Zero-arg callable yielding a fresh MiniMax access token. - The Anthropic SDK caches ``api_key`` as a static string at construction time, so a session that - resolves credentials once at startup would keep sending the same bearer until MiniMax returns - 401 — typically ~15 minutes in, because MiniMax issues short-lived access tokens. + The Anthropic SDK caches ``api_key`` at construction; MiniMax tokens live ~15 minutes, so a + static bearer would start 401-ing mid-session. """ def _provide() -> str: token = _minimax_fresh_state().get("access_token") @@ -317,11 +311,7 @@ def resolve_minimax_oauth_runtime_credentials( *, min_token_ttl_seconds: int = MINIMAX_OAUTH_REFRESH_SKEW_SECONDS, as_token_provider: bool = False, ) -> Dict[str, Any]: - """Return {provider, api_key, base_url, source} for minimax-oauth. - - The default (string ``api_key``) preserves the historical contract for diagnostic call sites - like ``hermes status`` that just want to know whether a valid token exists right now. - """ + """Return {provider, api_key, base_url, source}; string ``api_key`` by default (``hermes status`` contract).""" state = _minimax_fresh_state() return { "provider": "minimax-oauth", diff --git a/hermes_cli/auth_model_picker.py b/hermes_cli/auth_model_picker.py index cda7693f0c..181d7c5497 100644 --- a/hermes_cli/auth_model_picker.py +++ b/hermes_cli/auth_model_picker.py @@ -1,9 +1,7 @@ """Interactive model picker used after OAuth login. -Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so -``hermes_cli.auth.`` keeps resolving (and monkeypatching) as before. Origin-internal -helpers are imported lazily inside each function (no import cycle; patches on -``hermes_cli.auth.`` still intercept). +Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported +lazily per function so ``hermes_cli.auth.`` patches still intercept and no cycle forms. """ from __future__ import annotations @@ -13,7 +11,6 @@ import subprocess from typing import Dict, List, Optional from hermes_cli.auth_constants import DEFAULT_NOUS_PORTAL_URL -# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth"). logger = logging.getLogger("hermes_cli.auth") _CUSTOM_LABEL = "Enter custom model name" @@ -52,10 +49,10 @@ def _confirm_selection_guards( class _ModelPickerRows: - """Column-aligned model rows (name + $/Mtok prices + Nous sale chrome) for the model picker. + """Column-aligned picker rows (name + $/Mtok + Nous sale chrome). - Sale chrome (★ / -N% / was) is drawn as curses/ANSI segments (yellow % / dim "was"), not baked - into one plain string — curses addnstr would otherwise render escape bytes literally. + Sale chrome is emitted as styled segments, not ANSI baked into one string — curses addnstr + would render escape bytes literally. """ def __init__( @@ -164,15 +161,13 @@ def _prompt_model_selection( """ from hermes_cli.cli_output import line_input _unavailable = unavailable_models or [] - # Sale chrome (★ / -N% / was) is Nous Portal-only — never for OpenRouter or other providers - # even if pricing.original is somehow present. + # Sale chrome is Nous Portal-only, even if pricing.original is present for another provider. sale_chrome = (confirm_provider or "").strip().lower() == "nous" def _confirmed_selection(mid: str) -> Optional[str]: if not mid: return None - # The cost guard only runs when a provider is known (pricing lookups need one); id-keyed - # guards like the data-policy guard always run — even via a custom endpoint or gateway. + # Cost guard needs a known provider; id-keyed guards (data policy) always run. ok = _confirm_selection_guards( mid, provider=confirm_provider, base_url=confirm_base_url, api_key=confirm_api_key, include_kinds=None if confirm_provider else ["data_policy"], @@ -209,17 +204,14 @@ def _prompt_model_selection( if not unavailable_footer and _unavailable: unavailable_footer = f"Upgrade at {_upgrade_url} for paid models" - # The pricing column header (and any unavailable-models block) is shown as a multi-line - # description above the list so it survives the curses screen clear. menu_title already - # embeds the aligned price header; keep only the header/legend portion. + # Header/legend + unavailable block go in the description so they survive the curses clear. desc_lines: list[str] = menu_title.split("\n", 1)[1].splitlines() if rows.has_pricing else [] if _unavailable: desc_lines.extend(f" {rows.label(mid)}" for mid in _unavailable) desc_lines.append(f" ── {unavailable_footer} ──") - # Search haystacks keep pricing labels visible while adding aliases for brand-less wire - # ids (e.g. Kimi Coding `k3` ↔ query "kimi"). model_search_text always starts with the - # wire id; only append when aliases add tokens beyond the bare id already in the label. + # Search haystack = label + aliases for brand-less wire ids (Kimi `k3` ↔ "kimi"); skip when + # model_search_text adds nothing beyond the bare id. from hermes_cli.model_search import model_search_text model_search_labels = [] for mid in ordered: diff --git a/hermes_cli/auth_qwen.py b/hermes_cli/auth_qwen.py index 849dbfd066..860a7e37a3 100644 --- a/hermes_cli/auth_qwen.py +++ b/hermes_cli/auth_qwen.py @@ -1,9 +1,7 @@ """Qwen OAuth (qwen-cli token file) runtime credentials and status. -Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so -``hermes_cli.auth.`` keeps resolving (and monkeypatching) as before. Origin-internal -helpers are imported lazily inside each function (no import cycle; patches on -``hermes_cli.auth.`` still intercept). +Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported +lazily per function so ``hermes_cli.auth.`` patches still intercept and no cycle forms. """ from __future__ import annotations @@ -19,7 +17,6 @@ from hermes_cli.auth_constants import ( QWEN_OAUTH_TOKEN_URL, _FORM_JSON_HEADERS, _qwen_err, httpx, ) -# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth"). logger = logging.getLogger("hermes_cli.auth") _RERUN = "Re-run 'qwen auth qwen-oauth'." @@ -105,12 +102,7 @@ def _refresh_qwen_cli_tokens(tokens: Dict[str, Any], timeout_seconds: float = 20 def _mark_qwen_oauth_active(creds: Dict[str, Any]) -> None: - """Set active_provider to qwen-oauth in auth.json. - - Qwen tokens live in the Qwen CLI credential file, so this writes only a minimal provider-state - entry (base_url for display) and sets active_provider so ``get_active_provider()`` and the - setup wizard's credential check detect the provider. - """ + """Set active_provider to qwen-oauth with a minimal state entry (tokens stay in the Qwen CLI file).""" from hermes_cli.auth import _auth_store_lock, _load_auth_store, _save_auth_store, _save_provider_state with _auth_store_lock(): auth_store = _load_auth_store() @@ -146,9 +138,7 @@ def get_qwen_auth_status() -> Dict[str, Any]: from hermes_cli.auth import _qwen_cli_auth_path, resolve_qwen_runtime_credentials auth_path = _qwen_cli_auth_path() try: - # Validate the runtime credentials, including refresh when the cached CLI token is expired; - # otherwise stale tokens show up as "logged in" and `hermes model` walks users into a - # broken Qwen setup flow. + # Refresh-validate: otherwise stale CLI tokens read as "logged in" and break `hermes model`. creds = resolve_qwen_runtime_credentials(refresh_if_expiring=True) return { "logged_in": True, "auth_file": str(auth_path), "source": creds.get("source"), diff --git a/hermes_cli/auth_spotify.py b/hermes_cli/auth_spotify.py index 5131bc7851..7730e520ff 100644 --- a/hermes_cli/auth_spotify.py +++ b/hermes_cli/auth_spotify.py @@ -1,9 +1,7 @@ """Spotify OAuth (loopback PKCE) login, refresh and runtime credentials. -Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so -``hermes_cli.auth.`` keeps resolving (and monkeypatching) as before. Origin-internal -helpers are imported lazily inside each function (no import cycle; patches on -``hermes_cli.auth.`` still intercept). +Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported +lazily per function so ``hermes_cli.auth.`` patches still intercept and no cycle forms. """ from __future__ import annotations @@ -26,7 +24,6 @@ from hermes_cli.auth_constants import ( _spotify_err, httpx, ) -# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth"). logger = logging.getLogger("hermes_cli.auth") _CALLBACK_HTML = "

Spotify authorization {}.

You can close this tab." @@ -366,8 +363,7 @@ def _spotify_interactive_setup(redirect_uri_hint: str) -> str: print(f"\nNo Client ID entered. See {SPOTIFY_DOCS_URL} for the full guide.") raise SystemExit("Spotify setup cancelled: empty Client ID.") - # Persist so subsequent `hermes auth spotify` runs skip the wizard. Only persist a non-default - # redirect URI, to avoid pinning users to a value the default might later change to. + # Persist so later runs skip the wizard; only pin a NON-default redirect URI. save_env_value("HERMES_SPOTIFY_CLIENT_ID", raw) if redirect_uri_hint and redirect_uri_hint != DEFAULT_SPOTIFY_REDIRECT_URI: save_env_value("HERMES_SPOTIFY_REDIRECT_URI", redirect_uri_hint) @@ -380,8 +376,7 @@ def login_spotify_command(args) -> None: from hermes_cli.auth import _auth_store_lock, _can_open_graphical_browser, _is_remote_session, _load_auth_store, _print_loopback_ssh_hint, _save_auth_store, _store_provider_state, get_provider_auth_state existing_state = get_provider_auth_state("spotify") or {} - # Interactive wizard: if no client_id is configured anywhere, walk the user through creating - # the Spotify developer app instead of crashing with "HERMES_SPOTIFY_CLIENT_ID is required". + # No client_id anywhere -> wizard instead of "HERMES_SPOTIFY_CLIENT_ID is required". try: client_id = _spotify_client_id(getattr(args, "client_id", None), existing_state) except AuthError as exc: diff --git a/hermes_cli/auth_xai.py b/hermes_cli/auth_xai.py index 03c248b29f..b983fe68cc 100644 --- a/hermes_cli/auth_xai.py +++ b/hermes_cli/auth_xai.py @@ -1,9 +1,7 @@ """xAI Grok OAuth: token store, discovery, refresh, device-code login. -Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so -``hermes_cli.auth.`` keeps resolving (and monkeypatching) as before. Origin-internal -helpers are imported lazily inside each function (no import cycle; patches on -``hermes_cli.auth.`` still intercept). +Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported +lazily per function so ``hermes_cli.auth.`` patches still intercept and no cycle forms. """ from __future__ import annotations @@ -26,7 +24,6 @@ from utils import env_float if TYPE_CHECKING: # annotation-only; the runtime import would be a cycle from hermes_cli.auth import ProviderConfig -# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth"). logger = logging.getLogger("hermes_cli.auth") _RELOGIN = "Re-authenticate with `hermes model`." @@ -101,20 +98,18 @@ def _read_xai_oauth_tokens(*, _lock: bool = True) -> Dict[str, Any]: def _write_through_xai_oauth_to_global_root(state: Dict[str, Any]) -> None: - """Persist a rotated xAI OAuth ``state`` into the global-root auth.json (best effort). + """Best-effort persist of a rotated xAI grant into the global-root auth.json. - xAI rotates the refresh_token on every refresh, so when a profile session refreshes a grant it - resolved from the root fallback, the rotated chain must land back in root. Only updates - ``providers.xai-oauth`` in root; never touches the profile store (the caller saved that). - Swallows all errors — a failed write-through degrades to root-stale, never breaks the profile save. + xAI rotates refresh_token on every refresh, so a profile that refreshed a root-resolved grant + must write the chain back to root. Touches only root ``providers.xai-oauth``; swallows all + errors (root-stale is better than breaking the profile's own save). """ from hermes_cli.auth import _global_auth_file_path, _persist_provider_state_to_store global_path = _global_auth_file_path() if global_path is None: # classic mode (profile == root); the profile save already hit root return - # Seat belt: under pytest, refuse to write the real user's ~/.hermes/auth.json even when - # HERMES_HOME points at a profile path (mirrors the read-side guard in _load_global_auth_store). - # Uses the unmodified HOME env, not Path.home() which fixtures may monkeypatch. + # Seat belt: under pytest never write the real ~/.hermes/auth.json (mirrors the read-side guard + # in _load_global_auth_store). Uses raw HOME, not Path.home(), which fixtures may monkeypatch. real_home_env = os.environ.get("HOME", "") if os.environ.get("PYTEST_CURRENT_TEST") else "" if real_home_env: real_root = Path(real_home_env) / ".hermes" / "auth.json" @@ -134,21 +129,19 @@ def _save_xai_oauth_tokens( last_refresh: Optional[str] = None, auth_mode: str = "oauth_device_code", set_active: bool = True, ) -> None: - """Persist xAI OAuth tokens into the auth store. + """Persist xAI OAuth tokens; *set_active* also promotes ``xai-oauth`` to ``active_provider``. - *set_active* (default True) also promotes ``xai-oauth`` to ``active_provider`` — right for an - intentional login. Pass False for side-tool credential bootstrap (TTS/setup, tools config, - dashboard token save, token refresh) so inference routing is unchanged. + Pass ``set_active=False`` for side-tool bootstrap (TTS/setup, tools config, dashboard, refresh) + so inference routing is unchanged. """ from hermes_cli.auth import _auth_store_lock, _global_auth_file_path, _load_auth_store, _load_provider_state_with_source, _same_path, _save_auth_store, _store_provider_state, _utc_now_z, _write_through_xai_oauth_to_global_root if last_refresh is None: last_refresh = _utc_now_z() with _auth_store_lock(): auth_store = _load_auth_store() - # A profile without its own xai-oauth block reads the root grant through - # _load_provider_state's fallback. Refreshing that (rotating) grant must write the rotated - # chain back to root, or root keeps a revoked refresh token. Decide by where the grant was - # resolved FROM, not by key presence: _store_provider_state below would create the key. + # A profile lacking its own xai-oauth block reads root's grant via fallback; refreshing it + # must write the rotated chain back to root or root keeps a revoked refresh token. Decide by + # where the grant was resolved FROM (key presence lies: _store_provider_state creates it). state, source_path = _load_provider_state_with_source(auth_store, "xai-oauth") state = state if state is not None else {} state.update(tokens=tokens, last_refresh=last_refresh, auth_mode=auth_mode) @@ -158,8 +151,7 @@ def _save_xai_oauth_tokens( state["redirect_uri"] = redirect_uri global_root = _global_auth_file_path() if source_path is not None and global_root is not None and _same_path(source_path, global_root): - # Resolved from root — write back to root only. Storing on the profile auth_store would - # create a shadowing providers.xai-oauth key that disables write-through next refresh. + # Root-only write-back: a profile copy would shadow root and disable write-through. _write_through_xai_oauth_to_global_root(state) else: _store_provider_state(auth_store, "xai-oauth", state, set_active=set_active) @@ -187,12 +179,10 @@ def _xai_access_token_is_expiring(access_token: str, skew_seconds: int = 0) -> b def _xai_proactive_refresh_skew_seconds(access_token: str) -> int: - """How far before JWT ``exp`` to proactively refresh xAI OAuth tokens. + """Proactive-refresh lead time before JWT ``exp``. - SuperGrok sessions ship multi-hour tokens where the hour-long skew makes sense, but device-code - logins often return ~15-minute JWTs; the full skew would force a refresh on every credential - resolution, burning single-use refresh tokens and racing concurrent callers into - ``invalid_grant`` quarantine. + Device-code logins often return ~15-minute JWTs; the full hour-long skew would refresh on every + resolution, burning single-use refresh tokens and racing callers into ``invalid_grant``. """ max_skew = XAI_ACCESS_TOKEN_REFRESH_SKEW_SECONDS exp = _xai_jwt_exp(access_token) @@ -221,11 +211,10 @@ def _xai_url_problem(url: str) -> tuple[Optional[str], str]: def _xai_validate_oauth_endpoint(url: str, *, field: str) -> str: - """Refuse any OIDC discovery endpoint that isn't HTTPS on the xAI origin. + """Refuse a discovery endpoint that isn't HTTPS on the xAI origin. - The discovery result is cached in auth.json, so a single MITM at login could plant a malicious - ``token_endpoint`` that receives the refresh_token forever. Pinning scheme + host (RFC 8414 §2) - removes that persistence. + Discovery is cached in auth.json, so one MITM at login could plant a ``token_endpoint`` that + receives the refresh_token forever; pinning scheme + host (RFC 8414 §2) removes that. """ problem, host = _xai_url_problem(url) if problem is None: @@ -244,11 +233,9 @@ def _xai_validate_oauth_endpoint(url: str, *, field: str) -> str: def _xai_validate_inference_base_url(value: str, *, fallback: str) -> str: - """Refuse a non-xAI base_url for the OAuth-authenticated inference path. + """Pin the OAuth inference base_url to ``*.x.ai``; warn and use *fallback* on rejection. - Pins the inference origin to ``api.x.ai`` (or any ``*.x.ai`` subdomain). On rejection, fall - back to the default and log a warning rather than raise — a bad env var should not deadlock - authentication, but it must never leak the bearer. Empty input returns ``fallback``. + Warn-not-raise: a bad env var must not deadlock auth, but the bearer must never leak elsewhere. """ candidate = (value or "").strip().rstrip("/") if not candidate: @@ -321,9 +308,8 @@ def refresh_xai_oauth_pure( f"xAI OAuth is missing refresh_token. {_RELOGIN}", "xai_auth_missing_refresh_token", relogin=True, ) endpoint = token_endpoint.strip() or _xai_oauth_discovery(timeout_seconds)["token_endpoint"] - # Re-validate cached endpoints on the refresh hot path: an auth.json written by an older Hermes - # (or hand-edited) may carry a non-xAI token_endpoint that would receive every future - # refresh_token in plaintext if trusted blindly. + # Re-validate cached endpoints: an old/hand-edited auth.json may carry a non-xAI token_endpoint + # that would otherwise receive every future refresh_token. _xai_validate_oauth_endpoint(endpoint, field="token_endpoint") timeout = httpx.Timeout(max(5.0, float(timeout_seconds))) with httpx.Client(timeout=timeout, headers={"Accept": "application/json"}) as client: @@ -334,9 +320,8 @@ def refresh_xai_oauth_pure( if response.status_code != 200: detail = response.text.strip() suffix = f" Response: {detail}" if detail else "" - # 403 from xAI's token endpoint is almost always a tier / entitlement gate (the grant exists - # but the account isn't allowlisted for API access). Re-running `hermes model` won't fix - # that — use a separate code so format_auth_error skips the re-authenticate hint. + # 403 is almost always a tier/entitlement gate; re-login won't fix it, so use a separate + # code and format_auth_error skips the re-authenticate hint. if response.status_code == 403: raise _xai_err( "xAI token refresh failed with HTTP 403." + suffix @@ -365,8 +350,7 @@ def refresh_xai_oauth_pure( def _refresh_xai_oauth_tokens( tokens: Dict[str, Any], *, token_endpoint: str, redirect_uri: str = "", timeout_seconds: float ) -> Dict[str, Any]: - # Re-persist whatever auth_mode is already stored (legacy pre-device-code logins may still - # carry ``oauth_pkce``): the refresh hot path must not relabel how the grant was obtained. + # Keep the stored auth_mode (legacy logins may carry ``oauth_pkce``): refresh must not relabel it. from hermes_cli.auth import _load_auth_store, _load_provider_state, refresh_xai_oauth_pure try: state = _load_provider_state(_load_auth_store(), "xai-oauth") or {} @@ -386,8 +370,7 @@ def _refresh_xai_oauth_tokens( updated_tokens["expires_in"] = refreshed["expires_in"] if refreshed.get("token_type"): updated_tokens["token_type"] = refreshed["token_type"] - # set_active=False: refresh must not flip active_provider — TTS/side tools can refresh xAI - # tokens while chat still routes through another provider. + # set_active=False: side tools (TTS) refresh xAI tokens while chat routes elsewhere. _save_xai_oauth_tokens( updated_tokens, discovery={"token_endpoint": token_endpoint}, redirect_uri=redirect_uri, last_refresh=refreshed["last_refresh"], auth_mode=auth_mode, set_active=False, @@ -396,11 +379,9 @@ def _refresh_xai_oauth_tokens( def _quarantine_xai_oauth_tokens(exc: AuthError) -> None: - """Clear dead xAI tokens from auth.json after a terminal refresh failure. + """Clear dead xAI tokens after a terminal (400/401/403) refresh failure so later sessions fail fast. - Terminal = HTTP 400/401/403 (invalid_grant, token revoked). Subsequent sessions then fail fast - without a network retry. Best-effort: persistence failures are logged and swallowed (caller - re-raises the original error regardless). + Best-effort: persistence failures are logged and swallowed; the caller re-raises regardless. """ from hermes_cli.auth import _last_auth_error_marker, _load_auth_store, _load_provider_state, _save_auth_store, _store_provider_state try: @@ -471,8 +452,7 @@ def resolve_xai_oauth_runtime_credentials( "api_key": _clean(tokens.get("access_token")), "source": "hermes-auth-store", "last_refresh": data.get("last_refresh"), - # Display/telemetry only. Device-code is the only supported xAI OAuth flow, so report it - # unconditionally — auth.json may still carry a legacy ``oauth_pkce`` label. + # Display only; auth.json may still carry a legacy ``oauth_pkce`` label. "auth_mode": "oauth_device_code", } @@ -506,11 +486,9 @@ def _login_xai_oauth(args, pconfig: ProviderConfig, *, force_new_login: bool = F redirect_uri=creds.get("redirect_uri", ""), last_refresh=creds.get("last_refresh"), auth_mode="oauth_device_code", ) - # An explicit interactive re-login means the user wants the xAI credential re-enabled. - # ``hermes auth remove xai-oauth`` leaves a ``device_code`` suppression marker that otherwise - # stops the singleton seed from re-creating the pool entry. Kept OUT of _save_xai_oauth_tokens - # on purpose — that helper is shared with the refresh hot path, which must never mutate - # suppression state. + # Explicit re-login re-enables the credential: clear the ``device_code`` suppression marker left + # by ``hermes auth remove xai-oauth``. Deliberately NOT inside _save_xai_oauth_tokens — the + # refresh hot path shares that helper and must never mutate suppression state. unsuppress_credential_source("xai-oauth", "device_code") config_path = _update_config_for_provider("xai-oauth", creds.get("base_url", DEFAULT_XAI_OAUTH_BASE_URL)) _print_login_success("xai-oauth", config_path, show_auth_state=True) diff --git a/hermes_cli/auth_zai_kimi.py b/hermes_cli/auth_zai_kimi.py index c343cc696d..1985bb44ab 100644 --- a/hermes_cli/auth_zai_kimi.py +++ b/hermes_cli/auth_zai_kimi.py @@ -1,9 +1,7 @@ """Kimi Code and Z.AI endpoint auto-detection, LM Studio base-URL normalization. -Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so -``hermes_cli.auth.`` keeps resolving (and monkeypatching) as before. Origin-internal -helpers are imported lazily inside each function (no import cycle; patches on -``hermes_cli.auth.`` still intercept). +Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported +lazily per function so ``hermes_cli.auth.`` patches still intercept and no cycle forms. """ from __future__ import annotations @@ -13,13 +11,10 @@ import hashlib from typing import Dict, Optional from hermes_cli.auth_constants import httpx -# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth"). logger = logging.getLogger("hermes_cli.auth") -# Kimi Code (kimi.com/code) issues "sk-kimi-" keys that only work on api.kimi.com/coding; legacy -# platform.moonshot.ai keys work on api.moonshot.ai/v1 (the old default). Intentionally NO /v1 -# suffix: the /coding endpoint speaks Anthropic Messages and the SDK appends "/v1/messages" -# itself — "/coding/v1" would produce "/coding/v1/v1/messages" (a 404). +# "sk-kimi-" keys only work on api.kimi.com/coding; legacy moonshot keys use the old default. +# NO /v1 suffix: the anthropic SDK appends "/v1/messages" itself ("/coding/v1" would 404). KIMI_CODE_BASE_URL = "https://api.kimi.com/coding" @@ -32,10 +27,9 @@ def _resolve_kimi_base_url(api_key: str, default_url: str, env_override: str) -> return default_url -# Z.AI bills general vs coding plans, and global vs China endpoints, separately; a key that works -# on one may return "Insufficient balance" on another, so we probe at setup time and store the -# working endpoint. Each entry lists candidate models to try in order — newer coding-plan accounts -# may only have access to recent models while older ones still use glm-4.7. +# Z.AI bills general/coding plans and global/China endpoints separately ("Insufficient balance" on +# the wrong one), so probe once and cache. Candidate models are tried in order: newer coding-plan +# accounts may only have recent GLM slugs, older ones still glm-4.7. _ZAI_CODING_PROBE_MODELS = ["glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"] ZAI_ENDPOINTS = [ # (id, base_url, probe_models, label) @@ -69,8 +63,7 @@ def _probe_single_zai_endpoint(api_key: str, endpoint: tuple, timeout: float) -> def detect_zai_endpoint(api_key: str, timeout: float = 8.0) -> Optional[Dict[str, str]]: """Probe z.ai endpoints in parallel; first working one in ZAI_ENDPOINTS priority order, or None.""" from concurrent.futures import ThreadPoolExecutor, as_completed - # No `with` block: a context manager would join ALL probe threads on exit, defeating the early - # return below. shutdown(wait=False) lets surviving probes drain in the background. + # No `with`: it would join ALL probes on exit, defeating the early return below. pool = ThreadPoolExecutor(max_workers=len(ZAI_ENDPOINTS)) try: futures = {pool.submit(_probe_single_zai_endpoint, api_key, ep, timeout): ep[0] for ep in ZAI_ENDPOINTS} @@ -111,8 +104,7 @@ def _resolve_zai_base_url(api_key: str, default_url: str, env_override: str) -> from hermes_cli.auth import _auth_store_lock, _load_auth_store, _load_provider_state, _save_auth_store, _store_provider_state, detect_zai_endpoint if env_override: return env_override - # No API key → don't probe (N×M HTTPS requests with an empty Bearer, all 401). Hit during - # auxiliary-client auto-detection for users with no Z.AI credentials at all — pure latency. + # No key -> don't probe (N×M 401s); auxiliary-client auto-detection hits this for everyone. if not api_key: return default_url @@ -134,15 +126,13 @@ def _resolve_zai_base_url(api_key: str, default_url: str, env_override: str) -> "model": detected.get("model", ""), "label": detected.get("label", ""), "key_hash": key_hash, } - # Persist failure (disk full, permissions, lock timeout) must not break resolution — detection - # already succeeded; worst case the next start re-probes. + # Persist failure must not break resolution; worst case the next start re-probes. try: with _auth_store_lock(): auth_store = _load_auth_store() # reload under lock to avoid overwriting concurrent changes state_under_lock = _load_provider_state(auth_store, "zai") or {} state_under_lock["detected_endpoint"] = detected_endpoint - # set_active=False: this runs from credential-pool env seeding for ANY user with a Z.AI - # key in env, and caching a probe result must not flip their active provider. + # set_active=False: runs from credential-pool env seeding; must not flip active provider. _store_provider_state(auth_store, "zai", state_under_lock, set_active=False) _save_auth_store(auth_store) except Exception as exc: