refactor(hermes_cli): group C auth modules — compact WHY-preserving comments and docstrings

This commit is contained in:
Teknium
2026-09-02 21:59:38 -07:00
parent 105c4413d3
commit 16579014cb
6 changed files with 73 additions and 138 deletions
+9 -19
View File
@@ -1,9 +1,7 @@
"""MiniMax OAuth (user-code grant) login, refresh and runtime credentials.
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
helpers are imported lazily inside each function (no import cycle; patches on
``hermes_cli.auth.<helper>`` still intercept).
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
"""
from __future__ import annotations
@@ -23,7 +21,6 @@ from hermes_cli.auth_constants import (
if TYPE_CHECKING: # annotation-only; the runtime import would be a cycle
from hermes_cli.auth import ProviderConfig
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
logger = logging.getLogger("hermes_cli.auth")
_MINIMAX_OAUTH_ERROR_BODY_LIMIT = 16 * 1024
@@ -121,8 +118,7 @@ def _minimax_poll_token(
client: httpx.Client, *, portal_base_url: str, client_id: str,
user_code: str, code_verifier: str, expired_in: int, interval_ms: Optional[int],
) -> Dict[str, Any]:
# OpenClaw treats expired_in as a unix-ms timestamp; if it's small enough to be a duration,
# treat it as seconds.
# expired_in is a unix-ms timestamp upstream (OpenClaw) but small values are TTL seconds.
deadline = _minimax_resolve_token_expiry_unix(expired_in, now=datetime.now(timezone.utc))
interval = max(2.0, (interval_ms or 2000) / 1000.0)
@@ -247,9 +243,8 @@ def _refresh_minimax_oauth_state(state: Dict[str, Any], *, timeout_seconds: floa
data={"grant_type": "refresh_token", "client_id": state["client_id"], "refresh_token": state["refresh_token"]},
headers=_FORM_JSON_HEADERS,
)
# The non-200 branch reads a STREAMED body, so it must run while the client is still open
# (iter_bytes() after close raises StreamClosed). The 200 path was already read by
# _minimax_post_form, so response.json() below is safe outside.
# Non-200 reads a STREAMED body, so it must run inside the client context (iter_bytes()
# after close raises StreamClosed); the 200 body was already read by _minimax_post_form.
if response.status_code != 200:
body = _minimax_response_error_text(response)
body_lower = body.lower()
@@ -298,11 +293,10 @@ def _minimax_fresh_state() -> Dict[str, Any]:
def build_minimax_oauth_token_provider() -> Callable[[], str]:
"""Return a zero-arg callable that yields a fresh MiniMax access token.
"""Zero-arg callable yielding a fresh MiniMax access token.
The Anthropic SDK caches ``api_key`` as a static string at construction time, so a session that
resolves credentials once at startup would keep sending the same bearer until MiniMax returns
401 — typically ~15 minutes in, because MiniMax issues short-lived access tokens.
The Anthropic SDK caches ``api_key`` at construction; MiniMax tokens live ~15 minutes, so a
static bearer would start 401-ing mid-session.
"""
def _provide() -> str:
token = _minimax_fresh_state().get("access_token")
@@ -317,11 +311,7 @@ def resolve_minimax_oauth_runtime_credentials(
*, min_token_ttl_seconds: int = MINIMAX_OAUTH_REFRESH_SKEW_SECONDS,
as_token_provider: bool = False,
) -> Dict[str, Any]:
"""Return {provider, api_key, base_url, source} for minimax-oauth.
The default (string ``api_key``) preserves the historical contract for diagnostic call sites
like ``hermes status`` that just want to know whether a valid token exists right now.
"""
"""Return {provider, api_key, base_url, source}; string ``api_key`` by default (``hermes status`` contract)."""
state = _minimax_fresh_state()
return {
"provider": "minimax-oauth",
+10 -18
View File
@@ -1,9 +1,7 @@
"""Interactive model picker used after OAuth login.
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
helpers are imported lazily inside each function (no import cycle; patches on
``hermes_cli.auth.<helper>`` still intercept).
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
"""
from __future__ import annotations
@@ -13,7 +11,6 @@ import subprocess
from typing import Dict, List, Optional
from hermes_cli.auth_constants import DEFAULT_NOUS_PORTAL_URL
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
logger = logging.getLogger("hermes_cli.auth")
_CUSTOM_LABEL = "Enter custom model name"
@@ -52,10 +49,10 @@ def _confirm_selection_guards(
class _ModelPickerRows:
"""Column-aligned model rows (name + $/Mtok prices + Nous sale chrome) for the model picker.
"""Column-aligned picker rows (name + $/Mtok + Nous sale chrome).
Sale chrome (★ / -N% / was) is drawn as curses/ANSI segments (yellow % / dim "was"), not baked
into one plain string — curses addnstr would otherwise render escape bytes literally.
Sale chrome is emitted as styled segments, not ANSI baked into one string — curses addnstr
would render escape bytes literally.
"""
def __init__(
@@ -164,15 +161,13 @@ def _prompt_model_selection(
"""
from hermes_cli.cli_output import line_input
_unavailable = unavailable_models or []
# Sale chrome (★ / -N% / was) is Nous Portal-only — never for OpenRouter or other providers
# even if pricing.original is somehow present.
# Sale chrome is Nous Portal-only, even if pricing.original is present for another provider.
sale_chrome = (confirm_provider or "").strip().lower() == "nous"
def _confirmed_selection(mid: str) -> Optional[str]:
if not mid:
return None
# The cost guard only runs when a provider is known (pricing lookups need one); id-keyed
# guards like the data-policy guard always run — even via a custom endpoint or gateway.
# Cost guard needs a known provider; id-keyed guards (data policy) always run.
ok = _confirm_selection_guards(
mid, provider=confirm_provider, base_url=confirm_base_url, api_key=confirm_api_key,
include_kinds=None if confirm_provider else ["data_policy"],
@@ -209,17 +204,14 @@ def _prompt_model_selection(
if not unavailable_footer and _unavailable:
unavailable_footer = f"Upgrade at {_upgrade_url} for paid models"
# The pricing column header (and any unavailable-models block) is shown as a multi-line
# description above the list so it survives the curses screen clear. menu_title already
# embeds the aligned price header; keep only the header/legend portion.
# Header/legend + unavailable block go in the description so they survive the curses clear.
desc_lines: list[str] = menu_title.split("\n", 1)[1].splitlines() if rows.has_pricing else []
if _unavailable:
desc_lines.extend(f" {rows.label(mid)}" for mid in _unavailable)
desc_lines.append(f" ── {unavailable_footer} ──")
# Search haystacks keep pricing labels visible while adding aliases for brand-less wire
# ids (e.g. Kimi Coding `k3` ↔ query "kimi"). model_search_text always starts with the
# wire id; only append when aliases add tokens beyond the bare id already in the label.
# Search haystack = label + aliases for brand-less wire ids (Kimi `k3` ↔ "kimi"); skip when
# model_search_text adds nothing beyond the bare id.
from hermes_cli.model_search import model_search_text
model_search_labels = []
for mid in ordered:
+4 -14
View File
@@ -1,9 +1,7 @@
"""Qwen OAuth (qwen-cli token file) runtime credentials and status.
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
helpers are imported lazily inside each function (no import cycle; patches on
``hermes_cli.auth.<helper>`` still intercept).
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
"""
from __future__ import annotations
@@ -19,7 +17,6 @@ from hermes_cli.auth_constants import (
QWEN_OAUTH_TOKEN_URL, _FORM_JSON_HEADERS, _qwen_err, httpx,
)
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
logger = logging.getLogger("hermes_cli.auth")
_RERUN = "Re-run 'qwen auth qwen-oauth'."
@@ -105,12 +102,7 @@ def _refresh_qwen_cli_tokens(tokens: Dict[str, Any], timeout_seconds: float = 20
def _mark_qwen_oauth_active(creds: Dict[str, Any]) -> None:
"""Set active_provider to qwen-oauth in auth.json.
Qwen tokens live in the Qwen CLI credential file, so this writes only a minimal provider-state
entry (base_url for display) and sets active_provider so ``get_active_provider()`` and the
setup wizard's credential check detect the provider.
"""
"""Set active_provider to qwen-oauth with a minimal state entry (tokens stay in the Qwen CLI file)."""
from hermes_cli.auth import _auth_store_lock, _load_auth_store, _save_auth_store, _save_provider_state
with _auth_store_lock():
auth_store = _load_auth_store()
@@ -146,9 +138,7 @@ def get_qwen_auth_status() -> Dict[str, Any]:
from hermes_cli.auth import _qwen_cli_auth_path, resolve_qwen_runtime_credentials
auth_path = _qwen_cli_auth_path()
try:
# Validate the runtime credentials, including refresh when the cached CLI token is expired;
# otherwise stale tokens show up as "logged in" and `hermes model` walks users into a
# broken Qwen setup flow.
# Refresh-validate: otherwise stale CLI tokens read as "logged in" and break `hermes model`.
creds = resolve_qwen_runtime_credentials(refresh_if_expiring=True)
return {
"logged_in": True, "auth_file": str(auth_path), "source": creds.get("source"),
+4 -9
View File
@@ -1,9 +1,7 @@
"""Spotify OAuth (loopback PKCE) login, refresh and runtime credentials.
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
helpers are imported lazily inside each function (no import cycle; patches on
``hermes_cli.auth.<helper>`` still intercept).
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
"""
from __future__ import annotations
@@ -26,7 +24,6 @@ from hermes_cli.auth_constants import (
_spotify_err, httpx,
)
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
logger = logging.getLogger("hermes_cli.auth")
_CALLBACK_HTML = "<html><body><h1>Spotify authorization {}.</h1>You can close this tab.</body></html>"
@@ -366,8 +363,7 @@ def _spotify_interactive_setup(redirect_uri_hint: str) -> str:
print(f"\nNo Client ID entered. See {SPOTIFY_DOCS_URL} for the full guide.")
raise SystemExit("Spotify setup cancelled: empty Client ID.")
# Persist so subsequent `hermes auth spotify` runs skip the wizard. Only persist a non-default
# redirect URI, to avoid pinning users to a value the default might later change to.
# Persist so later runs skip the wizard; only pin a NON-default redirect URI.
save_env_value("HERMES_SPOTIFY_CLIENT_ID", raw)
if redirect_uri_hint and redirect_uri_hint != DEFAULT_SPOTIFY_REDIRECT_URI:
save_env_value("HERMES_SPOTIFY_REDIRECT_URI", redirect_uri_hint)
@@ -380,8 +376,7 @@ def login_spotify_command(args) -> None:
from hermes_cli.auth import _auth_store_lock, _can_open_graphical_browser, _is_remote_session, _load_auth_store, _print_loopback_ssh_hint, _save_auth_store, _store_provider_state, get_provider_auth_state
existing_state = get_provider_auth_state("spotify") or {}
# Interactive wizard: if no client_id is configured anywhere, walk the user through creating
# the Spotify developer app instead of crashing with "HERMES_SPOTIFY_CLIENT_ID is required".
# No client_id anywhere -> wizard instead of "HERMES_SPOTIFY_CLIENT_ID is required".
try:
client_id = _spotify_client_id(getattr(args, "client_id", None), existing_state)
except AuthError as exc:
+35 -57
View File
@@ -1,9 +1,7 @@
"""xAI Grok OAuth: token store, discovery, refresh, device-code login.
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
helpers are imported lazily inside each function (no import cycle; patches on
``hermes_cli.auth.<helper>`` still intercept).
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
"""
from __future__ import annotations
@@ -26,7 +24,6 @@ from utils import env_float
if TYPE_CHECKING: # annotation-only; the runtime import would be a cycle
from hermes_cli.auth import ProviderConfig
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
logger = logging.getLogger("hermes_cli.auth")
_RELOGIN = "Re-authenticate with `hermes model`."
@@ -101,20 +98,18 @@ def _read_xai_oauth_tokens(*, _lock: bool = True) -> Dict[str, Any]:
def _write_through_xai_oauth_to_global_root(state: Dict[str, Any]) -> None:
"""Persist a rotated xAI OAuth ``state`` into the global-root auth.json (best effort).
"""Best-effort persist of a rotated xAI grant into the global-root auth.json.
xAI rotates the refresh_token on every refresh, so when a profile session refreshes a grant it
resolved from the root fallback, the rotated chain must land back in root. Only updates
``providers.xai-oauth`` in root; never touches the profile store (the caller saved that).
Swallows all errors — a failed write-through degrades to root-stale, never breaks the profile save.
xAI rotates refresh_token on every refresh, so a profile that refreshed a root-resolved grant
must write the chain back to root. Touches only root ``providers.xai-oauth``; swallows all
errors (root-stale is better than breaking the profile's own save).
"""
from hermes_cli.auth import _global_auth_file_path, _persist_provider_state_to_store
global_path = _global_auth_file_path()
if global_path is None: # classic mode (profile == root); the profile save already hit root
return
# Seat belt: under pytest, refuse to write the real user's ~/.hermes/auth.json even when
# HERMES_HOME points at a profile path (mirrors the read-side guard in _load_global_auth_store).
# Uses the unmodified HOME env, not Path.home() which fixtures may monkeypatch.
# Seat belt: under pytest never write the real ~/.hermes/auth.json (mirrors the read-side guard
# in _load_global_auth_store). Uses raw HOME, not Path.home(), which fixtures may monkeypatch.
real_home_env = os.environ.get("HOME", "") if os.environ.get("PYTEST_CURRENT_TEST") else ""
if real_home_env:
real_root = Path(real_home_env) / ".hermes" / "auth.json"
@@ -134,21 +129,19 @@ def _save_xai_oauth_tokens(
last_refresh: Optional[str] = None, auth_mode: str = "oauth_device_code",
set_active: bool = True,
) -> None:
"""Persist xAI OAuth tokens into the auth store.
"""Persist xAI OAuth tokens; *set_active* also promotes ``xai-oauth`` to ``active_provider``.
*set_active* (default True) also promotes ``xai-oauth`` to ``active_provider`` — right for an
intentional login. Pass False for side-tool credential bootstrap (TTS/setup, tools config,
dashboard token save, token refresh) so inference routing is unchanged.
Pass ``set_active=False`` for side-tool bootstrap (TTS/setup, tools config, dashboard, refresh)
so inference routing is unchanged.
"""
from hermes_cli.auth import _auth_store_lock, _global_auth_file_path, _load_auth_store, _load_provider_state_with_source, _same_path, _save_auth_store, _store_provider_state, _utc_now_z, _write_through_xai_oauth_to_global_root
if last_refresh is None:
last_refresh = _utc_now_z()
with _auth_store_lock():
auth_store = _load_auth_store()
# A profile without its own xai-oauth block reads the root grant through
# _load_provider_state's fallback. Refreshing that (rotating) grant must write the rotated
# chain back to root, or root keeps a revoked refresh token. Decide by where the grant was
# resolved FROM, not by key presence: _store_provider_state below would create the key.
# A profile lacking its own xai-oauth block reads root's grant via fallback; refreshing it
# must write the rotated chain back to root or root keeps a revoked refresh token. Decide by
# where the grant was resolved FROM (key presence lies: _store_provider_state creates it).
state, source_path = _load_provider_state_with_source(auth_store, "xai-oauth")
state = state if state is not None else {}
state.update(tokens=tokens, last_refresh=last_refresh, auth_mode=auth_mode)
@@ -158,8 +151,7 @@ def _save_xai_oauth_tokens(
state["redirect_uri"] = redirect_uri
global_root = _global_auth_file_path()
if source_path is not None and global_root is not None and _same_path(source_path, global_root):
# Resolved from root — write back to root only. Storing on the profile auth_store would
# create a shadowing providers.xai-oauth key that disables write-through next refresh.
# Root-only write-back: a profile copy would shadow root and disable write-through.
_write_through_xai_oauth_to_global_root(state)
else:
_store_provider_state(auth_store, "xai-oauth", state, set_active=set_active)
@@ -187,12 +179,10 @@ def _xai_access_token_is_expiring(access_token: str, skew_seconds: int = 0) -> b
def _xai_proactive_refresh_skew_seconds(access_token: str) -> int:
"""How far before JWT ``exp`` to proactively refresh xAI OAuth tokens.
"""Proactive-refresh lead time before JWT ``exp``.
SuperGrok sessions ship multi-hour tokens where the hour-long skew makes sense, but device-code
logins often return ~15-minute JWTs; the full skew would force a refresh on every credential
resolution, burning single-use refresh tokens and racing concurrent callers into
``invalid_grant`` quarantine.
Device-code logins often return ~15-minute JWTs; the full hour-long skew would refresh on every
resolution, burning single-use refresh tokens and racing callers into ``invalid_grant``.
"""
max_skew = XAI_ACCESS_TOKEN_REFRESH_SKEW_SECONDS
exp = _xai_jwt_exp(access_token)
@@ -221,11 +211,10 @@ def _xai_url_problem(url: str) -> tuple[Optional[str], str]:
def _xai_validate_oauth_endpoint(url: str, *, field: str) -> str:
"""Refuse any OIDC discovery endpoint that isn't HTTPS on the xAI origin.
"""Refuse a discovery endpoint that isn't HTTPS on the xAI origin.
The discovery result is cached in auth.json, so a single MITM at login could plant a malicious
``token_endpoint`` that receives the refresh_token forever. Pinning scheme + host (RFC 8414 §2)
removes that persistence.
Discovery is cached in auth.json, so one MITM at login could plant a ``token_endpoint`` that
receives the refresh_token forever; pinning scheme + host (RFC 8414 §2) removes that.
"""
problem, host = _xai_url_problem(url)
if problem is None:
@@ -244,11 +233,9 @@ def _xai_validate_oauth_endpoint(url: str, *, field: str) -> str:
def _xai_validate_inference_base_url(value: str, *, fallback: str) -> str:
"""Refuse a non-xAI base_url for the OAuth-authenticated inference path.
"""Pin the OAuth inference base_url to ``*.x.ai``; warn and use *fallback* on rejection.
Pins the inference origin to ``api.x.ai`` (or any ``*.x.ai`` subdomain). On rejection, fall
back to the default and log a warning rather than raise — a bad env var should not deadlock
authentication, but it must never leak the bearer. Empty input returns ``fallback``.
Warn-not-raise: a bad env var must not deadlock auth, but the bearer must never leak elsewhere.
"""
candidate = (value or "").strip().rstrip("/")
if not candidate:
@@ -321,9 +308,8 @@ def refresh_xai_oauth_pure(
f"xAI OAuth is missing refresh_token. {_RELOGIN}", "xai_auth_missing_refresh_token", relogin=True,
)
endpoint = token_endpoint.strip() or _xai_oauth_discovery(timeout_seconds)["token_endpoint"]
# Re-validate cached endpoints on the refresh hot path: an auth.json written by an older Hermes
# (or hand-edited) may carry a non-xAI token_endpoint that would receive every future
# refresh_token in plaintext if trusted blindly.
# Re-validate cached endpoints: an old/hand-edited auth.json may carry a non-xAI token_endpoint
# that would otherwise receive every future refresh_token.
_xai_validate_oauth_endpoint(endpoint, field="token_endpoint")
timeout = httpx.Timeout(max(5.0, float(timeout_seconds)))
with httpx.Client(timeout=timeout, headers={"Accept": "application/json"}) as client:
@@ -334,9 +320,8 @@ def refresh_xai_oauth_pure(
if response.status_code != 200:
detail = response.text.strip()
suffix = f" Response: {detail}" if detail else ""
# 403 from xAI's token endpoint is almost always a tier / entitlement gate (the grant exists
# but the account isn't allowlisted for API access). Re-running `hermes model` won't fix
# that — use a separate code so format_auth_error skips the re-authenticate hint.
# 403 is almost always a tier/entitlement gate; re-login won't fix it, so use a separate
# code and format_auth_error skips the re-authenticate hint.
if response.status_code == 403:
raise _xai_err(
"xAI token refresh failed with HTTP 403." + suffix
@@ -365,8 +350,7 @@ def refresh_xai_oauth_pure(
def _refresh_xai_oauth_tokens(
tokens: Dict[str, Any], *, token_endpoint: str, redirect_uri: str = "", timeout_seconds: float
) -> Dict[str, Any]:
# Re-persist whatever auth_mode is already stored (legacy pre-device-code logins may still
# carry ``oauth_pkce``): the refresh hot path must not relabel how the grant was obtained.
# Keep the stored auth_mode (legacy logins may carry ``oauth_pkce``): refresh must not relabel it.
from hermes_cli.auth import _load_auth_store, _load_provider_state, refresh_xai_oauth_pure
try:
state = _load_provider_state(_load_auth_store(), "xai-oauth") or {}
@@ -386,8 +370,7 @@ def _refresh_xai_oauth_tokens(
updated_tokens["expires_in"] = refreshed["expires_in"]
if refreshed.get("token_type"):
updated_tokens["token_type"] = refreshed["token_type"]
# set_active=False: refresh must not flip active_provider — TTS/side tools can refresh xAI
# tokens while chat still routes through another provider.
# set_active=False: side tools (TTS) refresh xAI tokens while chat routes elsewhere.
_save_xai_oauth_tokens(
updated_tokens, discovery={"token_endpoint": token_endpoint}, redirect_uri=redirect_uri,
last_refresh=refreshed["last_refresh"], auth_mode=auth_mode, set_active=False,
@@ -396,11 +379,9 @@ def _refresh_xai_oauth_tokens(
def _quarantine_xai_oauth_tokens(exc: AuthError) -> None:
"""Clear dead xAI tokens from auth.json after a terminal refresh failure.
"""Clear dead xAI tokens after a terminal (400/401/403) refresh failure so later sessions fail fast.
Terminal = HTTP 400/401/403 (invalid_grant, token revoked). Subsequent sessions then fail fast
without a network retry. Best-effort: persistence failures are logged and swallowed (caller
re-raises the original error regardless).
Best-effort: persistence failures are logged and swallowed; the caller re-raises regardless.
"""
from hermes_cli.auth import _last_auth_error_marker, _load_auth_store, _load_provider_state, _save_auth_store, _store_provider_state
try:
@@ -471,8 +452,7 @@ def resolve_xai_oauth_runtime_credentials(
"api_key": _clean(tokens.get("access_token")),
"source": "hermes-auth-store",
"last_refresh": data.get("last_refresh"),
# Display/telemetry only. Device-code is the only supported xAI OAuth flow, so report it
# unconditionally — auth.json may still carry a legacy ``oauth_pkce`` label.
# Display only; auth.json may still carry a legacy ``oauth_pkce`` label.
"auth_mode": "oauth_device_code",
}
@@ -506,11 +486,9 @@ def _login_xai_oauth(args, pconfig: ProviderConfig, *, force_new_login: bool = F
redirect_uri=creds.get("redirect_uri", ""), last_refresh=creds.get("last_refresh"),
auth_mode="oauth_device_code",
)
# An explicit interactive re-login means the user wants the xAI credential re-enabled.
# ``hermes auth remove xai-oauth`` leaves a ``device_code`` suppression marker that otherwise
# stops the singleton seed from re-creating the pool entry. Kept OUT of _save_xai_oauth_tokens
# on purpose — that helper is shared with the refresh hot path, which must never mutate
# suppression state.
# Explicit re-login re-enables the credential: clear the ``device_code`` suppression marker left
# by ``hermes auth remove xai-oauth``. Deliberately NOT inside _save_xai_oauth_tokens — the
# refresh hot path shares that helper and must never mutate suppression state.
unsuppress_credential_source("xai-oauth", "device_code")
config_path = _update_config_for_provider("xai-oauth", creds.get("base_url", DEFAULT_XAI_OAUTH_BASE_URL))
_print_login_success("xai-oauth", config_path, show_auth_state=True)
+11 -21
View File
@@ -1,9 +1,7 @@
"""Kimi Code and Z.AI endpoint auto-detection, LM Studio base-URL normalization.
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
helpers are imported lazily inside each function (no import cycle; patches on
``hermes_cli.auth.<helper>`` still intercept).
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
"""
from __future__ import annotations
@@ -13,13 +11,10 @@ import hashlib
from typing import Dict, Optional
from hermes_cli.auth_constants import httpx
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
logger = logging.getLogger("hermes_cli.auth")
# Kimi Code (kimi.com/code) issues "sk-kimi-" keys that only work on api.kimi.com/coding; legacy
# platform.moonshot.ai keys work on api.moonshot.ai/v1 (the old default). Intentionally NO /v1
# suffix: the /coding endpoint speaks Anthropic Messages and the SDK appends "/v1/messages"
# itself — "/coding/v1" would produce "/coding/v1/v1/messages" (a 404).
# "sk-kimi-" keys only work on api.kimi.com/coding; legacy moonshot keys use the old default.
# NO /v1 suffix: the anthropic SDK appends "/v1/messages" itself ("/coding/v1" would 404).
KIMI_CODE_BASE_URL = "https://api.kimi.com/coding"
@@ -32,10 +27,9 @@ def _resolve_kimi_base_url(api_key: str, default_url: str, env_override: str) ->
return default_url
# Z.AI bills general vs coding plans, and global vs China endpoints, separately; a key that works
# on one may return "Insufficient balance" on another, so we probe at setup time and store the
# working endpoint. Each entry lists candidate models to try in order — newer coding-plan accounts
# may only have access to recent models while older ones still use glm-4.7.
# Z.AI bills general/coding plans and global/China endpoints separately ("Insufficient balance" on
# the wrong one), so probe once and cache. Candidate models are tried in order: newer coding-plan
# accounts may only have recent GLM slugs, older ones still glm-4.7.
_ZAI_CODING_PROBE_MODELS = ["glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"]
ZAI_ENDPOINTS = [
# (id, base_url, probe_models, label)
@@ -69,8 +63,7 @@ def _probe_single_zai_endpoint(api_key: str, endpoint: tuple, timeout: float) ->
def detect_zai_endpoint(api_key: str, timeout: float = 8.0) -> Optional[Dict[str, str]]:
"""Probe z.ai endpoints in parallel; first working one in ZAI_ENDPOINTS priority order, or None."""
from concurrent.futures import ThreadPoolExecutor, as_completed
# No `with` block: a context manager would join ALL probe threads on exit, defeating the early
# return below. shutdown(wait=False) lets surviving probes drain in the background.
# No `with`: it would join ALL probes on exit, defeating the early return below.
pool = ThreadPoolExecutor(max_workers=len(ZAI_ENDPOINTS))
try:
futures = {pool.submit(_probe_single_zai_endpoint, api_key, ep, timeout): ep[0] for ep in ZAI_ENDPOINTS}
@@ -111,8 +104,7 @@ def _resolve_zai_base_url(api_key: str, default_url: str, env_override: str) ->
from hermes_cli.auth import _auth_store_lock, _load_auth_store, _load_provider_state, _save_auth_store, _store_provider_state, detect_zai_endpoint
if env_override:
return env_override
# No API key → don't probe (N×M HTTPS requests with an empty Bearer, all 401). Hit during
# auxiliary-client auto-detection for users with no Z.AI credentials at all — pure latency.
# No key -> don't probe (N×M 401s); auxiliary-client auto-detection hits this for everyone.
if not api_key:
return default_url
@@ -134,15 +126,13 @@ def _resolve_zai_base_url(api_key: str, default_url: str, env_override: str) ->
"model": detected.get("model", ""), "label": detected.get("label", ""),
"key_hash": key_hash,
}
# Persist failure (disk full, permissions, lock timeout) must not break resolution — detection
# already succeeded; worst case the next start re-probes.
# Persist failure must not break resolution; worst case the next start re-probes.
try:
with _auth_store_lock():
auth_store = _load_auth_store() # reload under lock to avoid overwriting concurrent changes
state_under_lock = _load_provider_state(auth_store, "zai") or {}
state_under_lock["detected_endpoint"] = detected_endpoint
# set_active=False: this runs from credential-pool env seeding for ANY user with a Z.AI
# key in env, and caching a probe result must not flip their active provider.
# set_active=False: runs from credential-pool env seeding; must not flip active provider.
_store_provider_state(auth_store, "zai", state_under_lock, set_active=False)
_save_auth_store(auth_store)
except Exception as exc: