refactor(hermes_cli): group C auth modules — compact WHY-preserving comments and docstrings
This commit is contained in:
@@ -1,9 +1,7 @@
|
||||
"""MiniMax OAuth (user-code grant) login, refresh and runtime credentials.
|
||||
|
||||
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
|
||||
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
|
||||
helpers are imported lazily inside each function (no import cycle; patches on
|
||||
``hermes_cli.auth.<helper>`` still intercept).
|
||||
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
|
||||
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -23,7 +21,6 @@ from hermes_cli.auth_constants import (
|
||||
|
||||
if TYPE_CHECKING: # annotation-only; the runtime import would be a cycle
|
||||
from hermes_cli.auth import ProviderConfig
|
||||
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
|
||||
logger = logging.getLogger("hermes_cli.auth")
|
||||
|
||||
_MINIMAX_OAUTH_ERROR_BODY_LIMIT = 16 * 1024
|
||||
@@ -121,8 +118,7 @@ def _minimax_poll_token(
|
||||
client: httpx.Client, *, portal_base_url: str, client_id: str,
|
||||
user_code: str, code_verifier: str, expired_in: int, interval_ms: Optional[int],
|
||||
) -> Dict[str, Any]:
|
||||
# OpenClaw treats expired_in as a unix-ms timestamp; if it's small enough to be a duration,
|
||||
# treat it as seconds.
|
||||
# expired_in is a unix-ms timestamp upstream (OpenClaw) but small values are TTL seconds.
|
||||
deadline = _minimax_resolve_token_expiry_unix(expired_in, now=datetime.now(timezone.utc))
|
||||
interval = max(2.0, (interval_ms or 2000) / 1000.0)
|
||||
|
||||
@@ -247,9 +243,8 @@ def _refresh_minimax_oauth_state(state: Dict[str, Any], *, timeout_seconds: floa
|
||||
data={"grant_type": "refresh_token", "client_id": state["client_id"], "refresh_token": state["refresh_token"]},
|
||||
headers=_FORM_JSON_HEADERS,
|
||||
)
|
||||
# The non-200 branch reads a STREAMED body, so it must run while the client is still open
|
||||
# (iter_bytes() after close raises StreamClosed). The 200 path was already read by
|
||||
# _minimax_post_form, so response.json() below is safe outside.
|
||||
# Non-200 reads a STREAMED body, so it must run inside the client context (iter_bytes()
|
||||
# after close raises StreamClosed); the 200 body was already read by _minimax_post_form.
|
||||
if response.status_code != 200:
|
||||
body = _minimax_response_error_text(response)
|
||||
body_lower = body.lower()
|
||||
@@ -298,11 +293,10 @@ def _minimax_fresh_state() -> Dict[str, Any]:
|
||||
|
||||
|
||||
def build_minimax_oauth_token_provider() -> Callable[[], str]:
|
||||
"""Return a zero-arg callable that yields a fresh MiniMax access token.
|
||||
"""Zero-arg callable yielding a fresh MiniMax access token.
|
||||
|
||||
The Anthropic SDK caches ``api_key`` as a static string at construction time, so a session that
|
||||
resolves credentials once at startup would keep sending the same bearer until MiniMax returns
|
||||
401 — typically ~15 minutes in, because MiniMax issues short-lived access tokens.
|
||||
The Anthropic SDK caches ``api_key`` at construction; MiniMax tokens live ~15 minutes, so a
|
||||
static bearer would start 401-ing mid-session.
|
||||
"""
|
||||
def _provide() -> str:
|
||||
token = _minimax_fresh_state().get("access_token")
|
||||
@@ -317,11 +311,7 @@ def resolve_minimax_oauth_runtime_credentials(
|
||||
*, min_token_ttl_seconds: int = MINIMAX_OAUTH_REFRESH_SKEW_SECONDS,
|
||||
as_token_provider: bool = False,
|
||||
) -> Dict[str, Any]:
|
||||
"""Return {provider, api_key, base_url, source} for minimax-oauth.
|
||||
|
||||
The default (string ``api_key``) preserves the historical contract for diagnostic call sites
|
||||
like ``hermes status`` that just want to know whether a valid token exists right now.
|
||||
"""
|
||||
"""Return {provider, api_key, base_url, source}; string ``api_key`` by default (``hermes status`` contract)."""
|
||||
state = _minimax_fresh_state()
|
||||
return {
|
||||
"provider": "minimax-oauth",
|
||||
|
||||
@@ -1,9 +1,7 @@
|
||||
"""Interactive model picker used after OAuth login.
|
||||
|
||||
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
|
||||
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
|
||||
helpers are imported lazily inside each function (no import cycle; patches on
|
||||
``hermes_cli.auth.<helper>`` still intercept).
|
||||
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
|
||||
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -13,7 +11,6 @@ import subprocess
|
||||
from typing import Dict, List, Optional
|
||||
from hermes_cli.auth_constants import DEFAULT_NOUS_PORTAL_URL
|
||||
|
||||
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
|
||||
logger = logging.getLogger("hermes_cli.auth")
|
||||
|
||||
_CUSTOM_LABEL = "Enter custom model name"
|
||||
@@ -52,10 +49,10 @@ def _confirm_selection_guards(
|
||||
|
||||
|
||||
class _ModelPickerRows:
|
||||
"""Column-aligned model rows (name + $/Mtok prices + Nous sale chrome) for the model picker.
|
||||
"""Column-aligned picker rows (name + $/Mtok + Nous sale chrome).
|
||||
|
||||
Sale chrome (★ / -N% / was) is drawn as curses/ANSI segments (yellow % / dim "was"), not baked
|
||||
into one plain string — curses addnstr would otherwise render escape bytes literally.
|
||||
Sale chrome is emitted as styled segments, not ANSI baked into one string — curses addnstr
|
||||
would render escape bytes literally.
|
||||
"""
|
||||
|
||||
def __init__(
|
||||
@@ -164,15 +161,13 @@ def _prompt_model_selection(
|
||||
"""
|
||||
from hermes_cli.cli_output import line_input
|
||||
_unavailable = unavailable_models or []
|
||||
# Sale chrome (★ / -N% / was) is Nous Portal-only — never for OpenRouter or other providers
|
||||
# even if pricing.original is somehow present.
|
||||
# Sale chrome is Nous Portal-only, even if pricing.original is present for another provider.
|
||||
sale_chrome = (confirm_provider or "").strip().lower() == "nous"
|
||||
|
||||
def _confirmed_selection(mid: str) -> Optional[str]:
|
||||
if not mid:
|
||||
return None
|
||||
# The cost guard only runs when a provider is known (pricing lookups need one); id-keyed
|
||||
# guards like the data-policy guard always run — even via a custom endpoint or gateway.
|
||||
# Cost guard needs a known provider; id-keyed guards (data policy) always run.
|
||||
ok = _confirm_selection_guards(
|
||||
mid, provider=confirm_provider, base_url=confirm_base_url, api_key=confirm_api_key,
|
||||
include_kinds=None if confirm_provider else ["data_policy"],
|
||||
@@ -209,17 +204,14 @@ def _prompt_model_selection(
|
||||
if not unavailable_footer and _unavailable:
|
||||
unavailable_footer = f"Upgrade at {_upgrade_url} for paid models"
|
||||
|
||||
# The pricing column header (and any unavailable-models block) is shown as a multi-line
|
||||
# description above the list so it survives the curses screen clear. menu_title already
|
||||
# embeds the aligned price header; keep only the header/legend portion.
|
||||
# Header/legend + unavailable block go in the description so they survive the curses clear.
|
||||
desc_lines: list[str] = menu_title.split("\n", 1)[1].splitlines() if rows.has_pricing else []
|
||||
if _unavailable:
|
||||
desc_lines.extend(f" {rows.label(mid)}" for mid in _unavailable)
|
||||
desc_lines.append(f" ── {unavailable_footer} ──")
|
||||
|
||||
# Search haystacks keep pricing labels visible while adding aliases for brand-less wire
|
||||
# ids (e.g. Kimi Coding `k3` ↔ query "kimi"). model_search_text always starts with the
|
||||
# wire id; only append when aliases add tokens beyond the bare id already in the label.
|
||||
# Search haystack = label + aliases for brand-less wire ids (Kimi `k3` ↔ "kimi"); skip when
|
||||
# model_search_text adds nothing beyond the bare id.
|
||||
from hermes_cli.model_search import model_search_text
|
||||
model_search_labels = []
|
||||
for mid in ordered:
|
||||
|
||||
+4
-14
@@ -1,9 +1,7 @@
|
||||
"""Qwen OAuth (qwen-cli token file) runtime credentials and status.
|
||||
|
||||
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
|
||||
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
|
||||
helpers are imported lazily inside each function (no import cycle; patches on
|
||||
``hermes_cli.auth.<helper>`` still intercept).
|
||||
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
|
||||
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -19,7 +17,6 @@ from hermes_cli.auth_constants import (
|
||||
QWEN_OAUTH_TOKEN_URL, _FORM_JSON_HEADERS, _qwen_err, httpx,
|
||||
)
|
||||
|
||||
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
|
||||
logger = logging.getLogger("hermes_cli.auth")
|
||||
|
||||
_RERUN = "Re-run 'qwen auth qwen-oauth'."
|
||||
@@ -105,12 +102,7 @@ def _refresh_qwen_cli_tokens(tokens: Dict[str, Any], timeout_seconds: float = 20
|
||||
|
||||
|
||||
def _mark_qwen_oauth_active(creds: Dict[str, Any]) -> None:
|
||||
"""Set active_provider to qwen-oauth in auth.json.
|
||||
|
||||
Qwen tokens live in the Qwen CLI credential file, so this writes only a minimal provider-state
|
||||
entry (base_url for display) and sets active_provider so ``get_active_provider()`` and the
|
||||
setup wizard's credential check detect the provider.
|
||||
"""
|
||||
"""Set active_provider to qwen-oauth with a minimal state entry (tokens stay in the Qwen CLI file)."""
|
||||
from hermes_cli.auth import _auth_store_lock, _load_auth_store, _save_auth_store, _save_provider_state
|
||||
with _auth_store_lock():
|
||||
auth_store = _load_auth_store()
|
||||
@@ -146,9 +138,7 @@ def get_qwen_auth_status() -> Dict[str, Any]:
|
||||
from hermes_cli.auth import _qwen_cli_auth_path, resolve_qwen_runtime_credentials
|
||||
auth_path = _qwen_cli_auth_path()
|
||||
try:
|
||||
# Validate the runtime credentials, including refresh when the cached CLI token is expired;
|
||||
# otherwise stale tokens show up as "logged in" and `hermes model` walks users into a
|
||||
# broken Qwen setup flow.
|
||||
# Refresh-validate: otherwise stale CLI tokens read as "logged in" and break `hermes model`.
|
||||
creds = resolve_qwen_runtime_credentials(refresh_if_expiring=True)
|
||||
return {
|
||||
"logged_in": True, "auth_file": str(auth_path), "source": creds.get("source"),
|
||||
|
||||
@@ -1,9 +1,7 @@
|
||||
"""Spotify OAuth (loopback PKCE) login, refresh and runtime credentials.
|
||||
|
||||
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
|
||||
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
|
||||
helpers are imported lazily inside each function (no import cycle; patches on
|
||||
``hermes_cli.auth.<helper>`` still intercept).
|
||||
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
|
||||
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -26,7 +24,6 @@ from hermes_cli.auth_constants import (
|
||||
_spotify_err, httpx,
|
||||
)
|
||||
|
||||
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
|
||||
logger = logging.getLogger("hermes_cli.auth")
|
||||
|
||||
_CALLBACK_HTML = "<html><body><h1>Spotify authorization {}.</h1>You can close this tab.</body></html>"
|
||||
@@ -366,8 +363,7 @@ def _spotify_interactive_setup(redirect_uri_hint: str) -> str:
|
||||
print(f"\nNo Client ID entered. See {SPOTIFY_DOCS_URL} for the full guide.")
|
||||
raise SystemExit("Spotify setup cancelled: empty Client ID.")
|
||||
|
||||
# Persist so subsequent `hermes auth spotify` runs skip the wizard. Only persist a non-default
|
||||
# redirect URI, to avoid pinning users to a value the default might later change to.
|
||||
# Persist so later runs skip the wizard; only pin a NON-default redirect URI.
|
||||
save_env_value("HERMES_SPOTIFY_CLIENT_ID", raw)
|
||||
if redirect_uri_hint and redirect_uri_hint != DEFAULT_SPOTIFY_REDIRECT_URI:
|
||||
save_env_value("HERMES_SPOTIFY_REDIRECT_URI", redirect_uri_hint)
|
||||
@@ -380,8 +376,7 @@ def login_spotify_command(args) -> None:
|
||||
from hermes_cli.auth import _auth_store_lock, _can_open_graphical_browser, _is_remote_session, _load_auth_store, _print_loopback_ssh_hint, _save_auth_store, _store_provider_state, get_provider_auth_state
|
||||
existing_state = get_provider_auth_state("spotify") or {}
|
||||
|
||||
# Interactive wizard: if no client_id is configured anywhere, walk the user through creating
|
||||
# the Spotify developer app instead of crashing with "HERMES_SPOTIFY_CLIENT_ID is required".
|
||||
# No client_id anywhere -> wizard instead of "HERMES_SPOTIFY_CLIENT_ID is required".
|
||||
try:
|
||||
client_id = _spotify_client_id(getattr(args, "client_id", None), existing_state)
|
||||
except AuthError as exc:
|
||||
|
||||
+35
-57
@@ -1,9 +1,7 @@
|
||||
"""xAI Grok OAuth: token store, discovery, refresh, device-code login.
|
||||
|
||||
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
|
||||
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
|
||||
helpers are imported lazily inside each function (no import cycle; patches on
|
||||
``hermes_cli.auth.<helper>`` still intercept).
|
||||
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
|
||||
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -26,7 +24,6 @@ from utils import env_float
|
||||
|
||||
if TYPE_CHECKING: # annotation-only; the runtime import would be a cycle
|
||||
from hermes_cli.auth import ProviderConfig
|
||||
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
|
||||
logger = logging.getLogger("hermes_cli.auth")
|
||||
|
||||
_RELOGIN = "Re-authenticate with `hermes model`."
|
||||
@@ -101,20 +98,18 @@ def _read_xai_oauth_tokens(*, _lock: bool = True) -> Dict[str, Any]:
|
||||
|
||||
|
||||
def _write_through_xai_oauth_to_global_root(state: Dict[str, Any]) -> None:
|
||||
"""Persist a rotated xAI OAuth ``state`` into the global-root auth.json (best effort).
|
||||
"""Best-effort persist of a rotated xAI grant into the global-root auth.json.
|
||||
|
||||
xAI rotates the refresh_token on every refresh, so when a profile session refreshes a grant it
|
||||
resolved from the root fallback, the rotated chain must land back in root. Only updates
|
||||
``providers.xai-oauth`` in root; never touches the profile store (the caller saved that).
|
||||
Swallows all errors — a failed write-through degrades to root-stale, never breaks the profile save.
|
||||
xAI rotates refresh_token on every refresh, so a profile that refreshed a root-resolved grant
|
||||
must write the chain back to root. Touches only root ``providers.xai-oauth``; swallows all
|
||||
errors (root-stale is better than breaking the profile's own save).
|
||||
"""
|
||||
from hermes_cli.auth import _global_auth_file_path, _persist_provider_state_to_store
|
||||
global_path = _global_auth_file_path()
|
||||
if global_path is None: # classic mode (profile == root); the profile save already hit root
|
||||
return
|
||||
# Seat belt: under pytest, refuse to write the real user's ~/.hermes/auth.json even when
|
||||
# HERMES_HOME points at a profile path (mirrors the read-side guard in _load_global_auth_store).
|
||||
# Uses the unmodified HOME env, not Path.home() which fixtures may monkeypatch.
|
||||
# Seat belt: under pytest never write the real ~/.hermes/auth.json (mirrors the read-side guard
|
||||
# in _load_global_auth_store). Uses raw HOME, not Path.home(), which fixtures may monkeypatch.
|
||||
real_home_env = os.environ.get("HOME", "") if os.environ.get("PYTEST_CURRENT_TEST") else ""
|
||||
if real_home_env:
|
||||
real_root = Path(real_home_env) / ".hermes" / "auth.json"
|
||||
@@ -134,21 +129,19 @@ def _save_xai_oauth_tokens(
|
||||
last_refresh: Optional[str] = None, auth_mode: str = "oauth_device_code",
|
||||
set_active: bool = True,
|
||||
) -> None:
|
||||
"""Persist xAI OAuth tokens into the auth store.
|
||||
"""Persist xAI OAuth tokens; *set_active* also promotes ``xai-oauth`` to ``active_provider``.
|
||||
|
||||
*set_active* (default True) also promotes ``xai-oauth`` to ``active_provider`` — right for an
|
||||
intentional login. Pass False for side-tool credential bootstrap (TTS/setup, tools config,
|
||||
dashboard token save, token refresh) so inference routing is unchanged.
|
||||
Pass ``set_active=False`` for side-tool bootstrap (TTS/setup, tools config, dashboard, refresh)
|
||||
so inference routing is unchanged.
|
||||
"""
|
||||
from hermes_cli.auth import _auth_store_lock, _global_auth_file_path, _load_auth_store, _load_provider_state_with_source, _same_path, _save_auth_store, _store_provider_state, _utc_now_z, _write_through_xai_oauth_to_global_root
|
||||
if last_refresh is None:
|
||||
last_refresh = _utc_now_z()
|
||||
with _auth_store_lock():
|
||||
auth_store = _load_auth_store()
|
||||
# A profile without its own xai-oauth block reads the root grant through
|
||||
# _load_provider_state's fallback. Refreshing that (rotating) grant must write the rotated
|
||||
# chain back to root, or root keeps a revoked refresh token. Decide by where the grant was
|
||||
# resolved FROM, not by key presence: _store_provider_state below would create the key.
|
||||
# A profile lacking its own xai-oauth block reads root's grant via fallback; refreshing it
|
||||
# must write the rotated chain back to root or root keeps a revoked refresh token. Decide by
|
||||
# where the grant was resolved FROM (key presence lies: _store_provider_state creates it).
|
||||
state, source_path = _load_provider_state_with_source(auth_store, "xai-oauth")
|
||||
state = state if state is not None else {}
|
||||
state.update(tokens=tokens, last_refresh=last_refresh, auth_mode=auth_mode)
|
||||
@@ -158,8 +151,7 @@ def _save_xai_oauth_tokens(
|
||||
state["redirect_uri"] = redirect_uri
|
||||
global_root = _global_auth_file_path()
|
||||
if source_path is not None and global_root is not None and _same_path(source_path, global_root):
|
||||
# Resolved from root — write back to root only. Storing on the profile auth_store would
|
||||
# create a shadowing providers.xai-oauth key that disables write-through next refresh.
|
||||
# Root-only write-back: a profile copy would shadow root and disable write-through.
|
||||
_write_through_xai_oauth_to_global_root(state)
|
||||
else:
|
||||
_store_provider_state(auth_store, "xai-oauth", state, set_active=set_active)
|
||||
@@ -187,12 +179,10 @@ def _xai_access_token_is_expiring(access_token: str, skew_seconds: int = 0) -> b
|
||||
|
||||
|
||||
def _xai_proactive_refresh_skew_seconds(access_token: str) -> int:
|
||||
"""How far before JWT ``exp`` to proactively refresh xAI OAuth tokens.
|
||||
"""Proactive-refresh lead time before JWT ``exp``.
|
||||
|
||||
SuperGrok sessions ship multi-hour tokens where the hour-long skew makes sense, but device-code
|
||||
logins often return ~15-minute JWTs; the full skew would force a refresh on every credential
|
||||
resolution, burning single-use refresh tokens and racing concurrent callers into
|
||||
``invalid_grant`` quarantine.
|
||||
Device-code logins often return ~15-minute JWTs; the full hour-long skew would refresh on every
|
||||
resolution, burning single-use refresh tokens and racing callers into ``invalid_grant``.
|
||||
"""
|
||||
max_skew = XAI_ACCESS_TOKEN_REFRESH_SKEW_SECONDS
|
||||
exp = _xai_jwt_exp(access_token)
|
||||
@@ -221,11 +211,10 @@ def _xai_url_problem(url: str) -> tuple[Optional[str], str]:
|
||||
|
||||
|
||||
def _xai_validate_oauth_endpoint(url: str, *, field: str) -> str:
|
||||
"""Refuse any OIDC discovery endpoint that isn't HTTPS on the xAI origin.
|
||||
"""Refuse a discovery endpoint that isn't HTTPS on the xAI origin.
|
||||
|
||||
The discovery result is cached in auth.json, so a single MITM at login could plant a malicious
|
||||
``token_endpoint`` that receives the refresh_token forever. Pinning scheme + host (RFC 8414 §2)
|
||||
removes that persistence.
|
||||
Discovery is cached in auth.json, so one MITM at login could plant a ``token_endpoint`` that
|
||||
receives the refresh_token forever; pinning scheme + host (RFC 8414 §2) removes that.
|
||||
"""
|
||||
problem, host = _xai_url_problem(url)
|
||||
if problem is None:
|
||||
@@ -244,11 +233,9 @@ def _xai_validate_oauth_endpoint(url: str, *, field: str) -> str:
|
||||
|
||||
|
||||
def _xai_validate_inference_base_url(value: str, *, fallback: str) -> str:
|
||||
"""Refuse a non-xAI base_url for the OAuth-authenticated inference path.
|
||||
"""Pin the OAuth inference base_url to ``*.x.ai``; warn and use *fallback* on rejection.
|
||||
|
||||
Pins the inference origin to ``api.x.ai`` (or any ``*.x.ai`` subdomain). On rejection, fall
|
||||
back to the default and log a warning rather than raise — a bad env var should not deadlock
|
||||
authentication, but it must never leak the bearer. Empty input returns ``fallback``.
|
||||
Warn-not-raise: a bad env var must not deadlock auth, but the bearer must never leak elsewhere.
|
||||
"""
|
||||
candidate = (value or "").strip().rstrip("/")
|
||||
if not candidate:
|
||||
@@ -321,9 +308,8 @@ def refresh_xai_oauth_pure(
|
||||
f"xAI OAuth is missing refresh_token. {_RELOGIN}", "xai_auth_missing_refresh_token", relogin=True,
|
||||
)
|
||||
endpoint = token_endpoint.strip() or _xai_oauth_discovery(timeout_seconds)["token_endpoint"]
|
||||
# Re-validate cached endpoints on the refresh hot path: an auth.json written by an older Hermes
|
||||
# (or hand-edited) may carry a non-xAI token_endpoint that would receive every future
|
||||
# refresh_token in plaintext if trusted blindly.
|
||||
# Re-validate cached endpoints: an old/hand-edited auth.json may carry a non-xAI token_endpoint
|
||||
# that would otherwise receive every future refresh_token.
|
||||
_xai_validate_oauth_endpoint(endpoint, field="token_endpoint")
|
||||
timeout = httpx.Timeout(max(5.0, float(timeout_seconds)))
|
||||
with httpx.Client(timeout=timeout, headers={"Accept": "application/json"}) as client:
|
||||
@@ -334,9 +320,8 @@ def refresh_xai_oauth_pure(
|
||||
if response.status_code != 200:
|
||||
detail = response.text.strip()
|
||||
suffix = f" Response: {detail}" if detail else ""
|
||||
# 403 from xAI's token endpoint is almost always a tier / entitlement gate (the grant exists
|
||||
# but the account isn't allowlisted for API access). Re-running `hermes model` won't fix
|
||||
# that — use a separate code so format_auth_error skips the re-authenticate hint.
|
||||
# 403 is almost always a tier/entitlement gate; re-login won't fix it, so use a separate
|
||||
# code and format_auth_error skips the re-authenticate hint.
|
||||
if response.status_code == 403:
|
||||
raise _xai_err(
|
||||
"xAI token refresh failed with HTTP 403." + suffix
|
||||
@@ -365,8 +350,7 @@ def refresh_xai_oauth_pure(
|
||||
def _refresh_xai_oauth_tokens(
|
||||
tokens: Dict[str, Any], *, token_endpoint: str, redirect_uri: str = "", timeout_seconds: float
|
||||
) -> Dict[str, Any]:
|
||||
# Re-persist whatever auth_mode is already stored (legacy pre-device-code logins may still
|
||||
# carry ``oauth_pkce``): the refresh hot path must not relabel how the grant was obtained.
|
||||
# Keep the stored auth_mode (legacy logins may carry ``oauth_pkce``): refresh must not relabel it.
|
||||
from hermes_cli.auth import _load_auth_store, _load_provider_state, refresh_xai_oauth_pure
|
||||
try:
|
||||
state = _load_provider_state(_load_auth_store(), "xai-oauth") or {}
|
||||
@@ -386,8 +370,7 @@ def _refresh_xai_oauth_tokens(
|
||||
updated_tokens["expires_in"] = refreshed["expires_in"]
|
||||
if refreshed.get("token_type"):
|
||||
updated_tokens["token_type"] = refreshed["token_type"]
|
||||
# set_active=False: refresh must not flip active_provider — TTS/side tools can refresh xAI
|
||||
# tokens while chat still routes through another provider.
|
||||
# set_active=False: side tools (TTS) refresh xAI tokens while chat routes elsewhere.
|
||||
_save_xai_oauth_tokens(
|
||||
updated_tokens, discovery={"token_endpoint": token_endpoint}, redirect_uri=redirect_uri,
|
||||
last_refresh=refreshed["last_refresh"], auth_mode=auth_mode, set_active=False,
|
||||
@@ -396,11 +379,9 @@ def _refresh_xai_oauth_tokens(
|
||||
|
||||
|
||||
def _quarantine_xai_oauth_tokens(exc: AuthError) -> None:
|
||||
"""Clear dead xAI tokens from auth.json after a terminal refresh failure.
|
||||
"""Clear dead xAI tokens after a terminal (400/401/403) refresh failure so later sessions fail fast.
|
||||
|
||||
Terminal = HTTP 400/401/403 (invalid_grant, token revoked). Subsequent sessions then fail fast
|
||||
without a network retry. Best-effort: persistence failures are logged and swallowed (caller
|
||||
re-raises the original error regardless).
|
||||
Best-effort: persistence failures are logged and swallowed; the caller re-raises regardless.
|
||||
"""
|
||||
from hermes_cli.auth import _last_auth_error_marker, _load_auth_store, _load_provider_state, _save_auth_store, _store_provider_state
|
||||
try:
|
||||
@@ -471,8 +452,7 @@ def resolve_xai_oauth_runtime_credentials(
|
||||
"api_key": _clean(tokens.get("access_token")),
|
||||
"source": "hermes-auth-store",
|
||||
"last_refresh": data.get("last_refresh"),
|
||||
# Display/telemetry only. Device-code is the only supported xAI OAuth flow, so report it
|
||||
# unconditionally — auth.json may still carry a legacy ``oauth_pkce`` label.
|
||||
# Display only; auth.json may still carry a legacy ``oauth_pkce`` label.
|
||||
"auth_mode": "oauth_device_code",
|
||||
}
|
||||
|
||||
@@ -506,11 +486,9 @@ def _login_xai_oauth(args, pconfig: ProviderConfig, *, force_new_login: bool = F
|
||||
redirect_uri=creds.get("redirect_uri", ""), last_refresh=creds.get("last_refresh"),
|
||||
auth_mode="oauth_device_code",
|
||||
)
|
||||
# An explicit interactive re-login means the user wants the xAI credential re-enabled.
|
||||
# ``hermes auth remove xai-oauth`` leaves a ``device_code`` suppression marker that otherwise
|
||||
# stops the singleton seed from re-creating the pool entry. Kept OUT of _save_xai_oauth_tokens
|
||||
# on purpose — that helper is shared with the refresh hot path, which must never mutate
|
||||
# suppression state.
|
||||
# Explicit re-login re-enables the credential: clear the ``device_code`` suppression marker left
|
||||
# by ``hermes auth remove xai-oauth``. Deliberately NOT inside _save_xai_oauth_tokens — the
|
||||
# refresh hot path shares that helper and must never mutate suppression state.
|
||||
unsuppress_credential_source("xai-oauth", "device_code")
|
||||
config_path = _update_config_for_provider("xai-oauth", creds.get("base_url", DEFAULT_XAI_OAUTH_BASE_URL))
|
||||
_print_login_success("xai-oauth", config_path, show_auth_state=True)
|
||||
|
||||
+11
-21
@@ -1,9 +1,7 @@
|
||||
"""Kimi Code and Z.AI endpoint auto-detection, LM Studio base-URL normalization.
|
||||
|
||||
Split out of ``hermes_cli/auth.py``; every moved name is re-imported there, so
|
||||
``hermes_cli.auth.<name>`` keeps resolving (and monkeypatching) as before. Origin-internal
|
||||
helpers are imported lazily inside each function (no import cycle; patches on
|
||||
``hermes_cli.auth.<helper>`` still intercept).
|
||||
Re-exported from ``hermes_cli/auth.py`` (patch targets unchanged); origin helpers are imported
|
||||
lazily per function so ``hermes_cli.auth.<helper>`` patches still intercept and no cycle forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -13,13 +11,10 @@ import hashlib
|
||||
from typing import Dict, Optional
|
||||
from hermes_cli.auth_constants import httpx
|
||||
|
||||
# Log-record parity with the origin module (caplog tests pin "hermes_cli.auth").
|
||||
logger = logging.getLogger("hermes_cli.auth")
|
||||
|
||||
# Kimi Code (kimi.com/code) issues "sk-kimi-" keys that only work on api.kimi.com/coding; legacy
|
||||
# platform.moonshot.ai keys work on api.moonshot.ai/v1 (the old default). Intentionally NO /v1
|
||||
# suffix: the /coding endpoint speaks Anthropic Messages and the SDK appends "/v1/messages"
|
||||
# itself — "/coding/v1" would produce "/coding/v1/v1/messages" (a 404).
|
||||
# "sk-kimi-" keys only work on api.kimi.com/coding; legacy moonshot keys use the old default.
|
||||
# NO /v1 suffix: the anthropic SDK appends "/v1/messages" itself ("/coding/v1" would 404).
|
||||
KIMI_CODE_BASE_URL = "https://api.kimi.com/coding"
|
||||
|
||||
|
||||
@@ -32,10 +27,9 @@ def _resolve_kimi_base_url(api_key: str, default_url: str, env_override: str) ->
|
||||
return default_url
|
||||
|
||||
|
||||
# Z.AI bills general vs coding plans, and global vs China endpoints, separately; a key that works
|
||||
# on one may return "Insufficient balance" on another, so we probe at setup time and store the
|
||||
# working endpoint. Each entry lists candidate models to try in order — newer coding-plan accounts
|
||||
# may only have access to recent models while older ones still use glm-4.7.
|
||||
# Z.AI bills general/coding plans and global/China endpoints separately ("Insufficient balance" on
|
||||
# the wrong one), so probe once and cache. Candidate models are tried in order: newer coding-plan
|
||||
# accounts may only have recent GLM slugs, older ones still glm-4.7.
|
||||
_ZAI_CODING_PROBE_MODELS = ["glm-5.3", "glm-5.3-flash", "glm-5.2", "glm-5.1", "glm-5v-turbo", "glm-4.7"]
|
||||
ZAI_ENDPOINTS = [
|
||||
# (id, base_url, probe_models, label)
|
||||
@@ -69,8 +63,7 @@ def _probe_single_zai_endpoint(api_key: str, endpoint: tuple, timeout: float) ->
|
||||
def detect_zai_endpoint(api_key: str, timeout: float = 8.0) -> Optional[Dict[str, str]]:
|
||||
"""Probe z.ai endpoints in parallel; first working one in ZAI_ENDPOINTS priority order, or None."""
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
# No `with` block: a context manager would join ALL probe threads on exit, defeating the early
|
||||
# return below. shutdown(wait=False) lets surviving probes drain in the background.
|
||||
# No `with`: it would join ALL probes on exit, defeating the early return below.
|
||||
pool = ThreadPoolExecutor(max_workers=len(ZAI_ENDPOINTS))
|
||||
try:
|
||||
futures = {pool.submit(_probe_single_zai_endpoint, api_key, ep, timeout): ep[0] for ep in ZAI_ENDPOINTS}
|
||||
@@ -111,8 +104,7 @@ def _resolve_zai_base_url(api_key: str, default_url: str, env_override: str) ->
|
||||
from hermes_cli.auth import _auth_store_lock, _load_auth_store, _load_provider_state, _save_auth_store, _store_provider_state, detect_zai_endpoint
|
||||
if env_override:
|
||||
return env_override
|
||||
# No API key → don't probe (N×M HTTPS requests with an empty Bearer, all 401). Hit during
|
||||
# auxiliary-client auto-detection for users with no Z.AI credentials at all — pure latency.
|
||||
# No key -> don't probe (N×M 401s); auxiliary-client auto-detection hits this for everyone.
|
||||
if not api_key:
|
||||
return default_url
|
||||
|
||||
@@ -134,15 +126,13 @@ def _resolve_zai_base_url(api_key: str, default_url: str, env_override: str) ->
|
||||
"model": detected.get("model", ""), "label": detected.get("label", ""),
|
||||
"key_hash": key_hash,
|
||||
}
|
||||
# Persist failure (disk full, permissions, lock timeout) must not break resolution — detection
|
||||
# already succeeded; worst case the next start re-probes.
|
||||
# Persist failure must not break resolution; worst case the next start re-probes.
|
||||
try:
|
||||
with _auth_store_lock():
|
||||
auth_store = _load_auth_store() # reload under lock to avoid overwriting concurrent changes
|
||||
state_under_lock = _load_provider_state(auth_store, "zai") or {}
|
||||
state_under_lock["detected_endpoint"] = detected_endpoint
|
||||
# set_active=False: this runs from credential-pool env seeding for ANY user with a Z.AI
|
||||
# key in env, and caching a probe result must not flip their active provider.
|
||||
# set_active=False: runs from credential-pool env seeding; must not flip active provider.
|
||||
_store_provider_state(auth_store, "zai", state_under_lock, set_active=False)
|
||||
_save_auth_store(auth_store)
|
||||
except Exception as exc:
|
||||
|
||||
Reference in New Issue
Block a user