refactor(hermes_cli): keepalive/portal_cli/pairing — dict dispatch for pairing, compact comments (632->579 LOC)
This commit is contained in:
@@ -19,28 +19,16 @@ from hermes_cli.auth import (
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Two things have to line up for the keepalive to actually keep anything alive.
|
||||
#
|
||||
# 1. The tick has to be frequent enough to see the credential before it dies.
|
||||
# Nous credential lifetimes are not fixed: this varies by account, and
|
||||
# installs have been observed at both ~3594s and ~899s. The tick therefore
|
||||
# derives from the lifetime the server actually issued rather than assuming
|
||||
# an hour, capped by the configured interval and floored so a pathological
|
||||
# lifetime cannot spin the thread.
|
||||
#
|
||||
# 2. The refresh has to fire while the tick can still act on it. The refresh
|
||||
# only triggers once a credential is within a skew window of expiry, so a
|
||||
# tick spaced wider than that window steps straight over it. The keepalive
|
||||
# widens the window to "will this credential outlive my next tick?" instead
|
||||
# of the request-path default of 120s. Without this, ticking faster only
|
||||
# narrows the gap; it never closes it.
|
||||
#
|
||||
# The original 6-hour tick against a one-hour credential failed both tests, so
|
||||
# in practice every hour expired reactively into a 401 plus a re-auth retry.
|
||||
# Two things must line up for the keepalive to keep anything alive:
|
||||
# 1. The tick must be frequent enough to see the credential before it dies. Lifetimes vary by
|
||||
# account (~3594s and ~899s observed), so the tick derives from the lifetime the server issued,
|
||||
# capped by the configured interval and floored so a pathological lifetime can't spin the thread.
|
||||
# 2. The refresh must fire while the tick can still act on it: refresh triggers only within a skew
|
||||
# window of expiry, so the keepalive widens that window to "will this credential outlive my
|
||||
# next tick?" instead of the request path's 120s. Ticking faster alone never closes the gap.
|
||||
NOUS_AUTH_KEEPALIVE_INTERVAL_SECONDS = 15 * 60
|
||||
NOUS_AUTH_KEEPALIVE_MIN_INTERVAL_SECONDS = 60
|
||||
# Ticks per credential lifetime. Four keeps the refresh comfortably ahead of
|
||||
# expiry without making the thread chatty.
|
||||
# Ticks per credential lifetime: four keeps refresh comfortably ahead of expiry without chatter.
|
||||
NOUS_AUTH_KEEPALIVE_TICKS_PER_LIFETIME = 4
|
||||
NOUS_AUTH_KEEPALIVE_INITIAL_DELAY_SECONDS = 60
|
||||
NOUS_AUTH_KEEPALIVE_INTERVAL_CONFIG_KEY = "keepalive_interval_seconds"
|
||||
@@ -60,12 +48,7 @@ def _timeout_seconds(value: Optional[float]) -> float:
|
||||
|
||||
|
||||
def _nous_config() -> dict:
|
||||
"""Return the ``nous:`` section of config.yaml, or {} on any failure.
|
||||
|
||||
Imported lazily: this module is loaded by the gateway and the web server
|
||||
during startup, and the config loader pulls in a wider dependency graph
|
||||
than the keepalive itself needs.
|
||||
"""
|
||||
"""The ``nous:`` section of config.yaml, or {} on any failure (config loader imported lazily)."""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
|
||||
@@ -76,13 +59,8 @@ def _nous_config() -> dict:
|
||||
|
||||
|
||||
def _interval_seconds(value: Optional[int]) -> int:
|
||||
"""Resolve the keepalive tick interval.
|
||||
|
||||
Explicit argument wins, then ``nous.keepalive_interval_seconds`` in
|
||||
config.yaml, then the module default. This is a behavioural threshold
|
||||
rather than a credential, so it lives in config.yaml and not in .env.
|
||||
A non-positive result disables the keepalive thread entirely, which is
|
||||
the documented way to turn it off.
|
||||
"""Tick interval: explicit argument, then ``nous.keepalive_interval_seconds`` in config.yaml,
|
||||
then the module default. Non-positive disables the keepalive thread (the documented way off).
|
||||
"""
|
||||
if value is not None:
|
||||
try:
|
||||
@@ -98,19 +76,14 @@ def _interval_seconds(value: Optional[int]) -> int:
|
||||
except (TypeError, ValueError):
|
||||
logger.warning(
|
||||
"Ignoring invalid nous.%s=%r; using %ds",
|
||||
NOUS_AUTH_KEEPALIVE_INTERVAL_CONFIG_KEY,
|
||||
raw,
|
||||
NOUS_AUTH_KEEPALIVE_INTERVAL_SECONDS,
|
||||
NOUS_AUTH_KEEPALIVE_INTERVAL_CONFIG_KEY, raw, NOUS_AUTH_KEEPALIVE_INTERVAL_SECONDS,
|
||||
)
|
||||
return NOUS_AUTH_KEEPALIVE_INTERVAL_SECONDS
|
||||
|
||||
|
||||
def _observed_lifetime_seconds() -> Optional[int]:
|
||||
"""Lifetime the server issued for the current Nous credentials, in seconds.
|
||||
|
||||
Both the access token and the invoke agent key carry their own lifetime and
|
||||
they are not always equal, so the shorter one governs. Returns None when no
|
||||
usable value is stored, in which case the caller keeps its configured tick.
|
||||
"""Server-issued lifetime (seconds) of the current Nous credentials; the shorter of the access
|
||||
token and the invoke agent key governs. None when nothing usable is stored.
|
||||
"""
|
||||
state = get_provider_auth_state("nous") or {}
|
||||
lifetimes = []
|
||||
@@ -129,18 +102,12 @@ def _tick_seconds(configured_interval: int, lifetime: Optional[int]) -> int:
|
||||
if not lifetime or lifetime <= 0:
|
||||
return configured_interval
|
||||
derived = lifetime // NOUS_AUTH_KEEPALIVE_TICKS_PER_LIFETIME
|
||||
return max(
|
||||
NOUS_AUTH_KEEPALIVE_MIN_INTERVAL_SECONDS,
|
||||
min(configured_interval, derived),
|
||||
)
|
||||
return max(NOUS_AUTH_KEEPALIVE_MIN_INTERVAL_SECONDS, min(configured_interval, derived))
|
||||
|
||||
|
||||
def _refresh_horizon_seconds(tick_seconds: int, floor_seconds: int) -> int:
|
||||
"""How much life a credential needs to be left alone this tick.
|
||||
|
||||
A credential that will not survive until the next tick has to be refreshed
|
||||
now, because nothing will look at it again before it expires. Hence
|
||||
tick + skew rather than the request path's bare skew.
|
||||
"""Life a credential needs to be left alone this tick: it must survive until the next tick
|
||||
(nothing looks at it again before then), hence tick + skew rather than the bare skew.
|
||||
"""
|
||||
return max(floor_seconds, tick_seconds + ACCESS_TOKEN_REFRESH_SKEW_SECONDS)
|
||||
|
||||
@@ -158,10 +125,8 @@ def _refresh_selected_pool_entry(
|
||||
min_key_ttl_seconds: int,
|
||||
min_access_ttl_seconds: Optional[int] = None,
|
||||
) -> Optional[bool]:
|
||||
"""Refresh the current Nous credential pool entry when it is stale.
|
||||
|
||||
True = pool entry usable/refreshed; False = pool exists but no usable entry;
|
||||
None = no Nous pool.
|
||||
"""Refresh the current pool entry when stale. True = usable/refreshed; False = pool exists but
|
||||
no usable entry; None = no Nous pool.
|
||||
"""
|
||||
try:
|
||||
from agent.credential_pool import load_pool
|
||||
@@ -185,10 +150,7 @@ def _refresh_selected_pool_entry(
|
||||
|
||||
if min_access_ttl_seconds is None:
|
||||
min_access_ttl_seconds = ACCESS_TOKEN_REFRESH_SKEW_SECONDS
|
||||
access_expiring = _is_expiring(
|
||||
getattr(entry, "expires_at", None),
|
||||
min_access_ttl_seconds,
|
||||
)
|
||||
access_expiring = _is_expiring(getattr(entry, "expires_at", None), min_access_ttl_seconds)
|
||||
key_usable = _agent_key_is_usable(_entry_state(entry), min_key_ttl_seconds)
|
||||
if access_expiring or not key_usable:
|
||||
if pool.try_refresh_current() is None:
|
||||
@@ -207,8 +169,7 @@ def refresh_nous_auth_keepalive_once(
|
||||
min_key_ttl_seconds = max(60, int(min_key_ttl_seconds))
|
||||
|
||||
pool_result = _refresh_selected_pool_entry(
|
||||
min_key_ttl_seconds=min_key_ttl_seconds,
|
||||
min_access_ttl_seconds=min_access_ttl_seconds,
|
||||
min_key_ttl_seconds=min_key_ttl_seconds, min_access_ttl_seconds=min_access_ttl_seconds
|
||||
)
|
||||
if pool_result is not None:
|
||||
return pool_result
|
||||
@@ -241,15 +202,12 @@ def _keepalive_loop(
|
||||
return
|
||||
|
||||
while not stop_event.is_set():
|
||||
# Re-read each pass: the lifetime can change when the account, plan, or
|
||||
# server-side policy does, and a keepalive that caches it would go stale
|
||||
# in exactly the case it exists to cover.
|
||||
# Re-read each pass: the lifetime changes with account/plan/policy; caching it would go
|
||||
# stale in exactly the case the keepalive exists to cover.
|
||||
tick = _tick_seconds(interval_seconds, _observed_lifetime_seconds())
|
||||
horizon = _refresh_horizon_seconds(tick, min_key_ttl_seconds)
|
||||
refresh_nous_auth_keepalive_once(
|
||||
min_key_ttl_seconds=horizon,
|
||||
min_access_ttl_seconds=horizon,
|
||||
timeout_seconds=timeout_seconds,
|
||||
min_key_ttl_seconds=horizon, min_access_ttl_seconds=horizon, timeout_seconds=timeout_seconds
|
||||
)
|
||||
stop_event.wait(tick)
|
||||
|
||||
|
||||
Reference in New Issue
Block a user