;
+ # _load_provider_state has the same root fallback, so dropping the
+ # copy keeps the profile working while removing the fork.
+ for provider_id in ("openai-codex", "xai-oauth"):
+ block = providers.get(provider_id)
+ if isinstance(block, dict) and block:
+ del providers[provider_id]
+ stripped["providers"].append(provider_id)
+ changed = True
+ if not changed:
+ return stripped
+ try:
+ _save_auth_store(store, target_path=auth_path)
+ except Exception:
+ logger.debug(
+ "Failed to strip cloned single-use OAuth grants from %s",
+ auth_path,
+ exc_info=True,
+ )
+ return stripped
+
+
+# ── One-time heal for installs that ALREADY forked a single-use grant ────────
+#
+# Fleets created before the clone-strip / root-write-through above have
+# profile-local copies of the root grant. Those copies are the same credential
+# with several owners: whichever profile rotated last holds the only live
+# refresh token and every other copy (root included) is spent. Upgrading alone
+# does not fix that — the first load in each profile would keep using its own
+# doomed copy. ``heal_forked_single_use_oauth_grants`` runs at profile
+# ``load_pool()`` time: it finds the profile rows that share LINEAGE with a
+# root row (same pool id — clone-all and the old borrowed-persist both kept
+# it — or the same account identity / token material), keeps the copy most
+# likely to still be live (freshest rotation), writes that copy into ROOT when
+# root's is older, and strips the profile's copy so the profile borrows root
+# from then on. Idempotent (a healed profile has no matched rows), never
+# touches API-key rows, never deletes a row that has no root counterpart
+# (an independent ``hermes -p auth add`` grant, or the only surviving
+# copy), and reads only the two auth.json files the existing root fallback
+# already reads — no environ / secret-scope reads.
+
+_OAUTH_TOKEN_FIELDS = (
+ "access_token",
+ "refresh_token",
+ "expires_at",
+ "expires_at_ms",
+ "last_refresh",
+)
+
+_oauth_heal_notices: List[str] = []
+# provider -> (profile auth.json path, auth.json mtime_ns, singleton mtime_ns)
+# of the last store verified fork-free; lets load_pool() skip the locked scan.
+_oauth_heal_clean_marks: Dict[str, Tuple[str, Optional[int], Optional[int]]] = {}
+
+
+def consume_oauth_heal_notices() -> List[str]:
+ """Return (and clear) human-readable notes about heals run in this process.
+
+ ``hermes auth list`` / ``hermes auth status`` print them so the user sees
+ that a forked grant was consolidated rather than only finding it in logs.
+ """
+ notes = list(_oauth_heal_notices)
+ _oauth_heal_notices.clear()
+ return notes
+
+
+def _oauth_identity(entry: Dict[str, Any]) -> Optional[str]:
+ """Stable account identity for an OAuth row when the token carries one.
+
+ Codex / xAI access tokens are JWTs with ``sub`` / ``email`` /
+ ``chatgpt_account_id`` claims; Anthropic ``sk-ant-oat`` tokens carry no
+ claims (returns None — lineage then rests on id / token material).
+ """
+ if not isinstance(entry, dict):
+ return None
+ for token in (entry.get("access_token"), entry.get("id_token")):
+ claims = _decode_jwt_claims(token)
+ if not claims:
+ continue
+ nested = claims.get("https://api.openai.com/auth")
+ account = nested.get("chatgpt_account_id") if isinstance(nested, dict) else None
+ for value in (account, claims.get("sub"), claims.get("email")):
+ if isinstance(value, str) and value.strip():
+ return value.strip()
+ return None
+
+
+def _oauth_freshness(entry: Dict[str, Any]) -> float:
+ """Best-effort 'how recently was this pair issued' score (epoch seconds).
+
+ A rotation always issues a later-expiring access token, so ``expires_at``
+ ordering identifies the live copy; ``last_refresh`` and the JWT ``exp``
+ claim are fallbacks for rows that do not persist expiry.
+ """
+ from agent.credential_pool import _parse_absolute_timestamp
+
+ best = 0.0
+ for key in ("expires_at_ms", "expires_at", "last_refresh"):
+ ts = _parse_absolute_timestamp(entry.get(key))
+ if ts and ts > best:
+ best = ts
+ if best == 0.0:
+ exp = _decode_jwt_claims(entry.get("access_token")).get("exp")
+ ts = _parse_absolute_timestamp(exp)
+ if ts:
+ best = ts
+ return best
+
+
+def _find_root_counterpart(
+ profile_row: Dict[str, Any], root_rows: List[Dict[str, Any]]
+) -> Optional[int]:
+ """Index of the root OAuth row that shares a grant lineage with *profile_row*.
+
+ Strongest evidence first: same pool ``id`` (clone-all and the pre-fix
+ borrowed-persist both preserved it), same account identity from JWT
+ claims, same token material (an unrotated copy). Fallback per the
+ one-grant-at-root rule: same provider + same OAuth client — every
+ Anthropic ``hermes_pkce`` grant uses one client id and carries no claims,
+ so two Anthropic OAuth rows with no contrary identity are one lineage.
+ Only a row whose identity claims name a DIFFERENT account is left alone
+ (an independent ``hermes -p
auth add`` login for another account).
+ """
+ candidates = [i for i, r in enumerate(root_rows) if _is_oauth_pool_payload(r)]
+ if not candidates:
+ return None
+ pid = profile_row.get("id")
+ for i in candidates:
+ if pid and root_rows[i].get("id") == pid:
+ return i
+ p_ident = _oauth_identity(profile_row)
+ for i in candidates:
+ r_ident = _oauth_identity(root_rows[i])
+ if p_ident and r_ident and p_ident == r_ident:
+ return i
+ for key in ("refresh_token", "access_token"):
+ p_val = profile_row.get(key)
+ if not (isinstance(p_val, str) and p_val.strip()):
+ continue
+ for i in candidates:
+ if root_rows[i].get(key) == p_val:
+ return i
+ # Fallback: same provider + same client. Only a contradicting identity
+ # (both sides carry claims and they differ from every root row) blocks it.
+ if p_ident:
+ for i in candidates:
+ if not _oauth_identity(root_rows[i]):
+ return i
+ return None
+ return candidates[0]
+
+
+def _adopt_oauth_material(target: Dict[str, Any], winner: Dict[str, Any]) -> Dict[str, Any]:
+ """Return *target* carrying *winner*'s token pair, status markers cleared."""
+ merged = dict(target)
+ for key in _OAUTH_TOKEN_FIELDS:
+ if winner.get(key) is not None:
+ merged[key] = winner[key]
+ else:
+ merged.pop(key, None)
+ for status_field in _POOL_STATUS_FIELDS:
+ merged[status_field] = None
+ return merged
+
+
+def _singleton_as_row(path: Path) -> Optional[Dict[str, Any]]:
+ """Read a ``.anthropic_oauth.json`` as a pool-row-shaped dict, or None."""
+ try:
+ data = json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, ValueError):
+ return None
+ if not isinstance(data, dict) or not str(data.get("accessToken") or "").strip():
+ return None
+ return {
+ "access_token": data.get("accessToken"),
+ "refresh_token": data.get("refreshToken"),
+ "expires_at_ms": data.get("expiresAt"),
+ }
+
+
+def heal_forked_single_use_oauth_grants(provider_id: str) -> Optional[Dict[str, Any]]:
+ """Consolidate a profile's forked copy of a single-use OAuth grant into root.
+
+ Runs only in profile mode for ``SINGLE_USE_REFRESH_POOL_PROVIDERS``.
+ Returns a summary ``{"adopted": bool, "stripped_ids": [...], "files": [...],
+ "providers_block": bool}`` when something was healed, else ``None``.
+ Never raises.
+ """
+ if provider_id not in SINGLE_USE_REFRESH_POOL_PROVIDERS:
+ return None
+ try:
+ return _heal_forked_single_use_oauth_grants(provider_id)
+ except Exception:
+ logger.debug("%s: forked-OAuth heal skipped", provider_id, exc_info=True)
+ return None
+
+
+def _heal_forked_single_use_oauth_grants(provider_id: str) -> Optional[Dict[str, Any]]:
+ root_path = _global_auth_file_path()
+ if root_path is None:
+ return None # classic mode: nothing to consolidate into
+ if os.environ.get("PYTEST_CURRENT_TEST"):
+ # Same seat belt as the write-through paths: never touch the real
+ # user's ~/.hermes/auth.json from a test that forgot to isolate HOME.
+ real_home_env = os.environ.get("HOME", "")
+ if real_home_env and _same_path(root_path, Path(real_home_env) / ".hermes" / "auth.json"):
+ return None
+ profile_path = _auth_file_path()
+ profile_home = profile_path.parent
+ root_home = root_path.parent
+ profile_singleton = profile_home / ".anthropic_oauth.json" if provider_id == "anthropic" else None
+
+ # Hot-path short-circuit: load_pool() runs per model call. Once this
+ # profile's store was verified clean for *provider_id*, skip the locked
+ # read-modify-write until the profile's own files change (mtime key).
+ def _stamp(p: Optional[Path]) -> Optional[int]:
+ try:
+ return p.stat().st_mtime_ns if p is not None else None
+ except OSError:
+ return None
+
+ fingerprint = (str(profile_path), _stamp(profile_path), _stamp(profile_singleton))
+ if _oauth_heal_clean_marks.get(provider_id) == fingerprint:
+ return None
+ if fingerprint[1] is None and fingerprint[2] is None:
+ _oauth_heal_clean_marks[provider_id] = fingerprint
+ return None
+
+ summary: Dict[str, Any] = {"adopted": False, "stripped_ids": [], "files": [], "providers_block": False}
+ log_bits: List[str] = []
+
+ # Lock order: active (profile) store first, then the root source store —
+ # the same order ``_provider_state_transaction`` uses.
+ with _auth_store_lock():
+ profile_store = _load_auth_store(profile_path) if profile_path.exists() else {"providers": {}}
+ with _auth_store_lock(target_path=root_path):
+ root_store = _load_auth_store(root_path) if root_path.exists() else {"providers": {}}
+ profile_changed = False
+ root_changed = False
+
+ p_pool = profile_store.get("credential_pool")
+ p_rows = p_pool.get(provider_id) if isinstance(p_pool, dict) else None
+ p_rows = p_rows if isinstance(p_rows, list) else []
+ r_pool = root_store.get("credential_pool")
+ r_rows = r_pool.get(provider_id) if isinstance(r_pool, dict) else None
+ r_rows = r_rows if isinstance(r_rows, list) else []
+ r_oauth = [r for r in r_rows if _is_oauth_pool_payload(r)]
+
+ root_singleton = root_home / ".anthropic_oauth.json" if provider_id == "anthropic" else None
+ root_singleton_row = (
+ _singleton_as_row(root_singleton)
+ if root_singleton is not None and root_singleton.exists() else None
+ )
+
+ # ── credential_pool rows ────────────────────────────────────
+ kept_rows: List[Any] = []
+ for row in p_rows:
+ if not _is_oauth_pool_payload(row):
+ kept_rows.append(row) # API keys are safe to duplicate
+ continue
+ match_idx = _find_root_counterpart(row, r_rows)
+ if match_idx is not None:
+ root_row = r_rows[match_idx]
+ if _oauth_freshness(row) > _oauth_freshness(root_row):
+ r_rows[match_idx] = _adopt_oauth_material(root_row, row)
+ root_changed = True
+ summary["adopted"] = True
+ summary["stripped_ids"].append(row.get("id"))
+ profile_changed = True
+ continue
+ # No root pool counterpart. Root's grant may live only in its
+ # .anthropic_oauth.json (the ``hermes auth`` PKCE shape); a
+ # profile hermes_pkce-family row is that grant's copy.
+ is_pkce = str(row.get("source") or "").endswith("hermes_pkce")
+ if is_pkce and root_singleton_row is not None and not r_oauth:
+ if _oauth_freshness(row) > _oauth_freshness(root_singleton_row):
+ root_singleton_row = _adopt_oauth_material(root_singleton_row, row)
+ summary["adopted"] = True
+ summary["stripped_ids"].append(row.get("id"))
+ profile_changed = True
+ continue
+ # Root holds no copy of this lineage (independent account, or
+ # root never had the grant): the profile's row may be the
+ # only surviving copy — leave it alone.
+ kept_rows.append(row)
+ if profile_changed and isinstance(p_pool, dict):
+ if kept_rows:
+ p_pool[provider_id] = kept_rows
+ else:
+ p_pool.pop(provider_id, None)
+
+ # ── providers. device-code blocks (Codex / xAI) ─────────
+ if provider_id in ("openai-codex", "xai-oauth"):
+ p_providers = profile_store.get("providers")
+ r_providers = root_store.get("providers")
+ if isinstance(p_providers, dict) and isinstance(r_providers, dict):
+ p_block = p_providers.get(provider_id)
+ r_block = r_providers.get(provider_id)
+ else:
+ p_block = r_block = None
+ if isinstance(p_block, dict) and p_block and isinstance(r_block, dict) and r_block:
+ p_tokens = p_block.get("tokens") if isinstance(p_block.get("tokens"), dict) else {}
+ r_tokens = r_block.get("tokens") if isinstance(r_block.get("tokens"), dict) else {}
+ p_flat = {**p_tokens, "last_refresh": p_block.get("last_refresh")}
+ r_flat = {**r_tokens, "last_refresh": r_block.get("last_refresh")}
+ p_ident, r_ident = _oauth_identity(p_flat), _oauth_identity(r_flat)
+ same_account = (p_ident == r_ident) if (p_ident and r_ident) else True
+ if same_account:
+ if _oauth_freshness(p_flat) > _oauth_freshness(r_flat):
+ r_providers[provider_id] = dict(p_block)
+ root_changed = True
+ summary["adopted"] = True
+ del p_providers[provider_id]
+ profile_changed = True
+ summary["providers_block"] = True
+
+ # ── profile-local .anthropic_oauth.json singleton ───────────
+ if profile_singleton is not None and profile_singleton.exists():
+ p_single = _singleton_as_row(profile_singleton)
+ root_has_grant = bool(r_oauth) or root_singleton_row is not None
+ if p_single is not None and root_has_grant:
+ if root_singleton_row is not None:
+ if _oauth_freshness(p_single) > _oauth_freshness(root_singleton_row):
+ root_singleton_row = _adopt_oauth_material(root_singleton_row, p_single)
+ summary["adopted"] = True
+ else:
+ # Root only has pool rows: fold the singleton's pair
+ # into the freshest-matching root pkce row, if any.
+ idx = next(
+ (i for i, r in enumerate(r_rows)
+ if _is_oauth_pool_payload(r)
+ and str(r.get("source") or "").endswith("hermes_pkce")),
+ None,
+ )
+ if idx is not None and _oauth_freshness(p_single) > _oauth_freshness(r_rows[idx]):
+ r_rows[idx] = _adopt_oauth_material(r_rows[idx], p_single)
+ root_changed = True
+ summary["adopted"] = True
+ try:
+ profile_singleton.unlink()
+ summary["files"].append(profile_singleton.name)
+ except OSError:
+ logger.debug("could not remove %s", profile_singleton, exc_info=True)
+ # Otherwise root has NO grant for this provider (or the file
+ # is not a grant): the profile's singleton may be the only
+ # surviving copy — never delete it.
+
+ if not (profile_changed or root_changed or summary["adopted"]):
+ _oauth_heal_clean_marks[provider_id] = fingerprint
+ return None
+
+ if summary["adopted"] and root_singleton is not None and root_singleton_row is not None:
+ # Keep root's singleton and its ``hermes_pkce``-seeded pool row
+ # in step: root's next load_pool() re-seeds that row FROM the
+ # singleton file, so a stale file would resurrect the spent
+ # pair (and a stale row would be overwritten by a fresh file).
+ pkce_idx = next(
+ (i for i, r in enumerate(r_rows)
+ if _is_oauth_pool_payload(r) and r.get("source") == "hermes_pkce"),
+ None,
+ )
+ if pkce_idx is not None:
+ pkce_row = r_rows[pkce_idx]
+ if _oauth_freshness(pkce_row) > _oauth_freshness(root_singleton_row):
+ root_singleton_row = _adopt_oauth_material(root_singleton_row, pkce_row)
+ elif _oauth_freshness(root_singleton_row) > _oauth_freshness(pkce_row):
+ r_rows[pkce_idx] = _adopt_oauth_material(pkce_row, root_singleton_row)
+ root_changed = True
+
+ if root_changed:
+ if isinstance(r_pool, dict):
+ r_pool[provider_id] = r_rows
+ else:
+ root_store["credential_pool"] = {provider_id: r_rows}
+ _save_auth_store(root_store, target_path=root_path)
+ if summary["adopted"] and root_singleton is not None and root_singleton_row is not None:
+ from agent.anthropic_credentials import _write_hermes_oauth_credentials
+ _write_hermes_oauth_credentials(
+ root_singleton_row.get("access_token") or "",
+ root_singleton_row.get("refresh_token"),
+ root_singleton_row.get("expires_at_ms"),
+ target=root_singleton,
+ )
+ if profile_changed and profile_path.exists():
+ _save_auth_store(profile_store, target_path=profile_path)
+
+ if summary["stripped_ids"]:
+ log_bits.append(f"pool rows {summary['stripped_ids']}")
+ if summary["providers_block"]:
+ log_bits.append(f"providers.{provider_id} block")
+ if summary["files"]:
+ log_bits.append(", ".join(summary["files"]))
+ verdict = (
+ "profile copy was the live pair; root updated"
+ if summary["adopted"] else "root copy already newest; profile copy dropped"
+ )
+ message = (
+ f"profile {profile_home.name}: consolidated forked {provider_id} OAuth grant "
+ f"({'; '.join(log_bits) or 'no-op'}) into the root grant — {verdict}; "
+ f"this profile now borrows the root grant (#100339)"
+ )
+ logger.info(message)
+ _oauth_heal_notices.append(message)
+ return summary
+
+
def read_credential_pool(provider_id: Optional[str] = None) -> Dict[str, Any]:
"""Return the persisted credential pool, or one provider slice.
@@ -4051,6 +4549,32 @@ def _recover_codex_tokens_from_cli(reason: str) -> Optional[Dict[str, str]]:
return dict(imported)
+def _codex_http_client(**kwargs: Any) -> "httpx.Client":
+ """Build an ``httpx.Client`` for Codex OAuth/probe endpoints with racing.
+
+ Same broken-IPv6 failure mode as the chat transport (#13834): a host that
+ advertises AAAA records but blackholes IPv6 makes each serial connect
+ attempt eat the full connect timeout before IPv4 is tried, so token
+ refresh / device login / usage probes time out where the official Codex
+ CLI (which races families per RFC 8305) works. Install the same
+ Happy-Eyeballs sync backend #94388 added for the chat transport.
+
+ Best-effort: if the racing backend can't be installed (unexpected
+ httpx/httpcore internals, mocked client in tests), the client still works
+ with the default serial connect behavior. Proxy-backed transports are
+ intentionally left on the default backend (the TCP connect goes to the
+ proxy, not to auth.openai.com/chatgpt.com).
+ """
+ client = httpx.Client(**kwargs)
+ try:
+ from agent.process_bootstrap import enable_happy_eyeballs_on_client
+
+ enable_happy_eyeballs_on_client(client)
+ except Exception:
+ pass
+ return client
+
+
def refresh_codex_oauth_pure(
access_token: str,
refresh_token: str,
@@ -4068,7 +4592,7 @@ def refresh_codex_oauth_pure(
)
timeout = httpx.Timeout(max(5.0, float(timeout_seconds)))
- with httpx.Client(
+ with _codex_http_client(
timeout=timeout,
headers={
"Accept": "application/json",
@@ -4517,7 +5041,7 @@ def _probe_codex_quota_restored(
)
if isinstance(account_id, str) and account_id.strip():
headers["ChatGPT-Account-Id"] = account_id.strip()
- with httpx.Client(timeout=10.0) as client:
+ with _codex_http_client(timeout=10.0) as client:
response = client.get(_codex_usage_probe_url(base_url), headers=headers)
if response.status_code == 200:
payload = response.json() or {}
@@ -7317,8 +7841,73 @@ def get_api_key_provider_status(provider_id: str) -> Dict[str, Any]:
}
+def _external_process_auth_evidence(provider_id: str) -> tuple[bool, Optional[str]]:
+ """Best-effort POSITIVE evidence that an external-process provider's CLI
+ is authenticated.
+
+ Returns ``(verified, source)``. ``verified`` is only ever True on hard
+ evidence (a supported env token, or a known on-disk credential store).
+ False means "not verifiable from here", NOT "signed out" — the Copilot
+ CLI may hold its session in an OS keychain Hermes can't read. Callers
+ must therefore treat False as unknown, never as proof of absence.
+
+ Deliberately subprocess-free: this runs from status endpoints and pickers,
+ and spawning ``gh auth token`` there re-creates the cold-start stall
+ (#60800) that copilot_auth.py works to avoid.
+ """
+ if provider_id != "copilot-acp":
+ return False, None
+ # 1. Supported env tokens — the same vars the Copilot CLI itself honors.
+ try:
+ from hermes_cli.copilot_auth import COPILOT_ENV_VARS, validate_copilot_token
+ for env_var in COPILOT_ENV_VARS:
+ val = os.getenv(env_var, "").strip()
+ if val and validate_copilot_token(val)[0]:
+ return True, f"env: {env_var}"
+ except Exception as exc:
+ logger.debug("copilot-acp env token evidence check failed: %s", exc)
+ # 2. The Copilot CLI's own plaintext token store (~/.copilot/config.json,
+ # written by `copilot login` when no OS keychain is available). The file
+ # is JSONC — strip //-comment lines before parsing.
+ try:
+ cli_config = os.path.expanduser("~/.copilot/config.json")
+ if os.path.isfile(cli_config):
+ with open(cli_config, "r", encoding="utf-8", errors="ignore") as fh:
+ raw = "\n".join(
+ line for line in fh.read().splitlines()
+ if not line.lstrip().startswith("//")
+ )
+ data = json.loads(raw) if raw.strip() else {}
+ tokens = data.get("copilotTokens")
+ if isinstance(tokens, dict) and any(
+ isinstance(v, str) and v.strip() for v in tokens.values()
+ ):
+ return True, "~/.copilot/config.json"
+ except Exception as exc:
+ logger.debug("copilot-acp CLI config evidence check failed: %s", exc)
+ # 3. Known on-disk GitHub Copilot credential stores (the same locations
+ # models.py already fingerprints as external credential files).
+ for cred_path in (
+ "~/.config/github-copilot/hosts.json",
+ "~/.config/github-copilot/apps.json",
+ ):
+ try:
+ expanded = os.path.expanduser(cred_path)
+ if os.path.isfile(expanded) and os.path.getsize(expanded) > 2:
+ return True, cred_path
+ except OSError:
+ continue
+ return False, None
+
+
def get_external_process_provider_status(provider_id: str) -> Dict[str, Any]:
- """Status snapshot for providers that run a local subprocess."""
+ """Status snapshot for providers that run a local subprocess.
+
+ ``configured``/``logged_in`` stay structural (the executable resolves or a
+ TCP endpoint is set) because the spawned subprocess owns its real auth.
+ ``auth_verified``/``auth_source`` carry positive credential evidence when
+ Hermes can actually see some — absence of evidence is not absence of auth.
+ """
pconfig = PROVIDER_REGISTRY.get(provider_id)
if not pconfig or pconfig.auth_type != "external_process":
return {"configured": False}
@@ -7335,6 +7924,7 @@ def get_external_process_provider_status(provider_id: str) -> Dict[str, Any]:
base_url = pconfig.inference_base_url
resolved_command = shutil.which(command) if command else None
+ auth_verified, auth_source = _external_process_auth_evidence(provider_id)
return {
"configured": bool(resolved_command or base_url.startswith("acp+tcp://")),
"provider": provider_id,
@@ -7344,6 +7934,8 @@ def get_external_process_provider_status(provider_id: str) -> Dict[str, Any]:
"resolved_command": resolved_command,
"base_url": base_url,
"logged_in": bool(resolved_command or base_url.startswith("acp+tcp://")),
+ "auth_verified": auth_verified,
+ "auth_source": auth_source,
}
@@ -7364,12 +7956,16 @@ def get_auth_status(provider_id: Optional[str] = None) -> Dict[str, Any]:
return get_qwen_auth_status()
if target == "minimax-oauth":
return get_minimax_oauth_auth_status()
- if target == "copilot-acp":
- return get_external_process_provider_status(target)
if target == "azure-foundry":
return _get_azure_foundry_auth_status()
- # API-key providers
pconfig = PROVIDER_REGISTRY.get(target)
+ # External-process providers (copilot-acp today; kiro/devin/junie-style ACP
+ # backends tomorrow) — dispatch on auth_type, not a hardcoded slug, so every
+ # provider of this class gets a real status instead of the
+ # ``{"logged_in": False}`` fallthrough.
+ if pconfig and pconfig.auth_type == "external_process":
+ return get_external_process_provider_status(target)
+ # API-key providers
if pconfig and pconfig.auth_type == "api_key":
return get_api_key_provider_status(target)
# AWS SDK providers (Bedrock) — check via boto3 credential chain
@@ -8424,7 +9020,7 @@ def _codex_device_code_login() -> Dict[str, Any]:
max_attempts = 4
for attempt in range(1, max_attempts + 1):
try:
- with httpx.Client(timeout=httpx.Timeout(15.0)) as client:
+ with _codex_http_client(timeout=httpx.Timeout(15.0)) as client:
resp = client.post(
f"{issuer}/api/accounts/deviceauth/usercode",
json={"client_id": client_id},
@@ -8499,7 +9095,7 @@ def _codex_device_code_login() -> Dict[str, Any]:
code_resp = None
try:
- with httpx.Client(timeout=httpx.Timeout(15.0)) as client:
+ with _codex_http_client(timeout=httpx.Timeout(15.0)) as client:
while _time.monotonic() - start < max_wait:
_time.sleep(poll_interval)
poll_resp = client.post(
@@ -8540,7 +9136,7 @@ def _codex_device_code_login() -> Dict[str, Any]:
)
try:
- with httpx.Client(timeout=httpx.Timeout(15.0)) as client:
+ with _codex_http_client(timeout=httpx.Timeout(15.0)) as client:
token_resp = client.post(
CODEX_OAUTH_TOKEN_URL,
data={
@@ -9434,6 +10030,7 @@ def _login_nous(args, pconfig: ProviderConfig) -> None:
from hermes_cli.models import (
get_curated_nous_model_ids, get_pricing_for_provider,
check_nous_free_tier, partition_nous_models_by_tier,
+ nous_policy_allowed_ids, restrict_to_nous_policy,
union_with_portal_free_recommendations,
union_with_portal_paid_recommendations,
)
@@ -9448,6 +10045,10 @@ def _login_nous(args, pconfig: ProviderConfig) -> None:
# purchases are reflected immediately.
free_tier = check_nous_free_tier(force_fresh=True)
_portal_for_recs = auth_state.get("portal_base_url", "")
+ # Narrow before the tier split, so a rescued id still has to
+ # pass the free/paid predicate.
+ _policy_allowed = nous_policy_allowed_ids()
+ _policy_narrowed = False
if free_tier:
try:
from hermes_cli.nous_account import (
@@ -9473,6 +10074,11 @@ def _login_nous(args, pconfig: ProviderConfig) -> None:
model_ids, pricing = union_with_portal_free_recommendations(
model_ids, pricing, _portal_for_recs,
)
+ _before_policy = model_ids
+ model_ids = restrict_to_nous_policy(
+ model_ids, _policy_allowed, rescue_empty=True,
+ )
+ _policy_narrowed = model_ids != _before_policy
model_ids, unavailable_models = partition_nous_models_by_tier(
model_ids, pricing, free_tier=True,
)
@@ -9484,8 +10090,18 @@ def _login_nous(args, pconfig: ProviderConfig) -> None:
model_ids, pricing = union_with_portal_paid_recommendations(
model_ids, pricing, _portal_for_recs,
)
+ _before_policy = model_ids
+ model_ids = restrict_to_nous_policy(
+ model_ids, _policy_allowed, rescue_empty=True,
+ )
+ _policy_narrowed = model_ids != _before_policy
_portal = auth_state.get("portal_base_url", "")
if model_ids:
+ from hermes_cli.nous_account import nous_policy_notice
+
+ _policy_notice = nous_policy_notice(removed=_policy_narrowed)
+ if _policy_notice:
+ print(_policy_notice)
print(f"Showing {len(model_ids)} curated models — use \"Enter custom model name\" for others.")
selected_model = _prompt_model_selection(
model_ids, pricing=pricing,
diff --git a/hermes_cli/auth_commands.py b/hermes_cli/auth_commands.py
index 954c173cd2..3699032885 100644
--- a/hermes_cli/auth_commands.py
+++ b/hermes_cli/auth_commands.py
@@ -557,6 +557,13 @@ def auth_list_command(args) -> None:
source = _display_source(entry.source)
print(f" #{idx} {entry.label:<20} {entry.auth_type:<7} {source}{status} {marker}".rstrip())
print()
+ _print_oauth_heal_notices()
+
+
+def _print_oauth_heal_notices() -> None:
+ """Tell the user when load_pool() just consolidated a forked OAuth grant."""
+ for note in auth_mod.consume_oauth_heal_notices():
+ print(f"note: {note}")
def auth_remove_command(args) -> None:
@@ -608,7 +615,12 @@ def auth_status_command(args) -> None:
provider = _normalize_provider(getattr(args, "provider", "") or "")
if not provider:
raise SystemExit("Provider is required. Example: `hermes auth status spotify`.")
+ if provider in auth_mod.SINGLE_USE_REFRESH_POOL_PROVIDERS:
+ # load_pool() runs the forked-grant heal (#100339); do it before the
+ # status read so the report reflects the consolidated grant.
+ load_pool(provider)
status = auth_mod.get_auth_status(provider)
+ _print_oauth_heal_notices()
if not status.get("logged_in"):
reason = status.get("error")
if reason:
diff --git a/hermes_cli/backup.py b/hermes_cli/backup.py
index bed68e2a9d..1f49582a2e 100644
--- a/hermes_cli/backup.py
+++ b/hermes_cli/backup.py
@@ -107,6 +107,34 @@ _EXCLUDED_DIRS = {
".ruff_cache",
}
+# Hermes-managed runtime downloads that only exist at the top of a profile
+# home: local GGUF models, llama.cpp runtime binaries, and the managed Node
+# installation. All of them are re-downloaded on demand (model catalog,
+# runtime bootstrap, node installer) and routinely reach tens to hundreds of
+# GB, so zipping them turns a backup into an hours-long compress of
+# incompressible weights (the "backup stuck at N files" symptom). Matched
+# ONLY at the root of HERMES_HOME and at ``profiles//`` — a deeper
+# directory that happens to share one of these names (a skill's ``models/``,
+# a user checkout) is user data and stays in the backup.
+_EXCLUDED_ROOT_DIRS = {
+ "models",
+ "runtimes",
+ "node",
+}
+
+
+def _in_excluded_root_dir(rel_path: Path) -> bool:
+ """True when *rel_path* (relative to HERMES_HOME) is, or sits inside, a
+ Hermes-managed runtime tree at the top of a profile home."""
+ parts = rel_path.parts
+ if not parts:
+ return False
+ if parts[0] in _EXCLUDED_ROOT_DIRS:
+ return True
+ # Named profiles are profile homes too: profiles//models etc.
+ return len(parts) >= 3 and parts[0] == "profiles" and parts[2] in _EXCLUDED_ROOT_DIRS
+
+
# File-name suffixes to skip
_EXCLUDED_SUFFIXES = (
".pyc",
@@ -128,6 +156,16 @@ _EXCLUDED_NAMES = {
"cron.pid",
}
+# File-name prefixes to skip. The desktop updater's pre-flight drops
+# ``state.db.pre-update-emergency-.bak`` at the HERMES_HOME root
+# (apps/desktop/electron/main.ts preflightStateDb) — a backup artifact in
+# the same class as ``backups/`` and ``state-snapshots/``, so a full backup
+# must not re-ship it. Matched by prefix because the name carries a
+# timestamp; a plain ``.bak`` suffix rule would drop user files.
+_EXCLUDED_PREFIXES = (
+ "state.db.pre-update-emergency-",
+)
+
# File names that ``hermes import`` must never overwrite, matched by basename so
# they're caught for the root profile (``gateway_state.json``) and for named
# profiles alike (``profiles//gateway_state.json``).
@@ -335,6 +373,9 @@ def _should_exclude(rel_path: Path) -> bool:
"""Return True if *rel_path* (relative to hermes root) should be skipped."""
parts = rel_path.parts
+ if _in_excluded_root_dir(rel_path):
+ return True
+
for part in parts:
if part not in _EXCLUDED_DIRS:
continue
@@ -350,6 +391,9 @@ def _should_exclude(rel_path: Path) -> bool:
if name in _EXCLUDED_NAMES:
return True
+ if name.startswith(_EXCLUDED_PREFIXES):
+ return True
+
if name.endswith(_EXCLUDED_SUFFIXES):
return True
@@ -372,6 +416,48 @@ def _should_skip_backup_file(abs_path: Path, rel_path: Path, out_path: Path) ->
return False
+def _iter_backup_files(
+ hermes_root: Path,
+ out_path: Path,
+ skipped_dirs: Optional[set] = None,
+):
+ """Yield ``(abs_path, rel_path)`` for every file a full backup should hold.
+
+ The one owner of the backup walk policy: directory pruning (so os.walk
+ never descends a multi-GB excluded tree), the root-only ``hermes-agent``
+ carve-out, profile-home-root runtime trees, and the per-file exclusion
+ rules — shared by the manual ``hermes backup`` path and the automatic
+ pre-update/pre-migration path so the two can never drift.
+
+ ``skipped_dirs``, when given, collects pruned directories (root-relative,
+ as strings) for the end-of-run summary.
+ """
+ for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
+ rel_dir = Path(dirpath).relative_to(hermes_root)
+
+ # ``hermes-agent`` is only pruned at the root level; nested dirs
+ # with the same name (e.g. in skills/) must be preserved. Managed
+ # runtime trees (models/, runtimes/, node/) are pruned only at a
+ # profile-home root — see _EXCLUDED_ROOT_DIRS.
+ is_root = rel_dir == Path(".")
+ orig_dirnames = dirnames[:]
+ dirnames[:] = [
+ d for d in dirnames
+ if (d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root))
+ and not _in_excluded_root_dir(rel_dir / d)
+ ]
+ if skipped_dirs is not None:
+ for removed in set(orig_dirnames) - set(dirnames):
+ skipped_dirs.add(str(rel_dir / removed))
+
+ for fname in filenames:
+ rel = rel_dir / fname
+ fpath = hermes_root / rel
+ if _should_skip_backup_file(fpath, rel, out_path):
+ continue
+ yield fpath, rel
+
+
# ---------------------------------------------------------------------------
# SQLite safe copy
# ---------------------------------------------------------------------------
@@ -869,33 +955,10 @@ def _run_backup_locked(args, hermes_root: Path) -> None:
scan_started = time.monotonic()
logger.info("backup phase=scan status=started")
print(f"Scanning {display_hermes_home()} ...")
- files_to_add: list[tuple[Path, Path]] = [] # (absolute, relative)
- skipped_dirs = set()
-
- for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
- dp = Path(dirpath)
- rel_dir = dp.relative_to(hermes_root)
-
- # Prune excluded directories in-place so os.walk doesn't descend
- # ``hermes-agent`` is only pruned at the root level; nested dirs
- # with the same name (e.g. in skills/) must be preserved.
- is_root = rel_dir == Path(".")
- orig_dirnames = dirnames[:]
- dirnames[:] = [
- d for d in dirnames
- if d not in _EXCLUDED_DIRS or (d == "hermes-agent" and not is_root)
- ]
- for removed in set(orig_dirnames) - set(dirnames):
- skipped_dirs.add(str(rel_dir / removed))
-
- for fname in filenames:
- fpath = dp / fname
- rel = fpath.relative_to(hermes_root)
-
- if _should_skip_backup_file(fpath, rel, out_path):
- continue
-
- files_to_add.append((fpath, rel))
+ skipped_dirs: set = set()
+ files_to_add: list[tuple[Path, Path]] = list(
+ _iter_backup_files(hermes_root, out_path, skipped_dirs)
+ )
# External memory-provider state (e.g. ~/.honcho, ~/.hindsight) lives
# outside HERMES_HOME, so the walk above never sees it. Ask the active
@@ -2328,24 +2391,8 @@ def _write_full_zip_backup_locked(out_path: Path, hermes_root: Path) -> Optional
"""
scan_started = time.monotonic()
logger.info("automatic backup phase=scan status=started")
- files_to_add: list[tuple[Path, Path]] = []
try:
- for dirpath, dirnames, filenames in os.walk(hermes_root, followlinks=False):
- dp = Path(dirpath)
- # Prune excluded directories in-place so os.walk doesn't descend
- dirnames[:] = [d for d in dirnames if d not in _EXCLUDED_DIRS]
-
- for fname in filenames:
- fpath = dp / fname
- try:
- rel = fpath.relative_to(hermes_root)
- except ValueError:
- continue
-
- if _should_skip_backup_file(fpath, rel, out_path):
- continue
-
- files_to_add.append((fpath, rel))
+ files_to_add = list(_iter_backup_files(hermes_root, out_path))
except OSError as exc:
logger.warning("Full-zip backup: walk failed: %s", exc)
return None
diff --git a/hermes_cli/callbacks.py b/hermes_cli/callbacks.py
index aad0542d28..903bc6709b 100644
--- a/hermes_cli/callbacks.py
+++ b/hermes_cli/callbacks.py
@@ -120,6 +120,8 @@ def prompt_for_secret(cli, var_name: str, prompt: str, metadata=None) -> dict:
"response_queue": response_queue,
}
cli._secret_deadline = _time.monotonic() + timeout
+ if hasattr(cli, "_ring_bell"):
+ cli._ring_bell(prompt=True, context=f"secret needed ({var_name})")
# Avoid storing stale draft input as the secret when Enter is pressed.
if hasattr(cli, "_clear_secret_input_buffer"):
try:
diff --git a/hermes_cli/cli_agent_setup_mixin.py b/hermes_cli/cli_agent_setup_mixin.py
index 09aed5ed3e..46eae97b0d 100644
--- a/hermes_cli/cli_agent_setup_mixin.py
+++ b/hermes_cli/cli_agent_setup_mixin.py
@@ -353,12 +353,18 @@ class CLIAgentSetupMixin:
}
service_tier = getattr(self, "service_tier", None)
- if not service_tier:
+ if service_tier != "priority":
+ # None (normal) or auto/cold — the bounded window is applied per
+ # request by agent.fast_mode, not pinned into request_overrides.
route["request_overrides"] = None
return route
try:
- overrides = resolve_fast_mode_overrides(route["model"])
+ overrides = resolve_fast_mode_overrides(
+ route["model"],
+ provider=runtime["provider"],
+ base_url=runtime["base_url"],
+ )
except Exception:
overrides = None
route["request_overrides"] = overrides
@@ -646,29 +652,16 @@ class CLIAgentSetupMixin:
if not self._session_db:
return None
from hermes_state import (
- SessionExportTooLargeError,
SessionResumeTooLargeError,
- resolved_max_resume_messages,
)
try:
+ safety_check = getattr(self._session_db, "assert_resume_safe", None)
+ if not callable(safety_check):
+ return None
if tip_only:
- tip_check = getattr(self._session_db, "assert_export_safe", None)
- if not callable(tip_check):
- return None
- limit = resolved_max_resume_messages()
- if limit <= 0:
- return None
- try:
- tip_check(self.session_id, max_messages=limit)
- except SessionExportTooLargeError as exc:
- raise SessionResumeTooLargeError(
- exc.message_count, limit, scope="in its tip segment"
- ) from exc
+ safety_check(self.session_id, tip_only=True)
else:
- safety_check = getattr(self._session_db, "assert_resume_safe", None)
- if not callable(safety_check):
- return None
safety_check(self.session_id)
except SessionResumeTooLargeError as exc:
return str(exc)
diff --git a/hermes_cli/cli_commands_mixin.py b/hermes_cli/cli_commands_mixin.py
index 250949c94d..b78ff6157c 100644
--- a/hermes_cli/cli_commands_mixin.py
+++ b/hermes_cli/cli_commands_mixin.py
@@ -1986,7 +1986,13 @@ class CLICommandsMixin:
print(f" Skills: {', '.join(job['skills'])}")
print(f" Prompt: {job.get('prompt_preview', '')}")
if job.get("last_run_at"):
- print(f" Last run: {job['last_run_at']} ({job.get('last_status', '?')})")
+ status = job.get("last_status") or "?"
+ # delivery_failed: the agent ran fine but the output never
+ # reached the target — name the delivery reason, which
+ # lives in last_delivery_error (last_error is None).
+ if status == "delivery_failed" and job.get("last_delivery_error"):
+ status = f"delivery_failed: {job['last_delivery_error']}"
+ print(f" Last run: {job['last_run_at']} ({status})")
print()
return
@@ -3013,6 +3019,7 @@ class CLICommandsMixin:
review_memory=True,
review_skills=review_skills,
focus=focus or None,
+ explicit=True,
)
except Exception as exc:
_cprint(f" /refine failed to start: {exc}")
@@ -3982,9 +3989,9 @@ class CLICommandsMixin:
parts = cmd.strip().split(maxsplit=1)
if len(parts) < 2 or parts[1].strip().lower() == "status":
- status = "fast" if self.service_tier == "priority" else "normal"
+ status = {"priority": "fast", None: "normal"}.get(self.service_tier, self.service_tier)
_cprint(f" {_ACCENT}{feature_name}: {status}{_RST}")
- _cprint(f" {_DIM}Usage: /fast [normal|fast|status] [--global]{_RST}")
+ _cprint(f" {_DIM}Usage: /fast [normal|fast|auto|cold|status] [--global]{_RST}")
return
arg_tokens = parts[1].strip().lower().split()
@@ -4002,9 +4009,13 @@ class CLICommandsMixin:
self.service_tier = None
saved_value = "normal"
label = "NORMAL"
+ elif arg in {"auto", "cold"}:
+ self.service_tier = arg
+ saved_value = arg
+ label = arg.upper()
else:
_cprint(f" {_DIM}(._.) Unknown argument: {arg}{_RST}")
- _cprint(f" {_DIM}Usage: /fast [normal|fast|status] [--global]{_RST}")
+ _cprint(f" {_DIM}Usage: /fast [normal|fast|auto|cold|status] [--global]{_RST}")
return
self.agent = None # Force agent re-init with new service-tier config
diff --git a/hermes_cli/commands.py b/hermes_cli/commands.py
index 23af46886d..f2c2ea7faf 100644
--- a/hermes_cli/commands.py
+++ b/hermes_cli/commands.py
@@ -297,9 +297,9 @@ COMMAND_REGISTRY: list[CommandDef] = [
args_hint="[level|show|hide|full|clamp] [--global]",
subcommands=("none", "minimal", "low", "medium", "high", "xhigh", "max", "ultra", "show", "hide", "on", "off", "full", "clamp", "--global"),
desktop="advanced"),
- CommandDef("fast", "Toggle fast mode — OpenAI Priority Processing / Anthropic Fast Mode (Normal/Fast)", "Configuration",
- args_hint="[normal|fast|status] [--global]",
- subcommands=("normal", "fast", "status", "on", "off", "--global"),
+ CommandDef("fast", "Fast mode — OpenAI Priority Processing / Anthropic Fast Mode (normal/fast/auto/cold)", "Configuration",
+ args_hint="[normal|fast|auto|cold|status] [--global]",
+ subcommands=("normal", "fast", "auto", "cold", "status", "on", "off", "--global"),
desktop="advanced"),
CommandDef("skin", "Show or change the display skin/theme", "Configuration",
cli_only=True, args_hint="[name]", argument_mode="options"),
diff --git a/hermes_cli/config.py b/hermes_cli/config.py
index a3c934d050..add9141382 100644
--- a/hermes_cli/config.py
+++ b/hermes_cli/config.py
@@ -1111,6 +1111,7 @@ ENV_VARS_BY_VERSION: Dict[int, List[str]] = {
4: ["VOICE_TOOLS_OPENAI_KEY", "ELEVENLABS_API_KEY"],
5: ["WHATSAPP_ENABLED", "WHATSAPP_MODE", "WHATSAPP_ALLOWED_USERS",
"SLACK_BOT_TOKEN", "SLACK_APP_TOKEN", "SLACK_ALLOWED_USERS"],
+ 10: ["TAVILY_API_KEY"],
11: ["TERMINAL_MODAL_MODE"],
}
@@ -1456,7 +1457,7 @@ def _is_env_config_key(key: str) -> bool:
'OPENROUTER_API_KEY', 'OPENAI_API_KEY', 'ANTHROPIC_API_KEY', 'VOICE_TOOLS_OPENAI_KEY',
'EXA_API_KEY', 'PARALLEL_API_KEY', 'FIRECRAWL_API_KEY', 'FIRECRAWL_API_URL',
'FIRECRAWL_GATEWAY_URL', 'TOOL_GATEWAY_DOMAIN', 'TOOL_GATEWAY_SCHEME',
- 'TOOL_GATEWAY_USER_TOKEN',
+ 'TOOL_GATEWAY_USER_TOKEN', 'TAVILY_API_KEY', 'API_SERVER_KEY',
'BROWSERBASE_API_KEY', 'BROWSERBASE_PROJECT_ID', 'BROWSER_USE_API_KEY',
'FAL_KEY', 'TELEGRAM_BOT_TOKEN', 'DISCORD_BOT_TOKEN',
'TERMINAL_SSH_HOST', 'TERMINAL_SSH_USER', 'TERMINAL_SSH_KEY',
@@ -2526,6 +2527,32 @@ def validate_config_structure(config: Optional[Dict[str, Any]] = None) -> List["
f"Move '{key}' under the appropriate section",
))
+ # ── web backends that no longer ship in-tree ─────────────────────────
+ # A stale selection (e.g. web.backend: tavily after the #99199 removal)
+ # otherwise fails only at the first web_search/web_extract call, with a
+ # generic "no registered provider" error. Warn at startup instead.
+ web_cfg = config.get("web")
+ if isinstance(web_cfg, dict):
+ try:
+ from tools.tool_backend_helpers import removed_backend_note
+ except Exception:
+ removed_backend_note = None
+ if removed_backend_note is not None:
+ seen: set = set()
+ for _key in ("backend", "search_backend", "extract_backend"):
+ _val = str(web_cfg.get(_key) or "").strip().lower()
+ if not _val or _val in seen:
+ continue
+ seen.add(_val)
+ note = removed_backend_note("web", _val)
+ if note:
+ issues.append(ConfigIssue(
+ "warning",
+ f"web.{_key} is set to '{_val}', but {note} — "
+ "web_search/web_extract will fail until it is changed",
+ "Run 'hermes tools' and pick a different Web Search & Extract provider",
+ ))
+
return issues
@@ -2958,13 +2985,36 @@ def _strip_dotted_keys(cfg: dict, dotted_keys: set) -> Tuple[dict, set]:
return cfg, stripped
+def _env_ref_lookup(name: str) -> Optional[str]:
+ """Resolve the env var behind a ``${VAR}`` / ``${env:VAR}`` config ref.
+
+ Outside a profile secret scope this is a plain ``os.environ`` read — the
+ default profile and every single-profile caller keep their legacy
+ behavior. Inside a scope (a multiplexed gateway turn, a secondary
+ profile's config load, a cron job) the read goes through
+ ``agent.secret_scope.get_secret`` so the ref resolves against *that*
+ profile's ``.env``: under multiplexing a miss is a miss, never another
+ profile's ``os.environ`` value (#84079 — every profile "had" the default
+ profile's ``${MATRIX_ACCESS_TOKEN}`` and fanned out). Same policy as
+ ``gateway.config._getenv`` and ``get_env_value``.
+ """
+ try:
+ from agent.secret_scope import current_secret_scope, get_secret as _get_secret
+ except Exception:
+ return os.environ.get(name)
+ if current_secret_scope() is None:
+ return os.environ.get(name)
+ return _get_secret(name)
+
+
def _env_expand_match(m: re.Match) -> str:
"""Expand one ``${...}`` config reference.
Two accepted shapes, matching what MCP server config already resolves
(``tools/mcp_tool.py::_env_ref_name``):
- * ``${VAR}`` — legacy bare name, resolved via ``os.environ``.
+ * ``${VAR}`` — legacy bare name, resolved via ``_env_ref_lookup``
+ (``os.environ``, or the active profile secret scope).
* ``${env:VAR}`` — Cursor-style SecretRef, same resolution after the
``env:`` prefix is stripped. Before this, the prefixed form worked in
MCP config but stayed a literal string in config.yaml — a confusing
@@ -2982,7 +3032,7 @@ def _env_expand_match(m: re.Match) -> str:
name = inner[len("env:"):].strip()
if not name:
return raw
- val = os.environ.get(name)
+ val = _env_ref_lookup(name)
if val is not None:
return val
logger.warning(
@@ -3003,7 +3053,8 @@ def _env_expand_match(m: re.Match) -> str:
)
return raw
# Legacy ``${VAR}`` — bare name.
- return os.environ.get(inner, raw)
+ val = _env_ref_lookup(inner)
+ return val if val is not None else raw
def _env_ref_var_name(ref: str) -> Optional[str]:
@@ -3056,7 +3107,7 @@ def _env_ref_snapshot(obj, snapshot=None):
for raw in re.findall(r"\${([^}]+)}", obj):
name = _env_ref_var_name(raw)
if name is not None:
- snapshot[name] = os.environ.get(name)
+ snapshot[name] = _env_ref_lookup(name)
elif isinstance(obj, dict):
for value in obj.values():
_env_ref_snapshot(value, snapshot)
@@ -4070,7 +4121,7 @@ def _load_config_impl(*, want_deepcopy: bool) -> Dict[str, Any]:
# pins unexpanded literals (e.g. auxiliary..api_key) for the
# life of the process (#58514).
env_snapshot = cached[5] if len(cached) > 5 else {}
- if all(os.environ.get(k) == v for k, v in env_snapshot.items()):
+ if all(_env_ref_lookup(k) == v for k, v in env_snapshot.items()):
return copy.deepcopy(cached[4]) if want_deepcopy else cached[4]
config = copy.deepcopy(DEFAULT_CONFIG)
@@ -4631,6 +4682,38 @@ def _env_line_defines_key(
) == _env_var_policy_name(key, is_windows=is_windows)
+def _publish_env_value(key: str, value: Optional[str]) -> None:
+ """Publish a just-persisted ``.env`` change to the live process.
+
+ ``save_env_value`` / ``remove_env_value`` already target the right file
+ (``get_env_path()`` honors the profile-home override), but the in-process
+ mirror historically went straight to ``os.environ``. Under a multiplexed
+ gateway a routed profile's write (e.g. a ``/pair`` grant mirrored into
+ ``DISCORD_ALLOWED_USERS``) would then land in the SHARED process env and
+ be visible to every other profile (#88441, #77490). In that case update
+ the installed scope mapping instead so same-turn reads see the change,
+ and leave ``os.environ`` alone. Every other caller keeps the legacy
+ ``os.environ`` publish.
+ """
+ try:
+ from agent.secret_scope import current_secret_scope, is_multiplex_active
+
+ scope = current_secret_scope() if is_multiplex_active() else None
+ except Exception:
+ scope = None
+ if scope is not None:
+ if isinstance(scope, dict):
+ if value is None:
+ scope.pop(key, None)
+ else:
+ scope[key] = value
+ return
+ if value is None:
+ os.environ.pop(key, None)
+ else:
+ os.environ[key] = value
+
+
def save_env_value(key: str, value: str):
"""Save or update a value in ~/.hermes/.env."""
if is_managed():
@@ -4719,7 +4802,7 @@ def save_env_value(key: str, value: str):
pass
raise
- os.environ[key] = value
+ _publish_env_value(key, value)
invalidate_env_cache()
@@ -4766,7 +4849,7 @@ def remove_env_value(key: str) -> bool:
raise ValueError(f"Invalid environment variable name: {key!r}")
env_path = get_env_path()
if not env_path.exists():
- os.environ.pop(key, None)
+ _publish_env_value(key, None)
return False
read_kw = {"encoding": "utf-8-sig", "errors": "replace"}
@@ -4810,7 +4893,7 @@ def remove_env_value(key: str) -> bool:
pass
raise
- os.environ.pop(key, None)
+ _publish_env_value(key, None)
invalidate_env_cache()
return found
@@ -5064,6 +5147,7 @@ def show_config():
("EXA_API_KEY", "Exa"),
("PARALLEL_API_KEY", "Parallel"),
("FIRECRAWL_API_KEY", "Firecrawl"),
+ ("TAVILY_API_KEY", "Tavily"),
("BROWSERBASE_API_KEY", "Browserbase"),
("BROWSER_USE_API_KEY", "Browser Use"),
("FAL_KEY", "FAL"),
@@ -5108,7 +5192,10 @@ def show_config():
_active_personality = display.get('personality') or 'none'
print(f" Personality: {_active_personality}")
print(f" Reasoning: {'on' if display.get('show_reasoning', True) else 'off'}")
- print(f" Bell: {'on' if display.get('bell_on_complete', False) else 'off'}")
+ print(
+ f" Bell: complete={'on' if display.get('bell_on_complete', False) else 'off'}, "
+ f"prompt={'on' if display.get('bell_on_prompt', False) else 'off'}"
+ )
ump = display.get('user_message_preview', {}) if isinstance(display.get('user_message_preview', {}), dict) else {}
ump_first = ump.get('first_lines', 2)
ump_last = ump.get('last_lines', 2)
@@ -5813,6 +5900,37 @@ def _coerce_float(value: str):
return f
+def _redirect_platform_display_key(key: str) -> tuple[str, Optional[str]]:
+ """Canonicalize ``platforms..`` → ``display.platforms..``.
+
+ Per-platform *display* settings (streaming, show_reasoning, tool_progress,
+ …) are resolved by the gateway from ``display.platforms..``
+ (``gateway/display_config.py::resolve_display_setting``), while the
+ top-level ``platforms.`` block holds only connection config (token,
+ enabled, reply_to_mode, extra, …). Before #71047 a write such as
+ ``hermes config set platforms.telegram.streaming false`` landed on a key
+ the gateway never reads: ``config get`` echoed the new value back while
+ the runtime kept the old ``display.platforms`` one — a silent no-op that
+ looks like a duplicated key to the user.
+
+ Only known display settings (``OVERRIDEABLE_KEYS``) are redirected so real
+ connection keys stay put. Returns ``(canonical_key, note_or_None)``.
+ The gateway import is guarded: the CLI must keep working where the
+ gateway package is not importable.
+ """
+ segs = _split_key_path(key)
+ if len(segs) != 3 or segs[0] != "platforms":
+ return key, None
+ try:
+ from gateway.display_config import OVERRIDEABLE_KEYS as _display_keys
+ except Exception:
+ return key, None
+ if segs[2] not in _display_keys:
+ return key, None
+ canonical = f"display.platforms.{segs[1]}.{segs[2]}"
+ return canonical, f" (note: per-platform display setting — saved as {canonical})"
+
+
def set_config_value(key: str, value: str, force: bool = False):
"""Set a configuration value.
@@ -5878,6 +5996,12 @@ def set_config_value(key: str, value: str, force: bool = False):
# bare success and left the user debugging behavior that never changed.
# Warn after the write so the user gets immediate feedback plus a
# "did you mean" hint, without blocking legitimate unknown keys.
+ # Per-platform display settings live under display.platforms (#71047,
+ # Problem A) — canonicalize BEFORE validation/coercion so the type-aware
+ # coercion and the unknown-key hint both see the path the runtime reads.
+ key, _redirect_note = _redirect_platform_display_key(key)
+ if _redirect_note:
+ print(_redirect_note)
is_known, suggestion = _validate_config_key(key)
# Otherwise it goes to config.yaml
@@ -6093,6 +6217,9 @@ def get_config_value(key: str, *, as_json: bool = False):
env_value = get_env_value(key.upper())
value = _MISSING if env_value is None else env_value
else:
+ # Mirror set_config_value: read the canonical display.platforms path
+ # so ``config get`` reports what the gateway resolves (#71047).
+ key, _ = _redirect_platform_display_key(key)
value = _get_nested(load_config(), key)
if value is _MISSING:
@@ -6138,6 +6265,10 @@ def unset_config_value(key: str):
# refuse-write); returns the mapping so we do not re-parse / collapse.
user_config = require_readable_config_before_write(config_path)
+ # Mirror set_config_value's display.platforms canonicalization (#71047).
+ key, _redirect_note = _redirect_platform_display_key(key)
+ if _redirect_note:
+ print(_redirect_note.replace("saved as", "resolved as"))
removed = _unset_nested(user_config, key)
# Keep .env in sync for keys that terminal_tool reads directly from env vars.
diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py
index ae73c74eb5..6cfc67692f 100644
--- a/hermes_cli/config_defaults.py
+++ b/hermes_cli/config_defaults.py
@@ -151,7 +151,10 @@ DEFAULT_CONFIG = {
# leaves the budget untouched.
"cost_threshold_usd": 0.25,
},
+ # Fast mode: "" / "normal" (off), "fast" (always), "auto" (first
+ # fast_auto_seconds of every turn), "cold" (first turn of a session only).
"service_tier": "",
+ "fast_auto_seconds": 60,
# Tool-use enforcement: injects system prompt guidance that tells the
# model to actually call tools instead of describing intended actions.
# Values: "auto" (default — applies to gpt/codex models), true/false
@@ -555,7 +558,7 @@ DEFAULT_CONFIG = {
"extract_backend": "", # per-capability override for web_extract (e.g. "native")
"extract_char_limit": 15000, # per-page char budget for web_extract; larger pages truncate + store full text in cache/web
# Keyless free-tier ring: with NO web backend configured or keyed,
- # web_search/web_extract rotate round-robin across five vendors'
+ # web_search/web_extract rotate round-robin across four vendors'
# public free tiers (exa, parallel, firecrawl, keenable),
# failing over to the next ring vendor on rate limits. Never
# pre-empts a configured or keyed backend. Set false to disable.
@@ -565,10 +568,11 @@ DEFAULT_CONFIG = {
# free-tier ring — the next call attempts the chosen backend again
# (no sticky failover). Off when keyless_fallback is false.
"keyless_rescue": True,
- # Per-provider tier selection for ring vendors with both a keyless
+ # Per-provider tier selection for vendors with both a keyless
# free endpoint and a keyed paid path (exa, parallel,
- # firecrawl, keenable). Set by the `hermes tools` picker's
- # "Free (keyless)" / "Paid (API key)" rows.
+ # firecrawl, keenable on the ring; tavily is opt-in keyless via
+ # `hermes tools`, not a ring member). Set by the `hermes tools`
+ # picker's "Free (keyless)" / "Paid (API key)" rows.
# free — always use the anonymous free endpoint (even with a key)
# paid — always use the keyed path (missing key = error; vendor
# is also excluded from the keyless ring)
@@ -820,6 +824,10 @@ DEFAULT_CONFIG = {
"tool_loop_guardrails": {
"warnings_enabled": True,
"hard_stop_enabled": False,
+ # Unattended gateway/cron platforms get hard stops by default (nobody
+ # is present to /stop a model that ignores loop warnings); interactive
+ # cli/tui/desktop/acp stay warning-only unless hard_stop_enabled.
+ "non_interactive_hard_stop_enabled": True,
"warn_after": {
"exact_failure": 2,
"same_tool_failure": 3,
@@ -956,6 +964,10 @@ DEFAULT_CONFIG = {
# waiting. Kept well under chat-transport idle timeouts
# (Telegram ~30s). On expiry the turn proceeds
# uncompressed — an availability boundary, not a failure.
+ # The detached worker keeps its commit admission when its
+ # commit is watermark-fenced, so the finished summary is
+ # adopted at the next safe boundary instead of being
+ # discarded (#97963 — thinking summary models).
"context_timeout_seconds": 120, # inactivity budget for in-agent compress_context
# (conversation loop, /compress, preflight, etc.).
# Same progress-aware semantics as hygiene_timeout_seconds:
@@ -1457,12 +1469,18 @@ DEFAULT_CONFIG = {
# Mirrors `hermes -c` muscle memory. Default off so existing
# users aren't surprised. HERMES_TUI_RESUME= always wins.
"tui_auto_resume_recent": False,
+ # When true (default), the Desktop app reopens the last chat (or
+ # last page) on cold start. Set false to always land on a fresh
+ # new chat. Also a switch in Desktop Settings → Appearance.
+ "resume_last_session": True,
# When true (default), `hermes --tui` drops a one-time hint
# ("subagents working · /agents to watch live") the first time a turn
# starts delegating, nudging the user toward the live spawn-tree
# dashboard. Set false to suppress the hint.
"tui_agents_nudge": True,
"bell_on_complete": False,
+ # Bell when a blocking prompt opens (clarify/approval/sudo/secret).
+ "bell_on_prompt": False,
# Stream the model's reasoning/thinking live before the response.
# Default ON: on thinking models the reasoning phase can run tens of
# seconds, and with this off the user stares at a spinner the whole
@@ -1733,6 +1751,15 @@ DEFAULT_CONFIG = {
# override for backward compatibility. 0 disables the reap
# (park forever).
"ws_orphan_reap_grace_s": 20.0,
+ # Activity-staleness threshold (seconds) gating the WS-orphan
+ # interrupt of a detached RUNNING turn (#98028/#100325). A
+ # client-absent turn is only interrupted once its agent activity
+ # clock (the same one the agent.turn_liveness watchdog samples —
+ # stamped by API waits, stream tokens, tool heartbeats) has been
+ # idle at least this long; an actively-working detached turn runs
+ # to completion. Default matches agent.turn_liveness.timeout_s.
+ # 0 restores the old interrupt-at-grace-regardless behavior.
+ "ws_orphan_activity_stale_s": 600.0,
# Startup sweep of session rows orphaned by a dead gateway process
# (#65194). The ws-orphan grace timer above is in-process, so a
# gateway restart (update, crash, systemd) leaves disconnected
@@ -2796,6 +2823,15 @@ DEFAULT_CONFIG = {
# Wrap delivered cron responses with a header (task name) and footer
# ("The agent cannot see this message"). Set to false for clean output.
"wrap_response": True,
+ # Delivery behaviour for cron output sent through a live gateway adapter.
+ "delivery": {
+ # Mark cron deliveries as FINAL notifications so the platform pushes
+ # them (Telegram's "important" notification mode otherwise sends
+ # every non-notify message with disable_notification=True, and users
+ # report the silent brief as "never delivered"). Set to false to
+ # restore silent (no-push) cron deliveries.
+ "notify": True,
+ },
# Make cron deliveries CONTINUABLE: a user can reply to a cron brief
# and the agent has it in context (no "what is Task #2?" amnesia).
# Default False preserves the historical isolation guarantee (cron
@@ -3080,10 +3116,12 @@ DEFAULT_CONFIG = {
"model_catalog": {
"enabled": True,
"url": "https://hermes-agent.nousresearch.com/docs/api/model-catalog.json",
- # Disk cache TTL in hours. Beyond this, the CLI refetches on the
- # next /model or `hermes model` invocation; network failures
- # silently fall back to the stale cache.
- "ttl_hours": 1,
+ # Disk cache TTL in minutes. The gateway refreshes the catalogs on
+ # this cadence in the background; the CLI refetches on the next
+ # /model or `hermes model` invocation once the cache is older than
+ # this. Network failures silently fall back to the stale cache.
+ # (Legacy `ttl_hours` is still honoured when set explicitly.)
+ "ttl_minutes": 20,
# Optional per-provider override URLs for third parties that want
# to self-host their own curation list using the same schema.
# Example:
@@ -3337,6 +3375,17 @@ DEFAULT_CONFIG = {
# adapter. ``0`` disables the cap. Default 128 MiB.
"max_inbound_media_bytes": 134217728,
+ # Whether gateway platform adapters let aiohttp read proxy settings
+ # (HTTP_PROXY / HTTPS_PROXY / NO_PROXY, plus SSL_CERT_FILE) from the
+ # process environment, and whether generic proxy env / the macOS
+ # system proxy are auto-detected for adapter clients. Set to false
+ # when the gateway inherits a proxy it must not use — e.g. a Windows
+ # Scheduled Task picking up a Clash/V2Ray HTTP_PROXY the interactive
+ # shell never sees, producing "Cannot connect to host 127.0.0.1:7890"
+ # poll loops (#48820). Explicit per-platform vars (DISCORD_PROXY,
+ # TELEGRAM_PROXY, ...) are still honored. One knob for every adapter.
+ "trust_env": True,
+
# When false (default), any file path the agent emits is delivered
# as a native attachment as long as it isn't under the credential /
# system-path denylist (/etc, /proc, ~/.ssh, ~/.aws, ~/.hermes/.env,
@@ -3433,13 +3482,18 @@ DEFAULT_CONFIG = {
# reports 384MB+ databases with 68K+ messages, which slows down FTS5
# inserts, /resume listing, and insights queries.
"sessions": {
- # When true, prune ended sessions inactive for retention_days once
+ # When true, prune ENDED sessions inactive for retention_days once
# per (roughly) min_interval_hours at CLI/gateway/cron startup.
# Activity is the latest message timestamp, falling back to creation
- # time for empty sessions. Active sessions are always preserved.
- # Default false: session history is valuable for search recall, and
- # silently deleting it could surprise users. Opt in explicitly.
- "auto_prune": False,
+ # time for empty sessions. Sessions that are still open, pinned, or
+ # mid-turn are never deleted — the only open rows the sweep touches
+ # are stale automation sessions (cron/kanban/subagent/one-shot CLI)
+ # whose process died without closing them; those are *closed*, not
+ # deleted, and get a further full retention window before removal.
+ # Default true since #54189: without it state.db grows without bound
+ # (multi-GB installs reported within weeks). Set false to keep every
+ # ended session forever.
+ "auto_prune": True,
# How many inactive days of ended-session history to keep. Matches
# the default of ``hermes sessions prune``.
"retention_days": 90,
@@ -3457,7 +3511,9 @@ DEFAULT_CONFIG = {
# subsequent INSERTs — so without VACUUM the file stays bloated
# even after pruning. VACUUM blocks writes for a few seconds per
# 100MB, so it only runs at startup, and only when prune deleted
- # ≥1 session.
+ # ≥1 session AND the reclaimable fraction of the file
+ # (PRAGMA freelist_count / page_count) exceeds 25% — a dense DB
+ # never pays for a full rewrite to reclaim a few MB (#54189).
"vacuum_after_prune": True,
# Minimum days between successful VACUUM rewrites. Pruning can still
# run on its normal cadence while SQLite reuses the freed pages.
@@ -3532,11 +3588,28 @@ DEFAULT_CONFIG = {
"profile_build": "ask",
},
- # Privacy-safe aggregate metrics written only to this profile's local
- # telemetry directory. Collection is opt-in and no remote sink exists.
+ # Privacy-safe aggregate metrics written to this profile's local telemetry
+ # directory. Collection is opt-in (``enabled``). Transmission to the Nous
+ # telemetry service is a SEPARATE opt-in (``send``) and is off by default;
+ # see docs/observability/relay-shared-metrics.md, Appendix A, for the
+ # consent, identity, rotation, retention, and deletion decisions.
"telemetry": {
"shared_metrics": {
"enabled": False,
+ # Transmit exported packages to the Nous telemetry service.
+ # Requires ``enabled``: it never switches collection on by itself,
+ # and ``send`` without ``enabled`` is logged as an error rather
+ # than silently doing nothing. A package is only sent when its
+ # whole period falls inside a recorded consent window, so data
+ # collected before consent — or while it was withdrawn — stays
+ # local.
+ "send": False,
+ # Ingest endpoint. Production by default; override for staging or
+ # a local test server. Deliberately NOT overridable by an
+ # environment variable: that would let an inherited value silently
+ # redirect telemetry a user consented to send to Nous. Non-HTTPS
+ # is refused unless the host is localhost.
+ "endpoint": "https://telemetry.nousresearch.com/v1/telemetry",
},
},
@@ -4002,8 +4075,31 @@ DEFAULT_CONFIG = {
"region": "global",
},
+ # Managed llama.cpp local runtime (see docs: user-guide/local-models).
+ # Hermes downloads official llama.cpp release binaries, then spawns and
+ # supervises one llama-server in router mode. Context sizing is policy,
+ # not preference: there are deliberately no context/VRAM knobs here.
+ "local_runtime": {
+ # Master switch for the managed runtime. Off = detection-only
+ # (Hermes still finds an external llama-server you run yourself).
+ "enabled": False,
+ # Pinned llama.cpp release tag (rolling bNNNN). Bumped by Hermes
+ # releases after the validation suite re-runs, not tracked live.
+ "tag": "b10679",
+ # Inference backend: auto = CUDA on NVIDIA, Metal on macOS, Vulkan on
+ # other GPUs, else CPU. Explicit values: cuda|metal|vulkan|hip|cpu.
+ "backend": "auto",
+ # Router process: how many models may be resident at once.
+ "models_max": 4,
+ # Port for the managed server. 0 = pick a free port at spawn.
+ "port": 0,
+ # Extra ports detection probes for an external llama-server, in
+ # addition to the default 8080.
+ "detect_ports": [],
+ },
+
# Config schema version - bump this when adding new required fields
- "_config_version": 39,
+ "_config_version": 40,
}
# Optional environment variables that enhance functionality
@@ -4504,6 +4600,14 @@ OPTIONAL_ENV_VARS = {
"category": "tool",
"advanced": True,
},
+ "TAVILY_API_KEY": {
+ "description": "Tavily API key for AI-native web search and extract (optional — keyless works when Tavily is selected)",
+ "prompt": "Tavily API key",
+ "url": "https://app.tavily.com/home",
+ "tools": ["web_search", "web_extract"],
+ "password": True,
+ "category": "tool",
+ },
"KEENABLE_API_KEY": {
"description": "Keenable API key for fast independent-index web search and page fetch (optional — keyless free tier works without it)",
"prompt": "Keenable API key",
diff --git a/hermes_cli/config_migrations.py b/hermes_cli/config_migrations.py
index 3357aaec16..cb075be302 100644
--- a/hermes_cli/config_migrations.py
+++ b/hermes_cli/config_migrations.py
@@ -863,6 +863,28 @@ def _migrate_to_39(results: Dict[str, Any], quiet: bool) -> None:
)
+def _migrate_to_40(results: Dict[str, Any], quiet: bool) -> None:
+ # ── Version 39 → 40: model_catalog.ttl_hours → ttl_minutes (default 20) ──
+ # The picker catalogs now refresh every 20 minutes (and the gateway
+ # refreshes them in the background on that cadence). Only the OLD default
+ # (ttl_hours: 1, written by the v25 migration) is dropped so the new
+ # default applies; any other explicit ttl_hours is a deliberate choice
+ # and stays honoured by the loader.
+ _c = _cfg()
+ read_raw_config = _c.read_raw_config
+ _persist_migration = _c._persist_migration
+
+ config = read_raw_config()
+ raw_mc = config.get("model_catalog")
+ if isinstance(raw_mc, dict) and raw_mc.get("ttl_hours") == 1 and "ttl_minutes" not in raw_mc:
+ del raw_mc["ttl_hours"]
+ config["model_catalog"] = raw_mc
+ _persist_migration(config)
+ results["config_added"].append("model_catalog.ttl_hours 1 → ttl_minutes 20 (default)")
+ if not quiet:
+ print(" ✓ Model catalog now refreshes every 20 minutes (model_catalog.ttl_minutes)")
+
+
#: Registry of (target_version, migration_fn), strictly ascending. The driver
#: applies every entry whose target version is greater than the on-disk
#: observe earlier steps' writes via read_raw_config() (filesystem state).
@@ -890,6 +912,7 @@ MIGRATIONS: Tuple[Tuple[int, Callable[[Dict[str, Any], bool], None]], ...] = (
(37, _migrate_to_37),
(38, _migrate_to_38),
(39, _migrate_to_39),
+ (40, _migrate_to_40),
)
diff --git a/hermes_cli/container_boot.py b/hermes_cli/container_boot.py
index b0e9821b7b..cc5c6ff0c0 100644
--- a/hermes_cli/container_boot.py
+++ b/hermes_cli/container_boot.py
@@ -136,11 +136,23 @@ def reconcile_profile_gateways(
# for every profile. Named slots must still be registered (so explicit
# lifecycle management remains available), but booting them from their
# persisted run intent would create additional multiplex owners.
+ # Keep the boot reconciler aligned with the gateway that will own these
+ # slots. The runtime resolver gives a recognized environment override
+ # precedence over config.yaml and otherwise preserves the configured value.
+ from gateway.config import load_gateway_config
from utils import is_truthy_value
- multiplex_profiles = is_truthy_value(
- os.environ.get("GATEWAY_MULTIPLEX_PROFILES"),
- )
+ try:
+ multiplex_profiles = load_gateway_config().multiplex_profiles
+ except Exception:
+ log.warning(
+ "Unable to load gateway configuration during container boot; "
+ "using the GATEWAY_MULTIPLEX_PROFILES override if set.",
+ exc_info=True,
+ )
+ multiplex_profiles = is_truthy_value(
+ os.environ.get("GATEWAY_MULTIPLEX_PROFILES"),
+ )
# Default profile — always register, even if nothing has ever
# populated the root profile dir. The slot exists so
diff --git a/hermes_cli/cron.py b/hermes_cli/cron.py
index 8db53a707b..cc19e94c6b 100644
--- a/hermes_cli/cron.py
+++ b/hermes_cli/cron.py
@@ -264,6 +264,12 @@ def cron_list(show_all: bool = False):
last_run = job.get("last_run_at", "?")
if last_status == "ok":
status_display = color("ok", Colors.GREEN)
+ elif last_status == "delivery_failed":
+ # The agent succeeded but the result never reached the user —
+ # not green, and the detail lives in last_delivery_error
+ # (last_error is None for these runs).
+ detail = job.get("last_delivery_error") or "?"
+ status_display = color(f"delivery_failed: {detail}", Colors.YELLOW)
else:
status_display = color(f"{last_status}: {job.get('last_error', '?')}", Colors.RED)
streak = int(job.get("failure_streak") or 0)
@@ -286,6 +292,17 @@ def cron_list(show_all: bool = False):
if delivery_err:
print(f" {color('⚠ Delivery failed:', Colors.YELLOW)} {delivery_err}")
+ # A live adapter acked the last send but returned no message_id /
+ # raw_response (Slack/Matrix/Mattermost shape): accepted as delivered,
+ # but say so here rather than only in a WARNING log line.
+ unverified = job.get("last_delivery_unverified")
+ if unverified:
+ targets = ", ".join(str(t) for t in unverified) if isinstance(unverified, list) else str(unverified)
+ print(
+ f" {color('⚠ Delivery UNVERIFIED:', Colors.YELLOW)} "
+ f"adapter acked {targets} without message_id/raw_response"
+ )
+
fire_err = job.get("last_fire_error")
if isinstance(fire_err, dict) and fire_err.get("detail"):
print(
@@ -688,7 +705,10 @@ def _cron_doctor_issues_for_job(job: Dict[str, Any]) -> List[str]:
issues: List[str] = []
last_status = str(job.get("last_status") or "").strip().lower()
- if last_status and last_status != "ok":
+ # "delivery_failed" means the agent run itself succeeded, so it is not a
+ # failed last run — the dedicated delivery issue below reports it (and
+ # last_error is None, which would render as "unknown error" here).
+ if last_status and last_status not in {"ok", "delivery_failed"}:
err = str(job.get("last_error") or "unknown error").strip()
issues.append(f"last run failed: {err}")
@@ -696,6 +716,11 @@ def _cron_doctor_issues_for_job(job: Dict[str, Any]) -> List[str]:
if delivery_err:
issues.append(f"last delivery failed: {delivery_err}")
+ unverified = job.get("last_delivery_unverified")
+ if unverified:
+ targets = ", ".join(str(t) for t in unverified) if isinstance(unverified, list) else str(unverified)
+ issues.append(f"last delivery unverified (adapter acked without evidence): {targets}")
+
if job.get("enabled", True) and job.get("state") not in {"paused", "completed"}:
next_run = str(job.get("next_run_at") or "").strip()
if not next_run:
@@ -766,6 +791,7 @@ def cron_create(args):
prompt=args.prompt,
name=getattr(args, "name", None),
deliver=getattr(args, "deliver", None),
+ failure_deliver=getattr(args, "failure_deliver", None),
repeat=getattr(args, "repeat", None),
skill=getattr(args, "skill", None),
skills=_normalize_skills(getattr(args, "skill", None), getattr(args, "skills", None)),
@@ -842,6 +868,7 @@ def cron_edit(args):
prompt=getattr(args, "prompt", None),
name=getattr(args, "name", None),
deliver=getattr(args, "deliver", None),
+ failure_deliver=getattr(args, "failure_deliver", None),
repeat=getattr(args, "repeat", None),
skills=final_skills,
script=getattr(args, "script", None),
diff --git a/hermes_cli/dashboard_auth/__init__.py b/hermes_cli/dashboard_auth/__init__.py
index c07b2ade6f..9a997fd224 100644
--- a/hermes_cli/dashboard_auth/__init__.py
+++ b/hermes_cli/dashboard_auth/__init__.py
@@ -19,6 +19,7 @@ from hermes_cli.dashboard_auth.base import (
ProviderError,
RefreshExpiredError,
assert_protocol_compliance,
+ classify_jwks_lookup_error,
)
from hermes_cli.dashboard_auth.registry import (
register_provider,
@@ -39,6 +40,7 @@ __all__ = [
"ProviderError",
"RefreshExpiredError",
"assert_protocol_compliance",
+ "classify_jwks_lookup_error",
"register_provider",
"get_provider",
"list_providers",
diff --git a/hermes_cli/dashboard_auth/base.py b/hermes_cli/dashboard_auth/base.py
index 2d744c6cf3..02db55f65e 100644
--- a/hermes_cli/dashboard_auth/base.py
+++ b/hermes_cli/dashboard_auth/base.py
@@ -110,6 +110,45 @@ class RefreshExpiredError(Exception):
"""
+def classify_jwks_lookup_error(exc: BaseException) -> Exception:
+ """Map a ``PyJWKClient.get_signing_key_from_jwt`` failure to the protocol.
+
+ Only a genuine transport failure (the IDP's JWKS endpoint could not be
+ fetched) is a :class:`ProviderError` — middleware turns that into 503
+ "auth provider unreachable" so a flaky IDP never forces a logout.
+
+ Everything else means the token itself cannot be verified by this
+ provider and is an :class:`InvalidCodeError` (``verify_session`` returns
+ ``None``, the middleware tries the next provider / refresh / 401):
+
+ * ``jwt.DecodeError`` — the bearer is not a JWT at all (an opaque peer
+ key, a legacy session token, garbage). #94558: hosted agents answered
+ every non-JWT bearer with a fast 503 ``Auth provider 'nous'
+ unreachable`` even though Portal was healthy, because "cannot parse"
+ and "cannot reach" were folded into one branch.
+ * ``jwt.PyJWKSetError`` — the JWKS was fetched fine but holds no key for
+ this token's ``kid`` (rotated/foreign key). The provider was reached;
+ the token is simply not one of ours.
+
+ ``PyJWKClientConnectionError`` is the only ``PyJWKClientError`` subclass
+ that denotes unreachability; a bare ``PyJWKClientError`` (unexpected
+ JWKS shape) is kept as a provider fault since the IDP misbehaved.
+ """
+ try:
+ import jwt
+ except Exception: # pragma: no cover - jwt is a hard dep of these providers
+ return ProviderError(f"JWKS lookup failed: {exc!r}")
+ if isinstance(exc, jwt.PyJWKClientConnectionError):
+ return ProviderError(f"JWKS lookup failed: {exc}")
+ if isinstance(exc, (jwt.DecodeError, jwt.PyJWKSetError)):
+ return InvalidCodeError(f"token not verifiable by this provider: {exc}")
+ if isinstance(exc, jwt.PyJWKClientError):
+ return ProviderError(f"JWKS lookup failed: {exc}")
+ if isinstance(exc, jwt.InvalidTokenError):
+ return InvalidCodeError(f"token not verifiable by this provider: {exc}")
+ return ProviderError(f"JWKS lookup failed: {exc!r}")
+
+
class DashboardAuthProvider(ABC):
"""Protocol every dashboard-auth provider plugin implements.
diff --git a/hermes_cli/doctor.py b/hermes_cli/doctor.py
index 1af14d3882..fcb89ce5d0 100644
--- a/hermes_cli/doctor.py
+++ b/hermes_cli/doctor.py
@@ -405,6 +405,28 @@ def check_info(text: str):
print(f" {color('→', Colors.CYAN)} {text}")
+def _doctor_memory_config(hermes_home: Path | None = None) -> dict:
+ """Return the effective memory section used by doctor diagnostics."""
+ home = hermes_home if hermes_home is not None else HERMES_HOME
+ try:
+ from hermes_cli.config import _expand_env_vars, read_user_config_raw
+
+ config_path = home / "config.yaml"
+ if not config_path.exists():
+ return {}
+ config = _expand_env_vars(read_user_config_raw(config_path))
+ try:
+ from hermes_cli import managed_scope
+
+ config = managed_scope.apply_managed_overlay(config)
+ except Exception:
+ pass
+ section = config.get("memory") if isinstance(config, dict) else None
+ return section if isinstance(section, dict) else {}
+ except Exception:
+ return {}
+
+
# ── state.db health/stats thresholds (advisory only — module constants,
# deliberately NOT config: doctor warnings are guidance, not policy) ──
STATE_DB_SIZE_WARN_BYTES = 1 * 1024 * 1024 * 1024 # 1 GiB logical size
@@ -1980,8 +2002,19 @@ def run_doctor(args):
else:
check_warn(f"{_DHH} not found", "(will be created on first use)")
- # Check expected subdirectories
- expected_subdirs = ["cron", "sessions", "logs", "skills", "memories"]
+ from tools.memory_tool import get_builtin_memory_store_flags
+
+ _memory_config = _doctor_memory_config(hermes_home)
+ _memory_enabled, _user_profile_enabled = get_builtin_memory_store_flags(
+ {"memory": _memory_config}
+ )
+
+ # Check expected subdirectories. The built-in file store does not create or
+ # consume memories/ when both targets are disabled, so stale migration files
+ # are not an active diagnostic surface.
+ expected_subdirs = ["cron", "sessions", "logs", "skills"]
+ if _memory_enabled or _user_profile_enabled:
+ expected_subdirs.append("memories")
for subdir_name in expected_subdirs:
subdir_path = hermes_home / subdir_name
if subdir_path.exists():
@@ -2016,22 +2049,28 @@ def run_doctor(args):
check_ok(f"Created {_DHH}/SOUL.md with basic template")
fixed_count += 1
- # Check memory directory
+ # Check only enabled built-in stores. External providers are additive, but
+ # users can explicitly disable either legacy file target; stale files left
+ # by a migration must not be presented as active memory usage.
memories_dir = hermes_home / "memories"
- if memories_dir.exists():
+ if not (_memory_enabled or _user_profile_enabled):
+ check_info("Built-in memory files disabled by config")
+ elif memories_dir.exists():
check_ok(f"{_DHH}/memories/ directory exists")
memory_file = memories_dir / "MEMORY.md"
user_file = memories_dir / "USER.md"
- if memory_file.exists():
- size = len(memory_file.read_text(encoding="utf-8").strip())
- check_ok(f"MEMORY.md exists ({size} chars)")
- else:
- check_info("MEMORY.md not created yet (will be created when the agent first writes a memory)")
- if user_file.exists():
- size = len(user_file.read_text(encoding="utf-8").strip())
- check_ok(f"USER.md exists ({size} chars)")
- else:
- check_info("USER.md not created yet (will be created when the agent first writes a memory)")
+ if _memory_enabled:
+ if memory_file.exists():
+ size = len(memory_file.read_text(encoding="utf-8").strip())
+ check_ok(f"MEMORY.md exists ({size} chars)")
+ else:
+ check_info("MEMORY.md not created yet (will be created when the agent first writes a memory)")
+ if _user_profile_enabled:
+ if user_file.exists():
+ size = len(user_file.read_text(encoding="utf-8").strip())
+ check_ok(f"USER.md exists ({size} chars)")
+ else:
+ check_info("USER.md not created yet (will be created when the agent first writes a memory)")
else:
check_warn(f"{_DHH}/memories/ not found", "(will be created on first use)")
if should_fix:
@@ -3243,21 +3282,7 @@ def run_doctor(args):
check_warn("No GITHUB_TOKEN", f"(60 req/hr rate limit — set in {_DHH}/.env for better rates)")
_section("Memory Provider")
- _active_memory_provider = ""
- try:
- from hermes_cli.config import read_user_config_raw as _read_raw_mem
- _mem_cfg_path = HERMES_HOME / "config.yaml"
- if _mem_cfg_path.exists():
- # Raw-file diagnostic (+ managed overlay below, unchanged).
- _raw_cfg = _read_raw_mem(_mem_cfg_path)
- try:
- from hermes_cli import managed_scope
- _raw_cfg = managed_scope.apply_managed_overlay(_raw_cfg)
- except Exception:
- pass
- _active_memory_provider = (_raw_cfg.get("memory") or {}).get("provider", "")
- except Exception:
- pass
+ _active_memory_provider = _memory_config.get("provider", "")
if not _active_memory_provider:
check_ok("Built-in memory active", "(no external provider configured — this is fine)")
diff --git a/hermes_cli/dump.py b/hermes_cli/dump.py
index c7399f39f8..fa27044f43 100644
--- a/hermes_cli/dump.py
+++ b/hermes_cli/dump.py
@@ -388,6 +388,7 @@ def run_dump(args):
("COMMANDCODE_API_KEY", "commandcode"),
("KILOCODE_API_KEY", "kilocode"),
("FIRECRAWL_API_KEY", "firecrawl"),
+ ("TAVILY_API_KEY", "tavily"),
("KEENABLE_API_KEY", "keenable"),
("BROWSERBASE_API_KEY", "browserbase"),
("FAL_KEY", "fal"),
diff --git a/hermes_cli/env_loader.py b/hermes_cli/env_loader.py
index f926c6ed4e..a0e1fbfd95 100644
--- a/hermes_cli/env_loader.py
+++ b/hermes_cli/env_loader.py
@@ -51,6 +51,10 @@ _SECRET_SOURCE_VALUES_BY_HOME: dict[str, dict[str, str]] = {}
_APPLIED_HOMES: set[str] = set()
_SECRET_SOURCE_CACHE_LOCK = threading.RLock()
+# Routed profile homes whose dotenv load was skipped under multiplex, so the
+# skip is logged once per home rather than on every lazy import mid-turn.
+_SCOPED_SKIP_LOGGED: set[str] = set()
+
def _known_hermes_env_keys() -> set[str]:
"""Return the combined set of known Hermes env-var keys.
@@ -483,10 +487,42 @@ def load_hermes_dotenv(
- callers that only maintain the installation can set
``load_external_secrets=False`` to avoid loading optional secret-manager
dependencies into the process that replaces that same environment.
+ - routed multiplex profile loads hydrate external sources into the
+ profile's private secret snapshot without mutating the shared process
+ environment; unscoped startup loads retain the normal behavior above.
"""
- loaded: list[Path] = []
-
home_path = Path(hermes_home or os.getenv("HERMES_HOME", Path.home() / ".hermes"))
+
+ # A multiplex gateway hosts every profile in one process. While a routed
+ # profile-home override is active, copying that profile's .env into
+ # os.environ would expose its credentials to sibling turns and every
+ # subsequently spawned child. An unscoped startup load remains process
+ # configuration and must retain the normal loading path.
+ # External secret sources still need their normal refresh path, so resolve
+ # them against the existing profile-local mapping instead of simply
+ # returning before all hydration work.
+ from agent.secret_scope import is_multiplex_active
+ from hermes_constants import get_hermes_home_override
+
+ if is_multiplex_active() and get_hermes_home_override() is not None:
+ home_key = str(home_path.resolve())
+ if home_key not in _SCOPED_SKIP_LOGGED:
+ _SCOPED_SKIP_LOGGED.add(home_key)
+ import logging
+
+ logging.getLogger(__name__).debug(
+ "multiplex: skipping process-global dotenv load for routed "
+ "profile home %s (credentials resolve via the profile scope)",
+ home_path,
+ )
+ if load_external_secrets:
+ from hermes_cli import _early_recovery
+
+ if not _early_recovery._should_skip_external_secret_sources():
+ hydrate_profile_secret_sources(home_path)
+ return []
+
+ loaded: list[Path] = []
user_env = home_path / ".env"
project_env_path = Path(project_env) if project_env else None
diff --git a/hermes_cli/gateway.py b/hermes_cli/gateway.py
index 2d10dc4d13..6b155487cc 100644
--- a/hermes_cli/gateway.py
+++ b/hermes_cli/gateway.py
@@ -481,6 +481,46 @@ def _probe_loop_tick_socket(
pass
+def _probe_loop_tick_tcp(
+ port: int,
+ timeout: float = 1.0,
+) -> bool | None:
+ """Ping the loop-scheduling witness via TCP loopback (Windows).
+
+ Same protocol and semantics as the Unix socket variant: connect to
+ 127.0.0.1: and expect one byte "1" as proof the loop is
+ dispatching. Used on Windows / non-POSIX systems where AF_UNIX is not
+ available in asyncio.
+
+ Returns:
+ True — the loop answered.
+ False — the port was reachable but did not answer, or refused.
+ None — invalid port / could not connect for unrelated reasons.
+ """
+ try:
+ port_num = int(port)
+ if port_num <= 0 or port_num > 65535:
+ return None
+ except (TypeError, ValueError):
+ return None
+ sock = None
+ try:
+ sock = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
+ sock.settimeout(max(float(timeout), 0.0))
+ sock.connect(("127.0.0.1", port_num))
+ return sock.recv(1) == b"1"
+ except Exception:
+ # Connection refused, timeout, transient errors: witness exists
+ # but is silent (or the process is dead and the port is closed).
+ return False
+ finally:
+ if sock is not None:
+ try:
+ sock.close()
+ except Exception:
+ pass
+
+
def _probe_loop_tick_socket_sustained(
pid: int,
home: Path | None,
@@ -488,6 +528,7 @@ def _probe_loop_tick_socket_sustained(
timeout: float = 1.0,
strikes: int = 3,
gap_s: float = 0.2,
+ tcp_port: int | None = None,
) -> bool | None:
"""Probe the tick socket until a reply or the sustained-miss budget.
@@ -509,7 +550,10 @@ def _probe_loop_tick_socket_sustained(
"""
total = max(int(strikes), 0)
for attempt in range(total):
- result = _probe_loop_tick_socket(pid, home, timeout=timeout)
+ if tcp_port is not None:
+ result = _probe_loop_tick_tcp(tcp_port, timeout=timeout)
+ else:
+ result = _probe_loop_tick_socket(pid, home, timeout=timeout)
if result is True:
return True
if result is None:
@@ -579,14 +623,26 @@ def probe_gateway_loop_liveness(
# up, or a stale file from a previous PID. Not evidence of a wedge.
return GATEWAY_LOOP_UNKNOWN
- witness = _probe_loop_tick_socket(pid, home, timeout=tick_timeout)
+ # Pick the right witness probe: TCP loopback (Windows / non-POSIX)
+ # takes priority if the producer published a port, otherwise fall back
+ # to the AF_UNIX socket (POSIX / legacy).
+ tcp_port = payload.get("loop_tick_tcp_port")
+ try:
+ tcp_port_int = int(tcp_port) if tcp_port is not None else None
+ except (TypeError, ValueError):
+ tcp_port_int = None
+
+ if tcp_port_int is not None and tcp_port_int > 0:
+ witness = _probe_loop_tick_tcp(tcp_port_int, timeout=tick_timeout)
+ tick_armed = True
+ else:
+ witness = _probe_loop_tick_socket(pid, home, timeout=tick_timeout)
+ tick_armed = payload.get("loop_tick_socket", _LOOP_TICK_ABSENT)
if witness is True:
# The loop answered a ping — it is dispatching right now. A stale
# heartbeat file is a stalled write or a saturated executor, not a
# wedge (#90502).
return GATEWAY_LOOP_ALIVE
-
- tick_armed = payload.get("loop_tick_socket", _LOOP_TICK_ABSENT)
age = time.time() - mtime
if age <= stale_budget:
if witness is False:
@@ -620,6 +676,7 @@ def probe_gateway_loop_liveness(
timeout=tick_timeout,
strikes=tick_strikes - 1,
gap_s=tick_gap_s,
+ tcp_port=tcp_port_int,
)
if sustained is False:
# Both witnesses agree, sustained: the loop did not schedule for
@@ -1344,6 +1401,7 @@ def _spawn_gateway_restart_watcher(old_pid: int, run_argv: list[str]) -> bool:
import sys
import time
from hermes_cli._subprocess_compat import (
+ _WINDOWS_GATEWAY_BREAKAWAY_ENV,
windows_detach_flags,
windows_detach_flags_without_breakaway,
)
@@ -1361,6 +1419,24 @@ def _spawn_gateway_restart_watcher(old_pid: int, run_argv: list[str]) -> bool:
break
time.sleep(0.2)
+ # Route stray stdout/stderr from the respawned gateway to the same
+ # sidecar log _spawn_detached uses. DEVNULL here meant a gateway
+ # killed moments after respawn (e.g. parent Job Object teardown when
+ # breakaway is denied, #48820 4th repro) left ZERO trace anywhere —
+ # no gateway.log line, no exit-diag record, nothing. Best-effort:
+ # fall back to DEVNULL when the log dir is unavailable.
+ _stdio_target = subprocess.DEVNULL
+ _stdio_fh = None
+ try:
+ from hermes_cli.config import get_hermes_home
+ from pathlib import Path
+ _log_dir = Path(get_hermes_home()) / "logs"
+ _log_dir.mkdir(parents=True, exist_ok=True)
+ _stdio_fh = open(_log_dir / "gateway-stdio.log", "ab", buffering=0)
+ _stdio_target = _stdio_fh
+ except Exception:
+ pass
+
# Platform-appropriate detach for the respawned gateway. On POSIX
# start_new_session=True maps to os.setsid; on Windows we need
# explicit creationflags because start_new_session is a no-op there.
@@ -1369,8 +1445,8 @@ def _spawn_gateway_restart_watcher(old_pid: int, run_argv: list[str]) -> bool:
# without breakaway the respawned gateway would die when that job
# tears down. See _subprocess_compat.windows_detach_flags().
_popen_kwargs = {{
- "stdout": subprocess.DEVNULL,
- "stderr": subprocess.DEVNULL,
+ "stdout": _stdio_target,
+ "stderr": _stdio_target,
}}
# Anchor the respawned gateway at the stable working dir and overlay
# the env (VIRTUAL_ENV / PYTHONPATH / HERMES_HOME) the windowless
@@ -1378,23 +1454,45 @@ def _spawn_gateway_restart_watcher(old_pid: int, run_argv: list[str]) -> bool:
# the venv python resolves imports without help.
if _respawn_cwd:
_popen_kwargs["cwd"] = _respawn_cwd
- if _respawn_env_overlay:
- _popen_kwargs["env"] = {{**os.environ, **_respawn_env_overlay}}
- if sys.platform == "win32":
- try:
- _popen_kwargs["creationflags"] = windows_detach_flags()
+ _base_env = {{**os.environ, **_respawn_env_overlay}}
+ try:
+ if sys.platform == "win32":
+ try:
+ _popen_kwargs["creationflags"] = windows_detach_flags()
+ # Stamp the breakaway state exactly like the canonical
+ # gateway_windows._spawn_detached, so the respawned
+ # gateway's exit-diag / lifecycle records show whether it
+ # escaped the parent Job Object (#48820 4th repro:
+ # without the stamp, a job-teardown kill was
+ # indistinguishable from any other silent death).
+ _popen_kwargs["env"] = {{
+ **_base_env, _WINDOWS_GATEWAY_BREAKAWAY_ENV: "1",
+ }}
+ subprocess.Popen(cmd, **_popen_kwargs)
+ except OSError:
+ # CREATE_BREAKAWAY_FROM_JOB can be rejected with
+ # ERROR_ACCESS_DENIED when the parent's job object refuses
+ # breakaway. Retry without it — DETACHED_PROCESS et al.
+ # alone are enough in most setups. Mirrors the canonical
+ # fallback in gateway_windows._spawn_detached.
+ _popen_kwargs["creationflags"] = (
+ windows_detach_flags_without_breakaway()
+ )
+ _popen_kwargs["env"] = {{
+ **_base_env, _WINDOWS_GATEWAY_BREAKAWAY_ENV: "0",
+ }}
+ subprocess.Popen(cmd, **_popen_kwargs)
+ else:
+ if _respawn_env_overlay:
+ _popen_kwargs["env"] = _base_env
+ _popen_kwargs["start_new_session"] = True
subprocess.Popen(cmd, **_popen_kwargs)
- except OSError:
- # CREATE_BREAKAWAY_FROM_JOB can be rejected with
- # ERROR_ACCESS_DENIED when the parent's job object refuses
- # breakaway. Retry without it — DETACHED_PROCESS et al.
- # alone are enough in most setups. Mirrors the canonical
- # fallback in gateway_windows._spawn_detached.
- _popen_kwargs["creationflags"] = windows_detach_flags_without_breakaway()
- subprocess.Popen(cmd, **_popen_kwargs)
- else:
- _popen_kwargs["start_new_session"] = True
- subprocess.Popen(cmd, **_popen_kwargs)
+ finally:
+ if _stdio_fh is not None:
+ try:
+ _stdio_fh.close()
+ except OSError:
+ pass
"""
).strip().format(
respawn_cwd_literal=respawn_cwd_literal,
@@ -2121,14 +2219,17 @@ def _gateway_list() -> None:
label += " (current)"
parts = [f" {marker} {label:<24s}"]
if prof.gateway_running:
+ pid = None
try:
from gateway.status import get_running_pid
pid = get_running_pid(prof.path / "gateway.pid", cleanup_stale=False)
- if pid:
- parts.append(f"PID {pid}")
except Exception:
pass
+ if pid:
+ parts.append(f"PID {pid}")
+ elif named_profile_served_by_running_multiplexer(prof.name):
+ parts.append("served by the default multiplexer")
else:
parts.append("not running")
print(" — ".join(parts))
@@ -6178,18 +6279,20 @@ def _running_under_gateway_supervisor() -> bool:
return is_gateway_supervisor_process()
-def named_profile_served_by_running_multiplexer() -> bool:
+def named_profile_served_by_running_multiplexer(profile_name: str | None = None) -> bool:
"""True when a live default multiplexer already ticks this named profile.
- Shared by the named-profile start guard and cron liveness: a satellite
- profile has no gateway.pid of its own, but the default multiplexer's
- ticker still fires its jobs (#97120).
+ Shared by the named-profile start guard, cron liveness, and the
+ ``gateway status`` / ``gateway list`` / ``profile list`` reports: a
+ satellite profile has no gateway.pid of its own, but the default
+ multiplexer's ticker still fires its jobs (#97120) and serves its
+ platforms. ``profile_name`` defaults to the current HERMES_HOME profile.
"""
try:
- suffix = _profile_suffix()
+ suffix = profile_name if profile_name is not None else _profile_suffix()
except Exception:
return False
- if not suffix:
+ if not suffix or suffix == "default":
return False
try:
@@ -8973,7 +9076,12 @@ def _gateway_command_inner(args):
from hermes_cli import gateway_windows
_windows_service_installed = gateway_windows.is_installed()
- if supports_systemd_services() and (
+ if not snapshot.running and named_profile_served_by_running_multiplexer():
+ # Satellite profile: no gateway.pid / service of its own, but the
+ # default multiplexer is the live inbound process for it.
+ print("✓ Gateway is running via the default-profile multiplexer")
+ print(" Manage it from the default profile: hermes gateway status")
+ elif supports_systemd_services() and (
get_systemd_unit_path(system=False).exists()
or get_systemd_unit_path(system=True).exists()
):
diff --git a/hermes_cli/gateway_windows.py b/hermes_cli/gateway_windows.py
index b2ddf9fea6..3f86247613 100644
--- a/hermes_cli/gateway_windows.py
+++ b/hermes_cli/gateway_windows.py
@@ -1187,7 +1187,8 @@ def install(
def _confirm_gateway_stable(
- initial_pids: list[int], confirm_s: float, interval_s: float
+ initial_pids: list[int], confirm_s: float, interval_s: float,
+ all_profiles: bool = False,
) -> list[int]:
"""Re-check a freshly detected gateway for ``confirm_s`` seconds.
@@ -1206,7 +1207,7 @@ def _confirm_gateway_stable(
confirm_deadline = time.monotonic() + confirm_s
while time.monotonic() < confirm_deadline:
time.sleep(interval_s)
- pids = list(find_gateway_pids())
+ pids = list(find_gateway_pids(all_profiles=all_profiles))
if not pids:
return []
return pids
@@ -1216,6 +1217,7 @@ def _wait_for_gateway_ready(
timeout_s: float = 6.0,
interval_s: float = 0.4,
confirm_s: float = 2.0,
+ all_profiles: bool = False,
) -> list[int]:
"""Poll for a live gateway process for up to ``timeout_s`` seconds.
@@ -1225,6 +1227,10 @@ def _wait_for_gateway_ready(
after spawn must not earn a ✓, #91675). If it vanishes during the
confirmation window, polling resumes until the deadline.
+ ``all_profiles`` widens the scan across every profile's gateway — the
+ post-update resume path relaunches the whole fleet, not just the active
+ profile.
+
Returns the list of PIDs found. Empty list means nothing (stable) came
up in time — the caller should surface that to the user as a failed
start.
@@ -1233,9 +1239,11 @@ def _wait_for_gateway_ready(
deadline = time.monotonic() + timeout_s
while time.monotonic() < deadline:
- pids = list(find_gateway_pids())
+ pids = list(find_gateway_pids(all_profiles=all_profiles))
if pids:
- confirmed = _confirm_gateway_stable(pids, confirm_s, interval_s)
+ confirmed = _confirm_gateway_stable(
+ pids, confirm_s, interval_s, all_profiles=all_profiles
+ )
if confirmed:
return confirmed
continue # died during confirmation — keep polling until deadline
diff --git a/hermes_cli/goals.py b/hermes_cli/goals.py
index df1b86df44..74acc22c80 100644
--- a/hermes_cli/goals.py
+++ b/hermes_cli/goals.py
@@ -153,12 +153,22 @@ JUDGE_SYSTEM_PROMPT = (
"You are a strict judge evaluating whether an autonomous agent has "
"achieved a user's stated goal. You receive the goal text, the agent's "
"most recent response, and — when present — a list of background "
- "processes the agent has running. Decide one of three verdicts.\n\n"
+ "processes the agent has running. Decide one of four verdicts.\n\n"
"DONE — the goal is fully satisfied:\n"
"- The response explicitly confirms the goal was completed, OR\n"
- "- The response clearly shows the final deliverable was produced, OR\n"
- "- The response explains the goal is unachievable / blocked / needs "
- "user input (treat this as DONE with reason describing the block).\n\n"
+ "- The response clearly shows the final deliverable was produced.\n"
+ "DONE requires the deliverable to actually exist. If the response only "
+ "explains why the goal cannot be reached, the verdict is BLOCKED, not "
+ "DONE.\n\n"
+ "BLOCKED — the goal cannot be satisfied as stated:\n"
+ "- The response explains the goal is genuinely unachievable (impossible, "
+ "out of scope, no valid path to the deliverable), or refuses to "
+ "fabricate a deliverable that cannot exist, OR\n"
+ "- The response explains progress is blocked and the next step needs "
+ "user input to proceed.\n"
+ "Return BLOCKED with the reason describing what is blocking. BLOCKED is "
+ "a refusal, not a completion — never return BLOCKED for a goal that "
+ "was achieved.\n\n"
"WAIT — the goal is NOT done, but the next step is to wait for async "
"work to finish rather than act again. Choose this ONLY when the agent's "
"progress is genuinely gated on something running on its own:\n"
@@ -180,6 +190,7 @@ JUDGE_SYSTEM_PROMPT = (
"take right now. This is the default when in doubt.\n\n"
"Reply ONLY with a single JSON object on one line. Shapes:\n"
'{"verdict": "done", "reason": ""}\n'
+ '{"verdict": "blocked", "reason": ""}\n'
'{"verdict": "continue", "reason": ""}\n'
'{"verdict": "wait", "wait_on_session": "", "reason": ""}\n'
'{"verdict": "wait", "wait_on_pid": , "reason": ""}\n'
@@ -203,7 +214,7 @@ JUDGE_USER_PROMPT_TEMPLATE = (
"Agent's most recent response:\n{response}\n\n"
"{background_block}"
"Current time: {current_time}\n\n"
- "Is the goal satisfied — done, continue, or wait?"
+ "Is the goal satisfied — done, blocked, continue, or wait?"
)
# Used when the user has added /subgoal criteria. The judge must
@@ -247,11 +258,11 @@ JUDGE_USER_PROMPT_WITH_CONTRACT_TEMPLATE = (
"process to satisfy the Verification criterion (e.g. CI is the "
"verification and it's still running), return WAIT on that process "
"instead of re-poking — re-poking now would be pure busy-work.\n"
- "- If the response explains the work is blocked / unachievable / needs "
- "user input (e.g. the stated Stop condition was hit), treat it as DONE "
- "with the reason describing the block.\n"
+ "- If the response explains the work is genuinely unachievable or hits "
+ "the stated Stop condition and needs user input, the goal is NOT done — "
+ "return BLOCKED with the reason describing the block.\n"
"- Otherwise the goal is NOT done — CONTINUE.\n\n"
- "Is the goal satisfied per its completion contract — done, continue, or wait?"
+ "Is the goal satisfied per its completion contract — done, blocked, continue, or wait?"
)
@@ -553,7 +564,7 @@ class GoalState:
max_turns: int = DEFAULT_MAX_TURNS
created_at: float = 0.0
last_turn_at: float = 0.0
- last_verdict: Optional[str] = None # "done" | "continue" | "skipped"
+ last_verdict: Optional[str] = None # "done" | "blocked" | "continue" | "wait" | "skipped"
last_reason: Optional[str] = None
paused_reason: Optional[str] = None # why we auto-paused (budget, etc.)
consecutive_parse_failures: int = 0 # judge-output parse failures in a row
@@ -1027,7 +1038,7 @@ def _parse_judge_response(raw: str) -> Tuple[str, str, bool, Optional[Dict[str,
"""Parse the judge's reply. Fail-open on unusable output.
Returns ``(verdict, reason, parse_failed, wait_directive)`` where:
- - ``verdict`` is ``"done"``, ``"continue"``, or ``"wait"``.
+ - ``verdict`` is ``"done"``, ``"blocked"``, ``"continue"``, or ``"wait"``.
- ``parse_failed`` is True when the judge returned output that couldn't
be interpreted as the expected JSON verdict (empty body, prose,
malformed JSON). Callers use it to auto-pause after N consecutive
@@ -1084,7 +1095,7 @@ def _parse_judge_response(raw: str) -> Tuple[str, str, bool, Optional[Dict[str,
done = bool(done_val)
verdict = "done" if done else "continue"
- if verdict not in {"done", "continue", "wait"}:
+ if verdict not in {"done", "blocked", "continue", "wait"}:
verdict = "continue"
if verdict != "wait":
@@ -1178,7 +1189,7 @@ def judge_goal(
"""Ask the auxiliary model whether the goal is satisfied.
Returns ``(verdict, reason, parse_failed, wait_directive, transport_failed)`` where verdict
- is ``"done"``, ``"continue"``, ``"wait"``, or ``"skipped"`` (when the
+ is ``"done"``, ``"blocked"``, ``"continue"``, ``"wait"``, or ``"skipped"`` (when the
judge couldn't be reached). ``wait_directive`` is set only for ``"wait"``
(``{"pid": int}`` or ``{"seconds": int}``); ``None`` otherwise.
@@ -1882,7 +1893,7 @@ class GoalManager:
- ``status``: current goal status after update
- ``should_continue``: bool — caller should fire another turn
- ``continuation_prompt``: str or None
- - ``verdict``: "done" | "continue" | "wait" | "skipped" | "inactive"
+ - ``verdict``: "done" | "blocked" | "continue" | "wait" | "skipped" | "inactive"
- ``reason``: str
- ``message``: user-visible one-liner to print/send
"""
@@ -1999,6 +2010,28 @@ class GoalManager:
"message": f"⏳ Goal parked (judge) — waiting on {tgt}: {reason}",
}
+ # BLOCKED verdict: the judge ruled the goal genuinely cannot be
+ # satisfied as stated (impossible, out of scope, needs user input).
+ # This is NOT done — don't keep burning turns on an unachievable goal
+ # and don't wave it through as complete (#100954). Pause so the user
+ # sees the judge's reason and can re-scope (/goal set) or override
+ # (/goal resume).
+ if verdict == "blocked":
+ state.status = "paused"
+ state.paused_reason = f"judged unachievable: {reason}"
+ save_goal(self.session_id, state)
+ return {
+ "status": "paused",
+ "should_continue": False,
+ "continuation_prompt": None,
+ "verdict": "blocked",
+ "reason": reason,
+ "message": (
+ f"🚫 Goal judged unachievable — paused: {reason} "
+ "Re-scope with /goal set, or override with /goal resume."
+ ),
+ }
+
if verdict == "done":
state.status = "done"
save_goal(self.session_id, state)
@@ -2202,7 +2235,7 @@ def run_kanban_goal_loop(
Returns a decision dict: ``{"outcome", "turns_used", "reason"}`` where
outcome is one of ``"completed_by_worker"``, ``"review_requested_by_worker"``,
``"changes_requested_by_reviewer"``, ``"blocked_budget"``,
- ``"blocked_by_worker"``, or ``"stopped"``.
+ ``"blocked_unachievable"``, ``"blocked_by_worker"``, or ``"stopped"``.
"""
def _log(msg: str) -> None:
@@ -2258,6 +2291,22 @@ def run_kanban_goal_loop(
verdict = "continue"
_log(f"kanban goal loop: turn {turns_used}/{max_turns} verdict={verdict} reason={_truncate(reason, 120)}")
+ if verdict == "blocked":
+ # The judge ruled the goal cannot be satisfied at all — this is
+ # NOT done (#100954). Block the card now with the judge's reason
+ # instead of spending the remaining turns re-poking an impossible
+ # goal, and never let it land in done.
+ _log(f"kanban goal loop: task {task_id} judged unachievable; blocking")
+ try:
+ block_fn(f"Goal-mode judge ruled the goal unachievable: {reason}")
+ except Exception as exc:
+ _log(f"kanban goal loop: block_fn failed ({exc})")
+ return {
+ "outcome": "blocked_unachievable",
+ "turns_used": turns_used,
+ "reason": f"judge verdict blocked: {reason}",
+ }
+
if verdict == "done":
if nudged_to_finalize:
# Already asked once to call kanban_complete and it still
diff --git a/hermes_cli/gui_uninstall.py b/hermes_cli/gui_uninstall.py
index dc0d991ce3..90c8d5f549 100644
--- a/hermes_cli/gui_uninstall.py
+++ b/hermes_cli/gui_uninstall.py
@@ -156,10 +156,9 @@ def packaged_gui_app_paths() -> "list[Path]":
data_base / "applications" / "Hermes.desktop",
data_base / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png",
]
- # Fixed-size hicolor dirs: the icon is copied at its native size
- # (read from the PNG header), so sweep the standard ones plus the
- # 1024x1024 dir the shipped asset lands in.
- for size in ("256x256", "512x512", "1024x1024"):
+ # Fixed-size hicolor dirs the installer may have written (resized
+ # panel sizes plus leftover native-size copies from older builds).
+ for size in ("24x24", "32x32", "48x48", "256x256", "512x512", "1024x1024"):
paths.append(data_base / "icons" / "hicolor" / size / "apps" / "hermes.png")
return paths
diff --git a/hermes_cli/inventory.py b/hermes_cli/inventory.py
index 3f2de0820b..f9b0de4d90 100644
--- a/hermes_cli/inventory.py
+++ b/hermes_cli/inventory.py
@@ -206,6 +206,31 @@ def build_models_payload(
excluded_providers=ctx.excluded_providers or [],
)
+ # Managed local runtime: staged GGUFs are selectable like any provider's
+ # models. list_authenticated_providers can't know about them (no
+ # credential, no custom_providers entry — the credential is
+ # reachability), so inject the row here where every picker surface
+ # inherits it. Present whenever models are staged; picking one routes
+ # through the llamacpp alias -> managed/detected server resolution.
+ local_row = _local_runtime_row(ctx)
+ if local_row is not None:
+ rows = [r for r in rows if str(r.get("slug", "")).lower() != "llamacpp"]
+ rows.append(local_row)
+ # A live session on the managed server reports provider "custom"
+ # (the resolution seam's generic label for a raw base_url), which
+ # would otherwise materialize a duplicate "Custom endpoint" row
+ # carrying the same staged models and stealing the checkmark. The
+ # Local row owns the managed server's identity — drop custom rows
+ # that point at the managed endpoint.
+ if local_row.get("is_current"):
+ def _is_managed_custom(row: dict) -> bool:
+ if str(row.get("slug", "")).lower() != "custom":
+ return False
+ models = {str(m) for m in (row.get("models") or [])}
+ return bool(models) and models <= set(local_row["models"])
+
+ rows = [r for r in rows if not _is_managed_custom(r)]
+
moa_row = _moa_provider_row(ctx.current_provider)
if moa_row is not None:
rows = [moa_row] + [r for r in rows if str(r.get("slug", "")).lower() != "moa"]
@@ -217,9 +242,15 @@ def build_models_payload(
# has lost its credential, list_authenticated_providers() omits it;
# keep that one row visible so the UI can show the saved selection and
# a re-auth affordance instead of appearing to jump to another provider.
- rows = list(rows) + _append_unconfigured_rows(
- rows, ctx, current_only=True
- )
+ # Exception: a "custom" current whose endpoint is the managed local
+ # server is already represented (with the checkmark) by the Local row
+ # — the skeleton would resurrect the duplicate the dedup above removed.
+ _local_owns_current = bool(local_row and local_row.get("is_current")
+ and (ctx.current_provider or "").lower() == "custom")
+ if not _local_owns_current:
+ rows = list(rows) + _append_unconfigured_rows(
+ rows, ctx, current_only=True
+ )
# --- Deduplicate: remove models from aggregators that overlap with
# user-defined providers. When a local proxy (e.g. litellm-proxy)
@@ -735,6 +766,15 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list
if current_slug and slug == current_slug:
kept.append(row)
continue
+ if row.get("source") == "local-runtime":
+ # Managed local models are explicit configuration by existence:
+ # the user downloaded gigabytes into the machine-scoped models
+ # dir. There is deliberately no config credential to find
+ # (credential is reachability), so without this clause the row
+ # only survives on the profile where Use was last clicked —
+ # every other profile loses local models from its picker.
+ kept.append(row)
+ continue
if slug == "moa":
# MoA is a virtual routing mode, not an independently configured
# provider. Hide it from explicit-only pickers unless it is the
@@ -758,11 +798,35 @@ def _filter_explicit_provider_rows(rows: list[dict], ctx: ConfigContext) -> list
# just accepted those same credentials when building it.
kept.append(row)
continue
+ if _external_process_signed_in(slug):
+ # External-process providers (copilot-acp) authenticate through
+ # their own CLI (`copilot login`), which — like the Anthropic
+ # OAuth case above — leaves no trace in active_provider,
+ # model.provider, or env vars. Verified CLI credentials are a
+ # deliberate sign-in; without this the desktop picker drops the
+ # row the picker-discovery side just accepted.
+ kept.append(row)
+ continue
if is_provider_explicitly_configured(slug):
kept.append(row)
return kept
+def _external_process_signed_in(slug: str) -> bool:
+ """True when an external-process provider has verified CLI credentials."""
+ try:
+ from hermes_cli.auth import (
+ PROVIDER_REGISTRY,
+ get_external_process_provider_status,
+ )
+ pconfig = PROVIDER_REGISTRY.get(slug)
+ if not pconfig or pconfig.auth_type != "external_process":
+ return False
+ return bool(get_external_process_provider_status(slug).get("auth_verified"))
+ except Exception:
+ return False
+
+
def _provider_is_keyless(slug: str) -> bool:
"""True when the provider's Hermes overlay declares it keyless."""
try:
@@ -986,6 +1050,56 @@ def _apply_pricing(
row["unavailable_models"] = []
+def _local_runtime_row(ctx: "ConfigContext") -> dict | None:
+ """Build the ``llamacpp`` provider row from staged local models.
+
+ Present whenever GGUFs are staged in the managed models directory —
+ downloaded models must be selectable even before the server is running
+ (selection starts it via the runtime_provider seam / activate flow).
+ Returns ``None`` when nothing is staged.
+ """
+ try:
+ from hermes_cli.local_runtime.bootstrap import staged_model_ids
+
+ staged = staged_model_ids()
+ if not staged:
+ return None
+ current = (ctx.current_provider or "").strip().lower() in (
+ "llamacpp", "llama.cpp", "llama-cpp")
+ if not current:
+ # A LIVE session on the managed server reports provider "custom"
+ # (the resolution seam's label) with the managed base_url. Match
+ # on the endpoint so the picker still marks this row current —
+ # otherwise the session the user is chatting in shows no
+ # selection.
+ try:
+ from hermes_cli.local_runtime.endpoint import _state_endpoint
+
+ managed = _state_endpoint()
+ current = bool(
+ managed
+ and (ctx.current_base_url or "").strip().rstrip("/")
+ == managed["base_url"].rstrip("/"))
+ except Exception:
+ current = False
+ return {
+ "slug": "llamacpp",
+ # Bare "Local" everywhere user-facing: the engine name is an
+ # implementation detail (the pane brands this "Local models").
+ "name": "Local",
+ "is_current": current,
+ "is_user_defined": False,
+ "models": staged,
+ "total_models": len(staged),
+ "source": "local-runtime",
+ "authenticated": True, # the credential is reachability
+ "auth_type": "local",
+ "warning": None,
+ }
+ except Exception:
+ return None
+
+
def _moa_provider_row(current_provider: str = "") -> dict | None:
"""Build the virtual ``moa`` provider row for model pickers.
diff --git a/hermes_cli/kanban.py b/hermes_cli/kanban.py
index e23eedc7fa..7037941a46 100644
--- a/hermes_cli/kanban.py
+++ b/hermes_cli/kanban.py
@@ -2311,18 +2311,23 @@ def _worker_run_id_for(task_id: str) -> Optional[int]:
return None
-def _goal_mode_handoff_rejection(task: Optional[kb.Task], evidence: str) -> Optional[str]:
- """Apply the goal judge to every terminal worker handoff, including review."""
+def _goal_mode_handoff_rejection(task: Optional[kb.Task], evidence: str):
+ """Apply the goal judge to every terminal worker handoff, including review.
+
+ Returns ``(verdict, reason_or_None)`` — ``"done"`` allows the handoff;
+ ``"blocked"`` means the judge ruled the goal unachievable (#100954);
+ ``"continue"``/``"wait"`` reject with the judge's reason.
+ """
if task is None or not task.goal_mode:
- return None
+ return ("done", None)
try:
from agent.auxiliary_client import get_text_auxiliary_client
client, model = get_text_auxiliary_client("goal_judge")
except Exception:
- return None
+ return ("done", None)
if client is None or not model:
- return None
+ return ("done", None)
from hermes_cli.goals import judge_goal
@@ -2341,7 +2346,7 @@ def _goal_mode_handoff_rejection(task: Optional[kb.Task], evidence: str) -> Opti
judge_exc,
exc_info=True,
)
- return reason if verdict != "done" else None
+ return (verdict, None if verdict == "done" else reason)
def _cmd_complete(args: argparse.Namespace) -> int:
@@ -2379,10 +2384,20 @@ def _cmd_complete(args: argparse.Namespace) -> int:
# to every terminal handoff so request-review cannot bypass the
# acceptance contract that protects complete.
task = kb.get_task(conn, tid)
- rejection = _goal_mode_handoff_rejection(
+ gate_verdict, rejection = _goal_mode_handoff_rejection(
task,
(summary or args.result or "").strip(),
)
+ if gate_verdict == "blocked":
+ print(
+ f"kanban: goal completion of {tid} rejected: judge ruled "
+ f"the goal unachievable — {rejection}. Re-scope with "
+ f"kanban edit, or record the block with kanban block "
+ f"instead of completing.",
+ file=sys.stderr,
+ )
+ failed.append(tid)
+ continue
if rejection is not None:
print(
f"kanban: goal completion of {tid} rejected by judge: {rejection}. "
@@ -2532,10 +2547,18 @@ def _cmd_request_review(args: argparse.Namespace) -> int:
return 2
reviewer = getattr(args, "reviewer", None)
with kb.connect_closing() as conn:
- rejection = _goal_mode_handoff_rejection(
+ gate_verdict, rejection = _goal_mode_handoff_rejection(
kb.get_task(conn, tid),
summary or "",
)
+ if gate_verdict == "blocked":
+ print(
+ f"kanban: goal review handoff of {tid} rejected: judge ruled "
+ f"the goal unachievable — {rejection}. Record the block with "
+ f"kanban block instead of requesting review.",
+ file=sys.stderr,
+ )
+ return 1
if rejection is not None:
print(
f"kanban: goal review handoff of {tid} rejected by judge: "
diff --git a/hermes_cli/kanban_db.py b/hermes_cli/kanban_db.py
index cb3863466b..198669792e 100644
--- a/hermes_cli/kanban_db.py
+++ b/hermes_cli/kanban_db.py
@@ -11685,8 +11685,8 @@ def purge_stale_done_notify_subs(
*,
max_age_days: int = 30,
) -> int:
- """Delete notify subscriptions whose task has sat in ``done`` untouched
- for longer than ``max_age_days``.
+ """Delete notify subscriptions whose task has sat in ``done`` or
+ ``blocked`` untouched for longer than ``max_age_days``.
The notifier keeps subscriptions alive through ``done`` because a
completed task can be reopened (review corrections, continuation) and
@@ -11695,7 +11695,10 @@ def purge_stale_done_notify_subs(
subscription rows forever — each one scanned every notifier tick.
This GC bounds that: a task that has been ``done`` with no new events
for the retention window is treated as settled and its subscriptions
- are purged. Age is measured from the task's most recent event
+ are purged. ``blocked`` tasks (circuit-breaker trips, dead workers)
+ are reaped on the same clock — they are abandoned, not idle, unlike a
+ ``backlog``/``ready`` card that is merely waiting for pickup (#100955).
+ Age is measured from the task's most recent event
(falling back to ``completed_at`` then ``created_at``), so ANY
activity — including a reopen, which also moves the task off
``done`` — resets or exempts it.
@@ -11714,7 +11717,7 @@ def purge_stale_done_notify_subs(
cur = conn.execute(
"DELETE FROM kanban_notify_subs WHERE task_id IN ("
" SELECT t.id FROM tasks t"
- " WHERE t.status = 'done'"
+ " WHERE t.status IN ('done', 'blocked')"
" AND COALESCE("
" (SELECT MAX(e.created_at) FROM task_events e"
" WHERE e.task_id = t.id),"
diff --git a/hermes_cli/linux_desktop_entry.py b/hermes_cli/linux_desktop_entry.py
index 3a95229e76..e33675590e 100644
--- a/hermes_cli/linux_desktop_entry.py
+++ b/hermes_cli/linux_desktop_entry.py
@@ -15,9 +15,9 @@ Two values must be absolute for the entry to work:
checkout. Do not copy the icon: ``Exec`` already depends on that tree.
Cache refresh is best-effort and tool-gated: ``update-desktop-database``
-for the freedesktop menu cache, and ``kbuildsycoca6``/``kbuildsycoca5``
-for Plasma. Run each tool only when it exists. A missing tool is not an
-error.
+for the freedesktop menu cache, ``gtk-update-icon-cache`` for the user
+hicolor tree, and ``kbuildsycoca6``/``kbuildsycoca5`` for Plasma. Run
+each tool only when it exists. A missing tool is not an error.
Import-light and side-effect-free at import time: the uninstaller uses
this without loading the full CLI.
@@ -25,6 +25,7 @@ this without loading the full CLI.
from __future__ import annotations
+import io
import os
import shutil
import struct
@@ -624,31 +625,135 @@ def _run_quiet(cmd: "list[str]") -> bool:
return result.returncode == 0
+# Sizes a typical hicolor ``index.theme`` actually lists. ``scalable`` is
+# SVG-only — a raster PNG there is what Cinnamon's panel draws as a
+# mangled low-res blob. The shipped desktop asset is 1024×1024, which is
+# also not an indexed dir name, so a copy-only fallback lands in ``256x256``.
+_HICOLOR_INDEXED_SIZES = (16, 22, 24, 32, 36, 48, 64, 72, 96, 128, 192, 256, 512)
+# Cinnamon's panel is ~24–32px. Write these so the theme loads an exact
+# raster instead of downscaling a 1024px PNG at lookup time.
+_HICOLOR_INSTALL_SIZES = (24, 32, 48, 256)
+
+
+def _png_dimensions(raw: bytes) -> Optional[tuple[int, int]]:
+ """Return ``(width, height)`` from a PNG IHDR, or ``None`` if unreadable."""
+ if len(raw) >= 24 and raw[:8] == b"\x89PNG\r\n\x1a\n" and raw[12:16] == b"IHDR":
+ return struct.unpack(">II", raw[16:24])
+ return None
+
+
+def _hicolor_subdir(dimensions: Optional[tuple[int, int]]) -> str:
+ """Pick a fixed-size hicolor dir the theme indexes. Never ``scalable``."""
+ if dimensions is None:
+ return "256x256"
+ width, height = dimensions
+ if width != height or width <= 0:
+ return "256x256"
+ if width in _HICOLOR_INDEXED_SIZES:
+ return f"{width}x{width}"
+ if width > 256:
+ return "256x256"
+ nearest = min(_HICOLOR_INDEXED_SIZES, key=lambda size: abs(size - width))
+ return f"{nearest}x{nearest}"
+
+
+def _hicolor_icon_dest(subdir: str) -> Path:
+ return _xdg_data_home() / "icons" / "hicolor" / subdir / "apps" / "hermes.png"
+
+
+def _remove_stale_scalable_icon() -> bool:
+ """Drop a leftover PNG from ``scalable/`` (the pre-fix install path).
+
+ Return True when a file was removed so the caller can refresh the
+ icon cache. A missing file is not an error.
+ """
+ stale = _hicolor_icon_dest("scalable")
+ try:
+ if stale.is_file():
+ stale.unlink()
+ return True
+ except OSError:
+ return False
+ return False
+
+
+def _refresh_hicolor_cache() -> None:
+ """Best-effort reindex of the user hicolor tree. Missing tool is fine."""
+ hicolor = _xdg_data_home() / "icons" / "hicolor"
+ for tool in ("gtk-update-icon-cache", "gtk4-update-icon-cache"):
+ resolved = shutil.which(tool)
+ if resolved:
+ _run_quiet([resolved, "-f", "-t", str(hicolor)])
+ return
+
+
+def _resized_hicolor_pngs(raw: bytes) -> Optional[dict[str, bytes]]:
+ """Lanczos-resize *raw* to each panel size. ``None`` when it will not decode.
+
+ Pillow is a core dep but this module stays import-light: the import is
+ local so the uninstaller does not pay it. A truncated/fake PNG (tests,
+ interrupted copy) returns None and the caller falls back to a copy.
+ """
+ try:
+ from PIL import Image
+ except ImportError:
+ return None
+ try:
+ with Image.open(io.BytesIO(raw)) as im:
+ rgba = im.convert("RGBA")
+ out: dict[str, bytes] = {}
+ for size in _HICOLOR_INSTALL_SIZES:
+ resized = rgba.resize((size, size), Image.Resampling.LANCZOS)
+ buf = io.BytesIO()
+ resized.save(buf, format="PNG")
+ out[f"{size}x{size}"] = buf.getvalue()
+ return out
+ except (OSError, ValueError):
+ return None
+
+
+def _write_hicolor_pngs(files: dict[str, bytes]) -> bool:
+ """Write *files* keyed by hicolor size dir. Return True if any file changed."""
+ wrote = False
+ for subdir, data in files.items():
+ dest = _hicolor_icon_dest(subdir)
+ if dest.is_file() and dest.read_bytes() == data:
+ continue
+ dest.parent.mkdir(parents=True, exist_ok=True)
+ dest.write_bytes(data)
+ wrote = True
+ return wrote
+
+
def _install_icon_to_hicolor(icon: Path) -> bool:
- """Copy the app icon into the user's hicolor icon theme tree.
+ """Install the app icon into the user's hicolor icon theme tree.
The freedesktop icon lookup finds an installed ``apps/hermes.png``
by the unqualified name ``hermes``, so the entry can reference the
- icon without an absolute checkout path. The size subdirectory must
- be one the theme actually indexes (hicolor's index.theme lists
- fixed sizes and ``scalable`` — an unindexed dir like ``1024x1024``
- would never be found), so the icon lands in ``scalable`` unless the
- source is exactly 256x256, which goes to the fixed-size dir.
- Idempotent via content-compare; OSError caught internally (False) —
- the caller then falls back to the absolute path.
+ icon without an absolute checkout path. Raster PNGs go to indexed
+ fixed-size dirs, never ``scalable`` (SVG-only). When the source
+ decodes, it is Lanczos-resized to 24/32/48/256 so Cinnamon's panel
+ does not nearest-neighbor a 1024px PNG. Undecodable bytes fall back
+ to a copy into one indexed dir. Idempotent via content-compare;
+ OSError caught internally (False) — the caller then falls back to
+ the absolute path.
"""
try:
raw = icon.read_bytes()
- is_256 = False
- if len(raw) >= 24 and raw[:8] == b"\x89PNG\r\n\x1a\n" and raw[12:16] == b"IHDR":
- width, height = struct.unpack(">II", raw[16:24])
- is_256 = (width, height) == (256, 256)
- subdir = "256x256" if is_256 else "scalable"
- dest = _xdg_data_home() / "icons" / "hicolor" / subdir / "apps" / "hermes.png"
- if dest.is_file() and dest.read_bytes() == raw:
- return True
- dest.parent.mkdir(parents=True, exist_ok=True)
- shutil.copyfile(icon, dest)
+ resized = _resized_hicolor_pngs(raw)
+ if resized is not None:
+ wrote = _write_hicolor_pngs(resized)
+ else:
+ dest = _hicolor_icon_dest(_hicolor_subdir(_png_dimensions(raw)))
+ wrote = True
+ if dest.is_file() and dest.read_bytes() == raw:
+ wrote = False
+ else:
+ dest.parent.mkdir(parents=True, exist_ok=True)
+ shutil.copyfile(icon, dest)
+ removed_stale = _remove_stale_scalable_icon()
+ if wrote or removed_stale:
+ _refresh_hicolor_cache()
return True
except OSError:
return False
diff --git a/hermes_cli/local_runtime/__init__.py b/hermes_cli/local_runtime/__init__.py
new file mode 100644
index 0000000000..9793970b5f
--- /dev/null
+++ b/hermes_cli/local_runtime/__init__.py
@@ -0,0 +1,55 @@
+"""Managed llama.cpp runtime.
+
+Hermes downloads, verifies, supervises, and updates one llama-server, and
+decides per machine which model build and context window to run. Key
+modules:
+
+- ``binaries`` — resolve/download/verify official llama.cpp release zips
+ into ``$HERMES_HOME/runtimes/llamacpp//``.
+- ``supervisor``— spawn and supervise one llama-server in router mode;
+ readiness is a touch generation, never health-200 alone.
+- ``detect`` — find an already-running llama-server (external or ours).
+- ``estimator`` / ``context_policy`` / ``growth`` — price context memory
+ per architecture and run the window ladder (zero-spill start, grow
+ toward native max, compress only at the top).
+- ``catalog`` / ``presets`` — the curated model list and the per-model
+ launch flags that carry policy decisions to the router.
+
+Everything is driven by the ``local_runtime`` section of config.yaml.
+"""
+
+from hermes_cli.local_runtime.binaries import ( # noqa: F401
+ BinaryResolutionError,
+ ensure_runtime_installed,
+ resolve_assets,
+ select_backend,
+)
+from hermes_cli.local_runtime.bootstrap import ( # noqa: F401
+ ensure_local_runtime,
+ shutdown_local_runtime,
+)
+from hermes_cli.local_runtime.context_policy import ( # noqa: F401
+ FLOOR,
+ growth_decision,
+ initial_window,
+ ladder,
+ launch_args,
+)
+from hermes_cli.local_runtime.growth import ( # noqa: F401
+ clear_window_override,
+ load_window_overrides,
+ maybe_grow_window,
+ save_window_override,
+)
+from hermes_cli.local_runtime.detect import detect_server # noqa: F401
+from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint # noqa: F401
+from hermes_cli.local_runtime.estimator import ( # noqa: F401
+ HardwareBudget,
+ ctx_bytes,
+ physics_check,
+ profile_from_gguf,
+)
+from hermes_cli.local_runtime.gguf import read_gguf_header # noqa: F401
+from hermes_cli.local_runtime.hardware import probe_budget # noqa: F401
+from hermes_cli.local_runtime.presets import generate_presets # noqa: F401
+from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor # noqa: F401
diff --git a/hermes_cli/local_runtime/binaries.py b/hermes_cli/local_runtime/binaries.py
new file mode 100644
index 0000000000..bcd7d32a1d
--- /dev/null
+++ b/hermes_cli/local_runtime/binaries.py
@@ -0,0 +1,351 @@
+"""Binary acquisition for the managed llama.cpp runtime.
+
+llama.cpp publishes per-tag assets (rolling ``bNNNN`` tags, no semver).
+Backends are dlopen'd plugins, so a runtime = CPU/base zip + backend zip
+extracted into one directory, plus the cudart runtime zip on Windows CUDA
+(end users have no CUDA toolkit). We pin the tag in config, sha256-verify
+every download, and keep the previous tag for rollback (N-1).
+
+Layout: ``$HERMES_HOME/runtimes/llamacpp///``
+with a ``manifest.json`` recording zips, sha256s, and the verified
+llama-server version string.
+"""
+
+from __future__ import annotations
+
+import hashlib
+import json
+import logging
+import platform
+import shutil
+import subprocess
+import urllib.request
+import zipfile
+from dataclasses import dataclass, field
+from pathlib import Path
+from typing import Callable
+
+from hermes_constants import get_hermes_home
+
+logger = logging.getLogger(__name__)
+
+RELEASE_URL = "https://github.com/ggml-org/llama.cpp/releases/download/{tag}/{asset}"
+
+# Windows CUDA zips ship per CUDA major; the runtime zip must be paired with
+# its cudart zip so end users need no toolkit. 13.3 verified on 13.1 and
+# 13.2 drivers.
+_WIN_CUDA_VERSION = "13.3"
+# arm64 Windows CUDA prebuilts landed upstream (~b1036x) on CUDA 13.4 —
+# verified against live asset lists (b10362, b10630, b10679). Tags at or before
+# b10290 don't have them; resolution succeeds and the download 404s
+# honestly on such tags, which only arises if a user pins backward.
+_WIN_CUDA_VERSION_ARM64 = "13.4"
+
+
+# Fallback when the config section is missing entirely (deep-merge normally
+# guarantees the key). Single source: DEFAULT_CONFIG owns the shipped tag.
+def default_tag() -> str:
+ from hermes_cli.config_defaults import DEFAULT_CONFIG
+
+ return DEFAULT_CONFIG["local_runtime"]["tag"]
+
+
+class BinaryResolutionError(RuntimeError):
+ """No usable asset combination for this platform/backend."""
+
+
+@dataclass
+class AssetPlan:
+ """The exact zips one runtime install needs, in extraction order."""
+
+ tag: str
+ backend: str # cuda | metal | vulkan | hip | cpu
+ assets: list[str] = field(default_factory=list)
+
+ @property
+ def install_dir(self) -> Path:
+ return runtimes_root() / self.tag / self.backend
+
+
+def runtimes_root() -> Path:
+ """Machine-scoped, deliberately NOT profile-scoped. Engine binaries,
+ presets, and server state describe this machine's hardware and its one
+ managed server (stable port) — a second profile re-downloading the
+ engine or fighting over the port would be the bug. Profile-scoped
+ things (which model is the default, enabled) live in each profile's
+ config.yaml as ever."""
+ from hermes_constants import get_default_hermes_root
+
+ return get_default_hermes_root() / "runtimes" / "llamacpp"
+
+
+def installed_tags() -> list[str]:
+ """Tags with a verified install (manifest carries verified_version),
+ newest first by release number. The boot ladder and the update check
+ both read installed-ness from here — one resolver, every caller."""
+ root = runtimes_root()
+ if not root.exists():
+ return []
+ found: list[str] = []
+ for entry in root.iterdir():
+ if not entry.is_dir() or entry.name == "downloads":
+ continue
+ for manifest in entry.glob("*/manifest.json"):
+ try:
+ if json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"):
+ found.append(entry.name)
+ break
+ except (json.JSONDecodeError, OSError):
+ continue
+
+ def _release_number(tag: str) -> int:
+ digits = "".join(ch for ch in tag if ch.isdigit())
+ return int(digits) if digits else 0
+
+ return sorted(set(found), key=_release_number, reverse=True)
+
+
+def _host_os_arch() -> tuple[str, str]:
+ """(os, arch) normalized to release-asset vocabulary.
+
+ PITFALL: PROCESSOR_ARCHITECTURE lies under x64 emulation on
+ ARM64 Windows. platform.machine() reads the same env on some Pythons, so
+ on Windows prefer PROCESSOR_IDENTIFIER's text when present.
+ """
+ system = platform.system().lower()
+ os_name = {"windows": "win", "darwin": "macos", "linux": "ubuntu"}.get(system, system)
+ machine = platform.machine().lower()
+ arch = "arm64" if machine in ("arm64", "aarch64") else "x64"
+ if os_name == "win":
+ import os as _os
+ ident = _os.environ.get("PROCESSOR_IDENTIFIER", "")
+ if "armv8" in ident.lower() or "arm " in ident.lower():
+ arch = "arm64"
+ return os_name, arch
+
+
+def select_backend(gpu_vendor: str | None, os_name: str | None = None) -> str:
+ """Backend choice per design: CUDA if NVIDIA, Metal on macOS, Vulkan if
+ a non-NVIDIA GPU is present, else CPU. ``--list-devices`` validates the
+ choice post-install; the supervisor's touch generation is ground truth."""
+ if os_name is None:
+ os_name, _ = _host_os_arch()
+ if os_name == "macos":
+ return "metal"
+ vendor = (gpu_vendor or "").lower()
+ if "nvidia" in vendor:
+ return "cuda"
+ if vendor in ("amd", "intel") or "radeon" in vendor or "arc" in vendor:
+ return "vulkan"
+ return "cpu"
+
+
+def resolve_assets(tag: str, backend: str, os_name: str | None = None,
+ arch: str | None = None) -> AssetPlan:
+ """Compose the asset list for (tag, backend, platform).
+
+ Raises BinaryResolutionError for combinations the release does not ship
+ (a platform/backend pair upstream publishes no artifact for). Callers
+ fall back down the backend ladder: cuda -> vulkan -> cpu.
+ """
+ host_os, host_arch = _host_os_arch()
+ os_name = os_name or host_os
+ arch = arch or host_arch
+ plan = AssetPlan(tag=tag, backend=backend)
+
+ if os_name == "macos":
+ # macOS tarballs are unified (Metal built in).
+ plan.assets = [f"llama-{tag}-bin-macos-{arch}.tar.gz"]
+ return plan
+
+ if os_name == "ubuntu":
+ if backend == "cuda":
+ # No prebuilt Linux CUDA zips at current tags — Linux CUDA users
+ # build from source or use vulkan; resolver is honest about it.
+ raise BinaryResolutionError(
+ f"no prebuilt linux CUDA asset at {tag}; use vulkan/cpu or a source build")
+ suffix = {"vulkan": f"vulkan-{arch}", "hip": f"rocm-7.2-{arch}",
+ "cpu": arch}.get(backend)
+ if suffix is None:
+ raise BinaryResolutionError(f"unsupported linux backend {backend}")
+ plan.assets = [f"llama-{tag}-bin-ubuntu-{suffix}.tar.gz"]
+ return plan
+
+ if os_name == "win":
+ if backend == "cuda":
+ cuda_ver = _WIN_CUDA_VERSION_ARM64 if arch == "arm64" else _WIN_CUDA_VERSION
+ plan.assets = [
+ f"llama-{tag}-bin-win-cuda-{cuda_ver}-{arch}.zip",
+ f"cudart-llama-bin-win-cuda-{cuda_ver}-{arch}.zip",
+ ]
+ elif backend == "vulkan":
+ if arch == "arm64":
+ raise BinaryResolutionError(f"no win-vulkan-arm64 asset at {tag}")
+ plan.assets = [f"llama-{tag}-bin-win-vulkan-x64.zip"]
+ elif backend == "hip":
+ plan.assets = [f"llama-{tag}-bin-win-hip-radeon-x64.zip"]
+ elif backend == "cpu":
+ plan.assets = [f"llama-{tag}-bin-win-cpu-{arch}.zip"]
+ else:
+ raise BinaryResolutionError(f"unsupported windows backend {backend}")
+ return plan
+
+ raise BinaryResolutionError(f"unsupported platform {os_name}-{arch}")
+
+
+def _sha256(path: Path) -> str:
+ h = hashlib.sha256()
+ with open(path, "rb") as f:
+ for chunk in iter(lambda: f.read(1 << 22), b""):
+ h.update(chunk)
+ return h.hexdigest()
+
+
+def _download(url: str, dest: Path,
+ progress: "Callable[[int, int], None] | None" = None) -> None:
+ """Stream url -> dest. ``progress(done_bytes, total_bytes)`` ticks per
+ chunk (total 0 when the server sends no Content-Length) — a several-
+ hundred-MB archive on a slow line must never look hung."""
+ logger.info("downloading %s", url)
+ tmp = dest.with_suffix(dest.suffix + ".part")
+ with urllib.request.urlopen(url, timeout=120) as r, open(tmp, "wb") as f:
+ total = int(r.headers.get("Content-Length") or 0)
+ done = 0
+ while True:
+ chunk = r.read(1 << 20)
+ if not chunk:
+ break
+ f.write(chunk)
+ done += len(chunk)
+ if progress is not None:
+ progress(done, total)
+ tmp.replace(dest)
+
+
+def _extract(archive: Path, dest: Path,
+ progress: "Callable[[int, int], None] | None" = None) -> None:
+ """Extract member by member so ``progress(done, total)`` can tick in
+ uncompressed bytes — big archives take real time on laptop disks."""
+ if archive.name.endswith(".zip"):
+ with zipfile.ZipFile(archive) as z:
+ members = z.infolist()
+ total = sum(m.file_size for m in members)
+ done = 0
+ for m in members:
+ z.extract(m, dest)
+ done += m.file_size
+ if progress is not None:
+ progress(done, total)
+ else:
+ import tarfile
+ with tarfile.open(archive) as t:
+ members = t.getmembers()
+ total = sum(m.size for m in members)
+ done = 0
+ for m in members:
+ t.extract(m, dest, filter="data")
+ done += m.size
+ if progress is not None:
+ progress(done, total)
+
+
+def server_binary(install_dir: Path) -> Path:
+ """Locate llama-server within an extracted runtime (zips differ in
+ whether they nest a build/bin directory)."""
+ names = ("llama-server.exe", "llama-server")
+ for name in names:
+ direct = install_dir / name
+ if direct.exists():
+ return direct
+ for name in names:
+ hits = sorted(install_dir.rglob(name))
+ if hits:
+ return hits[0]
+ raise BinaryResolutionError(f"llama-server not found under {install_dir}")
+
+
+def verify_install(install_dir: Path, tag: str) -> str:
+ """Run --version; require the tag's build number in the output.
+ (The binary prints the tag WITHOUT the 'b' prefix.)"""
+ exe = server_binary(install_dir)
+ out = subprocess.run([str(exe), "--version"], capture_output=True,
+ text=True, encoding="utf-8", errors="replace",
+ timeout=60, cwd=str(exe.parent))
+ text = (out.stdout + out.stderr).strip()
+ if tag.lstrip("b") not in text:
+ raise BinaryResolutionError(
+ f"version check failed for {exe}: expected {tag}, got: {text[:120]}")
+ return text.splitlines()[0] if text else ""
+
+
+def prune_old_tags(keep: list[str]) -> None:
+ """Retain only the tags in ``keep`` (current + previous — N-1 rollback).
+ The shared ``downloads/`` archive cache is not a tag and always survives."""
+ root = runtimes_root()
+ if not root.exists():
+ return
+ for entry in root.iterdir():
+ if entry.is_dir() and entry.name != "downloads" and entry.name not in keep:
+ shutil.rmtree(entry, ignore_errors=True)
+ logger.info("pruned old runtime %s", entry.name)
+
+
+def ensure_runtime_installed(tag: str, backend: str,
+ expected_sha256: dict[str, str] | None = None,
+ progress: "Callable[[str, int, int, str], None] | None" = None) -> Path:
+ """Idempotent: resolve, download, verify, extract, version-check.
+
+ ``expected_sha256`` maps asset name -> hash when the catalog pins them;
+ without pins the computed hash is recorded in the manifest (trust on
+ first download, verified on every reinstall).
+ ``progress(stage, done_bytes, total_bytes, label)`` ticks through the
+ slow parts — stage is "download" | "extract" | "verify", label is the
+ asset counter ("1/2") when the plan has several archives.
+ Returns the install directory containing llama-server.
+ """
+ plan = resolve_assets(tag, backend)
+ install_dir = plan.install_dir
+ manifest_path = install_dir / "manifest.json"
+ if manifest_path.exists():
+ try:
+ manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
+ if manifest.get("verified_version"):
+ return install_dir
+ except (json.JSONDecodeError, OSError):
+ pass # damaged manifest -> reinstall
+
+ install_dir.mkdir(parents=True, exist_ok=True)
+ downloads = runtimes_root() / "downloads"
+ downloads.mkdir(parents=True, exist_ok=True)
+
+ recorded: dict[str, str] = {}
+ n_assets = len(plan.assets)
+ for i, asset in enumerate(plan.assets, 1):
+ label = f"{i}/{n_assets}" if n_assets > 1 else ""
+ archive = downloads / asset
+ if not archive.exists():
+ _download(RELEASE_URL.format(tag=tag, asset=asset), archive,
+ progress=(lambda d, t, _l=label: progress("download", d, t, _l))
+ if progress is not None else None)
+ if progress is not None:
+ progress("verify", 0, 0, label)
+ digest = _sha256(archive)
+ expected = (expected_sha256 or {}).get(asset)
+ if expected and digest != expected:
+ archive.unlink(missing_ok=True)
+ raise BinaryResolutionError(
+ f"sha256 mismatch for {asset}: expected {expected}, got {digest}")
+ recorded[asset] = digest
+ _extract(archive, install_dir,
+ progress=(lambda d, t, _l=label: progress("extract", d, t, _l))
+ if progress is not None else None)
+
+ if progress is not None:
+ progress("verify", 0, 0, "")
+ version = verify_install(install_dir, tag)
+ manifest_path.write_text(json.dumps({
+ "tag": tag, "backend": plan.backend, "assets": recorded,
+ "verified_version": version,
+ }, indent=2), encoding="utf-8")
+ logger.info("installed llama.cpp %s (%s): %s", tag, backend, version)
+ return install_dir
diff --git a/hermes_cli/local_runtime/bootstrap.py b/hermes_cli/local_runtime/bootstrap.py
new file mode 100644
index 0000000000..79402142ae
--- /dev/null
+++ b/hermes_cli/local_runtime/bootstrap.py
@@ -0,0 +1,337 @@
+"""Bootstrap for the managed runtime: config -> installed binaries ->
+running supervised server.
+
+One public call, ``ensure_local_runtime(config)``, safe to call at any
+session start:
+- disabled or already-running (state file answers /health) -> no-op
+- enabled -> install binaries if missing (idempotent), spawn supervisor
+
+Kept import-light: callers gate on config before importing this module so
+sessions with local_runtime disabled never pay the import.
+"""
+
+from __future__ import annotations
+
+import logging
+import os
+import subprocess
+import time
+from pathlib import Path
+
+from hermes_constants import get_hermes_home # noqa: F401 — config paths
+
+from hermes_cli.local_runtime.binaries import runtimes_root
+
+logger = logging.getLogger(__name__)
+
+_SUPERVISOR = None # process-wide singleton; one router per Hermes process
+
+
+def _detect_gpu_vendor() -> str | None:
+ """Best-effort GPU vendor for backend selection. NVIDIA via nvidia-smi
+ (resolved by the hardware probe's PATH-independent ladder — a stripped
+ service PATH must not demote an NVIDIA box to vulkan/cpu); anything
+ else defers to select_backend's fallback ladder."""
+ from hermes_cli.local_runtime.hardware import _nvidia_smi_path
+
+ smi = _nvidia_smi_path()
+ if smi is None:
+ return None
+ try:
+ out = subprocess.run(
+ [smi, "--query-gpu=name", "--format=csv,noheader"],
+ capture_output=True, text=True, timeout=10)
+ if out.returncode == 0 and out.stdout.strip():
+ return "nvidia " + out.stdout.strip().splitlines()[0]
+ except (OSError, subprocess.TimeoutExpired):
+ pass
+ return None
+
+
+def models_dir() -> Path:
+ """Machine-scoped, deliberately NOT profile-scoped: a 20 GB GGUF is a
+ machine asset, and every profile shares the one managed server that
+ serves it. See runtimes_root() for the same rule on the engine."""
+ from hermes_constants import get_default_hermes_root
+
+ return get_default_hermes_root() / "models"
+
+
+def assets_dir() -> Path:
+ """Non-model companion files (mmproj vision projectors, spec-decode
+ draft models). A subdirectory so the router's model listing — and our
+ staged_models() — never mistakes an asset for a servable model."""
+ return models_dir() / "assets"
+
+
+def staged_models() -> "list[Path]":
+ """Servable staged models: single-file GGUFs count when present; a
+ split GGUF counts once, by its first part, and only when EVERY part
+ is on disk — a mid-download split is not servable and must not
+ surface anywhere as a model. Continuation parts and assets/ never
+ count."""
+ import re
+
+ part = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$")
+ files = sorted(models_dir().glob("*.gguf"))
+ names = {p.name for p in files}
+ out = []
+ for p in files:
+ m = part.search(p.name)
+ if m is None:
+ out.append(p)
+ continue
+ if m.group(1) != "00001":
+ continue
+ stem = p.name[: m.start()]
+ total = int(m.group(2))
+ if all(f"{stem}-{i:05d}-of-{m.group(2)}.gguf" in names
+ for i in range(2, total + 1)):
+ out.append(p)
+ return out
+
+
+def staged_model_ids() -> "list[str]":
+ import re
+
+ return [re.sub(r"-\d{5}-of-\d{5}$", "", p.stem) for p in staged_models()]
+
+
+def _presets_stale() -> bool:
+ """True when a staged model has no section in the preset INI — it
+ would autoload with stock fit instead of a policy decision."""
+ try:
+ from hermes_cli.local_runtime.presets import read_preset_decisions
+
+ known = set(read_preset_decisions())
+ return any(mid not in known for mid in staged_model_ids())
+ except Exception: # noqa: BLE001
+ return False
+
+
+def _stop_state_server(state: dict) -> None:
+ """Best-effort stop of the server the state file points at (an
+ incumbent this process doesn't supervise). The state pid is ours by
+ contract — the file only ever describes the managed server."""
+ from hermes_cli.local_runtime.endpoint import _pid_alive
+
+ pid = state.get("pid")
+ try:
+ pid = int(pid)
+ except (TypeError, ValueError):
+ return
+ if pid <= 0:
+ return
+ try:
+ import signal
+
+ os.kill(pid, signal.SIGTERM)
+ except (OSError, ValueError):
+ return
+ # Give it a moment to release the port and the GPU. Liveness via
+ # psutil — on Windows os.kill(pid, 0) TERMINATES the process, it is
+ # not a probe (the endpoint.py pitfall note; #local-models review).
+ for _ in range(50):
+ if not _pid_alive(pid):
+ return
+ time.sleep(0.1)
+
+
+def refresh_local_runtime() -> bool:
+ """Restart the managed server so it rescans the models directory.
+
+ The router's model list is SPAWN-ONLY: a GGUF added after start is
+ invisible to GET /models and 400s on completion, so anything that
+ changes the staged set while the server runs must bounce it. Covers
+ both ownership shapes: a supervised server restarts in-process; an
+ ADOPTED server (started by a previous backend session — the normal
+ shape after any restart) is stopped via its state-file pid and
+ replaced with a supervised boot. Without the adopted branch, every
+ download/delete in a post-restart session silently no-ops the bounce
+ and the router serves a stale catalog. Returns False when there is
+ nothing to refresh (no server anywhere; next boot scans fresh).
+ """
+ global _SUPERVISOR
+ try:
+ from hermes_cli.config import load_config
+
+ if _SUPERVISOR is None:
+ from hermes_cli.local_runtime.endpoint import _state_endpoint
+
+ state = _state_endpoint()
+ if state is None:
+ return False
+ logger.info("bouncing adopted llama-server (pid=%s) to rescan models",
+ state.get("pid"))
+ _stop_state_server(state)
+ else:
+ shutdown_local_runtime()
+ return ensure_local_runtime(load_config(), force=True) is not None
+ except Exception as exc: # noqa: BLE001
+ logger.warning("local runtime refresh failed: %s", exc)
+ return False
+
+
+def ensure_local_runtime(config: dict, force: bool = False) -> "object | None":
+ """Idempotent boot of the managed runtime. Returns the supervisor (or
+ None when disabled/unavailable). Never raises into a session start —
+ failures log and return None; chat falls back to configured providers.
+
+ ``force=True`` skips the enabled gate — used by the explicit "Use this
+ model" action, where the click IS the opt-in (the caller records it in
+ config so future boots auto-start).
+ """
+ global _SUPERVISOR
+ section = (config or {}).get("local_runtime") or {}
+ if not force and not section.get("enabled"):
+ return None
+ if _SUPERVISOR is not None:
+ return _SUPERVISOR
+
+ # Residency: no staged models means nothing to serve — don't boot an
+ # empty server. The walked-away story handled with zero configuration
+ # (delete your last model and boots stop); Use force-boots as ever.
+ if not force and not staged_models():
+ logger.info("local runtime enabled but no models staged; not booting")
+ return None
+
+ # Another Hermes process may already be supervising — reuse via state,
+ # but ONLY while its launch policy still covers every staged model. A
+ # server whose preset file predates a download serves the new model
+ # with no policy at all (--models-autoload + stock fit: f16 KV at max
+ # context, no placement — the silent-demotion busy-wait on WDDM). A
+ # stale incumbent gets stopped and replaced by a fresh boot with
+ # regenerated presets; sessions ride through exactly like any other
+ # supervised restart (stable port + persisted key).
+ from hermes_cli.local_runtime.endpoint import _state_endpoint
+
+ state = _state_endpoint()
+ if state is not None:
+ if not _presets_stale():
+ logger.info("managed llama-server already running (another process)")
+ return None
+ logger.info("running server's presets predate the staged models; "
+ "replacing it so every model launches with a policy")
+ _stop_state_server(state)
+
+ try:
+ from hermes_cli.local_runtime.binaries import (
+ ensure_runtime_installed,
+ select_backend,
+ )
+ from hermes_cli.local_runtime.hardware import probe_budget
+ from hermes_cli.local_runtime.presets import generate_presets
+ from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
+
+ backend = section.get("backend", "auto")
+ if backend == "auto":
+ backend = select_backend(_detect_gpu_vendor())
+ # Boot ladder: serve what is INSTALLED, never download here. The
+ # configured tag (config root-of-trust; deep-merge supplies the
+ # Hermes-release default when unpinned) is preferred; when it isn't
+ # installed yet, the newest installed tag serves and the status
+ # endpoint reports the pending update — the download is a deliberate
+ # button click in the pane, not a boot-path surprise (a multi-minute
+ # inline download here is exactly how the onboarding bounce returns).
+ from hermes_cli.local_runtime.binaries import default_tag, installed_tags
+
+ tag = section.get("tag") or default_tag()
+ have = installed_tags()
+ if tag not in have:
+ if not have:
+ logger.info("local runtime enabled but no build installed; "
+ "install happens in the Local Models pane")
+ return None
+ logger.info("configured tag %s not installed; serving %s "
+ "(update is a click in Local Models)", tag, have[0])
+ tag = have[0]
+ install_dir = ensure_runtime_installed(tag, backend)
+
+ mdir = models_dir()
+ mdir.mkdir(parents=True, exist_ok=True)
+
+ # Context policy: one launch decision per staged model, carried to
+ # the router via the preset INI. Priced against CAPACITY, not live
+ # free VRAM: this runs while the outgoing server instance may still
+ # hold the card (restart, refresh after a download), and its memory
+ # is freed before the new instance loads anything. Pricing against
+ # live-free here once pinned a fitting model's weights to CPU
+ # because the probe saw the predecessor's VRAM as gone.
+ preset_path = runtimes_root() / "presets.ini"
+ try:
+ entries = generate_presets(mdir, probe_budget(planning=True), preset_path)
+ for entry in entries:
+ if entry.refusal:
+ logger.warning("model refused by physics check: %s", entry.refusal)
+ except Exception as exc: # noqa: BLE001 — policy failure must not block serving
+ # Degradation ladder: a STALE policy still beats no policy —
+ # stock fit (f16 KV at max context, no placement) is the
+ # silent-busy-wait failure on Windows. Keep serving with the
+ # previous INI when one exists; only a first boot with no INI
+ # at all falls to stock fit.
+ if preset_path.exists():
+ logger.error("preset generation failed (%s); serving with the "
+ "PREVIOUS launch policies — models staged since "
+ "the last successful generation run unpoliced "
+ "until this is fixed", exc)
+ else:
+ logger.error("preset generation failed (%s) and no previous "
+ "policy file exists; router runs stock fit", exc)
+ preset_path = None
+
+ sup = LlamaServerSupervisor(
+ install_dir, mdir,
+ models_max=int(section.get("models_max", 4)),
+ port=int(section.get("port", 0)) or None,
+ preset_path=preset_path,
+ )
+ try:
+ sup.start()
+ except Exception:
+ # start() can fail after the router process exists (health
+ # timeout, spawn error): leaving it running unsupervised
+ # strands its VRAM behind a port nothing will clean up.
+ try:
+ sup.stop()
+ except Exception: # noqa: BLE001 — cleanup is best-effort
+ pass
+ raise
+ _SUPERVISOR = sup
+ logger.info("managed llama-server up at %s (backend=%s tag=%s)",
+ sup.base_url, backend, tag)
+ _start_idle_sweeper(sup)
+ return sup
+ except Exception as exc: # noqa: BLE001 — never break session start
+ logger.warning("managed local runtime unavailable: %s", exc)
+ return None
+
+
+def shutdown_local_runtime() -> None:
+ global _SUPERVISOR
+ if _SUPERVISOR is not None:
+ _SUPERVISOR.stop()
+ _SUPERVISOR = None
+
+
+def get_supervisor():
+ """The process-local supervisor, or None (server may still be running
+ under another process — check the state file)."""
+ return _SUPERVISOR
+
+
+def _start_idle_sweeper(sup) -> None:
+ """Idle-residency loop: every couple of minutes, unload non-primary
+ models idle past the supervisor's threshold. Daemon thread tied to the
+ supervisor's lifetime — exits when the server stops."""
+ import threading
+
+ def _loop():
+ while sup.proc is not None and sup.proc.poll() is None:
+ time.sleep(120)
+ try:
+ sup.sweep_idle()
+ except Exception as exc: # noqa: BLE001
+ logger.debug("idle sweep skipped: %s", exc)
+
+ threading.Thread(target=_loop, daemon=True,
+ name="local-runtime-idle-sweep").start()
diff --git a/hermes_cli/local_runtime/capabilities.py b/hermes_cli/local_runtime/capabilities.py
new file mode 100644
index 0000000000..b30152fadf
--- /dev/null
+++ b/hermes_cli/local_runtime/capabilities.py
@@ -0,0 +1,118 @@
+"""Capability answers for models served by the managed runtime.
+
+Capability lookups (vision, and whatever comes next) consult cloud-shaped
+catalogs that have never heard of a local GGUF, so a vision-capable local
+model reads as text-only and images detour to an auxiliary cloud model —
+the wrong behavior twice over for a local-first user (broken feature, and
+a screenshot silently leaving the machine).
+
+The managed runtime can answer from ground truth instead, best source
+first:
+
+1. The RUNNING child's /props: llama-server reports a ``modalities`` block
+ when a vision projector is loaded. The server that will receive the
+ image says whether it can see — no inference, no catalog.
+2. The catalog entry's declared capability (the ``vision`` tag + mmproj
+ asset) for staged-but-unloaded models: what the model WILL support once
+ its projector loads beside it.
+3. None — not one of ours, or nothing known; the caller falls through to
+ its other sources.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import urllib.request
+
+logger = logging.getLogger(__name__)
+
+_LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"})
+
+# Image formats the managed server's decoder actually handles. llama.cpp
+# decodes with stb_image: PNG/JPEG/GIF/BMP yes, WebP NO — and a WebP part
+# fails SILENTLY (no HTTP error, no log line; the model just never sees an
+# image and confabulates a description). Anything outside this set must be
+# transcoded before the request. Measured against the live server: the
+# same red square answered 'Red' as PNG and 'Unseen' as WebP.
+ACCEPTED_IMAGE_MIMES = frozenset({"image/png", "image/jpeg"})
+
+
+def is_managed_provider(provider: str, base_url: str = "") -> bool:
+ """True when this provider/base_url pair points at the managed server.
+ ``custom`` only counts when the base_url IS the managed endpoint —
+ background lookups must never claim someone else's custom server."""
+ p = (provider or "").strip().lower()
+ if p in _LLAMACPP_ALIASES:
+ return True
+ if p == "custom" and base_url:
+ try:
+ from hermes_cli.local_runtime.growth import is_managed_endpoint
+
+ return is_managed_endpoint(base_url)
+ except Exception: # noqa: BLE001
+ return False
+ return False
+
+
+def _props_modalities(model_id: str) -> "bool | None":
+ """Ask the running server whether this loaded child sees images.
+ None when the server is down, the model isn't loaded, or the build
+ doesn't report modalities."""
+ try:
+ from hermes_cli.local_runtime.endpoint import _state_endpoint
+
+ state = _state_endpoint()
+ if state is None:
+ return None
+ base = state["base_url"].rsplit("/v1", 1)[0]
+ req = urllib.request.Request(
+ f"{base}/props?model={model_id}",
+ headers={"Authorization": f"Bearer {state.get('api_key', '')}"})
+ with urllib.request.urlopen(req, timeout=3) as r:
+ props = json.load(r)
+ modalities = props.get("modalities")
+ if isinstance(modalities, dict) and "vision" in modalities:
+ return bool(modalities["vision"])
+ return None
+ except Exception: # noqa: BLE001
+ return None
+
+
+def managed_model_supports_vision(model_id: str) -> "bool | None":
+ """Ground-truth vision capability for a staged model, or None when the
+ model isn't ours / nothing is known (caller keeps falling through)."""
+ if not model_id:
+ return None
+
+ # Only answer for models actually staged with us.
+ try:
+ from hermes_cli.local_runtime.bootstrap import staged_model_ids
+
+ if model_id not in staged_model_ids():
+ return None
+ except Exception: # noqa: BLE001
+ return None
+
+ live = _props_modalities(model_id)
+ if live is not None:
+ return live
+
+ # Staged but not loaded (or an older server build): the catalog knows
+ # whether this model ships a vision projector.
+ try:
+ from hermes_cli.local_runtime.bootstrap import assets_dir
+ from hermes_cli.local_runtime.catalog import find_entry_for_model
+
+ hit = find_entry_for_model(model_id)
+ if hit is None:
+ return None
+ entry = hit[0]
+ if entry.mmproj is None:
+ return False
+ # Capability requires the projector to actually be on disk — a
+ # model downloaded before its mmproj (partial delete, old layout)
+ # genuinely cannot see.
+ return (assets_dir() / entry.mmproj.local_name).exists()
+ except Exception: # noqa: BLE001
+ return None
diff --git a/hermes_cli/local_runtime/catalog.json b/hermes_cli/local_runtime/catalog.json
new file mode 100644
index 0000000000..587606e633
--- /dev/null
+++ b/hermes_cli/local_runtime/catalog.json
@@ -0,0 +1,174 @@
+{
+ "schema_version": 1,
+ "models": [
+ {
+ "id": "qwen3.8-27b",
+ "display_name": "Qwen3.8 27B",
+ "description": "Best all-round agent model; sees images; long context stays fast",
+ "repo": "unsloth/Qwen3.8-27B-GGUF",
+ "variants": [
+ {
+ "quant": "UD-Q4_K_M",
+ "files": [
+ {
+ "path": "Qwen3.8-27B-UD-Q4_K_M.gguf",
+ "size_bytes": 16464440224
+ }
+ ]
+ }
+ ],
+ "n_ctx_train": 262144,
+ "full_layers": 16,
+ "recurrent_layers": 48,
+ "per_layer_f16": 4096,
+ "n_vocab": 248320,
+ "mmproj": {
+ "path": "mmproj-BF16.gguf",
+ "size_bytes": 931146432,
+ "local": "mmproj-Qwen3.8-27B-BF16.gguf"
+ },
+ "mtp": true,
+ "mtp_draft_depth": 2,
+ "sampling": {
+ "temp": "1.0",
+ "top-p": "0.95",
+ "top-k": "20",
+ "min-p": "0.0"
+ },
+ "quality": 90,
+ "decode_fraction": 1.0
+ },
+ {
+ "id": "qwen3.8-flash-next",
+ "display_name": "Qwen3.8 Flash Next",
+ "description": "Frontier-scale model; needs a very large GPU to run well",
+ "repo": "unsloth/Qwen3.8-Flash-Next-GGUF",
+ "variants": [
+ {
+ "quant": "UD-Q4_K_XL",
+ "files": [
+ {
+ "path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00001-of-00004.gguf",
+ "size_bytes": 10946624
+ },
+ {
+ "path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00002-of-00004.gguf",
+ "size_bytes": 49859583136
+ },
+ {
+ "path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00003-of-00004.gguf",
+ "size_bytes": 49376141504
+ },
+ {
+ "path": "UD-Q4_K_XL/Qwen3.8-Flash-Next-UD-Q4_K_XL-00004-of-00004.gguf",
+ "size_bytes": 12087983520
+ }
+ ]
+ }
+ ],
+ "n_ctx_train": 262144,
+ "full_layers": 12,
+ "recurrent_layers": 36,
+ "per_layer_f16": 2048,
+ "moe": true,
+ "n_vocab": 248320,
+ "mmproj": {
+ "path": "mmproj-BF16.gguf",
+ "size_bytes": 907542944,
+ "local": "mmproj-Qwen3.8-Flash-Next-BF16.gguf"
+ },
+ "min_engine": "b10678",
+ "quality": 95,
+ "decode_fraction": 0.08
+ },
+ {
+ "id": "qwen3.6-35b-a3b",
+ "display_name": "Qwen3.6 35B-A3B",
+ "description": "Bigger mixture-of-experts with multi-token prediction; sees images",
+ "repo": "unsloth/Qwen3.6-35B-A3B-MTP-GGUF",
+ "variants": [
+ {
+ "quant": "UD-Q4_K_M",
+ "files": [
+ {
+ "path": "Qwen3.6-35B-A3B-UD-Q4_K_M.gguf",
+ "size_bytes": 22663387424
+ }
+ ],
+ "validated": true
+ }
+ ],
+ "n_ctx_train": 262144,
+ "full_layers": 10,
+ "recurrent_layers": 30,
+ "per_layer_f16": 2048,
+ "moe": true,
+ "mtp": true,
+ "n_vocab": 248320,
+ "mtp_draft_depth": 2,
+ "mmproj": {
+ "path": "mmproj-BF16.gguf",
+ "size_bytes": 902822528,
+ "local": "mmproj-Qwen3.6-35B-A3B-BF16.gguf"
+ },
+ "sampling": {
+ "temp": "1.0",
+ "top-p": "0.95",
+ "top-k": "20",
+ "min-p": "0.0"
+ },
+ "quality": 80,
+ "decode_fraction": 0.15
+ },
+ {
+ "id": "deepseek-v4-flash",
+ "display_name": "DeepSeek V4 Flash",
+ "description": "Frontier-class model for machines with 128GB+ memory",
+ "repo": "unsloth/DeepSeek-V4-Flash-0731-GGUF",
+ "variants": [
+ {
+ "quant": "UD-Q4_K_XL",
+ "files": [
+ {
+ "path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00001-of-00005.gguf",
+ "size_bytes": 5257408
+ },
+ {
+ "path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00002-of-00005.gguf",
+ "size_bytes": 48935523072
+ },
+ {
+ "path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00003-of-00005.gguf",
+ "size_bytes": 48980787136
+ },
+ {
+ "path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00004-of-00005.gguf",
+ "size_bytes": 49999168416
+ },
+ {
+ "path": "UD-Q4_K_XL/DeepSeek-V4-Flash-0731-UD-Q4_K_XL-00005-of-00005.gguf",
+ "size_bytes": 7174505088
+ }
+ ]
+ }
+ ],
+ "n_ctx_train": 1048576,
+ "full_layers": 43,
+ "recurrent_layers": 0,
+ "per_layer_f16": 1152,
+ "moe": true,
+ "n_vocab": 163840,
+ "draft": {
+ "path": "dspark-DeepSeek-V4-Flash-0731-Q8_0.gguf",
+ "size_bytes": 10896057440
+ },
+ "sampling": {
+ "temp": "1.0",
+ "top-p": "0.95",
+ "min-p": "0.01"
+ },
+ "quality": 85,
+ "decode_fraction": 0.1
+ }
+ ]
+}
diff --git a/hermes_cli/local_runtime/catalog.py b/hermes_cli/local_runtime/catalog.py
new file mode 100644
index 0000000000..715d83c07f
--- /dev/null
+++ b/hermes_cli/local_runtime/catalog.py
@@ -0,0 +1,464 @@
+"""Curated starter catalog for the managed local runtime.
+
+Small and honest: every entry carries the estimator inputs (measured on
+real GGUFs) so the picker can price a model BEFORE the user downloads
+gigabytes. Once a file is on disk, profile_from_gguf() is the authority
+and the catalog numbers are only used for the download decision. Entries
+whose base config is gated upstream carry a same-family conservative
+prior (commented) — the GGUF header corrects it at load time.
+
+Each model ships ONE build, Q4-class (UD-Q4_K_M where the repo has it,
+UD-Q4_K_XL elsewhere). Q4 is the quant class current engines optimize
+for and the sweet spot of the size/quality curve, so there is no quant
+ladder: headroom buys a bigger context window, never a bigger quant,
+and every machine runs the same well-tested build. Below Q4 the quality
+loss is too severe to ship as someone's first local-AI experience; the
+fit policy prices the build honestly (zero-spill, spilled, or refused by
+the physics check).
+
+Validation lifecycle: builds proven end-to-end on real hardware are
+marked validated. Day-0 entries ship before that proof (they simply lack
+the validated flag) — ensure_model_ready's touch generation still gates
+every first load at runtime.
+
+Multi-file models: variants may carry split-GGUF parts (llama-server loads
+from the first part; all parts download together). Entries may carry an
+mmproj (vision projector) and a speculative-decode draft model — both
+download alongside the weights. MTP-integrated models run spec decode
+wherever they load; a separate draft model attaches only when the launch
+decision spills, where its speedup is largest.
+
+File sizes come from HF LFS metadata and feed the estimator, the fit
+pills, and download progress. There is no download-time integrity check
+by design: a corrupt or truncated file surfaces as a llama.cpp
+load error at first use, and the reachability test catches upstream
+re-uploads by size drift before users do.
+
+This is deliberately not a live registry feed: entries are reviewed like a
+version bump (the same policy governs vendor recipe ingestion — parsed
+data, never executed commands).
+
+Vendor recipes overlay: a per-SKU recipes repo may SUPPLEMENT these
+entries where applicable — vendor SKUs only, never the base layer for
+other platforms. A recipe may enrich identity (GGUF/quant/sha), perf
+hints (-b/-ub, spec-decode), and sampling defaults; it never carries
+context/slots/placement/serving flags (the fit policy owns those).
+Resolution: exact SKU -> GPU-class bucket -> fit-only. Snapshot-synced,
+reviewed like a tag bump.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import re
+import threading
+import time
+import urllib.request
+from dataclasses import dataclass, field
+from pathlib import PurePosixPath
+
+from hermes_cli.local_runtime.context_policy import (
+ FLOOR,
+ RUNTIME_OVERHEAD_BYTES,
+ TARGET_WINDOW,
+ ub_logits_bytes,
+)
+from hermes_cli.local_runtime.estimator import (
+ HardwareBudget,
+ LayerKind,
+ ModelProfile,
+ ctx_bytes,
+)
+
+logger = logging.getLogger(__name__)
+
+_GIB = 1 << 30
+_PART_SUFFIX = re.compile(r"-\d{5}-of-\d{5}$")
+
+
+@dataclass(frozen=True)
+class AssetFile:
+ """One downloadable file: repo-relative path and exact bytes (the size
+ feeds the estimator and the download progress bar; there is no
+ download-time integrity check by design — a corrupt file surfaces as a
+ llama.cpp load error). ``local`` overrides the on-disk name (repos
+ reuse generic names like mmproj-BF16.gguf across models). Non-model
+ extras live under the models dir's assets/ subdirectory so the router
+ never lists them."""
+
+ path: str # repo-relative (may include a subdir)
+ size_bytes: int
+ local: str | None = None
+
+ @property
+ def local_name(self) -> str:
+ return self.local or PurePosixPath(self.path).name
+
+
+@dataclass(frozen=True)
+class QuantVariant:
+ """One downloadable build of a model. Split GGUFs list every part in
+ files; the model loads from the first part."""
+
+ quant: str # e.g. "UD-Q4_K_M"
+ files: tuple # AssetFile, first = the load target
+ validated: bool = False # proven end-to-end on real hardware
+
+ @property
+ def model_id(self) -> str:
+ stem = PurePosixPath(self.files[0].path).name.removesuffix(".gguf")
+ return _PART_SUFFIX.sub("", stem)
+
+ @property
+ def size_bytes(self) -> int:
+ return sum(f.size_bytes for f in self.files)
+
+ @property
+ def weights_bytes(self) -> int:
+ """Pre-download weights estimate: GGUF bytes ≈ tensor bytes + a
+ small header (<2%) — a safe, slightly conservative stand-in until
+ profile_from_gguf reads the real table."""
+ return self.size_bytes
+
+
+@dataclass(frozen=True)
+class CatalogEntry:
+ id: str # stable family id (variant-independent)
+ display_name: str
+ description: str # one line, plain language
+ repo: str # HF repo
+ variants: tuple # QuantVariant (exactly one, Q4-class)
+ # Estimator inputs (measured or config-derived; quant changes weights,
+ # never KV). Entries with gated upstream configs carry a conservative
+ # same-family prior — the GGUF header is the authority after download.
+ n_ctx_train: int
+ full_layers: int
+ recurrent_layers: int
+ per_layer_f16: int # KV bytes/token per full-attention layer
+ swa_layers: int = 0
+ swa_window: int = 0
+ moe: bool = False
+ mtp: bool = False # ships MTP heads (spec decode when loaded)
+ # Speculative draft depth for MTP models. Per-model and measured:
+ # deeper drafting pays only while draft acceptance holds, and the
+ # break-even depth differs by model.
+ mtp_draft_depth: int = 3
+ # Vocab size prices the GPU logits buffers (ubatch x vocab x fp32,
+ # doubled under MTP backend sampling) — a multi-GiB term at large
+ # vocab sizes that a weights-only fit would miss.
+ n_vocab: int = 0
+ mmproj: "AssetFile | None" = None # vision projector, downloads with model
+ draft: "AssetFile | None" = None # spec-decode draft model (e.g. DSpark)
+ sampling: dict = field(default_factory=dict) # INI long-form launch defaults
+ # Oldest llama.cpp release tag that can load this model (day-0
+ # architectures need the release where their support landed). Empty
+ # means any installed engine. The pane gates download/activate on it.
+ min_engine: str = ""
+ # Editorial quality ordering (higher = smarter), authored once,
+ # globally, at catalog-authoring time — Artificial Analysis-informed
+ # where they cover the model (scripts/aa_quality_sync.py proposes,
+ # the commit decides), editorial elsewhere. Ranks entries for the
+ # per-machine recommendation; never displayed as a score (it grades
+ # the full-precision model, not our Q4 build).
+ quality: int = 0
+ # Fraction of the build's bytes read per decoded token: 1.0 for dense
+ # models (every weight streams every token), the active slice for MoE
+ # (attention + shared + routed experts over total). With memory
+ # bandwidth this predicts decode speed — the physics half of the
+ # recommendation.
+ decode_fraction: float = 1.0
+
+ def profile(self, variant: QuantVariant) -> ModelProfile:
+ layers = ([(LayerKind.FULL, self.per_layer_f16)] * self.full_layers
+ + [(LayerKind.SWA, self.per_layer_f16)] * self.swa_layers
+ + [(LayerKind.RECURRENT, 0)] * self.recurrent_layers)
+ return ModelProfile(
+ name=variant.model_id, weights_bytes=variant.weights_bytes,
+ embd_table_bytes=0, n_ctx_train=self.n_ctx_train,
+ layers=layers, swa_window=self.swa_window, moe=self.moe,
+ n_vocab=self.n_vocab,
+ kv_scale=1.2 if self.mtp else 1.0)
+
+ def download_files(self, variant: QuantVariant) -> tuple:
+ """Everything a download job fetches for this variant, in order."""
+ extras = tuple(a for a in (self.mmproj, self.draft) if a is not None)
+ return tuple(variant.files) + extras
+
+ def download_bytes(self, variant: QuantVariant) -> int:
+ return sum(f.size_bytes for f in self.download_files(variant))
+
+
+@dataclass(frozen=True)
+class VariantChoice:
+ """Selection result: which build this machine should download and why.
+ reason_key is a UI-copy discriminator, not display text."""
+
+ variant: QuantVariant
+ zero_spill: bool
+ reason_key: str # "best-large-window" | "best-fits" | "smallest-fits-spilled"
+
+
+def select_variant(entry: CatalogEntry, budget: HardwareBudget) -> VariantChoice | None:
+ """Fit the entry's one build (Q4-class) to this machine.
+
+ Every entry ships exactly one variant (see the module docstring for
+ why there is no quant ladder); headroom buys a bigger window, never
+ a bigger quant. The fit shapes:
+
+ - "best-large-window": zero-spills at TARGET_WINDOW
+ - "best-fits": zero-spills at the 64K floor
+ - "smallest-fits-spilled": weights spill to host RAM, priced honestly
+ - None: even spilled, physics refuses (the machine can't run it)
+ """
+ overhead = (RUNTIME_OVERHEAD_BYTES
+ + (entry.mmproj.size_bytes if entry.mmproj else 0)
+ + ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
+ native = entry.n_ctx_train or FLOOR
+ variant = entry.variants[-1]
+ profile = entry.profile(variant)
+ need = variant.weights_bytes + overhead
+ if (need + ctx_bytes(profile, min(TARGET_WINDOW, native))
+ <= budget.usable_vram_bytes):
+ return VariantChoice(variant=variant, zero_spill=True,
+ reason_key="best-large-window")
+ floor_kv = ctx_bytes(profile, min(FLOOR, native))
+ if need + floor_kv <= budget.usable_vram_bytes:
+ return VariantChoice(variant=variant, zero_spill=True,
+ reason_key="best-fits")
+ if need + floor_kv <= budget.usable_vram_bytes + budget.ram_available_bytes:
+ return VariantChoice(variant=variant, zero_spill=False,
+ reason_key="smallest-fits-spilled")
+ return None
+
+
+# ── recommendation: best quality that fits and isn't miserably slow ──
+#
+# Two axes, each living where it belongs. QUALITY is a judgment made once,
+# globally, at authoring time (entry.quality — AA-informed, editorially
+# owned). SPEED is physics computed per machine: decode is memory-bound,
+# so predicted tok/s ≈ bandwidth / bytes-read-per-token, and the bytes per
+# token are the build's size scaled by its decode fraction (dense reads
+# everything; MoE reads the active slice). The pick: highest quality among
+# entries that run resident and clear a pleasant speed floor; else the
+# fastest resident entry; else the least-painful spilled one.
+#
+# The bandwidth axis is the `uma` flag for now: every discrete card that
+# matters is 900+ GB/s GDDR while the unified-memory class measures ~1/5th
+# of that, so the flag IS the high/low split. A measured per-machine
+# bandwidth (one cached memcpy probe) can replace these class constants
+# without touching the rule; predictions order candidates and gate the
+# floor — they are not display values.
+
+_DISCRETE_BANDWIDTH_GB_S = 1000.0 # representative GDDR6X/GDDR7 class
+_UMA_BANDWIDTH_GB_S = 210.0 # measured on unified-memory NVIDIA
+_HOST_BANDWIDTH_GB_S = 80.0 # spilled weights stream over host DRAM
+
+# The one editorial constant in the tree: below this predicted decode
+# speed a model stops feeling pleasant for agentic use (roughly reading
+# speed with headroom for tool-call bursts). Distinct from the growth
+# policy's 6 tok/s compress floor, which marks unusable, not unpleasant.
+PLEASANT_FLOOR_TOK_S = 20.0
+
+
+def predicted_decode_tok_s(entry: CatalogEntry, variant: QuantVariant,
+ budget: HardwareBudget, *,
+ spilled: bool = False) -> float:
+ """Memory-bound decode prediction for ordering and floor-gating."""
+ bandwidth = (_HOST_BANDWIDTH_GB_S if spilled
+ else _UMA_BANDWIDTH_GB_S if budget.uma
+ else _DISCRETE_BANDWIDTH_GB_S)
+ bytes_per_token = max(1.0, variant.size_bytes * entry.decode_fraction)
+ return bandwidth * 1e9 / bytes_per_token
+
+
+def recommended_entry(budget: HardwareBudget,
+ entries: "tuple[CatalogEntry, ...] | None" = None
+ ) -> "tuple[CatalogEntry, str] | None":
+ """The catalog's default pick for THIS machine, with its reason.
+
+ Callers pass pre-filtered entries when some are ineligible for
+ reasons the catalog can't know (engine too old); default is the full
+ catalog. Returns (entry, reason) — the reason is a key the UI turns
+ into the Recommended badge's tooltip, so the rationale shown to the
+ user is the branch that actually fired, never a parallel explanation
+ that can drift:
+
+ best-quality-resident quality won among resident entries that
+ clear the pleasant floor
+ speed-gated-quality same, but the floor eliminated a HIGHER
+ quality candidate — the exact 'why not the
+ big model?' a unified-memory owner asks
+ fastest-resident nothing resident clears the floor; the
+ quickest resident entry wins
+ least-painful-spilled nothing runs resident; fastest from host
+ memory (MoE by construction)
+
+ Returns None only when nothing fits at all.
+ """
+ pool = CATALOG if entries is None else entries
+ fitting: list[tuple[CatalogEntry, VariantChoice]] = []
+ for entry in pool:
+ choice = select_variant(entry, budget)
+ if choice is not None:
+ fitting.append((entry, choice))
+ if not fitting:
+ return None
+
+ resident = [(e, c) for e, c in fitting if c.zero_spill]
+ pleasant = [
+ (e, c) for e, c in resident
+ if predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S
+ ]
+ if pleasant:
+ pick = max(pleasant, key=lambda t: (t[0].quality, -t[1].variant.size_bytes))[0]
+ floor_gated = any(e.quality > pick.quality for e, _ in resident)
+ return (pick, "speed-gated-quality" if floor_gated
+ else "best-quality-resident")
+ if resident:
+ pick = max(resident,
+ key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget))[0]
+ return (pick, "fastest-resident")
+ # Everything spills: take the least painful — fastest predicted decode
+ # from host memory (MoE wins here by construction; a dense spill
+ # streams every weight over the host bus).
+ pick = max(fitting,
+ key=lambda t: predicted_decode_tok_s(t[0], t[1].variant, budget,
+ spilled=True))[0]
+ return (pick, "least-painful-spilled")
+
+
+def recommended_id(budget: HardwareBudget,
+ entries: "tuple[CatalogEntry, ...] | None" = None) -> str | None:
+ picked = recommended_entry(budget, entries)
+ return picked[0].id if picked is not None else None
+
+
+# ── catalog data: packaged JSON, refreshed from GitHub in memory ─
+#
+# The catalog DATA lives in catalog.json (checked in beside this module
+# and shipped as package data); this module keeps all policy. At import
+# we load the packaged copy — no network on the import path. A TTL-gated
+# background refresh fetches the same file from the repo's main branch
+# and swaps it in memory only: nothing on disk changes, so a git
+# checkout never sees a dirty tracked file and the packaged copy remains
+# the offline truth. A reverted commit on main heals every install on
+# its next fetch, and day-0 entries reach users without an app release.
+
+_CATALOG_URL = ("https://raw.githubusercontent.com/NousResearch/hermes-agent"
+ "/main/hermes_cli/local_runtime/catalog.json")
+_SCHEMA_VERSION = 1
+_REFRESH_TTL_S = 6 * 3600
+_refresh_lock = threading.Lock()
+_last_refresh_attempt = 0.0
+
+
+def _asset_from(d: "dict | None") -> "AssetFile | None":
+ if not d:
+ return None
+ return AssetFile(path=d["path"], size_bytes=int(d["size_bytes"]),
+ local=d.get("local"))
+
+
+def _load_catalog(doc: dict) -> "tuple[CatalogEntry, ...]":
+ """Parse a catalog document into entries. Unknown fields are ignored
+ (newer catalogs stay readable by older apps); a major schema bump is
+ the signal that they wouldn't be, and the caller skips the document."""
+ if int(doc.get("schema_version", 0)) != _SCHEMA_VERSION:
+ raise ValueError(f"catalog schema {doc.get('schema_version')!r} "
+ f"(this build reads {_SCHEMA_VERSION})")
+ entries = []
+ for m in doc["models"]:
+ variants = tuple(
+ QuantVariant(quant=v["quant"],
+ files=tuple(_asset_from(f) for f in v["files"]),
+ validated=bool(v.get("validated")))
+ for v in m["variants"])
+ entries.append(CatalogEntry(
+ id=m["id"], display_name=m["display_name"],
+ description=m["description"], repo=m["repo"], variants=variants,
+ n_ctx_train=int(m["n_ctx_train"]),
+ full_layers=int(m["full_layers"]),
+ recurrent_layers=int(m["recurrent_layers"]),
+ per_layer_f16=int(m["per_layer_f16"]),
+ swa_layers=int(m.get("swa_layers", 0)),
+ swa_window=int(m.get("swa_window", 0)),
+ moe=bool(m.get("moe")), mtp=bool(m.get("mtp")),
+ mtp_draft_depth=int(m.get("mtp_draft_depth", 3)),
+ n_vocab=int(m.get("n_vocab", 0)),
+ mmproj=_asset_from(m.get("mmproj")),
+ draft=_asset_from(m.get("draft")),
+ sampling=dict(m.get("sampling", {})),
+ min_engine=str(m.get("min_engine", "")),
+ quality=int(m.get("quality", 0)),
+ decode_fraction=float(m.get("decode_fraction", 1.0)),
+ ))
+ return tuple(entries)
+
+
+def _packaged_catalog() -> "tuple[CatalogEntry, ...]":
+ from importlib.resources import files
+
+ raw = files("hermes_cli.local_runtime").joinpath("catalog.json").read_text(
+ encoding="utf-8")
+ return _load_catalog(json.loads(raw))
+
+
+CATALOG: "tuple[CatalogEntry, ...]" = _packaged_catalog()
+
+
+def refresh_catalog(force: bool = False) -> bool:
+ """Fetch the current catalog from the repo and swap it in memory.
+
+ Best-effort by design: any failure (offline, GitHub down, unreadable
+ schema) leaves the running catalog untouched and retries after the
+ TTL. Returns True when a fetched document replaced the catalog."""
+ global CATALOG, _last_refresh_attempt
+
+ now = time.monotonic()
+ with _refresh_lock:
+ if not force and now - _last_refresh_attempt < _REFRESH_TTL_S:
+ return False
+ _last_refresh_attempt = now
+ try:
+ req = urllib.request.Request(
+ _CATALOG_URL, headers={"User-Agent": "hermes-local-runtime"})
+ with urllib.request.urlopen(req, timeout=10) as r:
+ fetched = _load_catalog(json.load(r))
+ except Exception as exc: # noqa: BLE001
+ logger.debug("catalog refresh skipped: %s", exc)
+ return False
+ if fetched != CATALOG:
+ logger.info("catalog refreshed from repo (%d models)", len(fetched))
+ CATALOG = fetched
+ return True
+
+
+def refresh_catalog_soon() -> None:
+ """TTL-gated background refresh; returns immediately. The caller's
+ current request serves the catalog it already has — the refresh
+ lands for the next one."""
+ if time.monotonic() - _last_refresh_attempt < _REFRESH_TTL_S:
+ return
+ threading.Thread(target=refresh_catalog, daemon=True,
+ name="catalog-refresh").start()
+
+
+def catalog_by_id() -> dict[str, CatalogEntry]:
+ return {entry.id: entry for entry in CATALOG}
+
+
+def find_variant(entry_id: str, model_id: str) -> QuantVariant | None:
+ entry = catalog_by_id().get(entry_id)
+ if entry is None:
+ return None
+ return next((v for v in entry.variants if v.model_id == model_id), None)
+
+
+def find_entry_for_model(model_id: str) -> "tuple[CatalogEntry, QuantVariant] | None":
+ """Locate the entry + variant that owns a staged model id."""
+ for entry in CATALOG:
+ for variant in entry.variants:
+ if variant.model_id == model_id:
+ return entry, variant
+ return None
diff --git a/hermes_cli/local_runtime/context_policy.py b/hermes_cli/local_runtime/context_policy.py
new file mode 100644
index 0000000000..1f08ae354b
--- /dev/null
+++ b/hermes_cli/local_runtime/context_policy.py
@@ -0,0 +1,286 @@
+"""Context policy — the window ladder for managed local models.
+
+One contract: any model runs at any window up to its native max; hardware
+and session depth only change tokens/s. Constants, not knobs — nothing in
+this module reads config.
+
+The policy encodes behavior measured on real hardware (llama.cpp,
+discrete NVIDIA GPUs on Windows/WDDM, and unified-memory devices):
+
+- Windows never over-allocates VRAM ahead of need. On WDDM, allocating
+ past residency slows decode roughly 9x even at identical conversation
+ depth — the driver silently demotes pages instead of failing. Every
+ window grant therefore re-fits against live memory at grant time.
+- Models launch at the largest window that fits entirely in GPU memory
+ (zero-spill) and grow toward their native max as the session needs
+ room, at request boundaries only.
+- Growth re-prefills the conversation into the larger window. Measured
+ cost is comparable to save/restore on discrete GPUs, and recurrent or
+ hybrid-attention models cannot rewind mid-sequence anyway, so
+ re-prefill is the only mechanism that works for every architecture.
+- Every recommended model gets at least a 64K window. When weights alone
+ exceed VRAM, the fit deliberately spills weights to host RAM to
+ protect that floor (measured: an explicit context size makes the fit
+ spill weights and hold the window rather than shrink it).
+- Below ~6 tok/s decode, growth stops and compression becomes the
+ default; deeper context is an explicit per-session choice. The deepest
+ measured host-spilled configuration bottomed out near this rate.
+- Spilled mixture-of-experts configs pin expert/FFN weights to host so
+ attention and KV stay GPU-resident — measured ~1.75x faster than
+ spilling layers naively at the same host byte count.
+- Speculative decoding (MTP) defaults on only for spilled configs, where
+ its speedup is largest (measured 1.43x spilled vs 1.35x resident).
+"""
+
+from __future__ import annotations
+
+from dataclasses import dataclass, field
+
+from hermes_cli.local_runtime.estimator import (
+ HardwareBudget,
+ ModelProfile,
+ PhysicsRefusal,
+ ctx_bytes,
+ physics_check,
+)
+
+FLOOR = 64 * 1024 # = target; one internal constant
+_LADDER_GROWTH = 1.5
+_GROW_AT_OCCUPANCY = 0.85 # of the current window, at turn boundary
+SPEED_FLOOR_TOK_S = 6.0 # deepest measured spill bottomed near this
+_EARLY_COST_CTX_FRACTION = 0.15 # bounded early cost when weights spill
+
+# TARGET_WINDOW: the smallest ladder rung at which compression becomes the
+# exception rather than the routine. Measured over 161 real agentic
+# sessions: 66% complete uncompressed in 64K, 82% in 96K, 91% in 144K —
+# and the marginal gain past 144K (+6 points for 216K) falls below the
+# quality cost of stepping down another quant. Quant selection prefers
+# the best build that reaches this; the FLOOR remains the guarantee.
+TARGET_WINDOW = 144 * 1024
+
+# What a load really costs beyond weights + KV: CUDA contexts and compute
+# buffers at the DEFAULT microbatch (-ub 512, no MTP). Measured on a
+# 32 GiB card: a model estimated at 29.3 GiB (weights+KV) loaded at
+# ~31.2 GiB resident and the server's own fit still shaved a layer to
+# CPU. Microbatch/MTP logits buffers are priced separately per model
+# (ub_logits_bytes — they scale with the model's vocab and doubled once
+# packed a card 3.9 GiB past this constant). Callers add mmproj bytes on
+# top.
+RUNTIME_OVERHEAD_BYTES = int(1.5 * (1 << 30))
+
+
+def ladder(native: int) -> list[int]:
+ """64K -> 96K -> 128K -> ... -> native (native always the last rung)."""
+ rungs: list[int] = []
+ step = float(FLOOR)
+ while step < native:
+ rungs.append(int(step))
+ step *= _LADDER_GROWTH
+ rungs.append(native)
+ return rungs
+
+
+@dataclass
+class WindowDecision:
+ window: int
+ spill_bytes: int # weights displaced to host at this window
+ kv_on_gpu: bool
+ reasons: list[str] = field(default_factory=list)
+
+ @property
+ def spilled(self) -> bool:
+ return self.spill_bytes > 0
+
+
+def initial_window(profile: ModelProfile, budget: HardwareBudget,
+ *, flash_attention: bool = True,
+ overhead_bytes: int = 0) -> WindowDecision | PhysicsRefusal:
+ """The launch decision: largest cheap rung, never below the floor.
+
+ Zero-spill rung: weights + ctx + overhead fit usable VRAM entirely.
+ Bounded-early-cost rung: weights already exceed VRAM; take the largest
+ rung whose ctx stays <= ~15% of usable VRAM.
+ Floor everywhere, capped at native.
+
+ ``overhead_bytes``: runtime cost beyond weights+KV (RUNTIME_OVERHEAD
+ plus the vision projector when one loads). Zero keeps this function
+ pure physics for decision-table tests; production callers pass it.
+ """
+ refusal = physics_check(profile, budget, FLOOR, flash_attention=flash_attention)
+ if refusal:
+ return refusal
+
+ native = profile.n_ctx_train or FLOOR
+ rungs = ladder(native)
+
+ reasons: list[str] = []
+ best_zero_spill: int | None = None
+ for rung in rungs:
+ need = (profile.weights_bytes + overhead_bytes
+ + ctx_bytes(profile, rung, flash_attention=flash_attention))
+ if need <= budget.usable_vram_bytes:
+ best_zero_spill = rung
+ else:
+ break
+
+ if best_zero_spill is not None and best_zero_spill >= min(FLOOR, native):
+ window = best_zero_spill
+ reasons.append(f"largest zero-spill rung ({window // 1024}K)")
+ else:
+ # Weights spill from turn one (steep-curve model on a small card) —
+ # hold the floor, bound the early ctx cost.
+ cap = int(budget.usable_vram_bytes * _EARLY_COST_CTX_FRACTION)
+ window = min(FLOOR, native)
+ for rung in rungs:
+ if rung < window:
+ continue
+ if ctx_bytes(profile, rung, flash_attention=flash_attention) <= cap:
+ window = rung
+ else:
+ break
+ reasons.append(f"floor held at {window // 1024}K; weights spill (deliberate price of the guarantee)")
+
+ kv = ctx_bytes(profile, window, flash_attention=flash_attention)
+ spill = max(0, profile.weights_bytes + kv - budget.usable_vram_bytes)
+ return WindowDecision(window=window, spill_bytes=spill,
+ kv_on_gpu=kv <= budget.usable_vram_bytes,
+ reasons=reasons)
+
+
+@dataclass
+class GrowthDecision:
+ action: str # "grow" | "hold" | "compress-default"
+ next_window: int | None = None
+ reason: str = ""
+
+
+def growth_decision(profile: ModelProfile, budget: HardwareBudget, *,
+ current_window: int, session_tokens: int,
+ measured_decode_tok_s: float | None,
+ server_idle: bool,
+ flash_attention: bool = True,
+ occupancy_confirmed: bool = False) -> GrowthDecision:
+ """One growth evaluation, END-OF-TURN ONLY (caller guarantees the turn
+ boundary; recurrent state cannot rewind mid-sequence).
+
+ Gate ordering:
+ 1. occupancy (~85%) — nothing to do before the edge;
+ 2. native cap — the contract tops out at trained context;
+ 3. idleness — growth re-grants only on an otherwise-idle
+ server (concurrency design);
+ 4. speed floor — below it, compression becomes the default and deeper
+ is an explicit user choice;
+ 5. re-fit against LIVE free memory (the rung must fit residency
+ NOW, not at launch time — over-allocation is the slow path).
+
+ ``occupancy_confirmed``: the caller has independently established that
+ the session is at its window's edge (the agent's compression gate fired
+ on its own threshold). Skips gate 1 so two separately-derived edge
+ definitions can't deadlock into compress-before-grow.
+ """
+ if not occupancy_confirmed and session_tokens < current_window * _GROW_AT_OCCUPANCY:
+ return GrowthDecision("hold", reason="session below growth occupancy")
+
+ native = profile.n_ctx_train or current_window
+ if current_window >= native:
+ return GrowthDecision("compress-default",
+ reason="at native window; compression is the only move")
+
+ if not server_idle:
+ return GrowthDecision("hold", reason="server busy; re-grant deferred to idle")
+
+ if measured_decode_tok_s is not None and measured_decode_tok_s < SPEED_FLOOR_TOK_S:
+ return GrowthDecision(
+ "compress-default",
+ reason=(f"decode {measured_decode_tok_s:.1f} tok/s below the "
+ f"~{SPEED_FLOOR_TOK_S:.0f} tok/s floor; growth is now an "
+ "explicit per-session choice"))
+
+ next_rung = next((r for r in ladder(native) if r > current_window), native)
+
+ # Re-fit against live free memory: allocation beyond residency is the
+ # slow path, so a rung that no longer fits doesn't get granted.
+ kv = ctx_bytes(profile, next_rung, flash_attention=flash_attention)
+ total_need = profile.weights_bytes + kv
+ if total_need > budget.usable_vram_bytes + budget.ram_available_bytes:
+ return GrowthDecision("compress-default",
+ reason="next rung exceeds physics; compression instead")
+
+ return GrowthDecision("grow", next_window=next_rung,
+ reason=f"rung {current_window // 1024}K -> {next_rung // 1024}K")
+
+
+def spill_overrides(profile: ModelProfile) -> list[str]:
+ """-ot placement for spilled configs: expert/FFN weights to host so
+ attention + KV stay GPU-resident. MoE gets the expert pattern;
+ hybrids push recurrent-layer FFNs (their n_head_kv==0 layers carry no
+ KV worth protecting)."""
+ if profile.moe:
+ return ["-ot", r"blk\.\d+\.ffn_.*_exps\.weight=CPU"]
+ if profile.recurrent_layer_count:
+ return ["-ot", r"blk\.\d+\.ffn_.*\.weight=CPU"]
+ return [] # dense: fit's back-to-front layer cut is the only axis
+
+
+def launch_args(profile: ModelProfile, decision: WindowDecision, *,
+ flash_attention: bool = True,
+ mtp_capable: bool = False,
+ mtp_draft_depth: int = 3,
+ uma: bool = False,
+ mtp_prefill: bool = False) -> list[str]:
+ """Per-model launch flags from a window decision. Explicit -c puts fit
+ into spill-weights-and-hold-ctx; q8 KV cache wherever flash attention
+ exists; -ot placement on spilled configs — DISCRETE cards only.
+
+ ``uma``: on unified memory there is no bus to protect tensors from —
+ "CPU" and "GPU" are the same silicon, and pinning FFN weights to the
+ host path just forces CPU compute (measured well over 2x slower than
+ letting the allocator place everything). The discrete
+ ~1.75x win the -ot pattern encodes does not transfer; a spilled UMA
+ config runs unpinned.
+
+ MTP and the large prefill microbatch both win, and whether they may
+ STACK is a fit question, not a rule: backend sampling keeps a
+ ubatch x vocab x fp32 logits buffer on the GPU and MTP's draft
+ context doubles it, so the stacked posture costs a few GiB extra at
+ large vocab. Where it fits, it measures best on both axes (Qwen3.8
+ Q4 on a 32 GiB card: 93.3 tok/s decode vs 89.5 at ub512, prefill
+ slightly better too); where it doesn't, ub512 keeps the decode win
+ without packing the card. ``mtp_prefill`` is that fit verdict —
+ presets decide it against the priced margin, and ub_logits_bytes()
+ prices the same choice so the flag and its cost travel together."""
+ args = ["-c", str(decision.window)]
+ if mtp_capable:
+ args += ["--spec-type", "draft-mtp",
+ "--spec-draft-n-max", str(mtp_draft_depth),
+ "--backend-sampling", "--spec-draft-backend-sampling"]
+ if mtp_prefill:
+ args += ["-b", "4096", "-ub", "2048"]
+ else:
+ args += ["-b", "2048", "-ub", "2048"]
+ if flash_attention:
+ args += ["-ctk", "q8_0", "-ctv", "q8_0", "-fa", "on"]
+ if decision.spilled and not uma:
+ args += spill_overrides(profile)
+ return args
+
+
+def ub_logits_bytes(n_vocab: int, *, mtp_capable: bool,
+ mtp_prefill: bool = False) -> int:
+ """GPU logits/compute-buffer cost of the microbatch posture chosen by
+ launch_args, priced from the model's own vocab and calibrated against
+ measured server RSS (Qwen3.8 Q4, both postures, three windows):
+
+ stacked (MTP + ub2048): ubatch x vocab x fp32 x 1.5 (~2.9 GiB at
+ 248K vocab; fitted 2.5, rounded up)
+ decode (MTP + ub512): ubatch x vocab x fp32 x 2 (~1.0 GiB)
+ plain (ub2048): ubatch x vocab x fp32 (~1.9 GiB)
+
+ Callers add this to RUNTIME_OVERHEAD per model — the flag and its
+ price travel together or the fit lies."""
+ v = max(0, int(n_vocab))
+ if mtp_capable and mtp_prefill:
+ return int(2048 * v * 4 * 1.5)
+ if mtp_capable:
+ return 512 * v * 4 * 2
+ return 2048 * v * 4
diff --git a/hermes_cli/local_runtime/detect.py b/hermes_cli/local_runtime/detect.py
new file mode 100644
index 0000000000..02131255ed
--- /dev/null
+++ b/hermes_cli/local_runtime/detect.py
@@ -0,0 +1,80 @@
+"""Detection of running llama-server instances.
+
+Probes well-known local roots and fingerprints genuine llama-server via
+/props (build_info + model fields — Ollama and LM Studio answer /v1/models
+but not /props). The credential is reachability; detection never needs a
+key, but honors one if the probed server requires it (401 -> detected,
+auth_required=True).
+"""
+
+from __future__ import annotations
+
+import json
+import urllib.error
+import urllib.request
+from dataclasses import dataclass
+
+# Always 127.0.0.1 — resolving localhost costs ~2s/request on Windows.
+DEFAULT_PROBE_PORTS = (8080,) # llama-server default; managed port comes from config
+
+
+@dataclass
+class DetectedServer:
+ base_url: str # OpenAI-compatible /v1 root
+ build_info: str # e.g. "b10290-c8e03ce81"
+ model_path: str # currently loaded model (may be empty in router mode)
+ n_ctx: int | None
+ router_mode: bool # GET /models answered -> router management available
+ auth_required: bool
+
+
+def _get(url: str, timeout_s: int = 3) -> tuple[int, dict | None]:
+ try:
+ with urllib.request.urlopen(url, timeout=timeout_s) as r:
+ raw = r.read()
+ return r.status, (json.loads(raw) if raw else None)
+ except urllib.error.HTTPError as exc:
+ return exc.code, None
+ except (urllib.error.URLError, OSError, TimeoutError, json.JSONDecodeError):
+ return 0, None
+
+
+def probe_port(port: int) -> DetectedServer | None:
+ """One port: /props fingerprint, then /models for router capability."""
+ root = f"http://127.0.0.1:{port}"
+ status, props = _get(f"{root}/props")
+ if status == 401:
+ return DetectedServer(base_url=f"{root}/v1", build_info="", model_path="",
+ n_ctx=None, router_mode=False, auth_required=True)
+ if status != 200 or not isinstance(props, dict):
+ return None
+ build = str(props.get("build_info", ""))
+ if not build:
+ return None # answers /props but isn't llama-server
+ n_ctx = None
+ dgs = props.get("default_generation_settings")
+ if isinstance(dgs, dict):
+ n_ctx = dgs.get("n_ctx")
+ models_status, models = _get(f"{root}/models")
+ return DetectedServer(
+ base_url=f"{root}/v1",
+ build_info=build,
+ model_path=str(props.get("model_path", "")),
+ n_ctx=n_ctx,
+ router_mode=(models_status == 200 and isinstance(models, dict)
+ and "data" in models),
+ auth_required=False,
+ )
+
+
+def detect_server(extra_ports: tuple[int, ...] = ()) -> DetectedServer | None:
+ """First hit across default + extra ports (managed port, config port)."""
+ seen = set()
+ for port in (*DEFAULT_PROBE_PORTS, *extra_ports):
+ if port in seen:
+ continue
+ seen.add(port)
+ hit = probe_port(port)
+ if hit:
+ return hit
+ return None
diff --git a/hermes_cli/local_runtime/endpoint.py b/hermes_cli/local_runtime/endpoint.py
new file mode 100644
index 0000000000..b80a663d03
--- /dev/null
+++ b/hermes_cli/local_runtime/endpoint.py
@@ -0,0 +1,194 @@
+"""Endpoint resolution for llamacpp-alias requests (provider integration).
+
+The seam between the existing provider mechanism and the managed runtime:
+``provider: llamacpp`` with no explicit base_url resolves, in order, to
+
+1. the managed server this Hermes is supervising (state file written by
+ LlamaServerSupervisor.start, removed on stop, staleness-checked), or
+2. a detected external llama-server.
+
+Returns None when neither exists — the caller falls through to the normal
+custom-provider path and its own error reporting.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import threading
+import time
+import urllib.error
+import urllib.request
+
+LLAMACPP_ALIASES = frozenset({"llamacpp", "llama.cpp", "llama-cpp"})
+
+logger = logging.getLogger(__name__)
+
+
+def _pid_alive(pid: int) -> bool:
+ """Liveness for the state file's supervisor-child pid.
+
+ psutil when available; otherwise fall back to True (optimistic) — on
+ Windows ``os.kill(pid, 0)`` TERMINATES the process, so it must never be
+ used as a probe (windows-git-bash interop pitfall).
+ """
+ if not pid or pid < 0:
+ return False
+ try:
+ import psutil # type: ignore
+
+ return psutil.pid_exists(pid)
+ except Exception: # noqa: BLE001
+ return True
+
+
+def _state_endpoint() -> dict | None:
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ path = state_path()
+ if not path.exists():
+ return None
+ try:
+ state = json.loads(path.read_text(encoding="utf-8"))
+ except (json.JSONDecodeError, OSError):
+ return None
+ base_url = state.get("base_url", "")
+ if not base_url:
+ return None
+ endpoint = {"base_url": base_url, "api_key": state.get("api_key", "")}
+ # Ownership proof: the stable port means a SECOND install (different
+ # HERMES_HOME — a scratch profile, say) can own 127.0.0.1:18434 with a
+ # different api key while this install's state file still points there.
+ # /health is a public route, so it answers 200 for ANYONE's server —
+ # trusting it alone sent every chat request and the load-progress
+ # watcher at a server that 401s our key, silently. The recorded
+ # supervisor pid is the tiebreaker: health-200 from a server whose
+ # recorded child is DEAD is someone else's server, never a starting one.
+ pid_ok = _pid_alive(int(state.get("pid") or 0))
+ # Healthy server: done (when it's ours).
+ try:
+ health = base_url.rsplit("/v1", 1)[0] + "/health"
+ with urllib.request.urlopen(health, timeout=3) as r:
+ if r.status == 200:
+ return endpoint if pid_ok else None
+ except (urllib.error.URLError, OSError, TimeoutError):
+ pass
+ # Not healthy YET: a live supervisor child is a STARTING server (state
+ # is written at spawn; llama-server takes seconds to listen). Resolve
+ # optimistically so readiness probes racing the boot see a configured
+ # provider, not missing credentials. A dead pid is a crashed-without-
+ # cleanup leftover — ignore it so requests don't blackhole.
+ if pid_ok:
+ return endpoint
+ return None
+
+
+def resolve_llamacpp_endpoint(config: dict | None = None,
+ wait_for_boot_s: float = 8.0) -> dict | None:
+ """Managed-first, detection-second endpoint for llamacpp aliases.
+
+ Returns {"base_url", "api_key"} or None. api_key is empty for keyless
+ external servers (callers substitute the SDK placeholder).
+
+ Boot-race rung: on a fresh backend start there is NO state file yet —
+ the lifespan boot thread is still spawning the server (config load +
+ preset generation + spawn ≈ 1-3 s) while the desktop's readiness probe
+ fires the moment the WebSocket connects. When the runtime is enabled
+ and installed, a missing endpoint means BOOTING, not unconfigured:
+ poll briefly for the state file instead of failing the probe (twice
+ observed as 'no usable credentials' → onboarding on restart).
+ """
+ managed = _state_endpoint()
+ if managed:
+ return managed
+
+ from hermes_cli.local_runtime.detect import detect_server
+
+ extra = ()
+ if config:
+ ports = (config.get("local_runtime") or {}).get("detect_ports") or []
+ extra = tuple(int(p) for p in ports)
+ hit = detect_server(extra_ports=extra)
+ if hit and not hit.auth_required:
+ return {"base_url": hit.base_url, "api_key": ""}
+
+ if wait_for_boot_s > 0 and _boot_in_flight(config):
+ _kick_managed_boot(config)
+ deadline = time.monotonic() + wait_for_boot_s
+ while time.monotonic() < deadline:
+ time.sleep(0.25)
+ managed = _state_endpoint()
+ if managed:
+ return managed
+ return None
+
+
+_KICK_LOCK = threading.Lock()
+
+
+def _kick_managed_boot(config: dict | None) -> None:
+ """Actively start the managed server when resolution finds it missing.
+
+ The wait loop above assumes some OTHER thread is bringing the server
+ up — true only at backend start (the lifespan boot thread). A router
+ that dies LATER leaves no boot in flight: the backend process was
+ killed with the router as part of its tree, or another install took
+ the stable port and the ownership guard rightly refused it. In those
+ states the wait just expired and agent init failed with 'no provider
+ configured', even though the fix is the same idempotent ensure call
+ the lifespan makes. Kick it here, off-thread (the resolver's wait
+ stays bounded; ensure's own state checks make a concurrent lifespan
+ boot harmless) and non-reentrant (racing resolutions kick once).
+ """
+ if not _KICK_LOCK.acquire(blocking=False):
+ return # a kick is already in flight
+
+ def _boot() -> None:
+ try:
+ cfg = config
+ if cfg is None:
+ from hermes_cli.config import load_config
+
+ cfg = load_config()
+ from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
+
+ ensure_local_runtime(cfg)
+ except Exception: # noqa: BLE001 — best-effort; resolution falls back
+ logger.warning("on-demand managed-server boot failed", exc_info=True)
+ finally:
+ _KICK_LOCK.release()
+
+ threading.Thread(target=_boot, daemon=True,
+ name="lr-on-demand-boot").start()
+
+
+def _boot_in_flight(config: dict | None) -> bool:
+ """True when the managed runtime is enabled and installed — the state
+ a lifespan boot thread is (or is about to be) bringing up.
+
+ Installed-ness is a verified-manifest scan under runtimes_root(), NOT a
+ server_binary() call — that helper requires an install_dir argument, and
+ calling it bare made this gate throw-and-return-False forever, silently
+ disabling the boot wait (the regression
+ test had monkeypatched this function instead of exercising it).
+ """
+ try:
+ if config is None:
+ from hermes_cli.config import load_config
+
+ config = load_config()
+ if not ((config or {}).get("local_runtime") or {}).get("enabled"):
+ return False
+ import json as _json
+
+ from hermes_cli.local_runtime.binaries import runtimes_root
+
+ for manifest in runtimes_root().glob("*/*/manifest.json"):
+ try:
+ if _json.loads(manifest.read_text(encoding="utf-8")).get("verified_version"):
+ return True
+ except (ValueError, OSError):
+ continue
+ return False
+ except Exception: # noqa: BLE001
+ return False
diff --git a/hermes_cli/local_runtime/estimator.py b/hermes_cli/local_runtime/estimator.py
new file mode 100644
index 0000000000..6778b88ed4
--- /dev/null
+++ b/hermes_cli/local_runtime/estimator.py
@@ -0,0 +1,179 @@
+"""Per-layer context-memory estimator + physics check.
+
+The whole-model dense formula misprices 1M-context hybrids by ~100x; the
+per-layer walk fixes that, and every column is measured on real GGUFs:
+
+- full-attention layer: linear in T (B1: 144.0 KiB/tok on Qwen3-4B
+ f16 — formula-exact)
+- SWA layer: capped at the sliding window
+- recurrent layer (n_head_kv == 0): constant (state is ~context-free)
+- q8_0 KV = exactly 34/64 of f16 (holds on CUDA and CPU)
+- weights: exact from the tensor table (within 0.01% of the loader)
+
+The estimator is ADVISORY: fit's allocation is authoritative at launch and
+the touch generation is ground truth after it. Unknown shapes round UP
+(never underestimate memory).
+"""
+
+from __future__ import annotations
+
+from dataclasses import dataclass
+from enum import Enum
+
+from hermes_cli.local_runtime.gguf import GGUFHeader
+
+# q8_0: 34-byte blocks of 32 f16-equivalent elements (exact).
+_Q8_BYTES_PER_ELEM = 34 / 32
+_F16_BYTES_PER_ELEM = 2.0
+
+# Architectures with a known SWA layer pattern: arch -> fraction of layers
+# that are sliding-window. Unknown SWA archs conservatively treat every
+# layer as full attention (overestimate; safe direction).
+_SWA_LAYER_FRACTION = {"gemma3": 5 / 6, "gemma2": 1 / 2}
+
+# Per-recurrent-layer state allowance (bytes/seq). Deliberately generous —
+# Measured: an entire hybrid slot state is ~99 MB including 8K tokens of
+# full-attn KV, so tens of MiB total is the right order; unknown SSM shapes
+# must never underestimate.
+_RECURRENT_STATE_PER_LAYER = 4 << 20
+
+
+class LayerKind(Enum):
+ FULL = "full"
+ SWA = "swa"
+ RECURRENT = "recurrent"
+
+
+@dataclass
+class ModelProfile:
+ """Everything the policy needs, decoupled from GGUF parsing so the
+ decision-table tests can construct profiles directly (design's
+ verification plan)."""
+
+ name: str
+ weights_bytes: int
+ embd_table_bytes: int
+ n_ctx_train: int
+ layers: list[tuple[LayerKind, int]] # (kind, kv_bytes_per_token_f16);
+ # SWA/recurrent reuse the same
+ # per-token figure, capped/ignored
+ swa_window: int = 0
+ moe: bool = False
+ architecture: str = ""
+ n_vocab: int = 0 # prices logits buffers (ubatch x vocab)
+ # Context-cost multiplier. MTP spec decode keeps a small draft
+ # context beside the main one. Calibrated against four measured
+ # server-RSS points on Qwen3.8 Q4 (128K/221K/256K, both postures):
+ # the draft adds ~17% to per-token KV; 1.2 rounds up so the error
+ # stays on the safe side (+250 MiB at 256K, never negative).
+ kv_scale: float = 1.0
+
+ @property
+ def per_token_kv_f16(self) -> int:
+ """Uncapped per-token KV cost (full + SWA share)."""
+ return sum(b for kind, b in self.layers if kind != LayerKind.RECURRENT)
+
+ @property
+ def recurrent_layer_count(self) -> int:
+ return sum(1 for kind, _ in self.layers if kind == LayerKind.RECURRENT)
+
+
+@dataclass
+class HardwareBudget:
+ """Memory the physics check may budget against.
+
+ Budget-source rule: discrete cards may trust the device query
+ (measured honest); unified-memory devices must budget from OS free
+ physical memory minus headroom — their device queries have been
+ observed off by 3x. Callers construct
+ this accordingly; the estimator just consumes it.
+ """
+
+ usable_vram_bytes: int # live free (discrete) / derived (UMA)
+ total_device_bytes: int
+ ram_available_bytes: int
+ uma: bool = False
+
+
+def profile_from_gguf(header: GGUFHeader) -> ModelProfile:
+ kv_heads = header.head_counts_kv()
+ dk, dv = header.head_dim_k, header.head_dim_v
+ swa_fraction = _SWA_LAYER_FRACTION.get(header.architecture, 0.0)
+ has_swa = header.sliding_window > 0 and swa_fraction > 0
+
+ layers: list[tuple[LayerKind, int]] = []
+ n_attn_seen = 0
+ n_attn_total = sum(1 for h in kv_heads if h > 0)
+ n_swa = round(n_attn_total * swa_fraction) if has_swa else 0
+ for heads in kv_heads:
+ if heads == 0:
+ layers.append((LayerKind.RECURRENT, 0))
+ continue
+ per_token = round(heads * (dk + dv) * _F16_BYTES_PER_ELEM)
+ # Distribute the SWA share across the first n_swa attention layers;
+ # only the full/SWA SPLIT matters to the totals, not which indexes.
+ kind = LayerKind.SWA if n_attn_seen < n_swa else LayerKind.FULL
+ layers.append((kind, per_token))
+ n_attn_seen += 1
+
+ return ModelProfile(
+ name=header.path,
+ weights_bytes=header.tensor_bytes,
+ embd_table_bytes=header.embd_table_bytes,
+ n_ctx_train=header.n_ctx_train,
+ layers=layers,
+ swa_window=header.sliding_window,
+ moe=header.expert_count > 0,
+ architecture=header.architecture,
+ n_vocab=header.n_vocab,
+ )
+
+
+def kv_dtype_factor(flash_attention: bool) -> float:
+ """q8_0 with FA (every backend we ship); f16 on exotic non-FA fallbacks
+ — the 64K guarantee stands either way, the physics check just prices
+ the doubled KV (design: KV dtype is behavior, not config)."""
+ return (_Q8_BYTES_PER_ELEM / _F16_BYTES_PER_ELEM) if flash_attention else 1.0
+
+
+def ctx_bytes(profile: ModelProfile, window: int, *,
+ flash_attention: bool = True) -> int:
+ """Context memory for one window: full layers linear in T, SWA layers
+ capped at the sliding window, recurrent layers constant. Scaled by
+ profile.kv_scale (MTP draft context)."""
+ factor = kv_dtype_factor(flash_attention)
+ total = 0.0
+ for kind, per_token_f16 in profile.layers:
+ if kind == LayerKind.RECURRENT:
+ total += _RECURRENT_STATE_PER_LAYER
+ elif kind == LayerKind.SWA:
+ total += per_token_f16 * factor * min(window, profile.swa_window)
+ else:
+ total += per_token_f16 * factor * window
+ return int(total * profile.kv_scale)
+
+
+@dataclass
+class PhysicsRefusal:
+ """The only true refusal: weights + floor-KV + state exceed VRAM + RAM.
+ The remedy is a smaller quant, never a smaller window."""
+
+ needed_bytes: int
+ available_bytes: int
+ message: str
+
+
+def physics_check(profile: ModelProfile, budget: HardwareBudget,
+ floor: int, *, flash_attention: bool = True) -> PhysicsRefusal | None:
+ needed = (profile.weights_bytes
+ + ctx_bytes(profile, min(floor, profile.n_ctx_train or floor),
+ flash_attention=flash_attention))
+ available = budget.usable_vram_bytes + budget.ram_available_bytes
+ if needed > available:
+ gib = 1 << 30
+ return PhysicsRefusal(
+ needed_bytes=needed, available_bytes=available,
+ message=(f"{profile.name}: needs ~{needed / gib:.1f} GiB at the "
+ f"{floor // 1024}K floor but only ~{available / gib:.1f} GiB "
+ "of VRAM+RAM exist — try a smaller quant (UD-Q3/Q2)"))
+ return None
diff --git a/hermes_cli/local_runtime/gguf.py b/hermes_cli/local_runtime/gguf.py
new file mode 100644
index 0000000000..a5159b9d56
--- /dev/null
+++ b/hermes_cli/local_runtime/gguf.py
@@ -0,0 +1,220 @@
+"""GGUF metadata + tensor-table reader (stdlib only).
+
+Feeds the per-layer context estimator: architecture, layer count, per-layer
+KV head counts (0 = recurrent layer — the hybrid discriminator), head dims,
+sliding-window config, trained context, and exact weight bytes summed from
+the tensor table (validated to within 0.01% of the loader's buffer).
+
+Reads the header only (metadata + tensor infos); never touches tensor data,
+so it is fast enough to run at picker time on multi-GB files.
+"""
+
+from __future__ import annotations
+
+import struct
+from dataclasses import dataclass, field
+from pathlib import Path
+
+_GGUF_MAGIC = b"GGUF"
+
+# ggml tensor type sizes: type_id -> (block_bytes, block_elems).
+# IQ-family sizes verified against ggml-common.h.
+_GGML_TYPE_SIZES = {
+ 0: (4, 1), 1: (2, 1), 2: (18, 32), 3: (20, 32), 6: (22, 32), 7: (24, 32),
+ 8: (34, 32), 9: (36, 32), 10: (84, 256), 11: (110, 256), 12: (144, 256),
+ 13: (176, 256), 14: (210, 256), 15: (292, 256), 16: (66, 256),
+ 17: (74, 256), 18: (98, 256), 19: (50, 256), 20: (18, 32),
+ 21: (110, 256), 22: (82, 256), 23: (136, 256), 24: (1, 1), 25: (2, 1),
+ 26: (4, 1), 27: (8, 1), 28: (8, 1), 29: (56, 256), 30: (2, 1),
+}
+
+# GGUF metadata value types.
+_V_UINT8, _V_INT8, _V_UINT16, _V_INT16 = 0, 1, 2, 3
+_V_UINT32, _V_INT32, _V_FLOAT32, _V_BOOL = 4, 5, 6, 7
+_V_STRING, _V_ARRAY, _V_UINT64, _V_INT64, _V_FLOAT64 = 8, 9, 10, 11, 12
+
+_SCALAR_FMT = {
+ _V_UINT8: " str:
+ return str(self.metadata.get("general.architecture", ""))
+
+ def _arch_key(self, suffix: str):
+ return self.metadata.get(f"{self.architecture}.{suffix}")
+
+ @property
+ def n_layer(self) -> int:
+ return int(self._arch_key("block_count") or 0)
+
+ @property
+ def n_vocab(self) -> int:
+ """Vocabulary size: prices the GPU logits buffers (they scale
+ ubatch x vocab). vocab_size metadata when present, else the
+ tokenizer list length."""
+ v = self._arch_key("vocab_size")
+ if v:
+ return int(v)
+ toks = self.metadata.get("tokenizer.ggml.tokens")
+ return len(toks) if isinstance(toks, list) else 0
+
+ @property
+ def n_ctx_train(self) -> int:
+ return int(self._arch_key("context_length") or 0)
+
+ @property
+ def sampling_defaults(self) -> dict:
+ """Upstream's recommended sampling, when the file carries it.
+
+ Model publishers bake general.sampling.* keys into the GGUF
+ (llama-server reads them as that model's default generation
+ settings), so the file itself is the source of truth for how its
+ publisher wants it run — it arrives with the download and updates
+ with every re-upload, no catalog required. Returned as preset INI
+ keys; empty when the file carries none.
+ """
+ ini_key = {"temp": "temp", "temperature": "temp", "top_p": "top-p",
+ "top_k": "top-k", "min_p": "min-p",
+ "repeat_penalty": "repeat-penalty",
+ "presence_penalty": "presence-penalty"}
+ out = {}
+ for key, value in self.metadata.items():
+ if not key.startswith("general.sampling."):
+ continue
+ name = ini_key.get(key.rsplit(".", 1)[-1])
+ if name is not None and isinstance(value, (int, float)):
+ num = round(float(value), 4)
+ out[name] = str(int(num)) if num == int(num) else str(num)
+ return out
+
+ @property
+ def n_embd(self) -> int:
+ return int(self._arch_key("embedding_length") or 0)
+
+ @property
+ def n_head(self) -> int:
+ v = self._arch_key("attention.head_count")
+ if isinstance(v, list):
+ return int(max(v))
+ return int(v or 0)
+
+ @property
+ def full_attention_interval(self) -> int:
+ """GDN-hybrid discriminator (qwen35 family): every Nth layer is full
+ attention, the rest are linear/recurrent. 0 = not present."""
+ return int(self._arch_key("full_attention_interval") or 0)
+
+ def head_counts_kv(self) -> list[int]:
+ """Per-layer KV head counts; 0 marks a recurrent/linear layer (the
+ n_head_kv == 0 discriminator).
+
+ Three GGUF shapes, each verified against real files:
+ - per-layer array (nemotron_h_moe): use as-is;
+ - scalar + full_attention_interval (qwen35): the scalar applies to
+ every INTERVAL-th layer (1-indexed: layers where (i+1) % N == 0),
+ zero elsewhere — pricing all layers as attention was a 4x
+ overestimate on Qwen3.6-27B;
+ - plain scalar (dense): broadcast to every layer.
+ """
+ v = self._arch_key("attention.head_count_kv")
+ if isinstance(v, list):
+ return [int(x) for x in v]
+ scalar = int(v or 0)
+ interval = self.full_attention_interval
+ if interval > 1:
+ return [scalar if (i + 1) % interval == 0 else 0
+ for i in range(self.n_layer)]
+ return [scalar] * self.n_layer
+
+ @property
+ def head_dim_k(self) -> int:
+ v = self._arch_key("attention.key_length")
+ if v:
+ return int(v)
+ return self.n_embd // self.n_head if self.n_head else 0
+
+ @property
+ def head_dim_v(self) -> int:
+ v = self._arch_key("attention.value_length")
+ if v:
+ return int(v)
+ return self.head_dim_k
+
+ @property
+ def sliding_window(self) -> int:
+ return int(self._arch_key("attention.sliding_window") or 0)
+
+ @property
+ def expert_count(self) -> int:
+ return int(self._arch_key("expert_count") or 0)
+
+
+def read_gguf_header(path: str | Path) -> GGUFHeader:
+ path = Path(path)
+
+ def read_str(f) -> str:
+ (n,) = struct.unpack(" dict:
+ """model_id -> granted window (int). Empty on any read problem."""
+ try:
+ with open(window_overrides_path(), encoding="utf-8") as fh:
+ data = json.load(fh)
+ return {str(k): int(v) for k, v in data.items()}
+ except Exception: # noqa: BLE001
+ return {}
+
+
+def save_window_override(model_id: str, window: int) -> None:
+ overrides = load_window_overrides()
+ overrides[model_id] = int(window)
+ path = window_overrides_path()
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text(json.dumps(overrides, indent=1), encoding="utf-8")
+
+
+def clear_window_override(model_id: str) -> None:
+ """Drop a model's growth state (delete/re-download paths)."""
+ overrides = load_window_overrides()
+ if model_id in overrides:
+ del overrides[model_id]
+ window_overrides_path().write_text(
+ json.dumps(overrides, indent=1), encoding="utf-8")
+
+
+def is_managed_endpoint(base_url: str) -> bool:
+ """True when base_url is the server this process's state file points at."""
+ try:
+ from hermes_cli.local_runtime.endpoint import _state_endpoint
+
+ state = _state_endpoint()
+ if state is None:
+ return False
+ return (base_url or "").rstrip("/") == str(
+ state.get("base_url", "")).rstrip("/")
+ except Exception: # noqa: BLE001
+ return False
+
+
+def maybe_grow_window(model_id: str, *, base_url: str, session_tokens: int,
+ current_window: int,
+ measured_decode_tok_s: float | None = None) -> int | None:
+ """One growth evaluation + execution. Returns the NEW window when the
+ ladder granted a bigger one, else None (hold / compress / not ours).
+
+ The caller sits at a request boundary by construction (the pre-API
+ compression gate), so re-prefill growth is safe at any call: the next
+ request rebuilds server state from scratch in the larger window —
+ nothing rewinds.
+ """
+ from hermes_cli.local_runtime.bootstrap import (
+ get_supervisor,
+ refresh_local_runtime,
+ staged_models,
+ )
+ from hermes_cli.local_runtime.context_policy import growth_decision
+ from hermes_cli.local_runtime.estimator import profile_from_gguf
+ from hermes_cli.local_runtime.gguf import read_gguf_header
+ from hermes_cli.local_runtime.hardware import probe_budget
+
+ sup = get_supervisor()
+ if sup is None or not is_managed_endpoint(base_url):
+ return None
+
+ gguf = next((p for p in staged_models()
+ if p.stem.startswith(model_id) or model_id in p.stem), None)
+ if gguf is None:
+ return None
+
+ try:
+ profile = profile_from_gguf(read_gguf_header(gguf))
+ except (ValueError, OSError) as exc:
+ logger.debug("growth skip %s: unreadable gguf (%s)", model_id, exc)
+ return None
+
+ try:
+ server_idle = sup.is_idle(model_id)
+ except Exception: # noqa: BLE001
+ server_idle = False
+
+ decision = growth_decision(
+ # Capacity budget, not live-free: growth executes via a server
+ # bounce, so the grown instance loads onto a freed card. Live-free
+ # here is distorted by the very model being grown — it reads its
+ # own residency as unavailable and vetoes rungs that fit.
+ profile, probe_budget(planning=True),
+ current_window=current_window,
+ session_tokens=session_tokens,
+ measured_decode_tok_s=measured_decode_tok_s,
+ server_idle=server_idle,
+ # The caller IS the occupancy signal: this runs from the agent's
+ # compression gate, which fired on its own threshold. Two
+ # separately-derived edges must not deadlock into
+ # compress-before-grow.
+ occupancy_confirmed=True,
+ )
+ if decision.action != "grow" or not decision.next_window:
+ logger.debug("growth %s: %s (%s)", model_id, decision.action, decision.reason)
+ return None
+
+ logger.info("context growth %s: %s", model_id, decision.reason)
+ save_window_override(model_id, decision.next_window)
+ if not refresh_local_runtime():
+ # The override still lands at the next boot; report no growth NOW
+ # so the caller compresses instead of overflowing a stale window.
+ logger.warning("growth %s: server refresh failed; compression proceeds", model_id)
+ return None
+ return decision.next_window
diff --git a/hermes_cli/local_runtime/hardware.py b/hermes_cli/local_runtime/hardware.py
new file mode 100644
index 0000000000..c6a316776d
--- /dev/null
+++ b/hermes_cli/local_runtime/hardware.py
@@ -0,0 +1,379 @@
+"""Live hardware budget probe.
+
+Budget-source rule: discrete cards may trust the device query (measured
+honest within rounding); unified-memory devices must budget from OS free
+physical memory minus headroom — their device queries have been observed
+off by 3x in both directions. The probe classifies the device and
+constructs the right HardwareBudget for the estimator.
+
+Vendor probe quirk (WDDM carve-out): on unified-memory NVIDIA devices
+under Windows, nvidia-smi answers from the legacy dedicated-VRAM
+carve-out — a fraction of the pool the CUDA allocator actually
+addresses uniformly at full bandwidth. The CUDA driver API is
+the tiebreaker: cuDeviceGetAttribute(INTEGRATED) is the vendor's own
+declaration and always wins — 1 budgets unified, 0 stays discrete no
+matter what any other number says. Only when the driver API is
+unreachable does the engine's --list-devices view apply, and then only
+behind two independent conditions no discrete card can meet.
+
+Every probe here must work under a stripped PATH — gateway and service
+sessions don't inherit the interactive environment. nvcuda/libcuda load
+through the system loader (PATH plays no part), so classification never
+depends on PATH; nvidia-smi resolves through an explicit candidate
+ladder (PATH first, then the driver's known install locations) and its
+absence only softens the live number, never the verdict.
+"""
+
+from __future__ import annotations
+
+import logging
+import os
+import re
+import shutil
+import subprocess
+import sys
+import time
+from pathlib import Path
+
+from hermes_cli.local_runtime.estimator import HardwareBudget
+
+logger = logging.getLogger(__name__)
+
+_GIB = 1 << 30
+# Reserve carved off the card before any grant: the desktop's own
+# co-residents (compositor, browser, Electron) measure ~2-2.5 GiB on a
+# working machine, and a window granted into that space demotes silently
+# under WDDM. 7% covers big cards; the 2 GiB floor is what the margin's
+# old 512 MiB floor failed to cover in practice (a 221K grant measured
+# 31.9/32.6 GiB with the desktop running — 'fits' by the math, demoted
+# in reality). Small cards give up window to this; spill mode is their
+# path to big models regardless.
+_MARGIN_FLOOR = 2 << 30
+_MARGIN_FRACTION = 0.09
+# UMA headroom: on unified-memory machines (Apple Silicon, unified-memory
+# NVIDIA) the model shares physical memory with the OS and every app, so
+# budget from RAM minus this fraction.
+_UMA_HEADROOM_FRACTION = 0.20
+
+# Engine-fallback gates for the unified-pool quirk — BOTH must hold, and
+# no discrete card can meet either: (1) the allocator's pool exceeds the
+# smi report by well past rounding/ECC slack (discrete cards agree within
+# ~2%; carve-out disagreement runs to whole multiples), and (2) the pool is
+# system-RAM-sized — a workstation card in a RAM-matched box fails (1)
+# because its smi and allocator AGREE, and a big discrete card in a
+# bigger box fails (2). The driver's INTEGRATED attribute, when
+# readable, bypasses both gates in whichever direction it points.
+_POOL_DISAGREEMENT_FACTOR = 1.5
+_POOL_RAM_FRACTION = 0.75
+
+# cuDeviceGetAttribute enum: device is integrated with host memory.
+_CU_DEVICE_ATTRIBUTE_INTEGRATED = 18
+
+# One probe per process once a device answers (silicon doesn't change);
+# a miss retries after this long so a runtime installed mid-session gets
+# picked up by the engine fallback.
+_POOL_NEGATIVE_TTL_S = 60.0
+_pool_probe_cache: tuple[float, "tuple[int, bool | None] | None"] | None = None
+
+# ' CUDA0: NVIDIA Example Device (1234-core Example GPU) (46464 MiB, 46284 MiB free)'
+# — greedy .* pins the LAST parenthesized group, so device names carrying
+# their own parentheses parse correctly.
+_DEVICE_LINE_RE = re.compile(r"CUDA\d+:.*\((\d+)\s*MiB,\s*\d+\s*MiB free\)\s*$")
+
+
+def _ram_bytes() -> tuple[int, int]:
+ """(total, available) physical memory, cross-platform stdlib."""
+ try:
+ import ctypes
+
+ class MEMORYSTATUSEX(ctypes.Structure):
+ _fields_ = [("dwLength", ctypes.c_ulong),
+ ("dwMemoryLoad", ctypes.c_ulong),
+ ("ullTotalPhys", ctypes.c_ulonglong),
+ ("ullAvailPhys", ctypes.c_ulonglong),
+ ("ullTotalPageFile", ctypes.c_ulonglong),
+ ("ullAvailPageFile", ctypes.c_ulonglong),
+ ("ullTotalVirtual", ctypes.c_ulonglong),
+ ("ullAvailVirtual", ctypes.c_ulonglong),
+ ("ullAvailExtendedVirtual", ctypes.c_ulonglong)]
+
+ stat = MEMORYSTATUSEX()
+ stat.dwLength = ctypes.sizeof(MEMORYSTATUSEX)
+ ctypes.windll.kernel32.GlobalMemoryStatusEx(ctypes.byref(stat))
+ return stat.ullTotalPhys, stat.ullAvailPhys
+ except (AttributeError, OSError):
+ pass
+ if sys.platform == "darwin":
+ # macOS getconf has no _PHYS_PAGES/_AVPHYS_PAGES (exit 64, "no such
+ # configuration parameter") — the POSIX branch below returns (0, 0)
+ # and every model reads unavailable. sysctl is the platform truth.
+ try:
+ total = int(subprocess.run(
+ ["/usr/sbin/sysctl", "-n", "hw.memsize"],
+ capture_output=True, text=True, timeout=5).stdout.strip() or 0)
+ if total <= 0:
+ return 0, 0
+ avail = total // 2 # conservative fallback
+ try:
+ out = subprocess.run(["/usr/bin/vm_stat"], capture_output=True,
+ text=True, timeout=5).stdout
+ page_m = re.search(r"page size of (\d+)", out)
+ page = int(page_m.group(1)) if page_m else 16384
+ pages = 0
+ # free + inactive + purgeable ≈ reclaimable-on-demand; the
+ # speculative pool is dropped by the OS under pressure too.
+ for key in ("Pages free", "Pages inactive", "Pages purgeable",
+ "Pages speculative"):
+ m = re.search(rf"{key}:\s+(\d+)\.", out)
+ if m:
+ pages += int(m.group(1))
+ if pages > 0:
+ avail = pages * page
+ except (OSError, ValueError):
+ pass
+ return total, avail
+ except (OSError, ValueError):
+ return 0, 0
+ # POSIX
+ try:
+ page = int(subprocess.run(["getconf", "PAGE_SIZE"], capture_output=True,
+ text=True, timeout=5).stdout or 4096)
+ total = int(subprocess.run(["getconf", "_PHYS_PAGES"], capture_output=True,
+ text=True, timeout=5).stdout or 0) * page
+ avail = total // 2 # conservative when _AVPHYS is unavailable
+ try:
+ avail = int(subprocess.run(["getconf", "_AVPHYS_PAGES"],
+ capture_output=True, text=True,
+ timeout=5).stdout or 0) * page or avail
+ except (OSError, ValueError):
+ pass
+ return total, avail
+ except (OSError, ValueError):
+ return 0, 0
+
+
+# nvidia-smi lives at a fixed path under the driver install; PATH presence
+# varies by session type (services and gateways often run with a minimal
+# environment) and by driver generation (legacy NVSMI dir was never on
+# PATH). Resolution result is cached: the driver doesn't move mid-process.
+_smi_path_cache: "tuple[str | None] | None" = None
+
+
+def _nvidia_smi_path() -> str | None:
+ """Absolute path to nvidia-smi, or None. PATH first (respects user
+ overrides), then the driver's known install locations on Windows;
+ on Linux/WSL the PATH lookup is the whole ladder."""
+ global _smi_path_cache
+ if _smi_path_cache is not None:
+ return _smi_path_cache[0]
+ found = shutil.which("nvidia-smi")
+ if found is None and os.name == "nt":
+ windir = os.environ.get("SystemRoot", r"C:\Windows")
+ for candidate in (
+ # DCH drivers (every modern install) place it in System32.
+ Path(windir) / "System32" / "nvidia-smi.exe",
+ # Legacy standalone drivers used NVSMI, never on PATH.
+ Path(os.environ.get("ProgramFiles", r"C:\Program Files"))
+ / "NVIDIA Corporation" / "NVSMI" / "nvidia-smi.exe",
+ ):
+ if candidate.exists():
+ found = str(candidate)
+ break
+ _smi_path_cache = (found,)
+ return found
+
+
+def _nvidia_vram() -> tuple[int, int] | None:
+ """(total, free) MiB->bytes from nvidia-smi, or None."""
+ exe = _nvidia_smi_path()
+ if exe is None:
+ return None
+ try:
+ out = subprocess.run(
+ [exe, "--query-gpu=memory.total,memory.free",
+ "--format=csv,noheader,nounits"],
+ capture_output=True, text=True, timeout=10)
+ if out.returncode != 0 or not out.stdout.strip():
+ return None
+ total_mib, free_mib = (int(x) for x in out.stdout.strip().splitlines()[0].split(","))
+ return total_mib << 20, free_mib << 20
+ except (OSError, ValueError, subprocess.TimeoutExpired):
+ return None
+
+
+def _cuda_driver_pool() -> "tuple[int, bool | None] | None":
+ """(allocator_total_bytes, integrated_or_None) from the CUDA driver
+ API, or None when unreachable. ctypes against the driver's own DLL/SO
+ — no toolkit, no subprocess, ~ms. INTEGRATED is the vendor's own
+ unified-memory declaration; total is the pool the allocator will
+ actually hand out (on carve-out devices, several times what
+ nvidia-smi reports)."""
+ import ctypes
+
+ for name in ("nvcuda.dll", "libcuda.so.1", "libcuda.so"):
+ try:
+ cuda = ctypes.CDLL(name)
+ break
+ except OSError:
+ continue
+ else:
+ return None
+ try:
+ if cuda.cuInit(0) != 0:
+ return None
+ dev = ctypes.c_int()
+ if cuda.cuDeviceGet(ctypes.byref(dev), 0) != 0:
+ return None
+ total = ctypes.c_size_t()
+ getter = getattr(cuda, "cuDeviceTotalMem_v2", None) or cuda.cuDeviceTotalMem
+ if getter(ctypes.byref(total), dev) != 0 or total.value <= 0:
+ return None
+ integrated: bool | None = None
+ attr = ctypes.c_int()
+ if cuda.cuDeviceGetAttribute(
+ ctypes.byref(attr), _CU_DEVICE_ATTRIBUTE_INTEGRATED, dev) == 0:
+ integrated = bool(attr.value)
+ return total.value, integrated
+ except (OSError, AttributeError):
+ return None
+
+
+def _engine_device_pool() -> "tuple[int, bool | None] | None":
+ """(engine_total_bytes, None) from the installed runtime's own
+ --list-devices, or None. The fallback truth source when the driver
+ API is unreachable: asks the exact binary that will do the
+ allocating. Carries no integrated verdict — callers must gate it."""
+ try:
+ from hermes_cli.local_runtime.binaries import (
+ installed_tags,
+ runtimes_root,
+ server_binary,
+ )
+
+ tags = installed_tags()
+ if not tags:
+ return None
+ tag_dir = runtimes_root() / tags[0]
+ backend_dirs = [d for d in tag_dir.iterdir() if d.is_dir()]
+ if not backend_dirs:
+ return None
+ exe = server_binary(backend_dirs[0])
+ out = subprocess.run([str(exe), "--list-devices"], capture_output=True,
+ text=True, timeout=30, cwd=str(exe.parent))
+ if out.returncode != 0:
+ return None
+ for line in (out.stdout + out.stderr).splitlines():
+ m = _DEVICE_LINE_RE.search(line)
+ if m:
+ return int(m.group(1)) << 20, None
+ return None
+ except Exception: # noqa: BLE001 — a probe miss must never block budgeting
+ return None
+
+
+def _device_pool_view() -> "tuple[int, bool | None] | None":
+ """Best available allocator-side view, cached: a hit is permanent for
+ the process, a miss retries after a short TTL (the engine binary can
+ appear mid-session via a pane install)."""
+ global _pool_probe_cache
+ now = time.monotonic()
+ if _pool_probe_cache is not None:
+ stamp, view = _pool_probe_cache
+ if view is not None or now - stamp < _POOL_NEGATIVE_TTL_S:
+ return view
+ view = _cuda_driver_pool() or _engine_device_pool()
+ _pool_probe_cache = (now, view)
+ return view
+
+
+def _unified_pool_bytes(smi_total: int, ram_total: int) -> int | None:
+ """The real pool size when this NVIDIA device is unified memory behind
+ a WDDM carve-out, else None (trust nvidia-smi as ever).
+
+ The driver's INTEGRATED attribute decides when readable — in BOTH
+ directions (0 pins discrete even if the numbers look weird; a driver
+ that declares integrated is believed even at modest pool sizes). Only
+ an attribute-less view (engine fallback) needs the two numeric gates;
+ both must hold and no discrete card meets either.
+ """
+ view = _device_pool_view()
+ if view is None:
+ return None
+ pool, integrated = view
+ if integrated is False:
+ return None
+ if integrated is True:
+ return pool
+ if (smi_total > 0 and pool >= int(smi_total * _POOL_DISAGREEMENT_FACTOR)
+ and ram_total > 0 and pool >= int(ram_total * _POOL_RAM_FRACTION)):
+ return pool
+ return None
+
+
+def probe_budget(*, planning: bool = False) -> HardwareBudget:
+ """Construct the budget per the source rules above.
+
+ ``planning=False`` (default): LIVE budget — free VRAM right now. The
+ right input for launch-time fit decisions and growth re-grants.
+
+ ``planning=True``: CAPACITY budget — what this machine can run once
+ the runtime manages placement (total device memory minus the margin).
+ The right input for catalog pricing and quant selection: pricing
+ against live-free while a model is already loaded made every row read
+ 'larger than your GPU memory' and degraded quant picks to Q2 on a
+ 32 GiB card. The managed server
+ unloads/relaunches models itself, so at load time the capacity is
+ genuinely available.
+ """
+ ram_total, ram_avail = _ram_bytes()
+ vram = _nvidia_vram()
+
+ # Unified-memory NVIDIA: the CUDA allocator pool is the real
+ # capacity. Classification comes from the driver API/engine — it
+ # must not require nvidia-smi (stripped-PATH sessions lose smi but
+ # nvcuda loads via the system loader regardless). Crossing the
+ # carve-out costs nothing (effective bandwidth is flat through the
+ # boundary; smi's used/total merely saturate at it) — the carve-out
+ # is an OS accounting knob, not a GPU limit. Deliberately NOT
+ # clamped to OS RAM: carved-out memory is invisible to
+ # GlobalMemoryStatusEx (the OS reports correspondingly less total
+ # RAM), so a RAM clamp would throw away exactly the carved capacity.
+ unified = _unified_pool_bytes(vram[0] if vram else 0, ram_total)
+ if unified is not None:
+ logger.info(
+ "unified-memory NVIDIA device: allocator pool %.1f GiB "
+ "(nvidia-smi carve-out: %s); budgeting from the pool",
+ unified / _GIB,
+ f"{vram[0] / _GIB:.1f} GiB" if vram else "unavailable")
+ if planning:
+ base = unified
+ else:
+ # Live: dedicated-free plus what the OS can still give. smi's
+ # free saturates at the carve-out so this under-counts a bit —
+ # the safe direction (the pool edge is a measured soft cliff:
+ # decode collapses ~3.5x when concurrent demand hits it).
+ # Without smi, OS-available alone is the honest floor.
+ live = (vram[1] + ram_avail) if vram else ram_avail
+ base = min(unified, live)
+ usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION)))
+ return HardwareBudget(usable_vram_bytes=usable,
+ total_device_bytes=unified,
+ ram_available_bytes=0, uma=True)
+
+ if vram is None:
+ # No NVIDIA device visible: Metal/Vulkan/CPU paths budget from RAM
+ # as UMA (Apple Silicon) — conservative for discrete AMD until a
+ # vendor probe lands (E3 hardware).
+ base = ram_total if planning else ram_avail
+ usable = max(0, int(base * (1 - _UMA_HEADROOM_FRACTION)))
+ return HardwareBudget(usable_vram_bytes=usable,
+ total_device_bytes=ram_total,
+ ram_available_bytes=0, uma=True)
+
+ total, free = vram
+ margin = max(_MARGIN_FLOOR, int(total * _MARGIN_FRACTION))
+ base = total if planning else free
+ return HardwareBudget(usable_vram_bytes=max(0, base - margin),
+ total_device_bytes=total,
+ ram_available_bytes=ram_avail if not planning else ram_total,
+ uma=False)
diff --git a/hermes_cli/local_runtime/hf_browse.py b/hermes_cli/local_runtime/hf_browse.py
new file mode 100644
index 0000000000..e554ee1fc4
--- /dev/null
+++ b/hermes_cli/local_runtime/hf_browse.py
@@ -0,0 +1,161 @@
+"""Browse Hugging Face for GGUF models the user can run.
+
+The curated catalog is the front page; this module is the firehose behind
+it — day-0 models not yet in the catalog, community quants,
+anything. Three rules keep it safe and honest:
+
+1. Acquisition only. Nothing here serves a model: a browsed download
+ lands in the machine-scoped models dir and from that moment the
+ normal machinery owns it — staleness bounce, preset generation from
+ the real GGUF header, fit policy, placement pills.
+2. The fit verdict shown BEFORE download is a rough cut priced from file
+ size alone (weights dominate; KV/overhead use conservative fill-ins).
+ After download the GGUF header is the authority, as everywhere.
+3. HF is queried directly with short timeouts and a small in-process
+ cache. No third-party proxy service; if HF rate limits ever bite at
+ fleet scale, revisit with a caching proxy then.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import re
+import time
+import urllib.parse
+import urllib.request
+from dataclasses import dataclass, field
+
+logger = logging.getLogger(__name__)
+
+_HF = "https://huggingface.co"
+_TIMEOUT_S = 15
+# Rough-fit fill-ins for pre-download pricing: a mid-size model's 64K-floor
+# KV plus runtime overhead. Deliberately round numbers — the verdict bands
+# are coarse (fits GPU / needs RAM / too big), not window grants.
+_ROUGH_KV_AND_OVERHEAD = 4 << 30
+
+# Tiny TTL cache: the pane fires a search per keystroke pause and re-opens
+# repos the user flips between. Process-local, size-capped, no invalidation
+# subtleties — upstream truth changes slowly at this granularity.
+_CACHE: dict[str, tuple[float, object]] = {}
+_CACHE_TTL_S = 300
+_CACHE_MAX = 128
+
+
+def _get_json(url: str) -> object:
+ now = time.monotonic()
+ hit = _CACHE.get(url)
+ if hit and now - hit[0] < _CACHE_TTL_S:
+ return hit[1]
+ req = urllib.request.Request(url, headers={"User-Agent": "hermes-local-models"})
+ with urllib.request.urlopen(req, timeout=_TIMEOUT_S) as r:
+ data = json.load(r)
+ if len(_CACHE) >= _CACHE_MAX:
+ _CACHE.pop(min(_CACHE, key=lambda k: _CACHE[k][0]))
+ _CACHE[url] = (now, data)
+ return data
+
+
+@dataclass(frozen=True)
+class HFModelHit:
+ repo: str # e.g. "unsloth/Qwen3.8-27B-GGUF"
+ downloads: int
+ likes: int
+ updated: str # ISO date from HF
+ gated: bool
+
+
+@dataclass(frozen=True)
+class HFFileGroup:
+ """One downloadable quant: a single GGUF or all parts of a split one."""
+
+ label: str # e.g. "Q4_K_M" or the file stem
+ paths: tuple[str, ...] # repo-relative, split parts in order
+ total_bytes: int
+ fit: str = "unknown" # fits-gpu | needs-ram | too-big | unknown
+
+
+_QUANT_RE = re.compile(
+ r"(?:IQ|Q)\d[_A-Z0-9]*|F16|BF16|F32", re.IGNORECASE)
+_SPLIT_RE = re.compile(r"-(\d{5})-of-(\d{5})\.gguf$", re.IGNORECASE)
+
+
+def search_models(query: str, limit: int = 20) -> list[HFModelHit]:
+ """Full-text search over HF models that ship GGUF files, most
+ downloaded first (the closest public signal to 'trending')."""
+ q = urllib.parse.quote(query.strip())
+ url = (f"{_HF}/api/models?search={q}&filter=gguf&sort=downloads"
+ f"&direction=-1&limit={max(1, min(int(limit), 50))}")
+ out: list[HFModelHit] = []
+ for m in _get_json(url):
+ out.append(HFModelHit(
+ repo=str(m.get("id", "")),
+ downloads=int(m.get("downloads") or 0),
+ likes=int(m.get("likes") or 0),
+ updated=str(m.get("lastModified") or ""),
+ gated=bool(m.get("gated")),
+ ))
+ return out
+
+
+def _quant_label(filename: str) -> str:
+ m = _QUANT_RE.search(filename)
+ return m.group(0).upper() if m else filename
+
+
+def repo_files(repo: str) -> list[HFFileGroup]:
+ """The servable GGUFs in a repo, grouped: split parts collapse into one
+ entry (first part is what llama.cpp loads), mmproj/draft companions are
+ excluded (they aren't standalone models). Largest quant first."""
+ url = f"{_HF}/api/models/{urllib.parse.quote(repo)}/tree/main?recursive=true"
+ files = _get_json(url)
+
+ singles: list[tuple[str, int]] = []
+ splits: dict[str, list[tuple[int, str, int]]] = {}
+ for f in files:
+ path = str(f.get("path", ""))
+ if not path.lower().endswith(".gguf"):
+ continue
+ name = path.rsplit("/", 1)[-1].lower()
+ if name.startswith("mmproj") or name.startswith("dspark") or "draft" in name:
+ continue
+ size = int(f.get("size") or 0)
+ m = _SPLIT_RE.search(path)
+ if m:
+ stem = path[: m.start()]
+ splits.setdefault(stem, []).append((int(m.group(1)), path, size))
+ else:
+ singles.append((path, size))
+
+ groups: list[HFFileGroup] = []
+ for path, size in singles:
+ groups.append(HFFileGroup(label=_quant_label(path), paths=(path,),
+ total_bytes=size))
+ for stem, parts in splits.items():
+ parts.sort()
+ groups.append(HFFileGroup(
+ label=_quant_label(stem),
+ paths=tuple(p for _, p, _ in parts),
+ total_bytes=sum(s for _, _, s in parts)))
+ groups.sort(key=lambda g: g.total_bytes, reverse=True)
+ return groups
+
+
+def rough_fit(total_bytes: int, budget) -> str:
+ """Coarse pre-download verdict from file size alone. The GGUF header
+ refines this after download; bands match the catalog pills' language.
+ File size ≈ in-memory weights for GGUF (mmap'd as-is)."""
+ need = total_bytes + _ROUGH_KV_AND_OVERHEAD
+ if need <= budget.usable_vram_bytes:
+ return "fits-gpu"
+ if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
+ return "needs-ram"
+ return "too-big"
+
+
+def priced_repo_files(repo: str, budget) -> list[HFFileGroup]:
+ from dataclasses import replace
+
+ return [replace(g, fit=rough_fit(g.total_bytes, budget))
+ for g in repo_files(repo)]
diff --git a/hermes_cli/local_runtime/load_progress.py b/hermes_cli/local_runtime/load_progress.py
new file mode 100644
index 0000000000..379955c6e8
--- /dev/null
+++ b/hermes_cli/local_runtime/load_progress.py
@@ -0,0 +1,198 @@
+"""Live model-load progress from the managed llama-server router.
+
+llama-server's child processes emit per-tensor load progress
+({stages, current, value}, throttled upstream to ~200ms) which the
+router relays ONLY over its /models/sse stream — GET /models carries
+just the coarse status string. This module owns one lazy background
+watcher on that stream and keeps an in-memory snapshot other code can
+poll cheaply:
+
+ get_loading_progress() -> {model_id: {"stage", "value", "percent"}}
+
+"percent" is a composite across stages so a bar doesn't sprint 0->100
+once per stage: the text model dominates load time (its weights dwarf
+the mmproj/spec extras), so it gets the lion's share of the range and
+the extras split the remainder.
+
+The watcher starts on first call, reconnects with backoff (the router
+bounces on model download/eject), and never raises into callers — no
+router, no state file, or no SSE support (older engines) all read as
+"nothing loading". Safe from any process on the machine: the endpoint
+comes from the supervisor's machine-scoped state file.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import threading
+import time
+import urllib.request
+
+logger = logging.getLogger(__name__)
+
+_TEXT_STAGE_SHARE = 0.85 # composite range share for the text model
+_RECONNECT_DELAY_S = 3.0
+_STALE_ENTRY_TTL_S = 120.0 # a loading entry with no events this long is dead
+
+_lock = threading.Lock()
+_watcher: threading.Thread | None = None
+_snapshot: dict[str, dict] = {}
+
+
+def _composite_percent(stages: list[str], current: str, value: float) -> int:
+ """Map (stage, in-stage value) onto one 0-100 range, text-heavy."""
+ if not stages or current not in stages or len(stages) == 1:
+ return max(0, min(100, round(value * 100)))
+ extras = [s for s in stages if s != "text_model"]
+ extra_share = (1.0 - _TEXT_STAGE_SHARE) / len(extras) if extras else 0.0
+ offset = 0.0
+ for stage in stages:
+ share = _TEXT_STAGE_SHARE if stage == "text_model" else extra_share
+ if stage == current:
+ return max(0, min(100, round((offset + share * value) * 100)))
+ offset += share
+ return max(0, min(100, round(value * 100)))
+
+
+def _endpoint() -> "tuple[str, str] | None":
+ """(base_root, api_key) of the managed router, or None.
+
+ Resolved through the endpoint module's ownership-guarded reader, not
+ a raw state-file read: on the shared stable port, a foreign install's
+ server answers /health for anyone, and a raw read would attach this
+ watcher to someone else's SSE stream (or spin on 401s against it).
+ The guard's dead-pid check is the ownership proof."""
+ try:
+ from hermes_cli.local_runtime.endpoint import _state_endpoint
+
+ state = _state_endpoint()
+ if state is None:
+ return None
+ base = str(state.get("base_url", "")).rsplit("/v1", 1)[0]
+ return (base, str(state.get("api_key", ""))) if base else None
+ except Exception: # noqa: BLE001
+ return None
+
+
+def _apply_event(model: str, event: str, data: dict) -> None:
+ with _lock:
+ status = str(data.get("status", ""))
+ if event in ("status_change", "model_status") and status == "loading":
+ progress = data.get("progress") or {}
+ stages = [str(s) for s in (progress.get("stages") or [])]
+ current = str(progress.get("current", ""))
+ value = progress.get("value")
+ entry = _snapshot.setdefault(model, {"stage": "", "value": 0.0,
+ "percent": 0, "ts": 0.0})
+ entry["ts"] = time.monotonic()
+ if current and isinstance(value, (int, float)):
+ entry["stage"] = current
+ entry["value"] = float(value)
+ entry["percent"] = _composite_percent(stages, current, float(value))
+ elif event in ("status_change", "model_status", "model_remove"):
+ # Any terminal status (loaded/unloaded/failed) ends the load.
+ if status != "loading":
+ _snapshot.pop(model, None)
+
+
+def _watch() -> None:
+ while True:
+ endpoint = _endpoint()
+ if endpoint is None:
+ with _lock:
+ _snapshot.clear()
+ time.sleep(_RECONNECT_DELAY_S)
+ continue
+ base, key = endpoint
+ try:
+ req = urllib.request.Request(
+ f"{base}/models/sse",
+ headers={"Authorization": f"Bearer {key}",
+ "Accept": "text/event-stream"})
+ with urllib.request.urlopen(req, timeout=60) as r:
+ buf = b""
+ while True:
+ chunk = r.read1(4096) if hasattr(r, "read1") else r.read(4096)
+ if not chunk:
+ break
+ buf += chunk
+ while b"\n" in buf:
+ line, buf = buf.split(b"\n", 1)
+ text = line.decode("utf-8", "replace").strip()
+ if not text.startswith("data:"):
+ continue
+ try:
+ msg = json.loads(text[5:].strip())
+ _apply_event(str(msg.get("model", "")),
+ str(msg.get("event", "")),
+ msg.get("data") or {})
+ except (json.JSONDecodeError, TypeError):
+ continue
+ except Exception as exc: # noqa: BLE001 — watcher must never die loud
+ logger.debug("load-progress SSE reconnecting: %s", exc)
+ # Stream ended (router bounce, timeout, error): loading entries from
+ # the dead connection are unverifiable — drop rather than freeze.
+ with _lock:
+ _snapshot.clear()
+ time.sleep(_RECONNECT_DELAY_S)
+
+
+def _ensure_watcher() -> None:
+ global _watcher
+ with _lock:
+ if _watcher is None or not _watcher.is_alive():
+ _watcher = threading.Thread(target=_watch, daemon=True,
+ name="llamacpp-load-progress")
+ _watcher.start()
+
+
+def get_loading_progress() -> dict[str, dict]:
+ """{model_id: {"stage", "value", "percent"}} for models loading right
+ now. Empty when nothing is loading (or nothing is knowable)."""
+ _ensure_watcher()
+ now = time.monotonic()
+ with _lock:
+ return {m: {"stage": e["stage"], "value": e["value"],
+ "percent": e["percent"]}
+ for m, e in _snapshot.items()
+ if now - e["ts"] < _STALE_ENTRY_TTL_S}
+
+
+def get_prefill_progress(model: str) -> "dict | None":
+ """{"processed": tokens} while the managed server is prompt-processing
+ for ``model``, or None (idle, decoding, unreachable, or foreign server).
+
+ llama-server's /slots reports ``n_prompt_tokens_processed`` climbing in
+ real time during prefill, but exposes no total — callers supply their
+ own denominator (the request's estimated token count). Busiest
+ processing slot wins when several are active: a parallel small request
+ (title generation) freezes its counter during decode while a live
+ prefill keeps climbing past it. One authenticated HTTP call per poll;
+ every failure reads as "no prefill" — this is garnish, never load-
+ bearing.
+ """
+ ep = _endpoint()
+ if ep is None:
+ return None
+ base, key = ep
+ try:
+ from urllib.parse import quote
+
+ req = urllib.request.Request(
+ f"{base}/slots?model={quote(model)}",
+ headers={"Authorization": f"Bearer {key}"})
+ with urllib.request.urlopen(req, timeout=2) as r:
+ slots = json.loads(r.read())
+ except Exception: # noqa: BLE001
+ return None
+ best = 0
+ for slot in slots if isinstance(slots, list) else []:
+ if not slot.get("is_processing"):
+ continue
+ try:
+ processed = int(slot.get("n_prompt_tokens_processed") or 0)
+ except (TypeError, ValueError):
+ continue
+ best = max(best, processed)
+ return {"processed": best} if best > 0 else None
diff --git a/hermes_cli/local_runtime/presets.py b/hermes_cli/local_runtime/presets.py
new file mode 100644
index 0000000000..a5aa96cfa3
--- /dev/null
+++ b/hermes_cli/local_runtime/presets.py
@@ -0,0 +1,265 @@
+"""Per-model preset generation (--models-preset INI) — the router-side
+carrier for context-policy launch decisions.
+
+The INI shape is what the router itself generates per child: a
+[model-id] section whose keys are long-form
+llama-server flag names without the leading dashes.
+"""
+
+from __future__ import annotations
+
+import logging
+from dataclasses import dataclass
+from pathlib import Path
+
+from hermes_cli.local_runtime.context_policy import (
+ RUNTIME_OVERHEAD_BYTES,
+ WindowDecision,
+ initial_window,
+ launch_args,
+ ub_logits_bytes,
+)
+from hermes_cli.local_runtime.estimator import (
+ HardwareBudget,
+ PhysicsRefusal,
+ profile_from_gguf,
+)
+from hermes_cli.local_runtime.gguf import read_gguf_header
+
+logger = logging.getLogger(__name__)
+
+# args list -> INI keys. Flags the policy owns; everything else stays out
+# of the preset (recipe sampling defaults merge in a later pass).
+_FLAG_TO_KEY = {
+ "-c": "ctx-size",
+ "-b": "batch-size",
+ "-ub": "ubatch-size",
+ "-ctk": "cache-type-k",
+ "-ctv": "cache-type-v",
+ "-fa": "flash-attn",
+ "-ot": "override-tensor",
+ "--spec-type": "spec-type",
+ "--spec-draft-n-max": "spec-draft-n-max",
+}
+
+
+@dataclass
+class PresetEntry:
+ model_id: str
+ window: int
+ spilled: bool
+ refusal: str | None = None
+ keys: dict[str, str] | None = None
+
+
+def _args_to_keys(args: list[str]) -> dict[str, str]:
+ keys: dict[str, str] = {}
+ i = 0
+ while i < len(args):
+ flag = args[i]
+ key = _FLAG_TO_KEY.get(flag)
+ if key is None:
+ i += 1
+ continue
+ keys[key] = args[i + 1]
+ i += 2
+ return keys
+
+
+def generate_presets(models_dir: Path, budget: HardwareBudget,
+ preset_path: Path,
+ mtp_capable: set[str] | None = None) -> list[PresetEntry]:
+ """Walk the staged models, run the launch decision per model, and
+ write one INI. Refused models get no section (the router simply won't
+ have policy for them; the picker surfaces the refusal + smaller-quant
+ suggestion from the returned entries).
+
+ Catalog-declared companions merge in here: sampling defaults (policy
+ keys always win), the vision projector when present, and a spec-decode
+ draft model iff the decision spilled — the rule: speculative
+ decode is a spill amplifier, so a resident draft accelerates a spilled
+ main model; a zero-spill model doesn't pay the draft's memory."""
+ from hermes_cli.local_runtime.bootstrap import assets_dir
+ from hermes_cli.local_runtime.catalog import find_entry_for_model
+
+ entries: list[PresetEntry] = []
+ sections: list[str] = []
+ for gguf in _staged_in(models_dir):
+ model_id = _strip_part(gguf.stem)
+ try:
+ header = read_gguf_header(gguf)
+ profile = profile_from_gguf(header)
+ except (ValueError, OSError) as exc:
+ logger.warning("preset skip %s: %s", gguf.name, exc)
+ continue
+ # Overhead beyond weights+KV: runtime buffers, the vision projector
+ # when this model ships one, and the logits buffers of whichever
+ # microbatch/MTP posture launch_args will choose — flag and price
+ # decided together, from the same facts.
+ hit = find_entry_for_model(model_id)
+ entry = hit[0] if hit is not None else None
+ is_mtp = (entry.mtp if entry is not None
+ else model_id in (mtp_capable or set()))
+ if is_mtp and profile.kv_scale == 1.0:
+ # Header-derived profiles don't know about MTP's draft
+ # context; apply the calibrated KV multiplier here so the
+ # launch fit prices what the server will actually allocate.
+ import dataclasses
+
+ profile = dataclasses.replace(profile, kv_scale=1.2)
+ mmproj_bytes = 0
+ if entry is not None and entry.mmproj is not None:
+ mmproj_path = assets_dir() / entry.mmproj.local_name
+ if mmproj_path.exists():
+ mmproj_bytes = entry.mmproj.size_bytes
+ # MTP posture ladder — window first, prefill second: price the
+ # launch under both postures and keep whichever grants the larger
+ # window (the stacked posture's bigger compute buffer buys ~3x
+ # short-prompt prefill but costs ~2 GiB that would otherwise be
+ # window; measured at 256K the ub512 posture still prefills at
+ # 2.7K tok/s, so window wins ties only one way: never trade
+ # context away for prefill). Same window -> stacked.
+ mtp_prefill = False
+ logits_bytes = ub_logits_bytes(profile.n_vocab, mtp_capable=is_mtp)
+ if is_mtp:
+ stacked_logits = ub_logits_bytes(profile.n_vocab, mtp_capable=True,
+ mtp_prefill=True)
+ stacked_probe = initial_window(
+ profile, budget,
+ overhead_bytes=(RUNTIME_OVERHEAD_BYTES + mmproj_bytes
+ + stacked_logits))
+ plain_probe = initial_window(
+ profile, budget,
+ overhead_bytes=(RUNTIME_OVERHEAD_BYTES + mmproj_bytes
+ + logits_bytes))
+ if (not isinstance(stacked_probe, PhysicsRefusal)
+ and not stacked_probe.spilled
+ and (isinstance(plain_probe, PhysicsRefusal)
+ or stacked_probe.window >= plain_probe.window)):
+ mtp_prefill = True
+ logits_bytes = stacked_logits
+ decision = initial_window(
+ profile, budget,
+ overhead_bytes=RUNTIME_OVERHEAD_BYTES + mmproj_bytes + logits_bytes)
+ if isinstance(decision, PhysicsRefusal):
+ entries.append(PresetEntry(model_id=model_id, window=0,
+ spilled=False, refusal=decision.message))
+ continue
+
+ # Session growth (growth.py): a persisted override lifts the launch
+ # window to where the ladder last grew it — capped at native, and
+ # only when physics still clears the bigger window on THIS boot's
+ # budget (a smaller-VRAM day re-fits honestly back down).
+ try:
+ from hermes_cli.local_runtime.estimator import ctx_bytes
+ from hermes_cli.local_runtime.growth import load_window_overrides
+
+ override = load_window_overrides().get(model_id)
+ native = profile.n_ctx_train or decision.window
+ if override and override > decision.window:
+ target = min(int(override), native)
+ kv = ctx_bytes(profile, target)
+ need = (profile.weights_bytes + kv
+ + RUNTIME_OVERHEAD_BYTES + mmproj_bytes + logits_bytes)
+ if need <= budget.usable_vram_bytes + budget.ram_available_bytes:
+ spill = max(0, need - budget.usable_vram_bytes)
+ decision = WindowDecision(
+ window=target, spill_bytes=spill,
+ kv_on_gpu=kv <= budget.usable_vram_bytes,
+ reasons=[f"grown window restored ({target // 1024}K)"])
+ except Exception as exc: # noqa: BLE001 — overrides are advisory
+ logger.debug("window override skipped for %s: %s", model_id, exc)
+
+ # (entry and is_mtp resolved above, where the overhead was priced —
+ # the launch flags below MUST match that pricing.)
+ args = launch_args(profile, decision, mtp_capable=is_mtp,
+ mtp_draft_depth=(entry.mtp_draft_depth
+ if entry is not None else 3),
+ uma=budget.uma, mtp_prefill=mtp_prefill)
+ keys = _args_to_keys(args)
+
+ if entry is not None and is_mtp:
+ # Integrated-MTP targets sample on the backend, and so does
+ # the draft (pairing validated against the vendor's published
+ # llama.cpp recipes for these models).
+ keys["backend-sampling"] = "on"
+ keys["spec-draft-backend-sampling"] = "on"
+
+ # Sampling deference ladder, under the policy keys (policy wins
+ # on clash). The GGUF's own general.sampling.* metadata is the
+ # publisher's recommendation — it arrives with the file, updates
+ # with every re-upload, and covers models the catalog has never
+ # heard of. Catalog sampling applies only where the file is
+ # silent; a model carrying neither runs llama.cpp defaults.
+ for k, v in header.sampling_defaults.items():
+ keys.setdefault(k, v)
+ if entry is not None:
+ for k, v in (entry.sampling or {}).items():
+ keys.setdefault(k, v)
+ if entry.mmproj is not None:
+ mmproj_path = assets_dir() / entry.mmproj.local_name
+ if mmproj_path.exists():
+ keys["mmproj"] = str(mmproj_path)
+ if entry.draft is not None and decision.spilled:
+ draft_path = assets_dir() / entry.draft.local_name
+ if draft_path.exists():
+ keys["model-draft"] = str(draft_path)
+ keys["spec-type"] = "draft-dspark"
+ # Unsloth's measured cliff: acceptance 83% at 2-3
+ # drafts, collapses at 4.
+ keys["spec-draft-n-max"] = "3"
+
+ entries.append(PresetEntry(model_id=model_id, window=decision.window,
+ spilled=decision.spilled, keys=keys))
+ body = "\n".join(f"{k} = {v}" for k, v in keys.items())
+ sections.append(f"[{model_id}]\n{body}\n")
+
+ preset_path.parent.mkdir(parents=True, exist_ok=True)
+ preset_path.write_text("\n".join(sections), encoding="utf-8")
+ logger.info("wrote %d preset sections to %s", len(sections), preset_path)
+ return entries
+
+
+def read_preset_decisions(preset_path: Path | None = None) -> dict[str, PresetEntry]:
+ """The launch decisions the running server was actually given, read
+ back from the preset INI (the INI is the record — it's what spawned
+ the children). Missing/unparseable file returns {}."""
+ import configparser
+
+ if preset_path is None:
+ from hermes_cli.local_runtime.binaries import runtimes_root
+
+ preset_path = runtimes_root() / "presets.ini"
+ out: dict[str, PresetEntry] = {}
+ try:
+ parser = configparser.ConfigParser()
+ parser.read(preset_path, encoding="utf-8")
+ for section in parser.sections():
+ window = parser.getint(section, "ctx-size", fallback=0)
+ spilled = parser.has_option(section, "override-tensor")
+ out[section] = PresetEntry(model_id=section, window=window,
+ spilled=spilled)
+ except Exception as exc: # noqa: BLE001
+ logger.debug("preset read-back failed: %s", exc)
+ return out
+
+
+def _strip_part(stem: str) -> str:
+ import re
+
+ return re.sub(r"-\d{5}-of-\d{5}$", "", stem)
+
+
+def _staged_in(models_dir: Path) -> "list[Path]":
+ """Servable models in an arbitrary directory (split first-parts only) —
+ the validation harness points at non-default dirs."""
+ import re
+
+ part = re.compile(r"-(\d{5})-of-\d{5}\.gguf$")
+ out = []
+ for p in sorted(models_dir.glob("*.gguf")):
+ m = part.search(p.name)
+ if m and m.group(1) != "00001":
+ continue
+ out.append(p)
+ return out
diff --git a/hermes_cli/local_runtime/supervisor.py b/hermes_cli/local_runtime/supervisor.py
new file mode 100644
index 0000000000..44f162cf48
--- /dev/null
+++ b/hermes_cli/local_runtime/supervisor.py
@@ -0,0 +1,500 @@
+"""Supervision of one llama-server in router mode.
+
+The router process is ours (restart with backoff on crash); router children
+are its problem — child failures surface via GET /models exit_code, never
+auto-retried here.
+
+Readiness rules (each learned the hard way on real hardware):
+- health-200 is NOT readiness; every readiness claim requires a touch
+ generation (temp-0, expected token, generous budget, reasoning_content
+ scanned).
+- Always dial 127.0.0.1 — resolving localhost adds ~2s per request on
+ Windows via IPv6 fallback.
+- /metrics is opt-in (--metrics) and carries no KV-usage metric;
+ idleness = requests_processing == 0 and no slot is_processing.
+- The router's LRU eviction has no pin for the primary model: until an
+ upstream pin exists, keep_primary_loaded re-touches the primary after
+ any other model load.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import secrets
+import socket
+import subprocess
+import threading
+import time
+import urllib.error
+import urllib.request
+from pathlib import Path
+
+from hermes_cli.local_runtime.binaries import server_binary, runtimes_root
+
+logger = logging.getLogger(__name__)
+
+TOUCH_PROMPT = "Reply with exactly one word: the capital of France."
+TOUCH_EXPECT = "paris"
+_RESTART_BACKOFF_S = (1, 5, 15, 60)
+
+
+def state_path() -> Path:
+ """Endpoint state for other Hermes processes (provider resolution reads
+ this to route llamacpp-alias requests at the managed server)."""
+ return runtimes_root() / "server.json"
+
+
+def _free_port() -> int:
+ with socket.socket() as s:
+ s.bind(("127.0.0.1", 0))
+ return s.getsockname()[1]
+
+
+# Default port for the managed server, chosen once and reused across
+# restarts. Sessions persist the resolved base_url; an ephemeral port
+# would strand every resumed session on a dead endpoint after each
+# restart. Deliberately NOT 8080 so we never collide with a user's own
+# llama-server/Ollama-adjacent stack.
+_DEFAULT_PORT = 18434
+
+
+def _stable_port() -> int:
+ """The stable default port, falling back to an ephemeral one only when
+ something else already listens there (and it isn't a leftover managed
+ server, which stop() would have cleaned up)."""
+ try:
+ with socket.socket() as s:
+ s.bind(("127.0.0.1", _DEFAULT_PORT))
+ return _DEFAULT_PORT
+ except OSError:
+ logger.warning(
+ "port %d busy; managed llama-server falling back to an ephemeral "
+ "port — existing sessions may need a model re-pick", _DEFAULT_PORT)
+ return _free_port()
+
+
+def _stable_api_key() -> str:
+ """One key for the life of the install, persisted beside the runtimes.
+
+ Endpoint identity must survive restarts as a UNIT — sessions persist the
+ resolved base_url + api_key, so a per-boot key strands every resumed
+ session on HTTP 401 exactly the way a per-boot port would strand them
+ on connection errors. Rotating it buys nothing: the key exists to stop
+ other loopback processes free-riding, and it lives on the same disk as
+ the state file that would leak it. Delete the file to rotate manually.
+ """
+ key_path = runtimes_root() / ".api_key"
+ try:
+ existing = key_path.read_text(encoding="utf-8").strip()
+ if len(existing) >= 16:
+ return existing
+ except OSError:
+ pass
+ key = secrets.token_urlsafe(24)
+ try:
+ key_path.parent.mkdir(parents=True, exist_ok=True)
+ key_path.write_text(key, encoding="utf-8")
+ except OSError as exc:
+ logger.warning("could not persist api key (%s); sessions will need "
+ "a re-pick after restart", exc)
+ return key
+
+
+class LlamaServerSupervisor:
+ """Own one llama-server router process for the life of a Hermes session.
+
+ Usage::
+
+ sup = LlamaServerSupervisor(install_dir, models_dir)
+ sup.start() # spawn + wait healthy
+ sup.ensure_model_ready(name) # load + touch-generate
+ ... sup.base_url is the /v1 endpoint, sup.api_key its key ...
+ sup.stop()
+ """
+
+ def __init__(self, install_dir: Path, models_dir: Path, *,
+ models_max: int = 4, port: int | None = None,
+ extra_args: list[str] | None = None,
+ log_path: Path | None = None,
+ preset_path: Path | None = None):
+ self.install_dir = Path(install_dir)
+ self.models_dir = Path(models_dir)
+ self.models_max = models_max
+ self.port = port or _stable_port()
+ self.api_key = _stable_api_key()
+ self.extra_args = list(extra_args or [])
+ self.log_path = log_path or (self.models_dir.parent / "logs" / "llama-server.log")
+ self.preset_path = preset_path
+ self.proc: subprocess.Popen | None = None
+ self.primary_model: str | None = None
+ self._restarts = 0
+ self._stopping = False
+ self._watchdog: threading.Thread | None = None
+ self._log_handle = None
+ self._idle_since: dict[str, float] = {}
+
+ # ── endpoints ────────────────────────────────────────────
+
+ @property
+ def base_url(self) -> str:
+ return f"http://127.0.0.1:{self.port}/v1"
+
+ def _url(self, route: str) -> str:
+ return f"http://127.0.0.1:{self.port}{route}"
+
+ def _request(self, route: str, body: dict | None = None, timeout_s: int = 30) -> dict:
+ req = urllib.request.Request(
+ self._url(route),
+ data=json.dumps(body).encode() if body is not None else None,
+ headers={"Content-Type": "application/json",
+ "Authorization": f"Bearer {self.api_key}"},
+ )
+ with urllib.request.urlopen(req, timeout=timeout_s) as r:
+ raw = r.read()
+ return json.loads(raw) if raw else {}
+
+ # ── lifecycle ────────────────────────────────────────────
+
+ def _spawn(self) -> None:
+ exe = server_binary(self.install_dir)
+ cmd = [
+ str(exe),
+ "--host", "127.0.0.1",
+ "--port", str(self.port),
+ "--api-key", self.api_key,
+ "--models-dir", str(self.models_dir),
+ "--models-max", str(self.models_max),
+ # The residency contract at the layer that sees every message:
+ # a chat request to a staged-but-unloaded model loads it (slow
+ # first token) instead of failing with 'model not found' —
+ # without this flag, chat after an eject is a bare 400/404.
+ "--models-autoload",
+ "--metrics", # opt-in flag; supervisor telemetry needs it
+ "--slots", # /slots endpoint is also opt-in; is_idle reads it
+ "--no-webui",
+ "--jinja",
+ # Direct I/O on model load: bypasses the page cache, so a
+ # multi-GB load doesn't evict half the OS cache — measured
+ # faster loads on NVMe, and our router bounces (download/
+ # delete/activate) reload models often enough to care.
+ "-dio",
+ ]
+ if self.preset_path and self.preset_path.exists():
+ cmd += ["--models-preset", str(self.preset_path)]
+ cmd += [
+ *self.extra_args,
+ ]
+ self.log_path.parent.mkdir(parents=True, exist_ok=True)
+ if self._log_handle is not None:
+ # The crash-restart loop calls _spawn repeatedly; without
+ # closing the prior handle each restart leaks one fd.
+ try:
+ self._log_handle.close()
+ except Exception: # noqa: BLE001 — best-effort
+ pass
+ self._log_handle = open(self.log_path, "a", encoding="utf-8", errors="replace")
+ self._log_handle.write(f"\n# spawn: {cmd}\n")
+ self._log_handle.flush()
+ # list-args, never a shell: spaced paths (user homes) must survive.
+ self.proc = subprocess.Popen(cmd, stdout=self._log_handle,
+ stderr=subprocess.STDOUT, cwd=str(exe.parent))
+ logger.info("llama-server router spawned pid=%s port=%s", self.proc.pid, self.port)
+ # State goes down at SPAWN, not after health: endpoint resolution
+ # treats a live-pid-but-not-yet-healthy server as "starting" rather
+ # than "unconfigured", so a readiness probe racing the boot doesn't
+ # throw the app back to onboarding (observed on first restart test).
+ self._write_state()
+
+ def start(self, timeout_s: int = 120) -> None:
+ self._stopping = False
+ self._spawn()
+ self._wait_health(timeout_s)
+ self._write_state()
+ self._watchdog = threading.Thread(target=self._watch, daemon=True,
+ name="llamacpp-supervisor")
+ self._watchdog.start()
+
+ def _write_state(self) -> None:
+ path = state_path()
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text(json.dumps({
+ "base_url": self.base_url,
+ "api_key": self.api_key,
+ "pid": self.proc.pid if self.proc else None,
+ }), encoding="utf-8")
+
+ def _wait_health(self, timeout_s: int) -> None:
+ deadline = time.monotonic() + timeout_s
+ while time.monotonic() < deadline:
+ if self.proc and self.proc.poll() is not None:
+ raise RuntimeError(
+ f"llama-server exited rc={self.proc.returncode} during startup "
+ f"(log: {self.log_path})")
+ try:
+ with urllib.request.urlopen(self._url("/health"), timeout=3) as r:
+ if r.status == 200:
+ return
+ except (urllib.error.URLError, OSError, TimeoutError):
+ pass
+ time.sleep(1)
+ raise TimeoutError(f"llama-server not healthy after {timeout_s}s (log: {self.log_path})")
+
+ def _watch(self) -> None:
+ """Restart the router (not its children) on crash, with backoff."""
+ while not self._stopping:
+ proc = self.proc
+ if proc is None:
+ return
+ rc = proc.poll()
+ if rc is None:
+ time.sleep(2)
+ continue
+ if self._stopping:
+ return
+ backoff = _RESTART_BACKOFF_S[min(self._restarts, len(_RESTART_BACKOFF_S) - 1)]
+ logger.warning("llama-server exited rc=%s; restart #%s in %ss",
+ rc, self._restarts + 1, backoff)
+ time.sleep(backoff)
+ self._restarts += 1
+ try:
+ self._reap_orphaned_children()
+ self._spawn()
+ self._wait_health(120)
+ if self.primary_model:
+ self.ensure_model_ready(self.primary_model)
+ except Exception as exc: # noqa: BLE001
+ logger.error("llama-server restart failed: %s", exc)
+
+ def stop(self) -> None:
+ self._stopping = True
+ state_path().unlink(missing_ok=True)
+ if self.proc and self.proc.poll() is None:
+ self._terminate_tree(self.proc)
+ if self._log_handle:
+ self._log_handle.close()
+ self._log_handle = None
+
+ @staticmethod
+ def _terminate_tree(proc: subprocess.Popen) -> None:
+ """Terminate the router AND its model children.
+
+ The router spawns one child llama-server per loaded model, each
+ holding gigabytes of VRAM. Terminating only the router (on
+ Windows, TerminateProcess — no signal handlers, no cleanup pass)
+ orphans those children: the port goes quiet but the weights stay
+ resident, and the next spawn re-loads models alongside a ghost
+ still holding the memory. Enumerate children FIRST (the parent
+ must be alive to walk them), then terminate parent and children
+ together, escalating to kill for stragglers.
+ """
+ children: list = []
+ try:
+ import psutil
+
+ children = psutil.Process(proc.pid).children(recursive=True)
+ except Exception: # noqa: BLE001 — no psutil view; still stop the router
+ children = []
+ proc.terminate()
+ for child in children:
+ try:
+ child.terminate()
+ except Exception: # noqa: BLE001
+ pass
+ try:
+ proc.wait(timeout=15)
+ except subprocess.TimeoutExpired:
+ proc.kill()
+ for child in children:
+ try:
+ if child.is_running():
+ child.kill()
+ except Exception: # noqa: BLE001
+ pass
+
+ def _reap_orphaned_children(self) -> None:
+ """Kill model children orphaned by a router crash, before respawn.
+
+ A crashed router can't clean up its children, and a dead parent
+ can't be walked — so match by identity instead: any process
+ running OUR llama-server binary whose parent is gone is an
+ orphan of a previous router. Their VRAM must come back before
+ the new router loads models next to the ghosts. External
+ llama-servers (different binary path) never match.
+ """
+ try:
+ import psutil
+
+ exe = str(server_binary(self.install_dir))
+ except Exception: # noqa: BLE001
+ return
+ for p in psutil.process_iter(["exe", "ppid"]):
+ try:
+ if p.info.get("exe") != exe:
+ continue
+ if self.proc is not None and p.pid == self.proc.pid:
+ continue
+ ppid = p.info.get("ppid") or 0
+ if ppid and psutil.pid_exists(ppid):
+ continue
+ logger.warning("reaping orphaned llama-server child pid=%s", p.pid)
+ p.kill()
+ except (psutil.NoSuchProcess, psutil.AccessDenied):
+ continue
+
+ # ── model management (router endpoints) ──────────────────
+
+ def models(self) -> dict:
+ """{model_id: status_value} from GET /models."""
+ data = self._request("/models")
+ return {m["id"]: m.get("status", {}).get("value", "unknown")
+ for m in data.get("data", [])}
+
+ def model_failures(self) -> dict:
+ """{model_id: exit_code} for children that died — surfaced to the
+ UI, never auto-retried (design: router children are its problem)."""
+ data = self._request("/models")
+ out = {}
+ for m in data.get("data", []):
+ status = m.get("status", {})
+ if status.get("value") == "failed" or status.get("exit_code"):
+ out[m["id"]] = status.get("exit_code")
+ return out
+
+ def load_model(self, model_id: str, timeout_s: int = 600) -> None:
+ self._request("/models/load", {"model": model_id}, timeout_s=timeout_s)
+
+ def unload_model(self, model_id: str) -> None:
+ """Free the child's VRAM now. Route existence verified empirically
+ on b10290 (POST /models/unload; bogus name -> 400 'model is not
+ found'). Momentary action: never touches primary_model — the
+ declaration is durable, an eject is not (residency design).
+
+ Settle before returning: for a few seconds after unload returns,
+ the router still routes to the dying child and answers chat with
+ 500 'proxy error: Could not establish connection' (probed on
+ b10362). Waiting for the model to report unloaded means the next
+ message autoloads cleanly instead of racing the teardown.
+ """
+ self._request("/models/unload", {"model": model_id}, timeout_s=120)
+ deadline = time.monotonic() + 15
+ while time.monotonic() < deadline:
+ try:
+ if self.models().get(model_id) not in ("loaded", "ready", "unloading"):
+ return
+ except Exception: # noqa: BLE001
+ return
+ time.sleep(0.3)
+
+ # ── idle residency (non-primary models) ──────────────────
+
+ # A model that has gone quiet gets its VRAM back after this long. A
+ # constant, not a knob: long enough that an active conversation never
+ # trips it, short enough that a wandered-off session frees ~20 GiB
+ # within the hour. No exemptions (residency v2): demand reloads
+ # anything the user comes back to.
+ IDLE_UNLOAD_S = 15 * 60
+
+ def sweep_idle(self, now: float | None = None) -> list[str]:
+ """Unload models idle past IDLE_UNLOAD_S. Returns the model ids
+ unloaded. Idle means no busy slots and no queued work, tracked
+ per model across calls; a model seen busy resets its clock."""
+ now = time.monotonic() if now is None else now
+ unloaded: list[str] = []
+ try:
+ statuses = self.models()
+ except Exception: # noqa: BLE001
+ return unloaded
+ for model_id, status in statuses.items():
+ if status not in ("loaded", "ready"):
+ self._idle_since.pop(model_id, None)
+ continue
+ if not self.is_idle(model_id):
+ self._idle_since.pop(model_id, None)
+ continue
+ first_idle = self._idle_since.setdefault(model_id, now)
+ if now - first_idle >= self.IDLE_UNLOAD_S:
+ try:
+ self.unload_model(model_id)
+ self._idle_since.pop(model_id, None)
+ unloaded.append(model_id)
+ logger.info("idle-unloaded %s (idle %ds)", model_id,
+ int(now - first_idle))
+ except Exception as exc: # noqa: BLE001
+ logger.warning("idle unload of %s failed: %s", model_id, exc)
+ return unloaded
+
+ def touch_generate(self, model_id: str, timeout_s: int = 300) -> bool:
+ """The readiness proof. Generous budget + reasoning_content scan —
+ small token budgets false-fail reasoning models, which spend their
+ first tokens thinking."""
+ try:
+ resp = self._request("/v1/chat/completions", {
+ "model": model_id,
+ "messages": [{"role": "user", "content": TOUCH_PROMPT}],
+ "max_tokens": 512, "temperature": 0,
+ }, timeout_s=timeout_s)
+ msg = resp["choices"][0]["message"]
+ blob = (msg.get("content") or "") + " " + (msg.get("reasoning_content") or "")
+ return TOUCH_EXPECT in blob.lower()
+ except Exception as exc: # noqa: BLE001
+ logger.warning("touch generation failed for %s: %s", model_id, exc)
+ return False
+
+ def ensure_model_ready(self, model_id: str, timeout_s: int = 600) -> bool:
+ """Load if needed, then prove readiness with a touch generation."""
+ status = self.models().get(model_id)
+ if status is None:
+ raise KeyError(f"model {model_id} not present in models dir")
+ if status not in ("loaded", "ready"):
+ self.load_model(model_id, timeout_s=timeout_s)
+ return self.touch_generate(model_id)
+
+ def actual_n_ctx(self, model_id: str) -> int | None:
+ """/props reconciliation: the granted window as the child reports
+ it — the compressor's budget and the picker's 'running at 87K of
+ 262K' both read THIS value, never the request (design step 4)."""
+ try:
+ props = self._request(f"/props?model={model_id}")
+ return props.get("default_generation_settings", {}).get("n_ctx")
+ except Exception: # noqa: BLE001
+ return None
+
+ def keep_primary_loaded(self) -> None:
+ """The router's LRU eviction has no pin, so after any other load
+ re-touch the primary to keep it most-recently-used. Best-effort
+ under bursty multi-model load — replaced when an upstream pin
+ exists."""
+ if self.primary_model and self.models().get(self.primary_model) in (
+ "loaded", "ready"):
+ self.touch_generate(self.primary_model, timeout_s=60)
+
+ # ── telemetry ────────────────────────────────────────────
+
+ def is_idle(self, model_id: str | None = None) -> bool:
+ """No processing requests and no busy slots. Router quirk: /slots
+ and /metrics are per-child and require ?model= (bare calls 400),
+ and no KV-usage metric exists. With ``model_id`` checks that one
+ child; without, every loaded child."""
+ try:
+ if model_id is not None:
+ loaded = [model_id]
+ else:
+ loaded = [m for m, status in self.models().items()
+ if status in ("loaded", "ready")]
+ for mid in loaded:
+ slots = self._request(f"/slots?model={mid}")
+ if any(s.get("is_processing") for s in slots):
+ return False
+ req = urllib.request.Request(
+ self._url(f"/metrics?model={mid}"),
+ headers={"Authorization": f"Bearer {self.api_key}"})
+ with urllib.request.urlopen(req, timeout=10) as r:
+ text = r.read().decode()
+ for line in text.splitlines():
+ if line.startswith("llamacpp:requests_processing"):
+ if float(line.split()[-1]) != 0.0:
+ return False
+ return True
+ except Exception: # noqa: BLE001
+ return False
diff --git a/hermes_cli/loops.py b/hermes_cli/loops.py
index 92cdefdfe3..04e8dc7c02 100644
--- a/hermes_cli/loops.py
+++ b/hermes_cli/loops.py
@@ -768,6 +768,18 @@ class LoopManager:
"reason": s.last_stop_reason,
"message": f"✓ Loop finished after {s.ticks_fired} tick{'s' if s.ticks_fired != 1 else ''} — {reason}",
}
+ if verdict == "blocked":
+ # Judge ruled the stop condition unachievable — don't spin
+ # until the tick budget; pause so the user can re-scope.
+ s.status = "paused"
+ s.paused_reason = f"stop condition judged unachievable: {reason}"
+ save_loop(self.session_id, s)
+ return {
+ "status": "paused",
+ "stopped": True,
+ "reason": s.paused_reason,
+ "message": f"⏸ Loop paused — {s.paused_reason}. /loop resume to keep going, /loop stop to end it.",
+ }
# 3. --times user cap.
if s.times and s.ticks_fired >= s.times:
diff --git a/hermes_cli/main.py b/hermes_cli/main.py
index 6ef648af25..52a2f92d4e 100644
--- a/hermes_cli/main.py
+++ b/hermes_cli/main.py
@@ -462,6 +462,7 @@ import shutil
import stat
import subprocess
import tempfile
+import time as _time_mod
from pathlib import Path
from typing import Optional
@@ -499,7 +500,7 @@ from hermes_cli.subcommands.skin import build_skin_parser
from hermes_cli.subcommands.console import build_console_parser
from hermes_cli.subcommands.update import build_update_parser
from hermes_cli.subcommands.uninstall import build_uninstall_parser
-from hermes_cli.subcommands.dashboard import build_dashboard_parser
+from hermes_cli.subcommands.dashboard import build_dashboard_parser, build_serve_parser
from hermes_cli.subcommands.gui import build_gui_parser
from hermes_cli.subcommands.logs import build_logs_parser
from hermes_cli.subcommands.prompt_size import build_prompt_size_parser
@@ -1053,8 +1054,14 @@ def _relative_time(ts) -> str:
return relative_time(ts)
-def _has_any_provider_configured() -> bool:
- """Check if at least one inference provider is usable."""
+def _has_any_provider_configured(*, strict_profile_scope: bool = False) -> bool:
+ """Check if at least one inference provider is usable.
+
+ ``strict_profile_scope``: the caller has bound a NAMED profile's home and
+ secret scope and wants an answer for that profile only — launch-process
+ env and host-wide fallbacks (gh auth, Claude Code credentials) must not
+ make it appear ready. Unscoped callers keep the legacy behavior.
+ """
from hermes_cli.config import get_env_path, get_hermes_home, load_config
from hermes_cli.auth import get_auth_status
@@ -1097,7 +1104,13 @@ def _has_any_provider_configured() -> bool:
for pconfig in PROVIDER_REGISTRY.values():
if pconfig.auth_type == "api_key":
provider_env_vars.update(pconfig.api_key_env_vars)
- if any(os.getenv(v) for v in provider_env_vars):
+ if strict_profile_scope:
+ from agent.secret_scope import current_secret_scope
+
+ read_provider_env = (current_secret_scope() or {}).get
+ else:
+ read_provider_env = os.getenv
+ if any(read_provider_env(v) for v in provider_env_vars):
return True
# Check .env file for keys
@@ -1129,7 +1142,10 @@ def _has_any_provider_configured() -> bool:
auth = json.loads(auth_file.read_text(encoding="utf-8-sig"))
active = auth.get("active_provider")
- if active:
+ active_config = PROVIDER_REGISTRY.get(str(active or "").strip().lower())
+ if active and not (
+ strict_profile_scope and active_config and active_config.auth_type == "api_key"
+ ):
status = get_auth_status(active)
if status.get("logged_in"):
return True
@@ -1148,20 +1164,21 @@ def _has_any_provider_configured() -> bool:
return True
# Check provider-specific auth fallbacks (for example, Copilot via gh auth).
- try:
- for provider_id, pconfig in PROVIDER_REGISTRY.items():
- if pconfig.auth_type != "api_key":
- continue
- status = get_auth_status(provider_id)
- if status.get("logged_in"):
- return True
- except Exception:
- pass
+ if not strict_profile_scope:
+ try:
+ for provider_id, pconfig in PROVIDER_REGISTRY.items():
+ if pconfig.auth_type != "api_key":
+ continue
+ status = get_auth_status(provider_id)
+ if status.get("logged_in"):
+ return True
+ except Exception:
+ pass
# Check for Claude Code OAuth credentials (~/.claude/.credentials.json)
# Only count these if Hermes has been explicitly configured — Claude Code
# being installed doesn't mean the user wants Hermes to use their tokens.
- if _has_hermes_config:
+ if _has_hermes_config and not strict_profile_scope:
try:
from agent.anthropic_adapter import (
read_claude_code_credentials,
@@ -5219,6 +5236,7 @@ _LAZY_COMMAND_EXPORTS = {
"_sync_with_upstream_if_needed",
"_update_node_dependencies",
"_update_via_zip",
+ "_warn_orphaned_update_autostashes",
"_upgrade_pip_before_lazy_refresh",
"_validate_critical_files_syntax",
"_validate_critical_modules_import",
@@ -6993,7 +7011,15 @@ def _write_desktop_build_stamp(project_root: Path, *, source_mode: bool) -> None
def _desktop_packaged_executable(desktop_dir: Path) -> Optional[Path]:
"""Return the current platform's unpacked Electron app executable."""
- release_dir = desktop_dir / "release"
+ return _desktop_packaged_executable_in(desktop_dir / "release")
+
+
+def _desktop_packaged_executable_in(release_dir: Path) -> Optional[Path]:
+ """Return the unpacked Electron app executable under *release_dir*.
+
+ *release_dir* is electron-builder's ``directories.output`` — the live
+ ``apps/desktop/release`` or a stage-and-swap staging dir (#86443).
+ """
if sys.platform == "darwin":
candidates = list(release_dir.glob("mac*/Hermes.app/Contents/MacOS/Hermes"))
elif sys.platform == "win32":
@@ -7027,6 +7053,91 @@ def _desktop_packaged_executable(desktop_dir: Path) -> Optional[Path]:
return max(existing, key=lambda p: p.stat().st_mtime)
+# ─── Desktop stage-and-swap pack (#86443) ───────────────────────────────────
+#
+# electron-builder packs IN PLACE: before-pack.mjs wipes ``release/-
+# unpacked`` (or the mac ``Hermes.app``) and the Electron unpack + asar + rename
+# then rebuild it. Any failure after that wipe — corrupt cached zip, blocked
+# download, missing dep, disk full — leaves the user with NO app, and
+# ``hermes update`` used to report "partially complete" over an empty
+# release/. Fix the class, not the predicate: build into a STAGING output
+# dir next to release/, verify the staged result, and only then swap it over
+# the live tree with renames. On any failure the live app is untouched.
+
+_DESKTOP_STAGING_PREFIX = ".staging-"
+_DESKTOP_PREVIOUS_SUFFIX = ".previous"
+
+
+def _desktop_staging_dir(desktop_dir: Path) -> Path:
+ """Fresh, unique staging output dir: ``apps/desktop/.staging--``.
+
+ A sibling of ``release/`` (same filesystem → the swap is a rename, not a
+ copy) but NOT inside it, so nothing globbing ``release/*-unpacked`` or
+ ``release/mac*`` can mistake the half-built tree for the live app.
+ Leftovers from a killed earlier build are swept first (best-effort).
+ """
+ for stale in desktop_dir.glob(f"{_DESKTOP_STAGING_PREFIX}*"):
+ shutil.rmtree(stale, ignore_errors=True)
+ return desktop_dir / f"{_DESKTOP_STAGING_PREFIX}{os.getpid()}-{int(_time_mod.time())}"
+
+
+def _desktop_unpacked_root(exe: Path, release_dir: Path) -> Path:
+ """The directory directly under *release_dir* that holds *exe*
+ (``linux-unpacked``, ``win-unpacked``, ``mac-arm64``…) — electron-builder's
+ ``appOutDir``, the unit that gets swapped as a whole."""
+ unpacked = exe
+ while unpacked.parent != release_dir:
+ if unpacked.parent == unpacked:
+ raise ValueError(f"{exe} is not under {release_dir}")
+ unpacked = unpacked.parent
+ return unpacked
+
+
+def _swap_staged_desktop_app(desktop_dir: Path, staging_dir: Path) -> Optional[Path]:
+ """Promote a VERIFIED staged pack over the live ``release/`` app.
+
+ ``release/`` → ``release/.previous``,
+ ``/`` → ``release/``, then drop ``.previous``.
+ Two renames; the only window with no live app is between them, and a
+ failure there rolls ``.previous`` back. Returns the live executable, or
+ ``None`` (live app untouched or restored) when the swap could not happen.
+ Best-effort cleanup of the staging dir; never raises.
+ """
+ staged_exe = _desktop_packaged_executable_in(staging_dir)
+ if staged_exe is None:
+ shutil.rmtree(staging_dir, ignore_errors=True)
+ return None
+ release_dir = desktop_dir / "release"
+ try:
+ staged_root = _desktop_unpacked_root(staged_exe, staging_dir)
+ live_root = release_dir / staged_root.name
+ previous = release_dir / (staged_root.name + _DESKTOP_PREVIOUS_SUFFIX)
+ release_dir.mkdir(parents=True, exist_ok=True)
+ shutil.rmtree(previous, ignore_errors=True)
+ moved_aside = False
+ if live_root.exists():
+ os.rename(live_root, previous)
+ moved_aside = True
+ try:
+ os.rename(staged_root, live_root)
+ except OSError:
+ if moved_aside:
+ os.rename(previous, live_root) # restore; live app back as it was
+ raise
+ if moved_aside:
+ shutil.rmtree(previous, ignore_errors=True)
+ except (OSError, ValueError) as exc:
+ logger.warning("desktop stage-and-swap failed, live app kept: %s", exc)
+ return None
+ finally:
+ shutil.rmtree(staging_dir, ignore_errors=True)
+ return live_root / staged_exe.relative_to(staged_root)
+
+
+def _discard_desktop_staging(staging_dir: Path) -> None:
+ shutil.rmtree(staging_dir, ignore_errors=True)
+
+
# ─── Desktop exe integrity gate (#69179) ────────────────────────────────────
#
# The desktop self-update chain (Desktop → hermes-setup --update →
@@ -7343,8 +7454,10 @@ def _ensure_desktop_exe_launchable(
# Self-heal setup for the retry: drop the (likely corrupt) cached Electron
# zip and the content stamp so the next rebuild is a genuine re-download +
- # re-stage rather than a replay of the same broken extraction.
- _purge_electron_build_cache(desktop_dir)
+ # re-stage rather than a replay of the same broken extraction. Only the
+ # exe's OWN output dir is purged (a stage-and-swap staging dir, #86443),
+ # never the live release/ tree that still holds the last working app.
+ _purge_electron_build_cache(desktop_dir, release_dir=packaged_executable.parent.parent)
try:
_desktop_stamp_path().unlink()
except OSError:
@@ -7400,7 +7513,9 @@ def _electron_download_cache_dirs() -> list[Path]:
return out
-def _purge_electron_build_cache(desktop_dir: Path) -> list[Path]:
+def _purge_electron_build_cache(
+ desktop_dir: Path, release_dir: Optional[Path] = None
+) -> list[Path]:
"""Clear the cached Electron download + half-written unpacked dir so the
next ``pack`` re-downloads and re-stages from scratch.
@@ -7446,8 +7561,11 @@ def _purge_electron_build_cache(desktop_dir: Path) -> list[Path]:
# Drop the half-written unpacked dir too: an interrupted prior pack leaves
# a partial tree that poisons the rename even after the zip is fixed.
# (before-pack.cjs also handles this, but clearing it here makes the retry
- # robust even if the hook is somehow skipped.)
- release_dir = desktop_dir / "release"
+ # robust even if the hook is somehow skipped.) ``release_dir`` lets a
+ # stage-and-swap caller point this at its STAGING output so a mid-retry
+ # purge never touches the live app under ``release/`` (#86443).
+ if release_dir is None:
+ release_dir = desktop_dir / "release"
if release_dir.is_dir():
for unpacked in release_dir.glob("*-unpacked"):
try:
@@ -7826,6 +7944,7 @@ def _desktop_macos_relaunchable_fixup(
desktop_dir: Path,
*,
publisher_signing_configured: Optional[bool] = None,
+ release_dir: Optional[Path] = None,
) -> bool:
"""Make a locally-built macOS desktop app survive in-place self-update
without resetting the user's TCC permission grants.
@@ -7857,7 +7976,9 @@ def _desktop_macos_relaunchable_fixup(
)
if publisher_signing_configured:
return True
- exe = _desktop_packaged_executable(desktop_dir)
+ # ``release_dir`` (stage-and-swap, #86443): sign the STAGED bundle before
+ # it is promoted, so the live app is never touched mid-sign.
+ exe = _desktop_packaged_executable_in(release_dir or (desktop_dir / "release"))
if exe is None:
return True
# exe = .../Hermes.app/Contents/MacOS/Hermes -> app bundle = .../Hermes.app
@@ -8539,7 +8660,16 @@ def cmd_gui(args: argparse.Namespace):
print(" → No Developer ID configured; ad-hoc signing this local rebuild "
"(CSC_IDENTITY_AUTO_DISCOVERY=false)")
npm_build_env = _npm_lifecycle_env(env)
+ # Stage-and-swap (#86443): electron-builder packs IN PLACE and
+ # before-pack.mjs wipes release/ first, so a pack that
+ # fails afterwards used to leave the user with NO app. Build into
+ # a fresh staging output dir instead; the live release/ tree is
+ # only replaced — by rename — after the staged result verifies.
+ staging_dir: Optional[Path] = None
+ build_cmd = [npm, "run", build_script]
if not source_mode:
+ staging_dir = _desktop_staging_dir(desktop_dir)
+ build_cmd += ["--", f"-c.directories.output={staging_dir}"]
# A running desktop instance launched from release/win-unpacked
# holds Hermes.exe locked on Windows, so the pack can't replace
# it ("Access is denied" / ERR_ELECTRON_BUILDER_CANNOT_EXECUTE).
@@ -8548,13 +8678,17 @@ def cmd_gui(args: argparse.Namespace):
stopped = _stop_desktop_processes_locking_build(desktop_dir)
if stopped:
print(f" ⚠ Stopped running desktop app to free the build output (pid {', '.join(map(str, stopped))})")
+
+ def _staged_exe() -> Optional[Path]:
+ return _desktop_packaged_executable_in(staging_dir) if staging_dir else None
+
build_result = subprocess.run(
- [npm, "run", build_script], cwd=desktop_dir, env=npm_build_env, check=False
+ build_cmd, cwd=desktop_dir, env=npm_build_env, check=False
)
if (
build_result.returncode != 0
and not source_mode
- and _desktop_packaged_executable(desktop_dir) is None
+ and _staged_exe() is None
):
# Corrupt cached Electron zip → partial unpack → ENOENT on rename.
# stdlib zipfile won't catch the common concat-junk case, so purge
@@ -8568,7 +8702,7 @@ def cmd_gui(args: argparse.Namespace):
purged: list[Path] = []
restored = False
if not _electron_dist_ok(PROJECT_ROOT):
- purged = _purge_electron_build_cache(desktop_dir)
+ purged = _purge_electron_build_cache(desktop_dir, release_dir=staging_dir)
restored = _redownload_electron_dist(PROJECT_ROOT, env)
if restored:
print(" ⚠ Desktop build failed; refreshed the Electron download and retrying once...")
@@ -8578,13 +8712,13 @@ def cmd_gui(args: argparse.Namespace):
# is still locked by a running instance; stop it before retry.
_stop_desktop_processes_locking_build(desktop_dir)
build_result = subprocess.run(
- [npm, "run", build_script], cwd=desktop_dir, env=npm_build_env, check=False
+ build_cmd, cwd=desktop_dir, env=npm_build_env, check=False
)
if (
build_result.returncode != 0
and not source_mode
and not env.get("ELECTRON_MIRROR")
- and _desktop_packaged_executable(desktop_dir) is None
+ and _staged_exe() is None
):
print(" ⚠ Desktop build still failing; the Electron download from "
"GitHub looks blocked. Re-downloading via a public mirror "
@@ -8595,9 +8729,13 @@ def cmd_gui(args: argparse.Namespace):
if not _electron_dist_ok(PROJECT_ROOT):
_redownload_electron_dist(PROJECT_ROOT, env, mirror=mirror)
_stop_desktop_processes_locking_build(desktop_dir)
- build_result = subprocess.run([npm, "run", build_script], cwd=desktop_dir, env=mirror_env, check=False)
+ build_result = subprocess.run(build_cmd, cwd=desktop_dir, env=mirror_env, check=False)
if build_result.returncode != 0:
print("✗ Desktop GUI build failed")
+ if staging_dir is not None:
+ _discard_desktop_staging(staging_dir)
+ if _desktop_packaged_executable(desktop_dir) is not None:
+ print(" ↩ The previous desktop app was left untouched and still works.")
print(f" Run manually: cd apps/desktop && npm run {build_script}")
if sys.platform == "win32":
print(" If this says \"Access is denied\" on Hermes.exe, close any")
@@ -8605,28 +8743,37 @@ def cmd_gui(args: argparse.Namespace):
print(" If the log shows Electron download retries, rebuild via a mirror:")
print(" ELECTRON_MIRROR= hermes desktop --force-build")
sys.exit(build_result.returncode or 1)
- packaged_executable = _desktop_packaged_executable(desktop_dir)
if not source_mode:
+ assert staging_dir is not None
+ staged_executable = _staged_exe()
# Locally-built apps are ad-hoc signed; make them relaunchable after
# an in-place self-update (otherwise macOS reports "Hermes is
# damaged"). No-op on non-macOS and on real-identity builds.
- _desktop_macos_relaunchable_fixup(desktop_dir)
+ # Signs the STAGED bundle so the live app is never half-signed.
+ _desktop_macos_relaunchable_fixup(desktop_dir, release_dir=staging_dir)
# Windows integrity gate (#69179): never declare the rebuild a
# success on a Hermes.exe Windows cannot load (truncated PE from
# a corrupt cached Electron zip, wrong-arch tree, interrupted
- # rcedit rewrite). Roll back to the .bak tree preserved by
- # before-pack.mjs when possible, then fail loudly so the
- # updater's retry-once rebuilds from a fresh Electron download
- # instead of silently shipping the broken exe.
+ # rcedit rewrite). Verified on the STAGED exe: a failure here
+ # simply discards the staging dir — the live app was never
+ # touched — and fails loudly so the updater's retry-once
+ # rebuilds from a fresh Electron download.
verified_executable, rolled_back = _ensure_desktop_exe_launchable(
- desktop_dir, packaged_executable
+ desktop_dir, staged_executable
)
- if packaged_executable is not None and (
- rolled_back or verified_executable is None
- ):
+ if staged_executable is None or rolled_back or verified_executable is None:
+ _discard_desktop_staging(staging_dir)
+ if staged_executable is None:
+ print(f"✗ Desktop build produced no launchable app in {staging_dir}")
+ print(" ↩ The previous desktop app was left untouched and still works.")
+ sys.exit(1)
+ # Verified: swap the staged tree over the live one (rename).
+ packaged_executable = _swap_staged_desktop_app(desktop_dir, staging_dir)
+ if packaged_executable is None:
+ print(f"✗ Could not install the rebuilt desktop app into {desktop_dir / 'release'}")
+ print(" ↩ The previous desktop app was left untouched and still works.")
sys.exit(1)
- packaged_executable = verified_executable
# Build succeeded — write the stamp so next run can skip
_write_desktop_build_stamp(PROJECT_ROOT, source_mode=source_mode)
@@ -8658,7 +8805,10 @@ def cmd_gui(args: argparse.Namespace):
if source_mode:
print("→ Launching Hermes Desktop from source build...")
- launch_result = subprocess.run([npm, "exec", "--", "electron", "."], cwd=desktop_dir, env=env, check=False)
+ electron_argv = [npm, "exec", "--", "electron", "."]
+ if getattr(args, "local", False):
+ electron_argv.append("--local")
+ launch_result = subprocess.run(electron_argv, cwd=desktop_dir, env=env, check=False)
sys.exit(launch_result.returncode)
if packaged_executable is None:
@@ -8677,6 +8827,8 @@ def cmd_gui(args: argparse.Namespace):
launch_command.append("--disable-setuid-sandbox")
launch_command.extend(config_electron_flags)
+ if getattr(args, "local", False):
+ launch_command.append("--local")
print(f"→ Launching packaged Hermes Desktop: {' '.join(launch_command)}")
launch_result = subprocess.run(launch_command, cwd=desktop_dir, env=env, check=False)
sys.exit(launch_result.returncode)
@@ -11452,6 +11604,7 @@ def cmd_profile(args):
profile_exists,
_read_config_model,
_check_gateway_running,
+ _served_by_running_multiplexer,
_count_skills,
_read_distribution_meta,
_get_wrapper_dir,
@@ -11465,7 +11618,7 @@ def cmd_profile(args):
sys.exit(1)
profile_dir = get_profile_dir(name)
model, provider = _read_config_model(profile_dir)
- gw = _check_gateway_running(profile_dir)
+ gw = _check_gateway_running(profile_dir) or _served_by_running_multiplexer(name)
skills = _count_skills(profile_dir)
dist_name, dist_version, dist_source = _read_distribution_meta(profile_dir)
alias_name = find_alias_for_profile(name)
@@ -12391,18 +12544,29 @@ def cmd_dashboard(args):
# this, a profile's configured MCP servers never connect, so desktop
# sessions show no MCP tools. Spawn discovery in the background here so a
# slow/dead server can't block dashboard startup.
- try:
- from hermes_cli.mcp_startup import start_background_mcp_discovery
+ #
+ # Desktop-spawned headless backends start it AFTER the socket binds
+ # instead (start_server's ready path): the thread's first act is the
+ # ~350ms `mcp` SDK import, which holds the GIL against the main thread's
+ # own web_server import and pushes the READY sentinel — and every
+ # renderer paint behind it — back by that much. The Desktop can't issue
+ # an agent turn until its WebSocket is up anyway, and _make_agent's
+ # bounded wait_for_mcp_discovery + the late-binding refresh cover a
+ # server that is still connecting when the first turn lands.
+ _mcp_discovery_after_bind = _headless_backend and os.environ.get("HERMES_DESKTOP") == "1"
+ if not _mcp_discovery_after_bind:
+ try:
+ from hermes_cli.mcp_startup import start_background_mcp_discovery
- start_background_mcp_discovery(
- logger=logger,
- thread_name="dashboard-mcp-discovery",
- )
- except Exception:
- logger.debug(
- "Background MCP tool discovery failed at dashboard startup",
- exc_info=True,
- )
+ start_background_mcp_discovery(
+ logger=logger,
+ thread_name="dashboard-mcp-discovery",
+ )
+ except Exception:
+ logger.debug(
+ "Background MCP tool discovery failed at dashboard startup",
+ exc_info=True,
+ )
from hermes_cli.web_server import start_server
@@ -12425,6 +12589,7 @@ def cmd_dashboard(args):
headless=_headless_backend,
ssh_session_token=_ssh_session_token,
ssh_owner_nonce=_ssh_owner_nonce,
+ start_mcp_discovery_after_bind=_mcp_discovery_after_bind,
)
@@ -12814,6 +12979,47 @@ def _set_chat_arg_defaults(args) -> None:
setattr(args, attr, default)
+def _try_fast_serve_launch() -> bool:
+ """Dispatch an unambiguous built-in ``serve`` without the full CLI tree.
+
+ Desktop launches this exact command on every cold start. Building parsers
+ for unrelated Hermes commands performs thousands of filesystem-backed
+ translation lookups on Windows even though none of those commands are
+ usable in this process. Unknown or globally-scoped arguments fall back to
+ normal parsing so compatibility and error reporting remain unchanged.
+ """
+ if os.environ.get("HERMES_DISABLE_FAST_SERVE_LAUNCH") == "1":
+ return False
+
+ argv = sys.argv[1:]
+ if not argv or argv[0] != "serve" or "-h" in argv or "--help" in argv:
+ return False
+
+ # Container routing is top-level policy and must run before host dispatch.
+ try:
+ from hermes_cli.config import get_container_exec_info
+
+ if get_container_exec_info():
+ return False
+ except Exception:
+ return False
+
+ parser = build_serve_parser(
+ cmd_dashboard=cmd_dashboard,
+ add_help=False,
+ exit_on_error=False,
+ )
+ try:
+ args, unknown = parser.parse_known_args(argv[1:])
+ except (argparse.ArgumentError, ValueError):
+ return False
+ if unknown:
+ return False
+
+ cmd_dashboard(args)
+ return True
+
+
def _try_fast_chat_launch() -> bool:
"""Fast path for unambiguous interactive chat launches (all hosts).
@@ -13361,6 +13567,8 @@ def main():
return
if _try_termux_fast_cli_launch():
return
+ if _try_fast_serve_launch():
+ return
if _try_fast_chat_launch():
return
diff --git a/hermes_cli/mcp_startup.py b/hermes_cli/mcp_startup.py
index c368805405..77a972591d 100644
--- a/hermes_cli/mcp_startup.py
+++ b/hermes_cli/mcp_startup.py
@@ -9,6 +9,7 @@ from typing import Optional
_mcp_discovery_lock = threading.Lock()
_mcp_discovery_started = False
_mcp_discovery_thread: Optional[threading.Thread] = None
+_mcp_discovery_deferred: Optional[threading.Timer] = None
def _has_configured_mcp_servers() -> bool:
@@ -172,6 +173,45 @@ def _discover_mcp_tools_without_interactive_oauth() -> None:
discover_mcp_tools()
+def defer_background_mcp_discovery(*, logger, thread_name: str, delay: float) -> None:
+ """Arm ``start_background_mcp_discovery`` to run ``delay`` seconds from now.
+
+ Used by the Desktop ``serve`` backend after its socket is announced: the
+ discovery thread's first act is the ~350ms ``mcp`` SDK import, which holds
+ the GIL against the renderer's connect + first hydration reads if it starts
+ at bind time, and against the web_server import if it starts before. Any
+ consumer that needs discovery sooner (``wait_for_mcp_discovery`` from an
+ agent build) fires the deferred start immediately, so the bounded join and
+ the late-binding refresh behave exactly as if it had been started eagerly.
+ """
+ global _mcp_discovery_deferred
+ with _mcp_discovery_lock:
+ if _mcp_discovery_started or _mcp_discovery_deferred is not None:
+ return
+
+ def _fire() -> None:
+ global _mcp_discovery_deferred
+ with _mcp_discovery_lock:
+ _mcp_discovery_deferred = None
+ start_background_mcp_discovery(logger=logger, thread_name=thread_name)
+
+ timer = threading.Timer(delay, _fire)
+ timer.daemon = True
+ timer.name = f"{thread_name}-deferred"
+ _mcp_discovery_deferred = timer
+ timer.start()
+
+
+def _start_deferred_mcp_discovery_now() -> None:
+ """Run an armed deferred start immediately (idempotent, thread-safe)."""
+ with _mcp_discovery_lock:
+ timer = _mcp_discovery_deferred
+ if timer is None:
+ return
+ timer.cancel()
+ timer.function()
+
+
def wait_for_mcp_discovery(
timeout: "float | None" = None, *, single_query: bool = False
) -> None:
@@ -188,6 +228,7 @@ def wait_for_mcp_discovery(
``mcp_single_query_discovery_timeout`` instead (default 15s vs 1.5s
interactive) because one-shot sessions have no second turn to recover.
"""
+ _start_deferred_mcp_discovery_now()
thread = _mcp_discovery_thread
if thread is None or not thread.is_alive():
return
diff --git a/hermes_cli/model_catalog.py b/hermes_cli/model_catalog.py
index 5aa479fa87..1e869b746a 100644
--- a/hermes_cli/model_catalog.py
+++ b/hermes_cli/model_catalog.py
@@ -74,7 +74,10 @@ DEFAULT_CATALOG_URL = (
DEFAULT_CATALOG_FALLBACK_URLS: tuple[str, ...] = (
"https://raw.githubusercontent.com/NousResearch/hermes-agent/main/website/static/api/model-catalog.json",
)
-DEFAULT_TTL_HOURS = 1
+DEFAULT_TTL_MINUTES = 20
+# Legacy key. ``ttl_hours`` is honoured only when the user set it explicitly;
+# the shipped default is ``ttl_minutes`` above.
+DEFAULT_TTL_HOURS = DEFAULT_TTL_MINUTES / 60.0
DEFAULT_FETCH_TIMEOUT = 8.0
SUPPORTED_SCHEMA_VERSION = 1
@@ -104,10 +107,28 @@ def _load_catalog_config() -> dict[str, Any]:
if not isinstance(raw, dict):
raw = {}
+ # ``ttl_minutes`` is the shipped default (20). ``ttl_hours`` is the legacy
+ # key: honoured when a user set it explicitly and ``ttl_minutes`` is still
+ # at its default (load_config() deep-merges the default in, so "present"
+ # alone doesn't mean "user-set"), so old customized configs keep their
+ # chosen window.
+ ttl_minutes = raw.get("ttl_minutes")
+ try:
+ ttl_minutes = float(ttl_minutes) if ttl_minutes not in (None, "") else DEFAULT_TTL_MINUTES
+ except (TypeError, ValueError):
+ ttl_minutes = DEFAULT_TTL_MINUTES
+ if ttl_minutes == DEFAULT_TTL_MINUTES and raw.get("ttl_hours"):
+ try:
+ ttl_minutes = float(raw["ttl_hours"]) * 60.0
+ except (TypeError, ValueError):
+ pass
+ if ttl_minutes <= 0:
+ ttl_minutes = DEFAULT_TTL_MINUTES
+
return {
"enabled": bool(raw.get("enabled", True)),
"url": str(raw.get("url") or DEFAULT_CATALOG_URL),
- "ttl_hours": float(raw.get("ttl_hours") or DEFAULT_TTL_HOURS),
+ "ttl_hours": ttl_minutes / 60.0,
"providers": raw.get("providers") if isinstance(raw.get("providers"), dict) else {},
}
@@ -330,6 +351,33 @@ def get_catalog(*, force_refresh: bool = False) -> dict[str, Any]:
return {}
+def refresh_interval_seconds() -> float:
+ """Return the configured catalog TTL in seconds (the gateway poll cadence)."""
+ return max(60.0, _load_catalog_config()["ttl_hours"] * 3600.0)
+
+
+def refresh_catalogs() -> bool:
+ """Force-refresh every remote model catalog the picker reads from.
+
+ Fetches the curated manifest, the OpenRouter live list (tool-support /
+ free-pricing filter) and the Nous Portal recommendations, writing each
+ to its disk cache so the next ``/model`` open in ANY process on this
+ machine sees the new lists. Blocking; run it off the event loop.
+ Returns True when the manifest refresh succeeded.
+ """
+ if not _load_catalog_config()["enabled"]:
+ return False
+ catalog = get_catalog(force_refresh=True)
+ try:
+ from hermes_cli.models import fetch_nous_recommended_models, fetch_openrouter_models
+
+ fetch_openrouter_models(force_refresh=True)
+ fetch_nous_recommended_models(force_refresh=True)
+ except Exception:
+ logger.debug("provider catalog refresh failed", exc_info=True)
+ return bool(catalog)
+
+
def _fetch_provider_override(provider: str) -> dict[str, Any] | None:
"""If ``model_catalog.providers..url`` is set, fetch that instead."""
cfg = _load_catalog_config()
diff --git a/hermes_cli/model_setup_flows.py b/hermes_cli/model_setup_flows.py
index 44aab0ca6c..adda8b6a6a 100644
--- a/hermes_cli/model_setup_flows.py
+++ b/hermes_cli/model_setup_flows.py
@@ -539,6 +539,15 @@ def _model_flow_nous(config, current_model="", args=None):
# of CLI release cadence.
unavailable_models: list[str] = []
unavailable_message = ""
+
+ # Neither the curated list nor the Portal's recommendations know what the
+ # org may reach. Narrow before the tier split, so an id the policy rescues
+ # still has to pass the free/paid predicate instead of going around it.
+ from hermes_cli.models import nous_policy_allowed_ids, restrict_to_nous_policy
+
+ _policy_allowed = nous_policy_allowed_ids()
+ _policy_narrowed = False
+
if free_tier:
try:
from hermes_cli.nous_account import (
@@ -559,6 +568,11 @@ def _model_flow_nous(config, current_model="", args=None):
model_ids, pricing = union_with_portal_free_recommendations(
model_ids, pricing, _nous_portal_url,
)
+ _before_policy = model_ids
+ model_ids = restrict_to_nous_policy(
+ model_ids, _policy_allowed, rescue_empty=True,
+ )
+ _policy_narrowed = model_ids != _before_policy
model_ids, unavailable_models = partition_nous_models_by_tier(
model_ids, pricing, free_tier=True
)
@@ -566,6 +580,11 @@ def _model_flow_nous(config, current_model="", args=None):
model_ids, pricing = union_with_portal_paid_recommendations(
model_ids, pricing, _nous_portal_url,
)
+ _before_policy = model_ids
+ model_ids = restrict_to_nous_policy(
+ model_ids, _policy_allowed, rescue_empty=True,
+ )
+ _policy_narrowed = model_ids != _before_policy
if not model_ids and not unavailable_models:
print("No models available for Nous Portal after filtering.")
@@ -580,6 +599,11 @@ def _model_flow_nous(config, current_model="", args=None):
print(unavailable_message or f"Upgrade at {_url} to access paid models.")
return
+ from hermes_cli.nous_account import nous_policy_notice
+
+ _policy_notice = nous_policy_notice(removed=_policy_narrowed)
+ if _policy_notice:
+ print(_policy_notice)
print(
f'Showing {len(model_ids)} curated models — use "Enter custom model name" for others.'
)
diff --git a/hermes_cli/model_switch.py b/hermes_cli/model_switch.py
index 6a13ce7b36..f70c71f30f 100644
--- a/hermes_cli/model_switch.py
+++ b/hermes_cli/model_switch.py
@@ -562,10 +562,12 @@ def _load_direct_aliases() -> dict[str, DirectAlias]:
neither is set the key is resolved from the alias HOST, never from the
previously active provider (#83612).
- Also reads ``model.aliases`` (set by ``hermes config set model.aliases.xxx``)
- and converts simple string entries (``ds-flash: deepseek/deepseek-v4-flash``)
- into DirectAlias objects. The provider is parsed from the ``provider/``
- prefix in the value; if no slash, the current provider is used.
+ Also reads ``model.aliases`` (set by ``hermes config set model.aliases.xxx``
+ or hand-written). String entries (``ds-flash: deepseek/deepseek-v4-flash``)
+ are converted into DirectAlias objects with the provider parsed from the
+ ``provider/`` prefix in the value; if no slash, the current provider is
+ used. Dict entries use the same shape as ``model_aliases:`` (``model``,
+ ``provider``, ``base_url`` keys).
"""
merged = dict(_BUILTIN_DIRECT_ALIASES)
try:
@@ -588,18 +590,34 @@ def _load_direct_aliases() -> dict[str, DirectAlias]:
key_env=str(entry.get("key_env", "") or "").strip(),
)
- # --- model.aliases (string-based format, from config set) ---
+ # --- model.aliases (from config set / hand-written config) ---
model_section = cfg.get("model", {})
if isinstance(model_section, dict):
simple_aliases = model_section.get("aliases")
if isinstance(simple_aliases, dict):
current_provider = model_section.get("provider", "")
for name, value in simple_aliases.items():
+ key = name.strip().lower()
+ if not key or key in merged:
+ continue # don't override explicit model_aliases entries
+ if isinstance(value, dict):
+ # Dict form mirrors the ``model_aliases:`` shape:
+ # localqwen: {model: qwen3.5:4b, provider: custom}.
+ # Hand-written configs already use it; honoring it
+ # here keeps aliases with an explicit provider from
+ # being silently dropped (#87189).
+ model = str(value.get("model") or "").strip()
+ if not model:
+ continue
+ provider = str(value.get("provider") or "").strip()
+ merged[key] = DirectAlias(
+ model=model,
+ provider=provider or current_provider or "custom",
+ base_url=str(value.get("base_url") or "").strip(),
+ )
+ continue
if not isinstance(value, str) or not value.strip():
continue
- key = name.strip().lower()
- if key in merged:
- continue # don't override explicit model_aliases entries
val = value.strip()
if "/" in val:
provider, model = val.split("/", 1)
@@ -744,6 +762,121 @@ def _may_reuse_session_credential(session_base_url: str, alias_base_url: str) ->
return scheme == "https" or hostname in _LOOPBACK_HOSTS
+class StartupModelRoute(NamedTuple):
+ """Model/provider pair resolved before an agent is constructed."""
+
+ model: str
+ provider: str = ""
+ base_url: str = ""
+ api_key: str = ""
+
+
+def resolve_startup_model_route(
+ raw_model: str,
+ *,
+ explicit_provider: str = "",
+ current_provider: str = "",
+ user_providers: Optional[dict] = None,
+ custom_providers: Optional[list] = None,
+) -> Optional[StartupModelRoute]:
+ """Resolve aliases and configured ``provider/model`` input at startup.
+
+ ``HermesCLI`` is constructed before the interactive ``/model`` pipeline
+ runs. Keeping this small resolver at the same boundary as
+ ``DIRECT_ALIASES`` prevents startup from attaching the configured default
+ provider to an explicitly requested model. Provider/model strings are
+ consumed only for providers present in user configuration; aggregator
+ namespaces remain untouched.
+
+ ``current_provider`` is the provider the session would otherwise use
+ (config ``model.provider`` / ``--provider``). When it is a routing
+ aggregator and the raw string is an aggregator-native slug
+ (``anthropic/claude-opus-4.6`` on OpenRouter), the input stays on the
+ aggregator — bare vendor slugs resolve WITHIN the aggregator first and a
+ ``providers:`` block for the same vendor must not steal the route.
+ """
+ raw = str(raw_model or "").strip()
+ if not raw:
+ return None
+
+ _ensure_direct_aliases()
+ direct = DIRECT_ALIASES.get(raw.lower())
+ if direct is not None:
+ if explicit_provider:
+ # An explicit --provider wins over the alias's own label; the
+ # alias contributes model/base_url only.
+ return StartupModelRoute(
+ model=direct.model,
+ provider=explicit_provider,
+ base_url=direct.base_url,
+ )
+ # Resolve through the SAME owner the interactive /model and oneshot
+ # paths use: a URL-bearing alias must resolve its credential for the
+ # alias HOST, never for its provider label — a label like
+ # ``anthropic`` on a foreign URL would otherwise reach that
+ # provider's explicit-runtime branch and put the live vendor token
+ # on the foreign wire (#28660).
+ alias_provider, alias_key = direct_alias_runtime_request(direct)
+ return StartupModelRoute(
+ model=direct.model,
+ provider=alias_provider,
+ base_url=direct.base_url,
+ api_key=alias_key or "",
+ )
+
+ if explicit_provider or "/" not in raw:
+ return None
+ prefix, model = (part.strip() for part in raw.split("/", 1))
+ if not prefix or not model:
+ return None
+
+ # Aggregator-native slugs stay on the aggregator. A user on OpenRouter
+ # whose config also has a ``providers.anthropic`` block must NOT have
+ # ``anthropic/claude-opus-4.6`` silently rerouted to native Anthropic.
+ if current_provider:
+ try:
+ from hermes_cli.providers import (
+ is_routing_aggregator as _is_routing_agg,
+ normalize_provider as _norm_prov,
+ )
+
+ if _is_routing_agg(_norm_prov(current_provider)):
+ from hermes_cli.models import _find_openrouter_slug
+
+ if _find_openrouter_slug(raw):
+ return None
+ except Exception:
+ pass
+
+ configured = {
+ str(name).strip().lower()
+ for name in (user_providers or {})
+ if str(name).strip()
+ }
+ configured.update(
+ f"custom:{entry.get('name', '').strip().lower()}"
+ for entry in (custom_providers or [])
+ if isinstance(entry, dict) and str(entry.get("name") or "").strip()
+ )
+ try:
+ from hermes_cli.models import normalize_provider
+
+ canonical = normalize_provider(prefix)
+ except Exception:
+ canonical = prefix.lower()
+
+ if prefix.lower() in configured:
+ provider = prefix
+ elif canonical.lower() in configured:
+ provider = canonical
+ else:
+ return None
+
+ if is_aggregator(canonical):
+ return None
+ return StartupModelRoute(model=model, provider=provider)
+
+
# ---------------------------------------------------------------------------
# Result dataclasses
# ---------------------------------------------------------------------------
@@ -883,11 +1016,18 @@ def resolve_persist_behavior(
1. ``--once`` explicitly opts out → ``False`` (next turn only).
2. ``--session`` explicitly opts out → ``False`` (this session only).
3. ``--global`` explicitly opts in → ``True``.
- 4. ``--provider`` given without an explicit persist flag → ``False``
+ 4. No default configured yet (neither ``model.default`` nor
+ ``model.provider`` set — a fresh install whose first-ever pick this
+ is) → ``True``. Without a persisted provider, ``resolve_provider``
+ falls through to whatever ``*_API_KEY`` env var is lying around on
+ the next launch (#86414), so the first pick becomes the default
+ instead of evaporating. Applies to every surface (CLI, gateway,
+ Desktop picker) so no client has to hardcode ``--global``.
+ 5. ``--provider`` given without an explicit persist flag → ``False``
(session only). Provider switches are typically exploratory — the
user is trying a different backend for this conversation, not
reconfiguring the default. ``--global`` can still force persist.
- 5. Otherwise defer to ``model.persist_switch_by_default`` in
+ 6. Otherwise defer to ``model.persist_switch_by_default`` in
``config.yaml`` (defaults to ``False``: a plain ``/model ``
affects only the current session). Users who want the old
persist-by-default behavior can set the key to ``true``; a one-off
@@ -903,17 +1043,20 @@ def resolve_persist_behavior(
return False
if is_global:
return True
- if explicit_provider:
- return False
try:
from hermes_cli.config import load_config
model_cfg = load_config().get("model")
- if isinstance(model_cfg, dict):
- return bool(model_cfg.get("persist_switch_by_default", False))
except Exception:
- pass
- return False
+ return False
+ if isinstance(model_cfg, dict):
+ if not (model_cfg.get("default") or model_cfg.get("provider")):
+ return True
+ if explicit_provider:
+ return False
+ return bool(model_cfg.get("persist_switch_by_default", False))
+ # Flat-string form: a non-empty string IS a configured default.
+ return not model_cfg
# ---------------------------------------------------------------------------
@@ -2801,7 +2944,10 @@ def _collect_authed_provider_slugs(
slugs.append(_cp.slug)
seen.add(_cp.slug.lower())
- return slugs
+ # Nous excluded: its picker branch builds from the curated list and it
+ # cannot reach the api_key-only pathway, so a prefetched entry is written
+ # and never read.
+ return [s for s in slugs if s != "nous"]
def list_authenticated_providers(
@@ -3218,6 +3364,19 @@ def list_authenticated_providers(
if any(os.environ.get(ev) for ev in pcfg.api_key_env_vars):
has_creds = True
break
+ # External-process providers (copilot-acp) hold no API key, OAuth
+ # token, or pool entry by design — the spawned ACP subprocess brings
+ # its own auth. "Configured" means the executable resolves, which is
+ # exactly what get_auth_status() reports for them; without this branch
+ # the has_creds filter below unconditionally hides the provider from
+ # every picker (#63662).
+ if not has_creds and overlay.auth_type == "external_process":
+ try:
+ from hermes_cli.auth import get_auth_status
+ _ext_status = get_auth_status(hermes_slug) or {}
+ has_creds = bool(_ext_status.get("logged_in") or _ext_status.get("configured"))
+ except Exception as exc:
+ logger.debug("External-process check failed for %s: %s", pid, exc)
# Check auth store and credential pool for non-env-var credentials.
# This applies to OAuth providers AND api_key providers that also
# support OAuth (e.g. anthropic supports both API key and Claude Code
@@ -3332,6 +3491,19 @@ def list_authenticated_providers(
# curated list alone (still correct, just may lag newly
# launched models, exactly like an offline CLI run).
pass
+ # Outside the try above, so a failed recommendation fetch still
+ # yields a policy-filtered curated list.
+ try:
+ from hermes_cli.models import (
+ nous_policy_allowed_ids as _nous_policy,
+ restrict_to_nous_policy as _nous_restrict,
+ )
+
+ model_ids = _nous_restrict(
+ model_ids, _nous_policy(), rescue_empty=True,
+ )
+ except Exception:
+ pass
else:
# Unified pathway — see Section 1 rationale. Fall back to the
# curated dict (with models.dev merge for preferred providers)
@@ -3379,7 +3551,20 @@ def list_authenticated_providers(
_cp_config = _auth_registry.get(_cp.slug)
_cp_has_creds = False
if _cp_config and _cp_config.api_key_env_vars:
- _cp_has_creds = any(os.environ.get(ev) for ev in _cp_config.api_key_env_vars)
+ _cp_lit = {ev for ev in _cp_config.api_key_env_vars if os.environ.get(ev)}
+ _cp_has_creds = bool(_cp_lit)
+ # A regional "-cn" twin lit only by key vars it shares with its
+ # non-CN sibling (e.g. alibaba-coding-plan-cn off the intl
+ # ALIBABA_CODING_PLAN_API_KEY) is a phantom picker row (#101122).
+ # Hide it unless the user configured that CN provider -- and only
+ # when it has a dedicated var of its own the user could set instead.
+ _sib = _auth_registry.get(_cp.slug[:-3]) if _cp.slug.endswith("-cn") else None
+ _sib_vars = set(_sib.api_key_env_vars) if _sib else set()
+ if (
+ _cp_lit and _cp_lit <= _sib_vars < set(_cp_config.api_key_env_vars)
+ and _cp.slug != current_provider
+ ):
+ continue
# Also check auth store and credential pool
if not _cp_has_creds:
try:
diff --git a/hermes_cli/models.py b/hermes_cli/models.py
index eac78f40a8..020e81c5c6 100644
--- a/hermes_cli/models.py
+++ b/hermes_cli/models.py
@@ -82,6 +82,7 @@ def _custom_provider_ssl_context(base_url: str):
# (model_id, display description shown in menus)
OPENROUTER_MODELS: list[tuple[str, str]] = [
# Anthropic
+ ("anthropic/claude-fable-5.1", ""),
("anthropic/claude-fable-5", ""),
("anthropic/claude-opus-5", ""),
("anthropic/claude-opus-5-fast", "2x price, higher output speed"),
@@ -266,6 +267,7 @@ _PROVIDER_MODELS: dict[str, list[str]] = {
"moa": ["default"],
"nous": [
# Anthropic
+ "anthropic/claude-fable-5.1",
"anthropic/claude-fable-5",
"anthropic/claude-opus-5",
"anthropic/claude-opus-4.8",
@@ -2343,12 +2345,22 @@ def _cached_catalog(cache_key: str) -> Optional[dict[str, dict[str, Any]]]:
def _cache_catalog(
- cache_key: str, result: dict[str, dict[str, Any]]
+ cache_key: str,
+ result: dict[str, dict[str, Any]],
+ ttl_seconds: Optional[float] = None,
) -> dict[str, dict[str, Any]]:
- """Cache a catalog result, giving an empty one an expiry."""
+ """Cache a catalog result, giving an empty one an expiry.
+
+ *ttl_seconds* expires a non-empty result too. Only a catalog whose contents
+ depend on server-side state the client cannot observe needs it — an org's
+ model policy can change while a long-lived process holds the entry.
+ """
_pricing_cache[cache_key] = result
if result:
- _pricing_cache_retry_after.pop(cache_key, None)
+ if ttl_seconds:
+ _pricing_cache_retry_after[cache_key] = time.monotonic() + ttl_seconds
+ else:
+ _pricing_cache_retry_after.pop(cache_key, None)
else:
_pricing_cache_retry_after[cache_key] = (
time.monotonic() + _FAILED_CATALOG_TTL_SECONDS
@@ -2356,6 +2368,46 @@ def _cache_catalog(
return result
+# NUL cannot appear in a URL, so this cannot collide with a real base URL.
+_PRICING_AUTH_KEY_PREFIX = "\x00auth:"
+
+
+def _pricing_auth_fingerprint(api_key: str | None) -> str:
+ """Key suffix identifying the credential a catalog was read with.
+
+ A governed endpoint answers each token with the catalog its org may reach,
+ so two credentials cannot share an entry. blake2b for cache-key
+ fingerprinting only, same rationale as :func:`_custom_endpoint_fingerprint`.
+ """
+ if not api_key:
+ return ""
+ import hashlib
+
+ digest = hashlib.blake2b(api_key.encode("utf-8", errors="replace"), digest_size=8)
+ return _PRICING_AUTH_KEY_PREFIX + digest.hexdigest()
+
+
+def peek_cached_pricing(base_url: str) -> dict[str, dict[str, Any]]:
+ """Pricing already cached for *base_url*, or ``{}``. Never fetches.
+
+ Accepts a ``/v1``-suffixed URL as well as the pre-``/v1`` root the fetchers
+ key on, and prefers an authenticated catalog. Scans rather than rebuilding a
+ key, because callers hold a base URL but no credential — newest first, and
+ skipping expired entries, so a rotated credential does not keep answering
+ from the catalog its predecessor read.
+ """
+ root = (base_url or "").rstrip("/")
+ if root.endswith("/v1"):
+ root = root[:-3].rstrip("/")
+ authed_prefix = root + _PRICING_AUTH_KEY_PREFIX
+ for key in reversed(list(_pricing_cache)):
+ if key.startswith(authed_prefix):
+ cached = _cached_catalog(key)
+ if cached:
+ return cached
+ return _cached_catalog(root) or {}
+
+
def _format_price_per_mtok(per_token_str: str) -> str:
"""Convert a per-token price string to a human-friendly $/Mtok string.
@@ -2489,10 +2541,12 @@ def fetch_models_with_pricing(
*,
force_refresh: bool = False,
include_sale_original: bool = False,
+ cache_ttl_seconds: Optional[float] = None,
) -> dict[str, dict[str, Any]]:
"""Fetch ``/v1/models`` and return ``{model_id: {prompt, completion, ...}}``.
- Results are cached per *base_url* so repeated calls are free.
+ Results are cached per *base_url* and per credential, so repeated calls are
+ free and one caller's catalog never answers another's read.
Works with any OpenRouter-compatible endpoint (OpenRouter, Nous Portal).
When *include_sale_original* is true (Nous Portal only) and the gateway
@@ -2503,13 +2557,14 @@ def fetch_models_with_pricing(
``{prompt, completion}`` shape even if a response happens to nest
``original``.
"""
- cache_key = (base_url or "").rstrip("/")
+ url_root = (base_url or "").rstrip("/")
+ cache_key = url_root + _pricing_auth_fingerprint(api_key)
if not force_refresh:
cached = _cached_catalog(cache_key)
if cached is not None:
return cached
- url = cache_key + "/v1/models"
+ url = url_root + "/v1/models"
headers: dict[str, str] = {
"Accept": "application/json",
"User-Agent": _HERMES_USER_AGENT,
@@ -2560,7 +2615,7 @@ def fetch_models_with_pricing(
entry["original"] = orig_entry
result[mid] = entry
- return _cache_catalog(cache_key, result)
+ return _cache_catalog(cache_key, result, cache_ttl_seconds)
def fetch_ai_gateway_pricing(
@@ -2670,6 +2725,86 @@ def _resolve_nous_pricing_credentials() -> tuple[str, str]:
return (api_key, base_url)
+def nous_policy_allowed_ids(*, force_refresh: bool = False) -> Optional[set[str]]:
+ """The Nous model ids the caller's org may reach, or ``None`` to not filter.
+
+ The gateway omits policy-blocked rows from an authenticated
+ ``GET /v1/models``, so that response's keys are the reachable set.
+
+ ``None`` means "leave the caller's list alone", for the three states that
+ cannot support narrowing one: no policy (or a token too old to say), an
+ anonymous read whose catalog is unfiltered, and an empty read, which is a
+ fetch failure rather than an org that may reach nothing.
+ """
+ try:
+ from hermes_cli.nous_account import nous_policy_present
+
+ if nous_policy_present() is not True:
+ return None
+ except Exception:
+ return None
+
+ api_key, base_url = _resolve_nous_pricing_credentials()
+ if not api_key or not base_url:
+ return None
+
+ # Same arguments as get_pricing_for_provider's nous branch, so a caller
+ # asking for pricing too shares this entry instead of paying for a second
+ # request.
+ pricing = fetch_models_with_pricing(
+ api_key=api_key,
+ base_url=base_url,
+ force_refresh=force_refresh,
+ include_sale_original=True,
+ cache_ttl_seconds=_NOUS_CATALOG_TTL_SECONDS,
+ )
+ return set(pricing) or None
+
+
+# Past this size an allowed set reads as a whole catalog rather than an
+# allowlist, and is not worth showing in place of an empty picker.
+_NOUS_POLICY_APPEND_MAX = 64
+
+# How long a Nous catalog stays trusted. Its contents depend on the org's
+# policy, which an admin can change at any time and the client cannot observe,
+# so a long-lived process must re-ask instead of holding the first answer for
+# its whole life. Other providers' catalogs carry no such state and keep the
+# default no-expiry caching.
+_NOUS_CATALOG_TTL_SECONDS = 300.0
+
+
+def restrict_to_nous_policy(
+ model_ids: list[str],
+ allowed: Optional[set[str]],
+ *,
+ rescue_empty: bool = False,
+) -> list[str]:
+ """*model_ids* narrowed to *allowed*, preserving the caller's order.
+
+ A ``None`` or empty *allowed* leaves the list untouched.
+
+ A ``:free`` sibling is kept when its base model is reachable, mirroring the
+ gateway, which admits a row when any of its requestable ids passes. Prefer
+ over-listing: that costs a 403 from the authoritative gate, while hiding a
+ row the gate would serve is unrecoverable from the client.
+ """
+ if not allowed:
+ return list(model_ids)
+ kept = [
+ mid
+ for mid in model_ids
+ if mid in allowed or mid.split(":", 1)[0] in allowed
+ ]
+
+ # An allowlist can name only models the curated manifest lacks, leaving an
+ # empty picker — worse than no filter, since the models the org may use are
+ # the ones dropped. Opt-in per list: an already-empty list (a paid tier's
+ # gated models) means "nothing to gate", not "nothing survived".
+ if rescue_empty and not kept and len(allowed) <= _NOUS_POLICY_APPEND_MAX:
+ return sorted(allowed)
+ return kept
+
+
def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> dict[str, dict[str, str]]:
"""Return live pricing for providers that support it (openrouter, nous, ai-gateway, novita)."""
normalized = normalize_provider(provider)
@@ -2696,6 +2831,7 @@ def get_pricing_for_provider(provider: str, *, force_refresh: bool = False) -> d
force_refresh=force_refresh,
# Sale chrome (pricing.original) is Nous Portal-only.
include_sale_original=True,
+ cache_ttl_seconds=_NOUS_CATALOG_TTL_SECONDS,
)
return {}
@@ -3612,6 +3748,69 @@ def detect_static_provider_for_model(
return None
+def _configured_provider_ids() -> set[str]:
+ """Provider ids defined in the user's config ``providers:`` block.
+
+ Includes both top-level ids (``ollama``, ``nous``) and ``custom:*``
+ profile ids. Returns an empty set when config is unreadable — callers
+ treat that as "no user-defined providers" and fall through to built-in
+ catalogs only.
+ """
+ try:
+ from hermes_cli.config import load_config
+
+ cfg = load_config() or {}
+ providers = cfg.get("providers")
+ if not isinstance(providers, dict):
+ return set()
+ ids: set[str] = set()
+ for pid in providers:
+ key = str(pid).strip().lower()
+ if key:
+ ids.add(key)
+ return ids
+ except Exception:
+ return set()
+
+
+def _resolve_provider_prefix(model_name: str) -> Optional[tuple[str, str]]:
+ """Resolve an explicit ``vendor/model`` prefix to a configured provider.
+
+ ``nous/deepseek-v4-pro`` or ``ollama/qwen3.5:4b`` should route to the
+ named provider instead of falling back to the configured default (which
+ silently sends non-default models to the wrong endpoint, #87189).
+
+ Only vendors the user actually defined in their ``providers:`` config
+ block (by raw name or alias) are routed here. Built-in vendor prefixes
+ (``google/gemini-2.5-flash``, ``deepseek/deepseek-chat``) deliberately
+ stay on the existing catalog / OpenRouter-slug / default-provider path:
+ those slug forms are aggregator-native, and rerouting them to the vendor
+ provider would change established provider-switch behavior (see
+ ``TestDenormalizeProviderSwitch`` in tests/hermes_cli/test_web_server.py).
+ The returned model is the suffix with the prefix stripped — the target
+ provider's API expects the bare id.
+ """
+ if "/" not in model_name:
+ return None
+ vendor, model = model_name.split("/", 1)
+ vendor = vendor.strip().lower()
+ model = model.strip()
+ if not vendor or not model:
+ return None
+ configured = _configured_provider_ids()
+ if not configured:
+ return None
+ # A provider block the user explicitly named (``ollama:``) wins over the
+ # built-in alias table, which may canonicalize the same name elsewhere
+ # (``ollama`` → ``custom``) and route to the wrong endpoint.
+ if vendor in configured:
+ return (vendor, model)
+ canonical = _PROVIDER_ALIASES.get(vendor, vendor)
+ if canonical in configured:
+ return (canonical, model)
+ return None
+
+
def detect_provider_for_model(
model_name: str,
current_provider: str,
@@ -3648,6 +3847,16 @@ def detect_provider_for_model(
return ("openrouter", or_slug)
return None # already on openrouter with matching name
+ # --- Step 3: explicit ``vendor/model`` prefix naming a configured provider ---
+ # Checked after the OpenRouter slug lookup so aggregator-native slugs
+ # (e.g. ``deepseek/deepseek-chat``) keep their existing routing; only
+ # vendors the user defined in their ``providers:`` block route here,
+ # so catalog/default behavior for built-in vendor prefixes is unchanged
+ # (#87189).
+ prefix_match = _resolve_provider_prefix(name)
+ if prefix_match is not None:
+ return prefix_match
+
return None
@@ -3760,23 +3969,60 @@ def model_supports_fast_mode(model_id: Optional[str]) -> bool:
def _is_anthropic_fast_model(model_id: Optional[str]) -> bool:
"""Return True if the model accepts the Anthropic Fast Mode ``speed`` param.
- This gates the *speed=fast request parameter*, which Anthropic supports on
- Opus 4.6 only (Opus 4.7 explicitly 400s). It is deliberately NOT a general
- "is this a fast model" check: for Opus 4.8 the fast offering is a SEPARATE
- model id (``…-opus-4.8-fast``) selected via the model field, not the speed
- parameter — see ``agent.anthropic_adapter._supports_fast_mode`` and its
- test. Keep this in lock-step with that adapter gate so the UI never shows a
- Fast toggle that the runtime would silently drop.
+ This gates the *speed=fast request parameter*, which Anthropic supports
+ on Opus 4.8 and Opus 5 (research preview, Claude API only). It is
+ deliberately NOT a general "is this a fast model" check:
+
+ - Opus 4.6 had fast mode at launch and LOST it (2026-06-29) — the param
+ is silently ignored (standard speed, standard billing), so exposing a
+ toggle for it would show users a switch that does nothing.
+ - Opus 4.7 hard-400s on the parameter.
+ - Dedicated ``…-fast`` model ids (e.g. OpenRouter's
+ ``claude-opus-4.8-fast``) select fast inference via the model field
+ and must not also receive the speed parameter.
+
+ Keep this in lock-step with ``agent.anthropic_adapter._supports_fast_mode``
+ so the UI never shows a Fast toggle that the runtime would drop.
"""
raw = _strip_vendor_prefix(str(model_id or ""))
base = raw.split(":")[0]
if not base.startswith("claude-"):
return False
- # Only Opus 4.6 supports the speed=fast parameter at present.
- return "opus-4-6" in base or "opus-4.6" in base
+ if "-fast" in base:
+ return False
+ return any(v in base for v in ("opus-4-8", "opus-4.8", "opus-5"))
-def resolve_fast_mode_overrides(model_id: Optional[str]) -> dict[str, Any] | None:
+def _fast_mode_route_supported(
+ model_id: Optional[str], provider: Optional[str], base_url: Optional[str]
+) -> bool:
+ """Only the first-party endpoint that bills for fast mode may receive its params.
+
+ OpenRouter, Nous, Copilot, Azure, Bedrock, and custom base_urls either
+ strip ``service_tier``/``speed`` (charging nothing) or 400 on them.
+ """
+ from urllib.parse import urlparse
+
+ from agent.model_metadata import is_grok_46_family
+
+ if _is_anthropic_fast_model(model_id):
+ allowed = {"anthropic": "api.anthropic.com"}
+ elif is_grok_46_family(str(model_id or "")):
+ allowed = {"xai": "api.x.ai"}
+ else:
+ allowed = {"openai": "api.openai.com", "openai-codex": "chatgpt.com"}
+ if provider and normalize_provider(provider) not in allowed:
+ return False
+ host = (urlparse(str(base_url or "")).hostname or "").lower()
+ return not host or host in allowed.values()
+
+
+def resolve_fast_mode_overrides(
+ model_id: Optional[str],
+ *,
+ provider: Optional[str] = None,
+ base_url: Optional[str] = None,
+) -> dict[str, Any] | None:
"""Return request_overrides for fast/priority mode, or None if unsupported.
Returns provider-appropriate overrides:
@@ -3784,12 +4030,21 @@ def resolve_fast_mode_overrides(model_id: Optional[str]) -> dict[str, Any] | Non
- Anthropic models: ``{"speed": "fast"}`` (Anthropic Fast Mode beta)
- Grok 4.6: ``{"service_tier": "priority"}`` (xAI Priority Processing)
+ When ``provider``/``base_url`` are given the result is also gated on the
+ route (see ``_fast_mode_route_supported``) so proxies never see the
+ params. This is the single fast-mode gate for static ``/fast fast`` and
+ the bounded ``auto``/``cold`` windows in ``agent.fast_mode``.
+
The overrides are injected into the API request kwargs by
- ``_build_api_kwargs`` in run_agent.py — each API path handles its own
- keys (service_tier for OpenAI/Codex, speed for Anthropic Messages).
+ ``build_api_kwargs`` — each API path handles its own keys
+ (service_tier for OpenAI/Codex, speed for Anthropic Messages).
"""
if not model_supports_fast_mode(model_id):
return None
+ if (provider or base_url) and not _fast_mode_route_supported(
+ model_id, provider, base_url
+ ):
+ return None
if _is_anthropic_fast_model(model_id):
return {"speed": "fast"}
return {"service_tier": "priority"}
@@ -3807,13 +4062,18 @@ def _resolve_copilot_catalog_api_key() -> str:
``auth.json`` under ``credential_pool.copilot[]``. The pool is
populated by ``hermes auth add copilot`` and by ``_seed_from_env``
when the env var is set in ``~/.hermes/.env``.
+ 3. ``~/.copilot/config.json`` ``copilotTokens`` — the GitHub Copilot
+ CLI's own store, written by ``copilot login`` on hosts without an
+ OS keychain. Without it, a user whose ONLY credential is the ACP
+ CLI login sees the copilot-acp picker fall back to the stale
+ curated list instead of the models their subscription serves.
- Without (2), users whose only Copilot credential is in the pool see
- the ``/model`` picker fall back to a stale hardcoded list because the
- live catalog fetch silently 401s. To avoid wedging on a malformed pool
- entry, each candidate is exchanged via ``exchange_copilot_token`` —
- only entries that actually exchange successfully are returned, so a
- later valid entry is reachable when an earlier one is unsupported.
+ Without (2)/(3), users without env-var credentials see the ``/model``
+ picker fall back to a stale hardcoded list because the live catalog
+ fetch silently 401s. To avoid wedging on a malformed entry, each
+ candidate is exchanged via ``exchange_copilot_token`` — only entries
+ that actually exchange successfully are returned, so a later valid
+ entry is reachable when an earlier one is unsupported.
"""
try:
from hermes_cli.auth import resolve_api_key_provider_credentials
@@ -3842,7 +4102,11 @@ def _resolve_copilot_catalog_api_key() -> str:
if not valid:
continue
try:
- api_token, _expires_at = exchange_copilot_token(raw)
+ # exchange_copilot_token returns (api_token, expires_at,
+ # base_url) — a 2-name unpack raises ValueError, which the
+ # except below silently swallowed, disabling this entire
+ # resolution path.
+ api_token = exchange_copilot_token(raw)[0]
except Exception:
continue
if api_token:
@@ -3850,6 +4114,41 @@ def _resolve_copilot_catalog_api_key() -> str:
except Exception:
pass
+ # 3. Copilot CLI plaintext token store (JSONC — strip //-comment lines).
+ try:
+ import json as _json
+
+ from hermes_cli.copilot_auth import (
+ exchange_copilot_token,
+ validate_copilot_token,
+ )
+
+ cli_config = os.path.expanduser("~/.copilot/config.json")
+ if os.path.isfile(cli_config):
+ with open(cli_config, "r", encoding="utf-8", errors="ignore") as fh:
+ raw_text = "\n".join(
+ line for line in fh.read().splitlines()
+ if not line.lstrip().startswith("//")
+ )
+ data = _json.loads(raw_text) if raw_text.strip() else {}
+ tokens = data.get("copilotTokens")
+ if isinstance(tokens, dict):
+ for raw in tokens.values():
+ raw = str(raw or "").strip()
+ if not raw:
+ continue
+ valid, _ = validate_copilot_token(raw)
+ if not valid:
+ continue
+ try:
+ api_token = exchange_copilot_token(raw)[0]
+ except Exception:
+ continue
+ if api_token:
+ return api_token
+ except Exception:
+ pass
+
return ""
@@ -4807,12 +5106,14 @@ def copilot_default_headers(*, is_agent_turn: bool = True) -> dict[str, str]:
}
-def _copilot_catalog_item_is_text_model(item: dict[str, Any]) -> bool:
+def _copilot_catalog_item_is_text_model(
+ item: dict[str, Any], *, ignore_picker_flag: bool = False
+) -> bool:
model_id = str(item.get("id") or "").strip()
if not model_id:
return False
- if item.get("model_picker_enabled") is False:
+ if not ignore_picker_flag and item.get("model_picker_enabled") is False:
return False
capabilities = item.get("capabilities")
@@ -4891,6 +5192,25 @@ def fetch_github_model_catalog(
continue
seen_ids.add(model_id)
models.append(item)
+ if not models and items:
+ # GitHub has been observed returning
+ # ``model_picker_enabled: false`` for EVERY model on some
+ # accounts/token types, which would silently reject the
+ # whole live catalog and strand the picker on the stale
+ # curated fallback. The flag is a display hint, not an
+ # availability contract — when honoring it empties the
+ # catalog, retry without it (chat/endpoint checks still
+ # apply, so embeddings and non-chat rows stay excluded).
+ for item in items:
+ if not _copilot_catalog_item_is_text_model(
+ item, ignore_picker_flag=True
+ ):
+ continue
+ model_id = str(item.get("id") or "").strip()
+ if not model_id or model_id in seen_ids:
+ continue
+ seen_ids.add(model_id)
+ models.append(item)
if models:
_github_model_catalog_cache = copy.deepcopy(models)
_github_model_catalog_cache_key = api_key
diff --git a/hermes_cli/nous_account.py b/hermes_cli/nous_account.py
index 654487e684..30247ff661 100644
--- a/hermes_cli/nous_account.py
+++ b/hermes_cli/nous_account.py
@@ -396,6 +396,50 @@ def get_nous_portal_account_info(
)
+def nous_policy_present() -> Optional[bool]:
+ """Whether the caller's org carries a restrictive model/provider policy.
+
+ Reads the ``policy_present`` claim off the access token, so it costs no
+ request; ``/api/oauth/account`` does not carry it. Stamped at mint time, so
+ it goes stale until the next token refresh.
+
+ ``None`` is unknown — an older mint or an unreadable claim — and must not be
+ reported as the absence of a policy.
+ """
+ try:
+ from hermes_cli.auth import get_provider_auth_state, _decode_jwt_claims
+
+ state = get_provider_auth_state("nous") or {}
+ access_token = state.get("access_token")
+ if not isinstance(access_token, str) or not access_token.strip():
+ return None
+ claims = _decode_jwt_claims(access_token)
+ if not claims:
+ return None
+ return _coerce_bool(claims.get("policy_present"))
+ except Exception:
+ return None
+
+
+def nous_policy_notice(*, removed: bool) -> str:
+ """A one-line notice for a list the org's policy narrowed, else ``""``.
+
+ A blocked model is omitted rather than marked, which reads as "Hermes does
+ not support this". This says which it is without enumerating the blocked
+ set, which under an allowlist is most of the catalog.
+
+ *removed* is whether the filter actually dropped anything. The catalog read
+ fails open — an anonymous or empty one narrows nothing — so the claim alone
+ would label a full list as filtered.
+ """
+ if not removed or nous_policy_present() is not True:
+ return ""
+ return (
+ "Your organization restricts which models are available — "
+ "models outside its policy are not listed."
+ )
+
+
def _fresh_account_info(
*,
state: dict[str, Any],
diff --git a/hermes_cli/nous_subscription.py b/hermes_cli/nous_subscription.py
index f9ca7ef35f..a930989e60 100644
--- a/hermes_cli/nous_subscription.py
+++ b/hermes_cli/nous_subscription.py
@@ -505,6 +505,10 @@ def get_nous_subscription_features(
direct_exa = bool(get_env_value("EXA_API_KEY"))
direct_firecrawl = bool(get_env_value("FIRECRAWL_API_KEY") or get_env_value("FIRECRAWL_API_URL"))
direct_parallel = bool(get_env_value("PARALLEL_API_KEY"))
+ direct_tavily = bool(get_env_value("TAVILY_API_KEY"))
+ # Keyless Tavily is opt-in: selecting it in `hermes tools` / setup writes
+ # web.backend (or a per-capability override) without requiring a key.
+ tavily_selected = "tavily" in {web_backend, web_search_backend, web_extract_backend}
direct_searxng = bool(get_env_value("SEARXNG_URL"))
direct_fal = fal_key_is_configured()
direct_fal_video = direct_fal # same FAL_KEY; separate var so use_gateway is independent
@@ -536,6 +540,8 @@ def get_nous_subscription_features(
direct_firecrawl = False
direct_exa = False
direct_parallel = False
+ direct_tavily = False
+ tavily_selected = False
if image_use_gateway:
direct_fal = False
if video_use_gateway:
@@ -624,6 +630,7 @@ def get_nous_subscription_features(
direct_camofox = False
+ tavily_ready = direct_tavily or tavily_selected
web_managed = web_backend == "firecrawl" and managed_web_available and not direct_firecrawl
web_active = bool(
web_tool_enabled
@@ -632,6 +639,7 @@ def get_nous_subscription_features(
or (web_backend == "exa" and direct_exa)
or (web_backend == "firecrawl" and direct_firecrawl)
or (web_backend == "parallel" and direct_parallel)
+ or (web_backend == "tavily" and tavily_ready)
or (web_backend == "searxng" and direct_searxng)
# Per-capability overrides: search_backend or extract_backend may be set
# without web.backend (using the new split config from #20061)
@@ -639,6 +647,8 @@ def get_nous_subscription_features(
or (web_search_backend == "exa" and direct_exa)
or (web_search_backend == "firecrawl" and direct_firecrawl)
or (web_search_backend == "parallel" and direct_parallel)
+ or (web_search_backend == "tavily" and tavily_ready)
+ or (web_extract_backend == "tavily" and tavily_ready)
)
)
web_available = bool(
@@ -646,6 +656,7 @@ def get_nous_subscription_features(
or direct_exa
or direct_firecrawl
or direct_parallel
+ or tavily_ready
or direct_searxng
)
@@ -889,6 +900,7 @@ def apply_nous_managed_defaults(
if "web" in selected_toolsets and not features.web.explicit_configured and not (
get_env_value("PARALLEL_API_KEY")
+ or get_env_value("TAVILY_API_KEY")
or get_env_value("FIRECRAWL_API_KEY")
or get_env_value("FIRECRAWL_API_URL")
):
@@ -986,6 +998,7 @@ def _get_gateway_direct_credentials() -> Dict[str, bool]:
get_env_value("FIRECRAWL_API_KEY")
or get_env_value("FIRECRAWL_API_URL")
or get_env_value("PARALLEL_API_KEY")
+ or get_env_value("TAVILY_API_KEY")
or get_env_value("EXA_API_KEY")
# Env-configured keyless local backend: a reachable self-hosted
# SearXNG is a working web setup even with no stored selection
diff --git a/hermes_cli/observability/relay_shared_metrics.py b/hermes_cli/observability/relay_shared_metrics.py
index 2ab88f51c3..5a97c8a18d 100644
--- a/hermes_cli/observability/relay_shared_metrics.py
+++ b/hermes_cli/observability/relay_shared_metrics.py
@@ -132,6 +132,9 @@ class _Runtime:
self._sessions: dict[str, _MetricsSession] = {}
self._task_creation_lock = threading.RLock()
self._task_sessions_lock = threading.RLock()
+ # Guards the opt-in send pass: at most one in flight per process.
+ self._send_lock = threading.RLock()
+ self._send_thread: threading.Thread | None = None
self._task_sessions: dict[tuple[str, str], _MetricsSession] = {}
self._turn_sessions: dict[tuple[str, str], _MetricsSession] = {}
self._subscriber_name = f"{SUBSCRIBER_NAME}.{self.host.runtime_id}"
@@ -668,6 +671,12 @@ class _Runtime:
self._safe(self.relay.subscribers.deregister, self._subscriber_name)
self.host.release_managed_execution(self._subscriber_name)
self._registered = False
+ # The final export above may have started a send. Give it the same
+ # bounded chance to finish that deactivate() gets — without this a
+ # short-lived CLI process exits immediately and kills the daemon
+ # thread mid-request, which is the common case for the one cadence
+ # this feature has.
+ self._join_send_thread()
try:
atexit.unregister(self.shutdown)
except Exception:
@@ -706,11 +715,29 @@ class _Runtime:
with self._task_sessions_lock:
self._task_sessions.clear()
self._turn_sessions.clear()
+ self._join_send_thread()
try:
atexit.unregister(self.shutdown)
except Exception:
pass
+ def _join_send_thread(self, timeout: float = 2.0) -> None:
+ """Give an in-flight send a brief chance to finish at exit.
+
+ Bounded on purpose: the packages stay pending in SQLite and go out on
+ the next run, so blocking a user's shutdown for a slow network is the
+ wrong trade. The thread is a daemon, so an unfinished pass dies with
+ the process rather than holding it open.
+ """
+ with self._send_lock:
+ thread = self._send_thread
+ if thread is None or not thread.is_alive():
+ return
+ try:
+ thread.join(timeout)
+ except Exception:
+ logger.debug("Shared-metrics send thread join failed", exc_info=True)
+
def _session(self, event: dict[str, Any]) -> _MetricsSession | None:
session_id = str(event.get("session_id") or "")
with self._sessions_lock:
@@ -1048,7 +1075,104 @@ class _Runtime:
return True
def _export(self) -> None:
- self._safe(self.subscriber.store.create_and_export_package_if_due)
+ exported = self._safe(self.subscriber.store.create_and_export_package_if_due)
+ # Sending is opt-in and must never delay the caller: _export runs on
+ # finish_task, which is the user's interactive path. Errors inside the
+ # sender are already swallowed there; the thread is about latency, not
+ # correctness.
+ if exported is not None:
+ self._safe(self._send_exported_packages)
+
+ def _observe_send_consent(self, send_enabled: bool) -> None:
+ """Reconcile consent windows with the observed config state.
+
+ Thin wrapper over the SINGLE consent writer. The old edge-detection
+ body (last-seen key, rising/falling branches) is gone: reconciliation
+ derives the correct window state from what it observes, so there is
+ no transition to miss and no ordering between callers to get wrong.
+
+ Failures must never break the export hook, but they are logged at
+ warning rather than debug: silently failing to close a consent window
+ is a privacy-relevant event, not routine bookkeeping.
+ """
+ try:
+ from hermes_cli.observability.shared_metrics_sender import (
+ reconcile_send_consent,
+ )
+ from hermes_cli.sqlite_util import write_txn
+
+ with self.subscriber.store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, send_enabled)
+ except Exception:
+ logger.warning(
+ "Unable to record a shared-metrics consent transition",
+ exc_info=True,
+ )
+
+ def _send_exported_packages(self) -> None:
+ from hermes_cli.observability.shared_metrics_send_config import (
+ resolve_send_config,
+ )
+
+ try:
+ from hermes_cli.config import read_raw_config_readonly
+
+ config = read_raw_config_readonly() or {}
+ except Exception:
+ logger.debug("Unable to read shared-metrics send policy", exc_info=True)
+ return
+
+ resolved = resolve_send_config(config)
+
+ # Observe the consent EDGE before deciding whether to send. Recording
+ # revocation inside the send loop (as an earlier fix did) can never
+ # work: the dominant case is the user turning sending off while no
+ # pass is running, and then this method returns below without ever
+ # constructing a sender. The window has to close on the transition,
+ # not on the next transmission that by definition will not happen.
+ self._observe_send_consent(resolved.send)
+
+ if not resolved.send:
+ return
+
+ with self._send_lock:
+ # One in-flight pass per process. A queued second pass would add
+ # nothing: the next hook fire picks up whatever is still pending.
+ if self._send_thread is not None and self._send_thread.is_alive():
+ return
+ thread = threading.Thread(
+ target=self._run_send_pass,
+ args=(resolved.endpoint,),
+ name="hermes-shared-metrics-send",
+ daemon=True,
+ )
+ self._send_thread = thread
+ thread.start()
+
+ def _run_send_pass(self, endpoint: str) -> None:
+ from hermes_cli.observability.shared_metrics_sender import (
+ SharedMetricsSender,
+ )
+
+ def still_consented() -> bool:
+ """Re-read consent so revoking `send` stops an in-flight pass."""
+ from hermes_cli.config import read_raw_config_readonly
+ from hermes_cli.observability.shared_metrics_send_config import (
+ resolve_send_config,
+ )
+
+ resolved = resolve_send_config(read_raw_config_readonly() or {})
+ return resolved.send and resolved.endpoint == endpoint
+
+ try:
+ SharedMetricsSender(
+ self.subscriber.store,
+ endpoint,
+ consent_check=still_consented,
+ ).send_pending()
+ except Exception:
+ logger.warning("Shared-metrics send pass failed", exc_info=True)
def _event_metadata(self) -> dict[str, str]:
return {
@@ -1101,8 +1225,62 @@ def handles_hook(hook_name: str) -> bool:
return hook_name in HANDLED_HOOKS and enabled()
+_consent_reconcile_done = False
+
+
+def _reconcile_send_consent_once() -> None:
+ """Reconcile consent windows with config, once per process.
+
+ Runs BEFORE and INDEPENDENT of the collection gate — that placement is
+ the fix for the round-5 D1 leak, where the only idle-path consent
+ observer sat behind ``handles_hook()`` and became dead code the moment
+ ``enabled: false`` was set. A user with collection off still gets their
+ send-consent windows reconciled here.
+
+ Skipped only when there is no store on disk AND consent is off: with no
+ store there are no packages, so there is nothing a window could protect,
+ and creating ``~/.hermes/telemetry`` for every fully-disabled user would
+ be a behaviour change in the wrong direction.
+ """
+ global _consent_reconcile_done
+ if _consent_reconcile_done:
+ return
+ _consent_reconcile_done = True
+ try:
+ from hermes_cli.config import read_raw_config_readonly
+ from hermes_cli.observability.shared_metrics import SharedMetricsStore
+ from hermes_cli.observability.shared_metrics_send_config import (
+ resolve_send_config,
+ )
+ from hermes_cli.observability.shared_metrics_sender import (
+ reconcile_send_consent,
+ )
+ from hermes_cli.sqlite_util import write_txn
+ from hermes_constants import get_hermes_home
+
+ resolved = resolve_send_config(read_raw_config_readonly() or {})
+ # Probe for an existing store WITHOUT constructing one: the
+ # constructor creates the directory and schema as a side effect,
+ # which round 6 caught making this skip dead code — every
+ # fully-disabled user was getting a ~/.hermes/telemetry directory.
+ default_path = (
+ get_hermes_home() / "telemetry" / "shared_metrics" / "metrics.sqlite3"
+ )
+ if not resolved.send and not default_path.exists():
+ return
+ store = SharedMetricsStore()
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, resolved.send)
+ except Exception:
+ logger.warning(
+ "Unable to reconcile shared-metrics send consent", exc_info=True
+ )
+
+
def observe_lifecycle(hook_name: str, **kwargs: Any) -> None:
"""Project one Hermes lifecycle event into the core Relay integration."""
+ _reconcile_send_consent_once()
if not handles_hook(hook_name):
return
if not relay_runtime.relay_instrumentation_enabled():
diff --git a/hermes_cli/observability/shared_metrics.py b/hermes_cli/observability/shared_metrics.py
index fd42b06230..87094922d9 100644
--- a/hermes_cli/observability/shared_metrics.py
+++ b/hermes_cli/observability/shared_metrics.py
@@ -337,6 +337,8 @@ class SharedMetricsStore:
)
"""
)
+ SharedMetricsStore._add_send_columns(connection)
+ SharedMetricsStore._add_consent_tables(connection)
connection.execute(
"""
INSERT INTO telemetry_state(key, value)
@@ -346,6 +348,98 @@ class SharedMetricsStore:
(_STORE_SCHEMA_VERSION,),
)
+ @staticmethod
+ def _add_send_columns(connection: sqlite3.Connection) -> None:
+ """Add transmission bookkeeping to ``package_outbox``, idempotently.
+
+ These columns are ADDITIVE and nullable, and the store schema version
+ is deliberately NOT bumped. ``_ensure_schema_in_transaction`` raises on
+ any version it does not recognise and has no forward-compatibility
+ branch, so bumping would make an older Hermes — a second profile on an
+ older build, or a rollback — hard-fail against the same database file.
+ Old readers select named columns and never ``SELECT *``, so extra
+ columns are invisible to them.
+ """
+ existing = {
+ str(row["name"])
+ for row in connection.execute("PRAGMA table_info(package_outbox)")
+ }
+ for column, declaration in (
+ # When the 202 was received. NULL = never acknowledged.
+ ("sent_at", "TEXT"),
+ # NULL/'pending' = eligible, 'sent' = done, 'rejected' = permanent 400.
+ ("send_state", "TEXT"),
+ ("send_attempts", "INTEGER NOT NULL DEFAULT 0"),
+ # Earliest next attempt; enforces backoff across process restarts.
+ ("next_attempt_at", "TEXT"),
+ ("last_error", "TEXT"),
+ # The identifier actually transmitted, frozen on the first
+ # attempt so retries stay byte-identical. Since the 2026-08-27
+ # product decision this is the stable install_id itself.
+ # Only the ~36-byte id is stored: the body is recomputed from
+ # payload_json, whose serialisation is deterministic.
+ ("sent_install_id", "TEXT"),
+ # NULL until first claimed; rewritten on every claim. Settlement
+ # and the pre-POST revalidation are compare-and-set on this, so a
+ # claimant whose lease lapsed loses authority the moment another
+ # process reclaims (PR-review finding: without it, a suspended
+ # sender resuming after a reclaim double-POSTs the package).
+ ("claim_token", "TEXT"),
+ ):
+ if column not in existing:
+ connection.execute(
+ f"ALTER TABLE package_outbox ADD COLUMN {column} {declaration}"
+ )
+
+ @staticmethod
+ def _add_consent_tables(connection: sqlite3.Connection) -> None:
+ """Create the consent-window tables, idempotently.
+
+ Additive like ``_add_send_columns`` — the schema version is
+ deliberately NOT bumped, and old readers never touch these tables.
+
+ ``send_consent_windows`` records consent as explicit intervals rather
+ than a moving day-stamp: a window is opened when send consent is
+ observed, heartbeat-confirmed on every later observation, and closed
+ at the LAST CONFIRMED moment (never "now") when consent is observed
+ withdrawn. Consent is asserted only for time that was actually
+ observed, so unobserved gaps — a hand-edited config with no process
+ running — fail closed by construction.
+
+ ``consent_marks`` holds two monotonic high-water marks with strictly
+ separated roles:
+
+ - ``obs``: the latest observation stamp ever seen. Advanced only by
+ the reconciler. Confirms consent and clamps window closes.
+ - ``data``: the latest package ``period_end`` ever stored. Advanced
+ only by the package writer. Clamps window OPENS, so a rolled-back
+ clock can never open a window underneath packages that already
+ exist on disk.
+
+ The separation is load-bearing: letting data stamps confirm consent
+ re-created a refused-window leak (packages stored during an off
+ window would vouch for it), and letting observation stamps clamp
+ opens is not enough on its own to stop a rollback sliding a window
+ under existing refused data.
+ """
+ connection.execute(
+ """
+ CREATE TABLE IF NOT EXISTS send_consent_windows (
+ opened_at TEXT NOT NULL,
+ last_confirmed_at TEXT NOT NULL,
+ closed_at TEXT
+ )
+ """
+ )
+ connection.execute(
+ """
+ CREATE TABLE IF NOT EXISTS consent_marks (
+ name TEXT PRIMARY KEY CHECK (name IN ('obs', 'data')),
+ stamp TEXT NOT NULL
+ )
+ """
+ )
+
@staticmethod
def _create_counter_aggregates_table(connection: sqlite3.Connection) -> None:
connection.execute(
@@ -580,6 +674,16 @@ class SharedMetricsStore:
payload["generated_at"],
),
)
+ # Advance the data high-water mark. This is the ONLY writer of the
+ # 'data' mark: it clamps consent-window opens so a rolled-back clock
+ # can never open a window underneath packages that already exist.
+ connection.execute(
+ """
+ INSERT INTO consent_marks(name, stamp) VALUES ('data', ?)
+ ON CONFLICT(name) DO UPDATE SET stamp = MAX(stamp, excluded.stamp)
+ """,
+ (payload["period_end"],),
+ )
for row in rows:
connection.execute(
"""
diff --git a/hermes_cli/observability/shared_metrics_send_config.py b/hermes_cli/observability/shared_metrics_send_config.py
new file mode 100644
index 0000000000..cb14027593
--- /dev/null
+++ b/hermes_cli/observability/shared_metrics_send_config.py
@@ -0,0 +1,114 @@
+"""Configuration for shared-metrics transmission.
+
+Collection (``telemetry.shared_metrics.enabled``) and transmission
+(``telemetry.shared_metrics.send``) are separate opt-ins. See
+``docs/observability/relay-shared-metrics.md`` Appendix A for the consent,
+identity, rotation, retention, and deletion decisions behind this module.
+"""
+
+from __future__ import annotations
+
+import logging
+from dataclasses import dataclass
+from urllib.parse import urlparse
+
+logger = logging.getLogger(__name__)
+
+#: Production ingest endpoint. Overridable through config only.
+#:
+#: Deliberately NOT overridable by an environment variable: AGENTS.md reserves
+#: HERMES_* env vars for secrets, and a behavioural override here would be a
+#: consent hazard — a user who agreed to send metrics to Nous could have them
+#: silently redirected to any host by an inherited variable, with nothing
+#: visible in their config to show it. Tests and the staging E2E write this
+#: key into a throwaway profile instead.
+DEFAULT_ENDPOINT = "https://telemetry.nousresearch.com/v1/telemetry"
+
+_LOCAL_HOSTS = frozenset({"localhost", "127.0.0.1", "::1", "[::1]"})
+
+# Module-level latch: the enabled/send mismatch is a static misconfiguration,
+# so it is reported once per process instead of on every hook fire.
+_warned_send_without_collection = False
+
+
+@dataclass(frozen=True)
+class SendConfig:
+ """Resolved transmission settings."""
+
+ #: Collection is on. Nothing is packaged or sent without it.
+ enabled: bool
+ #: Transmission is on AND permitted (that is, collection is also on).
+ send: bool
+ #: Where packages are POSTed.
+ endpoint: str
+
+
+def _endpoint_is_safe(endpoint: str) -> bool:
+ """Reject plaintext destinations unless they are loopback.
+
+ Telemetry must not leave a machine in clear text because of a typo in a
+ config file. Loopback stays allowed so tests can use a local HTTP server.
+ """
+ try:
+ parsed = urlparse(endpoint)
+ except ValueError:
+ return False
+ if parsed.scheme == "https":
+ return True
+ if parsed.scheme == "http":
+ return (parsed.hostname or "") in _LOCAL_HOSTS
+ return False
+
+
+def resolve_send_config(config: dict | None) -> SendConfig:
+ """Resolve transmission settings from config plus the environment.
+
+ Endpoint precedence: config > production default.
+
+ ``send`` is returned as False whenever transmission cannot legitimately
+ happen, so callers never have to re-check the combination.
+ """
+ global _warned_send_without_collection
+
+ raw = config if isinstance(config, dict) else {}
+ telemetry = raw.get("telemetry")
+ telemetry = telemetry if isinstance(telemetry, dict) else {}
+ shared = telemetry.get("shared_metrics")
+ shared = shared if isinstance(shared, dict) else {}
+
+ enabled = shared.get("enabled") is True
+ send_requested = shared.get("send") is True
+
+ if send_requested and not enabled:
+ # Loud, not silent: the user believes telemetry is being sent, and it
+ # never will be. Error level, once per process.
+ if not _warned_send_without_collection:
+ _warned_send_without_collection = True
+ logger.error(
+ "telemetry.shared_metrics.send is true but "
+ "telemetry.shared_metrics.enabled is false — nothing is "
+ "collected, so nothing can be sent. Enable collection or "
+ "turn sending off."
+ )
+ return SendConfig(enabled=False, send=False, endpoint=DEFAULT_ENDPOINT)
+
+ endpoint = shared.get("endpoint")
+ if not isinstance(endpoint, str) or not endpoint.strip():
+ endpoint = DEFAULT_ENDPOINT
+ endpoint = endpoint.strip()
+
+ if send_requested and not _endpoint_is_safe(endpoint):
+ logger.error(
+ "Refusing to send shared metrics to %r: telemetry must use https "
+ "(or a localhost http endpoint for testing).",
+ endpoint,
+ )
+ return SendConfig(enabled=enabled, send=False, endpoint=endpoint)
+
+ return SendConfig(enabled=enabled, send=send_requested, endpoint=endpoint)
+
+
+def reset_warning_latch_for_tests() -> None:
+ """Clear the once-per-process error latch (test support only)."""
+ global _warned_send_without_collection
+ _warned_send_without_collection = False
diff --git a/hermes_cli/observability/shared_metrics_sender.py b/hermes_cli/observability/shared_metrics_sender.py
new file mode 100644
index 0000000000..9418353c9b
--- /dev/null
+++ b/hermes_cli/observability/shared_metrics_sender.py
@@ -0,0 +1,792 @@
+"""Transmit exported shared-metrics packages to the Nous telemetry service.
+
+Implements the sender side of the ingest contract (see the telemetry repo's
+``CONTRACT.md``):
+
+* ``202`` — durably stored. Mark sent.
+* ``400`` — permanently malformed. Never retry.
+* ``429`` — keep, retry after ``Retry-After``.
+* ``5xx`` / timeout / connection error — keep, retry with backoff.
+
+Two properties are load-bearing and easy to get wrong:
+
+**The outbox directory is the user's local history, not a queue.** Packages
+are pruned by age; a ``202`` marks send state in SQLite and never deletes a
+file. See Appendix A.7 of ``docs/observability/relay-shared-metrics.md``.
+
+**Consent is gated on the package's PERIOD, not its creation time.** One
+period is split across packages created on different days, so a created-at
+gate would send a period's tail while dropping its head and silently
+undercount the first consented day. The gate itself is interval containment:
+the period must fall entirely inside a recorded consent window
+(``send_consent_windows``), maintained by the single ``reconcile_send_consent``
+writer below.
+"""
+
+from __future__ import annotations
+
+import gzip
+import json
+import logging
+import random
+import sqlite3
+import time
+import urllib.error
+import urllib.request
+import uuid
+from dataclasses import dataclass
+from datetime import datetime, timedelta, timezone
+
+from hermes_cli.sqlite_util import write_txn
+
+logger = logging.getLogger(__name__)
+
+#: Contract recommends timing out at 30s and treating a timeout as retryable.
+REQUEST_TIMEOUT_SECONDS = 30
+
+#: In-process attempts per package per pass, then the package waits for a
+#: later pass. Backoff is 1s/5s/25s with full jitter.
+MAX_ATTEMPTS = 3
+_BACKOFF_BASE_SECONDS = 1
+_BACKOFF_FACTOR = 5
+
+#: Contract recommends gzip above roughly this size.
+GZIP_THRESHOLD_BYTES = 4096
+
+#: Packages per pass. Bounds work on an interactive hook even after an outage.
+MAX_PACKAGES_PER_PASS = 20
+
+#: How long a claimed row is held. The claim writes a LEASE INTO THE FUTURE:
+#: selection requires `next_attempt_at <= now`, so for the length of the lease
+#: no other process can take the package.
+#:
+#: This must exceed the worst case for ONE package — three 30s request
+#: timeouts plus 1s+5s of backoff, about 96s — which is why packages are
+#: claimed one at a time, immediately before being sent. An earlier revision
+#: claimed up to 20 rows under a single shared lease; a full batch can legally
+#: run ~1900s, so the later rows' leases expired while the pass still held
+#: them in memory and another process re-sent them.
+_CLAIM_LEASE_SECONDS = 300
+
+#: Floor applied after a pass fails to deliver, so a hard-down service is not
+#: retried on every task completion.
+_FAILURE_BACKOFF_SECONDS = 15 * 60
+
+#: Statuses that are permanent per the ingest contract. Deliberately narrow:
+#: 400 means the envelope is malformed and will never validate. 413 is added
+#: because a package over the service's 1 MiB cap cannot shrink on retry.
+#: Everything else — including 403 from the origin guard and 404 from a bad
+#: path — is retried, because those are usually deployment or edge
+#: misconfiguration that resolves without the package changing.
+_PERMANENT_STATUSES = frozenset({400, 413})
+
+#: Attempts after which a package is abandoned. Without a ceiling a
+#: permanently-poisoned row is retried until 30-day retention deletes it —
+#: measured at ~160 requests — which wastes the user's bandwidth and keeps a
+#: doomed package at the head of the queue.
+MAX_SEND_ATTEMPTS = 25
+
+
+def _utc_now() -> datetime:
+ return datetime.now(timezone.utc)
+
+
+def _isoformat(value: datetime) -> str:
+ return value.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
+
+
+def _parse_stamp(value: str) -> datetime:
+ """Parse a stamp this module itself wrote (Z-suffixed ISO-8601, UTC)."""
+ return datetime.fromisoformat(value.replace("Z", "+00:00")).astimezone(
+ timezone.utc
+ )
+
+
+@dataclass
+class SendOutcome:
+ """What one pass did. Returned for tests and diagnostics."""
+
+ sent: int = 0
+ rejected: int = 0
+ deferred: int = 0
+
+
+class _Response:
+ __slots__ = ("status", "retry_after", "body")
+
+ def __init__(self, status: int, retry_after: str | None, body: str) -> None:
+ self.status = status
+ self.retry_after = retry_after
+ self.body = body
+
+
+def _post(endpoint: str, payload: bytes, *, timeout: int) -> _Response:
+ """POST one package. Raises on transport failure; never on HTTP status."""
+ headers = {
+ "Content-Type": "application/json",
+ "User-Agent": "hermes-agent-shared-metrics/1",
+ }
+ body = payload
+ if len(payload) > GZIP_THRESHOLD_BYTES:
+ # mtime=0: gzip embeds a timestamp by default, which would make two
+ # sends of one package differ on the wire. The service decompresses
+ # before storing so it would not change what lands in S3, but a
+ # deterministic body keeps "a resend is byte-identical" true at the
+ # transport layer too, and makes the property testable.
+ body = gzip.compress(payload, mtime=0)
+ headers["Content-Encoding"] = "gzip"
+
+ request = urllib.request.Request(
+ endpoint, data=body, headers=headers, method="POST"
+ )
+ try:
+ with urllib.request.urlopen(request, timeout=timeout) as response:
+ return _Response(
+ response.status,
+ response.headers.get("Retry-After"),
+ response.read(2048).decode("utf-8", "replace"),
+ )
+ except urllib.error.HTTPError as exc:
+ # An HTTP error status is a normal contract outcome, not a failure.
+ return _Response(
+ exc.code,
+ exc.headers.get("Retry-After") if exc.headers else None,
+ exc.read(2048).decode("utf-8", "replace") if exc.fp else "",
+ )
+
+
+def _retry_after_seconds(value: str | None, default: int) -> int:
+ if not value:
+ return default
+ try:
+ # Contract sends seconds. Clamp so a hostile or bogus value cannot
+ # park a package for years, and never go below one second.
+ return max(1, min(int(float(value)), 86_400))
+ except (TypeError, ValueError):
+ return default
+
+
+#: Maximum distance one reconcile call can advance the 'obs' mark. Honest
+#: heartbeats arrive hours apart at most, so the cap never binds in normal
+#: operation; a machine legitimately off for months catches up in a few
+#: hook fires (fail-closed latency only). What it bounds is FORWARD clock
+#: poison: without it, a single glitched sample (NTP flap reading 2099)
+#: permanently drags the mark — and with it every window open and every
+#: confirmation horizon — decades ahead, which round 6 reproduced as a
+#: refused-data leak. Capped, one insane sample moves the mark at most
+#: this far, and real time overtakes it again.
+MAX_OBS_ADVANCE_SECONDS = 30 * 24 * 3600
+
+
+def reconcile_send_consent(
+ connection: sqlite3.Connection,
+ send_enabled: bool,
+ *,
+ now: datetime | None = None,
+) -> None:
+ """Reconcile the consent-window table with the observed config state.
+
+ THE ONLY writer of consent state. Must run inside a write transaction.
+ A pure function of (config, now, store): call it from anywhere, any
+ number of times, in any order — the resulting windows are the same. This
+ replaces the previous edge-detection design, whose three partial
+ observers (wizard, relay, mid-pass) each covered a different subset of
+ transitions and repeatedly leaked the transitions between the subsets.
+
+ Timestamp discipline (each rule is load-bearing; see the validation
+ harness in tests/hermes_cli/test_shared_metrics_consent_windows.py):
+
+ - The 'obs' mark advances to every observation stamp, monotonically —
+ but by at most ``MAX_OBS_ADVANCE_SECONDS`` per call. Unbounded, the
+ mark is monotonic in the LEAK direction: one glitched-forward sample
+ would drag ``last_confirmed_at`` decades ahead, a later close would
+ stamp that horizon, and the closed window would contain every future
+ refused period (reproduced in round 6). Bounded, a poisoned sample
+ costs at most one cap's width, and real time overtakes it.
+ An open window's ``last_confirmed_at`` follows the mark: consent is
+ asserted only for time that was actually observed.
+ - A close is stamped at ``last_confirmed_at`` — never "now" — so an
+ unobserved gap (hand-edited config, machine off for 90 days) is never
+ inside a window and fails closed.
+ - An open clamps to ``max(now, obs, data)``: a rolled-back clock cannot
+ open a window underneath refused packages already on disk, and cannot
+ make the new window adjacent to the previous close.
+ """
+ stamp = _isoformat(now or _utc_now())
+ raw_stamp = stamp # pre-cap observation time, used to clamp closes
+ previous_obs = connection.execute(
+ "SELECT stamp FROM consent_marks WHERE name = 'obs'"
+ ).fetchone()
+ if previous_obs is not None:
+ ceiling = _isoformat(
+ _parse_stamp(str(previous_obs[0]))
+ + timedelta(seconds=MAX_OBS_ADVANCE_SECONDS)
+ )
+ stamp = min(stamp, ceiling)
+ connection.execute(
+ """
+ INSERT INTO consent_marks(name, stamp) VALUES ('obs', ?)
+ ON CONFLICT(name) DO UPDATE SET stamp = MAX(stamp, excluded.stamp)
+ """,
+ (stamp,),
+ )
+ marks = dict(
+ connection.execute("SELECT name, stamp FROM consent_marks").fetchall()
+ )
+ obs = marks["obs"] # >= stamp; immune to clock rollback
+ data = marks.get("data")
+
+ open_row = connection.execute(
+ "SELECT rowid FROM send_consent_windows WHERE closed_at IS NULL"
+ ).fetchone()
+
+ if send_enabled:
+ if open_row is None:
+ opened = max(x for x in (obs, data) if x is not None)
+ connection.execute(
+ "INSERT INTO send_consent_windows(opened_at, last_confirmed_at)"
+ " VALUES (?, ?)",
+ (opened, opened),
+ )
+ else:
+ connection.execute(
+ "UPDATE send_consent_windows"
+ " SET last_confirmed_at = MAX(last_confirmed_at, ?)"
+ " WHERE rowid = ?",
+ (obs, open_row[0]),
+ )
+ elif open_row is not None:
+ # Close at the last CONFIRMED moment, but never after the closing
+ # observation's own raw stamp. The two clamps serve different
+ # adversaries and both are load-bearing:
+ # - min with last_confirmed_at: an unobserved gap (machine off,
+ # hand-edited config) is never asserted as consented (v1's leak).
+ # - min with the RAW stamp (pre-cap, pre-MAX): if last_confirmed_at
+ # was poisoned by a glitched-forward sample, an honest clock at
+ # revoke time pulls the close back to the true revoke moment, so
+ # the refused era that follows falls OUTSIDE the closed window
+ # (round 6's D1 leak). A rolled-back clock at close time only
+ # closes EARLIER — fail-closed.
+ connection.execute(
+ "UPDATE send_consent_windows"
+ " SET closed_at = MIN(last_confirmed_at, ?)"
+ " WHERE rowid = ?",
+ (raw_stamp, open_row[0]),
+ )
+
+
+#: Claim-time consent predicate: the package's period must fall entirely
+#: inside SOME recorded consent window. An open window vouches only up to its
+#: last confirmed moment, so a package whose period runs past it waits for
+#: the next reconcile heartbeat (fail-closed; released within one hook fire).
+CONSENT_GATE_SQL = """EXISTS (
+ SELECT 1 FROM send_consent_windows w
+ WHERE package_outbox.period_start >= w.opened_at
+ AND package_outbox.period_end <=
+ CASE WHEN w.closed_at IS NULL THEN w.last_confirmed_at
+ ELSE w.closed_at END
+)"""
+
+
+def _state_get(connection: sqlite3.Connection, key: str) -> str | None:
+ row = connection.execute(
+ "SELECT value FROM telemetry_state WHERE key = ?", (key,)
+ ).fetchone()
+ return str(row[0]) if row is not None else None
+
+
+def _state_set(connection: sqlite3.Connection, key: str, value: str) -> None:
+ connection.execute(
+ """
+ INSERT INTO telemetry_state(key, value) VALUES (?, ?)
+ ON CONFLICT(key) DO UPDATE SET value = excluded.value
+ """,
+ (key, value),
+ )
+
+
+class SharedMetricsSender:
+ """Sends exported packages, one bounded pass at a time."""
+
+ def __init__(
+ self,
+ store,
+ endpoint: str,
+ *,
+ post=_post,
+ sleep=time.sleep,
+ now=_utc_now,
+ max_attempts: int = MAX_ATTEMPTS,
+ consent_check=None,
+ ) -> None:
+ self._store = store
+ self._endpoint = endpoint
+ self._post = post
+ self._sleep = sleep
+ self._now = now
+ self._max_attempts = max_attempts
+ # Called before every package. None disables the check for callers
+ # that have already established consent out of band (tests, E2E).
+ self._consent_check = consent_check
+
+ # -- selection ---------------------------------------------------------
+
+ def _claim_next(self, now: datetime, seen: set[str]) -> dict | None:
+ """Claim exactly ONE package, immediately before it is sent.
+
+ Claiming a whole batch up front does not work: a single shared lease
+ has to cover the entire pass, and 20 retrying packages can legally run
+ far longer than any sane lease (three 30s timeouts plus backoff each).
+ The later rows' leases then expire while this pass still holds them in
+ memory, and another process re-sends them. Taking one row at a time
+ keeps the lease covering only the package actually in flight.
+
+ ``seen`` holds packages this pass has already finished with. They are
+ excluded IN SQL rather than by rejecting the fetched row: with
+ ``LIMIT 1``, returning None for an already-seen row would make the
+ caller believe the queue was empty and abandon every healthy package
+ behind it. A row can legitimately become eligible again mid-pass (a
+ short Retry-After, or a pass that outlives the 15-minute failure
+ backoff), so this is reachable in normal operation, not just in tests.
+ """
+ with self._store._connection() as connection:
+ with write_txn(connection):
+ stamp = _isoformat(now)
+ lease_until = now + timedelta(seconds=_CLAIM_LEASE_SECONDS)
+
+ placeholders = ",".join("?" for _ in seen)
+ exclusion = (
+ f" AND package_id NOT IN ({placeholders})" if seen else ""
+ )
+ # Consent is a READ here — the claim must never mutate the
+ # window table. The old design's opt_in_period() call at this
+ # exact spot meant selecting a row could rewrite what was
+ # permitted to be sent (and did, under a rolled-back clock).
+ row = connection.execute(
+ f"""
+ SELECT package_id, payload_json, sent_install_id
+ FROM package_outbox
+ WHERE exported_at IS NOT NULL
+ AND (send_state IS NULL OR send_state = 'pending')
+ AND (next_attempt_at IS NULL OR next_attempt_at <= ?)
+ AND {CONSENT_GATE_SQL}
+ AND send_attempts < ?
+ {exclusion}
+ ORDER BY created_at, package_id
+ LIMIT 1
+ """,
+ (stamp, MAX_SEND_ATTEMPTS, *sorted(seen)),
+ ).fetchone()
+ if row is None:
+ return None
+
+ package_id = str(row[0])
+ derived = row[2]
+ if not derived:
+ derived = self._freeze_identity(
+ connection, package_id, row[1], now
+ )
+ if derived is None:
+ # Unusable row, already marked rejected. Signal the
+ # caller to continue rather than stop.
+ return {"package_id": package_id, "skip": True}
+
+ token = str(uuid.uuid4())
+ connection.execute(
+ """
+ UPDATE package_outbox
+ SET send_state = 'pending',
+ send_attempts = send_attempts + 1,
+ next_attempt_at = ?,
+ claim_token = ?
+ WHERE package_id = ?
+ """,
+ # Lease INTO THE FUTURE: selection requires
+ # next_attempt_at <= now, so no other process can take
+ # this row while it is in flight. Success or a real
+ # backoff overwrites it; if this process dies, it expires.
+ # The token is this claim's identity: a reclaim after
+ # expiry mints a new one, and every later write by THIS
+ # claimant is compare-and-set against it, so a lapsed
+ # claimant that resumes cannot settle or transmit.
+ (_isoformat(lease_until), token, package_id),
+ )
+ return {
+ "package_id": package_id,
+ "payload_json": str(row[1]),
+ "derived": str(derived),
+ "claim_token": token,
+ "skip": False,
+ }
+
+ def _freeze_identity(
+ self,
+ connection: sqlite3.Connection,
+ package_id: str,
+ payload_json,
+ now: datetime,
+ ) -> str | None:
+ """Record the transmitted id on the row, or reject an unusable one.
+
+ The stable install_id is transmitted as-is (product decision,
+ 2026-08-27 — see the doc's A.2). What remains of "freezing" is the
+ validation and the audit column: ``sent_install_id`` records exactly
+ what the wire will carry, and rejecting unusable rows here rather
+ than raising matters because an exception rolls back the claim
+ transaction and blocks every healthy package behind this one.
+ """
+ reason = None
+ install_id = None
+ try:
+ payload = json.loads(payload_json)
+ except (TypeError, ValueError):
+ reason = "unreadable payload"
+ else:
+ # Valid JSON is not enough: a top-level array, string, number or
+ # null parses cleanly and then has no .get().
+ if not isinstance(payload, dict):
+ reason = f"payload is {type(payload).__name__}, expected object"
+ else:
+ install_id = payload.get("install_id")
+ if not isinstance(install_id, str) or not install_id.strip():
+ reason = "payload has no usable install_id"
+
+ if reason is not None:
+ logger.warning(
+ "Shared-metrics package %s cannot be sent (%s)", package_id, reason
+ )
+ connection.execute(
+ """
+ UPDATE package_outbox
+ SET send_state = 'rejected', last_error = ?
+ WHERE package_id = ?
+ """,
+ (reason, package_id),
+ )
+ return None
+
+ connection.execute(
+ "UPDATE package_outbox SET sent_install_id = ? WHERE package_id = ?",
+ (install_id, package_id),
+ )
+ return str(install_id)
+
+ # -- transmission ------------------------------------------------------
+
+ def _body(self, payload_json: str, transmitted_id: str) -> bytes:
+ """Rebuild the exact bytes to send.
+
+ The payload is recomputed from the stored package rather than kept as
+ a second copy: json.dumps with these options is deterministic. The
+ install_id is written from the frozen ``sent_install_id`` column
+ rather than trusted implicitly, keeping "a resend is byte-identical"
+ anchored to one recorded value.
+ """
+ payload = json.loads(payload_json)
+ payload = dict(payload)
+ payload["install_id"] = transmitted_id
+ return json.dumps(payload, indent=2, sort_keys=True).encode("utf-8")
+
+ def _mark(
+ self,
+ package_id: str,
+ *,
+ only_if_pending: bool = True,
+ token: str | None = None,
+ **columns,
+ ) -> None:
+ """Write send state for one package.
+
+ Guarded on send_state so a pass whose lease lapsed cannot resurrect a
+ row another process has already finished: without this, a slow sender
+ could overwrite 'sent' back to 'pending' and cause a re-send.
+
+ When ``token`` is given, the write is additionally compare-and-set on
+ claim_token: it lands only if THIS claim is still the current one. A
+ claimant that lapsed and was superseded writes zero rows — its
+ settlement, backoff, and error strings all silently lose to the
+ newer claim's, which is the correct outcome.
+ """
+ assignments = ", ".join(f"{name} = ?" for name in columns)
+ predicate = (
+ " AND (send_state IS NULL OR send_state = 'pending')"
+ if only_if_pending
+ else ""
+ )
+ params: list = [*columns.values(), package_id]
+ if token is not None:
+ predicate += " AND claim_token = ?"
+ params.append(token)
+ with self._store._connection() as connection:
+ with write_txn(connection):
+ connection.execute(
+ f"UPDATE package_outbox SET {assignments} "
+ f"WHERE package_id = ?{predicate}",
+ params,
+ )
+
+ def _renew_claim(self, package_id: str, token: str | None) -> bool:
+ """Atomically re-assert ownership and extend the lease. CAS, one row.
+
+ A read-only ownership check is not enough: a claimant whose lease
+ expired while suspended can pass the check (its token is still in
+ the row if no one reclaimed yet) and then POST while another process
+ legitimately reclaims — the check-to-POST expiry race a seventh
+ review reproduced. Renewal closes it by requiring, in ONE statement:
+
+ - the token still matches (nobody reclaimed), AND
+ - the current lease is UNEXPIRED (this claimant is not stale), AND
+ - the row is still pending,
+
+ and only then pushing next_attempt_at a fresh lease into the future,
+ so the upcoming POST (30s timeout, well under the 300s lease) runs
+ entirely inside renewed authority. rowcount == 1 is the only grant.
+ A claimant that wakes past its own lease fails the unexpired
+ condition and yields even though its token was never replaced.
+ """
+ if token is None:
+ return False
+ try:
+ now = self._now()
+ lease_until = now + timedelta(seconds=_CLAIM_LEASE_SECONDS)
+ with self._store._connection() as connection:
+ with write_txn(connection):
+ cursor = connection.execute(
+ """
+ UPDATE package_outbox
+ SET next_attempt_at = ?
+ WHERE package_id = ?
+ AND claim_token = ?
+ AND (send_state IS NULL OR send_state = 'pending')
+ AND next_attempt_at > ?
+ """,
+ (
+ _isoformat(lease_until),
+ package_id,
+ token,
+ _isoformat(now),
+ ),
+ )
+ return cursor.rowcount == 1
+ except Exception:
+ # If renewal itself fails, do not transmit on unproven authority.
+ logger.warning(
+ "Unable to renew shared-metrics claim", exc_info=True
+ )
+ return False
+
+ def _defer(
+ self,
+ package_id: str,
+ delay_seconds: int,
+ reason: str,
+ *,
+ token: str | None = None,
+ ) -> None:
+ # Defence in depth: no current caller can pass a non-positive delay
+ # (Retry-After is already clamped to [1, 86400] when parsed, and every
+ # other call site passes a positive constant), so this clamp is
+ # deliberately unreachable today and no test can distinguish it. It
+ # stays because a past deadline would make the row instantly
+ # re-eligible and let a pass spin on it — a cheap guard against a
+ # future caller that forgets.
+ delay = max(1, int(delay_seconds))
+ retry_at = self._now().timestamp() + delay
+ self._mark(
+ package_id,
+ token=token,
+ send_state="pending",
+ next_attempt_at=_isoformat(
+ datetime.fromtimestamp(retry_at, tz=timezone.utc)
+ ),
+ last_error=reason[:500],
+ )
+
+ def _send_one(self, package: dict) -> str:
+ """Try one package. Returns 'sent', 'rejected', or 'deferred'.
+
+ Delivery is at-least-once. The pre-POST ownership check plus the
+ token-fenced writes close the claim->POST and settle-after-reclaim
+ gaps, but a suspension landing MID-POST (bytes already on the wire
+ when the machine sleeps) can still duplicate: no client-side check
+ can revoke a request in flight. The body is byte-identical across
+ retries by construction, so the residual duplicate is exactly one
+ redundant copy of identical content; collapsing it fully would need
+ package_id-keyed dedupe at the ingest service.
+ """
+ package_id = package["package_id"]
+ token = package.get("claim_token")
+ body = self._body(package["payload_json"], package["derived"])
+
+ for attempt in range(1, self._max_attempts + 1):
+ # Atomically renew the claim before EVERY external POST. The
+ # renewal is compare-and-set on (token, pending, lease unexpired)
+ # and extends the lease past the request, so a suspended-then-
+ # resumed claimant whose lease lapsed yields here even if nobody
+ # has reclaimed yet — a read-only ownership check passed in that
+ # state and still double-sent (check-to-POST expiry race). The
+ # ingest key is minute-prefixed, so duplicates become distinct
+ # stored objects, not overwrites.
+ if not self._renew_claim(package_id, token):
+ logger.info(
+ "Shared-metrics claim on %s superseded or expired; yielding",
+ package_id,
+ )
+ return "deferred"
+ try:
+ response = self._post(
+ self._endpoint, body, timeout=REQUEST_TIMEOUT_SECONDS
+ )
+ except Exception as exc: # transport failure: offline, DNS, TLS
+ reason = f"{type(exc).__name__}: {exc}"
+ if attempt >= self._max_attempts:
+ self._defer(
+ package_id, _FAILURE_BACKOFF_SECONDS, reason, token=token
+ )
+ return "deferred"
+ self._sleep(self._backoff(attempt))
+ continue
+
+ if response.status == 202:
+ self._mark(
+ package_id,
+ token=token,
+ send_state="sent",
+ sent_at=_isoformat(self._now()),
+ last_error=None,
+ )
+ return "sent"
+
+ if response.status in _PERMANENT_STATUSES:
+ # Only statuses the contract (or the envelope schema) makes
+ # terminal. Everything else retries: 403 in particular is the
+ # ingest service's origin guard, which returns 403 during an
+ # edge/Transform-Rule misconfiguration — treating that as
+ # permanent would discard every package sent during the
+ # incident instead of retrying after recovery.
+ logger.warning(
+ "Telemetry package %s rejected with HTTP %s; not retrying",
+ package_id,
+ response.status,
+ )
+ self._mark(
+ package_id,
+ token=token,
+ send_state="rejected",
+ last_error=f"HTTP {response.status}: {response.body[:400]}",
+ )
+ return "rejected"
+
+ if response.status == 429:
+ self._defer(
+ package_id,
+ _retry_after_seconds(response.retry_after, _FAILURE_BACKOFF_SECONDS),
+ "rate limited",
+ token=token,
+ )
+ return "deferred"
+
+ # 5xx and anything unexpected: retryable.
+ reason = f"HTTP {response.status}"
+ if attempt >= self._max_attempts:
+ self._defer(
+ package_id, _FAILURE_BACKOFF_SECONDS, reason, token=token
+ )
+ return "deferred"
+ self._sleep(self._backoff(attempt))
+
+ self._defer(
+ package_id, _FAILURE_BACKOFF_SECONDS, "attempts exhausted", token=token
+ )
+ return "deferred"
+
+ @staticmethod
+ def _backoff(attempt: int) -> float:
+ """1s, 5s, 25s with full jitter."""
+ ceiling = _BACKOFF_BASE_SECONDS * (_BACKOFF_FACTOR ** (attempt - 1))
+ return random.uniform(0, ceiling)
+
+ # -- entry point -------------------------------------------------------
+
+ def send_pending(self) -> SendOutcome:
+ """Run one bounded pass. Never raises.
+
+ Claims and sends ONE package at a time so each row's lease only has to
+ cover its own transmission, and re-checks consent before every send so
+ revoking `send` mid-pass stops the remaining packages.
+ """
+ outcome = SendOutcome()
+ seen: set[str] = set()
+
+ for _ in range(MAX_PACKAGES_PER_PASS):
+ if not self._still_consented():
+ # The user turned sending off while this pass was running.
+ # Stop without transmitting anything further, and reconcile
+ # so the window closes at its last confirmed moment. This is
+ # the same single writer every other observation point uses —
+ # not a separate recording mechanism.
+ logger.info("Shared-metrics sending disabled mid-pass; stopping")
+ self._reconcile(send_enabled=False)
+ break
+ try:
+ package = self._claim_next(self._now(), seen)
+ except Exception:
+ logger.warning(
+ "Unable to select shared-metrics packages", exc_info=True
+ )
+ break
+ if package is None:
+ break
+
+ seen.add(package["package_id"])
+ if package.get("skip"):
+ # Unusable row already marked rejected during the claim.
+ outcome.rejected += 1
+ continue
+
+ try:
+ result = self._send_one(package)
+ except Exception:
+ logger.warning("Unable to send shared-metrics package", exc_info=True)
+ outcome.deferred += 1
+ continue
+ if result == "sent":
+ outcome.sent += 1
+ elif result == "rejected":
+ outcome.rejected += 1
+ else:
+ outcome.deferred += 1
+ return outcome
+
+ def _reconcile(self, *, send_enabled: bool) -> None:
+ """Run the single consent writer from within a pass."""
+ try:
+ with self._store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(
+ connection, send_enabled, now=self._now()
+ )
+ except Exception:
+ logger.warning(
+ "Unable to reconcile shared-metrics consent", exc_info=True
+ )
+
+ def _still_consented(self) -> bool:
+ """Re-read profile-owned send consent.
+
+ Consent is a boundary, not cached configuration: the documentation
+ promises that setting `send: false` stops transmission immediately,
+ and a pass can run for minutes. Injected senders (tests, the staging
+ E2E) opt out by passing consent_check=None.
+ """
+ if self._consent_check is None:
+ return True
+ try:
+ return bool(self._consent_check())
+ except Exception:
+ # Fail CLOSED: if consent cannot be established, do not transmit.
+ logger.warning(
+ "Unable to confirm shared-metrics send consent; stopping",
+ exc_info=True,
+ )
+ return False
diff --git a/hermes_cli/platform_actions.py b/hermes_cli/platform_actions.py
index 52e0b68fce..de52a591e7 100644
--- a/hermes_cli/platform_actions.py
+++ b/hermes_cli/platform_actions.py
@@ -99,7 +99,36 @@ class PlatformActions:
platform_enum = Platform(str(platform).strip().lower())
except Exception:
return None, _err("unknown_platform", f"unknown platform {platform!r}")
- adapter = getattr(runner, "adapters", {}).get(platform_enum)
+ # Multiplex/Team-Gateway: a secondary profile's adapters live in
+ # runner._profile_adapters[profile], not runner.adapters (the default
+ # profile's registry) — every other adapter-resolution path in this
+ # codebase (_authorization_adapter, plugin message-injection) goes
+ # through this same profile-aware, fail-closed lookup so a plugin
+ # scoped to one profile can never act through another profile's bot
+ # identity. Falls back to the bare default-profile lookup only when
+ # the gateway runner predates this method (defensive, not expected).
+ resolve_fn = getattr(runner, "_authorization_adapter", None)
+ if callable(resolve_fn):
+ try:
+ from hermes_cli.profiles import get_active_profile_name
+
+ profile_name = get_active_profile_name()
+ except Exception:
+ # Fail closed: an unresolvable profile must not degrade to the
+ # default profile's bot (the same rule _authorization_adapter
+ # applies to a stamped profile with no registry entry).
+ logger.debug(
+ "platform_actions: profile resolution failed for %s",
+ self._plugin_id, exc_info=True,
+ )
+ return None, _err(
+ "adapter_not_registered",
+ f"no {platform_enum.value} adapter is registered "
+ "(active profile could not be resolved)",
+ )
+ adapter = resolve_fn(platform_enum, profile_name)
+ else:
+ adapter = getattr(runner, "adapters", {}).get(platform_enum)
if adapter is None:
return None, _err(
"adapter_not_registered",
diff --git a/hermes_cli/plugins.py b/hermes_cli/plugins.py
index 9028e28530..3188526912 100644
--- a/hermes_cli/plugins.py
+++ b/hermes_cli/plugins.py
@@ -4262,18 +4262,21 @@ class PluginManager:
# first process sees plugin backends (tracking #64177).
self._refresh_secret_sources_after_discovery()
if force:
- # config.yaml shell hooks live in ``_hooks`` but are
- # config-owned, not plugin-owned — the ledger-driven
- # unload() above wiped them and cannot restore them.
- # Re-register so force-reload is symmetric (#60036;
- # tracking #64178 — salvaged from PR #64188).
- self._re_register_shell_hooks_after_force()
+ # config.yaml shell hooks and outbound webhooks live in
+ # ``_hooks`` but are config-owned, not plugin-owned —
+ # the ledger-driven unload() above wiped them and
+ # cannot restore them. Re-register so force-reload is
+ # symmetric (#60036; tracking #64178 — salvaged from
+ # PR #64188; outbound webhooks added per #92682 review).
+ self._re_register_config_hooks_after_force()
except BaseException:
self._discovered = False
raise
- def _re_register_shell_hooks_after_force(self) -> None:
- """Restore config.yaml shell hooks wiped by force-clear of ``_hooks``."""
+ def _re_register_config_hooks_after_force(self) -> None:
+ """Restore config.yaml shell hooks/outbound webhooks wiped by
+ force-clear of ``_hooks``. Each re-register call is independently
+ guarded so one failing does not skip the other."""
try:
from agent.shell_hooks import re_register_config_hooks
@@ -4281,6 +4284,14 @@ class PluginManager:
except Exception as exc:
# Import cycle / missing module must not abort force reload.
logger.debug("force-reload shell-hook re-register skipped: %s", exc)
+ try:
+ from agent.outbound_webhooks import (
+ re_register_config_hooks as re_register_outbound_webhooks,
+ )
+
+ re_register_outbound_webhooks()
+ except Exception as exc:
+ logger.debug("force-reload outbound-webhook re-register skipped: %s", exc)
def _refresh_secret_sources_after_discovery(self) -> None:
"""If any plugin secret source is enabled, reset cache and re-apply.
diff --git a/hermes_cli/profiles.py b/hermes_cli/profiles.py
index 389cc7933b..39ad55555c 100644
--- a/hermes_cli/profiles.py
+++ b/hermes_cli/profiles.py
@@ -838,6 +838,22 @@ def _check_gateway_running(profile_dir: Path) -> bool:
return False
+def _served_by_running_multiplexer(profile_name: str) -> bool:
+ """True when the live default gateway multiplexes ``profile_name``.
+
+ A served named profile has no gateway.pid of its own, so
+ ``_check_gateway_running`` alone reports it stopped while the default
+ multiplexer is actually its inbound process. Single shared lookup with the
+ named-profile start guard and cron liveness (#97120).
+ """
+ try:
+ from hermes_cli.gateway import named_profile_served_by_running_multiplexer
+
+ return named_profile_served_by_running_multiplexer(profile_name)
+ except Exception:
+ return False
+
+
# In-process cache for skill counts. Walking ``skills_dir.rglob("SKILL.md")``
# recurses the entire skill tree (each skill carries references/scripts/assets
# sub-trees); the default profile alone has ~270 skills, and ``list_profiles``
@@ -1084,7 +1100,10 @@ def list_profiles() -> List[ProfileInfo]:
name=name,
path=entry,
is_default=False,
- gateway_running=_check_gateway_running(entry),
+ gateway_running=(
+ _check_gateway_running(entry)
+ or _served_by_running_multiplexer(name)
+ ),
model=model,
provider=provider,
has_env=(entry / ".env").exists(),
@@ -1261,6 +1280,18 @@ def create_profile(
# Strip runtime files
for stale in _CLONE_ALL_STRIP:
(profile_dir / stale).unlink(missing_ok=True)
+ # A clone-all copies auth.json and .anthropic_oauth.json verbatim.
+ # Single-use OAuth grants (Anthropic / Codex / xAI) forked that way
+ # are one credential with two owners: the first profile to refresh
+ # revokes the pair for every sibling (#100339). Drop the copies; the
+ # clone reads the root grant through the credential-pool fallback.
+ from hermes_cli.auth import strip_cloned_single_use_oauth_grants
+ stripped = strip_cloned_single_use_oauth_grants(profile_dir)
+ if any(stripped.values()):
+ logger.info(
+ "profile %s: dropped cloned single-use OAuth grants %s "
+ "(inherits the root grant instead)", canon, stripped,
+ )
else:
# Bootstrap directory structure
profile_dir.mkdir(parents=True, exist_ok=True)
@@ -2008,6 +2039,19 @@ def _stop_gateway_process(profile_dir: Path) -> None:
raw = pid_file.read_text(encoding="utf-8").strip()
data = json.loads(raw) if raw.startswith("{") else {"pid": int(raw)}
pid = int(data["pid"])
+ # Cross-profile kill refusal (#89315): the record's hermes_home stamp
+ # names the gateway's TRUE owner. A contaminated/poisoned gateway.pid
+ # inside this profile dir can point at another profile's live gateway
+ # — killing it starts the mutual SIGTERM restart loop from the issue.
+ from gateway.status import recorded_gateway_home_conflicts
+
+ if recorded_gateway_home_conflicts(data, expected_home=profile_dir):
+ print(
+ f"✗ Refusing to stop PID {pid}: its recorded HERMES_HOME "
+ f"belongs to a different profile than {profile_dir} "
+ "(stale/poisoned PID record, #89315)."
+ )
+ return
# Route through terminate_pid so Windows uses the appropriate
# primitive (taskkill / TerminateProcess) — raw os.kill with
# _signal.SIGKILL raises AttributeError at import time on Windows,
diff --git a/hermes_cli/providers.py b/hermes_cli/providers.py
index 87e2136a96..f3ecd99d22 100644
--- a/hermes_cli/providers.py
+++ b/hermes_cli/providers.py
@@ -1023,6 +1023,29 @@ def resolve_provider_full(
if custom_pdef is not None:
return custom_pdef
+ # 2c. Managed local runtime: the llamacpp aliases are a real provider
+ # whenever the managed server (or a detected external one) resolves —
+ # no credential and no providers: entry required, the credential is
+ # reachability. Without this rung the model-switch path rejected the
+ # very provider the Local Models 'Use' flow writes to config
+ # ("Unknown provider 'llamacpp'" from the desktop dropdown).
+ if raw in ("llamacpp", "llama.cpp", "llama-cpp"):
+ try:
+ from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint
+
+ endpoint = resolve_llamacpp_endpoint(wait_for_boot_s=0)
+ except Exception:
+ endpoint = None
+ if endpoint:
+ return ProviderDef(
+ id="llamacpp",
+ name="Local",
+ transport="openai_chat",
+ api_key_env_vars=(),
+ base_url=endpoint["base_url"],
+ source="local-runtime",
+ )
+
# 3. Try models.dev directly (for providers not in our ALIASES)
try:
from agent.models_dev import get_provider_info as _mdev_provider
diff --git a/hermes_cli/runtime_provider.py b/hermes_cli/runtime_provider.py
index 7a94fa8b57..605d2d7135 100644
--- a/hermes_cli/runtime_provider.py
+++ b/hermes_cli/runtime_provider.py
@@ -1228,6 +1228,53 @@ def _resolve_named_custom_runtime(
# `provider: ollama` with a LAN/WireGuard `base_url` doesn't silently
# fall through to OpenRouter.
requested_norm = (requested_provider or "").strip().lower()
+
+ # Managed llama.cpp runtime: a llamacpp-flavored alias with no explicit
+ # base_url resolves to the supervised server (or a detected external
+ # one) before the generic custom fallthrough. Explicit base_url always
+ # wins — a user pointing at a specific server means that server.
+ if requested_norm in ("llamacpp", "llama.cpp", "llama-cpp") and not explicit_base_url:
+ try:
+ from hermes_cli.local_runtime.endpoint import resolve_llamacpp_endpoint
+
+ endpoint = resolve_llamacpp_endpoint()
+ except Exception: # noqa: BLE001 — resolution is best-effort
+ endpoint = None
+ if endpoint:
+ return {
+ "provider": "custom",
+ "api_mode": "chat_completions",
+ "base_url": endpoint["base_url"],
+ "api_key": (explicit_api_key or "").strip()
+ or endpoint["api_key"] or "no-key-required",
+ "source": "local-runtime",
+ "requested_provider": requested_provider,
+ }
+ # No server to serve this model. Say so and stop — falling through
+ # to the generic custom path sends the request to whatever provider
+ # picks it up (OpenRouter with a placeholder key), and the user's
+ # "local server is off" surfaces as that provider's baffling
+ # "401 Invalid API key". The switch's own state picks the message:
+ # the user who turned the server off gets pointed at the switch,
+ # anyone else at the setup pane.
+ try:
+ from hermes_cli.config import load_config as _load_cfg
+
+ _lr_enabled = bool((_load_cfg().get("local_runtime") or {}).get("enabled"))
+ except Exception: # noqa: BLE001
+ _lr_enabled = False
+ if _lr_enabled:
+ raise ValueError(
+ "The local model server isn't running. It may still be "
+ "starting — try again in a moment, or check Settings → "
+ "Providers → Local models."
+ )
+ raise ValueError(
+ "The local model server is turned off. Turn it back on in "
+ "Settings → Providers → Local models, or switch to another "
+ "model."
+ )
+
if requested_norm and requested_norm != "custom":
try:
from hermes_cli.auth import resolve_provider as _resolve_provider
diff --git a/hermes_cli/session_lost_and_found.py b/hermes_cli/session_lost_and_found.py
index 90d8acba9a..b1fadcafe7 100644
--- a/hermes_cli/session_lost_and_found.py
+++ b/hermes_cli/session_lost_and_found.py
@@ -325,20 +325,25 @@ def _copy_direct_tables(
) -> dict[str, int]:
"""Copy rows .recover managed to attribute to real canonical tables."""
+ # Lazy import: session_recovery imports this module inside a function, so
+ # a module-level import here would be circular.
+ from hermes_cli.session_recovery import (
+ _AUXILIARY_TABLE_SCHEMAS,
+ _AUXILIARY_TABLES,
+ _CANONICAL_TABLES,
+ )
+
copied: dict[str, int] = {}
- for table in (
- "system_prompts",
- "sessions",
- "messages",
- "session_model_usage",
- "compression_locks",
- "gateway_routing",
- "async_delegations",
- ):
+ for table in (*_CANONICAL_TABLES, *_AUXILIARY_TABLES):
source_columns = _table_columns(lf_conn, table)
if not source_columns:
continue
dest_columns = _table_columns(dest, table)
+ if not dest_columns and table in _AUXILIARY_TABLE_SCHEMAS:
+ # Lazily-created gateway table: base SessionDB never made it on
+ # the fresh destination, so create it before copying.
+ _AUXILIARY_TABLE_SCHEMAS[table](dest)
+ dest_columns = _table_columns(dest, table)
columns = [c for c in dest_columns if c in source_columns]
if not columns:
continue
diff --git a/hermes_cli/session_recovery.py b/hermes_cli/session_recovery.py
index 6d5f5e8bf8..9a376550ad 100644
--- a/hermes_cli/session_recovery.py
+++ b/hermes_cli/session_recovery.py
@@ -45,6 +45,34 @@ _TOPIC_TABLES = (
"telegram_dm_topic_bindings",
)
+
+
+def _init_delivery_ledger_schema(conn: sqlite3.Connection) -> None:
+ from gateway.delivery_ledger import _initialize_schema
+
+ _initialize_schema(conn)
+
+
+# Tables that live in state.db but are created lazily by a gateway module on
+# first use, so base ``SessionDB`` never creates them on a fresh destination.
+# Every entry maps the table to the initializer that owns its DDL; recovery
+# creates the table on the destination before copying, so owed rows survive
+# instead of silently vanishing from a "complete" salvage (#100313, #86236).
+# Add new lazily-created state.db tables HERE, never as one-off ``if table ==``
+# branches.
+_AUXILIARY_TABLE_SCHEMAS: dict[str, Callable[[sqlite3.Connection], None]] = {
+ "delivery_obligations": _init_delivery_ledger_schema,
+}
+
+_AUXILIARY_TABLES = tuple(_AUXILIARY_TABLE_SCHEMAS)
+
+_INVENTORY_TABLES = (
+ *_CANONICAL_TABLES,
+ "state_meta",
+ *_TOPIC_TABLES,
+ *_AUXILIARY_TABLES,
+)
+
# These values describe derived indexes or the schema that owns an optional
# table. A fresh destination must generate them from its own current schema.
_GENERATED_META_KEYS = frozenset({
@@ -305,7 +333,7 @@ def _inspect_connection(conn: sqlite3.Connection) -> dict[str, Any]:
# A damaged journal pragma must not block rows that are still readable.
report["warnings"].append(f"journal mode: {exc}")
- for table in (*_CANONICAL_TABLES, "state_meta", *_TOPIC_TABLES):
+ for table in _INVENTORY_TABLES:
report["tables"][table] = _table_inventory(conn, table)
for required in ("sessions", "messages"):
@@ -385,6 +413,25 @@ def inspect_session_database(
temp_dir.cleanup()
+def _ensure_auxiliary_destination_schema(
+ destination: sqlite3.Connection,
+ table: str,
+) -> None:
+ """Create a lazy auxiliary table on the recovered destination.
+
+ Recovery initializes the destination through base ``SessionDB``, which
+ does not create gateway-owned tables. Copying into a missing dest table
+ would report ``missing`` / ``no compatible columns`` and drop the rows.
+ """
+
+ initialize = _AUXILIARY_TABLE_SCHEMAS.get(table)
+ if initialize is None:
+ raise SessionRecoverySafetyError(
+ f"no destination schema initializer registered for table {table!r}"
+ )
+ initialize(destination)
+
+
def _copy_table(
source: sqlite3.Connection,
destination: sqlite3.Connection,
@@ -1243,7 +1290,7 @@ def _verify_recovered_database(
)
counts: dict[str, int] = {}
- for table in (*_CANONICAL_TABLES, "state_meta", *_TOPIC_TABLES):
+ for table in _INVENTORY_TABLES:
columns = _table_columns(conn, table)
if columns:
counts[table] = int(
@@ -1251,7 +1298,7 @@ def _verify_recovered_database(
)
verification["table_counts"] = counts
- for table in ("sessions", "messages"):
+ for table in ("sessions", "messages", *_AUXILIARY_TABLES):
expected = expected_counts.get(table)
if expected is not None and counts.get(table) != expected:
message = (
@@ -1662,6 +1709,27 @@ def recover_session_database(
progress_cb=progress_cb,
source_rows=table_inspection.get("rows"),
)
+
+ for table in _AUXILIARY_TABLES:
+ table_inspection = inspection["tables"][table]
+ if not table_inspection.get("available"):
+ copy_report[table] = {
+ "status": "missing",
+ "copied_rows": 0,
+ }
+ continue
+ _ensure_auxiliary_destination_schema(destination_conn, table)
+ copy_function = (
+ _copy_table_salvage if allow_partial else _copy_table
+ )
+ copy_report[table] = copy_function(
+ source_conn,
+ destination_conn,
+ table,
+ chunk_size=chunk_size,
+ progress_cb=progress_cb,
+ source_rows=table_inspection.get("rows"),
+ )
orphan_cleanup = (
_cleanup_partial_orphans(destination_conn)
if allow_partial
@@ -1678,8 +1746,15 @@ def recover_session_database(
verification = _verify_recovered_database(
output,
expected_counts={
- table: inspection["tables"][table].get("rows")
- for table in _CANONICAL_TABLES
+ **{
+ table: inspection["tables"][table].get("rows")
+ for table in _CANONICAL_TABLES
+ },
+ **{
+ table: inspection["tables"][table].get("rows")
+ for table in _AUXILIARY_TABLES
+ if inspection["tables"].get(table, {}).get("available")
+ },
},
copy_report=copy_report,
allow_partial=allow_partial,
diff --git a/hermes_cli/setup.py b/hermes_cli/setup.py
index 390d71669a..1213eb1158 100644
--- a/hermes_cli/setup.py
+++ b/hermes_cli/setup.py
@@ -513,7 +513,7 @@ def _print_setup_summary(config: dict, hermes_home):
tool_status.append(("Vision (image analysis)", False, "run 'hermes setup' to configure"))
- # Web tools (Exa, Parallel, Firecrawl, or Keenable)
+ # Web tools (Exa, Parallel, Firecrawl, Tavily, or Keenable)
if subscription_features.web.managed_by_nous:
tool_status.append(("Web Search & Extract (Nous subscription)", True, None))
elif subscription_features.web.available:
@@ -522,7 +522,7 @@ def _print_setup_summary(config: dict, hermes_home):
label = f"Web Search & Extract ({subscription_features.web.current_provider})"
tool_status.append((label, True, None))
else:
- tool_status.append(("Web Search & Extract", False, "EXA_API_KEY, PARALLEL_API_KEY, FIRECRAWL_API_KEY/FIRECRAWL_API_URL, KEENABLE_API_KEY, or SEARXNG_URL"))
+ tool_status.append(("Web Search & Extract", False, "EXA_API_KEY, PARALLEL_API_KEY, FIRECRAWL_API_KEY/FIRECRAWL_API_URL, TAVILY_API_KEY, KEENABLE_API_KEY, or SEARXNG_URL"))
# Browser tools (local Chromium, Camofox, Browserbase, Browser Use, or Firecrawl)
browser_provider = subscription_features.browser.current_provider
@@ -2428,10 +2428,10 @@ def setup_tools(config: dict, first_install: bool = False):
def setup_telemetry(config: dict):
- """Configure the local, privacy-safe shared-metrics subscriber."""
+ """Configure the local shared-metrics subscriber and optional sending."""
print_header("Shared Metrics")
print_info("Shared metrics contain only bounded counters and histograms.")
- print_info("Packages stay under this Hermes profile and are not uploaded.")
+ print_info("Collection is local. Sending them to Nous is a separate opt-in.")
telemetry = config.get("telemetry")
if not isinstance(telemetry, dict):
@@ -2447,10 +2447,67 @@ def setup_telemetry(config: dict):
"Enable local shared metrics?",
default=current,
)
- if shared_metrics["enabled"]:
- print_success("Local shared metrics enabled.")
- else:
+ if not shared_metrics["enabled"]:
print_info("Local shared metrics disabled.")
+ # Sending cannot outlive collection: leaving send=true here would be a
+ # configuration that logs an error on every run and never transmits.
+ if shared_metrics.get("send") is True:
+ shared_metrics["send"] = False
+ print_info("Sending shared metrics disabled as well.")
+ # Turning collection off is also a withdrawal of send consent, and it
+ # has to close the window like any other. Recorded unconditionally:
+ # the send key may already be false in config while the consent window
+ # is still open, and that window must not survive to be reopened.
+ _record_send_consent_change(enabled=False)
+ return
+
+ print_success("Local shared metrics enabled.")
+ print_info("")
+ print_info("Sending uploads each daily package to the Nous telemetry")
+ print_info("service. Packages carry your profile-scoped install ID, a")
+ print_info("stable random UUID that identifies this profile across days")
+ print_info("(it contains no personal information and is reset by deleting")
+ print_info("the shared-metrics directory). Only packages whose entire")
+ print_info("collection period falls inside a recorded consent window are")
+ print_info("ever sent — data from before you opt in, or from any gap")
+ print_info("while sending was off, stays on this machine. Sending can be")
+ print_info("turned off again at any time.")
+ shared_metrics["send"] = prompt_yes_no(
+ "Send shared metrics to Nous?",
+ default=shared_metrics.get("send") is True,
+ )
+ if shared_metrics["send"]:
+ _record_send_consent_change(enabled=True)
+ print_success("Sending shared metrics enabled.")
+ else:
+ _record_send_consent_change(enabled=False)
+ print_info("Sending shared metrics disabled (collection stays local).")
+
+
+def _record_send_consent_change(*, enabled: bool) -> None:
+ """Reconcile consent windows at the moment the user decides.
+
+ Same single writer as the relay and the sender — reconciliation derives
+ the window state from the observation, so wizard, relay, and mid-pass
+ callers cannot disagree. The relay's once-per-process reconcile would
+ catch this on the next hook fire anyway; running it here just makes the
+ wizard's effect immediate.
+ """
+ try:
+ from hermes_cli.observability.shared_metrics import SharedMetricsStore
+ from hermes_cli.observability.shared_metrics_sender import (
+ reconcile_send_consent,
+ )
+ from hermes_cli.sqlite_util import write_txn
+
+ store = SharedMetricsStore()
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, enabled)
+ except Exception:
+ # Never block the wizard on telemetry bookkeeping. The relay runs the
+ # same reconciliation on the next lifecycle hook.
+ logger.debug("Unable to record shared-metrics consent change", exc_info=True)
# =============================================================================
diff --git a/hermes_cli/status.py b/hermes_cli/status.py
index 569c759835..f5c435d912 100644
--- a/hermes_cli/status.py
+++ b/hermes_cli/status.py
@@ -184,6 +184,7 @@ def show_status(args):
"MiniMax-CN": "MINIMAX_CN_API_KEY",
"DeepInfra": "DEEPINFRA_API_KEY",
"Firecrawl": "FIRECRAWL_API_KEY",
+ "Tavily": "TAVILY_API_KEY",
"Keenable": "KEENABLE_API_KEY",
"Browser Use": "BROWSER_USE_API_KEY", # Optional — local browser works without this
"Browserbase": "BROWSERBASE_API_KEY", # Optional — direct credentials only
diff --git a/hermes_cli/subcommands/cron.py b/hermes_cli/subcommands/cron.py
index b9d7f08569..4501578b2c 100644
--- a/hermes_cli/subcommands/cron.py
+++ b/hermes_cli/subcommands/cron.py
@@ -42,6 +42,16 @@ def build_cron_parser(subparsers, *, cmd_cron: Callable) -> None:
"local profile's canonical Bot Chat as a message the bot responds to)"
),
)
+ cron_create.add_argument(
+ "--failure-deliver",
+ dest="failure_deliver",
+ help=(
+ "Override target for FAILURE notices only (same grammar as "
+ "--deliver). 'local' suppresses failure notices entirely; run "
+ "state stays visible in `hermes cron list`. Omit = failures "
+ "follow --deliver."
+ ),
+ )
cron_create.add_argument("--repeat", type=int, help="Optional repeat count")
cron_create.add_argument(
"--skill",
@@ -142,6 +152,14 @@ def build_cron_parser(subparsers, *, cmd_cron: Callable) -> None:
cron_edit.add_argument("--prompt", help="New prompt/task instruction")
cron_edit.add_argument("--name", help="New job name")
cron_edit.add_argument("--deliver", help="New delivery target")
+ cron_edit.add_argument(
+ "--failure-deliver",
+ dest="failure_deliver",
+ help=(
+ "Override target for failure notices (same grammar as --deliver; "
+ "'local' suppresses; '' clears the override)"
+ ),
+ )
cron_edit.add_argument("--repeat", type=int, help="New repeat count")
cron_edit.add_argument(
"--skill",
diff --git a/hermes_cli/subcommands/dashboard.py b/hermes_cli/subcommands/dashboard.py
index 0b695e076a..8b2e6cc487 100644
--- a/hermes_cli/subcommands/dashboard.py
+++ b/hermes_cli/subcommands/dashboard.py
@@ -84,6 +84,60 @@ def _add_server_runtime_args(parser) -> None:
)
+def _configure_serve_parser(parser, *, cmd_dashboard: Callable) -> None:
+ """Attach the canonical ``serve`` arguments to *parser*.
+
+ Kept separate from the full subcommand tree so Desktop's hot path can parse
+ only the command it launches. Both callers use this exact function, keeping
+ the lean parser and normal CLI semantics in lockstep.
+ """
+ _add_server_runtime_args(parser)
+ # Accepted but redundant: ``serve`` is always headless. Kept so callers
+ # using the legacy flag do not trip an argparse error.
+ parser.add_argument("--no-open", action="store_true", help=argparse.SUPPRESS)
+ parser.add_argument(
+ "--ssh-session-token-file",
+ dest="ssh_session_token_file",
+ metavar="PATH",
+ default=None,
+ help="Read a one-shot Desktop SSH session token from PATH",
+ )
+ parser.add_argument(
+ "--ssh-owner-nonce",
+ dest="ssh_owner_nonce",
+ metavar="NONCE",
+ default=None,
+ help="Identify a Desktop-owned SSH backend process",
+ )
+ parser.set_defaults(
+ func=cmd_dashboard,
+ no_open=True,
+ headless_backend=True,
+ command="serve",
+ )
+
+
+def build_serve_parser(
+ *,
+ cmd_dashboard: Callable,
+ add_help: bool = True,
+ exit_on_error: bool = True,
+) -> argparse.ArgumentParser:
+ """Build the standalone parser used by the lean ``serve`` dispatch path."""
+ parser = argparse.ArgumentParser(
+ prog="hermes serve",
+ description=(
+ "Run the Hermes backend server - the JSON-RPC/WebSocket gateway the "
+ "desktop app and remote clients connect to. Headless: it never opens "
+ "a browser UI."
+ ),
+ add_help=add_help,
+ exit_on_error=exit_on_error,
+ )
+ _configure_serve_parser(parser, cmd_dashboard=cmd_dashboard)
+ return parser
+
+
def build_dashboard_parser(
subparsers, *, cmd_dashboard: Callable, cmd_dashboard_register: Callable
) -> None:
@@ -142,32 +196,7 @@ def build_dashboard_parser(
"a browser UI."
),
)
- _add_server_runtime_args(serve_parser)
- # Accepted but redundant: `serve` is always headless (see set_defaults
- # below). Kept so callers that pass the legacy `--no-open` flag (e.g. the
- # desktop backend spawn) don't trip "unrecognized arguments".
- serve_parser.add_argument(
- "--no-open", action="store_true", help=argparse.SUPPRESS
- )
- serve_parser.add_argument(
- "--ssh-session-token-file",
- dest="ssh_session_token_file",
- metavar="PATH",
- default=None,
- help="Read a one-shot Desktop SSH session token from PATH",
- )
- serve_parser.add_argument(
- "--ssh-owner-nonce",
- dest="ssh_owner_nonce",
- metavar="NONCE",
- default=None,
- help="Identify a Desktop-owned SSH backend process",
- )
- # `headless_backend` marks the lean path: desktop/remote clients speak pure
- # JSON-RPC/WS, so `serve` skips the web UI build AND never serves the SPA
- # (cmd_dashboard exports HERMES_SERVE_HEADLESS=1). `dashboard` leaves it
- # unset and serves the browser UI as before.
- serve_parser.set_defaults(func=cmd_dashboard, no_open=True, headless_backend=True)
+ _configure_serve_parser(serve_parser, cmd_dashboard=cmd_dashboard)
# `hermes dashboard register` — register a self-hosted dashboard OAuth
# client with Nous Portal and write the client_id into ~/.hermes/.env.
diff --git a/hermes_cli/subcommands/gui.py b/hermes_cli/subcommands/gui.py
index ec10ef117e..d856137a3c 100644
--- a/hermes_cli/subcommands/gui.py
+++ b/hermes_cli/subcommands/gui.py
@@ -55,6 +55,11 @@ def build_gui_parser(subparsers, *, cmd_gui: Callable) -> None:
action="store_true",
help="Skip npm install/package and launch the existing unpacked app from apps/desktop/release",
)
+ gui_parser.add_argument(
+ "--local",
+ action="store_true",
+ help="Show the local-models UI in the desktop app (models pane, quickstart, picker rows)",
+ )
gui_parser.add_argument(
"--force-build",
action="store_true",
diff --git a/hermes_cli/terminal_notify.py b/hermes_cli/terminal_notify.py
new file mode 100644
index 0000000000..6c1877cd8c
--- /dev/null
+++ b/hermes_cli/terminal_notify.py
@@ -0,0 +1,99 @@
+"""Terminal-native desktop notifications: OSC 9 and Warp's OSC 777 CLI-agent protocol.
+
+Both emitters ride on the existing ``display.bell_on_prompt`` /
+``display.bell_on_complete`` flags (see ``cli._ring_bell``) — no extra config.
+
+- **OSC 9** (``ESC ] 9 ; BEL``): Ghostty, iTerm2, Kitty and WezTerm
+ raise an OS notification; terminals that don't know the sequence drop it.
+- **OSC 777** (``ESC ] 777 ; notify ; warp://cli-agent ; BEL``): Warp's
+ structured CLI-agent protocol (tab status + notification mailbox). Only sent
+ when Warp advertises support and the build is newer than the last release
+ that set the protocol var without being able to render the payload.
+
+Sequences are written to ``/dev/tty`` because prompt_toolkit's stdout wrapper
+can buffer or strip raw escapes; when ``/dev/tty`` can't be opened (Windows,
+no controlling terminal) they fall back to ``sys.stdout``. Never raises.
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import re
+import sys
+
+_C0_AND_DEL = re.compile(r"[\x00-\x1f\x7f]")
+_WARP_PROTOCOL_VERSION = 1
+# Last Warp release per channel that set WARP_CLI_AGENT_PROTOCOL_VERSION but
+# could not render structured payloads (Warp's reference agent plugin,
+# should-use-structured.sh). Bash compares these lexicographically; so do we.
+_WARP_LAST_BROKEN = {
+ "stable": "v0.2026.03.25.08.24.stable_05",
+ "preview": "v0.2026.03.25.08.24.preview_05",
+}
+
+
+def _write_tty(seq: str) -> None:
+ """Write raw escapes to /dev/tty, falling back to sys.stdout. Never raises."""
+ try:
+ with open("/dev/tty", "w", encoding="utf-8") as tty:
+ tty.write(seq)
+ return
+ except OSError:
+ pass
+ try:
+ sys.stdout.write(seq)
+ sys.stdout.flush()
+ except Exception:
+ pass
+
+
+def osc9(body: str) -> str:
+ """OSC 9 sequence with C0 controls and DEL stripped from the body."""
+ return f"\x1b]9;{_C0_AND_DEL.sub('', body)}\x07"
+
+
+def warp_supported(env=None) -> bool:
+ """True when running in a Warp build that can render OSC 777 agent payloads."""
+ env = os.environ if env is None else env
+ if env.get("TERM_PROGRAM") != "WarpTerminal" or not env.get("WARP_CLI_AGENT_PROTOCOL_VERSION"):
+ return False
+ client = env.get("WARP_CLIENT_VERSION", "")
+ if not client:
+ return False
+ for channel, last_broken in _WARP_LAST_BROKEN.items():
+ if channel in client and client <= last_broken:
+ return False
+ return True
+
+
+def warp_osc777(event: str, detail: str, session_id: str = "") -> str:
+ """OSC 777 ``warp://cli-agent`` notification; ``event`` is ``stop`` or ``permission_request``.
+
+ Payload mirrors the reference plugin's build-payload.sh: common fields plus
+ ``summary`` (permission_request) or ``response`` (stop), truncated to 200.
+ """
+ try:
+ advertised = int(os.environ.get("WARP_CLI_AGENT_PROTOCOL_VERSION", "1"))
+ except ValueError:
+ advertised = 1
+ cwd = os.getcwd()
+ payload = {
+ "v": min(advertised, _WARP_PROTOCOL_VERSION),
+ "agent": "hermes",
+ "event": event,
+ "session_id": session_id,
+ "cwd": cwd,
+ "project": os.path.basename(cwd),
+ }
+ payload["summary" if event == "permission_request" else "response"] = detail[:200]
+ return f"\x1b]777;notify;warp://cli-agent;{json.dumps(payload, separators=(',', ':'))}\x07"
+
+
+def notify(context: str, *, prompt: bool, session_id: str = "", detail: str = "") -> None:
+ """Emit OSC 9 (plus Warp OSC 777 when supported) for a blocking prompt or turn end."""
+ seq = osc9(f"Hermes: {context}")
+ if warp_supported():
+ event = "permission_request" if prompt else "stop"
+ seq += warp_osc777(event, detail or context, session_id)
+ _write_tty(seq)
diff --git a/hermes_cli/tools_config.py b/hermes_cli/tools_config.py
index 10d433db7c..71d2ed2593 100644
--- a/hermes_cli/tools_config.py
+++ b/hermes_cli/tools_config.py
@@ -111,7 +111,7 @@ CONFIGURABLE_TOOLSETS = [
("tts", "🔊 Text-to-Speech", "text_to_speech"),
("stt", "🎙️ Speech-to-Text", "voice transcription (gateway voice messages + voice mode)"),
("skills", "📚 Skills", "list, view, manage"),
- ("todo", "📋 Task Planning", "todo"),
+ ("todo", "📋 Task Planning", "todo_list"),
("memory", "💾 Memory", "persistent memory across sessions"),
("context_engine", "🧩 Context Engine", "runtime tools from the active context engine"),
("session_search", "🔎 Session Search", "search past conversations"),
@@ -3331,8 +3331,8 @@ def _plugin_video_gen_providers() -> list[dict]:
# Mirror of _plugin_image_gen_providers for web search backends. Surfaces
# every plugin-registered web provider so it appears in the
-# "Web Search & Extract" picker. All seven providers (brave-free, ddgs,
-# searxng, exa, parallel, firecrawl, keenable) live as plugins after
+# "Web Search & Extract" picker. All bundled providers (brave-free, ddgs,
+# searxng, exa, parallel, tavily, firecrawl, keenable) live as plugins after
# PR #25182 — this helper is the sole source of truth for the category's
# provider rows. The hardcoded entries that used to drive the category
# were deleted in the same PR; only the two non-provider UX rows
@@ -3348,8 +3348,8 @@ def _plugin_web_search_providers() -> list[dict]:
marker) so the picker behaves identically whether a provider is
hardcoded or plugin-registered.
- After PR #25182, all seven web providers (brave-free, ddgs, searxng,
- exa, parallel, firecrawl, keenable) are plugins; this helper is the sole
+ After PR #25182, all bundled web providers (brave-free, ddgs, searxng,
+ exa, parallel, tavily, firecrawl, keenable) are plugins; this helper is the sole
source of provider rows for the Web Search & Extract category.
"""
try:
@@ -5622,6 +5622,43 @@ def _reconfigure_simple_requirements(ts_key: str):
# ─── Main Entry Point ─────────────────────────────────────────────────────────
+def _shared_metrics_state(config: dict) -> tuple[bool, bool]:
+ """Return (collection_enabled, send_enabled) from a config dict."""
+ telemetry = config.get("telemetry")
+ telemetry = telemetry if isinstance(telemetry, dict) else {}
+ shared = telemetry.get("shared_metrics")
+ shared = shared if isinstance(shared, dict) else {}
+ return shared.get("enabled") is True, shared.get("send") is True
+
+
+def _shared_metrics_menu_label(config: dict) -> str:
+ """Menu row for shared metrics, showing both consent states."""
+ enabled, send = _shared_metrics_state(config)
+ if not enabled:
+ state = "off"
+ elif send:
+ state = "collecting + sending to Nous"
+ else:
+ state = "collecting locally"
+ return f"Configure shared metrics ({state})"
+
+
+def _configure_shared_metrics_interactive(config: dict) -> None:
+ """Toggle shared-metrics collection and sending from `hermes tools`.
+
+ Delegates to the setup wizard's prompt so the consent rules live in one
+ place: sending requires collection, and turning collection off also turns
+ sending off.
+ """
+ from hermes_cli.setup import setup_telemetry
+
+ before = _shared_metrics_state(config)
+ setup_telemetry(config)
+ after = _shared_metrics_state(config)
+ if before != after:
+ save_config(config)
+
+
def tools_command(args=None, first_install: bool = False, config: dict = None):
"""Entry point for `hermes tools` and `hermes setup tools`.
@@ -5746,6 +5783,7 @@ def tools_command(args=None, first_install: bool = False, config: dict = None):
if len(platform_keys) > 1:
platform_choices.append("Configure all platforms (global)")
platform_choices.append("Reconfigure an existing tool's provider or API key")
+ platform_choices.append(_shared_metrics_menu_label(config))
# Show MCP option if any MCP servers are configured
_has_mcp = bool(config.get("mcp_servers"))
@@ -5757,8 +5795,9 @@ def tools_command(args=None, first_install: bool = False, config: dict = None):
# Index offsets for the extra options after per-platform entries
_global_idx = len(platform_keys) if len(platform_keys) > 1 else -1
_reconfig_idx = len(platform_keys) + (1 if len(platform_keys) > 1 else 0)
- _mcp_idx = (_reconfig_idx + 1) if _has_mcp else -1
- _done_idx = _reconfig_idx + (2 if _has_mcp else 1)
+ _metrics_idx = _reconfig_idx + 1
+ _mcp_idx = (_metrics_idx + 1) if _has_mcp else -1
+ _done_idx = _metrics_idx + (2 if _has_mcp else 1)
while True:
idx = _prompt_choice("Select an option:", platform_choices, default=0)
@@ -5773,6 +5812,13 @@ def tools_command(args=None, first_install: bool = False, config: dict = None):
print()
continue
+ # "Shared metrics" selected
+ if idx == _metrics_idx:
+ _configure_shared_metrics_interactive(config)
+ platform_choices[_metrics_idx] = _shared_metrics_menu_label(config)
+ print()
+ continue
+
# "Configure MCP tools" selected
if idx == _mcp_idx:
_configure_mcp_tools_interactive(config)
diff --git a/hermes_cli/update_cmd.py b/hermes_cli/update_cmd.py
index f0a825b32b..9bbfa17693 100644
--- a/hermes_cli/update_cmd.py
+++ b/hermes_cli/update_cmd.py
@@ -2083,6 +2083,90 @@ def _restore_state_db_from_snapshot(state_path: Path, snap_state: Path) -> bool:
return bool(restored.get("valid"))
+def _verify_and_restore_one_state_db(home: Path, *, label: str) -> None:
+ """Post-update integrity check + auto-restore for ONE home's state.db.
+
+ Shared by the root-DB and sibling-profile guards (ZIP update path and
+ git-pull path both route here). A corrupt live DB is restored from the
+ most recent valid snapshot under that home's own state-snapshots dir.
+ Never raises: a guard that crashes the update tail would be worse than
+ the corruption it detects.
+ """
+ try:
+ from hermes_cli.backup import _quick_snapshot_root, verify_sqlite_integrity
+
+ state_path = home / "state.db"
+ if not state_path.exists():
+ return
+ ok = verify_sqlite_integrity(state_path, check_header=True, run_pragma=True)
+ if ok.get("valid"):
+ logger.debug(
+ "Post-update state.db integrity OK (%s): %s",
+ label,
+ ok.get("message"),
+ )
+ return
+ print()
+ print(
+ f"⚠ state.db is corrupted after update ({label}): "
+ + ok.get("message", "unknown error")
+ )
+ snap_root = _quick_snapshot_root(home)
+ if not snap_root.exists():
+ print(" ⚠ No pre-update snapshot for this home")
+ return
+ for snap_dir in sorted(
+ (d for d in snap_root.iterdir() if d.is_dir()), reverse=True
+ ):
+ snap_state = snap_dir / "state.db"
+ if not snap_state.exists():
+ continue
+ snap_ok = verify_sqlite_integrity(
+ snap_state, check_header=True, run_pragma=True
+ )
+ if not snap_ok.get("valid"):
+ continue
+ try:
+ if _restore_state_db_from_snapshot(state_path, snap_state):
+ print(
+ f" ✓ Auto-restored from snapshot {snap_dir.name} ({label})"
+ )
+ else:
+ print(
+ " ✗ Auto-restore FAILED — restored copy also failed "
+ "integrity"
+ )
+ except OSError as exc:
+ print(f" ✗ Auto-restore file copy failed: {exc}")
+ return
+ print(" ⚠ No valid pre-update snapshot found for this home")
+ except Exception as exc:
+ logger.debug(
+ "Post-update state.db guard (%s) failed: %s", label, exc
+ )
+
+
+def _verify_and_restore_state_dbs_post_update() -> None:
+ """Post-update integrity guard for the ROOT state.db AND every sibling
+ profile's state.db (#97994).
+
+ The pre-update snapshot already covers every sibling profile
+ (#66140 create_pre_update_snapshots_all_profiles), but the post-update
+ guard only ever verified the root DB — a profile database corrupted by
+ the update was never detected and never auto-restored, leaving that
+ profile's sessions silently gone while the root DB passed.
+ """
+ home = get_hermes_home()
+ _verify_and_restore_one_state_db(home, label="default home")
+ try:
+ from hermes_cli.backup import _sibling_profile_homes
+
+ for name, profile_home in _sibling_profile_homes(home):
+ _verify_and_restore_one_state_db(profile_home, label=f"profile {name}")
+ except Exception as exc:
+ logger.debug("Sibling-profile state.db guard sweep failed: %s", exc)
+
+
def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> bool:
"""Update Hermes Agent by downloading a ZIP archive.
@@ -2433,55 +2517,12 @@ def _update_via_zip(args, *, had_desktop_app_before_update: bool = False) -> boo
except Exception as e:
logger.debug("Model catalog seed during zip update failed: %s", e)
- # ── Post-update state.db integrity guard (#68474) ─────────────────
- # Same as the git-pull path: verify state.db survived the ZIP update
- # and auto-restore from the most recent pre-update snapshot if needed.
+ # ── Post-update state.db integrity guard (#68474, #97994) ────────────
+ # Verify state.db survived the ZIP update in the root home AND every
+ # sibling profile, auto-restoring each from its own most recent valid
+ # pre-update snapshot when needed.
try:
- from hermes_cli.backup import _quick_snapshot_root, verify_sqlite_integrity
-
- _state_path = get_hermes_home() / "state.db"
- if _state_path.exists():
- _state_ok = verify_sqlite_integrity(
- _state_path, check_header=True, run_pragma=True
- )
- if not _state_ok.get("valid"):
- print()
- print(
- "⚠ state.db is corrupted after update: "
- + _state_ok.get("message", "unknown error")
- )
- _snap_root = _quick_snapshot_root(get_hermes_home())
- if _snap_root.exists():
- _snap_dirs = sorted(
- (d for d in _snap_root.iterdir() if d.is_dir()),
- reverse=True,
- )
- for _snap_dir in _snap_dirs:
- _snap_state = _snap_dir / "state.db"
- if _snap_state.exists():
- _snap_ok = verify_sqlite_integrity(
- _snap_state, check_header=True, run_pragma=True
- )
- if _snap_ok.get("valid"):
- try:
- if _restore_state_db_from_snapshot(
- _state_path, _snap_state
- ):
- print(
- " ✓ Auto-restored from snapshot "
- f"{_snap_dir.name}"
- )
- else:
- print(
- " ✗ Auto-restore FAILED — restored "
- "copy also failed integrity"
- )
- break
- except OSError as _exc:
- print(
- f" ✗ Auto-restore file copy failed: {_exc}"
- )
- break
+ _verify_and_restore_state_dbs_post_update()
except Exception as exc:
logger.debug(
"Post-update state.db integrity check (zip path) failed: %s", exc
@@ -2541,7 +2582,7 @@ def _stash_local_changes_if_needed(git_cmd: list[str], cwd: Path) -> Optional[st
from datetime import datetime, timezone
stash_name = datetime.now(timezone.utc).strftime(
- "hermes-update-autostash-%Y%m%d-%H%M%S"
+ f"{_AUTOSTASH_NAME_PREFIX}%Y%m%d-%H%M%S"
)
print("→ Local changes detected — stashing before update...")
prev_stash = subprocess.run(
@@ -2628,6 +2669,83 @@ def _resolve_stash_selector(
return selector.strip()
return None
+#: Producer/consumer contract for update autostash names: the stash subject is
+#: this prefix + a UTC YYYYMMDD-HHMMSS stamp (see _stash_local_changes_if_needed
+#: and _warn_orphaned_update_autostashes).
+_AUTOSTASH_NAME_PREFIX = "hermes-update-autostash-"
+
+#: Age past which a leftover ``hermes-update-autostash-*`` entry is called out
+#: at update time. Entries younger than this are normal (a parked stash from
+#: the desktop updater's --keep-stash run minutes ago); older ones are almost
+#: always forgotten (#63717 problem 6: an orphan persisted 9+ days unnoticed).
+_AUTOSTASH_WARN_AGE_DAYS = 7
+
+
+def _warn_orphaned_update_autostashes(git_cmd: list[str], cwd: Path) -> int:
+ """Surface leftover update autostashes older than the warn threshold.
+
+ Autostash entries legitimately outlive an update run (``--keep-stash``
+ parks them; a conflicted or failed restore preserves them for safety), but
+ nothing ever re-surfaces them afterwards — they sit in ``git stash``
+ invisibly for weeks (#63717 problem 6). This prints a short notice naming
+ the stale entries with recovery/cleanup guidance. Deliberately NOT a GC:
+ a stash entry can be the only copy of the user's uncommitted work, so
+ Hermes never drops one automatically.
+
+ Best-effort — any git failure returns 0 and must not block the update.
+ Returns the number of stale entries warned about.
+ """
+ from datetime import timedelta, timezone
+
+ try:
+ stash_list = subprocess.run(
+ git_cmd + ["stash", "list", "--format=%gd %s"],
+ cwd=cwd,
+ capture_output=True,
+ text=True, encoding="utf-8", errors="replace",
+ )
+ if stash_list.returncode != 0:
+ return 0
+ cutoff = datetime.now(timezone.utc) - timedelta(
+ days=_AUTOSTASH_WARN_AGE_DAYS
+ )
+ marker = _AUTOSTASH_NAME_PREFIX
+ stale: list[tuple[str, str]] = []
+ for line in stash_list.stdout.splitlines():
+ selector, _, subject = line.strip().partition(" ")
+ pos = subject.find(marker)
+ if pos < 0:
+ continue
+ stamp = subject[pos + len(marker):][:15] # "YYYYMMDD-HHMMSS"
+ try:
+ stash_time = datetime.strptime(stamp, "%Y%m%d-%H%M%S").replace(
+ tzinfo=timezone.utc
+ )
+ except ValueError:
+ # Unparseable name — age unknown; leave it alone rather than
+ # guess (same posture as _prune_orphan_rescue_refs).
+ continue
+ if stash_time < cutoff:
+ stale.append((selector, stamp))
+ if not stale:
+ return 0
+ print()
+ print(
+ f"⚠ {len(stale)} leftover update autostash entr"
+ f"{'y is' if len(stale) == 1 else 'ies are'} more than "
+ f"{_AUTOSTASH_WARN_AGE_DAYS} days old:"
+ )
+ for selector, stamp in stale:
+ print(f" {selector} ({_AUTOSTASH_NAME_PREFIX}{stamp})")
+ print(" These hold local changes stashed by earlier updates and never")
+ print(" restored. Review with: git stash show -p ")
+ print(" Restore with: git stash apply Discard with: git stash drop ")
+ return len(stale)
+ except Exception as exc:
+ logger.debug("Autostash age check failed: %s", exc)
+ return 0
+
+
def _print_stash_cleanup_guidance(
stash_ref: str, stash_selector: Optional[str] = None
) -> None:
@@ -7451,6 +7569,52 @@ def _resume_windows_gateways_after_update(token: dict | None) -> None:
token["unmapped"] = failed_unmapped
if failed_profiles or failed_unmapped:
raise RuntimeError("Could not restart every paused Windows gateway")
+
+ # A truthy return from the launch helpers only proves the detached
+ # watcher process was created — not that the gateway it respawns
+ # survived. A parent Job Object that denies CREATE_BREAKAWAY_FROM_JOB
+ # kills the freshly respawned gateway on updater teardown before it
+ # writes a single log line, yet "✓ Restarting" was printed anyway
+ # (#48820, 3rd/4th repro). Verify a stable gateway process actually
+ # exists before vouching for the resume, using the same
+ # provisional-hit + confirmation-window poll every other spawn path
+ # uses (#91675). all_profiles=True because the resume covers the fleet.
+ if relaunched or unmapped_relaunched:
+ try:
+ from hermes_cli import gateway_windows
+ except Exception as exc:
+ raise RuntimeError(
+ f"Could not load Windows gateway liveness helpers: {exc}"
+ ) from exc
+ ready_pids = gateway_windows._wait_for_gateway_ready(
+ timeout_s=30.0, all_profiles=True
+ )
+ if not ready_pids:
+ token["profiles"] = dict(profiles)
+ token["unmapped"] = list(unmapped)
+ print()
+ print(
+ " ⚠ Windows gateway restart could not be verified — no stable "
+ "gateway process appeared after relaunch."
+ )
+ print(
+ " (The respawned gateway may have been killed by a parent "
+ "Job Object during updater teardown, #48820.)"
+ )
+ print(" Recover with: hermes gateway restart")
+ raise RuntimeError(
+ "Windows gateway relaunch after update was not verified alive"
+ )
+ # Persist the PIDs this ✓ vouches for so a death AFTER the updater
+ # exits (parent Job Object teardown, #91675) is reported by the next
+ # CLI invocation instead of staying silent. Best-effort.
+ try:
+ gateway_windows._write_start_attestation(
+ ready_pids, "post-update relaunch"
+ )
+ except Exception:
+ pass
+
token["resume_needed"] = False
if relaunched:
@@ -8416,6 +8580,11 @@ def _cmd_update_impl(args, gateway_mode: bool):
if swept:
print(" (removed %d aborted-fetch pack temp file(s))" % len(swept))
+ # Surface autostash entries left behind by earlier updates (#63717
+ # problem 6) — parked --keep-stash runs and failed restores preserve
+ # the stash but nothing ever mentioned it again.
+ _m()._warn_orphaned_update_autostashes(git_cmd, _m().PROJECT_ROOT)
+
print("→ Fetching updates...")
fetch_result = subprocess.run(
git_cmd + ["fetch", "origin", branch],
@@ -9393,72 +9562,14 @@ def _cmd_update_impl(args, gateway_mode: bool):
except Exception:
logger.debug("macOS TCC anchor refresh skipped", exc_info=True)
- # ── Post-update state.db integrity guard (#68474) ─────────────────
- # Verify that state.db survived the update intact. If the live file
- # is now corrupted (zeroed, missing header, integrity failure),
- # automatically restore from the pre-update snapshot rather than
- # letting the user discover silently that their sessions are gone.
+ # ── Post-update state.db integrity guard (#68474, #97994) ─────────
+ # Verify that state.db survived the update intact in the root home
+ # AND every sibling profile. If a live file is now corrupted (zeroed,
+ # missing header, integrity failure), automatically restore from that
+ # home's own pre-update snapshot rather than letting the user discover
+ # silently that their sessions are gone.
try:
- from hermes_cli.backup import _quick_snapshot_root, verify_sqlite_integrity
-
- _state_path = get_hermes_home() / "state.db"
- if _state_path.exists():
- _state_ok = verify_sqlite_integrity(
- _state_path,
- check_header=True,
- run_pragma=True,
- )
- if _state_ok.get("valid"):
- logger.debug(
- "Post-update state.db integrity check: %s",
- _state_ok.get("message"),
- )
- else:
- print()
- print(
- "⚠ state.db is corrupted after update: "
- + _state_ok.get("message", "unknown error")
- )
- _pre_snap_id = pre_update_snapshot_id
- if _pre_snap_id:
- _snap_state = (
- _quick_snapshot_root(get_hermes_home())
- / _pre_snap_id
- / "state.db"
- )
- if _snap_state.exists():
- _snap_ok = verify_sqlite_integrity(
- _snap_state, check_header=True, run_pragma=True
- )
- if _snap_ok.get("valid"):
- try:
- if _restore_state_db_from_snapshot(
- _state_path, _snap_state
- ):
- print(
- " ✓ Auto-restored from pre-update "
- f"snapshot ({_pre_snap_id})"
- )
- else:
- print(
- " ✗ Auto-restore FAILED — restored "
- "copy also failed integrity"
- )
- except OSError as _exc:
- print(
- f" ✗ Auto-restore file copy failed: {_exc}"
- )
- else:
- print(
- " ✗ Pre-update snapshot also failed integrity"
- )
- else:
- print(
- " ⚠ Pre-update snapshot does not contain state.db"
- )
- else:
- print(" ⚠ No pre-update snapshot was taken")
- print()
+ _verify_and_restore_state_dbs_post_update()
except Exception as exc:
logger.debug("Post-update state.db integrity check failed: %s", exc)
@@ -10703,6 +10814,26 @@ def _cmd_update_impl(args, gateway_mode: bool):
node_failures, already_restarted_units=set(restarted_services)
)
+ # Check if any pre-update serve/dashboard runtimes survived on
+ # pre-update code generations (#100479). This is the SUCCESS-path
+ # twin of the abort-recovery probe above: the restart phase only
+ # restarts units, so an sshd-spawned `serve --isolated` or a manual
+ # `hermes serve` (no unit) is left running its pre-update
+ # sys.modules graph — and its cron ticker keeps firing agent jobs
+ # that ImportError on every symbol added in the pulled range. Runs
+ # AFTER the dashboard cleanup so a manual dashboard that cleanup
+ # killed and respawned is (correctly) not a survivor. The rows also
+ # feed the plan-vs-execution reconciliation below, so a survivor is
+ # escalated (exit 1) instead of merely printed. ``None`` means the
+ # probe itself failed; the reconciliation then stays fail-closed.
+ _stale_serve_rows: "list | None" = None
+ try:
+ _stale_serve_rows = _surviving_pre_update_serve_runtimes(_pre_update_plan)
+ if _stale_serve_rows:
+ _warn_stale_serve_runtimes(_stale_serve_rows)
+ except Exception as _serve_warn_exc:
+ logger.debug("Failed to check for surviving serve runtimes: %s", _serve_warn_exc)
+
print()
print("Tip: You can now select a provider and model:")
print(" hermes model # Select provider and model")
@@ -10812,6 +10943,13 @@ def _cmd_update_impl(args, gateway_mode: bool):
externally_supervised_profiles=externally_supervised_profiles,
killed_pids=killed_pids,
failed_units=failed_or_stale_units,
+ # Serve/dashboard runtimes reconcile by incarnation
+ # liveness, not by the gateway's unit names (#100479).
+ stale_serve_pids=(
+ {row.get("pid") for row in _stale_serve_rows}
+ if _stale_serve_rows is not None
+ else None
+ ),
)
if report_unaccounted_runtimes(_runtime_outcomes):
gateway_fleet_restart_incomplete = True
diff --git a/hermes_cli/update_inventory.py b/hermes_cli/update_inventory.py
index be03530f7d..1e2528e86c 100644
--- a/hermes_cli/update_inventory.py
+++ b/hermes_cli/update_inventory.py
@@ -425,6 +425,49 @@ def print_update_plan(plan: UpdatePlan) -> None:
)
+_SERVE_KINDS = ("serve", "dashboard")
+
+
+def _serve_unit_matches_profile(profile: str, unit: object) -> bool:
+ """Does *unit* name a ``hermes-serve*``/``hermes-dashboard*`` unit for *profile*?
+
+ Serve/dashboard runtimes have their OWN unit vocabulary; the gateway's
+ ``hermes-gateway*`` names never cover them (#100479). Exact names only —
+ ``work`` must not claim ``hermes-serve-workbench`` — and a scope prefix
+ (``user/hermes-serve``) is tolerated because the restart phase records
+ scope-qualified identities in some lists.
+ """
+ name = str(unit).removesuffix(".service")
+ if "/" in name:
+ name = name.rsplit("/", 1)[-1]
+ if profile == "default":
+ return name in {"hermes-serve", "hermes-dashboard"}
+ return name in {f"hermes-serve-{profile}", f"hermes-dashboard-{profile}"}
+
+
+def _serve_runtime_outcome(
+ r: RuntimeRecord,
+ *,
+ killed: set,
+ failed_set: set,
+ restarted_set: set,
+ stale_serves: "set | None",
+) -> str:
+ """Outcome for one serve/dashboard runtime — never the gateway's."""
+ if r.pid is not None and r.pid in killed:
+ return "stopped"
+ if any(_serve_unit_matches_profile(r.profile, u) for u in failed_set):
+ return "failed"
+ if stale_serves is not None:
+ # Incarnation-verified: the pre-update process is gone (replaced by
+ # its unit / the dashboard cleanup respawn / the Desktop app) or it
+ # is still alive on pre-update code.
+ return "unaccounted" if r.pid in stale_serves else "restarted"
+ if any(_serve_unit_matches_profile(r.profile, s) for s in restarted_set):
+ return "restarted"
+ return "unaccounted"
+
+
def match_runtime_outcomes(
plan: "UpdatePlan",
*,
@@ -433,6 +476,7 @@ def match_runtime_outcomes(
externally_supervised_profiles: list,
killed_pids: set,
failed_units: list,
+ stale_serve_pids: "set | None" = None,
) -> list[dict[str, Any]]:
"""Reconcile the plan's runtimes against what the restart phase DID.
@@ -450,6 +494,18 @@ def match_runtime_outcomes(
``unaccounted`` — the plan saw it and NO bookkeeping mentions it: the
blind-spot tripwire (same philosophy as the fleet matrix's DOWN row).
Never raises; on any probe error returns what it has.
+
+ Serve/dashboard runtimes are reconciled in their OWN vocabulary
+ (#100479): a ``hermes-serve*``/``hermes-dashboard*`` unit, a killed
+ PID, or — when the caller passes ``stale_serve_pids`` (the
+ ``(pid, create_time)``-verified survivor probe,
+ :func:`hermes_cli.update_abort_recovery._surviving_pre_update_serve_runtimes`)
+ — liveness: a pre-update serve whose incarnation is gone was replaced
+ (unit restart, dashboard cleanup respawn, Desktop respawn) and counts as
+ ``restarted``; one still alive is ``unaccounted``. They never borrow the
+ gateway's outcome: ``relaunched_profiles`` and ``hermes-gateway*`` name a
+ different process that shares the profile, nothing more. Without the
+ probe result, an untouched serve stays ``unaccounted`` (fail closed).
"""
outcomes: list[dict[str, Any]] = []
try:
@@ -458,23 +514,57 @@ def match_runtime_outcomes(
relaunched = set(relaunched_profiles or [])
external = set(externally_supervised_profiles or [])
killed = {int(p) for p in (killed_pids or set())}
+ stale_serves = (
+ {int(p) for p in stale_serve_pids} if stale_serve_pids is not None else None
+ )
for runtime in plan.runtimes:
r = runtime if isinstance(runtime, RuntimeRecord) else None
if r is None:
continue
+ if r.kind in _SERVE_KINDS:
+ outcomes.append(
+ {
+ "kind": r.kind,
+ "profile": r.profile,
+ "pid": r.pid,
+ "mechanism": r.restart_via,
+ "outcome": _serve_runtime_outcome(
+ r,
+ killed=killed,
+ failed_set=failed_set,
+ restarted_set=restarted_set,
+ stale_serves=stale_serves,
+ ),
+ }
+ )
+ continue
outcome = "unaccounted"
+ # The bare "hermes-gateway" unit name is gateway-specific: a
+ # serve/dashboard runtime that merely shares the default
+ # profile is a different process the gateway restart never
+ # touched, and must not borrow its outcome (#100479).
if r.profile in relaunched or r.profile in external:
outcome = "restarted"
elif r.pid is not None and r.pid in killed:
outcome = "stopped"
elif any(
- r.profile in unit or (r.profile == "default" and "hermes-gateway" in unit)
+ r.profile in unit
+ or (
+ r.kind == "gateway"
+ and r.profile == "default"
+ and "hermes-gateway" in unit
+ )
for unit in failed_set
):
outcome = "failed"
elif any(
- r.profile in svc or (r.profile == "default" and "hermes-gateway" in svc)
+ r.profile in svc
+ or (
+ r.kind == "gateway"
+ and r.profile == "default"
+ and "hermes-gateway" in svc
+ )
for svc in restarted_set
):
outcome = "restarted"
@@ -511,8 +601,14 @@ def report_unaccounted_runtimes(outcomes: list[dict[str, Any]]) -> bool:
f" — planned mechanism: {o['mechanism']}"
)
print(" Restart them manually, then verify:")
- print(" hermes gateway restart # active profile")
- print(" hermes -p gateway restart # named profile")
+ if any(o.get("kind") not in _SERVE_KINDS for o in missed):
+ print(" hermes gateway restart # active profile")
+ print(" hermes -p gateway restart # named profile")
+ if any(o.get("kind") in _SERVE_KINDS for o in missed):
+ # A serve/dashboard is not reachable by any `gateway restart`
+ # command (#100479): name the process, not the wrong verb.
+ print(" systemctl --user restart hermes-serve.service # unit-managed serve")
+ print(" relaunch `hermes serve` / `hermes dashboard` / the Desktop app")
return True
diff --git a/hermes_cli/web_models.py b/hermes_cli/web_models.py
index fa5dd37243..b03f649417 100644
--- a/hermes_cli/web_models.py
+++ b/hermes_cli/web_models.py
@@ -306,6 +306,17 @@ class TTSSpeakRequest(BaseModel):
text: str
+class TTSLeaseRequest(BaseModel):
+ """Body for ``POST /api/audio/tts-lease``.
+
+ ``lease`` names the toggle/surface holding the lease (``desktop:read-aloud``,
+ ``desktop:conversation``); ``active`` True acquires + warms, False releases.
+ """
+
+ lease: str
+ active: bool = True
+
+
# --- from web_server.py (originally lines 11549-11551) ---
class OAuthSubmitBody(BaseModel):
diff --git a/hermes_cli/web_routers/local_models.py b/hermes_cli/web_routers/local_models.py
new file mode 100644
index 0000000000..055e80a2db
--- /dev/null
+++ b/hermes_cli/web_routers/local_models.py
@@ -0,0 +1,1426 @@
+"""Local-models dashboard routes — the desktop's window into the managed
+llama.cpp runtime.
+
+Everything here is designed for a first-run user on an RTX laptop: every
+payload carries plain-language, pre-formatted facts the UI can show verbatim
+(what will this model do ON THIS MACHINE, how big is the download, what is
+the runtime doing right now), never raw internals the renderer would have to
+interpret.
+
+Long jobs (runtime install, model download) follow the repo's job pattern:
+start-POST -> {job_id} -> GET poll with byte progress. Downloads are
+byte-size checked against the catalog (no hash verification by design);
+a short download deletes the file and reports it plainly.
+"""
+
+from __future__ import annotations
+
+import json
+import logging
+import threading
+import time
+import urllib.parse
+import urllib.request
+import uuid
+from pathlib import Path
+from typing import Any, Dict, Optional
+
+from fastapi import APIRouter, HTTPException
+from pydantic import BaseModel
+
+from hermes_cli.local_runtime.endpoint import _state_endpoint
+
+logger = logging.getLogger(__name__)
+
+router = APIRouter()
+
+_GIB = 1 << 30
+_JOBS: Dict[str, Dict[str, Any]] = {}
+_JOBS_LOCK = threading.Lock()
+
+
+def _human_gb(n: int | float) -> str:
+ return f"{n / _GIB:.1f} GB"
+
+
+def _job(kind: str, target: str, model_id: str | None = None) -> Dict[str, Any]:
+ job = {
+ "job_id": uuid.uuid4().hex[:12],
+ "kind": kind, # "runtime-install" | "model-download"
+ "target": target,
+ "model_id": model_id, # catalog id for downloads; None otherwise
+ "status": "running", # running | done | error
+ "phase": "starting", # human-readable step name
+ "detail": "",
+ "total_bytes": None,
+ "done_bytes": 0,
+ "started_at": time.time(),
+ "error": None,
+ }
+ with _JOBS_LOCK:
+ _JOBS[job["job_id"]] = job
+ return job
+
+
+# ── fast download: ranged parallel streams ───────────────────
+
+# One TCP stream to a CDN rarely fills a fast line; 8 ranged connections
+# writing into a preallocated file saturate consumer gigabit.
+_DOWNLOAD_CONNECTIONS = 8
+_CHUNK = 4 << 20
+
+
+def _probe_range_support(url: str) -> int:
+ """Total size when the server honors Range requests, else 0.
+
+ Auth-shaped failures raise with a plain-language message — a 401/403
+ from the CDN means the repo is gated or the catalog entry names a
+ wrong repo, and the user deserves better than a bare status code.
+ """
+ req = urllib.request.Request(url, headers={"Range": "bytes=0-0"})
+ try:
+ with urllib.request.urlopen(req, timeout=60) as r:
+ if r.status == 206:
+ content_range = r.headers.get("Content-Range", "")
+ if "/" in content_range:
+ return int(content_range.rsplit("/", 1)[1])
+ except urllib.error.HTTPError as exc:
+ if exc.code in (401, 403):
+ raise RuntimeError(
+ "The model host refused the download (gated or moved). "
+ "This is a catalog problem, not yours — please report it.") from exc
+ raise
+ except Exception: # noqa: BLE001
+ pass
+ return 0
+
+
+def _model_id_for(gguf: Path) -> str:
+ """Variant model id for a staged file (strips split-part suffixes)."""
+ import re
+
+ return re.sub(r"-\d{5}-of-\d{5}$", "", gguf.stem)
+
+
+def _variant_files_on_disk(model_id: str) -> "list[Path]":
+ """Every local file belonging to a staged model: all split parts plus
+ its catalog-declared assets (mmproj/draft) when present."""
+ from hermes_cli.local_runtime.bootstrap import assets_dir
+ from hermes_cli.local_runtime.catalog import find_entry_for_model
+
+ mdir = _models_dir()
+ files = [p for p in mdir.glob("*.gguf") if _model_id_for(p) == model_id]
+ hit = find_entry_for_model(model_id)
+ if hit is not None:
+ entry, _variant = hit
+ for asset in (entry.mmproj, entry.draft):
+ if asset is not None:
+ p = assets_dir() / asset.local_name
+ if p.exists():
+ files.append(p)
+ return files
+
+
+def download_file(url: str, dest: Path, job: Dict[str, Any],
+ *,
+ base_done: int = 0, keep_totals: bool = False) -> None:
+ """Download url -> dest with byte progress on ``job``.
+
+ Ranged-parallel when the server supports it, single-stream fallback
+ otherwise. There is no integrity check against the CATALOG by
+ design: catalog sizes may lag an upstream re-upload, and a
+ newer file than we know about must download fine. Completeness is
+ checked only against what the SERVER declared for this transfer
+ (range-probe total / Content-Length) — self-consistent and always
+ current — so a dropped connection still errors instead of staging a
+ truncated file. Never leaves a .part behind.
+
+ Multi-file variants: ``base_done`` offsets the progress so this file's
+ bytes accumulate onto the files before it, and ``keep_totals=True``
+ stops the per-file size from overwriting the variant's total.
+ """
+ import shutil
+ import threading as _threading
+
+ tmp = dest.with_suffix(".part")
+ dest.parent.mkdir(parents=True, exist_ok=True)
+ file_done = [0]
+ progress_lock = _threading.Lock()
+
+ def bump(n: int) -> None:
+ with progress_lock:
+ file_done[0] += n
+ job["done_bytes"] = base_done + file_done[0]
+
+ try:
+ # The probe and the preallocation both take real seconds on a
+ # 20+ GB file — narrate them, or the pane shows a dead '— of X GB'
+ # until the first ranged byte lands.
+ job["detail"] = "Connecting"
+ total = _probe_range_support(url)
+ if total:
+ if not keep_totals:
+ job["total_bytes"] = total
+ # Preallocate so each worker writes at its own offset.
+ job["detail"] = f"Reserving {_human_gb(total)} of disk space"
+ with open(tmp, "wb") as f:
+ f.truncate(total)
+ job["detail"] = ""
+ errors: list[Exception] = []
+ bounds = [(i * total // _DOWNLOAD_CONNECTIONS,
+ (i + 1) * total // _DOWNLOAD_CONNECTIONS - 1)
+ for i in range(_DOWNLOAD_CONNECTIONS)]
+
+ def fetch_range(start: int, end: int) -> None:
+ try:
+ req = urllib.request.Request(
+ url, headers={"Range": f"bytes={start}-{end}"})
+ with urllib.request.urlopen(req, timeout=120) as r, \
+ open(tmp, "r+b") as f:
+ f.seek(start)
+ while True:
+ chunk = r.read(_CHUNK)
+ if not chunk:
+ break
+ f.write(chunk)
+ bump(len(chunk))
+ except Exception as exc: # noqa: BLE001
+ errors.append(exc)
+
+ threads = [_threading.Thread(target=fetch_range, args=b, daemon=True,
+ name=f"lm-dl-{i}")
+ for i, b in enumerate(bounds)]
+ for t in threads:
+ t.start()
+ for t in threads:
+ t.join()
+ if errors:
+ raise errors[0]
+ if file_done[0] != total:
+ raise RuntimeError(
+ f"download incomplete ({file_done[0]} of {total} bytes)")
+ else:
+ # No range support: single stream, large chunks. Completeness
+ # is judged by the server's own Content-Length when it sent
+ # one — never by the catalog, which may lag a re-upload.
+ with urllib.request.urlopen(url, timeout=120) as r, open(tmp, "wb") as f:
+ length = int(r.headers.get("Content-Length") or 0)
+ if length and not keep_totals:
+ job["total_bytes"] = length
+ while True:
+ chunk = r.read(_CHUNK)
+ if not chunk:
+ break
+ f.write(chunk)
+ bump(len(chunk))
+ if length and file_done[0] != length:
+ raise RuntimeError(
+ f"Download ended at {file_done[0]:,} bytes but the server "
+ f"said {length:,} — connection dropped? Removed; try again")
+
+ shutil.move(str(tmp), str(dest))
+ except Exception:
+ tmp.unlink(missing_ok=True)
+ raise
+
+
+def _models_dir() -> Path:
+ from hermes_cli.local_runtime.bootstrap import models_dir
+
+ return models_dir()
+
+
+def _engine_too_old(min_engine: str) -> bool:
+ """True when the installed llama.cpp predates a model's requirement.
+ Tags are release numbers (b10362); no engine installed compares as
+ too old only when the model states a requirement."""
+ if not min_engine:
+ return False
+ try:
+ from hermes_cli.local_runtime.binaries import default_tag, installed_tags
+
+ tags = installed_tags() or [default_tag()]
+ newest = max(int(t.lstrip("b")) for t in tags if t.lstrip("b").isdigit())
+ return newest < int(min_engine.lstrip("b"))
+ except Exception: # noqa: BLE001
+ return False
+
+
+def _load_config() -> dict:
+ from hermes_cli.config import load_config
+
+ try:
+ return load_config()
+ except Exception: # noqa: BLE001
+ return {}
+
+
+def _runtime_section() -> dict:
+ return (_load_config() or {}).get("local_runtime") or {}
+
+
+# ── status: the one call the pane opens with ─────────────────
+
+
+@router.get("/api/local-models/status")
+def local_models_status():
+ """Cheap, immediate, never blocks on probes (responsiveness standard):
+ config state + installed runtime + staged models + supervisor state.
+ GPU facts come from /api/local-models/hardware (slower, polled).
+
+ Sync def on purpose: the body does blocking urlopen/scans, so it runs
+ in FastAPI's threadpool instead of stalling the event loop."""
+ from hermes_cli.local_runtime.binaries import (
+ default_tag,
+ installed_tags,
+ runtimes_root,
+ server_binary,
+ )
+
+ section = _runtime_section()
+ configured_tag = section.get("tag") or default_tag()
+ have = installed_tags()
+
+ # The tag actually serving (boot ladder: configured if installed, else
+ # newest installed). Present tense for the pane header.
+ tag = configured_tag if configured_tag in have else (have[0] if have else configured_tag)
+
+ # A pending engine update exists when the user runs the local engine
+ # (enabled + something installed) and the configured tag — pinned or
+ # the Hermes-release default — is newer than anything on disk. The
+ # download is a button click, never automatic.
+ update_available = bool(
+ section.get("enabled") and have and configured_tag not in have)
+
+ runtime_installed = False
+ runtime_backend = None
+ root = runtimes_root() / tag
+ if root.exists():
+ for backend_dir in sorted(p for p in root.iterdir() if p.is_dir()):
+ try:
+ server_binary(backend_dir)
+ runtime_installed = True
+ runtime_backend = backend_dir.name
+ break
+ except Exception: # noqa: BLE001
+ continue
+
+ staged = []
+ mdir = _models_dir()
+ if mdir.exists():
+ from hermes_cli.local_runtime.bootstrap import staged_models
+
+ # Split models: report the whole variant's bytes, not one part's.
+ from hermes_cli.local_runtime.catalog import find_entry_for_model
+
+ for gguf in staged_models():
+ model_id = _model_id_for(gguf)
+ size = gguf.stat().st_size
+ hit = find_entry_for_model(model_id)
+ if hit is not None:
+ size = hit[1].size_bytes
+ staged.append({
+ "id": model_id,
+ "size_bytes": size,
+ "size_label": _human_gb(size),
+ })
+
+ running = _state_endpoint()
+
+ # Which staged models are resident right now (loaded in VRAM). Read
+ # from the live router when it's up; {} when down. Feeds the pane's
+ # Loaded pills and eject buttons.
+ loaded: Dict[str, str] = {}
+ placement: Dict[str, Any] = {}
+ if running is not None:
+ try:
+ import urllib.request as _url
+
+ req = _url.Request(
+ running["base_url"].rsplit("/v1", 1)[0] + "/models",
+ headers={"Authorization": f"Bearer {running.get('api_key', '')}"})
+ with _url.urlopen(req, timeout=3) as r:
+ data = json.loads(r.read())
+ loaded = {
+ m["id"]: m.get("status", {}).get("value", "unknown")
+ for m in data.get("data", [])
+ # Everything resident or becoming resident: 'loading' renders
+ # as its own state in the pane (a 20-GB load in flight is the
+ # single most important thing the pane can show).
+ if m.get("status", {}).get("value") in ("loaded", "ready", "loading")
+ }
+ # How each loaded model is actually running: the granted window
+ # from the child itself, and the plan's spill facts from the
+ # preset decision. The pane shows this verbatim — placement is
+ # the difference between 'fast' and 'why is my CPU busy', so it
+ # must be inspectable, not inferred from Task Manager.
+ from hermes_cli.local_runtime.presets import read_preset_decisions
+
+ decisions = read_preset_decisions()
+ for model_id in loaded:
+ entry_facts: Dict[str, Any] = {}
+ plan = decisions.get(model_id)
+ if plan is not None:
+ entry_facts["window"] = plan.window
+ entry_facts["window_label"] = f"{plan.window // 1024}K"
+ entry_facts["spilled"] = plan.spilled
+ if loaded[model_id] in ("loaded", "ready"):
+ try:
+ preq = _url.Request(
+ running["base_url"].rsplit("/v1", 1)[0]
+ + f"/props?model={model_id}",
+ headers={"Authorization":
+ f"Bearer {running.get('api_key', '')}"})
+ with _url.urlopen(preq, timeout=3) as pr:
+ props = json.loads(pr.read())
+ n_ctx = (props.get("default_generation_settings", {})
+ .get("n_ctx"))
+ if n_ctx:
+ entry_facts["granted_window"] = int(n_ctx)
+ entry_facts["granted_window_label"] = f"{int(n_ctx) // 1024}K"
+ except Exception: # noqa: BLE001
+ pass
+ if entry_facts:
+ placement[model_id] = entry_facts
+ except Exception as exc: # noqa: BLE001
+ # Never silent: an empty dict here renders as 'Not in memory'
+ # on a machine whose VRAM is visibly full.
+ logger.warning("loaded-models read failed: %r", exc)
+ loaded = {}
+
+ # The active main model, when it is one of ours (config authority: the
+ # same model.provider + model.default that /api/model/set writes).
+ active_model_id = None
+ try:
+ config = _load_config()
+ model_section = (config or {}).get("model") or {}
+ if str(model_section.get("provider", "")).strip().lower() in (
+ "llamacpp", "llama.cpp", "llama-cpp"):
+ active_model_id = str(
+ model_section.get("default") or model_section.get("name") or ""
+ ).strip() or None
+ except Exception: # noqa: BLE001
+ pass
+
+ return {
+ "enabled": bool(section.get("enabled")),
+ "tag": tag,
+ "configured_tag": configured_tag,
+ "update_available": update_available,
+ "runtime_installed": runtime_installed,
+ "runtime_backend": runtime_backend,
+ "server_running": running is not None,
+ "server_base_url": (running or {}).get("base_url"),
+ "active_model_id": active_model_id,
+ "loaded_models": loaded,
+ # Live load progress per model (SSE-fed): {model_id: {stage, value,
+ # percent}}. The chat's loading bar and the picker rows poll this.
+ "loading": _loading_progress(),
+ "placement": placement,
+ "models": staged,
+ "models_dir": str(mdir),
+ }
+
+
+def _loading_progress() -> Dict[str, Any]:
+ try:
+ from hermes_cli.local_runtime.load_progress import get_loading_progress
+
+ return get_loading_progress()
+ except Exception: # noqa: BLE001 — progress is garnish, never a 500
+ return {}
+
+
+# ── hardware: what this machine can do ───────────────────────
+
+
+@router.get("/api/local-models/hardware")
+def local_models_hardware():
+ """The budget as plain facts. Polled by the pane and the statusbar
+ resource item (throttled client-side). Sync def on purpose: the body
+ shells out to nvidia-smi and probes budgets — threadpool, not loop."""
+ from hermes_cli.local_runtime.hardware import probe_budget, _nvidia_vram, _ram_bytes
+
+ budget = probe_budget()
+ ram_total, ram_avail = _ram_bytes()
+ out = {
+ "uma": budget.uma,
+ "vram_total_bytes": budget.total_device_bytes,
+ "vram_usable_bytes": budget.usable_vram_bytes,
+ "ram_total_bytes": ram_total,
+ "ram_available_bytes": ram_avail,
+ "vram_label": _human_gb(budget.total_device_bytes),
+ "gpu_name": None,
+ "gpu_util_percent": None,
+ "vram_used_bytes": None,
+ }
+ # GPU identity + live utilization (NVIDIA; other vendors degrade to None
+ # and the UI hides those readouts).
+ try:
+ import subprocess
+
+ from hermes_cli.local_runtime.hardware import _nvidia_smi_path
+
+ smi_exe = _nvidia_smi_path()
+ smi = subprocess.run(
+ [smi_exe, "--query-gpu=name,utilization.gpu,memory.used",
+ "--format=csv,noheader,nounits"],
+ capture_output=True, text=True, timeout=5) if smi_exe else None
+ if smi and smi.returncode == 0 and smi.stdout.strip():
+ name, util, used_mib = (x.strip() for x in smi.stdout.strip().splitlines()[0].split(","))
+ out["gpu_name"] = name
+ out["gpu_util_percent"] = int(util)
+ out["vram_used_bytes"] = int(used_mib) << 20
+ except Exception: # noqa: BLE001
+ pass
+ return out
+
+
+# ── catalog: priced for THIS machine before download ─────────
+
+
+@router.get("/api/local-models/catalog")
+def local_models_catalog():
+ """Every entry answers the user's three questions up front: how big is
+ the download, will it fit, and what context/speed shape will I get —
+ computed from the catalog's measured numbers + this machine's
+ budget. Hardware-aware quant selection: the row advertises the BEST
+ build for this machine (highest quality that runs fully on the GPU at
+ the 64K floor; else the smallest that works, spilled and priced). No
+ entry is hidden; unaffordable models show WHY. Sync def on purpose:
+ probe_budget + catalog I/O block — threadpool, not loop."""
+ from hermes_cli.local_runtime.catalog import (
+ CATALOG,
+ recommended_entry,
+ refresh_catalog_soon,
+ select_variant,
+ )
+ from hermes_cli.local_runtime.context_policy import (
+ RUNTIME_OVERHEAD_BYTES,
+ initial_window,
+ ub_logits_bytes,
+ )
+ from hermes_cli.local_runtime.estimator import PhysicsRefusal
+ from hermes_cli.local_runtime.hardware import probe_budget
+
+ # This request serves the catalog already in memory; a TTL-gated
+ # background fetch from the repo lands new entries for the next one
+ # (day-0 models reach the pane without an app release).
+ refresh_catalog_soon()
+
+ # Planning budget: price against machine capacity, not live-free VRAM.
+ # A loaded model must not make the catalog call every row unaffordable.
+ budget = probe_budget(planning=True)
+ # The default pick for THIS machine: quality-ranked, fit- and
+ # speed-gated (recommended_entry). Engine-gated entries can't be
+ # activated today, so they can't be the recommendation either. The
+ # reason key ships with the row — the Recommended badge's tooltip is
+ # the branch that actually fired, not a re-derivation that can drift.
+ eligible = tuple(e for e in CATALOG if not _engine_too_old(e.min_engine))
+ picked = recommended_entry(budget, eligible)
+ recommended = picked[0].id if picked is not None else None
+ recommended_reason = picked[1] if picked is not None else None
+ # Completeness-checked staging (split parts all present) — the same
+ # answer the picker and the router see, so a mid-download model never
+ # reads as downloaded here.
+ from hermes_cli.local_runtime.bootstrap import staged_model_ids
+
+ staged_ids = set(staged_model_ids())
+ entries = []
+ for entry in CATALOG:
+ choice = select_variant(entry, budget)
+ # Any variant of this family already on disk counts as downloaded
+ # (split variants stage under their first part).
+ downloaded_variant = next(
+ (v for v in entry.variants if v.model_id in staged_ids), None)
+ row: Dict[str, Any] = {
+ "id": entry.id,
+ "display_name": entry.display_name,
+ "description": entry.description,
+ "native_context": entry.n_ctx_train,
+ "native_context_label": f"{entry.n_ctx_train // 1024}K",
+ "recommended": entry.id == recommended,
+ "recommended_reason": recommended_reason if entry.id == recommended else None,
+ "downloaded": downloaded_variant is not None,
+ "downloaded_model_id": downloaded_variant.model_id if downloaded_variant else None,
+ "downloaded_quant": downloaded_variant.quant if downloaded_variant else None,
+ "mtp": entry.mtp,
+ "vision": entry.mmproj is not None,
+ # Day-0 architectures need the llama.cpp release where their
+ # support landed. True gates download/activate in the pane
+ # until the engine updates; the row still renders (visible +
+ # explained beats hidden).
+ "needs_engine": _engine_too_old(entry.min_engine),
+ "min_engine": entry.min_engine or None,
+ }
+ if choice is None:
+ smallest = min(entry.variants, key=lambda v: v.size_bytes)
+ smallest_total = entry.download_bytes(smallest)
+ row.update({
+ "fits": False,
+ "size_bytes": smallest_total,
+ "size_label": _human_gb(smallest_total),
+ "fit_summary": "Needs more memory than this machine has",
+ "fit_detail": (f"even the most compact build ({smallest.quant}, "
+ f"{_human_gb(smallest_total)}) exceeds GPU + system memory"),
+ })
+ entries.append(row)
+ continue
+
+ variant = choice.variant
+ profile = entry.profile(variant)
+ # Same overhead the launch decision prices (runtime buffers +
+ # vision projector + the microbatch/MTP logits buffers): the row
+ # must advertise the window the model will actually get, not a
+ # paper number the server's own fit then shaves down.
+ overhead = (RUNTIME_OVERHEAD_BYTES
+ + (entry.mmproj.size_bytes if entry.mmproj else 0)
+ + ub_logits_bytes(entry.n_vocab, mtp_capable=entry.mtp))
+ decision = initial_window(profile, budget, overhead_bytes=overhead)
+ download_total = entry.download_bytes(variant)
+ row.update({
+ "fits": True,
+ "model_id": variant.model_id,
+ "quant": variant.quant,
+ "quant_validated": variant.validated,
+ "size_bytes": download_total,
+ "size_label": _human_gb(download_total),
+ "variant_count": len(entry.variants),
+ })
+ if choice.reason_key == "best-large-window":
+ row["quant_reason"] = (
+ f"Recommended build ({variant.quant}) — the quant class this "
+ "engine is optimized for; runs fully on your GPU with a "
+ "large context window")
+ elif choice.reason_key == "best-fits":
+ row["quant_reason"] = (
+ f"Recommended build ({variant.quant}) — the quant class this "
+ "engine is optimized for; runs fully on your GPU")
+ else:
+ row["quant_reason"] = (
+ f"Compact build sized for this machine ({variant.quant}) — "
+ "larger than GPU memory, runs slower")
+ if not isinstance(decision, PhysicsRefusal):
+ row["start_window"] = decision.window
+ row["start_window_label"] = f"{decision.window // 1024}K"
+ row["spilled"] = decision.spilled
+ if decision.window >= entry.n_ctx_train:
+ shape = f"runs at its full {row['native_context_label']} context"
+ else:
+ shape = (f"starts at {row['start_window_label']} and grows toward "
+ f"{row['native_context_label']} as you use it")
+ if decision.spilled:
+ shape += " (larger than your GPU memory — runs slower)"
+ row["fit_summary"] = shape
+ else:
+ row["fit_summary"] = row["quant_reason"]
+ entries.append(row)
+ return {"models": entries}
+
+
+# ── runtime install (job) ────────────────────────────────────
+
+
+class RuntimeInstallBody(BaseModel):
+ backend: Optional[str] = None # None/auto -> detect
+
+
+def _runtime_progress_hook(job: Dict[str, Any]):
+ """Adapter: ensure_runtime_installed's progress stream -> job fields.
+
+ Throttled to ~4 updates/s. The byte counters are CUMULATIVE across the
+ plan: a multi-asset engine (CUDA zip + cudart zip) reads as one growing
+ download, not a bar that restarts at zero per asset. The total grows as
+ each asset's size becomes known (sizes arrive with the response, not
+ the plan). Unpack/verify keep the download's counters in place — the
+ stage text says what's happening, and a bar that bounces back to zero
+ after the bytes finished reads as a failure."""
+ state = {"last": 0.0, "banked": 0, "asset": None, "asset_total": 0}
+
+ def hook(stage: str, done: int, total: int, label: str) -> None:
+ now = time.monotonic()
+ if now - state["last"] < 0.25 and done < total:
+ return
+ state["last"] = now
+ suffix = f" ({label})" if label else ""
+ if stage == "download":
+ if label != state["asset"]:
+ # Previous asset finished: bank its bytes so the counters
+ # keep climbing instead of restarting for the next asset.
+ state["banked"] += state["asset_total"]
+ state["asset"] = label
+ state["asset_total"] = total or done
+ plan_done = state["banked"] + done
+ plan_total = state["banked"] + (total or 0)
+ job["phase"] = "downloading-runtime"
+ if total:
+ job["detail"] = (f"Downloading the local engine{suffix} — "
+ f"{_human_gb(plan_done)} of {_human_gb(plan_total)}")
+ else:
+ job["detail"] = (f"Downloading the local engine{suffix} — "
+ f"{_human_gb(plan_done)}")
+ job["done_bytes"] = plan_done
+ job["total_bytes"] = plan_total or None
+ elif stage == "extract":
+ job["phase"] = "unpacking-runtime"
+ pct = f" — {min(100, round(done / total * 100))}%" if total else ""
+ job["detail"] = f"Unpacking the engine{suffix}{pct}"
+ else: # verify
+ job["phase"] = "verifying-runtime"
+ job["detail"] = f"Verifying the engine{suffix}"
+
+ return hook
+
+
+@router.post("/api/local-models/runtime/install")
+async def local_models_runtime_install(body: RuntimeInstallBody):
+ from hermes_cli.local_runtime.binaries import (
+ default_tag,
+ resolve_assets,
+ select_backend,
+ )
+ from hermes_cli.local_runtime.bootstrap import _detect_gpu_vendor
+
+ section = _runtime_section()
+ tag = section.get("tag") or default_tag()
+ backend = body.backend or section.get("backend", "auto")
+ if backend == "auto":
+ backend = select_backend(_detect_gpu_vendor())
+ # Resolve first so an impossible combination fails the POST, not the job.
+ try:
+ plan = resolve_assets(tag, backend)
+ except Exception as exc: # noqa: BLE001
+ raise HTTPException(status_code=400, detail=str(exc))
+
+ job = _job("runtime-install", f"llama.cpp {tag} ({backend})")
+
+ def _run():
+ try:
+ from hermes_cli.local_runtime.binaries import (
+ ensure_runtime_installed,
+ installed_tags,
+ prune_old_tags,
+ )
+
+ previous = installed_tags()
+ job["phase"] = "downloading"
+ job["detail"] = f"Fetching {len(plan.assets)} package(s) for {backend}"
+ ensure_runtime_installed(tag, backend,
+ progress=_runtime_progress_hook(job))
+
+ # Engine update path: a server already running on an older tag
+ # moves to the new one now — the click was the consent. Fresh
+ # installs (no server) skip this; Use/boot handles their start.
+ restarted = False
+ try:
+ from hermes_cli.local_runtime.bootstrap import (
+ ensure_local_runtime,
+ get_supervisor,
+ shutdown_local_runtime,
+ )
+
+ sup = get_supervisor()
+ if sup is not None and previous and tag not in previous:
+ job["phase"] = "restarting"
+ job["detail"] = "Switching the running server to the new build"
+ shutdown_local_runtime()
+ ensure_local_runtime(_load_config(), force=True)
+ restarted = True
+ except Exception as exc: # noqa: BLE001
+ # The new build is installed either way; the next boot serves
+ # it. Never fail the job on the restart nicety.
+ logger.warning("post-update restart skipped: %s", exc)
+
+ # N-1 retention, only after the new tag verified: keep it and the
+ # newest previous build as the rollback pin target.
+ try:
+ keep = [tag] + [t for t in previous if t != tag][:1]
+ prune_old_tags(keep)
+ except Exception as exc: # noqa: BLE001
+ logger.warning("runtime prune skipped: %s", exc)
+
+ job["phase"] = "done"
+ job["status"] = "done"
+ job["detail"] = (f"llama.cpp {tag} ready ({backend})"
+ + (" — server restarted on the new build" if restarted else ""))
+ except Exception as exc: # noqa: BLE001
+ logger.warning("runtime install failed: %s", exc)
+ job["status"] = "error"
+ job["error"] = str(exc)
+
+ threading.Thread(target=_run, daemon=True, name="lr-runtime-install").start()
+ return {"job_id": job["job_id"], "backend": backend, "tag": tag}
+
+
+# ── model download (job with byte progress) ──────────────────
+
+
+class ModelDownloadBody(BaseModel):
+ model_id: str
+
+
+@router.post("/api/local-models/download")
+async def local_models_download(body: ModelDownloadBody):
+ """Accepts either a family id (downloads this machine's selected
+ variant) or an exact variant model_id."""
+ from hermes_cli.local_runtime.catalog import (
+ CATALOG,
+ catalog_by_id,
+ select_variant,
+ )
+ from hermes_cli.local_runtime.hardware import probe_budget
+
+ entry = catalog_by_id().get(body.model_id)
+ variant = None
+ if entry is not None:
+ if _engine_too_old(entry.min_engine):
+ raise HTTPException(
+ status_code=409,
+ detail=(f"{entry.display_name} needs llama.cpp {entry.min_engine} "
+ f"or newer — update the engine first"))
+ # Same planning budget as the catalog — the user downloads exactly
+ # the build the row advertised.
+ choice = select_variant(entry, probe_budget(planning=True))
+ if choice is None:
+ raise HTTPException(status_code=409,
+ detail=f"no variant of {entry.id} fits this machine")
+ variant = choice.variant
+ else:
+ for candidate in CATALOG:
+ for v in candidate.variants:
+ if v.model_id == body.model_id:
+ entry, variant = candidate, v
+ break
+ if variant:
+ break
+ if entry is None or variant is None:
+ raise HTTPException(status_code=404, detail=f"unknown model {body.model_id}")
+
+ from hermes_cli.local_runtime.bootstrap import assets_dir, staged_model_ids
+
+ if variant.model_id in staged_model_ids():
+ return {"job_id": None, "already_downloaded": True, "model_id": variant.model_id}
+
+ # Everything this variant needs: split parts + mmproj/draft assets.
+ plan = [] # (url, dest, bytes)
+ for asset in variant.files:
+ plan.append((f"https://huggingface.co/{entry.repo}/resolve/main/{asset.path}",
+ _models_dir() / asset.local_name, asset.size_bytes))
+ for asset in (entry.mmproj, entry.draft):
+ if asset is not None:
+ plan.append((f"https://huggingface.co/{entry.repo}/resolve/main/{asset.path}",
+ assets_dir() / asset.local_name, asset.size_bytes))
+
+ total = sum(p[2] for p in plan)
+ job = _job("model-download", f"{entry.display_name} ({variant.quant})",
+ model_id=entry.id)
+ job["total_bytes"] = total
+
+ def _run():
+ try:
+ job["phase"] = "downloading"
+ job["detail"] = f"{entry.display_name} — {_human_gb(total)}"
+ done_before = 0
+ for url, dest, size in plan:
+ if dest.exists():
+ done_before += size
+ job["done_bytes"] = done_before
+ continue
+ download_file(url, dest, job,
+ base_done=done_before, keep_totals=True)
+ job["phase"] = "downloading"
+ done_before += size
+ job["done_bytes"] = done_before
+ job["phase"] = "done"
+ job["status"] = "done"
+ job["detail"] = f"{entry.display_name} ready"
+ # A running router only scans models at spawn —
+ # bounce it so the new model is servable
+ # immediately instead of 400ing until the next app restart.
+ try:
+ from hermes_cli.local_runtime.bootstrap import refresh_local_runtime
+
+ refresh_local_runtime()
+ except Exception: # noqa: BLE001
+ logger.debug("post-download runtime refresh skipped", exc_info=True)
+ except Exception as exc: # noqa: BLE001
+ logger.warning("model download failed: %s", exc)
+ job["status"] = "error"
+ job["error"] = str(exc)
+
+ threading.Thread(target=_run, daemon=True, name="lr-model-download").start()
+ return {"job_id": job["job_id"], "model_id": variant.model_id}
+
+
+@router.delete("/api/local-models/models/{model_id}")
+async def local_models_delete(model_id: str):
+ """Remove a staged model: every split part plus its private assets.
+ A running router keeps serving from its spawn-time scan, so bounce it
+ off the request thread — deleting the active file mid-serve is the
+ kind of stale state the refresh exists for."""
+ files = _variant_files_on_disk(model_id)
+ if not files:
+ raise HTTPException(status_code=404, detail="model not found")
+ for path in files:
+ path.unlink(missing_ok=True)
+ # Growth state dies with the model: a re-download starts back at its
+ # zero-spill window instead of inheriting a stale grown one.
+ try:
+ from hermes_cli.local_runtime.growth import clear_window_override
+
+ clear_window_override(model_id)
+ except Exception: # noqa: BLE001
+ logger.debug("window-override clear skipped", exc_info=True)
+
+ def _refresh():
+ try:
+ from hermes_cli.local_runtime.bootstrap import refresh_local_runtime
+
+ refresh_local_runtime()
+ except Exception: # noqa: BLE001
+ logger.debug("post-delete runtime refresh skipped", exc_info=True)
+
+ threading.Thread(target=_refresh, daemon=True, name="lr-post-delete").start()
+ return {"ok": True}
+
+
+# ── server lifecycle: turn the engine on/off ─────────────────
+
+
+class ServerActionBody(BaseModel):
+ action: str # "stop" | "start"
+
+
+# ── quickstart: one click from nothing to a working default ──
+
+
+class QuickstartBody(BaseModel):
+ model_id: str | None = None # default: the catalog's recommended entry
+
+
+# One quickstart at a time: the job sequences installs, downloads, a
+# server bounce, and a config write — two racing runs would interleave
+# all four. Held for the job's lifetime, released in the worker.
+_QUICKSTART_LOCK = threading.Lock()
+
+
+@router.post("/api/local-models/quickstart")
+async def local_models_quickstart(body: QuickstartBody):
+ """The dummy-proof path: one job that installs the runtime (if
+ missing), downloads this machine's build of the recommended model
+ (if missing), and makes it the default for new chats. Each leg is
+ the same code the individual routes run — this route only sequences
+ them, so 'Configure' (the existing pane) and quickstart can never
+ disagree about what gets installed.
+
+ Preflight rejects (no servable entry, engine too old) fail the POST
+ synchronously so the button can explain itself; everything slow runs
+ in the job with the usual phase/byte progress.
+ """
+ from hermes_cli.local_runtime.binaries import (
+ default_tag,
+ installed_tags,
+ resolve_assets,
+ select_backend,
+ )
+ from hermes_cli.local_runtime.bootstrap import (
+ _detect_gpu_vendor,
+ assets_dir,
+ staged_model_ids,
+ )
+ from hermes_cli.local_runtime.catalog import (
+ CATALOG,
+ catalog_by_id,
+ recommended_entry,
+ select_variant,
+ )
+ from hermes_cli.local_runtime.hardware import probe_budget
+
+ # Resolve the target entry: explicit id, else this machine's
+ # recommendation (quality-ranked, fit- and speed-gated), else the
+ # first catalog entry this machine can serve.
+ budget = probe_budget(planning=True)
+ entry = None
+ if body.model_id:
+ entry = catalog_by_id().get(body.model_id)
+ if entry is None:
+ raise HTTPException(status_code=404,
+ detail=f"unknown model {body.model_id}")
+ candidates = [entry]
+ else:
+ eligible = tuple(e for e in CATALOG if not _engine_too_old(e.min_engine))
+ picked = recommended_entry(budget, eligible)
+ best = picked[0] if picked is not None else None
+ candidates = ([best] if best is not None else []) + [
+ e for e in CATALOG if best is None or e.id != best.id]
+ chosen = None
+ for candidate in candidates:
+ choice = select_variant(candidate, budget)
+ if choice is not None and not _engine_too_old(candidate.min_engine):
+ chosen = (candidate, choice.variant)
+ break
+ if chosen is None:
+ raise HTTPException(
+ status_code=409,
+ detail="no catalog model fits this machine — open Local Models "
+ "to browse for a smaller build")
+ entry, variant = chosen
+
+ section = _runtime_section()
+ tag = section.get("tag") or default_tag()
+ backend = section.get("backend", "auto")
+ if backend == "auto":
+ backend = select_backend(_detect_gpu_vendor())
+ need_runtime = not installed_tags()
+ if need_runtime:
+ # Same preflight as /runtime/install: impossible combos fail the POST.
+ try:
+ resolve_assets(tag, backend)
+ except Exception as exc: # noqa: BLE001
+ raise HTTPException(status_code=400, detail=str(exc))
+
+ need_download = variant.model_id not in staged_model_ids()
+ download_plan = [] # (url, dest, bytes)
+ if need_download:
+ for asset in variant.files:
+ download_plan.append(
+ (f"https://huggingface.co/{entry.repo}/resolve/main/{asset.path}",
+ _models_dir() / asset.local_name, asset.size_bytes))
+ for asset in (entry.mmproj, entry.draft):
+ if asset is not None:
+ download_plan.append(
+ (f"https://huggingface.co/{entry.repo}/resolve/main/{asset.path}",
+ assets_dir() / asset.local_name, asset.size_bytes))
+
+ if not _QUICKSTART_LOCK.acquire(blocking=False):
+ raise HTTPException(status_code=409,
+ detail="Setup is already running")
+
+ job = _job("quickstart", entry.display_name, model_id=entry.id)
+ job["total_bytes"] = sum(p[2] for p in download_plan) or None
+
+ def _run():
+ try:
+ if need_runtime:
+ from hermes_cli.local_runtime.binaries import ensure_runtime_installed
+
+ job["phase"] = "installing-runtime"
+ job["detail"] = "Installing the local engine"
+ ensure_runtime_installed(tag, backend,
+ progress=_runtime_progress_hook(job))
+
+ if need_download:
+ job["phase"] = "downloading"
+ total = sum(p[2] for p in download_plan)
+ # The runtime leg repurposed the byte counters for its own
+ # stages — reset them to the model plan before download.
+ job["done_bytes"] = 0
+ job["total_bytes"] = total
+ job["detail"] = f"{entry.display_name} — {_human_gb(total)}"
+ done_before = 0
+ for url, dest, size in download_plan:
+ if dest.exists():
+ done_before += size
+ job["done_bytes"] = done_before
+ continue
+ download_file(url, dest, job,
+ base_done=done_before, keep_totals=True)
+ job["phase"] = "downloading"
+ done_before += size
+ job["done_bytes"] = done_before
+
+ # Activate: same sequence as /activate's job body.
+ from hermes_cli.config import load_config, save_config
+ from hermes_cli.local_runtime.bootstrap import (
+ ensure_local_runtime,
+ refresh_local_runtime,
+ )
+
+ job["phase"] = "starting-server"
+ job["detail"] = "Starting the local server"
+ config = load_config()
+ config.setdefault("local_runtime", {})["enabled"] = True
+ save_config(config)
+ sup = ensure_local_runtime(config, force=True)
+ if sup is None and _state_endpoint() is None:
+ raise RuntimeError(
+ "The local server could not start — open Local Models for details")
+ if sup is not None:
+ try:
+ if variant.model_id not in sup.models():
+ job["detail"] = "Refreshing the local server"
+ refresh_local_runtime()
+ except Exception: # noqa: BLE001
+ logger.debug("quickstart rescan check skipped", exc_info=True)
+
+ job["phase"] = "setting-default"
+ job["detail"] = "Making it your default"
+ from hermes_cli.web_deps import late
+
+ late("_apply_model_assignment_sync")(
+ "main", "llamacpp", variant.model_id, "", "", "")
+
+ job["phase"] = "done"
+ job["status"] = "done"
+ job["detail"] = f"{entry.display_name} is ready — new chats use it"
+ except Exception as exc: # noqa: BLE001
+ logger.warning("quickstart failed: %s", exc)
+ job["status"] = "error"
+ job["error"] = str(exc)
+ finally:
+ _QUICKSTART_LOCK.release()
+
+ threading.Thread(target=_run, daemon=True, name="lr-quickstart").start()
+ return {
+ "job_id": job["job_id"],
+ "model_id": entry.id,
+ "display_name": entry.display_name,
+ "needs_runtime": need_runtime,
+ "needs_download": need_download,
+ "download_bytes": sum(p[2] for p in download_plan),
+ }
+
+
+@router.post("/api/local-models/server")
+async def local_models_server(body: ServerActionBody):
+ """Turn the local engine off (stop the server, free ALL GPU memory,
+ and disable auto-start) or back on. The off switch is the whole-engine
+ counterpart of per-model eject — and unlike eject it IS durable: the
+ user said off, so boots stay off until they say on."""
+ import asyncio
+
+ from hermes_cli.config import load_config, save_config
+
+ action = (body.action or "").strip().lower()
+ if action not in ("stop", "start"):
+ raise HTTPException(status_code=400, detail="action must be 'stop' or 'start'")
+
+ def _stop():
+ from hermes_cli.local_runtime.bootstrap import (
+ get_supervisor,
+ shutdown_local_runtime,
+ )
+
+ sup = get_supervisor()
+ if sup is not None:
+ shutdown_local_runtime()
+ else:
+ # Server owned by another process (or an orphan): best-effort
+ # terminate via the state file's pid, then clear the state.
+ endpoint = _state_endpoint()
+ if endpoint is not None:
+ try:
+ import psutil # type: ignore
+
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ state = json.loads(state_path().read_text(encoding="utf-8"))
+ pid = int(state.get("pid") or 0)
+ if pid > 0 and psutil.pid_exists(pid):
+ psutil.Process(pid).terminate()
+ state_path().unlink(missing_ok=True)
+ except Exception: # noqa: BLE001
+ pass
+ config = load_config()
+ config.setdefault("local_runtime", {})["enabled"] = False
+ save_config(config)
+
+ def _start():
+ from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
+
+ config = load_config()
+ config.setdefault("local_runtime", {})["enabled"] = True
+ save_config(config)
+ sup = ensure_local_runtime(config, force=True)
+ if sup is None and _state_endpoint() is None:
+ raise RuntimeError("The local server could not start — check the "
+ "runtime is installed")
+
+ try:
+ await asyncio.to_thread(_stop if action == "stop" else _start)
+ except Exception as exc: # noqa: BLE001
+ raise HTTPException(status_code=502, detail=str(exc)) from exc
+ return {"ok": True, "action": action}
+
+
+# ── activate: make a downloaded model THE model ──────────────
+
+
+class ModelEjectBody(BaseModel):
+ model_id: str
+
+
+@router.post("/api/local-models/eject")
+def local_models_eject(body: ModelEjectBody):
+ """Free a loaded model's GPU memory now. Nothing reloads it except
+ demand — the next message to it (residency v2: no automatic loading
+ exists anywhere). Sync def on purpose: the fallback path blocks on a
+ urlopen with a 120s timeout — threadpool, never the event loop."""
+ from hermes_cli.local_runtime.bootstrap import get_supervisor
+
+ sup = get_supervisor()
+ if sup is not None:
+ try:
+ sup.unload_model(body.model_id)
+ return {"ok": True}
+ except Exception as exc: # noqa: BLE001
+ raise HTTPException(status_code=502, detail=str(exc)) from exc
+
+ # Server owned by another process (or state-file only): drive the
+ # router directly with the persisted endpoint.
+ endpoint = _state_endpoint()
+ if endpoint is None:
+ raise HTTPException(status_code=409, detail="local server is not running")
+ try:
+ import urllib.request as _url
+
+ req = _url.Request(
+ endpoint["base_url"].rsplit("/v1", 1)[0] + "/models/unload",
+ data=json.dumps({"model": body.model_id}).encode(),
+ headers={"Content-Type": "application/json",
+ "Authorization": f"Bearer {endpoint.get('api_key', '')}"},
+ method="POST")
+ with _url.urlopen(req, timeout=120):
+ pass
+ return {"ok": True}
+ except Exception as exc: # noqa: BLE001
+ raise HTTPException(status_code=502, detail=str(exc)) from exc
+
+
+class ModelActivateBody(BaseModel):
+ model_id: str # exact variant id (a staged .gguf stem)
+
+
+@router.post("/api/local-models/activate")
+async def local_models_activate(body: ModelActivateBody):
+ """Make a downloaded model the default for new chats. Pure selection
+ (residency v2): a config write through the same machinery as
+ /api/model/set, plus making sure the server is up. NO model loading —
+ models load on first inference, always; an empty router costs nothing.
+ Fast enough to be synchronous-feeling, but kept as a job for UI
+ continuity."""
+ # Split variants stage under their first part — resolve like the rest
+ # of the routes instead of assuming a single flat file.
+ from hermes_cli.local_runtime.bootstrap import staged_model_ids
+
+ if body.model_id not in staged_model_ids():
+ raise HTTPException(status_code=404, detail=f"{body.model_id} is not downloaded")
+
+ job = _job("model-activate", body.model_id, model_id=body.model_id)
+
+ def _run():
+ try:
+ from hermes_cli.config import load_config, save_config
+ from hermes_cli.local_runtime.bootstrap import (
+ ensure_local_runtime,
+ refresh_local_runtime,
+ )
+
+ job["phase"] = "starting-server"
+ job["detail"] = "Starting the local server"
+ config = load_config()
+ sup = ensure_local_runtime(config, force=True)
+ if sup is None:
+ if _state_endpoint() is None:
+ raise RuntimeError(
+ "The local server could not start — check the runtime is installed")
+
+ # Self-heal a stale router: the model list is spawn-only, so a
+ # server started before this model finished downloading can't
+ # serve it. If the router doesn't know the model, bounce it.
+ if sup is not None:
+ try:
+ if body.model_id not in sup.models():
+ job["detail"] = "Refreshing the local server"
+ refresh_local_runtime()
+ except Exception: # noqa: BLE001
+ logger.debug("activate rescan check skipped", exc_info=True)
+
+ job["phase"] = "setting-default"
+ job["detail"] = "Making it your default"
+ config = load_config()
+ config.setdefault("local_runtime", {})["enabled"] = True
+ save_config(config)
+ from hermes_cli.web_deps import late
+
+ late("_apply_model_assignment_sync")(
+ "main", "llamacpp", body.model_id, "", "", "")
+
+ job["phase"] = "done"
+ job["status"] = "done"
+ job["detail"] = f"{body.model_id} is the default for new chats"
+ except Exception as exc: # noqa: BLE001
+ logger.warning("model activate failed: %s", exc)
+ job["status"] = "error"
+ job["error"] = str(exc)
+
+ threading.Thread(target=_run, daemon=True, name="lr-model-activate").start()
+ return {"job_id": job["job_id"]}
+
+
+# ── job polling ──────────────────────────────────────────────
+
+
+@router.get("/api/local-models/jobs")
+async def local_models_jobs():
+ """All recent jobs, running first — the pane and the app-level poller
+ rediscover in-flight work here after a remount or app restart."""
+ with _JOBS_LOCK:
+ jobs = sorted(_JOBS.values(),
+ key=lambda j: (j["status"] != "running", -j["started_at"]))
+ out = []
+ for job in jobs[:20]:
+ entry = dict(job)
+ if entry["total_bytes"]:
+ entry["percent"] = min(100, round(entry["done_bytes"] / entry["total_bytes"] * 100))
+ out.append(entry)
+ return {"jobs": out}
+
+
+@router.get("/api/local-models/jobs/{job_id}")
+async def local_models_job(job_id: str):
+ with _JOBS_LOCK:
+ job = _JOBS.get(job_id)
+ if job is None:
+ raise HTTPException(status_code=404, detail="job not found")
+ out = dict(job)
+ if out["total_bytes"]:
+ out["percent"] = min(100, round(out["done_bytes"] / out["total_bytes"] * 100))
+ return out
+
+
+# ── Hugging Face browser: search, repo files, arbitrary download ─
+
+
+@router.get("/api/local-models/search")
+async def local_models_search(q: str, limit: int = 20):
+ """Full-text HF search over GGUF models — the firehose behind the
+ curated catalog. The pane's per-quant fit pills come from the
+ repo-files call once the user opens a hit."""
+ from starlette.concurrency import run_in_threadpool
+
+ from hermes_cli.local_runtime.hf_browse import search_models
+
+ if not q.strip():
+ return {"hits": []}
+ try:
+ hits = await run_in_threadpool(search_models, q, limit)
+ except Exception as exc: # noqa: BLE001
+ raise HTTPException(status_code=502,
+ detail=f"Hugging Face search unavailable: {exc}") from exc
+ return {"hits": [h.__dict__ for h in hits]}
+
+
+@router.get("/api/local-models/search/files")
+async def local_models_search_files(repo: str):
+ """The servable GGUFs in one HF repo with a rough pre-download fit
+ verdict per quant (file size + conservative fill-ins — the GGUF
+ header refines it after download)."""
+ from starlette.concurrency import run_in_threadpool
+
+ from hermes_cli.local_runtime.hardware import probe_budget
+ from hermes_cli.local_runtime.hf_browse import priced_repo_files
+
+ try:
+ groups = await run_in_threadpool(
+ priced_repo_files, repo, probe_budget(planning=True))
+ except Exception as exc: # noqa: BLE001
+ raise HTTPException(status_code=502,
+ detail=f"Could not list {repo}: {exc}") from exc
+ return {"files": [dict(g.__dict__, paths=list(g.paths)) for g in groups]}
+
+
+class BrowsedDownloadBody(BaseModel):
+ repo: str
+ paths: list[str] # one GGUF, or every part of a split, in order
+
+
+@router.post("/api/local-models/download-browsed")
+async def local_models_download_browsed(body: BrowsedDownloadBody):
+ """Download an arbitrary HF GGUF (browsed or pasted) into the managed
+ models dir. From the moment it lands it is a normal staged model: the
+ post-download bounce regenerates presets from its real header and the
+ fit policy owns its launch. No catalog entry — it serves 'unverified',
+ capabilities answered from the live server only."""
+ import re as _re
+
+ from hermes_cli.local_runtime.bootstrap import staged_model_ids
+
+ paths = [p for p in (body.paths or []) if p.lower().endswith(".gguf")]
+ if not paths:
+ raise HTTPException(status_code=422, detail="no .gguf files given")
+ first = paths[0].rsplit("/", 1)[-1]
+ model_id = _re.sub(r"-\d{5}-of-\d{5}\.gguf$", "", first, flags=_re.IGNORECASE)
+ model_id = model_id[:-5] if model_id.lower().endswith(".gguf") else model_id
+ if model_id in staged_model_ids():
+ return {"job_id": None, "already_downloaded": True, "model_id": model_id}
+
+ job = _job("model-download", f"{model_id} (from {body.repo})",
+ model_id=model_id)
+
+ def _run():
+ try:
+ job["phase"] = "downloading"
+ for p in paths:
+ url = (f"https://huggingface.co/{body.repo}"
+ f"/resolve/main/{urllib.parse.quote(p)}")
+ dest = _models_dir() / p.rsplit("/", 1)[-1]
+ if dest.exists():
+ continue
+ download_file(url, dest, job,
+ base_done=int(job.get("done_bytes") or 0),
+ keep_totals=bool(job.get("total_bytes")))
+ job["phase"] = "downloading"
+ job["phase"] = "done"
+ job["status"] = "done"
+ job["detail"] = f"{model_id} ready"
+ try:
+ from hermes_cli.local_runtime.bootstrap import refresh_local_runtime
+
+ refresh_local_runtime()
+ except Exception: # noqa: BLE001
+ logger.debug("post-download runtime refresh skipped", exc_info=True)
+ except Exception as exc: # noqa: BLE001
+ job["status"] = "error"
+ job["error"] = str(exc)
+
+ threading.Thread(target=_run, daemon=True, name="lm-download-browsed").start()
+ return {"job_id": job["job_id"], "model_id": model_id}
+
+
+class SideloadBody(BaseModel):
+ path: str # absolute path to a .gguf on this machine
+
+
+@router.post("/api/local-models/sideload")
+async def local_models_sideload(body: SideloadBody):
+ """Register a GGUF that already exists on this machine: link it into
+ the managed models dir (copy only when linking is impossible) and
+ bounce the router so it serves immediately. The original stays where
+ it is; delete-from-Hermes removes only our link."""
+ import os
+ import shutil
+
+ from starlette.concurrency import run_in_threadpool
+
+ src = Path(body.path)
+ if not src.is_file() or src.suffix.lower() != ".gguf":
+ raise HTTPException(status_code=422, detail="Pick a .gguf model file")
+ dest = _models_dir() / src.name
+ if dest.exists():
+ return {"ok": True, "model_id": dest.stem, "already_present": True}
+ dest.parent.mkdir(parents=True, exist_ok=True)
+ try:
+ os.link(src, dest) # hardlink: instant, no extra disk
+ except OSError:
+ try:
+ os.symlink(src, dest) # cross-volume fallback
+ except OSError:
+ await run_in_threadpool(shutil.copyfile, src, dest)
+ try:
+ from hermes_cli.local_runtime.bootstrap import refresh_local_runtime
+
+ refresh_local_runtime()
+ except Exception: # noqa: BLE001
+ logger.debug("post-sideload runtime refresh skipped", exc_info=True)
+ return {"ok": True, "model_id": dest.stem}
diff --git a/hermes_cli/web_routers/profiles.py b/hermes_cli/web_routers/profiles.py
index 0e08d63142..d037d86d6e 100644
--- a/hermes_cli/web_routers/profiles.py
+++ b/hermes_cli/web_routers/profiles.py
@@ -197,6 +197,11 @@ def _sidebar_singleflight_cache(func):
if cached is not miss:
return cached
result = func(*args, **kwargs)
+ # A 200 carrying errors[] is a FAILED profile scan, not a
+ # successful empty page. Caching it holds the empty recents in
+ # front of a store that has already recovered, for the whole TTL.
+ if isinstance(result, dict) and result.get("errors"):
+ return result
try:
snapshot = copy.deepcopy(result)
except Exception:
diff --git a/hermes_cli/web_routers/sessions.py b/hermes_cli/web_routers/sessions.py
index a4da40c3a1..657840ed62 100644
--- a/hermes_cli/web_routers/sessions.py
+++ b/hermes_cli/web_routers/sessions.py
@@ -31,7 +31,7 @@ from hermes_cli.web_models import (
SessionPrune,
SessionRename,
)
-from hermes_state import is_malformed_db_error
+from hermes_state import is_malformed_db_error, is_transient_sqlite_error
# Same logger the handlers used before extraction (identical logger object).
_log = logging.getLogger("hermes_cli.web_server")
@@ -197,6 +197,22 @@ def get_sessions(
db.close()
except HTTPException:
raise
+ except sqlite3.OperationalError as exc:
+ _log.exception("GET /api/sessions failed")
+ # 503, not 500: the store is busy, not gone. The desktop keeps the
+ # sidebar it already has instead of reading a 500 as an authoritative
+ # empty list. Retrying the OPEN here is deliberately not done — the
+ # bounded retry lives in SessionDB's read-only constructor, so every
+ # read-only opener gets it, not just this route.
+ transient = is_transient_sqlite_error(exc)
+ raise HTTPException(
+ status_code=503 if transient else 500,
+ detail=(
+ "Session store is busy (disk I/O or lock). Retry; the list was not cleared."
+ if transient
+ else "Internal server error"
+ ),
+ ) from exc
except Exception:
_log.exception("GET /api/sessions failed")
raise HTTPException(status_code=500, detail="Internal server error")
diff --git a/hermes_cli/web_server.py b/hermes_cli/web_server.py
index d5ad2e9ab6..b6a93a65dd 100644
--- a/hermes_cli/web_server.py
+++ b/hermes_cli/web_server.py
@@ -303,6 +303,17 @@ def _start_desktop_cron_ticker(stop_event: "threading.Event", interval: int = 60
profile_homes = list(profiles_to_serve(multiplex=True))
if len(profile_homes) > 1:
start_kwargs["profile_homes"] = profile_homes
+ # Stand down, per tick, for any profile whose OWN gateway is
+ # running: that gateway ticks it with live adapters, and the
+ # tick-lock race otherwise lets this adapter-less ticker win
+ # and deliver the job through the standalone path (#100489).
+ # Evaluated every cycle so a gateway starting/stopping later
+ # is picked up without a dashboard restart.
+ from hermes_cli.profiles import _check_gateway_running
+
+ start_kwargs["profile_gate"] = (
+ lambda _name, home: not _check_gateway_running(Path(home))
+ )
from hermes_logging import enable_profile_log_routing
enable_profile_log_routing(profile_homes)
@@ -320,6 +331,11 @@ def _start_desktop_cron_ticker(stop_event: "threading.Event", interval: int = 60
provider.start(stop_event, **start_kwargs)
+# Desktop `serve` only (start_server(start_mcp_discovery_after_bind=True)):
+# seconds after the READY sentinel before the MCP discovery thread starts.
+_DESKTOP_MCP_DISCOVERY_DELAY_S = 1.0
+
+
def _warm_gateway_module() -> None:
"""Pre-import heavy modules so the event loop is not stalled on first use.
@@ -511,6 +527,28 @@ async def _lifespan(app: "FastAPI"):
# sweeping stale sessions on schedule, independent of list requests.
auto_archive_task = asyncio.create_task(_auto_archive_ticker_loop())
+ # Managed local runtime: when the user opted in (local_runtime.enabled,
+ # set by the Local Models 'Use' action), bring the llama-server back up
+ # so a restart doesn't strand a llamacpp main model without a backend.
+ # Off-thread and best-effort: binary check + spawn + health poll must
+ # not delay the server socket, and failure falls back to configured
+ # cloud providers exactly like a cold start.
+ def _boot_local_runtime():
+ try:
+ from hermes_cli.config import load_config
+ from hermes_cli.local_runtime.bootstrap import ensure_local_runtime
+
+ # Server only — models load on first inference, always (residency
+ # design: downloaded = available; demand loads; idleness
+ # evicts). An empty router holds no VRAM; warming a model at
+ # boot would reload gigabytes nobody asked for yet.
+ ensure_local_runtime(load_config())
+ except Exception as exc: # noqa: BLE001
+ logging.getLogger(__name__).warning("local runtime boot failed: %s", exc)
+
+ threading.Thread(target=_boot_local_runtime, daemon=True,
+ name="local-runtime-boot").start()
+
try:
yield
finally:
@@ -523,6 +561,14 @@ async def _lifespan(app: "FastAPI"):
selftest_task.cancel()
auto_archive_task.cancel()
await PTY_REGISTRY.close_all()
+ # Stop the managed llama-server with its parent — a supervisor-less
+ # orphan would keep VRAM pinned after the app closes.
+ try:
+ from hermes_cli.local_runtime.bootstrap import shutdown_local_runtime
+
+ shutdown_local_runtime()
+ except Exception: # noqa: BLE001
+ pass
if os.getenv("HERMES_DESKTOP") == "1":
_terminate_desktop_managed_gateway()
@@ -1399,8 +1445,8 @@ _SCHEMA_OVERRIDES: Dict[str, Dict[str, Any]] = {
},
"agent.service_tier": {
"type": "select",
- "description": "API service tier (OpenAI/Anthropic)",
- "options": ["", "auto", "default", "flex"],
+ "description": "Fast mode: fast = always, auto = first N seconds of each turn, cold = first turn only",
+ "options": ["", "normal", "fast", "auto", "cold"],
},
"delegation.reasoning_effort": {
"type": "select",
@@ -1806,6 +1852,7 @@ from hermes_cli.web_models import ( # noqa: F401
LearningNodeEdit,
DebugShareRequest,
TTSSpeakRequest,
+ TTSLeaseRequest,
OAuthSubmitBody,
BulkDeleteSessions,
SessionImport,
@@ -3289,6 +3336,10 @@ def _git_path(path: str) -> str:
from hermes_cli.web_routers import git as _git_routes # noqa: E402
app.include_router(_git_routes.router)
+
+from hermes_cli.web_routers import local_models as _local_models_routes # noqa: E402
+
+app.include_router(_local_models_routes.router)
from hermes_cli.web_routers.git import ( # noqa: E402,F401 — legacy re-exports; tests call these via web_server.
git_status_route,
git_worktrees_route,
@@ -3363,6 +3414,7 @@ _PORT_BINDING_PLATFORM_PORTS: Dict[str, Tuple[str, int]] = {
"sms": ("webhook_port", 8080),
"whatsapp_cloud": ("webhook_port", 8090),
"line": ("port", 8646),
+ "teams": ("port", 3978),
}
# Platform states that mean the adapter is NOT serving its port right now.
@@ -5631,6 +5683,43 @@ async def speak_text(payload: TTSSpeakRequest, profile: Optional[str] = None):
}
+@app.post("/api/audio/tts-lease")
+async def tts_lease(payload: TTSLeaseRequest, profile: Optional[str] = None):
+ """Desktop TTS-output toggles as warm-up / release signals.
+
+ "Read replies aloud" and voice-conversation mode are explicit "speech is
+ about to be needed" gestures. ``active: true`` registers the toggle as a
+ lease on the TTS engine and pre-loads the configured provider (local
+ piper/kittentts model, lazily-installed SDK) so the first spoken reply
+ doesn't pay the load as dead air; ``active: false`` drops the lease and,
+ once no surface holds one, unloads resident local models.
+
+ Blocking work (model load, voice download) runs off the event loop.
+ Warm-up failures are reported in the body, never as an HTTP error — the
+ toggle must succeed even when the engine can't preload.
+ """
+ lease = (payload.lease or "").strip()
+ if not lease:
+ raise HTTPException(status_code=400, detail="lease is required")
+
+ def _apply():
+ from tools.tts_tool import acquire_tts_lease, release_tts_lease
+
+ if payload.active:
+ with _config_profile_scope(profile):
+ return acquire_tts_lease(lease)
+ return release_tts_lease(lease)
+
+ try:
+ result = await asyncio.get_running_loop().run_in_executor(None, _apply)
+ except HTTPException:
+ raise
+ except Exception as exc:
+ _log.warning("TTS lease %s (%s) failed: %s", lease, payload.active, exc)
+ result = {"leases": None, "action": "error", "error": str(exc)}
+ return {"ok": True, "lease": lease, "active": payload.active, **result}
+
+
def _split_text_for_speak_stream(text: str, cap: int) -> list:
"""Split *text* into provider-cap-sized pieces on sentence boundaries.
@@ -7501,7 +7590,11 @@ async def get_model_options(
# Keep the profile override inside the worker thread so the full
# sync picker build (config load, pricing, refresh probes) runs
# off the event loop under the requested profile.
- with _profile_scope(profile):
+ # Use _config_profile_scope (contextvar only, no skill-module
+ # lock) — the payload build can block for 15s on a models.dev
+ # cache miss, and _profile_scope's RLock held across that block
+ # starves concurrent /api/config and freezes the server (#58576).
+ with _config_profile_scope(profile):
return build_model_options_payload(
load_picker_context(),
explicit_only=bool(explicit_only),
@@ -7539,8 +7632,10 @@ def get_recommended_default_model(provider: str = ""):
get_curated_nous_model_ids,
get_pricing_for_provider,
check_nous_free_tier,
+ nous_policy_allowed_ids,
partition_nous_models_by_tier,
pick_silent_default_model,
+ restrict_to_nous_policy,
union_with_portal_free_recommendations,
union_with_portal_paid_recommendations,
)
@@ -7557,10 +7652,19 @@ def get_recommended_default_model(provider: str = ""):
except Exception:
portal_url = ""
+ # This endpoint picks the model a user lands on without choosing it,
+ # so an unreachable one here is worse than in a picker. Narrow before
+ # the tier split, so a rescued id still has to pass the free/paid
+ # predicate.
+ _policy_allowed = nous_policy_allowed_ids()
+
if free_tier:
model_ids, pricing = union_with_portal_free_recommendations(
model_ids, pricing, portal_url
)
+ model_ids = restrict_to_nous_policy(
+ model_ids, _policy_allowed, rescue_empty=True,
+ )
model_ids, _unavailable = partition_nous_models_by_tier(
model_ids, pricing, free_tier=True
)
@@ -7568,6 +7672,9 @@ def get_recommended_default_model(provider: str = ""):
model_ids, pricing = union_with_portal_paid_recommendations(
model_ids, pricing, portal_url
)
+ model_ids = restrict_to_nous_policy(
+ model_ids, _policy_allowed, rescue_empty=True,
+ )
model = pick_silent_default_model(model_ids, provider="nous")
return {"provider": "nous", "model": model, "free_tier": bool(free_tier)}
@@ -11023,20 +11130,63 @@ def _claude_code_only_status() -> Dict[str, Any]:
def _copilot_acp_status() -> Dict[str, Any]:
"""Status for copilot-acp — credentials are owned by the Copilot CLI.
- There is no cheap programmatic credential probe for the ACP subprocess, so
- this is a read-only "managed by the Copilot CLI" card (like claude-code):
- Hermes never claims a login state it can't verify.
+ ``logged_in`` is claimed only on positive evidence (a supported env token
+ or a known on-disk GitHub Copilot credential store, via
+ ``auth.get_external_process_provider_status``). The Copilot CLI may also
+ hold its session in an OS keychain Hermes can't read, so the unverified
+ state is presented as "managed by the Copilot CLI" — never as signed out.
"""
+ try:
+ from hermes_cli.auth import get_external_process_provider_status
+ status = get_external_process_provider_status("copilot-acp") or {}
+ except Exception:
+ status = {}
+ verified = bool(status.get("auth_verified"))
+ configured = bool(status.get("configured"))
+ if verified:
+ source_label = status.get("auth_source") or "Copilot credentials detected"
+ elif configured:
+ found = status.get("resolved_command") or status.get("command") or "copilot"
+ source_label = f"Managed by the GitHub Copilot CLI ({found})"
+ else:
+ source_label = "GitHub Copilot CLI not found on PATH"
return {
- "logged_in": False,
+ "logged_in": verified,
"source": "copilot_cli",
- "source_label": "Managed by the GitHub Copilot CLI",
+ "source_label": source_label,
"token_preview": None,
"expires_at": None,
"has_refresh_token": False,
+ "configured": configured,
}
+def _external_process_cli_command(provider_id: str, default: str) -> str:
+ """Render an external-process provider's sign-in command with the CLI the
+ user actually has configured.
+
+ The static catalog assumes the default executable name; users who point
+ Hermes at a custom binary (``HERMES_COPILOT_ACP_COMMAND`` /
+ ``COPILOT_CLI_PATH``) would otherwise be told to run a command that isn't
+ the one Hermes spawns. Non-external-process providers get ``default`` back
+ untouched.
+ """
+ try:
+ from hermes_cli.auth import PROVIDER_REGISTRY, get_external_process_provider_status
+ pconfig = PROVIDER_REGISTRY.get(provider_id)
+ if not pconfig or pconfig.auth_type != "external_process":
+ return default
+ status = get_external_process_provider_status(provider_id) or {}
+ command = str(status.get("command") or "").strip()
+ if command:
+ parts = default.split(" ", 1)
+ tail = f" {parts[1]}" if len(parts) > 1 else ""
+ return f"{command}{tail}"
+ except Exception:
+ pass
+ return default
+
+
# Explicit, hand-tuned OAuth/account provider cards. These carry the bits that
# can't be derived from the unified provider catalog: the OAuth ``flow`` shape,
# the per-provider ``status_fn``, the ``cli_command`` fallback, and curated
@@ -11102,7 +11252,11 @@ _OAUTH_PROVIDER_CATALOG: tuple[Dict[str, Any], ...] = (
"id": "copilot-acp",
"name": "GitHub Copilot (ACP)",
"flow": "external",
- "cli_command": "copilot /login",
+ # `copilot login` is the CLI's non-interactive device-code login
+ # subcommand; the previous `copilot /login` form is not a valid
+ # invocation (slash-commands only exist inside an interactive
+ # session, reachable as `copilot -i /login`).
+ "cli_command": "copilot login",
"docs_url": "https://docs.github.com/en/copilot",
"status_fn": _copilot_acp_status,
},
@@ -11361,7 +11515,7 @@ async def list_oauth_providers(profile: Optional[str] = None):
"id": p["id"],
"name": p["name"],
"flow": p["flow"],
- "cli_command": p["cli_command"],
+ "cli_command": _external_process_cli_command(p["id"], p["cli_command"]),
"docs_url": p["docs_url"],
"disconnect_hint": disconnect_hint,
"disconnect_command": _oauth_provider_disconnect_command(p),
@@ -12870,6 +13024,13 @@ def _normalize_dashboard_cron_updates(
)
if "deliver" in normalized:
normalized["deliver"] = _cron_optional_text(normalized["deliver"]) or "local"
+ if "failure_deliver" in normalized:
+ # Same text normalization as deliver, but empty CLEARS the override
+ # (failures fall back to deliver) rather than coalescing to a target
+ # — the field is optional by design (NS-788).
+ normalized["failure_deliver"] = _cron_optional_text(
+ normalized["failure_deliver"]
+ )
if "context_from" in normalized:
normalized["context_from"] = _cron_string_list(normalized["context_from"])
if "enabled_toolsets" in normalized:
@@ -13424,35 +13585,62 @@ def _gateway_fire_endpoint(profile: str, home: Path) -> str:
"""Resolve the loopback URL of the gateway api_server's cron-fire route.
Port resolution mirrors gateway/config.py's api_server load order for the
- TARGET profile: ``platforms.api_server.extra.port`` in the profile's
- config.yaml, then ``API_SERVER_PORT`` (process env for the active profile,
- the profile's own .env otherwise), then the adapter default 8642. The bind
- host is the adapter's loopback default — the dashboard and gateway share a
- network namespace in every supported deployment (same host process tree,
- or the same container under s6).
+ LISTENER-OWNER profile: ``platforms.api_server.extra.port`` in that
+ profile's config.yaml, then ``API_SERVER_PORT`` (process env for the
+ active profile, the profile's own .env otherwise), then the adapter
+ default 8642. The bind host is the adapter's loopback default — the
+ dashboard and gateway share a network namespace in every supported
+ deployment (same host process tree, or the same container under s6).
Multiplex mode (one gateway serving several profiles) exposes per-profile
mirrors under ``/p//…``, so a non-default profile routes through
- the default gateway's port with that prefix; per-profile-gateway mode
- (each profile its own process/port) uses the bare path on the profile's
- own port.
+ the default gateway's port with that prefix — only the DEFAULT profile's
+ api_server is bound in that mode, so the port must be read from the
+ default home, never the target profile's (a secondary's own
+ ``API_SERVER_PORT`` is a port nothing listens on). Per-profile-gateway
+ mode (each profile its own process/port) uses the bare path on the
+ profile's own port.
"""
import os as _os
+ multiplex = False
+ try:
+ from gateway.config import _env_multiplex_profiles_override
+
+ cfg = load_config()
+ multiplex = bool(cfg_get(cfg, "gateway", "multiplex_profiles", default=False))
+ env_flag = _env_multiplex_profiles_override()
+ if env_flag is not None:
+ multiplex = env_flag
+ except Exception:
+ _log.debug("cron fire: multiplex detection failed; assuming single-profile", exc_info=True)
+
+ listener_profile, listener_home = profile, home
+ if multiplex and profile != "default":
+ from hermes_constants import get_default_hermes_root
+
+ listener_profile, listener_home = "default", get_default_hermes_root()
+ _log.info(
+ "cron fire: multiplex gateway — resolving api_server port for %s "
+ "from the default profile's listener (%s)",
+ profile,
+ listener_home,
+ )
+
port = 0
try:
# Profile-scoped read through the CANONICAL loader (managed-scope
# overlay, ${ENV_VAR} expansion, profile pathing) — never a raw
# yaml.safe_load of config.yaml (tests/hermes_cli/
# test_config_read_guard.py). The HERMES_HOME override scopes
- # get_config_path() to the TARGET profile, same pattern the
+ # get_config_path() to the LISTENER-OWNER profile, same pattern the
# deprecated _fire_cron_job_for_profile used for its store scope.
from hermes_constants import (
reset_hermes_home_override,
set_hermes_home_override,
)
- token = set_hermes_home_override(str(home))
+ token = set_hermes_home_override(str(listener_home))
try:
profile_cfg = load_config()
finally:
@@ -13467,8 +13655,8 @@ def _gateway_fire_endpoint(profile: str, home: Path) -> str:
if not port:
raw = (
_os.getenv("API_SERVER_PORT", "")
- if profile == _cron_default_profile()
- else _profile_env_value(home, "API_SERVER_PORT")
+ if listener_profile == _cron_default_profile()
+ else _profile_env_value(listener_home, "API_SERVER_PORT")
)
try:
port = int(raw) if raw else 0
@@ -13477,18 +13665,6 @@ def _gateway_fire_endpoint(profile: str, home: Path) -> str:
if not port:
port = 8642
- multiplex = False
- try:
- cfg = load_config()
- multiplex = bool(cfg_get(cfg, "gateway", "multiplex_profiles", default=False))
- env_flag = _os.getenv("GATEWAY_MULTIPLEX_PROFILES", "").strip().lower()
- if env_flag in {"1", "true", "yes", "on"}:
- multiplex = True
- elif env_flag in {"0", "false", "no", "off"}:
- multiplex = False
- except Exception:
- pass
-
if multiplex and profile != "default":
return f"http://127.0.0.1:{port}/p/{profile}/api/cron/fire"
return f"http://127.0.0.1:{port}/api/cron/fire"
@@ -15021,7 +15197,13 @@ def _fallback_profile_dicts(profiles_mod) -> List[Dict[str, Any]]:
"provider": provider,
"has_env": _safe(lambda entry=entry_path: (entry / ".env").exists(), False),
"skill_count": _safe(lambda entry=entry_path: profiles_mod._count_skills(entry), 0),
- "gateway_running": _safe(lambda entry=entry_path: profiles_mod._check_gateway_running(entry), False),
+ "gateway_running": _safe(
+ lambda entry=entry_path, name=entry.name: (
+ profiles_mod._check_gateway_running(entry)
+ or profiles_mod._served_by_running_multiplexer(name)
+ ),
+ False,
+ ),
"description": _safe(lambda entry=entry_path: profiles_mod.read_profile_meta(entry).get("description", ""), ""),
"description_auto": _safe(lambda entry=entry_path: profiles_mod.read_profile_meta(entry).get("description_auto", False), False),
"distribution_name": None,
@@ -19578,6 +19760,7 @@ def start_server(
headless: bool = False,
ssh_session_token: Optional[str] = None,
ssh_owner_nonce: Optional[str] = None,
+ start_mcp_discovery_after_bind: bool = False,
):
"""Start the web UI server.
@@ -19592,6 +19775,10 @@ def start_server(
``ssh_session_token`` and ``ssh_owner_nonce`` are process-local Desktop SSH
bootstrap state. Neither is persisted or exported to child processes.
+
+ ``start_mcp_discovery_after_bind`` (Desktop ``serve``) defers the
+ background MCP discovery thread until the ready sentinel has been written,
+ so its SDK import cannot hold the GIL against the pre-bind import path.
"""
_apply_ssh_session_token(ssh_session_token or "")
_apply_ssh_owner_nonce(ssh_owner_nonce)
@@ -19969,6 +20156,27 @@ def start_server(
print(f" Hermes Web UI → http://{host}:{actual_port}")
_maybe_open_browser(host, actual_port, open_browser, initial_profile)
+ if start_mcp_discovery_after_bind:
+ # Deferred from cmd_dashboard for Desktop `serve` (see there).
+ # Not started at the bind itself either: the ~350ms `mcp` SDK
+ # import holds the GIL, and at bind time the renderer is doing
+ # its WebSocket handshake + first hydration reads against this
+ # loop (measured: starting it here gave back most of the
+ # READY gain as a slower connect). One second later the shell
+ # is painted and idle. An agent build inside that second fires
+ # the deferred start itself (wait_for_mcp_discovery), so its
+ # bounded join and the late-binding refresh are unchanged.
+ try:
+ from hermes_cli.mcp_startup import defer_background_mcp_discovery
+
+ defer_background_mcp_discovery(
+ logger=_log,
+ thread_name="dashboard-mcp-discovery",
+ delay=_DESKTOP_MCP_DISCOVERY_DELAY_S,
+ )
+ except Exception:
+ _log.debug("Deferred MCP discovery arm failed", exc_info=True)
+
# Collapse the peer-hangup teardown flood (#50005). When the Desktop
# forcibly closes its WebSocket mid-write, asyncio logs a full
# traceback per pending connection-lost callback — 50+ identical
diff --git a/hermes_state.py b/hermes_state.py
index b9794c7c79..f9696fc6fd 100644
--- a/hermes_state.py
+++ b/hermes_state.py
@@ -96,6 +96,11 @@ from hermes_state_common import ( # noqa: F401 (re-exported for back-compat)
_PREVIEW_MAX_CHARS,
_PREVIEW_SCAFFOLD_WINDOW,
_PREVIEW_SCAFFOLDED_SQL,
+ _acquire_db_flock,
+ _clear_lock_holder_record,
+ _describe_lock_holder,
+ _read_lock_holder_record,
+ is_advisory_lock_contention,
)
from hermes_state_portability import SessionPortabilityMixin
from hermes_state_schema import SessionSchemaMixin
@@ -111,6 +116,13 @@ logger = logging.getLogger(__name__)
MAX_SAFE_RESUME_MESSAGES = 20_000
MAX_SAFE_EXPORT_MESSAGES = 20_000
+# Auto-maintenance only VACUUMs when at least this fraction of the database
+# file is reclaimable (``PRAGMA freelist_count / PRAGMA page_count``). Below
+# it a full rewrite costs more I/O than it returns — pruning a handful of small
+# sessions on a dense multi-GB state.db should never rewrite the whole file to
+# reclaim a few MB (#54189). Composes with ``min_vacuum_interval_days``.
+AUTO_VACUUM_MIN_FREELIST_RATIO = 0.25
+
def _configured_transcript_limit(key: str, fallback: int) -> int:
"""Resolve a transcript safety limit from config at call time.
@@ -370,6 +382,20 @@ DEFAULT_DB_PATH = get_hermes_home() / "state.db"
# query; short enough that transient fd pressure doesn't strand the read pool.
_READ_OPEN_RETRY_SECONDS = 60.0
+# Transient SQLITE_IOERR retry budget for READ-ONLY opens (#100436). A WAL
+# database being actively written (checkpoint, WAL reset/truncate, frame
+# flush) can surface "disk I/O error" to a concurrent ``mode=ro`` reader in
+# a millisecond-wide transition window: the read-only connection cannot
+# perform the WAL recovery a read through a stale or mid-update -shm file
+# needs, because recovery requires writing the -shm index, which mode=ro
+# refuses. The window closes on its own (the writer finishes the transition),
+# so a bounded number of short retries makes the open succeed instead of
+# 500-ing the whole /api/sessions poll (or any other read-only opener).
+# Deliberately NOT attempted on writable opens: a writer owns the
+# transition, so an IOERR there means a real storage/fd problem.
+_READ_ONLY_IOERR_RETRY_ATTEMPTS = 3
+_READ_ONLY_IOERR_RETRY_BACKOFF_S = 0.05
+
# Hard ceiling on read-only connections ALIVE at once against one database
# FILE — pooled idle ones and checked-out ones together, summed over every
# SessionDB in this process that points at that file. See _PathReadBudget.
@@ -1115,6 +1141,17 @@ def _strip_stale_tool_call_markers(
return messages
+def _normalize_telegram_topic_profile_name(profile_name: Optional[str] = None) -> str:
+ """Normalize profile namespace for Telegram topic-mode tables.
+
+ Empty / missing values map to ``\"default\"`` so non-multiplexed gateways
+ keep a single namespace. Multiplexed callers must pass the *routed*
+ profile (``source.profile``), never the process-global active profile.
+ """
+ name = str(profile_name or "").strip()
+ return name if name else "default"
+
+
def format_session_db_unavailable(prefix: str = "Session database not available") -> str:
"""Format a user-facing 'session DB unavailable' message with cause.
@@ -1758,13 +1795,14 @@ def _log_wal_reset_bug_once(
# for git/pip/system Python installs (#75153).
repair_hint = _wal_reset_repair_hint()
logger.warning(
- "%s: linked SQLite %s is vulnerable to the WAL-reset corruption "
- "bug (https://sqlite.org/wal.html#walresetbug) — %s. "
+ "%s: linked SQLite %s (interpreter %s) is vulnerable to the WAL-reset "
+ "corruption bug (https://sqlite.org/wal.html#walresetbug) — %s. "
"Upgrade to SQLite 3.51.3+ (or backports 3.50.7 / 3.44.6); "
"%s. See `hermes doctor`. This warning fires once per "
"process per database.",
db_label,
sqlite3.sqlite_version,
+ sys.executable,
action,
repair_hint,
)
@@ -2078,6 +2116,52 @@ def is_malformed_db_error(exc: BaseException) -> bool:
return any(marker in str(exc).lower() for marker in _MALFORMED_DB_MARKERS)
+# SQLITE_IOERR, matched as a plain substring so wrapped error strings still
+# classify. Shared by the read-only open retry and the write-path BEGIN retry.
+_DISK_IO_ERROR_MARKER = "disk i/o error"
+
+# Broader set for HTTP classification: a read that failed for one of these
+# reasons found the store BUSY, not gone. Callers map it to 503 (retry, the
+# list was not cleared) instead of 500. Corruption is deliberately absent —
+# a malformed store must surface, not be retried into a timeout.
+_TRANSIENT_SQLITE_MARKERS = (
+ _DISK_IO_ERROR_MARKER,
+ "database is locked",
+ "database table is locked",
+ "busy",
+)
+
+
+def is_transient_sqlite_error(exc: BaseException) -> bool:
+ """True when a SQLite failure means "busy right now", not "damaged".
+
+ One predicate so the read paths cannot drift apart on what counts as
+ recoverable: the read-only open retry, and the HTTP 503-vs-500 split on
+ the session-list endpoints, classify the same way.
+ """
+ if not isinstance(exc, sqlite3.OperationalError):
+ return False
+ message = str(exc).lower()
+ return any(marker in message for marker in _TRANSIENT_SQLITE_MARKERS)
+
+
+def _is_transient_read_only_ioerr(exc: sqlite3.OperationalError, *, attempt: int) -> bool:
+ """True when a read-only open should be retried rather than raised.
+
+ A ``mode=ro`` connection cannot perform WAL recovery (recovery needs to
+ write the -shm index, which read-only mode refuses), so a concurrent WAL
+ checkpoint / reset / frame-flush can surface ``SQLITE_IOERR`` ("disk I/O
+ error") to a reader on an otherwise healthy database (#100436). The
+ transition is millisecond-scale, so a bounded number of short retries
+ clears it without changing classification for genuine storage failures —
+ a persistent IOERR still exhausts the budget and propagates.
+ """
+ return (
+ attempt < _READ_ONLY_IOERR_RETRY_ATTEMPTS
+ and _DISK_IO_ERROR_MARKER in str(exc).lower()
+ )
+
+
def is_malformed_schema_error(exc: BaseException) -> bool:
"""True only when SQLite explicitly reports malformed schema text.
@@ -2198,7 +2282,10 @@ def classify_persistence_error(exc_or_str) -> str:
if isinstance(exc_or_str, CompressionSessionBusyError):
return "compression"
if isinstance(exc_or_str, StateDbReplacedError):
+ # Includes DeletedWalGenerationError (subclass).
return "replaced"
+ if isinstance(exc_or_str, StateDbCorruptError):
+ return "corrupt"
text = str(exc_or_str).lower()
if "turn lease" in text:
return "turn_lease"
@@ -2208,6 +2295,8 @@ def classify_persistence_error(exc_or_str) -> str:
return "compression"
if "was replaced underneath" in text:
return "replaced"
+ if "deleted state.db-wal" in text or "deleted state.db-shm" in text:
+ return "replaced"
# Structural corruption BEFORE the lock and disk buckets: "database disk
# image is malformed" contains "disk" (and some wrapped corruption
# strings mention "locked" recovery attempts), so later buckets would
@@ -2272,14 +2361,19 @@ def _cross_process_repair_lock(db_path: Path):
"""Serialize state.db schema surgery across processes.
Yields True when this process holds the repair lock for *db_path*, False
- when the bounded acquire timed out. Unlike the kanban init lock — whose
- critical section is idempotent, so proceeding without the lock is merely
- redundant work — proceeding here would be exactly the unsafe interleaving
- we are trying to prevent, so a caller that gets False must NOT do surgery.
+ when the bounded acquire timed out or the lock file could not be opened at
+ all. Unlike the kanban init lock — whose critical section is idempotent,
+ so proceeding without the lock is merely redundant work — proceeding here
+ would be exactly the unsafe interleaving we are trying to prevent, so a
+ caller that gets False must NOT do surgery.
``flock`` is the right primitive for this: the kernel drops the lock when
the holding process dies, so a crashed repairer cannot leave a stale lock
- that wedges every future repair (a pidfile would). The acquire is still
+ that wedges every future repair (a pidfile would). One exception exists
+ (issue #100108): a forked child that inherited the lock fd keeps the
+ flock alive after the acquirer dies, so the acquire path records the
+ holder's pid + start time and breaks the lock when that holder is
+ provably dead (see ``_acquire_db_flock``). The acquire is still
bounded because a *live* repairer can legitimately sit in ``VACUUM`` for
minutes on a large DB, and an unbounded wait would hang the caller's open
with no traceback (the failure shape of #36644).
@@ -2289,42 +2383,67 @@ def _cross_process_repair_lock(db_path: Path):
lock_path.parent.mkdir(parents=True, exist_ok=True)
handle = lock_path.open("a+b")
except OSError as exc:
- # Read-only dir, exhausted fds, exotic filesystem: fall back to the
- # in-process behaviour that shipped before this lock existed rather
- # than refusing to repair a DB we could otherwise heal.
+ # Fail closed, exactly as a timed-out acquire does. A lock file we
+ # cannot even open means the filesystem is out of space, inodes or
+ # descriptors — and a sibling that opened ITS handle before the disk
+ # filled is still inside writable_schema surgery or VACUUM. Yielding
+ # True here let two processes run schema surgery on the same live
+ # state.db concurrently, which is itself the corruption source this
+ # lock exists to remove (#100368: the disk-full trigger, then a fresh
+ # corruption on every boot with other writers alive). Callers already
+ # handle False by re-probing and reporting, and on a read-only
+ # directory no repair strategy could have written anyway.
logger.warning(
- "Could not open state.db repair lock %s (%s) — proceeding with "
- "in-process serialisation only.", lock_path, exc,
+ "Could not open state.db repair lock %s (%s) — skipping schema "
+ "surgery rather than running it without cross-process authority.",
+ lock_path, exc,
)
- yield True
+ yield False
return
acquired = False
try:
- deadline = time.monotonic() + _REPAIR_LOCK_TIMEOUT_SECONDS
- while True:
- try:
- if _IS_WINDOWS:
+ if _IS_WINDOWS:
+ deadline = time.monotonic() + _REPAIR_LOCK_TIMEOUT_SECONDS
+ while True:
+ try:
import msvcrt
handle.seek(0)
msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1)
- else:
- import fcntl
-
- fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
- acquired = True
- break
- except (BlockingIOError, OSError):
- if time.monotonic() >= deadline:
+ acquired = True
break
- time.sleep(_REPAIR_LOCK_POLL_SECONDS)
- if not acquired:
+ except (BlockingIOError, OSError) as exc:
+ if not is_advisory_lock_contention(exc):
+ logger.warning(
+ "Could not acquire state.db repair lock %s (%s) — "
+ "skipping schema surgery on a non-contention error.",
+ lock_path, exc,
+ )
+ acquired = None
+ break
+ if time.monotonic() >= deadline:
+ break
+ time.sleep(_REPAIR_LOCK_POLL_SECONDS)
+ else:
+ acquired, handle = _acquire_db_flock(
+ str(lock_path),
+ handle,
+ _REPAIR_LOCK_TIMEOUT_SECONDS,
+ _REPAIR_LOCK_POLL_SECONDS,
+ "state.db repair lock",
+ )
+ if acquired is None:
+ # Non-contention failure already logged with its errno.
+ acquired = False
+ elif not acquired:
+ record = None if _IS_WINDOWS else _read_lock_holder_record(handle)
logger.warning(
"state.db repair lock %s held by another process for more "
"than %.0fs — skipping schema surgery in this process to "
- "avoid racing the repairer.",
+ "avoid racing the repairer. Recorded holder: %s.",
lock_path, _REPAIR_LOCK_TIMEOUT_SECONDS,
+ _describe_lock_holder(record),
)
yield acquired
finally:
@@ -2338,6 +2457,7 @@ def _cross_process_repair_lock(db_path: Path):
else:
import fcntl
+ _clear_lock_holder_record(handle)
fcntl.flock(handle.fileno(), fcntl.LOCK_UN)
except OSError: # pragma: no cover - best effort release
pass
@@ -2345,6 +2465,65 @@ def _cross_process_repair_lock(db_path: Path):
handle.close()
+def _try_acquire_auto_maintenance_lock(db_path: Path) -> Optional[Any]:
+ """Non-blocking cross-process lock for one auto-maintenance pass.
+
+ The kernel releases this advisory lock if the holder exits, unlike a
+ durable pid/meta marker. A caller that cannot acquire it must skip the
+ pass: otherwise two startups can both pass the interval check and the
+ second can prune a row the first has only just closed recoverably.
+ """
+ lock_path = db_path.with_name(db_path.name + ".auto-maintenance.lock")
+ try:
+ lock_path.parent.mkdir(parents=True, exist_ok=True)
+ handle = lock_path.open("a+b")
+ except OSError as exc:
+ logger.warning(
+ "Could not open state.db auto-maintenance lock %s (%s) — skipping "
+ "automatic maintenance.",
+ lock_path,
+ exc,
+ )
+ return None
+
+ try:
+ if _IS_WINDOWS:
+ import msvcrt
+
+ handle.seek(0)
+ msvcrt.locking( # type: ignore[attr-defined]
+ handle.fileno(), msvcrt.LK_NBLCK, 1 # type: ignore[attr-defined]
+ )
+ else:
+ import fcntl
+
+ fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except (BlockingIOError, OSError):
+ handle.close()
+ return None
+ return handle
+
+
+def _release_auto_maintenance_lock(handle: Any) -> None:
+ """Release a handle returned by :func:`_try_acquire_auto_maintenance_lock`."""
+ try:
+ if _IS_WINDOWS:
+ import msvcrt
+
+ handle.seek(0)
+ msvcrt.locking( # type: ignore[attr-defined]
+ handle.fileno(), msvcrt.LK_UNLCK, 1 # type: ignore[attr-defined]
+ )
+ else:
+ import fcntl
+
+ fcntl.flock(handle.fileno(), fcntl.LOCK_UN)
+ except OSError: # pragma: no cover - best effort release
+ pass
+ finally:
+ handle.close()
+
+
def _bump_schema_cookie(conn: sqlite3.Connection) -> None:
"""Increment the schema cookie after direct ``sqlite_master`` surgery.
@@ -3539,16 +3718,19 @@ def repair_state_db_schema(db_path: Path, *, backup: bool = True) -> Dict[str, A
result = report
with _cross_process_repair_lock(db_path) as holding_lock:
if not holding_lock:
- # Another process is still inside its critical section. It may
- # nonetheless have healed the file already (long VACUUM after a
- # successful strategy), so re-probe before reporting failure.
+ # Another process is still inside its critical section, or the
+ # lock file itself could not be opened (full disk / no fds). It
+ # may nonetheless have healed the file already (long VACUUM after
+ # a successful strategy), so re-probe before reporting failure.
if _db_opens_cleanly(db_path) is None:
report["repaired"] = True
report["strategy"] = "repaired_by_other_process"
else:
report["error"] = (
- "another process holds the state.db repair lock; skipped "
- "schema surgery to avoid racing it"
+ "could not obtain the state.db repair lock (held by "
+ "another process, or the lock file was unopenable); "
+ "skipped schema surgery to avoid racing a concurrent "
+ "repairer"
)
else:
# The fast check above avoids taking the lock for a known-exhausted
@@ -4190,6 +4372,23 @@ class StateDbReplacedError(RuntimeError):
"""
+class DeletedWalGenerationError(StateDbReplacedError):
+ """A live process holds a deleted state.db-wal / -shm generation.
+
+ Opening or writing through this handle would mint a second WAL inode
+ (or keep committing on the orphan) — the split-brain that produces
+ intermittent SQLITE_CORRUPT / SQLITE_IOERR. Stop the writers; do not
+ unlink the WAL yourself. ``database.journal_mode: delete`` is operator
+ containment, not a default change.
+
+ Subclasses :class:`StateDbReplacedError` so every downstream consumer
+ that already stops SQLite writes and diverts pending transcripts on a
+ replaced store (gateway retry queue, run_agent flush) handles the split
+ WAL generation identically — the correct response is the same: stop
+ writing, preserve the transcript tail on disk.
+ """
+
+
# SQLite header: 4-byte big-endian application_id at offset 68. Distinct from
# inode: ``cp`` onto the same path keeps st_ino and truncates+rewrites.
_STATE_DB_APPLICATION_ID_OFFSET = 68
@@ -4200,6 +4399,53 @@ _STATE_DB_REPLACED_MSG = (
"gateway pending_messages spool) and restore or reopen after operator "
"intervention."
)
+_DELETED_WAL_GENERATION_MSG = (
+ "FATAL: a live process holds a deleted state.db-wal or state.db-shm "
+ "inode while the path names a different (or missing) generation. "
+ "Refusing to open or write so a second WAL cannot be minted. "
+ "Stop the gateway, dashboard, and cron writers that hold the deleted "
+ "sidecar, then reopen. Do not delete the WAL yourself. "
+ "database.journal_mode: delete is operator containment, not a new default."
+)
+
+
+class StateDbCorruptError(sqlite3.DatabaseError):
+ """A live SessionDB observed structural (non-FTS) corruption and is quarantined.
+
+ Raised once a write on this handle reports bare ``SQLITE_CORRUPT`` /
+ ``SQLITE_NOTADB`` that is neither FTS-scoped (``_is_fts_write_corruption_error``)
+ nor a replaced-file case (``StateDbReplacedError``). Subclasses
+ ``sqlite3.DatabaseError`` so every existing ``except sqlite3.Error``
+ degrade path keeps working; ``sqlite_errorcode``/``sqlite_errorname``
+ are copied from the originating error.
+
+ The quarantine is sticky for the life of the handle: later writes fail
+ fast, the handle never reopens after ``close()``, and ``close()`` skips
+ its own WAL checkpoint. Field evidence (the #90837 lost/reordered-page
+ signature, the #90950 page-1 clobber): a handle that kept writing for ~50
+ minutes after the first structural error checkpointed 15 pages under the
+ wrong page numbers on shutdown, turning a still-readable file into
+ ``file is not a database``. Stopping the writes is what prevents that;
+ skipping the explicit checkpoint is the second line of defence. SQLite
+ still runs its own last-connection checkpoint inside ``close()`` (and
+ deletes the ``-wal`` sidecar) unless ``SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE``
+ is set — Python exposes it via ``Connection.setconfig()`` on 3.12+, so
+ quarantine disables the close-time checkpoint there and the WAL survives
+ on disk for forensics; on 3.11 the internal checkpoint is unavoidable
+ (post-quarantine it can only carry pre-corruption committed frames, since
+ no further writes are accepted). The
+ recovery boundary is a process restart on a repaired or restored file.
+ """
+
+
+_STATE_DB_CORRUPT_MSG = (
+ "FATAL: state.db reported structural corruption (database disk image is "
+ "malformed outside the FTS shadow tables) on a live handle; refusing further "
+ "writes, automatic reopen, and the close-time WAL checkpoint on this file. "
+ "Stop the gateway, then run `hermes sessions recover --source "
+ "--inspect-only` or restore a snapshot. Unwritten transcripts are diverted to "
+ "sessions/.jsonl (and the gateway pending_messages spool)."
+)
def divert_session_transcript_jsonl(session_id: str, messages) -> "Optional[Path]":
@@ -4224,13 +4470,80 @@ def divert_session_transcript_jsonl(session_id: str, messages) -> "Optional[Path
return path
-def _read_sqlite_application_id(db_path: Path) -> "Optional[int]":
- """Read application_id from the SQLite header without opening a connection."""
+# _read_sqlite_application_id runs on EVERY write via _raise_if_db_replaced,
+# against the LIVE state.db. A bare open()/read()/close() there is the
+# howtocorrupt §2.2 bug: close() cancels every POSIX advisory lock this
+# process holds on the file — measured on Linux/SQLite 3.53.1, one probe call
+# drops the WAL-mode DMS shared lock the writer connection holds on state.db
+# (see hermes_cli/sqlite_safe_read.py for the module built around this rule).
+# With the DMS lock gone, a fresh opener in another process can treat this
+# writer as dead and rerun WAL-index recovery underneath it.
+#
+# The probe therefore reads through a per-path fd cached for the life of the
+# process: opening an fd never cancels locks (only close() does), and
+# os.pread takes no shared file position. When the path is re-pointed at a
+# new inode (the very replacement this probe exists to detect), the stale fd
+# is RETIRED, never closed — closing it would cancel the live connection's
+# locks on the old file, the exact bug being avoided. Replacement events are
+# rare and halt writes anyway, so the leak is bounded.
+_HEADER_PROBE_LOCK = threading.Lock()
+_HEADER_PROBE_FDS: "dict[str, tuple[int, int, int]]" = {} # key -> (fd, dev, ino)
+_RETIRED_HEADER_PROBE_FDS: "list[int]" = [] # intentionally never closed
+
+
+def _pread_db_header(db_path: Path, length: int) -> "Optional[bytes]":
+ """Lock-safe raw header read of a possibly-live SQLite database.
+
+ POSIX: pread from a cached, never-closed fd (rebound when the path names
+ a new inode). Windows: plain read — advisory-lock cancellation is a
+ POSIX-only hazard and msvcrt locks do not share the failure mode.
+ """
+ if _IS_WINDOWS:
+ try:
+ with db_path.open("rb") as handle:
+ return handle.read(length)
+ except OSError:
+ return None
+ key = str(db_path)
try:
- with db_path.open("rb") as handle:
- header = handle.read(_STATE_DB_APPLICATION_ID_OFFSET + 4)
+ st = os.stat(db_path)
except OSError:
return None
+ with _HEADER_PROBE_LOCK:
+ cached = _HEADER_PROBE_FDS.get(key)
+ if cached is not None and (cached[1], cached[2]) != (st.st_dev, st.st_ino):
+ # Path re-pointed at a new file. Retire (never close) the old fd.
+ _RETIRED_HEADER_PROBE_FDS.append(cached[0])
+ cached = None
+ del _HEADER_PROBE_FDS[key]
+ if cached is None:
+ try:
+ fd = os.open(db_path, os.O_RDONLY)
+ except OSError:
+ return None
+ try:
+ fst = os.fstat(fd)
+ except OSError:
+ _RETIRED_HEADER_PROBE_FDS.append(fd)
+ return None
+ cached = (fd, fst.st_dev, fst.st_ino)
+ _HEADER_PROBE_FDS[key] = cached
+ try:
+ return os.pread(cached[0], length, 0)
+ except OSError:
+ return None
+
+
+def _read_sqlite_application_id(db_path: Path) -> "Optional[int]":
+ """Read application_id from the SQLite header without opening a connection.
+
+ Safe against live databases: routed through :func:`_pread_db_header`,
+ which never issues a ``close()`` that would cancel this process's POSIX
+ locks on the file (howtocorrupt §2.2).
+ """
+ header = _pread_db_header(db_path, _STATE_DB_APPLICATION_ID_OFFSET + 4)
+ if header is None:
+ return None
if len(header) < _STATE_DB_APPLICATION_ID_OFFSET + 4:
return None
if header[:16] != b"SQLite format 3\x00":
@@ -4257,6 +4570,86 @@ def _stat_db_file_identity(path: Path) -> "Optional[tuple]":
return (st.st_dev, st.st_ino)
+def _stat_sqlite_sidecar_identity(db_path: Path) -> Dict[str, tuple]:
+ """Snapshot ``(st_dev, st_ino)`` for existing WAL/SHM sidecars."""
+ identities: Dict[str, tuple] = {}
+ base = os.fspath(db_path)
+ for suffix in ("-wal", "-shm"):
+ ident = _stat_db_file_identity(Path(base + suffix))
+ if ident is not None:
+ identities[suffix] = ident
+ return identities
+
+
+def _canonical_sqlite_path(path: str) -> str:
+ """Normalize a /proc fd target, stripping the Linux `` (deleted)`` suffix."""
+ return os.path.normcase(os.path.abspath(path.removesuffix(" (deleted)")))
+
+
+def _watched_sqlite_sidecar_paths(db_path) -> Set[str]:
+ base = os.path.abspath(os.fspath(db_path))
+ return {
+ _canonical_sqlite_path(base + "-wal"),
+ _canonical_sqlite_path(base + "-shm"),
+ }
+
+
+def iter_deleted_sqlite_sidecar_holders(db_path) -> List[Tuple[int, str]]:
+ """Return processes holding an unlinked ``state.db-wal`` / ``-shm``.
+
+ Linux-only (``/proc//fd`` readlink). Windows and other hosts
+ return ``[]`` — Windows cannot unlink a sidecar another process still
+ holds, and macOS does not use the `` (deleted)`` suffix.
+
+ The scan includes this process: on the SessionDB open/write refuse
+ path, the in-process writer that still holds the orphan inode is the
+ one that must not mint a replacement WAL (and must stop committing).
+ ``_foreign_state_db_holders`` keeps skipping this PID for FTS
+ maintenance so a process does not block its own optional repair.
+ """
+ if not sys.platform.startswith("linux"):
+ return []
+
+ holders: List[Tuple[int, str]] = []
+ watched = _watched_sqlite_sidecar_paths(db_path)
+ try:
+ for pid_str in os.listdir("/proc"):
+ if not pid_str.isdigit():
+ continue
+ pid = int(pid_str)
+ fd_dir = f"/proc/{pid}/fd"
+ try:
+ fds = os.listdir(fd_dir)
+ except OSError:
+ continue
+ for fd in fds:
+ try:
+ target = os.readlink(f"{fd_dir}/{fd}")
+ except OSError:
+ continue
+ if " (deleted)" not in target:
+ continue
+ if _canonical_sqlite_path(target) in watched:
+ holders.append((pid, target))
+ except Exception as exc:
+ logger.debug("deleted-WAL holder scan failed for %s: %s", db_path, exc)
+ return holders
+ return holders
+
+
+def refuse_deleted_wal_generation(db_path) -> None:
+ """Raise if any process holds a deleted WAL/SHM generation for *db_path*.
+
+ Called *before* ``sqlite3.connect`` so a second opener cannot mint a
+ replacement WAL inode while a live writer still holds the orphan.
+ """
+ holders = iter_deleted_sqlite_sidecar_holders(db_path)
+ if not holders:
+ return
+ logger.error(_DELETED_WAL_GENERATION_MSG)
+ raise DeletedWalGenerationError(_DELETED_WAL_GENERATION_MSG)
+
+
# ── Process-wide shared SessionDB registry (#90837) ──
#
# The registry itself lives in hermes_state_registry.py — a bounded
@@ -4794,6 +5187,18 @@ def classify_session_status(
return SESSION_STATUS_COMPLETE
+# Parent→child ``profile_name`` inheritance fence (#88381). ``agent::...``
+# gateway keys encode the profile namespace; a keyless row (CLI / subagent
+# lineage) carries none and inherits freely. Two keyed rows must agree on
+# ``agent::`` — a default child (``agent:main:``) forked from a sibling
+# profile's row must not be durably mislabelled as that profile's.
+_SAME_KEY_NAMESPACE_SQL = (
+ "p.session_key IS NULL OR sessions.session_key IS NULL"
+ " OR substr(p.session_key, 1, instr(substr(p.session_key, 7), ':') + 6)"
+ " = substr(sessions.session_key, 1, instr(substr(sessions.session_key, 7), ':') + 6)"
+)
+
+
class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin):
"""
SQLite-backed session storage with FTS5 search.
@@ -4802,6 +5207,19 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
single writer via WAL mode). Each method opens its own cursor.
"""
+ # Only these state-owned producers participate in automatic stale-open
+ # reconciliation. Messaging-platform and UI/desktop sources have separate
+ # lifecycle owners; unknown/future sources fail closed (#60609).
+ _AUTO_PRUNE_STALE_OPEN_SOURCES: Tuple[str, ...] = (
+ "cli",
+ "cron",
+ "kanban",
+ "acp",
+ "api_server",
+ "subagent",
+ "tool",
+ )
+
# ── Write-contention tuning ──
# With multiple hermes processes (gateway + CLI sessions + worktree agents)
# all sharing one state.db, WAL write-lock contention causes visible TUI
@@ -5003,6 +5421,14 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self._db_file_application_id: int = 0
self._db_file_generation_token: str = ""
self._db_replaced = False
+ # Sticky: set once a write on THIS handle reports bare SQLITE_CORRUPT /
+ # NOTADB that is not FTS-scoped and not a replaced-file case. Never
+ # cleared; the recovery boundary is a process restart on a repaired or
+ # restored file (see StateDbCorruptError).
+ self._db_corrupt = False
+ self._db_corrupt_reason = ""
+ self._db_sidecar_identity: Dict[str, tuple] = {}
+ self._db_wal_generation_lost = False
# One-shot guard for the usermerge-floor config write on the
# incremental FTS merge cadence (see _merge_fts_incrementally).
self._fts_usermerge_floor_applied = False
@@ -5041,46 +5467,67 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# must already exist + be initialised (callers guard on
# db_path.exists()); a SELECT against an empty file raises and
# the caller degrades per-profile.
- self._conn = _connect_tracked_db(
- f"file:{self.db_path}?mode=ro",
- tracking_path=self.db_path,
- uri=True,
- check_same_thread=False,
- timeout=1.0,
- isolation_level=None,
- )
- self._conn.row_factory = sqlite3.Row
- # FTS capability flags normally come from writable schema
- # initialisation. Probe existing virtual tables with SELECTs
- # only so read-only search keeps its FTS and trigram paths.
- # Close the connection on ANY probe failure (e.g. malformed
- # schema raises DatabaseError, not the OperationalError the
- # probe handles). The constructor's outer finally also covers
- # failures before this probe and BaseException paths, so a
- # leaked tracked connection cannot block _backup_db_file's
- # raw-copy for the rest of the process — the writable heal
- # that follows would then repair WITHOUT its forensic backup.
- try:
- apply_database_pragmas(self._conn, db_label="state.db")
- cursor = self._conn.cursor()
- self._fts_enabled = (
- self._fts_table_probe(cursor, "messages_fts") is True
- )
- if self._fts_enabled:
- self._trigram_available = (
- self._fts_table_probe(
- cursor,
- "messages_fts_trigram",
- )
- is True
- )
- except BaseException:
- conn, self._conn = self._conn, None
+ open_attempt = 0
+ while True:
try:
- conn.close()
- except Exception:
- pass
- raise
+ self._conn = _connect_tracked_db(
+ f"file:{self.db_path}?mode=ro",
+ tracking_path=self.db_path,
+ uri=True,
+ check_same_thread=False,
+ timeout=1.0,
+ isolation_level=None,
+ )
+ self._conn.row_factory = sqlite3.Row
+ # FTS capability flags normally come from writable schema
+ # initialisation. Probe existing virtual tables with
+ # SELECTs only so read-only search keeps its FTS and
+ # trigram paths. Close the connection on ANY probe
+ # failure (e.g. malformed schema raises DatabaseError,
+ # not the OperationalError the probe handles). The
+ # constructor's outer finally also covers failures
+ # before this probe and BaseException paths, so a
+ # leaked tracked connection cannot block
+ # _backup_db_file's raw-copy for the rest of the
+ # process — the writable heal that follows would then
+ # repair WITHOUT its forensic backup.
+ try:
+ apply_database_pragmas(self._conn, db_label="state.db")
+ cursor = self._conn.cursor()
+ self._fts_enabled = (
+ self._fts_table_probe(cursor, "messages_fts")
+ is True
+ )
+ if self._fts_enabled:
+ self._trigram_available = (
+ self._fts_table_probe(
+ cursor,
+ "messages_fts_trigram",
+ )
+ is True
+ )
+ except BaseException:
+ conn, self._conn = self._conn, None
+ try:
+ conn.close()
+ except Exception:
+ pass
+ raise
+ break
+ except sqlite3.OperationalError as ioerr:
+ # A WAL checkpoint / reset / frame-flush in flight on
+ # the writer side can surface SQLITE_IOERR to a
+ # concurrent mode=ro reader (it cannot perform the
+ # recovery the read needs — recovery writes the -shm
+ # index, which mode=ro refuses). The transition closes
+ # in milliseconds, so retry a bounded number of times
+ # before classifying the store as failed (#100436).
+ if not _is_transient_read_only_ioerr(
+ ioerr, attempt=open_attempt
+ ):
+ raise
+ open_attempt += 1
+ time.sleep(_READ_ONLY_IOERR_RETRY_BACKOFF_S)
self._record_db_file_identity()
initialization_complete = True
return
@@ -5133,6 +5580,10 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
raise sqlite3.DatabaseError(msg)
def _connect_and_init():
+ # Refuse before sqlite3.connect (under the startup lock) so we
+ # cannot mint a replacement WAL while a live writer still
+ # holds a deleted sidecar inode.
+ refuse_deleted_wal_generation(self.db_path)
self._conn = _connect_tracked_db(
str(self.db_path),
check_same_thread=False,
@@ -5523,6 +5974,17 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# through stale WAL/shm assumptions (#89332). Refuse instead.
if self._db_replaced or self._db_file_was_replaced():
self._halt_db_replaced()
+ # A quarantined handle must never come back: reopening would hand a
+ # fresh connection (and its own close-time checkpoint) to a file we
+ # already know is structurally damaged.
+ if self._db_corrupt:
+ raise self._corrupt_error(
+ f"state.db connection for {self.db_path} is quarantined after "
+ f"structural corruption; refusing to reopen for a {context} "
+ "after close(). "
+ )
+ if self._db_wal_generation_lost or self._wal_generation_was_lost():
+ self._halt_deleted_wal_generation()
logger.warning(
"state.db connection for %s was closed while a %s was still in "
"flight — reopening (teardown/worker race, #94736)",
@@ -5848,6 +6310,13 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# Set on the first compression-busy collision so the short wait is
# measured from then, not from the start of the write.
compression_deadline: Optional[float] = None
+ # One retry for SQLITE_IOERR raised by BEGIN IMMEDIATE itself. The
+ # callback has not run at that point, so there is no durable effect
+ # to replay and the retry is exactly-once safe (#99502's contract).
+ # Once the callback starts, an IOERR leaves the write's settlement
+ # unknown and must propagate — this helper owns non-idempotent
+ # transcript/counter mutations, not just idempotent UPSERTs.
+ ioerr_begin_retried = False
# Transient engine-level error observed on contended WAL appends
# (dual gateway/agent writers; FTS5 trigram sync holds the write
@@ -5860,7 +6329,9 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
return "no more rows available" in str(exc).lower()
while True:
+ self._raise_if_db_corrupt()
self._raise_if_db_replaced()
+ fn_started = False
try:
with self._lock:
if self._conn is None:
@@ -5869,6 +6340,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self._reopen_after_close_locked(context="write")
self._conn.execute("BEGIN IMMEDIATE")
try:
+ fn_started = True
result = fn(self._conn)
self._conn.commit()
except BaseException:
@@ -5921,7 +6393,21 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
) from exc
if _is_no_more_rows(exc) and self._sleep_before_write_retry(deadline, patience_s):
continue
- # Non-lock error or patience exhausted — propagate.
+ if (
+ _DISK_IO_ERROR_MARKER in err_msg
+ and not fn_started
+ and not ioerr_begin_retried
+ and self._sleep_before_write_retry(deadline, patience_s)
+ ):
+ # BEGIN IMMEDIATE itself hit a transient WAL-transition
+ # IOERR. Nothing has been mutated, so retrying on the SAME
+ # connection replays nothing. Never close()+reopen to
+ # "heal" it: close() cancels this process's POSIX locks on
+ # the file for every sibling connection (howtocorrupt §2.2).
+ ioerr_begin_retried = True
+ continue
+ # Non-lock error, the callback already ran (settlement is
+ # unknown — do not replay), or patience exhausted.
raise
except sqlite3.DatabaseError as exc:
if _is_no_more_rows(exc) and self._sleep_before_write_retry(deadline, patience_s):
@@ -5946,6 +6432,11 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# explicit repair paths retain rebuild ownership.
if self._enter_fts_fail_open(exc):
continue
+ # Bare SQLITE_CORRUPT / NOTADB that survived the replaced-file
+ # check and the FTS-scoped fail-open is structural damage:
+ # quarantine the handle (see StateDbCorruptError).
+ if self._is_structural_corruption_error(exc):
+ self._halt_db_corrupt(exc)
raise
except sqlite3.Error as exc:
# Catch-all for builds that surface 'no more rows available'
@@ -6001,6 +6492,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
def _record_db_file_identity(self) -> None:
"""Snapshot inode plus the on-disk generation header when present."""
self._db_file_identity = _stat_db_file_identity(self.db_path)
+ self._db_sidecar_identity = _stat_sqlite_sidecar_identity(self.db_path)
disk_id = _read_sqlite_application_id(self.db_path)
if disk_id:
self._db_file_application_id = disk_id
@@ -6035,11 +6527,151 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
logger.error(_STATE_DB_REPLACED_MSG)
raise StateDbReplacedError(_STATE_DB_REPLACED_MSG)
+ def _wal_generation_was_lost(self) -> bool:
+ """True when the WAL/SHM generation this instance opened is gone.
+
+ Steady state (a sidecar generation is recorded): pure stat — a
+ recorded inode that is missing or replaced by a new file at the same
+ path means the generation split. No /proc walk on healthy writes.
+
+ Empty-identity state (fresh DB whose WAL appears only after open, or
+ identity cleared by a clean ``close()``): fall back to a
+ ``/proc/self/fd`` deleted-fd probe, and adopt the current sidecars as
+ this handle's generation once the probe comes back clean. The full
+ ``/proc/*/fd`` walk is reserved for
+ :func:`refuse_deleted_wal_generation` on open, where we must see
+ *foreign* deleted holders before ``sqlite3.connect`` mints a new WAL.
+ """
+ recorded = self._db_sidecar_identity or {}
+ base = os.fspath(self.db_path)
+ if recorded:
+ for suffix, recorded_ident in recorded.items():
+ current = _stat_db_file_identity(Path(base + suffix))
+ if current is None or current != recorded_ident:
+ return True
+ return False
+ if not self._wal_active:
+ # No WAL on this handle (journal_mode=delete/truncate fallback):
+ # there is no sidecar generation to lose, and probing every write
+ # would put a /proc walk on the hot path of exactly the
+ # delete-mode deployments the field report used as containment.
+ return False
+ if sys.platform.startswith("linux"):
+ watched = _watched_sqlite_sidecar_paths(self.db_path)
+ fd_dir = f"/proc/{os.getpid()}/fd"
+ try:
+ for fd in os.listdir(fd_dir):
+ try:
+ target = os.readlink(f"{fd_dir}/{fd}")
+ except OSError:
+ continue
+ if " (deleted)" in target and _canonical_sqlite_path(target) in watched:
+ return True
+ except OSError:
+ return False
+ # Probe clean (or unavailable on this platform): adopt whatever
+ # sidecar generation exists now so subsequent writes use the cheap
+ # stat check.
+ current_identity = _stat_sqlite_sidecar_identity(self.db_path)
+ if current_identity:
+ self._db_sidecar_identity = current_identity
+ return False
+
+ def _halt_deleted_wal_generation(self) -> None:
+ """Stop writes; do not mint or keep committing on a split WAL."""
+ self._db_wal_generation_lost = True
+ logger.error(_DELETED_WAL_GENERATION_MSG)
+ raise DeletedWalGenerationError(_DELETED_WAL_GENERATION_MSG)
+
def _raise_if_db_replaced(self) -> None:
if self._db_replaced:
raise StateDbReplacedError(_STATE_DB_REPLACED_MSG)
+ if self._db_wal_generation_lost:
+ raise DeletedWalGenerationError(_DELETED_WAL_GENERATION_MSG)
if self._db_file_was_replaced():
self._halt_db_replaced()
+ if self._wal_generation_was_lost():
+ self._halt_deleted_wal_generation()
+
+ @classmethod
+ def _is_structural_corruption_error(cls, exc: BaseException) -> bool:
+ """Bare SQLITE_CORRUPT/NOTADB with no FTS provenance.
+
+ ``_is_fts_write_corruption_error`` is the positive FTS classifier;
+ everything else in the ``corrupt`` bucket of
+ ``classify_persistence_error`` is damage to a canonical B-tree, the
+ schema, or the freelist — never repairable from the live write path.
+ """
+ if not isinstance(exc, sqlite3.DatabaseError):
+ return False
+ if isinstance(exc, StateDbCorruptError):
+ return False
+ if cls._is_fts_write_corruption_error(exc):
+ return False
+ return classify_persistence_error(exc) == "corrupt"
+
+ def _corrupt_error(self, prefix: str = "") -> "StateDbCorruptError":
+ """Build the quarantine error for this handle (message assembled once)."""
+ return StateDbCorruptError(
+ f"{prefix}{_STATE_DB_CORRUPT_MSG} (cause: {self._db_corrupt_reason})"
+ )
+
+ def _halt_db_corrupt(self, exc: BaseException) -> None:
+ """Quarantine this handle and raise; never run in-file repair here."""
+ self._db_corrupt = True
+ self._db_corrupt_reason = str(exc)
+ self._disable_close_time_checkpoint()
+ logger.error(
+ "state.db %s reported structural corruption outside the FTS "
+ "indexes (%s); quarantining this handle: no further writes, no "
+ "automatic reopen, no explicit WAL checkpoint at close. Stop the "
+ "gateway and run `hermes sessions recover --source %s "
+ "--inspect-only`.",
+ self.db_path,
+ exc,
+ self.db_path,
+ )
+ err = self._corrupt_error()
+ for attr in ("sqlite_errorcode", "sqlite_errorname"):
+ value = getattr(exc, attr, None)
+ if value is not None:
+ setattr(err, attr, value)
+ raise err from exc
+
+ def _disable_close_time_checkpoint(self) -> None:
+ """Best-effort: stop SQLite's own last-connection checkpoint on close.
+
+ Skipping our explicit ``PRAGMA wal_checkpoint(PASSIVE)`` in
+ ``close()`` is not enough on its own: ``sqlite3.Connection.close()``
+ still runs SQLite's internal last-connection PASSIVE checkpoint and
+ unlinks the ``-wal``/``-shm`` sidecars. On the field incident's file
+ that close-time checkpoint is exactly what wrote 15 pages under the
+ wrong page numbers. Python 3.12+ exposes the switch as
+ ``Connection.setconfig(SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE)``; on 3.11
+ neither the constant nor ``setconfig`` exists, so the internal
+ checkpoint remains (it can only carry pre-quarantine committed
+ frames — no further writes are accepted on this handle).
+ """
+ flag = getattr(sqlite3, "SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE", None)
+ if flag is None:
+ return
+ conn = self._conn
+ setconfig = getattr(conn, "setconfig", None)
+ if conn is None or setconfig is None:
+ return
+ try:
+ setconfig(flag, True)
+ except Exception:
+ logger.debug(
+ "Could not disable SQLite's close-time checkpoint on the "
+ "quarantined handle for %s",
+ self.db_path,
+ exc_info=True,
+ )
+
+ def _raise_if_db_corrupt(self) -> None:
+ if self._db_corrupt:
+ raise self._corrupt_error()
def _sleep_before_write_retry(
self, deadline: float, patience_s: float
@@ -6108,15 +6740,11 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
if psutil is None:
return [(-1, "open-file scan unavailable")]
- def _canonical(path: str) -> str:
- clean = path.removesuffix(" (deleted)")
- return os.path.normcase(os.path.abspath(clean))
-
db_path = os.path.abspath(os.fspath(self.db_path))
watched = {
- _canonical(db_path),
- _canonical(db_path + "-wal"),
- _canonical(db_path + "-shm"),
+ _canonical_sqlite_path(db_path),
+ _canonical_sqlite_path(db_path + "-wal"),
+ _canonical_sqlite_path(db_path + "-shm"),
}
holders: List[Tuple[int, str]] = []
@@ -6155,7 +6783,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
target = os.readlink(f"{fd_dir}/{fd}")
except OSError:
continue
- if _canonical(target) in watched:
+ if _canonical_sqlite_path(target) in watched:
holders.append((pid, target))
except Exception as exc:
logger.warning(
@@ -6180,7 +6808,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# Linux-specific (systemd units running as root).
for opened in info.get("open_files") or ():
path = getattr(opened, "path", "")
- if path and _canonical(path) in watched:
+ if path and _canonical_sqlite_path(path) in watched:
holders.append((pid, path))
except Exception as exc:
logger.warning(
@@ -6262,8 +6890,11 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
"""
if not self._fts_enabled or not self._is_fts_write_corruption_error(exc):
return False
+ self._raise_if_db_corrupt()
if self._db_replaced or self._db_file_was_replaced():
self._halt_db_replaced()
+ if self._db_wal_generation_lost or self._wal_generation_was_lost():
+ self._halt_deleted_wal_generation()
try:
with self._lock:
@@ -6329,6 +6960,8 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
databases (65K+ pages) due to the exclusive-lock I/O pressure
from checkpointing thousands of frames at once (issue #45383).
"""
+ if self._db_corrupt:
+ return # quarantined: never checkpoint over a damaged image
try:
with self._lock:
result = self._conn.execute(
@@ -6420,7 +7053,20 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self._close_read_conn(conn)
with self._lock:
if self._conn:
- if not self.read_only:
+ if self._db_corrupt:
+ # Quarantined handle (see StateDbCorruptError): no explicit
+ # checkpoint over a damaged page image.
+ logger.warning(
+ "Skipping the close-time WAL checkpoint for %s: this "
+ "handle observed structural corruption (%s). Take a "
+ "snapshot of state.db, -wal and -shm before restarting, "
+ "then run `hermes sessions recover --source %s "
+ "--inspect-only`.",
+ self.db_path,
+ self._db_corrupt_reason,
+ self.db_path,
+ )
+ elif not self.read_only:
# PASSIVE, not TRUNCATE. Every cron run_agent opens+closes a
# transient SessionDB, so a TRUNCATE here fires a full WAL
# reset many times/hour, racing the gateway's long-lived
@@ -6437,6 +7083,12 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
)
conn, self._conn = self._conn, None
self._close_connection_quietly(conn)
+ # A clean close of the last connection lets SQLite unlink the
+ # WAL/SHM sidecars — a legitimate end of this handle's sidecar
+ # generation, not a split (#94736 late writes must still
+ # self-heal). Drop the recorded generation so a teardown-race
+ # reopen re-adopts whatever exists then instead of halting.
+ self._db_sidecar_identity = {}
def __del__(self) -> None:
"""Safety net: close the connection if the caller forgot.
@@ -6705,7 +7357,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self._delete_unreferenced_system_prompts(conn)
if parent_session_id:
conn.execute(
- """UPDATE sessions
+ f"""UPDATE sessions
SET cwd = COALESCE(sessions.cwd,
(SELECT p.cwd FROM sessions p
WHERE p.id = sessions.parent_session_id)),
@@ -6717,7 +7369,8 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
WHERE p.id = sessions.parent_session_id)),
profile_name = COALESCE(sessions.profile_name,
(SELECT p.profile_name FROM sessions p
- WHERE p.id = sessions.parent_session_id))
+ WHERE p.id = sessions.parent_session_id
+ AND ({_SAME_KEY_NAMESPACE_SQL})))
WHERE id = ? AND parent_session_id IS NOT NULL""",
(session_id,),
)
@@ -7289,6 +7942,15 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# tuple so we never cross chats/threads/users.
if chat_id is None or chat_type is None:
return None
+ # Profile fence (#74285): a Telegram DM's peer tuple is identical
+ # for every bot (chat_id == user_id, no thread), so a sibling
+ # profile's row written into this store before the per-profile
+ # partition (legacy data) would otherwise be adopted here. Every
+ # profile-tree store has one owner; a row is ours when its
+ # profile_name is the owner or NULL (legacy rows this store
+ # minted). Stores outside the tree derive no owner and keep the
+ # historical unfenced behavior.
+ owner = self._own_profile_name()
row = conn.execute(
f"""
SELECT s.*,
@@ -7304,6 +7966,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
AND COALESCE(s.chat_id, '') = COALESCE(?, '')
AND COALESCE(s.chat_type, '') = COALESCE(?, '')
AND COALESCE(s.thread_id, '') = COALESCE(?, '')
+ AND (? IS NULL OR COALESCE(s.profile_name, ?) = ?)
AND (s.ended_at IS NULL OR s.end_reason IN ({_RECOVERABLE_END_REASONS_SQL}))
AND (COALESCE(s.message_count, 0) > 0 OR EXISTS (
SELECT 1 FROM messages WHERE messages.session_id = s.id LIMIT 1
@@ -7323,7 +7986,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
ORDER BY COALESCE(s.last_activity_at, s.started_at) DESC
LIMIT 1
""",
- (source, user_id, chat_id, chat_type, thread_id),
+ (source, user_id, chat_id, chat_type, thread_id, owner, owner, owner),
).fetchone()
return self._session_row_dict(row) if row else None
@@ -8476,6 +9139,54 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self._execute_write(_do)
+ def get_compression_recovery_deadline(self, session_id: str) -> float:
+ """Return the persisted anti-thrash recovery deadline (wall-clock epoch).
+
+ ``0.0`` means "not armed". The deadline is the durable half of the
+ #14694 recovery clock: the gateway rebuilds the compressor on every
+ turn / cache eviction, so a process-local deadline restarted the
+ wait on each rebuild and a tripped session never earned its probe
+ (#100185).
+ """
+ if not session_id:
+ return 0.0
+ with self._read_ctx() as conn:
+ if conn is None:
+ return 0.0
+ row = conn.execute(
+ "SELECT compression_recovery_deadline FROM sessions WHERE id = ?",
+ (session_id,),
+ ).fetchone()
+ if row is None:
+ return 0.0
+ value = (
+ row["compression_recovery_deadline"]
+ if isinstance(row, sqlite3.Row)
+ else row[0]
+ )
+ try:
+ return max(0.0, float(value or 0.0))
+ except (TypeError, ValueError):
+ return 0.0
+
+ def set_compression_recovery_deadline(self, session_id: str, deadline: float) -> None:
+ """Persist the anti-thrash recovery deadline; ``0`` / ``None`` disarms it."""
+ if not session_id:
+ return
+ try:
+ normalized = max(0.0, float(deadline or 0.0))
+ except (TypeError, ValueError):
+ normalized = 0.0
+ stored = normalized if normalized > 0.0 else None
+
+ def _do(conn):
+ conn.execute(
+ "UPDATE sessions SET compression_recovery_deadline = ? WHERE id = ?",
+ (stored, session_id),
+ )
+
+ self._execute_write(_do)
+
# ──────────────────────────────────────────────────────────────────────
# Compression locks
# ──────────────────────────────────────────────────────────────────────
@@ -9037,6 +9748,25 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self._delete_unreferenced_system_prompts(conn)
self._execute_write(_do)
+ def update_session_tool_names(
+ self, session_id: str, tool_names: Optional[List[str]]
+ ) -> None:
+ """Persist the session's resolved ``tools[]`` name order (JSON array).
+
+ Read back by ``tools.mcp_tool.restore_agent_tool_prefix`` when a fresh
+ ``AIAgent`` is rebuilt for an existing session (gateway agent-cache
+ eviction) so a flipped ``check_fn`` verdict can't fork the cached tool
+ prefix. ``None`` clears the pin.
+ """
+ payload = json.dumps(list(tool_names)) if tool_names is not None else None
+
+ def _do(conn):
+ conn.execute(
+ "UPDATE sessions SET tool_names = ? WHERE id = ?",
+ (payload, session_id),
+ )
+ self._execute_write(_do)
+
def update_session_model(
self, session_id: str, model: str, provider: Optional[str] = None
) -> None:
@@ -10054,57 +10784,54 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
max_idle_seconds: float,
sources: Tuple[str, ...] = ("tui", "desktop", "subagent"),
exclude_ids: Tuple[str, ...] = (),
+ exclude_pinned: bool = False,
heartbeat_staleness_seconds: Optional[float] = None,
heartbeat_ownership_grace_seconds: Optional[float] = None,
+ respect_gateway_heartbeats: bool = True,
) -> List[str]:
"""Close session rows orphaned by a dead gateway process (#65194, #94895).
The TUI/desktop gateway reaps disconnected websocket sessions with an
in-process ``threading.Timer`` grace timer; a gateway restart destroys
- the timer and leaves the row ``ended_at IS NULL`` forever. This is
- the startup-time complement: it closes rows for the given ``sources``
- whose ``started_at`` AND newest ``messages.timestamp`` are both older
- than ``max_idle_seconds``, with a distinct
+ the timer and leaves the row ``ended_at IS NULL`` forever. This is the
+ startup-time complement: it closes rows for the given ``sources`` whose
+ ``started_at`` and canonical last-activity time are both older than
+ ``max_idle_seconds``, with a distinct
``end_reason='startup_orphan_reap'`` for traceability.
- Both timestamps must be stale on purpose: message recency alone would
- sweep a freshly created compression/branch child carrying old copied
- message timestamps, while ``started_at`` alone would sweep a
- long-lived session that is still actively producing messages.
- Message-less rows fall back to ``started_at`` via COALESCE.
+ Canonical activity is the newest of ``last_activity_at`` (the in-turn
+ heartbeat) and the newest durable message timestamp, falling back to
+ ``started_at``. The separate ``started_at`` predicate protects freshly
+ created compression/branch children whose copied activity is old.
- Only pass sources owned by the local UI stack (never messaging-gateway
+ Only pass sources whose lifecycle the caller owns (never messaging-gateway
platforms like ``telegram`` — ending those triggers the #60609 routing
- loop). ``exclude_ids`` spares rows this process still holds in
- memory (a ``session.resume`` that landed during the startup grace
- window). Non-destructive: messages are preserved and the row remains
- resumable. First-reason-wins is preserved via ``ended_at IS NULL``.
+ loop). ``exclude_ids`` spares rows this process still holds in memory
+ (a ``session.resume`` that landed during the startup grace window).
+ ``exclude_pinned`` is intended for broad automatic sweeps; pinned rows
+ remain explicitly recoverable. Non-destructive: messages are preserved
+ and the row remains resumable. First-reason-wins is preserved via
+ ``ended_at IS NULL``.
- Cross-backend liveness (#94895): when one ``state.db`` is shared by
- N serve / gateway processes (isolated backends, fixed-port launchd
- ``hermes serve``, desktop WS sidecar), each backend registers a row
- in ``gateway_heartbeats`` refreshed every few seconds. A row is
- only reaped when ``started_at``/message staleness hold AND no live
- backend (heartbeat refreshed within ``heartbeat_staleness_seconds``,
- default ``2 * max_idle_seconds``) could plausibly own it.
+ Cross-backend liveness (#94895): when one ``state.db`` is shared by N
+ serve / gateway processes, each backend refreshes a row in
+ ``gateway_heartbeats``. With ``respect_gateway_heartbeats`` enabled, a
+ row is only reaped when activity staleness holds AND no live backend
+ (heartbeat refreshed within ``heartbeat_staleness_seconds``, default
+ ``2 * max_idle_seconds``) could plausibly own it. Disable that gate only
+ for sources whose lifecycle is explicitly owned by state.db itself.
Ownership inference: a live backend B ``owns`` a session S if
``B.started_at <= S.started_at + heartbeat_ownership_grace_seconds``
- (default ``heartbeat_staleness_seconds``). The grace window
- accommodates the deploy-time migration case where a backend just
- wrote its first heartbeat row while its existing open sessions
- predate the schema. The grace is bounded by the staleness window
- so a fresh PID-reuse respawn cannot indefinitely protect sessions
- inherited from a dead predecessor.
+ (default ``heartbeat_staleness_seconds``). The grace window covers a
+ migrating backend whose existing sessions predate its first heartbeat,
+ but is bounded so a fresh PID-reuse respawn cannot protect rows forever.
+ With no fresh heartbeat the predicate falls back to the legacy sweep.
- When NO backend has ever written a heartbeat (legacy deployment
- mid-upgrade before any process has registered) the predicate falls
- back to the original behavior so we never silently strand a row
- that pre-dates the schema.
-
- The SELECT + UPDATE run in one ``BEGIN IMMEDIATE`` write, so a sibling
- process cannot sneak a new message or end-reason between the
- staleness check and the close. Returns the swept session ids.
+ The SELECT, live-lease validation, and UPDATE run in one
+ ``BEGIN IMMEDIATE`` transaction. Active turn leases or compression
+ locks spare the row; expired/reclaimed guards are removed so their
+ former owner is fenced. Returns the swept session ids.
"""
srcs = tuple(s for s in sources if s)
if max_idle_seconds <= 0 or not srcs:
@@ -10116,56 +10843,76 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
)
hb_grace = (
heartbeat_ownership_grace_seconds
- if heartbeat_ownership_grace_seconds and heartbeat_ownership_grace_seconds >= 0
+ if heartbeat_ownership_grace_seconds is not None
+ and heartbeat_ownership_grace_seconds >= 0
else hb_staleness
)
- cutoff = time.time() - max_idle_seconds
- hb_cutoff = time.time() - hb_staleness
+ now = time.time()
+ cutoff = now - max_idle_seconds
+ hb_cutoff = now - hb_staleness
placeholders = ",".join("?" for _ in srcs)
staleness = (
- "started_at < ? AND COALESCE((SELECT MAX(m.timestamp) FROM messages m"
- " WHERE m.session_id = sessions.id), started_at) < ?"
+ f"started_at < ? AND {_sql_session_last_active('sessions')} < ?"
)
-
- def _do(conn):
- # Cross-process liveness gate (#94895). A session is "owned by
- # a live backend" if any row in gateway_heartbeats is fresh
- # (last_heartbeat >= hb_cutoff) AND was alive no later than
- # ``sessions.started_at + hb_grace`` (heartbeats.started_at <=
- # sessions.started_at + hb_grace). If at least one live backend
- # matches, the row is not orphaned.
- #
- # ``hb_cutoff`` and ``hb_grace`` are computed above so all
- # backends running concurrent sweep queries agree on the same
- # boundaries. We do NOT clear heartbeats here — that's each
- # backend's atexit responsibility via ``clear_backend_heartbeat``.
- orphan_predicate = (
- f"{staleness} AND NOT EXISTS ("
+ pin_scope = " AND COALESCE(pinned, 0) = 0" if exclude_pinned else ""
+ heartbeat_params: Tuple[float, ...] = ()
+ orphan_predicate = staleness
+ if respect_gateway_heartbeats:
+ orphan_predicate += (
+ " AND NOT EXISTS ("
"SELECT 1 FROM gateway_heartbeats h"
" WHERE h.last_heartbeat >= ?"
- f" AND h.started_at <= sessions.started_at + ?"
+ " AND h.started_at <= sessions.started_at + ?"
")"
)
+ heartbeat_params = (hb_cutoff, hb_grace)
+
+ def _do(conn):
rows = conn.execute(
f"SELECT id FROM sessions WHERE ended_at IS NULL"
- f" AND source IN ({placeholders}) AND {orphan_predicate}",
- (*srcs, cutoff, cutoff, hb_cutoff, hb_grace),
+ f" AND source IN ({placeholders}){pin_scope}"
+ f" AND {orphan_predicate}",
+ (*srcs, cutoff, cutoff, *heartbeat_params),
).fetchall()
excluded = {str(x) for x in exclude_ids if x}
- victims = [str(r["id"]) for r in rows if str(r["id"]) not in excluded]
+ victims = []
+ for row in rows:
+ sid = str(row["id"])
+ if sid in excluded:
+ continue
+ try:
+ self._check_transcript_write_guards(
+ conn,
+ sid,
+ compression_lock_holder=None,
+ turn_lease_holder=None,
+ reject_active_turn_lease=True,
+ reject_active_compression_lock=True,
+ )
+ except (
+ SessionCompressionInProgressError,
+ SessionTurnLeaseLostError,
+ ):
+ continue
+ victims.append(sid)
if not victims:
return []
- now = time.time()
+ closed_at = time.time()
marks = ",".join("?" for _ in victims)
- # Re-apply the same predicates under the write lock so a
- # row that raced to activity between SELECT and UPDATE is
- # spared (and so a freshly registered heartbeat from a sibling
- # that started during this transaction can still save the row).
+ # Re-apply every scope/liveness predicate under the write lock.
conn.execute(
f"UPDATE sessions SET ended_at = ?, end_reason = 'startup_orphan_reap'"
f" WHERE id IN ({marks}) AND ended_at IS NULL"
+ f" AND source IN ({placeholders}){pin_scope}"
f" AND {orphan_predicate}",
- (now, *victims, cutoff, cutoff, hb_cutoff, hb_grace),
+ (
+ closed_at,
+ *victims,
+ *srcs,
+ cutoff,
+ cutoff,
+ *heartbeat_params,
+ ),
)
return victims
@@ -10360,8 +11107,8 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# Bot Mode's forever-chat registry: the session titled exactly this, on a
# bot's profile, IS the bot's canonical chat — resolved by exact-title
# lookup on every open (no session-id pointer exists). The title is the
- # identity, which is why _set_session_title refuses user renames of a
- # hidden row holding it (#92473).
+ # identity, which is why _set_session_title refuses renames of a hidden
+ # row holding it (#92473).
CANONICAL_BOT_CHAT_TITLE = "Bot Chat"
@classmethod
@@ -10478,7 +11225,8 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
stored title has strictly lower authority, so the instant ``derived``
title upgrades to ``llm`` exactly once and neither can ever overwrite a
name the user typed. Re-running the titler on an already-``llm`` row is
- a no-op, which is what stops a session renaming itself.
+ a no-op, which is what stops a session renaming itself. The one thing
+ no writer may do is move a hidden canonical Bot Chat off its title.
The read and the write are one compare-and-swap inside a single
transaction, so a manual ``/title`` racing an in-flight generation
@@ -10503,18 +11251,21 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# surface funnels through (gateway session.title, /title, CLI
# rename, REST). Hidden is the discriminator: canonical chats are
# born hidden; an ordinary visible session a user happens to call
- # "Bot Chat" stays freely renameable.
+ # "Bot Chat" stays freely renameable. Provenance-blind: an
+ # automatic llm write outranks a derived title, so the auto-titler
+ # would otherwise rename the row too (#99517) — it no-ops instead.
if (
- is_user
- and (current["title"] or "") == self.CANONICAL_BOT_CHAT_TITLE
+ (current["title"] or "") == self.CANONICAL_BOT_CHAT_TITLE
and bool(current["hidden"])
and title != self.CANONICAL_BOT_CHAT_TITLE
):
- raise ValueError(
- "This is the bot's canonical Bot Chat — its name is its "
- "identity, and renaming it would orphan the conversation. "
- "To start fresh, create a new bot instead."
- )
+ if is_user:
+ raise ValueError(
+ "This is the bot's canonical Bot Chat — its name is its "
+ "identity, and renaming it would orphan the conversation. "
+ "To start fresh, create a new bot instead."
+ )
+ return 0
if not is_user and current["title"] is not None:
if self._title_rank(current["title_source"]) >= new_rank:
return 0
@@ -11050,8 +11801,13 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
return f"{base} #{max_num + 1}"
- def get_compression_tip(self, session_id: str) -> Optional[str]:
- """Walk the compression-continuation chain forward and return the tip.
+ def get_compression_chain(self, session_id: str) -> List[str]:
+ """Walk the compression-continuation chain forward and return every id.
+
+ Root-first order, ending at the tip; ``[session_id]`` when no
+ continuation exists. ``get_compression_tip`` is this walk's last
+ element — kept as the single implementation so the two can never
+ disagree about what the chain is.
A compression continuation is a child of a session whose
``end_reason = 'compression'``. Older builds tried to distinguish
@@ -11072,6 +11828,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
continuation exists.
"""
current = session_id
+ chain = [current] if current else []
seen = {current} if current else set()
# Bound the walk defensively — compression chains this deep are
# pathological and shouldn't happen in practice. 100 = plenty.
@@ -11102,13 +11859,21 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
)
row = cursor.fetchone()
if row is None:
- return current
+ return chain
child_id = row["id"]
if not child_id or child_id in seen:
- return current
+ return chain
seen.add(child_id)
current = child_id
- return current
+ chain.append(child_id)
+ return chain
+
+ def get_compression_tip(self, session_id: str) -> Optional[str]:
+ """The live tip of a compression-continuation chain (see
+ ``get_compression_chain`` for the walk's semantics). Returns the input
+ id when no continuation exists."""
+ chain = self.get_compression_chain(session_id)
+ return chain[-1] if chain else session_id
# Columns excluded from compact_rows projections: only the payload-heavy
# blob no list consumer renders. Everything else — including gateway
@@ -11486,12 +12251,15 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# call per compression root. Batch that half instead: resolve
# every tip id first, then fetch all tip rows in a single query.
tip_ids_by_root: Dict[str, str] = {}
+ chain_by_root: Dict[str, List[str]] = {}
for s in sessions:
if s.get("end_reason") != "compression":
continue
- tip_id = self.get_compression_tip(s["id"])
+ chain = self.get_compression_chain(s["id"])
+ tip_id = chain[-1] if chain else s["id"]
if tip_id != s["id"]:
tip_ids_by_root[s["id"]] = tip_id
+ chain_by_root[s["id"]] = chain
tip_rows = (
self._get_session_rich_rows_batch(
@@ -11519,6 +12287,13 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
if key in tip_row:
merged[key] = tip_row[key]
merged["_lineage_root_id"] = s["id"]
+ # Every id on the chain, intermediates included. Root and tip
+ # alone are not enough client-side: a persisted tile or route
+ # can hold a MIDDLE segment's id (it was the tip when opened,
+ # then rotated again), and with only the root/tip pair such a
+ # surface can no longer prove it names this conversation —
+ # which is how one chat ends up open twice after compaction.
+ merged["_lineage_ids"] = chain_by_root.get(s["id"]) or None
projected.append(merged)
sessions = projected
@@ -11675,6 +12450,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
turn_lease_ttl_seconds: float = 300.0,
reject_active_turn_lease: bool = False,
reject_active_compression_lock: bool = False,
+ allow_closed_compression_parent: bool = False,
) -> None:
"""Transcript-write admission checks, run INSIDE the write txn.
@@ -11774,6 +12550,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
session is not None
and session["ended_at"] is not None
and session["end_reason"] == "compression"
+ and not allow_closed_compression_parent
):
raise CompressionSessionClosedError(session_id)
@@ -13411,11 +14188,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
halves the resume's DB work versus two separate calls, with byte-identical
output (see test_get_resume_conversations_matches_separate_reads).
"""
- session_ids = (
- [session_id]
- if self._is_explicit_branch_session(session_id)
- else self._session_lineage_root_to_tip(session_id)
- )
+ session_ids = self._resume_lineage_ids(session_id)
with self._read_ctx() as conn:
placeholders = ",".join("?" for _ in session_ids)
rows = conn.execute(
@@ -13451,9 +14224,32 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
)
return model_history, display_history
- def get_resume_message_count(self, session_id: str) -> int:
- """Count active rows that a full resume would materialize."""
- session_ids = self._session_lineage_root_to_tip(session_id)
+ def _resume_lineage_ids(self, session_id: str) -> List[str]:
+ """Session ids a full (display) resume materializes for *session_id*.
+
+ Compression continuations need their ended ancestors' rows for the
+ display transcript; an explicit ``/branch`` copy already owns its
+ transcript, so its lineage is itself alone. This is the ONE definition
+ shared by the resume readers (``get_resume_conversations``,
+ ``get_ancestor_display_prefix``) and the resume guard
+ (``assert_resume_safe`` / ``get_resume_message_count``) — the guard must
+ count exactly the rows a resume would load, never a superset.
+ """
+ if self._is_explicit_branch_session(session_id):
+ return [session_id]
+ return self._session_lineage_root_to_tip(session_id)
+
+ def get_resume_message_count(
+ self, session_id: str, *, tip_only: bool = False
+ ) -> int:
+ """Count active rows that a resume would materialize.
+
+ ``tip_only=True`` counts only the tip segment — the set a model-history
+ restore loads (``get_messages_as_conversation`` without ancestors, or
+ the deferred Desktop resume that pages the display transcript over
+ REST and never materializes the ancestor prefix in memory).
+ """
+ session_ids = [session_id] if tip_only else self._resume_lineage_ids(session_id)
placeholders = ",".join("?" for _ in session_ids)
with self._read_ctx() as conn:
row = conn.execute(
@@ -13467,12 +14263,24 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self,
session_id: str,
max_messages: Optional[int] = None,
+ *,
+ tip_only: bool = False,
) -> int:
"""Return resume row count or reject a transcript too large to load.
``max_messages=None`` resolves the limit from config
(``sessions.max_resume_messages``); 0 disables the guard and returns
the (bounded) count without raising.
+
+ ``tip_only=True`` bounds only the tip segment, for callers that never
+ materialize the ancestor lineage in memory (tip-only model restore,
+ deferred Desktop resume whose display history is REST-paginated). A
+ heavily-compressed conversation — 85 compaction segments and ~29k
+ lineage rows behind a ~700-row tip — is exactly the shape compression
+ is supposed to produce; counting its whole lineage against a limit
+ sized for in-memory materialization rejected the healthiest sessions
+ (Desktop Bot Chat stuck on "Waking up…" with code 4130) while the
+ process would only ever have held the tip.
"""
if max_messages is None:
max_messages = resolved_max_resume_messages()
@@ -13484,7 +14292,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# return value, and an unbounded lineage COUNT here would do the
# exact pathological work the disable exists to avoid.
return 0
- session_ids = self._session_lineage_root_to_tip(session_id)
+ session_ids = [session_id] if tip_only else self._resume_lineage_ids(session_id)
placeholders = ",".join("?" for _ in session_ids)
with self._read_ctx() as conn:
row = conn.execute(
@@ -13496,7 +14304,11 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
).fetchone()
message_count = int(row[0] if row else 0)
if message_count > max_messages:
- raise SessionResumeTooLargeError(message_count, max_messages)
+ raise SessionResumeTooLargeError(
+ message_count,
+ max_messages,
+ scope="in its tip segment" if tip_only else "across its lineage",
+ )
return message_count
def assert_export_safe(
@@ -13555,10 +14367,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
returns ONLY the genuine ancestor messages, identified by
``session_id != tip_session_id``. (#65919)
"""
- if self._is_explicit_branch_session(session_id):
- return []
-
- session_ids = self._session_lineage_root_to_tip(session_id)
+ session_ids = self._resume_lineage_ids(session_id)
if len(session_ids) <= 1:
return []
with self._read_ctx() as conn:
@@ -14768,6 +15577,14 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
) < ?"""
)
params.append(last_active_before)
+ # An automatic orphan sweep closes a stale open row so the user can
+ # still recover it. Age those rows from the sweep, not from their old
+ # activity, or the next prune pass can delete them immediately.
+ clauses.append(
+ "(COALESCE(s.end_reason, '') != 'startup_orphan_reap' "
+ "OR s.ended_at < ?)"
+ )
+ params.append(last_active_before)
if last_active_after is not None:
clauses.append(
"""COALESCE(
@@ -15029,6 +15846,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
older_than_days: Optional[float] = 90,
source: str = None,
sessions_dir: Optional[Path] = None,
+ exclude_active_write_guards: bool = False,
**filters,
) -> int:
"""Delete sessions matching the filters. Returns count deleted.
@@ -15066,6 +15884,11 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
on-disk transcript files (``.json`` / ``.jsonl`` /
``request_dump_*``) for every pruned session, outside the DB
transaction.
+
+ ``exclude_active_write_guards`` is for destructive automatic
+ maintenance: rows protected by a live turn lease or compression lock
+ are skipped, while expired or provably dead holders are reclaimed and
+ fenced in the same write transaction.
"""
self._apply_prune_age_filter(older_than_days, filters)
where, where_params = self._prune_filter_where(source=source, **filters)
@@ -15077,6 +15900,26 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
)
session_ids = {row["id"] for row in cursor.fetchall()}
+ if exclude_active_write_guards:
+ protected = set()
+ for sid in session_ids:
+ try:
+ self._check_transcript_write_guards(
+ conn,
+ sid,
+ compression_lock_holder=None,
+ turn_lease_holder=None,
+ reject_active_turn_lease=True,
+ reject_active_compression_lock=True,
+ allow_closed_compression_parent=True,
+ )
+ except (
+ SessionCompressionInProgressError,
+ SessionTurnLeaseLostError,
+ ):
+ protected.add(sid)
+ session_ids.difference_update(protected)
+
if not session_ids:
return 0
@@ -15313,12 +16156,22 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
v1 — initial shape (no ON DELETE CASCADE on session_id FK)
v2 — session_id FK gets ON DELETE CASCADE so session pruning
automatically clears bindings.
+ v3 — ``profile_name`` dimension on both tables so multiplexed
+ gateways (shared ``state.db``) isolate topic mode/bindings
+ per Hermes profile (issue #76423).
"""
- def _do(conn):
- conn.executescript(
+ # (table, column list, DDL body). ``profile_name`` leads the primary
+ # key so multiplexed profiles sharing one state.db never collide on a
+ # private chat_id (which is the user id, identical across bots).
+ tables = (
+ (
+ "telegram_dm_topic_mode",
+ "profile_name, chat_id, user_id, enabled, activated_at, updated_at, "
+ "has_topics_enabled, allows_users_to_create_topics, "
+ "capability_checked_at, intro_message_id, pinned_message_id",
"""
- CREATE TABLE IF NOT EXISTS telegram_dm_topic_mode (
- chat_id TEXT PRIMARY KEY,
+ profile_name TEXT NOT NULL DEFAULT 'default',
+ chat_id TEXT NOT NULL,
user_id TEXT NOT NULL,
enabled INTEGER NOT NULL DEFAULT 1,
activated_at REAL NOT NULL,
@@ -15327,10 +16180,16 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
allows_users_to_create_topics INTEGER,
capability_checked_at REAL,
intro_message_id TEXT,
- pinned_message_id TEXT
- );
-
- CREATE TABLE IF NOT EXISTS telegram_dm_topic_bindings (
+ pinned_message_id TEXT,
+ PRIMARY KEY (profile_name, chat_id)
+ """,
+ ),
+ (
+ "telegram_dm_topic_bindings",
+ "profile_name, chat_id, thread_id, user_id, session_key, "
+ "session_id, managed_mode, linked_at, updated_at",
+ """
+ profile_name TEXT NOT NULL DEFAULT 'default',
chat_id TEXT NOT NULL,
thread_id TEXT NOT NULL,
user_id TEXT NOT NULL,
@@ -15339,65 +16198,50 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
managed_mode TEXT NOT NULL DEFAULT 'auto',
linked_at REAL NOT NULL,
updated_at REAL NOT NULL,
- PRIMARY KEY (chat_id, thread_id)
- );
+ PRIMARY KEY (profile_name, chat_id, thread_id)
+ """,
+ ),
+ )
+ def _do(conn):
+ for table, columns, ddl in tables:
+ # Fresh installs get the v3 shape immediately.
+ conn.execute(f"CREATE TABLE IF NOT EXISTS {table} ({ddl})")
+ have = {row[1] for row in conn.execute(f"PRAGMA table_info('{table}')")}
+ if "profile_name" in have:
+ continue
+ # Pre-profile shape (v1 or v2) → v3. SQLite can't ALTER a
+ # primary key (or a foreign key), so rebuild; this also
+ # supplies the v2 ON DELETE CASCADE for v1 bindings tables.
+ # Legacy rows land in the "default" namespace only — never
+ # replicated across profiles.
+ legacy_columns = columns.replace("profile_name, ", "", 1)
+ conn.executescript(
+ f"""
+ CREATE TABLE {table}_new ({ddl});
+ INSERT INTO {table}_new ({columns})
+ SELECT 'default', {legacy_columns} FROM {table};
+ DROP TABLE {table};
+ ALTER TABLE {table}_new RENAME TO {table};
+ """
+ )
+
+ # Indexes after any rebuild so they always target the v3 shape
+ # (a legacy table lacking profile_name can't take the user index).
+ conn.executescript(
+ """
CREATE UNIQUE INDEX IF NOT EXISTS idx_telegram_dm_topic_bindings_session
ON telegram_dm_topic_bindings(session_id);
CREATE INDEX IF NOT EXISTS idx_telegram_dm_topic_bindings_user
- ON telegram_dm_topic_bindings(user_id, chat_id);
+ ON telegram_dm_topic_bindings(profile_name, user_id, chat_id);
"""
)
- # v1 → v2: rebuild telegram_dm_topic_bindings if its session_id FK
- # lacks ON DELETE CASCADE. SQLite can't ALTER a foreign key, so we
- # rebuild the table. Only runs once per DB (version gate).
- current = conn.execute(
- "SELECT value FROM state_meta WHERE key = ?",
- ("telegram_dm_topic_schema_version",),
- ).fetchone()
- current_version = int(current[0]) if current and str(current[0]).isdigit() else 0
- if current_version < 2:
- fk_rows = conn.execute(
- "PRAGMA foreign_key_list('telegram_dm_topic_bindings')"
- ).fetchall()
- needs_rebuild = any(
- row[2] == "sessions" and (row[6] or "") != "CASCADE"
- for row in fk_rows
- )
- if needs_rebuild:
- conn.executescript(
- """
- CREATE TABLE telegram_dm_topic_bindings_new (
- chat_id TEXT NOT NULL,
- thread_id TEXT NOT NULL,
- user_id TEXT NOT NULL,
- session_key TEXT NOT NULL,
- session_id TEXT NOT NULL REFERENCES sessions(id) ON DELETE CASCADE,
- managed_mode TEXT NOT NULL DEFAULT 'auto',
- linked_at REAL NOT NULL,
- updated_at REAL NOT NULL,
- PRIMARY KEY (chat_id, thread_id)
- );
- INSERT INTO telegram_dm_topic_bindings_new
- SELECT chat_id, thread_id, user_id, session_key,
- session_id, managed_mode, linked_at, updated_at
- FROM telegram_dm_topic_bindings;
- DROP TABLE telegram_dm_topic_bindings;
- ALTER TABLE telegram_dm_topic_bindings_new
- RENAME TO telegram_dm_topic_bindings;
- CREATE UNIQUE INDEX idx_telegram_dm_topic_bindings_session
- ON telegram_dm_topic_bindings(session_id);
- CREATE INDEX idx_telegram_dm_topic_bindings_user
- ON telegram_dm_topic_bindings(user_id, chat_id);
- """
- )
-
conn.execute(
"INSERT INTO state_meta (key, value) VALUES (?, ?) "
"ON CONFLICT(key) DO UPDATE SET value = excluded.value",
- ("telegram_dm_topic_schema_version", "2"),
+ ("telegram_dm_topic_schema_version", "3"),
)
self._execute_write(_do)
@@ -15406,6 +16250,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
*,
chat_id: str,
user_id: str,
+ profile_name: str = "default",
has_topics_enabled: Optional[bool] = None,
allows_users_to_create_topics: Optional[bool] = None,
) -> None:
@@ -15413,9 +16258,15 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
This method intentionally owns the explicit topic migration. Ordinary
SessionDB startup must not create these side tables.
+
+ ``profile_name`` namespaces rows under a shared multiplex ``state.db``
+ (issue #76423). Callers handling a multiplexed event must pass the
+ routed profile from ``source.profile``, not the process-global active
+ profile.
"""
self.apply_telegram_topic_migration()
now = time.time()
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
def _to_int(value: Optional[bool]) -> Optional[int]:
if value is None:
@@ -15426,11 +16277,11 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
conn.execute(
"""
INSERT INTO telegram_dm_topic_mode (
- chat_id, user_id, enabled, activated_at, updated_at,
+ profile_name, chat_id, user_id, enabled, activated_at, updated_at,
has_topics_enabled, allows_users_to_create_topics,
capability_checked_at
- ) VALUES (?, ?, 1, ?, ?, ?, ?, ?)
- ON CONFLICT(chat_id) DO UPDATE SET
+ ) VALUES (?, ?, ?, 1, ?, ?, ?, ?, ?)
+ ON CONFLICT(profile_name, chat_id) DO UPDATE SET
user_id = excluded.user_id,
enabled = 1,
updated_at = excluded.updated_at,
@@ -15439,6 +16290,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
capability_checked_at = excluded.capability_checked_at
""",
(
+ profile_name,
str(chat_id),
str(user_id),
now,
@@ -15454,6 +16306,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self,
*,
chat_id: str,
+ profile_name: str = "default",
clear_bindings: bool = True,
) -> None:
"""Disable Telegram DM topic mode for one private chat.
@@ -15466,33 +16319,43 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
Never creates the topic-mode tables from scratch; if they don't
exist there is nothing to disable and the call is a no-op.
"""
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
+
def _do(conn):
try:
conn.execute(
"UPDATE telegram_dm_topic_mode SET enabled = 0, updated_at = ? "
- "WHERE chat_id = ?",
- (time.time(), str(chat_id)),
+ "WHERE profile_name = ? AND chat_id = ?",
+ (time.time(), profile_name, str(chat_id)),
)
if clear_bindings:
conn.execute(
- "DELETE FROM telegram_dm_topic_bindings WHERE chat_id = ?",
- (str(chat_id),),
+ "DELETE FROM telegram_dm_topic_bindings "
+ "WHERE profile_name = ? AND chat_id = ?",
+ (profile_name, str(chat_id)),
)
except sqlite3.OperationalError:
# Tables don't exist yet — nothing to disable.
return
self._execute_write(_do)
- def is_telegram_topic_mode_enabled(self, *, chat_id: str, user_id: str) -> bool:
+ def is_telegram_topic_mode_enabled(
+ self,
+ *,
+ chat_id: str,
+ user_id: str,
+ profile_name: str = "default",
+ ) -> bool:
"""Return whether Telegram DM topic mode is enabled for this chat/user."""
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
with self._read_ctx() as conn:
try:
row = conn.execute(
"""
SELECT enabled FROM telegram_dm_topic_mode
- WHERE chat_id = ? AND user_id = ?
+ WHERE profile_name = ? AND chat_id = ? AND user_id = ?
""",
- (str(chat_id), str(user_id)),
+ (profile_name, str(chat_id), str(user_id)),
).fetchone()
except sqlite3.OperationalError:
return False
@@ -15506,16 +16369,18 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
*,
chat_id: str,
thread_id: str,
+ profile_name: str = "default",
) -> Optional[Dict[str, Any]]:
"""Return the session binding for a Telegram DM topic, if present."""
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
with self._read_ctx() as conn:
try:
row = conn.execute(
"""
SELECT * FROM telegram_dm_topic_bindings
- WHERE chat_id = ? AND thread_id = ?
+ WHERE profile_name = ? AND chat_id = ? AND thread_id = ?
""",
- (str(chat_id), str(thread_id)),
+ (profile_name, str(chat_id), str(thread_id)),
).fetchone()
except sqlite3.OperationalError:
return None
@@ -15525,18 +16390,21 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self,
*,
chat_id: str,
+ profile_name: str = "default",
) -> List[Dict[str, Any]]:
"""All Telegram DM topic bindings for one chat, newest first.
Read-only; returns [] if the bindings table doesn't exist yet
(does not trigger the topic-mode migration).
"""
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
with self._read_ctx() as conn:
try:
rows = conn.execute(
"SELECT * FROM telegram_dm_topic_bindings "
- "WHERE chat_id = ? ORDER BY updated_at DESC",
- (str(chat_id),),
+ "WHERE profile_name = ? AND chat_id = ? "
+ "ORDER BY updated_at DESC",
+ (profile_name, str(chat_id)),
).fetchall()
except sqlite3.OperationalError:
return []
@@ -15571,6 +16439,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
*,
chat_id: str,
thread_id: str,
+ profile_name: str = "default",
) -> int:
"""Remove the binding row for a single (chat, thread) pair.
@@ -15601,6 +16470,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
"""
chat_id = str(chat_id)
thread_id = str(thread_id)
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
deleted = {"count": 0}
def _do(conn):
@@ -15608,9 +16478,9 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
cursor = conn.execute(
"""
DELETE FROM telegram_dm_topic_bindings
- WHERE chat_id = ? AND thread_id = ?
+ WHERE profile_name = ? AND chat_id = ? AND thread_id = ?
""",
- (chat_id, thread_id),
+ (profile_name, chat_id, thread_id),
)
deleted["count"] = cursor.rowcount or 0
except sqlite3.OperationalError:
@@ -15626,15 +16496,16 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
remaining = conn.execute(
"""
SELECT 1 FROM telegram_dm_topic_bindings
- WHERE chat_id = ? LIMIT 1
+ WHERE profile_name = ? AND chat_id = ? LIMIT 1
""",
- (chat_id,),
+ (profile_name, chat_id),
).fetchone()
if remaining is None:
conn.execute(
"UPDATE telegram_dm_topic_mode "
- "SET enabled = 0, updated_at = ? WHERE chat_id = ?",
- (time.time(), chat_id),
+ "SET enabled = 0, updated_at = ? "
+ "WHERE profile_name = ? AND chat_id = ?",
+ (time.time(), profile_name, chat_id),
)
except sqlite3.OperationalError:
# telegram_dm_topic_mode absent — binding prune still stands.
@@ -15652,6 +16523,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
session_key: str,
session_id: str,
managed_mode: str = "auto",
+ profile_name: str = "default",
) -> None:
"""Bind one Telegram DM topic thread to one Hermes session.
@@ -15666,28 +16538,38 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
user_id = str(user_id)
session_key = str(session_key)
session_id = str(session_id)
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
def _do(conn):
existing_session = conn.execute(
"""
- SELECT chat_id, thread_id FROM telegram_dm_topic_bindings
+ SELECT profile_name, chat_id, thread_id
+ FROM telegram_dm_topic_bindings
WHERE session_id = ?
""",
(session_id,),
).fetchone()
if existing_session is not None:
- linked_chat = existing_session["chat_id"] if isinstance(existing_session, sqlite3.Row) else existing_session[0]
- linked_thread = existing_session["thread_id"] if isinstance(existing_session, sqlite3.Row) else existing_session[1]
- if str(linked_chat) != chat_id or str(linked_thread) != thread_id:
+ if isinstance(existing_session, sqlite3.Row):
+ linked_profile = existing_session["profile_name"]
+ linked_chat = existing_session["chat_id"]
+ linked_thread = existing_session["thread_id"]
+ else:
+ linked_profile, linked_chat, linked_thread = existing_session
+ if (
+ str(linked_profile) != profile_name
+ or str(linked_chat) != chat_id
+ or str(linked_thread) != thread_id
+ ):
raise ValueError("session is already linked to another Telegram topic")
conn.execute(
"""
INSERT INTO telegram_dm_topic_bindings (
- chat_id, thread_id, user_id, session_key, session_id,
+ profile_name, chat_id, thread_id, user_id, session_key, session_id,
managed_mode, linked_at, updated_at
- ) VALUES (?, ?, ?, ?, ?, ?, ?, ?)
- ON CONFLICT(chat_id, thread_id) DO UPDATE SET
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)
+ ON CONFLICT(profile_name, chat_id, thread_id) DO UPDATE SET
user_id = excluded.user_id,
session_key = excluded.session_key,
session_id = excluded.session_id,
@@ -15695,6 +16577,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
updated_at = excluded.updated_at
""",
(
+ profile_name,
chat_id,
thread_id,
user_id,
@@ -15734,6 +16617,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
*,
chat_id: str,
user_id: str,
+ profile_name: str = "default",
limit: int = 10,
) -> List[Dict[str, Any]]:
"""List previous Telegram sessions for this user that are not bound to a topic.
@@ -15742,7 +16626,13 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
topic-mode tables are absent, fall back to a simpler query that
just returns this user's Telegram sessions — there can't be any
bindings yet.
+
+ Scoped by ``profile_name`` so multiplexed profiles do not surface
+ each other's unlinked sessions (issue #76423).
"""
+ profile_name = _normalize_telegram_topic_profile_name(profile_name)
+ # sessions.profile_name is NULL/empty for legacy rows → treat as default.
+ profile_clause = "AND COALESCE(NULLIF(TRIM(s.profile_name), ''), 'default') = ?"
with self._read_ctx() as conn:
try:
rows = conn.execute(
@@ -15764,6 +16654,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
ON sp.hash = s.system_prompt_hash
WHERE s.source = 'telegram'
AND s.user_id = ?
+ {profile_clause}
AND NOT EXISTS (
SELECT 1 FROM telegram_dm_topic_bindings b
WHERE b.session_id = s.id
@@ -15771,7 +16662,7 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
ORDER BY last_active DESC, s.started_at DESC
LIMIT ?
""",
- (str(user_id), int(limit)),
+ (str(user_id), profile_name, int(limit)),
).fetchall()
except sqlite3.OperationalError:
# telegram_dm_topic_bindings doesn't exist yet — no bindings
@@ -15844,6 +16735,31 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
logger.debug("Could not read logical DB size: %s", exc)
return None
+ def _freelist_ratio(self) -> Optional[float]:
+ """Fraction of database pages that are on the freelist (reclaimable).
+
+ ``PRAGMA freelist_count / PRAGMA page_count`` read over the existing
+ connection (never a byte-level probe of the live file — see
+ ``sqlite_safe_read``). This is what VACUUM would actually give back;
+ it is the gate :meth:`maybe_auto_prune_and_vacuum` uses to decide
+ whether a full rewrite pays off (#54189).
+
+ Returns None if the pragmas cannot be read (callers treat that as
+ "unknown" and fall back to the time throttle alone).
+ """
+ try:
+ with self._read_ctx() as conn:
+ if self._conn is None:
+ return None
+ page_count = int(conn.execute("PRAGMA page_count").fetchone()[0])
+ freelist = int(conn.execute("PRAGMA freelist_count").fetchone()[0])
+ if page_count <= 0:
+ return 0.0
+ return freelist / page_count
+ except Exception as exc:
+ logger.debug("Could not read freelist ratio: %s", exc)
+ return None
+
def vacuum(self) -> int:
"""Run VACUUM to reclaim disk space after large deletes.
@@ -15895,6 +16811,10 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
self._conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
except Exception as exc:
logger.debug("WAL checkpoint (TRUNCATE) after VACUUM failed: %s", exc)
+ # TRUNCATE may replace the WAL inode; adopt the post-VACUUM
+ # sidecars so the write-path generation guard does not halt a
+ # healthy exclusive maintenance connection.
+ self._record_db_file_identity()
return optimized
def maybe_auto_prune_and_vacuum(
@@ -15904,30 +16824,59 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
vacuum: bool = True,
sessions_dir: Optional[Path] = None,
min_vacuum_interval_days: int = 30,
+ min_vacuum_freelist_ratio: float = AUTO_VACUUM_MIN_FREELIST_RATIO,
) -> Dict[str, Any]:
"""Idempotent auto-maintenance: prune inactive sessions + optional VACUUM.
Records the last run timestamp in state_meta so subsequent calls
within ``min_interval_hours`` no-op. VACUUM has its own, typically
longer, throttle controlled by ``min_vacuum_interval_days`` so routine
- pruning does not repeatedly rewrite the database. Designed to be
- called once at startup from long-lived entrypoints (CLI, gateway, cron
- scheduler).
+ pruning does not repeatedly rewrite the database, and is additionally
+ gated on the reclaimable fraction of the file: it only runs when
+ ``PRAGMA freelist_count / PRAGMA page_count`` exceeds
+ ``min_vacuum_freelist_ratio`` (default
+ :data:`AUTO_VACUUM_MIN_FREELIST_RATIO`, 25%), so pruning a few small
+ sessions on a dense multi-GB database never triggers a full rewrite
+ (#54189). Designed to be called once at startup from long-lived
+ entrypoints (CLI, gateway, cron scheduler).
When *sessions_dir* is provided, on-disk transcript files
(``.json`` / ``.jsonl`` / ``request_dump_*``) for pruned sessions
are removed as part of the same sweep (issue #3015).
+ Stale-open reconciliation (#54189): several state-owned producers
+ (cron, kanban workers, subagents, one-shot CLI runs) never set
+ ``ended_at`` when their process dies, and ``prune_sessions`` only
+ deletes ended rows — so retention was a no-op exactly where growth
+ concentrates. After pruning, this pass closes open rows from
+ :attr:`_AUTO_PRUNE_STALE_OPEN_SOURCES` whose activity is older than
+ ``retention_days`` (``end_reason='startup_orphan_reap'``). Closed rows
+ stay resumable and are aged from their close, so they get one more
+ full retention window before a later pass deletes them. Messaging
+ and UI sources are never touched here.
+
Never raises. On any failure, logs a warning and returns a dict
with ``"error"`` set.
Returns a dict with keys:
- ``"skipped"`` (bool) — true if within min_interval_hours of last run
- ``"pruned"`` (int) — number of sessions deleted
+ - ``"closed"`` (int) — stale open state-owned sessions marked ended
- ``"vacuumed"`` (bool) — true if VACUUM ran
+ - ``"freelist_ratio"`` (float|None) — reclaimable fraction measured
+ when a VACUUM was considered (absent when it was not)
- ``"error"`` (str, optional) — present only on failure
"""
- result: Dict[str, Any] = {"skipped": False, "pruned": 0, "vacuumed": False}
+ result: Dict[str, Any] = {
+ "skipped": False,
+ "pruned": 0,
+ "closed": 0,
+ "vacuumed": False,
+ }
+ maintenance_lock = _try_acquire_auto_maintenance_lock(self.db_path)
+ if maintenance_lock is None:
+ result["skipped"] = True
+ return result
try:
# Skip if another process/call did maintenance recently.
last_raw = self.get_meta("last_auto_prune")
@@ -15941,19 +16890,40 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
except (TypeError, ValueError):
pass # corrupt meta; treat as no prior run
+ # Delete only sessions that were already explicitly closed. A
+ # startup orphan discovered by this pass is closed *after* pruning,
+ # preserving a full retention window in which it can be resumed.
pruned = self.prune_sessions(
older_than_days=retention_days,
sessions_dir=sessions_dir,
+ exclude_active_write_guards=True,
)
result["pruned"] = pruned
- # Only VACUUM if we actually freed rows, and no more often than
- # once every min_vacuum_interval_days -- a large prune (e.g. the
- # first one to cross retention_days on a DB with tens of
- # thousands of rows) can free enough pages that pruned > 0 fires
- # on every subsequent startup even though a VACUUM already ran
- # recently. VACUUM on this DB's size (FTS5 shadow tables) is not
- # cheap -- it holds an exclusive lock for the full rewrite.
+ # Reap stale state-owned rows only. Runtime-owned messaging sources
+ # are intentionally outside this automatic destructive scope.
+ closed = self.sweep_orphaned_sessions(
+ max_idle_seconds=float(retention_days) * 86400.0,
+ sources=self._AUTO_PRUNE_STALE_OPEN_SOURCES,
+ exclude_pinned=True,
+ # These sources are owned by state.db lifecycles, not by the
+ # dashboard/TUI gateway heartbeats used by startup recovery.
+ respect_gateway_heartbeats=False,
+ )
+ result["closed"] = len(closed)
+ # Only VACUUM if we actually freed rows, no more often than once
+ # every min_vacuum_interval_days, AND only when the rewrite pays
+ # off: the reclaimable fraction of the file (freelist_count /
+ # page_count) must exceed AUTO_VACUUM_MIN_FREELIST_RATIO (#54189).
+ # A large prune (e.g. the first one to cross retention_days on a
+ # DB with tens of thousands of rows) can free enough pages that
+ # pruned > 0 fires on every subsequent startup even though a
+ # VACUUM already ran recently; and pruning one tiny session on a
+ # dense multi-GB DB would otherwise rewrite the whole file to
+ # reclaim a few MB. VACUUM on this DB's size (FTS5 shadow tables)
+ # is not cheap -- it holds an exclusive lock for the full rewrite.
+ # The time throttle says "not too often"; the ratio gate says
+ # "only when it pays off". Both must pass.
last_vacuum_raw = self.get_meta("last_vacuum")
vacuum_due = True
if last_vacuum_raw:
@@ -15962,20 +16932,32 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
except (TypeError, ValueError):
vacuum_due = True
if vacuum and pruned > 0 and vacuum_due:
- try:
- self.vacuum()
- result["vacuumed"] = True
- self.set_meta("last_vacuum", str(now))
- except Exception as exc:
- logger.warning("state.db VACUUM failed: %s", exc)
+ ratio = self._freelist_ratio()
+ result["freelist_ratio"] = ratio
+ if ratio is None or ratio > min_vacuum_freelist_ratio:
+ try:
+ self.vacuum()
+ result["vacuumed"] = True
+ self.set_meta("last_vacuum", str(now))
+ except Exception as exc:
+ logger.warning("state.db VACUUM failed: %s", exc)
+ else:
+ logger.debug(
+ "state.db auto-maintenance: skipping VACUUM, only "
+ "%.1f%% of pages reclaimable (threshold %.0f%%)",
+ ratio * 100.0,
+ min_vacuum_freelist_ratio * 100.0,
+ )
# Record the attempt even if pruned == 0, so we don't retry
# every startup within the min_interval_hours window.
self.set_meta("last_auto_prune", str(now))
- if pruned > 0:
+ if closed or pruned > 0:
logger.info(
- "state.db auto-maintenance: pruned %d session(s) inactive for %d days%s",
+ "state.db auto-maintenance: closed %d stale open session(s), "
+ "pruned %d session(s) inactive for %d days%s",
+ len(closed),
pruned,
retention_days,
" + VACUUM" if result["vacuumed"] else "",
@@ -15984,6 +16966,8 @@ class SessionDB(SessionSearchMixin, SessionSchemaMixin, SessionPortabilityMixin)
# Maintenance must never block startup. Log and return error marker.
logger.warning("state.db auto-maintenance failed: %s", exc)
result["error"] = str(exc)
+ finally:
+ _release_auto_maintenance_lock(maintenance_lock)
return result
diff --git a/hermes_state_common.py b/hermes_state_common.py
index 2d2793bcf6..af12d322d3 100644
--- a/hermes_state_common.py
+++ b/hermes_state_common.py
@@ -7,6 +7,8 @@ hermes_state re-imports every name here for backward compatibility.
"""
import contextlib
+import errno
+import json
import logging
import os
import sys
@@ -352,7 +354,7 @@ def _sql_session_last_active_by_id(session_id_expr: str) -> str:
)
-SCHEMA_VERSION = 26
+SCHEMA_VERSION = 28
# FTS storage-layout version, tracked INDEPENDENTLY of SCHEMA_VERSION in the
@@ -443,12 +445,14 @@ CREATE TABLE IF NOT EXISTS sessions (
compression_failure_error TEXT,
compression_fallback_streak INTEGER NOT NULL DEFAULT 0,
compression_ineffective_count INTEGER NOT NULL DEFAULT 0,
+ compression_recovery_deadline REAL,
profile_name TEXT,
rewind_count INTEGER NOT NULL DEFAULT 0,
archived INTEGER NOT NULL DEFAULT 0,
pinned INTEGER NOT NULL DEFAULT 0,
hidden INTEGER NOT NULL DEFAULT 0,
last_read_at REAL,
+ tool_names TEXT,
FOREIGN KEY (parent_session_id) REFERENCES sessions(id),
FOREIGN KEY (system_prompt_hash) REFERENCES system_prompts(hash)
);
@@ -897,9 +901,15 @@ END;
# Semantics mirror `hermes_state._cross_process_repair_lock` (the schema-
# surgery authority): portable (msvcrt on Windows, flock elsewhere), bounded
# wait, and FAIL CLOSED — a caller that cannot acquire the lock must NOT
-# rebuild. The kernel drops both lock types when the holder dies, so a crashed
-# rebuilder cannot wedge future rebuilds. It lives here (not hermes_state)
-# because the search/schema mixins cannot import hermes_state (cycle).
+# rebuild. The kernel drops both lock types when the holder dies — UNLESS a
+# forked child inherited the lock fd (flock rides the open file description,
+# which fork() duplicates), in which case the orphaned descriptor holds the
+# lock forever (issue #100108). `_acquire_db_flock` therefore records the
+# holder's pid + start time under the lock and, when the recorded holder is
+# provably dead, breaks the orphaned lock by unlinking and retaking it on a
+# fresh inode; indeterminate liveness still defers. It lives here (not
+# hermes_state) because the search/schema mixins cannot import hermes_state
+# (cycle).
#
# The lock file is `.fts_rebuild.lock`, distinct from `.repair.lock`:
# schema surgery runs on an EXCLUSIVE offline connection and can legitimately
@@ -912,64 +922,365 @@ _FTS_REBUILD_LOCK_TIMEOUT_SECONDS = 120.0
_FTS_REBUILD_LOCK_POLL_SECONDS = 0.1
_IS_WINDOWS = sys.platform == "win32"
+# Post-break re-acquire budget: once a provably-orphaned lock has been broken
+# the fresh inode is uncontended (or contended only by live processes), so a
+# short bounded wait suffices — never re-enter the full timeout.
+_LOCK_BREAK_REACQUIRE_SECONDS = 5.0
+
+# errno set for "another process holds this advisory lock". flock() reports
+# contention as EWOULDBLOCK/EAGAIN; msvcrt.locking() as EACCES (and EDEADLK
+# when its internal retry gives up). Anything else — ESTALE on a dropped NFS
+# handle, ENOTSUP/ENOLCK on a filesystem without advisory locks, EIO — is a
+# persistent environment failure that no amount of polling turns into an
+# acquire. Treating every OSError as contention made such a failure look
+# like a live holder and burned the full 120s admission timeout on every
+# attempt (#100108, PR #100130).
+_LOCK_CONTENTION_ERRNOS = {errno.EAGAIN, errno.EACCES, errno.EWOULDBLOCK}
+if hasattr(errno, "EDEADLK"):
+ _LOCK_CONTENTION_ERRNOS.add(errno.EDEADLK)
+
+
+def is_advisory_lock_contention(exc: BaseException) -> bool:
+ """True when *exc* means another process holds the advisory lock.
+
+ False for every other ``OSError`` (ESTALE, ENOTSUP, ENOLCK, EIO, ...):
+ callers must fail closed IMMEDIATELY rather than poll to the deadline,
+ because retrying cannot succeed and the wait only stalls the caller.
+ """
+ if isinstance(exc, BlockingIOError):
+ return True
+ if not isinstance(exc, OSError):
+ return False
+ return exc.errno in _LOCK_CONTENTION_ERRNOS
+
+
+def _proc_start_ticks(pid: int):
+ """Kernel start time of *pid* in clock ticks, or None when unknowable.
+
+ Field 22 of ``/proc//stat`` (``starttime``) uniquely identifies a
+ process together with its PID: a recycled PID gets a different start
+ time. Returns None off Linux or on any read/parse failure — callers must
+ treat None as "unknowable" and FAIL CLOSED.
+ """
+ try:
+ with open(f"/proc/{pid}/stat", "rb") as fh:
+ stat = fh.read()
+ # comm (field 2) may contain spaces/parens; split after the LAST ')'.
+ return int(stat.rsplit(b")", 1)[1].split()[19])
+ except (OSError, ValueError, IndexError):
+ return None
+
+
+def _read_lock_holder_record(handle):
+ """Best-effort parse of the holder metadata JSON in a lock file."""
+ try:
+ handle.seek(0)
+ raw = handle.read(4096)
+ except (OSError, ValueError):
+ return None
+ if not raw:
+ return None
+ try:
+ record = json.loads(raw.decode("utf-8", "replace"))
+ except (ValueError, UnicodeDecodeError):
+ return None
+ return record if isinstance(record, dict) else None
+
+
+def _write_lock_holder_record(handle) -> None:
+ """Record this process as the lock holder (advisory, best effort).
+
+ Written under the flock so contenders that time out can tell an
+ orphaned-fd holder (recorded process dead, flock inherited by a forked
+ child — issue #100108) from a live wedged holder.
+ """
+ try:
+ record = {
+ "pid": os.getpid(),
+ "start_ticks": _proc_start_ticks(os.getpid()),
+ "acquired_at": time.time(),
+ }
+ handle.seek(0)
+ handle.truncate()
+ handle.write(json.dumps(record, sort_keys=True).encode("utf-8"))
+ handle.flush()
+ except (OSError, ValueError):
+ pass
+
+
+def _clear_lock_holder_record(handle) -> None:
+ """Erase holder metadata before a normal release.
+
+ Guarantees that a surviving record always describes an ABNORMAL exit
+ (holder died without releasing), which is the only condition under which
+ a contender may break the lock.
+ """
+ try:
+ handle.seek(0)
+ handle.truncate()
+ handle.flush()
+ except (OSError, ValueError):
+ pass
+
+
+def _lock_holder_provably_dead(record) -> bool:
+ """True ONLY when the recorded holder is provably dead or PID-recycled.
+
+ Any indeterminate state (no record, malformed record, PID owned by
+ another user, /proc unavailable, start-time unknowable) returns False —
+ the caller must FAIL CLOSED and defer, never break a possibly-live
+ holder's lock.
+ """
+ if not isinstance(record, dict):
+ return False
+ try:
+ pid = int(record["pid"])
+ except (KeyError, TypeError, ValueError):
+ return False
+ if pid <= 0:
+ return False
+ try:
+ os.kill(pid, 0)
+ except ProcessLookupError:
+ return True
+ except OSError:
+ # PermissionError et al.: the PID exists (or is unknowable) — closed.
+ return False
+ recorded_ticks = record.get("start_ticks")
+ if recorded_ticks is None:
+ return False
+ current_ticks = _proc_start_ticks(pid)
+ if current_ticks is None:
+ return False
+ # Same PID, different kernel start time: the recorded holder is dead and
+ # its PID was recycled by an unrelated process.
+ return current_ticks != recorded_ticks
+
+
+def _acquire_db_flock(lock_path, handle, timeout_seconds, poll_seconds, description):
+ """Bounded POSIX flock acquire with orphaned-holder staleness break.
+
+ Returns ``(acquired, handle)``; *handle* may have been re-opened (the
+ caller owns closing whichever handle comes back). *acquired* is True on
+ success, False when a holder kept the lock past the deadline, and None
+ when a non-contention ``OSError`` (ESTALE/ENOTSUP/EIO) made acquisition
+ impossible — already logged here; callers treat None as "not acquired"
+ without emitting the held-by-another-process warning.
+
+ Why breaking exists at all (issue #100108): ``flock`` belongs to the open
+ file DESCRIPTION, which ``fork()`` duplicates into every child. A holder
+ that forks (multiprocessing worker, daemonized helper) and then dies
+ leaves the flock held by a child that will never release it — the
+ kernel's holder-death release never triggers, and every contender defers
+ forever. The recorded-holder liveness check distinguishes exactly that
+ case: the process that ACQUIRED is provably dead (so its critical section
+ died with it), yet the flock is still held. Only then is the lock file
+ unlinked and retaken on a fresh inode; the orphan's flock stays on the
+ old unlinked inode where it blocks nobody. Every successful acquire
+ verifies its inode still names *lock_path*, so a racer that locked a dead
+ inode retries instead of running concurrently with the breaker.
+ Indeterminate liveness always defers (fail closed).
+ """
+ import fcntl
+
+ deadline = time.monotonic() + timeout_seconds
+ broke_lock = False
+ while True:
+ try:
+ fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
+ except (BlockingIOError, OSError) as exc:
+ if not is_advisory_lock_contention(exc):
+ # ESTALE / ENOTSUP / EIO: not a holder, and polling cannot
+ # fix it. Defer NOW instead of pretending a live process
+ # held the lock for the whole timeout (#100108).
+ logger.warning(
+ "Could not acquire %s %s (%s) — deferring rather than "
+ "waiting out the %.0fs holder timeout on a "
+ "non-contention error.",
+ description,
+ lock_path,
+ exc,
+ timeout_seconds,
+ )
+ return None, handle
+ if time.monotonic() < deadline:
+ time.sleep(poll_seconds)
+ continue
+ if broke_lock:
+ return False, handle
+ record = _read_lock_holder_record(handle)
+ if not _lock_holder_provably_dead(record):
+ return False, handle
+ logger.warning(
+ "%s %s is held by an orphaned file descriptor (recorded "
+ "holder pid %s is dead — a forked child inherited the lock "
+ "fd); breaking the stale lock and retaking it on a fresh "
+ "file.",
+ description,
+ lock_path,
+ (record or {}).get("pid"),
+ )
+ try:
+ os.unlink(lock_path)
+ handle.close()
+ handle = open(lock_path, "a+b")
+ except OSError as exc:
+ logger.warning(
+ "Could not break stale %s %s (%s) — deferring.",
+ description,
+ lock_path,
+ exc,
+ )
+ return False, handle
+ broke_lock = True
+ deadline = time.monotonic() + _LOCK_BREAK_REACQUIRE_SECONDS
+ continue
+ # flock acquired — verify the path still names our inode: a breaker
+ # may have unlinked/replaced the file while we waited, and a lock on
+ # a dead inode excludes nobody.
+ try:
+ fd_stat = os.fstat(handle.fileno())
+ path_stat = os.stat(lock_path)
+ same_file = (
+ fd_stat.st_dev == path_stat.st_dev
+ and fd_stat.st_ino == path_stat.st_ino
+ )
+ except OSError:
+ same_file = False
+ if same_file:
+ _write_lock_holder_record(handle)
+ return True, handle
+ try:
+ handle.close()
+ handle = open(lock_path, "a+b")
+ except OSError:
+ return False, handle
+ if time.monotonic() >= deadline:
+ return False, handle
+
+
+def _describe_lock_holder(record) -> str:
+ """Human-readable holder identity for deferral warnings."""
+ if not isinstance(record, dict) or "pid" not in record:
+ return "unknown (no holder record; pre-fix writer or non-Hermes)"
+ pid = record.get("pid")
+ acquired_at = record.get("acquired_at")
+ age = ""
+ try:
+ if acquired_at is not None:
+ age = f", acquired {time.time() - float(acquired_at):.0f}s ago"
+ except (TypeError, ValueError):
+ pass
+ return f"pid {pid}{age}"
+
@contextlib.contextmanager
-def fts_rebuild_admission(db_path):
+def fts_rebuild_admission(db_path, *, timeout_seconds=None):
"""Serialize full structural FTS rebuilds on *db_path* across processes.
Yields True when this process holds the rebuild authority, False when the
- bounded acquire timed out. A caller that gets False must NOT perform a
- full rebuild — proceeding is exactly the concurrent-rebuild interleaving
- this lock exists to prevent (fail closed). The deferred/stale breadcrumb
- machinery already guarantees a skipped rebuild is retried later.
+ bounded acquire timed out or the lock file could not be opened at all. A
+ caller that gets False must NOT perform a full rebuild — proceeding is
+ exactly the concurrent-rebuild interleaving this lock exists to prevent
+ (fail closed). The deferred/stale breadcrumb machinery already guarantees
+ a skipped rebuild is retried later.
``db_path`` may be a str or Path; None (in-memory DB / tests without a
file path) yields True — a private in-memory DB has no cross-process
surface.
+
+ *timeout_seconds* defaults to ``_FTS_REBUILD_LOCK_TIMEOUT_SECONDS``.
+ Opportunistic in-process retries (``retry_deferred_fts_recovery``) pass
+ ``0`` so a live holder never stalls a long-lived writer for two minutes;
+ the orphaned-holder break still applies on the single attempt.
"""
if db_path is None:
yield True
return
+ timeout = (
+ _FTS_REBUILD_LOCK_TIMEOUT_SECONDS
+ if timeout_seconds is None
+ else max(float(timeout_seconds), 0.0)
+ )
lock_path = f"{db_path}.fts_rebuild.lock"
try:
handle = open(lock_path, "a+b")
except OSError as exc:
- # Read-only dir, exhausted fds, exotic filesystem: fall back to the
- # pre-lock behaviour rather than refusing a rebuild we could run.
+ # Fail closed, exactly as a timed-out acquire does. A lock file we
+ # cannot even open means the filesystem is out of space, inodes or
+ # descriptors — and a sibling process that opened ITS handle before
+ # the disk filled is still holding the authority and rebuilding.
+ # Yielding True here handed every process on a full disk a concurrent
+ # structural rebuild of the same live state.db with no cross-process
+ # authority at all: the disk-full trigger and the re-corruption on
+ # every multi-writer boot in #100368. Deferring costs nothing that
+ # was reachable anyway — the breadcrumb retries, and on a read-only
+ # directory the rebuild's own writes could not have committed either.
logger.warning(
- "Could not open FTS rebuild lock %s (%s) — proceeding with "
- "in-process serialisation only.", lock_path, exc,
+ "Could not open FTS rebuild lock %s (%s) — deferring this rebuild "
+ "rather than running it without cross-process authority.",
+ lock_path, exc,
)
- yield True
+ yield False
return
acquired = False
try:
- deadline = time.monotonic() + _FTS_REBUILD_LOCK_TIMEOUT_SECONDS
- while True:
- try:
- if _IS_WINDOWS:
+ if _IS_WINDOWS:
+ deadline = time.monotonic() + timeout
+ while True:
+ try:
import msvcrt
handle.seek(0)
msvcrt.locking(handle.fileno(), msvcrt.LK_NBLCK, 1)
- else:
- import fcntl
-
- fcntl.flock(handle.fileno(), fcntl.LOCK_EX | fcntl.LOCK_NB)
- acquired = True
- break
- except (BlockingIOError, OSError):
- if time.monotonic() >= deadline:
+ acquired = True
break
- time.sleep(_FTS_REBUILD_LOCK_POLL_SECONDS)
- if not acquired:
- logger.warning(
- "FTS rebuild lock %s held by another process for more than "
- "%.0fs — deferring this rebuild to avoid racing the holder "
- "(the stale-FTS breadcrumb keeps it retryable).",
- lock_path, _FTS_REBUILD_LOCK_TIMEOUT_SECONDS,
+ except (BlockingIOError, OSError) as exc:
+ if not is_advisory_lock_contention(exc):
+ logger.warning(
+ "Could not acquire FTS rebuild lock %s (%s) — "
+ "deferring on a non-contention error.",
+ lock_path, exc,
+ )
+ acquired = None
+ break
+ if time.monotonic() >= deadline:
+ break
+ time.sleep(_FTS_REBUILD_LOCK_POLL_SECONDS)
+ else:
+ acquired, handle = _acquire_db_flock(
+ lock_path,
+ handle,
+ timeout,
+ _FTS_REBUILD_LOCK_POLL_SECONDS,
+ "FTS rebuild lock",
)
+ if acquired is None:
+ # Non-contention failure: already logged with the real errno;
+ # a "held by another process" line here would be a lie.
+ acquired = False
+ elif not acquired:
+ record = None if _IS_WINDOWS else _read_lock_holder_record(handle)
+ if timeout <= 0:
+ # Non-blocking probe from an in-process retry: a busy lock
+ # is expected and will be tried again, so keep it quiet.
+ logger.info(
+ "FTS rebuild lock %s is busy — deferring this retry "
+ "(the stale-FTS breadcrumb keeps it retryable). "
+ "Recorded holder: %s.",
+ lock_path,
+ _describe_lock_holder(record),
+ )
+ else:
+ logger.warning(
+ "FTS rebuild lock %s held by another process for more than "
+ "%.0fs — deferring this rebuild to avoid racing the holder "
+ "(the stale-FTS breadcrumb keeps it retryable). "
+ "Recorded holder: %s.",
+ lock_path, timeout,
+ _describe_lock_holder(record),
+ )
yield acquired
finally:
try:
@@ -982,6 +1293,7 @@ def fts_rebuild_admission(db_path):
else:
import fcntl
+ _clear_lock_holder_record(handle)
fcntl.flock(handle.fileno(), fcntl.LOCK_UN)
except OSError: # pragma: no cover - best effort release
pass
diff --git a/hermes_state_registry.py b/hermes_state_registry.py
index 3c5b6ec24e..0f3bcedc20 100644
--- a/hermes_state_registry.py
+++ b/hermes_state_registry.py
@@ -42,7 +42,7 @@ from __future__ import annotations
import logging
import threading
from pathlib import Path
-from typing import TYPE_CHECKING, Dict, Optional, Tuple
+from typing import TYPE_CHECKING, Dict, List, Optional, Tuple
if TYPE_CHECKING: # pragma: no cover - import cycle guard, typed only
from hermes_state import SessionDB
@@ -91,6 +91,11 @@ _lock = threading.Lock()
_generations: Dict[Path, _Generation] = {}
# Object-keyed retired generations still draining holders.
_retired: Dict[int, _Generation] = {} # id(db) → generation
+# Paths whose next generation is currently being constructed. Construction
+# stays outside _lock because schema reconciliation can take seconds, but peers
+# for the SAME file must wait: otherwise every cold caller opens a writable
+# SQLite connection before the registry chooses one winner.
+_opening: Dict[Path, threading.Event] = {}
def _open_session_db(path: Path) -> "SessionDB":
@@ -132,42 +137,67 @@ def acquire(db_path: Optional[Path] = None) -> "SessionDB":
"""
from hermes_state import _default_db_path
- path = Path(db_path) if db_path is not None else Path(_default_db_path())
+ raw_path = Path(db_path) if db_path is not None else Path(_default_db_path())
+ try:
+ path = raw_path.resolve()
+ except OSError:
+ path = raw_path
- with _lock:
- generation = _generations.get(path)
- if generation is not None:
- current = _stat_db_file_identity(path)
- if (
- current is not None
- and generation.identity is not None
- and current != generation.identity
- ):
- # File replaced: retire the live generation (its
- # holders keep it until they release) and fall
- # through to opening a fresh one below.
- _retire_generation_locked(path, generation)
- else:
- generation.refcount += 1
- return generation.db
+ while True:
+ with _lock:
+ generation = _generations.get(path)
+ if generation is not None:
+ current = _stat_db_file_identity(path)
+ if (
+ current is not None
+ and generation.identity is not None
+ and current != generation.identity
+ ):
+ # File replaced: retire the live generation (its
+ # holders keep it until they release) and elect one
+ # caller to construct the replacement below.
+ _retire_generation_locked(path, generation)
+ else:
+ generation.refcount += 1
+ return generation.db
+
+ opening = _opening.get(path)
+ if opening is None:
+ opening = threading.Event()
+ _opening[path] = opening
+ break
+
+ # Another caller is constructing this path. Do not hold the global
+ # registry lock while waiting: unrelated databases continue opening.
+ # A failed opener signals too, so one waiter can retry as the successor.
+ opening.wait()
+
+ # Open a fresh generation OUTSIDE the lock. The per-path opening marker
+ # prevents redundant writer connections without serialising other files.
+ try:
+ db = _open_session_db(path)
+ db._shared_registry_owned = True
+ identity = _stat_db_file_identity(path)
+ except BaseException:
+ with _lock:
+ if _opening.get(path) is opening:
+ _opening.pop(path, None)
+ opening.set()
+ raise
- # Open a fresh generation OUTSIDE the lock: construction can
- # take seconds (write-lock patience) and must not block every
- # other state.db acquisition in the process.
- db = _open_session_db(path)
- db._shared_registry_owned = True
- identity = _stat_db_file_identity(path)
with _lock:
existing = _generations.get(path)
if existing is not None:
- # Someone else opened a generation while we were
- # constructing (or retired ours and installed a new one).
- # Ours loses — close it (outside the lock) and use theirs.
+ # Defensive: a generation may have been installed by explicit
+ # registry manipulation while this open was in flight.
existing.refcount += 1
winner = existing.db
else:
_generations[path] = _Generation(db, identity)
winner = db
+ if _opening.get(path) is opening:
+ _opening.pop(path, None)
+ opening.set()
if winner is not db:
_teardown(db)
return winner
@@ -259,6 +289,18 @@ def close_all() -> int:
return closed
+def live_shared_session_dbs() -> List["SessionDB"]:
+ """Snapshot of every live (non-retired) shared SessionDB in this process.
+
+ For periodic in-process maintenance (the gateway housekeeping tick's
+ deferred-FTS retry). Refcounts are NOT touched: the caller only invokes
+ a method on an instance that some holder already keeps alive; a
+ concurrent final release closes it and the callee sees ``_conn is None``.
+ """
+ with _lock:
+ return [g.db for g in _generations.values() if not g.retired]
+
+
def stats() -> Dict[str, int]:
"""Registry census for tests and diagnostics (no locks held long)."""
with _lock:
diff --git a/hermes_state_schema.py b/hermes_state_schema.py
index 9813b12785..01801a4870 100644
--- a/hermes_state_schema.py
+++ b/hermes_state_schema.py
@@ -43,6 +43,15 @@ logger = logging.getLogger("hermes_state")
_FTS_HOLDER_ESCALATE_ATTEMPTS = 3
_FTS_HOLDER_ESCALATE_SECONDS = 60.0
+# Minimum spacing between in-process retries of a deferred stale-FTS rebuild
+# (``retry_deferred_fts_recovery``). The startup open already paid the full
+# admission wait once; later retries are non-blocking probes on this cadence
+# so a live holder never stalls a long-lived writer.
+_FTS_STALE_RETRY_SECONDS = 60.0
+# Each failed retry doubles the spacing up to this cap, so a holder that never
+# goes away (a second long-lived writer) costs one deferral warning per hour,
+# not one per minute. A successful rebuild clears the stale state entirely.
+_FTS_STALE_RETRY_MAX_SECONDS = 3600.0
# Cache for schema_read_probe_statements() — parsing SCHEMA_SQL spins up an
# in-memory SQLite database, so derive the statements once per process.
@@ -422,8 +431,14 @@ class SessionSchemaMixin:
)
return None
- def _recover_stale_fts(self, cursor: sqlite3.Cursor, *, legacy: bool) -> bool:
- """Atomically rebuild stale base/trigram indexes and resume syncing."""
+ def _recover_stale_fts(
+ self, cursor: sqlite3.Cursor, *, legacy: bool, timeout_seconds=None
+ ) -> bool:
+ """Atomically rebuild stale base/trigram indexes and resume syncing.
+
+ *timeout_seconds* bounds the cross-process admission wait; None uses
+ the full startup budget, ``0`` is the non-blocking in-process retry.
+ """
foreign_holders = self._foreign_state_db_holders()
if foreign_holders:
now = time.time()
@@ -502,7 +517,9 @@ class SessionSchemaMixin:
# authority (fail closed). Losing the race means another process is
# already performing this exact recovery; the stale breadcrumb stays
# set, so this process simply keeps FTS detached and retries later.
- with fts_rebuild_admission(getattr(self, "db_path", None)) as admitted:
+ with fts_rebuild_admission(
+ getattr(self, "db_path", None), timeout_seconds=timeout_seconds
+ ) as admitted:
if not admitted:
logger.warning(
"Deferred stale state.db FTS rebuild: another process "
@@ -512,6 +529,66 @@ class SessionSchemaMixin:
return False
return self._recover_stale_fts_locked(cursor, legacy=legacy)
+ def retry_deferred_fts_recovery(self) -> bool:
+ """Retry a deferred stale-FTS rebuild on this open SessionDB.
+
+ ``_recover_stale_fts`` runs at open and fails closed when foreign
+ holders or the rebuild lock are busy, leaving ``_fts_stale`` set and
+ search on the LIKE fallback. Live write/search paths must never start
+ a full rebuild (#97940), so on a short-lived CLI that deferral is
+ cleared by the next process open — but a gateway opens state.db
+ once and stays up for days, so "next open" never came (#100108).
+ This is the in-process retry: bounded backoff from
+ ``_FTS_STALE_RETRY_SECONDS`` doubling to ``_FTS_STALE_RETRY_MAX_SECONDS``,
+ non-blocking admission (``timeout=0``) so a live holder is skipped and
+ tried again later, no new thread — the caller is an existing periodic
+ tick (gateway housekeeping).
+
+ Returns True only when the index was rebuilt and sync triggers
+ restored. Never raises.
+ """
+ if not getattr(self, "_fts_stale", False):
+ return False
+ if getattr(self, "read_only", False) or getattr(self, "_conn", None) is None:
+ return False
+ now = time.monotonic()
+ if now < getattr(self, "_fts_stale_retry_after", 0.0):
+ return False
+ interval = float(getattr(self, "_fts_stale_retry_interval", 0.0))
+ if interval <= 0.0:
+ interval = _FTS_STALE_RETRY_SECONDS
+ self._fts_stale_retry_after = now + interval
+ self._fts_stale_retry_interval = min(
+ max(interval, _FTS_STALE_RETRY_SECONDS, 1.0) * 2.0,
+ _FTS_STALE_RETRY_MAX_SECONDS,
+ )
+ try:
+ with self._lock:
+ if self._conn is None or not self._fts_stale:
+ return False
+ cursor = self._conn.cursor()
+ legacy = self._db_has_legacy_inline_fts(cursor)
+ recovered = self._recover_stale_fts(
+ cursor, legacy=legacy, timeout_seconds=0.0
+ )
+ if recovered:
+ # CJK was detached alongside the base indexes; its own
+ # ensure path decides when it comes back online.
+ self._ensure_fts_cjk_schema(cursor)
+ self._fts_stale_retry_interval = 0.0
+ try:
+ self._conn.commit()
+ except sqlite3.Error:
+ pass
+ return recovered
+ except Exception: # noqa: BLE001 - background retry must never raise
+ logger.warning(
+ "In-process retry of the deferred stale state.db FTS rebuild "
+ "failed; will retry later.",
+ exc_info=True,
+ )
+ return False
+
def _recover_stale_fts_locked(
self, cursor: sqlite3.Cursor, *, legacy: bool
) -> bool:
@@ -1510,7 +1587,8 @@ class SessionSchemaMixin:
breadcrumb is persisted, mirroring ``_enter_fts_fail_open``'s
ordering contract: triggers must never be live over an index with an
unrebuilt gap. FTS stays detached for this instance; the winner's
- rebuild — or ``_recover_stale_fts`` at the next startup — restores
+ rebuild — or ``retry_deferred_fts_recovery`` from the gateway
+ housekeeping tick, or ``_recover_stale_fts`` at the next startup — restores
the index and triggers atomically.
"""
with fts_rebuild_admission(getattr(self, "db_path", None)) as admitted:
diff --git a/hermes_state_search.py b/hermes_state_search.py
index 40fddb70fd..3dafeebc4a 100644
--- a/hermes_state_search.py
+++ b/hermes_state_search.py
@@ -2382,7 +2382,9 @@ class SessionSearchMixin:
FAILS CLOSED: if another process holds the rebuild lock beyond the
bounded wait, this call defers (returns 0) rather than racing it.
Callers already treat 0 as "rebuild made no progress" and fall back
- to the stale-FTS breadcrumb path, which retries at next startup.
+ to the stale-FTS breadcrumb path, which retries in-process from the
+ gateway housekeeping tick (``retry_deferred_fts_recovery``) and at
+ next startup.
Safe to call when FTS tables don't exist (skips them).
Returns the number of FTS indexes that were rebuilt.
diff --git a/locales/af.yaml b/locales/af.yaml
index 21806156ab..363563e9a4 100644
--- a/locales/af.yaml
+++ b/locales/af.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nHuidige modus: `{mode}`\\n\\nKies \\'n opsie:"
choice_fast: "fast — Priority Processing aan"
choice_normal: "normal — standaardverwerking"
+ choice_auto: "auto — vinnig vir die eerste sekondes van elke beurt"
+ choice_cold: "cold — vinnig slegs vir die eerste beurt van 'n sessie"
footer:
status: "📎 Looptyd-voetstuk: **{state}**\nVelde: `{fields}`\nPlatform: `{platform}`"
diff --git a/locales/ar.yaml b/locales/ar.yaml
index 1f68628a3b..abdc371ead 100644
--- a/locales/ar.yaml
+++ b/locales/ar.yaml
@@ -160,6 +160,8 @@ gateway:
picker_title: "⚡ **المعالجة ذات الأولوية**\n\nالوضع الحالي: `{mode}`\n\nاختر خيارًا:"
choice_fast: "fast — المعالجة ذات الأولوية مُفعّلة"
choice_normal: "normal — المعالجة القياسية"
+ choice_auto: "auto — سريع في الثواني الأولى من كل دور"
+ choice_cold: "cold — سريع في الدور الأول من الجلسة فقط"
footer:
status: "📎 تذييل التشغيل: **{state}**\nالحقول: `{fields}`\nالمنصّة: `{platform}`"
diff --git a/locales/de.yaml b/locales/de.yaml
index bc00bfe32f..d6e1528088 100644
--- a/locales/de.yaml
+++ b/locales/de.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nAktueller Modus: `{mode}`\\n\\nOption wählen:"
choice_fast: "fast — Priority Processing an"
choice_normal: "normal — Standardverarbeitung"
+ choice_auto: "auto — schnell in den ersten Sekunden jedes Zugs"
+ choice_cold: "cold — schnell nur im ersten Zug einer Sitzung"
footer:
status: "📎 Laufzeit-Fußzeile: **{state}**\nFelder: `{fields}`\nPlattform: `{platform}`"
diff --git a/locales/en.yaml b/locales/en.yaml
index 2adac023f2..9b06ae1e96 100644
--- a/locales/en.yaml
+++ b/locales/en.yaml
@@ -141,8 +141,8 @@ gateway:
fast:
not_supported: "⚡ /fast is only available for OpenAI models that support Priority Processing."
- status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast `"
- unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, status"
+ status: "⚡ Priority Processing\n\nCurrent mode: `{mode}`\n\n_Usage:_ `/fast `"
+ unknown_arg: "⚠️ Unknown argument: `{arg}`\n\n**Valid options:** normal, fast, auto, cold, status"
saved: "⚡ ✓ Priority Processing: **{label}** (saved to config)\n_(takes effect on next message)_"
session_only: "⚡ ✓ Priority Processing: **{label}** (this session only)"
label_fast: "FAST"
@@ -152,6 +152,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\n\nCurrent mode: `{mode}`\n\nPick an option:"
choice_fast: "fast — Priority Processing on"
choice_normal: "normal — standard processing"
+ choice_auto: "auto — fast for the first seconds of every turn"
+ choice_cold: "cold — fast for the first turn of a session only"
footer:
status: "📎 Runtime footer: **{state}**\nFields: `{fields}`\nPlatform: `{platform}`"
diff --git a/locales/es.yaml b/locales/es.yaml
index 6b06a52afb..06cd2e9e23 100644
--- a/locales/es.yaml
+++ b/locales/es.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nModo actual: `{mode}`\\n\\nElige una opción:"
choice_fast: "fast — Priority Processing activado"
choice_normal: "normal — procesamiento estándar"
+ choice_auto: "auto — rápido en los primeros segundos de cada turno"
+ choice_cold: "cold — rápido solo en el primer turno de una sesión"
footer:
status: "📎 Pie de ejecución: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`"
diff --git a/locales/fr.yaml b/locales/fr.yaml
index 4ce9760969..4f1faa6cbf 100644
--- a/locales/fr.yaml
+++ b/locales/fr.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nMode actuel : `{mode}`\\n\\nChoisissez une option :"
choice_fast: "fast — Priority Processing activé"
choice_normal: "normal — traitement standard"
+ choice_auto: "auto — rapide pendant les premières secondes de chaque tour"
+ choice_cold: "cold — rapide uniquement au premier tour d'une session"
footer:
status: "📎 Pied de page d'exécution : **{state}**\nChamps : `{fields}`\nPlateforme : `{platform}`"
diff --git a/locales/ga.yaml b/locales/ga.yaml
index 92ef5363ea..843dc5a1c2 100644
--- a/locales/ga.yaml
+++ b/locales/ga.yaml
@@ -141,6 +141,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nMód reatha: `{mode}`\\n\\nRoghnaigh rogha:"
choice_fast: "fast — Priority Processing ar siúl"
choice_normal: "normal — gnáthphróiseáil"
+ choice_auto: "auto — tapa do na chéad soicindí de gach seal"
+ choice_cold: "cold — tapa don chéad seal de sheisiún amháin"
footer:
status: "📎 Buntásc rite: **{state}**\nRéimsí: `{fields}`\nArdán: `{platform}`"
diff --git a/locales/hu.yaml b/locales/hu.yaml
index b8feb1b994..d134d4372f 100644
--- a/locales/hu.yaml
+++ b/locales/hu.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nJelenlegi mód: `{mode}`\\n\\nVálassz egy opciót:"
choice_fast: "fast — Priority Processing bekapcsolva"
choice_normal: "normal — normál feldolgozás"
+ choice_auto: "auto — gyors minden kör első másodperceiben"
+ choice_cold: "cold — gyors csak a munkamenet első körében"
footer:
status: "📎 Futási idejű lábléc: **{state}**\nMezők: `{fields}`\nPlatform: `{platform}`"
diff --git a/locales/it.yaml b/locales/it.yaml
index 758be5d8a7..a3480541bc 100644
--- a/locales/it.yaml
+++ b/locales/it.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nModalità attuale: `{mode}`\\n\\nScegli un\\'opzione:"
choice_fast: "fast — Priority Processing attivo"
choice_normal: "normal — elaborazione standard"
+ choice_auto: "auto — veloce nei primi secondi di ogni turno"
+ choice_cold: "cold — veloce solo nel primo turno di una sessione"
footer:
status: "📎 Footer di runtime: **{state}**\nCampi: `{fields}`\nPiattaforma: `{platform}`"
diff --git a/locales/ja.yaml b/locales/ja.yaml
index 28b41682aa..b691daf93d 100644
--- a/locales/ja.yaml
+++ b/locales/ja.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\n現在のモード: `{mode}`\\n\\nオプションを選択:"
choice_fast: "fast — Priority Processing オン"
choice_normal: "normal — 標準処理"
+ choice_auto: "auto — 各ターンの最初の数秒間だけ高速"
+ choice_cold: "cold — セッションの最初のターンのみ高速"
footer:
status: "📎 ランタイムフッター: **{state}**\nフィールド: `{fields}`\nプラットフォーム: `{platform}`"
diff --git a/locales/ko.yaml b/locales/ko.yaml
index ecf58bbc68..f7f4f25a06 100644
--- a/locales/ko.yaml
+++ b/locales/ko.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\n현재 모드: `{mode}`\\n\\n옵션을 선택하세요:"
choice_fast: "fast — Priority Processing 켜기"
choice_normal: "normal — 표준 처리"
+ choice_auto: "auto — 매 턴의 처음 몇 초 동안 빠름"
+ choice_cold: "cold — 세션의 첫 턴에만 빠름"
footer:
status: "📎 런타임 푸터: **{state}**\n필드: `{fields}`\n플랫폼: `{platform}`"
diff --git a/locales/pt.yaml b/locales/pt.yaml
index 1ac1fd4b00..f18100340e 100644
--- a/locales/pt.yaml
+++ b/locales/pt.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nModo atual: `{mode}`\\n\\nEscolha uma opção:"
choice_fast: "fast — Priority Processing ativado"
choice_normal: "normal — processamento padrão"
+ choice_auto: "auto — rápido nos primeiros segundos de cada turno"
+ choice_cold: "cold — rápido apenas no primeiro turno de uma sessão"
footer:
status: "📎 Rodapé de execução: **{state}**\nCampos: `{fields}`\nPlataforma: `{platform}`"
diff --git a/locales/ru.yaml b/locales/ru.yaml
index 51c892e02e..7fd2c285f2 100644
--- a/locales/ru.yaml
+++ b/locales/ru.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nТекущий режим: `{mode}`\\n\\nВыберите вариант:"
choice_fast: "fast — Priority Processing включён"
choice_normal: "normal — стандартная обработка"
+ choice_auto: "auto — быстро в первые секунды каждого хода"
+ choice_cold: "cold — быстро только на первом ходе сессии"
footer:
status: "📎 Нижний колонтитул среды выполнения: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`"
diff --git a/locales/tr.yaml b/locales/tr.yaml
index a88b1d0586..7af791fb74 100644
--- a/locales/tr.yaml
+++ b/locales/tr.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nMevcut mod: `{mode}`\\n\\nBir seçenek seçin:"
choice_fast: "fast — Priority Processing açık"
choice_normal: "normal — standart işleme"
+ choice_auto: "auto — her turun ilk saniyelerinde hızlı"
+ choice_cold: "cold — yalnızca oturumun ilk turunda hızlı"
footer:
status: "📎 Çalışma zamanı altbilgisi: **{state}**\nAlanlar: `{fields}`\nPlatform: `{platform}`"
diff --git a/locales/uk.yaml b/locales/uk.yaml
index 730a052cd5..5e63f0bdad 100644
--- a/locales/uk.yaml
+++ b/locales/uk.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\nПоточний режим: `{mode}`\\n\\nОберіть варіант:"
choice_fast: "fast — Priority Processing увімкнено"
choice_normal: "normal — стандартна обробка"
+ choice_auto: "auto — швидко в перші секунди кожного ходу"
+ choice_cold: "cold — швидко лише на першому ході сесії"
footer:
status: "📎 Нижній колонтитул середовища: **{state}**\nПоля: `{fields}`\nПлатформа: `{platform}`"
diff --git a/locales/zh-hant.yaml b/locales/zh-hant.yaml
index 9468fbba1c..b9091c2aa2 100644
--- a/locales/zh-hant.yaml
+++ b/locales/zh-hant.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **Priority Processing**\\n\\n目前模式:`{mode}`\\n\\n請選擇:"
choice_fast: "fast — 開啟 Priority Processing"
choice_normal: "normal — 標準處理"
+ choice_auto: "auto — 每輪的前幾秒快速"
+ choice_cold: "cold — 僅會話的第一輪快速"
footer:
status: "📎 執行階段頁尾:**{state}**\n欄位:`{fields}`\n平台:`{platform}`"
diff --git a/locales/zh.yaml b/locales/zh.yaml
index f659de9a24..dde25d0e82 100644
--- a/locales/zh.yaml
+++ b/locales/zh.yaml
@@ -137,6 +137,8 @@ gateway:
picker_title: "⚡ **优先处理**\\n\\n当前模式:`{mode}`\\n\\n请选择:"
choice_fast: "fast — 开启优先处理"
choice_normal: "normal — 标准处理"
+ choice_auto: "auto — 每轮的前几秒快速"
+ choice_cold: "cold — 仅会话的第一轮快速"
footer:
status: "📎 运行时页脚:**{state}**\n字段:`{fields}`\n平台:`{platform}`"
diff --git a/model_tools.py b/model_tools.py
index 20ce327a2a..e5e4e657e8 100644
--- a/model_tools.py
+++ b/model_tools.py
@@ -279,7 +279,7 @@ _LEGACY_TOOLSET_MAP = {
"browser_press", "browser_get_images",
"browser_vision", "browser_console"
],
- "cronjob_tools": ["cronjob"],
+ "cronjob_tools": ["cronjob_manage"],
"file_tools": ["read_file", "write_file", "patch", "search_files"],
"tts_tools": ["text_to_speech"],
}
@@ -602,7 +602,7 @@ def _compute_tool_definitions(
# Same session-level seam as the browser_exec gate above.
if "delegate_task" in available_tool_names:
blocked_present = [
- t for t in ("clarify", "memory", "cronjob") if t in available_tool_names
+ t for t in ("clarify", "memory", "cronjob_manage") if t in available_tool_names
]
if len(blocked_present) < 3:
full_offvariant = "delegate_task, clarify, memory, or cronjob"
@@ -788,7 +788,18 @@ def _resolve_active_context_length() -> int:
# because they need agent-level state (TodoStore, MemoryStore, etc.).
# The registry still holds their schemas; dispatch just returns a stub error
# so if something slips through, the LLM sees a sensible message.
-_AGENT_LOOP_TOOLS = {"todo", "memory", "session_search", "delegate_task"}
+_AGENT_LOOP_TOOLS = {"todo_list", "memory", "session_search", "delegate_task"}
+
+# Legacy tool-name aliases (2026-08 renames): accepted at every dispatch seam
+# (handle_function_call + both executors) so old sessions and saved prompts
+# keep working; schemas only advertise the new names.
+_LEGACY_TOOL_ALIASES = {
+ "todo": "todo_list",
+ "cronjob": "cronjob_manage",
+ "process": "process_manage",
+ "tour": "gui_tour",
+ "tip": "show_tip",
+}
_READ_SEARCH_TOOLS = {"read_file", "search_files"}
@@ -1284,6 +1295,13 @@ def handle_function_call(
function_args = {}
_tool_middleware_trace = list(tool_request_middleware_trace or [])
+ # ── Legacy tool-name aliases (2026-08 renames) ────────────────────
+ # Old sessions resuming mid-conversation (and users' muscle memory in
+ # saved skills/cron prompts) still emit the pre-rename names. Alias at
+ # the dispatch seam so every replay keeps working; new schemas only
+ # advertise the new names, so fresh sessions never see the old ones.
+ function_name = _LEGACY_TOOL_ALIASES.get(function_name, function_name)
+
# ── Tool Search bridge dispatch ──────────────────────────────────
# tool_search and tool_describe are pure catalog reads — handle them
# inline. tool_call is unwrapped to the underlying tool so that every
@@ -1368,9 +1386,9 @@ def handle_function_call(
"Use tool_search to find tools you can call."
)
)
- # Probe-validate against the deferred tool's schema (ironclaw#5149):
- # a blind call missing required arguments returns the parameter
- # schema instead of dispatching into an opaque downstream failure.
+ # Validate against the deferred tool's concrete schema before
+ # dispatch. This covers constraints the provider cannot enforce
+ # through the generic tool_call ``arguments: object`` bridge.
_probe_err = _ts_mod.validate_deferred_call_args(underlying_name, underlying_args)
if _probe_err is not None:
return _return_bridge_result(_probe_err)
diff --git a/plugins/dashboard_auth/nous/__init__.py b/plugins/dashboard_auth/nous/__init__.py
index 69acd18e36..fdb28173ee 100644
--- a/plugins/dashboard_auth/nous/__init__.py
+++ b/plugins/dashboard_auth/nous/__init__.py
@@ -85,6 +85,7 @@ from hermes_cli.dashboard_auth import (
LoginStart,
ProviderError,
RefreshExpiredError,
+ classify_jwks_lookup_error,
Session,
)
@@ -436,10 +437,11 @@ class NousDashboardAuthProvider(DashboardAuthProvider):
signing_key = self._get_jwks_client().get_signing_key_from_jwt(
access_token
)
- except jwt.PyJWKClientError as exc:
- raise ProviderError(f"JWKS lookup failed: {exc}") from exc
- except Exception as exc: # pragma: no cover - defensive
- raise ProviderError(f"JWKS lookup failed: {exc!r}") from exc
+ except Exception as exc:
+ # Unreachable JWKS -> ProviderError (503); a bearer that is not
+ # one of our JWTs (opaque peer key, foreign kid) -> InvalidCodeError
+ # (None / next provider). Folding both into 503 produced #94558.
+ raise classify_jwks_lookup_error(exc) from exc
try:
claims = jwt.decode(
diff --git a/plugins/dashboard_auth/self_hosted/__init__.py b/plugins/dashboard_auth/self_hosted/__init__.py
index 2672006571..7b7283b2df 100644
--- a/plugins/dashboard_auth/self_hosted/__init__.py
+++ b/plugins/dashboard_auth/self_hosted/__init__.py
@@ -93,6 +93,7 @@ from hermes_cli.dashboard_auth import (
LoginStart,
ProviderError,
RefreshExpiredError,
+ classify_jwks_lookup_error,
Session,
)
@@ -617,10 +618,11 @@ class SelfHostedOIDCProvider(DashboardAuthProvider):
signing_key = self._get_jwks_client().get_signing_key_from_jwt(
id_token
)
- except jwt.PyJWKClientError as exc:
- raise ProviderError(f"JWKS lookup failed: {exc}") from exc
- except Exception as exc: # pragma: no cover - defensive
- raise ProviderError(f"JWKS lookup failed: {exc!r}") from exc
+ except Exception as exc:
+ # Unreachable JWKS -> ProviderError (503); a bearer that is not
+ # one of our JWTs (opaque peer key, foreign kid) -> InvalidCodeError
+ # (None / next provider). Folding both into 503 produced #94558.
+ raise classify_jwks_lookup_error(exc) from exc
try:
claims = jwt.decode(
diff --git a/plugins/image_gen/meta-ai/__init__.py b/plugins/image_gen/meta-ai/__init__.py
new file mode 100644
index 0000000000..e5c2d43993
--- /dev/null
+++ b/plugins/image_gen/meta-ai/__init__.py
@@ -0,0 +1,297 @@
+"""Meta Model API image generation backend.
+
+Exposes Meta's ``muse-image`` model(s) as an :class:`ImageGenProvider`.
+The Meta Model API (https://api.meta.ai/v1) is OpenAI-compatible, so we reuse
+the OpenAI Python SDK pointed at Meta's base URL and authenticate with
+``META_MODEL_API_KEY``.
+
+Output is base64 JSON (WebP) -> saved under ``$HERMES_HOME/cache/images/``.
+
+Selection precedence (first hit wins):
+ 1. ``model`` kwarg forwarded by the dispatcher (the ``hermes tools`` pick)
+ 2. ``META_IMAGE_MODEL`` env var (escape hatch for scripts / tests)
+ 3. ``image_gen.meta-ai.model`` in ``config.yaml``
+ 4. ``image_gen.model`` in ``config.yaml`` (when it's one of our IDs)
+ 5. :data:`DEFAULT_MODEL`
+"""
+
+from __future__ import annotations
+
+import logging
+import os
+from typing import Any, Dict, List, Optional, Tuple
+
+from agent.secret_scope import get_secret
+from agent.image_gen_provider import (
+ DEFAULT_ASPECT_RATIO,
+ ImageGenProvider,
+ error_response,
+ normalize_reference_images,
+ resolve_aspect_ratio,
+ save_b64_image,
+ save_url_image,
+ success_response,
+)
+
+logger = logging.getLogger(__name__)
+
+DEFAULT_BASE_URL = "https://api.meta.ai/v1"
+# Auth env vars, in priority order. Mirrors the bundled ``meta-ai`` chat
+# provider (plugins/model-providers/meta-ai): MODEL_API_KEY is Meta's
+# documented var; the rest are accepted aliases.
+API_KEY_ENVS = ("MODEL_API_KEY", "META_API_KEY", "META_MODEL_API_KEY")
+# Primary key shown in setup prompts / error messages.
+API_KEY_ENV = "META_MODEL_API_KEY"
+# Optional base-url override (same var the chat provider honors).
+BASE_URL_ENV = "META_BASE_URL"
+
+
+def _resolve_api_key() -> Optional[str]:
+ """First non-empty auth env var, checked in priority order."""
+ for env in API_KEY_ENVS:
+ val = get_secret(env)
+ if val:
+ return val
+ return None
+
+
+def _resolve_base_url() -> str:
+ return (os.environ.get(BASE_URL_ENV) or "").strip() or DEFAULT_BASE_URL
+
+
+# ---------------------------------------------------------------------------
+# Model catalog
+# ---------------------------------------------------------------------------
+# Catalog shown in `hermes tools` and matched against `image_gen.model`.
+# The model id is sent verbatim to the Meta Model API (`/v1/images/generations`).
+_MODELS: Dict[str, Dict[str, Any]] = {
+ "muse-image-1.0": {
+ "display": "Muse Image 1.0",
+ "speed": "~10s",
+ "strengths": "Meta Model API image generation",
+ "price": "$0.01/image",
+ },
+}
+DEFAULT_MODEL = "muse-image-1.0"
+
+# aspect_ratio -> OpenAI-style size string
+_SIZES: Dict[str, str] = {
+ "square": "1024x1024",
+ "landscape": "1536x1024",
+ "portrait": "1024x1536",
+}
+
+
+def _resolve_model(caller_model: Optional[str] = None) -> Tuple[str, Dict[str, Any]]:
+ """Return (model_id, metadata) using the documented precedence chain.
+
+ ``caller_model`` is the ``model`` kwarg the dispatcher forwards from the
+ top-level ``image_gen.model`` config key (what ``hermes tools`` writes).
+ It wins when it names one of our models, mirroring the xai/krea/openrouter
+ providers, so a user's picker choice is never silently dropped.
+ """
+ if caller_model and caller_model in _MODELS:
+ return caller_model, _MODELS[caller_model]
+
+ env_model = os.environ.get("META_IMAGE_MODEL")
+ if env_model and env_model in _MODELS:
+ return env_model, _MODELS[env_model]
+
+ try:
+ from hermes_cli.config import load_config
+
+ cfg = load_config() or {}
+ ig = cfg.get("image_gen") or {}
+ scoped = (ig.get("meta-ai") or {}).get("model")
+ if scoped and scoped in _MODELS:
+ return scoped, _MODELS[scoped]
+ top = ig.get("model")
+ if top and top in _MODELS:
+ return top, _MODELS[top]
+ except Exception:
+ logger.debug("Could not read image_gen model from config", exc_info=True)
+
+ return DEFAULT_MODEL, _MODELS[DEFAULT_MODEL]
+
+
+class MetaImageGenProvider(ImageGenProvider):
+ """Meta Model API ``images.generate`` backend (muse-image)."""
+
+ @property
+ def name(self) -> str:
+ return "meta-ai"
+
+ @property
+ def display_name(self) -> str:
+ return "Meta Model API"
+
+ def is_available(self) -> bool:
+ if not _resolve_api_key():
+ return False
+ try:
+ import openai # noqa: F401
+ except ImportError:
+ return False
+ return True
+
+ def list_models(self) -> List[Dict[str, Any]]:
+ return [
+ {
+ "id": mid,
+ "display": m["display"],
+ "speed": m["speed"],
+ "strengths": m["strengths"],
+ "price": m["price"],
+ }
+ for mid, m in _MODELS.items()
+ ]
+
+ def default_model(self) -> Optional[str]:
+ return DEFAULT_MODEL
+
+ def get_setup_schema(self) -> Dict[str, Any]:
+ return {
+ "name": "Meta Model API",
+ "badge": "paid",
+ "tag": "Muse Image via Meta Model API (api.meta.ai)",
+ "env_vars": [
+ {
+ "key": API_KEY_ENV,
+ "prompt": "Meta Model API key (LLM|... token)",
+ "url": "https://api.meta.ai",
+ },
+ ],
+ }
+
+ def capabilities(self) -> Dict[str, Any]:
+ # Text-to-image only for now. Bump this once image-to-image is verified
+ # against the Meta endpoint.
+ return {"modalities": ["text"], "max_reference_images": 0}
+
+ def generate(
+ self,
+ prompt: str,
+ aspect_ratio: str = DEFAULT_ASPECT_RATIO,
+ *,
+ image_url: Optional[str] = None,
+ reference_image_urls: Optional[List[str]] = None,
+ **kwargs: Any,
+ ) -> Dict[str, Any]:
+ prompt = (prompt or "").strip()
+ aspect = resolve_aspect_ratio(aspect_ratio)
+
+ if not prompt:
+ return error_response(
+ error="Prompt is required and must be a non-empty string",
+ error_type="invalid_argument",
+ provider="meta-ai",
+ aspect_ratio=aspect,
+ )
+
+ api_key = _resolve_api_key()
+ if not api_key:
+ return error_response(
+ error=(
+ f"{API_KEY_ENV} not set. Run `hermes tools` -> Image "
+ "Generation -> Meta Model API to configure."
+ ),
+ error_type="auth_required",
+ provider="meta-ai",
+ aspect_ratio=aspect,
+ )
+
+ try:
+ import openai
+ except ImportError:
+ return error_response(
+ error="openai Python package not installed (pip install openai)",
+ error_type="missing_dependency",
+ provider="meta-ai",
+ aspect_ratio=aspect,
+ )
+
+ model_id, _meta = _resolve_model(kwargs.get("model"))
+ size = _SIZES.get(aspect, _SIZES["square"])
+
+ client = openai.OpenAI(api_key=api_key, base_url=_resolve_base_url())
+
+ payload: Dict[str, Any] = {
+ "model": model_id,
+ "prompt": prompt,
+ "size": size,
+ "n": 1,
+ }
+
+ try:
+ response = client.images.generate(**payload)
+ except Exception as exc:
+ logger.debug("Meta image generation failed", exc_info=True)
+ return error_response(
+ error=f"Meta image generation failed: {exc}",
+ error_type="api_error",
+ provider="meta-ai",
+ model=model_id,
+ prompt=prompt,
+ aspect_ratio=aspect,
+ )
+
+ try:
+ first = response.data[0]
+ except (AttributeError, IndexError, TypeError):
+ return error_response(
+ error="Meta response contained no image data",
+ error_type="empty_response",
+ provider="meta-ai",
+ model=model_id,
+ prompt=prompt,
+ aspect_ratio=aspect,
+ )
+
+ b64 = getattr(first, "b64_json", None)
+ url = getattr(first, "url", None)
+
+ try:
+ if b64:
+ path = save_b64_image(b64, prefix="meta", extension="webp")
+ image_ref = str(path)
+ elif url:
+ path = save_url_image(url, prefix="meta")
+ image_ref = str(path)
+ else:
+ return error_response(
+ error="Meta response contained neither b64_json nor URL",
+ error_type="empty_response",
+ provider="meta-ai",
+ model=model_id,
+ prompt=prompt,
+ aspect_ratio=aspect,
+ )
+ except Exception as exc:
+ return error_response(
+ error=f"Failed to save Meta image: {exc}",
+ error_type="io_error",
+ provider="meta-ai",
+ model=model_id,
+ prompt=prompt,
+ aspect_ratio=aspect,
+ )
+
+ revised_prompt = getattr(first, "revised_prompt", None)
+ extra: Dict[str, Any] = {"size": size}
+ if revised_prompt:
+ extra["revised_prompt"] = revised_prompt
+
+ return success_response(
+ image=image_ref,
+ model=model_id,
+ prompt=prompt,
+ aspect_ratio=aspect,
+ provider="meta-ai",
+ modality="text",
+ extra=extra,
+ )
+
+
+def register(ctx) -> None:
+ """Plugin entry point -- wire ``MetaImageGenProvider`` into the registry."""
+ ctx.register_image_gen_provider(MetaImageGenProvider())
diff --git a/plugins/image_gen/meta-ai/plugin.yaml b/plugins/image_gen/meta-ai/plugin.yaml
new file mode 100644
index 0000000000..2d5bb36649
--- /dev/null
+++ b/plugins/image_gen/meta-ai/plugin.yaml
@@ -0,0 +1,7 @@
+name: meta-ai-image-gen
+version: 1.0.0
+description: "Meta Model API image generation backend (muse-image). OpenAI-compatible /v1/images/generations. Saves images to $HERMES_HOME/cache/images/."
+author: Meta Platforms, Inc.
+kind: backend
+requires_env:
+ - META_MODEL_API_KEY
diff --git a/plugins/memory/hindsight/__init__.py b/plugins/memory/hindsight/__init__.py
index 7f9079ee27..cf3b9f618f 100644
--- a/plugins/memory/hindsight/__init__.py
+++ b/plugins/memory/hindsight/__init__.py
@@ -32,6 +32,7 @@ from __future__ import annotations
import asyncio
import atexit
+import contextvars
import importlib
import json
import logging
@@ -1314,8 +1315,17 @@ class HindsightMemoryProvider(MemoryProvider):
# If the previous writer exited (e.g. after a prior shutdown), reset
# the flag so this fresh writer is allowed to drain new jobs.
self._shutting_down.clear()
+ # Per-provider background threads start with an EMPTY contextvars
+ # Context. Under multiplex_profiles the spawning thread carries the
+ # profile's secret scope + HERMES_HOME override (gateway/run.py wraps
+ # the agent turn in copy_context().run), and get_secret fails closed
+ # without it (#92608). Snapshot the spawner's context into the thread.
+ # (The shared ``hindsight-loop`` thread needs no wrap: coroutines
+ # scheduled via run_coroutine_threadsafe inherit the submitter's
+ # context per call, so one loop can serve every profile.)
thread = threading.Thread(
- target=self._writer_loop,
+ target=contextvars.copy_context().run,
+ args=(self._writer_loop,),
daemon=True,
name="hindsight-writer",
)
@@ -1835,7 +1845,12 @@ class HindsightMemoryProvider(MemoryProvider):
f.write(f"\n=== Daemon startup failed: {e} ===\n")
traceback.print_exc(file=f)
- t = threading.Thread(target=_start_daemon, daemon=True, name="hindsight-daemon-start")
+ t = threading.Thread(
+ target=contextvars.copy_context().run,
+ args=(_start_daemon,),
+ daemon=True,
+ name="hindsight-daemon-start",
+ )
t.start()
def system_prompt_block(self) -> str:
@@ -1992,7 +2007,12 @@ class HindsightMemoryProvider(MemoryProvider):
self._prefetch_result = recalled.text
self._prefetch_count = recalled.count
- self._prefetch_thread = threading.Thread(target=_run, daemon=True, name="hindsight-prefetch")
+ self._prefetch_thread = threading.Thread(
+ target=contextvars.copy_context().run,
+ args=(_run,),
+ daemon=True,
+ name="hindsight-prefetch",
+ )
self._prefetch_thread.start()
def _build_turn_messages(self, user_content: str, assistant_content: str) -> List[Dict[str, str]]:
diff --git a/plugins/model-providers/alibaba-coding-plan/__init__.py b/plugins/model-providers/alibaba-coding-plan/__init__.py
index b420fbbbd9..4723606d0e 100644
--- a/plugins/model-providers/alibaba-coding-plan/__init__.py
+++ b/plugins/model-providers/alibaba-coding-plan/__init__.py
@@ -9,6 +9,10 @@ Region split, mirroring the base DashScope pair (#73265):
Profile names match the models.dev catalog keys exactly so model metadata
lines up and ``model.provider: alibaba-coding-plan-cn`` resolves at runtime.
+
+The CN profile checks its own ``ALIBABA_CODING_PLAN_CN_API_KEY`` first (#101122,
+mirroring kimi-coding-cn) and keeps the shared vars as ordered fallbacks so
+existing CN users configured with the shared key keep working.
"""
from providers import register_provider
@@ -31,7 +35,7 @@ alibaba_coding_plan_cn = ProviderProfile(
display_name="Alibaba Cloud (Coding Plan, China)",
description="Alibaba Cloud Coding Plan, mainland-China endpoint",
signup_url="https://help.aliyun.com/zh/model-studio/",
- env_vars=("ALIBABA_CODING_PLAN_API_KEY", "DASHSCOPE_API_KEY", "ALIBABA_CODING_PLAN_CN_BASE_URL"),
+ env_vars=("ALIBABA_CODING_PLAN_CN_API_KEY", "ALIBABA_CODING_PLAN_API_KEY", "DASHSCOPE_API_KEY", "ALIBABA_CODING_PLAN_CN_BASE_URL"),
base_url="https://coding.dashscope.aliyuncs.com/v1",
auth_type="api_key",
)
diff --git a/plugins/model-providers/alibaba/__init__.py b/plugins/model-providers/alibaba/__init__.py
index 6135945d50..a9198d3931 100644
--- a/plugins/model-providers/alibaba/__init__.py
+++ b/plugins/model-providers/alibaba/__init__.py
@@ -55,7 +55,7 @@ alibaba_token_plan_cn = ProviderProfile(
display_name="Alibaba Cloud (Token Plan, China)",
description="Alibaba Cloud Model Studio Token Plan, mainland-China endpoint",
signup_url="https://help.aliyun.com/zh/model-studio/",
- env_vars=("ALIBABA_TOKEN_PLAN_API_KEY", "ALIBABA_TOKEN_PLAN_CN_BASE_URL"),
+ env_vars=("ALIBABA_TOKEN_PLAN_CN_API_KEY", "ALIBABA_TOKEN_PLAN_API_KEY", "ALIBABA_TOKEN_PLAN_CN_BASE_URL"),
base_url="https://token-plan.cn-beijing.maas.aliyuncs.com/compatible-mode/v1",
auth_type="api_key",
)
diff --git a/plugins/platforms/a2a/adapter.py b/plugins/platforms/a2a/adapter.py
index 79842c88c6..7a9b6b1142 100644
--- a/plugins/platforms/a2a/adapter.py
+++ b/plugins/platforms/a2a/adapter.py
@@ -74,8 +74,30 @@ def _reply_timeout() -> float:
return 300.0
+def _profile_scoped() -> bool:
+ """True when running inside a multiplexed secondary profile's scope.
+
+ Secondary-profile adapters are constructed inside ``_profile_runtime_scope``
+ (secret scope installed + multiplex active) — the same discriminator the
+ Buzz/SimpleX adapters use for this bug class (#98738). The DEFAULT profile
+ under multiplexing runs unscoped: ``os.environ`` holds its own bridge
+ output there and keeps its legacy precedence.
+ """
+ try:
+ from agent.secret_scope import current_secret_scope, is_multiplex_active
+
+ return bool(is_multiplex_active() and current_secret_scope() is not None)
+ except Exception:
+ return False
+
+
def _default_agent_name() -> str:
- name = os.getenv("A2A_AGENT_NAME", "").strip()
+ # Scope-aware: inside a secondary multiplex profile, os.environ holds the
+ # DEFAULT profile's bridged A2A_AGENT_NAME — borrowing it would brand a
+ # secondary profile's Agent Card with another profile's identity. There
+ # is no per-profile config.yaml equivalent yet, so a scoped profile just
+ # falls through to the hostname-based default below instead.
+ name = "" if _profile_scoped() else os.getenv("A2A_AGENT_NAME", "").strip()
if name:
return name
try:
@@ -227,7 +249,7 @@ class A2ARequestHandler(BaseHTTPRequestHandler):
}
# Do not leak profile/tenant topology on remote unauthenticated GETs.
# Agent Cards are intentionally public; health topology is not.
- if security.localhost_only() or security.authenticate(
+ if self.adapter._security_context.localhost_only() or self.adapter._security_context.authenticate(
self.headers.get("Authorization"),
self.client_address[0] if self.client_address else "",
) is not None:
@@ -246,7 +268,9 @@ class A2ARequestHandler(BaseHTTPRequestHandler):
# Identity comes from the presented credential (or the socket in
# localhost-only mode) — never from the request body.
- identity = security.authenticate(self.headers.get("Authorization"), client_ip)
+ identity = adapter._security_context.authenticate(
+ self.headers.get("Authorization"), client_ip
+ )
if identity is None:
self._json(401, protocol.jsonrpc_error(None, protocol.ERR_UNAUTHORIZED, "unauthorized"))
return
@@ -292,7 +316,7 @@ class A2ARequestHandler(BaseHTTPRequestHandler):
self._json(429, protocol.jsonrpc_error(req_id, protocol.ERR_RATE_LIMITED, "rate limit exceeded"))
return
- if not security.is_trusted_peer(identity):
+ if not adapter._security_context.is_trusted_peer(identity):
self._json(403, protocol.jsonrpc_error(
req_id, protocol.ERR_UNTRUSTED_PEER, f"peer '{identity}' not trusted"))
return
@@ -343,8 +367,17 @@ class A2AAdapter(BasePlatformAdapter):
super().__init__(config=config, platform=platform)
extra = getattr(config, "extra", {}) or {}
- self.port = int(os.getenv("A2A_PORT") or extra.get("port", _DEFAULT_PORT))
- self.host = security.resolve_bind_host()
+ # Scope-aware: a secondary multiplex profile must not borrow the
+ # default profile's bridged A2A_PORT (mirrors the Buzz/SimpleX fix
+ # for #98738) — an unconfigured profile falls closed to the module
+ # default port instead. (advertised_toolsets has the same env-leak
+ # shape but is left unscoped here — see the "Scope note" in this
+ # fix's PR description: open PR #98937 is actively rewriting this
+ # field's None-vs-empty-list semantics.)
+ self._security_context = security.A2ASecurityContext.capture()
+ _port_env = None if _profile_scoped() else os.getenv("A2A_PORT")
+ self.port = int(_port_env or extra.get("port", _DEFAULT_PORT))
+ self.host = self._security_context.resolve_bind_host()
self.agent_name = _default_agent_name()
self._advertised_toolsets = [
t.strip() for t in (
@@ -440,7 +473,11 @@ class A2AAdapter(BasePlatformAdapter):
self._mark_connected()
- exposure = "localhost-only" if security.localhost_only() else "REMOTE (bearer auth)"
+ exposure = (
+ "localhost-only"
+ if self._security_context.localhost_only()
+ else "REMOTE (bearer auth)"
+ )
logger.info(
"A2A: serving Agent Card + JSON-RPC on http://%s:%s (%s) as %r; %d routed agent(s)",
self.host, self.port, exposure, self.agent_name, len(self._agents),
@@ -502,9 +539,15 @@ class A2AAdapter(BasePlatformAdapter):
raw = cfg.get("a2a_served_agents") or (cfg.get("a2a") or {}).get("served_agents")
agents: dict[str, dict] = {}
- default_desc = os.getenv(
- "A2A_AGENT_DESCRIPTION",
- "Hermes Agent — a general-purpose agent reachable over A2A.",
+ # Scope-aware for the same reason as port/toolsets above: a secondary
+ # profile must not inherit the default profile's A2A_AGENT_DESCRIPTION.
+ default_desc = (
+ "Hermes Agent — a general-purpose agent reachable over A2A."
+ if _profile_scoped()
+ else os.getenv(
+ "A2A_AGENT_DESCRIPTION",
+ "Hermes Agent — a general-purpose agent reachable over A2A.",
+ )
)
agents[""] = {
"slug": "",
@@ -613,7 +656,7 @@ class A2AAdapter(BasePlatformAdapter):
skills=self._advertised_skills(agent),
streaming=bool(agent.get("local", True)),
push_notifications=True,
- auth_required=not security.localhost_only(),
+ auth_required=not self._security_context.localhost_only(),
tenant=str(agent.get("tenant") or ""),
)
@@ -1192,7 +1235,10 @@ class A2AAdapter(BasePlatformAdapter):
if not callback_url:
return
- if not security.is_safe_callback_url(callback_url):
+ if not security.is_safe_callback_url(
+ callback_url,
+ localhost_mode=self._security_context.localhost_only(),
+ ):
logger.warning("A2A: push notification for task %s blocked — unsafe callback URL: %s",
task_id, callback_url)
protocol.metrics.push_failed += 1
@@ -1201,7 +1247,7 @@ class A2AAdapter(BasePlatformAdapter):
# Push payload uses the StreamResponse format (same as streaming).
payload = protocol.status_update(task_id, context_id, state, (reply or "")[:2000])
- signature = security.sign_push_payload(payload)
+ signature = self._security_context.sign_push_payload(payload)
headers = {"Content-Type": "application/json"}
if signature:
headers["X-A2A-Signature"] = signature
diff --git a/plugins/platforms/a2a/security.py b/plugins/platforms/a2a/security.py
index 753c202a54..350031b9df 100644
--- a/plugins/platforms/a2a/security.py
+++ b/plugins/platforms/a2a/security.py
@@ -31,29 +31,44 @@ import logging
import os
import re
import time
+from dataclasses import dataclass
from pathlib import Path
from typing import Optional
logger = logging.getLogger(__name__)
-# --------------------------------------------------------------------------
-# Bearer auth + peer identity
-# --------------------------------------------------------------------------
+def _profile_scoped() -> bool:
+ """True when running inside a multiplexed secondary profile's scope.
-def get_bearer_token() -> str:
- """Return the configured shared inbound bearer token (empty if none)."""
- return os.getenv("A2A_BEARER_TOKEN", "").strip()
-
-
-def get_peer_tokens() -> dict[str, str]:
- """Parse A2A_PEER_TOKENS ("alice:tok1,bob:tok2") into {token: peer_name}.
-
- Per-peer tokens give each remote agent its own credential, so the identity
- used for rate limiting, trust, and audit is authenticated — not whatever
- the request body claims.
+ Same discriminator as the Buzz/SimpleX/Raft adapters (#98738): secret
+ scope installed + multiplex active. The DEFAULT profile under
+ multiplexing (and every single-profile process) runs unscoped and keeps
+ its legacy ``os.environ`` precedence.
"""
- raw = os.getenv("A2A_PEER_TOKENS", "").strip()
+ try:
+ from agent.secret_scope import current_secret_scope, is_multiplex_active
+
+ return bool(is_multiplex_active() and current_secret_scope() is not None)
+ except Exception:
+ return False
+
+
+def _startup_env(name: str) -> str:
+ """Read one A2A setting from the active profile's scope, else the env.
+
+ Inside a secondary profile's scope the scope is authoritative: a miss
+ yields "" and never falls through to ``os.environ`` (which holds the
+ default profile's tokens in a multiplexer).
+ """
+ if _profile_scoped():
+ from agent.secret_scope import get_secret
+
+ return (get_secret(name) or "").strip()
+ return os.getenv(name, "").strip()
+
+
+def _parse_peer_tokens(raw: str) -> dict[str, str]:
out: dict[str, str] = {}
for pair in raw.split(","):
pair = pair.strip()
@@ -66,6 +81,115 @@ def get_peer_tokens() -> dict[str, str]:
return out
+def _configured_trusted_peers() -> frozenset[str]:
+ raw = _startup_env("A2A_TRUSTED_PEERS")
+ if raw:
+ return frozenset(p.strip() for p in raw.split(",") if p.strip())
+ try:
+ from hermes_cli.config import load_config
+
+ cfg = load_config() or {}
+ peers = (cfg.get("a2a") or {}).get("trusted_peers", [])
+ if isinstance(peers, list):
+ return frozenset(str(peer).strip() for peer in peers if str(peer).strip())
+ except Exception:
+ pass
+ return frozenset()
+
+
+@dataclass(frozen=True)
+class A2ASecurityContext:
+ """Immutable, profile-scoped security settings captured at adapter startup.
+
+ ``ThreadingHTTPServer`` handles requests on fresh threads that do not inherit
+ the gateway's profile ContextVars. Keeping the resolved settings on the
+ adapter prevents those threads from falling back to another profile's
+ process-global environment.
+ """
+
+ bearer_token: str
+ peer_tokens: tuple[tuple[str, str], ...]
+ trusted_peers: frozenset[str]
+ allow_all_users: bool
+ requested_host: str
+ push_secret: str
+
+ @classmethod
+ def capture(cls) -> "A2ASecurityContext":
+ bearer_token = _startup_env("A2A_BEARER_TOKEN")
+ return cls(
+ bearer_token=bearer_token,
+ peer_tokens=tuple(_parse_peer_tokens(_startup_env("A2A_PEER_TOKENS")).items()),
+ trusted_peers=_configured_trusted_peers(),
+ allow_all_users=_startup_env("A2A_ALLOW_ALL_USERS").lower()
+ in {"1", "true", "yes"},
+ requested_host=_startup_env("A2A_HOST") or "127.0.0.1",
+ push_secret=_startup_env("A2A_PUSH_SECRET") or bearer_token,
+ )
+
+ def localhost_only(self) -> bool:
+ return not (self.bearer_token or self.peer_tokens)
+
+ def resolve_bind_host(self) -> str:
+ loopback = {"127.0.0.1", "localhost", "::1"}
+ if self.requested_host in loopback:
+ return self.requested_host
+ if self.localhost_only():
+ logger.warning(
+ "A2A: A2A_HOST=%s ignored — no A2A_BEARER_TOKEN or "
+ "A2A_PEER_TOKENS set; binding to 127.0.0.1. Configure a token "
+ "to expose A2A remotely.",
+ self.requested_host,
+ )
+ return "127.0.0.1"
+ return self.requested_host
+
+ def authenticate(self, auth_header: Optional[str], client_ip: str = "") -> Optional[str]:
+ if self.localhost_only():
+ return f"ip:{client_ip or 'local'}"
+ presented = _parse_bearer(auth_header)
+ if presented is None:
+ return None
+ for token, name in self.peer_tokens:
+ if hmac.compare_digest(presented, token):
+ return name
+ if self.bearer_token and hmac.compare_digest(presented, self.bearer_token):
+ return f"ip:{client_ip or 'unknown'}"
+ return None
+
+ def is_trusted_peer(self, identity: str) -> bool:
+ if self.allow_all_users or self.localhost_only() or not self.trusted_peers:
+ return True
+ return identity in self.trusted_peers
+
+ def sign_push_payload(self, payload: dict) -> str:
+ if not self.push_secret:
+ return ""
+ body = json.dumps(payload, sort_keys=True, ensure_ascii=False).encode("utf-8")
+ return hmac.new(
+ self.push_secret.encode("utf-8"), body, hashlib.sha256
+ ).hexdigest()
+
+
+# --------------------------------------------------------------------------
+# Bearer auth + peer identity
+# --------------------------------------------------------------------------
+
+def get_bearer_token() -> str:
+ """Return the configured shared inbound bearer token (empty if none)."""
+ return _startup_env("A2A_BEARER_TOKEN")
+
+
+def get_peer_tokens() -> dict[str, str]:
+ """Parse A2A_PEER_TOKENS ("alice:tok1,bob:tok2") into {token: peer_name}.
+
+ Per-peer tokens give each remote agent its own credential, so the identity
+ used for rate limiting, trust, and audit is authenticated — not whatever
+ the request body claims.
+ """
+ return _parse_peer_tokens(_startup_env("A2A_PEER_TOKENS"))
+
+
def _parse_bearer(auth_header: Optional[str]) -> Optional[str]:
if not auth_header:
return None
@@ -85,24 +209,12 @@ def authenticate(auth_header: Optional[str], client_ip: str = "") -> Optional[st
Comparisons are constant-time (hmac.compare_digest).
"""
- peer_tokens = get_peer_tokens()
- shared = get_bearer_token()
- if not peer_tokens and not shared:
- return f"ip:{client_ip or 'local'}"
- presented = _parse_bearer(auth_header)
- if presented is None:
- return None
- for token, name in peer_tokens.items():
- if hmac.compare_digest(presented, token):
- return name
- if shared and hmac.compare_digest(presented, shared):
- return f"ip:{client_ip or 'unknown'}"
- return None
+ return A2ASecurityContext.capture().authenticate(auth_header, client_ip)
def localhost_only() -> bool:
"""True when we must refuse non-loopback binds (no token of any kind set)."""
- return not (get_bearer_token() or get_peer_tokens())
+ return A2ASecurityContext.capture().localhost_only()
def resolve_bind_host() -> str:
@@ -112,18 +224,7 @@ def resolve_bind_host() -> str:
per-peer) AND explicitly asked for a wider host. A token alone does not
widen the bind — opting into remote exposure must be deliberate.
"""
- requested = os.getenv("A2A_HOST", "").strip() or "127.0.0.1"
- loopback = {"127.0.0.1", "localhost", "::1"}
- if requested in loopback:
- return requested
- if localhost_only():
- logger.warning(
- "A2A: A2A_HOST=%s ignored — no A2A_BEARER_TOKEN or A2A_PEER_TOKENS "
- "set; binding to 127.0.0.1. Configure a token to expose A2A remotely.",
- requested,
- )
- return "127.0.0.1"
- return requested
+ return A2ASecurityContext.capture().resolve_bind_host()
# --------------------------------------------------------------------------
@@ -138,18 +239,7 @@ def get_trusted_peers() -> set[str]:
names from ``authenticate()`` — peer-token names, or ``ip:`` for
shared-token callers.
"""
- env_peers = os.getenv("A2A_TRUSTED_PEERS", "").strip()
- if env_peers:
- return {p.strip() for p in env_peers.split(",") if p.strip()}
- try:
- from hermes_cli.config import load_config
- cfg = load_config() or {}
- peers_list = (cfg.get("a2a") or {}).get("trusted_peers", [])
- if isinstance(peers_list, list):
- return {str(p).strip() for p in peers_list if p}
- except Exception:
- pass
- return set()
+ return set(_configured_trusted_peers())
def is_trusted_peer(identity: str) -> bool:
@@ -160,14 +250,7 @@ def is_trusted_peer(identity: str) -> bool:
otherwise any *authenticated* identity is allowed (authentication is the
primary gate — the allow-list is an optional restriction on top).
"""
- if os.getenv("A2A_ALLOW_ALL_USERS", "").strip().lower() in ("1", "true", "yes"):
- return True
- if localhost_only():
- return True
- trusted = get_trusted_peers()
- if not trusted:
- return True
- return identity in trusted
+ return A2ASecurityContext.capture().is_trusted_peer(identity)
# --------------------------------------------------------------------------
@@ -259,10 +342,7 @@ def get_push_secret() -> str:
Falls back to the bearer token if no dedicated push secret is set.
If neither is configured, push notifications are unsigned (localhost-only mode).
"""
- secret = os.getenv("A2A_PUSH_SECRET", "").strip()
- if secret:
- return secret
- return get_bearer_token()
+ return A2ASecurityContext.capture().push_secret
def sign_push_payload(payload: dict) -> str:
@@ -304,12 +384,14 @@ _BLOCKED_PREFIXES = (
)
-def is_safe_callback_url(url: str) -> bool:
+def is_safe_callback_url(url: str, *, localhost_mode: Optional[bool] = None) -> bool:
"""Check if a push notification callback URL is safe from SSRF.
Blocks internal/private/loopback/metadata addresses.
Only allows http:// and https:// schemes.
"""
+ if localhost_mode is None:
+ localhost_mode = localhost_only()
if not url or not isinstance(url, str):
return False
try:
@@ -324,16 +406,16 @@ def is_safe_callback_url(url: str) -> bool:
hostname_lower = hostname.lower()
if hostname_lower == "localhost":
# Loopback callbacks only make sense for local testing.
- return localhost_only()
+ return localhost_mode
for prefix in _BLOCKED_PREFIXES:
if hostname_lower.startswith(prefix.lower()):
- if localhost_only() and prefix in ("127.", "::1"):
+ if localhost_mode and prefix in ("127.", "::1"):
return True
return False
try:
ip = ipaddress.ip_address(hostname)
if ip.is_loopback or ip.is_link_local or ip.is_private or ip.is_reserved:
- if localhost_only() and ip.is_loopback:
+ if localhost_mode and ip.is_loopback:
return True
return False
except ValueError:
diff --git a/plugins/platforms/discord/adapter.py b/plugins/platforms/discord/adapter.py
index 79268849df..1839d57dd0 100644
--- a/plugins/platforms/discord/adapter.py
+++ b/plugins/platforms/discord/adapter.py
@@ -142,6 +142,47 @@ import sys
from pathlib import Path as _Path
sys.path.insert(0, str(_Path(__file__).resolve().parents[3]))
+
+def _is_discord_transport_error(exc: BaseException) -> bool:
+ """Return True for connection-shaped send failures (dead/dropping WS).
+
+ These are the failures where the message demonstrably did NOT reach
+ Discord because the transport itself was down — the delivery-obligation
+ ledger can safely replay them after reconnect (#95382). HTTP-level
+ rejections (permissions, formatting, 4xx) are NOT transport errors and
+ must keep their original error string. Timeouts are excluded: a timed-out
+ send may have reached Discord, so replaying it risks a duplicate.
+ """
+ if isinstance(exc, asyncio.TimeoutError):
+ return False
+ if isinstance(exc, (ConnectionError, OSError)):
+ return True
+ if DISCORD_AVAILABLE and discord is not None:
+ _transport_types = tuple(
+ t
+ for t in (
+ getattr(discord, "ConnectionClosed", None),
+ getattr(discord, "GatewayNotFound", None),
+ getattr(discord, "DiscordServerError", None),
+ )
+ if isinstance(t, type)
+ )
+ if _transport_types and isinstance(exc, _transport_types):
+ return True
+ text = str(exc).lower()
+ return any(
+ marker in text
+ for marker in (
+ "websocket closed",
+ "connection reset",
+ "connection closed",
+ "session is closed",
+ "cannot write to closing transport",
+ "not connected",
+ )
+ )
+
+
try:
from .ffmpeg_utils import resolve_ffmpeg_executable
except ImportError:
@@ -3451,7 +3492,15 @@ class DiscordAdapter(BasePlatformAdapter):
created automatically.
"""
if not self._client:
- return SendResult(success=False, error="Not connected")
+ # Dead transport (client gone / gateway reconnecting): classify as
+ # send_path_degraded so the delivery-obligation ledger's reconnect
+ # sweep (_redeliver_failed_obligations_for_platform) can replay
+ # this final response once the adapter is live again — a generic
+ # "Not connected" error is not runtime-retryable and left the
+ # turn's output stranded until a full process restart (#95382).
+ return SendResult(
+ success=False, error="send_path_degraded", retryable=True
+ )
if not (content or "").strip():
logger.warning(
"[%s] Dropped empty message to chat=%s (caller bug). Call site:\n%s",
@@ -3582,7 +3631,16 @@ class DiscordAdapter(BasePlatformAdapter):
except Exception as e: # pragma: no cover - defensive logging
logger.error("[%s] Failed to send Discord message: %s", self.name, e, exc_info=True)
- result = SendResult(success=False, error=str(e))
+ if _is_discord_transport_error(e):
+ # Connection-shaped failure (WS drop / closed session): use
+ # the ledger's runtime-retryable marker so the reconnect
+ # sweep can replay this final response instead of stranding
+ # it until a process restart (#95382 silent partial loss).
+ result = SendResult(
+ success=False, error="send_path_degraded", retryable=True
+ )
+ else:
+ result = SendResult(success=False, error=str(e))
await asyncio.to_thread(
self._record_discord_response,
reply_to=reply_to,
@@ -6398,6 +6456,14 @@ class DiscordAdapter(BasePlatformAdapter):
)
return (len(self._skill_entries), self._skill_group_hidden_count)
+ def _interaction_guild_id(self, interaction: discord.Interaction) -> Optional[str]:
+ """Resolve the guild id of a slash interaction (mirrors the message path)."""
+ guild_id = getattr(interaction, "guild_id", None)
+ if guild_id is None:
+ guild = getattr(getattr(interaction, "channel", None), "guild", None)
+ guild_id = getattr(guild, "id", None)
+ return str(guild_id) if guild_id else None
+
def _build_slash_event(self, interaction: discord.Interaction, text: str) -> MessageEvent:
"""Build a MessageEvent from a Discord slash command interaction."""
is_dm = isinstance(interaction.channel, discord.DMChannel)
@@ -6422,6 +6488,12 @@ class DiscordAdapter(BasePlatformAdapter):
# For forum threads, inherit the parent forum's topic.
chat_topic = self._get_effective_topic(interaction.channel, is_thread=is_thread)
+ # guild_id/parent_chat_id feed profile_routes matching in build_source,
+ # exactly as on_message passes them — without them a guild- or
+ # channel-routed profile never matches a native slash command (#69178).
+ parent_id = (
+ self._get_parent_channel_id(interaction.channel) if is_thread else None
+ ) or ""
source = self.build_source(
chat_id=str(interaction.channel_id),
chat_name=chat_name,
@@ -6430,11 +6502,12 @@ class DiscordAdapter(BasePlatformAdapter):
user_name=interaction.user.display_name,
thread_id=thread_id,
chat_topic=chat_topic,
+ guild_id=self._interaction_guild_id(interaction),
+ parent_chat_id=parent_id or None,
)
msg_type = MessageType.COMMAND if text.startswith("/") else MessageType.TEXT
channel_id = str(interaction.channel_id)
- parent_id = str(getattr(getattr(interaction, "channel", None), "parent_id", "") or "")
return MessageEvent(
text=text,
message_type=msg_type,
@@ -6516,6 +6589,8 @@ class DiscordAdapter(BasePlatformAdapter):
_chan = getattr(interaction, "channel", None)
chat_topic = self._get_effective_topic(_chan, is_thread=True) if _chan else None
+ _parent_channel = self._thread_parent_channel(getattr(interaction, "channel", None))
+ _parent_id = str(getattr(_parent_channel, "id", "") or "")
source = self.build_source(
chat_id=thread_id,
chat_name=chat_name,
@@ -6524,10 +6599,10 @@ class DiscordAdapter(BasePlatformAdapter):
user_name=interaction.user.display_name,
thread_id=thread_id,
chat_topic=chat_topic,
+ guild_id=self._interaction_guild_id(interaction),
+ parent_chat_id=_parent_id or None,
)
- _parent_channel = self._thread_parent_channel(getattr(interaction, "channel", None))
- _parent_id = str(getattr(_parent_channel, "id", "") or "")
_skills = self._resolve_channel_skills(thread_id, _parent_id or None)
_channel_prompt = self._resolve_channel_prompt(thread_id, _parent_id or None)
event = MessageEvent(
diff --git a/plugins/platforms/email/adapter.py b/plugins/platforms/email/adapter.py
index 89ead8a82a..228cad281f 100644
--- a/plugins/platforms/email/adapter.py
+++ b/plugins/platforms/email/adapter.py
@@ -7,8 +7,12 @@ Uses IMAP to receive and SMTP to send messages.
Environment variables:
EMAIL_IMAP_HOST — IMAP server host (e.g., imap.gmail.com)
EMAIL_IMAP_PORT — IMAP server port (default: 993)
+ EMAIL_IMAP_SECURITY — IMAP transport: tls, starttls, or plain (default: tls)
+ EMAIL_IMAP_TLS_VERIFY — Verify the IMAP TLS certificate (default: true)
EMAIL_SMTP_HOST — SMTP server host (e.g., smtp.gmail.com)
EMAIL_SMTP_PORT — SMTP server port (default: 587)
+ EMAIL_SMTP_SECURITY — SMTP transport: tls, starttls, or plain (port-based default)
+ EMAIL_SMTP_TLS_VERIFY — Verify the SMTP TLS certificate (default: true)
EMAIL_ADDRESS — Email address for the agent
EMAIL_PASSWORD — Email password or app-specific password
EMAIL_POLL_INTERVAL — Seconds between mailbox checks (default: 15)
@@ -88,7 +92,40 @@ def _esecret_int(name: str, default: int) -> int:
def _esecret_bool(name: str, default: bool = False) -> bool:
"""Scope-aware boolean read (``env_bool`` variant of ``_get_esecret``)."""
- return is_truthy_value(_get_esecret(name, ""), default=default)
+ raw = str(_get_esecret(name, "")).strip()
+ return is_truthy_value(raw, default=default) if raw else default
+
+
+_SECURITY_ALIASES = {
+ "tls": "tls", "ssl": "tls", "implicit": "tls",
+ "starttls": "starttls",
+ "plain": "plain", "none": "plain",
+}
+
+
+def _normalize_security(value: Any, default: str = "tls") -> str:
+ """Map an IMAP/SMTP security setting to ``tls`` | ``starttls`` | ``plain``.
+
+ Unknown values log a warning and fall back to *default* rather than
+ failing the connection, so a typo never silently downgrades to plaintext.
+ """
+ raw = str(value or "").strip().lower().replace("-", "").replace("_", "")
+ if not raw:
+ return default
+ mode = _SECURITY_ALIASES.get(raw)
+ if mode is None:
+ logger.warning("Unknown email security mode %r; using %r", value, default)
+ return default
+ return mode
+
+
+def _tls_context(verify: bool, host: str) -> ssl.SSLContext:
+ """Verified context by default; unverified only when explicitly opted out."""
+ if verify:
+ return ssl.create_default_context()
+ if host not in ("127.0.0.1", "::1", "localhost"):
+ logger.warning("TLS verification disabled for non-loopback host %s", host)
+ return ssl._create_unverified_context()
# Automated sender patterns — emails from these are silently ignored
@@ -554,8 +591,23 @@ class EmailAdapter(BasePlatformAdapter):
self._password = _get_secret("EMAIL_PASSWORD", "")
self._imap_host = (_get_secret("EMAIL_IMAP_HOST", "") or extra.get("imap_host", "")).strip()
self._imap_port = _esecret_int("EMAIL_IMAP_PORT", 993)
+ self._imap_security = _normalize_security(
+ _get_secret("EMAIL_IMAP_SECURITY", "") or extra.get("imap_security", "")
+ )
+ self._imap_tls_verify = _esecret_bool(
+ "EMAIL_IMAP_TLS_VERIFY",
+ is_truthy_value(extra.get("imap_tls_verify"), default=True),
+ )
self._smtp_host = (_get_secret("EMAIL_SMTP_HOST", "") or extra.get("smtp_host", "")).strip()
self._smtp_port = _esecret_int("EMAIL_SMTP_PORT", 587)
+ self._smtp_security = _normalize_security(
+ _get_secret("EMAIL_SMTP_SECURITY", "") or extra.get("smtp_security", ""),
+ default="tls" if self._smtp_port == 465 else "starttls",
+ )
+ self._smtp_tls_verify = _esecret_bool(
+ "EMAIL_SMTP_TLS_VERIFY",
+ is_truthy_value(extra.get("smtp_tls_verify"), default=True),
+ )
self._poll_interval = _esecret_int("EMAIL_POLL_INTERVAL", 15)
# Skip attachments — configured via config.yaml:
@@ -627,6 +679,25 @@ class EmailAdapter(BasePlatformAdapter):
# Fallback: just clear old entries if sort fails
self._seen_uids = set(list(self._seen_uids)[-self._seen_uids_max // 2:])
+ def _connect_imap(self) -> imaplib.IMAP4:
+ """Create an IMAP connection using implicit TLS, STARTTLS, or plaintext."""
+ if self._imap_security == "tls":
+ return imaplib.IMAP4_SSL(
+ self._imap_host,
+ self._imap_port,
+ timeout=30,
+ ssl_context=_tls_context(self._imap_tls_verify, self._imap_host),
+ )
+
+ imap = imaplib.IMAP4(self._imap_host, self._imap_port, timeout=30)
+ if self._imap_security == "starttls":
+ try:
+ imap.starttls(ssl_context=_tls_context(self._imap_tls_verify, self._imap_host))
+ except Exception:
+ _close_imap(imap)
+ raise
+ return imap
+
def _connect_smtp(self) -> smtplib.SMTP:
"""Create an SMTP connection, selecting the correct protocol for the port.
@@ -642,22 +713,24 @@ class EmailAdapter(BasePlatformAdapter):
Returns a connected SMTP object with TLS established — callers
can proceed directly to ``login()``.
"""
- ctx = ssl.create_default_context()
host = self._smtp_host
port = self._smtp_port
+ security = self._smtp_security
+ ctx = _tls_context(self._smtp_tls_verify, host)
def _connect(*, ipv4_only: bool = False) -> smtplib.SMTP:
"""Attempt one SMTP connection."""
smtp_cls = _IPv4SMTP if ipv4_only else smtplib.SMTP
smtp_ssl_cls = _IPv4SMTP_SSL if ipv4_only else smtplib.SMTP_SSL
- if port == 465:
+ if security == "tls":
return smtp_ssl_cls(host, port, timeout=SMTP_CONNECT_TIMEOUT, context=ctx)
smtp = smtp_cls(host, port, timeout=SMTP_CONNECT_TIMEOUT)
- try:
- smtp.starttls(context=ctx)
- except Exception:
- smtp.close()
- raise
+ if security == "starttls":
+ try:
+ smtp.starttls(context=ctx)
+ except Exception:
+ smtp.close()
+ raise
return smtp
try:
@@ -711,7 +784,7 @@ class EmailAdapter(BasePlatformAdapter):
# (#79889).
imap = None
try:
- imap = imaplib.IMAP4_SSL(self._imap_host, self._imap_port, timeout=30)
+ imap = self._connect_imap()
imap.login(self._address, self._password)
_send_imap_id(imap)
imap.select("INBOX")
@@ -855,7 +928,7 @@ class EmailAdapter(BasePlatformAdapter):
results = []
imap: Optional[imaplib.IMAP4] = None
try:
- imap = imaplib.IMAP4_SSL(self._imap_host, self._imap_port, timeout=30)
+ imap = self._connect_imap()
try:
imap.login(self._address, self._password)
_send_imap_id(imap)
@@ -1438,7 +1511,6 @@ async def _standalone_send(
"""Out-of-process Email delivery via SMTP (one-shot). Implements the
standalone_sender_fn contract; replaces the legacy _send_email helper."""
import smtplib
- import ssl as _ssl
from email.mime.text import MIMEText
from email.utils import formatdate
@@ -1450,6 +1522,14 @@ async def _standalone_send(
smtp_port = int(_get_secret("EMAIL_SMTP_PORT", "587") or "587")
except (ValueError, TypeError):
smtp_port = 587
+ smtp_security = _normalize_security(
+ _get_secret("EMAIL_SMTP_SECURITY", "") or extra.get("smtp_security"),
+ default="tls" if smtp_port == 465 else "starttls",
+ )
+ smtp_tls_verify = _esecret_bool(
+ "EMAIL_SMTP_TLS_VERIFY",
+ is_truthy_value(extra.get("smtp_tls_verify"), default=True),
+ )
if not all([address, password, smtp_host]):
return {"error": "Email not configured (EMAIL_ADDRESS, EMAIL_PASSWORD, EMAIL_SMTP_HOST required)"}
@@ -1461,8 +1541,17 @@ async def _standalone_send(
msg["Subject"] = "Hermes Agent"
msg["Date"] = formatdate(localtime=True)
- server = smtplib.SMTP(smtp_host, smtp_port)
- server.starttls(context=_ssl.create_default_context())
+ ctx = _tls_context(smtp_tls_verify, smtp_host)
+ if smtp_security == "tls":
+ server = smtplib.SMTP_SSL(smtp_host, smtp_port, context=ctx)
+ else:
+ server = smtplib.SMTP(smtp_host, smtp_port)
+ if smtp_security == "starttls":
+ try:
+ server.starttls(context=ctx)
+ except Exception:
+ server.close()
+ raise
server.login(address, password)
server.send_message(msg)
server.quit()
diff --git a/plugins/platforms/feishu/adapter.py b/plugins/platforms/feishu/adapter.py
index 153f2d4d0c..d0546f6339 100644
--- a/plugins/platforms/feishu/adapter.py
+++ b/plugins/platforms/feishu/adapter.py
@@ -436,6 +436,9 @@ class FeishuAdapterSettings:
group_rules: Dict[str, FeishuGroupRule] = field(default_factory=dict)
allow_bots: str = "none" # "none" | "mentions" | "all"
require_mention: bool = True
+ # DM allow-all (FEISHU_ALLOW_ALL_USERS / GATEWAY_ALLOW_ALL_USERS), resolved
+ # per-profile so multiplexed secondary adapters honor their own .env.
+ allow_all_dm: bool = False
@dataclass
@@ -1322,16 +1325,96 @@ def _strip_edge_self_mentions(
return remaining
+# ---------------------------------------------------------------------------
+# Multiplex isolation for the lark_oapi WebSocket client (#73779)
+# ---------------------------------------------------------------------------
+#
+# ``lark_oapi.ws.client`` keeps the asyncio loop used by ``Client.start()``
+# and every coroutine it spawns in a *module-level global* (``loop``), and
+# Hermes also monkey-patches ``websockets.connect`` on the shared
+# ``websockets`` module to inject per-adapter ping settings. In multiplex
+# mode every profile runs its own WS client on a dedicated thread, so the N
+# threads overwrite each other's module globals (last-write-wins): a client
+# ends up scheduling tasks on a sibling profile's loop ("Future attached to
+# a different loop" crashes) or binds to the wrong loop at construction time
+# and goes deaf from the start.
+#
+# The fix installs process-wide, thread-dispatching shims exactly once:
+#
+# * ``ws_client_module.loop`` becomes a proxy that forwards every attribute
+# access to the loop registered by the *current thread*. All SDK reads of
+# the global happen on the thread that owns the loop (``start()`` blocks
+# in ``run_until_complete`` and every ``create_task`` callback runs on
+# the loop's own thread), so each profile transparently sees its own
+# loop. Threads that never registered one (single-profile installs, CLI)
+# fall back to the SDK's original module loop.
+# * ``websockets.connect`` becomes a single dispatcher that merges the
+# per-thread ping overrides registered by the calling profile, so
+# profiles no longer race over the global patch or restore each other's
+# hooks while a sibling is still connected.
+
+_WS_ISOLATION_LOCK = threading.Lock()
+_WS_ISOLATION_INSTALLED = False
+# Per-WS-thread registration: ``.loop`` (the thread's asyncio loop) and
+# ``.connect_kwargs`` (websockets.connect overrides, e.g. ping settings).
+_ws_isolation_state = threading.local()
+
+
+class _ThreadLocalLoopProxy:
+ """Forwards attribute access to the current thread's registered loop."""
+
+ def __init__(self, fallback: Any) -> None:
+ self._fallback = fallback
+
+ def _target(self) -> Any:
+ return getattr(_ws_isolation_state, "loop", None) or self._fallback
+
+ def __getattr__(self, name: str) -> Any:
+ return getattr(self._target(), name)
+
+ def __repr__(self) -> str: # pragma: no cover - debugging aid
+ return f""
+
+
+def _install_lark_ws_isolation(ws_client_module: Any) -> None:
+ """Install the thread-dispatching shims once per process (idempotent)."""
+ global _WS_ISOLATION_INSTALLED
+ with _WS_ISOLATION_LOCK:
+ if _WS_ISOLATION_INSTALLED:
+ return
+
+ ws_client_module.loop = _ThreadLocalLoopProxy(ws_client_module.loop)
+
+ real_connect = ws_client_module.websockets.connect
+
+ def _dispatch_connect(*args: Any, **kwargs: Any) -> Any:
+ overrides = getattr(_ws_isolation_state, "connect_kwargs", None) or {}
+ for key, value in overrides.items():
+ kwargs.setdefault(key, value)
+ return real_connect(*args, **kwargs)
+
+ # Keep ``inspect.signature(websockets.connect)`` honest: the SDK's
+ # ``_ws_connect_kwargs()`` probes the real signature to decide whether
+ # the installed websockets generation supports the ``proxy`` kwarg.
+ _dispatch_connect.__wrapped__ = real_connect
+ _dispatch_connect.__name__ = getattr(real_connect, "__name__", "connect")
+ ws_client_module.websockets.connect = _dispatch_connect
+ _WS_ISOLATION_INSTALLED = True
+
+
def _run_official_feishu_ws_client(ws_client: Any, adapter: Any) -> None:
- """Run the official Lark WS client in its own thread-local event loop."""
+ """Run the official Lark WS client in its own thread-local event loop.
+
+ In multiplex mode several profiles run this concurrently; the shims
+ installed by ``_install_lark_ws_isolation`` make each thread see its own
+ loop and connect overrides (see the isolation comment block above).
+ """
import lark_oapi.ws.client as ws_client_module
loop = asyncio.new_event_loop()
asyncio.set_event_loop(loop)
- ws_client_module.loop = loop
adapter._ws_thread_loop = loop
- original_connect = ws_client_module.websockets.connect
original_configure = getattr(ws_client, "_configure", None)
def _apply_runtime_ws_overrides() -> None:
@@ -1343,12 +1426,15 @@ def _run_official_feishu_ws_client(ws_client: Any, adapter: Any) -> None:
except Exception:
logger.debug("[Feishu] Failed to apply websocket runtime overrides", exc_info=True)
- def _connect_with_overrides(*args: Any, **kwargs: Any) -> Any:
- if adapter._ws_ping_interval is not None and "ping_interval" not in kwargs:
- kwargs["ping_interval"] = adapter._ws_ping_interval
- if adapter._ws_ping_timeout is not None and "ping_timeout" not in kwargs:
- kwargs["ping_timeout"] = adapter._ws_ping_timeout
- return original_connect(*args, **kwargs)
+ connect_overrides: Dict[str, Any] = {}
+ if adapter._ws_ping_interval is not None:
+ connect_overrides["ping_interval"] = adapter._ws_ping_interval
+ if adapter._ws_ping_timeout is not None:
+ connect_overrides["ping_timeout"] = adapter._ws_ping_timeout
+
+ _install_lark_ws_isolation(ws_client_module)
+ _ws_isolation_state.loop = loop
+ _ws_isolation_state.connect_kwargs = connect_overrides
def _configure_with_overrides(conf: Any) -> Any:
if original_configure is None:
@@ -1357,7 +1443,6 @@ def _run_official_feishu_ws_client(ws_client: Any, adapter: Any) -> None:
_apply_runtime_ws_overrides()
return result
- ws_client_module.websockets.connect = _connect_with_overrides
if original_configure is not None:
setattr(ws_client, "_configure", _configure_with_overrides)
_apply_runtime_ws_overrides()
@@ -1366,7 +1451,8 @@ def _run_official_feishu_ws_client(ws_client: Any, adapter: Any) -> None:
except Exception:
pass
finally:
- ws_client_module.websockets.connect = original_connect
+ _ws_isolation_state.loop = None
+ _ws_isolation_state.connect_kwargs = None
if original_configure is not None:
setattr(ws_client, "_configure", original_configure)
pending = [t for t in asyncio.all_tasks(loop) if not t.done()]
@@ -1517,6 +1603,8 @@ class FeishuAdapter(BasePlatformAdapter):
self._sdk_executor_closing = False
self._ws_client: Optional[Any] = None
self._ws_future: Optional[asyncio.Future] = None
+ self._ws_supervisor: Optional[asyncio.Task] = None
+ self._ws_restart_backoff = 5.0
self._ws_thread_loop: Optional[asyncio.AbstractEventLoop] = None
self._loop: Optional[asyncio.AbstractEventLoop] = None
self._webhook_runner: Optional[Any] = None
@@ -1591,7 +1679,9 @@ class FeishuAdapter(BasePlatformAdapter):
# Env-only so adapter and gateway auth bypass share one source; yaml
# feishu.allow_bots is bridged to this env var at config load.
- allow_bots = os.getenv("FEISHU_ALLOW_BOTS", "none").strip().lower()
+ # Scope-aware read: under multiplex a secondary profile's .env must
+ # govern its own adapter (same pattern as app_secret below) — #86905.
+ allow_bots = _get_scoped_secret("FEISHU_ALLOW_BOTS", "none").strip().lower()
if allow_bots not in {"none", "mentions", "all"}:
logger.warning(
"[Feishu] Unknown allow_bots=%r, falling back to 'none'. Valid: none, mentions, all.",
@@ -1599,8 +1689,13 @@ class FeishuAdapter(BasePlatformAdapter):
)
allow_bots = "none"
+ allow_all_dm = any(
+ _get_scoped_secret(var, "").strip().lower() in {"true", "1", "yes"}
+ for var in ("FEISHU_ALLOW_ALL_USERS", "GATEWAY_ALLOW_ALL_USERS")
+ )
+
return FeishuAdapterSettings(
- app_id=str(extra.get("app_id") or os.getenv("FEISHU_APP_ID", "")).strip(),
+ app_id=str(extra.get("app_id") or _get_scoped_secret("FEISHU_APP_ID", "")).strip(),
app_secret=str(extra.get("app_secret") or _get_scoped_secret("FEISHU_APP_SECRET", "")).strip(),
domain_name=str(extra.get("domain") or os.getenv("FEISHU_DOMAIN", "feishu")).strip().lower(),
connection_mode=str(
@@ -1610,15 +1705,15 @@ class FeishuAdapter(BasePlatformAdapter):
verification_token=str(
extra.get("verification_token") or _get_scoped_secret("FEISHU_VERIFICATION_TOKEN", "")
).strip(),
- group_policy=os.getenv("FEISHU_GROUP_POLICY", "allowlist").strip().lower(),
+ group_policy=_get_scoped_secret("FEISHU_GROUP_POLICY", "allowlist").strip().lower(),
allowed_group_users=frozenset(
item.strip()
- for item in os.getenv("FEISHU_ALLOWED_USERS", "").split(",")
+ for item in _get_scoped_secret("FEISHU_ALLOWED_USERS", "").split(",")
if item.strip()
),
- bot_open_id=os.getenv("FEISHU_BOT_OPEN_ID", "").strip(),
- bot_user_id=os.getenv("FEISHU_BOT_USER_ID", "").strip(),
- bot_name=os.getenv("FEISHU_BOT_NAME", "").strip(),
+ bot_open_id=_get_scoped_secret("FEISHU_BOT_OPEN_ID", "").strip(),
+ bot_user_id=_get_scoped_secret("FEISHU_BOT_USER_ID", "").strip(),
+ bot_name=_get_scoped_secret("FEISHU_BOT_NAME", "").strip(),
dedup_cache_size=max(
32,
env_int("HERMES_FEISHU_DEDUP_CACHE_SIZE", _DEFAULT_DEDUP_CACHE_SIZE),
@@ -1658,8 +1753,9 @@ class FeishuAdapter(BasePlatformAdapter):
default_group_policy=default_group_policy,
group_rules=group_rules,
allow_bots=allow_bots,
+ allow_all_dm=allow_all_dm,
require_mention=_to_boolean(
- extra.get("require_mention", os.getenv("FEISHU_REQUIRE_MENTION", "true"))
+ extra.get("require_mention", _get_scoped_secret("FEISHU_REQUIRE_MENTION", "true"))
),
)
@@ -1692,6 +1788,7 @@ class FeishuAdapter(BasePlatformAdapter):
self._ws_ping_interval = settings.ws_ping_interval
self._ws_ping_timeout = settings.ws_ping_timeout
self._allow_bots = settings.allow_bots
+ self._allow_all_dm = settings.allow_all_dm
self._require_mention = settings.require_mention
def _build_event_handler(self) -> Any:
@@ -1814,6 +1911,13 @@ class FeishuAdapter(BasePlatformAdapter):
self._loop = asyncio.get_running_loop()
await self._connect_with_retry()
+ if self._connection_mode == "websocket":
+ # Supervised reconnect (#73779): the WS thread can die without
+ # any external signal; keep a watcher alive for as long as this
+ # adapter is supposed to be connected.
+ self._ws_supervisor = asyncio.ensure_future(
+ self._supervise_websocket_thread()
+ )
self._mark_connected()
logger.info("[Feishu] Connected in %s mode (%s)", self._connection_mode, self._domain_name)
# Plugin-registered native handlers (lark_oapi client).
@@ -1829,6 +1933,9 @@ class FeishuAdapter(BasePlatformAdapter):
async def disconnect(self) -> None:
"""Disconnect from Feishu/Lark."""
self._running = False
+ if self._ws_supervisor is not None:
+ self._ws_supervisor.cancel()
+ self._ws_supervisor = None
await self._cancel_pending_tasks(self._pending_text_batch_tasks)
await self._cancel_pending_tasks(self._pending_media_batch_tasks)
self._reset_batch_buffers()
@@ -4390,9 +4497,10 @@ class FeishuAdapter(BasePlatformAdapter):
return "bot_not_mentioned"
if not is_group:
- if os.getenv("FEISHU_ALLOW_ALL_USERS", "").strip().lower() in {"true", "1", "yes"}:
- return None
- if os.getenv("GATEWAY_ALLOW_ALL_USERS", "").strip().lower() in {"true", "1", "yes"}:
+ # Snapshotted per-profile in _load_settings: _admit runs on the
+ # lark_oapi WS thread with no secret scope, and a bare os.getenv
+ # here would read the default profile's value (#86905).
+ if self._allow_all_dm:
return None
# Empty FEISHU_ALLOWED_USERS is the pairing-mode default from setup:
# forward DMs to gateway intake so the pairing handshake can run.
@@ -4946,6 +5054,52 @@ class FeishuAdapter(BasePlatformAdapter):
)
await asyncio.sleep(wait_seconds)
+ async def _supervise_websocket_thread(self) -> None:
+ """Restart the WS client thread if it dies while the adapter is up.
+
+ ``lark_oapi``'s ``start()`` blocks forever on a healthy connection
+ and only returns on fatal errors. Before this watcher existed the
+ executor future was awaited solely by ``disconnect()``, so a dead
+ thread left the profile silently deaf until a gateway restart
+ (#73779). Watch the future and, on unexpected exit, rebuild the
+ client with capped exponential backoff.
+ """
+ backoff = initial_backoff = float(self._ws_restart_backoff)
+ last_dead: Optional[asyncio.Future] = None
+ while self._running:
+ ws_future = self._ws_future
+ if ws_future is None:
+ return
+ try:
+ await asyncio.shield(ws_future)
+ except asyncio.CancelledError:
+ raise
+ except Exception:
+ pass
+ # Deliberate disconnect paths nil ``_ws_client`` / ``_running``
+ # before the thread exits; only restart when the link is still
+ # expected to be up.
+ if not self._running or self._ws_client is None:
+ return
+ if ws_future is not last_dead:
+ logger.error(
+ "[Feishu] WebSocket client thread exited unexpectedly; "
+ "restarting in %.0fs",
+ backoff,
+ )
+ last_dead = ws_future
+ await asyncio.sleep(backoff)
+ if not self._running:
+ return
+ try:
+ await self._connect_websocket()
+ backoff = initial_backoff
+ except Exception as exc:
+ logger.warning(
+ "[Feishu] WebSocket restart failed (retrying): %s", exc
+ )
+ backoff = min(backoff * 2, 60.0)
+
async def _connect_websocket(self) -> None:
if not FEISHU_WEBSOCKET_AVAILABLE:
raise RuntimeError("websockets not installed; websocket mode unavailable")
diff --git a/plugins/platforms/google_chat/adapter.py b/plugins/platforms/google_chat/adapter.py
index 41c9b65500..7be34ceee2 100644
--- a/plugins/platforms/google_chat/adapter.py
+++ b/plugins/platforms/google_chat/adapter.py
@@ -48,6 +48,45 @@ import time
from pathlib import Path as _Path
from typing import Any, Callable, Dict, List, Optional, Tuple
+from agent.secret_scope import UnscopedSecretError as _UnscopedSecretError
+from agent.secret_scope import get_secret as _scoped_get_secret
+from agent.secret_scope import is_multiplex_active
+
+
+def _get_scoped_secret(name: str, default: Optional[str] = None) -> Optional[str]:
+ """Scope-aware config/credential read with the default-profile fallback.
+
+ Secondary profiles construct their adapters under a profile secret
+ scope -- the scope is authoritative and a scoped miss returns ``default``
+ (no cross-profile borrow from ``os.environ``, which may hold another
+ profile's value). The DEFAULT profile's adapter constructs and connects
+ *unscoped* under multiplexing, where a bare ``get_secret`` would raise
+ ``UnscopedSecretError`` and crash startup/reconnect (#70652 class); there
+ ``os.environ`` is that profile's own value, so fall back to it. Same
+ pattern as ``whatsapp_common._get_wsecret`` and the WeCom/IRC/ntfy
+ plugin adapters.
+ """
+ try:
+ val = _scoped_get_secret(name, default)
+ except _UnscopedSecretError:
+ val = os.getenv(name)
+ return val if val is not None else default
+
+
+def _adc_would_borrow_foreign_credentials() -> bool:
+ """True when ADC would silently read another profile's SA from process env.
+
+ ``google.auth.default()`` consults ``os.environ`` directly. Under
+ multiplexing a scoped profile only reaches the ADC branch after its own
+ scope had no service-account setting -- if the process env still carries
+ one (the default profile's), ADC would authenticate this profile as that
+ other identity. Fail closed instead.
+ """
+ return is_multiplex_active() and bool(
+ os.environ.get("GOOGLE_CHAT_SERVICE_ACCOUNT_JSON")
+ or os.environ.get("GOOGLE_APPLICATION_CREDENTIALS")
+ )
+
# Heavy google-cloud + googleapiclient imports are deferred to first
# adapter use. Importing them eagerly here added ~110ms wall and ~33MB
# RSS to *every* CLI invocation (the plugin loader imports this module at
@@ -184,6 +223,7 @@ from gateway.config import Platform, PlatformConfig
Platform("google_chat")
from gateway.platforms.helpers import MessageDeduplicator
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -736,28 +776,48 @@ class GoogleChatAdapter(BasePlatformAdapter):
# end-of-turn by on_processing_complete via patch-to-empty so
# they don't sit in the chat forever as "Hermes is thinking…".
self._orphan_typing_messages: Dict[str, List[str]] = {}
- # FlowControl knobs (env-configurable).
+ # Snapshot profile-scoped settings while adapter construction still
+ # runs inside _profile_runtime_scope. Pub/Sub invokes callbacks from
+ # its own threads, where the ContextVar secret scope is intentionally
+ # unavailable; callbacks must use these instance values rather than
+ # consulting process-global environment state.
+ extra = self.config.extra
try:
- self._max_messages = int(os.getenv("GOOGLE_CHAT_MAX_MESSAGES", "1"))
+ self._max_messages = int(
+ extra.get("max_messages")
+ or _get_scoped_secret("GOOGLE_CHAT_MAX_MESSAGES", "1")
+ )
except (ValueError, TypeError):
self._max_messages = 1
try:
- self._max_bytes = int(os.getenv("GOOGLE_CHAT_MAX_BYTES", str(16 * 1024 * 1024)))
+ self._max_bytes = int(
+ extra.get("max_bytes")
+ or _get_scoped_secret("GOOGLE_CHAT_MAX_BYTES", str(16 * 1024 * 1024))
+ )
except (ValueError, TypeError):
self._max_bytes = 16 * 1024 * 1024
+ self._bootstrap_spaces = str(
+ extra.get("bootstrap_spaces")
+ or _get_scoped_secret("GOOGLE_CHAT_BOOTSTRAP_SPACES", "")
+ or ""
+ ).strip()
+ self._debug_raw = bool(
+ extra.get("debug_raw")
+ or _get_scoped_secret("GOOGLE_CHAT_DEBUG_RAW")
+ )
self._http_events_url = (
- self.config.extra.get("http_events_url")
- or os.getenv("GOOGLE_CHAT_HTTP_EVENTS_URL", "")
+ extra.get("http_events_url")
+ or _get_scoped_secret("GOOGLE_CHAT_HTTP_EVENTS_URL", "")
or ""
).strip()
self._http_events_audience = (
- self.config.extra.get("http_events_audience")
- or os.getenv("GOOGLE_CHAT_HTTP_EVENTS_AUDIENCE", "")
+ extra.get("http_events_audience")
+ or _get_scoped_secret("GOOGLE_CHAT_HTTP_EVENTS_AUDIENCE", "")
or self._http_events_url
).strip()
self._http_events_service_account_email = (
- self.config.extra.get("http_events_service_account_email")
- or os.getenv("GOOGLE_CHAT_HTTP_EVENTS_SERVICE_ACCOUNT_EMAIL", "")
+ extra.get("http_events_service_account_email")
+ or _get_scoped_secret("GOOGLE_CHAT_HTTP_EVENTS_SERVICE_ACCOUNT_EMAIL", "")
or ""
).strip().lower()
@@ -779,7 +839,7 @@ class GoogleChatAdapter(BasePlatformAdapter):
"""
sa_path = (
self.config.extra.get("service_account_json")
- or os.getenv("GOOGLE_APPLICATION_CREDENTIALS")
+ or _get_scoped_secret("GOOGLE_APPLICATION_CREDENTIALS")
)
if sa_path:
# Inline JSON (rare, but supported).
@@ -811,6 +871,13 @@ class GoogleChatAdapter(BasePlatformAdapter):
# No explicit SA configured — try ADC. This is the Cloud Run / GCE
# path; google-auth picks up the workload identity automatically.
+ if _adc_would_borrow_foreign_credentials():
+ raise ValueError(
+ "Google Chat ADC skipped for this profile: service-account "
+ "credentials are set in the process environment but not in "
+ "this profile's secret scope. Set "
+ "GOOGLE_CHAT_SERVICE_ACCOUNT_JSON in this profile's .env."
+ )
try:
import google.auth as google_auth
except ImportError:
@@ -916,8 +983,12 @@ class GoogleChatAdapter(BasePlatformAdapter):
# ------------------------------------------------------------------
def _bot_id_cache_path(self) -> _Path:
"""Location where the resolved bot user_id is cached across restarts."""
- base = os.getenv("HERMES_HOME", str(_Path.home() / ".hermes"))
- return _Path(base) / "google_chat_bot_id.json"
+ # Resolve at call time (connect() runs inside the profile scope) so
+ # multiplexed profiles do not share one bot-identity cache file; the
+ # thread-count store above already resolves the same way.
+ from hermes_constants import get_hermes_home as _get_hermes_home
+
+ return _get_hermes_home() / "google_chat_bot_id.json"
def _load_cached_bot_id(self) -> Optional[str]:
path = self._bot_id_cache_path()
@@ -952,7 +1023,7 @@ class GoogleChatAdapter(BasePlatformAdapter):
if self.config.home_channel and self.config.home_channel.chat_id:
candidate_spaces.append(self.config.home_channel.chat_id)
# Env-configured allowed spaces (comma-separated). Optional.
- extra_spaces = os.getenv("GOOGLE_CHAT_BOOTSTRAP_SPACES", "").strip()
+ extra_spaces = self._bootstrap_spaces
if extra_spaces:
candidate_spaces.extend(
s.strip() for s in extra_spaces.split(",") if s.strip()
@@ -1401,7 +1472,7 @@ class GoogleChatAdapter(BasePlatformAdapter):
list(envelope.keys()),
ce_type,
)
- if os.getenv("GOOGLE_CHAT_DEBUG_RAW"):
+ if self._debug_raw:
# Dangerous flag: contains message text and sender email. Route
# through the global redaction filter and gate at DEBUG level so
# default log configurations never surface it. Operators must
@@ -3362,14 +3433,14 @@ def _check_for_registry() -> bool:
if not check_google_chat_requirements():
return False
project = (
- os.getenv("GOOGLE_CHAT_PROJECT_ID")
- or os.getenv("GOOGLE_CLOUD_PROJECT")
+ _get_scoped_secret("GOOGLE_CHAT_PROJECT_ID")
+ or _get_scoped_secret("GOOGLE_CLOUD_PROJECT")
)
subscription = (
- os.getenv("GOOGLE_CHAT_SUBSCRIPTION_NAME")
- or os.getenv("GOOGLE_CHAT_SUBSCRIPTION")
+ _get_scoped_secret("GOOGLE_CHAT_SUBSCRIPTION_NAME")
+ or _get_scoped_secret("GOOGLE_CHAT_SUBSCRIPTION")
)
- http_events_url = os.getenv("GOOGLE_CHAT_HTTP_EVENTS_URL")
+ http_events_url = _get_scoped_secret("GOOGLE_CHAT_HTTP_EVENTS_URL")
return bool(http_events_url or (project and subscription))
@@ -3393,14 +3464,14 @@ def _env_enablement() -> Optional[Dict[str, Any]]:
``PlatformConfig`` rather than being merged into ``extra``.
"""
project = (
- os.getenv("GOOGLE_CHAT_PROJECT_ID")
- or os.getenv("GOOGLE_CLOUD_PROJECT")
+ _get_scoped_secret("GOOGLE_CHAT_PROJECT_ID")
+ or _get_scoped_secret("GOOGLE_CLOUD_PROJECT")
)
subscription = (
- os.getenv("GOOGLE_CHAT_SUBSCRIPTION_NAME")
- or os.getenv("GOOGLE_CHAT_SUBSCRIPTION")
+ _get_scoped_secret("GOOGLE_CHAT_SUBSCRIPTION_NAME")
+ or _get_scoped_secret("GOOGLE_CHAT_SUBSCRIPTION")
)
- http_events_url = os.getenv("GOOGLE_CHAT_HTTP_EVENTS_URL")
+ http_events_url = _get_scoped_secret("GOOGLE_CHAT_HTTP_EVENTS_URL")
if not (http_events_url or (project and subscription)):
return None
seed: Dict[str, Any] = {}
@@ -3410,23 +3481,32 @@ def _env_enablement() -> Optional[Dict[str, Any]]:
seed["subscription_name"] = subscription
if http_events_url:
seed["http_events_url"] = http_events_url
- http_events_audience = os.getenv("GOOGLE_CHAT_HTTP_EVENTS_AUDIENCE")
+ http_events_audience = _get_scoped_secret("GOOGLE_CHAT_HTTP_EVENTS_AUDIENCE")
if http_events_audience:
seed["http_events_audience"] = http_events_audience
- http_events_sa_email = os.getenv("GOOGLE_CHAT_HTTP_EVENTS_SERVICE_ACCOUNT_EMAIL")
+ http_events_sa_email = _get_scoped_secret("GOOGLE_CHAT_HTTP_EVENTS_SERVICE_ACCOUNT_EMAIL")
if http_events_sa_email:
seed["http_events_service_account_email"] = http_events_sa_email
+ for env_name, extra_name in (
+ ("GOOGLE_CHAT_MAX_MESSAGES", "max_messages"),
+ ("GOOGLE_CHAT_MAX_BYTES", "max_bytes"),
+ ("GOOGLE_CHAT_BOOTSTRAP_SPACES", "bootstrap_spaces"),
+ ("GOOGLE_CHAT_DEBUG_RAW", "debug_raw"),
+ ):
+ value = _get_scoped_secret(env_name)
+ if value:
+ seed[extra_name] = value
sa_json = (
- os.getenv("GOOGLE_CHAT_SERVICE_ACCOUNT_JSON")
- or os.getenv("GOOGLE_APPLICATION_CREDENTIALS")
+ _get_scoped_secret("GOOGLE_CHAT_SERVICE_ACCOUNT_JSON")
+ or _get_scoped_secret("GOOGLE_APPLICATION_CREDENTIALS")
)
if sa_json:
seed["service_account_json"] = sa_json
- home = os.getenv("GOOGLE_CHAT_HOME_CHANNEL")
+ home = _get_scoped_secret("GOOGLE_CHAT_HOME_CHANNEL")
if home:
seed["home_channel"] = {
"chat_id": home,
- "name": os.getenv("GOOGLE_CHAT_HOME_CHANNEL_NAME", "Home"),
+ "name": _get_scoped_secret("GOOGLE_CHAT_HOME_CHANNEL_NAME", "Home"),
}
return seed
@@ -3576,8 +3656,8 @@ async def _standalone_send(
extra = getattr(pconfig, "extra", {}) or {}
sa_value = (
extra.get("service_account_json")
- or os.getenv("GOOGLE_CHAT_SERVICE_ACCOUNT_JSON")
- or os.getenv("GOOGLE_APPLICATION_CREDENTIALS")
+ or _get_scoped_secret("GOOGLE_CHAT_SERVICE_ACCOUNT_JSON")
+ or _get_scoped_secret("GOOGLE_APPLICATION_CREDENTIALS")
)
if service_account is None:
@@ -3607,6 +3687,12 @@ async def _standalone_send(
return {"error": f"Google Chat standalone send: SA JSON file is invalid: {exc}"}
creds = service_account.Credentials.from_service_account_info(info, scopes=_CHAT_SCOPES)
else:
+ if _adc_would_borrow_foreign_credentials():
+ return {"error": (
+ "Google Chat standalone send: ADC skipped for this profile: "
+ "service-account credentials are set in the process environment "
+ "but not in this profile's secret scope"
+ )}
try:
import google.auth as _google_auth
except ImportError:
@@ -3655,7 +3741,7 @@ async def _standalone_send(
return {"error": "Google Chat standalone send: aiohttp not installed"}
try:
- async with _aiohttp.ClientSession(timeout=_aiohttp.ClientTimeout(total=30.0), trust_env=True) as session:
+ async with _aiohttp.ClientSession(timeout=_aiohttp.ClientTimeout(total=30.0), trust_env=gateway_trust_env()) as session:
async with session.post(
url,
json=body,
diff --git a/plugins/platforms/homeassistant/adapter.py b/plugins/platforms/homeassistant/adapter.py
index 37a7397d4b..bfdd136cdf 100644
--- a/plugins/platforms/homeassistant/adapter.py
+++ b/plugins/platforms/homeassistant/adapter.py
@@ -30,6 +30,7 @@ except ImportError:
from gateway.config import Platform, PlatformConfig
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -141,7 +142,8 @@ class HomeAssistantAdapter(BasePlatformAdapter):
# Dedicated REST session for send() calls
self._rest_session = aiohttp.ClientSession(
- timeout=aiohttp.ClientTimeout(total=30)
+ timeout=aiohttp.ClientTimeout(total=30),
+ trust_env=gateway_trust_env(),
)
# Warn if no event filters are configured
@@ -171,7 +173,8 @@ class HomeAssistantAdapter(BasePlatformAdapter):
ws_url = f"{ws_url}/api/websocket"
self._session = aiohttp.ClientSession(
- timeout=aiohttp.ClientTimeout(total=30)
+ timeout=aiohttp.ClientTimeout(total=30),
+ trust_env=gateway_trust_env(),
)
self._ws = await self._session.ws_connect(ws_url, heartbeat=30, timeout=30)
@@ -447,7 +450,7 @@ class HomeAssistantAdapter(BasePlatformAdapter):
body = await resp.text()
return SendResult(success=False, error=f"HTTP {resp.status}: {body}")
else:
- async with aiohttp.ClientSession() as session:
+ async with aiohttp.ClientSession(trust_env=gateway_trust_env()) as session:
async with session.post(
url,
headers=headers,
@@ -532,7 +535,8 @@ async def _standalone_send(
try:
async with aiohttp.ClientSession(
- timeout=aiohttp.ClientTimeout(total=30)
+ timeout=aiohttp.ClientTimeout(total=30),
+ trust_env=gateway_trust_env(),
) as session:
async with session.post(url, headers=headers, json=payload) as resp:
if resp.status not in {200, 201}:
diff --git a/plugins/platforms/irc/adapter.py b/plugins/platforms/irc/adapter.py
index ce3ec4ed59..8afce0ff19 100644
--- a/plugins/platforms/irc/adapter.py
+++ b/plugins/platforms/irc/adapter.py
@@ -130,16 +130,17 @@ class IRCAdapter(BasePlatformAdapter):
extra = getattr(config, "extra", {}) or {}
# Connection settings (env vars override config.yaml)
- self.server = os.getenv("IRC_SERVER") or extra.get("server", "")
+ self.server = _get_scoped_secret("IRC_SERVER") or extra.get("server", "")
try:
- self.port = int(os.getenv("IRC_PORT") or extra.get("port", 6697))
+ self.port = int(_get_scoped_secret("IRC_PORT") or extra.get("port", 6697))
except (ValueError, TypeError):
self.port = 6697
- self.nickname = os.getenv("IRC_NICKNAME") or extra.get("nickname", "hermes-bot")
- self.channel = os.getenv("IRC_CHANNEL") or extra.get("channel", "")
+ self.nickname = _get_scoped_secret("IRC_NICKNAME") or extra.get("nickname", "hermes-bot")
+ self.channel = _get_scoped_secret("IRC_CHANNEL") or extra.get("channel", "")
+ _use_tls_raw = _get_scoped_secret("IRC_USE_TLS")
self.use_tls = (
- os.getenv("IRC_USE_TLS", "").lower() in {"1", "true", "yes"}
- if os.getenv("IRC_USE_TLS")
+ _use_tls_raw.lower() in {"1", "true", "yes"}
+ if _use_tls_raw
else extra.get("use_tls", True)
)
self.server_password = _get_scoped_secret("IRC_SERVER_PASSWORD") or extra.get("server_password", "")
@@ -545,8 +546,8 @@ def check_requirements() -> bool:
Only requires the server and channel — no external pip packages needed.
"""
- server = os.getenv("IRC_SERVER", "")
- channel = os.getenv("IRC_CHANNEL", "")
+ server = _get_scoped_secret("IRC_SERVER", "")
+ channel = _get_scoped_secret("IRC_CHANNEL", "")
# Also accept config.yaml-only configuration (no env vars).
# The gateway passes PlatformConfig; we just check env for the
# hermes setup / requirements check path.
@@ -556,8 +557,8 @@ def check_requirements() -> bool:
def validate_config(config) -> bool:
"""Validate that the platform config has enough info to connect."""
extra = getattr(config, "extra", {}) or {}
- server = os.getenv("IRC_SERVER") or extra.get("server", "")
- channel = os.getenv("IRC_CHANNEL") or extra.get("channel", "")
+ server = _get_scoped_secret("IRC_SERVER") or extra.get("server", "")
+ channel = _get_scoped_secret("IRC_CHANNEL") or extra.get("channel", "")
return bool(server and channel)
@@ -671,8 +672,8 @@ def interactive_setup() -> None:
def is_connected(config) -> bool:
"""Check whether IRC is configured (env or config.yaml)."""
extra = getattr(config, "extra", {}) or {}
- server = os.getenv("IRC_SERVER") or extra.get("server", "")
- channel = os.getenv("IRC_CHANNEL") or extra.get("channel", "")
+ server = _get_scoped_secret("IRC_SERVER") or extra.get("server", "")
+ channel = _get_scoped_secret("IRC_CHANNEL") or extra.get("channel", "")
return bool(server and channel)
@@ -689,24 +690,24 @@ def _env_enablement() -> dict | None:
the core hook — it becomes a proper ``HomeChannel`` dataclass on the
``PlatformConfig`` rather than being merged into ``extra``.
"""
- server = os.getenv("IRC_SERVER", "").strip()
- channel = os.getenv("IRC_CHANNEL", "").strip()
+ server = _get_scoped_secret("IRC_SERVER", "").strip()
+ channel = _get_scoped_secret("IRC_CHANNEL", "").strip()
if not (server and channel):
return None
seed: dict = {
"server": server,
"channel": channel,
}
- port = os.getenv("IRC_PORT", "").strip()
+ port = _get_scoped_secret("IRC_PORT", "").strip()
if port:
try:
seed["port"] = int(port)
except ValueError:
pass
- nickname = os.getenv("IRC_NICKNAME", "").strip()
+ nickname = _get_scoped_secret("IRC_NICKNAME", "").strip()
if nickname:
seed["nickname"] = nickname
- use_tls = os.getenv("IRC_USE_TLS", "").strip().lower()
+ use_tls = _get_scoped_secret("IRC_USE_TLS", "").strip().lower()
if use_tls:
seed["use_tls"] = use_tls in {"1", "true", "yes"}
# Passwords live in PlatformConfig.extra as well for back-compat with
@@ -718,11 +719,11 @@ def _env_enablement() -> dict | None:
# Optional home-channel (usually the same as IRC_CHANNEL, but can be a
# dedicated reports channel). Defaults to IRC_CHANNEL so cron jobs
# with ``deliver=irc`` have a sensible target without extra config.
- home = os.getenv("IRC_HOME_CHANNEL") or channel
+ home = _get_scoped_secret("IRC_HOME_CHANNEL") or channel
if home:
seed["home_channel"] = {
"chat_id": home,
- "name": os.getenv("IRC_HOME_CHANNEL_NAME", home),
+ "name": _get_scoped_secret("IRC_HOME_CHANNEL_NAME", home),
}
return seed
@@ -770,19 +771,19 @@ async def _standalone_send(
primitive.
"""
extra = getattr(pconfig, "extra", {}) or {}
- server = os.getenv("IRC_SERVER") or extra.get("server", "")
- channel = os.getenv("IRC_CHANNEL") or extra.get("channel", "")
+ server = _get_scoped_secret("IRC_SERVER") or extra.get("server", "")
+ channel = _get_scoped_secret("IRC_CHANNEL") or extra.get("channel", "")
if not server or not channel:
return {"error": "IRC standalone send: IRC_SERVER and IRC_CHANNEL must be configured"}
- port_value = os.getenv("IRC_PORT") or extra.get("port", 6697)
+ port_value = _get_scoped_secret("IRC_PORT") or extra.get("port", 6697)
try:
port = int(port_value)
except (TypeError, ValueError):
return {"error": f"IRC standalone send: invalid port {port_value!r}"}
- nickname = os.getenv("IRC_NICKNAME") or extra.get("nickname", "hermes-bot")
- use_tls_env = os.getenv("IRC_USE_TLS")
+ nickname = _get_scoped_secret("IRC_NICKNAME") or extra.get("nickname", "hermes-bot")
+ use_tls_env = _get_scoped_secret("IRC_USE_TLS")
if use_tls_env is not None:
use_tls = use_tls_env.lower() in {"1", "true", "yes"}
else:
diff --git a/plugins/platforms/line/adapter.py b/plugins/platforms/line/adapter.py
index b8d3ae10cd..1150556b4e 100644
--- a/plugins/platforms/line/adapter.py
+++ b/plugins/platforms/line/adapter.py
@@ -113,6 +113,7 @@ logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -514,7 +515,7 @@ class _LineClient:
async def reply(self, reply_token: str, messages: List[Dict[str, Any]]) -> None:
import aiohttp
timeout = aiohttp.ClientTimeout(total=self._timeout)
- async with aiohttp.ClientSession(timeout=timeout, trust_env=True) as session:
+ async with aiohttp.ClientSession(timeout=timeout, trust_env=gateway_trust_env()) as session:
async with session.post(
LINE_REPLY_URL,
headers=self._headers,
@@ -527,7 +528,7 @@ class _LineClient:
async def push(self, chat_id: str, messages: List[Dict[str, Any]]) -> None:
import aiohttp
timeout = aiohttp.ClientTimeout(total=self._timeout)
- async with aiohttp.ClientSession(timeout=timeout, trust_env=True) as session:
+ async with aiohttp.ClientSession(timeout=timeout, trust_env=gateway_trust_env()) as session:
async with session.post(
LINE_PUSH_URL,
headers=self._headers,
@@ -546,7 +547,7 @@ class _LineClient:
clamped = max(5, min(60, (seconds // 5) * 5 or 5))
try:
timeout = aiohttp.ClientTimeout(total=5.0)
- async with aiohttp.ClientSession(timeout=timeout, trust_env=True) as session:
+ async with aiohttp.ClientSession(timeout=timeout, trust_env=gateway_trust_env()) as session:
await session.post(
LINE_LOADING_URL,
headers=self._headers,
@@ -560,7 +561,7 @@ class _LineClient:
import aiohttp
url = LINE_CONTENT_URL_FMT.format(message_id=message_id)
timeout = aiohttp.ClientTimeout(total=30.0)
- async with aiohttp.ClientSession(timeout=timeout, trust_env=True) as session:
+ async with aiohttp.ClientSession(timeout=timeout, trust_env=gateway_trust_env()) as session:
async with session.get(url, headers={"Authorization": f"Bearer {self._token}"}) as resp:
if resp.status >= 400:
raise RuntimeError(f"LINE content {resp.status}")
@@ -571,7 +572,7 @@ class _LineClient:
import aiohttp
timeout = aiohttp.ClientTimeout(total=10.0)
try:
- async with aiohttp.ClientSession(timeout=timeout, trust_env=True) as session:
+ async with aiohttp.ClientSession(timeout=timeout, trust_env=gateway_trust_env()) as session:
async with session.get(LINE_BOT_INFO_URL, headers=self._headers) as resp:
if resp.status >= 400:
return None
diff --git a/plugins/platforms/matrix/adapter.py b/plugins/platforms/matrix/adapter.py
index 3268fd9d19..6a2eb362ff 100644
--- a/plugins/platforms/matrix/adapter.py
+++ b/plugins/platforms/matrix/adapter.py
@@ -128,6 +128,7 @@ except ImportError:
from gateway.config import Platform, PlatformConfig
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -593,13 +594,14 @@ def _resolve_max_message_length(config) -> int:
# Back-compat alias for callers/tests that import the module constant.
MAX_MESSAGE_LENGTH = DEFAULT_MAX_MESSAGE_LENGTH
-# Store directory for E2EE keys and sync state.
-# Uses get_hermes_home() so each profile gets its own Matrix store.
+# Store directory for E2EE keys and sync state. Resolved per adapter in
+# ``connect()`` (see ``_resolve_store_dir``), NOT at module scope: the
+# multiplex gateway imports this module once, so a module-level constant
+# would pin the root HERMES_HOME for every profile and all bots' Olm
+# identities would collide in one crypto.db (#89168). Mirrors the
+# pairing-store fix (a6397c379).
from hermes_constants import get_hermes_dir as _get_hermes_dir
-_STORE_DIR = _get_hermes_dir("platforms/matrix/store", "matrix/store")
-_CRYPTO_DB_PATH = _STORE_DIR / "crypto.db"
-
# Grace period: ignore messages older than this many seconds before startup.
_STARTUP_GRACE_SECONDS = 5
@@ -763,7 +765,7 @@ def _create_matrix_session(proxy_url: str | None):
import aiohttp
if not proxy_url:
- return aiohttp.ClientSession(trust_env=True)
+ return aiohttp.ClientSession(trust_env=gateway_trust_env())
if proxy_url.split("://")[0].lower().startswith("socks"):
try:
@@ -778,7 +780,7 @@ def _create_matrix_session(proxy_url: str | None):
"Run: pip install aiohttp-socks",
proxy_url,
)
- return aiohttp.ClientSession(trust_env=True)
+ return aiohttp.ClientSession(trust_env=gateway_trust_env())
return aiohttp.ClientSession(proxy=proxy_url)
@@ -1018,7 +1020,7 @@ def check_matrix_requirements() -> bool:
"""
token = _startup_env_secret("MATRIX_ACCESS_TOKEN")
password = _startup_env_secret("MATRIX_PASSWORD")
- homeserver = os.getenv("MATRIX_HOMESERVER", "")
+ homeserver = _startup_env_secret("MATRIX_HOMESERVER")
if not token and not password:
logger.debug("Matrix: neither MATRIX_ACCESS_TOKEN nor MATRIX_PASSWORD set")
@@ -1187,6 +1189,23 @@ class MatrixAdapter(BasePlatformAdapter):
max_message_length = DEFAULT_MAX_MESSAGE_LENGTH
_split_threshold = DEFAULT_MAX_MESSAGE_LENGTH - 100
+ def _resolve_store_dir(self) -> Path:
+ """Pin this adapter's crypto-store directory to the active profile.
+
+ Called from ``connect()``, which the multiplex gateway runs inside
+ ``_profile_runtime_scope`` -- ``get_hermes_dir`` honors that
+ context-local HERMES_HOME, so each profile's adapter gets its own
+ store. Cached on the instance so later reads (diagnostics, error
+ logs) outside the scope still report the store actually in use.
+ """
+ self._store_dir = _get_hermes_dir("platforms/matrix/store", "matrix/store")
+ return self._store_dir
+
+ @property
+ def _crypto_db_path(self) -> Path:
+ store_dir = self._store_dir or _get_hermes_dir("platforms/matrix/store", "matrix/store")
+ return store_dir / "crypto.db"
+
def __init__(self, config: PlatformConfig):
super().__init__(config, Platform.MATRIX)
@@ -1217,6 +1236,7 @@ class MatrixAdapter(BasePlatformAdapter):
self._client: Any = None # mautrix.client.Client
self._crypto_db: Any = None # mautrix.util.async_db.Database
+ self._store_dir: Optional[Path] = None # pinned per profile in connect()
self._sync_task: Optional[asyncio.Task] = None
self._invite_join_tasks: Dict[str, asyncio.Task] = {}
self._closing = False
@@ -1672,7 +1692,7 @@ class MatrixAdapter(BasePlatformAdapter):
"Matrix: server has different identity keys for device %s — "
"local crypto state is stale. Delete %s and restart.",
client.device_id,
- _CRYPTO_DB_PATH,
+ str(self._crypto_db_path),
)
return False
@@ -1728,8 +1748,9 @@ class MatrixAdapter(BasePlatformAdapter):
logger.error("Matrix: homeserver URL not configured")
return False
- # Ensure store dir exists for E2EE key persistence.
- _STORE_DIR.mkdir(parents=True, exist_ok=True)
+ # Ensure store dir exists for E2EE key persistence (resolved here,
+ # inside the profile scope, so multiplexed profiles never share it).
+ self._resolve_store_dir().mkdir(parents=True, exist_ok=True)
# Create the HTTP API layer.
client_session = _create_matrix_session(self._proxy_url)
@@ -1886,7 +1907,7 @@ class MatrixAdapter(BasePlatformAdapter):
from mautrix.crypto.store.asyncpg import PgCryptoStore
from mautrix.util.async_db import Database
- _STORE_DIR.mkdir(parents=True, exist_ok=True)
+ self._store_dir.mkdir(parents=True, exist_ok=True)
except Exception as exc:
if self._e2ee_mode == "optional":
logger.warning(
@@ -1907,7 +1928,7 @@ class MatrixAdapter(BasePlatformAdapter):
if self._encryption:
try:
# Remove legacy pickle file from pre-SQLite era.
- legacy_pickle = _STORE_DIR / "crypto_store.pickle"
+ legacy_pickle = self._store_dir / "crypto_store.pickle"
if legacy_pickle.exists():
logger.info(
"Matrix: removing legacy crypto_store.pickle (migrated to SQLite)"
@@ -1915,7 +1936,7 @@ class MatrixAdapter(BasePlatformAdapter):
legacy_pickle.unlink()
crypto_db = Database.create(
- f"sqlite:///{_CRYPTO_DB_PATH}",
+ f"sqlite:///{self._crypto_db_path}",
upgrade_table=PgCryptoStore.upgrade_table,
)
await crypto_db.start()
@@ -2043,7 +2064,7 @@ class MatrixAdapter(BasePlatformAdapter):
client.crypto = olm
logger.info(
"Matrix: E2EE enabled (store: %s%s)",
- str(_CRYPTO_DB_PATH),
+ str(self._crypto_db_path),
f", device_id={client.device_id}" if client.device_id else "",
)
except Exception as exc:
@@ -2287,7 +2308,7 @@ class MatrixAdapter(BasePlatformAdapter):
"mode": self._e2ee_mode,
"enabled": bool(self._encryption),
"deps_available": _check_e2ee_deps(),
- "crypto_store_path": str(_CRYPTO_DB_PATH),
+ "crypto_store_path": str(self._crypto_db_path),
"recovery_key_configured": bool(
_scoped_recovery_key().strip()
),
diff --git a/plugins/platforms/mattermost/adapter.py b/plugins/platforms/mattermost/adapter.py
index 6962fbf615..6f5172bb01 100644
--- a/plugins/platforms/mattermost/adapter.py
+++ b/plugins/platforms/mattermost/adapter.py
@@ -24,6 +24,7 @@ from typing import Any, Dict, List, Optional, Tuple
from gateway.config import Platform, PlatformConfig
from gateway.platforms.helpers import MessageDeduplicator
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -100,7 +101,7 @@ def validate_mattermost_config(config: PlatformConfig) -> bool:
"""Return True when Mattermost has enough config to connect."""
extra = getattr(config, "extra", {}) or {}
token = (getattr(config, "token", None) or _get_scoped_secret("MATTERMOST_TOKEN", "")).strip()
- url = (extra.get("url", "") or os.getenv("MATTERMOST_URL", "")).strip()
+ url = (extra.get("url", "") or _get_scoped_secret("MATTERMOST_URL", "")).strip()
if not token:
logger.debug("Mattermost: MATTERMOST_TOKEN not set")
return False
@@ -120,7 +121,7 @@ class MattermostAdapter(BasePlatformAdapter):
self._base_url: str = (
config.extra.get("url", "")
- or os.getenv("MATTERMOST_URL", "")
+ or _get_scoped_secret("MATTERMOST_URL", "")
).rstrip("/")
self._token: str = config.token or _get_scoped_secret("MATTERMOST_TOKEN", "")
@@ -137,7 +138,7 @@ class MattermostAdapter(BasePlatformAdapter):
# Reply mode: "thread" to nest replies, "off" for flat messages.
self._reply_mode: str = (
config.extra.get("reply_mode", "")
- or os.getenv("MATTERMOST_REPLY_MODE", "off")
+ or _get_scoped_secret("MATTERMOST_REPLY_MODE", "off")
).lower()
self._last_post_status: Optional[int] = None
@@ -316,7 +317,8 @@ class MattermostAdapter(BasePlatformAdapter):
return False
self._session = aiohttp.ClientSession(
- timeout=aiohttp.ClientTimeout(total=30)
+ timeout=aiohttp.ClientTimeout(total=30),
+ trust_env=gateway_trust_env(),
)
self._closing = False
@@ -870,7 +872,7 @@ class MattermostAdapter(BasePlatformAdapter):
# ignored, even if @mentioned. DMs are already excluded above.
allowed_raw = self.config.extra.get("allowed_channels") if self.config.extra else None
if allowed_raw is None:
- allowed_raw = os.getenv("MATTERMOST_ALLOWED_CHANNELS", "")
+ allowed_raw = _get_scoped_secret("MATTERMOST_ALLOWED_CHANNELS", "")
if isinstance(allowed_raw, list):
allowed_channels = {str(c).strip() for c in allowed_raw if str(c).strip()}
else:
@@ -884,12 +886,18 @@ class MattermostAdapter(BasePlatformAdapter):
)
return
- require_mention = os.getenv(
- "MATTERMOST_REQUIRE_MENTION", "true"
- ).lower() not in {"false", "0", "no"}
+ require_mention_raw = self.config.extra.get("require_mention") if self.config.extra else None
+ if require_mention_raw is None:
+ require_mention_raw = _get_scoped_secret("MATTERMOST_REQUIRE_MENTION", "true")
+ require_mention = str(require_mention_raw).lower() not in {"false", "0", "no"}
- free_channels_raw = os.getenv("MATTERMOST_FREE_RESPONSE_CHANNELS", "")
- free_channels = {ch.strip() for ch in free_channels_raw.split(",") if ch.strip()}
+ free_channels_raw = self.config.extra.get("free_response_channels") if self.config.extra else None
+ if free_channels_raw is None:
+ free_channels_raw = _get_scoped_secret("MATTERMOST_FREE_RESPONSE_CHANNELS", "")
+ if isinstance(free_channels_raw, list):
+ free_channels = {str(ch).strip() for ch in free_channels_raw if str(ch).strip()}
+ else:
+ free_channels = {ch.strip() for ch in str(free_channels_raw).split(",") if ch.strip()}
is_free_channel = channel_id in free_channels
mention_patterns = [
@@ -1057,7 +1065,7 @@ async def _standalone_send(
base_url = (
(getattr(pconfig, "extra", {}) or {}).get("url")
- or os.getenv("MATTERMOST_URL", "")
+ or _get_scoped_secret("MATTERMOST_URL", "")
).rstrip("/")
token = (getattr(pconfig, "token", None) or _get_scoped_secret("MATTERMOST_TOKEN", "")).strip()
if not base_url or not token:
@@ -1232,40 +1240,62 @@ def interactive_setup() -> None:
# ---------------------------------------------------------------------------
+def _profile_scoped_config_load() -> bool:
+ """True when running inside a multiplexed secondary profile's scope.
+
+ Secondary-profile adapters are constructed and connected inside
+ ``_profile_runtime_scope`` (secret scope installed + multiplex active) --
+ the same discriminator the Buzz/Discord/Telegram/WhatsApp/LINE/DingTalk
+ adapters use for this bug class (#98738 / #72348 / #80099). The DEFAULT
+ profile under multiplexing runs unscoped: ``os.environ`` holds its own
+ bridge output there and keeps its legacy precedence.
+ """
+ try:
+ from agent.secret_scope import current_secret_scope, is_multiplex_active
+
+ return bool(is_multiplex_active() and current_secret_scope() is not None)
+ except Exception:
+ return False
+
+
def _apply_yaml_config(yaml_cfg: dict, mattermost_cfg: dict) -> dict | None:
- """Translate ``config.yaml`` ``mattermost:`` keys into env vars.
+ """Translate ``config.yaml`` ``mattermost:`` keys into env vars and
+ ``PlatformConfig.extra`` entries.
Implements the ``apply_yaml_config_fn`` contract (#24836 / #25443).
Mirrors the legacy ``mattermost_cfg`` block that used to live in
``gateway/config.py::load_gateway_config()`` before this migration.
- The MattermostAdapter reads its runtime configuration via
- ``os.getenv()`` for ``MATTERMOST_REQUIRE_MENTION``,
- ``MATTERMOST_FREE_RESPONSE_CHANNELS``, and
- ``MATTERMOST_ALLOWED_CHANNELS``. Rather than rewrite those call sites
- to read from ``PlatformConfig.extra``, this hook keeps the env-driven
- model and merely owns the YAML→env translation here, next to the
- adapter that consumes it.
-
- Env vars take precedence over YAML — every assignment is guarded
- by ``not os.getenv(...)`` so an explicit env var survives a config.yaml
- update. Returns ``None`` because no extras are seeded into
- ``PlatformConfig.extra`` directly (everything flows through env).
+ Env vars take precedence over YAML for single-profile deployments --
+ each env write is guarded by ``not os.getenv(...)`` so an explicit env
+ var survives a config.yaml update. Under a multiplexed secondary
+ profile's scope, the env write is skipped entirely (it would otherwise
+ leak into the process-global ``os.environ`` and be inherited by every
+ other profile); instead the values are returned so the caller merges
+ them into this profile's own ``PlatformConfig.extra``, which the
+ require_mention/free_response_channels/allowed_channels read sites now
+ check first.
"""
- if "require_mention" in mattermost_cfg and not os.getenv("MATTERMOST_REQUIRE_MENTION"):
- os.environ["MATTERMOST_REQUIRE_MENTION"] = str(mattermost_cfg["require_mention"]).lower()
+ _skip_env_bridge = _profile_scoped_config_load()
+ seeded: dict = {}
+ if "require_mention" in mattermost_cfg:
+ seeded["require_mention"] = mattermost_cfg["require_mention"]
+ if not _skip_env_bridge and not os.getenv("MATTERMOST_REQUIRE_MENTION"):
+ os.environ["MATTERMOST_REQUIRE_MENTION"] = str(mattermost_cfg["require_mention"]).lower()
frc = mattermost_cfg.get("free_response_channels")
- if frc is not None and not os.getenv("MATTERMOST_FREE_RESPONSE_CHANNELS"):
- if isinstance(frc, list):
- frc = ",".join(str(v) for v in frc)
- os.environ["MATTERMOST_FREE_RESPONSE_CHANNELS"] = str(frc)
+ if frc is not None:
+ seeded["free_response_channels"] = frc
+ if not _skip_env_bridge and not os.getenv("MATTERMOST_FREE_RESPONSE_CHANNELS"):
+ _frc = ",".join(str(v) for v in frc) if isinstance(frc, list) else str(frc)
+ os.environ["MATTERMOST_FREE_RESPONSE_CHANNELS"] = _frc
# allowed_channels: if set, bot ONLY responds in these channels (whitelist)
ac = mattermost_cfg.get("allowed_channels")
- if ac is not None and not os.getenv("MATTERMOST_ALLOWED_CHANNELS"):
- if isinstance(ac, list):
- ac = ",".join(str(v) for v in ac)
- os.environ["MATTERMOST_ALLOWED_CHANNELS"] = str(ac)
- return None # all settings flow through env; nothing to merge into extras
+ if ac is not None:
+ seeded["allowed_channels"] = ac
+ if not _skip_env_bridge and not os.getenv("MATTERMOST_ALLOWED_CHANNELS"):
+ _ac = ",".join(str(v) for v in ac) if isinstance(ac, list) else str(ac)
+ os.environ["MATTERMOST_ALLOWED_CHANNELS"] = _ac
+ return seeded or None
# ---------------------------------------------------------------------------
diff --git a/plugins/platforms/ntfy/adapter.py b/plugins/platforms/ntfy/adapter.py
index b9fb08c7ef..87986416ef 100644
--- a/plugins/platforms/ntfy/adapter.py
+++ b/plugins/platforms/ntfy/adapter.py
@@ -155,21 +155,21 @@ def check_requirements() -> bool:
"""
if not HTTPX_AVAILABLE:
return False
- topic = os.getenv("NTFY_TOPIC", "").strip()
+ topic = _get_scoped_secret("NTFY_TOPIC", "").strip()
return bool(topic)
def validate_config(config) -> bool:
"""Validate that the configured ntfy platform has a topic set."""
extra = getattr(config, "extra", {}) or {}
- topic = extra.get("topic") or os.getenv("NTFY_TOPIC", "")
+ topic = extra.get("topic") or _get_scoped_secret("NTFY_TOPIC", "")
return bool(topic)
def is_connected(config) -> bool:
"""Check whether ntfy is configured (env or config.yaml)."""
extra = getattr(config, "extra", {}) or {}
- topic = os.getenv("NTFY_TOPIC") or extra.get("topic", "")
+ topic = _get_scoped_secret("NTFY_TOPIC") or extra.get("topic", "")
return bool(topic)
@@ -189,12 +189,12 @@ class NtfyAdapter(BasePlatformAdapter):
extra = config.extra or {}
self._server: str = (
extra.get("server")
- or os.getenv("NTFY_SERVER_URL", DEFAULT_SERVER)
+ or _get_scoped_secret("NTFY_SERVER_URL", DEFAULT_SERVER)
).rstrip("/")
- self._topic: str = extra.get("topic") or os.getenv("NTFY_TOPIC", "")
+ self._topic: str = extra.get("topic") or _get_scoped_secret("NTFY_TOPIC", "")
self._publish_topic: str = (
extra.get("publish_topic")
- or os.getenv("NTFY_PUBLISH_TOPIC", "")
+ or _get_scoped_secret("NTFY_PUBLISH_TOPIC", "")
or self._topic
)
self._token: str = extra.get("token") or _get_scoped_secret("NTFY_TOKEN", "")
@@ -488,27 +488,27 @@ def _env_enablement() -> dict | None:
core hook — it becomes a proper ``HomeChannel`` dataclass on the
``PlatformConfig`` rather than being merged into ``extra``.
"""
- topic = os.getenv("NTFY_TOPIC", "").strip()
+ topic = _get_scoped_secret("NTFY_TOPIC", "").strip()
if not topic:
return None
seed: dict = {
"topic": topic,
- "server": os.getenv("NTFY_SERVER_URL", DEFAULT_SERVER).rstrip("/"),
+ "server": _get_scoped_secret("NTFY_SERVER_URL", DEFAULT_SERVER).rstrip("/"),
}
- publish_topic = os.getenv("NTFY_PUBLISH_TOPIC", "").strip()
+ publish_topic = _get_scoped_secret("NTFY_PUBLISH_TOPIC", "").strip()
if publish_topic:
seed["publish_topic"] = publish_topic
token = _get_scoped_secret("NTFY_TOKEN", "").strip()
if token:
seed["token"] = token
- markdown = os.getenv("NTFY_MARKDOWN", "").strip().lower()
+ markdown = _get_scoped_secret("NTFY_MARKDOWN", "").strip().lower()
if markdown:
seed["markdown"] = markdown in ("1", "true", "yes")
- home = os.getenv("NTFY_HOME_CHANNEL", "").strip() or topic
+ home = _get_scoped_secret("NTFY_HOME_CHANNEL", "").strip() or topic
if home:
seed["home_channel"] = {
"chat_id": home,
- "name": os.getenv("NTFY_HOME_CHANNEL_NAME", home),
+ "name": _get_scoped_secret("NTFY_HOME_CHANNEL_NAME", home),
}
return seed
@@ -540,20 +540,20 @@ async def _standalone_send(
extra = getattr(pconfig, "extra", {}) or {}
server = (
extra.get("server")
- or os.getenv("NTFY_SERVER_URL", DEFAULT_SERVER)
+ or _get_scoped_secret("NTFY_SERVER_URL", DEFAULT_SERVER)
).rstrip("/")
publish_topic = (
chat_id
or extra.get("publish_topic")
- or os.getenv("NTFY_PUBLISH_TOPIC", "").strip()
+ or _get_scoped_secret("NTFY_PUBLISH_TOPIC", "").strip()
or extra.get("topic")
- or os.getenv("NTFY_TOPIC", "").strip()
+ or _get_scoped_secret("NTFY_TOPIC", "").strip()
)
if not publish_topic:
return {"error": "ntfy standalone send: NTFY_TOPIC not configured"}
token = extra.get("token") or _get_scoped_secret("NTFY_TOKEN", "")
- markdown_env = os.getenv("NTFY_MARKDOWN", "").strip().lower()
+ markdown_env = _get_scoped_secret("NTFY_MARKDOWN", "").strip().lower()
markdown_enabled = bool(extra.get("markdown")) or markdown_env in ("1", "true", "yes")
headers = {"Content-Type": "text/plain; charset=utf-8", "X-Tags": _ECHO_TAG, **_build_auth_header(token)}
diff --git a/plugins/platforms/photon/adapter.py b/plugins/platforms/photon/adapter.py
index ddc32c569d..0d07a15009 100644
--- a/plugins/platforms/photon/adapter.py
+++ b/plugins/platforms/photon/adapter.py
@@ -423,10 +423,10 @@ def check_requirements() -> bool:
if not HTTPX_AVAILABLE:
logger.warning("photon: httpx not installed — pip install httpx")
return False
- if not shutil.which(os.getenv("PHOTON_NODE_BIN") or "node"):
+ if not shutil.which(_get_scoped_secret("PHOTON_NODE_BIN") or "node"):
logger.warning(
"photon: node binary '%s' not found on PATH",
- os.getenv("PHOTON_NODE_BIN") or "node",
+ _get_scoped_secret("PHOTON_NODE_BIN") or "node",
)
return False
if not sidecar_deps_installed():
@@ -551,7 +551,7 @@ def _reinstall_sidecar_deps() -> None:
def validate_config(cfg: PlatformConfig) -> bool:
extra = cfg.extra or {}
- project_id = extra.get("project_id") or os.getenv("PHOTON_PROJECT_ID")
+ project_id = extra.get("project_id") or _get_scoped_secret("PHOTON_PROJECT_ID")
project_secret = extra.get("project_secret") or _get_scoped_secret("PHOTON_PROJECT_SECRET")
if not project_id or not project_secret:
# Fall back to auth.json
@@ -574,11 +574,11 @@ def _env_enablement() -> Optional[dict]:
if not (project_id and project_secret):
return None
seed: dict = {"project_id": project_id, "project_secret": project_secret}
- home = os.getenv("PHOTON_HOME_CHANNEL", "").strip()
+ home = _get_scoped_secret("PHOTON_HOME_CHANNEL", "").strip()
if home:
seed["home_channel"] = {
"chat_id": home,
- "name": os.getenv("PHOTON_HOME_CHANNEL_NAME", "Home"),
+ "name": _get_scoped_secret("PHOTON_HOME_CHANNEL_NAME", "Home"),
}
return seed
@@ -591,7 +591,7 @@ def _markdown_enabled() -> bool:
``PHOTON_MARKDOWN=false`` is the kill-switch back to stripped plain
text without a release.
"""
- return os.getenv("PHOTON_MARKDOWN", "true").strip().lower() not in {
+ return _get_scoped_secret("PHOTON_MARKDOWN", "true").strip().lower() not in {
"false", "0", "no",
}
@@ -729,7 +729,7 @@ class PhotonAdapter(BasePlatformAdapter):
# the spectrum-ts SDK authenticates with.
stored_id, stored_sec = load_project_credentials()
self._project_id: str = (
- os.getenv("PHOTON_PROJECT_ID")
+ _get_scoped_secret("PHOTON_PROJECT_ID")
or extra.get("project_id")
or stored_id
or ""
@@ -743,7 +743,7 @@ class PhotonAdapter(BasePlatformAdapter):
# Sidecar
self._sidecar_port = _coerce_port(
- extra.get("sidecar_port") or os.getenv("PHOTON_SIDECAR_PORT"),
+ extra.get("sidecar_port") or _get_scoped_secret("PHOTON_SIDECAR_PORT"),
_DEFAULT_SIDECAR_PORT,
)
self._sidecar_bind = _DEFAULT_SIDECAR_BIND
@@ -751,9 +751,9 @@ class PhotonAdapter(BasePlatformAdapter):
_get_scoped_secret("PHOTON_SIDECAR_TOKEN") or secrets.token_hex(16)
)
self._autostart_sidecar = str(
- os.getenv("PHOTON_SIDECAR_AUTOSTART", "true")
+ _get_scoped_secret("PHOTON_SIDECAR_AUTOSTART", "true")
).lower() not in ("0", "false", "no")
- self._node_bin = os.getenv("PHOTON_NODE_BIN") or shutil.which("node") or "node"
+ self._node_bin = _get_scoped_secret("PHOTON_NODE_BIN") or shutil.which("node") or "node"
# Presence watchdog. spectrum-ts only reconnects when its inbound
# iterator throws or ends; a half-open ("zombie") gRPC socket makes the
@@ -776,21 +776,21 @@ class PhotonAdapter(BasePlatformAdapter):
self._probe_interval = _coerce_float(
_first_set(
extra.get("probe_interval_seconds"),
- os.getenv("PHOTON_PROBE_INTERVAL_SECONDS"),
+ _get_scoped_secret("PHOTON_PROBE_INTERVAL_SECONDS"),
),
600.0,
)
self._probe_timeout = _coerce_float(
_first_set(
extra.get("probe_timeout_seconds"),
- os.getenv("PHOTON_PROBE_TIMEOUT_SECONDS"),
+ _get_scoped_secret("PHOTON_PROBE_TIMEOUT_SECONDS"),
),
10.0,
)
self._probe_max_failures = _coerce_int(
_first_set(
extra.get("probe_max_failures"),
- os.getenv("PHOTON_PROBE_MAX_FAILURES"),
+ _get_scoped_secret("PHOTON_PROBE_MAX_FAILURES"),
),
3,
)
@@ -843,14 +843,14 @@ class PhotonAdapter(BasePlatformAdapter):
# always processed. Config key wins, then env var.
_require_mention = extra.get("require_mention")
if _require_mention is None:
- _require_mention = os.getenv("PHOTON_REQUIRE_MENTION")
+ _require_mention = _get_scoped_secret("PHOTON_REQUIRE_MENTION")
self.require_mention = str(_require_mention).strip().lower() in {
"true", "1", "yes", "on",
}
self._mention_patterns = self._compile_mention_patterns(
extra["mention_patterns"]
if "mention_patterns" in extra
- else os.getenv("PHOTON_MENTION_PATTERNS")
+ else _get_scoped_secret("PHOTON_MENTION_PATTERNS")
)
# -- Group-mention gating (parity with BlueBubbles) -------------------
@@ -2274,7 +2274,7 @@ class PhotonAdapter(BasePlatformAdapter):
return True
def _reactions_enabled(self) -> bool:
- return os.getenv("PHOTON_REACTIONS", "false").strip().lower() in {
+ return _get_scoped_secret("PHOTON_REACTIONS", "false").strip().lower() in {
"true", "1", "yes", "on",
}
@@ -2805,7 +2805,7 @@ async def _standalone_send(
if not HTTPX_AVAILABLE:
return {"error": "httpx not installed"}
port = _coerce_port(
- (pconfig.extra or {}).get("sidecar_port") or os.getenv("PHOTON_SIDECAR_PORT"),
+ (pconfig.extra or {}).get("sidecar_port") or _get_scoped_secret("PHOTON_SIDECAR_PORT"),
_DEFAULT_SIDECAR_PORT,
)
token = _get_scoped_secret("PHOTON_SIDECAR_TOKEN")
diff --git a/plugins/platforms/photon/auth.py b/plugins/platforms/photon/auth.py
index 34b573a2d8..14fecee80f 100644
--- a/plugins/platforms/photon/auth.py
+++ b/plugins/platforms/photon/auth.py
@@ -254,7 +254,7 @@ def load_project_credentials() -> Tuple[Optional[str], Optional[str]]:
use. This is the pair the Node sidecar feeds to ``spectrum-ts``; the id
is the unified project id (dashboard id == spectrumProjectId).
"""
- env_id = os.getenv("PHOTON_PROJECT_ID")
+ env_id = _get_scoped_secret("PHOTON_PROJECT_ID")
env_sec = _get_scoped_secret("PHOTON_PROJECT_SECRET")
if env_id and env_sec:
return env_id, env_sec
@@ -277,7 +277,7 @@ def load_dashboard_project_id() -> Optional[str]:
rewrote (it now 404s), while the Spectrum id always matches the live row.
Falls back to the legacy keys for older records.
"""
- env_id = os.getenv("PHOTON_DASHBOARD_PROJECT_ID")
+ env_id = _get_scoped_secret("PHOTON_DASHBOARD_PROJECT_ID")
if env_id:
return env_id
auth = _load_auth()
diff --git a/plugins/platforms/raft/adapter.py b/plugins/platforms/raft/adapter.py
index d31ee4601a..49f9224675 100644
--- a/plugins/platforms/raft/adapter.py
+++ b/plugins/platforms/raft/adapter.py
@@ -97,6 +97,52 @@ _RAFT_TURN_IDS: set[str] = set()
_RAFT_PROMPT_TURN_IDS: set[str] = set()
+def _profile_scoped() -> bool:
+ """True when running inside a multiplexed secondary profile's scope.
+
+ Secondary-profile adapters are constructed, connected, and reloaded
+ inside ``_profile_runtime_scope`` (secret scope installed + multiplex
+ active) — the same discriminator the Buzz/SimpleX adapters use for this
+ bug class (#98738). The DEFAULT profile under multiplexing runs
+ unscoped: ``os.environ`` holds its own bridge output there and keeps its
+ legacy precedence.
+ """
+ try:
+ from agent.secret_scope import current_secret_scope, is_multiplex_active
+
+ return bool(is_multiplex_active() and current_secret_scope() is not None)
+ except Exception:
+ return False
+
+
+def _resolve_raft_profile() -> str:
+ """Scope-aware resolution of the ``RAFT_PROFILE`` slug.
+
+ Raft has no ``config.yaml`` equivalent for this value (env-only), so a
+ secondary multiplex profile's only way to configure Raft is via its own
+ ``.env`` file — which the installed secret scope (built from that
+ profile's ``.env`` by ``_profile_runtime_scope``) already carries.
+ Reading raw ``os.environ.get("RAFT_PROFILE")`` here would instead return
+ the DEFAULT profile's bridged value, misdirecting the bridge subprocess
+ or CLI hint at another profile's external Raft workspace/agent identity.
+
+ ``get_secret()`` is only called when ``_profile_scoped()`` is True — the
+ callers of this helper (``connect()``/``register()``) run inside
+ ``_profile_runtime_scope`` for secondary profiles, but the DEFAULT
+ profile's own startup path never installs a scope, where ``get_secret()``
+ would raise ``UnscopedSecretError``; the guard keeps that path on the
+ unchanged ``os.environ`` read.
+ """
+ if _profile_scoped():
+ try:
+ from agent.secret_scope import get_secret
+
+ return (get_secret("RAFT_PROFILE") or "").strip()
+ except Exception:
+ return ""
+ return os.environ.get("RAFT_PROFILE", "").strip()
+
+
def check_raft_requirements() -> bool:
"""Check if Raft channel dependencies are available.
@@ -533,7 +579,7 @@ class RaftAdapter(BasePlatformAdapter):
logger.warning("[raft] raft CLI not found in PATH; bridge not spawned — wake-only polling mode")
return
- profile = os.environ.get("RAFT_PROFILE", "")
+ profile = _resolve_raft_profile()
if not profile:
logger.warning("[raft] RAFT_PROFILE not set; bridge not spawned")
return
@@ -777,8 +823,12 @@ def _env_enablement() -> Optional[dict]:
"""Seed PlatformConfig.extra from env vars during gateway config load.
Auto-enables when RAFT_PROFILE is set (the adapter needs it anyway).
+ Scope-aware: consults the active profile's own RAFT_PROFILE (env, or a
+ secondary profile's own .env via the secret scope) instead of the
+ default profile's bridged env value (mirrors the Buzz/SimpleX fix for
+ #98738) — see ``_resolve_raft_profile``.
"""
- if not os.getenv("RAFT_PROFILE"):
+ if not _resolve_raft_profile():
return None
return {"enabled": True}
@@ -839,12 +889,18 @@ def register(ctx) -> None:
setup_fn=interactive_setup,
env_enablement_fn=_env_enablement,
emoji="🔔",
+ # Scope-aware (mirrors _resolve_raft_profile's docstring): register()
+ # runs inside _profile_runtime_scope for a secondary multiplex
+ # profile (via discover_plugins() in
+ # gateway/run.py::_start_one_profile_adapters), so this resolves
+ # that profile's own RAFT_PROFILE instead of the default profile's
+ # bridged env value baked into a shared registry entry.
platform_hint=(
"You are connected to Raft via an external-agent channel. "
"Run `raft --profile {profile} profile show` to confirm which agent profile is active. "
"Run `raft --profile {profile} manual get raft-cli-overview` to learn available Raft commands. "
"Always pass `--profile {profile}` to every raft CLI call."
- ).format(profile=os.environ.get("RAFT_PROFILE", "your-agent-profile")),
+ ).format(profile=_resolve_raft_profile() or "your-agent-profile"),
)
ctx.register_hook("on_session_start", _on_session_start)
ctx.register_hook("pre_llm_call", _on_pre_llm_call)
diff --git a/plugins/platforms/simplex/adapter.py b/plugins/platforms/simplex/adapter.py
index b4f493e456..979c1e6ea3 100644
--- a/plugins/platforms/simplex/adapter.py
+++ b/plugins/platforms/simplex/adapter.py
@@ -56,6 +56,28 @@ from datetime import datetime, timezone
from pathlib import Path
from typing import Any, Dict, List, Optional
+from agent.secret_scope import UnscopedSecretError as _UnscopedSecretError
+from agent.secret_scope import get_secret as _scoped_get_secret
+
+
+def _get_scoped_secret(name, default=None):
+ """Scope-aware env read with the default-profile startup fallback.
+
+ Secondary profiles construct their adapters under a profile secret
+ scope -- the scope is authoritative and a scoped miss returns ``default``
+ (no cross-profile borrow from ``os.environ``, which holds the DEFAULT
+ profile's YAML-to-env bridge output under multiplexing). The default
+ profile's adapter constructs *unscoped*, where a bare ``get_secret``
+ would raise ``UnscopedSecretError``; there ``os.environ`` is that
+ profile's own value, so fall back to it. Same helper as the IRC/ntfy/
+ Mattermost plugins.
+ """
+ try:
+ val = _scoped_get_secret(name, default)
+ except _UnscopedSecretError:
+ val = os.getenv(name)
+ return val if val is not None else default
+
# Lazy import: BasePlatformAdapter and friends live in the main repo.
# Imported at module top because they're stdlib-only inside Hermes — no
# external dependency that would block the plugin from loading.
@@ -153,7 +175,7 @@ class SimplexAdapter(BasePlatformAdapter):
# Contact-request auto-accept (on by default — matches the way most
# bot deployments expect to behave). Read from env first, then fall
# back to the value seeded by ``_env_enablement``.
- env_auto = os.getenv("SIMPLEX_AUTO_ACCEPT")
+ env_auto = _get_scoped_secret("SIMPLEX_AUTO_ACCEPT")
if env_auto is not None:
self.auto_accept = env_auto.strip().lower() not in {"0", "false", "no", ""}
else:
@@ -162,7 +184,7 @@ class SimplexAdapter(BasePlatformAdapter):
# Group allowlist. Without ``SIMPLEX_GROUP_ALLOWED``, group messages
# are ignored entirely (safer default — a bot in a group otherwise
# processes every member's traffic). Use ``*`` to accept any group.
- group_allowed_str = os.getenv("SIMPLEX_GROUP_ALLOWED", "") or extra.get(
+ group_allowed_str = _get_scoped_secret("SIMPLEX_GROUP_ALLOWED", "") or extra.get(
"group_allowed", ""
)
self.group_allow_from = set(_parse_comma_list(group_allowed_str))
@@ -1172,7 +1194,7 @@ def check_requirements() -> bool:
so the gateway never instantiates the adapter when the dependency is
missing or no daemon URL is configured.
"""
- if not os.getenv("SIMPLEX_WS_URL"):
+ if not _get_scoped_secret("SIMPLEX_WS_URL"):
return False
try:
import websockets # noqa: F401
@@ -1184,14 +1206,14 @@ def check_requirements() -> bool:
def validate_config(config) -> bool:
"""Validate that the platform config has enough info to connect."""
extra = getattr(config, "extra", {}) or {}
- ws_url = os.getenv("SIMPLEX_WS_URL") or extra.get("ws_url", "")
+ ws_url = _get_scoped_secret("SIMPLEX_WS_URL") or extra.get("ws_url", "")
return bool(ws_url)
def is_connected(config) -> bool:
"""Check whether SimpleX is configured (env or config.yaml)."""
extra = getattr(config, "extra", {}) or {}
- ws_url = os.getenv("SIMPLEX_WS_URL") or extra.get("ws_url", "")
+ ws_url = _get_scoped_secret("SIMPLEX_WS_URL") or extra.get("ws_url", "")
return bool(ws_url)
@@ -1207,24 +1229,24 @@ def _env_enablement() -> Optional[dict]:
becomes a proper ``HomeChannel`` dataclass on the ``PlatformConfig``
rather than being merged into ``extra``.
"""
- ws_url = os.getenv("SIMPLEX_WS_URL", "").strip()
+ ws_url = _get_scoped_secret("SIMPLEX_WS_URL", "").strip()
if not ws_url:
return None
seed: dict = {"ws_url": ws_url}
- auto_accept = os.getenv("SIMPLEX_AUTO_ACCEPT", "").strip().lower()
+ auto_accept = _get_scoped_secret("SIMPLEX_AUTO_ACCEPT", "").strip().lower()
if auto_accept:
seed["auto_accept"] = auto_accept not in {"0", "false", "no"}
- group_allowed = os.getenv("SIMPLEX_GROUP_ALLOWED", "").strip()
+ group_allowed = _get_scoped_secret("SIMPLEX_GROUP_ALLOWED", "").strip()
if group_allowed:
seed["group_allowed"] = group_allowed
- home = os.getenv("SIMPLEX_HOME_CHANNEL", "").strip()
+ home = _get_scoped_secret("SIMPLEX_HOME_CHANNEL", "").strip()
if home:
seed["home_channel"] = {
"chat_id": home,
- "name": os.getenv("SIMPLEX_HOME_CHANNEL_NAME", "").strip() or home,
+ "name": _get_scoped_secret("SIMPLEX_HOME_CHANNEL_NAME", "").strip() or home,
}
return seed
@@ -1257,7 +1279,7 @@ async def _standalone_send(
return {"error": "websockets not installed. Run: pip install websockets"}
extra = getattr(pconfig, "extra", {}) or {}
- ws_url = os.getenv("SIMPLEX_WS_URL") or extra.get(
+ ws_url = _get_scoped_secret("SIMPLEX_WS_URL") or extra.get(
"ws_url", "ws://127.0.0.1:5225"
)
if not ws_url:
diff --git a/plugins/platforms/slack/adapter.py b/plugins/platforms/slack/adapter.py
index cd8e237d3b..b4cd21c143 100644
--- a/plugins/platforms/slack/adapter.py
+++ b/plugins/platforms/slack/adapter.py
@@ -43,6 +43,7 @@ from agent.secret_scope import UnscopedSecretError, get_secret
from gateway.config import Platform, PlatformConfig
from gateway.platforms.helpers import MessageDeduplicator
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -1825,7 +1826,7 @@ class SlackAdapter(BasePlatformAdapter):
"Slack's ephemeral reply limit.]_"
)
try:
- async with aiohttp.ClientSession(trust_env=True) as session:
+ async with aiohttp.ClientSession(trust_env=gateway_trust_env()) as session:
for idx, chunk in enumerate(chunks):
payload = {
"response_type": "ephemeral",
@@ -3838,6 +3839,30 @@ class SlackAdapter(BasePlatformAdapter):
return "none"
return value
+ def _slack_api_human_users(self) -> frozenset:
+ """Slack user IDs whose Web-API posts count as human-authored.
+
+ A message posted with a *user* token (``xoxp-``) is authored by a real
+ person, but Slack still stamps it with the posting ``app_id`` and it
+ carries no ``client_msg_id`` — exactly the #35777 app/bot signature in
+ ``_event_declares_bot_sender``. Operators running their own front-end
+ (dashboard, mobile shell) allowlist those *users* via
+ ``platforms.slack.extra.api_human_users`` (``SLACK_API_HUMAN_USERS``
+ fallback) instead of ``allow_bots: all``. Users only — an app-id
+ allowlist would also admit the app's own ``xoxb`` bot posts, which
+ carry the same user+app_id shape.
+ """
+ cached = getattr(self, "_api_human_users_cache", None)
+ if cached is None:
+ raw = self.config.extra.get("api_human_users")
+ if raw is None:
+ raw = os.getenv("SLACK_API_HUMAN_USERS", "")
+ parts = raw if isinstance(raw, (list, tuple, set)) else str(raw).split(",")
+ cached = self._api_human_users_cache = frozenset(
+ str(p).strip() for p in parts if str(p).strip()
+ )
+ return cached
+
def _event_declares_bot_sender(self, event: dict) -> bool:
"""Return True when the Slack event itself identifies a bot sender."""
if event.get("bot_id") or event.get("bot_profile"):
@@ -3852,7 +3877,11 @@ class SlackAdapter(BasePlatformAdapter):
# human-authored messages normally carry client_msg_id, so treat the
# combination as app/bot-authored (#35777).
if event.get("app_id") and not event.get("client_msg_id"):
- return True
+ # ...unless the operator allowlisted this user's API posts
+ # (_slack_api_human_users). ``user`` is required so classic bot
+ # posts (no ``user``) never match; bot_message/bot_id already
+ # returned True above.
+ return event.get("user") not in self._slack_api_human_users()
return False
def _resolve_thread_ts(
@@ -6324,9 +6353,19 @@ class SlackAdapter(BasePlatformAdapter):
# or file downloads. The final gateway runner auth check happens
# after MessageEvent construction, so adapter-side media fetches need
# the same auth chain up front.
+ # Prefer the injected profile-bound check (survives the multiplex
+ # closure handler, which has no ``__self__``); fall back to runner
+ # introspection for adapters wired without one.
+ _early_decision = (
+ self._is_sender_authorized(
+ user_id, "dm" if is_dm else "group", channel_id
+ )
+ if user_id and getattr(self, "_authorization_check", None) is not None
+ else None
+ )
_runner = getattr(getattr(self, "_message_handler", None), "__self__", None)
_auth_fn = getattr(_runner, "_is_user_authorized", None)
- if user_id and callable(_auth_fn):
+ if _early_decision is None and user_id and callable(_auth_fn):
_source = self.build_source(
chat_id=channel_id,
chat_name="",
@@ -6334,13 +6373,14 @@ class SlackAdapter(BasePlatformAdapter):
user_id=user_id,
user_name="",
)
- if not _auth_fn(_source):
- logger.warning(
- "[Slack] Early reject of unauthorized user %s in channel %s",
- user_id,
- channel_id,
- )
- return
+ _early_decision = bool(_auth_fn(_source))
+ if _early_decision is False:
+ logger.warning(
+ "[Slack] Early reject of unauthorized user %s in channel %s",
+ user_id,
+ channel_id,
+ )
+ return
# Build thread_ts for session keying.
# In channels: fall back to ts so each top-level @mention starts a
@@ -7049,7 +7089,9 @@ class SlackAdapter(BasePlatformAdapter):
# subtype=bot_message with user=None; flag them so the
# gateway SLACK_ALLOW_BOTS bypass can authorize them
# (they carry no user_id to match against the allowlist).
- is_bot=bool(event.get("bot_id")) or event.get("subtype") == "bot_message",
+ # Same predicate as the drop gate above, so an api_human_users
+ # post is a plain human here too.
+ is_bot=self._event_declares_bot_sender(event),
)
# Per-channel ephemeral prompt
@@ -7459,6 +7501,23 @@ class SlackAdapter(BasePlatformAdapter):
if not normalized_user_id:
return False
+ chat_type = "dm" if str(channel_id or "").startswith("D") else "group"
+
+ # Preferred path: the auth callback GatewayRunner injects at connect
+ # time (``set_authorization_check``) runs the full, profile-bound
+ # ``_is_user_authorized`` chain. Unlike the ``__self__`` introspection
+ # below it also resolves on a multiplexed adapter, whose message
+ # handler is a profile closure with no ``__self__`` (#72657, same
+ # class as Telegram's #86296).
+ # ``getattr``: adapters built via ``object.__new__`` never ran
+ # ``BasePlatformAdapter.__init__``.
+ if getattr(self, "_authorization_check", None) is not None:
+ injected = self._is_sender_authorized(
+ normalized_user_id, chat_type, str(channel_id or "")
+ )
+ if injected is not None:
+ return injected
+
runner = getattr(getattr(self, "_message_handler", None), "__self__", None)
auth_fn = getattr(runner, "_is_user_authorized", None)
if callable(auth_fn):
@@ -7468,7 +7527,7 @@ class SlackAdapter(BasePlatformAdapter):
source = SessionSource(
platform=Platform.SLACK,
chat_id=str(channel_id or normalized_user_id),
- chat_type="dm" if str(channel_id or "").startswith("D") else "group",
+ chat_type=chat_type,
user_id=normalized_user_id,
user_name=str(user_name).strip() if user_name else None,
scope_id=str(team_id) if team_id else None,
@@ -7481,21 +7540,15 @@ class SlackAdapter(BasePlatformAdapter):
exc_info=True,
)
- if os.getenv("SLACK_ALLOW_ALL_USERS", "").lower() in {"true", "1", "yes"}:
+ # Env-only fallback (no injected check, no bound runner). Gate reads go
+ # through the shared per-profile accessor: under multiplex a scoped
+ # miss returns "" instead of falling through to ``os.environ``, which
+ # holds the DEFAULT profile's allow-all flag / allowlist.
+ from gateway.authz_mixin import _platform_gate_env as _env
+
+ if _env("SLACK_ALLOW_ALL_USERS").lower() in {"true", "1", "yes"}:
return True
- def _env(name: str) -> str:
- # Multiplex: profile .env is in secret_scope, not process environ.
- try:
- from agent.secret_scope import get_secret
-
- val = get_secret(name)
- if val is not None and str(val).strip():
- return str(val).strip()
- except Exception:
- pass
- return (os.getenv(name) or "").strip()
-
allowed_ids = set()
platform_allowlist = _env("SLACK_ALLOWED_USERS")
if platform_allowlist:
@@ -7507,8 +7560,6 @@ class SlackAdapter(BasePlatformAdapter):
if allowed_ids:
return "*" in allowed_ids or normalized_user_id in allowed_ids
- if _env("SLACK_ALLOW_ALL_USERS").lower() in {"true", "1", "yes"}:
- return True
return _env("GATEWAY_ALLOW_ALL_USERS").lower() in {"true", "1", "yes"}
async def _handle_slash_confirm_action(self, ack, body, action) -> None:
@@ -8184,7 +8235,7 @@ class SlackAdapter(BasePlatformAdapter):
skip_for_delta = bool(after_ts and msg_ts and msg_ts <= after_ts)
if skip_for_delta and not is_parent:
continue
- is_bot = bool(msg.get("bot_id")) or msg.get("subtype") == "bot_message"
+ is_bot = self._event_declares_bot_sender(msg)
msg_user = msg.get("user", "")
# Identify "our own" bot for this workspace (multi-workspace safe).
diff --git a/plugins/platforms/sms/adapter.py b/plugins/platforms/sms/adapter.py
index 37db336e7a..8d2592bc7b 100644
--- a/plugins/platforms/sms/adapter.py
+++ b/plugins/platforms/sms/adapter.py
@@ -29,6 +29,7 @@ from typing import Any, Dict, Optional
from gateway.config import Platform, PlatformConfig
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -156,7 +157,7 @@ class SmsAdapter(BasePlatformAdapter):
await site.start()
self._http_session = aiohttp.ClientSession(
timeout=aiohttp.ClientTimeout(total=30),
- trust_env=True,
+ trust_env=gateway_trust_env(),
)
self._running = True
@@ -200,7 +201,7 @@ class SmsAdapter(BasePlatformAdapter):
session = self._http_session or aiohttp.ClientSession(
timeout=aiohttp.ClientTimeout(total=30),
- trust_env=True,
+ trust_env=gateway_trust_env(),
)
try:
for chunk in chunks:
diff --git a/plugins/platforms/teams/adapter.py b/plugins/platforms/teams/adapter.py
index f6b357208f..172d89d946 100644
--- a/plugins/platforms/teams/adapter.py
+++ b/plugins/platforms/teams/adapter.py
@@ -103,6 +103,7 @@ TextBlock = None # type: ignore[assignment,misc]
from gateway.config import Platform, PlatformConfig
from gateway.platforms.helpers import MessageDeduplicator
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -641,7 +642,7 @@ async def _standalone_send(
# Per-request timeouts so a slow STS endpoint cannot starve the
# subsequent activity POST of its budget.
per_request_timeout = _aiohttp.ClientTimeout(total=15.0)
- async with _aiohttp.ClientSession(trust_env=True) as session:
+ async with _aiohttp.ClientSession(trust_env=gateway_trust_env()) as session:
async with session.post(
token_url,
data={
diff --git a/plugins/platforms/telegram/adapter.py b/plugins/platforms/telegram/adapter.py
index b8ed8d11f0..15d2541d30 100644
--- a/plugins/platforms/telegram/adapter.py
+++ b/plugins/platforms/telegram/adapter.py
@@ -1213,18 +1213,40 @@ class TelegramAdapter(BasePlatformAdapter):
if not normalized_user_id:
return False
+ normalized_chat_type = str(chat_type or "dm").strip().lower() or "dm"
+ if normalized_chat_type == "private":
+ normalized_chat_type = "dm"
+ elif normalized_chat_type == "supergroup":
+ normalized_chat_type = "forum" if thread_id is not None else "group"
+
+ # Preferred path: the auth callback GatewayRunner injects at
+ # connection time (set_authorization_check), which delegates to the
+ # full _is_user_authorized chain -- env allowlists, group allowlists,
+ # pairing store, allow-all flags. Unlike the __self__ introspection
+ # below, this also works for a secondary multiplexed adapter, whose
+ # _message_handler is a profile closure with no __self__ (the same
+ # gap the admin-tier check had -- resolved the same way). The getattr
+ # tolerates partially-constructed adapters (object.__new__ in tests)
+ # that never ran BasePlatformAdapter.__init__.
+ if getattr(self, "_authorization_check", None) is not None:
+ injected = self._is_sender_authorized(
+ normalized_user_id,
+ chat_type=normalized_chat_type,
+ chat_id=str(chat_id or normalized_user_id),
+ thread_id=str(thread_id) if thread_id is not None else None,
+ )
+ if injected is not None:
+ return injected
+
+ # Legacy path: resolve the runner off the bound message handler.
+ # Still reachable for adapters wired without set_authorization_check
+ # (bare-adapter tests, direct embedding).
runner = getattr(getattr(self, "_message_handler", None), "__self__", None)
auth_fn = getattr(runner, "_is_user_authorized", None)
if callable(auth_fn):
try:
from gateway.session import SessionSource
- normalized_chat_type = str(chat_type or "dm").strip().lower() or "dm"
- if normalized_chat_type == "private":
- normalized_chat_type = "dm"
- elif normalized_chat_type == "supergroup":
- normalized_chat_type = "forum" if thread_id is not None else "group"
-
source = SessionSource(
platform=Platform.TELEGRAM,
chat_id=str(chat_id or normalized_user_id),
@@ -1264,6 +1286,9 @@ class TelegramAdapter(BasePlatformAdapter):
user = getattr(message, "from_user", None)
chat = getattr(message, "chat", None)
user_id = str(getattr(user, "id", "")).strip() or None
+ # Carry the bot flag so the runner's ``*_ALLOW_BOTS`` policy branch is
+ # reachable from this prefilter, exactly as it is for ``build_source``.
+ is_bot = bool(getattr(user, "is_bot", False)) if user is not None else False
user_name = (
str(getattr(user, "username", "") or getattr(user, "full_name", "") or "").strip()
or None
@@ -1309,6 +1334,7 @@ class TelegramAdapter(BasePlatformAdapter):
user_id=user_id,
user_name=user_name,
thread_id=thread_id,
+ is_bot=is_bot,
)
def _source_from_reaction_for_auth(self, update):
@@ -1390,14 +1416,20 @@ class TelegramAdapter(BasePlatformAdapter):
if source.chat_type != "dm":
return False
- runner = getattr(getattr(self, "_message_handler", None), "__self__", None)
+ # The bound-handler ``__self__`` is None under multiplex (the handler is
+ # a profile closure); ``gateway_runner`` is injected on every adapter
+ # by ``GatewayRunner._create_adapter`` and survives that wrapping.
+ runner = getattr(
+ getattr(self, "_message_handler", None), "__self__", None
+ ) or getattr(self, "gateway_runner", None)
behavior_fn = getattr(runner, "_get_unauthorized_dm_behavior", None)
if callable(behavior_fn):
try:
return (
behavior_fn(
Platform.TELEGRAM,
- profile=getattr(source, "profile", None),
+ profile=getattr(source, "profile", None)
+ or getattr(self, "_owner_profile", None),
)
== "pair"
)
@@ -1490,6 +1522,8 @@ class TelegramAdapter(BasePlatformAdapter):
user_id,
chat_type=source.chat_type,
chat_id=source.chat_id,
+ is_bot=source.is_bot,
+ thread_id=source.thread_id,
)
if has_callback
else None
@@ -1686,7 +1720,11 @@ class TelegramAdapter(BasePlatformAdapter):
return "thread not found" in str(error).lower()
def _prune_stale_dm_topic_binding(
- self, chat_id: Any, thread_id: Any,
+ self,
+ chat_id: Any,
+ thread_id: Any,
+ *,
+ metadata: Optional[Dict[str, Any]] = None,
) -> None:
"""Drop the stale ``telegram_dm_topic_bindings`` row for a
topic Telegram has confirmed deleted.
@@ -1699,6 +1737,12 @@ class TelegramAdapter(BasePlatformAdapter):
on to a fresh topic). Best-effort: we never raise from a
send-fallback path — a failed cleanup must not turn into a
failed user-facing send.
+
+ Rows are namespaced by profile (#76423). Under
+ ``gateway.profile_routes`` the transport adapter may not be the
+ profile that wrote the binding, so the send's ``hermes_profile``
+ metadata wins over the adapter's own profile stamp; single-profile
+ bots fall back to ``"default"``.
"""
if chat_id is None or thread_id is None:
return
@@ -1709,8 +1753,15 @@ class TelegramAdapter(BasePlatformAdapter):
if db is None or not hasattr(db, "delete_telegram_topic_binding"):
return
try:
+ profile_name = (
+ (metadata or {}).get("hermes_profile")
+ or getattr(self, "_hermes_profile_name", None)
+ or "default"
+ )
removed = db.delete_telegram_topic_binding(
- chat_id=str(chat_id), thread_id=str(thread_id),
+ chat_id=str(chat_id),
+ thread_id=str(thread_id),
+ profile_name=profile_name,
)
except Exception:
logger.debug(
@@ -5590,7 +5641,9 @@ class TelegramAdapter(BasePlatformAdapter):
self.name, effective_thread_id,
)
self._prune_stale_dm_topic_binding(
- chat_id, effective_thread_id,
+ chat_id,
+ effective_thread_id,
+ metadata=metadata,
)
used_thread_fallback = True
effective_thread_id = None
@@ -6380,7 +6433,8 @@ class TelegramAdapter(BasePlatformAdapter):
# Same prune as the streaming send path — the
# control-message retry tells us the topic is gone,
# so the binding row in state.db must go too
- # (#31501).
+ # (#31501). Control sends carry no gateway metadata, so
+ # the prune namespaces by this adapter's profile stamp.
self._prune_stale_dm_topic_binding(
kwargs.get("chat_id"), message_thread_id,
)
diff --git a/plugins/platforms/wecom/adapter.py b/plugins/platforms/wecom/adapter.py
index c26d4a8350..27fbf52e2b 100644
--- a/plugins/platforms/wecom/adapter.py
+++ b/plugins/platforms/wecom/adapter.py
@@ -63,6 +63,7 @@ except ImportError:
from gateway.config import Platform, PlatformConfig
from gateway.platforms.helpers import MessageDeduplicator
from gateway.platforms.base import (
+ gateway_trust_env,
BasePlatformAdapter,
MessageEvent,
MessageType,
@@ -317,12 +318,12 @@ class WeComAdapter(BasePlatformAdapter):
super().__init__(config, Platform.WECOM)
extra = config.extra or {}
- self._bot_id = str(extra.get("bot_id") or os.getenv("WECOM_BOT_ID", "")).strip()
+ self._bot_id = str(extra.get("bot_id") or _get_scoped_secret("WECOM_BOT_ID", "")).strip()
self._secret = str(extra.get("secret") or _get_scoped_secret("WECOM_SECRET", "")).strip()
self._ws_url = str(
extra.get("websocket_url")
or extra.get("websocketUrl")
- or os.getenv("WECOM_WEBSOCKET_URL", DEFAULT_WS_URL)
+ or _get_scoped_secret("WECOM_WEBSOCKET_URL", DEFAULT_WS_URL)
).strip() or DEFAULT_WS_URL
self._dm_policy = str(extra.get("dm_policy") or _get_scoped_secret("WECOM_DM_POLICY", "pairing")).strip().lower()
@@ -723,7 +724,7 @@ class WeComAdapter(BasePlatformAdapter):
except ImportError:
_ssl_ctx = _ssl.create_default_context()
_connector = aiohttp.TCPConnector(ssl=_ssl_ctx)
- self._session = aiohttp.ClientSession(trust_env=True, connector=_connector)
+ self._session = aiohttp.ClientSession(trust_env=gateway_trust_env(), connector=_connector)
self._ws = await self._session.ws_connect(
self._ws_url,
heartbeat=HEARTBEAT_INTERVAL_SECONDS * 2,
diff --git a/plugins/web/brave_free/provider.py b/plugins/web/brave_free/provider.py
index 769a850587..0da8d11c99 100644
--- a/plugins/web/brave_free/provider.py
+++ b/plugins/web/brave_free/provider.py
@@ -34,7 +34,7 @@ class BraveFreeWebSearchProvider(WebSearchProvider):
"""Search-only Brave provider using the free-tier Data-for-Search API.
Free tier is 2,000 queries/month (1 qps). No content-extraction capability —
- users pair this with Firecrawl/Keenable/Exa for ``web_extract``.
+ users pair this with Firecrawl/Tavily/Exa for ``web_extract``.
"""
@property
diff --git a/plugins/web/searxng/__init__.py b/plugins/web/searxng/__init__.py
index 62e12a5c7d..cea8eabb18 100644
--- a/plugins/web/searxng/__init__.py
+++ b/plugins/web/searxng/__init__.py
@@ -1,7 +1,7 @@
"""SearXNG search plugin — bundled, auto-loaded.
Backed by a user-hosted SearXNG instance (URL configured via ``SEARXNG_URL``).
-Search-only — pair with an extract provider (firecrawl/keenable/exa) for
+Search-only — pair with an extract provider (firecrawl/tavily/exa) for
``web_extract`` calls.
"""
diff --git a/plugins/web/tavily/__init__.py b/plugins/web/tavily/__init__.py
new file mode 100644
index 0000000000..1e0ced61d1
--- /dev/null
+++ b/plugins/web/tavily/__init__.py
@@ -0,0 +1,10 @@
+"""Tavily web search + extract plugin — bundled, auto-loaded."""
+
+from __future__ import annotations
+
+from plugins.web.tavily.provider import TavilyWebSearchProvider
+
+
+def register(ctx) -> None:
+ """Register the Tavily provider with the plugin context."""
+ ctx.register_web_search_provider(TavilyWebSearchProvider())
diff --git a/plugins/web/tavily/plugin.yaml b/plugins/web/tavily/plugin.yaml
new file mode 100644
index 0000000000..3ac90594e5
--- /dev/null
+++ b/plugins/web/tavily/plugin.yaml
@@ -0,0 +1,7 @@
+name: web-tavily
+version: 1.0.0
+description: "Tavily web search + extract. Opt-in keyless via hermes tools; set TAVILY_API_KEY for higher limits — https://app.tavily.com/home."
+author: NousResearch
+kind: backend
+provides_web_providers:
+ - tavily
diff --git a/plugins/web/tavily/provider.py b/plugins/web/tavily/provider.py
new file mode 100644
index 0000000000..621aa3ea03
--- /dev/null
+++ b/plugins/web/tavily/provider.py
@@ -0,0 +1,313 @@
+"""Tavily web search + content extraction — plugin form.
+
+Subclasses :class:`agent.web_search_provider.WebSearchProvider`. Two
+capabilities advertised:
+
+- ``supports_search()`` -> True (Tavily ``/search``)
+- ``supports_extract()`` -> True (Tavily ``/extract``)
+
+Both are sync — the underlying call is ``httpx.post(...)``.
+
+Config keys this provider responds to::
+
+ web:
+ search_backend: "tavily" # explicit per-capability
+ extract_backend: "tavily" # explicit per-capability
+ backend: "tavily" # shared fallback for both
+
+Env vars::
+
+ TAVILY_API_KEY=... # https://app.tavily.com/home (optional)
+ TAVILY_BASE_URL=... # optional override of https://api.tavily.com
+
+Auth is header-based. A key uses ``Authorization: Bearer``; without a
+key the request is keyless (``X-Tavily-Access-Mode: keyless``). Both
+paths send ``X-Client-Name: hermes-agent``.
+
+Tavily is **not** a member of the zero-config keyless ring
+(``plugins.web.keyless_mcp._KEYLESS_RING``). Keyless access is opt-in:
+select Tavily in ``hermes tools`` (or set ``web.backend: tavily``).
+Fresh installs with no web credentials rotate across Exa / Parallel /
+Firecrawl / Keenable instead.
+"""
+
+from __future__ import annotations
+
+import logging
+from typing import Any, Dict, List, Optional
+
+import httpx
+
+from agent.web_search_provider import WebSearchProvider
+
+logger = logging.getLogger(__name__)
+
+_CLIENT_NAME = "hermes-agent"
+
+_SEARCH_PAYLOAD = {
+ "include_raw_content": False,
+ "include_images": False,
+}
+
+
+def _tavily_headers(api_key: str) -> Dict[str, str]:
+ """Build Tavily request headers for keyed or keyless access."""
+ headers = {"X-Client-Name": _CLIENT_NAME}
+ if api_key:
+ headers["Authorization"] = f"Bearer {api_key}"
+ else:
+ headers["X-Tavily-Access-Mode"] = "keyless"
+ return headers
+
+
+def _tavily_request(
+ endpoint: str,
+ payload: Dict[str, Any],
+ *,
+ api_key: Optional[str] = None,
+) -> Dict[str, Any]:
+ """POST to the Tavily API and return the parsed JSON response.
+
+ Keyed when *api_key* (or ``TAVILY_API_KEY``) is set (Bearer auth);
+ otherwise keyless. Pass ``api_key=""`` to force the keyless header even
+ when a key is present (``web.provider_tier.tavily: free``). Non-2xx
+ responses raise ``ValueError`` with the response body so Tavily's
+ keyless rate-limit / upgrade text reaches the model.
+ """
+ from agent.web_search_provider import get_provider_env
+
+ if api_key is None:
+ api_key = get_provider_env("TAVILY_API_KEY")
+ base_url = get_provider_env("TAVILY_BASE_URL") or "https://api.tavily.com"
+ url = f"{base_url}/{endpoint.lstrip('/')}"
+ logger.info("Tavily %s request to %s", endpoint, url)
+
+ response = httpx.post(
+ url,
+ json=payload,
+ timeout=60,
+ headers=_tavily_headers(api_key),
+ )
+ if response.status_code >= 400:
+ body = (response.text or "").strip()
+ detail = body or f"HTTP {response.status_code}"
+ raise ValueError(detail)
+ return response.json()
+
+
+def _normalize_tavily_search_results(response: Dict[str, Any]) -> Dict[str, Any]:
+ """Map Tavily ``/search`` response to ``{success, data: {web: [...]}}``."""
+ web_results = []
+ for i, result in enumerate(response.get("results", [])):
+ web_results.append(
+ {
+ "title": result.get("title", ""),
+ "url": result.get("url", ""),
+ "description": result.get("content", ""),
+ "position": i + 1,
+ }
+ )
+ return {"success": True, "data": {"web": web_results}}
+
+
+def _normalize_tavily_documents(
+ response: Dict[str, Any], fallback_url: str = ""
+) -> List[Dict[str, Any]]:
+ """Map Tavily ``/extract`` response to standard documents.
+
+ Documents follow the legacy LLM post-processing shape::
+
+ {"url", "title", "content", "raw_content", "metadata"}
+
+ Failures (``failed_results``, ``failed_urls``) become result entries
+ with an ``error`` field rather than raising.
+ """
+ documents: List[Dict[str, Any]] = []
+ for result in response.get("results", []):
+ url = result.get("url", fallback_url)
+ raw = result.get("raw_content", "") or result.get("content", "")
+ documents.append(
+ {
+ "url": url,
+ "title": result.get("title", ""),
+ "content": raw,
+ "raw_content": raw,
+ "metadata": {"sourceURL": url, "title": result.get("title", "")},
+ }
+ )
+ for fail in response.get("failed_results", []):
+ documents.append(
+ {
+ "url": fail.get("url", fallback_url),
+ "title": "",
+ "content": "",
+ "raw_content": "",
+ "error": fail.get("error", "extraction failed"),
+ "metadata": {"sourceURL": fail.get("url", fallback_url)},
+ }
+ )
+ for fail_url in response.get("failed_urls", []):
+ url_str = fail_url if isinstance(fail_url, str) else str(fail_url)
+ documents.append(
+ {
+ "url": url_str,
+ "title": "",
+ "content": "",
+ "raw_content": "",
+ "error": "extraction failed",
+ "metadata": {"sourceURL": url_str},
+ }
+ )
+ return documents
+
+
+def _missing_key_error(action: str) -> str:
+ return (
+ f"TAVILY_API_KEY is not set. Get a key at https://app.tavily.com/home "
+ f"or select Tavily in `hermes tools` for opt-in keyless {action}."
+ )
+
+
+class TavilyWebSearchProvider(WebSearchProvider):
+ """Tavily search + extract provider (keyed, or opt-in keyless)."""
+
+ @property
+ def name(self) -> str:
+ return "tavily"
+
+ @property
+ def display_name(self) -> str:
+ return "Tavily"
+
+ def is_available(self) -> bool:
+ """Return True when ``TAVILY_API_KEY`` is set to a non-empty value."""
+ from agent.web_search_provider import get_provider_env
+
+ return bool(get_provider_env("TAVILY_API_KEY"))
+
+ def is_keyless_available(self) -> bool:
+ """Tavily serves anonymous keyless requests (X-Tavily-Access-Mode).
+
+ Opt-in only — Tavily is not a member of the zero-config keyless
+ ring. ``is_keyless_available`` is True so an explicit
+ ``web.backend: tavily`` (or ``hermes tools`` pick) works without a
+ key. False when the user pinned ``web.provider_tier.tavily: paid``.
+ """
+ from plugins.web.keyless_mcp import keyless_enabled, provider_tier
+
+ return keyless_enabled() and provider_tier("tavily") != "paid"
+
+ def supports_search(self) -> bool:
+ return True
+
+ def supports_extract(self) -> bool:
+ return True
+
+ def search(self, query: str, limit: int = 5) -> Dict[str, Any]:
+ """Execute a Tavily search (keyed path or opt-in keyless)."""
+ try:
+ from tools.interrupt import is_interrupted
+
+ if is_interrupted():
+ return {"success": False, "error": "Interrupted"}
+
+ from agent.web_search_provider import get_provider_env
+
+ from plugins.web.keyless_mcp import use_keyless
+
+ api_key = get_provider_env("TAVILY_API_KEY")
+ force_keyless = use_keyless("tavily", api_key)
+ if not force_keyless and not api_key:
+ return {"success": False, "error": _missing_key_error("search")}
+
+ logger.info(
+ "Tavily %ssearch: '%s' (limit=%d)",
+ "keyless " if force_keyless else "",
+ query,
+ limit,
+ )
+ raw = _tavily_request(
+ "search",
+ {
+ "query": query,
+ "max_results": min(limit, 20),
+ **_SEARCH_PAYLOAD,
+ },
+ api_key="" if force_keyless else api_key,
+ )
+ return _normalize_tavily_search_results(raw)
+ except ValueError as exc:
+ return {"success": False, "error": str(exc)}
+ except Exception as exc: # noqa: BLE001 — including httpx errors
+ logger.warning("Tavily search error: %s", exc)
+ return {"success": False, "error": f"Tavily search failed: {exc}"}
+
+ def extract(self, urls: List[str], **kwargs: Any) -> List[Dict[str, Any]]:
+ """Extract content from one or more URLs via Tavily.
+
+ Sync — the underlying call is httpx.post(...). Returns the legacy
+ list-of-results shape; per-URL failures become items with ``error``.
+ Keyless uses Tavily's own endpoint, not the keyless ring.
+ """
+ try:
+ from tools.interrupt import is_interrupted
+
+ if is_interrupted():
+ return [
+ {"url": u, "error": "Interrupted", "title": ""} for u in urls
+ ]
+
+ from agent.web_search_provider import get_provider_env
+
+ from plugins.web.keyless_mcp import use_keyless
+
+ api_key = get_provider_env("TAVILY_API_KEY")
+ force_keyless = use_keyless("tavily", api_key)
+ if not force_keyless and not api_key:
+ err = _missing_key_error("extract")
+ return [
+ {"url": u, "title": "", "content": "", "error": err}
+ for u in urls
+ ]
+
+ logger.info(
+ "Tavily %sextract: %d URL(s)",
+ "keyless " if force_keyless else "",
+ len(urls),
+ )
+ raw = _tavily_request(
+ "extract",
+ {
+ "urls": urls,
+ "include_images": False,
+ },
+ api_key="" if force_keyless else api_key,
+ )
+ return _normalize_tavily_documents(
+ raw, fallback_url=urls[0] if urls else ""
+ )
+ except ValueError as exc:
+ return [{"url": u, "title": "", "content": "", "error": str(exc)} for u in urls]
+ except Exception as exc: # noqa: BLE001
+ logger.warning("Tavily extract error: %s", exc)
+ return [
+ {"url": u, "title": "", "content": "", "error": f"Tavily extract failed: {exc}"}
+ for u in urls
+ ]
+
+ def get_setup_schema(self) -> Dict[str, Any]:
+ return {
+ "name": "Tavily",
+ "badge": "free · key optional",
+ "tag": (
+ "Search + extract. Opt-in keyless; "
+ "set TAVILY_API_KEY for higher limits."
+ ),
+ "env_vars": [
+ {
+ "key": "TAVILY_API_KEY",
+ "prompt": "Tavily API key (optional — keyless works when Tavily is selected)",
+ "url": "https://app.tavily.com/home",
+ },
+ ],
+ }
diff --git a/plugins/web/xai/provider.py b/plugins/web/xai/provider.py
index 922b9856be..77d80a4398 100644
--- a/plugins/web/xai/provider.py
+++ b/plugins/web/xai/provider.py
@@ -101,12 +101,12 @@ class XAIWebSearchProvider(WebSearchProvider):
back to the Responses API ``citations`` list if Grok ignores the JSON
schema instruction (rare for grok-4.3 but cheap insurance).
- No extract capability — pair with Firecrawl / Keenable / Exa for
+ No extract capability — pair with Firecrawl / Tavily / Exa for
``web_extract`` if you need page content.
Trust model
-----------
- Unlike index-backed providers (Brave / Keenable / Exa) which return
+ Unlike index-backed providers (Brave / Tavily / Exa) which return
verbatim search-engine results, this backend is an LLM in a trench
coat: Grok decides which URLs to surface, generates the titles and
descriptions itself, and is influenced by the *content of the query*.
diff --git a/pyproject.toml b/pyproject.toml
index f455c40ed5..fbd324439c 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -587,7 +587,7 @@ py-modules = [
include = ["agent", "agent.*", "tools", "tools.*", "hermes_cli", "hermes_cli.*", "gateway", "gateway.*", "tui_gateway", "tui_gateway.*", "cron", "cron.*", "acp_adapter", "plugins", "plugins.*", "providers", "providers.*"]
[tool.setuptools.package-data]
-hermes_cli = ["observability/schemas/*.json", "data/*.json"]
+hermes_cli = ["observability/schemas/*.json", "data/*.json", "local_runtime/*.json"]
# gateway/assets/ ships status_phrases.yaml and the Telegram BotFather
# screenshot. Without this, sealed venvs (uv2nix) silently lose both —
# status phrases fall back to the tiny hardcoded set and the Telegram
diff --git a/run_agent.py b/run_agent.py
index 261945efd2..b3d2bf89cf 100644
--- a/run_agent.py
+++ b/run_agent.py
@@ -876,6 +876,7 @@ class AIAgent:
# transcript — a fresh/branched/resumed session must fall back to
# full estimation until its first provider response re-anchors.
self._usage_anchor = None
+ self._turn_base_usage_anchor = None
# Turn counter (added after reset_session_state was first written — #2635)
self._user_turn_count = 0
@@ -1884,12 +1885,26 @@ class AIAgent:
(LiteLLM/sglang/vLLM/LM Studio proxies, Tailscale boxes), which
report finish_reason correctly and were the source of #13971's
false-positive truncation continuations.
+
+ Also excludes Ollama Cloud — the hosted service correctly reports
+ finish_reason and is not affected by the local Ollama stop-reason
+ bug (GH-72316). Two signatures identify it: the ``ollama.com`` host
+ (provider ``ollama-cloud``) and the ``:cloud`` model suffix (cloud
+ generation proxied through a local 11434 endpoint, #98406). Applying
+ the stop→length rewrite to them manufactures false truncations and
+ causes the continuation nudge to consume the model's output budget
+ on the next retry, making further false-positives more likely.
"""
model_lower = (self.model or "").lower()
provider_lower = (self.provider or "").lower()
if "glm" not in model_lower and provider_lower != "zai":
return False
- if "ollama" in self._base_url_lower or ":11434" in self._base_url_lower:
+ base = self._base_url_lower
+ # Ollama Cloud (hosted service or :cloud proxy) forwards finish_reason
+ # faithfully — do not rewrite.
+ if "ollama.com" in base or ":cloud" in model_lower:
+ return False
+ if "ollama" in base or ":11434" in base:
return True
return provider_lower == "ollama"
@@ -1976,6 +1991,71 @@ class AIAgent:
review_memory: bool = False,
review_skills: bool = False,
focus: Optional[str] = None,
+ explicit: bool = False,
+ ) -> None:
+ """Post-turn review entry point: decide WHEN, then spawn.
+
+ The decision to review (nudge intervals, enabled gate) already
+ happened at the call site. This wrapper adds one policy: a review
+ whose runtime resolves to the MANAGED LOCAL llama-server is queued
+ for machine idle instead of spawned into the user's GPU mid-session
+ (auxiliary.background_review.defer: auto|never). Everything else —
+ cloud runtimes, external local servers, explicit /refine — spawns
+ immediately, exactly as before.
+
+ ``explicit`` marks a user-initiated review (/refine, with or
+ without focus text): never deferred. It does NOT touch the
+ delegate/enabled gates below — those stay keyed on ``focus`` so a
+ bare /refine keeps its historical gating behavior.
+ """
+ # Delegation-subagent and enabled gates run here at enqueue/spawn
+ # time; the idle dispatcher re-checks the enabled gate again at
+ # dispatch time so a review queued for minutes cannot be
+ # resurrected after the user disables reviews.
+ if focus is None and getattr(self, "_delegate_depth", 0) > 0:
+ return
+ task_cfg = None
+ if focus is None:
+ from agent.background_review import load_background_review_settings
+ enabled, task_cfg = load_background_review_settings()
+ if not enabled:
+ return
+
+ # Structural clone at the single chokepoint every review path
+ # (automatic, /refine, idle-queue deferral) goes through. The fork
+ # sanitizes its transcript in place; a shallow copy would alias the
+ # nested tool_calls/content containers of the live history (#100795).
+ from agent.turn_finalizer import _clone_background_review_messages
+ messages_snapshot = _clone_background_review_messages(messages_snapshot)
+
+ kwargs = dict(
+ messages_snapshot=messages_snapshot,
+ review_memory=review_memory,
+ review_skills=review_skills,
+ focus=focus,
+ task_cfg=task_cfg,
+ )
+ if focus is None and not explicit:
+ from agent.review_idle_queue import (
+ QUEUE,
+ defer_mode,
+ review_targets_managed_local,
+ )
+ if (defer_mode(task_cfg) == "auto"
+ and review_targets_managed_local(self, task_cfg)):
+ session_key = str(getattr(self, "session_id", None) or id(self))
+ QUEUE.enqueue(self, session_key, kwargs)
+ return
+ self._spawn_background_review_now(**kwargs)
+
+ def _spawn_background_review_now(
+ self,
+ messages_snapshot: List[Dict],
+ review_memory: bool = False,
+ review_skills: bool = False,
+ focus: Optional[str] = None,
+ task_cfg: Optional[Dict[str, Any]] = None,
+ _requeue_attempts: int = 0,
) -> None:
"""Spawn the background memory/skill review thread.
@@ -1988,28 +2068,17 @@ class AIAgent:
``focus`` is optional user-supplied steering (from ``/refine``)
appended to the review prompt — e.g. "save the deploy workflow as a
skill". The automatic post-turn triggers never set it.
+
+ ``task_cfg`` is the pre-loaded ``auxiliary.background_review``
+ block from the entry wrapper (None on direct calls, e.g. /refine —
+ the spawn path reads config itself then).
+
+ A deferred review preempted by a live turn is REQUEUED (bounded by
+ ``_requeue_attempts``) instead of lost: on the managed local
+ runtime a review takes minutes, so cancel-and-forget — harmless on
+ cloud, where reviews finish in seconds — would silently discard
+ most learning on an active session.
"""
- # A delegation subagent (``_delegate_depth > 0``) must not run the
- # automatic post-turn review. Subagents are ephemeral workers already
- # barred from writing shared MEMORY.md (``DELEGATE_BLOCKED_TOOLS``) and
- # are spawned with ``skip_memory=True``, so a review here has little to
- # persist — yet it inherits the subagent's (often premium) delegation
- # model and replays the whole conversation at premium rates, silently
- # inflating token cost (#85859). An explicit ``/refine`` (``focus`` set)
- # is a deliberate user request and still runs.
- if focus is None and getattr(self, "_delegate_depth", 0) > 0:
- return
- # Explicit off-switch for automatic post-turn forks
- # (``auxiliary.background_review.enabled: false``). Manual ``/refine``
- # still works — same contract as zeroing the nudge intervals (#87250).
- # Load the task block once here and pass it into the spawn path so
- # aux routing does not re-read config.
- task_cfg = None
- if focus is None:
- from agent.background_review import load_background_review_settings
- enabled, task_cfg = load_background_review_settings()
- if not enabled:
- return
from agent.background_review import (
finish_background_review_run,
prepare_background_review_run,
@@ -2030,10 +2099,25 @@ class AIAgent:
task_cfg=task_cfg,
review_run=review_run,
)
+
+ def _target_with_requeue() -> None:
+ target()
+ self._maybe_requeue_preempted_review(
+ review_run,
+ dict(
+ messages_snapshot=messages_snapshot,
+ review_memory=review_memory,
+ review_skills=review_skills,
+ focus=focus,
+ task_cfg=task_cfg,
+ _requeue_attempts=_requeue_attempts + 1,
+ ),
+ )
+
# Carry the active profile into the review thread so MEMORY.md /
# skill review writes land in the right profile (#54937).
t = threading.Thread(
- target=propagate_context_to_thread(target),
+ target=propagate_context_to_thread(_target_with_requeue),
daemon=True,
name="bg-review",
)
@@ -2042,6 +2126,42 @@ class AIAgent:
finish_background_review_run(self, review_run)
raise
+ _REVIEW_REQUEUE_MAX_ATTEMPTS = 3
+
+ def _maybe_requeue_preempted_review(self, review_run, kwargs) -> None:
+ """Requeue a deferred-mode review that a live turn cancelled.
+
+ Only fires for automatic reviews whose runtime targets the managed
+ local server (the deferred population); bounded attempts prevent a
+ busy box from cycling one review forever — past the cap it is
+ dropped exactly like the pre-deferral behavior dropped every
+ cancelled review.
+ """
+ try:
+ if not review_run.cancel_requested.is_set():
+ return # ran to completion (or never admitted for other reasons)
+ if kwargs.get("focus") is not None:
+ return
+ if kwargs.get("_requeue_attempts", 0) > self._REVIEW_REQUEUE_MAX_ATTEMPTS:
+ logger.info("Preempted background review dropped after %d requeues",
+ self._REVIEW_REQUEUE_MAX_ATTEMPTS)
+ return
+ from agent.review_idle_queue import (
+ QUEUE,
+ defer_mode,
+ review_targets_managed_local,
+ )
+ task_cfg = kwargs.get("task_cfg")
+ if (defer_mode(task_cfg) != "auto"
+ or not review_targets_managed_local(self, task_cfg)):
+ return
+ session_key = str(getattr(self, "session_id", None) or id(self))
+ # kwargs carries the incremented _requeue_attempts through the
+ # queue so the cap survives the round trip.
+ QUEUE.enqueue(self, session_key, dict(kwargs))
+ except Exception: # noqa: BLE001 — requeue is best-effort
+ logger.debug("Preempted-review requeue failed", exc_info=True)
+
def _build_memory_write_metadata(
self,
*,
@@ -2550,13 +2670,17 @@ class AIAgent:
# ("storage was busy, send it again") from disk-full/read-only.
from hermes_state import (
CompressionSessionClosedError,
+ StateDbCorruptError,
StateDbReplacedError,
classify_persistence_error,
divert_session_transcript_jsonl,
)
self._last_persistence_error_cause = classify_persistence_error(e)
- if isinstance(e, StateDbReplacedError):
+ if isinstance(e, (StateDbReplacedError, StateDbCorruptError)):
+ # Replaced generation or quarantined (structurally corrupt)
+ # handle: SQLite will not take this batch again, so keep it
+ # on disk instead of only in RAM.
try:
divert_session_transcript_jsonl(
getattr(self, "session_id", "") or "",
@@ -2564,7 +2688,8 @@ class AIAgent:
)
except Exception:
logger.warning(
- "JSONL divert failed after state.db replace for %s",
+ "JSONL divert failed after state.db %s for %s",
+ self._last_persistence_error_cause,
getattr(self, "session_id", None),
exc_info=True,
)
@@ -5599,6 +5724,75 @@ class AIAgent:
exc,
)
+ def _drain_transports_after_abandonment(self, *, reason: str) -> int:
+ """FD-safe transport drain for an abandoned (timed-out) worker (#94248).
+
+ A delegation deadline abandons this agent's daemon worker while it may
+ still be blocked inside an in-flight OpenSSL ``read`` (Codex Responses
+ stream, httpx request). The timeout thread must never hard-close those
+ transports — ``client.close()`` releases raw FDs under a live SSL BIO,
+ the #29507 / #67142 / #70773 native-corruption family and the SIGSEGV
+ shape reported in #94248. This helper only ``shutdown()``s pooled
+ sockets (safe from any thread), settling blocked reads with EOF/EPIPE
+ so the worker can unwind and run the real close from its own thread.
+
+ Returns the number of sockets shut down across all transports.
+ """
+ drained = 0
+ # Shared primary client (codex-direct / MoA stream on it directly).
+ try:
+ client = getattr(self, "client", None)
+ if client is not None:
+ drained += self._force_close_tcp_sockets(client)
+ except Exception:
+ logger.debug("Abandoned-worker drain: shared client sweep failed",
+ exc_info=True)
+ # Cached per-request wire clients: abort (shutdown + poison the reuse
+ # slot) so the unwinding worker discards them instead of re-caching.
+ try:
+ with self._openai_client_lock():
+ cache = getattr(self, "_request_client_cache", None)
+ cached = cache["client"] if cache else None
+ if cached is not None:
+ self._abort_request_openai_client(cached, reason=reason)
+ except Exception:
+ logger.debug("Abandoned-worker drain: request client abort failed",
+ exc_info=True)
+ try:
+ with self._openai_client_lock():
+ cache = getattr(self, "_request_anthropic_client_cache", None)
+ cached = cache["client"] if cache else None
+ if cached is not None:
+ self._abort_request_anthropic_client(cached, reason=reason)
+ except Exception:
+ logger.debug("Abandoned-worker drain: anthropic client abort failed",
+ exc_info=True)
+ # Codex app-server session watches a private interrupt event.
+ try:
+ codex_session = getattr(self, "_codex_session", None)
+ request_interrupt = getattr(codex_session, "request_interrupt", None)
+ if callable(request_interrupt):
+ request_interrupt()
+ except Exception:
+ logger.debug("Abandoned-worker drain: codex interrupt failed",
+ exc_info=True)
+ # Inline (cron-style) request abort hook, when registered.
+ try:
+ abort_active = getattr(self, "_active_request_abort", None)
+ if callable(abort_active):
+ abort_active(reason)
+ except Exception:
+ logger.debug("Abandoned-worker drain: active request abort failed",
+ exc_info=True)
+ logger.info(
+ "Abandoned-worker transports drained (%s, tcp_shutdown=%d, "
+ "fd_release=deferred_to_worker) %s",
+ reason,
+ drained,
+ self._client_log_context(),
+ )
+ return drained
+
def _build_primary_client_for_active_provider(self, *, reason: str) -> Any:
"""Build the shared client shape required by the active provider.
@@ -8352,6 +8546,7 @@ class AIAgent:
task_id: str = "default",
focus_topic: str = None,
force: bool = False,
+ bypass_cooldown: bool = False,
defer_context_engine_notification: bool = False,
commit_fence=None,
) -> tuple:
@@ -8360,7 +8555,9 @@ class AIAgent:
``force=True`` is passed by the manual ``/compress`` slash command
so users can bypass the summary-failure cooldown after an
auto-compress abort. Auto-compress callers use the default
- ``force=False``.
+ ``force=False``. ``bypass_cooldown=True`` is passed by the
+ provider-proven overflow recovery path so one real attempt runs while
+ the cooldown is armed (#100661) — without clearing it.
"""
# Per-attempt signal consumed by turn-start preflight (#98424) and the
# in-loop pre-API/overflow consumers. A stalled compression must not
@@ -8441,6 +8638,7 @@ class AIAgent:
approx_tokens=approx_tokens, task_id=task_id,
focus_topic=focus_topic,
force=force,
+ bypass_cooldown=bypass_cooldown,
defer_context_engine_notification=(
defer_context_engine_notification
),
@@ -8821,6 +9019,13 @@ class AIAgent:
function_result = append_toolguard_guidance(function_result, decision)
if decision.should_halt:
self._set_tool_guardrail_halt(decision)
+ else:
+ # observe_call may have raised the identical-call streak halt
+ # (hard_stop_enabled, tool-agnostic) — surface it the same way.
+ streak_halt = self._tool_guardrails.halt_decision
+ if streak_halt is not None and streak_halt.code == "identical_call_streak_halt":
+ function_result = append_toolguard_guidance(function_result, streak_halt)
+ self._set_tool_guardrail_halt(streak_halt)
if stall_notice:
function_result = (function_result or "") + "\n\n" + stall_notice
return function_result
@@ -9026,6 +9231,13 @@ class AIAgent:
cancel_background_review_for_live_turn(self)
+ # Turn liveness for the deferred-review idle queue: a queued review
+ # must not dispatch into the settle gap between two quick prompts.
+ # Marked inside the try below so the balancing note_turn_finished in
+ # its finally covers every exit; the actual start-mark happens as the
+ # first statement of the try.
+ from agent.review_idle_queue import QUEUE as _review_queue
+
from agent.aux_accounting import (
reset_accounting_context,
set_accounting_context,
@@ -9111,6 +9323,7 @@ class AIAgent:
_clear_if_owned()
try:
+ _review_queue.note_turn_started()
# Serialize the full load -> run -> flush region across Hermes
# processes. Gateway's asyncio lease closes alias routing inside one
# process; this durable lease covers Desktop, CLI resume, gateway,
@@ -9657,6 +9870,13 @@ class AIAgent:
reset_conversation_context(token)
if affinity_token is not None:
reset_affinity_scope(affinity_token)
+ # Balance the note_turn_started above — every exit path
+ # lands here, so the idle queue's live-turn count cannot
+ # leak upward and starve deferred reviews.
+ try:
+ _review_queue.note_turn_finished()
+ except Exception:
+ pass
def chat(self, message: str, stream_callback: Optional[callable] = None) -> str:
"""
diff --git a/scripts/aa_quality_sync.py b/scripts/aa_quality_sync.py
new file mode 100644
index 0000000000..66e61c7f7d
--- /dev/null
+++ b/scripts/aa_quality_sync.py
@@ -0,0 +1,94 @@
+"""Propose catalog quality updates from Artificial Analysis.
+
+Authoring-time helper — NEVER called at runtime (their terms forbid
+client-side keys, the fleet would burn the rate limit, and a
+recommendation must not change because a third-party endpoint
+hiccuped). Run it when adding a model or refreshing the ordering;
+review the printed diff and edit catalog.json yourself. The script
+proposes, the commit decides.
+
+The catalog's `quality` stays OUR field: AA-informed where they cover a
+model, editorially set where they don't (day-0 releases lag their evals;
+some entries never appear). AA's Intelligence Index grades the
+full-precision cloud model, not our Q4 build — fine for ordering, never
+for display.
+
+Usage:
+ export AA_API_KEY=... # from https://artificialanalysis.ai (free tier)
+ python scripts/aa_quality_sync.py
+
+Attribution: scores by Artificial Analysis (https://artificialanalysis.ai).
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import sys
+import urllib.request
+from pathlib import Path
+
+REPO_ROOT = Path(__file__).resolve().parent.parent
+CATALOG_PATH = REPO_ROOT / "hermes_cli" / "local_runtime" / "catalog.json"
+AA_URL = "https://artificialanalysis.ai/api/v2/data/llms/models"
+
+# Catalog entry id -> AA slug. Hand-maintained: AA's naming rarely matches
+# HF repo names, and a wrong match silently mis-ranks a model. An entry
+# absent here (or mapped to None) is editorial-only and never overwritten.
+AA_SLUG_BY_ENTRY = {
+ "qwen3.8-27b": "qwen3-8-27b",
+ "qwen3.8-flash-next": "qwen3-8-flash-next",
+ "qwen3.6-35b-a3b": "qwen3-6-35b-a3b",
+ "deepseek-v4-flash": "deepseek-v4-flash",
+}
+
+
+def fetch_aa_models(api_key: str) -> dict[str, dict]:
+ req = urllib.request.Request(AA_URL, headers={"x-api-key": api_key})
+ with urllib.request.urlopen(req, timeout=30) as r:
+ doc = json.load(r)
+ return {m["slug"]: m for m in doc.get("data", [])}
+
+
+def main() -> int:
+ api_key = os.environ.get("AA_API_KEY", "").strip()
+ if not api_key:
+ print("AA_API_KEY not set — create a free key at "
+ "https://artificialanalysis.ai and export it.", file=sys.stderr)
+ return 2
+
+ catalog = json.loads(CATALOG_PATH.read_text(encoding="utf-8"))
+ aa = fetch_aa_models(api_key)
+
+ print(f"{'entry':24s} {'catalog q':>9s} {'AA index':>9s} note")
+ print("-" * 70)
+ for model in catalog["models"]:
+ entry_id = model["id"]
+ current = model.get("quality", 0)
+ slug = AA_SLUG_BY_ENTRY.get(entry_id)
+ if not slug:
+ print(f"{entry_id:24s} {current:>9d} {'—':>9s} editorial only (no AA mapping)")
+ continue
+ hit = aa.get(slug)
+ if hit is None:
+ print(f"{entry_id:24s} {current:>9d} {'—':>9s} not in AA data (slug {slug!r})")
+ continue
+ index = (hit.get("evaluations") or {}).get(
+ "artificial_analysis_intelligence_index")
+ if index is None:
+ print(f"{entry_id:24s} {current:>9d} {'—':>9s} AA row lacks the index")
+ continue
+ proposed = round(float(index))
+ marker = "" if proposed == current else " <-- proposes change"
+ print(f"{entry_id:24s} {current:>9d} {proposed:>9d}{marker}")
+
+ print("\nReview against the decision table before editing: a quality "
+ "change that flips cells in tests/hermes_cli/"
+ "test_local_recommendation.py is the actual decision being made.")
+ print("Attribution: scores by Artificial Analysis "
+ "(https://artificialanalysis.ai).")
+ return 0
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/scripts/add_contributor.py b/scripts/add_contributor.py
index cb64b331cb..6e408b2a83 100644
--- a/scripts/add_contributor.py
+++ b/scripts/add_contributor.py
@@ -55,6 +55,24 @@ def _legacy_login(email: str) -> str | None:
return None
+def _case_collision(email: str) -> str | None:
+ """An existing mapping whose filename differs from `email` only in case.
+
+ Returns the colliding filename, or None. Exact matches are not collisions --
+ that is the ordinary "already mapped" path handled by the caller.
+ """
+ if not EMAILS_DIR.is_dir():
+ return None
+
+ # casefold (not lower) matches how macOS/Windows fold non-ASCII text —
+ # same key scripts/check-case-collisions.py uses repo-wide.
+ folded = email.casefold()
+ for entry in EMAILS_DIR.iterdir():
+ if entry.name != email and entry.name.casefold() == folded:
+ return entry.name
+ return None
+
+
def add_contributor(email: str, login: str, comment: str = "") -> int:
email = email.strip()
login = login.strip().lstrip("@")
@@ -67,6 +85,23 @@ def add_contributor(email: str, login: str, comment: str = "") -> int:
return 2
path = EMAILS_DIR / email
+
+ # One file per email means the FILENAME is the key, and on a
+ # case-insensitive filesystem (Windows, default macOS) two emails differing
+ # only in case are the same file. Creating both makes the repo impossible to
+ # check out cleanly there -- `git status` reports a phantom modification
+ # forever, because whichever file git wrote second wins on disk. Refuse for
+ # the same reason a conflicting login is refused: resolve it deliberately.
+ collision = _case_collision(email)
+ if collision is not None:
+ print(
+ f"error: {email} collides with existing mapping {collision} on "
+ "case-insensitive filesystems (Windows/macOS) — the two are the same "
+ "file there. Reuse that mapping, or resolve manually.",
+ file=sys.stderr,
+ )
+ return 1
+
existing = read_mapping_file(path) if path.is_file() else None
if existing is None:
existing = _legacy_login(email)
diff --git a/scripts/check-case-collisions.py b/scripts/check-case-collisions.py
new file mode 100644
index 0000000000..0ef0becfb8
--- /dev/null
+++ b/scripts/check-case-collisions.py
@@ -0,0 +1,114 @@
+#!/usr/bin/env python3
+"""
+Blocking check for tracked files whose paths collide when case is ignored.
+
+Linux is case-sensitive; Windows and macOS (default) are not. Two tracked
+paths that differ only by case — ``README.md`` and ``readme.md``, or
+``src/Foo.py`` and ``SRC/foo.py`` — coexist happily in a Linux checkout and
+silently break every clone on a case-insensitive host: the filesystem can
+hold only one of them, so checkout either refuses or whichever file is
+written last wins and clobbers the other. Git itself won't stop the pair
+from landing — it only warns at checkout time, on a case-insensitive FS,
+for whichever client happens to do the checkout, and the collision is
+invisible on Linux. This check is the enforcement point: scan the index,
+fail the build, name the offenders.
+
+Usage:
+ # Check the checkout this script lives in (CI + the common local case)
+ python scripts/check-case-collisions.py
+
+ # Check an arbitrary git checkout (tests, other worktrees)
+ python scripts/check-case-collisions.py /path/to/other/repo
+
+Exit status:
+ 0 — no case-colliding tracked paths
+ 1 — at least one collision group (paths printed to stdout)
+ 2 — not in a git repository / git failed
+
+Comparison key: the casefolded FULL path (``str.casefold``), not the
+basename — on a case-insensitive filesystem the entire path is
+case-insensitive, so ``dir/Foo.txt`` and ``DIR/foo.txt`` collide just like
+same-directory pairs. ``casefold`` (not ``lower``) is used because it
+matches how the OSes fold case for non-ASCII text (straße vs strasse,
+sigma variants); a pair it flags is a genuine collision on macOS/Windows
+even when Linux disagrees.
+
+Deliberately out of scope: Unicode NFC/NFD normalization collisions (macOS
+stores NFD, Linux NFC). git already handles those at checkout via
+``core.precomposeunicode``; this check is strictly about case.
+"""
+
+from __future__ import annotations
+
+import argparse
+import os
+import subprocess
+import sys
+from collections import defaultdict
+from pathlib import Path
+
+REPO_ROOT = Path(__file__).resolve().parent.parent
+
+
+def main() -> int:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument(
+ "root",
+ nargs="?",
+ default=str(REPO_ROOT),
+ help="git checkout to scan (default: the repo this script lives in)",
+ )
+ args = parser.parse_args()
+
+ try:
+ os.chdir(args.root)
+ except OSError as exc:
+ print(f"::error::cannot enter {args.root}: {exc}")
+ return 2
+
+ proc = subprocess.run(["git", "ls-files", "-z"], capture_output=True)
+ if proc.returncode != 0:
+ msg = proc.stderr.decode("utf-8", errors="replace").strip()
+ print(f"::error::git ls-files failed in {args.root}: {msg}")
+ return 2
+
+ paths = [
+ p.decode("utf-8", errors="surrogateescape")
+ for p in proc.stdout.split(b"\0")
+ if p
+ ]
+
+ by_casefold: dict[str, list[str]] = defaultdict(list)
+ for path in paths:
+ by_casefold[path.casefold()].append(path)
+
+ collisions = {key: group for key, group in by_casefold.items() if len(group) > 1}
+
+ if not collisions:
+ print(f"::notice::{len(paths)} tracked files, no case-colliding paths.")
+ return 0
+
+ print(
+ f"::error::Found {len(collisions)} case-collision group(s) among "
+ f"{len(paths)} tracked files."
+ )
+ print(
+ "Paths that differ only by case are ONE file on Windows/macOS but "
+ "several on Linux - the pair breaks every clone on a case-insensitive "
+ "host. Rename one member of each group so the paths differ beyond case."
+ )
+ print()
+ for key, group in sorted(collisions.items()):
+ for path in sorted(group):
+ print(f" {path}")
+ print()
+ print(
+ "Fix: `git mv` one path in each group to a name that doesn't collide. "
+ "On Windows/macOS you may need two steps (`git mv a.txt tmp && git mv "
+ "tmp A.txt`) because the filesystem can't hold both spellings at once."
+ )
+ return 1
+
+
+if __name__ == "__main__":
+ sys.exit(main())
diff --git a/scripts/ci/classify_changes.py b/scripts/ci/classify_changes.py
index 935703c870..71afcda686 100644
--- a/scripts/ci/classify_changes.py
+++ b/scripts/ci/classify_changes.py
@@ -25,6 +25,12 @@ Lanes:
must not run it.
* ``npm_lock`` — semantic package-lock.json diff PR comment.
* ``installer`` — PowerShell installer tests (Windows runner).
+* ``desktop_updater`` — the Windows desktop-update hand-off script and the
+ tests that drive the REAL ``windows.ps1`` (``-SelfTestUi`` / pipe drain /
+ retry policy). These are integration tests of a PowerShell process on a
+ shared runner; running them on every Python PR made their timing noise
+ everyone's problem. They still run on push (fail-open) and whenever the
+ script, its siblings, or their tests change.
* ``rust`` — ``cargo test`` for the Tauri bootstrap installer. ``.rs``
lives under ``apps/``, so without this lane a Rust change matched ``frontend``
and only the TypeScript matrix ran.
@@ -110,6 +116,17 @@ _MCP_CATALOG_FILES = {"hermes_cli/mcp_catalog.py"}
_INSTALLER_PATHS = ("scripts/tests/",)
_INSTALLER_FILES = {"scripts/install.ps1", "scripts/install.cmd"}
+# Windows desktop-update hand-off (scripts/desktop-update/windows.ps1 + the
+# Electron side that launches it) and the pytest files that spawn it.
+_DESKTOP_UPDATER_PATHS = ("scripts/desktop-update/",)
+_DESKTOP_UPDATER_TEST_PREFIX = "tests/test_desktop_update_"
+_DESKTOP_UPDATER_FILES = {
+ "apps/desktop/electron/updater-process.ts",
+ "apps/desktop/electron/managed-ssh-update.ts",
+ "tests/conftest.py",
+ "pyproject.toml",
+}
+
# Rust crates — currently just the Tauri bootstrap installer (Hermes-Setup).
# These live under ``apps/``, so before this lane existed a ``.rs`` edit matched
# ``frontend`` and nothing more: the TypeScript matrix built, cargo never ran,
@@ -163,6 +180,14 @@ def _is_installer(p: str) -> bool:
return p.startswith(_INSTALLER_PATHS) or p in _INSTALLER_FILES
+def _is_desktop_updater(p: str) -> bool:
+ return (
+ p.startswith(_DESKTOP_UPDATER_PATHS)
+ or p.startswith(_DESKTOP_UPDATER_TEST_PREFIX)
+ or p in _DESKTOP_UPDATER_FILES
+ )
+
+
def _is_rust(p: str) -> bool:
return (
p.endswith(".rs")
@@ -206,6 +231,7 @@ def classify(files: list[str]) -> dict[str, bool]:
"uv_lock": any(f in ("pyproject.toml", "uv.lock") for f in files),
"npm_lock": npm_lock,
"installer": any(_is_installer(f) for f in files),
+ "desktop_updater": any(_is_desktop_updater(f) for f in files),
"rust": any(_is_rust(f) for f in files),
"mcp_catalog": any(_is_mcp_catalog(f) for f in files),
"ci_review": any(_is_ci_review(f) for f in files),
@@ -223,6 +249,7 @@ def classify(files: list[str]) -> dict[str, bool]:
ret["uv_lock"] = True
ret["npm_lock"] = True
ret["installer"] = True
+ ret["desktop_updater"] = True
ret["rust"] = True
ret["nix"] = True
ret["ci_review"] = True
diff --git a/scripts/desktop-update/windows.ps1 b/scripts/desktop-update/windows.ps1
index 7067cff335..19c9d8615a 100644
--- a/scripts/desktop-update/windows.ps1
+++ b/scripts/desktop-update/windows.ps1
@@ -212,6 +212,36 @@ function Start-UiServer([string]$HtmlPath) {
})
[void]$ps.BeginInvoke()
+ # Readiness handshake. BeginInvoke returns before the runspace has
+ # opened its pipeline and JIT'd the script block — on a loaded machine
+ # that is seconds, during which the kernel ACCEPTS connections into
+ # the listener's backlog and nobody answers them. Anything that
+ # trusted "listener bound" as "server serving" (the browser window
+ # opening to a page that never loads; the -SelfTestUi URL that CI
+ # polls) raced that gap. Prove one /progress round-trip before
+ # handing the port out, so the URL means "serving", not "bound".
+ $ready = $false
+ $readyDeadline = [DateTime]::UtcNow.AddSeconds(15)
+ while (-not $ready -and [DateTime]::UtcNow -lt $readyDeadline) {
+ try {
+ $probe = [System.Net.HttpWebRequest]::Create("http://127.0.0.1:$port/progress")
+ $probe.Timeout = 1000
+ $probe.ReadWriteTimeout = 1000
+ $probe.KeepAlive = $false
+ $resp = $probe.GetResponse()
+ try { $ready = ([int]$resp.StatusCode -eq 200) } finally { $resp.Close() }
+ } catch {
+ Start-Sleep -Milliseconds 100
+ }
+ }
+ if (-not $ready) {
+ Write-HandoffLog "progress server did not answer /progress within 15s; continuing without UI"
+ try { $listener.Stop() } catch {}
+ try { $ps.Stop() } catch {}
+ try { $rs.Close() } catch {}
+ return $null
+ }
+
return @{ Listener = $listener; Runspace = $rs; PowerShell = $ps; Port = $port; BrowserProc = $null; Profile = $null }
} catch {
try { if ($listener) { $listener.Stop() } } catch {}
diff --git a/scripts/e2e_shared_metrics_staging.py b/scripts/e2e_shared_metrics_staging.py
new file mode 100644
index 0000000000..666c9e51e9
--- /dev/null
+++ b/scripts/e2e_shared_metrics_staging.py
@@ -0,0 +1,198 @@
+"""Live staging E2E for the shared-metrics exporter.
+
+Sends REAL packages through the REAL sender to the REAL staging ingest
+service, then reports what the service acknowledged. Uses a throwaway
+HERMES_HOME so the operator's own telemetry state is untouched.
+
+Usage:
+ .venv/bin/python scripts/e2e_shared_metrics_staging.py
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import sys
+import tempfile
+import uuid
+from datetime import datetime, timezone
+from pathlib import Path
+
+REPO = Path(__file__).resolve().parents[1]
+sys.path.insert(0, str(REPO))
+
+STAGING = "https://telemetry.staging-nousresearch.com/v1/telemetry"
+
+
+def main() -> int:
+ scratch = Path(tempfile.mkdtemp(prefix="hermes-telemetry-e2e-"))
+ os.environ["HERMES_HOME"] = str(scratch)
+
+ # Staging is selected by writing config into the THROWAWAY profile, not by
+ # an environment override: a runtime env var that can retarget consented
+ # telemetry would be a consent hazard in production.
+ (scratch / "config.yaml").write_text(
+ "telemetry:\n"
+ " shared_metrics:\n"
+ " enabled: true\n"
+ " send: true\n"
+ f" endpoint: {STAGING}\n",
+ encoding="utf-8",
+ )
+
+ from hermes_cli.observability.shared_metrics import SharedMetricsStore
+ from hermes_cli.observability.shared_metrics_send_config import (
+ resolve_send_config,
+ )
+ from hermes_cli.observability.shared_metrics_sender import SharedMetricsSender
+
+ # Resolve through the real config path so this exercises what a user gets.
+ import yaml
+
+ resolved = resolve_send_config(
+ yaml.safe_load((scratch / "config.yaml").read_text(encoding="utf-8"))
+ )
+ if not resolved.send or resolved.endpoint != STAGING:
+ print(f"FAIL: config did not resolve to staging: {resolved}")
+ return 1
+
+ store = SharedMetricsStore(
+ database_path=scratch / "metrics.sqlite3",
+ outbox_directory=scratch / "outbox",
+ )
+
+ today = datetime.now(timezone.utc).date().isoformat()
+ # The generator only exports COMPLETED periods, so the realistic E2E
+ # package is yesterday's. It also has to be: the consent gate only
+ # releases a package once its whole period is confirmed consented, and
+ # today's period cannot be confirmed before it ends.
+ from datetime import timedelta
+
+ period_day = (
+ datetime.now(timezone.utc).date() - timedelta(days=1)
+ ).isoformat()
+
+ # Open the consent window before the period, confirm it after — exactly
+ # what the runtime reconciler does across two days of hook fires.
+ from hermes_cli.observability.shared_metrics_sender import (
+ reconcile_send_consent,
+ )
+ from hermes_cli.sqlite_util import write_txn
+
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(
+ connection,
+ True,
+ now=datetime.now(timezone.utc) - timedelta(days=2),
+ )
+ reconcile_send_consent(connection, True)
+ real_install_id = str(uuid.uuid4())
+ packages = []
+
+ # Two packages for today's period: the "head" and a later "tail", which is
+ # the real shape the outbox produces and the case the period gate exists
+ # for. One is large enough to exercise gzip.
+ for index, metric_count in ((0, 3), (1, 140)):
+ package_id = str(uuid.uuid4())
+ payload = {
+ "schema_version": "hermes.shared_metrics.v2",
+ "package_id": package_id,
+ "install_id": real_install_id,
+ "generated_at": datetime.now(timezone.utc).isoformat().replace(
+ "+00:00", "Z"
+ ),
+ "period_start": f"{period_day}T00:00:00Z",
+ "period_end": f"{period_day}T23:59:59Z",
+ "resource": {
+ "hermes_version": "e2e-test",
+ "os_family": "macos",
+ "architecture": "arm64",
+ "install_method": "git",
+ },
+ "metrics": [
+ {
+ "name": f"hermes.e2e.metric.{i}",
+ "type": "counter",
+ "dimensions": {"outcome": "ok", "surface": "e2e"},
+ "value": i + 1,
+ }
+ for i in range(metric_count)
+ ],
+ }
+ with store._connection() as connection:
+ connection.execute(
+ """
+ INSERT INTO package_outbox(
+ package_id, period_start, period_end, payload_json,
+ created_at, exported_at
+ ) VALUES (?, ?, ?, ?, ?, ?)
+ """,
+ (
+ package_id,
+ f"{period_day}T00:00:00Z",
+ f"{period_day}T23:59:59Z",
+ json.dumps(payload),
+ f"{period_day}T0{index}:00:00Z",
+ f"{period_day}T0{index}:00:01Z",
+ ),
+ )
+ packages.append((package_id, metric_count))
+
+ print(f"scratch HERMES_HOME : {scratch}")
+ print(f"endpoint : {STAGING}")
+ print(f"local install_id : {real_install_id}")
+ print(f"packages queued : {len(packages)}")
+ for package_id, count in packages:
+ print(f" - {package_id} ({count} metrics)")
+ print()
+
+ outcome = SharedMetricsSender(store, resolved.endpoint).send_pending()
+ print(f"outcome: sent={outcome.sent} rejected={outcome.rejected} "
+ f"deferred={outcome.deferred}")
+ print()
+
+ failures = []
+ with store._connection() as connection:
+ rows = connection.execute(
+ """
+ SELECT package_id, send_state, sent_at, send_attempts,
+ sent_install_id, last_error
+ FROM package_outbox ORDER BY created_at
+ """
+ ).fetchall()
+
+ for row in rows:
+ print(f"package : {row[0]}")
+ print(f" send_state : {row[1]}")
+ print(f" sent_at : {row[2]}")
+ print(f" attempts : {row[3]}")
+ print(f" transmitted : {row[4]}")
+ print(f" last_error : {row[5]}")
+ if row[1] != "sent":
+ failures.append(f"{row[0]} is {row[1]}: {row[5]}")
+ # Product decision 2026-08-27: the stable install_id is transmitted
+ # as-is; the transmitted value must be exactly the local id.
+ if row[4] != real_install_id:
+ failures.append(
+ f"{row[0]} transmitted {row[4]!r}, expected the install_id"
+ )
+ print()
+
+ if failures:
+ print("FAILURES:")
+ for failure in failures:
+ print(f" ✗ {failure}")
+ return 1
+
+ print("PASS: every package acknowledged 202 with the stable install_id.")
+ print()
+ print("Verify the objects in S3 with the package ids above:")
+ print(" aws s3 ls --recursive "
+ "s3://hermes-agent-telemetry-staging-767397871023-us-west-2-an/raw/ "
+ "| tail -20")
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main())
diff --git a/scripts/install.sh b/scripts/install.sh
index 6f717b011b..eec308a5de 100755
--- a/scripts/install.sh
+++ b/scripts/install.sh
@@ -618,21 +618,64 @@ install_uv() {
check_python() {
if [ "$DISTRO" = "termux" ]; then
log_info "Checking Termux Python..."
- if command -v python >/dev/null 2>&1; then
- PYTHON_PATH="$(command -v python)"
- if "$PYTHON_PATH" -c 'import sys; raise SystemExit(0 if sys.version_info >= (3, 11) else 1)' 2>/dev/null; then
- PYTHON_FOUND_VERSION="$("$PYTHON_PATH" --version 2>/dev/null)"
- log_success "Python found: $PYTHON_FOUND_VERSION"
- return 0
+ # Hermes currently declares requires-python >=3.11,<3.14. Termux can
+ # expose a newer default `python` before dependencies have compatible
+ # wheels, so do not accept the default interpreter until the upper bound
+ # is verified. Prefer the project's pinned minor when present, then
+ # other explicit compatible interpreters.
+ for python_cmd in python3.11 python3.12 python3.13 python; do
+ if command -v "$python_cmd" >/dev/null 2>&1; then
+ local candidate_path
+ candidate_path="$(command -v "$python_cmd")"
+ if "$candidate_path" -c 'import sys; raise SystemExit(0 if (3, 11) <= sys.version_info[:2] < (3, 14) else 1)' 2>/dev/null; then
+ PYTHON_PATH="$candidate_path"
+ PYTHON_FOUND_VERSION="$("$PYTHON_PATH" --version 2>/dev/null)"
+ log_success "Python found: $PYTHON_FOUND_VERSION"
+ return 0
+ fi
fi
- fi
+ done
log_info "Installing Python via pkg..."
pkg install -y python >/dev/null
PYTHON_PATH="$(command -v python)"
- PYTHON_FOUND_VERSION="$("$PYTHON_PATH" --version 2>/dev/null)"
- log_success "Python installed: $PYTHON_FOUND_VERSION"
- return 0
+ if "$PYTHON_PATH" -c 'import sys; raise SystemExit(0 if (3, 11) <= sys.version_info[:2] < (3, 14) else 1)' 2>/dev/null; then
+ PYTHON_FOUND_VERSION="$("$PYTHON_PATH" --version 2>/dev/null)"
+ log_success "Python installed: $PYTHON_FOUND_VERSION"
+ return 0
+ fi
+
+ # Termux's default `python` package is outside the supported range
+ # (e.g. 3.14.x before Rust transitives ship cp314 wheels). The Termux
+ # User Repository (TUR) publishes versioned CPython packages
+ # (python3.13, python3.11), so try to provision a supported
+ # interpreter from there before giving up.
+ PYTHON_FOUND_VERSION="$("$PYTHON_PATH" --version 2>/dev/null || true)"
+ log_warn "Termux Python $PYTHON_FOUND_VERSION is outside the supported range (>=3.11,<3.14)"
+ log_info "Trying the Termux User Repository (TUR) for a supported Python..."
+ pkg install -y tur-repo >/dev/null 2>&1 || true
+ local tur_pkg
+ for tur_pkg in python3.13 python3.12 python3.11; do
+ if ! pkg install -y "$tur_pkg" >/dev/null 2>&1; then
+ continue
+ fi
+ if ! command -v "$tur_pkg" >/dev/null 2>&1; then
+ continue
+ fi
+ local tur_path
+ tur_path="$(command -v "$tur_pkg")"
+ if "$tur_path" -c 'import sys; raise SystemExit(0 if (3, 11) <= sys.version_info[:2] < (3, 14) else 1)' 2>/dev/null; then
+ PYTHON_PATH="$tur_path"
+ PYTHON_FOUND_VERSION="$("$PYTHON_PATH" --version 2>/dev/null)"
+ log_success "Python installed from TUR: $PYTHON_FOUND_VERSION"
+ return 0
+ fi
+ done
+
+ log_error "Termux Python $PYTHON_FOUND_VERSION is not supported; Hermes requires Python >=3.11,<3.14"
+ log_info "Install a supported interpreter and re-run this script:"
+ log_info " pkg install tur-repo && pkg install python3.13"
+ exit 1
fi
log_info "Checking Python $PYTHON_VERSION..."
diff --git a/setup-hermes.sh b/setup-hermes.sh
index 0358bf1f7b..4f71fade62 100755
--- a/setup-hermes.sh
+++ b/setup-hermes.sh
@@ -133,19 +133,32 @@ fi
echo -e "${CYAN}→${NC} Checking Python $PYTHON_VERSION..."
if is_termux; then
- if command -v python >/dev/null 2>&1; then
- PYTHON_PATH="$(command -v python)"
- if "$PYTHON_PATH" -c 'import sys; raise SystemExit(0 if sys.version_info >= (3, 11) else 1)' 2>/dev/null; then
- PYTHON_FOUND_VERSION=$($PYTHON_PATH --version 2>/dev/null)
- echo -e "${GREEN}✓${NC} $PYTHON_FOUND_VERSION found"
- else
- echo -e "${RED}✗${NC} Termux Python must be 3.11+"
- echo " Run: pkg install python"
- exit 1
+ # Hermes currently declares requires-python >=3.11,<3.14. Termux can expose
+ # a newer default `python` before dependencies have compatible wheels, so
+ # prefer explicit compatible minors and verify the upper bound before using
+ # the interpreter to create the venv.
+ for python_cmd in python3.11 python3.12 python3.13 python; do
+ if command -v "$python_cmd" >/dev/null 2>&1; then
+ CANDIDATE_PATH="$(command -v "$python_cmd")"
+ if "$CANDIDATE_PATH" -c 'import sys; raise SystemExit(0 if (3, 11) <= sys.version_info[:2] < (3, 14) else 1)' 2>/dev/null; then
+ PYTHON_PATH="$CANDIDATE_PATH"
+ PYTHON_FOUND_VERSION=$($PYTHON_PATH --version 2>/dev/null)
+ echo -e "${GREEN}✓${NC} $PYTHON_FOUND_VERSION found"
+ break
+ fi
+ fi
+ done
+
+ if [ -z "${PYTHON_PATH:-}" ]; then
+ if command -v python >/dev/null 2>&1; then
+ PYTHON_FOUND_VERSION="$(python --version 2>/dev/null || true)"
+ echo -e "${RED}✗${NC} Termux Python $PYTHON_FOUND_VERSION is not supported; Hermes requires Python >=3.11,<3.14"
+ echo " Install a supported interpreter and re-run this script:"
+ echo " pkg install tur-repo && pkg install python3.13"
+ else
+ echo -e "${RED}✗${NC} Python not found in Termux"
+ echo " Run: pkg install python"
fi
- else
- echo -e "${RED}✗${NC} Python not found in Termux"
- echo " Run: pkg install python"
exit 1
fi
else
diff --git a/tests/agent/test_acp_openai_bridge.py b/tests/agent/test_acp_openai_bridge.py
index d1c0402604..f687461818 100644
--- a/tests/agent/test_acp_openai_bridge.py
+++ b/tests/agent/test_acp_openai_bridge.py
@@ -200,7 +200,10 @@ def test_copilot_prompt_still_carries_the_contract_and_the_tools():
assert "{...}" in prompt
assert '"name": "memory"' in prompt
assert '"name": "read_file"' in prompt # copilot forwards everything
- assert "Hermes requested model hint: gpt-5" in prompt
+ # No prompt-text model mention: the model is applied via ACP
+ # session/set_model, and a prompt hint makes a substituted backend
+ # falsely self-identify as the requested model.
+ assert "model hint" not in prompt
assert "hi" in prompt
diff --git a/tests/agent/test_anthropic_adapter.py b/tests/agent/test_anthropic_adapter.py
index 9619925194..61754a4307 100644
--- a/tests/agent/test_anthropic_adapter.py
+++ b/tests/agent/test_anthropic_adapter.py
@@ -983,19 +983,23 @@ class TestBuildAnthropicKwargs:
def test_supports_fast_mode_predicate(self):
- """Fast mode is Opus 4.6 only — Opus 4.7 and others must be excluded.
+ """The speed-param allowlist tracks the live fast-mode docs.
- For Opus 4.8 the fast variant is a separate model ID
- (anthropic/claude-opus-4.8-fast) routed through the normal model
- field, NOT via the ``speed: "fast"`` request parameter. So
- ``_supports_fast_mode`` (which gates the parameter) must stay
- False for both opus-4-8 and opus-4-8-fast.
+ Per https://platform.claude.com/docs/en/build-with-claude/fast-mode:
+ Opus 4.8 and Opus 5 support ``speed: "fast"``. Opus 4.6 LOST fast
+ mode (param silently ignored → standard speed at standard billing);
+ Opus 4.7 hard-400s. Dedicated ``…-fast`` model ids select fast
+ inference via the model field and must not also get the param.
"""
from agent.anthropic_adapter import _supports_fast_mode
- assert _supports_fast_mode("claude-opus-4-6") is True
- assert _supports_fast_mode("anthropic/claude-opus-4-6") is True
+ assert _supports_fast_mode("claude-opus-4-8") is True
+ assert _supports_fast_mode("claude-opus-4.8") is True
+ assert _supports_fast_mode("anthropic/claude-opus-4-8") is True
+ assert _supports_fast_mode("claude-opus-5") is True
+ assert _supports_fast_mode("anthropic/claude-opus-5") is True
+ assert _supports_fast_mode("claude-opus-4-6") is False
+ assert _supports_fast_mode("anthropic/claude-opus-4-6") is False
assert _supports_fast_mode("claude-opus-4-7") is False
- assert _supports_fast_mode("claude-opus-4-8") is False
assert _supports_fast_mode("claude-opus-4-8-fast") is False
assert _supports_fast_mode("claude-sonnet-4-6") is False
assert _supports_fast_mode("claude-haiku-4-5") is False
diff --git a/tests/agent/test_aux_stream_host_deadline.py b/tests/agent/test_aux_stream_host_deadline.py
new file mode 100644
index 0000000000..924b0766a8
--- /dev/null
+++ b/tests/agent/test_aux_stream_host_deadline.py
@@ -0,0 +1,295 @@
+"""#99692 — the streamed auxiliary summary must not outlive its compression host.
+
+Background
+----------
+``run_compress_context_with_progress_timeout`` arms a wall-clock deadline on the
+``CompressionCommitFence`` (``set_total_ceiling_seconds``), whose docstring calls
+it "the wall-clock deadline **shared by the host and worker**". Only the host
+ever read it.
+
+``8207862212`` (fix(compression): stop timeout paths from blocking retries)
+closed the first half: a cancelled fence now releases the compression OWNER,
+which frees the pool slot and the session lease. It left the second half open
+by design — its own comment says the isolated provider daemon runs on "until
+the auxiliary stream's longer absolute ceiling expires".
+
+That ceiling is ``_aux_stream_total_ceiling`` = ``max(600, 4 * aux_timeout)``:
+>= the default host ceiling (600s) for every configured timeout, and it starts
+counting later (after pool admission, serialization, prompt build and TTFT).
+So the daemon holding the socket is *always* still streaming when its host gives
+up — 2400s with the reporter's ``auxiliary.compression.timeout: 600`` — billing
+every token of a summary the fence is already guaranteed to refuse, and stacking
+one fresh orphan per turn because the session never shrank.
+
+These tests pin the missing half of that shared deadline: the stream consumer
+must stop at the host's deadline, including on the isolated provider daemon
+that ``_run_protected_sync_provider_call`` spawns.
+"""
+
+from __future__ import annotations
+
+import ast
+import asyncio
+import inspect
+import threading
+import time
+from pathlib import Path
+from types import SimpleNamespace
+
+import pytest
+
+from agent import auxiliary_client as aux
+from agent.conversation_compression import (
+ DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS,
+ CompressionCommitFence,
+)
+
+
+def _chunk(text: str) -> SimpleNamespace:
+ return SimpleNamespace(
+ id="resp-1",
+ model="test-model",
+ usage=None,
+ choices=[
+ SimpleNamespace(
+ index=0,
+ finish_reason=None,
+ delta=SimpleNamespace(content=text, tool_calls=None),
+ )
+ ],
+ )
+
+
+class _Stream:
+ """Chunk iterator that records how far the consumer drained it."""
+
+ def __init__(self, count: int = 50) -> None:
+ self._count = count
+ self.yielded = 0
+ self.closed = False
+
+ def __iter__(self):
+ for _ in range(self._count):
+ self.yielded += 1
+ yield _chunk("x")
+
+ def close(self) -> None:
+ self.closed = True
+
+
+class _AsyncStream(_Stream):
+ async def __aiter__(self): # pragma: no cover - exercised via asyncio.run
+ for _ in range(self._count):
+ self.yielded += 1
+ yield _chunk("x")
+
+
+# ── The structural gap the bug lives in ──────────────────────────────────
+
+
+def test_stream_ceiling_structurally_outlives_the_default_host_ceiling():
+ """The worker's own budget is >= the host's for every configured timeout.
+
+ This is the arithmetic that guarantees the orphan: there is no aux timeout
+ for which ``_aux_stream_total_ceiling`` lands below the 600s default host
+ ceiling, and the reporter's ``auxiliary.compression.timeout: 600`` puts it
+ at 2400s — a 30-minute window in which an abandoned provider daemon keeps
+ streaming a summary nobody can commit.
+ """
+ for aux_timeout in (None, 0, 30.0, 120.0, 300.0):
+ assert (
+ aux._aux_stream_total_ceiling(aux_timeout)
+ >= DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS
+ )
+ assert aux._aux_stream_total_ceiling(600.0) == 2400.0
+ assert (
+ aux._aux_stream_total_ceiling(600.0)
+ - DEFAULT_CONTEXT_TOTAL_CEILING_SECONDS
+ == 1800.0
+ )
+
+
+# ── The fence must publish the deadline it already owns ──────────────────
+
+
+def test_commit_fence_publishes_its_shared_deadline():
+ fence = CompressionCommitFence()
+ assert fence.deadline_monotonic is None
+
+ fence.set_total_ceiling_seconds(600.0)
+ published = fence.deadline_monotonic
+ assert published is not None
+ assert 590.0 < published - time.monotonic() <= 600.0
+ assert not fence.deadline_exceeded
+
+ fence.set_total_ceiling_seconds(0.001)
+ time.sleep(0.01)
+ assert fence.deadline_exceeded
+ assert fence.deadline_monotonic <= time.monotonic()
+
+
+# ── The stream consumer must honour it ───────────────────────────────────
+
+
+def test_streamed_summary_stops_at_an_elapsed_host_deadline():
+ """A host that already gave up must not leave the worker streaming on."""
+ stream = _Stream(count=50)
+ with aux.aux_stream_deadline(time.monotonic() - 1.0):
+ with pytest.raises(TimeoutError) as excinfo:
+ aux._aggregate_chat_stream(stream, model="m", total_ceiling=2400.0)
+
+ # "timed out" keeps _is_timeout_error classification identical to a
+ # request timeout, so the existing recovery chains are unchanged.
+ assert "timed out" in str(excinfo.value)
+ assert "host compression deadline" in str(excinfo.value)
+ # Stopped on the first frame instead of draining the whole stream, and the
+ # HTTP response was closed rather than left dangling.
+ assert stream.yielded == 1
+ assert stream.closed is True
+
+
+def test_streamed_summary_runs_to_completion_under_a_live_host_deadline():
+ stream = _Stream(count=5)
+ with aux.aux_stream_deadline(time.monotonic() + 600.0):
+ response = aux._aggregate_chat_stream(
+ stream, model="m", total_ceiling=2400.0
+ )
+ assert response.choices[0].message.content == "xxxxx"
+ assert stream.yielded == 5
+
+
+def test_no_host_deadline_keeps_the_historical_ceiling_behaviour():
+ """Every non-compression aux caller must be byte-for-byte unchanged."""
+ stream = _Stream(count=5)
+ response = aux._aggregate_chat_stream(stream, model="m", total_ceiling=2400.0)
+ assert response.choices[0].message.content == "xxxxx"
+ assert stream.yielded == 5
+
+ # An installed-then-exited scope must not leak into the next call.
+ with aux.aux_stream_deadline(time.monotonic() - 1.0):
+ pass
+ stream2 = _Stream(count=3)
+ assert (
+ aux._aggregate_chat_stream(
+ stream2, model="m", total_ceiling=2400.0
+ ).choices[0].message.content
+ == "xxx"
+ )
+
+
+def test_none_deadline_is_a_no_op_passthrough():
+ """Callers wire the scope unconditionally; a fenceless call must not break."""
+ stream = _Stream(count=3)
+ with aux.aux_stream_deadline(None):
+ response = aux._aggregate_chat_stream(
+ stream, model="m", total_ceiling=2400.0
+ )
+ assert response.choices[0].message.content == "xxx"
+
+
+def test_nested_none_inherits_rather_than_escaping_the_host_deadline():
+ """A fenceless aux call nested inside a fenced one stays bounded.
+
+ ``None`` means "I have no deadline of my own", not "clear the one in
+ force" — mirroring ``_aux_thread_local_hook``'s passthrough contract. If it
+ cleared, any nested auxiliary call made during compression would escape the
+ host ceiling that the whole attempt is supposed to live inside.
+ """
+ outer = time.monotonic() - 1.0
+ stream = _Stream(count=50)
+ with aux.aux_stream_deadline(outer):
+ with aux.aux_stream_deadline(None):
+ assert aux._current_aux_stream_deadline() == outer
+ with pytest.raises(TimeoutError):
+ aux._aggregate_chat_stream(stream, model="m", total_ceiling=2400.0)
+ assert stream.yielded == 1
+
+
+def test_deadline_scope_restores_the_previous_value():
+ outer = time.monotonic() + 900.0
+ with aux.aux_stream_deadline(outer):
+ assert aux._current_aux_stream_deadline() == outer
+ with aux.aux_stream_deadline(time.monotonic() + 10.0):
+ assert aux._current_aux_stream_deadline() != outer
+ assert aux._current_aux_stream_deadline() == outer
+ assert aux._current_aux_stream_deadline() is None
+
+
+def test_async_stream_mirror_honours_the_host_deadline():
+ """The async consumer must not drift from the sync one."""
+ stream = _AsyncStream(count=50)
+
+ async def _run():
+ with aux.aux_stream_deadline(time.monotonic() - 1.0):
+ return await aux._aggregate_chat_stream_async(
+ stream, model="m", total_ceiling=2400.0
+ )
+
+ with pytest.raises(TimeoutError):
+ asyncio.run(_run())
+ assert stream.yielded == 1
+
+
+# ── The isolated provider daemon must inherit it ─────────────────────────
+
+
+def test_protected_provider_daemon_inherits_the_host_deadline():
+ """``_run_protected_sync_provider_call`` runs the stream on ANOTHER thread.
+
+ Thread-locals do not cross that boundary, so without explicit propagation
+ the fix would be inert on exactly the path large-session compression takes
+ (protected + hard-cancel source installed).
+ """
+ seen: dict[str, object] = {}
+
+ def _callback(_kwargs):
+ seen["deadline"] = aux._current_aux_stream_deadline()
+ seen["thread"] = threading.current_thread().name
+ return "ok"
+
+ deadline = time.monotonic() + 42.0
+ cancel_event = threading.Event()
+ with aux.aux_progress_hook(lambda: None), aux.aux_interrupt_protection(
+ cancel_event=cancel_event
+ ), aux.aux_stream_deadline(deadline):
+ assert aux._run_protected_sync_provider_call(_callback, {}) == "ok"
+
+ assert seen["thread"] == "hermes-protected-aux-provider"
+ assert seen["deadline"] == deadline
+
+
+# ── The compression worker must actually install it ──────────────────────
+
+
+def _summary_dispatch_source() -> str:
+ from agent import conversation_compression
+
+ path = Path(inspect.getsourcefile(conversation_compression))
+ return path.read_text(encoding="utf-8")
+
+
+def test_compression_summary_dispatch_installs_the_fence_deadline():
+ """Source guard: the wiring is one line and trivially droppable.
+
+ A behavioural test would have to drive the whole ``compress_context`` body
+ (durable lock, watermark, telemetry, commit). This asserts the seam itself:
+ the same ``with`` statement that installs the progress hook must also
+ install the stream deadline.
+ """
+ tree = ast.parse(_summary_dispatch_source())
+ wired = False
+ for node in ast.walk(tree):
+ if not isinstance(node, ast.With):
+ continue
+ names = set()
+ for item in node.items:
+ call = item.context_expr
+ if isinstance(call, ast.Call) and isinstance(call.func, ast.Name):
+ names.add(call.func.id)
+ if "aux_progress_hook" in names:
+ assert "aux_stream_deadline" in names, (
+ "the summary dispatch scope installs the progress hook but not "
+ "the host stream deadline — #99692 would regress"
+ )
+ wired = True
+ assert wired, "summary dispatch scope not found"
diff --git a/tests/agent/test_aux_stream_host_deadline_sibling_wires.py b/tests/agent/test_aux_stream_host_deadline_sibling_wires.py
new file mode 100644
index 0000000000..250b5ac132
--- /dev/null
+++ b/tests/agent/test_aux_stream_host_deadline_sibling_wires.py
@@ -0,0 +1,189 @@
+"""#99692 sibling wires — the host compression deadline must stop EVERY aux
+stream consumer, not only the chat.completions accumulator.
+
+``aux_stream_deadline`` (salvaged from PR #99779 by @JoaoMarcos44) publishes
+the ``CompressionCommitFence`` ceiling to the streamed chat.completions path.
+Two other auxiliary wires consume their streams internally and were left with
+their own, always-larger budgets:
+
+* the Codex Responses adapter (``_CodexCompletionsAdapter.create``) — its
+ re-armable watchdog only knew ``_aux_stream_total_ceiling`` (>= 600s);
+* the Anthropic Messages adapter — its ``on_stream_event`` hook only ticked
+ progress and never stopped the stream at all (nor honoured a hard cancel).
+
+Both now stop at the host's absolute deadline, so an abandoned summary is not
+billed to completion on a socket nobody is waiting for.
+"""
+
+from __future__ import annotations
+
+import time
+from types import SimpleNamespace
+from unittest.mock import patch
+
+import pytest
+
+from agent import auxiliary_client as aux
+from agent.anthropic_adapter import create_anthropic_message
+
+
+# ── Codex Responses wire ─────────────────────────────────────────────────
+
+
+def _codex_content_event(text="tok"):
+ return SimpleNamespace(type="response.output_text.delta", delta=text)
+
+
+def _consume_codex(stream, *, model, on_event):
+ del model
+ for event in stream:
+ on_event(event)
+ return SimpleNamespace(
+ output=[SimpleNamespace(
+ type="message",
+ content=[SimpleNamespace(type="output_text", text="summary")],
+ )],
+ usage=None,
+ )
+
+
+def _make_codex_adapter(event_iter):
+ real_client = SimpleNamespace(
+ base_url="https://chatgpt.com/backend-api/codex",
+ responses=SimpleNamespace(create=lambda **_kwargs: event_iter),
+ close=lambda: None,
+ )
+ return aux._CodexCompletionsAdapter(real_client, "gpt-5.6-sol")
+
+
+def test_codex_stream_stops_at_the_host_deadline_not_its_own_ceiling():
+ """A live (re-arming) Codex stream must die at the host's deadline even
+ though its own hard ceiling is >= 600s and every token re-arms the
+ no-progress window."""
+ yielded = [0]
+
+ def _live_forever():
+ while True:
+ time.sleep(0.02)
+ yielded[0] += 1
+ yield _codex_content_event()
+
+ adapter = _make_codex_adapter(_live_forever())
+ start = time.monotonic()
+ with (
+ patch("agent.codex_runtime._consume_codex_event_stream", _consume_codex),
+ aux.aux_stream_deadline(time.monotonic() + 0.4),
+ pytest.raises(TimeoutError, match="hard ceiling"),
+ ):
+ adapter.create(
+ messages=[{"role": "user", "content": "summarize"}],
+ timeout=300,
+ )
+ elapsed = time.monotonic() - start
+ assert elapsed < 5.0, f"stream outlived the host deadline by {elapsed:.1f}s"
+ assert yielded[0] < 100
+
+
+def test_codex_stream_without_host_deadline_keeps_its_ceiling():
+ def _short():
+ for _ in range(3):
+ yield _codex_content_event()
+
+ adapter = _make_codex_adapter(_short())
+ with patch("agent.codex_runtime._consume_codex_event_stream", _consume_codex):
+ response = adapter.create(
+ messages=[{"role": "user", "content": "summarize"}], timeout=300,
+ )
+ assert response.choices[0].message.content == "summary"
+
+
+# ── Anthropic Messages wire ──────────────────────────────────────────────
+
+
+class _AnthropicStream:
+ def __init__(self, count=10_000, delay=0.01):
+ self._count, self._delay = count, delay
+ self.yielded = 0
+ self.exited = False
+ self.response = None
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *exc):
+ self.exited = True
+ return False
+
+ def __iter__(self):
+ for _ in range(self._count):
+ time.sleep(self._delay)
+ self.yielded += 1
+ yield SimpleNamespace(
+ type="content_block_delta", delta=SimpleNamespace(text="tok"),
+ )
+
+ def get_final_message(self):
+ return SimpleNamespace(content=[SimpleNamespace(type="text", text="summary")])
+
+
+def _anthropic_client(stream):
+ return SimpleNamespace(
+ messages=SimpleNamespace(
+ stream=lambda **_kw: stream,
+ create=lambda **_kw: pytest.fail("must not fall back to create()"),
+ )
+ )
+
+
+def test_anthropic_stream_stops_at_the_host_deadline():
+ stream = _AnthropicStream()
+ ticks = []
+ with (
+ aux.aux_progress_hook(lambda: ticks.append(1)),
+ aux.aux_stream_deadline(time.monotonic() + 0.3),
+ ):
+ hook = aux._anthropic_aux_stream_event_hook()
+ start = time.monotonic()
+ with pytest.raises(TimeoutError, match="timed out at the host compression deadline"):
+ create_anthropic_message(
+ _anthropic_client(stream), {"model": "m", "messages": []},
+ on_stream_event=hook,
+ )
+ assert time.monotonic() - start < 5.0
+ assert stream.exited, "stream context must be closed on the deadline"
+ assert ticks, "substantive deltas must still tick the progress hook"
+ assert stream.yielded < 1000
+
+
+def test_anthropic_stream_honours_an_explicit_hard_cancel():
+ stream = _AnthropicStream()
+ cancelled = {"v": False}
+ with (
+ aux.aux_progress_hook(lambda: None),
+ aux.aux_interrupt_protection(cancel_check=lambda: cancelled["v"]),
+ ):
+ hook = aux._anthropic_aux_stream_event_hook()
+
+ def _flip_after_first(event, _inner=hook):
+ cancelled["v"] = True
+ _inner(event)
+
+ with pytest.raises(aux.AuxiliaryExplicitCancellation):
+ create_anthropic_message(
+ _anthropic_client(stream), {"model": "m", "messages": []},
+ on_stream_event=_flip_after_first,
+ )
+ assert stream.yielded == 1
+ assert stream.exited
+
+
+def test_anthropic_stream_without_host_deadline_runs_to_completion():
+ stream = _AnthropicStream(count=5, delay=0)
+ with aux.aux_progress_hook(lambda: None):
+ hook = aux._anthropic_aux_stream_event_hook()
+ message = create_anthropic_message(
+ _anthropic_client(stream), {"model": "m", "messages": []},
+ on_stream_event=hook,
+ )
+ assert message.content[0].text == "summary"
+ assert stream.yielded == 5
diff --git a/tests/agent/test_codex_happy_eyeballs.py b/tests/agent/test_codex_happy_eyeballs.py
index 48e0bbe38f..91804555af 100644
--- a/tests/agent/test_codex_happy_eyeballs.py
+++ b/tests/agent/test_codex_happy_eyeballs.py
@@ -146,3 +146,180 @@ def test_connection_staggers_past_blackholed_ipv6(monkeypatch):
assert clock[0] == process_bootstrap._HAPPY_EYEBALLS_DELAY_SECONDS
assert sockets[0].closed is True
assert sockets[1].closed is False
+
+
+def test_async_codex_client_relies_on_native_anyio_racing(no_proxy_env):
+ """The async transport needs no custom backend — anyio races natively.
+
+ httpcore's ``AnyIOBackend.connect_tcp`` delegates to
+ ``anyio.connect_tcp``, whose ``happy_eyeballs_delay`` default (0.25s)
+ implements RFC 8305 staggered family racing. This pins the contract the
+ ``async_mode`` branch of ``build_keepalive_http_client`` documents: if
+ anyio ever drops the parameter (or the default stops racing), this fails
+ and the async path needs an explicit backend like the sync one.
+ """
+ import inspect
+
+ import anyio
+
+ params = inspect.signature(anyio.connect_tcp).parameters
+ assert "happy_eyeballs_delay" in params
+ assert params["happy_eyeballs_delay"].default == pytest.approx(0.25)
+
+ client = process_bootstrap.build_keepalive_http_client(
+ "https://chatgpt.com/backend-api/codex", async_mode=True
+ )
+ try:
+ assert all(
+ not isinstance(backend, process_bootstrap._HappyEyeballsSyncBackend)
+ for backend in _client_backends(client)
+ )
+ finally:
+ import asyncio
+
+ asyncio.get_event_loop_policy().new_event_loop().run_until_complete(
+ client.aclose()
+ )
+
+
+def test_async_connect_races_past_blackholed_ipv6(monkeypatch):
+ """IPv4 completes ~250ms after a hanging IPv6 attempt on the async path.
+
+ Mirrors ``test_connection_staggers_past_blackholed_ipv6`` for the async
+ transport: resolve a fake host to a blackholed IPv6 address plus a live
+ local IPv4 listener and assert httpcore's async backend connects fast
+ instead of serially waiting out the IPv6 connect timeout.
+ """
+ import asyncio
+ import threading
+ import time as _time
+
+ server = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
+ server.bind(("127.0.0.1", 0))
+ server.listen(5)
+ port = server.getsockname()[1]
+
+ def _accept_loop():
+ while True:
+ try:
+ conn, _ = server.accept()
+ conn.close()
+ except OSError:
+ return
+
+ thread = threading.Thread(target=_accept_loop, daemon=True)
+ thread.start()
+
+ real_getaddrinfo = socket.getaddrinfo
+
+ def fake_getaddrinfo(host, *args, **kwargs):
+ name = host.decode() if isinstance(host, (bytes, bytearray)) else str(host)
+ if name == "codex-he-async.test":
+ return [
+ (
+ socket.AF_INET6,
+ socket.SOCK_STREAM,
+ socket.IPPROTO_TCP,
+ "",
+ ("100::1", port, 0, 0), # RFC 6666 discard prefix: blackhole
+ ),
+ (
+ socket.AF_INET,
+ socket.SOCK_STREAM,
+ socket.IPPROTO_TCP,
+ "",
+ ("127.0.0.1", port),
+ ),
+ ]
+ return real_getaddrinfo(host, *args, **kwargs)
+
+ monkeypatch.setattr(socket, "getaddrinfo", fake_getaddrinfo)
+
+ async def _connect():
+ from httpcore._backends.auto import AutoBackend
+
+ backend = AutoBackend()
+ start = _time.monotonic()
+ stream = await backend.connect_tcp(
+ "codex-he-async.test", port, timeout=30.0
+ )
+ elapsed = _time.monotonic() - start
+ await stream.aclose()
+ return elapsed
+
+ try:
+ elapsed = asyncio.run(_connect())
+ finally:
+ server.close()
+
+ # Native anyio racing: IPv6 is attempted first, IPv4 starts 0.25s later
+ # and wins immediately. Serial behavior would block until the IPv6
+ # connect timeout (tens of seconds). Generous bound for slow CI hosts.
+ assert elapsed < 5.0
+
+
+class _RecordingPool:
+ def __init__(self):
+ self._network_backend = "default"
+
+
+class _RecordingTransport:
+ def __init__(self):
+ self._pool = _RecordingPool()
+
+
+def test_enable_happy_eyeballs_on_client_covers_transport_and_mounts():
+ class _Client:
+ pass
+
+ client = _Client()
+ client._transport = _RecordingTransport()
+ client._mounts = {"https://": _RecordingTransport(), "http://": None}
+
+ process_bootstrap.enable_happy_eyeballs_on_client(client)
+
+ assert isinstance(
+ client._transport._pool._network_backend,
+ process_bootstrap._HappyEyeballsSyncBackend,
+ )
+ assert isinstance(
+ client._mounts["https://"]._pool._network_backend,
+ process_bootstrap._HappyEyeballsSyncBackend,
+ )
+
+
+def test_enable_happy_eyeballs_on_client_skips_proxy_pools(no_proxy_env):
+ import httpcore
+ import httpx
+
+ client = httpx.Client(proxy="http://127.0.0.1:3128")
+ try:
+ process_bootstrap.enable_happy_eyeballs_on_client(client)
+ proxy_pools = [
+ transport._pool
+ for transport in client._mounts.values()
+ if transport is not None
+ and isinstance(getattr(transport, "_pool", None), httpcore.HTTPProxy)
+ ]
+ assert proxy_pools # the all:// mount is proxy-backed
+ assert all(
+ not isinstance(
+ pool._network_backend, process_bootstrap._HappyEyeballsSyncBackend
+ )
+ for pool in proxy_pools
+ )
+ finally:
+ client.close()
+
+
+def test_codex_auth_http_client_uses_happy_eyeballs_backend(no_proxy_env):
+ from hermes_cli.auth import _codex_http_client
+
+ client = _codex_http_client(timeout=5.0)
+ try:
+ assert any(
+ isinstance(backend, process_bootstrap._HappyEyeballsSyncBackend)
+ for backend in _client_backends(client)
+ )
+ finally:
+ client.close()
diff --git a/tests/agent/test_codex_ttfb_watchdog.py b/tests/agent/test_codex_ttfb_watchdog.py
index 66208a8e1a..bdf53061f3 100644
--- a/tests/agent/test_codex_ttfb_watchdog.py
+++ b/tests/agent/test_codex_ttfb_watchdog.py
@@ -108,6 +108,89 @@ def test_ttfb_includes_silent_hang_hint_for_gpt_5_5(tmp_path, monkeypatch):
stop["flag"] = True
+def test_ttfb_installs_and_retires_the_codex_request_token(tmp_path, monkeypatch):
+ """The watchdog must publish a per-request token and clear it on the kill.
+
+ ``run_codex_stream`` reads ``agent._active_codex_stream_request_token`` to
+ tell whether it is still the owning attempt. Without an install here the
+ whole retirement guard would be inert, and without the clear on kill a
+ retired worker would keep normalizing partial deltas into a "completed"
+ response.
+
+ The worker also unwinds with its own local error after the force-close;
+ that error must not replace the watchdog's retryable ``TimeoutError``.
+ """
+ from agent import chat_completion_helpers as h
+
+ agent = _make_codex_agent(tmp_path, monkeypatch)
+ monkeypatch.setenv("HERMES_CODEX_TTFB_TIMEOUT_SECONDS", "1")
+
+ closes: list = []
+ seen = {"token_while_running": None}
+ dummy_client = SimpleNamespace()
+ monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client)
+ monkeypatch.setattr(
+ agent,
+ "_abort_request_openai_client",
+ lambda c, reason=None: closes.append(reason),
+ )
+ monkeypatch.setattr(
+ agent,
+ "_close_request_openai_client",
+ lambda c, reason=None: closes.append(reason),
+ )
+
+ def fake_stream(api_kwargs, client=None, on_first_delta=None):
+ seen["token_while_running"] = getattr(
+ agent, "_active_codex_stream_request_token", None
+ )
+ deadline = time.time() + 30
+ while time.time() < deadline:
+ if getattr(agent, "_active_codex_stream_request_token", None) is None:
+ # Retired by the watchdog — mimic the transport unwinding.
+ raise RuntimeError("retired worker stream ended without terminal")
+ time.sleep(0.02)
+ raise RuntimeError("test timed out waiting for retirement")
+
+ monkeypatch.setattr(agent, "_run_codex_stream", fake_stream)
+
+ with pytest.raises(TimeoutError) as excinfo:
+ h.interruptible_api_call(agent, {"model": "gpt-5.5", "input": "hi"})
+
+ assert seen["token_while_running"] is not None, (
+ "interruptible_api_call must install a request token before the worker runs"
+ )
+ assert "TTFB" in str(excinfo.value)
+ assert "retired worker" not in str(excinfo.value)
+ assert "codex_ttfb_kill" in closes
+ assert getattr(agent, "_active_codex_stream_request_token", None) is None
+
+
+def test_non_codex_api_mode_installs_no_request_token(tmp_path, monkeypatch):
+ """The token is codex_responses-only — other api_modes stay untouched."""
+ from agent import chat_completion_helpers as h
+
+ agent = _make_codex_agent(tmp_path, monkeypatch)
+ agent.api_mode = "chat_completions"
+
+ seen = {"token": "unset"}
+ dummy_client = SimpleNamespace()
+ monkeypatch.setattr(agent, "_create_request_openai_client", lambda **k: dummy_client)
+
+ def fake_dispatch(_agent, _api_kwargs, *, make_client):
+ make_client("test")
+ seen["token"] = getattr(
+ _agent, "_active_codex_stream_request_token", "absent"
+ )
+ return SimpleNamespace(choices=[])
+
+ monkeypatch.setattr(h, "_dispatch_nonstreaming_api_request", fake_dispatch)
+
+ h.interruptible_api_call(agent, {"model": "gpt-5.5", "messages": []})
+
+ assert seen["token"] in (None, "absent")
+
+
def test_ttfb_does_not_kill_when_events_flow(tmp_path, monkeypatch):
diff --git a/tests/agent/test_compression_anti_thrash_recovery.py b/tests/agent/test_compression_anti_thrash_recovery.py
index 109f23c18e..cf245ac9a5 100644
--- a/tests/agent/test_compression_anti_thrash_recovery.py
+++ b/tests/agent/test_compression_anti_thrash_recovery.py
@@ -17,10 +17,13 @@ The recovery contract pinned here:
next recovery waits a FULL fresh window (no immediate re-probe loop).
* An effective probe (or any fitting real-usage reading) fully clears the
counters through the existing ``update_from_response`` path.
-* The recovery clock is armed lazily on the first blocked evaluation and is
- NOT durable: a process restart that loads a durable tripped counter
- (#69872) starts a full fresh window blocked — a restart must never disarm
- or shorten the guard (#54923).
+* The recovery clock is armed lazily on the first blocked evaluation and
+ persisted on the session row as a wall-clock deadline (#100185): a fresh
+ compressor that loads a durable tripped counter (#69872) with NO stored
+ deadline starts a full window blocked — a restart must never disarm or
+ shorten the guard (#54923) — while one that loads an armed deadline
+ resumes that window instead of restarting it, so gateway agent rebuilds
+ cannot block a session forever.
* The protection itself is preserved: inside the window the gate stays
blocked exactly as before.
"""
@@ -57,10 +60,10 @@ class TestRecoveryWindow:
cc = _compressor()
_trip(cc)
base = 1000.0
- with patch("agent.context_compressor.time.monotonic", return_value=base):
+ with patch("agent.context_compressor.time.time", return_value=base):
assert cc.should_compress(cc.threshold_tokens + 1) is False
with patch(
- "agent.context_compressor.time.monotonic",
+ "agent.context_compressor.time.time",
return_value=base + cc._ANTI_THRASH_RECOVERY_SECONDS + 1,
):
assert cc.should_compress(cc.threshold_tokens + 1) is True
@@ -73,10 +76,10 @@ class TestRecoveryWindow:
cc = _compressor()
cc._fallback_compression_streak = 2
base = 1000.0
- with patch("agent.context_compressor.time.monotonic", return_value=base):
+ with patch("agent.context_compressor.time.time", return_value=base):
assert cc.should_compress(cc.threshold_tokens + 1) is False
with patch(
- "agent.context_compressor.time.monotonic",
+ "agent.context_compressor.time.time",
return_value=base + cc._ANTI_THRASH_RECOVERY_SECONDS + 1,
):
assert cc.should_compress(cc.threshold_tokens + 1) is True
@@ -95,13 +98,13 @@ class TestRestartSemantics:
cc = _compressor()
cc.bind_session_state(session_db=db, session_id="sess-1")
assert cc._ineffective_compression_count == 2
- # The recovery clock is process-local and must come up disarmed.
+ # No stored deadline yet -> the clock comes up disarmed.
assert cc._anti_thrash_recovery_deadline == 0.0
base = 5000.0
- with patch("agent.context_compressor.time.monotonic", return_value=base):
+ with patch("agent.context_compressor.time.time", return_value=base):
assert cc.should_compress(cc.threshold_tokens + 1) is False
with patch(
- "agent.context_compressor.time.monotonic",
+ "agent.context_compressor.time.time",
return_value=base + cc._ANTI_THRASH_RECOVERY_SECONDS + 1,
):
assert cc.should_compress(cc.threshold_tokens + 1) is True
@@ -113,9 +116,92 @@ class TestRestartSemantics:
cc = _compressor()
_trip(cc)
base = 1000.0
- with patch("agent.context_compressor.time.monotonic", return_value=base):
+ with patch("agent.context_compressor.time.time", return_value=base):
assert cc.should_compress(cc.threshold_tokens + 1) is False
assert cc._anti_thrash_recovery_deadline > 0.0
cc.on_session_reset()
assert cc._anti_thrash_recovery_deadline == 0.0
assert cc._ineffective_compression_count == 0
+
+
+class TestDurableDeadline:
+ """#100185: the gateway rebuilds the compressor on every cache eviction."""
+
+ def _bound(self, db, session_id="sess-1"):
+ cc = _compressor()
+ cc.bind_session_state(session_db=db, session_id=session_id)
+ return cc
+
+ def test_fresh_compressors_resume_the_same_window(self, tmp_path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ db.create_session(session_id="sess-1", source="telegram")
+ db.set_compression_ineffective_count("sess-1", 2)
+ base = 5000.0
+ first = self._bound(db)
+ with patch("agent.context_compressor.time.time", return_value=base):
+ assert first.should_compress(first.threshold_tokens + 1) is False
+ # Deadline is durable, as a wall-clock epoch.
+ assert db.get_compression_recovery_deadline("sess-1") == (
+ base + first._ANTI_THRASH_RECOVERY_SECONDS
+ )
+ # Fresh compressor (gateway rebuilt the agent) well past the window:
+ # before the fix it re-armed a new window and stayed blocked forever.
+ second = self._bound(db)
+ assert second._anti_thrash_recovery_deadline == (
+ base + first._ANTI_THRASH_RECOVERY_SECONDS
+ )
+ with patch(
+ "agent.context_compressor.time.time",
+ return_value=base + first._ANTI_THRASH_RECOVERY_SECONDS + 1,
+ ):
+ assert second.should_compress(second.threshold_tokens + 1) is True
+ assert db.get_compression_ineffective_count("sess-1") == 1
+ assert db.get_compression_recovery_deadline("sess-1") == 0.0
+
+ def test_fresh_compressor_inside_window_stays_blocked(self, tmp_path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ db.create_session(session_id="sess-1", source="telegram")
+ db.set_compression_ineffective_count("sess-1", 2)
+ base = 5000.0
+ first = self._bound(db)
+ with patch("agent.context_compressor.time.time", return_value=base):
+ assert first.should_compress(first.threshold_tokens + 1) is False
+ second = self._bound(db)
+ with patch("agent.context_compressor.time.time", return_value=base + 10):
+ assert second.should_compress(second.threshold_tokens + 1) is False
+ assert db.get_compression_ineffective_count("sess-1") == 2
+
+ def test_backward_clock_jump_is_bounded_to_one_window(self, tmp_path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ db.create_session(session_id="sess-1", source="telegram")
+ db.set_compression_ineffective_count("sess-1", 2)
+ window = ContextCompressor._ANTI_THRASH_RECOVERY_SECONDS
+ db.set_compression_recovery_deadline("sess-1", 1_000_000.0)
+ cc = self._bound(db)
+ # Wall clock now far BEFORE the stored deadline (clock stepped back).
+ with patch("agent.context_compressor.time.time", return_value=100.0):
+ assert cc.should_compress(cc.threshold_tokens + 1) is False
+ assert db.get_compression_recovery_deadline("sess-1") == 100.0 + window
+
+ def test_clearing_the_guard_disarms_the_durable_deadline(self, tmp_path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ db.create_session(session_id="sess-1", source="telegram")
+ db.set_compression_ineffective_count("sess-1", 2)
+ cc = self._bound(db)
+ with patch("agent.context_compressor.time.time", return_value=5000.0):
+ assert cc.should_compress(cc.threshold_tokens + 1) is False
+ assert db.get_compression_recovery_deadline("sess-1") > 0.0
+ cc._record_ineffective_compression_verdict(0)
+ with patch("agent.context_compressor.time.time", return_value=5001.0):
+ assert cc.should_compress(cc.threshold_tokens + 1) is True
+ assert db.get_compression_recovery_deadline("sess-1") == 0.0
+
+ def test_session_db_round_trip(self, tmp_path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ db.create_session(session_id="sess-1", source="cli")
+ assert db.get_compression_recovery_deadline("sess-1") == 0.0
+ db.set_compression_recovery_deadline("sess-1", 1234.5)
+ assert db.get_compression_recovery_deadline("sess-1") == 1234.5
+ db.set_compression_recovery_deadline("sess-1", 0.0)
+ assert db.get_compression_recovery_deadline("sess-1") == 0.0
+ assert db.get_compression_recovery_deadline("missing") == 0.0
diff --git a/tests/agent/test_compression_attempt_lifecycle.py b/tests/agent/test_compression_attempt_lifecycle.py
index d7cf68be4f..84879151fb 100644
--- a/tests/agent/test_compression_attempt_lifecycle.py
+++ b/tests/agent/test_compression_attempt_lifecycle.py
@@ -287,3 +287,73 @@ class TestTransientBlockIsNotExhaustion:
mock_agent = MagicMock()
# MagicMock auto-attributes are truthy but not str.
assert compression_blocked_transiently(mock_agent) is False
+
+
+def _summary_response(content: str):
+ from unittest.mock import MagicMock
+
+ response = MagicMock()
+ response.choices = [MagicMock()]
+ response.choices[0].message.content = content
+ return response
+
+
+class TestProviderOverflowBypassesCooldown:
+ """#100661: a provider-proven overflow must get one REAL summary attempt
+ while the summary-failure cooldown is armed. Before the fix every turn of
+ a wedged session hit the cooldown gate, returned the soft "temporarily
+ paused" deferral, and the next failure extended the ladder — 4 long
+ sessions were lost this way. Ordinary (non-overflow) automatic passes
+ must still defer."""
+
+ def _armed_agent(self, tmp_path: Path, session_id: str):
+ db, agent = _build_agent(tmp_path, session_id)
+ # Realistic arming: a failed/stalled attempt recorded the ladder.
+ agent.context_compressor.record_timeout_failure(
+ "stall", failure_kind="stalled"
+ )
+ assert agent.context_compressor.should_compress_info(500_000)[0] is False
+ return db, agent
+
+ def test_overflow_attempt_invokes_summarizer_while_cooldown_armed(
+ self, tmp_path: Path
+ ):
+ db, agent = self._armed_agent(tmp_path, "OVERFLOW_BYPASS")
+ calls = []
+
+ def fake_call_llm(**kwargs):
+ calls.append(kwargs)
+ return _summary_response("## Goal\nRecovered after overflow.")
+
+ # Bulky turns so the compacted transcript is genuinely smaller.
+ live = [
+ {"role": "user" if i % 2 == 0 else "assistant", "content": f"m{i} " * 400}
+ for i in range(20)
+ ]
+ with patch("agent.context_compressor.call_llm", fake_call_llm):
+ out, _ = compress_context(
+ agent, live, "sys", approx_tokens=500_000, bypass_cooldown=True
+ )
+ assert len(calls) == 1, (
+ "provider-proven overflow must reach the summary LLM even while "
+ "the failure cooldown is armed (#100661)"
+ )
+ assert compression_blocked_transiently(agent) is False
+ assert len(out) < len(live), "the attempt must actually compact"
+
+ def test_non_overflow_pass_still_deferred_by_cooldown(self, tmp_path: Path):
+ db, agent = self._armed_agent(tmp_path, "OVERFLOW_ORDINARY")
+ calls = []
+
+ def fake_call_llm(**kwargs): # pragma: no cover - must not run
+ calls.append(kwargs)
+ return _summary_response("unexpected")
+
+ live = _messages()
+ before = copy.deepcopy(live)
+ with patch("agent.context_compressor.call_llm", fake_call_llm):
+ out, _ = compress_context(agent, live, "sys", approx_tokens=500_000)
+ assert calls == [] and out == before
+ assert compression_blocked_transiently(agent) is True, (
+ "ordinary threshold pressure keeps honoring the cooldown (#11529)"
+ )
diff --git a/tests/agent/test_compression_busy_steer_anchor.py b/tests/agent/test_compression_busy_steer_anchor.py
new file mode 100644
index 0000000000..3aed06bece
--- /dev/null
+++ b/tests/agent/test_compression_busy_steer_anchor.py
@@ -0,0 +1,146 @@
+"""Regression coverage for busy-steer preservation across compaction (#100053).
+
+With ``display.busy_input_mode: steer`` the follow-up rides inside the latest
+``role=tool`` result (``apply_pending_steer_to_tool_results``), never as a
+``role=user`` row. ``_ensure_compressed_has_user_turn`` must treat that marker
+as live user intent — and must pick whichever intent-bearing row is LAST in
+the original transcript, so an older steer never outranks a newer real user
+request.
+"""
+
+import pytest
+
+from agent.context_compressor import (
+ COMPRESSION_CONTINUATION_USER_CONTENT,
+ SUMMARY_PREFIX,
+)
+from agent.conversation_compression import (
+ _compressed_has_busy_steer,
+ _ensure_compressed_has_user_turn,
+)
+from agent.prompt_builder import STEER_MARKER_OPEN, format_steer_marker
+
+REQUEST_A = "Historical request A: audit the auth module."
+STEER_B = "Steer B: stop, switch to fixing the login bug instead."
+REQUEST_C = "Newer real user request C: now write the release notes."
+
+
+def _tool_turns(start: int, count: int, *, steer_at: int | None = None) -> list[dict]:
+ turns: list[dict] = []
+ for idx in range(start, start + count):
+ turns.append(
+ {
+ "role": "assistant",
+ "content": "Working.",
+ "tool_calls": [
+ {
+ "id": f"call-{idx}",
+ "function": {"name": "terminal", "arguments": "{}"},
+ }
+ ],
+ }
+ )
+ content = f"tool output {idx}"
+ if steer_at == idx:
+ content += format_steer_marker(STEER_B)
+ turns.append({"role": "tool", "tool_call_id": f"call-{idx}", "content": content})
+ return turns
+
+
+def _summary_row() -> dict:
+ return {"role": "user", "content": f"{SUMMARY_PREFIX}\n\nEarlier work summarized."}
+
+
+def _assert_alternation(messages: list[dict]) -> None:
+ roles = [m.get("role") for m in messages]
+ for left, right in zip(roles, roles[1:]):
+ assert not (left == right == "user"), f"user/user adjacency in {roles}"
+ assert not (left == right == "assistant"), f"assistant/assistant adjacency in {roles}"
+
+
+def _user_rows(messages: list[dict]) -> list[str]:
+ return [str(m.get("content")) for m in messages if m.get("role") == "user"]
+
+
+def test_s1_steer_summarized_away_becomes_anchor_not_historical_request():
+ """S1: the steer lived in a tool row that compaction dropped; the only
+ ``role=user`` row in history is the already-consumed request A. The steer
+ must be restored as the anchor, and A must not be replayed."""
+ original = [{"role": "user", "content": REQUEST_A}] + _tool_turns(0, 6, steer_at=2)
+ compressed = [_summary_row(), *_tool_turns(5, 1)]
+
+ outcome = _ensure_compressed_has_user_turn(original, compressed)
+
+ assert outcome == "inserted"
+ _assert_alternation(compressed)
+ users = _user_rows(compressed)
+ assert STEER_B in users, users
+ assert REQUEST_A not in users, "historical request replayed as new input"
+ assert COMPRESSION_CONTINUATION_USER_CONTENT not in users
+ # Steer text is used exactly once across the whole compressed transcript.
+ assert sum(str(m.get("content")).count(STEER_B) for m in compressed) == 1
+
+
+def test_s2_steer_surviving_in_tail_tool_row_counts_as_present():
+ """S2: the steer-bearing tool row survived into the tail. No anchor may be
+ inserted (the intent is already there) and A must not be cloned."""
+ original = [{"role": "user", "content": REQUEST_A}] + _tool_turns(0, 6, steer_at=5)
+ compressed = [_summary_row(), *_tool_turns(5, 1, steer_at=5)]
+ before = [dict(m) for m in compressed]
+
+ outcome = _ensure_compressed_has_user_turn(original, compressed)
+
+ assert outcome == "already_present"
+ assert compressed == before, "transcript mutated despite live steer present"
+ assert REQUEST_A not in _user_rows(compressed)
+ assert sum(str(m.get("content")).count(STEER_B) for m in compressed) == 1
+
+
+def test_s3_newer_real_user_turn_outranks_older_steer():
+ """S3: ``[user A, tool(steer B), ..., user C]`` — C is the newest intent.
+ A steer-first scan would anchor the consumed steer B and replay it."""
+ original = (
+ [{"role": "user", "content": REQUEST_A}]
+ + _tool_turns(0, 3, steer_at=1)
+ + [{"role": "user", "content": REQUEST_C}]
+ + _tool_turns(3, 4)
+ )
+ compressed = [_summary_row(), *_tool_turns(6, 1)]
+
+ outcome = _ensure_compressed_has_user_turn(original, compressed)
+
+ assert outcome == "inserted"
+ _assert_alternation(compressed)
+ users = _user_rows(compressed)
+ assert REQUEST_C in users, users
+ assert STEER_B not in users, "older consumed steer replayed over newer user turn"
+ assert REQUEST_A not in users
+ assert not any(STEER_B in u for u in users)
+
+
+def test_newer_steer_outranks_older_real_user_turn():
+ """Mirror of S3: ``[user A, ..., tool(steer B)]`` — the steer is newest."""
+ original = [{"role": "user", "content": REQUEST_A}] + _tool_turns(0, 4, steer_at=3)
+ compressed = [_summary_row(), *_tool_turns(4, 1)]
+
+ outcome = _ensure_compressed_has_user_turn(original, compressed)
+
+ assert outcome == "inserted"
+ _assert_alternation(compressed)
+ users = _user_rows(compressed)
+ assert STEER_B in users
+ assert REQUEST_A not in users
+
+
+@pytest.mark.parametrize(
+ "role",
+ ["user", "assistant"],
+)
+def test_compressed_steer_presence_only_counts_tool_rows(role):
+ """A summary or assistant row that merely quotes the marker text is not a
+ live steer delivery — only ``role=tool`` rows carry real steers."""
+ quoted = {"role": role, "content": f"{SUMMARY_PREFIX}\n{format_steer_marker(STEER_B)}"}
+ assert _compressed_has_busy_steer([quoted]) is False
+ assert STEER_MARKER_OPEN in quoted["content"]
+ live = {"role": "tool", "tool_call_id": "c", "content": f"ok{format_steer_marker(STEER_B)}"}
+ assert _compressed_has_busy_steer([live]) is True
diff --git a/tests/agent/test_context_compressor.py b/tests/agent/test_context_compressor.py
index 7e373f209e..10997fe94a 100644
--- a/tests/agent/test_context_compressor.py
+++ b/tests/agent/test_context_compressor.py
@@ -855,6 +855,19 @@ class TestAuthFailureAborts:
)
assert _is_summary_access_or_quota_error(err) is True
+ def test_unscoped_secret_read_is_terminal_access_failure(self):
+ # Multiplexed gateway: a credential read reached get_secret() from a
+ # worker thread without the profile scope. The summary model is
+ # unreachable until the spawn site is fixed — abort and preserve the
+ # session rather than truncating the middle window (#100849 bundle).
+ from agent.secret_scope import UnscopedSecretError
+
+ err = UnscopedSecretError(
+ "get_secret('SURPLUS_API_KEY') called with no profile secret scope "
+ "active while multiplexing is on."
+ )
+ assert _is_summary_access_or_quota_error(err) is True
+
diff --git a/tests/agent/test_copilot_acp_client.py b/tests/agent/test_copilot_acp_client.py
index 100dca67f4..1b2b08f01a 100644
--- a/tests/agent/test_copilot_acp_client.py
+++ b/tests/agent/test_copilot_acp_client.py
@@ -317,3 +317,107 @@ def test_probe_skipped_for_custom_args_without_acp():
with _patch("agent.copilot_acp_client.subprocess.run") as run_mock:
assert _acp_supported("mycli", ["--custom-transport"]) is True
run_mock.assert_not_called()
+
+
+# --- session/set_model: honor the picker-selected model ----------------------
+#
+# `copilot --acp` validates but IGNORES the `--model` spawn flag; the ACP
+# session runs the CLI's own default unless the client issues the ACP-native
+# `session/set_model` call. Without it, picking gpt-5.6-terra in Hermes
+# visibly answers as the CLI's default model.
+
+
+# --- session model selection -------------------------------------------------
+
+
+def _session_with_config_options():
+ return {
+ "sessionId": "s1",
+ "configOptions": [
+ {
+ "id": "model",
+ "category": "model",
+ "type": "select",
+ "currentValue": "auto",
+ "options": [
+ {"value": "auto", "name": "Auto"},
+ {"value": "gpt-5.6-terra", "name": "GPT-5.6 Terra"},
+ {
+ "value": "claude-fable-5",
+ "name": "Claude Fable 5",
+ "_meta": {"copilotEnablement": "disabled"},
+ },
+ ],
+ }
+ ],
+ }
+
+
+def test_model_selection_prefers_stable_config_option():
+ from agent.copilot_acp_client import _model_selection_request
+
+ assert _model_selection_request(
+ _session_with_config_options(), "gpt-5.6-terra"
+ ) == (
+ "session/set_config_option",
+ {"sessionId": "s1", "configId": "model", "value": "gpt-5.6-terra"},
+ )
+
+
+def test_model_selection_rejects_disabled_config_option():
+ from agent.copilot_acp_client import _model_selection_request
+
+ assert _model_selection_request(
+ _session_with_config_options(), "claude-fable-5"
+ ) is None
+
+
+def test_model_selection_rejects_unknown_config_option():
+ from agent.copilot_acp_client import _model_selection_request
+
+ assert _model_selection_request(
+ _session_with_config_options(), "not-served-here"
+ ) is None
+
+
+def test_model_selection_falls_back_to_legacy_extension():
+ from agent.copilot_acp_client import _model_selection_request
+
+ legacy_session = {
+ "sessionId": "s1",
+ "models": {
+ "availableModels": [
+ {"modelId": "auto"},
+ {"modelId": "gpt-5.6-terra"},
+ ]
+ },
+ }
+ assert _model_selection_request(legacy_session, "gpt-5.6-terra") == (
+ "session/set_model",
+ {"sessionId": "s1", "modelId": "gpt-5.6-terra"},
+ )
+
+
+def test_model_selection_skips_provider_virtual_slug():
+ from agent.copilot_acp_client import _model_selection_request
+
+ assert _model_selection_request(
+ _session_with_config_options(), "copilot-acp"
+ ) is None
+
+
+def test_run_prompt_receives_picker_model():
+ # _create_chat_completion must forward `model` into _run_prompt — the
+ # original wiring dropped it, reducing the selection to prompt text.
+ client = CopilotACPClient(acp_cwd="/tmp")
+ seen = {}
+
+ def fake_run_prompt(prompt_text, *, timeout_seconds, model=None):
+ seen["model"] = model
+ return "ok", ""
+
+ with patch.object(CopilotACPClient, "_run_prompt", side_effect=fake_run_prompt):
+ client._create_chat_completion(
+ model="gpt-5.6-terra", messages=[{"role": "user", "content": "hi"}]
+ )
+ assert seen["model"] == "gpt-5.6-terra"
diff --git a/tests/agent/test_credential_pool_profile_oauth_fork.py b/tests/agent/test_credential_pool_profile_oauth_fork.py
new file mode 100644
index 0000000000..057db3021d
--- /dev/null
+++ b/tests/agent/test_credential_pool_profile_oauth_fork.py
@@ -0,0 +1,452 @@
+"""Regression tests for #100339: cloned / borrowed single-use Anthropic OAuth
+grants must never fork across profiles.
+
+Real imports, real temp HERMES_HOME root + named profile, real auth.json I/O.
+The Anthropic token endpoint is replaced at the ``urllib.request.urlopen``
+boundary with genuine single-use semantics (a refresh token redeems once;
+a second POST returns ``invalid_grant``).
+"""
+from __future__ import annotations
+
+import io
+import json
+import os
+import time
+import urllib.error
+import urllib.request
+
+import pytest
+
+
+@pytest.fixture
+def fleet(tmp_path, monkeypatch):
+ """Root HERMES_HOME with an expired-but-refreshable Anthropic pool row."""
+ root = tmp_path / "hermes-root"
+ root.mkdir()
+ (tmp_path / "fakehome").mkdir()
+ # Keep host ~/.claude and host auth.json out of the picture.
+ monkeypatch.setenv("HOME", str(tmp_path / "fakehome"))
+ monkeypatch.setenv("CLAUDE_CONFIG_DIR", str(tmp_path / "fakehome"))
+ for var in ("ANTHROPIC_TOKEN", "ANTHROPIC_API_KEY", "CLAUDE_CODE_OAUTH_TOKEN"):
+ monkeypatch.delenv(var, raising=False)
+ monkeypatch.setenv("HERMES_HOME", str(root))
+ # The pytest seat-belt in the root write-through compares the global path
+ # against $HOME/.hermes/auth.json; our root is elsewhere, so writes go.
+ import hermes_constants
+ hermes_constants._default_hermes_root_memo = None # type: ignore[attr-defined]
+
+ expired = int((time.time() - 3600) * 1000)
+ store = {
+ "version": 1,
+ "providers": {},
+ "credential_pool": {
+ "anthropic": [{
+ "id": "abc123", "label": "team-grant", "auth_type": "oauth",
+ "priority": 0, "source": "manual:hermes_pkce",
+ "access_token": "sk-ant-oat01-AT0", "refresh_token": "sk-ant-ort-RT0",
+ "expires_at_ms": expired, "base_url": "https://api.anthropic.com",
+ }],
+ "openai": [{
+ "id": "key001", "label": "static", "auth_type": "api_key",
+ "priority": 0, "source": "manual", "access_token": "sk-static-key",
+ }],
+ },
+ }
+ (root / "auth.json").write_text(json.dumps(store))
+
+ server = {"valid": {"sk-ant-ort-RT0"}, "spent": set(), "n": 0, "log": []}
+
+ class _Resp(io.BytesIO):
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ def fake_urlopen(req, timeout=None):
+ assert "oauth/token" in req.full_url
+ body = req.data.decode()
+ if req.get_header("Content-type", "").startswith("application/json"):
+ rt = json.loads(body)["refresh_token"]
+ else:
+ from urllib.parse import parse_qsl
+ rt = dict(parse_qsl(body))["refresh_token"]
+ if rt in server["spent"] or rt not in server["valid"]:
+ server["log"].append(("REUSE", rt))
+ raise urllib.error.HTTPError(
+ req.full_url, 400, "Bad Request", {},
+ io.BytesIO(b'{"error":"invalid_grant","error_description":"refresh_token_reused"}'),
+ )
+ server["n"] += 1
+ server["spent"].add(rt)
+ server["valid"].discard(rt)
+ new_rt = f"sk-ant-ort-RT{server['n']}"
+ server["valid"].add(new_rt)
+ server["log"].append(("ROTATE", rt, new_rt))
+ return _Resp(json.dumps({
+ "access_token": f"sk-ant-oat01-AT{server['n']}",
+ "refresh_token": new_rt, "expires_in": 28800, "token_type": "Bearer",
+ }).encode())
+
+ monkeypatch.setattr(urllib.request, "urlopen", fake_urlopen)
+
+ def use(home):
+ """Switch the process to *home* (root or a profile dir)."""
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ hermes_constants._default_hermes_root_memo = None # type: ignore[attr-defined]
+ import hermes_cli.auth as auth_mod
+ auth_mod._global_auth_store_cache = None
+ auth_mod._oauth_heal_clean_marks.clear()
+
+ # Process-wide notice buffer: start each test clean.
+ import hermes_cli.auth as _auth_mod
+ _auth_mod._oauth_heal_notices.clear()
+ _auth_mod._oauth_heal_clean_marks.clear()
+
+ def pool_rows(home):
+ p = home / "auth.json"
+ if not p.exists():
+ return None
+ return (json.loads(p.read_text()).get("credential_pool") or {}).get("anthropic")
+
+ return {"root": root, "server": server, "use": use, "rows": pool_rows}
+
+
+def _profile(fleet, name, **kw):
+ from hermes_cli.profiles import create_profile
+ fleet["use"](fleet["root"])
+ return create_profile(name, **kw)
+
+
+# ── A. cloning never copies single-use OAuth grants ──────────────────────
+
+def test_clone_all_strips_oauth_grant_but_keeps_api_keys(fleet):
+ (fleet["root"] / ".anthropic_oauth.json").write_text(
+ json.dumps({"accessToken": "sk-ant-oat01-AT0", "refreshToken": "sk-ant-ort-RT0", "expiresAt": 1})
+ )
+ pdir = _profile(fleet, "forge", clone_all=True)
+ store = json.loads((pdir / "auth.json").read_text())
+ assert "anthropic" not in store["credential_pool"], "OAuth grant was forked into the clone"
+ assert store["credential_pool"]["openai"][0]["access_token"] == "sk-static-key"
+ assert not (pdir / ".anthropic_oauth.json").exists()
+
+
+def test_strip_helper_drops_device_code_blocks_and_reports(tmp_path):
+ from hermes_cli.auth import strip_cloned_single_use_oauth_grants
+ pdir = tmp_path / "p"
+ pdir.mkdir()
+ (pdir / "auth.json").write_text(json.dumps({
+ "version": 1,
+ "providers": {"openai-codex": {"access_token": "a", "refresh_token": "r"}, "nous": {"agent_key": "k"}},
+ "credential_pool": {
+ "xai-oauth": [{"id": "x", "auth_type": "oauth", "access_token": "t", "refresh_token": "r"}],
+ "anthropic": [
+ {"id": "legacy", "access_token": "sk-ant-oat01-legacy"}, # no auth_type field
+ {"id": "key", "auth_type": "api_key", "access_token": "sk-ant-api03-x"},
+ ],
+ },
+ }))
+ summary = strip_cloned_single_use_oauth_grants(pdir)
+ store = json.loads((pdir / "auth.json").read_text())
+ assert sorted(summary["pool"]) == ["anthropic", "xai-oauth"]
+ assert summary["providers"] == ["openai-codex"]
+ assert "xai-oauth" not in store["credential_pool"]
+ assert [e["id"] for e in store["credential_pool"]["anthropic"]] == ["key"]
+ assert "openai-codex" not in store["providers"] and "nous" in store["providers"]
+
+
+def test_strip_helper_is_a_noop_without_credentials(tmp_path):
+ from hermes_cli.auth import strip_cloned_single_use_oauth_grants
+ assert strip_cloned_single_use_oauth_grants(tmp_path) == {"pool": [], "providers": [], "files": []}
+
+
+# ── B. borrowed rotation commits to root, never a profile copy ───────────
+
+def test_first_profile_rotation_does_not_strand_root_or_siblings(fleet):
+ from agent.credential_pool import load_pool
+
+ forge = _profile(fleet, "forge")
+ atlas = _profile(fleet, "atlas")
+
+ fleet["use"](forge)
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT1"
+ # The rotated pair landed in ROOT; forge did not grow a local copy.
+ assert fleet["rows"](forge) is None
+ assert fleet["rows"](fleet["root"])[0]["refresh_token"] == "sk-ant-ort-RT1"
+
+ for home in (atlas, fleet["root"], forge):
+ fleet["use"](home)
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT1", home
+ assert [e[0] for e in fleet["server"]["log"]] == ["ROTATE"], fleet["server"]["log"]
+ assert fleet["rows"](atlas) is None and fleet["rows"](forge) is None
+
+
+def test_agent_init_resolver_sees_sibling_rotation(fleet):
+ from agent.anthropic_credentials import resolve_anthropic_token
+ from agent.credential_pool import load_pool
+
+ forge = _profile(fleet, "forge")
+ atlas = _profile(fleet, "atlas")
+ fleet["use"](forge)
+ load_pool("anthropic").select()
+ fleet["use"](atlas)
+ assert resolve_anthropic_token() == "sk-ant-oat01-AT1"
+
+
+def test_borrowing_profile_load_pool_does_not_materialize_local_copy(fleet):
+ from agent.credential_pool import load_pool
+
+ fresh = _profile(fleet, "fresh")
+ fleet["use"](fresh)
+ pool = load_pool("anthropic")
+ assert [e.id for e in pool.entries()] == ["abc123"]
+ assert pool._borrowed_root_ids == {"abc123"}
+ assert fleet["rows"](fresh) is None
+
+
+def test_borrower_prune_never_deletes_root_singleton_grant(fleet, tmp_path):
+ """Root's hermes_pkce row is seeded from ROOT's .anthropic_oauth.json; a
+ profile without that file must not prune (and write-through-delete) it."""
+ from agent.credential_pool import load_pool
+
+ root = fleet["root"]
+ (root / ".anthropic_oauth.json").write_text(json.dumps({
+ "accessToken": "sk-ant-oat01-AT0", "refreshToken": "sk-ant-ort-RT0",
+ "expiresAt": int((time.time() - 3600) * 1000),
+ }))
+ store = json.loads((root / "auth.json").read_text())
+ store["active_provider"] = "anthropic"
+ del store["credential_pool"]["anthropic"]
+ (root / "auth.json").write_text(json.dumps(store))
+ fleet["use"](root)
+ root_rows = [e for e in load_pool("anthropic").entries()]
+ assert [e.source for e in root_rows] == ["hermes_pkce"]
+
+ kid = _profile(fleet, "kid")
+ fleet["use"](kid)
+ pool = load_pool("anthropic")
+ assert [e.source for e in pool.entries()] == ["hermes_pkce"], "borrowed root grant was pruned"
+ assert fleet["rows"](root) and fleet["rows"](root)[0]["source"] == "hermes_pkce"
+ assert fleet["rows"](kid) is None
+
+ # Rotating from the profile commits BOTH the pool row and the singleton at ROOT.
+ sel = pool.select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT1"
+ assert json.loads((root / ".anthropic_oauth.json").read_text())["refreshToken"] == "sk-ant-ort-RT1"
+ assert not (kid / ".anthropic_oauth.json").exists()
+ assert fleet["rows"](root)[0]["refresh_token"] == "sk-ant-ort-RT1"
+
+
+def test_profile_auth_add_owns_only_its_own_rows(fleet):
+ from agent.credential_pool import AUTH_TYPE_OAUTH, PooledCredential, load_pool
+
+ kid = _profile(fleet, "kid")
+ fleet["use"](kid)
+ pool = load_pool("anthropic")
+ pool.add_entry(PooledCredential(
+ provider="anthropic", id="own001", label="mine", auth_type=AUTH_TYPE_OAUTH,
+ priority=0, source="manual:hermes_pkce", access_token="sk-ant-oat01-MINE",
+ refresh_token="rt-mine",
+ ))
+ assert [e["id"] for e in fleet["rows"](kid)] == ["own001"], "borrowed root row was copied into the profile"
+ assert [e["id"] for e in fleet["rows"](fleet["root"])] == ["abc123"]
+
+
+def test_classic_mode_persist_is_unchanged(fleet):
+ from agent.credential_pool import load_pool
+
+ fleet["use"](fleet["root"])
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT1"
+ assert fleet["rows"](fleet["root"])[0]["refresh_token"] == "sk-ant-ort-RT1"
+
+
+# ── C. one-time heal for installs that ALREADY forked the grant ──────────
+#
+# Fleets created on pre-fix code hold profile-local copies of the root grant
+# (verbatim --clone-all, or the old borrowed-persist). The heal runs inside
+# the profile's load_pool(): consolidate to ROOT (freshest rotation wins),
+# strip the profile copy, borrow root from then on.
+
+def _fork(fleet, name, *, rotated_to=None):
+ """Create *name* with a pre-fix style verbatim copy of root's auth.json.
+
+ ``rotated_to=N`` makes the copy the LIVE pair (RT, spent RT0 server-side)
+ to emulate a profile that already refreshed on the old code.
+ """
+ pdir = _profile(fleet, name)
+ pdir.mkdir(parents=True, exist_ok=True)
+ store = json.loads((fleet["root"] / "auth.json").read_text())
+ if rotated_to is not None:
+ row = store["credential_pool"]["anthropic"][0]
+ row["access_token"] = f"sk-ant-oat01-AT{rotated_to}"
+ row["refresh_token"] = f"sk-ant-ort-RT{rotated_to}"
+ row["expires_at_ms"] = int((time.time() - 60) * 1000) # newer, still expired
+ srv = fleet["server"]
+ srv["spent"].add("sk-ant-ort-RT0")
+ srv["valid"].discard("sk-ant-ort-RT0")
+ srv["valid"].add(f"sk-ant-ort-RT{rotated_to}")
+ srv["n"] = rotated_to
+ (pdir / "auth.json").write_text(json.dumps(store))
+ return pdir
+
+
+def test_heal_consolidates_existing_forks_to_the_live_copy(fleet, caplog):
+ """root + atlas hold spent RT0; forge already rotated to RT1 on old code."""
+ import logging
+ from agent.credential_pool import load_pool
+
+ forge = _fork(fleet, "forge", rotated_to=1)
+ atlas = _fork(fleet, "atlas")
+ assert fleet["rows"](forge)[0]["refresh_token"] == "sk-ant-ort-RT1"
+ assert fleet["rows"](atlas)[0]["refresh_token"] == "sk-ant-ort-RT0"
+
+ with caplog.at_level(logging.INFO, logger="hermes_cli.auth"):
+ fleet["use"](forge)
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT2"
+ # forge's live pair was adopted by ROOT, then rotated there; forge holds nothing.
+ assert fleet["rows"](forge) is None
+ assert fleet["rows"](fleet["root"])[0]["refresh_token"] == "sk-ant-ort-RT2"
+ assert fleet["rows"](fleet["root"])[0]["id"] == "abc123"
+ healed = [r.message for r in caplog.records if "consolidated forked anthropic OAuth grant" in r.message]
+ assert len(healed) == 1 and "profile forge" in healed[0] and "root updated" in healed[0]
+
+ for home in (atlas, fleet["root"], forge):
+ fleet["use"](home)
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT2", home
+ assert fleet["rows"](atlas) is None and fleet["rows"](forge) is None
+ # Exactly one rotation by us (RT1 -> RT2); the spent RT0 was never replayed.
+ assert [e[0] for e in fleet["server"]["log"]] == ["ROTATE"], fleet["server"]["log"]
+ # API-key rows in the profiles were not touched.
+ for home in (forge, atlas):
+ store = json.loads((home / "auth.json").read_text())
+ assert store["credential_pool"]["openai"][0]["access_token"] == "sk-static-key"
+
+
+def test_heal_is_idempotent_and_logs_once(fleet, caplog):
+ import logging
+ from agent.credential_pool import load_pool
+ from hermes_cli.auth import consume_oauth_heal_notices, heal_forked_single_use_oauth_grants
+
+ kid = _fork(fleet, "kid")
+ fleet["use"](kid)
+ with caplog.at_level(logging.INFO, logger="hermes_cli.auth"):
+ load_pool("anthropic")
+ assert fleet["rows"](kid) is None
+ notices = consume_oauth_heal_notices()
+ assert len(notices) == 1 and "profile kid" in notices[0]
+ root_before = (fleet["root"] / "auth.json").read_text()
+ # Second and third loads: nothing to do, nothing written, nothing logged.
+ assert heal_forked_single_use_oauth_grants("anthropic") is None
+ load_pool("anthropic")
+ assert consume_oauth_heal_notices() == []
+ assert (fleet["root"] / "auth.json").read_text() == root_before
+ assert sum("consolidated forked" in r.message for r in caplog.records) == 1
+
+
+def test_heal_never_deletes_the_only_surviving_copy(fleet):
+ """Root lost its grant (user ran `hermes auth remove` at root); the profile's
+ copy is the only one left — and an independent second account stays put."""
+ from agent.credential_pool import load_pool
+
+ kid = _fork(fleet, "kid", rotated_to=1)
+ store = json.loads((fleet["root"] / "auth.json").read_text())
+ del store["credential_pool"]["anthropic"]
+ (fleet["root"] / "auth.json").write_text(json.dumps(store))
+
+ fleet["use"](kid)
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT2"
+ assert fleet["rows"](kid) and fleet["rows"](kid)[0]["refresh_token"] == "sk-ant-ort-RT2"
+ assert "anthropic" not in (json.loads((fleet["root"] / "auth.json").read_text())["credential_pool"])
+
+
+def test_heal_leaves_a_different_account_alone(fleet):
+ """A profile row whose JWT identity names ANOTHER account is not root's grant."""
+ import base64
+ from agent.credential_pool import load_pool
+
+ def jwt(sub):
+ payload = base64.urlsafe_b64encode(json.dumps({"sub": sub, "exp": int(time.time()) + 3600}).encode()).rstrip(b"=")
+ return "h." + payload.decode() + ".s"
+
+ root_store = json.loads((fleet["root"] / "auth.json").read_text())
+ root_store["credential_pool"]["xai-oauth"] = [{
+ "id": "rootx", "auth_type": "oauth", "priority": 0, "source": "manual:device_code",
+ "access_token": jwt("alice"), "refresh_token": "xr-alice",
+ }]
+ (fleet["root"] / "auth.json").write_text(json.dumps(root_store))
+ kid = _profile(fleet, "kid")
+ kid.mkdir(parents=True, exist_ok=True)
+ (kid / "auth.json").write_text(json.dumps({
+ "version": 1, "providers": {},
+ "credential_pool": {"xai-oauth": [
+ {"id": "kidx", "auth_type": "oauth", "priority": 0, "source": "manual:device_code",
+ "access_token": jwt("bob"), "refresh_token": "xr-bob"},
+ {"id": "kidk", "auth_type": "api_key", "priority": 1, "source": "manual",
+ "access_token": "xai-static"},
+ ]},
+ }))
+ fleet["use"](kid)
+ load_pool("xai-oauth")
+ rows = (json.loads((kid / "auth.json").read_text())["credential_pool"])["xai-oauth"]
+ assert [r["id"] for r in rows] == ["kidx", "kidk"]
+ assert json.loads((fleet["root"] / "auth.json").read_text())["credential_pool"]["xai-oauth"][0]["refresh_token"] == "xr-alice"
+
+
+def test_heal_pkce_singleton_shape_commits_live_pair_to_root_singleton(fleet):
+ """`hermes auth` PKCE shape: root + profile each have .anthropic_oauth.json +
+ a hermes_pkce-seeded row; the profile's copy is the rotated (live) one."""
+ from agent.credential_pool import load_pool
+
+ root = fleet["root"]
+ store = json.loads((root / "auth.json").read_text())
+ store["active_provider"] = "anthropic"
+ del store["credential_pool"]["anthropic"]
+ (root / "auth.json").write_text(json.dumps(store))
+ (root / ".anthropic_oauth.json").write_text(json.dumps({
+ "accessToken": "sk-ant-oat01-AT0", "refreshToken": "sk-ant-ort-RT0",
+ "expiresAt": int((time.time() - 3600) * 1000),
+ }))
+ fleet["use"](root)
+ load_pool("anthropic") # seeds root's hermes_pkce row from the singleton
+
+ kid = _profile(fleet, "kid")
+ kid.mkdir(parents=True, exist_ok=True)
+ import shutil
+ shutil.copy2(root / "auth.json", kid / "auth.json")
+ (kid / ".anthropic_oauth.json").write_text(json.dumps({
+ "accessToken": "sk-ant-oat01-AT1", "refreshToken": "sk-ant-ort-RT1",
+ "expiresAt": int((time.time() - 60) * 1000),
+ }))
+ kstore = json.loads((kid / "auth.json").read_text())
+ kstore["credential_pool"]["anthropic"][0].update(
+ access_token="sk-ant-oat01-AT1", refresh_token="sk-ant-ort-RT1",
+ expires_at_ms=int((time.time() - 60) * 1000),
+ )
+ (kid / "auth.json").write_text(json.dumps(kstore))
+ srv = fleet["server"]
+ srv["spent"].add("sk-ant-ort-RT0"); srv["valid"] = {"sk-ant-ort-RT1"}; srv["n"] = 1
+
+ fleet["use"](kid)
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT2"
+ assert not (kid / ".anthropic_oauth.json").exists()
+ assert fleet["rows"](kid) is None
+ assert json.loads((root / ".anthropic_oauth.json").read_text())["refreshToken"] == "sk-ant-ort-RT2"
+ fleet["use"](root)
+ sel = load_pool("anthropic").select()
+ assert sel is not None and sel.access_token == "sk-ant-oat01-AT2"
+ assert [e[0] for e in srv["log"]] == ["ROTATE"], srv["log"]
+
+
+def test_heal_is_a_noop_in_classic_mode(fleet):
+ from hermes_cli.auth import heal_forked_single_use_oauth_grants
+ fleet["use"](fleet["root"])
+ before = (fleet["root"] / "auth.json").read_text()
+ assert heal_forked_single_use_oauth_grants("anthropic") is None
+ assert (fleet["root"] / "auth.json").read_text() == before
diff --git a/tests/agent/test_curator.py b/tests/agent/test_curator.py
index eac61e01c5..f14ef73ea9 100644
--- a/tests/agent/test_curator.py
+++ b/tests/agent/test_curator.py
@@ -931,7 +931,7 @@ def test_review_fork_toolset_surface_excludes_execution_tools():
# The incident class stays out: no command execution, no background
# process steering (stdin is a second unguarded write sink), and no
# generic filesystem-write tool.
- for tool in ("terminal", "process", "write_file", "patch",
+ for tool in ("terminal", "process_manage", "write_file", "patch",
"execute_code", "computer_use", "browser_exec"):
assert tool not in surface, (
f"execution/write tool {tool!r} leaked into the curator fork's "
diff --git a/tests/agent/test_display_todo_progress.py b/tests/agent/test_display_todo_progress.py
index d182be9269..3d6d657ca5 100644
--- a/tests/agent/test_display_todo_progress.py
+++ b/tests/agent/test_display_todo_progress.py
@@ -26,7 +26,7 @@ class TestTodoRead:
"""get_cute_tool_message(…, result=…) when todos_arg is None (read path)."""
def test_read_no_result(self):
- msg = get_cute_tool_message("todo", {}, 0.5)
+ msg = get_cute_tool_message("todo_list", {}, 0.5)
assert "reading tasks" in msg
assert "0.5s" in msg
@@ -34,7 +34,7 @@ class TestTodoRead:
def test_read_zero_total(self):
"""Edge case: empty todo list returns summary with total=0."""
- msg = get_cute_tool_message("todo", {}, 0.5,
+ msg = get_cute_tool_message("todo_list", {}, 0.5,
result=_todo_result(0, 0))
assert "reading tasks" in msg
@@ -46,7 +46,7 @@ class TestTodoCreate:
def test_create_default(self):
"""Brand-new plan: all pending, no result — plain count."""
- msg = get_cute_tool_message("todo",
+ msg = get_cute_tool_message("todo_list",
{"todos": [
{"id": "a", "content": "x", "status": "pending"},
]}, 0.3)
@@ -58,7 +58,7 @@ class TestTodoCreate:
def test_create_with_result_zero_done(self):
"""New plan with 0 done — plain count, no progress fraction."""
- msg = get_cute_tool_message("todo",
+ msg = get_cute_tool_message("todo_list",
{"todos": [
{"id": "a", "content": "x", "status": "pending"},
{"id": "b", "content": "y", "status": "pending"},
@@ -74,7 +74,7 @@ class TestTodoUpdate:
def test_update_no_result(self):
"""No result available — plain update N task(s)."""
- msg = get_cute_tool_message("todo",
+ msg = get_cute_tool_message("todo_list",
{"todos": [{"id": "a", "status": "completed"}],
"merge": True}, 0.5)
assert "update 1 task(s)" in msg
@@ -82,7 +82,7 @@ class TestTodoUpdate:
def test_update_halfway(self):
"""2/4 — midpoint progress."""
- msg = get_cute_tool_message("todo",
+ msg = get_cute_tool_message("todo_list",
{"todos": [{"id": "b", "status": "in_progress"}],
"merge": True},
0.7,
@@ -96,7 +96,7 @@ class TestTodoUpdate:
def test_update_total_not_in_summary(self):
"""Result summary missing total key."""
- msg = get_cute_tool_message("todo",
+ msg = get_cute_tool_message("todo_list",
{"todos": [{"id": "a", "status": "completed"}],
"merge": True},
0.3,
@@ -111,7 +111,7 @@ class TestTodoEdgeCases:
def test_merge_default_value(self):
"""merge defaults to False in function signature, should be False when absent."""
- msg = get_cute_tool_message("todo",
+ msg = get_cute_tool_message("todo_list",
{"todos": [{"id": "a", "content": "x", "status": "pending"}]},
1.0)
assert "1 task(s)" in msg
@@ -120,7 +120,7 @@ class TestTodoEdgeCases:
def test_large_task_count(self):
"""Many tasks should not break formatting."""
many = [{"id": str(i), "content": "x", "status": "pending"} for i in range(50)]
- msg = get_cute_tool_message("todo", {"todos": many}, 0.5)
+ msg = get_cute_tool_message("todo_list", {"todos": many}, 0.5)
assert "50 task(s)" in msg
@@ -131,7 +131,7 @@ class TestTodoSkinIntegration:
"""
def test_default_skin_prefix(self):
- msg = get_cute_tool_message("todo", {}, 0.5)
+ msg = get_cute_tool_message("todo_list", {}, 0.5)
assert msg.startswith("┊")
diff --git a/tests/agent/test_fast_mode_auto.py b/tests/agent/test_fast_mode_auto.py
new file mode 100644
index 0000000000..a9af81a8df
--- /dev/null
+++ b/tests/agent/test_fast_mode_auto.py
@@ -0,0 +1,143 @@
+"""Bounded /fast auto|cold windows and the shared route-aware gate."""
+
+from types import SimpleNamespace
+
+from agent import fast_mode
+
+
+def _agent(**kw):
+ base = dict(
+ service_tier="auto",
+ model="gpt-5.4",
+ provider="openai",
+ base_url="https://api.openai.com/v1",
+ api_mode="chat_completions",
+ request_overrides={"extra_body": {"keep": 1}},
+ fast_auto_seconds=60,
+ )
+ base.update(kw)
+ return SimpleNamespace(**base)
+
+
+def test_bounded_fast_window_policy(monkeypatch):
+ clock = [1000.0]
+ monkeypatch.setattr(fast_mode.time, "monotonic", lambda: clock[0])
+
+ # auto: window open -> fast override layered over existing overrides
+ agent = _agent()
+ fast_mode.begin_turn(agent, conversation_history=[])
+ assert fast_mode.effective_request_overrides(agent) == {
+ "extra_body": {"keep": 1},
+ "service_tier": "priority",
+ }
+ assert agent.request_overrides == {"extra_body": {"keep": 1}} # never mutated
+
+ # window expired -> override absent
+ clock[0] += 61
+ assert fast_mode.effective_request_overrides(agent) == {"extra_body": {"keep": 1}}
+
+ # auto re-opens on the next turn
+ fast_mode.begin_turn(agent, conversation_history=[{"role": "user", "content": "x"}])
+ assert "service_tier" in fast_mode.effective_request_overrides(agent)
+
+ # cold: prior history -> no window at all
+ cold = _agent(service_tier="cold")
+ fast_mode.begin_turn(cold, conversation_history=[{"role": "user", "content": "x"}])
+ assert "service_tier" not in fast_mode.effective_request_overrides(cold)
+ fast_mode.begin_turn(cold, conversation_history=None)
+ assert fast_mode.effective_request_overrides(cold)["service_tier"] == "priority"
+
+ # Anthropic route uses the speed param
+ anth = _agent(
+ service_tier="auto",
+ model="claude-opus-5",
+ provider="anthropic",
+ base_url="https://api.anthropic.com",
+ api_mode="anthropic_messages",
+ )
+ fast_mode.begin_turn(anth, conversation_history=[])
+ assert fast_mode.effective_request_overrides(anth)["speed"] == "fast"
+
+ # unsupported routes never get fast params, in auto or static mode
+ from hermes_cli.models import resolve_fast_mode_overrides
+
+ for provider, base_url in (
+ ("openrouter", "https://openrouter.ai/api/v1"),
+ ("nous", "https://inference-api.nousresearch.com/v1"),
+ ("copilot", "https://api.githubcopilot.com"),
+ ("azure", "https://foo.openai.azure.com"),
+ ("custom", "http://10.0.0.1:8000/v1"),
+ ("openai", "https://proxy.example.com/v1"),
+ ):
+ proxied = _agent(provider=provider, base_url=base_url)
+ fast_mode.begin_turn(proxied, conversation_history=[])
+ assert "service_tier" not in fast_mode.effective_request_overrides(proxied), provider
+ assert resolve_fast_mode_overrides("gpt-5.4", provider=provider, base_url=base_url) is None
+ assert resolve_fast_mode_overrides(
+ "claude-opus-5", provider="bedrock", base_url="https://bedrock-runtime.us-east-1.amazonaws.com"
+ ) is None
+ # first-party routes (and the legacy model-only call) still resolve
+ assert resolve_fast_mode_overrides("gpt-5.4", provider="openai-codex", base_url="https://chatgpt.com/backend-api/codex")
+ assert resolve_fast_mode_overrides("grok-4.6", provider="xai", base_url="https://api.x.ai/v1")
+ assert resolve_fast_mode_overrides("gpt-5.4") == {"service_tier": "priority"}
+
+ # normal / static modes are untouched by the window logic
+ static = _agent(service_tier="priority", request_overrides={"service_tier": "priority"})
+ fast_mode.begin_turn(static, conversation_history=[])
+ assert fast_mode.effective_request_overrides(static) == {"service_tier": "priority"}
+ off = _agent(service_tier=None)
+ fast_mode.begin_turn(off, conversation_history=[])
+ assert fast_mode.effective_request_overrides(off) == {"extra_body": {"keep": 1}}
+
+
+def test_fast_auto_and_cold_parse_and_slash_command(monkeypatch):
+ import hermes_cli.config as config_mod
+
+ if not hasattr(config_mod, "save_env_value_secure"):
+ config_mod.save_env_value_secure = lambda key, value: {"success": True}
+ import cli as cli_mod
+ from gateway.run import GatewayRunner
+ from hermes_cli.commands import COMMAND_REGISTRY
+ from hermes_cli.config import DEFAULT_CONFIG
+
+ # config parsing: CLI, gateway, TUI all accept auto/cold; default stays off
+ for raw, expected in (("auto", "auto"), ("COLD", "cold"), ("fast", "priority"), ("", None), ("bogus", None)):
+ assert cli_mod._parse_service_tier_config(raw) == expected
+ monkeypatch.setattr(
+ "gateway.run._load_gateway_runtime_config", lambda: {"agent": {"service_tier": raw}}
+ )
+ assert GatewayRunner._load_service_tier() == expected
+ assert DEFAULT_CONFIG["agent"]["service_tier"] == ""
+ assert DEFAULT_CONFIG["agent"]["fast_auto_seconds"] == 60
+
+ # /fast auto — session-scoped, agent rebuilt, status reports the mode
+ fast_cmd = next(c for c in COMMAND_REGISTRY if c.name == "fast")
+ assert {"auto", "cold"} <= set(fast_cmd.subcommands)
+ printed = []
+ monkeypatch.setattr(cli_mod, "_cprint", lambda *a, **k: printed.append(" ".join(map(str, a))))
+ monkeypatch.setattr(cli_mod, "save_config_value", lambda *a, **k: (_ for _ in ()).throw(AssertionError("no config write")))
+ stub = SimpleNamespace(
+ service_tier=None, model="gpt-5.4", agent=object(), _fast_command_available=lambda: True
+ )
+ cli_mod.HermesCLI._handle_fast_command(stub, "/fast auto")
+ assert stub.service_tier == "auto"
+ assert stub.agent is None
+ cli_mod.HermesCLI._handle_fast_command(stub, "/fast status")
+ assert any("auto" in line for line in printed)
+ cli_mod.HermesCLI._handle_fast_command(stub, "/fast cold")
+ assert stub.service_tier == "cold"
+
+ # auto/cold do NOT pin a static override into the turn route
+ route_stub = SimpleNamespace(
+ model="gpt-5.4", api_key="k", base_url="https://api.openai.com/v1", provider="openai",
+ api_mode="chat_completions", acp_command=None, acp_args=[], _credential_pool=None,
+ service_tier="auto",
+ )
+ assert cli_mod.HermesCLI._resolve_turn_agent_config(route_stub, "hi")["request_overrides"] is None
+ route_stub.service_tier = "priority"
+ assert cli_mod.HermesCLI._resolve_turn_agent_config(route_stub, "hi")["request_overrides"] == {
+ "service_tier": "priority"
+ }
+ route_stub.base_url = "https://openrouter.ai/api/v1"
+ route_stub.provider = "openrouter"
+ assert cli_mod.HermesCLI._resolve_turn_agent_config(route_stub, "hi")["request_overrides"] is None
diff --git a/tests/agent/test_model_metadata.py b/tests/agent/test_model_metadata.py
index a6334eb71c..3d0ccd4101 100644
--- a/tests/agent/test_model_metadata.py
+++ b/tests/agent/test_model_metadata.py
@@ -799,6 +799,36 @@ class TestFetchEndpointModelMetadata:
not_found.close.assert_called_once()
success.close.assert_called_once()
+ def test_remote_probe_is_memoized_on_disk_across_processes(self, tmp_path, monkeypatch):
+ """A fresh process (cleared in-memory cache) must answer from the disk
+ memo within the TTL instead of re-probing the endpoint — the cost every
+ one-shot Bot Mode DM hop paid on startup. Expired memos re-probe."""
+ import agent.model_metadata as mm
+
+ monkeypatch.setattr(
+ mm, "_get_endpoint_metadata_cache_path", lambda: tmp_path / "endpoint_model_metadata.json"
+ )
+ success = MagicMock()
+ success.status_code = 200
+ success.json.return_value = {"data": [{"id": "test/model", "context_length": 32768}]}
+
+ with patch("agent.model_metadata.requests.get", return_value=success) as mock_get:
+ assert mm.fetch_endpoint_model_metadata("https://custom.example/v1")["test/model"]["context_length"] == 32768
+ # "New process": drop the in-memory cache only.
+ mm._endpoint_model_metadata_cache.clear()
+ mm._endpoint_model_metadata_cache_time.clear()
+ assert mm.fetch_endpoint_model_metadata("https://custom.example/v1")["test/model"]["context_length"] == 32768
+ mock_get.assert_called_once()
+
+ # Past the TTL the memo is stale and the endpoint is probed again.
+ mm._endpoint_model_metadata_cache.clear()
+ mm._endpoint_model_metadata_cache_time.clear()
+ with patch("agent.model_metadata.time.time", return_value=time.time() + mm._ENDPOINT_MODEL_CACHE_TTL + 1), patch(
+ "agent.model_metadata.requests.get", return_value=success
+ ) as mock_get:
+ mm.fetch_endpoint_model_metadata("https://custom.example/v1")
+ mock_get.assert_called_once()
+
# =========================================================================
# Nous Portal context-window resolution (provider="nous")
diff --git a/tests/agent/test_outbound_webhooks.py b/tests/agent/test_outbound_webhooks.py
index 29e05a4cdb..39c8a25221 100644
--- a/tests/agent/test_outbound_webhooks.py
+++ b/tests/agent/test_outbound_webhooks.py
@@ -301,6 +301,23 @@ class TestPayload:
assert payload["delivery_id"] == "did_1234"
assert payload["timestamp"].endswith("Z")
+ def test_profile_field_reflects_bound_profile_home(self, tmp_path, monkeypatch):
+ """Receivers behind a multiplexed gateway need to know which profile
+ fired (#92674): ``profile`` follows the bound home at fire time."""
+ from hermes_constants import reset_hermes_home_override, set_hermes_home_override
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ profile_home = tmp_path / "profiles" / "b"
+ profile_home.mkdir(parents=True)
+ token = set_hermes_home_override(profile_home)
+ try:
+ body = outbound_webhooks._serialize_payload("on_session_end", {}, "did_1")
+ finally:
+ reset_hermes_home_override(token)
+ assert json.loads(body)["profile"] == "b"
+ body = outbound_webhooks._serialize_payload("on_session_end", {}, "did_2")
+ assert json.loads(body)["profile"] == "default"
+
def test_unserialisable_values_stringified(self):
body = outbound_webhooks._serialize_payload(
"on_session_end", {"weird": object()}, "did_1"
@@ -345,6 +362,46 @@ class TestRegistration:
assert len(http_server.captured) == 1
+class TestForceReloadHomeScoping:
+ """Force-reloading one profile's plugin manager must restore that
+ profile's own outbound webhook and leave it firing exactly once —
+ the mirror of the shell-hook force-reload symmetry fix (#92682
+ review: outbound webhooks were the "same symptom class... after a
+ supported lifecycle transition instead of initial startup").
+ """
+
+ def test_force_reload_restores_webhook_and_fires_once(
+ self, monkeypatch, http_server,
+ ):
+ from hermes_cli import plugins
+
+ cfg = _cfg({"url": _url(http_server), "events": ["on_session_end"]})
+ monkeypatch.setattr("hermes_cli.config.load_config", lambda: cfg)
+
+ monkeypatch.setenv("HERMES_HOME", "/tmp/profile-b-webhook")
+ mgr_b = plugins.PluginManager()
+ plugins._plugin_manager = mgr_b
+ outbound_webhooks.register_from_config(cfg)
+ assert len(mgr_b._hooks.get("on_session_end", [])) == 1
+
+ # Force-reload: unload() wipes _hooks (config-owned webhook
+ # callbacks included, same as the ledger-driven plugin sweep), so
+ # without the fix the idempotence key alone would survive and a
+ # later register_from_config() call would see it and skip
+ # re-wiring — leaving the webhook silently inert.
+ mgr_b.unload()
+ assert mgr_b._hooks.get("on_session_end", []) == []
+
+ outbound_webhooks.re_register_config_hooks()
+ assert len(mgr_b._hooks.get("on_session_end", [])) == 1
+
+ plugins.get_plugin_manager().invoke_hook(
+ "on_session_end", session_id="s1",
+ )
+ assert outbound_webhooks.flush()
+ assert len(http_server.captured) == 1
+
+
# ── E2E delivery against a real HTTP server ──────────────────────────────
diff --git a/tests/agent/test_phantom_tool_references.py b/tests/agent/test_phantom_tool_references.py
index 045f356198..836522827a 100644
--- a/tests/agent/test_phantom_tool_references.py
+++ b/tests/agent/test_phantom_tool_references.py
@@ -65,8 +65,8 @@ class TestCodingBriefTodoGating:
return prefix[0]
def test_todo_kept_when_tool_available(self):
- brief = self._brief({"todo", "terminal", "read_file"})
- assert "Track multi-step work with `todo`" in brief
+ brief = self._brief({"todo_list", "terminal", "read_file"})
+ assert "Track multi-step work with `todo_list`" in brief
def test_todo_dropped_when_tool_missing(self):
brief = self._brief({"terminal", "read_file"})
@@ -76,7 +76,7 @@ class TestCodingBriefTodoGating:
def test_unknown_toolset_keeps_full_brief(self):
brief = self._brief(None)
- assert "Track multi-step work with `todo`" in brief
+ assert "Track multi-step work with `todo_list`" in brief
class TestEssentialSkillsUndisableable:
diff --git a/tests/agent/test_redact.py b/tests/agent/test_redact.py
index e51dfdf12b..d70c8146c1 100644
--- a/tests/agent/test_redact.py
+++ b/tests/agent/test_redact.py
@@ -100,6 +100,37 @@ class TestEnvAssignments:
result = redact_sensitive_text(text)
assert result == text
+ @pytest.mark.parametrize(
+ "text",
+ [
+ 'IDENTITY_TOKEN="bailu"',
+ "--override-tensor per_layer_token_embd.weight=CPU",
+ 'runtime.token="local"',
+ '{"token": "CPU"}',
+ "token: CPU",
+ ],
+ )
+ def test_ambiguous_key_preserves_obviously_noncredential_value(self, text):
+ assert redact_sensitive_text(text, force=True) == text
+
+ @pytest.mark.parametrize(
+ "text, cleartext",
+ [
+ ("PASSWORD=hunter2", "hunter2"),
+ ("SECRET_TOKEN=bailu", "bailu"),
+ ("id_token=local", "local"),
+ ("CUSTOM_TOKEN=opaqueValue123456789", "opaqueValue123456789"),
+ ('{"token": "opaqueValue123456789"}', "opaqueValue123456789"),
+ ('{"key_material": "CPU"}', "CPU"),
+ ('{"bearer": "local"}', "local"),
+ ("TOKEN=" + "sk-" + "a" * 30, "a" * 20),
+ ],
+ )
+ def test_strong_key_or_credential_shaped_value_still_redacts(
+ self, text, cleartext
+ ):
+ assert cleartext not in redact_sensitive_text(text, force=True)
+
@@ -1081,3 +1112,69 @@ class TestMaskSecretControlStripping:
def test_all_control_value_returns_empty_fallback(self):
assert mask_secret("\n\x85\u200b") == ""
assert mask_secret("\n\x85\u200b", empty="(not set)") == "(not set)"
+
+
+class TestValueAwareGatingCorpus:
+ """Issue #96607: corpus-level before/after for value-aware gating.
+
+ Redaction must mask a keyword-named assignment ONLY when the value has
+ credential shape (vendor prefix, hex/base64/high-entropy, or a strong
+ credential-specific key name). Bare technical vocabulary — ``token``,
+ ``key``, ``cpu`` — in ordinary technical prose/config must pass through
+ byte-for-byte, on every assignment family (ENV, dotted config, JSON,
+ YAML).
+ """
+
+ # Realistic technical prose. On pre-fix main every line was corrupted
+ # to ``***`` despite containing no secret.
+ TECHNICAL_CORPUS = [
+ 'IDENTITY_TOKEN="bailu"',
+ "--override-tensor per_layer_token_embd.weight=CPU",
+ "MAX_TOKENS=4096",
+ "runtime.token=local",
+ "The tokenizer splits on whitespace; set max_new_tokens=256.",
+ "num_key_value_heads=8",
+ "token: CPU",
+ "llm_load_tensors: per_layer_token_embd.weight=CPU buffer",
+ ]
+
+ # Obviously-fake but shape-realistic secrets: every one of these must
+ # STAY masked after the gating change (fail-closed on credential shape
+ # or strong key names).
+ FAKE_SECRET_CORPUS = [
+ ("API_KEY=sk-fakefakefakefakefake1234567890abcd", "fakefake"),
+ ("GITHUB_TOKEN=ghp_FAKEfakeFAKEfake1234567890fake", "FAKEfake"),
+ ("MY_SERVICE_TOKEN=A9f3kZq7Lm2Xw8Rt4Yv6", "A9f3kZq7"),
+ ("TOKEN=6f1d2a9c8b3e4f5a6d7c8b9a0e1f2d3c", "6f1d2a9c"),
+ ("password=hunter2", "hunter2"),
+ ("db_password: hunter2", "hunter2"),
+ ("auth_token: 9f8e7d6c5b4a39281706f5e4d3c2b1a0", "9f8e7d6c"),
+ ('"token": "Zx9Qw8Er7Ty6Ui5Op4As3"', "Zx9Qw8Er"),
+ ("SESSION_TOKEN=shrt", "shrt"),
+ ("client_secret=abc", "abc"),
+ ("spring.datasource.password=fakePass123", "fakePass123"),
+ ]
+
+ def test_technical_prose_survives_intact(self):
+ for line in self.TECHNICAL_CORPUS:
+ assert redact_sensitive_text(line, force=True) == line, line
+
+ def test_technical_corpus_as_one_block_survives_intact(self):
+ # The multi-line shape a model actually reads from tool output.
+ block = "\n".join(self.TECHNICAL_CORPUS)
+ assert redact_sensitive_text(block, force=True) == block
+
+ def test_shape_realistic_fake_secrets_still_masked(self):
+ for line, cleartext in self.FAKE_SECRET_CORPUS:
+ result = redact_sensitive_text(line, force=True)
+ assert result != line, line
+ assert cleartext not in result, line
+
+ def test_mixed_block_masks_only_the_secret_lines(self):
+ # Precondition guard: both halves must actually exercise the gate.
+ secret_line = "MY_SERVICE_TOKEN=A9f3kZq7Lm2Xw8Rt4Yv6"
+ prose_line = 'IDENTITY_TOKEN="bailu"'
+ block = f"{prose_line}\n{secret_line}"
+ result = redact_sensitive_text(block, force=True)
+ assert prose_line in result
+ assert "A9f3kZq7Lm2Xw8Rt4Yv6" not in result
diff --git a/tests/agent/test_refine_snapshot_isolation.py b/tests/agent/test_refine_snapshot_isolation.py
new file mode 100644
index 0000000000..234beb1993
--- /dev/null
+++ b/tests/agent/test_refine_snapshot_isolation.py
@@ -0,0 +1,101 @@
+"""Every review path hands the fork a snapshot that cannot alias the live transcript.
+
+``AIAgent._spawn_background_review`` is the single chokepoint the automatic
+post-turn review, the idle-queue deferral and both explicit ``/refine`` entry
+points (CLI mixin + gateway slash command) go through; it clones the snapshot
+structurally there. A shallow ``list()`` would share the nested
+``tool_calls`` / ``content`` containers with the persisted history, so the
+fork's in-place transcript sanitization would rewrite the parent's messages
+(#100795). These tests drive the real /refine handlers into the real
+chokepoint and capture what reaches the spawn.
+"""
+
+import threading
+from unittest.mock import MagicMock
+
+import pytest
+
+
+def _agent_with_real_chokepoint():
+ """MagicMock agent whose _spawn_background_review is the REAL method.
+
+ Everything below the chokepoint (thread spawn) is captured at
+ ``_spawn_background_review_now`` so no fork actually runs.
+ """
+ from run_agent import AIAgent
+
+ agent = MagicMock()
+ agent.valid_tool_names = {"memory"}
+ agent._delegate_depth = 0
+ agent._spawn_background_review = AIAgent._spawn_background_review.__get__(agent)
+ return agent
+
+
+def _nested_history():
+ return [
+ {"role": "user", "content": [{"type": "text", "text": "ask"}]},
+ {
+ "role": "assistant",
+ "content": "ok",
+ "tool_calls": [{
+ "id": "call-1",
+ "function": {"name": "read_file", "arguments": '{"path":"x"}'},
+ }],
+ },
+ ]
+
+
+def _assert_isolated(live, snapshot):
+ assert snapshot == live # same shape/bytes …
+ for live_msg, snap_msg in zip(live, snapshot):
+ assert snap_msg is not live_msg # … but no shared containers
+ for key in ("content", "tool_calls"):
+ if isinstance(live_msg.get(key), (dict, list)):
+ assert snap_msg[key] is not live_msg[key]
+ # Mutating the snapshot the way the fork's sanitizers do must not leak.
+ snapshot[0]["content"][0]["text"] = "mutated"
+ snapshot[1]["tool_calls"][0]["function"]["arguments"] = "{}"
+ assert live[0]["content"][0]["text"] == "ask"
+ assert live[1]["tool_calls"][0]["function"]["arguments"] == '{"path":"x"}'
+
+
+def test_cli_refine_snapshot_does_not_alias_live_history(monkeypatch):
+ from hermes_cli.cli_commands_mixin import CLICommandsMixin
+
+ monkeypatch.setattr("cli._cprint", lambda *a, **k: None, raising=False)
+ agent = _agent_with_real_chokepoint()
+ cli = object.__new__(CLICommandsMixin)
+ cli.agent = agent
+ cli.conversation_history = _nested_history()
+
+ cli._handle_refine_command("/refine")
+
+ agent._spawn_background_review_now.assert_called_once()
+ snapshot = agent._spawn_background_review_now.call_args.kwargs["messages_snapshot"]
+ _assert_isolated(cli.conversation_history, snapshot)
+
+
+@pytest.mark.asyncio
+async def test_gateway_refine_snapshot_does_not_alias_live_history():
+ from gateway.run import GatewayRunner
+
+ key = "agent:main:test:dm:1"
+ agent = _agent_with_real_chokepoint()
+ agent._session_messages = _nested_history()
+
+ runner = object.__new__(GatewayRunner)
+ runner._running_agents = {}
+ runner._agent_cache = {key: agent}
+ runner._agent_cache_lock = threading.Lock()
+ runner._session_key_for_source = lambda source: key
+
+ event = MagicMock()
+ event.source = object()
+ event.get_command_args.return_value = ""
+
+ out = await runner._handle_refine_command(event)
+
+ assert out.startswith("⚗")
+ agent._spawn_background_review_now.assert_called_once()
+ snapshot = agent._spawn_background_review_now.call_args.kwargs["messages_snapshot"]
+ _assert_isolated(agent._session_messages, snapshot)
diff --git a/tests/agent/test_review_idle_queue.py b/tests/agent/test_review_idle_queue.py
new file mode 100644
index 0000000000..43082f76e9
--- /dev/null
+++ b/tests/agent/test_review_idle_queue.py
@@ -0,0 +1,347 @@
+"""Deferred background review on the managed local runtime.
+
+Behavior contracts for agent/review_idle_queue.py and the decision
+wrapper in run_agent.AIAgent._spawn_background_review:
+
+- defer: auto + review runtime == managed local -> queued, not spawned
+- defer: never, or non-managed runtime, or /refine -> immediate spawn
+- queue coalesces per session (newest snapshot wins, age preserved)
+- dispatch requires sustained process-quiet AND server idle
+- aged-out items dispatch regardless of idleness (delay, never lose)
+- preempted deferred reviews requeue with a bounded attempt cap
+"""
+
+import threading
+import time
+import types
+
+import pytest
+
+from agent.review_idle_queue import (
+ ReviewIdleQueue,
+ _IDLE_SETTLE_S,
+ defer_max_age_s,
+ defer_mode,
+)
+
+
+# ── config parsing ───────────────────────────────────────────────
+
+
+def test_defer_mode_values():
+ assert defer_mode(None) == "auto"
+ assert defer_mode({}) == "auto"
+ assert defer_mode({"defer": "auto"}) == "auto"
+ assert defer_mode({"defer": "never"}) == "never"
+ assert defer_mode({"defer": "NEVER"}) == "never"
+ # Unknown values fall back to auto (the safe, documented default).
+ assert defer_mode({"defer": "sometimes"}) == "auto"
+ assert defer_mode({"defer": 3}) == "auto"
+
+
+def test_defer_max_age_parsing():
+ assert defer_max_age_s(None) == 30 * 60
+ assert defer_max_age_s({"defer_max_age_s": 120}) == 120.0
+ assert defer_max_age_s({"defer_max_age_s": "600"}) == 600.0
+ # Nonsense and non-positive fall back to the default.
+ assert defer_max_age_s({"defer_max_age_s": "soon"}) == 30 * 60
+ assert defer_max_age_s({"defer_max_age_s": 0}) == 30 * 60
+ assert defer_max_age_s({"defer_max_age_s": -5}) == 30 * 60
+
+
+# ── queue harness ────────────────────────────────────────────────
+
+
+class _FakeAgent:
+ def __init__(self):
+ self.spawned = []
+ self.session_id = "sess-x"
+
+ def _spawn_background_review_now(self, **kwargs):
+ self.spawned.append(kwargs)
+
+
+def _make_queue(now=None, server_idle=True):
+ q = ReviewIdleQueue()
+ clock = {"t": 0.0}
+ if now is None:
+ q._now = lambda: clock["t"]
+ else:
+ q._now = now
+ q._server_idle = lambda: server_idle
+ # Never start the real dispatcher thread in unit tests.
+ q._ensure_thread = lambda: None
+ return q, clock
+
+
+def test_enqueue_coalesces_per_session_newest_wins_oldest_age():
+ q, clock = _make_queue()
+ agent = _FakeAgent()
+
+ clock["t"] = 100.0
+ q.enqueue(agent, "s1", {"messages_snapshot": ["old"], "task_cfg": {}})
+ clock["t"] = 200.0
+ q.enqueue(agent, "s1", {"messages_snapshot": ["new"], "task_cfg": {}})
+ q.enqueue(agent, "s2", {"messages_snapshot": ["other"], "task_cfg": {}})
+
+ assert q.pending_count() == 2
+ with q._lock:
+ item = q._pending["s1"]
+ # Newest snapshot won, but the age clock kept the ORIGINAL enqueue
+ # time so a busy session cannot push its own age-out forever.
+ assert item.kwargs["messages_snapshot"] == ["new"]
+ assert item.enqueued_at == 100.0
+
+
+def test_dispatch_waits_for_sustained_quiet():
+ q, clock = _make_queue()
+ agent = _FakeAgent()
+ q.enqueue(agent, "s1", {"task_cfg": {}})
+
+ # A live turn: nothing dispatches.
+ q.note_turn_started()
+ assert q._pop_dispatchable() is None
+
+ # Turn finished, but the settle window hasn't elapsed.
+ q.note_turn_finished()
+ assert q._pop_dispatchable() is None
+
+ # Quiet long enough -> dispatchable.
+ clock["t"] += _IDLE_SETTLE_S + 1
+ item = q._pop_dispatchable()
+ assert item is not None and item.session_key == "s1"
+ assert q.pending_count() == 0
+
+
+def test_dispatch_blocked_by_busy_server():
+ q, clock = _make_queue(server_idle=False)
+ agent = _FakeAgent()
+ q.enqueue(agent, "s1", {"task_cfg": {}})
+ q.note_turn_started()
+ q.note_turn_finished()
+ clock["t"] += _IDLE_SETTLE_S + 1
+ # Process is quiet but the managed server has a processing slot
+ # (another profile's session, a live prefill): hold.
+ assert q._pop_dispatchable() is None
+ assert q.pending_count() == 1
+
+
+def test_aged_out_item_dispatches_despite_busy_server():
+ q, clock = _make_queue(server_idle=False)
+ agent = _FakeAgent()
+ q.enqueue(agent, "s1", {"task_cfg": {"defer_max_age_s": 60}})
+ q.note_turn_started() # never goes quiet
+ clock["t"] += 61
+ item = q._pop_dispatchable()
+ assert item is not None
+ assert item.session_key == "s1"
+
+
+def test_new_turn_resets_the_quiet_clock():
+ q, clock = _make_queue()
+ agent = _FakeAgent()
+ q.enqueue(agent, "s1", {"task_cfg": {}})
+ q.note_turn_started()
+ q.note_turn_finished()
+ clock["t"] += _IDLE_SETTLE_S - 2
+ # A new prompt arrives just before the settle window closes.
+ q.note_turn_started()
+ clock["t"] += 30
+ assert q._pop_dispatchable() is None # still live
+ q.note_turn_finished()
+ assert q._pop_dispatchable() is None # settle restarts
+ clock["t"] += _IDLE_SETTLE_S + 1
+ assert q._pop_dispatchable() is not None
+
+
+def test_nested_turns_require_all_to_finish():
+ q, clock = _make_queue()
+ agent = _FakeAgent()
+ q.enqueue(agent, "s1", {"task_cfg": {}})
+ q.note_turn_started()
+ q.note_turn_started()
+ q.note_turn_finished()
+ clock["t"] += _IDLE_SETTLE_S + 1
+ assert q._pop_dispatchable() is None # one turn still live
+ q.note_turn_finished()
+ clock["t"] += _IDLE_SETTLE_S + 1
+ assert q._pop_dispatchable() is not None
+
+
+# ── the decision wrapper ─────────────────────────────────────────
+
+
+def _wrapper_agent(monkeypatch, defer="auto", managed=True):
+ """A minimal object wearing the real _spawn_background_review."""
+ import run_agent
+ from agent import review_idle_queue as riq
+
+ agent = _FakeAgent()
+ agent._delegate_depth = 0
+ calls = {"enqueued": [], "spawned": []}
+
+ monkeypatch.setattr(
+ "agent.background_review.load_background_review_settings",
+ lambda: (True, {"defer": defer}),
+ )
+ monkeypatch.setattr(
+ riq, "review_targets_managed_local", lambda a, cfg: managed
+ )
+ monkeypatch.setattr(
+ riq.QUEUE, "enqueue",
+ lambda a, key, kw: calls["enqueued"].append((key, kw)),
+ )
+ agent._spawn_background_review_now = (
+ lambda **kw: calls["spawned"].append(kw)
+ )
+ bound = types.MethodType(run_agent.AIAgent._spawn_background_review, agent)
+ return bound, calls
+
+
+def test_wrapper_defers_managed_local_auto(monkeypatch):
+ spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
+ spawn([{"role": "user", "content": "hi"}], review_memory=True)
+ assert len(calls["enqueued"]) == 1
+ assert calls["spawned"] == []
+ key, kwargs = calls["enqueued"][0]
+ assert key == "sess-x"
+ assert kwargs["review_memory"] is True
+
+
+def test_wrapper_spawns_immediately_for_non_managed(monkeypatch):
+ spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=False)
+ spawn([{"role": "user", "content": "hi"}], review_skills=True)
+ assert calls["enqueued"] == []
+ assert len(calls["spawned"]) == 1
+
+
+def test_wrapper_defer_never_is_old_behavior(monkeypatch):
+ spawn, calls = _wrapper_agent(monkeypatch, defer="never", managed=True)
+ spawn([{"role": "user", "content": "hi"}], review_memory=True)
+ assert calls["enqueued"] == []
+ assert len(calls["spawned"]) == 1
+
+
+def test_wrapper_refine_bypasses_queue(monkeypatch):
+ spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
+ spawn([{"role": "user", "content": "hi"}], review_memory=True,
+ focus="save the deploy workflow")
+ assert calls["enqueued"] == []
+ assert len(calls["spawned"]) == 1
+ assert calls["spawned"][0]["focus"] == "save the deploy workflow"
+
+
+def test_wrapper_bare_refine_bypasses_queue(monkeypatch):
+ """/refine with no focus text is still explicit: never deferred."""
+ spawn, calls = _wrapper_agent(monkeypatch, defer="auto", managed=True)
+ spawn([{"role": "user", "content": "hi"}], review_memory=True,
+ focus=None, explicit=True)
+ assert calls["enqueued"] == []
+ assert len(calls["spawned"]) == 1
+
+
+def test_wrapper_cloud_fast_path_skips_runtime_resolution(monkeypatch):
+ """No managed server on the machine -> the classifier answers from the
+ TTL-cached netloc probe alone, without resolving the review runtime.
+ Guards the cloud-only turn tail from growing new work."""
+ from agent import review_idle_queue as riq
+
+ resolved = {"count": 0}
+
+ def _explode(agent, cfg):
+ resolved["count"] += 1
+ raise AssertionError("runtime resolution must not run")
+
+ monkeypatch.setattr(
+ "agent.auxiliary_client._managed_local_netloc", lambda: "")
+ monkeypatch.setattr(
+ "agent.background_review._resolve_review_runtime", _explode)
+ assert riq.review_targets_managed_local(object(), {}) is False
+ assert resolved["count"] == 0
+
+
+def test_dispatcher_rechecks_enabled_gate(monkeypatch):
+ """A review disabled while queued must not be resurrected at dispatch."""
+ from agent import review_idle_queue as riq
+
+ q, clock = _make_queue()
+ agent = _FakeAgent()
+ q.enqueue(agent, "s1", {"task_cfg": {}})
+ monkeypatch.setattr(
+ "agent.background_review.load_background_review_settings",
+ lambda: (False, {}),
+ )
+ item = None
+ clock["t"] += _IDLE_SETTLE_S + 1
+ q.note_turn_started()
+ q.note_turn_finished()
+ clock["t"] += _IDLE_SETTLE_S + 1
+ item = q._pop_dispatchable()
+ assert item is not None
+ assert q._still_enabled(item) is False
+
+
+# ── requeue on preemption ────────────────────────────────────────
+
+
+class _Run:
+ def __init__(self, cancelled):
+ self.cancel_requested = threading.Event()
+ if cancelled:
+ self.cancel_requested.set()
+
+
+def _requeue_agent(monkeypatch, managed=True):
+ import run_agent
+ from agent import review_idle_queue as riq
+
+ agent = _FakeAgent()
+ calls = {"enqueued": []}
+ monkeypatch.setattr(
+ riq, "review_targets_managed_local", lambda a, cfg: managed
+ )
+ monkeypatch.setattr(
+ riq.QUEUE, "enqueue",
+ lambda a, key, kw: calls["enqueued"].append(kw),
+ )
+ agent._REVIEW_REQUEUE_MAX_ATTEMPTS = (
+ run_agent.AIAgent._REVIEW_REQUEUE_MAX_ATTEMPTS
+ )
+ bound = types.MethodType(
+ run_agent.AIAgent._maybe_requeue_preempted_review, agent
+ )
+ return bound, calls
+
+
+def test_preempted_review_requeues(monkeypatch):
+ requeue, calls = _requeue_agent(monkeypatch)
+ requeue(_Run(cancelled=True),
+ {"task_cfg": {"defer": "auto"}, "focus": None,
+ "_requeue_attempts": 1})
+ assert len(calls["enqueued"]) == 1
+ # The attempt counter rides along so the cap survives the round trip.
+ assert calls["enqueued"][0]["_requeue_attempts"] == 1
+
+
+def test_completed_review_does_not_requeue(monkeypatch):
+ requeue, calls = _requeue_agent(monkeypatch)
+ requeue(_Run(cancelled=False),
+ {"task_cfg": {"defer": "auto"}, "focus": None,
+ "_requeue_attempts": 1})
+ assert calls["enqueued"] == []
+
+
+def test_requeue_attempt_cap(monkeypatch):
+ requeue, calls = _requeue_agent(monkeypatch)
+ requeue(_Run(cancelled=True),
+ {"task_cfg": {"defer": "auto"}, "focus": None,
+ "_requeue_attempts": 4})
+ assert calls["enqueued"] == []
+
+
+def test_requeue_skips_non_managed(monkeypatch):
+ requeue, calls = _requeue_agent(monkeypatch, managed=False)
+ requeue(_Run(cancelled=True),
+ {"task_cfg": {"defer": "auto"}, "focus": None,
+ "_requeue_attempts": 1})
+ assert calls["enqueued"] == []
diff --git a/tests/agent/test_secret_scope.py b/tests/agent/test_secret_scope.py
index 7e73f12dbc..5a42f842d8 100644
--- a/tests/agent/test_secret_scope.py
+++ b/tests/agent/test_secret_scope.py
@@ -347,3 +347,34 @@ class TestRelayRoutingStampGlobals:
ss.set_multiplex_active(False)
for name in self.AUTH_VARS:
assert not ss._is_global_env(name), name
+
+
+class TestSecretScopeAcrossExecutorThreads:
+ """Multiplexed profile state must reach pool workers (see #95119).
+
+ The context-compression timeout fence runs auxiliary LLM calls in a
+ daemon thread pool. Bundled CPython runtime builds omit
+ ``ThreadPoolExecutor``'s context propagation, so the profile secret
+ scope was absent in the worker and ``get_secret`` failed closed with
+ ``UnscopedSecretError``, silently degrading compression to lossy
+ deterministic summaries. ``DaemonThreadPoolExecutor.submit`` restores
+ stdlib context semantics; these tests lock that in.
+ """
+
+ def test_scoped_read_works_in_daemon_pool_worker(self, monkeypatch):
+ from tools.daemon_pool import DaemonThreadPoolExecutor
+
+ monkeypatch.setenv("SURPLUS_API_KEY", "env-key")
+ ss.set_multiplex_active(True)
+ token = ss.set_secret_scope({"SURPLUS_API_KEY": "scope-key"})
+ pool = DaemonThreadPoolExecutor(max_workers=1)
+ try:
+ # The scope (authoritative under multiplex) must reach the worker.
+ seen = pool.submit(ss.get_secret, "SURPLUS_API_KEY").result(timeout=10)
+ assert seen == "scope-key"
+ # A scoped miss must still not borrow the (cross-profile) env value.
+ monkeypatch.setenv("OPENAI_API_KEY", "env-leak")
+ assert pool.submit(ss.get_secret, "OPENAI_API_KEY").result(timeout=10) is None
+ finally:
+ pool.shutdown(wait=True)
+ ss.reset_secret_scope(token)
diff --git a/tests/agent/test_skill_commands.py b/tests/agent/test_skill_commands.py
index 623f5a5c05..02e7bfd149 100644
--- a/tests/agent/test_skill_commands.py
+++ b/tests/agent/test_skill_commands.py
@@ -255,6 +255,41 @@ class TestScanSkillCommands:
assert "/b-only" in profile_b_commands
assert "/a-only" not in profile_b_commands
+ def test_get_skill_commands_scans_profile_skills_dir_not_frozen_import_dir(self, tmp_path):
+ """Under a profile home override the scan must read /skills/,
+ not the launch home's import-time ``SKILLS_DIR`` (#67277): a
+ multiplexed webhook routed to profile B otherwise sees default's skills.
+ Deliberately does NOT patch ``tools.skills_tool.SKILLS_DIR``.
+ """
+ import agent.skill_commands as sc_mod
+ from agent.skill_commands import build_skill_invocation_message, get_skill_commands
+ from hermes_constants import reset_hermes_home_override, set_hermes_home_override
+
+ profile_b = tmp_path / "profiles" / "b"
+ _make_skill(profile_b / "skills", "b-only", body="Body of b-only.")
+ (profile_b / "config.yaml").write_text("{}\n")
+
+ with (
+ patch.object(sc_mod, "_skill_commands", {}),
+ patch.object(sc_mod, "_skill_commands_platform", None),
+ patch.object(sc_mod, "_skill_commands_home", None),
+ ):
+ token = set_hermes_home_override(profile_b)
+ try:
+ commands = dict(get_skill_commands())
+ assert "/b-only" in commands
+ # Frozen SKILLS_DIR (the launch home) must not leak in.
+ launch_dir = str(skills_tool_module._SKILLS_DIR_AT_IMPORT)
+ assert not any(
+ info["skill_dir"].startswith(launch_dir) for info in commands.values()
+ )
+ # And the absolute skill_dir round-trips through skill_view
+ # (normalize_skill_lookup_name must use the same live root).
+ msg = build_skill_invocation_message("/b-only", user_instruction="go")
+ finally:
+ reset_hermes_home_override(token)
+ assert msg is not None and "Body of b-only." in msg
+
def test_get_skill_commands_rescans_when_leaving_platform_scope(self, tmp_path, monkeypatch):
"""Returning to no-platform-scope (CLI / cron / RL) after a gateway
session must rescan so the unfiltered view is repopulated (#14536).
diff --git a/tests/agent/test_stall_guards.py b/tests/agent/test_stall_guards.py
index b79ba55c5b..013f8b83f8 100644
--- a/tests/agent/test_stall_guards.py
+++ b/tests/agent/test_stall_guards.py
@@ -92,7 +92,7 @@ def test_arg_canonicalization_ignores_key_order():
def test_allowlisted_pollers_never_fire():
c = ToolCallGuardrailController()
- for tool in ("process", "vendor_get_result", "job_poll"):
+ for tool in ("process_manage", "vendor_get_result", "job_poll"):
for _ in range(STALL_GUARD_IDENTICAL_CALL_THRESHOLD + 2):
assert c.observe_identical_call(tool, {"id": "j1"}, "Generating") is None
diff --git a/tests/agent/test_subagent_progress.py b/tests/agent/test_subagent_progress.py
index 4ec939780b..8fc656ac06 100644
--- a/tests/agent/test_subagent_progress.py
+++ b/tests/agent/test_subagent_progress.py
@@ -132,11 +132,12 @@ class TestBuildChildProgressCallback:
parent._delegate_spinner = spinner
parent.tool_progress_callback = None
- # task_index=0 in a batch of 3 → prefix "[1]"
+ # task_index=0 in a batch of 3 → prefix "[1/3]" (batch slot; a
+ # delegation batch tag is prepended when the id is known)
cb0 = _build_child_progress_callback(0, "test goal", parent, task_count=3)
cb0("tool.started", "web_search", "test", {})
output = buf.getvalue()
- assert "[1]" in output
+ assert "[1/3]" in output
# task_index=2 in a batch of 3 → prefix "[3]"
buf.truncate(0)
@@ -144,7 +145,7 @@ class TestBuildChildProgressCallback:
cb2 = _build_child_progress_callback(2, "test goal", parent, task_count=3)
cb2("tool.started", "web_search", "test", {})
output = buf.getvalue()
- assert "[3]" in output
+ assert "[3/3]" in output
diff --git a/tests/agent/test_summarize_tool_result_type_safety.py b/tests/agent/test_summarize_tool_result_type_safety.py
index 2899c9be27..f05cca0c0c 100644
--- a/tests/agent/test_summarize_tool_result_type_safety.py
+++ b/tests/agent/test_summarize_tool_result_type_safety.py
@@ -112,7 +112,7 @@ class TestBackstopWrapper:
"terminal", "read_file", "write_file", "search_files", "patch",
"browser_navigate", "web_search", "web_extract", "delegate_task",
"execute_code", "skill_view", "vision_analyze", "memory",
- "cronjob", "process", "totally_unknown_tool",
+ "cronjob_manage", "process_manage", "totally_unknown_tool",
]
keys = ["command", "path", "content", "pattern", "url", "query",
"urls", "goal", "code", "name", "question", "action",
@@ -151,12 +151,12 @@ class TestDisplayPreviewTypeSafety:
def test_process_preview_non_string_data(self):
from agent.display import build_tool_preview
result = build_tool_preview(
- "process", {"action": "submit", "session_id": "abc", "data": 42}
+ "process_manage", {"action": "submit", "session_id": "abc", "data": 42}
)
assert result == 'submit abc "42"'
def test_process_preview_none_action(self):
from agent.display import build_tool_preview
- result = build_tool_preview("process", {"action": None, "session_id": "abc"})
+ result = build_tool_preview("process_manage", {"action": None, "session_id": "abc"})
assert isinstance(result, str)
diff --git a/tests/agent/test_tool_guardrails.py b/tests/agent/test_tool_guardrails.py
index dbeb2d9d3f..63ae5debd3 100644
--- a/tests/agent/test_tool_guardrails.py
+++ b/tests/agent/test_tool_guardrails.py
@@ -33,6 +33,18 @@ def test_tool_call_signature_hashes_canonical_nested_unicode_args_without_exposi
assert "☤" not in json.dumps(metadata)
+def test_default_config_is_soft_warning_only_with_hard_stop_disabled():
+ cfg = ToolCallGuardrailConfig()
+
+ assert cfg.warnings_enabled is True
+ assert cfg.hard_stop_enabled is False
+ assert cfg.non_interactive_hard_stop_enabled is True
+ assert cfg.exact_failure_warn_after == 2
+ assert cfg.same_tool_failure_warn_after == 3
+ assert cfg.no_progress_warn_after == 2
+ assert cfg.exact_failure_block_after == 5
+ assert cfg.same_tool_failure_halt_after == 8
+ assert cfg.no_progress_block_after == 5
def test_config_parses_nested_warn_and_hard_stop_thresholds():
@@ -63,6 +75,29 @@ def test_config_parses_nested_warn_and_hard_stop_thresholds():
assert cfg.no_progress_block_after == 8
+def test_gateway_platform_defaults_to_hard_stop_without_changing_interactive_defaults():
+ interactive_configs = [
+ ToolCallGuardrailConfig.from_mapping({}, platform=platform)
+ for platform in ("cli", "tui", "desktop", "acp")
+ ]
+ telegram_cfg = ToolCallGuardrailConfig.from_mapping({}, platform="telegram")
+ cron_cfg = ToolCallGuardrailConfig.from_mapping({}, platform="cron")
+
+ assert all(cfg.hard_stop_enabled is False for cfg in interactive_configs)
+ assert telegram_cfg.hard_stop_enabled is True
+ assert cron_cfg.hard_stop_enabled is True
+
+
+def test_non_interactive_hard_stop_can_be_disabled_explicitly():
+ cfg = ToolCallGuardrailConfig.from_mapping(
+ {"non_interactive_hard_stop_enabled": False},
+ platform="telegram",
+ )
+
+ assert cfg.hard_stop_enabled is False
+ assert cfg.non_interactive_hard_stop_enabled is False
+
+
def test_default_repeated_identical_failed_call_warns_without_blocking():
controller = ToolCallGuardrailController()
args = {"query": "same"}
@@ -119,6 +154,41 @@ def test_hard_stop_enabled_blocks_repeated_exact_failure_before_next_execution()
+def test_skill_read_tools_are_idempotent_and_block_repeated_identical_success_output():
+ cases = [
+ (
+ "skill_view",
+ {"name": "gui-agent-ml-operations"},
+ '{"success":true,"name":"gui-agent-ml-operations","content":"same"}',
+ ),
+ (
+ "skills_list",
+ {"category": "mlops"},
+ '{"success":true,"skills":[{"name":"gui-agent-ml-operations"}]}',
+ ),
+ ]
+
+ for tool_name, args, result in cases:
+ controller = ToolCallGuardrailController(
+ ToolCallGuardrailConfig(
+ hard_stop_enabled=True,
+ no_progress_warn_after=2,
+ no_progress_block_after=2,
+ )
+ )
+
+ assert controller.before_call(tool_name, args).action == "allow"
+ assert controller.after_call(tool_name, args, result, failed=False).action == "allow"
+ assert controller.before_call(tool_name, args).action == "allow"
+ warn = controller.after_call(tool_name, args, result, failed=False)
+ assert warn.action == "warn"
+ assert warn.code == "idempotent_no_progress_warning"
+
+ blocked = controller.before_call(tool_name, args)
+ assert blocked.action == "block"
+ assert blocked.code == "idempotent_no_progress_block"
+
+
def test_mutating_or_unknown_tools_are_not_blocked_for_repeated_identical_success_output_by_default():
controller = ToolCallGuardrailController(
ToolCallGuardrailConfig(no_progress_warn_after=2, no_progress_block_after=2)
@@ -131,6 +201,49 @@ def test_mutating_or_unknown_tools_are_not_blocked_for_repeated_identical_succes
assert controller.after_call("custom_tool", {"x": 1}, "ok", failed=False).action == "allow"
+def test_identical_call_streak_halts_any_tool_when_hard_stop_enabled():
+ # #89069 / #100849 bundle: a model replaying the same SUCCESSFUL
+ # terminal/skill_view call with a byte-identical result is not covered by
+ # the idempotent_tools no-progress block. The consecutive-identical
+ # streak (observe_call) is tool-agnostic; under hard_stop it must halt.
+ controller = ToolCallGuardrailController(
+ ToolCallGuardrailConfig(hard_stop_enabled=True, no_progress_block_after=5)
+ )
+ args = {"command": "hermes config get memory.provider"}
+ for i in range(1, 5):
+ controller.after_call("terminal", args, "local\n", failed=False)
+ controller.observe_call("terminal", args, "local\n", failed=False)
+ assert controller.halt_decision is None, f"halted early at {i}"
+
+ controller.after_call("terminal", args, "local\n", failed=False)
+ controller.observe_call("terminal", args, "local\n", failed=False)
+ halt = controller.halt_decision
+ assert halt is not None and halt.should_halt
+ assert halt.code == "identical_call_streak_halt"
+ assert halt.tool_name == "terminal" and halt.count == 5
+
+
+def test_identical_call_streak_never_halts_when_hard_stop_disabled_or_for_pollers():
+ soft = ToolCallGuardrailController(
+ ToolCallGuardrailConfig(hard_stop_enabled=False, no_progress_block_after=2)
+ )
+ for _ in range(6):
+ soft.observe_call("terminal", {"command": "ls"}, "a\nb\n", failed=False)
+ assert soft.halt_decision is None # notice-only in interactive sessions
+
+ hard = ToolCallGuardrailController(
+ ToolCallGuardrailConfig(hard_stop_enabled=True, no_progress_block_after=2)
+ )
+ for _ in range(6):
+ hard.observe_call("process_manage", {"action": "poll", "session_id": "p1"}, "running", failed=False)
+ assert hard.halt_decision is None # an unchanged poll is legitimate progress
+
+ # A changed result resets the streak.
+ for i in range(6):
+ hard.observe_call("terminal", {"command": "date"}, f"t{i}", failed=False)
+ assert hard.halt_decision is None
+
+
@@ -177,3 +290,89 @@ def test_web_search_cap_blocks_after_limit_regardless_of_hard_stop():
+
+
+# ── Legitimate flows must survive hard stops (Teknium, Sep 2026) ────────────
+# Hard stops default ON for unattended platforms. These pin the flows that
+# must NEVER be cut off there: edit -> re-run loops, diagnostic sweeps of
+# distinct red commands, and browser retry-after-action — while the pure
+# replay (same call, nothing changed between attempts) is still stopped.
+
+_HARD = lambda: ToolCallGuardrailController( # noqa: E731
+ ToolCallGuardrailConfig(hard_stop_enabled=True)
+)
+_PYTEST = {"command": "pytest tests/test_x.py -q"}
+_RED = '{"output": "1 failed", "exit_code": 1}'
+
+
+def _run_red(c, args=_PYTEST):
+ assert c.before_call("terminal", args).allows_execution
+ return c.after_call("terminal", args, _RED, failed=True)
+
+
+def test_fix_retest_loop_is_never_hard_stopped():
+ c = _HARD()
+ for i in range(12):
+ d = _run_red(c)
+ assert not d.should_halt, f"halted on red run {i + 1}"
+ # the model edits between runs — a landed mutation is progress
+ c.after_call("patch", {"path": "x.py", "old_string": "a", "new_string": f"b{i}"},
+ '{"success": true, "diff": "..."}', failed=False)
+ assert c.halt_decision is None
+ assert c.before_call("terminal", _PYTEST).allows_execution
+
+
+def test_pure_replay_with_no_intervening_change_is_still_blocked():
+ c = _HARD()
+ for _ in range(5):
+ _run_red(c)
+ d = c.before_call("terminal", _PYTEST)
+ assert d.action == "block" and d.code == "repeated_exact_failure_block"
+
+
+def test_intervening_mutation_resets_the_replay_streak_only_once():
+ # 4 reds, one edit, then 4 reds with NO edit: the second run of 4 is a
+ # fresh streak, and the 5th unchanged retry after it is blocked.
+ c = _HARD()
+ for _ in range(4):
+ _run_red(c)
+ c.after_call("write_file", {"path": "x.py", "content": "y"}, '{"bytes_written": 1}', failed=False)
+ for _ in range(5):
+ assert c.before_call("terminal", _PYTEST).allows_execution
+ c.after_call("terminal", _PYTEST, _RED, failed=True)
+ assert c.before_call("terminal", _PYTEST).action == "block"
+
+
+def test_distinct_failing_terminal_commands_warn_but_never_halt():
+ # A diagnostic sweep: grep with no matches, missing binaries, red builds.
+ c = _HARD()
+ for i in range(12):
+ args = {"command": f"grep -q needle{i} haystack.txt"}
+ d = c.after_call("terminal", args, _RED, failed=True)
+ assert not d.should_halt, f"same_tool halt on distinct command #{i + 1}"
+ assert c.halt_decision is None
+ # ...while a non-tolerant tool failing 8 distinct ways still halts.
+ c2 = _HARD()
+ last = None
+ for i in range(8):
+ last = c2.after_call("send_message", {"to": f"u{i}"}, '{"error": "no route"}', failed=True)
+ assert last.should_halt and last.code == "same_tool_failure_halt"
+
+
+def test_browser_retry_after_action_is_not_a_replay():
+ c = _HARD()
+ nav = {"url": "https://example.test/app"}
+ for _ in range(8):
+ assert c.before_call("browser_navigate", nav).allows_execution
+ c.after_call("browser_navigate", nav, '{"error": "timeout"}', failed=True)
+ c.after_call("browser_click", {"selector": "#retry"}, '{"ok": true}', failed=False)
+ assert c.halt_decision is None
+
+
+def test_supervised_task_platforms_keep_warning_only_default():
+ for platform in ("subagent", "api_server", "cli"):
+ cfg = ToolCallGuardrailConfig.from_mapping({}, platform=platform)
+ assert cfg.hard_stop_enabled is False, platform
+ for platform in ("telegram", "discord", "cron", "kanban"):
+ cfg = ToolCallGuardrailConfig.from_mapping({}, platform=platform)
+ assert cfg.hard_stop_enabled is True, platform
diff --git a/tests/agent/test_turn_base_display_anchor.py b/tests/agent/test_turn_base_display_anchor.py
new file mode 100644
index 0000000000..6c1a6f1183
--- /dev/null
+++ b/tests/agent/test_turn_base_display_anchor.py
@@ -0,0 +1,201 @@
+"""Turn-base display anchor: the context meter shows durable-transcript cost.
+
+On reasoning models a long tool loop replays the current turn's thinking +
+scaffolding on every request, so the LAST request's ``prompt_tokens`` can
+exceed the durable transcript by hundreds of K — all of which evaporates at
+the turn boundary. Display surfaces (CLI status bar, /context breakdown)
+therefore anchor on the turn's FIRST response (``_turn_base_usage_anchor``)
+plus a stale-thinking-free delta estimate, instead of the raw last-request
+figure. Compression trigger math is unchanged (real last-request usage).
+
+Covers:
+ * anchored_context_tokens(charge_stale_thinking=False) excludes stale
+ reasoning text in the delta while keeping the newest assistant turn;
+ * the CLI status snapshot prefers the turn-base anchored figure over
+ compressor.last_prompt_tokens and falls back cleanly without an anchor;
+ * compute_session_context_breakdown prefers the turn-base anchor over the
+ last-response anchor;
+ * invalidation sites clear _turn_base_usage_anchor alongside _usage_anchor.
+"""
+
+from types import SimpleNamespace
+
+from agent.model_metadata import (
+ anchored_context_tokens,
+ capture_usage_anchor,
+ estimate_messages_tokens_rough,
+)
+
+
+def _msg(role, content, **extra):
+ m = {"role": role, "content": content}
+ m.update(extra)
+ return m
+
+
+class TestChargeStaleThinkingKwarg:
+ def test_delta_excludes_stale_reasoning(self):
+ messages = [_msg("user", "start"), _msg("assistant", "base reply")]
+ anchor = capture_usage_anchor(10_000, 100, messages)
+ assert anchor is not None
+
+ # Simulate a tool loop appending reasoning-heavy assistant turns.
+ big_thinking = "deliberation " * 5_000 # ~65K chars ≈ 16K tokens
+ messages.append(_msg("assistant", "the anchored reply itself"))
+ messages.append(
+ _msg("assistant", "step one", reasoning_content=big_thinking)
+ )
+ messages.append(_msg("tool", "tool output", tool_call_id="c1"))
+ messages.append(
+ _msg("assistant", "step two", reasoning_content=big_thinking)
+ )
+
+ charged = anchored_context_tokens(messages, anchor)
+ uncharged = anchored_context_tokens(
+ messages, anchor, charge_stale_thinking=False
+ )
+ assert charged is not None and uncharged is not None
+ # Stale thinking on the non-newest assistant message is excluded;
+ # the newest assistant message keeps its reasoning charge.
+ one_thinking_tokens = estimate_messages_tokens_rough(
+ [_msg("assistant", "", reasoning_content=big_thinking)]
+ )
+ assert charged - uncharged >= one_thinking_tokens * 0.9
+ assert uncharged >= 10_000 + 100 # anchor base still counted exactly
+
+ def test_default_remains_full_charge(self):
+ messages = [_msg("user", "s"), _msg("assistant", "r")]
+ anchor = capture_usage_anchor(1_000, 10, messages)
+ messages.append(_msg("assistant", "reply"))
+ assert anchored_context_tokens(messages, anchor) == anchored_context_tokens(
+ messages, anchor, charge_stale_thinking=True
+ )
+
+
+class TestCliStatusSnapshotPrefersTurnBaseAnchor:
+ def _agent_with(self, last_prompt_tokens, messages, anchor):
+ compressor = SimpleNamespace(
+ last_prompt_tokens=last_prompt_tokens,
+ context_length=1_000_000,
+ compression_count=0,
+ )
+ return SimpleNamespace(
+ context_compressor=compressor,
+ _session_messages=messages,
+ _turn_base_usage_anchor=anchor,
+ )
+
+ def _snapshot_context_tokens(self, agent):
+ """Mirror the cli.py snapshot block's context_tokens resolution."""
+ compressor = agent.context_compressor
+ context_tokens = getattr(compressor, "last_prompt_tokens", 0) or 0
+ if context_tokens < 0:
+ context_tokens = 0
+ msgs = getattr(agent, "_session_messages", None)
+ anchored = anchored_context_tokens(
+ msgs if isinstance(msgs, list) else [],
+ getattr(agent, "_turn_base_usage_anchor", None),
+ charge_stale_thinking=False,
+ )
+ if anchored is not None and anchored > 0:
+ context_tokens = anchored
+ return context_tokens
+
+ def test_turn_base_anchor_wins_over_inflated_last_request(self):
+ messages = [_msg("user", "start"), _msg("assistant", "reply")]
+ anchor = capture_usage_anchor(600_000, 500, messages)
+ messages.append(_msg("assistant", "anchored reply"))
+ agent = self._agent_with(850_000, messages, anchor)
+ # Bar shows the durable figure, not the inflated last request.
+ tokens = self._snapshot_context_tokens(agent)
+ assert 600_000 <= tokens < 650_000
+
+ def test_fallback_without_anchor(self):
+ agent = self._agent_with(123_456, [_msg("user", "x")], None)
+ assert self._snapshot_context_tokens(agent) == 123_456
+
+ def test_stale_anchor_falls_back(self):
+ messages = [_msg("user", "start"), _msg("assistant", "reply")]
+ anchor = capture_usage_anchor(50_000, 10, messages)
+ agent = self._agent_with(77_000, [_msg("user", "rebuilt")], anchor)
+ # Compaction rebuilt the list: structural check fails, raw fallback.
+ assert self._snapshot_context_tokens(agent) == 77_000
+
+ def test_negative_sentinel_still_clamped(self):
+ agent = self._agent_with(-1, [], None)
+ assert self._snapshot_context_tokens(agent) == 0
+
+
+class TestContextBreakdownPrefersTurnBaseAnchor:
+ def test_breakdown_uses_turn_base_over_last_response(self, monkeypatch):
+ from agent import context_breakdown as cb
+
+ messages = [_msg("user", "start"), _msg("assistant", "reply")]
+ turn_base = capture_usage_anchor(400_000, 200, messages)
+ messages.append(_msg("assistant", "anchored reply"))
+ last_anchor = capture_usage_anchor(900_000, 50, messages)
+
+ agent = SimpleNamespace(
+ _usage_anchor=last_anchor,
+ _turn_base_usage_anchor=turn_base,
+ _memory_store=None,
+ tools=[],
+ model="test/model",
+ context_compressor=SimpleNamespace(
+ context_length=1_000_000, last_prompt_tokens=900_000
+ ),
+ )
+ monkeypatch.setattr(
+ "agent.system_prompt.build_system_prompt_parts",
+ lambda a: {"stable": "sys", "context": "", "volatile": ""},
+ )
+ payload = cb.compute_session_context_breakdown(agent, messages)
+ assert 400_000 <= payload["context_used"] < 450_000
+
+ def test_breakdown_falls_back_to_last_response_anchor(self, monkeypatch):
+ from agent import context_breakdown as cb
+
+ messages = [_msg("user", "start"), _msg("assistant", "reply")]
+ last_anchor = capture_usage_anchor(300_000, 50, messages)
+
+ agent = SimpleNamespace(
+ _usage_anchor=last_anchor,
+ _turn_base_usage_anchor=None,
+ _memory_store=None,
+ tools=[],
+ model="test/model",
+ context_compressor=SimpleNamespace(
+ context_length=1_000_000, last_prompt_tokens=1
+ ),
+ )
+ monkeypatch.setattr(
+ "agent.system_prompt.build_system_prompt_parts",
+ lambda a: {"stable": "sys", "context": "", "volatile": ""},
+ )
+ payload = cb.compute_session_context_breakdown(agent, messages)
+ assert payload["context_used"] >= 300_000
+
+
+class TestInvalidationSitesClearTurnBaseAnchor:
+ def test_compression_invalidation_clears_both(self):
+ import inspect
+ from agent import conversation_compression
+
+ src = inspect.getsource(conversation_compression)
+ block = src.split("agent._usage_anchor = None", 1)[1][:200]
+ assert "_turn_base_usage_anchor = None" in block
+
+ def test_codex_native_invalidation_clears_both(self):
+ import inspect
+ from agent import codex_runtime
+
+ src = inspect.getsource(codex_runtime)
+ block = src.split("agent._usage_anchor = None", 1)[1][:200]
+ assert "_turn_base_usage_anchor = None" in block
+
+ def test_agent_init_defines_turn_base_anchor(self):
+ import inspect
+ from agent import agent_init
+
+ src = inspect.getsource(agent_init)
+ assert "_turn_base_usage_anchor = None" in src
diff --git a/tests/ci/test_classify_changes.py b/tests/ci/test_classify_changes.py
index 81e93d6809..43e33dd620 100644
--- a/tests/ci/test_classify_changes.py
+++ b/tests/ci/test_classify_changes.py
@@ -41,13 +41,14 @@ DEFAULT = {
"uv_lock": True,
"npm_lock": True,
"installer": True,
+ "desktop_updater": True,
"rust": True,
"mcp_catalog": False,
"ci_review": True,
}
-def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, uv_lock=False, npm_lock=False, installer=False, rust=False, mcp_catalog=False, docker_meta=False, ci_review=False, python_prod=None, nix=None, docker=None) -> dict[str, bool]:
+def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, uv_lock=False, npm_lock=False, installer=False, desktop_updater=False, rust=False, mcp_catalog=False, docker_meta=False, ci_review=False, python_prod=None, nix=None, docker=None) -> dict[str, bool]:
# python_prod tracks python except for tests-only diffs; default it to
# python so the majority of cases don't need to spell it out.
#
@@ -69,6 +70,7 @@ def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, uv_
"uv_lock": uv_lock,
"npm_lock": npm_lock,
"installer": installer,
+ "desktop_updater": desktop_updater,
"rust": rust,
"mcp_catalog": mcp_catalog,
"ci_review": ci_review,
@@ -78,7 +80,9 @@ def _lanes(python=False, frontend=False, site=False, scan=False, deps=False, uv_
CASES = {
"docs-only → nothing heavy": (["README.md", "docs/guide.md"], _lanes()),
"python source → python": (["run_agent.py"], _lanes(python=True, scan=True)),
- "dep manifest → python": (["pyproject.toml"], _lanes(python=True, scan=True, deps=True, uv_lock=True)),
+ # pyproject.toml declares the pytest markers the OS lanes select on, so it
+ # also re-arms the desktop_updater integration tests (fail-open).
+ "dep manifest → python": (["pyproject.toml"], _lanes(python=True, scan=True, deps=True, uv_lock=True, desktop_updater=True)),
"uv.lock → python": (["uv.lock"], _lanes(python=True, uv_lock=True)),
"ts package → frontend": (["apps/desktop/src/app.tsx"], _lanes(frontend=True)),
"ui-tui → frontend": (["ui-tui/src/entry.ts"], _lanes(frontend=True)),
@@ -141,6 +145,23 @@ CASES = {
_lanes(python=True, installer=True),
),
"python source alone → no installer lane": (["run_agent.py"], _lanes(python=True, scan=True)),
+ # The Windows desktop-update hand-off is a PowerShell integration surface:
+ # its tests spawn the real script and poll its loopback server. They run
+ # when the script, the Electron side that launches it, or their own test
+ # files change — not on every hermes_state.py PR.
+ "windows.ps1 → desktop_updater": (
+ ["scripts/desktop-update/windows.ps1"],
+ _lanes(python=True, desktop_updater=True),
+ ),
+ "desktop-update test → desktop_updater": (
+ ["tests/test_desktop_update_windows_progress.py"],
+ _lanes(python=True, python_prod=False, scan=True, desktop_updater=True),
+ ),
+ "updater-process.ts → desktop_updater": (
+ ["apps/desktop/electron/updater-process.ts"],
+ _lanes(frontend=True, desktop_updater=True),
+ ),
+ "python source alone → no desktop_updater lane": (["hermes_state.py"], _lanes(python=True, scan=True)),
# `.rs` lives under apps/, so it matches `frontend` too. That lane builds
# TypeScript and cannot notice a Rust error — before `rust` existed it was
# the ONLY lane a Rust change ran, and the crate's tests never executed.
@@ -168,9 +189,15 @@ CASES = {
# tests-only diffs: pytest lanes stay ON, product jobs (Desktop E2E,
# Docker) gate on python_prod and skip.
"tests-only → python without python_prod": (
- ["tests/agent/test_foo.py", "tests/conftest.py"],
+ ["tests/agent/test_foo.py"],
_lanes(python=True, python_prod=False, scan=True),
),
+ # conftest.py owns the _OS_MARKS skip logic, so it re-arms the
+ # desktop_updater integration tests too (fail-open).
+ "conftest → python + desktop_updater": (
+ ["tests/conftest.py"],
+ _lanes(python=True, python_prod=False, scan=True, desktop_updater=True),
+ ),
"tests + prod source → both lanes": (
["tests/agent/test_foo.py", "agent/x.py"],
_lanes(python=True, scan=True),
diff --git a/tests/cli/test_cli_clarify_batch.py b/tests/cli/test_cli_clarify_batch.py
index c839780bbb..a19575956c 100644
--- a/tests/cli/test_cli_clarify_batch.py
+++ b/tests/cli/test_cli_clarify_batch.py
@@ -341,3 +341,33 @@ class TestClarifyBatchNavigation:
thread.join(timeout=2)
assert result["value"] == {"answers": {"q0": "red", "q1": "small"}}
+
+
+class TestClarifyBellOnPrompt:
+ """display.bell_on_prompt rings BEL when a clarify modal opens; off is silent."""
+
+ @staticmethod
+ def _run_clarify(bell_on_prompt):
+ import io
+
+ cli = _make_cli_stub()
+ cli.bell_on_prompt = bell_on_prompt
+ out = io.StringIO()
+ with patch("cli.sys.stdout", out), patch(
+ "tools.clarify_gateway.resolve_clarify_timeout", return_value=60
+ ):
+ thread = threading.Thread(
+ target=cli._clarify_callback, args=("Color?", ["red", "blue"]), daemon=True
+ )
+ thread.start()
+ deadline = time.time() + 2
+ while cli._clarify_state is None and time.time() < deadline:
+ time.sleep(0.01)
+ assert cli._clarify_state is not None
+ cli._clarify_state["response_queue"].put("red")
+ thread.join(timeout=2)
+ return out.getvalue()
+
+ def test_bell_on_prompt_rings_and_off_is_silent(self):
+ assert "\a" in self._run_clarify(True)
+ assert "\a" not in self._run_clarify(False)
diff --git a/tests/cli/test_cli_init.py b/tests/cli/test_cli_init.py
index cca40f831e..059508991f 100644
--- a/tests/cli/test_cli_init.py
+++ b/tests/cli/test_cli_init.py
@@ -465,6 +465,25 @@ class TestNestedDictModelDefaultPairing:
assert "unrestricted" in output
assert "Slash commands: all available" in output
+ def test_provider_prefixed_startup_model_overrides_stale_provider(self):
+ cli = _make_cli(
+ config_overrides={
+ "model": {
+ "default": "anthropic/claude-opus-4.6",
+ "provider": "anthropic",
+ },
+ "providers": {
+ "nous": {
+ "base_url": "https://inference-api.nousresearch.com/v1",
+ },
+ },
+ },
+ model="nous/deepseek-v4-pro",
+ )
+
+ assert cli.model == "deepseek-v4-pro"
+ assert cli.requested_provider == "nous"
+
class TestRootLevelProviderOverride:
"""Root-level provider/base_url in config.yaml must NOT override model.provider."""
diff --git a/tests/cli/test_fast_command.py b/tests/cli/test_fast_command.py
index 87a6b2689c..203dfe329d 100644
--- a/tests/cli/test_fast_command.py
+++ b/tests/cli/test_fast_command.py
@@ -159,8 +159,8 @@ class TestFastModeRouting(unittest.TestCase):
stub = SimpleNamespace(
model="gpt-5.4",
api_key="primary-key",
- base_url="https://openrouter.ai/api/v1",
- provider="openrouter",
+ base_url="https://api.openai.com/v1",
+ provider="openai",
api_mode="chat_completions",
acp_command=None,
acp_args=[],
@@ -171,11 +171,16 @@ class TestFastModeRouting(unittest.TestCase):
route = cli_mod.HermesCLI._resolve_turn_agent_config(stub, "hi")
# Provider should NOT have changed
- assert route["runtime"]["provider"] == "openrouter"
+ assert route["runtime"]["provider"] == "openai"
assert route["runtime"]["api_mode"] == "chat_completions"
# But request_overrides should be set
assert route["request_overrides"] == {"service_tier": "priority"}
+ # Proxied routes (OpenRouter etc.) strip/400 on the param — never sent.
+ stub.base_url = "https://openrouter.ai/api/v1"
+ stub.provider = "openrouter"
+ assert cli_mod.HermesCLI._resolve_turn_agent_config(stub, "hi")["request_overrides"] is None
+
def test_turn_route_keeps_primary_runtime_when_model_has_no_fast_backend(self):
cli_mod = _import_cli()
stub = SimpleNamespace(
@@ -202,28 +207,36 @@ class TestAnthropicFastMode(unittest.TestCase):
def test_anthropic_opus_supported(self):
from hermes_cli.models import model_supports_fast_mode
+ # Per the live fast-mode docs: Opus 4.8 + Opus 5, Claude API only.
# Native Anthropic format (hyphens)
- assert model_supports_fast_mode("claude-opus-4-6") is True
+ assert model_supports_fast_mode("claude-opus-4-8") is True
# OpenRouter format (dots)
- assert model_supports_fast_mode("claude-opus-4.6") is True
+ assert model_supports_fast_mode("claude-opus-4.8") is True
# With vendor prefix
- assert model_supports_fast_mode("anthropic/claude-opus-4-6") is True
- assert model_supports_fast_mode("anthropic/claude-opus-4.6") is True
+ assert model_supports_fast_mode("anthropic/claude-opus-4-8") is True
+ assert model_supports_fast_mode("anthropic/claude-opus-4.8") is True
+ assert model_supports_fast_mode("claude-opus-5") is True
+ assert model_supports_fast_mode("anthropic/claude-opus-5") is True
- def test_anthropic_non_opus46_models_excluded(self):
- """The speed=fast parameter is gated to Opus 4.6 — others excluded.
+ def test_anthropic_unsupported_models_excluded(self):
+ """The speed=fast parameter is gated to Opus 4.8 / Opus 5.
- Per https://platform.claude.com/docs/en/build-with-claude/fast-mode,
- sending speed=fast to Opus 4.7, Sonnet, or Haiku returns HTTP 400.
- Opus 4.8 uses a separate ``…-fast`` model id, not this parameter.
+ Per https://platform.claude.com/docs/en/build-with-claude/fast-mode:
+ Opus 4.6 LOST fast mode 2026-06-29 (the param is silently ignored —
+ standard speed at standard billing — so a toggle would do nothing);
+ Opus 4.7 hard-400s; Sonnet/Haiku never had it; dedicated ``…-fast``
+ ids select fast inference via the model field, not the parameter.
"""
from hermes_cli.models import model_supports_fast_mode
assert model_supports_fast_mode("claude-sonnet-4-6") is False
assert model_supports_fast_mode("claude-sonnet-4.6") is False
assert model_supports_fast_mode("claude-haiku-4-5") is False
+ assert model_supports_fast_mode("claude-opus-4-6") is False
+ assert model_supports_fast_mode("claude-opus-4.6") is False
assert model_supports_fast_mode("claude-opus-4-7") is False
- assert model_supports_fast_mode("claude-opus-4-8") is False
+ assert model_supports_fast_mode("claude-opus-4-8-fast") is False
+ assert model_supports_fast_mode("anthropic/claude-opus-4.8-fast") is False
assert model_supports_fast_mode("anthropic/claude-sonnet-4.6") is False
assert model_supports_fast_mode("anthropic/claude-opus-4-7") is False
@@ -232,10 +245,10 @@ class TestAnthropicFastMode(unittest.TestCase):
def test_resolve_overrides_returns_speed_for_anthropic(self):
from hermes_cli.models import resolve_fast_mode_overrides
- result = resolve_fast_mode_overrides("claude-opus-4-6")
+ result = resolve_fast_mode_overrides("claude-opus-4-8")
assert result == {"speed": "fast"}
- result = resolve_fast_mode_overrides("anthropic/claude-opus-4.6")
+ result = resolve_fast_mode_overrides("anthropic/claude-opus-4.8")
assert result == {"speed": "fast"}
@@ -243,7 +256,7 @@ class TestAnthropicFastMode(unittest.TestCase):
def test_fast_command_hidden_for_anthropic_sonnet(self):
- """Sonnet doesn't support fast mode (Opus 4.6 only) — /fast must be hidden."""
+ """Sonnet doesn't support fast mode (Opus 4.8/5 only) — /fast must be hidden."""
cli_mod = _import_cli()
stub = SimpleNamespace(
provider="anthropic", requested_provider="anthropic",
@@ -257,7 +270,7 @@ class TestAnthropicFastMode(unittest.TestCase):
"""Anthropic models should get speed:'fast' override, not service_tier."""
cli_mod = _import_cli()
stub = SimpleNamespace(
- model="claude-opus-4-6",
+ model="claude-opus-4-8",
api_key="sk-ant-test",
base_url="https://api.anthropic.com",
provider="anthropic",
@@ -281,7 +294,7 @@ class TestAnthropicFastModeAdapter(unittest.TestCase):
from agent.anthropic_adapter import build_anthropic_kwargs, _FAST_MODE_BETA
kwargs = build_anthropic_kwargs(
- model="claude-opus-4-6",
+ model="claude-opus-4-8",
messages=[{"role": "user", "content": [{"type": "text", "text": "hi"}]}],
tools=None,
max_tokens=None,
@@ -297,7 +310,7 @@ class TestAnthropicFastModeAdapter(unittest.TestCase):
from agent.anthropic_adapter import build_anthropic_kwargs
kwargs = build_anthropic_kwargs(
- model="claude-opus-4-6",
+ model="claude-opus-4-8",
messages=[{"role": "user", "content": [{"type": "text", "text": "hi"}]}],
tools=None,
max_tokens=None,
@@ -312,7 +325,7 @@ class TestAnthropicFastModeAdapter(unittest.TestCase):
from agent.anthropic_adapter import build_anthropic_kwargs
kwargs = build_anthropic_kwargs(
- model="claude-opus-4-6",
+ model="claude-opus-4-8",
messages=[{"role": "user", "content": [{"type": "text", "text": "hi"}]}],
tools=None,
max_tokens=None,
diff --git a/tests/cli/test_personality_none.py b/tests/cli/test_personality_none.py
index ba4847607c..5a8752122c 100644
--- a/tests/cli/test_personality_none.py
+++ b/tests/cli/test_personality_none.py
@@ -87,7 +87,6 @@ class TestGatewayPersonalityNone:
def _make_runner(self, personalities=None):
from gateway.run import GatewayRunner
runner = GatewayRunner.__new__(GatewayRunner)
- runner._ephemeral_system_prompt = "You are kawaii~"
runner.config = {
"agent": {
"personalities": personalities or {"helpful": "You are helpful."}
@@ -125,7 +124,9 @@ class TestGatewayPersonalityNone:
saved = yaml.safe_load(config_file.read_text())
assert saved["agent"]["system_prompt"] == "manual forever"
assert saved.get("display", {}).get("personality", None) == ""
- assert runner._ephemeral_system_prompt == "manual forever"
+ # The next turn re-resolves from config (no in-memory snapshot).
+ with p1, p2:
+ assert runner._get_system_prompt_for_channel(None, "c") == "manual forever"
@pytest.mark.asyncio
async def test_set_persists_display_personality_not_system_prompt(self, tmp_path):
@@ -147,7 +148,8 @@ class TestGatewayPersonalityNone:
saved = yaml.safe_load(config_file.read_text())
assert saved["agent"]["system_prompt"] == "manual forever"
assert saved["display"]["personality"] == "helpful"
- assert runner._ephemeral_system_prompt == "You are helpful."
+ with p1, p2:
+ assert runner._get_system_prompt_for_channel(None, "c") == "You are helpful."
assert "helpful" in result.lower()
@pytest.mark.asyncio
diff --git a/tests/cli/test_resume_display.py b/tests/cli/test_resume_display.py
index 3c8755c31b..e1c8fb2e5b 100644
--- a/tests/cli/test_resume_display.py
+++ b/tests/cli/test_resume_display.py
@@ -323,6 +323,27 @@ class TestPreloadResumedSession:
assert "safe resume limit is 20000" in output.getvalue()
mock_db.get_resume_conversations.assert_not_called()
+ def test_tip_only_guard_goes_through_the_shared_resume_guard(self):
+ """The mid-setup path loads only the tip, so it asks the ONE resume
+ guard for a tip-only bound instead of borrowing the export guard."""
+ from hermes_state import SessionResumeTooLargeError
+
+ cli = _make_cli(resume="deep-lineage")
+ cli.session_id = "deep-lineage"
+ mock_db = MagicMock()
+ guard = MagicMock(return_value=666)
+ mock_db.assert_resume_safe = guard
+ cli._session_db = mock_db
+
+ assert cli._resume_history_limit_error(tip_only=True) is None
+ guard.assert_called_once_with("deep-lineage", tip_only=True)
+
+ guard.side_effect = SessionResumeTooLargeError(
+ 20_001, 20_000, scope="in its tip segment"
+ )
+ error = cli._resume_history_limit_error(tip_only=True)
+ assert error and "in its tip segment" in error
+
# ── Tests for _handle_resume_command recap display ───────────────────
diff --git a/tests/conftest.py b/tests/conftest.py
index ffa5e7b928..e4dfb0ed06 100644
--- a/tests/conftest.py
+++ b/tests/conftest.py
@@ -186,7 +186,7 @@ _CREDENTIAL_NAMES = frozenset({
"FIRECRAWL_API_KEY",
"PARALLEL_API_KEY",
"EXA_API_KEY",
- "TAVILY_API_KEY", # removed backend; still blanked for hermeticity
+ "TAVILY_API_KEY",
"WANDB_API_KEY",
"ELEVENLABS_API_KEY",
"HONCHO_API_KEY",
diff --git a/tests/cron/test_cleanup_timeout.py b/tests/cron/test_cleanup_timeout.py
index 6d17502e34..6b967afc2a 100644
--- a/tests/cron/test_cleanup_timeout.py
+++ b/tests/cron/test_cleanup_timeout.py
@@ -70,8 +70,8 @@ def test_run_job_bounds_sessiondb_finalization(tmp_path):
success, _output, final_response, error = run_job(job)
elapsed = time.monotonic() - started
- assert fake_db.entered.wait(timeout=0.5)
- assert elapsed < 0.5
+ assert fake_db.entered.wait(timeout=2.0)
+ assert elapsed < 5.0
assert success is True
assert final_response == "ok"
assert error is None
@@ -88,8 +88,8 @@ def test_agent_teardown_is_bounded():
_teardown_cron_agent(agent, "cleanup-agent-hang", timeout_seconds=0.02)
elapsed = time.monotonic() - started
- assert agent.entered.wait(timeout=0.5)
- assert elapsed < 0.5
+ assert agent.entered.wait(timeout=2.0)
+ assert elapsed < 5.0
finally:
release.set()
diff --git a/tests/cron/test_cron_drift_alert_once.py b/tests/cron/test_cron_drift_alert_once.py
index a6014f2591..c3e57eab29 100644
--- a/tests/cron/test_cron_drift_alert_once.py
+++ b/tests/cron/test_cron_drift_alert_once.py
@@ -48,7 +48,7 @@ def _tick(job, tmp_path, current_provider, deliveries):
"""Run one run_one_job tick with the provider resolution pinned."""
fake_db = MagicMock()
- def fake_deliver(job, content, adapters=None, loop=None):
+ def fake_deliver(job, content, adapters=None, loop=None, **kwargs):
deliveries.append(content)
return None
@@ -129,7 +129,7 @@ class TestDriftAlertOnce:
job = _job(provider_snapshot=None, drift_alerted=True)
deliveries = []
- def fake_deliver(jb, content, adapters=None, loop=None):
+ def fake_deliver(jb, content, adapters=None, loop=None, **kwargs):
deliveries.append(content)
return None
diff --git a/tests/cron/test_cron_failure_deliver.py b/tests/cron/test_cron_failure_deliver.py
new file mode 100644
index 0000000000..1695447ad0
--- /dev/null
+++ b/tests/cron/test_cron_failure_deliver.py
@@ -0,0 +1,462 @@
+"""Per-job ``failure_deliver`` routing (NS-788).
+
+A job's FAILURE notices (run failed, escaped scheduler exception, drift-skip /
+blocked-config alerts) resolve their delivery targets from ``failure_deliver``
+when the job sets it, falling back to ``deliver`` when unset — so existing
+jobs behave byte-identically. ``failure_deliver: local`` is structural silence
+for failures: nothing is sent, but state (last_status, run history, output
+file) is still recorded. Success-path delivery never reads ``failure_deliver``.
+
+The grammar is exactly the ``deliver`` grammar — same normalization, same
+validation — reused, not duplicated.
+"""
+
+import json
+
+import pytest
+
+import cron.scheduler as s
+from cron.scheduler import _resolve_delivery_targets
+
+
+@pytest.fixture
+def cron_env(tmp_path, monkeypatch):
+ """Isolated cron environment with temp HERMES_HOME."""
+ hermes_home = tmp_path / ".hermes"
+ hermes_home.mkdir()
+ (hermes_home / "cron").mkdir()
+ (hermes_home / "cron" / "output").mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(hermes_home))
+
+ import cron.jobs as jobs_mod
+ monkeypatch.setattr(jobs_mod, "HERMES_DIR", hermes_home)
+ monkeypatch.setattr(jobs_mod, "CRON_DIR", hermes_home / "cron")
+ monkeypatch.setattr(jobs_mod, "JOBS_FILE", hermes_home / "cron" / "jobs.json")
+ monkeypatch.setattr(jobs_mod, "OUTPUT_DIR", hermes_home / "cron" / "output")
+
+ return hermes_home
+
+
+@pytest.fixture
+def run_env(monkeypatch, tmp_path):
+ """Drive run_one_job with the REAL delivery path down to a fake sender.
+
+ Bookkeeping primitives are stubbed (recorded), but _deliver_result and
+ _resolve_delivery_targets are the genuine articles — the send that would
+ leave the process is captured at the platform-registry sender seam,
+ exactly where a real slack delivery exits.
+ """
+ home = tmp_path / "hermes-home"
+ home.mkdir()
+ (home / "config.yaml").write_text(
+ "platforms:\n slack:\n enabled: true\n token: xoxb-test\n"
+ )
+ monkeypatch.setenv("HERMES_HOME", str(home))
+
+ send_calls = []
+
+ async def fake_sender(pconfig, chat_id, message, *, thread_id=None,
+ media_files=None, force_document=False, caption=None):
+ send_calls.append({"chat_id": chat_id, "message": message})
+ return {"success": True, "chat_id": chat_id, "message_id": "1.2"}
+
+ import gateway.platform_registry as reg
+ import hermes_cli.plugins as hp
+
+ entry = reg.platform_registry.get("slack")
+ if entry is None:
+ hp.discover_plugins()
+ entry = reg.platform_registry.get("slack")
+ if entry is None:
+ pytest.skip("slack platform entry not registered")
+ monkeypatch.setattr(entry, "standalone_sender_fn", fake_sender)
+ monkeypatch.setattr(hp, "discover_plugins", lambda *a, **k: None)
+
+ state = {"send": send_calls, "marked": [], "saved": [], "finished": []}
+
+ monkeypatch.setattr(s, "create_execution", lambda *_a, **_kw: {"id": "exec-t"})
+ monkeypatch.setattr(s, "claim_dispatch", lambda _job_id: True)
+ monkeypatch.setattr(s, "mark_execution_running", lambda _execution_id: None)
+ monkeypatch.setattr(
+ s, "save_job_output",
+ lambda jid, out: state["saved"].append(jid) or f"/tmp/{jid}.txt",
+ )
+ monkeypatch.setattr(
+ s, "mark_job_run",
+ lambda *a, **kw: state["marked"].append((a, kw)) or True,
+ )
+ monkeypatch.setattr(
+ s, "finish_execution",
+ lambda *a, **kw: state["finished"].append((a, kw)),
+ )
+ # No durable incident store in play: never acked, no id.
+ monkeypatch.setattr(
+ s, "_upsert_incident_for_failure", lambda *_a, **_kw: (False, None)
+ )
+ monkeypatch.setattr(s, "load_config", lambda: {})
+ return state
+
+
+def _failing_run_job(error="provider exploded"):
+ def _fake(job, **_kw):
+ return (False, "raw output", "", error)
+ return _fake
+
+
+def _succeeding_run_job(final="all good, here is the brief"):
+ def _fake(job, **_kw):
+ return (True, "raw output", final, None)
+ return _fake
+
+
+class TestFailureDeliverRouting:
+ def test_failure_without_failure_deliver_goes_to_deliver_targets(
+ self, run_env, monkeypatch
+ ):
+ """(a) Unset failure_deliver = today's behavior: failure summary to
+ the job's deliver targets."""
+ monkeypatch.setattr(s, "run_job", _failing_run_job())
+
+ s.run_one_job({"id": "j1", "name": "scout", "deliver": "slack:D0MAIN"})
+
+ assert [c["chat_id"] for c in run_env["send"]] == ["D0MAIN"]
+ assert "failed" in run_env["send"][0]["message"].lower()
+
+ def test_failure_deliver_local_is_silent_but_state_is_recorded(
+ self, run_env, monkeypatch
+ ):
+ """(b) failure_deliver: local — no delivery leaves the process, but
+ the run is still saved and marked failed."""
+ monkeypatch.setattr(s, "run_job", _failing_run_job())
+
+ s.run_one_job({
+ "id": "j2", "name": "scout",
+ "deliver": "slack:D0MAIN", "failure_deliver": "local",
+ })
+
+ assert run_env["send"] == []
+ # State recording is untouched by the silence.
+ assert run_env["saved"] == ["j2"]
+ assert len(run_env["marked"]) == 1
+ args, _kw = run_env["marked"][0]
+ assert args[0] == "j2" and args[1] is False
+ assert "provider exploded" in args[2]
+
+ def test_failure_deliver_explicit_target_wins_over_deliver(
+ self, run_env, monkeypatch
+ ):
+ """(c) failure_deliver set to a different target: the failure notice
+ goes THERE, and nothing goes to the deliver target."""
+ monkeypatch.setattr(s, "run_job", _failing_run_job())
+
+ s.run_one_job({
+ "id": "j3", "name": "scout",
+ "deliver": "slack:D0MAIN", "failure_deliver": "slack:D0ALERTS",
+ })
+
+ assert [c["chat_id"] for c in run_env["send"]] == ["D0ALERTS"]
+ assert "failed" in run_env["send"][0]["message"].lower()
+
+ def test_success_ignores_failure_deliver(self, run_env, monkeypatch):
+ """(d) Success output still goes to deliver — failure_deliver is
+ never consulted on the success path."""
+ monkeypatch.setattr(s, "run_job", _succeeding_run_job())
+
+ ok = s.run_one_job({
+ "id": "j4", "name": "scout",
+ "deliver": "slack:D0MAIN", "failure_deliver": "slack:D0ALERTS",
+ })
+
+ assert ok is True
+ assert [c["chat_id"] for c in run_env["send"]] == ["D0MAIN"]
+ assert "all good, here is the brief" in run_env["send"][0]["message"]
+
+
+class TestEscapedExceptionPath:
+ """The scheduler-layer exception handler is the second failure-delivery
+ site — it must honor failure_deliver identically."""
+
+ def _raise_run_job(self, monkeypatch):
+ monkeypatch.setattr(
+ s, "run_job",
+ lambda *_a, **_kw: (_ for _ in ()).throw(
+ RuntimeError("cannot import name X")
+ ),
+ )
+
+ def test_escaped_failure_honors_failure_deliver_target(
+ self, run_env, monkeypatch
+ ):
+ self._raise_run_job(monkeypatch)
+
+ ok = s.run_one_job({
+ "id": "j5", "name": "scout",
+ "deliver": "slack:D0MAIN", "failure_deliver": "slack:D0ALERTS",
+ })
+
+ assert ok is False
+ assert [c["chat_id"] for c in run_env["send"]] == ["D0ALERTS"]
+
+ def test_escaped_failure_with_failure_deliver_local_is_silent(
+ self, run_env, monkeypatch
+ ):
+ self._raise_run_job(monkeypatch)
+
+ ok = s.run_one_job({
+ "id": "j6", "name": "scout",
+ "deliver": "slack:D0MAIN", "failure_deliver": "local",
+ })
+
+ assert ok is False
+ assert run_env["send"] == []
+ # Failure is still recorded.
+ assert len(run_env["marked"]) == 1
+ args, _kw = run_env["marked"][0]
+ assert args[1] is False and "cannot import name X" in args[2]
+
+
+class TestResolutionGrammar:
+ """(e) failure_deliver shares deliver's exact value grammar — the same
+ normalization/expansion path, not a parallel one."""
+
+ def test_for_failure_resolves_failure_deliver_value(self):
+ job = {"deliver": "local", "failure_deliver": "slack:D0ALERTS"}
+ targets = _resolve_delivery_targets(job, for_failure=True)
+ assert [(t["platform"], t["chat_id"]) for t in targets] == [
+ ("slack", "D0ALERTS")
+ ]
+
+ def test_for_failure_falls_back_to_deliver_when_unset(self):
+ job = {"deliver": "slack:D0MAIN"}
+ targets = _resolve_delivery_targets(job, for_failure=True)
+ assert [(t["platform"], t["chat_id"]) for t in targets] == [
+ ("slack", "D0MAIN")
+ ]
+
+ def test_success_resolution_never_reads_failure_deliver(self):
+ job = {"deliver": "slack:D0MAIN", "failure_deliver": "slack:D0ALERTS"}
+ targets = _resolve_delivery_targets(job)
+ assert [(t["platform"], t["chat_id"]) for t in targets] == [
+ ("slack", "D0MAIN")
+ ]
+
+ def test_local_yields_zero_failure_targets(self):
+ job = {"deliver": "slack:D0MAIN", "failure_deliver": "local"}
+ assert _resolve_delivery_targets(job, for_failure=True) == []
+
+ def test_comma_list_and_thread_grammar(self):
+ """The comma-combine + platform:chat:thread forms deliver's grammar
+ supports work identically for failure_deliver."""
+ job = {
+ "deliver": "local",
+ "failure_deliver": "slack:D0ALERTS,telegram:-1001:17",
+ }
+ targets = _resolve_delivery_targets(job, for_failure=True)
+ assert [(t["platform"], t["chat_id"], t.get("thread_id")) for t in targets] == [
+ ("slack", "D0ALERTS", None),
+ ("telegram", "-1001", "17"),
+ ]
+
+ def test_legacy_list_value_is_flattened_like_deliver(self):
+ """Same list/tuple tolerance _normalize_deliver_value grants deliver."""
+ job = {"deliver": "local", "failure_deliver": ["slack:D0ALERTS"]}
+ targets = _resolve_delivery_targets(job, for_failure=True)
+ assert [(t["platform"], t["chat_id"]) for t in targets] == [
+ ("slack", "D0ALERTS")
+ ]
+
+
+class TestToolSurface:
+ """cronjob(action=create/update) accepts failure_deliver with deliver's
+ validation — reusing the same normalize/validate helpers."""
+
+ def test_create_stores_failure_deliver(self, cron_env):
+ from tools.cronjob_tools import cronjob
+ from cron.jobs import get_job
+
+ result = json.loads(cronjob(
+ action="create",
+ prompt="scan",
+ schedule="every 1h",
+ deliver="slack:D0MAIN",
+ failure_deliver="local",
+ ))
+ assert result["success"] is True
+ assert get_job(result["job_id"])["failure_deliver"] == "local"
+
+ def test_create_without_failure_deliver_does_not_persist_the_key(self, cron_env):
+ """Existing-job byte-identity: the field only exists when set."""
+ from tools.cronjob_tools import cronjob
+ from cron.jobs import get_job
+
+ result = json.loads(cronjob(
+ action="create", prompt="scan", schedule="every 1h",
+ ))
+ assert result["success"] is True
+ assert "failure_deliver" not in get_job(result["job_id"])
+
+ def test_create_flattens_list_value_like_deliver(self, cron_env):
+ from tools.cronjob_tools import cronjob
+ from cron.jobs import get_job
+
+ result = json.loads(cronjob(
+ action="create",
+ prompt="scan",
+ schedule="every 1h",
+ failure_deliver=["slack", "telegram"],
+ ))
+ assert result["success"] is True
+ assert get_job(result["job_id"])["failure_deliver"] == "slack,telegram"
+
+ def test_create_rejects_bad_bot_chat_profile_same_as_deliver(self, cron_env):
+ from tools.cronjob_tools import cronjob
+
+ via_failure = json.loads(cronjob(
+ action="create", prompt="scan", schedule="every 1h",
+ failure_deliver="bot-chat:no-such-profile-xyz",
+ ))
+ via_deliver = json.loads(cronjob(
+ action="create", prompt="scan", schedule="every 1h",
+ deliver="bot-chat:no-such-profile-xyz",
+ ))
+ assert via_failure["success"] is False
+ assert via_deliver["success"] is False
+ # Same validator, same message.
+ assert via_failure["error"] == via_deliver["error"]
+
+ def test_update_sets_and_clears_failure_deliver(self, cron_env):
+ from cron.jobs import create_job, get_job
+ from tools.cronjob_tools import cronjob
+
+ job = create_job(prompt="scan", schedule="every 1h")
+ result = json.loads(cronjob(
+ action="update", job_id=job["id"], failure_deliver="slack:D0ALERTS",
+ ))
+ assert result["success"] is True
+ assert get_job(job["id"])["failure_deliver"] == "slack:D0ALERTS"
+
+ # '' clears — job falls back to deliver on failures again.
+ result = json.loads(cronjob(
+ action="update", job_id=job["id"], failure_deliver="",
+ ))
+ assert result["success"] is True
+ assert not get_job(job["id"]).get("failure_deliver")
+
+
+class TestOutcomeBookkeeping:
+ """Review finding B1 (NS-788): delivery bookkeeping — outcome
+ classification, unresolved-origin, incident 'alerted' marking — must
+ read the SAME lane the notice was actually routed through, or the
+ execution history and incident store record lies (silenced failures
+ logged 'delivered'; delivered failures logged 'not_configured')."""
+
+ @staticmethod
+ def _outcome(state):
+ assert state["finished"], "finish_execution never called"
+ _a, kw = state["finished"][-1]
+ return kw.get("delivery_outcome")
+
+ def test_fd_local_failure_records_suppressed_not_delivered(
+ self, run_env, monkeypatch
+ ):
+ alerted = []
+ monkeypatch.setattr(s, "_mark_incident_alerted", alerted.append)
+ monkeypatch.setattr(s, "run_job", _failing_run_job())
+
+ s.run_one_job({
+ "id": "b1a", "name": "scout",
+ "deliver": "slack:D0MAIN", "failure_deliver": "local",
+ })
+
+ assert run_env["send"] == []
+ assert self._outcome(run_env) == "suppressed"
+ assert alerted == [], "silenced failure must NOT mark incident alerted"
+
+ def test_fd_explicit_target_failure_records_delivered(
+ self, run_env, monkeypatch
+ ):
+ """deliver=origin (unresolvable) + failure_deliver=explicit target:
+ the notice IS delivered — outcome must say so, not 'not_configured'."""
+ alerted = []
+ monkeypatch.setattr(s, "_mark_incident_alerted", alerted.append)
+ monkeypatch.setattr(
+ s, "_upsert_incident_for_failure", lambda *_a, **_kw: (False, "inc-b1")
+ )
+ monkeypatch.setattr(s, "run_job", _failing_run_job())
+
+ s.run_one_job({
+ "id": "b1b", "name": "scout",
+ "deliver": "origin", "failure_deliver": "slack:D0OPS",
+ })
+
+ assert [c["chat_id"] for c in run_env["send"]] == ["D0OPS"]
+ assert self._outcome(run_env) == "delivered"
+ assert alerted == ["inc-b1"], "delivered failure ping must mark incident alerted"
+
+ def test_success_outcome_still_reads_deliver_lane(self, run_env, monkeypatch):
+ """Success bookkeeping is untouched: fd set, success delivers to
+ deliver and records 'delivered'."""
+ monkeypatch.setattr(s, "run_job", _succeeding_run_job())
+
+ s.run_one_job({
+ "id": "b1c", "name": "scout",
+ "deliver": "slack:D0MAIN", "failure_deliver": "local",
+ })
+
+ assert [c["chat_id"] for c in run_env["send"]] == ["D0MAIN"]
+ assert self._outcome(run_env) == "delivered"
+
+
+class TestPreflightAndDashboardLanes:
+ """Follow-up (salvage): the failure lane is validated everywhere the
+ deliver lane is — preflight config checks and the dashboard update
+ normalizer — so a typo'd failure target is caught before a failure
+ needs it."""
+
+ def test_preflight_blocks_unknown_failure_platform(self, monkeypatch):
+ """A bogus failure_deliver platform blocks at preflight, exactly
+ like a bogus deliver platform would."""
+ monkeypatch.setattr(s, "_is_known_delivery_platform", lambda _p: False)
+ err = s._preflight_check_delivery({
+ "id": "p1", "deliver": "local",
+ "failure_deliver": "nonexistent-platform:C1",
+ })
+ assert err is not None and "not a known" in err
+
+ def test_preflight_failure_deliver_local_adds_no_platforms(self):
+ """failure_deliver: local adds nothing to check — a deliver=local
+ job with suppressed failures stays zero-cost at preflight."""
+ assert s._preflight_check_delivery({
+ "id": "p2", "deliver": "local", "failure_deliver": "local",
+ }) is None
+
+ def test_preflight_duplicate_lane_not_checked_twice(self, monkeypatch):
+ """failure_deliver equal to deliver must not double-check (or
+ double-report) the same platform."""
+ seen = []
+
+ def _known(p):
+ seen.append(p)
+ return False
+
+ monkeypatch.setattr(s, "_is_known_delivery_platform", _known)
+ s._preflight_check_delivery({
+ "id": "p3", "deliver": "ghost:C1", "failure_deliver": "ghost:C1",
+ })
+ assert seen == ["ghost"]
+
+ def test_dashboard_update_normalizes_failure_deliver(self, tmp_path):
+ """The dashboard update lane normalizes failure_deliver like
+ deliver: text stripped, empty clears (None) instead of
+ coalescing to a target."""
+ from hermes_cli.web_server import _normalize_dashboard_cron_updates
+
+ out = _normalize_dashboard_cron_updates(
+ {"failure_deliver": " slack:D0ALERTS "}, tmp_path
+ )
+ assert out["failure_deliver"] == "slack:D0ALERTS"
+
+ cleared = _normalize_dashboard_cron_updates(
+ {"failure_deliver": ""}, tmp_path
+ )
+ assert cleared["failure_deliver"] is None
diff --git a/tests/cron/test_cron_incidents.py b/tests/cron/test_cron_incidents.py
index c495506336..22ec1c1d95 100644
--- a/tests/cron/test_cron_incidents.py
+++ b/tests/cron/test_cron_incidents.py
@@ -46,7 +46,7 @@ def _tick_failing(job, tmp_path, deliveries, error="boom unrelated"):
harness so the incident gating is exercised through the real scheduler."""
fake_db = MagicMock()
- def fake_deliver(jb, content, adapters=None, loop=None):
+ def fake_deliver(jb, content, adapters=None, loop=None, **kwargs):
deliveries.append(content)
return None
diff --git a/tests/cron/test_cron_live_delivery_confirmation.py b/tests/cron/test_cron_live_delivery_confirmation.py
new file mode 100644
index 0000000000..b2822edc24
--- /dev/null
+++ b/tests/cron/test_cron_live_delivery_confirmation.py
@@ -0,0 +1,404 @@
+"""Live-adapter delivery confirmation for cron (#77763).
+
+A ``no_agent`` job fired, the scheduler logged
+``delivered to telegram: via live adapter``, and the user received
+nothing — no message row, no delivery obligation. The log line was not
+evidence of a send:
+
+* the silence-narration filter returns ``{"success": True, "delivered": False}``
+ (a successful *drop*), and the normalization block read only ``success``;
+* an empty payload skipped the send entirely and still fell into the
+ "delivered" branch;
+* the log line named the chat but not the lane, so a wrong-thread delivery and
+ a phantom one look identical after the fact.
+
+These tests pin the confirmation contract: positive evidence, honest logging,
+and fail-closed on nothing-to-send.
+"""
+
+import asyncio
+import logging
+from concurrent.futures import Future
+from unittest.mock import MagicMock, patch
+
+import pytest
+
+from cron import scheduler as sched
+from cron.scheduler import _confirm_adapter_delivery, _deliver_result
+from gateway.config import Platform, PlatformConfig
+
+
+# ---------------------------------------------------------------------------
+# _confirm_adapter_delivery: the contract in isolation
+# ---------------------------------------------------------------------------
+
+class _SendResult:
+ """Minimal stand-in for an adapter SendResult."""
+
+ def __init__(self, success=True, message_id=None, raw_response=None, **extra):
+ self.success = success
+ self.message_id = message_id
+ self.raw_response = raw_response
+ for key, value in extra.items():
+ setattr(self, key, value)
+
+
+class TestConfirmAdapterDelivery:
+ def test_none_is_not_delivered(self):
+ assert _confirm_adapter_delivery(None, "j1") is False
+
+ def test_missing_success_is_not_delivered(self):
+ assert _confirm_adapter_delivery(object(), "j1") is False
+ assert _confirm_adapter_delivery({"message_id": 7}, "j1") is False
+
+ def test_explicit_failure_is_not_delivered(self):
+ assert _confirm_adapter_delivery(_SendResult(success=False), "j1") is False
+ assert _confirm_adapter_delivery({"success": False}, "j1") is False
+
+ def test_filtered_dict_is_not_delivered(self):
+ """The exact silence-filter shape: a successful DROP is not a delivery."""
+ filtered = {"success": True, "filtered": "silence_narration", "delivered": False}
+ assert _confirm_adapter_delivery(filtered, "j1") is False
+
+ def test_delivered_false_on_an_object_is_not_delivered(self):
+ result = _SendResult(success=True, message_id=42, delivered=False)
+ assert _confirm_adapter_delivery(result, "j1") is False
+
+ def test_positive_evidence_is_delivered_without_warning(self, caplog):
+ with caplog.at_level(logging.WARNING, logger="cron.scheduler"):
+ assert _confirm_adapter_delivery(_SendResult(message_id=1234), "j1") is True
+ assert "UNVERIFIED" not in caplog.text
+
+ def test_raw_response_alone_counts_as_evidence(self, caplog):
+ with caplog.at_level(logging.WARNING, logger="cron.scheduler"):
+ result = _SendResult(raw_response={"ok": True})
+ assert _confirm_adapter_delivery(result, "j1") is True
+ assert "UNVERIFIED" not in caplog.text
+
+ def test_evidence_free_success_is_accepted_but_warned(self, caplog):
+ """Not proof of failure either — accept it, but say so in the log."""
+ with caplog.at_level(logging.WARNING, logger="cron.scheduler"):
+ assert _confirm_adapter_delivery(_SendResult(), "92e639af907f") is True
+ assert "UNVERIFIED" in caplog.text
+ assert "92e639af907f" in caplog.text
+
+ def test_evidence_free_success_dict_is_accepted_but_warned(self, caplog):
+ with caplog.at_level(logging.WARNING, logger="cron.scheduler"):
+ assert _confirm_adapter_delivery({"success": True}, "j1") is True
+ assert "UNVERIFIED" in caplog.text
+
+
+# ---------------------------------------------------------------------------
+# _deliver_result: the live lane end to end
+# ---------------------------------------------------------------------------
+
+CHAT_ID = "-1001234567890"
+
+
+def _job(thread_id=None):
+ origin = {"platform": "telegram", "chat_id": CHAT_ID}
+ if thread_id is not None:
+ origin["thread_id"] = thread_id
+ return {
+ "id": "92e639af907f",
+ "name": "Ghost Delivery",
+ "deliver": "origin",
+ "origin": origin,
+ }
+
+
+def _gateway_config(relay=False):
+ config = MagicMock()
+ platforms = {Platform.TELEGRAM: PlatformConfig(enabled=True)}
+ if relay:
+ platforms[Platform.RELAY] = PlatformConfig(enabled=True)
+ config.platforms = platforms
+ config.get_home_channel = lambda p: None
+ return config
+
+
+def _adapters(relay=False):
+ adapter = MagicMock()
+ if relay:
+ adapter.fronts_platform = lambda p: p == Platform.TELEGRAM
+ return {Platform.RELAY: adapter}
+ return {Platform.TELEGRAM: adapter}
+
+
+RECORDED_VERIFICATION = []
+
+
+def _record_verification(job, unverified_targets):
+ RECORDED_VERIFICATION.append((job["id"], list(unverified_targets)))
+
+
+def _run(job, content, send_result, relay=False, standalone_result=None, cron_cfg=None):
+ """Drive ``_deliver_result`` over the live lane with a stubbed router.
+
+ Returns ``(error, router_calls, standalone_calls)``. ``cron_cfg`` extends
+ the ``cron:`` section handed to the scheduler (default: unwrapped output).
+ """
+ loop = MagicMock()
+ loop.is_running.return_value = True
+
+ def fake_run_coro(coro, _loop):
+ future = Future()
+ try:
+ future.set_result(asyncio.run(coro))
+ except BaseException as e: # noqa: BLE001
+ future.set_exception(e)
+ return future
+
+ router_calls = []
+ standalone_calls = []
+ RECORDED_VERIFICATION.clear()
+
+ router = MagicMock()
+
+ async def _deliver_to_platform(target, text, metadata):
+ router_calls.append({"target": target, "text": text, "metadata": metadata})
+ return send_result
+
+ router._deliver_to_platform = _deliver_to_platform
+
+ async def _fake_send_to_platform(platform, pconfig, chat_id, text, **kwargs):
+ standalone_calls.append({"chat_id": chat_id, "text": text, "kwargs": kwargs})
+ return standalone_result if standalone_result is not None else {}
+
+ with patch("gateway.config.load_gateway_config", return_value=_gateway_config(relay)), \
+ patch("cron.scheduler.load_config",
+ return_value={"cron": {"wrap_response": False, **(cron_cfg or {})}}), \
+ patch("cron.scheduler._record_delivery_verification", side_effect=_record_verification), \
+ patch("gateway.delivery.DeliveryRouter", return_value=router), \
+ patch("tools.send_message_tool._send_to_platform", _fake_send_to_platform), \
+ patch("asyncio.run_coroutine_threadsafe", side_effect=fake_run_coro):
+ error = _deliver_result(job, content, adapters=_adapters(relay), loop=loop)
+ return error, router_calls, standalone_calls
+
+
+class TestFilteredResultIsNotDelivered:
+ FILTERED = {"success": True, "filtered": "silence_narration", "delivered": False}
+
+ def test_filtered_dict_does_not_log_a_live_delivery(self, caplog):
+ with caplog.at_level(logging.INFO, logger="cron.scheduler"):
+ _, router_calls, standalone_calls = _run(_job(), "...", self.FILTERED)
+
+ assert len(router_calls) == 1 # the live send was attempted
+ assert "via live adapter" not in caplog.text # but never claimed as delivered
+ assert len(standalone_calls) == 1 # fell back instead of lying
+
+ def test_filtered_dict_fails_closed_on_the_relay_lane(self):
+ """Relay owns the destination, so there is no fallback — report it."""
+ error, _, standalone_calls = _run(_job(), "...", self.FILTERED, relay=True)
+
+ assert error is not None
+ assert "unconfirmed result" in error
+ assert "silence_narration" in error # names the filter, not "unknown"
+ assert standalone_calls == []
+
+ def test_confirmed_send_result_still_delivers(self, caplog):
+ with caplog.at_level(logging.INFO, logger="cron.scheduler"):
+ error, router_calls, standalone_calls = _run(
+ _job(), "Nightly report.", _SendResult(message_id=1234),
+ )
+
+ assert error is None
+ assert len(router_calls) == 1
+ assert standalone_calls == []
+ assert "via live adapter" in caplog.text
+
+
+class TestEmptyPayloadFailsClosed:
+ def test_empty_payload_never_reaches_the_adapter(self, caplog):
+ with caplog.at_level(logging.INFO, logger="cron.scheduler"):
+ _, router_calls, _ = _run(_job(), " ", _SendResult(message_id=1))
+
+ assert router_calls == [] # nothing was sent
+ assert "via live adapter" not in caplog.text # and nothing was claimed
+ assert "empty text and no media" in caplog.text
+
+ def test_empty_payload_never_reaches_the_standalone_sender(self, caplog):
+ """The native fallback must not re-open the hole the live lane closed.
+
+ Telegram's adapter returns ``SendResult(success=True)`` for empty
+ content without an API call, so an unguarded fallback would log a
+ standalone "delivered" for the same phantom payload (#77763).
+ """
+ with caplog.at_level(logging.INFO, logger="cron.scheduler"):
+ error, router_calls, standalone_calls = _run(
+ _job(), " ", _SendResult(message_id=1),
+ )
+
+ assert router_calls == []
+ assert standalone_calls == [] # _send_to_platform never called
+ assert error is not None
+ assert "standalone send skipped (empty text and no media)" in error
+ assert "delivered to" not in caplog.text
+
+ def test_empty_payload_is_reported_on_the_relay_lane(self):
+ error, router_calls, _ = _run(_job(), "", _SendResult(message_id=1), relay=True)
+
+ assert router_calls == []
+ assert error is not None
+ assert "live adapter send skipped (empty text and no media)" in error
+
+
+class TestDeliveredLogNamesTheLane:
+ def test_log_includes_thread_and_message_id(self, caplog):
+ with caplog.at_level(logging.INFO, logger="cron.scheduler"):
+ error, _, _ = _run(
+ _job(thread_id="99"), "Nightly report.", _SendResult(message_id=1234),
+ )
+
+ assert error is None
+ assert "via live adapter thread=99 message_id=1234" in caplog.text
+
+ def test_log_uses_a_dash_when_the_lane_is_unknown(self, caplog):
+ """No thread and an evidence-free result must still be attributable."""
+ with caplog.at_level(logging.INFO, logger="cron.scheduler"):
+ error, _, _ = _run(_job(), "Nightly report.", _SendResult())
+
+ assert error is None
+ assert "via live adapter thread=- message_id=-" in caplog.text
+ assert "UNVERIFIED" in caplog.text
+
+
+class TestLiveDeliveryIsAFinalNotification:
+ """Cron output is a final user-visible delivery, not a progress send.
+
+ Telegram's adapter defaults to ``_notifications_mode = "important"`` and
+ sends with ``disable_notification=True`` unless ``metadata["notify"]`` is
+ set — so a cron brief without the marker lands silently, which users
+ report as "never delivered" (#77763 thread, #58258 typing bubble). The
+ marker must ride both the text route and the media route, in every
+ Telegram routing mode.
+ """
+
+ def test_text_route_metadata_carries_notify(self):
+ _, router_calls, _ = _run(_job(), "Nightly report.", _SendResult(message_id=1))
+ assert len(router_calls) == 1
+ metadata = router_calls[0]["metadata"]
+ assert metadata["job_id"] == "92e639af907f"
+ assert metadata["notify"] is True
+
+ def test_forum_topic_route_keeps_thread_and_notify(self):
+ _, router_calls, _ = _run(
+ _job(thread_id="99"), "Nightly report.", _SendResult(message_id=1),
+ )
+ metadata = router_calls[0]["metadata"]
+ assert metadata["thread_id"] == "99"
+ assert metadata["notify"] is True
+
+ def test_media_route_metadata_carries_notify(self, tmp_path):
+ media = tmp_path / "report.png"
+ media.write_bytes(b"\x89PNG\r\n\x1a\n")
+ sent = []
+
+ def fake_send_media(adapter, chat_id, media_files, metadata, loop, job, platform=None):
+ sent.append({"media": list(media_files), "metadata": metadata})
+ return []
+
+ with patch("cron.scheduler._send_media_via_adapter", side_effect=fake_send_media), \
+ patch("gateway.platforms.base.BasePlatformAdapter.filter_media_delivery_paths",
+ side_effect=lambda files: files):
+ error, router_calls, _ = _run(
+ _job(), f"Nightly report.\nMEDIA:{media}", _SendResult(message_id=1),
+ )
+
+ assert error is None
+ assert len(router_calls) == 1
+ assert len(sent) == 1
+ assert sent[0]["metadata"]["notify"] is True
+
+
+class TestNotifyIsConfigurable:
+ """``cron.delivery.notify`` (config.yaml) gates the notify marker.
+
+ The current behaviour (push notification) stays the default; only an
+ explicit ``false`` restores silent deliveries. The knob rides both the
+ text route and the media route so the two never disagree.
+ """
+
+ def test_default_is_notify(self):
+ _, router_calls, _ = _run(_job(), "Nightly report.", _SendResult(message_id=1))
+ assert router_calls[0]["metadata"]["notify"] is True
+
+ def test_explicit_false_disables_notify_on_text_route(self):
+ _, router_calls, _ = _run(
+ _job(thread_id="99"), "Nightly report.", _SendResult(message_id=1),
+ cron_cfg={"delivery": {"notify": False}},
+ )
+ metadata = router_calls[0]["metadata"]
+ assert metadata["notify"] is False
+ assert metadata["thread_id"] == "99" # routing untouched
+
+ def test_explicit_false_disables_notify_on_media_route(self, tmp_path):
+ media = tmp_path / "report.png"
+ media.write_bytes(b"\x89PNG\r\n\x1a\n")
+ sent = []
+
+ def fake_send_media(adapter, chat_id, media_files, metadata, loop, job, platform=None):
+ sent.append(metadata)
+ return []
+
+ with patch("cron.scheduler._send_media_via_adapter", side_effect=fake_send_media), \
+ patch("gateway.platforms.base.BasePlatformAdapter.filter_media_delivery_paths",
+ side_effect=lambda files: files):
+ _run(
+ _job(), f"Nightly report.\nMEDIA:{media}", _SendResult(message_id=1),
+ cron_cfg={"delivery": {"notify": False}},
+ )
+ assert sent[0]["notify"] is False
+
+ @pytest.mark.parametrize("cron_cfg", [
+ {"delivery": None}, # `delivery:` with no body parses to null
+ {"delivery": "yes"}, # malformed scalar
+ {"delivery": {"notify": None}}, # `notify:` with no value
+ ])
+ def test_malformed_section_keeps_the_default(self, cron_cfg):
+ _, router_calls, _ = _run(_job(), "Nightly report.", _SendResult(message_id=1), cron_cfg=cron_cfg)
+ assert router_calls[0]["metadata"]["notify"] is True
+
+ def test_default_config_ships_notify_true(self):
+ from hermes_cli.config_defaults import DEFAULT_CONFIG
+
+ assert DEFAULT_CONFIG["cron"]["delivery"]["notify"] is True
+
+
+class TestUnverifiedDeliveryIsRecordedOnTheJob:
+ """An evidence-free ack is accepted, but the state must reach the job
+ record (and from there ``hermes cron list`` / ``cron doctor``), not only a
+ WARNING log line."""
+
+ def test_evidence_free_ack_records_the_target(self):
+ error, _, _ = _run(_job(), "Nightly report.", _SendResult())
+ assert error is None
+ assert RECORDED_VERIFICATION == [("92e639af907f", [f"telegram:{CHAT_ID}"])]
+
+ def test_positive_evidence_clears_the_marker(self):
+ error, _, _ = _run(_job(), "Nightly report.", _SendResult(message_id=1234))
+ assert error is None
+ assert RECORDED_VERIFICATION == [("92e639af907f", [])]
+
+ def test_recorder_skips_the_write_when_nothing_changed(self):
+ with patch("cron.jobs.update_job") as update_job:
+ sched._record_delivery_verification({"id": "j1", "last_delivery_unverified": None}, [])
+ update_job.assert_not_called()
+ sched._record_delivery_verification({"id": "j1", "last_delivery_unverified": None}, ["slack:C1"])
+ update_job.assert_called_once_with("j1", {"last_delivery_unverified": ["slack:C1"]})
+
+ def test_recorder_clears_a_stale_marker(self):
+ with patch("cron.jobs.update_job") as update_job:
+ sched._record_delivery_verification({"id": "j1", "last_delivery_unverified": ["slack:C1"]}, [])
+ update_job.assert_called_once_with("j1", {"last_delivery_unverified": None})
+
+ def test_tool_listing_exposes_the_field(self):
+ from tools.cronjob_tools import _format_job
+
+ assert _format_job({"id": "j1", "name": "n", "prompt": "p",
+ "last_delivery_unverified": ["slack:C1"]})["last_delivery_unverified"] == ["slack:C1"]
+
+
+def test_scheduler_module_exposes_the_confirmation_helper():
+ """Guard the import surface the delivery block depends on."""
+ assert callable(sched._confirm_adapter_delivery)
diff --git a/tests/cron/test_cron_multiplex_desktop_ticker_scope.py b/tests/cron/test_cron_multiplex_desktop_ticker_scope.py
new file mode 100644
index 0000000000..a01aa0ae3c
--- /dev/null
+++ b/tests/cron/test_cron_multiplex_desktop_ticker_scope.py
@@ -0,0 +1,139 @@
+"""Regression tests for #100489 — desktop multiplex ticker must not deliver a
+secondary profile's cron output through the default profile's identity.
+
+Two halves:
+
+1. ``_deliver_result``'s standalone fallback pool (taken when the caller has a
+ RUNNING event loop — the desktop dashboard shape) spawns a fresh thread that
+ did not inherit the profile ContextVars; it must run inside a copy of the
+ active context so the sender reads THIS profile's home + secrets.
+2. The desktop ticker must stand down, per tick, for a profile whose OWN
+ gateway is running — that gateway ticks it with live adapters, and racing it
+ on the tick lock lets the adapter-less desktop ticker deliver standalone.
+"""
+import asyncio
+import threading
+from unittest.mock import patch
+
+
+
+def test_standalone_fallback_pool_keeps_profile_scope(tmp_path, monkeypatch):
+ from agent.secret_scope import (
+ get_secret,
+ set_multiplex_active,
+ set_secret_scope,
+ )
+ from hermes_constants import get_hermes_home, set_hermes_home_override
+ import cron.scheduler as sched
+ import tools.send_message_tool as smt
+
+ default_home = tmp_path / "default"
+ sec_home = tmp_path / "profiles" / "ops"
+ for home in (default_home, sec_home):
+ (home / "cron").mkdir(parents=True)
+ (home / "config.yaml").write_text("platforms:\n telegram:\n enabled: true\n")
+ monkeypatch.setenv("HERMES_HOME", str(default_home))
+ monkeypatch.setenv("TELEGRAM_BOT_TOKEN", "DEFAULT-TOKEN")
+ set_multiplex_active(True)
+
+ seen = {}
+
+ async def fake_send(platform, pconfig, chat_id, message, **kwargs):
+ seen["home"] = str(get_hermes_home())
+ seen["token"] = get_secret("TELEGRAM_BOT_TOKEN", None)
+ return {"success": True, "message_id": "1"}
+
+ job = {"id": "j1", "name": "probe", "deliver": "telegram:12345", "schedule": {"kind": "cron"}}
+
+ async def _inside_running_loop():
+ # Emulate the multiplex ticker's per-profile scope on the caller.
+ set_hermes_home_override(str(sec_home))
+ set_secret_scope({"TELEGRAM_BOT_TOKEN": "OPS-TOKEN"})
+ return sched._deliver_result(job, "hello", adapters={}, loop=None)
+
+ try:
+ with patch.object(smt, "_send_to_platform", fake_send):
+ err = asyncio.run(_inside_running_loop())
+ finally:
+ set_multiplex_active(False)
+
+ assert err is None, err
+ assert seen["home"] == str(sec_home.resolve())
+ assert seen["token"] == "OPS-TOKEN"
+
+
+def test_multiplex_ticker_profile_gate_skips_rejected_profile(tmp_path):
+ from cron.scheduler_provider import InProcessCronScheduler
+ from hermes_constants import get_hermes_home
+
+ own_gateway = tmp_path / "own-gateway"
+ orphan = tmp_path / "orphan"
+ for home in (own_gateway, orphan):
+ (home / "cron").mkdir(parents=True)
+
+ stop = threading.Event()
+ ticked: list[str] = []
+
+ def _tick(*args, **kwargs):
+ ticked.append(str(get_hermes_home()))
+ if len(ticked) >= 3:
+ stop.set()
+ return 0
+
+ provider = InProcessCronScheduler()
+ with patch("cron.scheduler.tick", side_effect=_tick):
+ thread = threading.Thread(
+ target=provider.start,
+ args=(stop,),
+ kwargs={
+ "interval": 0,
+ "profile_homes": [("own-gateway", own_gateway), ("orphan", orphan)],
+ "profile_gate": lambda name, home: name != "own-gateway",
+ },
+ daemon=True,
+ )
+ thread.start()
+ thread.join(timeout=5)
+ stop.set()
+ thread.join(timeout=5)
+
+ assert not thread.is_alive()
+ assert set(ticked) == {str(orphan)}
+ # The gated profile gets no tick-loop success marker either: its own
+ # gateway owns that status surface.
+ assert not (own_gateway / "cron" / "ticker_last_success").exists()
+ assert (orphan / "cron" / "ticker_last_success").exists()
+
+
+def test_desktop_ticker_gates_on_profile_gateway_running(tmp_path, monkeypatch):
+ """The desktop ticker wires the gate to ``_check_gateway_running``."""
+ from hermes_cli import web_server
+
+ homes = [("default", tmp_path / "default"), ("ops", tmp_path / "ops")]
+ monkeypatch.setattr(
+ "hermes_cli.profiles.profiles_to_serve", lambda multiplex=False: list(homes)
+ )
+ monkeypatch.setattr(
+ "hermes_cli.profiles._check_gateway_running", lambda home: home.name == "ops"
+ )
+ captured = {}
+
+ class _Provider:
+ name = "fake"
+
+ def start(self, stop_event, **kwargs):
+ captured.update(kwargs)
+
+ from cron import scheduler_provider as sp
+
+ monkeypatch.setattr(web_server, "resolve_cron_scheduler", lambda: _Provider(), raising=False)
+ monkeypatch.setattr(sp, "resolve_cron_scheduler", lambda: _Provider())
+ monkeypatch.setattr(sp, "InProcessCronScheduler", _Provider)
+ monkeypatch.setattr("hermes_logging.enable_profile_log_routing", lambda homes: None)
+
+ web_server._start_desktop_cron_ticker(threading.Event(), interval=0)
+
+ gate = captured.get("profile_gate")
+ assert gate is not None, "desktop ticker did not install a profile gate"
+ assert gate("default", tmp_path / "default") is True
+ assert gate("ops", tmp_path / "ops") is False
diff --git a/tests/cron/test_cron_multiplex_shared_route_delivery.py b/tests/cron/test_cron_multiplex_shared_route_delivery.py
new file mode 100644
index 0000000000..92aa6b4f7b
--- /dev/null
+++ b/tests/cron/test_cron_multiplex_shared_route_delivery.py
@@ -0,0 +1,112 @@
+"""Regression tests for #101113 — a credentialless satellite profile under
+``gateway.profile_routes`` delivers cron output through the PRIMARY adapter
+for exactly the targets the primary routes to it, and fails closed otherwise.
+
+The multiplex ticker hands such a profile a ``SharedRouteAdapters`` view over
+the primary adapter map; ``_deliver_result`` resolves a transport from it per
+target using the same ``ProfileRoute.matches`` predicate as inbound routing.
+"""
+import asyncio
+from concurrent.futures import Future
+from unittest.mock import MagicMock, patch
+
+import yaml
+
+from cron.scheduler import (
+ SharedRouteAdapters,
+ _deliver_result,
+ _primary_profile_routes_for_current_home,
+)
+from gateway.config import Platform, PlatformConfig
+from hermes_constants import reset_hermes_home_override, set_hermes_home_override
+
+PRIMARY_YAML = {
+ "gateway": {
+ "multiplex_profiles": True,
+ "profile_routes": [
+ {"name": "fit", "platform": "discord", "chat_id": "1543065293755256852", "profile": "fitness"},
+ {"name": "off", "platform": "discord", "chat_id": "999", "profile": "fitness", "enabled": False},
+ {"name": "other", "platform": "discord", "chat_id": "777", "profile": "other"},
+ ],
+ }
+}
+
+
+def _job(chat_id: str) -> dict:
+ return {"id": "a7ae1520356c", "name": "brief", "deliver": f"discord:{chat_id}"}
+
+
+def _run(job, adapters):
+ """Drive ``_deliver_result`` with a live loop and a real DeliveryRouter."""
+ loop = MagicMock()
+ loop.is_running.return_value = True
+
+ def fake_run_coro(coro, _loop):
+ future = Future()
+ future.set_result(asyncio.run(coro))
+ return future
+
+ standalone = []
+
+ async def _fake_send_to_platform(platform, pconfig, chat_id, text, **kwargs):
+ standalone.append(chat_id)
+ return {"success": False, "error": "DISCORD_BOT_TOKEN is not set"}
+
+ config = MagicMock()
+ config.platforms = {Platform.DISCORD: PlatformConfig(enabled=True)}
+ config.get_home_channel = lambda p: None
+ with patch("gateway.config.load_gateway_config", return_value=config), \
+ patch("cron.scheduler.load_config", return_value={"cron": {"wrap_response": False}}), \
+ patch("tools.send_message_tool._send_to_platform", _fake_send_to_platform), \
+ patch("asyncio.run_coroutine_threadsafe", side_effect=fake_run_coro):
+ error = _deliver_result(job, "hello", adapters=adapters, loop=loop)
+ return error, standalone
+
+
+def _primary_adapter():
+ adapter = MagicMock()
+ adapter.sent = []
+
+ async def send(chat_id, content, metadata=None):
+ adapter.sent.append(chat_id)
+ return {"success": True, "message_id": "m1"}
+
+ adapter.send = send
+ return adapter
+
+
+def test_satellite_routes_exact_target_through_primary_adapter(tmp_path, monkeypatch):
+ root = tmp_path / "root"
+ fitness_home = root / "profiles" / "fitness"
+ fitness_home.mkdir(parents=True)
+ (root / "config.yaml").write_text(yaml.safe_dump(PRIMARY_YAML), encoding="utf-8")
+ monkeypatch.setattr("hermes_constants.get_default_hermes_root", lambda: root)
+ primary = _primary_adapter()
+
+ token = set_hermes_home_override(str(fitness_home))
+ try:
+ shared = SharedRouteAdapters(
+ {Platform.DISCORD: primary}, _primary_profile_routes_for_current_home()
+ )
+ # exact enabled route → primary adapter sends, no standalone attempt
+ error, standalone = _run(_job("1543065293755256852"), shared)
+ assert error is None, error
+ assert primary.sent == ["1543065293755256852"]
+ assert standalone == []
+
+ # unmatched chat, disabled route, route for another profile → the
+ # primary bot is NEVER used; delivery stays on the satellite's own
+ # (credentialless) standalone path and reports its failure.
+ for chat in ("424242", "999", "777"):
+ primary.sent.clear()
+ error, standalone = _run(_job(chat), shared)
+ assert error is not None and "DISCORD_BOT_TOKEN" in error
+ assert primary.sent == []
+ assert standalone == [chat]
+ finally:
+ reset_hermes_home_override(token)
+
+
+def test_shared_view_is_falsy_without_routes_or_primary_adapters():
+ assert not SharedRouteAdapters({}, [])
+ assert SharedRouteAdapters({Platform.DISCORD: object()}, []).get(Platform.DISCORD) is None
diff --git a/tests/cron/test_cron_reasoning_effort.py b/tests/cron/test_cron_reasoning_effort.py
index cea6d11230..4f45549a05 100644
--- a/tests/cron/test_cron_reasoning_effort.py
+++ b/tests/cron/test_cron_reasoning_effort.py
@@ -194,7 +194,7 @@ class TestCronjobToolReasoningEffort:
def _tool_handler(self):
import tools.cronjob_tools as mod
- return mod.registry._tools["cronjob"].handler
+ return mod.registry._tools["cronjob_manage"].handler
def test_schema_does_not_expose_reasoning_effort(self):
"""Policy pin: the model-facing surface must NOT offer the
diff --git a/tests/cron/test_cron_timezone_migration_catchup.py b/tests/cron/test_cron_timezone_migration_catchup.py
new file mode 100644
index 0000000000..ef4cf4a108
--- /dev/null
+++ b/tests/cron/test_cron_timezone_migration_catchup.py
@@ -0,0 +1,213 @@
+"""Timezone-migration silent misfire on the cron fire path.
+
+Production incident: after upgrading from a build that scheduled in UTC to
+one that honours the profile timezone (Europe/Brussels), daily cron jobs
+stopped running. Their ``jobs.json`` rows still held pre-migration instants
+like ``2026-09-02T04:00:00+00:00`` for expr ``0 4 * * *``. ``_ensure_aware``
+normalizes that to ``06:00+02``, which ``0 4 * * *`` excludes, so the
+stale-expression guard (#93049) classified it as a direct ``jobs.json`` edit,
+logged exactly that, and re-anchored to tomorrow WITHOUT firing — the due
+occurrence vanished with no failure anywhere.
+
+The fix classifies the mismatch instead of assuming an edit: an instant whose
+own wall clock is a legal occurrence, and which only left the lattice because
+normalization changed its offset, is a representation migration and fires.
+
+These exercise the real store against a temp ``HERMES_HOME`` (no mocks) per
+the E2E-over-mocks discipline for file-touching code.
+"""
+
+from __future__ import annotations
+
+from datetime import datetime
+
+import pytest
+
+
+@pytest.fixture
+def temp_home(tmp_path, monkeypatch):
+ """Isolated HERMES_HOME so jobs.json doesn't touch the real store."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ yield tmp_path
+
+
+@pytest.fixture(autouse=True)
+def _reset_migration_counters(monkeypatch):
+ """Module-level telemetry counters must not leak between tests."""
+ from cron import jobs as J
+
+ monkeypatch.setattr(J, "_timezone_migration_catchups", 0)
+ monkeypatch.setattr(J, "_timezone_migration_catchups_recent", [])
+ yield
+
+
+# Europe/Brussels is +02:00 on this date; the legacy row was written by a
+# build that scheduled everything at the UTC offset.
+_BRUSSELS_NOW = datetime.fromisoformat("2026-09-02T06:05:00+02:00")
+_LEGACY_UTC_NEXT_RUN = "2026-09-02T04:00:00+00:00"
+_DAILY_0400 = "0 4 * * *"
+
+
+def _write_cron_job(expr: str, next_run_at: str, name: str = "t") -> str:
+ """Persist a cron job with a pinned next_run_at (the legacy-row shape)."""
+ from cron.jobs import create_job, load_jobs, save_jobs
+
+ job = create_job(prompt="x", schedule="every 5m", name=name)
+ jobs = load_jobs()
+ for j in jobs:
+ if j["id"] == job["id"]:
+ j["schedule"] = {"kind": "cron", "expr": expr}
+ j["next_run_at"] = next_run_at
+ save_jobs(jobs)
+ return job["id"]
+
+
+def test_legacy_utc_offset_next_run_still_fires(temp_home, monkeypatch):
+ """The incident case: a pre-migration +00:00 instant for a Brussels
+ ``0 4 * * *`` job must fire its due occurrence, not be re-anchored away."""
+ from cron.jobs import get_due_jobs, get_timezone_migration_catchup_stats
+
+ monkeypatch.setattr("cron.jobs._hermes_now", lambda: _BRUSSELS_NOW)
+ jid = _write_cron_job(_DAILY_0400, _LEGACY_UTC_NEXT_RUN)
+
+ due = get_due_jobs()
+
+ assert jid in [j["id"] for j in due]
+ stats = get_timezone_migration_catchup_stats()
+ assert stats["timezone_migration_catchups"] == 1
+ record = stats["recent"][0]
+ assert record["job_id"] == jid
+ assert record["expr"] == _DAILY_0400
+ assert record["stored_next_run_at"] == _LEGACY_UTC_NEXT_RUN
+ assert record["normalized_next_run_at"] == "2026-09-02T06:00:00+02:00"
+
+
+def test_legacy_offset_catchup_fires_at_most_once(temp_home, monkeypatch):
+ """The catch-up run is a single fire: once the scheduler advances the
+ job, the legacy instant is gone and a second scan finds nothing due."""
+ from cron.jobs import advance_next_run, get_due_jobs, get_job
+
+ monkeypatch.setattr("cron.jobs._hermes_now", lambda: _BRUSSELS_NOW)
+ jid = _write_cron_job(_DAILY_0400, _LEGACY_UTC_NEXT_RUN)
+
+ assert jid in [j["id"] for j in get_due_jobs()]
+ assert advance_next_run(jid) is True
+
+ # Re-anchored to tomorrow's occurrence, expressed in the configured zone.
+ assert get_job(jid)["next_run_at"] == "2026-09-03T04:00:00+02:00"
+ assert [j["id"] for j in get_due_jobs() if j["id"] == jid] == []
+
+
+def test_genuine_expr_edit_still_reanchors_without_firing(temp_home, monkeypatch):
+ """#93049 protection intact: a stale instant in the CURRENT offset (no
+ representation change) is still treated as an edit and does not fire."""
+ from cron.jobs import get_due_jobs, get_job, get_timezone_migration_catchup_stats
+
+ monkeypatch.setattr("cron.jobs._hermes_now", lambda: _BRUSSELS_NOW)
+ # Stored at the configured offset, but the expr was edited to 09:00.
+ jid = _write_cron_job("0 9 * * *", "2026-09-02T04:00:00+02:00")
+
+ due = get_due_jobs()
+
+ assert [j["id"] for j in due if j["id"] == jid] == []
+ assert get_job(jid)["next_run_at"] == "2026-09-02T09:00:00+02:00"
+ assert (
+ get_timezone_migration_catchup_stats()["timezone_migration_catchups"] == 0
+ )
+
+
+def test_expr_edit_on_a_legacy_offset_row_still_does_not_fire(temp_home, monkeypatch):
+ """A legacy +00:00 row whose expr was ALSO edited must not fire: the
+ stored wall clock is not an occurrence of the new expression either, so
+ the migration escape hatch does not open."""
+ from cron.jobs import get_due_jobs, get_job, get_timezone_migration_catchup_stats
+
+ monkeypatch.setattr("cron.jobs._hermes_now", lambda: _BRUSSELS_NOW)
+ jid = _write_cron_job("0 9 * * *", _LEGACY_UTC_NEXT_RUN)
+
+ due = get_due_jobs()
+
+ assert [j["id"] for j in due if j["id"] == jid] == []
+ assert get_job(jid)["next_run_at"] == "2026-09-02T09:00:00+02:00"
+ assert (
+ get_timezone_migration_catchup_stats()["timezone_migration_catchups"] == 0
+ )
+
+
+def test_future_local_wall_clock_is_left_scheduled(temp_home, monkeypatch):
+ """A legacy row whose normalized instant has not arrived yet is simply
+ not due — no catch-up, no re-anchor, no telemetry."""
+ from cron.jobs import get_due_jobs, get_job, get_timezone_migration_catchup_stats
+
+ before_due = datetime.fromisoformat("2026-09-02T05:00:00+02:00")
+ monkeypatch.setattr("cron.jobs._hermes_now", lambda: before_due)
+ jid = _write_cron_job(_DAILY_0400, _LEGACY_UTC_NEXT_RUN)
+
+ due = get_due_jobs()
+
+ assert [j["id"] for j in due if j["id"] == jid] == []
+ assert get_job(jid)["next_run_at"] == _LEGACY_UTC_NEXT_RUN
+ assert (
+ get_timezone_migration_catchup_stats()["timezone_migration_catchups"] == 0
+ )
+
+
+def test_future_stored_wall_clock_still_takes_the_offset_repair_path(
+ temp_home, monkeypatch
+):
+ """#28934 regression: a westward TZ move (+10 -> +02) that makes a still-
+ future wall clock look due recomputes rather than firing early, and is
+ NOT reclassified as a migration catch-up."""
+ from cron.jobs import get_due_jobs, get_job, get_timezone_migration_catchup_stats
+
+ scan_time = datetime.fromisoformat("2026-09-02T14:00:00+02:00")
+ monkeypatch.setattr("cron.jobs._hermes_now", lambda: scan_time)
+ jid = _write_cron_job("0 21 * * *", "2026-09-02T21:00:00+10:00")
+
+ due = get_due_jobs()
+
+ assert [j["id"] for j in due if j["id"] == jid] == []
+ assert get_job(jid)["next_run_at"] == "2026-09-02T21:00:00+02:00"
+ assert (
+ get_timezone_migration_catchup_stats()["timezone_migration_catchups"] == 0
+ )
+
+
+def test_classifier_separates_migration_from_edit(temp_home):
+ """Unit-level: the three classifications the fire path branches on."""
+ from cron.jobs import (
+ STALE_CRON_EXPR_EDIT,
+ STALE_CRON_MATCH,
+ STALE_CRON_TIMEZONE_MIGRATION,
+ _classify_stale_cron_next_run,
+ )
+
+ daily = {"kind": "cron", "expr": _DAILY_0400}
+ raw_legacy = datetime.fromisoformat(_LEGACY_UTC_NEXT_RUN)
+ normalized = datetime.fromisoformat("2026-09-02T06:00:00+02:00")
+ on_lattice = datetime.fromisoformat("2026-09-02T04:00:00+02:00")
+
+ # Stored instant already occurs under the current expression.
+ assert (
+ _classify_stale_cron_next_run(daily, on_lattice, on_lattice)
+ == STALE_CRON_MATCH
+ )
+ # Only the offset representation changed.
+ assert (
+ _classify_stale_cron_next_run(daily, raw_legacy, normalized)
+ == STALE_CRON_TIMEZONE_MIGRATION
+ )
+ # Wall clock never moved, so a mismatch can only be a schedule edit.
+ assert (
+ _classify_stale_cron_next_run(
+ {"kind": "cron", "expr": "0 9 * * *"}, on_lattice, on_lattice
+ )
+ == STALE_CRON_EXPR_EDIT
+ )
+ # Wall clock moved, but the stored wall clock is not an occurrence either.
+ assert (
+ _classify_stale_cron_next_run(
+ {"kind": "cron", "expr": "0 9 * * *"}, raw_legacy, normalized
+ )
+ == STALE_CRON_EXPR_EDIT
+ )
diff --git a/tests/cron/test_jobs.py b/tests/cron/test_jobs.py
index 6a3492f390..f993928890 100644
--- a/tests/cron/test_jobs.py
+++ b/tests/cron/test_jobs.py
@@ -645,6 +645,8 @@ class TestMarkJobRun:
assert updated is not None
assert updated["state"] == "completed"
assert updated["last_delivery_error"] == "platform 'telegram' not configured"
+ # A terminal completion that never reached the user is not a success.
+ assert updated["last_status"] == "delivery_failed"
def test_completed_oneshot_visible_in_list(self, tmp_cron_dir):
"""list_jobs(include_disabled=True) surfaces the completed record."""
@@ -654,6 +656,7 @@ class TestMarkJobRun:
assert job["id"] in listed
assert listed[job["id"]]["state"] == "completed"
assert listed[job["id"]]["last_delivery_error"] == "send failed: 502"
+ assert listed[job["id"]]["last_status"] == "delivery_failed"
# Default (enabled-only) listing hides it, matching paused/disabled jobs.
assert job["id"] not in {j["id"] for j in list_jobs()}
@@ -672,13 +675,53 @@ class TestMarkJobRun:
assert updated["last_error"] == "timeout"
def test_delivery_error_tracked_separately(self, tmp_cron_dir):
- """Agent succeeds but delivery fails — both tracked independently."""
+ """Agent succeeds but delivery fails — surfaced, not hidden behind ok.
+
+ Regression guard for #83993: recording ``last_status="ok"`` made a run
+ the user never received look like a quiet success everywhere that keys
+ off "ok". The agent error stays independent of the delivery error, and
+ the delivery failure is not an agent failure (no streak).
+ """
job = create_job(prompt="Report", schedule="every 1h")
- mark_job_run(job["id"], success=True, delivery_error="platform 'telegram' not configured")
+ mark_job_run(job["id"], success=True, delivery_error="send failed: 502")
updated = get_job(job["id"])
- assert updated["last_status"] == "ok"
+ assert updated["last_status"] == "delivery_failed"
assert updated["last_error"] is None
- assert updated["last_delivery_error"] == "platform 'telegram' not configured"
+ assert updated["last_delivery_error"] == "send failed: 502"
+ assert updated["failure_streak"] == 0
+
+ def test_success_without_delivery_error_stays_ok(self, tmp_cron_dir):
+ """A fully successful run is still plain "ok"."""
+ job = create_job(prompt="Report", schedule="every 1h")
+ mark_job_run(job["id"], success=True)
+ assert get_job(job["id"])["last_status"] == "ok"
+ # An empty delivery error is no error at all.
+ mark_job_run(job["id"], success=True, delivery_error="")
+ assert get_job(job["id"])["last_status"] == "ok"
+
+ def test_agent_failure_still_error_with_delivery_error(self, tmp_cron_dir):
+ """An agent failure outranks delivery: still "error", still a streak."""
+ job = create_job(prompt="Report", schedule="every 1h")
+ mark_job_run(
+ job["id"], success=False, error="timeout",
+ delivery_error="send failed: 502",
+ )
+ updated = get_job(job["id"])
+ assert updated["last_status"] == "error"
+ assert updated["last_error"] == "timeout"
+ assert updated["failure_streak"] == 1
+
+ def test_explicit_status_override_wins_over_delivery_failed(self, tmp_cron_dir):
+ """An explicit terminal status (T1-26 blocked_config) still wins."""
+ job = create_job(prompt="Report", schedule="every 1h")
+ mark_job_run(
+ job["id"], success=True,
+ delivery_error="send failed: 502",
+ status="blocked_config",
+ )
+ updated = get_job(job["id"])
+ assert updated["last_status"] == "blocked_config"
+ assert updated["last_delivery_error"] == "send failed: 502"
def test_failure_streak_increments_and_resets(self, tmp_cron_dir):
"""failure_streak counts consecutive agent failures; success resets."""
diff --git a/tests/cron/test_notepad.py b/tests/cron/test_notepad.py
index 9d140e5099..e91c1ac139 100644
--- a/tests/cron/test_notepad.py
+++ b/tests/cron/test_notepad.py
@@ -8,6 +8,7 @@ use the notepad, and the `hermes cron notepad` CLI handler.
from __future__ import annotations
import argparse
+import importlib
import sys
from pathlib import Path
@@ -103,6 +104,33 @@ class TestNotepadCrud:
assert not notepad.NOTEPAD_FILE.exists()
+class TestNotepadProfileIsolation:
+ def test_profile_override_routes_writes_to_current_home(self, tmp_path):
+ from hermes_constants import (
+ reset_hermes_home_override,
+ set_hermes_home_override,
+ )
+ import cron.notepad as notepad_mod
+
+ profile_a = tmp_path / "profile-a"
+ profile_b = tmp_path / "profile-b"
+
+ import_token = set_hermes_home_override(profile_a)
+ try:
+ importlib.reload(notepad_mod)
+ finally:
+ reset_hermes_home_override(import_token)
+
+ runtime_token = set_hermes_home_override(profile_b)
+ try:
+ notepad_mod.set_note("job-1", "cursor", "page=7")
+ finally:
+ reset_hermes_home_override(runtime_token)
+
+ assert (profile_b / "cron" / "notepad.db").exists()
+ assert not (profile_a / "cron" / "notepad.db").exists()
+
+
class TestJobRemovalCleanup:
def test_remove_job_clears_notepad(self, cron_env, notepad):
"""remove_job must clear the job's notepad rows — without this,
diff --git a/tests/cron/test_preflight_config.py b/tests/cron/test_preflight_config.py
index 0a12721d8b..4e56b3d6e9 100644
--- a/tests/cron/test_preflight_config.py
+++ b/tests/cron/test_preflight_config.py
@@ -129,7 +129,7 @@ class TestMissingProviderKeyBlocks:
job = _job()
deliveries = []
- def fake_deliver(job, content, adapters=None, loop=None):
+ def fake_deliver(job, content, adapters=None, loop=None, **kwargs):
deliveries.append(content)
return None
@@ -233,7 +233,7 @@ class TestOptOut:
job = _job()
deliveries = []
- def fake_deliver(job, content, adapters=None, loop=None):
+ def fake_deliver(job, content, adapters=None, loop=None, **kwargs):
deliveries.append(content)
return None
diff --git a/tests/cron/test_run_one_job.py b/tests/cron/test_run_one_job.py
index a3b8bb425a..93bcef00a5 100644
--- a/tests/cron/test_run_one_job.py
+++ b/tests/cron/test_run_one_job.py
@@ -29,7 +29,7 @@ def _patch_pipeline(monkeypatch, *, success=True, output="out", final="final res
calls.append(("save", jid))
return f"/tmp/{jid}.txt"
- def fake_deliver(job, content, adapters=None, loop=None):
+ def fake_deliver(job, content, adapters=None, loop=None, **kwargs):
calls.append(("deliver", job["id"]))
return None
diff --git a/tests/cron/test_scheduler_provider.py b/tests/cron/test_scheduler_provider.py
index 12ac73560c..6bf710e4ed 100644
--- a/tests/cron/test_scheduler_provider.py
+++ b/tests/cron/test_scheduler_provider.py
@@ -786,3 +786,109 @@ def test_multiplex_missing_secondary_does_not_fall_back_to_shared(tmp_path):
assert default_ad is shared
assert sec_ad is not shared
assert not sec_ad
+
+
+def test_multiplex_ticker_isolates_profile_failures(tmp_path):
+ """A failing profile's tick must not skip healthy siblings in the same
+ cycle, nor darken their status (#74878)."""
+ from cron.jobs import get_ticker_last_error, record_ticker_error, use_cron_store
+ from cron.scheduler_provider import InProcessCronScheduler
+ from hermes_constants import get_hermes_home
+
+ failing_home = tmp_path / "failing"
+ healthy_home = tmp_path / "healthy"
+ for home in (failing_home, healthy_home):
+ (home / "cron").mkdir(parents=True)
+ with use_cron_store(home):
+ record_ticker_error("RuntimeError: stale failure")
+
+ stop = threading.Event()
+ tick_homes: list[str] = []
+
+ def _tick(*args, **kwargs):
+ home = str(get_hermes_home())
+ tick_homes.append(home)
+ if home == str(failing_home):
+ raise RuntimeError("profile-local failure")
+ stop.set()
+ return 0
+
+ provider = InProcessCronScheduler()
+ with patch("cron.scheduler.tick", side_effect=_tick):
+ thread = threading.Thread(
+ target=provider.start,
+ args=(stop,),
+ kwargs={
+ "interval": 0,
+ "profile_homes": [("failing", failing_home), ("healthy", healthy_home)],
+ },
+ daemon=True,
+ )
+ thread.start()
+ thread.join(timeout=5)
+ stop.set()
+ thread.join(timeout=5)
+
+ assert not thread.is_alive()
+ assert str(healthy_home) in tick_homes, "healthy sibling was skipped"
+ assert not (failing_home / "cron" / "ticker_last_success").exists()
+ assert (healthy_home / "cron" / "ticker_last_success").exists()
+ with use_cron_store(failing_home):
+ assert get_ticker_last_error() == "RuntimeError: profile-local failure"
+ with use_cron_store(healthy_home):
+ assert get_ticker_last_error() is None
+
+
+def test_multiplex_recovery_isolates_profile_failures(tmp_path):
+ """A startup-recovery error in one profile's ledger must not kill the
+ ticker thread before it ever ticks (#74878)."""
+ import sqlite3
+
+ from cron.scheduler_provider import InProcessCronScheduler
+ from hermes_constants import get_hermes_home
+
+ failing_home = tmp_path / "failing"
+ healthy_home = tmp_path / "healthy"
+ for home in (failing_home, healthy_home):
+ (home / "cron").mkdir(parents=True)
+
+ stop = threading.Event()
+ recovery_homes: list[str] = []
+ tick_homes: list[str] = []
+
+ def _recover():
+ home = str(get_hermes_home())
+ recovery_homes.append(home)
+ if home == str(failing_home):
+ raise sqlite3.OperationalError("unable to open database file")
+ return 0
+
+ def _tick(*args, **kwargs):
+ tick_homes.append(str(get_hermes_home()))
+ if len(tick_homes) >= 2:
+ stop.set()
+ return 0
+
+ provider = InProcessCronScheduler()
+ with (
+ patch.object(provider, "recover_interrupted", side_effect=_recover),
+ patch("cron.scheduler.tick", side_effect=_tick),
+ ):
+ thread = threading.Thread(
+ target=provider.start,
+ args=(stop,),
+ kwargs={
+ "interval": 0,
+ "profile_homes": [("failing", failing_home), ("healthy", healthy_home)],
+ },
+ daemon=True,
+ )
+ thread.start()
+ thread.join(timeout=5)
+ stop.set()
+ thread.join(timeout=5)
+
+ assert not thread.is_alive()
+ assert recovery_homes == [str(failing_home), str(healthy_home)]
+ # The failing profile stays in rotation: its ledger may still hold jobs.
+ assert set(tick_homes) == {str(failing_home), str(healthy_home)}
diff --git a/tests/cron/test_suggestions.py b/tests/cron/test_suggestions.py
index 605686f52c..3abaf54d31 100644
--- a/tests/cron/test_suggestions.py
+++ b/tests/cron/test_suggestions.py
@@ -19,7 +19,6 @@ def store(tmp_path, monkeypatch):
home = tmp_path / ".hermes"
home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(home))
- # Reload so module-level CRON_DIR/SUGGESTIONS_FILE pick up the temp home.
import hermes_constants
importlib.reload(hermes_constants)
import cron.suggestions as s
@@ -38,6 +37,51 @@ def _add(store, key="k1", title="Test", source="catalog", schedule="0 9 * * *"):
class TestStore:
+ def test_explicit_file_override_wins_over_profile_home(self, tmp_path, monkeypatch):
+ from hermes_constants import (
+ reset_hermes_home_override,
+ set_hermes_home_override,
+ )
+ import cron.suggestions as suggestions_mod
+
+ explicit_file = tmp_path / "explicit" / "suggestions.json"
+ profile_home = tmp_path / "profile"
+ monkeypatch.setattr(suggestions_mod, "SUGGESTIONS_FILE", explicit_file)
+
+ token = set_hermes_home_override(profile_home)
+ try:
+ _add(suggestions_mod, key="explicit-file")
+ finally:
+ reset_hermes_home_override(token)
+
+ assert explicit_file.exists()
+ assert not (profile_home / "cron" / "suggestions.json").exists()
+
+ def test_profile_override_routes_writes_to_current_home(self, tmp_path):
+ from hermes_constants import (
+ reset_hermes_home_override,
+ set_hermes_home_override,
+ )
+ import cron.suggestions as suggestions_mod
+
+ profile_a = tmp_path / "profile-a"
+ profile_b = tmp_path / "profile-b"
+
+ import_token = set_hermes_home_override(profile_a)
+ try:
+ importlib.reload(suggestions_mod)
+ finally:
+ reset_hermes_home_override(import_token)
+
+ runtime_token = set_hermes_home_override(profile_b)
+ try:
+ _add(suggestions_mod, key="profile-b")
+ finally:
+ reset_hermes_home_override(runtime_token)
+
+ assert (profile_b / "cron" / "suggestions.json").exists()
+ assert not (profile_a / "cron" / "suggestions.json").exists()
+
def test_add_and_list_pending(self, store):
rec = _add(store)
assert rec is not None
diff --git a/tests/gateway/feishu_helpers.py b/tests/gateway/feishu_helpers.py
index ae8a4bfc37..97771daaa3 100644
--- a/tests/gateway/feishu_helpers.py
+++ b/tests/gateway/feishu_helpers.py
@@ -34,6 +34,7 @@ def make_adapter_skeleton(
allow_bots: str = "none",
require_mention: bool = True,
group_policy: str = "allowlist",
+ allow_all_dm: bool = False,
) -> Any:
from plugins.platforms.feishu.adapter import FeishuAdapter
@@ -48,6 +49,7 @@ def make_adapter_skeleton(
adapter._default_group_policy = group_policy
adapter._allowed_group_users = frozenset()
adapter._allow_bots = allow_bots
+ adapter._allow_all_dm = allow_all_dm
adapter._require_mention = require_mention
return adapter
diff --git a/tests/gateway/relay/test_relay_passthrough.py b/tests/gateway/relay/test_relay_passthrough.py
index 2150e9bf0b..a8e27e4335 100644
--- a/tests/gateway/relay/test_relay_passthrough.py
+++ b/tests/gateway/relay/test_relay_passthrough.py
@@ -44,7 +44,7 @@ def adapter():
return RelayAdapter(PlatformConfig(), _desc(), transport=StubConnector(_desc()))
-def _interaction_forward(payload: dict) -> PassthroughForward:
+def _interaction_forward(payload: dict, *, profile: str | None = None) -> PassthroughForward:
body = json.dumps(payload).encode("utf-8")
return PassthroughForward(
platform="discord",
@@ -53,6 +53,7 @@ def _interaction_forward(payload: dict) -> PassthroughForward:
path="/interactions/discord/appShared",
headers=[("content-type", "application/json")],
body=body,
+ profile=profile,
)
@@ -75,6 +76,27 @@ def test_passthrough_from_wire_byte_preserves_body():
assert fwd.headers == [("content-type", "application/json")]
+def test_passthrough_from_wire_stamps_routed_profile():
+ """A connector-routed profile on the wire frame lands on PassthroughForward.
+
+ Mirrors _event_from_wire's profile stamping for the ``inbound`` frame
+ (#60586) — the passthrough plane needs the same carry-through so a
+ Team-Gateway's Discord interactions route to the same profile a plain
+ message would.
+ """
+ wire = {
+ "platform": "discord",
+ "botId": "appShared",
+ "method": "POST",
+ "path": "/interactions/discord/appShared",
+ "headers": [],
+ "bodyB64": "",
+ "profile": "reviewer",
+ }
+ fwd = _passthrough_from_wire(wire)
+ assert fwd.profile == "reviewer"
+
+
@pytest.mark.asyncio
async def test_connect_wires_passthrough_handler_over_ws(adapter):
"""connect() registers the passthrough handler on the transport so a
@@ -137,6 +159,40 @@ async def test_discord_interaction_routes_through_handle_message(adapter, monkey
assert adapter._platform_by_chat.get("chan-9") == "discord"
+@pytest.mark.asyncio
+async def test_discord_interaction_stamps_routed_profile(adapter, monkeypatch):
+ """A connector-routed profile on the passthrough forward lands on the
+ resulting event's SessionSource, the same way it does for a plain relayed
+ message (#60586) — so a Team-Gateway's Discord slash-command/button/modal
+ routes to the same profile a plain message would, instead of always
+ falling back to agent:main."""
+ await adapter.connect()
+ stub = adapter._transport
+
+ seen = []
+
+ async def fake_handle(event):
+ seen.append(event)
+
+ monkeypatch.setattr(adapter, "handle_message", fake_handle)
+
+ fwd = _interaction_forward(
+ {
+ "id": "interaction-2",
+ "type": 2, # APPLICATION_COMMAND
+ "channel_id": "chan-9",
+ "guild_id": "guild-7",
+ "data": {"name": "summarize"},
+ "member": {"user": {"id": "user-3", "username": "ben"}},
+ },
+ profile="reviewer",
+ )
+ await stub.push_passthrough(fwd, buffer_id=None)
+
+ assert len(seen) == 1
+ assert seen[0].source.profile == "reviewer"
+
+
@pytest.mark.asyncio
async def test_application_command_subcommand_nesting_renders_names_then_values(
adapter, monkeypatch
diff --git a/tests/gateway/test_42039_duplicate_user_message.py b/tests/gateway/test_42039_duplicate_user_message.py
index 13a73181f6..3ddc30d90e 100644
--- a/tests/gateway/test_42039_duplicate_user_message.py
+++ b/tests/gateway/test_42039_duplicate_user_message.py
@@ -24,7 +24,7 @@ import pytest
import gateway.run as gateway_run
from gateway.config import GatewayConfig, Platform
from gateway.platforms.base import MessageEvent
-from gateway.session import SessionEntry, SessionSource
+from gateway.session import SessionEntry, SessionSource, TranscriptReadError
def _bootstrap(monkeypatch, tmp_path):
@@ -185,6 +185,24 @@ async def test_not_new_messages_skip_db_when_agent_has_session_db(
)
+@pytest.mark.asyncio
+async def test_transcript_read_failure_stops_turn_before_agent_or_append(
+ monkeypatch, tmp_path
+):
+ runner = _bootstrap(monkeypatch, tmp_path)
+ runner.session_store.load_transcript.side_effect = TranscriptReadError("sess-dedup")
+ runner._run_agent = AsyncMock()
+
+ response = await runner._handle_message_with_agent(
+ _event(), _source(), "agent:main:telegram:group:-1001:12345", 1
+ )
+
+ assert "history is temporarily unavailable" in response
+ assert "not processed" in response
+ runner._run_agent.assert_not_awaited()
+ runner.session_store.append_to_transcript.assert_not_called()
+
+
# ── Post-stream MEDIA delivery keeps prior-turn deduplication ──────────
diff --git a/tests/gateway/test_64674_multiplex_primary_token_scope.py b/tests/gateway/test_64674_multiplex_primary_token_scope.py
index 44398aec19..b11fedb177 100644
--- a/tests/gateway/test_64674_multiplex_primary_token_scope.py
+++ b/tests/gateway/test_64674_multiplex_primary_token_scope.py
@@ -119,6 +119,60 @@ class TestPlatformHasBotCredential:
Platform.TELEGRAM, PlatformConfig(enabled=True, token=None)
) is False
+ def test_matrix_password_login_is_a_credential(self):
+ """Matrix password auth has no .token but is fully reconnectable.
+
+ MATRIX_USER_ID + MATRIX_PASSWORD with no MATRIX_ACCESS_TOKEN is a
+ supported setup (build_config puts it on extra). Treating it as
+ credential-less evicted it from the reconnect queue on the first
+ transient failure, so a momentary DNS blip took Matrix down until
+ the gateway was restarted by hand.
+ """
+ from gateway.run import _platform_has_bot_credential
+
+ cfg = PlatformConfig(enabled=True)
+ cfg.extra = {
+ "homeserver": "https://matrix.example.org",
+ "user_id": "@bot:matrix.example.org",
+ "password": "hunter2",
+ }
+ assert _platform_has_bot_credential(Platform.MATRIX, cfg) is True
+
+ @pytest.mark.parametrize(
+ "extra",
+ [
+ {},
+ {"homeserver": "https://matrix.example.org", "password": "hunter2"},
+ {"user_id": "@bot:matrix.example.org", "password": "hunter2"},
+ {"homeserver": "https://matrix.example.org", "user_id": "@bot:m.example.org"},
+ {"homeserver": " ", "user_id": " ", "password": " "},
+ ],
+ ids=["empty", "no-user-id", "no-homeserver", "no-password", "blank"],
+ )
+ def test_matrix_incomplete_password_config_still_dropped(self, extra, monkeypatch):
+ """An incomplete Matrix config can never connect — keep evicting it.
+
+ Guards the #64674 intent, and specifically pins "read extra, not the
+ environment". A fully-populated MATRIX_* environment is set here on
+ purpose: on a real host those vars are present (build_config exports
+ them, and importing gateway.run loads ~/.hermes/.env), so an
+ implementation that falls back to os.getenv would report every Matrix
+ config as credentialed and never evict anything.
+
+ conftest sandboxes HERMES_HOME and scrubs MATRIX_* from the
+ environment, so without these explicit setenv calls this test would
+ pass against an env-reading implementation and guard nothing.
+ """
+ from gateway.run import _platform_has_bot_credential
+
+ monkeypatch.setenv("MATRIX_HOMESERVER", "https://env.example.org")
+ monkeypatch.setenv("MATRIX_USER_ID", "@envbot:env.example.org")
+ monkeypatch.setenv("MATRIX_PASSWORD", "env-password")
+
+ cfg = PlatformConfig(enabled=True)
+ cfg.extra = dict(extra)
+ assert _platform_has_bot_credential(Platform.MATRIX, cfg) is False
+
class TestPrimaryStartupSkipsEmptyTokenUnderMultiplex:
@pytest.mark.asyncio
diff --git a/tests/gateway/test_73771_media_resend_dedup.py b/tests/gateway/test_73771_media_resend_dedup.py
index d0490e8056..3c54e1c055 100644
--- a/tests/gateway/test_73771_media_resend_dedup.py
+++ b/tests/gateway/test_73771_media_resend_dedup.py
@@ -353,18 +353,19 @@ async def test_bare_path_history_lookup_timeout_fails_open(tmp_path, monkeypatch
started = time.monotonic()
await adapter._process_message_background(event, build_session_key(event.source))
- # The lookup times out after 0.02s and fails open; the generous 1.0s
- # bound only guards against delivery hanging on the wedged read
- # indefinitely, without flaking on loaded CI hosts. Delivery of the
- # document below is the real fail-open assertion.
- assert time.monotonic() - started < 1.0
+ # The lookup times out after 0.02s and fails open; the bound only guards
+ # against delivery hanging on the wedged read indefinitely. 1.0s still
+ # flaked on loaded CI runners (observed 1.55s on main run 33455779041),
+ # so keep it >= 5s per the flake policy. Delivery of the document below
+ # is the real fail-open assertion.
+ assert time.monotonic() - started < 5.0
assert adapter.documents == [str(pdf)]
@pytest.mark.asyncio
async def test_history_lookup_saturation_fails_open_without_new_worker(monkeypatch):
"""Wedged lookups are bounded and cannot consume unbounded worker threads."""
- monkeypatch.setattr("gateway.platforms.base._HISTORY_MEDIA_LOOKUP_TIMEOUT_SECONDS", 1.0)
+ monkeypatch.setattr("gateway.platforms.base._HISTORY_MEDIA_LOOKUP_TIMEOUT_SECONDS", 5.0)
monkeypatch.setattr(
"gateway.platforms.base._HISTORY_MEDIA_LOOKUP_ADMISSION",
threading.BoundedSemaphore(2),
@@ -381,13 +382,13 @@ async def test_history_lookup_saturation_fails_open_without_new_worker(monkeypat
calls += 1
if calls == 2:
two_started.set()
- release.wait(timeout=1)
+ release.wait(timeout=10)
return None
monkeypatch.setattr(adapter, "_history_media_paths_for_session", blocked_lookup)
first = asyncio.create_task(adapter._bounded_history_media_paths_for_session("one"))
second = asyncio.create_task(adapter._bounded_history_media_paths_for_session("two"))
- deadline = time.monotonic() + 1
+ deadline = time.monotonic() + 5
while not two_started.is_set() and time.monotonic() < deadline:
await asyncio.sleep(0.005)
assert two_started.is_set()
@@ -397,9 +398,10 @@ async def test_history_lookup_saturation_fails_open_without_new_worker(monkeypat
elapsed = time.monotonic() - began
assert third is None
- # Saturation must fail open immediately (no waiting on the 1.0s lookup
- # timeout); 0.5s is a generous bound that stays flake-free on loaded CI.
- assert elapsed < 0.5
+ # Saturation must fail open immediately (no waiting on the 5.0s lookup
+ # timeout); 2.0s keeps the distinction while staying flake-free on
+ # loaded CI runners.
+ assert elapsed < 2.0
assert calls == 2
release.set()
await asyncio.gather(first, second)
diff --git a/tests/gateway/test_api_server.py b/tests/gateway/test_api_server.py
index ecb49cd7a8..56e298dc06 100644
--- a/tests/gateway/test_api_server.py
+++ b/tests/gateway/test_api_server.py
@@ -611,7 +611,9 @@ class TestDisconnectedAgentReap:
adapter._active_run_agents["run_x"] = agent
request = MagicMock()
+ request.headers = {}
request.match_info = {"run_id": "run_x"}
+ adapter._run_owners["run_x"] = adapter._run_idempotency_scope(request)
resp = await adapter._handle_stop_run(request)
assert resp.status == 200
diff --git a/tests/gateway/test_api_server_runs.py b/tests/gateway/test_api_server_runs.py
index 0f9f573ac4..8d68f919c9 100644
--- a/tests/gateway/test_api_server_runs.py
+++ b/tests/gateway/test_api_server_runs.py
@@ -22,6 +22,7 @@ from aiohttp.test_utils import TestClient, TestServer
from gateway.config import PlatformConfig
from gateway.platforms.api_server import (
APIServerAdapter,
+ _api_request_profile,
_approval_event_choices,
cors_middleware,
security_headers_middleware,
@@ -68,6 +69,13 @@ def _make_adapter(api_key: str = "") -> APIServerAdapter:
return adapter
+def _claim_run(adapter: APIServerAdapter, run_id: str) -> None:
+ """Stamp *run_id* as owned by the unprefixed (default) request scope."""
+ request = MagicMock()
+ request.headers = {}
+ adapter._run_owners[run_id] = adapter._run_idempotency_scope(request)
+
+
def _create_runs_app(adapter: APIServerAdapter) -> web.Application:
"""Create an aiohttp app with /v1/runs routes registered."""
mws = [mw for mw in (cors_middleware, security_headers_middleware) if mw is not None]
@@ -468,6 +476,7 @@ class TestSteerRun:
adapter._active_run_agents["run_123"] = agent
adapter._run_streams["run_123"] = queue
adapter._set_run_status("run_123", "running")
+ _claim_run(adapter, "run_123")
async with TestClient(TestServer(app)) as cli:
resp = await cli.post("/v1/runs/run_123/steer", json={"input": "tighten the ending"})
@@ -500,6 +509,7 @@ class TestSteerRun:
async def test_steer_inactive_run_returns_409(self, adapter):
app = _create_runs_app(adapter)
adapter._set_run_status("run_done", "completed")
+ _claim_run(adapter, "run_done")
async with TestClient(TestServer(app)) as cli:
resp = await cli.post("/v1/runs/run_done/steer", json={"input": "hello"})
@@ -515,6 +525,7 @@ class TestSteerRun:
agent.steer.return_value = True
adapter._active_run_agents["run_123"] = agent
adapter._set_run_status("run_123", "running")
+ _claim_run(adapter, "run_123")
async with TestClient(TestServer(app)) as cli:
resp = await cli.post("/v1/runs/run_123/steer", json={"input": ""})
@@ -681,6 +692,92 @@ class TestRunLifecycleSweep:
mock_agent.interrupt.assert_called_once_with("Stop requested via API")
+# ---------------------------------------------------------------------------
+# Run ownership across served profiles (#93689 / #90415)
+# ---------------------------------------------------------------------------
+
+
+class TestRunOwnershipAcrossProfiles:
+ """Every served profile holds a valid key under multiplex; only the
+ creating profile may see or control a run."""
+
+ KEYS = {"victim": "sk-victim-profile-key-0001", "attacker": "sk-attacker-profile-key-01"}
+
+ @classmethod
+ def _profile_app(cls, adapter: APIServerAdapter) -> web.Application:
+ """Runs routes behind a stand-in for the /p// middleware:
+ the routed profile arrives in ``X-Test-Profile`` and each profile
+ authenticates with its own key, as under gateway.multiplex_profiles."""
+
+ @web.middleware
+ async def stamp_profile(request, handler):
+ token = _api_request_profile.set(request.headers.get("X-Test-Profile"))
+ try:
+ return await handler(request)
+ finally:
+ _api_request_profile.reset(token)
+
+ adapter._expected_api_key = lambda: cls.KEYS.get(_api_request_profile.get(), "")
+ app = _create_runs_app(adapter)
+ app.middlewares.append(stamp_profile)
+ app.router.add_post(
+ "/api/sessions/{session_id}/chat/stream", adapter._handle_session_chat_stream
+ )
+ return app
+
+ @pytest.mark.asyncio
+ async def test_unstamped_run_state_fails_closed(self, adapter):
+ """Run state with no owner stamp is nobody's — not everybody's."""
+ app = _create_runs_app(adapter)
+ adapter._active_run_agents["run_unstamped"] = MagicMock()
+ adapter._set_run_status("run_unstamped", "running")
+
+ async with TestClient(TestServer(app)) as cli:
+ get_resp = await cli.get("/v1/runs/run_unstamped")
+ stop_resp = await cli.post("/v1/runs/run_unstamped/stop")
+
+ assert (get_resp.status, stop_resp.status) == (404, 404)
+
+ @pytest.mark.asyncio
+ async def test_session_chat_stream_run_is_owned_by_creating_profile(self, adapter):
+ """The session-chat-stream run mint claims ownership like /v1/runs does."""
+ app = self._profile_app(adapter)
+ victim = {"X-Test-Profile": "victim", "Authorization": f"Bearer {self.KEYS['victim']}"}
+ attacker = {"X-Test-Profile": "attacker", "Authorization": f"Bearer {self.KEYS['attacker']}"}
+ gate = asyncio.Event()
+
+ async def slow_run_agent(**kwargs):
+ await gate.wait()
+ return {"final_response": "ok"}, {}
+
+ async with TestClient(TestServer(app)) as cli:
+ with (
+ patch.object(adapter, "_get_existing_session_or_404", new=AsyncMock(return_value=({"id": "s1"}, None))),
+ patch.object(adapter, "_conversation_history_for_session", new=AsyncMock(return_value=[])),
+ patch.object(adapter, "_run_agent", new=slow_run_agent),
+ ):
+ stream = await cli.post(
+ "/api/sessions/s1/chat/stream", json={"message": "hi"}, headers=victim
+ )
+ await stream.content.readline()
+ (run_id,) = list(adapter._run_statuses)
+ assert run_id in adapter._run_owners
+
+ foreign_get = await cli.get(f"/v1/runs/{run_id}", headers=attacker)
+ foreign_stop = await cli.post(f"/v1/runs/{run_id}/stop", headers=attacker)
+ own_get = await cli.get(f"/v1/runs/{run_id}", headers=victim)
+ assert (foreign_get.status, foreign_stop.status, own_get.status) == (404, 404, 200)
+
+ gate.set()
+ await stream.text()
+
+ # The owner outlives the terminal status and goes with the last surface.
+ assert run_id in adapter._run_owners
+ adapter._run_statuses.pop(run_id)
+ adapter._release_run_owner_if_forgotten(run_id)
+ assert run_id not in adapter._run_owners
+
+
# ---------------------------------------------------------------------------
# POST /v1/runs/{run_id}/stop — interrupt a running agent
# ---------------------------------------------------------------------------
diff --git a/tests/gateway/test_api_server_toolset.py b/tests/gateway/test_api_server_toolset.py
index fb9fe9176b..debdbbfb52 100644
--- a/tests/gateway/test_api_server_toolset.py
+++ b/tests/gateway/test_api_server_toolset.py
@@ -17,11 +17,11 @@ class TestHermesApiServerToolset:
def test_toolset_includes_core_tools(self):
tools = resolve_toolset("hermes-api-server")
expected = [
- "terminal", "process",
+ "terminal", "process_manage",
"read_file", "write_file", "patch", "search_files",
"vision_analyze", "image_generate",
"execute_code", "delegate_task",
- "todo", "memory", "session_search", "cronjob",
+ "todo_list", "memory", "session_search", "cronjob_manage",
]
for tool in expected:
assert tool in tools, f"Missing expected tool: {tool}"
diff --git a/tests/gateway/test_choice_picker.py b/tests/gateway/test_choice_picker.py
index a2c9a52961..c8e6712ec0 100644
--- a/tests/gateway/test_choice_picker.py
+++ b/tests/gateway/test_choice_picker.py
@@ -126,7 +126,7 @@ class TestFastChoicePicker:
assert result is None
values = [c["value"] for c in adapter.calls[0]["choices"]]
- assert values == ["fast", "normal"]
+ assert values == ["fast", "normal", "auto", "cold"]
@pytest.mark.asyncio
async def test_fast_picker_selection_is_session_scoped(self, tmp_path, monkeypatch):
diff --git a/tests/gateway/test_codex_hygiene_compaction.py b/tests/gateway/test_codex_hygiene_compaction.py
index 71dd907f3b..7fa795d694 100644
--- a/tests/gateway/test_codex_hygiene_compaction.py
+++ b/tests/gateway/test_codex_hygiene_compaction.py
@@ -336,3 +336,46 @@ def test_manual_compress_without_live_thread_reports_honestly():
host._compress_codex_app_server_session("tg:123", "sess-1")
)
assert "Nothing to compact" in reply
+
+
+# ---------------------------------------------------------------------------
+# Multiplexed gateway: the hygiene worker must see the caller's ContextVars
+# (profile secret scope / HERMES_HOME override). A bare run_in_executor worker
+# starts with an EMPTY Context, so get_secret(_API_KEY) inside the
+# summary path fails closed and every hygiene compaction degrades to a lossy
+# truncation (#100849 bundle).
+# ---------------------------------------------------------------------------
+
+def test_hygiene_worker_inherits_caller_contextvars(tmp_path):
+ import contextvars
+ import threading
+
+ marker = contextvars.ContextVar("hygiene_scope_marker", default=None)
+ seen = {}
+
+ class ScopeProbeAgent(LiveCodexAgent):
+ def _compress_context(self, messages, system_message, **kwargs):
+ seen["value"] = marker.get()
+ seen["thread"] = threading.current_thread().name
+ return super()._compress_context(messages, system_message, **kwargs)
+
+ agent = ScopeProbeAgent(mode="hermes")
+ key = "tg:ctx"
+ gw, _db = _gateway(tmp_path, key, agent)
+
+ async def _scoped():
+ token = marker.set("profile-scope")
+ try:
+ return await run_codex_hygiene_compaction(
+ gw, key, agent.session_id, auto_mode="hermes",
+ history=_history(), approx_tokens=345_000, timeout_seconds=30.0,
+ )
+ finally:
+ marker.reset(token)
+
+ assert asyncio.run(_scoped()) == "compacted"
+ assert seen["thread"] != "MainThread", "compaction must still run off-loop"
+ assert seen["value"] == "profile-scope", (
+ "hygiene worker lost the caller's ContextVars — under multiplex_profiles "
+ "this is the UnscopedSecretError / lossy-truncation regression"
+ )
diff --git a/tests/gateway/test_config.py b/tests/gateway/test_config.py
index 2e4285f68f..480e26d48c 100644
--- a/tests/gateway/test_config.py
+++ b/tests/gateway/test_config.py
@@ -1409,3 +1409,58 @@ class TestApiServerEnvOverride:
assert config.platforms[Platform.API_SERVER].enabled is False
# The key is still wired through for the shared listener.
assert config.platforms[Platform.API_SERVER].extra.get("key") == api_server_key
+
+
+class TestWebhookEnvOverride:
+ def test_env_key_does_not_reenable_explicitly_disabled_webhook(self):
+ """An explicit ``platforms.webhook.enabled: false`` must survive
+ _apply_env_overrides() even when WEBHOOK_ENABLED is truthy in the env.
+
+ Regression (#85637): _apply_env_overrides() force-set
+ webhook.enabled = True whenever WEBHOOK_ENABLED was truthy. In
+ multiplex mode a secondary profile pins ``webhook.enabled: false`` so
+ it shares the default profile's listener instead of binding its own
+ port, but it still inherits the process-level WEBHOOK_ENABLED
+ (or carries one in its own .env). The unconditional re-enable
+ flipped it back on and tripped the MultiplexConfigError check.
+
+ The fix honors the explicit disable, flagged by ``_enabled_explicit``
+ in the platform's extra (set when the config.yaml pins enabled).
+ The MSGRAPH_WEBHOOK branch shares the shape and the fix.
+ """
+ config = GatewayConfig(
+ platforms={
+ Platform.WEBHOOK: PlatformConfig(
+ enabled=False,
+ extra={"_enabled_explicit": True},
+ ),
+ Platform.MSGRAPH_WEBHOOK: PlatformConfig(
+ enabled=False,
+ extra={"_enabled_explicit": True},
+ ),
+ },
+ )
+
+ with patch.dict(
+ os.environ,
+ {
+ "WEBHOOK_ENABLED": "true",
+ "WEBHOOK_PORT": "9999",
+ "WEBHOOK_SECRET": "shared-secret",
+ "MSGRAPH_WEBHOOK_ENABLED": "true",
+ "MSGRAPH_WEBHOOK_PORT": "9998",
+ },
+ clear=True,
+ ):
+ _apply_env_overrides(config)
+
+ # Explicit disable wins over the env-var presence.
+ assert config.platforms[Platform.WEBHOOK].enabled is False
+ assert config.platforms[Platform.MSGRAPH_WEBHOOK].enabled is False
+ assert config.platforms[Platform.MSGRAPH_WEBHOOK].extra.get("port") == 9998
+ # Port/secret are still wired through for the shared listener.
+ assert config.platforms[Platform.WEBHOOK].extra.get("port") == 9999
+ assert (
+ config.platforms[Platform.WEBHOOK].extra.get("secret")
+ == "shared-secret"
+ )
diff --git a/tests/gateway/test_cron_interrupt_notification.py b/tests/gateway/test_cron_interrupt_notification.py
index bde4738c20..f157e17478 100644
--- a/tests/gateway/test_cron_interrupt_notification.py
+++ b/tests/gateway/test_cron_interrupt_notification.py
@@ -120,6 +120,37 @@ class TestNotifyInterruptedCronJobs:
assert sent == 0
assert adapter.sent == []
+ @pytest.mark.asyncio
+ async def test_failure_deliver_local_suppresses_interrupt_notice(self):
+ """Interrupted notices are failure-category engine status (NS-788):
+ a job with failure_deliver='local' opted out of failure pings, and
+ the shutdown notice must honor that. Real target resolution — no
+ _resolve_delivery_targets patch — so the failure_deliver override is
+ actually exercised."""
+ runner, adapter = make_restart_runner()
+ _bind_notifier(runner)
+ job = dict(_telegram_job(), failure_deliver="local")
+
+ with patch("cron.jobs.get_job", return_value=job):
+ sent = await runner._notify_interrupted_cron_jobs([job["id"]])
+
+ assert sent == 0
+ assert adapter.sent == []
+
+ @pytest.mark.asyncio
+ async def test_failure_deliver_unset_notice_reaches_deliver_target(self):
+ """Control for the suppress test: same job without failure_deliver,
+ same real resolution path — the notice goes to the deliver target."""
+ runner, adapter = make_restart_runner()
+ _bind_notifier(runner)
+ job = _telegram_job()
+
+ with patch("cron.jobs.get_job", return_value=job):
+ sent = await runner._notify_interrupted_cron_jobs([job["id"]])
+
+ assert sent == 1
+ assert adapter.sent_calls[0][0] == "123456"
+
@pytest.mark.asyncio
async def test_empty_job_list_is_a_noop(self):
runner, adapter = make_restart_runner()
diff --git a/tests/gateway/test_delivery_silence_filter.py b/tests/gateway/test_delivery_silence_filter.py
index 1013e4bc75..11b7ba3296 100644
--- a/tests/gateway/test_delivery_silence_filter.py
+++ b/tests/gateway/test_delivery_silence_filter.py
@@ -124,6 +124,50 @@ async def test_env_override_enables_filter_over_config(tmp_path, monkeypatch):
assert result["filtered"] == "silence_narration"
+# --- Cron artifacts are exempt ----------------------------------------------
+#
+# The filter exists to stop bot-to-bot mirror loops of *model chatter*. Cron
+# output is an artifact: a job that legitimately emits "..." (a quiet script,
+# a terse digest) has no loop partner, and dropping it while returning
+# {"success": True} produced a cron the scheduler logged as delivered and the
+# user never received (#77763). Cron sends carry job_id in metadata.
+
+
+@pytest.mark.asyncio
+async def test_cron_job_id_metadata_bypasses_the_filter(tmp_path, monkeypatch):
+ monkeypatch.setattr("gateway.delivery.get_hermes_home", lambda: tmp_path)
+ monkeypatch.delenv("HERMES_FILTER_SILENCE_NARRATION", raising=False)
+ adapter = RecordingAdapter()
+ router = DeliveryRouter(GatewayConfig(), adapters={Platform.DISCORD: adapter})
+ target = DeliveryTarget.parse("discord:99887766")
+
+ result = await router._deliver_to_platform(
+ target, "*(silent)*", metadata={"job_id": "92e639af907f"},
+ )
+
+ assert len(adapter.calls) == 1
+ assert adapter.calls[0]["content"] == "*(silent)*"
+ assert result.get("filtered") is None
+ assert result.get("delivered") is not False
+
+
+@pytest.mark.asyncio
+async def test_non_cron_metadata_still_filters(tmp_path, monkeypatch):
+ """The exemption keys on job_id alone — everything else is unchanged."""
+ monkeypatch.setattr("gateway.delivery.get_hermes_home", lambda: tmp_path)
+ monkeypatch.delenv("HERMES_FILTER_SILENCE_NARRATION", raising=False)
+ adapter = RecordingAdapter()
+ router = DeliveryRouter(GatewayConfig(), adapters={Platform.DISCORD: adapter})
+ target = DeliveryTarget.parse("discord:99887766")
+
+ result = await router._deliver_to_platform(
+ target, "*(silent)*", metadata={"thread_id": "42", "user_id": "u1"},
+ )
+
+ assert adapter.calls == []
+ assert result["filtered"] == "silence_narration"
+
+
# --- Config round-trip ------------------------------------------------------
diff --git a/tests/gateway/test_discord_slash_commands.py b/tests/gateway/test_discord_slash_commands.py
index e3e5a39ff5..4f602a8768 100644
--- a/tests/gateway/test_discord_slash_commands.py
+++ b/tests/gateway/test_discord_slash_commands.py
@@ -602,3 +602,43 @@ def test_register_skill_command_payload_fits_discord_8kb_limit(adapter):
)
+
+
+# ------------------------------------------------------------------
+# _build_slash_event — guild/parent ids reach profile_routes (#69178, #91633)
+# ------------------------------------------------------------------
+
+
+def test_build_slash_event_routes_guild_profile_like_messages(adapter, monkeypatch):
+ """A guild-keyed profile route must match a native slash command exactly
+ as it matches a regular message: build_source needs guild_id (and the
+ thread's parent_chat_id) or the route never fires and /new resets the
+ default profile's session instead of the routed one."""
+ from gateway import run as gateway_run
+ from gateway.config import GatewayConfig
+ from gateway.profile_routing import ProfileRoute
+
+ runner = object.__new__(gateway_run.GatewayRunner)
+ runner.config = GatewayConfig(
+ multiplex_profiles=True,
+ profile_routes=[ProfileRoute(name="work", profile="work", platform="discord", guild_id="1")],
+ )
+ monkeypatch.setattr(gateway_run, "_multiplex_profile_homes", lambda _cfg: [("work", None)])
+ adapter.gateway_runner = runner
+ user = SimpleNamespace(display_name="Jezza", id=42)
+
+ channel_event = adapter._build_slash_event(
+ SimpleNamespace(channel=SimpleNamespace(id=200, name="general", guild=SimpleNamespace(id=1, name="G"), topic=None),
+ channel_id=200, guild_id=1, user=user),
+ "/new",
+ )
+ thread_event = adapter._build_slash_event(
+ SimpleNamespace(channel=_FakeThreadChannel(channel_id=555), channel_id=555, guild_id=None, user=user),
+ "/status",
+ )
+
+ assert channel_event.source.guild_id == "1"
+ assert channel_event.source.profile == "work"
+ assert thread_event.source.guild_id == "1"
+ assert thread_event.source.parent_chat_id == "100"
+ assert thread_event.source.profile == "work"
diff --git a/tests/gateway/test_email_robustness.py b/tests/gateway/test_email_robustness.py
index c1266196c6..5df979fb78 100644
--- a/tests/gateway/test_email_robustness.py
+++ b/tests/gateway/test_email_robustness.py
@@ -77,5 +77,39 @@ class TestMessageIdDomain(unittest.TestCase):
self.assertEqual(adapter._message_id_domain(), "localhost")
+class TestTransportSecurity(unittest.TestCase):
+ """platforms.email.extra.imap_security / smtp_security select the transport (#99641)."""
+
+ def _adapter(self, **extra):
+ from gateway.config import PlatformConfig
+
+ with patch.dict(os.environ, {
+ "EMAIL_ADDRESS": "hermes@test.com", "EMAIL_PASSWORD": "secret",
+ "EMAIL_IMAP_HOST": "127.0.0.1", "EMAIL_IMAP_PORT": "1143",
+ "EMAIL_SMTP_HOST": "127.0.0.1", "EMAIL_SMTP_PORT": "1025",
+ }, clear=True):
+ from plugins.platforms.email.adapter import EmailAdapter
+
+ return EmailAdapter(PlatformConfig(enabled=True, extra=extra))
+
+ def test_starttls_builds_plain_imap_then_upgrades(self):
+ adapter = self._adapter(imap_security="starttls", imap_tls_verify=False)
+ imap = MagicMock()
+ with patch("imaplib.IMAP4", return_value=imap) as imap_cls, \
+ patch("imaplib.IMAP4_SSL") as imap_ssl_cls:
+ self.assertIs(adapter._connect_imap(), imap)
+ imap_cls.assert_called_once_with("127.0.0.1", 1143, timeout=30)
+ imap_ssl_cls.assert_not_called()
+ imap.starttls.assert_called_once()
+
+ def test_unknown_mode_falls_back_to_secure_default(self):
+ adapter = self._adapter(imap_security="bogus", smtp_security="bogus")
+ self.assertEqual(adapter._imap_security, "tls")
+ self.assertEqual(adapter._smtp_security, "starttls") # port 1025 != 465
+ # verification stays ON unless explicitly opted out
+ self.assertTrue(adapter._imap_tls_verify)
+ self.assertTrue(adapter._smtp_tls_verify)
+
+
if __name__ == "__main__":
unittest.main()
diff --git a/tests/gateway/test_env_override_explicit_disable_48820.py b/tests/gateway/test_env_override_explicit_disable_48820.py
new file mode 100644
index 0000000000..1d8452c5a0
--- /dev/null
+++ b/tests/gateway/test_env_override_explicit_disable_48820.py
@@ -0,0 +1,199 @@
+"""Regression tests for #48820 Bug 2: an explicit ``platforms..enabled: false``
+in config.yaml must survive ``_apply_env_overrides`` when that platform's
+credentials are present in the environment.
+
+Before the fix, twelve credential-presence branches (weixin, whatsapp_cloud,
+homeassistant, email, sms, dingtalk, feishu, wecom, wecom_callback, bluebubbles,
+qqbot, yuanbao) force-set ``enabled = True`` unconditionally, while Telegram /
+Discord / Slack routed through ``_enable_from_env`` and honored the
+``_enabled_explicit`` marker. These tests drive the real ``load_gateway_config``
+against a temp HERMES_HOME — real YAML I/O, no mocks of the code under test.
+"""
+
+import logging
+
+import pytest
+
+from gateway import config as gateway_config
+from gateway.config import Platform, load_gateway_config
+
+
+# platform -> env credentials that trigger its env-enable branch
+CRED_ENV = {
+ "weixin": {
+ "WEIXIN_TOKEN": "wx_9f8e7d6c5b4a3f2e1d0c9b8a7f6e5d4c3b2a1f0e",
+ "WEIXIN_ACCOUNT_ID": "acct_12345",
+ },
+ "whatsapp_cloud": {
+ "WHATSAPP_CLOUD_PHONE_NUMBER_ID": "1234567890",
+ "WHATSAPP_CLOUD_ACCESS_TOKEN": "EAAB-test-access-token",
+ },
+ "homeassistant": {"HASS_TOKEN": "hass-long-lived-token"},
+ "email": {
+ "EMAIL_ADDRESS": "bot@example.com",
+ "EMAIL_PASSWORD": "app-password",
+ "EMAIL_IMAP_HOST": "imap.example.com",
+ "EMAIL_SMTP_HOST": "smtp.example.com",
+ },
+ "sms": {"TWILIO_ACCOUNT_SID": "ACxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx"},
+ "dingtalk": {"DINGTALK_CLIENT_ID": "ding-id", "DINGTALK_CLIENT_SECRET": "ding-secret"},
+ "feishu": {"FEISHU_APP_ID": "cli_feishu", "FEISHU_APP_SECRET": "feishu-secret"},
+ "wecom": {"WECOM_BOT_ID": "wecom-bot", "WECOM_SECRET": "wecom-secret"},
+ "wecom_callback": {
+ "WECOM_CALLBACK_CORP_ID": "corp-id",
+ "WECOM_CALLBACK_CORP_SECRET": "corp-secret",
+ },
+ "bluebubbles": {
+ "BLUEBUBBLES_SERVER_URL": "http://127.0.0.1:1234",
+ "BLUEBUBBLES_PASSWORD": "bb-password",
+ },
+ "qqbot": {"QQ_APP_ID": "qq-app", "QQ_CLIENT_SECRET": "qq-secret"},
+ "yuanbao": {"YUANBAO_APP_ID": "yb-app", "YUANBAO_APP_SECRET": "yb-secret"},
+ # control: the pattern that always honored the explicit disable
+ "telegram": {"TELEGRAM_BOT_TOKEN": "123456:ABC-DEF1234ghIkl-zyx57W2v1u123ew11"},
+}
+
+_PLATFORM_ENV_PREFIXES = (
+ "TELEGRAM_", "DISCORD_", "SLACK_", "WEIXIN_", "WHATSAPP_", "HASS_", "EMAIL_",
+ "TWILIO_", "DINGTALK_", "FEISHU_", "WECOM_", "BLUEBUBBLES_", "QQ_", "QQBOT_",
+ "YUANBAO_", "GATEWAY_RELAY", "SIGNAL_", "MATTERMOST_", "MATRIX_",
+)
+
+
+def _isolate(monkeypatch, tmp_path, env):
+ import os
+
+ for key in list(os.environ):
+ if key.startswith(_PLATFORM_ENV_PREFIXES):
+ monkeypatch.delenv(key, raising=False)
+ hermes_home = tmp_path / ".hermes"
+ hermes_home.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(hermes_home))
+ for k, v in env.items():
+ monkeypatch.setenv(k, v)
+ return hermes_home
+
+
+@pytest.mark.parametrize("platform", sorted(CRED_ENV))
+def test_yaml_explicit_disable_survives_env_credentials(platform, tmp_path, monkeypatch):
+ """``platforms..enabled: false`` + credentials in env -> stays disabled."""
+ hermes_home = _isolate(monkeypatch, tmp_path, CRED_ENV[platform])
+ (hermes_home / "config.yaml").write_text(
+ f"platforms:\n {platform}:\n enabled: false\n", encoding="utf-8"
+ )
+
+ config = load_gateway_config()
+
+ cfg = config.platforms.get(Platform(platform))
+ assert cfg is not None
+ assert cfg.enabled is False, (
+ f"{platform}: env credentials re-enabled a platform the user explicitly "
+ "disabled in config.yaml (#48820 Bug 2)"
+ )
+
+
+@pytest.mark.parametrize("platform", sorted(CRED_ENV))
+def test_env_credentials_still_enable_without_yaml_opinion(platform, tmp_path, monkeypatch):
+ """No ``enabled`` key in YAML + credentials in env -> env-only setup still works."""
+ hermes_home = _isolate(monkeypatch, tmp_path, CRED_ENV[platform])
+ (hermes_home / "config.yaml").write_text("platforms: {}\n", encoding="utf-8")
+
+ config = load_gateway_config()
+
+ cfg = config.platforms.get(Platform(platform))
+ assert cfg is not None and cfg.enabled is True, (
+ f"{platform}: env-only configuration must still enable the platform"
+ )
+
+
+def test_env_credentials_still_populate_extra_when_yaml_disables(tmp_path, monkeypatch):
+ """The disable only gates ``enabled``; credentials are still wired through
+ (mirrors the Slack/API-server contract so send-only tooling keeps working)."""
+ hermes_home = _isolate(monkeypatch, tmp_path, CRED_ENV["weixin"])
+ (hermes_home / "config.yaml").write_text(
+ "platforms:\n weixin:\n enabled: false\n", encoding="utf-8"
+ )
+
+ config = load_gateway_config()
+
+ cfg = config.platforms[Platform.WEIXIN]
+ assert cfg.enabled is False
+ assert cfg.token == CRED_ENV["weixin"]["WEIXIN_TOKEN"]
+ assert cfg.extra.get("account_id") == "acct_12345"
+ # marker never leaks out of config load
+ assert "_enabled_explicit" not in cfg.extra
+
+
+@pytest.fixture()
+def _fresh_warn_dedup(monkeypatch):
+ """The explicit-disable notice is one-time per process; start each test clean."""
+ monkeypatch.setattr(gateway_config, "_EXPLICIT_DISABLE_WARNED", set())
+
+
+@pytest.mark.usefixtures("_fresh_warn_dedup")
+@pytest.mark.parametrize("platform", sorted(CRED_ENV))
+def test_explicit_disable_with_env_credentials_warns_once(platform, tmp_path, monkeypatch, caplog):
+ """Users who relied on 'creds in .env = platform on' must be told why it went
+ dark: one WARNING naming the platform, the winning config key, and the env
+ credential(s) — emitted once per process, not on every config reload."""
+ hermes_home = _isolate(monkeypatch, tmp_path, CRED_ENV[platform])
+ (hermes_home / "config.yaml").write_text(
+ f"platforms:\n {platform}:\n enabled: false\n", encoding="utf-8"
+ )
+
+ with caplog.at_level(logging.WARNING, logger="gateway.config"):
+ load_gateway_config()
+ load_gateway_config() # reload: must not repeat
+
+ hits = [
+ r for r in caplog.records
+ if r.levelno == logging.WARNING and f"platforms.{platform}.enabled: false" in r.getMessage()
+ ]
+ assert len(hits) == 1, [r.getMessage() for r in caplog.records]
+ msg = hits[0].getMessage()
+ assert f"Platform '{platform}'" in msg
+ for env_name in CRED_ENV[platform]:
+ assert env_name in msg
+ assert f"platforms.{platform}.enabled: true" in msg # the remedy
+
+
+@pytest.mark.usefixtures("_fresh_warn_dedup")
+def test_no_warning_when_yaml_has_no_opinion_or_is_enabled(tmp_path, monkeypatch, caplog):
+ hermes_home = _isolate(monkeypatch, tmp_path, {**CRED_ENV["weixin"], **CRED_ENV["homeassistant"]})
+ (hermes_home / "config.yaml").write_text(
+ "platforms:\n homeassistant:\n enabled: true\n", encoding="utf-8"
+ )
+
+ with caplog.at_level(logging.WARNING, logger="gateway.config"):
+ config = load_gateway_config()
+
+ assert config.platforms[Platform.WEIXIN].enabled is True
+ assert config.platforms[Platform.HOMEASSISTANT].enabled is True
+ assert not [r for r in caplog.records if "explicitly disabled" in r.getMessage()]
+
+
+@pytest.mark.usefixtures("_fresh_warn_dedup")
+def test_no_warning_when_disabled_and_no_env_credentials(tmp_path, monkeypatch, caplog):
+ """The notice is about credentials being IGNORED; a plain disable is silent."""
+ hermes_home = _isolate(monkeypatch, tmp_path, {})
+ (hermes_home / "config.yaml").write_text(
+ "platforms:\n weixin:\n enabled: false\n", encoding="utf-8"
+ )
+
+ with caplog.at_level(logging.WARNING, logger="gateway.config"):
+ config = load_gateway_config()
+
+ assert config.platforms[Platform.WEIXIN].enabled is False
+ assert not [r for r in caplog.records if "explicitly disabled" in r.getMessage()]
+
+
+def test_every_env_enable_branch_is_named_for_the_warning():
+ """Each platform routed through ``_enable_from_env`` needs a credential
+ entry so the WARNING can name what is being ignored."""
+ import inspect, re
+
+ src = inspect.getsource(gateway_config._apply_env_overrides)
+ routed = {Platform[name] for name in re.findall(r"_enable_from_env\(Platform\.([A-Z_]+)\)", src)}
+ routed.add(Platform.SLACK) # Slack has its own inline copy of the logic
+ missing = {p.value for p in routed} - {p.value for p in gateway_config._ENV_ENABLE_CREDENTIALS}
+ assert not missing, f"platforms without a credential entry for the explicit-disable warning: {missing}"
diff --git a/tests/gateway/test_fast_command.py b/tests/gateway/test_fast_command.py
index c714b76e84..b8792ecce4 100644
--- a/tests/gateway/test_fast_command.py
+++ b/tests/gateway/test_fast_command.py
@@ -109,8 +109,8 @@ def test_turn_route_injects_priority_processing_without_changing_runtime():
runner._service_tier = "priority"
runtime_kwargs = {
"api_key": "***",
- "base_url": "https://openrouter.ai/api/v1",
- "provider": "openrouter",
+ "base_url": "https://api.openai.com/v1",
+ "provider": "openai",
"api_mode": "chat_completions",
"command": None,
"args": [],
@@ -119,10 +119,15 @@ def test_turn_route_injects_priority_processing_without_changing_runtime():
route = gateway_run.GatewayRunner._resolve_turn_agent_config(runner, "hi", "gpt-5.4", runtime_kwargs)
- assert route["runtime"]["provider"] == "openrouter"
+ assert route["runtime"]["provider"] == "openai"
assert route["runtime"]["api_mode"] == "chat_completions"
assert route["request_overrides"] == {"service_tier": "priority"}
+ # Proxied routes never receive the param (OpenRouter strips it / others 400).
+ runtime_kwargs.update(base_url="https://openrouter.ai/api/v1", provider="openrouter")
+ route = gateway_run.GatewayRunner._resolve_turn_agent_config(runner, "hi", "gpt-5.4", runtime_kwargs)
+ assert route["request_overrides"] == {}
+
@pytest.mark.asyncio
async def test_handle_fast_command_global_flag_persists_config(monkeypatch, tmp_path):
diff --git a/tests/gateway/test_feishu_bot_admission.py b/tests/gateway/test_feishu_bot_admission.py
index 09157c757e..b396e90d29 100644
--- a/tests/gateway/test_feishu_bot_admission.py
+++ b/tests/gateway/test_feishu_bot_admission.py
@@ -566,3 +566,77 @@ def test_handle_message_event_data_forwards_sender_when_admitted():
assert captured.get("sender_id") is sender.sender_id
assert captured.get("is_bot") is True
assert captured.get("message_id") == "om_bot_ok"
+
+
+# --- Profile-scoped admission config (#86905) -------------------------------
+
+
+def test_dm_admission_config_resolves_from_profile_scope_under_multiplex(tmp_path, monkeypatch):
+ """os.environ holds the DEFAULT profile's admission view; a secondary
+ profile's .env must govern its own adapter — Feishu open_ids are
+ app-scoped, so the default allow-list can never match the role app's
+ senders, and the role profile's allow-all flag must be honored."""
+ import agent.secret_scope as ss
+ from plugins.platforms.feishu.adapter import FeishuAdapter
+
+ monkeypatch.setenv("FEISHU_APP_ID", "cli_default")
+ monkeypatch.setenv("FEISHU_APP_SECRET", "secret_default")
+ monkeypatch.setenv("FEISHU_ALLOWED_USERS", "ou_default_view")
+ monkeypatch.setenv("FEISHU_ALLOW_BOTS", "all")
+ monkeypatch.delenv("GATEWAY_ALLOW_ALL_USERS", raising=False)
+ monkeypatch.delenv("FEISHU_ALLOW_ALL_USERS", raising=False)
+ (tmp_path / ".env").write_text(
+ "FEISHU_APP_ID=cli_role\nFEISHU_APP_SECRET=secret_role\n"
+ "FEISHU_ALLOWED_USERS=ou_role_view\n",
+ encoding="utf-8",
+ )
+
+ ss.set_multiplex_active(True)
+ tok = ss.set_secret_scope(ss.build_profile_secret_scope(tmp_path))
+ try:
+ settings = FeishuAdapter._load_settings(extra={})
+ finally:
+ ss.reset_secret_scope(tok)
+ (tmp_path / ".env").write_text(
+ "FEISHU_APP_ID=cli_role\nFEISHU_APP_SECRET=secret_role\nGATEWAY_ALLOW_ALL_USERS=true\n",
+ encoding="utf-8",
+ )
+ tok = ss.set_secret_scope(ss.build_profile_secret_scope(tmp_path))
+ try:
+ allow_all = FeishuAdapter._load_settings(extra={})
+ finally:
+ ss.reset_secret_scope(tok)
+ ss.set_multiplex_active(False)
+
+ assert settings.app_id == "cli_role"
+ assert settings.allowed_group_users == frozenset({"ou_role_view"})
+ assert settings.allow_bots == "none" # default's "all" must not leak in
+ assert settings.allow_all_dm is False
+
+ # _admit runs on the WS thread with no scope: the snapshot must carry.
+ adapter = object.__new__(FeishuAdapter)
+ adapter._apply_settings(settings)
+ assert adapter._admit(make_sender(open_id="ou_role_view"), make_message(chat_type="p2p")) is None
+ assert adapter._admit(make_sender(open_id="ou_default_view"), make_message(chat_type="p2p")) == "dm_policy_rejected"
+
+ assert allow_all.allow_all_dm is True
+ adapter = object.__new__(FeishuAdapter)
+ adapter._apply_settings(allow_all)
+ assert adapter._admit(make_sender(open_id="ou_anyone"), make_message(chat_type="p2p")) is None
+
+
+def test_dm_admission_config_falls_back_to_os_environ_when_unscoped(monkeypatch):
+ """Single-profile behavior unchanged: process env still configures DMs."""
+ from plugins.platforms.feishu.adapter import FeishuAdapter
+
+ monkeypatch.setenv("FEISHU_APP_ID", "cli_test")
+ monkeypatch.setenv("FEISHU_APP_SECRET", "secret_test")
+ monkeypatch.setenv("GATEWAY_ALLOW_ALL_USERS", "true")
+ monkeypatch.setenv("FEISHU_ALLOWED_USERS", "ou_a,ou_b")
+
+ settings = FeishuAdapter._load_settings(extra={})
+ assert settings.allow_all_dm is True
+ assert settings.allowed_group_users == frozenset({"ou_a", "ou_b"})
+ adapter = object.__new__(FeishuAdapter)
+ adapter._apply_settings(settings)
+ assert adapter._admit(make_sender(open_id="ou_anyone"), make_message(chat_type="p2p")) is None
diff --git a/tests/gateway/test_feishu_ws_multiplex_isolation.py b/tests/gateway/test_feishu_ws_multiplex_isolation.py
new file mode 100644
index 0000000000..ca7d514cc3
--- /dev/null
+++ b/tests/gateway/test_feishu_ws_multiplex_isolation.py
@@ -0,0 +1,167 @@
+"""Multiplex isolation for the lark_oapi WS client (issue #73779).
+
+``lark_oapi.ws.client`` keeps the loop used by ``Client.start()`` in a
+module-level global and Hermes monkey-patches ``websockets.connect`` on the
+shared module. With N profile WS threads the globals were last-write-wins:
+"Future attached to a different loop" crashes or a client bound to a
+sibling's loop that never hears anything again.
+"""
+
+import asyncio
+import sys
+import threading
+import types
+from types import SimpleNamespace
+from unittest.mock import MagicMock
+
+from plugins.platforms.feishu import adapter as feishu_adapter
+
+
+def _inject_fake_lark_module(monkeypatch, connect=None):
+ """Make ``import lark_oapi.ws.client`` resolve to a module with the SDK's
+ global layout (``loop`` + ``websockets.connect``)."""
+ if connect is None:
+ connect = MagicMock(name="real-connect")
+ lark = types.ModuleType("lark_oapi")
+ lark_ws = types.ModuleType("lark_oapi.ws")
+ client_mod = types.ModuleType("lark_oapi.ws.client")
+ client_mod.loop = SimpleNamespace(name="sdk-default-loop")
+ client_mod.websockets = SimpleNamespace(connect=connect)
+ lark.ws = lark_ws
+ lark_ws.client = client_mod
+ monkeypatch.setitem(sys.modules, "lark_oapi", lark)
+ monkeypatch.setitem(sys.modules, "lark_oapi.ws", lark_ws)
+ monkeypatch.setitem(sys.modules, "lark_oapi.ws.client", client_mod)
+ monkeypatch.setattr(feishu_adapter, "_WS_ISOLATION_INSTALLED", False)
+ return client_mod
+
+
+def _adapter_stub(**overrides):
+ stub = SimpleNamespace(
+ _ws_thread_loop=None,
+ _ws_reconnect_nonce=None,
+ _ws_reconnect_interval=None,
+ _ws_ping_interval=None,
+ _ws_ping_timeout=None,
+ )
+ for key, value in overrides.items():
+ setattr(stub, key, value)
+ return stub
+
+
+def test_two_concurrent_clients_each_use_their_own_loop_and_overrides(monkeypatch):
+ """Two profiles start() concurrently through the module global: each must
+ run on its own loop, and websockets.connect must receive only the
+ calling profile's ping overrides. On main both are last-write-wins."""
+ real_connect = MagicMock(name="real-connect")
+ client_mod = _inject_fake_lark_module(monkeypatch, connect=real_connect)
+
+ results = {}
+ barrier = threading.Barrier(2)
+
+ class FakeClient:
+ def __init__(self, name):
+ self._name = name
+
+ def start(self):
+ barrier.wait(timeout=10) # both threads past the global "assign"
+
+ async def probe():
+ await asyncio.sleep(0.02)
+ return id(asyncio.get_running_loop())
+
+ results[self._name] = client_mod.loop.run_until_complete(probe())
+ client_mod.websockets.connect(f"wss://{self._name}")
+
+ pings = {"p0": 10, "p1": 20}
+
+ def run(name):
+ feishu_adapter._run_official_feishu_ws_client(
+ FakeClient(name), _adapter_stub(_ws_ping_interval=pings[name])
+ )
+
+ threads = [threading.Thread(target=run, args=(f"p{i}",)) for i in range(2)]
+ for t in threads:
+ t.start()
+ for t in threads:
+ t.join(timeout=15)
+ assert not t.is_alive()
+
+ assert results["p0"] != results["p1"]
+ calls = {c.args[0]: c.kwargs for c in real_connect.call_args_list}
+ assert calls == {"wss://p0": {"ping_interval": 10}, "wss://p1": {"ping_interval": 20}}
+ # Thread-local registrations are cleared for the pooled executor thread.
+ assert getattr(feishu_adapter._ws_isolation_state, "loop", None) is None
+ assert getattr(feishu_adapter._ws_isolation_state, "connect_kwargs", None) is None
+
+
+def _supervisor_stub():
+ stub = SimpleNamespace(
+ _running=True,
+ _ws_future=None,
+ _ws_client=object(),
+ _ws_restart_backoff=0.01,
+ connect_calls=0,
+ connect_should_fail=0,
+ )
+
+ async def _connect_websocket():
+ stub.connect_calls += 1
+ if stub.connect_should_fail > 0:
+ stub.connect_should_fail -= 1
+ raise RuntimeError("simulated restart failure")
+ fut = asyncio.get_running_loop().create_future()
+ fut.set_result(None) # new thread dies immediately too
+ stub._ws_future = fut
+
+ stub._connect_websocket = _connect_websocket
+ return stub
+
+
+def test_supervisor_restarts_a_dead_ws_thread_with_backoff():
+ """A dead WS thread used to leave the profile silently deaf (the future
+ was awaited only by disconnect()). The supervisor must rebuild the client
+ and survive a failed restart without hot-looping."""
+
+ async def scenario():
+ stub = _supervisor_stub()
+ stub.connect_should_fail = 1
+ fut = asyncio.get_running_loop().create_future()
+ fut.set_result(None) # the WS "thread" is already dead
+ stub._ws_future = fut
+
+ task = asyncio.ensure_future(
+ feishu_adapter.FeishuAdapter._supervise_websocket_thread(stub)
+ )
+ for _ in range(300):
+ await asyncio.sleep(0.01)
+ if stub.connect_calls >= 2:
+ break
+ task.cancel()
+ try:
+ await task
+ except asyncio.CancelledError:
+ pass
+ return stub.connect_calls
+
+ assert asyncio.run(scenario()) == 2 # failed restart, then a successful one
+
+
+def test_supervisor_stops_when_disconnect_nils_the_client():
+ async def scenario():
+ stub = _supervisor_stub()
+ fut = asyncio.get_running_loop().create_future() # thread "alive"
+ stub._ws_future = fut
+
+ task = asyncio.ensure_future(
+ feishu_adapter.FeishuAdapter._supervise_websocket_thread(stub)
+ )
+ await asyncio.sleep(0.01)
+ stub._ws_client = None # deliberate disconnect ...
+ fut.set_result(None) # ... then the thread exits
+ await asyncio.wait_for(asyncio.shield(task), timeout=2.0)
+ return stub, task
+
+ stub, task = asyncio.run(scenario())
+ assert task.done()
+ assert stub.connect_calls == 0
diff --git a/tests/gateway/test_fifo_overflow_rescue.py b/tests/gateway/test_fifo_overflow_rescue.py
new file mode 100644
index 0000000000..e1f3efd110
--- /dev/null
+++ b/tests/gateway/test_fifo_overflow_rescue.py
@@ -0,0 +1,159 @@
+"""Regression tests for #99882: FIFO overflow orphan rescue.
+
+When a follow-up is demoted to /queue during compression-in-flight,
+it lands in SessionState.conversation.queued_events (overflow) with
+the current turn's event occupying adapter._pending_messages[session_key]
+(slot). After the slot's turn completes, _promote_queued_event moves
+the overflow head into the slot. When that drain never runs — the
+busy window ended through an exit that skipped the promotion site
+(/stop, turn exception, generation bump) — the overflow is silently
+orphaned: never dispatched, never persisted, never logged.
+
+The rescue in GatewayRunner._rescue_orphaned_overflow pops the oldest
+orphan for the caller to run as the current turn and stages the next
+orphan in the slot, so FIFO order (#28503) holds and nothing runs twice.
+"""
+
+from unittest.mock import MagicMock
+
+from gateway.platforms.base import (
+ BasePlatformAdapter,
+ MessageEvent,
+ MessageType,
+ Platform,
+ PlatformConfig,
+)
+from gateway.run import GatewayRunner
+
+
+class _StubAdapter(BasePlatformAdapter):
+ def __init__(self):
+ super().__init__(PlatformConfig(enabled=True, token="test"), Platform.TELEGRAM)
+
+ async def connect(self, *, is_reconnect: bool = False) -> bool:
+ return True
+
+ async def disconnect(self) -> None:
+ self._mark_disconnected()
+
+ async def send(self, chat_id, content, reply_to=None, metadata=None):
+ from gateway.platforms.base import SendResult
+
+ return SendResult(success=True, message_id="msg-1")
+
+ async def get_chat_info(self, chat_id):
+ return {"id": chat_id, "type": "dm"}
+
+
+def _text_event(text: str, msg_id: str) -> MessageEvent:
+ return MessageEvent(
+ text=text,
+ message_type=MessageType.TEXT,
+ source=MagicMock(chat_id="123", platform=Platform.TELEGRAM, profile=None),
+ message_id=msg_id,
+ )
+
+
+def _runner() -> GatewayRunner:
+ runner = GatewayRunner.__new__(GatewayRunner)
+ runner._queued_events = {}
+ return runner
+
+
+class TestRescueOrphanedOverflow:
+ def test_single_orphan_is_returned_and_removed_from_both_stores(self):
+ runner = _runner()
+ adapter = _StubAdapter()
+ session_key = "telegram:user:1"
+ runner._session_state(session_key).conversation.queued_events.append(
+ _text_event("orphan-1", "o1")
+ )
+ assert session_key not in adapter._pending_messages
+
+ rescued = runner._rescue_orphaned_overflow(session_key, adapter)
+
+ assert rescued is not None and rescued.text == "orphan-1"
+ # The rescued event runs as the current turn, so it must NOT also
+ # sit in the slot — the post-turn drain would run it a second time.
+ assert session_key not in adapter._pending_messages
+ assert runner._session_state(session_key).conversation.queued_events == []
+
+ def test_two_orphans_return_oldest_and_stage_next_in_slot(self):
+ runner = _runner()
+ adapter = _StubAdapter()
+ session_key = "telegram:user:1b"
+ runner._session_state(session_key).conversation.queued_events.extend(
+ [_text_event("orphan-1", "o1"), _text_event("orphan-2", "o2")]
+ )
+
+ rescued = runner._rescue_orphaned_overflow(session_key, adapter)
+
+ assert rescued is not None and rescued.text == "orphan-1"
+ # Slot now holds the NEXT orphan so the drain continues the chain.
+ assert adapter._pending_messages[session_key].text == "orphan-2"
+ assert runner._session_state(session_key).conversation.queued_events == []
+
+ def test_noop_when_slot_occupied(self):
+ runner = _runner()
+ adapter = _StubAdapter()
+ session_key = "telegram:user:2"
+ runner._session_state(session_key).conversation.queued_events.append(
+ _text_event("orphan", "o1")
+ )
+ adapter._pending_messages[session_key] = _text_event("busy-slot", "slot")
+
+ rescued = runner._rescue_orphaned_overflow(session_key, adapter)
+
+ assert rescued is None
+ assert adapter._pending_messages[session_key].text == "busy-slot"
+ assert len(runner._session_state(session_key).conversation.queued_events) == 1
+
+ def test_noop_when_no_overflow(self):
+ runner = _runner()
+ adapter = _StubAdapter()
+ session_key = "telegram:user:3"
+
+ rescued = runner._rescue_orphaned_overflow(session_key, adapter)
+
+ assert rescued is None
+ assert session_key not in adapter._pending_messages
+
+ def test_fifo_order_preserved_across_rescue_and_new_message(self):
+ """Oldest orphan runs first, new arrival last — FIFO (#28503).
+
+ Mirrors the idle-arrival call site: rescue → _enqueue_fifo(new).
+ """
+ runner = _runner()
+ adapter = _StubAdapter()
+ session_key = "telegram:user:4"
+ runner._session_state(session_key).conversation.queued_events.extend(
+ [_text_event("orphan-1", "o1"), _text_event("orphan-2", "o2")]
+ )
+
+ rescued = runner._rescue_orphaned_overflow(session_key, adapter)
+ assert rescued is not None and rescued.text == "orphan-1"
+ runner._enqueue_fifo(session_key, _text_event("new-msg", "new1"), adapter)
+
+ # Drain order after this turn: slot (orphan-2), then overflow (new-msg)
+ assert adapter._pending_messages[session_key].text == "orphan-2"
+ overflow_texts = [
+ e.text for e in runner._session_state(session_key).conversation.queued_events
+ ]
+ assert overflow_texts == ["new-msg"]
+
+ def test_single_orphan_then_new_message_lands_in_slot(self):
+ """With one orphan the slot is free after rescue, so the incoming
+ message must go to the slot (not overflow) or the drain never sees it."""
+ runner = _runner()
+ adapter = _StubAdapter()
+ session_key = "telegram:user:5"
+ runner._session_state(session_key).conversation.queued_events.append(
+ _text_event("orphan-1", "o1")
+ )
+
+ rescued = runner._rescue_orphaned_overflow(session_key, adapter)
+ assert rescued is not None and rescued.text == "orphan-1"
+ runner._enqueue_fifo(session_key, _text_event("new-msg", "new1"), adapter)
+
+ assert adapter._pending_messages[session_key].text == "new-msg"
+ assert runner._session_state(session_key).conversation.queued_events == []
diff --git a/tests/gateway/test_gateway_trust_env.py b/tests/gateway/test_gateway_trust_env.py
new file mode 100644
index 0000000000..78965ee66b
--- /dev/null
+++ b/tests/gateway/test_gateway_trust_env.py
@@ -0,0 +1,47 @@
+"""gateway.trust_env — one config key controls aiohttp proxy-env honoring at every adapter site (#48820)."""
+import re
+from pathlib import Path
+
+import pytest
+
+from gateway.platforms import base as gw_base
+
+REPO = Path(__file__).resolve().parents[2]
+_ADAPTER_FILES = sorted(
+ list((REPO / "gateway" / "platforms").rglob("*.py"))
+ + list((REPO / "plugins" / "platforms").rglob("*.py"))
+)
+
+
+def _write_config(tmp_path, monkeypatch, body: str) -> None:
+ # load_config caches on (path, mtime) — a fresh tmp HERMES_HOME per test is a fresh cache key.
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ (tmp_path / "config.yaml").write_text(body)
+
+
+@pytest.mark.parametrize(
+ "yaml_body, expected",
+ [("gateway:\n trust_env: false\n", False), ("gateway:\n trust_env: true\n", True), ("{}\n", True)],
+)
+def test_gateway_trust_env_reads_config(tmp_path, monkeypatch, yaml_body, expected):
+ """gateway.trust_env in config.yaml drives the shared helper; absent → True (default)."""
+ _write_config(tmp_path, monkeypatch, yaml_body)
+ assert gw_base.gateway_trust_env() is expected
+ # The generic-proxy discovery path is gated by the same knob; explicit per-platform vars are not.
+ monkeypatch.setenv("HTTPS_PROXY", "http://127.0.0.1:7890")
+ monkeypatch.delenv("NO_PROXY", raising=False)
+ monkeypatch.delenv("no_proxy", raising=False)
+ assert (gw_base.resolve_proxy_url() is not None) is expected
+ monkeypatch.setenv("X_PLATFORM_PROXY", "http://127.0.0.1:1080")
+ assert gw_base.resolve_proxy_url("X_PLATFORM_PROXY") == "http://127.0.0.1:1080"
+
+
+def test_no_bare_trust_env_literal_in_adapters():
+ """Every aiohttp session in gateway/ + plugins/platforms/ must go through gateway_trust_env()."""
+ bare = re.compile(r"trust_env\s*=\s*(True|False)\b")
+ offenders = []
+ for path in _ADAPTER_FILES:
+ for lineno, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
+ if bare.search(line) and "httpx" not in line:
+ offenders.append(f"{path.relative_to(REPO)}:{lineno}: {line.strip()}")
+ assert not offenders, "hard-coded aiohttp trust_env literal(s); use gateway_trust_env():\n" + "\n".join(offenders)
diff --git a/tests/gateway/test_goal_continuation_drain.py b/tests/gateway/test_goal_continuation_drain.py
index 662f541fc6..6a969c07e3 100644
--- a/tests/gateway/test_goal_continuation_drain.py
+++ b/tests/gateway/test_goal_continuation_drain.py
@@ -87,6 +87,11 @@ def hermes_home(tmp_path, monkeypatch):
from hermes_cli import goals
goals._DB_CACHE.clear()
+ # Pre-warm the SessionDB cache from this sync (non-loop) context so the
+ # async tests' GoalManager.set() never races the bounded loop-thread
+ # bootstrap window on loaded CI runners (goal silently not persisted →
+ # continuation never enqueued; flaked on main run 33455779041).
+ goals._get_session_db()
yield home
goals._DB_CACHE.clear()
diff --git a/tests/gateway/test_goal_max_turns_config.py b/tests/gateway/test_goal_max_turns_config.py
index 4e4f8657d9..3d3c81e3a0 100644
--- a/tests/gateway/test_goal_max_turns_config.py
+++ b/tests/gateway/test_goal_max_turns_config.py
@@ -59,6 +59,10 @@ async def test_gateway_goal_uses_goals_max_turns_from_full_config(tmp_path, monk
(home / "config.yaml").write_text("goals:\n max_turns: 7\n", encoding="utf-8")
monkeypatch.setenv("HERMES_HOME", str(home))
goals._DB_CACHE.clear()
+ # Pre-warm from sync context: the /goal handler runs on the event loop,
+ # where a cold cache only waits the bounded bootstrap window — under CI
+ # load the goal write can be dropped and the state assertion flakes.
+ goals._get_session_db()
runner = _make_runner()
diff --git a/tests/gateway/test_goal_resume_restart.py b/tests/gateway/test_goal_resume_restart.py
index 7b9be97ad3..2fcf1f34c5 100644
--- a/tests/gateway/test_goal_resume_restart.py
+++ b/tests/gateway/test_goal_resume_restart.py
@@ -46,6 +46,10 @@ def hermes_home(tmp_path, monkeypatch):
token = set_hermes_home_override(str(home))
goals._DB_CACHE.clear()
+ # Pre-warm the SessionDB cache from sync context so async GoalManager
+ # writes never race the bounded loop-thread bootstrap window on loaded
+ # CI runners (goal silently unpersisted; main run 33455779041).
+ goals._get_session_db()
yield home
try:
reset_hermes_home_override(token)
diff --git a/tests/gateway/test_google_chat.py b/tests/gateway/test_google_chat.py
index 19aa5163e6..d3a05ea00c 100644
--- a/tests/gateway/test_google_chat.py
+++ b/tests/gateway/test_google_chat.py
@@ -270,6 +270,61 @@ class TestEnvConfigLoading:
cfg = load_gateway_config()
assert _GC not in cfg.platforms
+ def test_multiplex_scoped_profile_never_borrows_process_env(
+ self, monkeypatch, tmp_path
+ ):
+ """Under multiplex a scoped profile sees ONLY its own Google Chat
+ settings, and the ADC branch fails closed instead of authenticating
+ as the default profile's service account (#73439)."""
+ from agent.secret_scope import (
+ build_profile_secret_scope,
+ set_multiplex_active,
+ set_secret_scope,
+ )
+
+ self._clean_env(monkeypatch)
+ monkeypatch.setenv("GOOGLE_CHAT_PROJECT_ID", "default-proj")
+ monkeypatch.setenv("GOOGLE_CHAT_SUBSCRIPTION_NAME", "default-sub")
+ monkeypatch.setenv("GOOGLE_APPLICATION_CREDENTIALS", "/secrets/default.json")
+ monkeypatch.setenv("GOOGLE_CHAT_BOOTSTRAP_SPACES", "spaces/DEFAULT")
+ profile_home = tmp_path / "beta"
+ profile_home.mkdir()
+ (profile_home / ".env").write_text(
+ "GOOGLE_CHAT_PROJECT_ID=beta-proj\nGOOGLE_CHAT_SUBSCRIPTION_NAME=beta-sub\n"
+ )
+ set_multiplex_active(True)
+ token = set_secret_scope(build_profile_secret_scope(profile_home))
+ try:
+ seed = _gc_mod._env_enablement() or {}
+ beta = GoogleChatAdapter(
+ PlatformConfig(enabled=True, extra={"project_id": "beta-proj", "subscription_name": "beta-sub"})
+ )
+ with pytest.raises(ValueError, match="ADC skipped"):
+ beta._load_sa_credentials()
+ finally:
+ from agent.secret_scope import reset_secret_scope
+
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+ assert seed["project_id"] == "beta-proj"
+ assert "service_account_json" not in seed
+ assert beta._bootstrap_spaces == ""
+
+ def test_multiplex_default_profile_constructs_unscoped(self, monkeypatch):
+ """The default profile's adapter is built OUTSIDE any scope while
+ multiplex is active (gateway startup/reconnect); it must keep reading
+ its own process env instead of raising UnscopedSecretError."""
+ from agent.secret_scope import set_multiplex_active
+
+ self._clean_env(monkeypatch)
+ monkeypatch.setenv("GOOGLE_CHAT_BOOTSTRAP_SPACES", "spaces/DEFAULT")
+ set_multiplex_active(True)
+ try:
+ default = GoogleChatAdapter(_base_config())
+ finally:
+ set_multiplex_active(False)
+ assert default._bootstrap_spaces == "spaces/DEFAULT"
+
# ===========================================================================
# Pure helpers
diff --git a/tests/gateway/test_handoff_secondary_profile_adapter.py b/tests/gateway/test_handoff_secondary_profile_adapter.py
index e637b293c1..b65e75bfd6 100644
--- a/tests/gateway/test_handoff_secondary_profile_adapter.py
+++ b/tests/gateway/test_handoff_secondary_profile_adapter.py
@@ -167,6 +167,34 @@ async def test_default_profile_handoff_keeps_primary_adapter(monkeypatch):
assert used["home_chat_id"] == "1111"
+@pytest.mark.asyncio
+async def test_secondary_profile_config_load_failure_fails_closed(monkeypatch):
+ """A secondary profile whose config cannot load must fail the handoff.
+
+ Falling back to the primary's config delivers through the right bot to
+ the WRONG chat (the primary's home channel) and reports completed.
+ """
+ runner, _ = _make_multiplex_runner()
+ used = {}
+
+ def _boom():
+ raise RuntimeError("config.yaml exploded")
+
+ monkeypatch.setattr(
+ "gateway.run.resolve_delivery_transport", _spy_transport_factory(used),
+ )
+ monkeypatch.setattr("gateway.run.load_gateway_config", _boom)
+
+ with pytest.raises(RuntimeError, match="could not load config"):
+ await runner._process_handoff(
+ {"id": "cli-session", "title": "work", "handoff_platform": "telegram"},
+ profile_name="medicina",
+ )
+ assert used == {}, (
+ "nothing may be delivered when the profile config fails to load"
+ )
+
+
@pytest.mark.asyncio
async def test_secondary_profile_without_live_adapters_fails_loudly(monkeypatch):
"""Never silently fall back to the primary's bot — that ships to the wrong chat.
diff --git a/tests/gateway/test_handoff_watcher_multiprofile.py b/tests/gateway/test_handoff_watcher_multiprofile.py
index 08bcf95710..bea32bc7bc 100644
--- a/tests/gateway/test_handoff_watcher_multiprofile.py
+++ b/tests/gateway/test_handoff_watcher_multiprofile.py
@@ -15,6 +15,7 @@ These tests pin the two halves of the fix:
"""
import asyncio
+import threading
import types
from pathlib import Path
@@ -111,14 +112,14 @@ async def test_watcher_enters_profile_scope_for_each_home(monkeypatch):
def __init__(self, home):
self.home = home
- def __enter__(self):
+ async def __aenter__(self):
entered.append(self.home)
return self
- def __exit__(self, *exc):
+ async def __aexit__(self, *exc):
return False
- monkeypatch.setattr(run, "_profile_runtime_scope", _SpyScope)
+ monkeypatch.setattr(run, "_async_profile_runtime_scope", _SpyScope)
async def _no_sleep(_seconds):
return None
@@ -159,6 +160,68 @@ async def test_watcher_enters_profile_scope_for_each_home(monkeypatch):
assert db.polls == 3, "root + both profiles polled once each per tick"
+@pytest.mark.asyncio
+async def test_slow_profile_secret_load_does_not_block_event_loop(monkeypatch, tmp_path):
+ """A slow profile ``.env`` read must not stall unrelated loop work."""
+ profile_home = tmp_path / "profiles" / "slow"
+ profile_home.mkdir(parents=True)
+ monkeypatch.setattr(
+ run,
+ "_handoff_watch_scopes",
+ lambda _runner: [(None, None), ("slow", profile_home)],
+ )
+
+ from agent import secret_scope
+
+ load_started = threading.Event()
+ ticker_progressed = threading.Event()
+ ticker_progressed_while_loading = []
+
+ def _slow_build(_home):
+ load_started.set()
+ ticker_progressed_while_loading.append(
+ ticker_progressed.wait(timeout=2)
+ )
+ return {}
+
+ monkeypatch.setattr(secret_scope, "build_profile_secret_scope", _slow_build)
+
+ class _DB:
+ async def list_pending_handoffs(self):
+ return []
+
+ fake = types.SimpleNamespace(
+ _session_db=_DB(),
+ _running=False,
+ )
+
+ async def _process_handoff(_row, _profile_name=None):
+ return None
+
+ fake._process_handoff = _process_handoff
+
+ real_sleep = asyncio.sleep
+
+ async def _skip_initial_delay(seconds):
+ await real_sleep(0 if seconds == 5 else seconds)
+
+ monkeypatch.setattr(run.asyncio, "sleep", _skip_initial_delay)
+ async def _ticker():
+ assert await asyncio.to_thread(load_started.wait, 5)
+ ticker_progressed.set()
+
+ watcher = asyncio.create_task(
+ run.GatewayRunner._handoff_watcher(fake, interval=0.0)
+ )
+ ticker = asyncio.create_task(_ticker())
+ await asyncio.wait_for(asyncio.gather(watcher, ticker), timeout=5)
+
+ assert ticker_progressed_while_loading == [True], (
+ "profile secret loading blocked the asyncio event loop until the "
+ "filesystem operation completed"
+ )
+
+
@pytest.mark.asyncio
async def test_each_scope_resolves_its_own_store_and_profile(monkeypatch):
"""The whole point: a DIFFERENT ``state.db`` per scope, and the profile
@@ -183,15 +246,15 @@ async def test_each_scope_resolves_its_own_store_and_profile(monkeypatch):
def __init__(self, home):
self.home = home
- def __enter__(self):
+ async def __aenter__(self):
active["home"] = self.home
return self
- def __exit__(self, *exc):
+ async def __aexit__(self, *exc):
active["home"] = None
return False
- monkeypatch.setattr(run, "_profile_runtime_scope", _SpyScope)
+ monkeypatch.setattr(run, "_async_profile_runtime_scope", _SpyScope)
async def _no_sleep(_seconds):
return None
diff --git a/tests/gateway/test_handoff_watcher_resilience.py b/tests/gateway/test_handoff_watcher_resilience.py
index cfe672b784..b7b4e25d73 100644
--- a/tests/gateway/test_handoff_watcher_resilience.py
+++ b/tests/gateway/test_handoff_watcher_resilience.py
@@ -219,13 +219,13 @@ async def test_reclaim_runs_per_profile_store(monkeypatch):
def __init__(self, home):
self.home = home
- def __enter__(self):
+ async def __aenter__(self):
return self
- def __exit__(self, *exc):
+ async def __aexit__(self, *exc):
return False
- monkeypatch.setattr(run, "_profile_runtime_scope", _Scope)
+ monkeypatch.setattr(run, "_async_profile_runtime_scope", _Scope)
async def _no_sleep(_seconds):
return None
diff --git a/tests/gateway/test_hosted_room_gateway_lifecycle.py b/tests/gateway/test_hosted_room_gateway_lifecycle.py
index fe692d41cc..679d7ac145 100644
--- a/tests/gateway/test_hosted_room_gateway_lifecycle.py
+++ b/tests/gateway/test_hosted_room_gateway_lifecycle.py
@@ -185,7 +185,7 @@ def test_gateway_restart_resumes_queued_room_for_multiplexed_profile(tmp_path):
)
)
finally:
- assert resumed.stop(timeout=1.0)
+ assert resumed.stop(timeout=5.0)
assert rpc.submits == ["ops"]
assert hosted_room_driver.list_tasks(db, room_id="room-1", status="settled")
@@ -226,8 +226,8 @@ def test_dashboard_and_gateway_workers_share_one_fenced_execution_owner(tmp_path
)
time.sleep(0.05)
finally:
- assert gateway.stop(timeout=1.0)
- assert dashboard.stop(timeout=1.0)
+ assert gateway.stop(timeout=5.0)
+ assert dashboard.stop(timeout=5.0)
assert len(gateway_rpc.submits) + len(dashboard_rpc.submits) == 1
events = hosted_rooms.read_events(db, room_id="room-1", since_seq=0)["events"]
diff --git a/tests/gateway/test_irc_adapter.py b/tests/gateway/test_irc_adapter.py
index f08cf73614..e703f5e1fd 100644
--- a/tests/gateway/test_irc_adapter.py
+++ b/tests/gateway/test_irc_adapter.py
@@ -18,6 +18,8 @@ check_requirements = _irc_mod.check_requirements
validate_config = _irc_mod.validate_config
register = _irc_mod.register
_standalone_send = _irc_mod._standalone_send
+is_connected = _irc_mod.is_connected
+_env_enablement = _irc_mod._env_enablement
class TestIRCProtocolHelpers:
@@ -406,3 +408,96 @@ class TestIRCStandaloneSend:
assert "registration" in result["error"].lower() or "timeout" in result["error"].lower()
+# ---------------------------------------------------------------------------
+# Multiplex secondary-profile scope
+# ---------------------------------------------------------------------------
+#
+# __init__'s server/port/nickname/channel/use_tls, check_requirements/
+# validate_config/is_connected's server/channel, and _env_enablement's
+# server/channel/port/nickname/use_tls/home_channel, all previously read raw
+# os.getenv unconditionally (only IRC_SERVER_PASSWORD/IRC_NICKSERV_PASSWORD
+# were already scoped). Under multiplex, os.environ holds the DEFAULT
+# profile's YAML-to-env bridge output -- a secondary profile with its own
+# (different or absent) IRC config would silently connect to the default
+# profile's server/channel, or (for _env_enablement) get auto-enabled using
+# the default's channel as its cron home_channel -- a real message-
+# misdelivery risk, not just cosmetic. Mirrors the LINE/Buzz/SimpleX fix for
+# #98738.
+
+@pytest.fixture
+def multiplex_scope():
+ """Install multiplex + a secondary-profile secret scope; restore after."""
+ tokens = []
+
+ def install(scope=None):
+ from agent.secret_scope import set_multiplex_active, set_secret_scope
+
+ set_multiplex_active(True)
+ tokens.append(set_secret_scope(scope or {}))
+ return tokens[-1]
+
+ yield install
+
+ from agent.secret_scope import reset_secret_scope, set_multiplex_active
+
+ for token in reversed(tokens):
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+
+@pytest.fixture
+def default_profile_env(monkeypatch):
+ """The default profile's YAML-to-env bridge output in os.environ."""
+ monkeypatch.setenv("IRC_SERVER", "default.example.net")
+ monkeypatch.setenv("IRC_CHANNEL", "#default")
+ monkeypatch.setenv("IRC_PORT", "6667")
+ monkeypatch.setenv("IRC_NICKNAME", "default-bot")
+ monkeypatch.setenv("IRC_USE_TLS", "false")
+
+
+class TestMultiplexProfileScope:
+
+ def test_secondary_extra_wins_over_default_profile_env(
+ self, multiplex_scope, default_profile_env
+ ):
+ """The secondary profile's own config.yaml extra is authoritative,
+ not the default profile's bridged server/channel/port/nick/tls."""
+ from gateway.config import PlatformConfig
+
+ multiplex_scope()
+ cfg = PlatformConfig(
+ enabled=True,
+ extra={
+ "server": "profile.example.net",
+ "channel": "#profile",
+ "port": 6697,
+ "nickname": "profile-bot",
+ "use_tls": True,
+ },
+ )
+ adapter = IRCAdapter(cfg)
+ assert adapter.server == "profile.example.net"
+ assert adapter.channel == "#profile"
+ assert adapter.port == 6697
+ assert adapter.nickname == "profile-bot"
+ assert adapter.use_tls is True
+
+ def test_secondary_missing_keys_fail_closed(
+ self, multiplex_scope, default_profile_env
+ ):
+ """Keys absent from the profile's own scope must NOT borrow the
+ default profile's bridged env values -- that would silently connect
+ the secondary profile's bot to the wrong IRC server/channel."""
+ from gateway.config import PlatformConfig
+
+ multiplex_scope()
+ adapter = IRCAdapter(PlatformConfig(enabled=True, extra={}))
+ assert adapter.server == ""
+ assert adapter.channel == ""
+ assert adapter.port == 6697 # falls through to the hardcoded default
+ assert adapter.nickname == "hermes-bot"
+ assert adapter.use_tls is True # extra.get("use_tls", True) default
+ # Nor may the registry auto-enable IRC for this profile off the default's channel.
+ assert _env_enablement() is None
+ assert is_connected(PlatformConfig(enabled=True, extra={})) is False
+
diff --git a/tests/gateway/test_loop_command.py b/tests/gateway/test_loop_command.py
index f18d1c5d5c..7e8caee8fc 100644
--- a/tests/gateway/test_loop_command.py
+++ b/tests/gateway/test_loop_command.py
@@ -34,6 +34,13 @@ def loop_env(tmp_path, monkeypatch):
home.mkdir()
monkeypatch.setenv("HERMES_HOME", str(home))
goals._DB_CACHE.clear()
+ # Pre-warm the SessionDB cache from this sync (non-loop) context. Inside
+ # the async tests, a cold cache makes GoalManager.set() kick the bounded
+ # background bootstrap (loop-thread path) and wait only
+ # _DB_BOOTSTRAP_INIT_WAIT_S — on a loaded CI runner the init overruns the
+ # window, the goal is never persisted, and the active-goal assertion
+ # flakes (main run 33455779041). Warming here removes the race entirely.
+ goals._get_session_db()
yield home
goals._DB_CACHE.clear()
diff --git a/tests/gateway/test_loop_liveness_watchdog.py b/tests/gateway/test_loop_liveness_watchdog.py
index ae07106b27..d763fbc461 100644
--- a/tests/gateway/test_loop_liveness_watchdog.py
+++ b/tests/gateway/test_loop_liveness_watchdog.py
@@ -3,8 +3,11 @@
from __future__ import annotations
import asyncio
+import json
+import os
import pathlib
import inspect
+import tempfile
import threading
import time
from unittest.mock import MagicMock, patch
@@ -403,3 +406,100 @@ def test_loop_scheduling_witness_is_served_by_the_loop_itself():
assert "await asyncio.start_unix_server(" in body, (
"the loop-scheduling witness socket is not armed by the loop task"
)
+
+
+def test_windows_tcp_witness_arms_and_publishes_port():
+ """On non-POSIX platforms the witness must arm over TCP loopback.
+
+ ``asyncio.start_unix_server`` does not exist on Windows (no AF_UNIX
+ event-loop support), so the producer arm fell into the broad except and
+ recorded ``loop_tick_socket=False`` — every stale-file probe then
+ classified UNKNOWN forever, disabling the wedge interlock on Windows
+ entirely. The TCP loopback witness restores the same contract: armed by
+ the loop task (an awaited ``asyncio.start_server`` is structurally
+ loop-owned exactly like the Unix variant), answered only while the loop
+ dispatches, port published in the heartbeat payload.
+ """
+ if os.name == "posix":
+ pytest.skip("TCP loopback witness is the non-POSIX arm")
+
+ async def scenario() -> tuple[dict, bool]:
+ task = asyncio.create_task(
+ loop_heartbeat_forever(interval_s=1.0, home=tmp_home)
+ )
+ try:
+ deadline = time.monotonic() + 5.0
+ payload = None
+ while time.monotonic() < deadline:
+ hb = tmp_home.joinpath(*("state", "gateway.heartbeat"))
+ if hb.exists():
+ try:
+ payload = json.loads(hb.read_text(encoding="utf-8"))
+ except Exception:
+ payload = None
+ if payload and payload.get("loop_tick_tcp_port"):
+ break
+ await asyncio.sleep(0.02)
+ assert payload is not None, "heartbeat never appeared"
+ assert payload.get("loop_tick_socket") is True, (
+ "witness reported unarmed on a platform where the TCP arm "
+ "must work"
+ )
+ port = int(payload["loop_tick_tcp_port"])
+ assert 0 < port <= 65535, "published port out of range"
+
+ # Probe from a worker thread so the blocking connect/recv never
+ # stalls the very loop we are witnessing (an external process
+ # probes from its own loop/thread — reproduce that shape).
+ from hermes_cli.gateway import _probe_loop_tick_tcp
+
+ result_box: dict[str, object] = {}
+
+ def _probe() -> None:
+ result_box["r"] = _probe_loop_tick_tcp(port, timeout=2.0)
+
+ worker = threading.Thread(target=_probe)
+ worker.start()
+ while worker.is_alive():
+ await asyncio.sleep(0.05)
+ worker.join()
+ return payload, bool(result_box.get("r") is True)
+ finally:
+ task.cancel()
+ try:
+ await task
+ except asyncio.CancelledError:
+ pass
+
+ with tempfile.TemporaryDirectory(prefix="lw-tcp-") as raw:
+ tmp_home = pathlib.Path(raw)
+ payload, answered = asyncio.run(scenario())
+ assert answered, (
+ "the loop-tick TCP witness did not answer a probe while the loop "
+ "was dispatching — the two-witness interlock would misclassify "
+ "this gateway as UNKNOWN"
+ )
+
+
+def test_windows_tcp_witness_arms_on_loop_task_source_shape():
+ """The TCP arm must be awaited by the loop task, never thread-owned.
+
+ Structural companion to ``test_loop_scheduling_witness_is_served_by_the_
+ loop_itself``: the same property that makes the Unix socket an honest
+ witness (a coroutine cannot run inside a thread) must hold for the TCP
+ loopback arm, or a wedged loop could keep answering pings and the
+ interlock would be void on Windows.
+ """
+ src = pathlib.Path(
+ inspect.getsourcefile(loop_heartbeat_forever) or ""
+ ).read_text()
+ body = src[src.index("async def loop_heartbeat_forever("):]
+ body = body[: body.index("\ndef ") if "\ndef " in body else len(body)]
+ assert "await asyncio.start_server(" in body, (
+ "the TCP loop-scheduling witness is not armed by the loop task"
+ )
+ # The Unix arm must stay gated to POSIX-only code paths so the missing
+ # attribute can never raise on Windows again.
+ assert 'os.name == "posix"' in body, (
+ "the AF_UNIX witness arm is not gated to POSIX platforms"
+ )
diff --git a/tests/gateway/test_matrix_crypto_store_per_profile.py b/tests/gateway/test_matrix_crypto_store_per_profile.py
new file mode 100644
index 0000000000..3705689260
--- /dev/null
+++ b/tests/gateway/test_matrix_crypto_store_per_profile.py
@@ -0,0 +1,47 @@
+"""Matrix crypto store must be pinned per profile at connect(), not at import.
+
+Under ``gateway.multiplex_profiles`` one process imports
+``plugins.platforms.matrix.adapter`` once; the old module-level
+``_STORE_DIR``/``_CRYPTO_DB_PATH`` resolved against the root HERMES_HOME at
+import time, so every profile's adapter opened the SAME crypto.db and inbound
+E2EE failed with "no session found" (#89168). ``connect()`` calls
+``_resolve_store_dir()`` inside ``_profile_runtime_scope`` (context-local
+HERMES_HOME), so resolving there -- and caching on the instance -- gives each
+profile its own store. Exercised via ``_resolve_store_dir`` directly so the
+test needs no mautrix install.
+"""
+from gateway.config import PlatformConfig
+from hermes_constants import reset_hermes_home_override, set_hermes_home_override
+from plugins.platforms.matrix import adapter as matrix_adapter
+
+
+def _make_adapter() -> matrix_adapter.MatrixAdapter:
+ return matrix_adapter.MatrixAdapter(
+ PlatformConfig(
+ enabled=True,
+ token="syt_test_token",
+ extra={"homeserver": "https://matrix.example.org", "user_id": "@bot:example.org"},
+ )
+ )
+
+
+def test_store_dir_pinned_to_each_profile_home(tmp_path):
+ """Two profiles resolving in one process get two stores, and each
+ adapter keeps reporting its own store after the scope is gone."""
+ stores = {}
+ for profile in ("accountant", "engineering-lead"):
+ home = tmp_path / "profiles" / profile
+ home.mkdir(parents=True)
+ adapter = _make_adapter()
+ token = set_hermes_home_override(str(home))
+ try:
+ adapter._resolve_store_dir().mkdir(parents=True, exist_ok=True)
+ finally:
+ reset_hermes_home_override(token)
+ # Cached on the instance: correct even when read outside the scope.
+ path = adapter.get_diagnostics()["e2ee"]["crypto_store_path"]
+ assert path.startswith(str(home)), f"store not profile-scoped: {path}"
+ assert adapter._store_dir.is_dir()
+ stores[profile] = path
+
+ assert stores["accountant"] != stores["engineering-lead"]
diff --git a/tests/gateway/test_mattermost.py b/tests/gateway/test_mattermost.py
index 3166ddea53..9cb56073a8 100644
--- a/tests/gateway/test_mattermost.py
+++ b/tests/gateway/test_mattermost.py
@@ -594,3 +594,118 @@ async def test_mattermost_top_level_channel_post_is_thread_root():
assert msg_event.message_id == "top_post_123"
+# ---------------------------------------------------------------------------
+# Multiplex secondary-profile scope
+# ---------------------------------------------------------------------------
+#
+# __init__'s url/reply_mode, validate_mattermost_config's url,
+# _standalone_send's url, and _handle_ws_event's require_mention/
+# free_response_channels/allowed_channels, all previously read raw
+# os.getenv unconditionally (only MATTERMOST_TOKEN was already scoped).
+# _apply_yaml_config also wrote MATTERMOST_REQUIRE_MENTION/
+# MATTERMOST_FREE_RESPONSE_CHANNELS/MATTERMOST_ALLOWED_CHANNELS into the
+# process-global os.environ unconditionally. Under multiplex, os.environ
+# holds the DEFAULT profile's YAML-to-env bridge output -- a secondary
+# profile with its own (different or absent) Mattermost config would
+# silently connect to the default profile's server, or have its
+# mention-gating/channel-allowlist decisions driven by the default
+# profile's settings. Mirrors the LINE/DingTalk/IRC fix for #98738.
+
+@pytest.fixture
+def multiplex_scope():
+ """Install multiplex + a secondary-profile secret scope; restore after."""
+ tokens = []
+
+ def install(scope=None):
+ from agent.secret_scope import set_multiplex_active, set_secret_scope
+
+ set_multiplex_active(True)
+ tokens.append(set_secret_scope(scope or {}))
+ return tokens[-1]
+
+ yield install
+
+ from agent.secret_scope import reset_secret_scope, set_multiplex_active
+
+ for token in reversed(tokens):
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+
+@pytest.fixture
+def default_profile_env(monkeypatch):
+ """The default profile's YAML-to-env bridge output in os.environ."""
+ monkeypatch.setenv("MATTERMOST_URL", "https://default.example.com")
+ monkeypatch.setenv("MATTERMOST_REPLY_MODE", "thread")
+ monkeypatch.setenv("MATTERMOST_REQUIRE_MENTION", "false")
+ monkeypatch.setenv("MATTERMOST_FREE_RESPONSE_CHANNELS", "chan_default")
+ monkeypatch.setenv("MATTERMOST_ALLOWED_CHANNELS", "chan_default")
+
+
+class TestMultiplexProfileScope:
+
+ @pytest.mark.asyncio
+ async def test_ws_event_gating_uses_scoped_settings_not_default(
+ self, monkeypatch
+ ):
+ """A secondary profile's own require_mention/free_response_channels/
+ allowed_channels (installed via the scope) must gate its messages --
+ not the default profile's bridged settings."""
+ from agent.secret_scope import (
+ reset_secret_scope,
+ set_multiplex_active,
+ set_secret_scope,
+ )
+ from plugins.platforms.mattermost.adapter import MattermostAdapter
+
+ monkeypatch.setenv("MATTERMOST_REQUIRE_MENTION", "true")
+ monkeypatch.delenv("MATTERMOST_FREE_RESPONSE_CHANNELS", raising=False)
+
+ adapter = _make_adapter()
+ adapter._bot_user_id = "bot_user_id"
+ adapter._bot_username = "hermes-bot"
+ adapter.handle_message = AsyncMock()
+
+ post_data = {
+ "id": "post_scoped",
+ "user_id": "user_123",
+ "channel_id": "chan_456",
+ "message": "hello with no mention",
+ }
+ event = {
+ "event": "posted",
+ "data": {
+ "post": json.dumps(post_data),
+ "channel_type": "O",
+ "sender_name": "@alice",
+ },
+ }
+
+ set_multiplex_active(True)
+ token = set_secret_scope({"MATTERMOST_REQUIRE_MENTION": "false"})
+ try:
+ await adapter._handle_ws_event(event)
+ finally:
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+ # The profile's own scope disables require_mention -- the message
+ # must be dispatched even without an @mention, despite the default
+ # profile's env bridge saying require_mention=true.
+ assert adapter.handle_message.called
+
+ def test_apply_yaml_config_scoped_skips_env_write_and_seeds_extra(
+ self, multiplex_scope
+ ):
+ from plugins.platforms.mattermost.adapter import _apply_yaml_config
+
+ multiplex_scope()
+ with patch.dict(os.environ, {}, clear=False):
+ os.environ.pop("MATTERMOST_REQUIRE_MENTION", None)
+ seeded = _apply_yaml_config({}, {"require_mention": False, "allowed_channels": ["c1"]})
+ assert seeded == {"require_mention": False, "allowed_channels": ["c1"]}
+ # Under a secondary profile's scope the env bridge must be
+ # skipped -- writing here would leak into every other profile's
+ # os.environ.
+ assert "MATTERMOST_REQUIRE_MENTION" not in os.environ
+
diff --git a/tests/gateway/test_multiplex_adapter_registry.py b/tests/gateway/test_multiplex_adapter_registry.py
index 3d0c196cbf..972701ddaf 100644
--- a/tests/gateway/test_multiplex_adapter_registry.py
+++ b/tests/gateway/test_multiplex_adapter_registry.py
@@ -1,6 +1,9 @@
"""Phase 3: secondary-profile adapter registry + same-token conflict detection."""
import logging
import asyncio
+import threading
+import time
+import types
from contextlib import contextmanager
from pathlib import Path
from unittest.mock import AsyncMock, MagicMock
@@ -39,6 +42,50 @@ class TestCredentialFingerprint:
assert fp1 is not None
assert "shared-project-secret" not in fp1
+ def test_reads_feishu_app_id(self):
+ """Feishu/Lark authenticates via app_id/app_secret, not a token.
+
+ Without _app_id in the fingerprint attribute list, every Feishu
+ adapter in a multiplexed gateway returns None here and the
+ same-credential conflict check is silently skipped — N profiles
+ spawn WebSocket clients against the same app, which evict each
+ other in a 1000 bye loop until all go offline.
+ """
+ class _FeishuAdapter:
+ def __init__(self):
+ self._app_id = "cli_a1b2c3"
+ self._app_secret = "top-secret"
+
+ fp1 = GatewayRunner._adapter_credential_fingerprint(_FeishuAdapter())
+ fp2 = GatewayRunner._adapter_credential_fingerprint(_FeishuAdapter())
+
+ assert fp1 is not None
+ assert fp1 == fp2 # same app -> same fingerprint -> conflict detected
+ assert "cli_a1b2c3" not in fp1 # log-safe, never the raw credential
+
+ def test_distinct_feishu_app_ids_distinct_fp(self):
+ class _FeishuAdapter:
+ def __init__(self, app_id):
+ self._app_id = app_id
+ self._app_secret = "s"
+
+ fp_a = GatewayRunner._adapter_credential_fingerprint(_FeishuAdapter("app-A"))
+ fp_b = GatewayRunner._adapter_credential_fingerprint(_FeishuAdapter("app-B"))
+
+ assert fp_a is not None and fp_b is not None
+ assert fp_a != fp_b
+
+ @pytest.mark.parametrize("attr", ["_client_id", "_bot_id"])
+ def test_reads_app_style_ids_teams_wecom(self, attr):
+ """Teams (_client_id) and WeCom (_bot_id) are the same class as Feishu:
+ id/secret pairs, no token — cloned profiles must collide."""
+ a = types.SimpleNamespace(**{attr: "app-1"})
+ b = types.SimpleNamespace(**{attr: "app-1"})
+ c = types.SimpleNamespace(**{attr: "app-2"})
+ fp = GatewayRunner._adapter_credential_fingerprint
+ assert fp(a) is not None and fp(a) == fp(b)
+ assert fp(a) != fp(c)
+ assert "app-1" not in fp(a)
def test_reads_config_token(self):
"""Adapters like Discord store token on `config`, not on self.
@@ -182,11 +229,15 @@ def _secondary_recovery_runner(*, running=True):
return runner
-def _install_secondary_reconnect_context(monkeypatch, runner, adapter, scoped_homes=None):
+def _install_secondary_reconnect_context(
+ monkeypatch, runner, adapter, scoped_homes=None, hydration_flags=None
+):
@contextmanager
- def fake_scope(profile_home):
+ def fake_scope(profile_home, *, hydrate_secrets=True):
if scoped_homes is not None:
scoped_homes.append(Path(profile_home))
+ if hydration_flags is not None:
+ hydration_flags.append(hydrate_secrets)
yield
monkeypatch.setattr(gateway_run, "_profile_runtime_scope", fake_scope)
@@ -208,6 +259,93 @@ def _install_secondary_reconnect_context(monkeypatch, runner, adapter, scoped_ho
class TestSecondaryProfileFatalRecovery:
+ @pytest.mark.asyncio
+ @pytest.mark.parametrize("entry", ["startup", "reconnect"])
+ async def test_secondary_hydrates_secrets_off_the_event_loop(self, monkeypatch, entry):
+ """#99519 class: both secondary entry points (initial start + reconnect)
+ hydrate external secret sources in a worker thread, exactly once, and
+ enter the runtime scope with hydration disabled."""
+ runner = _secondary_recovery_runner()
+ replacement = _SecondaryRecoveryAdapter()
+ hydration_flags = []
+ _install_secondary_reconnect_context(
+ monkeypatch, runner, replacement, hydration_flags=hydration_flags
+ )
+ loop_thread_id = threading.get_ident()
+ hydration_started = threading.Event()
+ hydration_finished = threading.Event()
+ hydration_thread_ids = []
+ stop_ticker = asyncio.Event()
+ ticks_during_hydration = 0
+
+ def slow_hydrate(profile_home):
+ hydration_thread_ids.append(threading.get_ident())
+ hydration_started.set()
+ time.sleep(0.05)
+ hydration_finished.set()
+
+ async def ticker():
+ nonlocal ticks_during_hydration
+ while not stop_ticker.is_set():
+ if hydration_started.is_set() and not hydration_finished.is_set():
+ ticks_during_hydration += 1
+ await asyncio.sleep(0)
+
+ async def connect(adapter, platform, **_kwargs):
+ assert adapter is replacement
+ assert platform is Platform.DISCORD
+ return True
+
+ monkeypatch.setattr(
+ "hermes_cli.env_loader.hydrate_profile_secret_sources", slow_hydrate
+ )
+ monkeypatch.setattr(runner, "_connect_adapter_with_timeout", connect)
+ monkeypatch.setattr(runner, "_connect_initial_adapter_with_timeout", connect)
+ monkeypatch.setattr(gateway_run, "_load_gateway_runtime_config", lambda: {})
+ monkeypatch.setattr(runner, "_snapshot_profile_busy_modes", lambda *a, **k: None)
+ monkeypatch.setattr("hermes_cli.plugins.discover_plugins", lambda: None)
+ if entry == "startup":
+ coro = runner._start_one_profile_adapters(
+ "reviewer", Path("/profiles/reviewer"), {}
+ )
+ else:
+ coro = runner._run_secondary_profile_reconnect("reviewer", Platform.DISCORD)
+ ticker_task = asyncio.create_task(ticker())
+ work = asyncio.create_task(coro)
+ try:
+ assert await asyncio.to_thread(hydration_started.wait, 1.0)
+ await work
+ finally:
+ stop_ticker.set()
+ await ticker_task
+
+ assert len(hydration_thread_ids) == 1
+ assert hydration_thread_ids[0] != loop_thread_id
+ assert ticks_during_hydration > 0
+ assert hydration_flags and set(hydration_flags) == {False}
+ assert runner._profile_adapters["reviewer"][Platform.DISCORD] is replacement
+
+ @pytest.mark.asyncio
+ async def test_secondary_initial_connect_syncs_voice_mode_state(self, monkeypatch):
+ """#84872: a secondary bot gets its persisted /voice state at INITIAL
+ connect, not only on reconnect."""
+ runner = _secondary_recovery_runner()
+ adapter = _SecondaryRecoveryAdapter()
+ _install_secondary_reconnect_context(monkeypatch, runner, adapter)
+ synced = []
+ runner._sync_voice_mode_state_to_adapter = synced.append
+ monkeypatch.setattr("hermes_cli.env_loader.hydrate_profile_secret_sources", lambda h: {})
+ monkeypatch.setattr(gateway_run, "_load_gateway_runtime_config", lambda: {})
+ monkeypatch.setattr(runner, "_snapshot_profile_busy_modes", lambda *a, **k: None)
+ monkeypatch.setattr("hermes_cli.plugins.discover_plugins", lambda: None)
+
+ async def connect(a, platform):
+ return True
+
+ monkeypatch.setattr(runner, "_connect_initial_adapter_with_timeout", connect)
+ assert await runner._start_one_profile_adapters("reviewer", Path("/profiles/reviewer"), {}) == 1
+ assert synced == [adapter]
+
@pytest.mark.asyncio
async def test_retryable_secondary_fatal_reconnects_with_its_profile_scope(
self, monkeypatch
@@ -341,13 +479,15 @@ class TestSecondaryStartupFailureRecovery:
# gateway is already running) to the regular reconnect task, which
# publishes the replacement and clears its own slot.
await asyncio.wait_for(bridge[0], timeout=0.5)
- for _ in range(20):
- if (
- runner._profile_adapters.get("reviewer", {}).get(Platform.DISCORD)
- is replacement
- ):
- break
- await asyncio.sleep(0)
+ # The reconnect runner hops to a worker thread for secret hydration,
+ # so wait on a deadline rather than a fixed number of loop turns.
+ deadline = time.monotonic() + 1.0
+ while (
+ runner._profile_adapters.get("reviewer", {}).get(Platform.DISCORD)
+ is not replacement
+ and time.monotonic() < deadline
+ ):
+ await asyncio.sleep(0.005)
assert (
runner._profile_adapters["reviewer"][Platform.DISCORD] is replacement
)
@@ -394,13 +534,15 @@ class TestSecondaryStartupFailureRecovery:
bridge = list(runner._background_tasks)
assert len(bridge) == 1
await asyncio.wait_for(bridge[0], timeout=0.5)
- for _ in range(20):
- if (
- runner._profile_adapters.get("reviewer", {}).get(Platform.DISCORD)
- is replacement
- ):
- break
- await asyncio.sleep(0)
+ # The reconnect runner hops to a worker thread for secret hydration,
+ # so wait on a deadline rather than a fixed number of loop turns.
+ deadline = time.monotonic() + 1.0
+ while (
+ runner._profile_adapters.get("reviewer", {}).get(Platform.DISCORD)
+ is not replacement
+ and time.monotonic() < deadline
+ ):
+ await asyncio.sleep(0.005)
assert (
runner._profile_adapters["reviewer"][Platform.DISCORD] is replacement
)
@@ -435,6 +577,55 @@ class TestSecondaryStartupFailureRecovery:
assert runner._background_tasks == set()
assert runner._profile_failed_platforms == {}
+ @pytest.mark.asyncio
+ async def test_token_lock_initial_failure_parks_fatal_not_retried(
+ self, monkeypatch
+ ):
+ """Salvage of #83183 claim 2: a secondary whose token is held by a live
+ foreign gateway (``{scope}_lock``, emitted retryable by
+ ``_acquire_platform_lock``) is an ownership conflict — park it fatal
+ like ``duplicate_credential`` instead of retry-storming the token."""
+ runner = _secondary_recovery_runner()
+ failed = _SecondaryRecoveryAdapter()
+ failed.fatal_error_code = "discord-bot-token_lock"
+ failed.fatal_error_message = "Discord bot token already in use (PID 4242)."
+ _install_secondary_reconnect_context(
+ monkeypatch, runner, _SecondaryRecoveryAdapter()
+ )
+ monkeypatch.setattr(runner, "_create_adapter", lambda platform, config: failed)
+ statuses = []
+ monkeypatch.setattr(
+ runner,
+ "_update_platform_runtime_status",
+ lambda key, **kw: statuses.append((key, kw)),
+ )
+
+ async def fail_initial_connect(adapter, platform):
+ return False
+
+ monkeypatch.setattr(
+ runner, "_connect_initial_adapter_with_timeout", fail_initial_connect
+ )
+
+ connected = await runner._start_one_profile_adapters(
+ "reviewer", "/tmp/reviewer", {}
+ )
+
+ assert connected == 0
+ assert failed.disconnected is True
+ assert runner._background_tasks == set()
+ assert runner._profile_failed_platforms == {}
+ assert statuses == [
+ (
+ "reviewer:discord",
+ {
+ "platform_state": "fatal",
+ "error_code": "discord-bot-token_lock",
+ "error_message": failed.fatal_error_message,
+ },
+ )
+ ]
+
@pytest.mark.asyncio
async def test_handoff_failure_is_logged_not_raised(self, monkeypatch, caplog):
"""If the scheduler raises at bridge handoff, the parked task must not
@@ -820,6 +1011,113 @@ class TestSecondaryProfileConfigHandling:
assert second == 1
assert runner._profile_adapters["later"][photon] is later
+ @pytest.mark.asyncio
+ async def test_secondary_teams_uses_degradable_error(self, monkeypatch):
+ from gateway.config import GatewayConfig, Platform, PlatformConfig
+ from gateway.run import SecondaryPortBindingConfigError
+
+ runner = GatewayRunner.__new__(GatewayRunner)
+ runner.config = GatewayConfig(multiplex_profiles=True)
+ runner._profile_adapters = {}
+
+ reviewer_cfg = GatewayConfig(multiplex_profiles=True)
+ reviewer_cfg.platforms = {
+ Platform("teams"): PlatformConfig(enabled=True, extra={"port": 3978}),
+ }
+ monkeypatch.setattr(
+ "gateway.config.load_gateway_config", lambda: reviewer_cfg
+ )
+
+ with pytest.raises(SecondaryPortBindingConfigError) as exc_info:
+ await runner._start_one_profile_adapters("reviewer", "/tmp/x", {})
+ assert "teams" in str(exc_info.value)
+ assert "reviewer" in str(exc_info.value)
+ assert "reviewer" not in runner._profile_adapters
+
+ @pytest.mark.asyncio
+ async def test_secondary_profile_adapter_start_skips_whatsapp(self, monkeypatch):
+ """WhatsApp is shared process-level ingress like Relay: the bridge is
+ one authenticated session tied to a single phone number, so a
+ credential-less secondary profile must be skipped (not stall startup
+ in a connect/retry loop) while its other platforms start normally."""
+ runner = _secondary_recovery_runner()
+ direct = _SecondaryRecoveryAdapter()
+ _install_secondary_reconnect_context(monkeypatch, runner, direct)
+ monkeypatch.setattr(
+ "gateway.config.load_gateway_config",
+ lambda: GatewayConfig(
+ multiplex_profiles=True,
+ platforms={
+ Platform.WHATSAPP: PlatformConfig(enabled=True),
+ Platform.DISCORD: PlatformConfig(enabled=True, token="profile-token"),
+ },
+ ),
+ )
+ factory_calls = []
+
+ def _create_adapter(platform, config):
+ factory_calls.append(platform)
+ return direct
+
+ async def _connect(adapter, platform):
+ return True
+
+ monkeypatch.setattr(runner, "_create_adapter", _create_adapter)
+ monkeypatch.setattr(runner, "_connect_initial_adapter_with_timeout", _connect)
+
+ connected = await runner._start_one_profile_adapters("clientbot", "/tmp/x", {})
+
+ assert connected == 1
+ assert factory_calls == [Platform.DISCORD]
+ assert runner._profile_adapters["clientbot"] == {Platform.DISCORD: direct}
+
+
+class TestSecondaryProfileHookRegistration:
+ """A secondary profile's own `hooks:` block must register on ITS
+ plugin manager, not just the root/default profile's (#92672).
+
+ Startup only calls agent.shell_hooks/outbound_webhooks
+ register_from_config() once, against the root config, before any
+ profile scope exists. Without a matching call inside
+ _start_one_profile_adapters, a secondary profile's config.yaml
+ `hooks:` block (shell hooks and outbound webhooks) never registers.
+ """
+
+ @pytest.mark.asyncio
+ async def test_registers_shell_hooks_and_webhooks_for_secondary_profile(
+ self, monkeypatch
+ ):
+ runner = _secondary_recovery_runner()
+ config = GatewayConfig(multiplex_profiles=True, platforms={})
+ monkeypatch.setattr("gateway.config.load_gateway_config", lambda: config)
+
+ profile_cfg = {
+ "hooks": {
+ "pre_tool_call": [
+ {"matcher": "write_file", "command": "~/.hermes/deny.sh"}
+ ],
+ "outbound": [
+ {"url": "http://127.0.0.1:9000/hook", "events": ["on_session_end"]}
+ ],
+ }
+ }
+ monkeypatch.setattr("hermes_cli.config.load_config", lambda: profile_cfg)
+
+ seen = []
+ monkeypatch.setattr(
+ "agent.shell_hooks.register_from_config",
+ lambda cfg, **kwargs: seen.append(("shell", cfg)) or [],
+ )
+ monkeypatch.setattr(
+ "agent.outbound_webhooks.register_from_config",
+ lambda cfg: seen.append(("webhook", cfg)) or [],
+ )
+
+ await runner._start_one_profile_adapters("second", "/tmp/second", {})
+
+ assert ("shell", profile_cfg) in seen
+ assert ("webhook", profile_cfg) in seen
+
class TestFeishuPortBindingConditional:
"""Feishu websocket mode does NOT bind a port; only webhook mode does (#52563)."""
@@ -848,3 +1146,78 @@ class TestFeishuPortBindingConditional:
assert connected == 0 # no error, just nothing connected
+class TestSecondarySkipsCredentiallessPlatforms:
+ """#84079 — multiplex must not build adapters for platforms a profile
+ has no credential for.
+
+ The shared config.yaml enables a platform once; under multiplex every
+ secondary profile reloads it inside its own secret scope, so a profile
+ whose scope lacks the platform credential resolves ``enabled=True`` with
+ an empty token. Constructing an adapter anyway treats every profile as
+ configured for the platform — one inbound message fans out across all of
+ them. These tests lock the credential gate on the secondary startup path
+ (the primary path got the same gate in #64674; the reconnect path shares
+ the helper). Also reported independently in #72313.
+ """
+
+ def _make_runner(self, monkeypatch, profile_cfg):
+ runner = GatewayRunner.__new__(GatewayRunner)
+ runner.config = GatewayConfig(multiplex_profiles=True)
+ runner._profile_adapters = {}
+ runner.adapters = {}
+ created = []
+
+ def fake_create(platform, platform_config):
+ created.append((platform, platform_config))
+ return _FakeAdapter(token=platform_config.token or None)
+
+ monkeypatch.setattr("gateway.config.load_gateway_config", lambda: profile_cfg)
+ monkeypatch.setattr(runner, "_create_adapter", fake_create)
+ monkeypatch.setattr(runner, "_configure_profile_adapter", lambda *a, **k: None)
+ monkeypatch.setattr(
+ runner,
+ "_connect_initial_adapter_with_timeout",
+ AsyncMock(return_value=True),
+ )
+ return runner, created
+
+ @pytest.mark.asyncio
+ async def test_credentialless_platform_builds_no_adapter(self, monkeypatch, tmp_path):
+ """Enabled-in-YAML but no credential in the profile scope -> no adapter."""
+ from gateway.config import GatewayConfig, Platform, PlatformConfig
+
+ profile_cfg = GatewayConfig(multiplex_profiles=True)
+ profile_cfg.platforms = {
+ # Shared config.yaml enables Slack; profile-b's .env has no
+ # SLACK_BOT_TOKEN, so its scoped load resolves token="" but
+ # keeps enabled=True (#84079).
+ Platform.SLACK: PlatformConfig(enabled=True, token=""),
+ Platform.TELEGRAM: PlatformConfig(enabled=True, token="telegram-token-b"),
+ }
+ runner, created = self._make_runner(monkeypatch, profile_cfg)
+
+ connected = await runner._start_one_profile_adapters("profile-b", tmp_path, {})
+
+ # Only Telegram (which profile-b has its own credential for) gets an
+ # adapter; Slack is skipped instead of fanning out a turn per profile.
+ assert [p for p, _ in created] == [Platform.TELEGRAM]
+ assert connected == 1
+ assert Platform.TELEGRAM in runner._profile_adapters["profile-b"]
+ assert Platform.SLACK not in runner._profile_adapters["profile-b"]
+
+ @pytest.mark.asyncio
+ async def test_profile_with_own_credential_still_connects(self, monkeypatch, tmp_path):
+ """A profile that defines its own credential keeps its adapter."""
+ from gateway.config import GatewayConfig, Platform, PlatformConfig
+
+ profile_cfg = GatewayConfig(multiplex_profiles=True)
+ profile_cfg.platforms = {
+ Platform.SLACK: PlatformConfig(enabled=True, token="slack-token-b"),
+ }
+ runner, created = self._make_runner(monkeypatch, profile_cfg)
+
+ connected = await runner._start_one_profile_adapters("profile-b", tmp_path, {})
+
+ assert connected == 1
+ assert created == [(Platform.SLACK, profile_cfg.platforms[Platform.SLACK])]
+ assert Platform.SLACK in runner._profile_adapters["profile-b"]
diff --git a/tests/gateway/test_multiplex_credential_isolation.py b/tests/gateway/test_multiplex_credential_isolation.py
index f5d2d5d8f6..a43ee8f440 100644
--- a/tests/gateway/test_multiplex_credential_isolation.py
+++ b/tests/gateway/test_multiplex_credential_isolation.py
@@ -88,6 +88,54 @@ class TestProfilePathResolutionUnderMultiplexScope:
assert b_seen == prof_b / "skills"
+def test_turn_scoped_dotenv_reload_does_not_pollute_process_env(tmp_path, monkeypatch):
+ """A routed profile reload must stay inside its context-local scope.
+
+ ``load_hermes_dotenv`` has several lazy-import and cron call sites beyond
+ the gateway's guarded reload helper. Any one of them can run during a
+ multiplexed turn, so the loader itself must not copy the active profile's
+ ``.env`` into the shared process environment.
+ """
+ import os
+
+ from agent.secret_scope import get_secret
+ from gateway.run import _profile_runtime_scope
+ from hermes_cli.env_loader import load_hermes_dotenv
+ from hermes_constants import get_hermes_home
+
+ profile_a = tmp_path / "profiles" / "a"
+ profile_b = tmp_path / "profiles" / "b"
+ profile_a.mkdir(parents=True)
+ profile_b.mkdir(parents=True)
+ (profile_a / ".env").write_text(
+ "PROFILE_SCOPED_API_KEY=secret-a\n"
+ "DISCORD_ALLOWED_CHANNELS=profile-a-only\n",
+ encoding="utf-8",
+ )
+ (profile_b / ".env").write_text(
+ "PROFILE_SCOPED_API_KEY=secret-b\n"
+ "DISCORD_ALLOWED_CHANNELS=profile-b-only\n",
+ encoding="utf-8",
+ )
+ monkeypatch.delenv("PROFILE_SCOPED_API_KEY", raising=False)
+ monkeypatch.setenv("DISCORD_ALLOWED_CHANNELS", "all-channels")
+
+ ss.set_multiplex_active(True)
+ with _profile_runtime_scope(profile_a):
+ assert get_secret("PROFILE_SCOPED_API_KEY") == "secret-a"
+ assert get_secret("DISCORD_ALLOWED_CHANNELS") == "profile-a-only"
+ assert load_hermes_dotenv(hermes_home=get_hermes_home()) == []
+ assert "PROFILE_SCOPED_API_KEY" not in os.environ
+ assert os.environ["DISCORD_ALLOWED_CHANNELS"] == "all-channels"
+
+ with _profile_runtime_scope(profile_b):
+ assert get_secret("PROFILE_SCOPED_API_KEY") == "secret-b"
+ assert get_secret("DISCORD_ALLOWED_CHANNELS") == "profile-b-only"
+ assert load_hermes_dotenv(hermes_home=get_hermes_home()) == []
+ assert "PROFILE_SCOPED_API_KEY" not in os.environ
+ assert os.environ["DISCORD_ALLOWED_CHANNELS"] == "all-channels"
+
+
def test_cold_profile_hydrates_external_source_without_global_env(
tmp_path, monkeypatch
):
@@ -164,5 +212,3 @@ def test_cold_profile_hydrates_external_source_without_global_env(
assert calls["count"] == 1
assert "TEST_PROVIDER_API_KEY" not in os.environ
assert "EXPLICIT_API_KEY" not in os.environ
-
-
diff --git a/tests/gateway/test_multiplex_interactive_auth.py b/tests/gateway/test_multiplex_interactive_auth.py
new file mode 100644
index 0000000000..5ed97aa17c
--- /dev/null
+++ b/tests/gateway/test_multiplex_interactive_auth.py
@@ -0,0 +1,170 @@
+"""Multiplex interactive-auth regressions (#86296, #92840, #72657, #87240 egress).
+
+Real ``GatewayRunner`` methods on an ``object.__new__`` runner, real
+``PairingStore`` files under a temp HERMES_HOME, multiplex active.
+"""
+
+from pathlib import Path
+from types import SimpleNamespace
+
+import pytest
+
+from gateway.config import GatewayConfig, Platform, PlatformConfig
+from gateway.pairing import PairingStore
+from gateway.profile_routing import ProfileRoute
+
+
+@pytest.fixture
+def mux_home(tmp_path, monkeypatch):
+ from agent import secret_scope
+
+ home = tmp_path / "hh"
+ (home / "profiles" / "secondary").mkdir(parents=True)
+ (home / ".env").write_text("")
+ (home / "profiles" / "secondary" / ".env").write_text("")
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ for key in (
+ "TELEGRAM_ALLOWED_USERS",
+ "TELEGRAM_ALLOW_BOTS",
+ "GATEWAY_ALLOW_ALL_USERS",
+ "GATEWAY_ALLOWED_USERS",
+ "SLACK_ALLOW_ALL_USERS",
+ "SLACK_ALLOWED_USERS",
+ ):
+ monkeypatch.delenv(key, raising=False)
+ prev = secret_scope.is_multiplex_active()
+ secret_scope.set_multiplex_active(True)
+ yield home
+ secret_scope.set_multiplex_active(prev)
+
+
+def _runner(home):
+ from gateway.run import GatewayRunner
+
+ runner = object.__new__(GatewayRunner)
+ runner.config = GatewayConfig(multiplex_profiles=True)
+ runner.config.profile_routes = [
+ ProfileRoute(name="r", platform="telegram", chat_id="-100555", profile="secondary")
+ ]
+ runner.config.platforms = {Platform.TELEGRAM: PlatformConfig(enabled=True, extra={})}
+ runner.pairing_store = PairingStore(profile="default")
+ runner.pairing_stores = {
+ "default": runner.pairing_store,
+ "secondary": PairingStore(profile="secondary"),
+ }
+ runner._primary_profile_name = "default"
+ runner._profile_adapters = {"secondary": {}}
+ return runner
+
+
+def _telegram(runner):
+ from plugins.platforms.telegram.adapter import TelegramAdapter
+
+ tg = object.__new__(TelegramAdapter)
+ tg.config = PlatformConfig(enabled=True, extra={})
+ tg._authorization_check = None
+ tg._message_handler = runner._primary_message_handler() # closure, no __self__
+ runner.adapters = {Platform.TELEGRAM: tg}
+ tg.set_authorization_check(runner._make_adapter_auth_check(Platform.TELEGRAM))
+ return tg
+
+
+def test_routed_primary_callback_uses_routed_pairing_store_and_transport_allowlist(mux_home):
+ """#86296: shared primary bot + profile_routes → the inline-button caller
+ is authorized by the ROUTED profile's pairing store, while env allowlists
+ resolve under the transport (launch) home, exactly like inbound messages."""
+ runner = _runner(mux_home)
+ store = runner.pairing_stores["secondary"]
+ store._save_json(store._approved_path("telegram"), {"777": {}})
+ (mux_home / ".env").write_text("TELEGRAM_ALLOWED_USERS=999\n")
+ tg = _telegram(runner)
+
+ # Paired only in the routed profile → allowed in the routed chat only.
+ assert tg._is_callback_user_authorized("777", chat_id="-100555", chat_type="supergroup") is True
+ assert tg._is_callback_user_authorized("777", chat_id="-100999", chat_type="supergroup") is False
+ # Transport-home allowlist honored in the routed chat (not the routed profile's empty scope).
+ assert tg._is_callback_user_authorized("999", chat_id="-100555", chat_type="supergroup") is True
+ assert tg._is_callback_user_authorized("888", chat_id="-100555", chat_type="supergroup") is False
+
+
+def test_bot_sender_reaches_allow_bots_policy_through_callback(mux_home):
+ """#92840: the early prefilter must carry ``is_bot`` so TELEGRAM_ALLOW_BOTS
+ admits bot-authored messages under the multiplex closure handler."""
+ from gateway.run import _profile_runtime_scope
+
+ runner = _runner(mux_home)
+ (mux_home / ".env").write_text("TELEGRAM_ALLOWED_USERS=999\nTELEGRAM_ALLOW_BOTS=all\n")
+ tg = _telegram(runner)
+
+ def msg(uid, is_bot):
+ return SimpleNamespace(
+ from_user=SimpleNamespace(id=uid, is_bot=is_bot, username="x", full_name="X"),
+ chat=SimpleNamespace(id=-100777, type="supergroup", is_forum=False),
+ sender_chat=None,
+ message_thread_id=None,
+ is_topic_message=False,
+ )
+
+ with _profile_runtime_scope(mux_home):
+ assert tg._is_user_authorized_from_message(msg(4242, True)) is True
+ assert tg._is_user_authorized_from_message(msg(4343, False)) is False
+
+
+def test_slack_interactive_auth_prefers_wired_profile_check(mux_home, monkeypatch):
+ """#72657: a multiplexed Slack adapter's button gate resolves through the
+ wired ``_make_adapter_auth_check`` for its own profile; the DEFAULT
+ profile's process-env allow-all never leaks in — not through the
+ injected path, and not through the env-only fallback either."""
+ from gateway.run import _profile_runtime_scope
+ from plugins.platforms.slack.adapter import SlackAdapter
+
+ runner = _runner(mux_home)
+ runner.adapters = {}
+ sec_home = mux_home / "profiles" / "secondary"
+ (sec_home / ".env").write_text("SLACK_ALLOWED_USERS=U_SEC\n")
+ monkeypatch.setenv("SLACK_ALLOW_ALL_USERS", "true")
+
+ def slack(with_check):
+ sl = object.__new__(SlackAdapter)
+ sl.config = PlatformConfig(enabled=True, extra={})
+ sl._authorization_check = None
+ sl._message_handler = runner._make_profile_message_handler("secondary")
+ if with_check:
+ runner._profile_adapters = {"secondary": {Platform.SLACK: sl}}
+ sl.set_authorization_check(
+ runner._make_adapter_auth_check(Platform.SLACK, profile_name="secondary")
+ )
+ return sl
+
+ with _profile_runtime_scope(sec_home):
+ wired = slack(True)
+ assert wired._is_interactive_user_authorized("U_SEC", channel_id="C1") is True
+ assert wired._is_interactive_user_authorized("U_X", channel_id="C1") is False
+ assert slack(False)._is_interactive_user_authorized("U_X", channel_id="C1") is False
+
+
+def test_authorization_adapter_ignores_per_turn_active_profile(mux_home):
+ """#87240 egress half: inside a secondary profile's runtime scope the
+ default bot must not be handed to that profile (fail-closed None); the
+ launch profile still resolves ``self.adapters``."""
+ from gateway.run import _profile_runtime_scope
+
+ runner = _runner(mux_home)
+ default_bot = object()
+ runner.adapters = {Platform.TELEGRAM: default_bot}
+
+ with _profile_runtime_scope(mux_home / "profiles" / "secondary"):
+ assert runner._authorization_adapter(Platform.TELEGRAM, profile="secondary") is None
+ assert runner._authorization_adapter(Platform.TELEGRAM, profile="default") is default_bot
+
+
+def test_channel_directory_path_follows_current_home(mux_home):
+ """#87240: the directory file resolves against the CURRENT profile home,
+ not the home that happened to import the module."""
+ import gateway.channel_directory as cd
+ from gateway.run import _profile_runtime_scope
+
+ assert cd.DIRECTORY_PATH is None
+ with _profile_runtime_scope(mux_home / "profiles" / "secondary"):
+ assert cd._directory_path() == Path(mux_home / "profiles" / "secondary" / "channel_directory.json")
+ assert cd._directory_path() == Path(mux_home / "channel_directory.json")
diff --git a/tests/gateway/test_multiplex_log_routing.py b/tests/gateway/test_multiplex_log_routing.py
new file mode 100644
index 0000000000..529018b51a
--- /dev/null
+++ b/tests/gateway/test_multiplex_log_routing.py
@@ -0,0 +1,67 @@
+"""Multiplex gateway log routing (#82936, salvage of #84954).
+
+``setup_logging(mode="gateway")`` binds agent.log/errors.log/gateway.log to
+the launch home. Under ``multiplex_profiles`` every secondary profile's
+records — emitted inside ``_profile_runtime_scope`` — used to fan out into
+the DEFAULT profile's files. The gateway now enables the #99440 profile
+routers at startup so each record lands in its owner's ``logs/``.
+"""
+
+import logging
+import types
+from pathlib import Path
+
+import pytest
+
+import hermes_logging
+from gateway import run
+
+
+@pytest.fixture
+def clean_logging():
+ hermes_logging._reset_queued_handlers()
+ hermes_logging._logging_initialized = False
+ yield
+ hermes_logging._reset_queued_handlers()
+ hermes_logging._logging_initialized = False
+
+
+def _emit_under(home: Path, name: str, level: int, msg: str) -> None:
+ from hermes_constants import reset_hermes_home_override, set_hermes_home_override
+
+ token = set_hermes_home_override(home)
+ try:
+ logging.getLogger(name).log(level, msg)
+ finally:
+ reset_hermes_home_override(token)
+
+
+def _contains(home: Path, filename: str, needle: str) -> bool:
+ path = home / "logs" / filename
+ return path.exists() and needle in path.read_text()
+
+
+def test_multiplex_gateway_routes_profile_records_to_their_own_logs(
+ tmp_path, monkeypatch, clean_logging
+):
+ default_home = tmp_path / "default"
+ beta_home = tmp_path / "default" / "profiles" / "beta"
+ beta_home.mkdir(parents=True)
+ homes = [("default", default_home), ("beta", beta_home)]
+ monkeypatch.setattr(run, "_multiplex_profile_homes", lambda _cfg: homes)
+
+ hermes_logging.setup_logging(hermes_home=default_home, mode="gateway")
+
+ # Single-profile gateway: wiring is inert and handlers stay static.
+ assert run._enable_multiplex_log_routing(types.SimpleNamespace(multiplex_profiles=False)) is False
+ assert run._enable_multiplex_log_routing(types.SimpleNamespace(multiplex_profiles=True)) is True
+
+ _emit_under(beta_home, "gateway.run", logging.WARNING, "BETA-GATEWAY-WARN")
+ _emit_under(default_home, "gateway.run", logging.INFO, "DEFAULT-GATEWAY-INFO")
+ hermes_logging.flush_log_queue()
+
+ for filename in ("agent.log", "errors.log", "gateway.log"):
+ assert _contains(beta_home, filename, "BETA-GATEWAY-WARN"), filename
+ assert not _contains(default_home, filename, "BETA-GATEWAY-WARN"), filename
+ assert _contains(default_home, "gateway.log", "DEFAULT-GATEWAY-INFO")
+ assert not _contains(beta_home, "gateway.log", "DEFAULT-GATEWAY-INFO")
diff --git a/tests/gateway/test_multiplex_mcp_discovery.py b/tests/gateway/test_multiplex_mcp_discovery.py
new file mode 100644
index 0000000000..633b39b755
--- /dev/null
+++ b/tests/gateway/test_multiplex_mcp_discovery.py
@@ -0,0 +1,121 @@
+"""Multiplexed gateways discover and reload MCP servers per profile (#95518)."""
+
+from __future__ import annotations
+
+import threading
+from pathlib import Path
+from types import SimpleNamespace
+from unittest.mock import MagicMock
+
+import pytest
+
+from gateway.config import GatewayConfig, Platform
+from gateway.platforms.base import MessageEvent
+from gateway.session import SessionSource
+from hermes_constants import get_hermes_home, hermes_home_key
+
+
+@pytest.mark.asyncio
+async def test_gateway_boot_discovers_mcp_under_every_profile_home(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ import gateway.run as gateway_run
+ from tools import mcp_tool
+
+ homes = [("default", tmp_path / "default"), ("worker", tmp_path / "worker")]
+ for _name, home in homes:
+ home.mkdir()
+ seen: list[tuple[Path, str]] = []
+
+ def fake_discover() -> list[str]:
+ seen.append((get_hermes_home(), threading.current_thread().name))
+ return []
+
+ monkeypatch.setattr(
+ "hermes_cli.profiles.profiles_to_serve",
+ lambda multiplex, profile_allowlist=None: homes,
+ )
+ monkeypatch.setattr(mcp_tool, "discover_mcp_tools", fake_discover)
+
+ await gateway_run._discover_gateway_mcp_tools(GatewayConfig(multiplex_profiles=True))
+
+ # Ran once per profile, under that profile's home, off the loop thread.
+ assert [home for home, _ in seen] == [home for _, home in homes]
+ assert all(thread != threading.current_thread().name for _, thread in seen)
+
+
+@pytest.mark.asyncio
+async def test_reload_mcp_only_touches_requesting_profile(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ from gateway.run import GatewayRunner
+ from tools import mcp_tool
+
+ worker_home = tmp_path / "profiles" / "worker"
+ worker_home.mkdir(parents=True)
+ worker_scope = hermes_home_key(worker_home)
+
+ runner = GatewayRunner.__new__(GatewayRunner)
+ runner.config = GatewayConfig(multiplex_profiles=True)
+ runner._resolve_profile_home_for_source = MagicMock(return_value=worker_home)
+ runner._agent_cache = {}
+ runner._agent_cache_lock = None
+ runner._async_session_store = SimpleNamespace(
+ get_or_create_session=MagicMock(side_effect=RuntimeError("skip transcript")),
+ )
+
+ monkeypatch.setattr(mcp_tool, "_servers", {"default-srv": object(), "worker-srv": object()})
+ monkeypatch.setattr(
+ mcp_tool, "_server_scope_keys",
+ {"default-srv": hermes_home_key(tmp_path), "worker-srv": worker_scope},
+ )
+ seen: list[tuple] = []
+
+ def fake_shutdown(*, scope=None) -> None:
+ seen.append(("shutdown", scope, get_hermes_home()))
+
+ def fake_discover() -> list[str]:
+ seen.append(("discover", get_hermes_home()))
+ return []
+
+ monkeypatch.setattr(mcp_tool, "shutdown_mcp_servers", fake_shutdown)
+ monkeypatch.setattr(mcp_tool, "discover_mcp_tools", fake_discover)
+
+ event = MessageEvent(
+ text="/reload-mcp", message_id="m1",
+ source=SessionSource(
+ platform=Platform.TELEGRAM, user_id="u1", chat_id="c1",
+ chat_type="dm", profile="worker",
+ ),
+ )
+ result = await runner._execute_mcp_reload(event)
+
+ # Entered worker's scope itself, shut down only worker's servers, and
+ # reported only worker's servers (default's untouched connection is not
+ # "removed").
+ assert seen == [
+ ("shutdown", worker_scope, worker_home),
+ ("discover", worker_home),
+ ]
+ assert "default-srv" not in result
+
+
+def test_deregister_scope_kwarg_targets_overlay_and_keeps_plugin_confinement() -> None:
+ from tools.registry import ToolRegistry
+
+ reg = ToolRegistry()
+ reg.register("mcp__s__t", "mcp-s", {"name": "mcp__s__t", "description": "d"},
+ lambda **kw: None, scope="/home/p1")
+ assert reg.snapshot_registration("mcp__s__t", scope="/home/p1") is not None
+
+ reg.deregister("mcp__s__t") # unscoped: global slot only, overlay untouched
+ assert reg.snapshot_registration("mcp__s__t", scope="/home/p1") is not None
+
+ reg.deregister("mcp__s__t", scope="/home/p1")
+ assert reg.snapshot_registration("mcp__s__t", scope="/home/p1") is None
+
+ # A plugin module may not name another profile's overlay.
+ reg._plugin_module_scopes["hermes_plugins.p"] = {"/home/p1"}
+ reg._caller_module = staticmethod(lambda: "hermes_plugins.p")
+ with pytest.raises(PermissionError):
+ reg.deregister("anything", scope="/home/p2")
diff --git a/tests/gateway/test_multiplex_pairing_stores.py b/tests/gateway/test_multiplex_pairing_stores.py
index 63c4a9ea9a..78dc2f3e1d 100644
--- a/tests/gateway/test_multiplex_pairing_stores.py
+++ b/tests/gateway/test_multiplex_pairing_stores.py
@@ -85,3 +85,40 @@ def test_pairing_store_scoped_to_profile_dir(tmp_path, monkeypatch):
assert "profiles/ops/platforms/pairing" in str(store._dir).replace("\\", "/"), (
f"store not profile-scoped: {store._dir}"
)
+
+
+def test_routed_pairing_grant_mirror_stays_in_profile_scope(tmp_path, monkeypatch):
+ """A /pair grant mirrored under a routed profile scope must update THAT
+ profile's .env and installed scope, never the shared os.environ (#88441,
+ #77490). Outside multiplex the legacy os.environ publish is unchanged."""
+ import os
+
+ from agent import secret_scope as ss
+ from gateway.pairing import _sync_allowlist_add
+ from gateway.run import _profile_runtime_scope
+ from hermes_cli.config import save_env_value
+
+ root = tmp_path / ".hermes"
+ prof = root / "profiles" / "b"
+ prof.mkdir(parents=True)
+ (root / ".env").write_text("DISCORD_ALLOWED_USERS=default-admin\n")
+ (prof / ".env").write_text("DISCORD_ALLOWED_USERS=b-admin\n")
+ monkeypatch.setenv("HERMES_HOME", str(root))
+ monkeypatch.setenv("DISCORD_ALLOWED_USERS", "default-admin")
+
+ was_active = ss.is_multiplex_active()
+ ss.set_multiplex_active(True)
+ try:
+ with _profile_runtime_scope(prof):
+ _sync_allowlist_add("discord", "111")
+ assert ss.get_secret("DISCORD_ALLOWED_USERS") == "b-admin,111"
+ finally:
+ ss.set_multiplex_active(was_active)
+
+ assert (prof / ".env").read_text().strip() == "DISCORD_ALLOWED_USERS=b-admin,111"
+ assert (root / ".env").read_text().strip() == "DISCORD_ALLOWED_USERS=default-admin"
+ assert os.environ["DISCORD_ALLOWED_USERS"] == "default-admin"
+
+ # Single-profile: no multiplex -> save still publishes to the process env.
+ save_env_value("DISCORD_ALLOWED_USERS", "default-admin,222")
+ assert os.environ["DISCORD_ALLOWED_USERS"] == "default-admin,222"
diff --git a/tests/gateway/test_multiplex_phase0.py b/tests/gateway/test_multiplex_phase0.py
index 836a0d6355..6558d67f54 100644
--- a/tests/gateway/test_multiplex_phase0.py
+++ b/tests/gateway/test_multiplex_phase0.py
@@ -172,3 +172,27 @@ class TestSessionStoreUnmultiplexedRecovery:
assert recovered.session_id == "sess-coder"
assert recovered.session_key == "agent:main:telegram:dm:99"
assert store._db.reopened == ["sess-coder"]
+
+ @pytest.mark.parametrize(
+ ("recovered_key", "adopted"),
+ [
+ ("agent:coder:telegram:dm:99", False), # sibling namespace → fail closed
+ ("agent:main:telegram:dm:99:v1", True), # same namespace → adoptable
+ ],
+ ids=["sibling-profile", "same-profile"],
+ )
+ def test_flag_on_fences_recovery_by_requested_namespace(
+ self, tmp_path, recovered_key, adopted
+ ):
+ """#74285: under multiplexing the guard compares the recovered row's
+ ``agent::`` against the REQUESTED key, never the active profile."""
+ row = {"id": "sess", "started_at": 1700000000, "session_key": recovered_key}
+ store = self._store_with_row(tmp_path, row, multiplex_profiles=True)
+ store._db_pinned = store._db
+ with patch("hermes_cli.profiles.get_active_profile_name", return_value="coder"):
+ recovered = store._recover_session_from_db(
+ session_key="agent:main:telegram:dm:99",
+ source=_src(chat_id="99", chat_type="dm"),
+ now=datetime.fromtimestamp(1700000001),
+ )
+ assert (recovered is not None) is adopted
diff --git a/tests/gateway/test_multiplex_profile_authz.py b/tests/gateway/test_multiplex_profile_authz.py
index e20176efcb..cd1bbf5a59 100644
--- a/tests/gateway/test_multiplex_profile_authz.py
+++ b/tests/gateway/test_multiplex_profile_authz.py
@@ -74,6 +74,52 @@ def test_active_profile_stamp_resolves_primary_adapter(monkeypatch):
assert runner._authorization_adapter(Platform.WECOM, profile="dev") is default_adapter
+def test_scoped_secondary_profile_still_uses_profile_adapters(monkeypatch):
+ """Runtime scope must not redirect secondary authz to primary adapters.
+
+ ``_make_profile_message_handler`` wraps ``_handle_message`` in
+ ``_profile_runtime_scope``, which overrides HERMES_HOME so
+ ``get_active_profile_name()`` equals the secondary profile for that turn.
+ Authorization must still read ``_profile_adapters[profile]``, not the
+ empty primary ``self.adapters`` map — otherwise upstream-auth platforms
+ such as A2A default-deny an already-authenticated peer (#80884). A
+ secondary profile with NO registry entry still fails closed.
+ """
+ from gateway.run import GatewayRunner
+
+ _clear_auth_env(monkeypatch)
+
+ runner = object.__new__(GatewayRunner)
+ runner.config = GatewayConfig(multiplex_profiles=True)
+ runner.adapters = {}
+ runner.pairing_store = MagicMock()
+ runner.pairing_store.is_approved.return_value = False
+
+ secondary = SimpleNamespace(
+ authorization_is_upstream=True,
+ enforces_own_access_policy=False,
+ )
+ runner._profile_adapters = {"beta": {Platform("a2a"): secondary}}
+ # Simulate the scoped turn: active profile name collapses to the secondary.
+ runner._active_profile_name = lambda: "beta"
+
+ assert runner._authorization_adapter(Platform("a2a"), profile="beta") is secondary
+
+ source = SessionSource(
+ platform=Platform("a2a"),
+ chat_id="a2a-context",
+ user_id="alpha",
+ user_name="alpha",
+ chat_type="dm",
+ profile="beta",
+ )
+ assert runner._is_user_authorized(source) is True
+
+ # Fail-closed guard is untouched: no registry entry -> no default fallback.
+ runner._profile_adapters = {"beta": {}}
+ assert runner._authorization_adapter(Platform("a2a"), profile="beta") is None
+
+
def test_secondary_allowlist_dm_behavior_ignores_unauthorized(monkeypatch):
"""Unauthorized-DM behavior must read the secondary adapter's dm_policy."""
runner, _default_adapter, secondary_adapter = _make_multiplex_runner(monkeypatch)
diff --git a/tests/gateway/test_ntfy_plugin.py b/tests/gateway/test_ntfy_plugin.py
index 9e992eeb3e..8cb86b7f8f 100644
--- a/tests/gateway/test_ntfy_plugin.py
+++ b/tests/gateway/test_ntfy_plugin.py
@@ -491,3 +491,82 @@ class TestTruncateHelper:
assert _ntfy._truncate_body("hi", context="test") == b"hi"
+# ---------------------------------------------------------------------------
+# 13. Multiplex secondary-profile scope
+# ---------------------------------------------------------------------------
+#
+# __init__'s server/topic/publish_topic, _env_enablement's topic/server/
+# publish_topic/markdown/home_channel, and check_requirements/validate_config/
+# is_connected's topic reads, all previously read raw os.getenv
+# unconditionally (only NTFY_TOKEN was already scoped). Under multiplex,
+# os.environ holds the DEFAULT profile's YAML-to-env bridge output -- a
+# secondary profile with its own (different or absent) ntfy config would
+# silently subscribe to / publish on the default profile's topic, or get
+# auto-enabled using the default profile's topic entirely. Mirrors the
+# LINE/Buzz/SimpleX fix for #98738.
+
+@pytest.fixture
+def multiplex_scope():
+ """Install multiplex + a secondary-profile secret scope; restore after."""
+ tokens = []
+
+ def install(scope=None):
+ from agent.secret_scope import set_multiplex_active, set_secret_scope
+
+ set_multiplex_active(True)
+ tokens.append(set_secret_scope(scope or {}))
+ return tokens[-1]
+
+ yield install
+
+ from agent.secret_scope import reset_secret_scope, set_multiplex_active
+
+ for token in reversed(tokens):
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+
+@pytest.fixture
+def default_profile_env(monkeypatch):
+ """The default profile's YAML-to-env bridge output in os.environ."""
+ monkeypatch.setenv("NTFY_TOPIC", "default-topic")
+ monkeypatch.setenv("NTFY_SERVER_URL", "https://default.example.com")
+ monkeypatch.setenv("NTFY_PUBLISH_TOPIC", "default-out")
+
+
+class TestMultiplexProfileScope:
+
+ def test_secondary_extra_wins_over_default_profile_env(
+ self, multiplex_scope, default_profile_env
+ ):
+ """The secondary profile's own config.yaml extra is authoritative,
+ not the default profile's bridged topic/server/publish_topic."""
+ multiplex_scope()
+ cfg = PlatformConfig(
+ enabled=True,
+ extra={
+ "topic": "profile-topic",
+ "server": "https://profile.example.com",
+ "publish_topic": "profile-out",
+ },
+ )
+ adapter = NtfyAdapter(cfg)
+ assert adapter._topic == "profile-topic"
+ assert adapter._server == "https://profile.example.com"
+ assert adapter._publish_topic == "profile-out"
+
+ def test_secondary_missing_keys_fail_closed(
+ self, multiplex_scope, default_profile_env
+ ):
+ """Keys absent from the profile's own scope must NOT borrow the
+ default profile's bridged env values -- that would silently
+ subscribe/publish on the wrong topic."""
+ multiplex_scope()
+ adapter = NtfyAdapter(PlatformConfig(enabled=True, extra={}))
+ assert adapter._topic == ""
+ assert adapter._server == DEFAULT_SERVER
+ assert adapter._publish_topic == ""
+ # Nor may the registry auto-enable ntfy for this profile off the default's topic.
+ assert _env_enablement() is None
+ assert is_connected(PlatformConfig(enabled=True, extra={})) is False
+
diff --git a/tests/gateway/test_pending_drain_no_recursion.py b/tests/gateway/test_pending_drain_no_recursion.py
index a406c9d602..acedc6ae4b 100644
--- a/tests/gateway/test_pending_drain_no_recursion.py
+++ b/tests/gateway/test_pending_drain_no_recursion.py
@@ -112,7 +112,9 @@ async def test_in_band_drain_does_not_grow_stack():
# Drain the chain. Each turn schedules the next via the in-band
# drain block, so we wait until N handler runs have completed and
# the session has been released.
- for _ in range(400):
+ # 2000 * 0.01s = 20s budget: the old 4s budget flaked on loaded CI
+ # runners (11/12 turns completed; main run 33455779041).
+ for _ in range(2000):
if len(depths) >= N and sk not in adapter._active_sessions:
break
await asyncio.sleep(0.01)
@@ -278,7 +280,9 @@ async def test_late_arrival_drain_still_fires_when_no_in_band_drain():
await adapter.handle_message(_make_event(text="first"))
# Wait for the late-arrival drain task to finish the second event.
- for _ in range(400):
+ # 2000 * 0.01s = 20s budget: the old 4s budget flaked on loaded CI
+ # runners (11/12 turns completed; main run 33455779041).
+ for _ in range(2000):
if "late" in results and sk not in adapter._active_sessions:
break
await asyncio.sleep(0.01)
diff --git a/tests/gateway/test_pending_drain_race.py b/tests/gateway/test_pending_drain_race.py
index 10ca90dc75..479e769264 100644
--- a/tests/gateway/test_pending_drain_race.py
+++ b/tests/gateway/test_pending_drain_race.py
@@ -109,7 +109,7 @@ async def test_pending_drain_keeps_active_session_guard_live():
await adapter.handle_message(_make_event(text="M1"))
# Wait until M1 is actively running inside the handler.
- await asyncio.wait_for(first_started.wait(), timeout=1.0)
+ await asyncio.wait_for(first_started.wait(), timeout=5.0)
# Assert: session is active.
assert sk in adapter._active_sessions
@@ -126,7 +126,7 @@ async def test_pending_drain_keeps_active_session_guard_live():
try:
# Pause inside the handoff's typing cleanup. Production has already
# cleared the guard and has not yet transferred task ownership.
- await asyncio.wait_for(handoff_entered.wait(), timeout=2.0)
+ await asyncio.wait_for(handoff_entered.wait(), timeout=5.0)
# Across the drain transition, the Event object must be the SAME
# reference (not replaced, not deleted).
@@ -141,7 +141,7 @@ async def test_pending_drain_keeps_active_session_guard_live():
# Finish drain without relying on scheduler speed.
release_handoff.set()
- await asyncio.wait_for(second_processed.wait(), timeout=2.0)
+ await asyncio.wait_for(second_processed.wait(), timeout=5.0)
finally:
release_handoff.set()
await adapter.cancel_background_tasks()
@@ -190,7 +190,7 @@ async def test_finally_cleanup_drains_late_arrival_pending():
await adapter.handle_message(_make_event(text="M1"))
# Drain: wait for the late-drain task itself to process LATE.
- await asyncio.wait_for(late_processed.wait(), timeout=2.0)
+ await asyncio.wait_for(late_processed.wait(), timeout=5.0)
await adapter.cancel_background_tasks()
@@ -218,7 +218,7 @@ async def test_no_pending_cleans_up_normally():
# Await the task that owns this session rather than sampling cleanup after
# an arbitrary wall-clock delay.
owner_task = adapter._session_tasks[sk]
- await asyncio.wait_for(asyncio.shield(owner_task), timeout=2.0)
+ await asyncio.wait_for(asyncio.shield(owner_task), timeout=5.0)
assert sk not in adapter._active_sessions, (
"_active_sessions was not cleaned up after a normal turn with no pending"
diff --git a/tests/gateway/test_personality_routed_profile.py b/tests/gateway/test_personality_routed_profile.py
new file mode 100644
index 0000000000..9ad41b3628
--- /dev/null
+++ b/tests/gateway/test_personality_routed_profile.py
@@ -0,0 +1,45 @@
+"""#89161: a routed multiplex profile's personality must reach its turns.
+
+``GatewayRunner`` used to snapshot ``_ephemeral_system_prompt`` once at boot
+from the launch profile's config and hand that string to every routed turn,
+so a secondary profile's ``display.personality`` / ``agent.system_prompt`` never
+injected. ``_get_system_prompt_for_channel`` now resolves from the config of
+the profile currently in scope (``run_sync`` runs inside
+``_profile_runtime_scope``).
+"""
+
+from __future__ import annotations
+
+import gateway.run as gateway_run
+from gateway.config import Platform
+from gateway.run import GatewayRunner, _profile_runtime_scope
+
+
+def test_routed_profile_prompt_resolves_from_its_own_config(tmp_path, monkeypatch):
+ default_home = tmp_path / "default"
+ routed_home = tmp_path / "profiles" / "beta"
+ default_home.mkdir()
+ routed_home.mkdir(parents=True)
+ (default_home / "config.yaml").write_text("agent:\n system_prompt: DEFAULT-PERSONA\n")
+ (routed_home / "config.yaml").write_text(
+ "agent:\n system_prompt: BETA-PERSONA\n personalities:\n pirate: ARR\n"
+ )
+ monkeypatch.setattr(gateway_run, "_hermes_home", default_home)
+ monkeypatch.setenv("HERMES_HOME", str(default_home))
+ monkeypatch.delenv("HERMES_EPHEMERAL_SYSTEM_PROMPT", raising=False)
+
+ runner = object.__new__(GatewayRunner)
+ runner.config = None
+
+ with _profile_runtime_scope(routed_home):
+ assert runner._get_system_prompt_for_channel(Platform.TELEGRAM, "c") == "BETA-PERSONA"
+ assert runner._get_system_prompt_for_channel(Platform.TELEGRAM, "c") == "DEFAULT-PERSONA"
+
+ # /personality from the routed chat writes the routed profile and only it.
+ from hermes_cli.personality import persist_personality
+
+ with _profile_runtime_scope(routed_home):
+ assert persist_personality("pirate")
+ assert runner._get_system_prompt_for_channel(Platform.TELEGRAM, "c") == "ARR"
+ assert "pirate" not in (default_home / "config.yaml").read_text()
+ assert runner._get_system_prompt_for_channel(Platform.TELEGRAM, "c") == "DEFAULT-PERSONA"
diff --git a/tests/gateway/test_profile_resolution.py b/tests/gateway/test_profile_resolution.py
index 695b9c7b89..e79e0415ca 100644
--- a/tests/gateway/test_profile_resolution.py
+++ b/tests/gateway/test_profile_resolution.py
@@ -248,6 +248,69 @@ class TestGatewayRunnerInjection:
assert hasattr(BasePlatformAdapter, "gateway_runner")
assert BasePlatformAdapter.gateway_runner is None
+ def test_factory_binds_every_adapter_to_runner(self, monkeypatch):
+ """``_create_adapter`` binds the runner regardless of which branch
+ built the adapter (plugin registry OR built-in if/elif) — every
+ lifecycle path (startup, reconnect, secondary profiles) goes through
+ it, so this is the single seam that makes profile_routes reachable
+ for built-ins like Signal (#68332 / #70831)."""
+ from gateway.config import PlatformConfig
+
+ runner = object.__new__(GatewayRunner)
+ adapter = MagicMock(spec=BasePlatformAdapter)
+ monkeypatch.setattr(runner, "_instantiate_adapter", lambda platform, config: adapter)
+ assert runner._create_adapter(Platform.SIGNAL, PlatformConfig(enabled=True)) is adapter
+ assert adapter.gateway_runner is runner
+ monkeypatch.setattr(runner, "_instantiate_adapter", lambda platform, config: None)
+ assert runner._create_adapter(Platform.SIGNAL, PlatformConfig(enabled=True)) is None
+
+ @pytest.mark.asyncio
+ async def test_real_signal_factory_routes_inbound_group_event(self, monkeypatch):
+ """A factory-built (built-in) Signal adapter resolves profile_routes
+ for a real inbound envelope — fails on main where the Signal branch
+ returned a bare ``SignalAdapter(config)`` with no runner."""
+ from gateway.config import PlatformConfig
+
+ group_id = "test-signal-route"
+ monkeypatch.setenv("SIGNAL_GROUP_ALLOWED_USERS", group_id)
+ runner = object.__new__(GatewayRunner)
+ runner.config = GatewayConfig(
+ multiplex_profiles=True,
+ profile_routes=[
+ ProfileRoute(name="signal", platform="signal", profile="ops", chat_id=f"group:{group_id}"),
+ ],
+ )
+ adapter = runner._create_adapter(
+ Platform.SIGNAL,
+ PlatformConfig(enabled=True, extra={"http_url": "http://127.0.0.1:18080", "account": "+15555550123"}),
+ )
+ assert adapter is not None and adapter.gateway_runner is runner
+
+ captured = {}
+
+ async def capture_event(event):
+ captured["event"] = event
+
+ adapter.handle_message = capture_event
+ with patch(
+ "hermes_cli.profiles.profiles_to_serve",
+ return_value=[("default", Path("/profiles/default")), ("ops", Path("/profiles/ops"))],
+ ):
+ await adapter._handle_envelope({
+ "envelope": {
+ "sourceNumber": "+15555550124",
+ "sourceName": "Test Operator",
+ "timestamp": 1700000000000,
+ "dataMessage": {
+ "message": "diagnose the cluster",
+ "groupInfo": {"groupId": group_id, "groupName": "US East 7"},
+ },
+ },
+ })
+ source = captured["event"].source
+ assert source.profile == "ops"
+ assert build_session_key(source, profile=source.profile).startswith("agent:ops:")
+
# A concrete adapter we can instantiate without the full platform stack.
# ``build_source`` only reads ``self.platform`` and ``self.gateway_runner``, so a
diff --git a/tests/gateway/test_profile_routing.py b/tests/gateway/test_profile_routing.py
index 73934ac52d..37ebb8f69a 100644
--- a/tests/gateway/test_profile_routing.py
+++ b/tests/gateway/test_profile_routing.py
@@ -1,5 +1,7 @@
"""Tests for gateway/profile_routing.py — profile-based routing."""
+import json
+
import pytest
from gateway.profile_routing import (
ProfileRoute,
@@ -50,6 +52,40 @@ class TestParseProfileRoutes:
assert parse_profile_routes(None) == []
assert parse_profile_routes([]) == []
+ def test_coerces_yaml_native_int_ids_to_str(self):
+ # PyYAML loads unquoted snowflakes / negative Telegram ids as int;
+ # inbound SessionSource ids are str, so un-coerced routes never match.
+ routes = parse_profile_routes([
+ {"name": "server", "platform": "discord", "profile": "p",
+ "guild_id": 111, "chat_id": 222, "thread_id": 333},
+ {"name": "tg", "platform": "telegram", "profile": "p",
+ "chat_id": -1001234567890},
+ {"name": "platform-only", "platform": "discord", "profile": "p"},
+ ])
+ by_name = {r.name: r for r in routes}
+ assert (by_name["server"].guild_id, by_name["server"].chat_id,
+ by_name["server"].thread_id) == ("111", "222", "333")
+ assert match_profile_route(
+ routes, "discord", guild_id="111", chat_id="222", thread_id="333",
+ ).name == "server"
+ assert match_profile_route(
+ routes, "telegram", chat_id="-1001234567890",
+ ).name == "tg"
+ assert (by_name["platform-only"].guild_id, by_name["platform-only"].chat_id,
+ by_name["platform-only"].thread_id) == (None, None, None)
+
+ def test_non_int_numeric_ids_warn_instead_of_silently_coercing(self, caplog):
+ # #86470 nuance: float/bool stringify to values that can never match
+ # an inbound id, so surface the misconfiguration at load time.
+ with caplog.at_level("WARNING", logger="gateway.profile_routing"):
+ routes = parse_profile_routes([
+ {"name": "f", "platform": "discord", "profile": "p", "chat_id": 123.0},
+ {"name": "b", "platform": "discord", "profile": "p", "guild_id": True},
+ ])
+ assert {r.name for r in routes} == {"f", "b"}
+ assert match_profile_route(routes, "discord", chat_id="123") is None
+ assert sum("can never match" in rec.message for rec in caplog.records) == 2
+
class TestMatchProfileRoute:
@@ -106,3 +142,48 @@ class TestForumPostMatching:
parent_chat_id="forum_channel_123")
assert m is not None
assert m.profile == "forum_profile"
+
+
+class TestWhatsAppChatIdIdentityMatching:
+ """WhatsApp ``chat_id`` routes match across number / JID / LID forms (the
+ same alias canonicalization allowlists and session keys already use);
+ every other platform, and WhatsApp groups, stay exact-compare."""
+
+ PHONE = "15551234567"
+ LID = "999999999999999"
+
+ def _write_lid_mapping(self, tmp_path, monkeypatch):
+ mapping_dir = tmp_path / "platforms" / "whatsapp" / "session"
+ mapping_dir.mkdir(parents=True)
+ (mapping_dir / f"lid-mapping-{self.PHONE}.json").write_text(json.dumps(f"{self.LID}@lid"))
+ (mapping_dir / f"lid-mapping-{self.LID}_reverse.json").write_text(
+ json.dumps(f"{self.PHONE}@s.whatsapp.net")
+ )
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+
+ def test_number_route_matches_jid_and_mapped_lid_forms(self, tmp_path, monkeypatch):
+ self._write_lid_mapping(tmp_path, monkeypatch)
+ for platform in ("whatsapp", "whatsapp_cloud"):
+ r = ProfileRoute(name="owner", platform=platform, profile="owner", chat_id=self.PHONE)
+ assert r.matches(platform, chat_id=f"{self.PHONE}@s.whatsapp.net")
+ assert r.matches(platform, chat_id=f"{self.PHONE}:47@s.whatsapp.net")
+ assert r.matches(platform, chat_id=f"{self.LID}@lid")
+ # Alias fallback also applies to the thread-parent slot.
+ assert r.matches(platform, chat_id="thread-1", parent_chat_id=f"{self.LID}@lid")
+ assert not r.matches(platform, chat_id="15550001111@s.whatsapp.net")
+
+ def test_groups_and_other_platforms_stay_exact(self, tmp_path, monkeypatch):
+ self._write_lid_mapping(tmp_path, monkeypatch)
+ group = "120363012345678901@g.us"
+ owner = ProfileRoute(name="owner", platform="whatsapp", profile="owner", chat_id=self.PHONE)
+ assert not owner.matches("whatsapp", chat_id=group)
+ grp = ProfileRoute(name="grp", platform="whatsapp", profile="grp", chat_id=group)
+ assert grp.matches("whatsapp", chat_id=group)
+ assert not grp.matches("whatsapp", chat_id=f"{self.PHONE}@s.whatsapp.net")
+ # Stripping @g.us must never turn a group into a phone-identity match.
+ assert not ProfileRoute(
+ name="oops", platform="whatsapp", profile="owner", chat_id=group.split("@", 1)[0]
+ ).matches("whatsapp", chat_id=group)
+ tg = ProfileRoute(name="tg", platform="telegram", profile="owner", chat_id="640466638")
+ assert tg.matches("telegram", chat_id="640466638")
+ assert not tg.matches("telegram", chat_id="640466638@s.whatsapp.net")
diff --git a/tests/gateway/test_raft_adapter.py b/tests/gateway/test_raft_adapter.py
index 34a739f6e2..552c0da291 100644
--- a/tests/gateway/test_raft_adapter.py
+++ b/tests/gateway/test_raft_adapter.py
@@ -3,6 +3,7 @@
import asyncio
import json
import os
+from types import SimpleNamespace
from unittest.mock import AsyncMock, patch
import pytest
@@ -218,3 +219,104 @@ class TestRaftConfig:
assert os.environ["RAFT_PROFILE"] == "existing"
assert "Keeping RAFT_PROFILE=existing" in capsys.readouterr().out
+
+# ---------------------------------------------------------------------------
+# Multiplex secondary-profile scope (RAFT_PROFILE resolution)
+# ---------------------------------------------------------------------------
+#
+# _spawn_bridge, _env_enablement, and register()'s platform_hint all
+# previously read RAFT_PROFILE via raw os.environ.get unconditionally. Under
+# a multiplexed secondary profile, os.environ holds the DEFAULT profile's
+# YAML-to-env bridge output — a secondary profile with its own RAFT_PROFILE
+# (set only in its own .env, resolved via the installed secret scope) would
+# silently connect the bridge subprocess / CLI hint to the default profile's
+# external Raft workspace/agent identity instead of its own. Mirrors the
+# Buzz/SimpleX fix for #98738.
+
+@pytest.fixture
+def multiplex_scope():
+ """Install multiplex + a secondary-profile secret scope; restore after."""
+ tokens = []
+
+ def install(scope=None):
+ from agent.secret_scope import set_multiplex_active, set_secret_scope
+
+ set_multiplex_active(True)
+ tokens.append(set_secret_scope(scope or {}))
+ return tokens[-1]
+
+ yield install
+
+ from agent.secret_scope import reset_secret_scope, set_multiplex_active
+
+ for token in reversed(tokens):
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+
+@pytest.fixture
+def default_profile_env(monkeypatch):
+ """The default profile's YAML-to-env bridge output in os.environ."""
+ monkeypatch.setenv("RAFT_PROFILE", "default-profile-slug")
+
+
+class _FakeCtx:
+ """Minimal ``ctx`` capturing ``register_platform``'s kwargs."""
+
+ def __init__(self):
+ self.platform_kwargs = None
+
+ def register_platform(self, **kwargs):
+ self.platform_kwargs = kwargs
+
+ def register_hook(self, *args, **kwargs):
+ pass
+
+
+class TestMultiplexProfileScope:
+
+ def test_secondary_profile_uses_its_own_slug_and_never_borrows_default(
+ self, multiplex_scope, default_profile_env, monkeypatch
+ ):
+ """Bridge spawn, env-enablement and the register() hint all resolve
+ the secondary profile's own RAFT_PROFILE; with none of its own the
+ profile fails closed (bridge not spawned, not auto-enabled)."""
+ import plugins.platforms.raft.adapter as raft_mod
+
+ monkeypatch.setattr(raft_mod.shutil, "which", lambda name: "/usr/bin/raft")
+ spawned = []
+ monkeypatch.setattr(
+ raft_mod.subprocess, "Popen",
+ lambda cmd, **kwargs: spawned.append(cmd) or SimpleNamespace(pid=1),
+ )
+
+ multiplex_scope({"RAFT_PROFILE": "secondary-profile-slug"})
+ _make_adapter()._spawn_bridge(9999)
+ assert spawned[-1][:3] == ["/usr/bin/raft", "--profile", "secondary-profile-slug"]
+ assert _env_enablement() == {"enabled": True}
+ ctx = _FakeCtx()
+ register(ctx)
+ assert "--profile secondary-profile-slug" in ctx.platform_kwargs["platform_hint"]
+ assert "default-profile-slug" not in ctx.platform_kwargs["platform_hint"]
+
+ spawned.clear()
+ multiplex_scope({})
+ _make_adapter()._spawn_bridge(9999)
+ assert spawned == []
+ assert _env_enablement() is None
+
+ def test_default_profile_unscoped_keeps_env_precedence(
+ self, monkeypatch, default_profile_env
+ ):
+ """Multiplex ON but no scope (the DEFAULT profile constructs
+ unscoped): env is its own bridge output and still wins."""
+ from agent.secret_scope import set_multiplex_active
+
+ set_multiplex_active(True)
+ try:
+ assert _env_enablement() == {"enabled": True}
+ ctx = _FakeCtx()
+ register(ctx)
+ assert "--profile default-profile-slug" in ctx.platform_kwargs["platform_hint"]
+ finally:
+ set_multiplex_active(False)
diff --git a/tests/gateway/test_runner_startup_failures.py b/tests/gateway/test_runner_startup_failures.py
index 68fbfdf162..c3f906f171 100644
--- a/tests/gateway/test_runner_startup_failures.py
+++ b/tests/gateway/test_runner_startup_failures.py
@@ -3,11 +3,31 @@ from unittest.mock import AsyncMock
from gateway.config import GatewayConfig, Platform, PlatformConfig
from gateway.platforms.base import BasePlatformAdapter
-from gateway.restart import GATEWAY_FATAL_CONFIG_EXIT_CODE
+from gateway.restart import GATEWAY_FATAL_CONFIG_EXIT_CODE, is_global_startup_conflict
from gateway.run import GatewayRunner
from gateway.status import read_runtime_status
+@pytest.mark.parametrize(
+ "code, expected",
+ [
+ ("telegram-bot-token_lock", True), # BasePlatformAdapter._acquire_platform_lock
+ ("discord-bot-token_lock", True),
+ ("whatsapp-session_lock", True),
+ ("feishu_app_lock", True),
+ ("lock_conflict", True), # buzz / irc / line identity conflicts
+ ("telegram_connect_error", False),
+ ("telegram_auth_error", False),
+ ("relay_membership_required", False),
+ ("duplicate_credential", False),
+ ("", False),
+ (None, False),
+ ],
+)
+def test_is_global_startup_conflict_matches_lock_code_families(code, expected):
+ assert is_global_startup_conflict(code) is expected
+
+
class _RetryableFailureAdapter(BasePlatformAdapter):
def __init__(self):
super().__init__(PlatformConfig(enabled=True, token="***"), Platform.TELEGRAM)
@@ -443,3 +463,115 @@ async def test_start_gateway_propagates_fatal_config_exit_code(monkeypatch, tmp_
await start_gateway(config=GatewayConfig(), replace=False, verbosity=0)
assert exc_info.value.code == GATEWAY_FATAL_CONFIG_EXIT_CODE
+
+
+class _ForeignTokenLockAdapter(BasePlatformAdapter):
+ """Connects exactly like telegram/discord do: production
+ ``_acquire_platform_lock`` first, which emits ``{scope}_lock`` with
+ ``retryable=True`` (so a mid-run reconnect can recover, #54167)."""
+
+ def __init__(self):
+ super().__init__(PlatformConfig(enabled=True, token="***"), Platform.TELEGRAM)
+
+ async def connect(self, *, is_reconnect: bool = False) -> bool:
+ return self._acquire_platform_lock(
+ "telegram-bot-token", self.config.token, "Telegram bot token"
+ )
+
+ async def disconnect(self) -> None:
+ self._release_platform_lock()
+ self._mark_disconnected()
+
+ async def send(self, chat_id, content, reply_to=None, metadata=None):
+ raise NotImplementedError
+
+ async def get_chat_info(self, chat_id):
+ return {"id": chat_id}
+
+
+@pytest.mark.asyncio
+async def test_live_foreign_token_lock_at_startup_exits_ex_config(monkeypatch, tmp_path):
+ """Salvage of #83183 claim 1: a LIVE foreign holder of the bot token at
+ zero-connected startup is a single-writer conflict, not a transient blip.
+
+ ``_acquire_platform_lock`` deliberately emits the conflict retryable so a
+ *mid-run* reconnect can recover once the holder exits. The startup router
+ used to key solely off that flag, so the gateway stayed alive, deaf, and
+ retry-queued forever instead of exiting 78 (EX_CONFIG)."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ monkeypatch.setenv("HERMES_GATEWAY_LOCK_DIR", str(tmp_path / "locks"))
+ # A live foreign holder: acquire_scoped_lock reports (False, record).
+ monkeypatch.setattr(
+ "gateway.status.acquire_scoped_lock",
+ lambda scope, identity, metadata=None: (
+ False,
+ {"pid": 424242, "start_time": 1, "hermes_home": "/other/home", "profile": "other"},
+ ),
+ )
+ config = GatewayConfig(
+ platforms={Platform.TELEGRAM: PlatformConfig(enabled=True, token="***")},
+ sessions_dir=tmp_path / "sessions",
+ )
+ runner = GatewayRunner(config)
+ monkeypatch.setattr(
+ runner, "_create_adapter", lambda platform, platform_config: _ForeignTokenLockAdapter()
+ )
+
+ ok = await runner.start()
+
+ assert ok is True
+ assert runner.should_exit_cleanly is True
+ assert runner.exit_code == GATEWAY_FATAL_CONFIG_EXIT_CODE
+ assert runner._failed_platforms == {}
+ state = read_runtime_status()
+ assert state["gateway_state"] == "startup_failed"
+ assert state["platforms"]["telegram"]["state"] == "fatal"
+ assert state["platforms"]["telegram"]["error_code"] == "telegram-bot-token_lock"
+
+
+@pytest.mark.asyncio
+async def test_token_lock_plus_retryable_peer_stays_alive(monkeypatch, tmp_path):
+ """A lock conflict alongside a genuinely transient peer failure is the
+ NS-609 mixed mode: the lock is parked fatal, the peer keeps its retry, and
+ the gateway stays alive (no exit 78)."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ monkeypatch.setenv("HERMES_GATEWAY_LOCK_DIR", str(tmp_path / "locks"))
+ monkeypatch.setattr(
+ "gateway.status.acquire_scoped_lock",
+ lambda scope, identity, metadata=None: (False, {"pid": 424242, "start_time": 1}),
+ )
+ config = GatewayConfig(
+ platforms={
+ Platform.TELEGRAM: PlatformConfig(enabled=True, token="***"),
+ Platform.DISCORD: PlatformConfig(enabled=True, token="***"),
+ },
+ sessions_dir=tmp_path / "sessions",
+ )
+ runner = GatewayRunner(config)
+
+ class _DiscordBlip(_RetryableFailureAdapter):
+ def __init__(self):
+ BasePlatformAdapter.__init__(
+ self, PlatformConfig(enabled=True, token="***"), Platform.DISCORD
+ )
+
+ monkeypatch.setattr(
+ runner,
+ "_create_adapter",
+ lambda platform, cfg: (
+ _ForeignTokenLockAdapter() if platform is Platform.TELEGRAM else _DiscordBlip()
+ ),
+ )
+
+ ok = await runner.start()
+ try:
+ assert ok is True
+ assert runner.should_exit_cleanly is False
+ assert runner.exit_code is None
+ assert set(runner._failed_platforms) == {Platform.DISCORD}
+ state = read_runtime_status()
+ assert state["gateway_state"] == "running"
+ assert state["platforms"]["telegram"]["state"] == "fatal"
+ assert state["platforms"]["discord"]["state"] == "retrying"
+ finally:
+ await runner.stop()
diff --git a/tests/gateway/test_scale_to_zero_dashboard_client.py b/tests/gateway/test_scale_to_zero_dashboard_client.py
new file mode 100644
index 0000000000..93dbd73f1c
--- /dev/null
+++ b/tests/gateway/test_scale_to_zero_dashboard_client.py
@@ -0,0 +1,284 @@
+"""Scale-to-zero: an attached dashboard/desktop/TUI WS client counts as activity.
+
+Background (2026-09-02 fleet audit): 13 of 72 active opted-in prod instances
+flapped suspend -> proxy-wake every ~60s. The gateway only stamped
+``_last_inbound_at`` for messaging inbound, so it suspended under an open
+dashboard client; the client's reconnect loop re-poked the Fly-proxied hostname
+and autostart resumed the box. The dashboard runs in a separate process on
+hosted instances, so the signal crosses over as a marker-file mtime.
+
+These tests exercise the REAL seams — the pure helpers with a real temp
+HERMES_HOME, GatewayRunner._scale_to_zero_is_idle's composition, and
+tui_gateway.ws.handle_ws — rather than stubbing the collection under test
+(the F25 / #84327 lesson: bugs live at the call site, not in the pure predicate).
+"""
+from __future__ import annotations
+
+import asyncio
+import os
+import time
+
+import pytest
+
+from gateway import scale_to_zero as s2z
+from gateway.run import GatewayRunner
+
+
+@pytest.fixture
+def hermes_home(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ return tmp_path
+
+
+def _stat_denying(target):
+ """os.stat replacement that denies ONLY the marker path (stdlib callers unaffected)."""
+ real = os.stat
+
+ def _stat(path, *a, **k):
+ if os.fspath(path) == os.fspath(target):
+ raise PermissionError("nope")
+ return real(path, *a, **k)
+
+ return _stat
+
+
+# --- pure helpers -----------------------------------------------------------
+
+
+def test_heartbeat_path_lives_under_hermes_home_state(hermes_home):
+ p = s2z.dashboard_client_heartbeat_path()
+ assert p == hermes_home / "state" / "dashboard_clients.heartbeat"
+
+
+def test_last_seen_missing_marker_is_none_not_fail_awake(hermes_home):
+ # Steady state for a box nobody has the dashboard open on: must read as
+ # "no client", otherwise no instance would ever suspend.
+ assert s2z.dashboard_client_last_seen() is None
+
+
+def test_touch_creates_state_dir_and_marker(hermes_home):
+ assert s2z.touch_dashboard_client_heartbeat() is True
+ p = s2z.dashboard_client_heartbeat_path()
+ assert p.exists()
+ seen = s2z.dashboard_client_last_seen()
+ assert seen is not None and abs(seen - time.time()) < 5
+
+
+def test_last_seen_returns_raw_mtime_without_staleness_cutoff(hermes_home):
+ # No liveness cutoff here on purpose: is_idle decides recency. A 1h-old
+ # marker still reports its mtime; the gateway then finds it outside
+ # idle_timeout, same as an old _last_inbound_at.
+ s2z.touch_dashboard_client_heartbeat()
+ p = s2z.dashboard_client_heartbeat_path()
+ mtime = os.stat(p).st_mtime
+ assert s2z.dashboard_client_last_seen(now=mtime + 10) == mtime
+ assert s2z.dashboard_client_last_seen(now=mtime + 3600) == mtime
+
+
+def test_last_seen_future_mtime_is_clamped_to_now(hermes_home):
+ # A wall-clock step-back can leave the marker in the future; it must not
+ # extend the idle window past "now".
+ s2z.touch_dashboard_client_heartbeat()
+ p = s2z.dashboard_client_heartbeat_path()
+ future = time.time() + 600
+ os.utime(p, (future, future))
+ now = time.time()
+ assert s2z.dashboard_client_last_seen(now=now) == now
+
+
+def test_last_seen_unreadable_marker_fails_awake(hermes_home, monkeypatch):
+ s2z.touch_dashboard_client_heartbeat()
+
+ monkeypatch.setattr(s2z.os, "stat", _stat_denying(s2z.dashboard_client_heartbeat_path()))
+ now = 1_000_000.0
+ # Unreadable (not missing) => counts as activity right now.
+ assert s2z.dashboard_client_last_seen(now=now) == now
+
+
+def test_touch_never_raises(hermes_home, monkeypatch):
+ monkeypatch.setattr(s2z.os, "utime", lambda *a, **k: (_ for _ in ()).throw(OSError("ro")))
+ assert s2z.touch_dashboard_client_heartbeat() is False
+
+
+# --- gateway side: the idle predicate composition ---------------------------
+
+
+def _runner(monkeypatch, *, last_inbound_at):
+ r = GatewayRunner.__new__(GatewayRunner)
+ r._running = True
+ r._last_inbound_at = last_inbound_at
+ r._running_agents = {}
+ r._background_tasks = set()
+ r.adapters = {}
+ monkeypatch.setattr(r, "_scale_to_zero_idle_timeout_seconds", lambda: 120.0, raising=False)
+ monkeypatch.setattr(r, "_scale_to_zero_has_live_background_work", lambda: False, raising=False)
+ monkeypatch.setattr("cron.scheduler.get_running_job_ids", lambda: [])
+ return r
+
+
+def test_idle_without_dashboard_client_unchanged(hermes_home, monkeypatch):
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 600)
+ assert r._scale_to_zero_is_idle() is True
+
+
+def test_attached_dashboard_client_blocks_idle(hermes_home, monkeypatch):
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 600)
+ s2z.touch_dashboard_client_heartbeat()
+ assert r._scale_to_zero_is_idle() is False
+
+
+def test_client_gets_the_same_idle_grace_as_a_message(hermes_home, monkeypatch):
+ """Last WS frame 100s ago with a 120s idle_timeout => still inside the
+ window => NOT idle. This is the 2-minute-after-the-app-closes contract; an
+ earlier draft cut the marker off at 45s and suspended ~50s after
+ disconnect (observed live on staging)."""
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 600)
+ s2z.touch_dashboard_client_heartbeat()
+ p = s2z.dashboard_client_heartbeat_path()
+ old = time.time() - 100
+ os.utime(p, (old, old))
+ assert r._scale_to_zero_is_idle() is False
+
+
+def test_client_gone_longer_than_idle_timeout_is_idle(hermes_home, monkeypatch):
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 600)
+ s2z.touch_dashboard_client_heartbeat()
+ p = s2z.dashboard_client_heartbeat_path()
+ old = time.time() - 121
+ os.utime(p, (old, old))
+ assert r._scale_to_zero_is_idle() is True
+
+
+def test_marker_predating_gateway_inbound_does_not_matter(hermes_home, monkeypatch):
+ # Ancient marker from a client that left hours ago, gateway idle 600s.
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 600)
+ s2z.touch_dashboard_client_heartbeat()
+ p = s2z.dashboard_client_heartbeat_path()
+ old = time.time() - 7200
+ os.utime(p, (old, old))
+ assert r._scale_to_zero_is_idle() is True
+
+
+def test_dashboard_client_seen_recently_extends_inbound_clock(hermes_home, monkeypatch):
+ # Marker 30s old: inbound clock moves to 30s ago, which is
+ # inside the 120s window => not idle, even though the gateway's own
+ # _last_inbound_at is ancient.
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 600)
+ s2z.touch_dashboard_client_heartbeat()
+ p = s2z.dashboard_client_heartbeat_path()
+ t = time.time() - 30
+ os.utime(p, (t, t))
+ assert r._scale_to_zero_is_idle() is False
+
+
+def test_newer_gateway_inbound_wins_over_older_marker(hermes_home, monkeypatch):
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 5)
+ monkeypatch.setattr(r, "_scale_to_zero_idle_timeout_seconds", lambda: 10.0, raising=False)
+ s2z.touch_dashboard_client_heartbeat()
+ p = s2z.dashboard_client_heartbeat_path()
+ t = time.time() - 40
+ os.utime(p, (t, t))
+ # Picking the marker (40s > 10s) would read idle; the chat message (5s) wins.
+ assert r._scale_to_zero_is_idle() is False
+ # _last_inbound_at itself is not mutated by the read.
+ assert time.time() - r._last_inbound_at < 10
+
+
+def test_unreadable_marker_keeps_gateway_awake(hermes_home, monkeypatch):
+ r = _runner(monkeypatch, last_inbound_at=time.time() - 600)
+ s2z.touch_dashboard_client_heartbeat()
+ monkeypatch.setattr(s2z.os, "stat", _stat_denying(s2z.dashboard_client_heartbeat_path()))
+ assert r._scale_to_zero_is_idle() is False
+
+
+# --- dashboard side: the real handle_ws path touches the marker -------------
+
+
+def test_handle_ws_connect_touches_marker(hermes_home, monkeypatch):
+ from tui_gateway import server, ws as ws_mod
+
+ monkeypatch.setattr(server, "_start_backend_heartbeat_refresher", lambda: None)
+ monkeypatch.setattr(server, "_schedule_startup_orphan_sweep", lambda: None, raising=False)
+ monkeypatch.setattr(server, "resolve_skin", lambda: "default")
+ monkeypatch.setattr(server, "_ensure_skin_watcher", lambda: None)
+ monkeypatch.setattr(server, "register_live_transport", lambda *_a, **_k: None)
+ monkeypatch.setattr(server, "_WS_ORPHAN_REAP_GRACE_S", 0)
+ monkeypatch.setattr(ws_mod, "_dashboard_client_touched_at", 0.0)
+
+ class FakeWS:
+ async def accept(self):
+ pass
+
+ async def send_text(self, line):
+ pass
+
+ async def receive_text(self):
+ raise ws_mod._WebSocketDisconnect()
+
+ async def close(self):
+ pass
+
+ assert s2z.dashboard_client_last_seen() is None
+ asyncio.run(ws_mod.handle_ws(FakeWS()))
+ seen = s2z.dashboard_client_last_seen()
+ assert seen is not None and abs(seen - time.time()) < 5
+
+
+def test_handle_ws_inbound_frames_refresh_marker(hermes_home, monkeypatch):
+ from tui_gateway import server, ws as ws_mod
+
+ monkeypatch.setattr(server, "_start_backend_heartbeat_refresher", lambda: None)
+ monkeypatch.setattr(server, "_schedule_startup_orphan_sweep", lambda: None, raising=False)
+ monkeypatch.setattr(server, "resolve_skin", lambda: "default")
+ monkeypatch.setattr(server, "_ensure_skin_watcher", lambda: None)
+ monkeypatch.setattr(server, "register_live_transport", lambda *_a, **_k: None)
+ monkeypatch.setattr(server, "_WS_ORPHAN_REAP_GRACE_S", 0)
+ monkeypatch.setattr(ws_mod, "_dashboard_client_touched_at", 0.0)
+ # Disable the throttle so each frame is observable.
+ monkeypatch.setattr(ws_mod, "_DASHBOARD_CLIENT_TOUCH_MIN_INTERVAL_S", 0.0)
+
+ frames = ['{"jsonrpc":"2.0","method":"gateway.ping","id":1}'] * 2
+ touches: list[float] = []
+ real_touch = s2z.touch_dashboard_client_heartbeat
+
+ def _spy(path=None):
+ touches.append(time.time())
+ return real_touch(path)
+
+ monkeypatch.setattr(s2z, "touch_dashboard_client_heartbeat", _spy)
+
+ class FakeWS:
+ async def accept(self):
+ pass
+
+ async def send_text(self, line):
+ pass
+
+ async def receive_text(self):
+ if frames:
+ return frames.pop()
+ raise ws_mod._WebSocketDisconnect()
+
+ async def close(self):
+ pass
+
+ asyncio.run(ws_mod.handle_ws(FakeWS()))
+ # 1 on connect + 1 per inbound frame.
+ assert len(touches) == 3
+ assert s2z.dashboard_client_last_seen() is not None
+
+
+def test_note_activity_is_throttled(hermes_home, monkeypatch):
+ from tui_gateway import ws as ws_mod
+
+ calls = {"n": 0}
+ monkeypatch.setattr(
+ s2z, "touch_dashboard_client_heartbeat", lambda path=None: calls.__setitem__("n", calls["n"] + 1) or True
+ )
+ monkeypatch.setattr(ws_mod, "_dashboard_client_touched_at", 0.0)
+ ws_mod._note_dashboard_client_activity(force=True)
+ ws_mod._note_dashboard_client_activity()
+ ws_mod._note_dashboard_client_activity()
+ assert calls["n"] == 1
+ ws_mod._note_dashboard_client_activity(force=True)
+ assert calls["n"] == 2
diff --git a/tests/gateway/test_session_continuity_82616.py b/tests/gateway/test_session_continuity_82616.py
index d99bccfa77..7f9a498dc6 100644
--- a/tests/gateway/test_session_continuity_82616.py
+++ b/tests/gateway/test_session_continuity_82616.py
@@ -201,6 +201,28 @@ class TestPeerResolutionRecency:
class TestLoadTranscriptReroutes:
+ def test_load_transcript_raises_when_message_read_fails(self, tmp_path, monkeypatch):
+ from gateway.session import SessionStore, TranscriptReadError
+
+ from gateway.config import GatewayConfig
+
+ store = SessionStore(sessions_dir=tmp_path / "gw-failed-read", config=GatewayConfig())
+ db = store._db
+ assert db is not None
+ monkeypatch.setattr(db, "get_compression_tip", lambda _session_id: None)
+
+ def _malformed(_session_id, *, repair_alternation):
+ assert repair_alternation is True
+ raise RuntimeError("database disk image is malformed")
+
+ monkeypatch.setattr(db, "get_messages_as_conversation", _malformed)
+
+ with pytest.raises(TranscriptReadError) as exc_info:
+ store.load_transcript("existing-session")
+
+ assert exc_info.value.session_id == "existing-session"
+ assert isinstance(exc_info.value.__cause__, RuntimeError)
+
def test_load_transcript_follows_reroute_chain(self, tmp_path):
from gateway.session import SessionStore
diff --git a/tests/gateway/test_session_db_corrupt_fallback.py b/tests/gateway/test_session_db_corrupt_fallback.py
new file mode 100644
index 0000000000..9a6be9a114
--- /dev/null
+++ b/tests/gateway/test_session_db_corrupt_fallback.py
@@ -0,0 +1,70 @@
+"""Gateway SessionStore must divert, not retry forever, after structural corruption.
+
+Mirrors ``test_session_db_replaced_fallback.py``: once the SessionDB handle
+is quarantined (``StateDbCorruptError``) the pending transcript goes to the
+JSONL/spool fallback and no FTS surgery runs on the damaged file.
+"""
+
+import json
+import sqlite3
+
+from gateway.config import GatewayConfig
+from gateway.session import SessionStore
+
+
+class _MalformedConn:
+ def __init__(self, real_conn):
+ self._real = real_conn
+
+ def execute(self, *args, **kwargs):
+ raise sqlite3.DatabaseError("database disk image is malformed")
+
+ def __getattr__(self, name):
+ return getattr(self._real, name)
+
+
+def _assert_diverted(tmp_path, sid, needle):
+ pending = list((tmp_path / "pending_messages").glob("pending-*.json"))
+ assert pending, "expected pending_messages/pending-*.json spool"
+ spooled = False
+ for path in pending:
+ payload = json.loads(path.read_text(encoding="utf-8"))
+ message = (payload.get("data") or {}).get("message") or {}
+ if needle in str(message.get("content", "")):
+ spooled = True
+ break
+ assert spooled, f"{needle!r} missing from pending spool"
+ jsonl = tmp_path / "sessions" / f"{sid}.jsonl"
+ assert jsonl.is_file()
+ assert needle in jsonl.read_text(encoding="utf-8")
+
+
+def test_corrupt_state_db_diverts_pending_without_fts_rebuild(tmp_path, monkeypatch):
+ import hermes_state
+
+ live = tmp_path / "state.db"
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ monkeypatch.setattr(hermes_state, "DEFAULT_DB_PATH", live)
+
+ store = SessionStore(sessions_dir=tmp_path, config=GatewayConfig())
+ sid = "gw-corrupt"
+ store._db.create_session(session_id=sid, source="cli")
+ store.append_to_transcript(
+ sid, {"role": "user", "content": "before", "timestamp": 1.0}
+ )
+ real_conn = store._db._conn
+ store._db._conn = _MalformedConn(real_conn)
+ try:
+ store.append_to_transcript(
+ sid, {"role": "user", "content": "after-corrupt", "timestamp": 2.0}
+ )
+ assert store._db._db_corrupt is True
+ # No FTS surgery ran on either layer.
+ assert store._db._fts_enabled is True
+ assert store._db._fts_stale is False
+ assert store._fts_rebuild_attempted is False
+ assert sid not in store._dirty_transcripts
+ _assert_diverted(tmp_path, sid, "after-corrupt")
+ finally:
+ store._db._conn = real_conn
+ store.close_all_db_handles()
diff --git a/tests/gateway/test_session_hygiene_turnhold_adoption.py b/tests/gateway/test_session_hygiene_turnhold_adoption.py
new file mode 100644
index 0000000000..1b6c209cbd
--- /dev/null
+++ b/tests/gateway/test_session_hygiene_turnhold_adoption.py
@@ -0,0 +1,431 @@
+"""Regression tests for #97963 — hygiene turn-hold must not burn a
+watermark-fenced compression attempt.
+
+The 10s ``hygiene_max_turn_hold_seconds`` budget (#92318) releases the
+arriving user turn while a thinking summary model is still streaming its
+reasoning prefix. Before the fix, that release ALWAYS cancelled the commit
+fence, so 100% of the summary attempt (including the full thinking prefix)
+was discarded on every turn — auto-compression permanently failed for any
+deployment whose summary model thinks longer than the hold.
+
+The fix decouples the turn from the compression: when the worker's commit is
+watermark-fenced (rows appended after compression start survive its commit
+verbatim as concurrent tail), the detached worker KEEPS its commit admission
+and the summary is adopted at its own watermark-fenced commit boundary. The
+turn is still released at the same budget — the invariant pinned by
+``test_session_hygiene_turn_hold_budget_abandons_streaming_wait`` (#90845)
+is untouched (that test's worker is NOT watermark-fenced and still takes the
+cancel path).
+"""
+
+import asyncio
+import importlib
+import sys
+import threading
+import time
+import types
+from datetime import datetime
+from types import SimpleNamespace
+from unittest.mock import AsyncMock, MagicMock
+
+import pytest
+
+from gateway.config import GatewayConfig, Platform, PlatformConfig
+from gateway.platforms.base import BasePlatformAdapter, MessageEvent, SendResult
+from gateway.session import SessionEntry, SessionSource
+
+
+def _make_history(n_messages: int, content_size: int = 100) -> list:
+ history = []
+ content = "x" * content_size
+ for i in range(n_messages):
+ role = "user" if i % 2 == 0 else "assistant"
+ history.append({"role": role, "content": content, "timestamp": f"t{i}"})
+ return history
+
+
+class _CaptureAdapter(BasePlatformAdapter):
+ def __init__(self):
+ super().__init__(
+ PlatformConfig(enabled=True, token="fake-token"), Platform.TELEGRAM
+ )
+ self.sent = []
+
+ async def connect(self, *, is_reconnect: bool = False) -> bool:
+ return True
+
+ async def disconnect(self) -> None:
+ return None
+
+ async def send(self, chat_id, content, reply_to=None, metadata=None):
+ self.sent.append({"chat_id": chat_id, "content": content})
+ return SendResult(success=True, message_id="x")
+
+ async def get_chat_info(self, chat_id: str):
+ return {"id": chat_id}
+
+
+def _write_turnhold_config(tmp_path):
+ cfg_path = tmp_path / "config.yaml"
+ cfg_path.write_text(
+ "compression:\n"
+ " enabled: true\n"
+ " hygiene_timeout_seconds: 60\n"
+ " hygiene_total_ceiling_seconds: 600\n"
+ " hygiene_max_turn_hold_seconds: 0.3\n"
+ " hygiene_failure_cooldown_seconds: 120\n"
+ )
+
+
+def _build_runner(gateway_run, adapter, fake_db):
+ runner = object.__new__(gateway_run.GatewayRunner)
+ runner.config = GatewayConfig(
+ platforms={
+ Platform.TELEGRAM: PlatformConfig(enabled=True, token="fake-token")
+ }
+ )
+ runner.adapters = {Platform.TELEGRAM: adapter}
+ runner._voice_mode = {}
+ runner.hooks = SimpleNamespace(emit=AsyncMock(), loaded_hooks=False)
+ runner.session_store = MagicMock()
+ runner.session_store.get_or_create_session.return_value = SessionEntry(
+ session_key="agent:main:telegram:dm:12345",
+ session_id="sess-97963",
+ created_at=datetime.now(),
+ updated_at=datetime.now(),
+ platform=Platform.TELEGRAM,
+ chat_type="dm",
+ )
+ runner.session_store.load_transcript.return_value = _make_history(
+ 6, content_size=400
+ )
+ runner.session_store.has_any_sessions.return_value = True
+ runner.session_store.rewrite_transcript = MagicMock()
+ runner.session_store.append_to_transcript = MagicMock()
+ runner._running_agents = {}
+ runner._pending_messages = {}
+ runner._pending_approvals = {}
+ runner._session_db = SimpleNamespace(_db=fake_db)
+ runner._is_user_authorized = lambda _source: True
+ runner._set_session_env = lambda _context: None
+ runner._run_agent = AsyncMock(
+ return_value={
+ "final_response": "ok",
+ "messages": [],
+ "tools": [],
+ "history_offset": 0,
+ "last_prompt_tokens": 0,
+ }
+ )
+ return runner
+
+
+def _make_event():
+ return MessageEvent(
+ text="hello",
+ source=SessionSource(
+ platform=Platform.TELEGRAM,
+ chat_id="12345",
+ chat_type="dm",
+ user_id="12345",
+ ),
+ message_id="1",
+ )
+
+
+def _install_fakes(monkeypatch, gateway_run, tmp_path, agent_cls):
+ fake_dotenv = types.ModuleType("dotenv")
+ fake_dotenv.load_dotenv = lambda *args, **kwargs: None
+ monkeypatch.setitem(sys.modules, "dotenv", fake_dotenv)
+ fake_run_agent = types.ModuleType("run_agent")
+ fake_run_agent.AIAgent = agent_cls
+ monkeypatch.setitem(sys.modules, "run_agent", fake_run_agent)
+ monkeypatch.setattr(gateway_run, "_hermes_home", tmp_path)
+ monkeypatch.setattr(
+ gateway_run, "_resolve_runtime_agent_kwargs", lambda: {"api_key": "fake"}
+ )
+ monkeypatch.setattr(
+ "agent.model_metadata.get_model_context_length",
+ lambda *_args, **_kwargs: 100,
+ )
+
+
+async def _drain_deferred(runner, timeout=10.0):
+ tasks = getattr(runner, "_deferred_agent_cleanup_tasks", None) or set()
+ if tasks:
+ await asyncio.wait_for(
+ asyncio.gather(*list(tasks), return_exceptions=True), timeout
+ )
+
+
+@pytest.mark.asyncio
+async def test_turn_hold_keeps_admission_and_adopts_watermark_fenced_summary(
+ monkeypatch, tmp_path
+):
+ """A watermark-fenced worker keeps its commit admission at turn-hold
+ expiry; its late summary is ADOPTED (committed), not discarded — while
+ the turn itself is still released at the budget (#90845 invariant).
+ """
+ worker_started = threading.Event()
+ release_worker = threading.Event()
+ committed = threading.Event()
+ cleanup_done = threading.Event()
+ fake_db = MagicMock()
+ fake_db.get_compression_failure_cooldown.return_value = None
+
+ class FencedStreamingAgent:
+ last_instance = None
+
+ def __init__(self, **kwargs):
+ self.session_id = kwargs.get("session_id", "sess-97963")
+ self._session_db = kwargs.get("session_db")
+ self._last_compaction_in_place = False
+ self.context_compressor = SimpleNamespace(
+ bind_session_state=MagicMock(),
+ _last_compress_aborted=False,
+ _last_aux_model_failure_model=None,
+ )
+ self.shutdown_memory_provider = MagicMock()
+ self.close = MagicMock(side_effect=cleanup_done.set)
+ type(self).last_instance = self
+
+ def _compress_context(
+ self, messages, *_args, commit_fence=None, **_kwargs
+ ):
+ # Real compress_context marks the fence right after capturing
+ # the active-row watermark under the durable compression lock.
+ if commit_fence is not None:
+ commit_fence.mark_commit_watermark_fenced()
+ worker_started.set()
+ # Thinking-model shape: continuous progress, no commit yet —
+ # only the turn-hold budget can release the waiting turn.
+ # Bounded spin: a failing assertion before release_worker.set()
+ # must not leave this executor thread alive forever (pytest
+ # would hang at interpreter exit joining executor threads).
+ _spin_started = time.monotonic()
+ while not release_worker.is_set():
+ if time.monotonic() - _spin_started > 20:
+ return (messages, None)
+ if commit_fence is not None:
+ commit_fence.touch_progress()
+ time.sleep(0.01)
+ if commit_fence is not None and not commit_fence.begin_commit():
+ return (messages, None)
+ try:
+ self._session_db.archive_and_compact(
+ self.session_id,
+ [{"role": "assistant", "content": "summary"}],
+ watermark=6,
+ )
+ self._last_compaction_in_place = True
+ committed.set()
+ return ([{"role": "assistant", "content": "summary"}], None)
+ finally:
+ if commit_fence is not None:
+ commit_fence.finish_commit()
+
+ gateway_run = importlib.import_module("gateway.run")
+ _write_turnhold_config(tmp_path)
+ _install_fakes(monkeypatch, gateway_run, tmp_path, FencedStreamingAgent)
+
+ adapter = _CaptureAdapter()
+ runner = _build_runner(gateway_run, adapter, fake_db)
+
+ started = time.monotonic()
+ result = await asyncio.wait_for(runner._handle_message(_make_event()), timeout=15)
+ elapsed = time.monotonic() - started
+
+ # #90845/#92318 invariant intact: the turn is released at the budget.
+ assert result == "ok"
+ assert elapsed < 5.0, f"turn held for {elapsed:.1f}s despite the turn-hold budget"
+ assert worker_started.is_set()
+ assert runner._run_agent.await_count == 1
+
+ # (b) NO retry-after was armed while the attempt is still running —
+ # arming it would block the agent-side preflight from adopting the
+ # finished summary ("same-session cooldown active", #97963).
+ assert not fake_db.record_compression_failure_cooldown.called, (
+ "keep-admission path must not arm the retry-after while the "
+ "detached attempt is still running"
+ )
+
+ # The detached worker finishes late; its commit is ADMITTED (adoption),
+ # not refused — the summary attempt is no longer burned.
+ release_worker.set()
+ await asyncio.wait_for(asyncio.to_thread(committed.wait, 5), timeout=6)
+ assert committed.is_set(), (
+ "watermark-fenced worker must keep its commit admission after "
+ "turn-hold expiry (fence was cancelled — attempt burned)"
+ )
+ fake_db.archive_and_compact.assert_called_once()
+ # The commit went through the watermark-fenced path (concurrent tail
+ # rows above the watermark survive the compaction).
+ assert fake_db.archive_and_compact.call_args.kwargs.get("watermark") == 6
+
+ await _drain_deferred(runner)
+ await asyncio.wait_for(asyncio.to_thread(cleanup_done.wait, 5), timeout=6)
+ FencedStreamingAgent.last_instance.close.assert_called_once()
+
+ # Successful adoption resets the hygiene failure streak and still never
+ # advances it (the deferral is not a failure).
+ assert not fake_db.increment_hygiene_failure_streak.called
+ assert fake_db.reset_hygiene_failure_streak.called
+ # Deferral notice still reaches the user.
+ sent = [m["content"] for m in adapter.sent]
+ assert any(
+ "deferred" in c.lower() or "still streaming" in c.lower() for c in sent
+ ), f"turn-hold must send deferral notice, got: {sent}"
+
+
+@pytest.mark.asyncio
+async def test_turn_hold_kept_admission_arms_flat_retry_only_when_nothing_commits(
+ monkeypatch, tmp_path
+):
+ """If the kept-admission worker ends WITHOUT committing (summary failed
+ / attempt superseded), the flat non-escalating retry-after is restored so
+ sustained traffic does not spawn-and-abandon a compressor every turn —
+ but only AFTER the attempt truly ended, and without touching the streak.
+ """
+ worker_started = threading.Event()
+ release_worker = threading.Event()
+ fake_db = MagicMock()
+ fake_db.get_compression_failure_cooldown.return_value = None
+
+ class FencedNoCommitAgent:
+ def __init__(self, **kwargs):
+ self.session_id = kwargs.get("session_id", "sess-97963")
+ self._session_db = kwargs.get("session_db")
+ self._last_compaction_in_place = False
+ self.context_compressor = SimpleNamespace(
+ bind_session_state=MagicMock(),
+ _last_compress_aborted=False,
+ _last_aux_model_failure_model=None,
+ )
+ self.shutdown_memory_provider = MagicMock()
+ self.close = MagicMock()
+
+ def _compress_context(
+ self, messages, *_args, commit_fence=None, **_kwargs
+ ):
+ if commit_fence is not None:
+ commit_fence.mark_commit_watermark_fenced()
+ worker_started.set()
+ _spin_started = time.monotonic()
+ while not release_worker.is_set():
+ if time.monotonic() - _spin_started > 20:
+ return (messages, None)
+ if commit_fence is not None:
+ commit_fence.touch_progress()
+ time.sleep(0.01)
+ # Summary failed — return unchanged, no commit.
+ return (messages, None)
+
+ gateway_run = importlib.import_module("gateway.run")
+ _write_turnhold_config(tmp_path)
+ _install_fakes(monkeypatch, gateway_run, tmp_path, FencedNoCommitAgent)
+
+ adapter = _CaptureAdapter()
+ runner = _build_runner(gateway_run, adapter, fake_db)
+
+ result = await asyncio.wait_for(runner._handle_message(_make_event()), timeout=15)
+ assert result == "ok"
+ assert worker_started.is_set()
+ # While the attempt still runs: no cooldown, so preflight adoption
+ # stays possible.
+ assert not fake_db.record_compression_failure_cooldown.called
+
+ release_worker.set()
+ await _drain_deferred(runner)
+ # Let the done-callback fire.
+ for _ in range(100):
+ if fake_db.record_compression_failure_cooldown.called:
+ break
+ await asyncio.sleep(0.05)
+
+ # Nothing committed → flat retry-after restored (spacing), streak intact.
+ assert fake_db.record_compression_failure_cooldown.called, (
+ "a kept-admission attempt that ends without committing must restore "
+ "the flat turn-hold retry-after spacing"
+ )
+ args = fake_db.record_compression_failure_cooldown.call_args[0]
+ retry = args[1] - time.time()
+ assert retry <= 120, (
+ f"retry-after must stay flat (~60s), got {retry:.0f}s"
+ )
+ assert "turn-hold" in (args[2] or "")
+ assert not fake_db.increment_hygiene_failure_streak.called, (
+ "turn-hold deferral must never advance the failure streak"
+ )
+
+
+@pytest.mark.asyncio
+async def test_turn_hold_without_watermark_fence_still_cancels(
+ monkeypatch, tmp_path
+):
+ """A worker whose commit is NOT watermark-fenced (no session_db /
+ watermark capture failed) must still be cancelled at turn-hold expiry —
+ a late unfenced commit could clobber newer turns. Never worse than the
+ status quo. (Complements the pinned #90845 test, which exercises the
+ same path through the public surface.)
+ """
+ worker_started = threading.Event()
+ release_worker = threading.Event()
+ fake_db = MagicMock()
+ fake_db.get_compression_failure_cooldown.return_value = None
+
+ class UnfencedStreamingAgent:
+ def __init__(self, **kwargs):
+ self.session_id = kwargs.get("session_id", "sess-97963")
+ self._session_db = kwargs.get("session_db")
+ self._last_compaction_in_place = False
+ self.context_compressor = SimpleNamespace(
+ bind_session_state=MagicMock(),
+ _last_compress_aborted=False,
+ _last_aux_model_failure_model=None,
+ )
+ self.shutdown_memory_provider = MagicMock()
+ self.close = MagicMock()
+
+ def _compress_context(
+ self, messages, *_args, commit_fence=None, **_kwargs
+ ):
+ # Deliberately NO mark_commit_watermark_fenced().
+ worker_started.set()
+ _spin_started = time.monotonic()
+ while not release_worker.is_set():
+ if time.monotonic() - _spin_started > 20:
+ return (messages, None)
+ if commit_fence is not None:
+ commit_fence.touch_progress()
+ time.sleep(0.01)
+ if commit_fence is not None and not commit_fence.begin_commit():
+ return (messages, None)
+ try:
+ self._session_db.archive_and_compact(
+ self.session_id,
+ [{"role": "assistant", "content": "too late"}],
+ )
+ return ([{"role": "assistant", "content": "too late"}], None)
+ finally:
+ if commit_fence is not None:
+ commit_fence.finish_commit()
+
+ gateway_run = importlib.import_module("gateway.run")
+ _write_turnhold_config(tmp_path)
+ _install_fakes(monkeypatch, gateway_run, tmp_path, UnfencedStreamingAgent)
+
+ adapter = _CaptureAdapter()
+ runner = _build_runner(gateway_run, adapter, fake_db)
+
+ result = await asyncio.wait_for(runner._handle_message(_make_event()), timeout=15)
+ assert result == "ok"
+ assert worker_started.is_set()
+
+ release_worker.set()
+ await _drain_deferred(runner)
+ await asyncio.sleep(0.2)
+ # The unfenced late commit was refused — discard as before the fix.
+ fake_db.archive_and_compact.assert_not_called()
+ # Legacy path still records the flat retry-after immediately.
+ assert fake_db.record_compression_failure_cooldown.called
+ assert not fake_db.increment_hygiene_failure_streak.called
diff --git a/tests/gateway/test_shutdown_executor_quiesce.py b/tests/gateway/test_shutdown_executor_quiesce.py
new file mode 100644
index 0000000000..4f0233f0ec
--- /dev/null
+++ b/tests/gateway/test_shutdown_executor_quiesce.py
@@ -0,0 +1,266 @@
+"""Gateway shutdown quiesces its thread pool before closing state.db (#101093).
+
+``_shutdown_executor()`` used to run *after* the SessionDB close block in
+``_stop_impl``, and it never waited: ``cancel_futures`` only drops work that has
+not started, and cancelling the awaiting task does not stop the worker thread
+behind a ``run_in_executor`` future. So blocking DB work could still be running
+when ``SessionDB.close()`` checkpointed the WAL and let SQLite unlink the
+sidecar. The late write then reopens the handle (#94736) and mints a fresh WAL
+generation behind that checkpoint, leaving teardown to checkpoint the same file
+a second time from a connection the shutdown log never accounts for -- the
+close-time page-write damage in #101093 and the split WAL generation in #101064.
+
+The order is now: quiesce (bounded) -> close.
+"""
+
+import asyncio
+import concurrent.futures
+import threading
+import time
+from collections import OrderedDict
+
+import pytest
+
+import gateway.run as gw_mod
+
+
+class _FakeSessionDB:
+ """Records when the gateway closed it, on a shared event log."""
+
+ def __init__(self, events, name):
+ self._events = events
+ self._name = name
+
+ def close(self):
+ self._events.append(f"close:{self._name}")
+
+
+class _FakeGateway:
+ """Minimal stand-in with just enough state for ``stop()`` to run."""
+
+ def __init__(self, events):
+ self._events = events
+ self._running = True
+ self._draining = False
+ self._restart_requested = False
+ self._restart_detached = False
+ self._restart_via_service = False
+ self._stop_task = None
+ self._exit_cleanly = False
+ self._exit_with_failure = False
+ self._exit_reason = None
+ self._exit_code = None
+ self._restart_drain_timeout = 0.01
+ self._running_agents = {}
+ self._running_agents_ts = {}
+ self._agent_cache = OrderedDict()
+ self._agent_cache_lock = threading.Lock()
+ self.adapters = {}
+ self._background_tasks = set()
+ self._failed_platforms = []
+ self._shutdown_event = asyncio.Event()
+ self._pending_messages = {}
+ self._pending_approvals = {}
+ self._busy_ack_ts = {}
+ self._executor_lock = threading.Lock()
+ self._executor_closing = False
+ self._executor = concurrent.futures.ThreadPoolExecutor(
+ max_workers=2, thread_name_prefix="quiesce-test"
+ )
+ self._session_db = _FakeSessionDB(events, "session_db")
+ self.session_store = None
+
+ # -- shutdown collaborators the real stop() reaches into ---------------
+
+ def _running_agent_count(self):
+ return len(self._running_agents)
+
+ def _active_cron_job_count(self):
+ return 0
+
+ def _active_api_run_count(self):
+ return 0
+
+ def _update_runtime_status(self, *_a, **_kw):
+ pass
+
+ def _clear_plugin_message_injector(self):
+ pass
+
+ async def _run_in_executor_with_context(self, func, *args):
+ return func(*args)
+
+ async def _cleanup_agent_resources_off_loop(self, agent, *, context=""):
+ self._cleanup_agent_resources(agent)
+
+ async def _notify_active_sessions_of_shutdown(self):
+ pass
+
+ async def _cancel_secondary_profile_reconnect_tasks(self):
+ pass
+
+ async def _drain_active_agents(self, timeout, cron_timeout=None):
+ return {}, False
+
+ async def _finalize_shutdown_agents(self, agents):
+ pass
+
+ def _cleanup_agent_resources(self, agent):
+ pass
+
+ def _evict_cached_agent(self, key):
+ pass
+
+ def _release_running_agent_state(self, session_key, **_kwargs):
+ self._running_agents.pop(session_key, None)
+ self._running_agents_ts.pop(session_key, None)
+ return False
+
+ def close_all_session_db_handles(self):
+ pass
+
+
+@pytest.mark.asyncio
+async def test_running_executor_work_finishes_before_session_db_close():
+ """A future already running when stop() begins writes before the close."""
+ events = []
+ gw = _FakeGateway(events)
+ started = threading.Event()
+
+ def _blocking_db_write():
+ started.set()
+ # Longer than the rest of the shutdown tail (~0.4s), shorter than the
+ # 2s quiesce ceiling: without the wait the close lands first.
+ time.sleep(1.0)
+ events.append("worker_write")
+
+ future = gw._executor.submit(_blocking_db_write)
+ assert started.wait(2.0), "worker never started"
+
+ await gw_mod.GatewayRunner.stop(gw)
+ future.result(timeout=5)
+
+ assert "worker_write" in events, "worker never ran"
+ assert "close:session_db" in events, "SessionDB was never closed"
+ assert events.index("worker_write") < events.index("close:session_db"), (
+ f"state.db was closed while a worker was still writing: {events}"
+ )
+
+
+@pytest.mark.asyncio
+async def test_executor_refuses_new_work_before_session_db_close():
+ """``_executor_closing`` is set before the close, so no fresh pool is minted."""
+ events = []
+ gw = _FakeGateway(events)
+
+ real_close = gw._session_db.close
+
+ def _close_and_probe():
+ # The flag must already be set by the time the DB is closed, or a
+ # coroutine reaching _get_executor() here would spin up a new pool and
+ # run more blocking DB work against the handle being torn down.
+ events.append(f"closing_flag:{gw._executor_closing}")
+ real_close()
+
+ gw._session_db.close = _close_and_probe
+
+ await gw_mod.GatewayRunner.stop(gw)
+
+ assert "closing_flag:True" in events, events
+ with pytest.raises(RuntimeError):
+ gw_mod.GatewayRunner._get_executor(gw)
+
+
+@pytest.mark.asyncio
+async def test_stuck_worker_skips_the_session_db_close():
+ """A worker that outlives the quiesce budget must not be raced by close().
+
+ Reporting the live worker with a "may reopen state.db" warning is not
+ enough: the close()/checkpoint itself is the operation that raced the
+ late write and produced the wrong-page-number corruption in #101093,
+ so the close path has to be skipped whenever a worker survives the
+ budget, not merely logged around.
+ """
+ events = []
+ gw = _FakeGateway(events)
+ release = threading.Event()
+ started = threading.Event()
+
+ def _stuck():
+ started.set()
+ release.wait(5.0)
+ events.append("worker_write")
+
+ future = gw._executor.submit(_stuck)
+ assert started.wait(2.0), "worker never started"
+
+ # Force the quiesce budget to 0 so the worker is deterministically still
+ # alive when `_shutdown_executor` returns, without sleeping through the
+ # real 2s ceiling.
+ original_timeout = gw_mod._EXECUTOR_QUIESCE_TIMEOUT
+ gw_mod._EXECUTOR_QUIESCE_TIMEOUT = 0.0
+ try:
+ await gw_mod.GatewayRunner.stop(gw)
+ finally:
+ gw_mod._EXECUTOR_QUIESCE_TIMEOUT = original_timeout
+
+ assert "close:session_db" not in events, (
+ f"SessionDB was closed/checkpointed while a worker was still alive: {events}"
+ )
+
+ release.set()
+ future.result(timeout=5)
+ assert "worker_write" in events, "worker never finished"
+
+
+def test_shutdown_executor_defaults_to_no_wait():
+ """The no-argument call keeps the historical fire-and-forget contract."""
+ gw = _FakeGateway([])
+ release = threading.Event()
+ started = threading.Event()
+
+ def _slow():
+ started.set()
+ release.wait(5.0)
+
+ future = gw._executor.submit(_slow)
+ assert started.wait(2.0)
+
+ began = time.monotonic()
+ still_live = gw_mod.GatewayRunner._shutdown_executor(gw)
+ elapsed = time.monotonic() - began
+
+ assert elapsed < 0.5, f"default call waited {elapsed:.2f}s"
+ assert still_live == 1
+ release.set()
+ future.result(timeout=5)
+
+
+def test_shutdown_executor_reports_a_stuck_worker():
+ """A worker that outlives the budget is reported, not waited on forever."""
+ gw = _FakeGateway([])
+ release = threading.Event()
+ started = threading.Event()
+
+ def _stuck():
+ started.set()
+ release.wait(5.0)
+
+ future = gw._executor.submit(_stuck)
+ assert started.wait(2.0)
+
+ began = time.monotonic()
+ still_live = gw_mod.GatewayRunner._shutdown_executor(gw, drain_timeout=0.2)
+ elapsed = time.monotonic() - began
+
+ assert still_live == 1
+ assert 0.15 <= elapsed < 2.0, f"budget not honoured: {elapsed:.2f}s"
+ release.set()
+ future.result(timeout=5)
+
+
+def test_shutdown_executor_without_executor_returns_zero():
+ gw = _FakeGateway([])
+ gw._executor.shutdown(wait=True)
+ gw._executor = None
+ assert gw_mod.GatewayRunner._shutdown_executor(gw, drain_timeout=1.0) == 0
diff --git a/tests/gateway/test_shutdown_flush.py b/tests/gateway/test_shutdown_flush.py
index efe6f59572..f966ea896d 100644
--- a/tests/gateway/test_shutdown_flush.py
+++ b/tests/gateway/test_shutdown_flush.py
@@ -11,6 +11,7 @@ import pytest
from gateway.shutdown_flush import (
_serialise_value,
+ flush_overflow_to_file,
flush_pending_to_file,
recover_pending_to_db,
)
@@ -168,3 +169,72 @@ def test_get_flush_dir_uses_get_hermes_home(tmp_path, monkeypatch):
assert result == tmp_path / "pending_messages"
+
+
+# ── FIFO overflow tail durability (#99882) ─────────────────────────────
+
+
+def _overflow_event(text: str, session_id: str = "20260901_120000_fifo"):
+ event = MagicMock()
+ event.text = text
+ event.session_id = session_id
+ event.platform = "telegram"
+ event.sender_id = "1572286605"
+ event.sender_name = "tester"
+ event.reply_to = None
+ event.media = None
+ event.raw_event = None
+ return event
+
+
+def test_flush_overflow_writes_one_payload_per_event_in_arrival_order(tmp_path, monkeypatch):
+ """The FIFO tail (queued_events) must survive shutdown like the slot does.
+
+ Each overflow entry is its own recover_pending_to_db-compatible payload,
+ with ``seq`` recording arrival order inside the session.
+ """
+ flush_dir = _make_flush_dir(tmp_path)
+ monkeypatch.setattr("gateway.shutdown_flush._get_flush_dir", lambda: flush_dir)
+
+ count = flush_overflow_to_file(
+ {
+ "agent:main:telegram:dm:1": [
+ _overflow_event("follow-up B"),
+ _overflow_event("follow-up C"),
+ ],
+ "agent:main:telegram:dm:2": [],
+ "": [_overflow_event("keyless — skipped")],
+ },
+ reason="shutdown",
+ )
+ assert count == 2
+ payloads = sorted(
+ (json.loads(f.read_text(encoding="utf-8")) for f in flush_dir.glob("*.json")),
+ key=lambda p: p["seq"],
+ )
+ assert [p["data"]["text"] for p in payloads] == ["follow-up B", "follow-up C"]
+ assert {p["session_key"] for p in payloads} == {"agent:main:telegram:dm:1"}
+ assert all(p["reason"] == "shutdown" for p in payloads)
+
+
+def test_flushed_overflow_is_replayed_by_recover_pending_to_db(tmp_path, monkeypatch):
+ """Round-trip: overflow payloads use the slot-flush shape, so the existing
+ startup recovery inserts them as user rows without any new reader."""
+ flush_dir = _make_flush_dir(tmp_path)
+ monkeypatch.setattr("gateway.shutdown_flush._get_flush_dir", lambda: flush_dir)
+ flush_overflow_to_file({"agent:main:telegram:dm:1": [_overflow_event("orphan-1")]})
+
+ db = MagicMock()
+ recovered = recover_pending_to_db(session_db=db)
+ assert recovered == 1
+ db.append_message.assert_called_once()
+ kwargs = db.append_message.call_args.kwargs
+ assert kwargs["session_id"] == "20260901_120000_fifo"
+ assert kwargs["role"] == "user"
+ assert kwargs["content"] == "orphan-1"
+ assert list(flush_dir.glob("*.json")) == []
+
+
+def test_flush_overflow_noop_on_empty():
+ assert flush_overflow_to_file({}) == 0
+ assert flush_overflow_to_file({"k": []}) == 0
diff --git a/tests/gateway/test_shutdown_watchdog.py b/tests/gateway/test_shutdown_watchdog.py
index b46437383b..ec0e93e5c3 100644
--- a/tests/gateway/test_shutdown_watchdog.py
+++ b/tests/gateway/test_shutdown_watchdog.py
@@ -8,11 +8,18 @@ structurally unable to fire. These tests pin the out-of-loop backstop
from __future__ import annotations
import asyncio
+import contextlib
import json
+import logging
+import os
+import shutil
+import tempfile
import threading
import time
+from pathlib import Path
from unittest.mock import patch
+import gateway.shutdown_watchdog as shutdown_watchdog_module
import pytest
from gateway.shutdown_watchdog import (
@@ -66,3 +73,105 @@ def test_arm_shutdown_watchdog_fires_with_dump_and_exit(tmp_path):
assert get_shutdown_watchdog_dump_path(tmp_path).name == "gateway-shutdown-watchdog.log"
+
+
+async def _run_heartbeat_until_payload(tmp_path, timeout_s=10.0):
+ """Run loop_heartbeat_forever as a task until a heartbeat payload exists.
+
+ Returns (task, payload). Cancels the task and awaits it (suppressing
+ CancelledError) before returning so the tick server is closed cleanly.
+ """
+ task = asyncio.ensure_future(
+ loop_heartbeat_forever(interval_s=1.0, home=tmp_path)
+ )
+ heartbeat_path = get_loop_heartbeat_path(tmp_path)
+ deadline = time.monotonic() + timeout_s
+ payload = None
+ while time.monotonic() < deadline:
+ if heartbeat_path.is_file():
+ with contextlib.suppress(OSError, json.JSONDecodeError):
+ payload = json.loads(heartbeat_path.read_text(encoding="utf-8"))
+ if payload:
+ break
+ payload = None
+ await asyncio.sleep(0.05)
+ task.cancel()
+ with contextlib.suppress(asyncio.CancelledError):
+ await task
+ if payload is None:
+ pytest.fail(
+ f"heartbeat payload did not appear at {heartbeat_path} within "
+ f"{timeout_s}s"
+ )
+ return payload
+
+
+@pytest.fixture()
+def short_home():
+ """Short HERMES_HOME for tests that bind a real AF_UNIX socket.
+
+ pytest's tmp_path nests deep enough on CI runners / macOS that
+ ``state/gateway.loop-tick..sock`` exceeds the sockaddr_un limit and
+ bind() raises ``OSError: AF_UNIX path too long`` — which the producer
+ swallows into ``loop_tick_socket=False``, falsely failing the POSIX arm
+ test. Same pattern as tests/hermes_cli/test_update_wedged_gateway.py.
+ """
+ path = Path(tempfile.mkdtemp(prefix="hsw-"))
+ try:
+ yield path
+ finally:
+ shutil.rmtree(path, ignore_errors=True)
+
+
+@pytest.mark.asyncio
+async def test_loop_tick_witness_arms_over_tcp_on_windows(
+ short_home, caplog, monkeypatch
+):
+ """Non-POSIX never touches AF_UNIX; the witness arms over TCP loopback."""
+ tmp_path = short_home
+ # Pretend the platform is Windows as seen from the module under test.
+ # A plain monkeypatch of the global os.name would flip pathlib.Path
+ # dispatch (Path.__new__ reads os.name at runtime) and crash pytest's
+ # own tmp-dir machinery, so swap the module's `os` binding for a proxy
+ # whose `.name` is "nt" and which delegates everything else to real os.
+ class _WindowsOsProxy:
+ name = "nt"
+
+ def __getattr__(self, item):
+ return getattr(os, item)
+
+ monkeypatch.setattr(shutdown_watchdog_module, "os", _WindowsOsProxy())
+
+ start_unix_server_calls = []
+
+ def _forbid_start_unix_server(*args, **kwargs):
+ start_unix_server_calls.append((args, kwargs))
+ raise AssertionError("start_unix_server must not be called on non-POSIX")
+
+ with patch.object(
+ shutdown_watchdog_module.asyncio,
+ "start_unix_server",
+ side_effect=_forbid_start_unix_server,
+ ), caplog.at_level(logging.DEBUG, logger="gateway.shutdown_watchdog"):
+ payload = await _run_heartbeat_until_payload(tmp_path)
+
+ # (a) the AF_UNIX server was never attempted
+ assert start_unix_server_calls == []
+ # (b) no warning about an unavailable tick socket
+ assert not [
+ r
+ for r in caplog.records
+ if r.levelname == "WARNING"
+ and "Loop tick socket unavailable" in r.getMessage()
+ ]
+ # (c) the witness is armed over TCP and the port is published
+ assert payload["loop_tick_socket"] is True
+ assert 0 < int(payload["loop_tick_tcp_port"]) <= 65535
+ # (d) the POSIX socket node was never created
+ assert not list(tmp_path.glob("**/gateway.loop-tick.*.sock"))
+
+
+@pytest.mark.asyncio
+async def test_loop_tick_witness_arms_on_posix(short_home):
+ payload = await _run_heartbeat_until_payload(short_home)
+ assert payload["loop_tick_socket"] is True
diff --git a/tests/gateway/test_silent_partial_delivery_95382.py b/tests/gateway/test_silent_partial_delivery_95382.py
new file mode 100644
index 0000000000..1f41063771
--- /dev/null
+++ b/tests/gateway/test_silent_partial_delivery_95382.py
@@ -0,0 +1,481 @@
+"""Regression coverage for #95382 / #98552 — silent partial delivery.
+
+#95382 (Discord): the WebSocket drops after the first streaming edit (which
+carried only a prefix). The consumer's delivery flags could suppress the
+gateway's normal final send even though no recorded payload proved the
+COMPLETE ``final_response`` ever reached the platform; and when the normal
+final send then failed on the dead transport, the failure was recorded with a
+non-retryable error string, so the delivery-obligation ledger's reconnect
+sweep never replayed it — the turn's output was silently lost until a full
+process restart.
+
+#98552 (Telegram): a finalize path that sets ``final_content_delivered=True``
+without recording what was actually delivered produced the same false
+positive on a 624-char message truncated at 333 chars.
+
+Class contract under test:
+
+1. ``delivered_final_matches`` judges a payload-less delivery flag against
+ the FINAL content (via ``has_delivered_text``) instead of returning the
+ legacy-trust ``None`` — only the explicitly-marked ambiguous-timeout path
+ keeps legacy trust.
+2. Every flag-setting site records its delivered payload (fresh-final and
+ the optimistic native finalize were the record-less holdouts).
+3. Discord transport-shaped send failures are classified as
+ ``send_path_degraded`` (retryable) so the ledger reconnect sweep can
+ replay the stranded final response.
+
+Boundary tests drive the REAL ``GatewayRunner._run_agent`` with a live
+``GatewayStreamConsumer`` (pattern from test_stale_finalize_suppression.py).
+"""
+
+import asyncio
+import importlib
+import sys
+import types
+from types import SimpleNamespace
+
+import pytest
+
+from gateway.config import Platform, PlatformConfig, StreamingConfig
+from gateway.platforms.base import BasePlatformAdapter, SendResult
+from gateway.session import SessionSource
+from gateway.stream_consumer import GatewayStreamConsumer, StreamConsumerConfig
+
+
+STREAMED_PREFIX = "Deploy summary: 713 items published (578 as of 08-26"
+MISSING_TAIL = ", another 135 over the past 4 days). All checks green."
+FULL_RESPONSE = STREAMED_PREFIX + MISSING_TAIL
+
+
+# ---------------------------------------------------------------------------
+# Unit coverage — delivered_final_matches tri-state tightening
+# ---------------------------------------------------------------------------
+
+
+def _make_consumer(adapter=None, **overrides):
+ adapter = adapter or SimpleNamespace(
+ MAX_MESSAGE_LENGTH=4096,
+ splits_long_messages=True,
+ )
+ consumer = GatewayStreamConsumer.__new__(GatewayStreamConsumer)
+ consumer.adapter = adapter
+ consumer.chat_id = "c1"
+ consumer.cfg = StreamConsumerConfig(cursor="▉")
+ consumer._final_response_sent = True
+ consumer._final_content_delivered = True
+ consumer._delivered_final_text = None
+ consumer._turn_split_delivery = False
+ consumer._delivery_ambiguous = False
+ consumer._delivered_commentary_texts = []
+ consumer._delivered_segment_texts = []
+ consumer._last_sent_text = ""
+ consumer._accumulated = ""
+ consumer._stream_ledger = ""
+ consumer._initial_reply_to_id = None
+ consumer.metadata = None
+ consumer._already_sent = True
+ for key, value in overrides.items():
+ setattr(consumer, key, value)
+ return consumer
+
+
+class TestDeliveredFinalMatchesRecordless:
+ def test_recordless_flag_with_partial_visible_is_mismatch(self):
+ """#95382 core: flag set, no record, visible text is only a prefix —
+ the matcher must return False (recover), not None (legacy trust)."""
+ consumer = _make_consumer(_last_sent_text=STREAMED_PREFIX + "▉")
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is False
+
+ def test_recordless_flag_with_no_visible_text_is_mismatch(self):
+ """Flag set but nothing visibly delivered at all — mismatch."""
+ consumer = _make_consumer()
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is False
+
+ def test_recordless_flag_with_equal_visible_text_matches(self):
+ """Duplicate-suppression control: the visible text IS the final
+ answer — suppression must be retained (True)."""
+ consumer = _make_consumer(_last_sent_text=FULL_RESPONSE + "▉")
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is True
+
+ def test_ambiguous_timeout_keeps_legacy_trust(self):
+ """The explicitly-marked ambiguous full-final timeout is the ONE
+ record-less case that keeps legacy trust (None) — re-sending there
+ risks a duplicate, not a recovery."""
+ consumer = _make_consumer(_delivery_ambiguous=True)
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is None
+
+ def test_recorded_payload_still_wins_over_visible(self):
+ consumer = _make_consumer(
+ _delivered_final_text=FULL_RESPONSE,
+ _last_sent_text="something else entirely",
+ )
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is True
+
+ def test_payloadless_split_still_refuses_trust(self):
+ """#78541 behavior preserved by the tightening."""
+ consumer = _make_consumer(_turn_split_delivery=True)
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is False
+
+ def test_delivered_segment_text_matches(self):
+ """A segment-finalized delivery of the final text still suppresses."""
+ consumer = _make_consumer(
+ _delivered_segment_texts=[FULL_RESPONSE],
+ )
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is True
+
+
+class TestFlagSettingSitesRecordPayload:
+ @pytest.mark.asyncio
+ async def test_fresh_final_records_delivered_payload(self):
+ """_try_fresh_final must record what it sent (#95382 holdout)."""
+
+ class FreshAdapter:
+ MAX_MESSAGE_LENGTH = 4096
+ splits_long_messages = True
+
+ def __init__(self):
+ self.sent = []
+
+ async def send(self, chat_id, content, reply_to=None, metadata=None):
+ self.sent.append(content)
+ return SendResult(success=True, message_id="m-1")
+
+ adapter = FreshAdapter()
+ consumer = _make_consumer(adapter)
+ consumer._final_response_sent = False
+ consumer._final_content_delivered = False
+ consumer._preview_message_ids = set()
+ consumer._message_id = "m-0"
+ consumer._message_created_ts = None
+ consumer.metadata = None
+ consumer._already_sent = False
+
+ ok = await consumer._try_fresh_final(STREAMED_PREFIX, is_turn_final=True)
+ assert ok is True
+ assert consumer._final_response_sent is True
+ # The recorded payload lets the gateway detect a stale fresh-final.
+ assert consumer._delivered_final_text is not None
+ assert STREAMED_PREFIX in consumer._delivered_final_text
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is False
+ assert consumer.delivered_final_matches(STREAMED_PREFIX) is True
+
+
+# ---------------------------------------------------------------------------
+# Gateway-boundary regression — record-less flags must not swallow the reply
+# ---------------------------------------------------------------------------
+
+
+class CaptureAdapter(BasePlatformAdapter):
+ def __init__(self, platform=Platform.DISCORD):
+ super().__init__(PlatformConfig(enabled=True, token="***"), platform)
+ self.sent = []
+ self.edits = []
+ self._next_id = 0
+ self.fail_edits = False
+
+ async def connect(self, *, is_reconnect: bool = False) -> bool:
+ return True
+
+ async def disconnect(self) -> None:
+ return None
+
+ def _mint_id(self) -> str:
+ self._next_id += 1
+ return f"m-{self._next_id}"
+
+ async def send(self, chat_id, content, reply_to=None, metadata=None) -> SendResult:
+ self.sent.append({"chat_id": chat_id, "content": content})
+ return SendResult(success=True, message_id=self._mint_id())
+
+ async def edit_message(
+ self, chat_id, message_id, content, *, finalize: bool = False, metadata=None
+ ) -> SendResult:
+ if self.fail_edits:
+ return SendResult(success=False, error="websocket closed")
+ self.edits.append(
+ {"message_id": message_id, "content": content, "finalize": finalize}
+ )
+ return SendResult(success=True, message_id=message_id)
+
+ async def send_typing(self, chat_id, metadata=None) -> None:
+ return None
+
+ async def stop_typing(self, chat_id) -> None:
+ return None
+
+ async def get_chat_info(self, chat_id: str):
+ return {"id": chat_id}
+
+
+class PrefixOnlyAgent:
+ """Streams only a prefix; the completed response has a longer tail."""
+
+ def __init__(self, **kwargs):
+ self.stream_delta_callback = kwargs.get("stream_delta_callback")
+ self.tools = []
+
+ def run_conversation(self, message, conversation_history=None, task_id=None):
+ if self.stream_delta_callback:
+ self.stream_delta_callback(STREAMED_PREFIX)
+ return {
+ "final_response": FULL_RESPONSE,
+ "response_previewed": False,
+ "messages": [],
+ "api_calls": 1,
+ }
+
+
+class _RecordlessFlagConsumer(GatewayStreamConsumer):
+ """Sabotage subclass: models the #95382/#98552 incident state.
+
+ After a normal drain, claim final delivery via the flags but scrub the
+ recorded payload — the pre-fix gateway read matcher ``None`` as legacy
+ trust and suppressed the corrective send even though only the prefix was
+ ever visible.
+ """
+
+ async def run(self):
+ await super().run()
+ self._final_response_sent = True
+ self._final_content_delivered = True
+ self._turn_split_delivery = False
+ self._delivered_final_text = None
+ # Only the prefix was ever on screen.
+ self._last_sent_text = STREAMED_PREFIX
+ self._delivered_segment_texts = []
+ self._delivered_commentary_texts = []
+
+
+def _make_runner(adapter):
+ gateway_run = importlib.import_module("gateway.run")
+ runner = object.__new__(gateway_run.GatewayRunner)
+ runner.adapters = {adapter.platform: adapter}
+ runner._voice_mode = {}
+ runner._prefill_messages = []
+ runner._ephemeral_system_prompt = ""
+ runner._reasoning_config = None
+ runner._provider_routing = {}
+ runner._fallback_model = None
+ runner._session_db = None
+ runner._running_agents = {}
+ runner._session_run_generation = {}
+ runner.session_store = SimpleNamespace(_entries={}, _save=lambda: None)
+ runner.hooks = SimpleNamespace(loaded_hooks=False)
+ runner.config = SimpleNamespace(
+ thread_sessions_per_user=False,
+ group_sessions_per_user=False,
+ stt_enabled=False,
+ streaming=StreamingConfig.from_dict(
+ {"enabled": True, "edit_interval": 0.01, "buffer_threshold": 1}
+ ),
+ )
+ return runner
+
+
+async def _run_turn(monkeypatch, tmp_path, *, consumer_cls=None, session_id):
+ import yaml
+
+ (tmp_path / "config.yaml").write_text(
+ yaml.dump(
+ {
+ "display": {"tool_progress": "off", "interim_assistant_messages": False},
+ "streaming": {
+ "enabled": True,
+ "edit_interval": 0.01,
+ "buffer_threshold": 1,
+ },
+ }
+ ),
+ encoding="utf-8",
+ )
+
+ fake_dotenv = types.ModuleType("dotenv")
+ fake_dotenv.load_dotenv = lambda *args, **kwargs: None
+ monkeypatch.setitem(sys.modules, "dotenv", fake_dotenv)
+
+ fake_run_agent = types.ModuleType("run_agent")
+ fake_run_agent.AIAgent = PrefixOnlyAgent
+ monkeypatch.setitem(sys.modules, "run_agent", fake_run_agent)
+
+ gateway_run = importlib.import_module("gateway.run")
+ if consumer_cls is not None:
+ stream_consumer_mod = importlib.import_module("gateway.stream_consumer")
+ monkeypatch.setattr(
+ stream_consumer_mod, "GatewayStreamConsumer", consumer_cls
+ )
+ monkeypatch.setattr(gateway_run, "_hermes_home", tmp_path)
+ monkeypatch.setattr(
+ gateway_run, "_resolve_runtime_agent_kwargs", lambda: {"api_key": "***"}
+ )
+
+ adapter = CaptureAdapter()
+ runner = _make_runner(adapter)
+ source = SessionSource(
+ platform=Platform.DISCORD, chat_id="1534932197436424204", chat_type="group"
+ )
+ result = await runner._run_agent(
+ message="deploy status?",
+ context_prompt="",
+ history=[],
+ source=source,
+ session_id=session_id,
+ session_key=f"agent:main:discord:group:{session_id}",
+ )
+ return adapter, result
+
+
+@pytest.mark.asyncio
+async def test_recordless_delivery_flag_does_not_suppress_complete_response(
+ monkeypatch, tmp_path
+):
+ """#95382 boundary: flags claim delivery, nothing recorded, only the
+ prefix visible — the complete response must NOT be suppressed."""
+ adapter, result = await _run_turn(
+ monkeypatch,
+ tmp_path,
+ consumer_cls=_RecordlessFlagConsumer,
+ session_id="sess-95382-recordless",
+ )
+ assert result["final_response"] == FULL_RESPONSE
+ # Pre-fix behavior: already_sent=True and the tail appears in NO platform
+ # call (silent partial delivery). Post-fix: either the gateway performed
+ # the reconciliation edit itself (full text on the wire), or it declined
+ # to claim delivery so the caller's normal final send delivers it.
+ all_payloads = [c["content"] for c in adapter.sent] + [
+ e["content"] for e in adapter.edits
+ ]
+ delivered_here = any(FULL_RESPONSE in p for p in all_payloads)
+ assert delivered_here or not result.get("already_sent"), (
+ "silent partial delivery: gateway claimed delivery but the complete "
+ f"response never reached the platform; payloads={all_payloads!r}"
+ )
+
+
+@pytest.mark.asyncio
+async def test_normal_streaming_turn_still_suppresses_exactly_once(
+ monkeypatch, tmp_path
+):
+ """Control: an honest streaming turn (finalize edit carries the full
+ response) must still suppress the duplicate normal send."""
+ adapter, result = await _run_turn(
+ monkeypatch, tmp_path, session_id="sess-95382-control"
+ )
+ assert result["final_response"] == FULL_RESPONSE
+ all_payloads = [c["content"] for c in adapter.sent] + [
+ e["content"] for e in adapter.edits
+ ]
+ assert any(FULL_RESPONSE in p for p in all_payloads)
+ full_sends = [c for c in adapter.sent if FULL_RESPONSE in c["content"]]
+ assert len(full_sends) <= 1, f"duplicate final delivery: {full_sends!r}"
+
+
+@pytest.mark.asyncio
+async def test_recordless_flag_with_dead_transport_leaves_normal_send(
+ monkeypatch, tmp_path
+):
+ """#95382 incident shape: the reconciliation edit ALSO fails (dead
+ transport). The gateway must NOT claim already_sent — the normal final
+ send (and, on failure there, the delivery ledger) owns recovery."""
+
+ class _DeadEditRecordlessConsumer(_RecordlessFlagConsumer):
+ async def run(self):
+ await super().run()
+ # Transport dies after the stream drained: every further edit
+ # fails, like a dropped Discord WebSocket.
+ self.adapter.fail_edits = True
+
+ adapter, result = await _run_turn(
+ monkeypatch,
+ tmp_path,
+ consumer_cls=_DeadEditRecordlessConsumer,
+ session_id="sess-95382-dead-transport",
+ )
+ assert result["final_response"] == FULL_RESPONSE
+ assert not result.get("already_sent"), (
+ "gateway claimed delivery although neither the stream nor the "
+ "reconciliation edit put the complete response on the wire"
+ )
+
+
+# ---------------------------------------------------------------------------
+# Discord transport classification + ledger reconnect replay (#95382 lane 2)
+# ---------------------------------------------------------------------------
+
+
+class TestDiscordTransportClassification:
+ def _adapter_module(self):
+ import plugins.platforms.discord.adapter as mod
+
+ return mod
+
+ def test_connection_error_is_transport(self):
+ mod = self._adapter_module()
+ assert mod._is_discord_transport_error(ConnectionError("websocket closed"))
+ assert mod._is_discord_transport_error(
+ RuntimeError("Session is closed")
+ )
+ assert mod._is_discord_transport_error(OSError(104, "Connection reset"))
+
+ def test_http_and_timeout_errors_are_not_transport(self):
+ mod = self._adapter_module()
+ assert not mod._is_discord_transport_error(
+ RuntimeError("error code: 50013: Missing Permissions")
+ )
+ assert not mod._is_discord_transport_error(asyncio.TimeoutError())
+
+ @pytest.mark.asyncio
+ async def test_send_without_client_reports_send_path_degraded(self):
+ mod = self._adapter_module()
+ adapter = mod.DiscordAdapter.__new__(mod.DiscordAdapter)
+ adapter._client = None
+ result = await mod.DiscordAdapter.send(adapter, "c1", "hello")
+ assert result.success is False
+ assert result.error == "send_path_degraded"
+ assert result.retryable is True
+
+
+class TestLedgerReplaysDegradedDiscordSend:
+ def test_reconnect_sweep_claims_degraded_discord_row(self, tmp_path, monkeypatch):
+ """End-to-end ledger check: a final response rejected with
+ ``send_path_degraded`` on Discord is claimed by the runtime
+ reconnect sweep; a generic 'Not connected' row (pre-fix error
+ string) is stranded. This is the exact silent-loss mechanism from
+ the #95382 field logs."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ import gateway.delivery_ledger as dl
+
+ importlib.reload(dl)
+
+ oid_degraded = dl.compute_obligation_id("sess-a", "msg-1", FULL_RESPONSE)
+ dl.record_obligation(
+ obligation_id=oid_degraded,
+ session_key="agent:main:discord:group:c1",
+ platform="discord",
+ chat_id="c1",
+ thread_id=None,
+ content=FULL_RESPONSE,
+ )
+ dl.mark_attempting(oid_degraded)
+ dl.mark_failed(oid_degraded, "send_path_degraded")
+
+ oid_generic = dl.compute_obligation_id("sess-b", "msg-2", FULL_RESPONSE)
+ dl.record_obligation(
+ obligation_id=oid_generic,
+ session_key="agent:main:discord:group:c2",
+ platform="discord",
+ chat_id="c2",
+ thread_id=None,
+ content=FULL_RESPONSE,
+ )
+ dl.mark_attempting(oid_generic)
+ dl.mark_failed(oid_generic, "Not connected")
+
+ claimed = dl.sweep_failed_for_runtime("discord")
+ claimed_ids = {row["obligation_id"] for row in claimed}
+ assert oid_degraded in claimed_ids, (
+ "send_path_degraded Discord row must be replayable after reconnect"
+ )
+ assert oid_generic not in claimed_ids, (
+ "non-transport errors must not be blindly replayed"
+ )
diff --git a/tests/gateway/test_simplex_plugin.py b/tests/gateway/test_simplex_plugin.py
index 1a88d56513..90d3aa8ed1 100644
--- a/tests/gateway/test_simplex_plugin.py
+++ b/tests/gateway/test_simplex_plugin.py
@@ -388,3 +388,73 @@ def _make_file_chat_item(file_path: str, file_name: str) -> dict:
}
+
+
+# ---------------------------------------------------------------------------
+# Multiplex secondary-profile scope
+# ---------------------------------------------------------------------------
+#
+# Every SIMPLEX_* read (auto_accept / group_allowed in __init__, ws_url in the
+# registry gates, everything in _env_enablement) went through raw os.getenv,
+# which under multiplexing holds the DEFAULT profile's YAML-to-env bridge
+# output -- a secondary profile silently borrowed the default's daemon URL,
+# group allowlist and auto-accept setting. Reads now go through the module's
+# ``_get_scoped_secret`` (profile .env AND extra both honored; scoped miss
+# fails closed; unscoped default profile keeps env precedence).
+
+
+@pytest.fixture
+def multiplex_scope():
+ """Install multiplex + a secondary-profile secret scope; restore after."""
+ from agent.secret_scope import (
+ reset_secret_scope,
+ set_multiplex_active,
+ set_secret_scope,
+ )
+
+ tokens = []
+
+ def install(scope=None):
+ set_multiplex_active(True)
+ tokens.append(set_secret_scope(scope or {}))
+
+ yield install
+ for token in reversed(tokens):
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+
+@pytest.fixture
+def default_profile_env(monkeypatch):
+ """The default profile's YAML-to-env bridge output in os.environ."""
+ monkeypatch.setenv("SIMPLEX_WS_URL", "ws://default:5225")
+ monkeypatch.setenv("SIMPLEX_GROUP_ALLOWED", "*")
+ monkeypatch.setenv("SIMPLEX_AUTO_ACCEPT", "true")
+
+
+def test_multiplex_scoped_miss_does_not_borrow_default_profile_env(
+ multiplex_scope, default_profile_env
+):
+ """A secondary profile with no SimpleX config of its own must not be
+ auto-enabled off the default's daemon URL, nor inherit its wide-open
+ group allowlist."""
+ from gateway.config import PlatformConfig
+
+ multiplex_scope({"SOMETHING_ELSE": "x"})
+ assert _env_enablement() is None
+ assert check_requirements() is False
+ assert is_connected(PlatformConfig(enabled=True, extra={})) is False
+ adapter = SimplexAdapter(PlatformConfig(enabled=True, extra={"auto_accept": False}))
+ assert adapter.group_allow_from == set()
+ assert adapter.auto_accept is False
+
+
+def test_multiplex_scope_reads_profile_own_env_not_default(
+ multiplex_scope, default_profile_env
+):
+ """A secondary profile's own .env (installed as the scope) is honored --
+ the extra-only shape would have ignored it."""
+ multiplex_scope({"SIMPLEX_WS_URL": "ws://profile:5225", "SIMPLEX_GROUP_ALLOWED": "g1"})
+ seeded = _env_enablement()
+ assert seeded == {"ws_url": "ws://profile:5225", "group_allowed": "g1"}
+ assert check_requirements() is True
diff --git a/tests/gateway/test_slack_api_human_senders.py b/tests/gateway/test_slack_api_human_senders.py
new file mode 100644
index 0000000000..5bde8bd103
--- /dev/null
+++ b/tests/gateway/test_slack_api_human_senders.py
@@ -0,0 +1,94 @@
+"""Tests for the Slack ``api_human_users`` allowlist.
+
+A message posted through the Web API with a *user* token (``xoxp-``) is
+authored by a real person, but it arrives with the posting ``app_id`` and no
+``client_msg_id`` — the #35777 app/bot signature — so
+``_event_declares_bot_sender`` drops it. ``platforms.slack.extra.api_human_users``
+allowlists those *users* (never apps: an app's own ``xoxb`` bot posts carry
+the same user+app_id shape).
+"""
+
+import sys
+from unittest.mock import MagicMock
+
+import pytest
+
+
+# Mock slack-bolt / slack-sdk the same way test_slack_mention.py does.
+def _ensure_slack_mock():
+ if "slack_bolt" in sys.modules and hasattr(sys.modules["slack_bolt"], "__file__"):
+ return
+ slack_bolt = MagicMock()
+ slack_bolt.async_app.AsyncApp = MagicMock
+ slack_bolt.adapter.socket_mode.async_handler.AsyncSocketModeHandler = MagicMock
+ slack_sdk = MagicMock()
+ slack_sdk.web.async_client.AsyncWebClient = MagicMock
+ for name, mod in [
+ ("slack_bolt", slack_bolt),
+ ("slack_bolt.async_app", slack_bolt.async_app),
+ ("slack_bolt.adapter", slack_bolt.adapter),
+ ("slack_bolt.adapter.socket_mode", slack_bolt.adapter.socket_mode),
+ (
+ "slack_bolt.adapter.socket_mode.async_handler",
+ slack_bolt.adapter.socket_mode.async_handler,
+ ),
+ ("slack_sdk", slack_sdk),
+ ("slack_sdk.web", slack_sdk.web),
+ ("slack_sdk.web.async_client", slack_sdk.web.async_client),
+ ]:
+ sys.modules.setdefault(name, mod)
+ sys.modules.setdefault("aiohttp", MagicMock())
+
+
+_ensure_slack_mock()
+
+import plugins.platforms.slack.adapter as _slack_mod # noqa: E402
+
+_slack_mod.SLACK_AVAILABLE = True
+
+from plugins.platforms.slack.adapter import SlackAdapter # noqa: E402
+
+from gateway.config import Platform, PlatformConfig # noqa: E402
+
+
+HUMAN_ID = "U_human"
+
+
+def _make_adapter(extra=None):
+ adapter = object.__new__(SlackAdapter)
+ adapter.platform = Platform.SLACK
+ adapter.config = PlatformConfig(enabled=True, extra=dict(extra or {}))
+ return adapter
+
+
+def _api_post(**overrides):
+ """A user-token chat.postMessage as delivered over Socket Mode:
+ real ``user``, app_id stamp, no ``client_msg_id``."""
+ event = {"type": "message", "user": HUMAN_ID, "app_id": "A_frontend", "text": "hi"}
+ event.update(overrides)
+ return event
+
+
+@pytest.fixture(autouse=True)
+def _clean_env(monkeypatch):
+ monkeypatch.delenv("SLACK_API_HUMAN_USERS", raising=False)
+
+
+def test_api_post_is_bot_by_default():
+ assert _make_adapter()._event_declares_bot_sender(_api_post()) is True
+
+
+def test_allowlisted_user_api_post_is_human():
+ adapter = _make_adapter({"api_human_users": ["U_other", HUMAN_ID]})
+ assert adapter._event_declares_bot_sender(_api_post()) is False
+ # Same predicate everywhere: no other user, and no user-less app post, rides it.
+ assert adapter._event_declares_bot_sender(_api_post(user="U_stranger")) is True
+ assert adapter._event_declares_bot_sender({"app_id": "A_frontend", "text": "hi"}) is True
+
+
+def test_bot_markers_win_over_allowlist():
+ """Allowlisting a user never admits genuine bot posts, so the app's own
+ ``xoxb`` traffic (bot_id / subtype=bot_message) cannot loop back in."""
+ adapter = _make_adapter({"api_human_users": HUMAN_ID})
+ assert adapter._event_declares_bot_sender(_api_post(subtype="bot_message")) is True
+ assert adapter._event_declares_bot_sender(_api_post(bot_id="B_stamp")) is True
diff --git a/tests/gateway/test_slash_command_profile_scope.py b/tests/gateway/test_slash_command_profile_scope.py
new file mode 100644
index 0000000000..5a3dba7397
--- /dev/null
+++ b/tests/gateway/test_slash_command_profile_scope.py
@@ -0,0 +1,94 @@
+"""Gateway slash commands must do their blocking work inside the routed profile.
+
+The multiplexed inbound handler wraps the whole message in
+``_profile_runtime_scope``, which installs the routed profile's ``HERMES_HOME``
+override and its secret scope as **contextvars**. A bare
+``loop.run_in_executor(None, ...)`` starts the worker with an EMPTY context, so
+``SessionDB()`` / ``get_hermes_home()`` inside the worker resolve the LAUNCH
+home — /insights reported the default profile's conversations from another
+profile's chat. ``/compress`` already routes through
+``_run_in_executor_with_context``; every other hop in the mixin must too.
+
+Drives the real mixin methods and the real ``_profile_runtime_scope``: the
+contextvar loss is a property of the hop, so mocking the hop away would test
+nothing.
+"""
+
+from __future__ import annotations
+
+from pathlib import Path
+
+import pytest
+
+
+@pytest.fixture
+def profile_home(tmp_path, monkeypatch):
+ root = tmp_path / ".hermes"
+ home = root / "profiles" / "coder"
+ home.mkdir(parents=True)
+ monkeypatch.setattr(Path, "home", lambda: tmp_path)
+ monkeypatch.setenv("HERMES_HOME", str(root))
+ return home
+
+
+@pytest.fixture
+def runner():
+ """Minimal host exposing the mixin plus the runner's executor helpers."""
+ from gateway.run import GatewayRunner
+ from gateway.slash_commands import GatewaySlashCommandsMixin
+
+ class _Runner(GatewaySlashCommandsMixin):
+ _run_in_executor_with_context = GatewayRunner._run_in_executor_with_context
+ _get_executor = GatewayRunner._get_executor
+
+ r = _Runner()
+ r.adapters = {}
+ r._pending_skills_reload_notes = {}
+ return r
+
+
+class _Event:
+ def __init__(self, args: str = ""):
+ self._args = args
+ self.source = None
+
+ def get_command_args(self) -> str:
+ return self._args
+
+
+@pytest.mark.asyncio
+async def test_insights_opens_session_db_under_the_routed_home(
+ runner, profile_home, monkeypatch
+):
+ import agent.insights as insights_mod
+ import hermes_state
+ from gateway.run import _profile_runtime_scope
+ from hermes_constants import get_hermes_home
+
+ seen: dict = {}
+
+ class _RecordingDB:
+ def __init__(self, *a, **kw):
+ seen["home"] = str(get_hermes_home())
+
+ def close(self):
+ pass
+
+ class _Engine:
+ def __init__(self, db):
+ pass
+
+ def generate(self, **kw):
+ return {}
+
+ def format_gateway(self, report):
+ return "ok"
+
+ monkeypatch.setattr(hermes_state, "SessionDB", _RecordingDB)
+ monkeypatch.setattr(insights_mod, "InsightsEngine", _Engine)
+
+ with _profile_runtime_scope(profile_home):
+ result = await runner._handle_insights_command(_Event(""))
+
+ assert result == "ok"
+ assert seen["home"] == str(profile_home)
diff --git a/tests/gateway/test_slash_config_writes_routed_profile.py b/tests/gateway/test_slash_config_writes_routed_profile.py
new file mode 100644
index 0000000000..c1893156f5
--- /dev/null
+++ b/tests/gateway/test_slash_config_writes_routed_profile.py
@@ -0,0 +1,70 @@
+"""Slash-command config writes must land in the routed profile's config.yaml.
+
+Regression for #87939 / #75684: the multiplexed inbound handler already runs
+every slash handler inside ``_profile_runtime_scope`` (routed HERMES_HOME
+override), but several handlers built their write path from the module
+constant ``gateway.run._hermes_home`` — the LAUNCH home — so ``/reasoning
+--global``, ``/fast``, ``/memory approval``, ``/skills approval``, ``/verbose``
+and ``/footer`` persisted into the default profile's config.yaml. They now go
+through ``_gateway_config_home()`` like the reads do.
+"""
+
+from __future__ import annotations
+
+import pytest
+import yaml
+
+import gateway.run as gateway_run
+from gateway.run import GatewayRunner, _profile_runtime_scope
+from gateway.slash_commands import GatewaySlashCommandsMixin
+
+
+class _Runner(GatewaySlashCommandsMixin):
+ _run_in_executor_with_context = GatewayRunner._run_in_executor_with_context
+ _get_executor = GatewayRunner._get_executor
+
+ def _session_key_for_source(self, _source):
+ return "k"
+
+ def _evict_cached_agent(self, _session_key):
+ pass
+
+
+class _Event:
+ def __init__(self, args: str = ""):
+ self._args = args
+ self.source = None
+
+ def get_command_args(self) -> str:
+ return self._args
+
+
+@pytest.fixture
+def homes(tmp_path, monkeypatch):
+ default_home = tmp_path / "default"
+ routed_home = tmp_path / "profiles" / "beta"
+ default_home.mkdir()
+ routed_home.mkdir(parents=True)
+ (default_home / "config.yaml").write_text("agent:\n reasoning_effort: medium\n")
+ (routed_home / "config.yaml").write_text("agent:\n reasoning_effort: none\n")
+ monkeypatch.setattr(gateway_run, "_hermes_home", default_home)
+ monkeypatch.setenv("HERMES_HOME", str(default_home))
+ return default_home, routed_home
+
+
+@pytest.mark.asyncio
+async def test_slash_config_writes_hit_routed_profile_and_leave_default_untouched(homes):
+ default_home, routed_home = homes
+ default_before = (default_home / "config.yaml").read_bytes()
+ runner = _Runner()
+
+ with _profile_runtime_scope(routed_home):
+ assert runner._save_gateway_config_key("agent.reasoning_effort", "high")
+ await runner._handle_memory_command(_Event("approval on"))
+ await runner._handle_skills_command(_Event("approval on"))
+
+ routed = yaml.safe_load((routed_home / "config.yaml").read_text())
+ assert routed["agent"]["reasoning_effort"] == "high"
+ assert routed["memory"]["write_approval"] is True
+ assert routed["skills"]["write_approval"] is True
+ assert (default_home / "config.yaml").read_bytes() == default_before
diff --git a/tests/gateway/test_stale_finalize_suppression.py b/tests/gateway/test_stale_finalize_suppression.py
index 15d591c8c9..679e4f7e3e 100644
--- a/tests/gateway/test_stale_finalize_suppression.py
+++ b/tests/gateway/test_stale_finalize_suppression.py
@@ -373,8 +373,23 @@ def _consumer():
class TestDeliveredFinalMatches:
- def test_no_record_returns_none(self):
+ def test_no_record_no_visible_text_returns_false(self):
+ """#95382 tightening: a record-less consumer with no visible match
+ for the final text is a demonstrable non-delivery, not legacy trust."""
consumer = _consumer()
+ assert consumer.delivered_final_matches("anything") is False
+
+ def test_no_record_but_visible_final_returns_true(self):
+ """Ambiguous-dedup control: visible text equals the final answer."""
+ consumer = _consumer()
+ consumer._already_sent = True
+ consumer._last_sent_text = FULL_RESPONSE
+ assert consumer.delivered_final_matches(FULL_RESPONSE) is True
+
+ def test_no_record_ambiguous_timeout_returns_none(self):
+ """The explicitly-marked ambiguous timeout keeps legacy trust."""
+ consumer = _consumer()
+ consumer._delivery_ambiguous = True
assert consumer.delivered_final_matches("anything") is None
def test_matching_record_returns_true(self):
diff --git a/tests/gateway/test_telegram_callback_auth_fail_closed.py b/tests/gateway/test_telegram_callback_auth_fail_closed.py
index ee92721e4e..db7d60d2a8 100644
--- a/tests/gateway/test_telegram_callback_auth_fail_closed.py
+++ b/tests/gateway/test_telegram_callback_auth_fail_closed.py
@@ -87,3 +87,46 @@ class TestCallbackAuthFailClosed:
assert adapter._is_callback_user_authorized("12345") is True
+class TestCallbackAuthPrefersInjectedCheck:
+ """_is_callback_user_authorized must use the auth callback GatewayRunner
+ injects via set_authorization_check before the _message_handler.__self__
+ introspection.
+
+ A secondary multiplexed adapter's _message_handler is a profile closure
+ (no __self__), so the introspection path resolves to nothing and the old
+ code fell through to the env-only fallback — which knows nothing about
+ profile config allowlists or the pairing store. The injected callback is
+ registered for every gateway-connected adapter, including multiplexed
+ secondaries, and delegates to the full _is_user_authorized chain.
+ """
+
+ def test_injected_check_used_when_handler_is_a_closure(self, monkeypatch):
+ """Multiplexed shape: closure handler (no __self__) + injected check
+ registered → the injected check decides, not the env fallback."""
+ monkeypatch.delenv("TELEGRAM_ALLOWED_USERS", raising=False)
+ monkeypatch.delenv("GATEWAY_ALLOW_ALL_USERS", raising=False)
+ adapter = _make_adapter()
+ adapter._message_handler = lambda *a, **kw: None # no __self__
+ seen = {}
+
+ def _check(user_id, chat_type=None, chat_id=None):
+ seen.update(user_id=user_id, chat_type=chat_type, chat_id=chat_id)
+ return user_id == "999"
+
+ adapter._authorization_check = _check
+
+ # Env fallback would deny (empty allowlist); the injected check allows.
+ assert adapter._is_callback_user_authorized(
+ "999", chat_id="777", chat_type="supergroup"
+ ) is True
+ assert seen == {"user_id": "999", "chat_type": "group", "chat_id": "777"}
+
+ def test_injected_check_deny_wins_over_env_allowlist(self, monkeypatch):
+ """The injected check is authoritative when registered — an env
+ allowlist entry must not override its deny."""
+ monkeypatch.setenv("TELEGRAM_ALLOWED_USERS", "12345")
+ adapter = _make_adapter()
+ adapter._message_handler = lambda *a, **kw: None
+ adapter._authorization_check = lambda user_id, chat_type=None, chat_id=None: False
+
+ assert adapter._is_callback_user_authorized("12345") is False
diff --git a/tests/gateway/test_telegram_final_delivery.py b/tests/gateway/test_telegram_final_delivery.py
index e5d378dcee..b5c85e73c3 100644
--- a/tests/gateway/test_telegram_final_delivery.py
+++ b/tests/gateway/test_telegram_final_delivery.py
@@ -122,6 +122,33 @@ async def test_empty_tail_commit_honors_retry_after(monkeypatch):
assert consumer.final_content_delivered is True
+@pytest.mark.asyncio
+async def test_complete_preview_survives_long_flood_fallback_failure(monkeypatch):
+ """A complete ACKed preview must not trigger a duplicate normal final."""
+ adapter = _adapter()
+ adapter.send.return_value = SendResult(
+ success=False,
+ error="flood_control:20.0",
+ retry_after=20.0,
+ )
+ sleep = AsyncMock()
+ monkeypatch.setattr("gateway.stream_consumer.asyncio.sleep", sleep)
+
+ consumer = GatewayStreamConsumer(adapter, "chat-1")
+ consumer._message_id = "preview-1"
+ consumer._last_sent_text = "Final answer"
+ consumer._already_sent = True
+ consumer._fallback_final_send = True
+
+ await consumer._send_fallback_final("Final answer")
+
+ adapter.send.assert_awaited_once()
+ sleep.assert_not_awaited()
+ assert consumer.final_response_sent is False
+ assert consumer.final_content_delivered is True
+ assert consumer.delivered_final_matches("Final answer") is True
+
+
@pytest.mark.asyncio
async def test_telegram_long_flood_result_keeps_retry_after():
"""The real adapter contract preserves the server delay for consumers."""
@@ -139,3 +166,91 @@ async def test_telegram_long_flood_result_keeps_retry_after():
assert result.retry_after == 30.0
+
+
+@pytest.mark.asyncio
+async def test_empty_fallback_resend_preserves_reply_anchor():
+ """The fresh-commit resend must carry the turn's reply anchor (#71047).
+
+ With reply_to_mode='first' the streamed preview is delivered as a reply
+ to the user's message. When a failed finalize edit forces the fresh
+ resend, the replacement message must use the same anchor so the visible
+ behavior matches the preview (and the non-streaming path).
+ """
+ adapter = _adapter()
+ adapter.send.return_value = SendResult(success=True, message_id="final-1")
+
+ consumer = GatewayStreamConsumer(
+ adapter, "chat-1", initial_reply_to_id="111",
+ )
+ consumer._message_id = "preview-1"
+ consumer._last_sent_text = "Final answer"
+ consumer._already_sent = True
+ consumer._fallback_final_send = True
+
+ await consumer._send_fallback_final("Final answer")
+
+ adapter.send.assert_awaited_once()
+ kwargs = adapter.send.await_args.kwargs
+ assert kwargs.get("reply_to") == "111"
+ # Preview replaced: deleted after the fresh final succeeded.
+ adapter.delete_message.assert_awaited_once_with("chat-1", "preview-1")
+ assert consumer.final_response_sent is True
+ assert consumer.final_content_delivered is True
+
+
+@pytest.mark.asyncio
+async def test_empty_fallback_preview_delete_retries_once(monkeypatch):
+ """A False (flood-rejected) preview delete gets one bounded retry."""
+ adapter = _adapter()
+ adapter.send.return_value = SendResult(success=True, message_id="final-1")
+ adapter.delete_message = AsyncMock(side_effect=[False, True])
+ sleep = AsyncMock()
+ monkeypatch.setattr("gateway.stream_consumer.asyncio.sleep", sleep)
+
+ consumer = GatewayStreamConsumer(adapter, "chat-1")
+ consumer._message_id = "preview-1"
+ consumer._last_sent_text = "Final answer"
+ consumer._already_sent = True
+ consumer._fallback_final_send = True
+
+ await consumer._send_fallback_final("Final answer")
+
+ assert adapter.delete_message.await_count == 2
+ sleep.assert_awaited_once_with(1.0)
+ assert consumer.final_response_sent is True
+
+
+@pytest.mark.asyncio
+async def test_flood_capped_resend_keeps_single_bubble_reply_first(monkeypatch):
+ """#71047 Problem B end-to-end shape: preview as reply, finalize edit and
+ fresh resend both flood-capped — the gateway suppression decision must
+ keep the complete ACKed preview as the single visible bubble instead of
+ letting the normal final send create a second one.
+ """
+ adapter = _adapter()
+ # Fresh-commit resend flood-capped past the inline retry budget.
+ adapter.send.return_value = SendResult(
+ success=False,
+ error="flood_control:41.0",
+ retry_after=41.0,
+ )
+ sleep = AsyncMock()
+ monkeypatch.setattr("gateway.stream_consumer.asyncio.sleep", sleep)
+
+ consumer = GatewayStreamConsumer(
+ adapter, "chat-1", initial_reply_to_id="111",
+ )
+ consumer._message_id = "preview-1"
+ consumer._last_sent_text = "Final answer"
+ consumer._already_sent = True
+ consumer._fallback_final_send = True
+
+ await consumer._send_fallback_final("Final answer")
+
+ # Preview must NOT be deleted — it is the only copy of the answer.
+ adapter.delete_message.assert_not_awaited()
+ # Mirror the gateway/run.py suppression decision: content delivered and
+ # the recorded payload reconciles, so the normal final send is skipped.
+ assert consumer.final_content_delivered is True
+ assert consumer.delivered_final_matches("Final answer") is True
diff --git a/tests/gateway/test_telegram_topic_profile_isolation_76423.py b/tests/gateway/test_telegram_topic_profile_isolation_76423.py
new file mode 100644
index 0000000000..32d8323335
--- /dev/null
+++ b/tests/gateway/test_telegram_topic_profile_isolation_76423.py
@@ -0,0 +1,118 @@
+"""Issue #76423 — SessionDB: telegram topic tables namespace by profile."""
+
+from __future__ import annotations
+
+import sqlite3
+from pathlib import Path
+
+from hermes_state import SessionDB
+
+
+CHAT = "208214988"
+
+
+def _session(db, sid, profile_name=None):
+ db.create_session(session_id=sid, source="telegram", user_id=CHAT, profile_name=profile_name)
+
+
+def test_legacy_rows_migrate_only_to_default(tmp_path: Path):
+ """v1 shape (no CASCADE, old user index) → v3: rows land in 'default' only."""
+ db_path = tmp_path / "legacy.db"
+ conn = sqlite3.connect(str(db_path))
+ conn.executescript(
+ f"""
+ CREATE TABLE state_meta (key TEXT PRIMARY KEY, value TEXT);
+ INSERT INTO state_meta(key, value) VALUES ('telegram_dm_topic_schema_version', '1');
+ CREATE TABLE sessions (
+ id TEXT PRIMARY KEY, source TEXT, user_id TEXT, model TEXT,
+ model_config TEXT, system_prompt TEXT, parent_session_id TEXT,
+ started_at REAL, ended_at REAL, end_reason TEXT,
+ message_count INTEGER DEFAULT 0, tool_call_count INTEGER DEFAULT 0,
+ input_tokens INTEGER DEFAULT 0, output_tokens INTEGER DEFAULT 0
+ );
+ INSERT INTO sessions(id, source, user_id, started_at)
+ VALUES ('legacy-sess', 'telegram', '{CHAT}', 1.0);
+ CREATE TABLE telegram_dm_topic_mode (
+ chat_id TEXT PRIMARY KEY, user_id TEXT NOT NULL,
+ enabled INTEGER NOT NULL DEFAULT 1,
+ activated_at REAL NOT NULL, updated_at REAL NOT NULL,
+ has_topics_enabled INTEGER, allows_users_to_create_topics INTEGER,
+ capability_checked_at REAL, intro_message_id TEXT, pinned_message_id TEXT
+ );
+ INSERT INTO telegram_dm_topic_mode(chat_id, user_id, enabled, activated_at, updated_at)
+ VALUES ('{CHAT}', '{CHAT}', 1, 1.0, 1.0);
+ CREATE TABLE telegram_dm_topic_bindings (
+ chat_id TEXT NOT NULL, thread_id TEXT NOT NULL, user_id TEXT NOT NULL,
+ session_key TEXT NOT NULL,
+ session_id TEXT NOT NULL REFERENCES sessions(id),
+ managed_mode TEXT NOT NULL DEFAULT 'auto',
+ linked_at REAL NOT NULL, updated_at REAL NOT NULL,
+ PRIMARY KEY (chat_id, thread_id)
+ );
+ CREATE INDEX idx_telegram_dm_topic_bindings_user
+ ON telegram_dm_topic_bindings(user_id, chat_id);
+ INSERT INTO telegram_dm_topic_bindings
+ VALUES ('{CHAT}', '99', '{CHAT}', 'k', 'legacy-sess', 'auto', 1.0, 1.0);
+ """
+ )
+ conn.close()
+
+ db = SessionDB(db_path=db_path)
+ db.apply_telegram_topic_migration()
+ assert db.get_meta("telegram_dm_topic_schema_version") == "3"
+ assert db.is_telegram_topic_mode_enabled(
+ chat_id=CHAT, user_id=CHAT, profile_name="default",
+ )
+ assert not db.is_telegram_topic_mode_enabled(
+ chat_id=CHAT, user_id=CHAT, profile_name="coder",
+ )
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="99", profile_name="default",
+ )["session_id"] == "legacy-sess"
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="99", profile_name="coder",
+ ) is None
+ fk = db._conn.execute("PRAGMA foreign_key_list('telegram_dm_topic_bindings')").fetchall()
+ assert any(row[2] == "sessions" and row[6] == "CASCADE" for row in fk)
+ db.close()
+
+
+def test_mode_and_bindings_isolated_across_profiles(tmp_path: Path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ _session(db, "sess-a", "alpha")
+ _session(db, "sess-b", "beta")
+
+ db.enable_telegram_topic_mode(chat_id=CHAT, user_id=CHAT, profile_name="alpha")
+ db.enable_telegram_topic_mode(chat_id=CHAT, user_id=CHAT, profile_name="beta")
+ db.disable_telegram_topic_mode(chat_id=CHAT, profile_name="alpha")
+ assert not db.is_telegram_topic_mode_enabled(chat_id=CHAT, user_id=CHAT, profile_name="alpha")
+ assert db.is_telegram_topic_mode_enabled(chat_id=CHAT, user_id=CHAT, profile_name="beta")
+
+ db.bind_telegram_topic(
+ chat_id=CHAT, thread_id="77", user_id=CHAT,
+ session_key="ka", session_id="sess-a", profile_name="alpha",
+ )
+ db.bind_telegram_topic(
+ chat_id=CHAT, thread_id="77", user_id=CHAT,
+ session_key="kb", session_id="sess-b", profile_name="beta",
+ )
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="77", profile_name="alpha",
+ )["session_id"] == "sess-a"
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="77", profile_name="beta",
+ )["session_id"] == "sess-b"
+
+ assert db.delete_telegram_topic_binding(
+ chat_id=CHAT, thread_id="77", profile_name="alpha",
+ ) == 1
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="77", profile_name="alpha",
+ ) is None
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="77", profile_name="beta",
+ ) is not None
+ # Omitted kwarg == the single-profile "default" namespace, not a wildcard.
+ assert not db.is_telegram_topic_mode_enabled(chat_id=CHAT, user_id=CHAT)
+ assert db.get_telegram_topic_binding(chat_id=CHAT, thread_id="77") is None
+ db.close()
diff --git a/tests/gateway/test_telegram_topic_profile_routing_76423.py b/tests/gateway/test_telegram_topic_profile_routing_76423.py
new file mode 100644
index 0000000000..9f9f16e1ac
--- /dev/null
+++ b/tests/gateway/test_telegram_topic_profile_routing_76423.py
@@ -0,0 +1,93 @@
+"""Issue #76423 — Gateway routes source.profile into telegram topic state."""
+
+from __future__ import annotations
+
+from pathlib import Path
+from types import SimpleNamespace
+
+from hermes_state import SessionDB
+from gateway.config import Platform
+from gateway.session import SessionSource
+
+
+CHAT = "208214988"
+
+
+def _source(profile=None, thread_id="42"):
+ return SessionSource(
+ platform=Platform.TELEGRAM,
+ user_id=CHAT,
+ chat_id=CHAT,
+ user_name="tester",
+ chat_type="dm",
+ thread_id=thread_id,
+ profile=profile,
+ )
+
+
+def test_gateway_uses_source_profile_not_global(tmp_path: Path):
+ from gateway.run import GatewayRunner
+
+ assert GatewayRunner._telegram_topic_profile_name(_source("coder")) == "coder"
+ assert GatewayRunner._telegram_topic_profile_name(_source(None)) == "default"
+
+ db = SessionDB(db_path=tmp_path / "state.db")
+ db.create_session(session_id="sess-coder", source="telegram", user_id=CHAT, profile_name="coder")
+ db.enable_telegram_topic_mode(chat_id=CHAT, user_id=CHAT, profile_name="coder")
+
+ runner = object.__new__(GatewayRunner)
+ runner._session_db = db
+ assert runner._telegram_topic_mode_enabled(_source("coder")) is True
+ assert runner._telegram_topic_mode_enabled(_source("other")) is False
+ assert runner._telegram_topic_mode_enabled(_source(None)) is False
+
+ runner._record_telegram_topic_binding(
+ _source("coder", "42"),
+ SimpleNamespace(session_key="k", session_id="sess-coder"),
+ )
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="42", profile_name="coder",
+ ) is not None
+ assert db.get_telegram_topic_binding(
+ chat_id=CHAT, thread_id="42", profile_name="default",
+ ) is None
+ db.close()
+
+
+def test_routed_profile_flows_into_prune_via_send_metadata(tmp_path: Path):
+ """profile_routes: the transport adapter may be the primary (default) bot
+ while the turn is routed to another profile — the outbound metadata built
+ by the gateway carries the routed profile, and prune uses it over the
+ adapter's own stamp (#76423)."""
+ from gateway.run import GatewayRunner
+ from plugins.platforms.telegram.adapter import TelegramAdapter
+
+ runner = object.__new__(GatewayRunner)
+ runner._thread_metadata_for_target = lambda *a, **k: {"thread_id": "99"}
+ meta = runner._thread_metadata_for_source(_source("coder", "99"))
+ assert meta["hermes_profile"] == "coder"
+ assert "hermes_profile" not in runner._thread_metadata_for_source(_source(None, "99"))
+
+ # Cooldowns are keyed (profile, chat): alpha's reminder must not gag beta.
+ assert runner._should_send_telegram_lobby_reminder(_source("alpha")) is True
+ assert runner._should_send_telegram_lobby_reminder(_source("beta")) is True
+ assert runner._should_send_telegram_lobby_reminder(_source("alpha")) is False
+
+ db = SessionDB(db_path=tmp_path / "state.db")
+ db.create_session(session_id="sess-default", source="telegram", user_id=CHAT)
+ db.create_session(session_id="sess-coder", source="telegram", user_id=CHAT, profile_name="coder")
+ for prof, sid in (("default", "sess-default"), ("coder", "sess-coder")):
+ db.bind_telegram_topic(
+ chat_id=CHAT, thread_id="99", user_id=CHAT,
+ session_key=f"k-{prof}", session_id=sid, profile_name=prof,
+ )
+
+ adapter = object.__new__(TelegramAdapter)
+ adapter.platform = Platform.TELEGRAM
+ adapter._session_store = SimpleNamespace(_db=db)
+ adapter._hermes_profile_name = "default" # transport = primary bot
+ adapter._prune_stale_dm_topic_binding(CHAT, "99", metadata=meta)
+
+ assert db.get_telegram_topic_binding(chat_id=CHAT, thread_id="99", profile_name="coder") is None
+ assert db.get_telegram_topic_binding(chat_id=CHAT, thread_id="99", profile_name="default") is not None
+ db.close()
diff --git a/tests/gateway/test_turn_request_overrides.py b/tests/gateway/test_turn_request_overrides.py
index c985125176..bd5da602d8 100644
--- a/tests/gateway/test_turn_request_overrides.py
+++ b/tests/gateway/test_turn_request_overrides.py
@@ -53,7 +53,7 @@ def test_provider_request_overrides_merged_under_fast_mode(monkeypatch):
"""/fast active: provider extra_body AND the service-tier marker both survive."""
monkeypatch.setattr(
"hermes_cli.models.resolve_fast_mode_overrides",
- lambda model_id: {"service_tier": "priority"},
+ lambda model_id, **_route: {"service_tier": "priority"},
)
runner = _runner(service_tier="priority")
rk = _runtime_kwargs(request_overrides=PROVIDER_OVERRIDES)
diff --git a/tests/gateway/test_voice_mode_platform_isolation.py b/tests/gateway/test_voice_mode_platform_isolation.py
index 68485ee14c..799029911f 100644
--- a/tests/gateway/test_voice_mode_platform_isolation.py
+++ b/tests/gateway/test_voice_mode_platform_isolation.py
@@ -9,7 +9,9 @@ same key. The fix prefixes keys with platform value: 'telegram:123' vs
import json
import tempfile
from pathlib import Path
-from unittest.mock import MagicMock, patch
+from unittest.mock import AsyncMock, MagicMock, patch
+
+import pytest
from gateway.config import Platform
@@ -115,6 +117,88 @@ class TestSyncVoiceModeStateToAdapter:
assert mock_adapter._auto_tts_disabled_chats == {"123"}
+class TestVoiceModeProfileIsolation:
+ """Two multiplexed bots in one Discord channel keep independent /voice
+ state and voice transcripts dispatch through the bot that heard them
+ (#75198 voice half)."""
+
+ @staticmethod
+ def _discord_adapter(owner=None):
+ from unittest.mock import AsyncMock
+
+ a = MagicMock()
+ a.platform = Platform.DISCORD
+ a._owner_profile = owner
+ a._voice_text_channels = {111: 123}
+ a._voice_sources = {}
+ a._voice_input_callback = None
+ a._on_voice_disconnect = None
+ a._voice_mode_getter = None
+ a._auto_tts_enabled_chats = set()
+ a._auto_tts_disabled_chats = set()
+ a._client = MagicMock()
+ a._client.get_channel = MagicMock(return_value=None)
+ a.handle_message = AsyncMock()
+ return a
+
+ @pytest.mark.asyncio
+ async def test_voice_state_and_transcripts_stay_with_the_owning_bot(self, tmp_path):
+ from types import SimpleNamespace
+
+ from gateway.platforms.base import MessageEvent, MessageType, SessionSource
+
+ runner = _make_runner()
+ runner._VOICE_MODE_PATH = tmp_path / "voice.json"
+ runner._is_user_authorized = lambda source: True
+ default_ad = self._discord_adapter()
+ bot2_ad = self._discord_adapter(owner="bot2")
+ runner.adapters = {Platform.DISCORD: default_ad}
+ runner._profile_adapters = {"bot2": {Platform.DISCORD: bot2_ad}}
+ # Inbound event from bot2's transport in channel 123 (same id the
+ # default bot also sees).
+ src = SessionSource(platform=Platform.DISCORD, chat_id="123", user_id="u1",
+ chat_type="channel", profile="bot2")
+ src._transport_adapter_ref = lambda: bot2_ad
+
+ await runner._handle_voice_command(
+ MessageEvent(text="/voice tts", message_type=MessageType.TEXT, source=src)
+ )
+ assert runner._voice_mode == {"bot2:discord:123": "all"}
+ assert "123" in bot2_ad._auto_tts_enabled_chats
+ assert "123" not in default_ad._auto_tts_enabled_chats
+
+ # A transcript captured by bot2's adapter runs through bot2, not default.
+ runner._bind_voice_input_callback(bot2_ad)
+ await bot2_ad._voice_input_callback(guild_id=111, user_id=42, transcript="hi")
+ bot2_ad.handle_message.assert_awaited_once()
+ default_ad.handle_message.assert_not_awaited()
+ assert bot2_ad.handle_message.call_args[0][0].source.profile == "bot2"
+
+ # Timeout cleanup from bot2's channel disables bot2's auto-TTS only.
+ join = MessageEvent(text="/voice channel", message_type=MessageType.TEXT, source=src)
+ join.raw_message = SimpleNamespace(guild_id=111, guild=None)
+ bot2_ad.join_voice_channel = AsyncMock(return_value=True)
+ ch = MagicMock(); ch.name = "General"
+ bot2_ad.get_user_voice_channel = AsyncMock(return_value=ch)
+ await runner._handle_voice_channel_join(join)
+ bot2_ad._on_voice_disconnect("123")
+ assert runner._voice_mode["bot2:discord:123"] == "off"
+ assert "123" in bot2_ad._auto_tts_disabled_chats
+ assert "123" not in default_ad._auto_tts_disabled_chats
+
+ def test_sync_restores_only_the_owning_profiles_chats(self):
+ runner = _make_runner()
+ runner._voice_mode = {"discord:1": "all", "bot2:discord:2": "all"}
+ default_ad = MagicMock(); default_ad.platform = Platform.DISCORD
+ default_ad._owner_profile = None; default_ad._auto_tts_enabled_chats = set()
+ bot2_ad = MagicMock(); bot2_ad.platform = Platform.DISCORD
+ bot2_ad._owner_profile = "bot2"; bot2_ad._auto_tts_enabled_chats = set()
+ runner._sync_voice_mode_state_to_adapter(default_ad)
+ runner._sync_voice_mode_state_to_adapter(bot2_ad)
+ assert default_ad._auto_tts_enabled_chats == {"1"}
+ assert bot2_ad._auto_tts_enabled_chats == {"2"}
+
+
# ---------------------------------------------------------------------------
# Helper
# ---------------------------------------------------------------------------
diff --git a/tests/gateway/test_webhook_adapter.py b/tests/gateway/test_webhook_adapter.py
index 4f5cdb1390..7ea8e8acde 100644
--- a/tests/gateway/test_webhook_adapter.py
+++ b/tests/gateway/test_webhook_adapter.py
@@ -1011,6 +1011,63 @@ class TestMultiplexProfileWebhookAuthentication:
)
assert default_profile.status == 404
+ @pytest.mark.asyncio
+ async def test_routed_profile_skills_resolve_under_that_profile(
+ self, tmp_path, monkeypatch
+ ):
+ """A /p// route's ``skills:`` must load from that profile's
+ skills/ dir (#67277). Before the fix the lookup ran with no profile
+ scope, so it scanned the launch profile and logged "Skill not found".
+ """
+ import agent.skill_commands as sc_mod
+
+ worker = tmp_path / "profiles" / "worker"
+ skill_dir = worker / "skills" / "worker-only"
+ skill_dir.mkdir(parents=True)
+ (skill_dir / "SKILL.md").write_text(
+ "---\nname: worker-only\ndescription: w\n---\n\nBody of worker-only.\n"
+ )
+ (worker / "config.yaml").write_text("{}\n")
+ (worker / ".env").write_text("")
+ monkeypatch.setattr(
+ "hermes_cli.profiles.get_profile_dir", lambda name: tmp_path / "profiles" / name
+ )
+ route_secret = "worker-route-secret-abc123"
+ adapter = _make_adapter(
+ routes={
+ "gh": {
+ "profile": "worker",
+ "secret": route_secret,
+ "prompt": "PR: {action}",
+ "skills": ["worker-only"],
+ }
+ },
+ host="127.0.0.1",
+ )
+ self._configure_profiles(adapter, tmp_path, monkeypatch)
+ seen = []
+
+ async def _capture(event):
+ seen.append(event)
+
+ adapter.handle_message = _capture
+ body = b'{"action":"opened"}'
+ headers = {
+ "Content-Type": "application/json",
+ "X-Hub-Signature-256": _github_signature(body, route_secret),
+ }
+ with (
+ patch.object(sc_mod, "_skill_commands", {}),
+ patch.object(sc_mod, "_skill_commands_home", None),
+ ):
+ async with TestClient(TestServer(self._app(adapter))) as cli:
+ resp = await cli.post("/p/worker/webhooks/gh", data=body, headers=headers)
+ assert resp.status == 202
+ await asyncio.sleep(0.05)
+ assert len(seen) == 1
+ assert seen[0].source.profile == "worker"
+ assert "Body of worker-only." in seen[0].text
+
def test_route_profile_validation_fails_closed():
assert WebhookAdapter._route_allows_profile({}, None) is True
diff --git a/tests/gateway/test_wecom.py b/tests/gateway/test_wecom.py
index a46a1caded..96886df995 100644
--- a/tests/gateway/test_wecom.py
+++ b/tests/gateway/test_wecom.py
@@ -76,6 +76,40 @@ class TestWeComAdapterAuthzScope:
assert adapter._dm_policy == "pairing"
assert adapter._allow_from == []
+ def test_scoped_construction_reads_bot_id_from_scope_not_environ(self, multiplex_on, monkeypatch):
+ """bot_id must honor the same scope as its neighboring _secret read
+ (both are read on adjacent lines in __init__) -- a secondary profile's
+ own bot_id must never fall back to the default profile's os.environ
+ value."""
+ from agent import secret_scope
+ from plugins.platforms.wecom.adapter import WeComAdapter
+
+ monkeypatch.setenv("WECOM_BOT_ID", "default-profile-bot-id")
+ monkeypatch.setenv("WECOM_SECRET", "default-profile-secret")
+ token = secret_scope.set_secret_scope(
+ {"WECOM_BOT_ID": "scoped-bot-id", "WECOM_SECRET": "scoped-secret"}
+ )
+ try:
+ adapter = WeComAdapter(PlatformConfig(enabled=True))
+ finally:
+ secret_scope.reset_secret_scope(token)
+ assert adapter._bot_id == "scoped-bot-id"
+ assert adapter._secret == "scoped-secret"
+
+ def test_scoped_miss_does_not_leak_default_profiles_bot_id(self, multiplex_on, monkeypatch):
+ from agent import secret_scope
+ from plugins.platforms.wecom.adapter import DEFAULT_WS_URL, WeComAdapter
+
+ monkeypatch.setenv("WECOM_BOT_ID", "default-profile-bot-id")
+ monkeypatch.setenv("WECOM_WEBSOCKET_URL", "wss://default-profile.example/ws")
+ token = secret_scope.set_secret_scope({"SOMETHING_ELSE": "x"})
+ try:
+ adapter = WeComAdapter(PlatformConfig(enabled=True))
+ finally:
+ secret_scope.reset_secret_scope(token)
+ assert adapter._bot_id == ""
+ assert adapter._ws_url == DEFAULT_WS_URL
+
class TestWeComConnect:
diff --git a/tests/hermes_cli/test_alibaba_coding_plan_cn_provider_listing.py b/tests/hermes_cli/test_alibaba_coding_plan_cn_provider_listing.py
new file mode 100644
index 0000000000..6dc59e3338
--- /dev/null
+++ b/tests/hermes_cli/test_alibaba_coding_plan_cn_provider_listing.py
@@ -0,0 +1,30 @@
+"""alibaba-coding-plan and alibaba-coding-plan-cn must not both appear in the
+/model picker off a single shared key (#101122).
+
+The CN profile now has its own ALIBABA_CODING_PLAN_CN_API_KEY (checked first),
+keeping the shared ALIBABA_CODING_PLAN_API_KEY / DASHSCOPE_API_KEY as ordered
+fallbacks so existing CN users are not broken. The picker hides a ``-cn`` row
+whose only lit vars are shared with a lit non-CN sibling row.
+"""
+
+import os
+from unittest.mock import patch
+
+from hermes_cli.model_switch import list_authenticated_providers
+
+_CLEAR = {k: "" for k in ("ALIBABA_CODING_PLAN_API_KEY", "ALIBABA_CODING_PLAN_CN_API_KEY", "DASHSCOPE_API_KEY")}
+
+
+def _alibaba_slugs(current_provider=""):
+ return [p["slug"] for p in list_authenticated_providers(current_provider=current_provider) if "coding-plan" in p["slug"]]
+
+
+@patch.dict(os.environ, {**_CLEAR, "ALIBABA_CODING_PLAN_CN_API_KEY": "sk-cn-fake"}, clear=False)
+def test_alibaba_cn_appears_when_only_cn_key_set():
+ assert _alibaba_slugs() == ["alibaba-coding-plan-cn"]
+
+
+@patch.dict(os.environ, {**_CLEAR, "ALIBABA_CODING_PLAN_API_KEY": "sk-intl-fake"}, clear=False)
+def test_alibaba_cn_does_not_appear_when_only_intl_key_set():
+ """#101122: the shared intl key alone must light only the intl row."""
+ assert _alibaba_slugs() == ["alibaba-coding-plan"]
diff --git a/tests/hermes_cli/test_backup.py b/tests/hermes_cli/test_backup.py
index f0e8e52e4d..aa4c80056b 100644
--- a/tests/hermes_cli/test_backup.py
+++ b/tests/hermes_cli/test_backup.py
@@ -158,6 +158,94 @@ class TestShouldExclude:
# The .db itself is still included (and safe-copied separately)
assert not _should_exclude(Path("state.db"))
+ def test_excludes_managed_runtime_trees_at_root(self):
+ """models/, runtimes/, and node/ at a profile-home root hold
+ re-downloadable GGUF weights and runtime binaries that reach
+ hundreds of GB — zipping them is the 20-minute-hang symptom."""
+ from hermes_cli.backup import _should_exclude
+ assert _should_exclude(Path("models/Qwen3.6-27B-Q4_K_M.gguf"))
+ assert _should_exclude(Path("models/assets/mmproj.gguf"))
+ assert _should_exclude(Path("runtimes/llamacpp/b10362/cuda/ggml-cuda.dll"))
+ assert _should_exclude(Path("node/node.exe"))
+ # Named profiles download their own copies.
+ assert _should_exclude(Path("profiles/clean/models/big.gguf"))
+ assert _should_exclude(Path("profiles/clean/runtimes/llamacpp/x.dll"))
+
+ def test_keeps_nested_dirs_named_like_runtime_trees(self):
+ """A deeper directory that happens to be called models/ or node/ is
+ user data (a skill's assets, project files) and must survive."""
+ from hermes_cli.backup import _should_exclude
+ assert not _should_exclude(Path("skills/mlops/models/notes.md"))
+ assert not _should_exclude(Path("scratch/node/index.js"))
+ assert not _should_exclude(Path("profiles/clean/skills/x/models/a.txt"))
+
+ def test_excludes_desktop_emergency_state_db_baks(self):
+ """The desktop updater's pre-flight drops timestamped
+ state.db.pre-update-emergency-*.bak files at the HERMES_HOME root —
+ backup artifacts in the same class as backups/, so a full backup
+ must not re-ship them."""
+ from hermes_cli.backup import _should_exclude
+ assert _should_exclude(
+ Path("state.db.pre-update-emergency-2026-08-15T04-55-33-619Z.bak")
+ )
+ assert _should_exclude(
+ Path("profiles/coder/state.db.pre-update-emergency-2026-08-15T04-55-33-619Z.bak")
+ )
+ # Other .bak files are user data and stay.
+ assert not _should_exclude(Path("config.yaml.bak"))
+
+
+# ---------------------------------------------------------------------------
+# _iter_backup_files tests
+# ---------------------------------------------------------------------------
+
+class TestIterBackupFiles:
+ def test_manual_and_automatic_paths_share_one_walk(self, tmp_path):
+ """Both backup entry points must select the identical file set.
+
+ Before the walks were unified, the automatic pre-update path pruned
+ ``hermes-agent`` at ANY depth, silently dropping nested skill dirs
+ like ``skills/autonomous-ai-agents/hermes-agent/`` that the manual
+ path preserved. One shared iterator makes that drift impossible;
+ this test pins the contract."""
+ from hermes_cli.backup import _iter_backup_files
+
+ root = tmp_path / ".hermes"
+ root.mkdir()
+ _make_hermes_tree(root)
+
+ # The case the old automatic walk got wrong: a nested dir named
+ # hermes-agent holding real skill content.
+ nested = root / "skills" / "autonomous-ai-agents" / "hermes-agent"
+ nested.mkdir(parents=True)
+ (nested / "SKILL.md").write_text("# nested skill\n")
+
+ # A root-level managed runtime tree that both paths must prune.
+ (root / "models").mkdir()
+ (root / "models" / "big.gguf").write_bytes(b"\x00" * 64)
+
+ out_path = tmp_path / "out.zip"
+ selected = {str(rel) for _, rel in _iter_backup_files(root, out_path)}
+
+ rel_nested = str(Path("skills/autonomous-ai-agents/hermes-agent/SKILL.md"))
+ assert rel_nested in selected
+ assert str(Path("models/big.gguf")) not in selected
+ assert not any(s.startswith("hermes-agent") for s in selected)
+
+ def test_skipped_dirs_collected_for_summary(self, tmp_path):
+ from hermes_cli.backup import _iter_backup_files
+
+ root = tmp_path / ".hermes"
+ root.mkdir()
+ _make_hermes_tree(root)
+ (root / "models").mkdir()
+ (root / "models" / "big.gguf").write_bytes(b"\x00")
+
+ skipped: set = set()
+ list(_iter_backup_files(root, tmp_path / "out.zip", skipped))
+ assert "models" in skipped
+ assert "hermes-agent" in skipped
+
# ---------------------------------------------------------------------------
# Backup tests
diff --git a/tests/hermes_cli/test_boot_preset_staleness.py b/tests/hermes_cli/test_boot_preset_staleness.py
new file mode 100644
index 0000000000..fd7256c8bf
--- /dev/null
+++ b/tests/hermes_cli/test_boot_preset_staleness.py
@@ -0,0 +1,126 @@
+"""Every staged model must launch with a policy decision, never stock fit.
+
+The managed server autoloads any GGUF in its models dir; a model missing
+from the preset INI loads with llama-server defaults (f16 KV at max
+context, no placement) — on Windows/WDDM that silently demotes VRAM and
+decodes at a crawl. Boot must therefore refuse to adopt a running server
+whose presets predate the staged set."""
+
+from __future__ import annotations
+
+import pytest
+
+
+@pytest.fixture
+def hermes_home(tmp_path, monkeypatch):
+ home = tmp_path / ".hermes"
+ home.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ return home
+
+
+def _stage(home, name):
+ mdir = home / "models"
+ mdir.mkdir(parents=True, exist_ok=True)
+ (mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 32)
+
+
+def _write_presets(home, *model_ids):
+ pdir = home / "runtimes" / "llamacpp"
+ pdir.mkdir(parents=True, exist_ok=True)
+ body = "\n".join(f"[{m}]\nctx-size = 65536\n" for m in model_ids)
+ (pdir / "presets.ini").write_text(body, encoding="utf-8")
+
+
+def test_presets_stale_when_a_staged_model_has_no_section(hermes_home):
+ from hermes_cli.local_runtime.bootstrap import _presets_stale
+
+ _stage(hermes_home, "model-a")
+ _stage(hermes_home, "model-b")
+ _write_presets(hermes_home, "model-a")
+ assert _presets_stale() is True
+
+
+def test_presets_current_when_every_staged_model_is_covered(hermes_home):
+ from hermes_cli.local_runtime.bootstrap import _presets_stale
+
+ _stage(hermes_home, "model-a")
+ _write_presets(hermes_home, "model-a")
+ assert _presets_stale() is False
+
+
+def test_no_models_is_never_stale(hermes_home):
+ from hermes_cli.local_runtime.bootstrap import _presets_stale
+
+ _write_presets(hermes_home, "model-a")
+ assert _presets_stale() is False
+
+
+def test_boot_replaces_incumbent_with_stale_presets(hermes_home, monkeypatch):
+ """ensure_local_runtime must not adopt a running server whose presets
+ miss a staged model — it stops it and boots fresh (boot itself is
+ stubbed; the contract under test is the adopt/replace decision)."""
+ import hermes_cli.local_runtime.bootstrap as boot
+
+ _stage(hermes_home, "model-a")
+ _stage(hermes_home, "model-b")
+ _write_presets(hermes_home, "model-a")
+
+ stopped = {}
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.endpoint._state_endpoint",
+ lambda: {"base_url": "http://127.0.0.1:18434/v1", "pid": 12345})
+ monkeypatch.setattr(boot, "_stop_state_server",
+ lambda state: stopped.setdefault("pid", state["pid"]))
+
+ sentinel = object()
+
+ def fake_boot(*a, **k):
+ raise _BootReached()
+
+ class _BootReached(Exception):
+ pass
+
+ # Fail fast once boot proper begins — reaching it IS the assertion.
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.ensure_runtime_installed", fake_boot)
+
+ result = boot.ensure_local_runtime({"local_runtime": {"enabled": True}})
+ assert stopped.get("pid") == 12345, "stale incumbent was not stopped"
+ # Boot proceeded past adoption (our fake raised inside the try block,
+ # which ensure_local_runtime swallows into a None return).
+ assert result is None or result is sentinel
+
+
+def test_refresh_bounces_an_adopted_server(hermes_home, monkeypatch):
+ """refresh_local_runtime with no in-process supervisor but a running
+ state-file server (the post-restart shape) must stop that server and
+ boot fresh — NOT silently no-op. Regression: the no-op meant every
+ download/delete after a backend restart left the router serving a
+ stale model catalog, and picking the new model failed with
+ 'not found in this provider's model listing'."""
+ import hermes_cli.local_runtime.bootstrap as boot
+
+ stopped = {}
+ monkeypatch.setattr(boot, "_SUPERVISOR", None)
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.endpoint._state_endpoint",
+ lambda: {"base_url": "http://127.0.0.1:18434/v1", "pid": 4242})
+ monkeypatch.setattr(boot, "_stop_state_server",
+ lambda state: stopped.setdefault("pid", state["pid"]))
+ booted = {}
+ monkeypatch.setattr(boot, "ensure_local_runtime",
+ lambda cfg, force=False: booted.setdefault("force", force) or object())
+
+ assert boot.refresh_local_runtime() is True
+ assert stopped.get("pid") == 4242, "adopted server was not stopped"
+ assert booted.get("force") is True, "fresh boot did not follow the stop"
+
+
+def test_refresh_no_server_anywhere_is_a_noop(hermes_home, monkeypatch):
+ import hermes_cli.local_runtime.bootstrap as boot
+
+ monkeypatch.setattr(boot, "_SUPERVISOR", None)
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.endpoint._state_endpoint", lambda: None)
+ assert boot.refresh_local_runtime() is False
diff --git a/tests/hermes_cli/test_budget_source.py b/tests/hermes_cli/test_budget_source.py
new file mode 100644
index 0000000000..744997f2be
--- /dev/null
+++ b/tests/hermes_cli/test_budget_source.py
@@ -0,0 +1,51 @@
+"""Launch and growth decisions must price against CAPACITY, not live-free
+VRAM. Both execute through a server bounce — the outgoing instance's memory
+is freed before the new one loads — so a probe that reads the predecessor's
+(or the grown model's own) residency as 'gone' vetoes configurations that
+genuinely fit. Symptom when this regresses: a model the pane promised
+'144K on GPU' launches with its weights pinned to CPU and single-digit
+tokens/s while the card sits 60% empty."""
+
+from __future__ import annotations
+
+import ast
+import inspect
+
+
+def _planning_probe_calls(source: str) -> list[bool]:
+ """Every probe_budget(...) call's planning= value in the source."""
+ tree = ast.parse(source)
+ out = []
+ for node in ast.walk(tree):
+ if (isinstance(node, ast.Call)
+ and getattr(node.func, "id", getattr(node.func, "attr", ""))
+ == "probe_budget"):
+ planning = any(
+ kw.arg == "planning"
+ and isinstance(kw.value, ast.Constant)
+ and kw.value.value is True
+ for kw in node.keywords)
+ out.append(planning)
+ return out
+
+
+def test_bootstrap_presets_price_against_capacity():
+ import hermes_cli.local_runtime.bootstrap as bootstrap
+
+ calls = _planning_probe_calls(inspect.getsource(bootstrap))
+ assert calls, "bootstrap no longer probes a budget? update this test"
+ assert all(calls), (
+ "bootstrap prices launch decisions against live-free VRAM; a "
+ "restart/refresh probes while the outgoing server still holds the "
+ "card, pinning fitting models to CPU")
+
+
+def test_growth_refit_prices_against_capacity():
+ import hermes_cli.local_runtime.growth as growth
+
+ calls = _planning_probe_calls(inspect.getsource(growth))
+ assert calls, "growth no longer probes a budget? update this test"
+ assert all(calls), (
+ "growth re-fits against live-free VRAM; the grown model's own "
+ "residency reads as unavailable and vetoes rungs that fit the "
+ "post-bounce card")
diff --git a/tests/hermes_cli/test_catalog_json.py b/tests/hermes_cli/test_catalog_json.py
new file mode 100644
index 0000000000..9aa8973b10
--- /dev/null
+++ b/tests/hermes_cli/test_catalog_json.py
@@ -0,0 +1,118 @@
+"""The pulled catalog: packaged JSON is the offline truth, a GitHub fetch
+swaps entries in memory only, and min_engine gates day-0 models.
+
+Nothing here touches disk beyond the packaged file — the design constraint
+is that a git checkout must never see a dirty tracked catalog.json."""
+
+from __future__ import annotations
+
+import dataclasses
+import io
+import json
+import urllib.request
+
+import pytest
+
+import hermes_cli.local_runtime.catalog as cat
+
+
+@pytest.fixture(autouse=True)
+def _reset_refresh_state(monkeypatch):
+ """Each test starts outside the TTL window with the packaged catalog."""
+ monkeypatch.setattr(cat, "_last_refresh_attempt", 0.0)
+ packaged = cat._packaged_catalog()
+ monkeypatch.setattr(cat, "CATALOG", packaged)
+ yield
+
+
+def _doc_from(entries):
+ """A fetchable catalog document built by mutating the packaged JSON."""
+ from importlib.resources import files
+
+ doc = json.loads(files("hermes_cli.local_runtime")
+ .joinpath("catalog.json").read_text(encoding="utf-8"))
+ doc["models"] = entries(doc["models"])
+ return doc
+
+
+def _fetch_returns(monkeypatch, doc):
+ body = json.dumps(doc).encode()
+
+ class R(io.BytesIO):
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ monkeypatch.setattr(urllib.request, "urlopen",
+ lambda *a, **k: R(body))
+
+
+def test_packaged_json_round_trips_the_catalog():
+ """The packaged JSON must produce a complete, selection-ready catalog:
+ every entry carries estimator inputs and at least one variant, and the
+ known invariants (best-first ordering, Q4 floor) hold — the same
+ contract the literals obeyed."""
+ assert len(cat.CATALOG) >= 4
+ for e in cat.CATALOG:
+ assert e.variants and e.n_ctx_train > 0 and e.per_layer_f16 >= 0
+ sizes = [v.size_bytes for v in e.variants]
+ assert sizes == sorted(sizes, reverse=True), f"{e.id} not best-first"
+
+
+def test_refresh_swaps_in_memory_only(monkeypatch, tmp_path):
+ """A fetched catalog replaces CATALOG in memory; the packaged file on
+ disk is untouched (checkout stays clean)."""
+ from importlib.resources import files
+
+ packaged_path = files("hermes_cli.local_runtime").joinpath("catalog.json")
+ before = packaged_path.read_text(encoding="utf-8")
+
+ def add_day0(models):
+ day0 = dict(models[0])
+ day0.update(id="day0-model", display_name="Day 0",
+ description="new", min_engine="b99999")
+ return models + [day0]
+
+ _fetch_returns(monkeypatch, _doc_from(add_day0))
+ assert cat.refresh_catalog(force=True) is True
+ assert "day0-model" in {e.id for e in cat.CATALOG}
+ assert cat.catalog_by_id()["day0-model"].min_engine == "b99999"
+ assert packaged_path.read_text(encoding="utf-8") == before
+
+
+def test_refresh_failure_keeps_current_catalog(monkeypatch):
+ def boom(*a, **k):
+ raise OSError("offline")
+
+ monkeypatch.setattr(urllib.request, "urlopen", boom)
+ ids_before = [e.id for e in cat.CATALOG]
+ assert cat.refresh_catalog(force=True) is False
+ assert [e.id for e in cat.CATALOG] == ids_before
+
+
+def test_refresh_rejects_wrong_schema(monkeypatch):
+ doc = _doc_from(lambda m: m)
+ doc["schema_version"] = 2
+ _fetch_returns(monkeypatch, doc)
+ ids_before = [e.id for e in cat.CATALOG]
+ assert cat.refresh_catalog(force=True) is False
+ assert [e.id for e in cat.CATALOG] == ids_before
+
+
+def test_loader_ignores_unknown_fields():
+ doc = _doc_from(lambda m: m)
+ doc["models"][0]["future_field"] = {"anything": True}
+ entries = cat._load_catalog(doc)
+ assert entries[0].id == doc["models"][0]["id"]
+
+
+def test_min_engine_gate(monkeypatch):
+ from hermes_cli.web_routers.local_models import _engine_too_old
+
+ monkeypatch.setattr("hermes_cli.local_runtime.binaries.installed_tags",
+ lambda: ["b10362"])
+ assert _engine_too_old("") is False, "no requirement, no gate"
+ assert _engine_too_old("b10000") is False, "installed engine suffices"
+ assert _engine_too_old("b10363") is True, "newer requirement gates"
diff --git a/tests/hermes_cli/test_catalog_reachability.py b/tests/hermes_cli/test_catalog_reachability.py
new file mode 100644
index 0000000000..1b7fa5b991
--- /dev/null
+++ b/tests/hermes_cli/test_catalog_reachability.py
@@ -0,0 +1,59 @@
+"""Catalog reachability: every entry's repo and files must exist upstream.
+
+Existence is the contract; SIZES are advisory (they feed the estimator and
+progress bars, and downloads deliberately tolerate a stale size when
+upstream re-uploads — completeness is judged against the server's own
+declared length, never the catalog). Size drift prints as a warning so a
+catalog refresh can be batched deliberately; only a MISSING file or repo
+fails.
+
+Network-marked (skipped in hermetic CI unless explicitly enabled) — this is
+the test that catches wrong repo names (the Nemotron 401) and moved files.
+Run before any catalog commit:
+
+ HERMES_TEST_NETWORK=1 scripts/run_tests.sh tests/hermes_cli/test_catalog_reachability.py
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import urllib.request
+
+import pytest
+
+pytestmark = pytest.mark.skipif(
+ not os.environ.get("HERMES_TEST_NETWORK"),
+ reason="network test; set HERMES_TEST_NETWORK=1 to run",
+)
+
+
+def test_every_catalog_file_resolves():
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ problems = []
+ drift = []
+ for entry in CATALOG:
+ url = f"https://huggingface.co/api/models/{entry.repo}/tree/main?recursive=true"
+ try:
+ with urllib.request.urlopen(url, timeout=30) as r:
+ files = {f["path"]: f.get("size") for f in json.load(r)}
+ except Exception as exc: # noqa: BLE001
+ problems.append(f"{entry.id}: repo {entry.repo} unreachable ({exc})")
+ continue
+ for variant in entry.variants:
+ for asset in entry.download_files(variant):
+ if asset.path not in files:
+ problems.append(
+ f"{entry.id}/{variant.quant}: {asset.path} not in {entry.repo}")
+ continue
+ live_size = files[asset.path]
+ if live_size and live_size != asset.size_bytes:
+ drift.append(
+ f"{entry.id}/{variant.quant}: size drift on {asset.path} — "
+ f"catalog {asset.size_bytes} vs live {live_size}")
+ if drift:
+ print("\nADVISORY size drift (downloads tolerate this; refresh when convenient):")
+ print("\n".join(drift))
+ assert not problems, "\n".join(problems)
+
diff --git a/tests/hermes_cli/test_catalog_variants.py b/tests/hermes_cli/test_catalog_variants.py
new file mode 100644
index 0000000000..4912be4ec4
--- /dev/null
+++ b/tests/hermes_cli/test_catalog_variants.py
@@ -0,0 +1,182 @@
+"""Variant-selection contracts: fit the catalog's single Q4-class build
+to a machine and price it honestly. Pure decision-table tests over
+synthetic budgets."""
+
+from __future__ import annotations
+
+import pytest
+
+from hermes_cli.local_runtime.catalog import (
+ CATALOG,
+ catalog_by_id,
+ find_entry_for_model,
+ select_variant,
+)
+from hermes_cli.local_runtime.estimator import HardwareBudget
+
+GIB = 1 << 30
+
+
+def budget(vram_gib: float, ram_gib: float = 64) -> HardwareBudget:
+ return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
+ total_device_bytes=int(vram_gib * GIB),
+ ram_available_bytes=int(ram_gib * GIB))
+
+
+def test_every_entry_ships_exactly_one_q4_build():
+ """No quant ladder: one Q4-class build per entry (K_M where the repo
+ ships it, XL elsewhere) — the quant class current engines optimize
+ for. Nothing below Q4 ever ships. Validation status is explicit per
+ variant in catalog.json; unvalidated builds are permitted (day-0
+ entries) and surface as unbadged rows in the pane."""
+ for entry in CATALOG:
+ assert len(entry.variants) == 1, (
+ f"{entry.id}: {len(entry.variants)} variants — expected exactly one")
+ build = entry.variants[0]
+ assert build.quant.startswith(("UD-Q4", "Q4")), (
+ f"{entry.id}: ships {build.quant}, not a Q4-class build")
+ for asset in entry.download_files(build):
+ assert asset.size_bytes > 0, f"{entry.id}: no size on {asset.path}"
+
+
+def test_split_variants_have_coherent_parts():
+ """Multi-file variants: same model_id from every part, exact sizes,
+ first file is the load target."""
+ entry = catalog_by_id()["deepseek-v4-flash"]
+ for v in entry.variants:
+ assert len(v.files) >= 2, "deepseek ships split GGUFs"
+ assert "00001-of" in v.files[0].path, "first part must be the load target"
+ assert v.size_bytes == sum(f.size_bytes for f in v.files)
+ assert entry.draft is not None, "DSpark draft rides along"
+
+
+def test_selection_is_the_q4_build_even_with_headroom():
+ """The selector picks the Q4 build even when bigger quants would fit
+ with room to spare — headroom buys window, not quant. Larger builds
+ stay one tile click away in the pane."""
+ entry = catalog_by_id()["qwen3.8-27b"]
+ choice = select_variant(entry, budget(60))
+ assert choice is not None
+ assert choice.zero_spill
+ assert choice.variant.quant == entry.variants[-1].quant # the Q4 rung
+ assert choice.reason_key == "best-large-window"
+
+
+def test_selected_build_constant_and_fit_shape_monotone_in_vram():
+ """More VRAM never changes the selected build (always the Q4 rung);
+ what improves is the fit shape: spilled -> floor -> target window."""
+ entry = catalog_by_id()["qwen3.8-27b"]
+ quants = set()
+ shapes = []
+ rank = {"smallest-fits-spilled": 0, "best-fits": 1, "best-large-window": 2}
+ for vram in (8, 12, 16, 24, 32, 48):
+ choice = select_variant(entry, budget(vram))
+ assert choice is not None
+ quants.add(choice.variant.quant)
+ shapes.append(rank[choice.reason_key])
+ assert quants == {entry.variants[-1].quant}, f"selection not constant: {quants}"
+ assert shapes == sorted(shapes), f"fit shape not monotone in VRAM: {shapes}"
+
+
+def test_small_card_gets_q4_spilled_never_below():
+ """8 GiB card + 27B: nothing zero-spills. The floor holds — the
+ selector offers Q4 spilled (priced honestly), never a sub-Q4 build."""
+ entry = catalog_by_id()["qwen3.8-27b"]
+ choice = select_variant(entry, budget(8))
+ assert choice is not None
+ assert not choice.zero_spill
+ assert choice.reason_key == "smallest-fits-spilled"
+ assert choice.variant.quant == "UD-Q4_K_M"
+
+
+def test_frontier_model_refused_on_consumer_card_offered_on_big_ram():
+ """DeepSeek V4 Flash (161 GB at Q4): refused outright on a 32 GiB-RAM
+ desktop; offered spilled on a 192 GiB-RAM workstation. The catalog
+ carries frontier hardware honestly instead of hiding the model."""
+ entry = catalog_by_id()["deepseek-v4-flash"]
+ assert select_variant(entry, budget(32, ram_gib=32)) is None
+ big = select_variant(entry, budget(32, ram_gib=192))
+ assert big is not None and not big.zero_spill
+
+
+def test_selection_accounts_for_kv_not_just_weights():
+ """The zero-spill check prices weights + KV, not weights alone: give a
+ machine exactly enough VRAM for the build's weights and the fit must
+ come back spilled, not zero-spill."""
+ entry = catalog_by_id()["qwen3.8-27b"]
+ build = entry.variants[0]
+ exactly_weights = HardwareBudget(
+ usable_vram_bytes=build.size_bytes + (100 << 20),
+ total_device_bytes=build.size_bytes + (100 << 20),
+ ram_available_bytes=64 * GIB)
+ choice = select_variant(entry, exactly_weights)
+ assert choice is not None
+ assert not choice.zero_spill, "KV cost ignored — weights alone can't zero-spill"
+
+
+def test_floor_fallback_when_target_window_does_not_fit():
+ """Cards where nothing clears the target keep the old rule: highest
+ quality that zero-spills at the 64K floor (reason 'best-fits'), never
+ a needless step down."""
+ entry = catalog_by_id()["qwen3.8-27b"]
+ # ~23.5 GiB usable: Q4 weights (16.7 GiB in-memory) + floor KV (2.2)
+ # + overhead (1.5 + 0.9 mmproj + ~1.0 MTP-posture logits) fits, but
+ # the 144K-target KV (+2.7 more) does not.
+ choice = select_variant(entry, budget(23.5))
+ assert choice is not None and choice.zero_spill
+ assert choice.reason_key == "best-fits"
+ assert choice.variant.quant == "UD-Q4_K_M"
+
+
+def test_target_never_degrades_below_floor_choice():
+ """The target preference may only IMPROVE the window, never the
+ floor guarantees: whenever the old floor rule found a zero-spill pick,
+ the new rule also finds one (possibly a smaller quant, never spill)."""
+ for entry in CATALOG:
+ for vram in (8, 12, 16, 24, 32, 48, 96):
+ choice = select_variant(entry, budget(vram, ram_gib=256))
+ if choice is None:
+ continue
+ # Rule 2: whatever was chosen zero-spill must genuinely clear
+ # the floor (the selector's own invariant, re-checked).
+ if choice.zero_spill:
+ assert choice.reason_key in ("best-large-window", "best-fits")
+
+
+def test_find_entry_for_model_resolves_split_ids():
+ hit = find_entry_for_model("DeepSeek-V4-Flash-0731-UD-Q4_K_XL")
+ assert hit is not None
+ entry, variant = hit
+ assert entry.id == "deepseek-v4-flash"
+ assert variant.quant == "UD-Q4_K_XL"
+
+
+def test_hybrid_long_context_stays_cheap():
+ """The reason Nemotron/Qwen3.6 headline the catalog: their priced
+ 64K-floor KV must be a small fraction of a dense model's."""
+ from hermes_cli.local_runtime.catalog import FLOOR
+ from hermes_cli.local_runtime.estimator import ctx_bytes
+
+ from hermes_cli.local_runtime.estimator import LayerKind, ModelProfile
+
+ hybrid = catalog_by_id()["qwen3.6-35b-a3b"]
+ hybrid_profile = hybrid.profile(hybrid.variants[-1])
+ # A fully-dense profile of the same layer count and per-layer cost:
+ # the contract is about LAYER ECONOMICS (recurrent layers pay no
+ # per-token KV), not about any particular catalog entry.
+ n_layers = len(hybrid_profile.layers)
+ dense_profile = ModelProfile(
+ name="synthetic-dense", weights_bytes=hybrid_profile.weights_bytes,
+ embd_table_bytes=0, n_ctx_train=hybrid.n_ctx_train,
+ layers=[(LayerKind.FULL, hybrid.per_layer_f16)] * n_layers)
+ dense_kv = ctx_bytes(dense_profile, FLOOR)
+ hybrid_kv = ctx_bytes(hybrid_profile, FLOOR)
+ # The contract is structural: recurrent layers pay no per-token KV,
+ # so the hybrid's KV must track its full-attention share (x kv_scale
+ # for MTP's draft context), not its total layer count.
+ full = sum(1 for kind, _ in hybrid_profile.layers if kind == LayerKind.FULL)
+ expected = dense_kv * full / n_layers * hybrid_profile.kv_scale
+ assert hybrid_kv < dense_kv, "hybrid must be cheaper than dense"
+ assert abs(hybrid_kv - expected) / expected < 0.25, (
+ f"hybrid KV ({hybrid_kv:,}) should track its full-attention share "
+ f"(expected ~{expected:,.0f})")
diff --git a/tests/hermes_cli/test_config.py b/tests/hermes_cli/test_config.py
index 968649e58a..0db5db5226 100644
--- a/tests/hermes_cli/test_config.py
+++ b/tests/hermes_cli/test_config.py
@@ -666,13 +666,23 @@ class TestOptionalEnvVarsRegistry:
from hermes_cli.config import OPTIONAL_ENV_VARS
assert OPTIONAL_ENV_VARS["KEENABLE_API_KEY"]["url"] == "https://keenable.ai"
- def test_removed_tavily_var_not_in_env_vars_by_version(self):
- """TAVILY_API_KEY was removed with the Tavily backend."""
+ def test_tavily_api_key_registered(self):
+ """TAVILY_API_KEY is listed in OPTIONAL_ENV_VARS."""
+ from hermes_cli.config import OPTIONAL_ENV_VARS
+ assert "TAVILY_API_KEY" in OPTIONAL_ENV_VARS
+
+ def test_tavily_api_key_has_url(self):
+ """TAVILY_API_KEY has a URL."""
+ from hermes_cli.config import OPTIONAL_ENV_VARS
+ assert OPTIONAL_ENV_VARS["TAVILY_API_KEY"]["url"] == "https://app.tavily.com/home"
+
+ def test_tavily_in_env_vars_by_version(self):
+ """TAVILY_API_KEY is listed in ENV_VARS_BY_VERSION."""
from hermes_cli.config import ENV_VARS_BY_VERSION
all_vars = []
for vars_list in ENV_VARS_BY_VERSION.values():
all_vars.extend(vars_list)
- assert "TAVILY_API_KEY" not in all_vars
+ assert "TAVILY_API_KEY" in all_vars
def test_max_iterations_not_offered_as_env_var(self):
"""HERMES_MAX_ITERATIONS must NOT be in OPTIONAL_ENV_VARS (issue #17534).
@@ -886,7 +896,9 @@ class TestConfigSupportFloor:
},
"memory": {"write_approval": True},
"model": {"default": "openai/gpt-5.4", "provider": "openrouter"},
- "model_catalog": {"ttl_hours": 1},
+ # v25 lowered the old 24h default to 1h; v40 drops that 1h default so
+ # the shipped ttl_minutes (20) applies.
+ "model_catalog": {},
"plugins": {"enabled": []},
"stt": {"provider": "local"},
}
@@ -905,7 +917,7 @@ class TestConfigSupportFloor:
# default (opt-in) so the write invariant strips it from disk.
"agent": {},
"model": {"default": "anthropic/claude-fable-5", "provider": "nous"},
- "model_catalog": {"ttl_hours": 1},
+ "model_catalog": {},
"plugins": {"disabled": ["foo"], "enabled": []},
}
diff --git a/tests/hermes_cli/test_config_env_expansion.py b/tests/hermes_cli/test_config_env_expansion.py
index 207ae5625f..6571015245 100644
--- a/tests/hermes_cli/test_config_env_expansion.py
+++ b/tests/hermes_cli/test_config_env_expansion.py
@@ -122,3 +122,31 @@ class TestLoadCliConfigExpansion:
config = load_cli_config()
assert config["auxiliary"]["vision"]["api_key"] == "${UNSET_CLI_VAR_ABC}"
+
+
+class TestExpansionUnderProfileScope:
+ """``${VAR}`` refs must resolve against the active profile's secret scope,
+ not the shared process environment (#84079): under multiplex every
+ secondary profile otherwise "had" the default profile's token and fanned
+ out. Outside multiplex the scope is an overlay and environ still applies."""
+
+ def test_scoped_ref_never_reads_another_profiles_environ(self, monkeypatch):
+ from agent import secret_scope as ss
+
+ monkeypatch.setenv("MATRIX_ACCESS_TOKEN", "default-token")
+ was_active = ss.is_multiplex_active()
+ ss.set_multiplex_active(True)
+ token = ss.set_secret_scope({"OTHER_KEY": "x"}) # profile-b: no matrix token
+ try:
+ assert _expand_env_vars("${MATRIX_ACCESS_TOKEN}") == "${MATRIX_ACCESS_TOKEN}"
+ assert _expand_env_vars("${env:MATRIX_ACCESS_TOKEN}") == "${env:MATRIX_ACCESS_TOKEN}"
+ finally:
+ ss.reset_secret_scope(token)
+ token = ss.set_secret_scope({"MATRIX_ACCESS_TOKEN": "c-token"})
+ try:
+ assert _expand_env_vars("${MATRIX_ACCESS_TOKEN}") == "c-token"
+ finally:
+ ss.reset_secret_scope(token)
+ ss.set_multiplex_active(was_active)
+ # Unscoped (default profile / single-profile CLI): legacy environ read.
+ assert _expand_env_vars("${MATRIX_ACCESS_TOKEN}") == "default-token"
diff --git a/tests/hermes_cli/test_config_set_platforms_redirect.py b/tests/hermes_cli/test_config_set_platforms_redirect.py
new file mode 100644
index 0000000000..803a70e58f
--- /dev/null
+++ b/tests/hermes_cli/test_config_set_platforms_redirect.py
@@ -0,0 +1,157 @@
+"""Regression tests for #71047 (Problem A): per-platform display settings.
+
+`hermes config set platforms.. ` must write to
+`display.platforms..` — the path the gateway actually
+reads (gateway/display_config.py::resolve_display_setting). Writing to the
+top-level `platforms.` block is silently ignored by the runtime, so the
+edit appeared to succeed while having no effect.
+"""
+
+from pathlib import Path
+
+import pytest
+import yaml
+
+
+def _write_config(hermes_home: Path, data: dict) -> Path:
+ hermes_home.mkdir(parents=True, exist_ok=True)
+ config_path = hermes_home / "config.yaml"
+ config_path.write_text(yaml.dump(data))
+ return config_path
+
+
+def _set(monkeypatch, hermes_home, key, value, force=False):
+ """Isolated call to set_config_value against a temp HERMES_HOME."""
+ monkeypatch.setenv("HERMES_HOME", str(hermes_home))
+ # set_config_value resolves the home live via get_config_path()/get_hermes_home()
+ from hermes_cli.config import set_config_value
+ set_config_value(key, value, force=force)
+
+
+@pytest.fixture
+def hermes_home(tmp_path, monkeypatch):
+ home = tmp_path / ".hermes"
+ # A config that already has a top-level platforms block (connection keys)
+ # AND a display.platforms block, mirroring the real-world report.
+ cfg = {
+ "model": {"default": "test-model", "provider": "openrouter"},
+ "platforms": {
+ "telegram": {"token": "secret-bot-token"},
+ },
+ "display": {
+ "skin": "default",
+ "platforms": {
+ "telegram": {"show_reasoning": True},
+ },
+ },
+ }
+ _write_config(home, cfg)
+ return home
+
+
+class TestPerPlatformDisplayRedirect:
+ def test_streaming_redirects_to_display_platforms(self, hermes_home, monkeypatch):
+ """platforms.telegram.streaming must land under display.platforms."""
+ _set(monkeypatch, hermes_home, "platforms.telegram.streaming", "false")
+
+ result = yaml.safe_load((hermes_home / "config.yaml").read_text())
+ # Redirected target exists and is correct
+ assert result["display"]["platforms"]["telegram"]["streaming"] is False
+ # Top-level platforms.telegram must NOT gain a streaming key
+ assert "streaming" not in result["platforms"]["telegram"]
+ # Connection key untouched
+ assert result["platforms"]["telegram"]["token"] == "secret-bot-token"
+
+ def test_show_reasoning_redirects(self, hermes_home, monkeypatch):
+ _set(monkeypatch, hermes_home, "platforms.telegram.show_reasoning", "false")
+ result = yaml.safe_load((hermes_home / "config.yaml").read_text())
+ assert result["display"]["platforms"]["telegram"]["show_reasoning"] is False
+
+ def test_tool_progress_redirects(self, hermes_home, monkeypatch):
+ # ``off`` is coerced to False by the bool-aware coercion in
+ # set_config_value; gateway/display_config._normalise turns False back
+ # into the canonical "off" string at read time, so the persisted value
+ # is the bool.
+ _set(monkeypatch, hermes_home, "platforms.discord.tool_progress", "off")
+ result = yaml.safe_load((hermes_home / "config.yaml").read_text())
+ assert result["display"]["platforms"]["discord"]["tool_progress"] is False
+
+ def test_connection_key_not_redirected(self, hermes_home, monkeypatch):
+ """A real connection key (token) stays in top-level platforms.."""
+ _set(monkeypatch, hermes_home, "platforms.telegram.token", "new-token")
+ result = yaml.safe_load((hermes_home / "config.yaml").read_text())
+ assert result["platforms"]["telegram"]["token"] == "new-token"
+ # Nothing leaked into display.platforms.telegram.token
+ assert "token" not in result["display"]["platforms"]["telegram"]
+
+ def test_no_top_level_platforms_created_when_missing(self, tmp_path, monkeypatch):
+ """When there is no pre-existing top-level platforms block, a display
+ setting write must not invent one."""
+ home = tmp_path / ".hermes"
+ _write_config(home, {"model": {"default": "m"}})
+ _set(monkeypatch, home, "platforms.telegram.streaming", "true")
+ result = yaml.safe_load((home / "config.yaml").read_text())
+ assert result["display"]["platforms"]["telegram"]["streaming"] is True
+ assert "platforms" not in result # no stray top-level platforms block
+
+
+class TestRedirectSiblingSurfaces:
+ """The canonicalization must hold for every CLI surface that takes a dotted
+ key — set, get, unset — and the written value must be what the gateway's
+ resolver actually reads (the #71047 symptom was CLI and runtime disagreeing).
+ """
+
+ def test_get_mirrors_gateway_resolution_after_set(self, hermes_home, monkeypatch, capsys):
+ from gateway.display_config import resolve_display_setting
+ from hermes_cli.config import get_config_value
+
+ _set(monkeypatch, hermes_home, "platforms.telegram.streaming", "false")
+ capsys.readouterr()
+ get_config_value("platforms.telegram.streaming")
+ assert capsys.readouterr().out.strip() == "false"
+
+ raw = yaml.safe_load((hermes_home / "config.yaml").read_text())
+ assert resolve_display_setting(raw, "telegram", "streaming") is False
+
+ def test_unset_removes_the_redirected_leaf(self, hermes_home, monkeypatch):
+ from hermes_cli.config import unset_config_value
+
+ _set(monkeypatch, hermes_home, "platforms.telegram.streaming", "false")
+ unset_config_value("platforms.telegram.streaming")
+ result = yaml.safe_load((hermes_home / "config.yaml").read_text())
+ assert "streaming" not in result["display"]["platforms"]["telegram"]
+ # Sibling display override and connection block untouched.
+ assert result["display"]["platforms"]["telegram"]["show_reasoning"] is True
+ assert result["platforms"]["telegram"] == {"token": "secret-bot-token"}
+
+ def test_unset_missing_redirected_leaf_exits_nonzero(self, hermes_home, monkeypatch):
+ from hermes_cli.config import unset_config_value
+
+ monkeypatch.setenv("HERMES_HOME", str(hermes_home))
+ with pytest.raises(SystemExit) as exc:
+ unset_config_value("platforms.telegram.streaming")
+ assert exc.value.code == 1
+
+ def test_set_prints_redirect_note(self, hermes_home, monkeypatch, capsys):
+ _set(monkeypatch, hermes_home, "platforms.telegram.streaming", "false")
+ out = capsys.readouterr().out
+ assert "saved as display.platforms.telegram.streaming" in out
+ assert "Set display.platforms.telegram.streaming = False" in out
+
+ def test_redirect_helper_only_touches_known_display_keys(self):
+ from gateway.display_config import OVERRIDEABLE_KEYS
+ from hermes_cli.config import _redirect_platform_display_key
+
+ for setting in OVERRIDEABLE_KEYS:
+ canonical, note = _redirect_platform_display_key(f"platforms.discord.{setting}")
+ assert canonical == f"display.platforms.discord.{setting}"
+ assert note
+ for key in (
+ "platforms.telegram.token",
+ "platforms.telegram.reply_to_mode",
+ "platforms.telegram.extra.foo", # 4 segments — not a display leaf
+ "platforms.telegram",
+ "display.platforms.telegram.streaming", # already canonical
+ "streaming.enabled",
+ ):
+ assert _redirect_platform_display_key(key) == (key, None)
diff --git a/tests/hermes_cli/test_container_boot.py b/tests/hermes_cli/test_container_boot.py
index 5838ffdb45..cb772eeb65 100644
--- a/tests/hermes_cli/test_container_boot.py
+++ b/tests/hermes_cli/test_container_boot.py
@@ -128,6 +128,49 @@ def test_running_profile_is_registered_and_autostarted(tmp_path: Path) -> None:
assert not (svc / "down").exists()
+@pytest.mark.parametrize(
+ "config_value,env_value,expected",
+ [
+ pytest.param("true", None, "registered", id="config-only-multiplex"),
+ pytest.param("true", "false", "started", id="env-false-overrides-config"),
+ ],
+)
+def test_boot_honors_config_multiplex_profiles(
+ tmp_path: Path,
+ monkeypatch: pytest.MonkeyPatch,
+ config_value: str,
+ env_value: str | None,
+ expected: str,
+) -> None:
+ """Container boot must resolve multiplex_profiles like the gateway does:
+ config.yaml opt-in honored (#85413), env override keeps precedence."""
+ scandir = tmp_path / "run-service"
+ scandir.mkdir()
+ _make_profile(tmp_path, "coder", state="running")
+ (tmp_path / "config.yaml").write_text(
+ f"multiplex_profiles: {config_value}\n",
+ encoding="utf-8",
+ )
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ if env_value is None:
+ monkeypatch.delenv("GATEWAY_MULTIPLEX_PROFILES", raising=False)
+ else:
+ monkeypatch.setenv("GATEWAY_MULTIPLEX_PROFILES", env_value)
+
+ actions = reconcile_profile_gateways(
+ hermes_home=tmp_path,
+ scandir=scandir,
+ dry_run=False,
+ )
+
+ assert _named_actions(actions) == [ReconcileAction(
+ profile="coder",
+ prior_state="running",
+ action=expected,
+ )]
+ assert (scandir / "gateway-coder" / "down").exists() is (expected == "registered")
+
+
def test_registered_profile_has_finish_script(tmp_path: Path) -> None:
"""The finish script must be written so s6 stops restarting on
fatal config errors (exit 78 → exit 125). See #51228."""
@@ -299,5 +342,3 @@ def _write_lifecycle_sentinel(profile_dir: Path, payload: dict) -> None:
(state_dir / "gateway.lifecycle.json").write_text(json.dumps(payload))
-
-
diff --git a/tests/hermes_cli/test_context_policy.py b/tests/hermes_cli/test_context_policy.py
new file mode 100644
index 0000000000..b98bcd49c1
--- /dev/null
+++ b/tests/hermes_cli/test_context_policy.py
@@ -0,0 +1,410 @@
+"""Context-policy decision-table tests (Rollout 3).
+
+Per the design's verification plan: synthetic per-layer profiles ->
+relationships, never exact numbers. Real-model spot checks pin the
+estimator to constants measured on real GGUFs (those ARE relationships —
+constants with tolerance bands, not change-detecting catalog snapshots).
+"""
+
+from __future__ import annotations
+
+import pytest
+
+from hermes_cli.local_runtime.context_policy import (
+ FLOOR,
+ SPEED_FLOOR_TOK_S,
+ GrowthDecision,
+ WindowDecision,
+ growth_decision,
+ initial_window,
+ ladder,
+ launch_args,
+ spill_overrides,
+ ub_logits_bytes,
+)
+from hermes_cli.local_runtime.estimator import (
+ HardwareBudget,
+ LayerKind,
+ ModelProfile,
+ PhysicsRefusal,
+ ctx_bytes,
+ kv_dtype_factor,
+ physics_check,
+)
+
+GIB = 1 << 30
+KIB = 1024
+
+
+# ── synthetic profiles (per-layer tuples, per the verification plan) ──
+
+
+def dense(name="dense-32b", layers=64, per_token_f16=4096, weights_gib=20,
+ native=128 * 1024) -> ModelProfile:
+ return ModelProfile(
+ name=name, weights_bytes=weights_gib * GIB, embd_table_bytes=0,
+ n_ctx_train=native,
+ layers=[(LayerKind.FULL, per_token_f16)] * layers)
+
+
+def hybrid(name="hybrid-30b", full_layers=16, recurrent_layers=48,
+ per_token_f16=4096, weights_gib=22, native=1024 * 1024) -> ModelProfile:
+ layers = ([(LayerKind.FULL, per_token_f16)] * full_layers
+ + [(LayerKind.RECURRENT, 0)] * recurrent_layers)
+ return ModelProfile(name=name, weights_bytes=weights_gib * GIB,
+ embd_table_bytes=0, n_ctx_train=native, layers=layers)
+
+
+def moe(name="moe-30b", layers=48, per_token_f16=3072, weights_gib=17,
+ native=256 * 1024) -> ModelProfile:
+ return ModelProfile(name=name, weights_bytes=weights_gib * GIB,
+ embd_table_bytes=0, n_ctx_train=native,
+ layers=[(LayerKind.FULL, per_token_f16)] * layers,
+ moe=True)
+
+
+def card(vram_gib, ram_gib=64, uma=False) -> HardwareBudget:
+ return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
+ total_device_bytes=int(vram_gib * GIB),
+ ram_available_bytes=int(ram_gib * GIB), uma=uma)
+
+
+# ── estimator invariants ─────────────────────────────────────
+
+
+def test_full_attention_linear_in_window():
+ p = dense()
+ b32, b64, b128 = (ctx_bytes(p, w * KIB) for w in (32, 64, 128))
+ assert abs(b64 / b32 - 2) < 0.01
+ assert abs(b128 / b64 - 2) < 0.01
+
+
+def test_recurrent_state_constant_in_window():
+ p = hybrid()
+ full_share_32 = ctx_bytes(p, 32 * KIB)
+ full_share_1m = ctx_bytes(p, 1024 * KIB)
+ # Grows only through the 16 full-attn layers — the recurrent share is
+ # identical, so the ratio tracks the full-attn ratio exactly.
+ full_only = ModelProfile(name="x", weights_bytes=0, embd_table_bytes=0,
+ n_ctx_train=p.n_ctx_train,
+ layers=[(LayerKind.FULL, 4096)] * 16)
+ expected_delta = ctx_bytes(full_only, 1024 * KIB) - ctx_bytes(full_only, 32 * KIB)
+ assert abs((full_share_1m - full_share_32) - expected_delta) <= 1
+
+
+def test_swa_layers_capped_at_window():
+ p = ModelProfile(
+ name="swa", weights_bytes=0, embd_table_bytes=0, n_ctx_train=128 * KIB,
+ layers=[(LayerKind.SWA, 4096)] * 5 + [(LayerKind.FULL, 4096)] * 1,
+ swa_window=1024)
+ small, big = ctx_bytes(p, 4 * KIB), ctx_bytes(p, 32 * KIB)
+ # Full layer grew 8x; the 5 SWA layers stayed capped at 1024 — total
+ # growth must land well under the all-full 8x (here ~4.1x).
+ assert big / small < 0.6 * 8
+
+
+def test_q8_factor_is_exactly_34_over_64():
+ assert kv_dtype_factor(True) == pytest.approx(34 / 64)
+ assert kv_dtype_factor(False) == 1.0
+
+
+def test_non_fa_fallback_doubles_ctx_cost():
+ p = dense()
+ assert ctx_bytes(p, FLOOR, flash_attention=False) == pytest.approx(
+ ctx_bytes(p, FLOOR, flash_attention=True) * 64 / 34, rel=0.001)
+
+
+def test_hybrid_vs_dense_100x_class_spread():
+ """The whole reason for the per-layer walk: equal-size models, ~100x
+ per-token spread between classic dense and a mostly-recurrent hybrid."""
+ d = dense(layers=64, per_token_f16=8192) # 256 KiB/tok class
+ h = hybrid(full_layers=4, recurrent_layers=60, per_token_f16=8192)
+ window = 256 * KIB
+ dense_cost = ctx_bytes(d, window)
+ hybrid_cost = ctx_bytes(h, window)
+ assert dense_cost / hybrid_cost > 10
+
+
+# ── measured-constant spot checks (real models, tolerance bands) ──
+
+
+def test_measured_dense_4b_per_token():
+ """Qwen3-4B: 36 layers x 8 kv-heads x (128+128) x 2B = 144 KiB/tok f16."""
+ p = ModelProfile(name="qwen3-4b", weights_bytes=0, embd_table_bytes=0,
+ n_ctx_train=262144,
+ layers=[(LayerKind.FULL, 8 * 256 * 2)] * 36)
+ per_token_bytes = ctx_bytes(p, 32 * KIB, flash_attention=False) / (32 * KIB)
+ assert per_token_bytes == pytest.approx(144 * KIB, rel=0.02)
+
+
+def test_measured_gdn_27b_per_token_q8():
+ """Qwen3.6-27B: 16 full-attn of 64; measured 34.0 KiB/tok @ q8 (B4).
+ Per-layer f16 = 34 KiB * 64/34 / 16 = 4 KiB."""
+ per_layer_f16 = 4 * KIB
+ kv_only = ModelProfile(name="kv", weights_bytes=0, embd_table_bytes=0,
+ n_ctx_train=262144,
+ layers=[(LayerKind.FULL, per_layer_f16)] * 16)
+ per_token_bytes = ctx_bytes(kv_only, 128 * KIB) / (128 * KIB)
+ assert per_token_bytes == pytest.approx(34 * KIB, rel=0.02)
+
+
+def test_measured_nemotron_1m_within_band():
+ """1M @ q8 measured 3264 MiB KV (B3): ~3.19 KiB/token TOTAL across the
+ 16 full-attn layers -> per-layer f16 ~384 B. Estimator must land in the
+ measured band, not the dense-formula 100x miss."""
+ p = hybrid(full_layers=16, recurrent_layers=46, per_token_f16=384,
+ native=1024 * KIB)
+ total = ctx_bytes(p, 1024 * KIB)
+ assert 2.5 * GIB < total < 4.0 * GIB
+
+
+# ── physics check ────────────────────────────────────────────
+
+
+def test_physics_refusal_only_past_vram_plus_ram():
+ p = dense(weights_gib=60)
+ ok = physics_check(p, card(24, ram_gib=64), FLOOR)
+ assert ok is None # 60 GiB weights fit in 24+64
+ refused = physics_check(p, card(24, ram_gib=16), FLOOR)
+ assert isinstance(refused, PhysicsRefusal)
+ assert "smaller quant" in refused.message
+
+
+def test_physics_check_prices_at_floor_not_native():
+ """A 1M-native hybrid must not be refused for its native window —
+ the check prices the floor only."""
+ p = hybrid(weights_gib=22)
+ assert physics_check(p, card(24, ram_gib=8), FLOOR) is None
+
+
+# ── ladder + initial window ──────────────────────────────────
+
+
+def test_ladder_shape():
+ rungs = ladder(262144)
+ assert rungs[0] == FLOOR
+ assert rungs[-1] == 262144
+ assert all(a < b for a, b in zip(rungs, rungs[1:]))
+ # geometric-ish: each step grows, none more than 2x
+ assert all(b / a <= 2.0 for a, b in zip(rungs, rungs[1:]))
+
+
+def test_initial_window_never_below_floor_and_never_above_native():
+ for profile in (dense(), hybrid(), moe(), dense(native=32 * KIB)):
+ for vram in (8, 16, 24, 32):
+ d = initial_window(profile, card(vram))
+ if isinstance(d, WindowDecision):
+ assert d.window >= min(FLOOR, profile.n_ctx_train)
+ assert d.window <= profile.n_ctx_train
+
+
+def test_initial_window_monotone_in_vram():
+ p = dense()
+ windows = []
+ for vram in (8, 12, 16, 24, 32, 48):
+ d = initial_window(p, card(vram))
+ assert isinstance(d, WindowDecision)
+ windows.append(d.window)
+ assert all(a <= b for a, b in zip(windows, windows[1:]))
+
+
+def test_flat_curve_reaches_native_where_dense_does_not():
+ """Design invariant: equal-size hybrid rides to native spill-free where
+ the dense model cannot. Hybrid KV priced at the B3 class (~3.2 KiB/tok
+ total: per-layer f16 384 B x 16 layers)."""
+ h = hybrid(weights_gib=18, native=1024 * KIB, per_token_f16=384)
+ d = dense(weights_gib=18, per_token_f16=8192, native=1024 * KIB)
+ vram = card(24)
+ dh = initial_window(h, vram)
+ dd = initial_window(d, vram)
+ assert isinstance(dh, WindowDecision) and isinstance(dd, WindowDecision)
+ assert dh.window == 1024 * KIB and not dh.spilled
+ assert dd.window < 1024 * KIB
+
+
+def test_dense_on_small_card_holds_floor_and_spills():
+ """The deliberate price of the guarantee (design table: dense 32B on
+ 24 GB starts at the floor with a few GiB spilled)."""
+ d = initial_window(dense(weights_gib=20), card(12))
+ assert isinstance(d, WindowDecision)
+ assert d.window == FLOOR
+ assert d.spilled
+
+
+def test_uma_budget_caps_the_window_through_physics():
+ """Unified memory needs no special context rule: the budget already
+ encodes the constraint (usable = RAM minus headroom, ram_available=0),
+ so the ladder stops where weights + KV genuinely stop fitting."""
+ p = hybrid(weights_gib=8, native=1024 * KIB)
+ unified = card(38.4, ram_gib=0, uma=True) # 48 GiB machine, 20% headroom
+ d = initial_window(p, unified)
+ assert isinstance(d, WindowDecision)
+ assert not d.spilled, "UMA budget must produce a resident decision"
+ need = 8 * GIB + ctx_bytes(p, d.window)
+ assert need <= unified.usable_vram_bytes
+
+
+# ── growth ───────────────────────────────────────────────────
+
+
+def _grow(profile, budget, **kw):
+ defaults = dict(current_window=FLOOR, session_tokens=int(FLOOR * 0.9),
+ measured_decode_tok_s=40.0, server_idle=True)
+ defaults.update(kw)
+ return growth_decision(profile, budget, **defaults)
+
+
+def test_growth_holds_below_occupancy():
+ d = _grow(dense(), card(24), session_tokens=int(FLOOR * 0.5))
+ assert d.action == "hold"
+
+
+def test_growth_requires_idle_server():
+ d = _grow(dense(), card(24), server_idle=False)
+ assert d.action == "hold"
+ assert "idle" in d.reason
+
+
+def test_growth_steps_one_rung_and_is_monotone():
+ p = dense(native=262144)
+ d = _grow(p, card(32))
+ assert d.action == "grow"
+ assert d.next_window > FLOOR
+ rungs = ladder(262144)
+ assert d.next_window == rungs[rungs.index(FLOOR) + 1]
+
+
+def test_growth_stops_at_native():
+ p = dense(native=128 * KIB)
+ d = _grow(p, card(48), current_window=128 * KIB,
+ session_tokens=int(128 * KIB * 0.9))
+ assert d.action == "compress-default"
+
+
+def test_speed_floor_flips_default_to_compression():
+ d = _grow(dense(), card(24), measured_decode_tok_s=SPEED_FLOOR_TOK_S - 2)
+ assert d.action == "compress-default"
+ assert "explicit per-session choice" in d.reason
+
+
+def test_growth_refits_against_live_budget():
+ """V3C: a rung that no longer fits (external pressure ate the memory)
+ is not granted."""
+ p = dense(weights_gib=20)
+ starved = card(2, ram_gib=1)
+ d = _grow(p, starved)
+ assert d.action == "compress-default"
+ assert "physics" in d.reason
+
+
+# ── spill placement + launch args ────────────────────────────
+
+
+def test_spill_overrides_prefer_expert_and_recurrent_ffn():
+ assert "exps" in " ".join(spill_overrides(moe()))
+ assert "ffn" in " ".join(spill_overrides(hybrid()))
+ assert spill_overrides(dense()) == []
+
+
+def test_launch_args_contract():
+ p = moe()
+ spilled = WindowDecision(window=FLOOR, spill_bytes=4 * GIB, kv_on_gpu=True)
+ resident = WindowDecision(window=131072, spill_bytes=0, kv_on_gpu=True)
+
+ a = launch_args(p, spilled, mtp_capable=True)
+ assert a[:2] == ["-c", str(FLOOR)] # explicit window, always
+ assert "q8_0" in a # q8 KV under flash attn
+ assert "-ot" in a # spill placement
+ assert "--spec-type" in a # MTP on spilled
+
+ # MTP is not gated on spill: resident decode measured +16% at depth 2.
+ b = launch_args(p, resident, mtp_capable=True, mtp_draft_depth=2)
+ assert "-ot" not in b, "placement is spill-only"
+ assert "--spec-type" in b, "MTP must run on resident configs too"
+ assert b[b.index("--spec-draft-n-max") + 1] == "2"
+ assert "--backend-sampling" in b
+ assert "--spec-draft-backend-sampling" in b
+
+ # Stacking MTP with the large microbatch is a FIT question, decided
+ # by the caller (presets' posture ladder) and passed as mtp_prefill.
+ # Default (no headroom proven): decode posture, small ubatch — the
+ # stacked logits buffers once packed a 32 GiB card 3.9 GiB past a
+ # fit that ignored them.
+ assert "-ub" not in b, "default MTP posture stays at the small ubatch"
+
+ # Headroom proven: the stacked posture carries the large microbatch
+ # (measured best on both axes where it fits: 93.3 vs 89.5 tok/s
+ # decode on Qwen3.8 Q4). ub_logits_bytes must price the same choice.
+ s = launch_args(p, resident, mtp_capable=True, mtp_draft_depth=2,
+ mtp_prefill=True)
+ assert "-ub" in s and s[s.index("-ub") + 1] == "2048"
+ assert "--spec-type" in s
+ v = 248320
+ assert ub_logits_bytes(v, mtp_capable=True) == 512 * v * 4 * 2
+ assert ub_logits_bytes(v, mtp_capable=True, mtp_prefill=True) == int(2048 * v * 4 * 1.5)
+
+ c = launch_args(p, resident, mtp_capable=False)
+ assert "-ub" in c and c[c.index("-ub") + 1] == "2048" # prefill hint
+ assert "--spec-type" not in c
+
+ d = launch_args(p, spilled, flash_attention=False, mtp_capable=False)
+ assert "q8_0" not in d # f16 on non-FA fallback
+
+
+def test_launch_args_uma_never_pins_tensors():
+ """On unified memory, -ot pinning is off even for spilled decisions:
+ "CPU" and "GPU" are the same silicon, and forcing FFN weights down
+ the host compute path measures far slower than letting the
+ allocator place everything. The discrete ~1.75x -ot win does not
+ transfer. Everything else about the launch shape is identical to
+ discrete."""
+ p = moe()
+ spilled = WindowDecision(window=FLOOR, spill_bytes=4 * GIB, kv_on_gpu=True)
+
+ u = launch_args(p, spilled, mtp_capable=False, uma=True)
+ assert "-ot" not in u, "UMA must never pin tensors to the host path"
+ assert u[:2] == ["-c", str(FLOOR)] # window contract unchanged
+ assert "q8_0" in u # KV policy unchanged
+
+ # Same call on discrete keeps the pinning — the flag is the ONLY delta.
+ disc = launch_args(p, spilled, mtp_capable=False, uma=False)
+ assert "-ot" in disc
+ assert [x for x in disc if x != "-ot" and not x.startswith("blk")] == \
+ [x for x in u if x != "-ot" and not x.startswith("blk")]
+
+
+def test_ub_logits_bytes_prices_the_flag_choice():
+ """The logits-buffer price must match the microbatch launch_args
+ chooses: 2048 x vocab x 4 for non-MTP, 512 x vocab x 4 x 2 for MTP
+ (draft context doubles it). 248320-vocab receipts: ~1.9 GiB at
+ ub2048, ~0.95 GiB under MTP."""
+ v = 248320
+ assert ub_logits_bytes(v, mtp_capable=False) == 2048 * v * 4
+ assert ub_logits_bytes(v, mtp_capable=True) == 512 * v * 4 * 2
+ assert ub_logits_bytes(0, mtp_capable=True) == 0 # unknown vocab: no charge
+
+
+def test_no_refusal_branch_past_physics():
+ """Design invariant: anything past the physics check is servable —
+ initial_window never refuses on its own."""
+ for vram in (4, 6, 8, 12):
+ d = initial_window(dense(weights_gib=20), card(vram, ram_gib=64))
+ assert isinstance(d, WindowDecision)
+
+
+def test_kv_scale_prices_mtp_draft_context():
+ """MTP profiles carry kv_scale > 1 (the draft context's KV share,
+ calibrated from measured server RSS); ctx_bytes must scale with it so
+ every consumer — launch fit, catalog rows, growth — prices what the
+ server actually allocates. Four-point calibration held within
+ +1.4 GiB conservative, never optimistic."""
+ import dataclasses
+
+ p = moe()
+ base = ctx_bytes(p, 131072)
+ scaled = ctx_bytes(dataclasses.replace(p, kv_scale=1.2), 131072)
+ assert scaled == int(base * 1.2)
+
+ # The safety direction: the estimate must never be BELOW measured.
+ # (Calibration receipts: predicted-measured was +233..+1400 MiB.)
+ assert scaled > base
diff --git a/tests/hermes_cli/test_copilot_in_model_list.py b/tests/hermes_cli/test_copilot_in_model_list.py
index 83832b0c33..2889627964 100644
--- a/tests/hermes_cli/test_copilot_in_model_list.py
+++ b/tests/hermes_cli/test_copilot_in_model_list.py
@@ -3,6 +3,8 @@
import os
from unittest.mock import patch
+import pytest
+
from hermes_cli.model_switch import list_authenticated_providers
@@ -20,3 +22,63 @@ def test_copilot_picker_uses_live_catalog_when_available():
assert copilot is not None
assert copilot["models"] == live_models
assert copilot["total_models"] == len(live_models)
+
+
+# --- copilot-acp: external_process availability (#63662) -------------------
+#
+# copilot-acp holds no API key, OAuth token, or credential-pool entry by
+# design — the spawned `copilot --acp --stdio` subprocess brings its own auth.
+# The picker loop used to filter it out unconditionally (has_creds never had
+# an external_process branch), so the provider was invisible in every picker
+# even with a perfectly resolvable executable.
+
+
+@pytest.fixture()
+def _no_other_copilot_creds(monkeypatch):
+ """Make sure copilot-acp visibility comes ONLY from executable resolution:
+ no env tokens, no configured ACP endpoint, no auth-store entry, no seeded
+ credential pool."""
+ # COPILOT_ACP_BASE_URL is not a credential, but an `acp+tcp://` value marks
+ # the provider configured with no executable at all (hermes_cli/auth.py), so
+ # a host that sets it would decide the outcome instead of the test.
+ for var in ("GH_TOKEN", "GITHUB_TOKEN", "HERMES_COPILOT_ACP_COMMAND",
+ "COPILOT_CLI_PATH", "COPILOT_ACP_BASE_URL"):
+ monkeypatch.delenv(var, raising=False)
+ import hermes_cli.auth as auth
+ import hermes_cli.model_switch as model_switch
+
+ monkeypatch.setattr(auth, "_load_auth_store", lambda: {})
+ monkeypatch.setattr(model_switch, "_credential_pool_is_usable", lambda *a, **k: False)
+
+
+def test_copilot_acp_listed_when_executable_resolves(tmp_path, monkeypatch, _no_other_copilot_creds):
+ fake = tmp_path / ("copilot.exe" if os.name == "nt" else "copilot")
+ fake.write_text("", encoding="utf-8")
+ fake.chmod(0o755)
+ monkeypatch.setenv("HERMES_COPILOT_ACP_COMMAND", str(fake))
+
+ with patch("agent.models_dev.fetch_models_dev", return_value={}), \
+ patch("hermes_cli.models._resolve_copilot_catalog_api_key", return_value=None), \
+ patch("hermes_cli.models._fetch_github_models", return_value=[]):
+ providers = list_authenticated_providers(current_provider="openrouter", max_models=50)
+
+ acp = next((p for p in providers if p["slug"] == "copilot-acp"), None)
+
+ assert acp is not None, "copilot-acp must be listed when its executable resolves"
+ assert acp["models"], "copilot-acp row must offer at least the curated fallback models"
+
+
+def test_copilot_acp_hidden_when_executable_missing(monkeypatch, _no_other_copilot_creds):
+ # `copilot` may genuinely be installed on a dev machine — force the
+ # resolution miss so the test pins behaviour, not the host's PATH.
+ import hermes_cli.auth as auth
+
+ monkeypatch.setattr(auth.shutil, "which", lambda *_a, **_k: None)
+
+ with patch("agent.models_dev.fetch_models_dev", return_value={}), \
+ patch("hermes_cli.models._resolve_copilot_catalog_api_key", return_value=None), \
+ patch("hermes_cli.models._fetch_github_models", return_value=[]):
+ providers = list_authenticated_providers(current_provider="openrouter", max_models=50)
+
+ assert all(p["slug"] != "copilot-acp" for p in providers), \
+ "copilot-acp must stay hidden when no executable resolves"
diff --git a/tests/hermes_cli/test_cron.py b/tests/hermes_cli/test_cron.py
index f48ce6e0e6..1712e0ddab 100644
--- a/tests/hermes_cli/test_cron.py
+++ b/tests/hermes_cli/test_cron.py
@@ -129,6 +129,44 @@ class TestCronCommandLifecycle:
assert jobs[0]["name"] == "Skill combo"
+class TestUnverifiedDeliveryVisibility:
+ """An evidence-free live-adapter ack (Slack/Matrix/Mattermost bare
+ ``SendResult(success=True)``) is accepted as delivered, but the UNVERIFIED
+ state must be visible in ``hermes cron list`` and ``hermes cron doctor``,
+ not only in a WARNING log line."""
+
+ def _seed(self):
+ job = create_job(prompt="Nightly brief", schedule="every 1h", deliver="slack:C0123456")
+ jobs = load_jobs()
+ jobs[0]["last_status"] = "ok"
+ jobs[0]["last_delivery_unverified"] = ["slack:C0123456"]
+ save_jobs(jobs)
+ return job
+
+ def test_list_shows_unverified_delivery(self, tmp_cron_dir, capsys):
+ job = self._seed()
+ cron_command(Namespace(cron_command="list", all=True, json=False))
+ out = capsys.readouterr().out
+ assert job["id"] in out
+ assert "Delivery UNVERIFIED" in out
+ assert "slack:C0123456" in out
+ assert "without message_id/raw_response" in out
+
+ def test_list_is_quiet_when_delivery_was_verified(self, tmp_cron_dir, capsys):
+ create_job(prompt="Nightly brief", schedule="every 1h", deliver="slack:C0123456")
+ cron_command(Namespace(cron_command="list", all=True, json=False))
+ assert "UNVERIFIED" not in capsys.readouterr().out
+
+ def test_doctor_reports_unverified_delivery(self, tmp_cron_dir, capsys):
+ job = self._seed()
+ rc = cron_command(Namespace(cron_command="doctor"))
+ out = capsys.readouterr().out
+ assert rc == 1
+ assert job["id"] in out
+ assert "last delivery unverified" in out
+ assert "slack:C0123456" in out
+
+
class TestCronDoctor:
def test_doctor_reports_cron_health_issues(self, tmp_cron_dir, capsys):
job = create_job(prompt="Daily digest", schedule="every 1h", script="missing.py")
@@ -160,6 +198,28 @@ class TestCronDoctor:
assert rc == 0
assert "✓ Cron doctor found no issues" in out
+ def test_doctor_reports_delivery_failure_once(self, tmp_cron_dir, capsys):
+ """A delivery_failed run is a delivery issue, not a failed agent run.
+
+ The agent succeeded (last_error is None), so the generic last-run-failed
+ line would only ever say "unknown error" — double-reporting the same
+ incident (#83993).
+ """
+ create_job(prompt="Daily digest", schedule="every 1h")
+ jobs = load_jobs()
+ jobs[0]["last_status"] = "delivery_failed"
+ jobs[0]["last_error"] = None
+ jobs[0]["last_delivery_error"] = "telegram timeout"
+ save_jobs(jobs)
+
+ rc = cron_command(Namespace(cron_command="doctor"))
+
+ out = capsys.readouterr().out
+ assert rc == 1
+ assert "last delivery failed: telegram timeout" in out
+ assert "last run failed" not in out
+ assert "unknown error" not in out
+
def test_doctor_flags_overdue_next_run(self, tmp_cron_dir, capsys):
from datetime import datetime, timedelta, timezone
@@ -193,6 +253,48 @@ class TestCronDoctor:
assert "✓ Cron doctor found no issues" in out
+class TestCronListStatusRendering:
+ """`cron list` must never paint an undelivered run as a success (#83993)."""
+
+ def test_delivery_failed_is_not_green_ok(self, tmp_cron_dir, capsys, monkeypatch):
+ monkeypatch.setattr("hermes_cli.gateway.find_gateway_pids", lambda: [1])
+ # capsys is not a tty, so force colors on to check the paint itself.
+ monkeypatch.setattr("hermes_cli.colors.should_use_color", lambda: True)
+ create_job(prompt="Daily digest", schedule="every 1h")
+ jobs = load_jobs()
+ jobs[0]["last_run_at"] = "2026-09-01T09:00:00+00:00"
+ jobs[0]["last_status"] = "delivery_failed"
+ jobs[0]["last_error"] = None
+ jobs[0]["last_delivery_error"] = "telegram timeout"
+ save_jobs(jobs)
+
+ cron_command(Namespace(cron_command="list", all=True))
+
+ out = capsys.readouterr().out
+ last_run_line = next(l for l in out.splitlines() if "Last run:" in l)
+ assert "delivery_failed" in last_run_line
+ assert "telegram timeout" in last_run_line, (
+ "the delivery detail lives in last_delivery_error, not last_error"
+ )
+ assert cron_cli.Colors.GREEN not in last_run_line
+
+ def test_ok_run_still_green(self, tmp_cron_dir, capsys, monkeypatch):
+ monkeypatch.setattr("hermes_cli.gateway.find_gateway_pids", lambda: [1])
+ monkeypatch.setattr("hermes_cli.colors.should_use_color", lambda: True)
+ create_job(prompt="Daily digest", schedule="every 1h")
+ jobs = load_jobs()
+ jobs[0]["last_run_at"] = "2026-09-01T09:00:00+00:00"
+ jobs[0]["last_status"] = "ok"
+ save_jobs(jobs)
+
+ cron_command(Namespace(cron_command="list", all=True))
+
+ out = capsys.readouterr().out
+ last_run_line = next(l for l in out.splitlines() if "Last run:" in l)
+ assert f"{cron_cli.Colors.GREEN}ok" in last_run_line
+ assert "delivery_failed" not in last_run_line
+
+
class TestGatewayNotRunningWarning:
"""`cron create` / `cron list` must warn when the gateway (and thus the
cron ticker) isn't running, since jobs only fire inside the gateway.
@@ -446,3 +548,41 @@ class TestCronRunBackgroundDispatch:
assert rc == 0
assert "Running in background (delegation del-xyz)." in out
assert "failed" not in out.lower()
+
+
+class TestSlashCronListLastStatus:
+ """The in-chat ``/cron list`` (cli_commands_mixin) renders every
+ ``last_status`` literal explicitly — ``delivery_failed`` names the delivery
+ reason (last_error is None for those runs) instead of printing the bare
+ literal next to a run that looks otherwise fine."""
+
+ def _run_list(self, tmp_cron_dir, capsys):
+ from hermes_cli.cli_commands_mixin import CLICommandsMixin
+
+ class _Host(CLICommandsMixin):
+ pass
+
+ _Host()._handle_cron_command("/cron list --all")
+ return capsys.readouterr().out
+
+ def test_delivery_failed_names_the_delivery_error(self, tmp_cron_dir, capsys):
+ create_job(prompt="Nightly brief", schedule="every 1h", deliver="telegram:1")
+ jobs = load_jobs()
+ jobs[0]["last_run_at"] = "2026-09-01T07:00:00+00:00"
+ jobs[0]["last_status"] = "delivery_failed"
+ jobs[0]["last_error"] = None
+ jobs[0]["last_delivery_error"] = "telegram: 502 Bad Gateway"
+ save_jobs(jobs)
+
+ out = self._run_list(tmp_cron_dir, capsys)
+ assert "Last run: 2026-09-01T07:00:00+00:00 (delivery_failed: telegram: 502 Bad Gateway)" in out
+
+ def test_ok_stays_plain(self, tmp_cron_dir, capsys):
+ create_job(prompt="Nightly brief", schedule="every 1h")
+ jobs = load_jobs()
+ jobs[0]["last_run_at"] = "2026-09-01T07:00:00+00:00"
+ jobs[0]["last_status"] = "ok"
+ save_jobs(jobs)
+
+ out = self._run_list(tmp_cron_dir, capsys)
+ assert "(ok)" in out
diff --git a/tests/hermes_cli/test_cron_fire_dashboard.py b/tests/hermes_cli/test_cron_fire_dashboard.py
index aa898bf78e..d6d406398d 100644
--- a/tests/hermes_cli/test_cron_fire_dashboard.py
+++ b/tests/hermes_cli/test_cron_fire_dashboard.py
@@ -250,6 +250,41 @@ def test_fire_endpoint_multiplex_profile_prefix(tmp_path, monkeypatch):
assert url == "http://127.0.0.1:8642/p/worker_alpha/api/cron/fire"
+def test_fire_endpoint_multiplex_reads_port_from_default_listener(tmp_path, monkeypatch):
+ """Multiplex mode: only the DEFAULT profile's api_server is bound, so a
+ secondary's fire URL must use the default home's port — not the
+ secondary's own config.yaml/.env port, which nothing listens on
+ (PR #84755). Real config files, real load_config()."""
+ default_home = tmp_path / "root"
+ worker_home = default_home / "profiles" / "worker_alpha"
+ default_home.mkdir()
+ worker_home.mkdir(parents=True)
+ (default_home / "config.yaml").write_text(
+ "gateway:\n multiplex_profiles: true\n"
+ "platforms:\n api_server:\n extra:\n port: 8650\n",
+ encoding="utf-8",
+ )
+ (worker_home / "config.yaml").write_text(
+ "platforms:\n api_server:\n enabled: false\n extra:\n port: 8702\n",
+ encoding="utf-8",
+ )
+ (worker_home / ".env").write_text("API_SERVER_PORT=8701\n", encoding="utf-8")
+ monkeypatch.setenv("HERMES_HOME", str(default_home))
+ monkeypatch.delenv("API_SERVER_PORT", raising=False)
+ monkeypatch.delenv("GATEWAY_MULTIPLEX_PROFILES", raising=False)
+ monkeypatch.setattr(web_server, "_cron_default_profile", lambda: "default")
+
+ url = web_server._gateway_fire_endpoint("worker_alpha", worker_home)
+
+ assert url == "http://127.0.0.1:8650/p/worker_alpha/api/cron/fire"
+ # The GATEWAY_MULTIPLEX_PROFILES env override is still honored (parity
+ # with gateway/config.py): forcing it off restores per-profile routing.
+ monkeypatch.setenv("GATEWAY_MULTIPLEX_PROFILES", "0")
+ assert web_server._gateway_fire_endpoint("worker_alpha", worker_home) == (
+ "http://127.0.0.1:8702/api/cron/fire"
+ )
+
+
# ── OOF-266: intentional-stop drop + Retry-After on transient 503 ─────────
diff --git a/tests/hermes_cli/test_cross_profile_kill_refusal.py b/tests/hermes_cli/test_cross_profile_kill_refusal.py
new file mode 100644
index 0000000000..7b05e98d65
--- /dev/null
+++ b/tests/hermes_cli/test_cross_profile_kill_refusal.py
@@ -0,0 +1,217 @@
+"""Cross-profile kill refusal regression tests (#89315).
+
+A poisoned/contaminated ``gateway.pid`` inside one profile's HERMES_HOME can
+truthfully name ANOTHER profile's live gateway (its ``hermes_home`` stamp
+records the real owner). ``gateway stop`` / the restart force-kill escalation
+/ ``profile delete`` must refuse to signal such a PID instead of starting the
+mutual cross-profile SIGTERM restart loop from the issue report.
+
+These tests exercise the REAL code paths against real PID files, a real
+flock-held gateway lock, and a real dummy child process — no mocks of the
+code under test.
+"""
+
+import json
+import os
+import subprocess
+import sys
+import time
+from pathlib import Path
+
+import pytest
+
+from gateway.status import recorded_gateway_home_conflicts
+
+
+def _spawn_gateway_lookalike(bin_dir: Path, lock_path: Path) -> subprocess.Popen:
+ """Real child process whose argv matches the gateway runtime matcher."""
+ bin_dir.mkdir(parents=True, exist_ok=True)
+ lock_path.parent.mkdir(parents=True, exist_ok=True)
+ script = bin_dir / "hermes"
+ if sys.platform == "win32":
+ body = "import time\ntime.sleep(120)\n"
+ else:
+ body = (
+ "import fcntl, time\n"
+ f"fh = open({str(lock_path)!r}, 'a+')\n"
+ "fcntl.flock(fh, fcntl.LOCK_EX | fcntl.LOCK_NB)\n"
+ "time.sleep(120)\n"
+ )
+ script.write_text(f"#!{sys.executable}\n{body}", encoding="utf-8")
+ if sys.platform != "win32":
+ script.chmod(0o755)
+ cmd = [str(script), "gateway", "run"]
+ else:
+ cmd = [sys.executable, str(script), "gateway", "run"]
+ proc = subprocess.Popen(
+ cmd, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL
+ )
+ deadline = time.monotonic() + 10.0
+ while time.monotonic() < deadline and not lock_path.exists():
+ if proc.poll() is not None:
+ raise RuntimeError("gateway lookalike died at startup")
+ time.sleep(0.05)
+ return proc
+
+
+def _pid_record(proc: subprocess.Popen, script: Path, owner_home: Path) -> dict:
+ from gateway.status import get_process_start_time
+
+ return {
+ "pid": proc.pid,
+ "kind": "hermes-gateway",
+ "argv": [str(script), "gateway", "run"],
+ "start_time": get_process_start_time(proc.pid),
+ "hermes_home": str(owner_home),
+ }
+
+
+class TestRecordedGatewayHomeConflicts:
+ def test_conflicting_home_detected(self, tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / "profiles" / "tim"))
+ record = {"pid": 1, "hermes_home": str(tmp_path)}
+ assert recorded_gateway_home_conflicts(record) is True
+
+ def test_same_home_accepted(self, tmp_path, monkeypatch):
+ home = tmp_path / "profiles" / "tim"
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ record = {"pid": 1, "hermes_home": str(home)}
+ assert recorded_gateway_home_conflicts(record) is False
+
+ def test_legacy_record_without_home_proves_nothing(self, tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ assert recorded_gateway_home_conflicts({"pid": 1}) is False
+ assert recorded_gateway_home_conflicts(None) is False
+ assert recorded_gateway_home_conflicts({"pid": 1, "hermes_home": " "}) is False
+
+ def test_expected_home_override(self, tmp_path):
+ target = tmp_path / "profiles" / "tim"
+ record = {"pid": 1, "hermes_home": str(tmp_path)}
+ assert (
+ recorded_gateway_home_conflicts(record, expected_home=target) is True
+ )
+ assert (
+ recorded_gateway_home_conflicts(record, expected_home=tmp_path) is False
+ )
+
+
+@pytest.mark.skipif(sys.platform == "win32", reason="POSIX flock harness")
+class TestCrossProfileStopRefusal:
+ def test_stop_profile_gateway_refuses_other_profiles_pid(
+ self, tmp_path, monkeypatch
+ ):
+ """Profile B's ``gateway stop`` must not SIGTERM profile A's gateway.
+
+ On main this path is already safe upstream of any guard:
+ ``get_running_pid()`` filters a pid record owned by another profile
+ (and unlinks the poisoned pid file) before ``stop_profile_gateway``
+ ever sees a pid — so the contract here is "returns False, other
+ profile's process untouched, poisoned pid file gone", not a printed
+ refusal.
+ """
+ root_home = tmp_path / "root-home"
+ tim_home = tmp_path / "root-home" / "profiles" / "tim"
+ tim_home.mkdir(parents=True)
+ monkeypatch.setenv("HERMES_HOME", str(tim_home))
+
+ proc = _spawn_gateway_lookalike(
+ tmp_path / "bin", tim_home / "gateway.lock"
+ )
+ try:
+ record = _pid_record(proc, tmp_path / "bin" / "hermes", root_home)
+ (tim_home / "gateway.pid").write_text(json.dumps(record))
+
+ from hermes_cli import gateway as gateway_cli
+
+ assert gateway_cli.stop_profile_gateway() is False
+ assert not (tim_home / "gateway.pid").exists(), (
+ "poisoned cross-profile pid file should have been unlinked"
+ )
+ time.sleep(0.5)
+ assert proc.poll() is None, (
+ "cross-profile SIGTERM fired: profile A's gateway was killed"
+ )
+ finally:
+ proc.kill()
+ proc.wait(timeout=10)
+
+ def test_stop_profile_gateway_still_stops_own_gateway(
+ self, tmp_path, monkeypatch
+ ):
+ """Same-home records keep stopping normally (no false refusal)."""
+ tim_home = tmp_path / "profiles" / "tim"
+ tim_home.mkdir(parents=True)
+ monkeypatch.setenv("HERMES_HOME", str(tim_home))
+
+ proc = _spawn_gateway_lookalike(
+ tmp_path / "bin", tim_home / "gateway.lock"
+ )
+ try:
+ record = _pid_record(proc, tmp_path / "bin" / "hermes", tim_home)
+ (tim_home / "gateway.pid").write_text(json.dumps(record))
+
+ from hermes_cli import gateway as gateway_cli
+
+ assert gateway_cli.stop_profile_gateway() is True
+ deadline = time.monotonic() + 15.0
+ while time.monotonic() < deadline and proc.poll() is None:
+ time.sleep(0.1)
+ assert proc.poll() is not None, "own gateway was not stopped"
+ finally:
+ if proc.poll() is None:
+ proc.kill()
+ proc.wait(timeout=10)
+
+
+@pytest.mark.skipif(sys.platform == "win32", reason="POSIX flock harness")
+class TestProfileDeleteStopRefusal:
+ def test_stop_gateway_process_refuses_other_profiles_pid(
+ self, tmp_path, capsys
+ ):
+ """``profile delete`` must not kill a gateway owned by another home."""
+ root_home = tmp_path / "root-home"
+ tim_home = root_home / "profiles" / "tim"
+ tim_home.mkdir(parents=True)
+
+ proc = _spawn_gateway_lookalike(
+ tmp_path / "bin", tim_home / "gateway.lock"
+ )
+ try:
+ record = _pid_record(proc, tmp_path / "bin" / "hermes", root_home)
+ (tim_home / "gateway.pid").write_text(json.dumps(record))
+
+ from hermes_cli.profiles import _stop_gateway_process
+
+ _stop_gateway_process(tim_home)
+ out = capsys.readouterr().out
+ assert "Refusing to stop" in out
+ time.sleep(0.5)
+ assert proc.poll() is None, (
+ "profile delete killed another profile's gateway"
+ )
+ finally:
+ proc.kill()
+ proc.wait(timeout=10)
+
+ def test_stop_gateway_process_still_stops_own_gateway(self, tmp_path):
+ tim_home = tmp_path / "profiles" / "tim"
+ tim_home.mkdir(parents=True)
+
+ proc = _spawn_gateway_lookalike(
+ tmp_path / "bin", tim_home / "gateway.lock"
+ )
+ try:
+ record = _pid_record(proc, tmp_path / "bin" / "hermes", tim_home)
+ (tim_home / "gateway.pid").write_text(json.dumps(record))
+
+ from hermes_cli.profiles import _stop_gateway_process
+
+ _stop_gateway_process(tim_home)
+ deadline = time.monotonic() + 15.0
+ while time.monotonic() < deadline and proc.poll() is None:
+ time.sleep(0.1)
+ assert proc.poll() is not None, "own gateway was not stopped"
+ finally:
+ if proc.poll() is None:
+ proc.kill()
+ proc.wait(timeout=10)
diff --git a/tests/hermes_cli/test_desktop_exe_integrity.py b/tests/hermes_cli/test_desktop_exe_integrity.py
index 6e9d3dd06e..bdff61e62a 100644
--- a/tests/hermes_cli/test_desktop_exe_integrity.py
+++ b/tests/hermes_cli/test_desktop_exe_integrity.py
@@ -277,8 +277,12 @@ def _ns(**kw):
@pytest.mark.windows_only
def test_build_only_fails_when_pack_produces_corrupt_exe(tmp_path, monkeypatch, capsys):
"""The updater chain's contract: a rebuild whose Hermes.exe cannot launch
- must exit nonzero (so hermes-setup's retry-once kicks in) and must restore
- the previous working build instead of leaving the corrupt one.
+ must exit nonzero (so hermes-setup's retry-once kicks in) and must leave
+ the previous working build in place instead of installing the corrupt one.
+
+ Stage-and-swap (#86443): the pack lands in a staging dir; the integrity
+ gate runs on the STAGED exe and a failure discards staging without ever
+ touching the live ``win-unpacked`` tree.
``windows_only``: the whole chain is Windows-gated — ``win-unpacked``
candidate discovery in ``_desktop_packaged_executable`` and the integrity
@@ -290,12 +294,20 @@ def test_build_only_fails_when_pack_produces_corrupt_exe(tmp_path, monkeypatch,
(desktop_dir / "package.json").write_text("{}", encoding="utf-8")
monkeypatch.setattr(cli_main, "PROJECT_ROOT", root)
- exe = desktop_dir / "release" / "win-unpacked" / "Hermes.exe"
- make_pe(exe, PE_AMD64, truncate_to=0x300) # what the failed pack produced
- make_pe(desktop_dir / "release" / "win-unpacked.bak" / "Hermes.exe", PE_AMD64)
+ live_exe = desktop_dir / "release" / "win-unpacked" / "Hermes.exe"
+ make_pe(live_exe, PE_AMD64) # the previous, working app
+ live_bytes = live_exe.read_bytes()
install_ok = subprocess.CompletedProcess(["npm", "ci"], 0)
- pack_ok = subprocess.CompletedProcess(["npm", "run", "pack"], 0)
+
+ def pack_into_staging(cmd, *args, **kwargs):
+ # electron-builder honours -c.directories.output=; emulate a
+ # pack that "succeeds" but writes a truncated exe there.
+ out_flag = next((a for a in cmd if str(a).startswith("-c.directories.output=")), None)
+ assert out_flag is not None, "pack must be redirected into a staging dir"
+ staging = Path(str(out_flag).split("=", 1)[1])
+ make_pe(staging / "win-unpacked" / "Hermes.exe", PE_AMD64, truncate_to=0x300)
+ return subprocess.CompletedProcess(list(cmd), 0)
with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \
patch("hermes_cli.main._resolve_node_runtime_npm", return_value="npm.cmd"), \
@@ -306,16 +318,17 @@ def test_build_only_fails_when_pack_produces_corrupt_exe(tmp_path, monkeypatch,
patch("hermes_cli.main._desktop_stamp_path", return_value=tmp_path / "stamp.json"), \
patch("hermes_cli.main._write_desktop_build_stamp") as mock_stamp, \
patch("hermes_cli.main._windows_native_machine", return_value="AMD64"), \
- patch("hermes_cli.main.subprocess.run", return_value=pack_ok), \
+ patch("hermes_cli.main.subprocess.run", side_effect=pack_into_staging), \
pytest.raises(SystemExit) as exc:
cli_main.cmd_gui(_ns())
assert exc.value.code == 1
- # The previous working exe was restored...
- assert cli_main._parse_pe_machine(exe) == PE_AMD64
+ # The previous working exe was never touched...
+ assert live_exe.read_bytes() == live_bytes
+ assert cli_main._parse_pe_machine(live_exe) == PE_AMD64
+ # ...the staged corrupt tree was discarded...
+ assert not list((desktop_dir / "release").glob(".staging-*"))
# ...and the poisoned build was never stamped as good.
mock_stamp.assert_not_called()
out = capsys.readouterr().out
assert "integrity check" in out
-
-
diff --git a/tests/hermes_cli/test_desktop_local_flag.py b/tests/hermes_cli/test_desktop_local_flag.py
new file mode 100644
index 0000000000..c581705ebd
--- /dev/null
+++ b/tests/hermes_cli/test_desktop_local_flag.py
@@ -0,0 +1,39 @@
+"""The desktop subcommand's --local launch flag.
+
+Local models ship on main behind this flag: `hermes desktop --local` (or
+`Hermes.exe --local` directly) shows the local-models GUI surfaces; without
+it the desktop hides them all, even when local models are configured. These
+tests pin the argparse contract; the pass-through to the Electron argv lives
+in cmd_gui's launch paths.
+"""
+
+import argparse
+
+from hermes_cli.subcommands.gui import build_gui_parser
+
+
+def _parser() -> argparse.ArgumentParser:
+ parser = argparse.ArgumentParser(prog="hermes")
+ subparsers = parser.add_subparsers(dest="command")
+ build_gui_parser(subparsers, cmd_gui=lambda args: None)
+
+ return parser
+
+
+def test_local_flag_parses():
+ args = _parser().parse_args(["desktop", "--local"])
+
+ assert args.local is True
+
+
+def test_local_flag_defaults_off():
+ args = _parser().parse_args(["desktop"])
+
+ assert args.local is False
+
+
+def test_local_flag_composes_with_build_flags():
+ args = _parser().parse_args(["desktop", "--local", "--force-build"])
+
+ assert args.local is True
+ assert args.force_build is True
diff --git a/tests/hermes_cli/test_doctor.py b/tests/hermes_cli/test_doctor.py
index 9e37dc2cc2..267388f369 100644
--- a/tests/hermes_cli/test_doctor.py
+++ b/tests/hermes_cli/test_doctor.py
@@ -285,18 +285,34 @@ def test_doctor_reports_vercel_backend_diagnostics(monkeypatch, tmp_path):
class TestDoctorMemoryProviderSection:
"""The ◆ Memory Provider section should respect memory.provider config."""
- def _make_hermes_home(self, tmp_path, provider=""):
+ def _make_hermes_home(self, tmp_path, provider="", memory_config=None):
"""Create a minimal HERMES_HOME with config.yaml."""
home = tmp_path / ".hermes"
home.mkdir(parents=True, exist_ok=True)
import yaml
- config = {"memory": {"provider": provider}} if provider else {"memory": {}}
+ config = dict(memory_config or {})
+ if provider:
+ config["provider"] = provider
+ config = {"memory": config}
(home / "config.yaml").write_text(yaml.dump(config))
return home
- def _run_doctor_and_capture(self, monkeypatch, tmp_path, provider=""):
+ def _run_doctor_and_capture(
+ self,
+ monkeypatch,
+ tmp_path,
+ provider="",
+ *,
+ memory_config=None,
+ stale_builtin_files=False,
+ ):
"""Run doctor and capture stdout."""
- home = self._make_hermes_home(tmp_path, provider)
+ home = self._make_hermes_home(tmp_path, provider, memory_config)
+ if stale_builtin_files:
+ memories = home / "memories"
+ memories.mkdir()
+ (memories / "MEMORY.md").write_text("stale memory", encoding="utf-8")
+ (memories / "USER.md").write_text("stale user", encoding="utf-8")
monkeypatch.setattr(doctor_mod, "HERMES_HOME", home)
monkeypatch.setattr(doctor_mod, "PROJECT_ROOT", tmp_path / "project")
monkeypatch.setattr(doctor_mod, "_DHH", str(home))
@@ -340,6 +356,26 @@ class TestDoctorMemoryProviderSection:
assert "Memory Provider" in out
assert "Built-in memory active" not in out
+ @pytest.mark.parametrize("memory_enabled", [False, True])
+ def test_stale_builtin_files_reported_only_when_store_enabled(
+ self, monkeypatch, tmp_path, memory_enabled
+ ):
+ # #100668: disabled built-in stores must not surface stale files as active.
+ out = self._run_doctor_and_capture(
+ monkeypatch,
+ tmp_path,
+ provider="mnemosyne",
+ memory_config={
+ "memory_enabled": memory_enabled,
+ "user_profile_enabled": False,
+ },
+ stale_builtin_files=True,
+ )
+
+ assert ("MEMORY.md exists" in out) is memory_enabled
+ assert "USER.md exists" not in out
+ assert ("Built-in memory files disabled by config" in out) is not memory_enabled
+
def test_run_doctor_termux_treats_docker_and_browser_warnings_as_expected(monkeypatch, tmp_path):
helper = TestDoctorMemoryProviderSection()
diff --git a/tests/hermes_cli/test_dump_env_visibility.py b/tests/hermes_cli/test_dump_env_visibility.py
index 40feba0cec..ba98cfa3a5 100644
--- a/tests/hermes_cli/test_dump_env_visibility.py
+++ b/tests/hermes_cli/test_dump_env_visibility.py
@@ -47,6 +47,7 @@ def test_dump_leaves_unset_key_untouched(monkeypatch, capsys, tmp_path):
monkeypatch.setattr(dump, "get_project_root", lambda: tmp_path / "noproject")
monkeypatch.delenv("KEENABLE_API_KEY", raising=False)
+ monkeypatch.delenv("TAVILY_API_KEY", raising=False)
home = get_hermes_home()
home.mkdir(parents=True, exist_ok=True)
diff --git a/tests/hermes_cli/test_external_process_auth_status.py b/tests/hermes_cli/test_external_process_auth_status.py
new file mode 100644
index 0000000000..221459b10d
--- /dev/null
+++ b/tests/hermes_cli/test_external_process_auth_status.py
@@ -0,0 +1,292 @@
+"""Tests for external-process provider auth status and Accounts-tab wiring.
+
+Covers the copilot-acp fix class:
+ * ``get_auth_status()`` dispatches on ``auth_type == "external_process"``
+ (not a hardcoded slug), so future ACP-style providers inherit the
+ behaviour automatically.
+ * ``auth_verified``/``auth_source`` carry positive credential evidence
+ (env token or on-disk GitHub Copilot credential store) while remaining
+ honest — no evidence means unknown, never "signed out".
+ * The Accounts-tab sign-in ``cli_command`` reflects the executable the
+ user actually configured (``HERMES_COPILOT_ACP_COMMAND`` /
+ ``COPILOT_CLI_PATH``), and its default is a valid Copilot CLI
+ invocation (``copilot login`` — ``copilot /login`` is not a command).
+"""
+
+import os
+
+import pytest
+
+from hermes_cli.auth import (
+ get_auth_status,
+ get_external_process_provider_status,
+)
+
+
+@pytest.fixture()
+def _clean_copilot_env(monkeypatch):
+ """Neutralize host state so tests pin behaviour, not this machine."""
+ for var in (
+ "COPILOT_GITHUB_TOKEN", "GH_TOKEN", "GITHUB_TOKEN",
+ "HERMES_COPILOT_ACP_COMMAND", "COPILOT_CLI_PATH",
+ "HERMES_COPILOT_ACP_ARGS", "COPILOT_ACP_BASE_URL",
+ ):
+ monkeypatch.delenv(var, raising=False)
+
+
+# --- get_auth_status dispatches on auth_type, not slug ----------------------
+
+
+def test_get_auth_status_dispatches_external_process_by_auth_type(
+ tmp_path, monkeypatch, _clean_copilot_env
+):
+ fake = tmp_path / ("copilot.exe" if os.name == "nt" else "copilot")
+ fake.write_text("", encoding="utf-8")
+ fake.chmod(0o755)
+ monkeypatch.setenv("HERMES_COPILOT_ACP_COMMAND", str(fake))
+ # Point HOME somewhere empty so on-disk credential stores don't leak in.
+ monkeypatch.setenv("HOME", str(tmp_path))
+
+ status = get_auth_status("copilot-acp")
+
+ # The external_process status shape, not the {"logged_in": False}
+ # fallthrough — proves the dispatcher reached the right branch.
+ assert status.get("provider") == "copilot-acp"
+ assert status.get("configured") is True
+ assert status.get("resolved_command") == str(fake)
+ assert "auth_verified" in status
+
+
+def test_external_process_status_rejects_wrong_auth_type():
+ # A provider that exists but is not external_process must be refused —
+ # the generic dispatcher relies on this guard.
+ assert get_external_process_provider_status("openrouter") == {"configured": False}
+ assert get_external_process_provider_status("no-such-provider") == {"configured": False}
+
+
+# --- auth_verified: positive evidence only ----------------------------------
+
+
+def test_auth_verified_false_without_evidence(tmp_path, monkeypatch, _clean_copilot_env):
+ monkeypatch.setenv("HOME", str(tmp_path)) # no ~/.config/github-copilot
+ status = get_external_process_provider_status("copilot-acp")
+ assert status["auth_verified"] is False
+ assert status["auth_source"] is None
+
+
+def test_auth_verified_from_supported_env_token(tmp_path, monkeypatch, _clean_copilot_env):
+ monkeypatch.setenv("HOME", str(tmp_path))
+ monkeypatch.setenv("GH_TOKEN", "gho_" + "x" * 36) # supported OAuth prefix
+
+ status = get_external_process_provider_status("copilot-acp")
+
+ assert status["auth_verified"] is True
+ assert status["auth_source"] == "env: GH_TOKEN"
+
+
+def test_classic_pat_is_not_login_evidence(tmp_path, monkeypatch, _clean_copilot_env):
+ # ghp_* classic PATs are rejected by the Copilot API — presence of one
+ # must not be presented as a working login.
+ monkeypatch.setenv("HOME", str(tmp_path))
+ monkeypatch.setenv("GH_TOKEN", "ghp_" + "x" * 36)
+
+ status = get_external_process_provider_status("copilot-acp")
+
+ assert status["auth_verified"] is False
+
+
+def test_auth_verified_from_on_disk_credential_store(tmp_path, monkeypatch, _clean_copilot_env):
+ monkeypatch.setenv("HOME", str(tmp_path))
+ store = tmp_path / ".config" / "github-copilot"
+ store.mkdir(parents=True)
+ (store / "hosts.json").write_text(
+ '{"github.com": {"oauth_token": "gho_test"}}', encoding="utf-8"
+ )
+
+ status = get_external_process_provider_status("copilot-acp")
+
+ assert status["auth_verified"] is True
+ assert status["auth_source"] == "~/.config/github-copilot/hosts.json"
+
+
+def test_empty_credential_store_is_not_evidence(tmp_path, monkeypatch, _clean_copilot_env):
+ monkeypatch.setenv("HOME", str(tmp_path))
+ store = tmp_path / ".config" / "github-copilot"
+ store.mkdir(parents=True)
+ (store / "hosts.json").write_text("{}", encoding="utf-8") # logged out
+
+ status = get_external_process_provider_status("copilot-acp")
+
+ assert status["auth_verified"] is False
+
+
+def test_auth_verified_from_copilot_cli_plaintext_store(tmp_path, monkeypatch, _clean_copilot_env):
+ # `copilot login` without an OS keychain writes the token into
+ # ~/.copilot/config.json (JSONC, with //-comment header lines).
+ monkeypatch.setenv("HOME", str(tmp_path))
+ cfg_dir = tmp_path / ".copilot"
+ cfg_dir.mkdir()
+ (cfg_dir / "config.json").write_text(
+ "// User settings belong in settings.json.\n"
+ "// This file is managed automatically.\n"
+ "{\n"
+ ' "copilotTokens": {"https://github.com:someuser": "gho_test"},\n'
+ ' "lastLoggedInUser": {"host": "https://github.com", "login": "someuser"}\n'
+ "}\n",
+ encoding="utf-8",
+ )
+
+ status = get_external_process_provider_status("copilot-acp")
+
+ assert status["auth_verified"] is True
+ assert status["auth_source"] == "~/.copilot/config.json"
+
+
+def test_copilot_cli_store_without_tokens_is_not_evidence(tmp_path, monkeypatch, _clean_copilot_env):
+ # A config.json exists after first launch even before any login —
+ # its presence alone must not read as signed-in.
+ monkeypatch.setenv("HOME", str(tmp_path))
+ cfg_dir = tmp_path / ".copilot"
+ cfg_dir.mkdir()
+ (cfg_dir / "config.json").write_text(
+ '// managed\n{"firstLaunchAt": "2026-01-01T00:00:00Z", "copilotTokens": {}}\n',
+ encoding="utf-8",
+ )
+
+ status = get_external_process_provider_status("copilot-acp")
+
+ assert status["auth_verified"] is False
+
+
+# --- desktop picker explicit-only filter ------------------------------------
+
+
+def test_explicit_filter_keeps_signed_in_external_process_row(tmp_path, monkeypatch, _clean_copilot_env):
+ # A verified CLI login leaves no trace in active_provider/config/env —
+ # the explicit-only desktop filter must treat it like the Anthropic OAuth
+ # carve-out and keep the row.
+ from hermes_cli.inventory import _filter_explicit_provider_rows
+
+ monkeypatch.setenv("HOME", str(tmp_path))
+ cfg_dir = tmp_path / ".copilot"
+ cfg_dir.mkdir()
+ (cfg_dir / "config.json").write_text(
+ '{"copilotTokens": {"https://github.com:u": "gho_test"}}', encoding="utf-8"
+ )
+
+ class _Ctx:
+ current_provider = "nous"
+
+ rows = [{"slug": "copilot-acp", "models": ["gpt-5.4"]}]
+ kept = _filter_explicit_provider_rows(rows, _Ctx())
+
+ assert any(r["slug"] == "copilot-acp" for r in kept), \
+ "signed-in copilot-acp must survive the explicit-only picker filter"
+
+
+def test_explicit_filter_drops_unverified_external_process_row(tmp_path, monkeypatch, _clean_copilot_env):
+ # Merely having the executable on PATH is ambient discovery, not an
+ # explicit configuration — the desktop filter keeps its narrower contract.
+ from hermes_cli.inventory import _filter_explicit_provider_rows
+
+ monkeypatch.setenv("HOME", str(tmp_path)) # no credential stores
+
+ class _Ctx:
+ current_provider = "nous"
+
+ rows = [{"slug": "copilot-acp", "models": ["gpt-5.4"]}]
+ kept = _filter_explicit_provider_rows(rows, _Ctx())
+
+ assert all(r["slug"] != "copilot-acp" for r in kept)
+
+
+# --- Accounts-tab cli_command ------------------------------------------------
+
+
+def test_catalog_sign_in_command_is_a_valid_copilot_invocation():
+ from hermes_cli.web_server import _OAUTH_PROVIDER_CATALOG
+
+ entry = next(e for e in _OAUTH_PROVIDER_CATALOG if e["id"] == "copilot-acp")
+ # `copilot /login` is not a valid invocation — slash-commands only exist
+ # inside an interactive session. The catalog must hand users a command
+ # that actually starts a login flow.
+ assert entry["cli_command"] == "copilot login"
+
+
+def test_cli_command_reflects_configured_executable(tmp_path, monkeypatch, _clean_copilot_env):
+ from hermes_cli.web_server import _external_process_cli_command
+
+ fake = tmp_path / ("copilot.exe" if os.name == "nt" else "copilot")
+ fake.write_text("", encoding="utf-8")
+ fake.chmod(0o755)
+ monkeypatch.setenv("HERMES_COPILOT_ACP_COMMAND", str(fake))
+
+ rendered = _external_process_cli_command("copilot-acp", "copilot login")
+
+ assert rendered == f"{fake} login"
+
+
+def test_cli_command_untouched_for_non_external_providers(_clean_copilot_env):
+ from hermes_cli.web_server import _external_process_cli_command
+
+ assert _external_process_cli_command("nous", "hermes auth add nous") == "hermes auth add nous"
+
+
+def test_cli_command_default_when_no_override(monkeypatch, _clean_copilot_env):
+ from hermes_cli.web_server import _external_process_cli_command
+
+ assert _external_process_cli_command("copilot-acp", "copilot login") == "copilot login"
+
+
+# --- live catalog key from the Copilot CLI store -----------------------------
+
+
+def test_catalog_key_resolves_from_copilot_cli_store(tmp_path, monkeypatch, _clean_copilot_env):
+ # A user whose ONLY credential is `copilot login` must still get the live
+ # model catalog — otherwise the picker silently falls back to the stale
+ # curated list (visibly wrong vs. what their subscription serves).
+ from unittest.mock import patch as mock_patch
+
+ from hermes_cli import models as models_mod
+
+ monkeypatch.setenv("HOME", str(tmp_path))
+ cfg_dir = tmp_path / ".copilot"
+ cfg_dir.mkdir()
+ (cfg_dir / "config.json").write_text(
+ "// managed\n"
+ '{"copilotTokens": {"https://github.com:u": "gho_' + "x" * 36 + '"}}\n',
+ encoding="utf-8",
+ )
+
+ with mock_patch.object(
+ models_mod, "_resolve_copilot_catalog_api_key", wraps=models_mod._resolve_copilot_catalog_api_key
+ ), mock_patch(
+ "hermes_cli.copilot_auth.exchange_copilot_token",
+ return_value=("exchanged-api-token", 0.0, None),
+ ), mock_patch(
+ "hermes_cli.auth.resolve_api_key_provider_credentials",
+ side_effect=Exception("no env creds"),
+ ), mock_patch(
+ "hermes_cli.auth.read_credential_pool", return_value=[]
+ ):
+ key = models_mod._resolve_copilot_catalog_api_key()
+
+ assert key == "exchanged-api-token"
+
+
+def test_catalog_key_empty_when_cli_store_absent(tmp_path, monkeypatch, _clean_copilot_env):
+ from unittest.mock import patch as mock_patch
+
+ from hermes_cli import models as models_mod
+
+ monkeypatch.setenv("HOME", str(tmp_path)) # no ~/.copilot at all
+
+ with mock_patch(
+ "hermes_cli.auth.resolve_api_key_provider_credentials",
+ side_effect=Exception("no env creds"),
+ ), mock_patch(
+ "hermes_cli.auth.read_credential_pool", return_value=[]
+ ):
+ key = models_mod._resolve_copilot_catalog_api_key()
+
+ assert key == ""
diff --git a/tests/hermes_cli/test_fast_serve_launch.py b/tests/hermes_cli/test_fast_serve_launch.py
new file mode 100644
index 0000000000..a0f3961596
--- /dev/null
+++ b/tests/hermes_cli/test_fast_serve_launch.py
@@ -0,0 +1,51 @@
+from __future__ import annotations
+
+import argparse
+import sys
+
+import hermes_cli.config as config_mod
+import hermes_cli.main as main_mod
+from hermes_cli.subcommands.dashboard import build_dashboard_parser, build_serve_parser
+
+
+def _capture(_args) -> None:
+ return None
+
+
+def test_lean_serve_parser_matches_full_subcommand_parser() -> None:
+ root = argparse.ArgumentParser()
+ subparsers = root.add_subparsers(dest="command")
+ build_dashboard_parser(subparsers, cmd_dashboard=_capture, cmd_dashboard_register=_capture)
+ lean = build_serve_parser(cmd_dashboard=_capture)
+
+ argv = [
+ "--host", "127.0.0.1", "--port", "0", "--no-open",
+ "--ssh-session-token-file", "token.txt", "--ssh-owner-nonce", "0123456789abcdef",
+ ]
+
+ assert vars(lean.parse_args(argv)) == vars(root.parse_args(["serve", *argv]))
+
+
+def test_fast_serve_launch_dispatches_only_unambiguous_serve(monkeypatch) -> None:
+ captured = []
+ monkeypatch.setattr(config_mod, "get_container_exec_info", lambda: None)
+ monkeypatch.setattr(main_mod, "cmd_dashboard", captured.append)
+
+ monkeypatch.setattr(sys, "argv", ["hermes", "serve", "--host", "127.0.0.1", "--port", "0"])
+ assert main_mod._try_fast_serve_launch() is True
+ assert (captured[0].command, captured[0].headless_backend, captured[0].no_open, captured[0].port) == (
+ "serve", True, True, 0,
+ )
+
+ # Every ambiguous shape falls back to the full parser: unknown flags,
+ # help, the opt-out, and container routing.
+ for argv in (["serve", "--future-flag"], ["serve", "--help"], ["chat"]):
+ monkeypatch.setattr(sys, "argv", ["hermes", *argv])
+ assert main_mod._try_fast_serve_launch() is False
+ monkeypatch.setenv("HERMES_DISABLE_FAST_SERVE_LAUNCH", "1")
+ monkeypatch.setattr(sys, "argv", ["hermes", "serve"])
+ assert main_mod._try_fast_serve_launch() is False
+ monkeypatch.delenv("HERMES_DISABLE_FAST_SERVE_LAUNCH")
+ monkeypatch.setattr(config_mod, "get_container_exec_info", lambda: {"name": "managed"})
+ assert main_mod._try_fast_serve_launch() is False
+ assert len(captured) == 1
diff --git a/tests/hermes_cli/test_gateway_job_teardown_live.py b/tests/hermes_cli/test_gateway_job_teardown_live.py
new file mode 100644
index 0000000000..e6245c10c6
--- /dev/null
+++ b/tests/hermes_cli/test_gateway_job_teardown_live.py
@@ -0,0 +1,335 @@
+"""LIVE Windows E2E for #48820 (4th repro): Job-Object teardown vs the
+gateway restart watcher, with real processes on a real windows-latest runner.
+
+Three live proofs (no mocks of the code under test):
+
+1. ``TestJobObjectMechanismLive`` — the mechanism everything rests on:
+ a child spawned with ``windows_detach_flags()`` (CREATE_BREAKAWAY_FROM_JOB)
+ from inside a kill-on-close Job Object SURVIVES the job teardown, while a
+ child spawned with ``windows_detach_flags_without_breakaway()`` is killed
+ by it. This is exactly the reporter's suspected kill path.
+
+2. ``TestWatcherRespawnLive`` — drives the REAL
+ ``hermes_cli.gateway._spawn_gateway_restart_watcher`` end to end with a
+ real stub gateway process, against a temp HERMES_HOME:
+ - the respawned process's stderr must land in ``logs/gateway-stdio.log``
+ (on unfixed main it went to DEVNULL: a job-teardown kill left ZERO trace);
+ - the respawn env must carry ``_HERMES_GATEWAY_BREAKAWAY=1`` (the stamp
+ that makes a later job-teardown death diagnosable in exit-diag).
+
+3. ``TestResumeVerificationLive`` — the user-visible symptom: the updater's
+ ``_resume_windows_gateways_after_update`` must NOT print
+ "✓ Restarting Windows gateway profile(s)" when the relaunched gateway is
+ dead. The relaunch chain runs for real; the "gateway" is a stub that exits
+ immediately (standing in for the job-teardown kill). On unfixed main the ✓
+ is printed anyway; after the fix the resume raises "not verified alive".
+"""
+
+from __future__ import annotations
+
+import ctypes
+import os
+import subprocess
+import sys
+import time
+from ctypes import wintypes
+from pathlib import Path
+
+import pytest
+
+pytestmark = [
+ pytest.mark.windows_only,
+ pytest.mark.skipif(sys.platform != "win32", reason="native Windows only"),
+]
+
+_REPO_ROOT = Path(__file__).resolve().parents[2]
+
+JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE = 0x00002000
+JOB_OBJECT_LIMIT_BREAKAWAY_OK = 0x00000800
+JobObjectExtendedLimitInformation = 9
+PROCESS_ALL_ACCESS = 0x001FFFFF
+
+
+class IO_COUNTERS(ctypes.Structure):
+ _fields_ = [
+ ("ReadOperationCount", ctypes.c_ulonglong),
+ ("WriteOperationCount", ctypes.c_ulonglong),
+ ("OtherOperationCount", ctypes.c_ulonglong),
+ ("ReadTransferCount", ctypes.c_ulonglong),
+ ("WriteTransferCount", ctypes.c_ulonglong),
+ ("OtherTransferCount", ctypes.c_ulonglong),
+ ]
+
+
+class JOBOBJECT_BASIC_LIMIT_INFORMATION(ctypes.Structure):
+ _fields_ = [
+ ("PerProcessUserTimeLimit", ctypes.c_longlong),
+ ("PerJobUserTimeLimit", ctypes.c_longlong),
+ ("LimitFlags", wintypes.DWORD),
+ ("MinimumWorkingSetSize", ctypes.c_size_t),
+ ("MaximumWorkingSetSize", ctypes.c_size_t),
+ ("ActiveProcessLimit", wintypes.DWORD),
+ ("Affinity", ctypes.c_size_t),
+ ("PriorityClass", wintypes.DWORD),
+ ("SchedulingClass", wintypes.DWORD),
+ ]
+
+
+class JOBOBJECT_EXTENDED_LIMIT_INFORMATION(ctypes.Structure):
+ _fields_ = [
+ ("BasicLimitInformation", JOBOBJECT_BASIC_LIMIT_INFORMATION),
+ ("IoInfo", IO_COUNTERS),
+ ("ProcessMemoryLimit", ctypes.c_size_t),
+ ("JobMemoryLimit", ctypes.c_size_t),
+ ("PeakProcessMemoryUsed", ctypes.c_size_t),
+ ("PeakJobMemoryUsed", ctypes.c_size_t),
+ ]
+
+
+def _make_kill_on_close_job(allow_breakaway: bool) -> int:
+ kernel32 = ctypes.windll.kernel32
+ job = kernel32.CreateJobObjectW(None, None)
+ assert job, "CreateJobObjectW failed"
+ info = JOBOBJECT_EXTENDED_LIMIT_INFORMATION()
+ flags = JOB_OBJECT_LIMIT_KILL_ON_JOB_CLOSE
+ if allow_breakaway:
+ flags |= JOB_OBJECT_LIMIT_BREAKAWAY_OK
+ info.BasicLimitInformation.LimitFlags = flags
+ ok = kernel32.SetInformationJobObject(
+ job,
+ JobObjectExtendedLimitInformation,
+ ctypes.byref(info),
+ ctypes.sizeof(info),
+ )
+ assert ok, "SetInformationJobObject failed"
+ return job
+
+
+def _assign_to_job(job: int, proc: subprocess.Popen) -> None:
+ kernel32 = ctypes.windll.kernel32
+ ok = kernel32.AssignProcessToJobObject(job, int(proc._handle))
+ assert ok, f"AssignProcessToJobObject failed (winerror={ctypes.GetLastError()})"
+
+
+def _pid_alive(pid: int) -> bool:
+ kernel32 = ctypes.windll.kernel32
+ PROCESS_QUERY_LIMITED_INFORMATION = 0x1000
+ h = kernel32.OpenProcess(PROCESS_QUERY_LIMITED_INFORMATION, False, pid)
+ if not h:
+ return False
+ try:
+ code = wintypes.DWORD()
+ kernel32.GetExitCodeProcess(h, ctypes.byref(code))
+ return code.value == 259 # STILL_ACTIVE
+ finally:
+ kernel32.CloseHandle(h)
+
+
+_SLEEPER = "import time; time.sleep(120)"
+
+
+def _wait_for(predicate, timeout_s: float = 30.0, interval_s: float = 0.25):
+ deadline = time.monotonic() + timeout_s
+ while time.monotonic() < deadline:
+ if predicate():
+ return True
+ time.sleep(interval_s)
+ return False
+
+
+class TestJobObjectMechanismLive:
+ """Real Job Objects, real children — the #48820 kill mechanism."""
+
+ def _driver_source(self, flags_helper: str, pid_file: str) -> str:
+ # The driver runs INSIDE the job and spawns a grandchild "gateway"
+ # with the flag bundle under test, then exits — mirroring the
+ # updater/watcher exiting while its job tears down.
+ return (
+ "import subprocess, sys, pathlib\n"
+ "sys.path.insert(0, r'%s')\n"
+ "from hermes_cli._subprocess_compat import (\n"
+ " windows_detach_flags, windows_detach_flags_without_breakaway)\n"
+ "flags = %s()\n"
+ "p = subprocess.Popen([sys.executable, '-c', %r],\n"
+ " creationflags=flags, stdin=subprocess.DEVNULL,\n"
+ " stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)\n"
+ "pathlib.Path(r'%s').write_text(str(p.pid), encoding='utf-8')\n"
+ ) % (str(_REPO_ROOT), flags_helper, _SLEEPER, pid_file)
+
+ def _run_in_job(self, tmp_path: Path, flags_helper: str) -> int:
+ pid_file = tmp_path / f"{flags_helper}.pid"
+ job = _make_kill_on_close_job(allow_breakaway=True)
+ kernel32 = ctypes.windll.kernel32
+ try:
+ driver = subprocess.Popen(
+ [
+ sys.executable,
+ "-c",
+ # Handshake: wait until the test has assigned us to the
+ # job before spawning the grandchild.
+ "import pathlib, sys, time\n"
+ f"go = pathlib.Path(r'{tmp_path / 'go.marker'}')\n"
+ "deadline = time.monotonic() + 30\n"
+ "while not go.exists():\n"
+ " assert time.monotonic() < deadline, 'no go marker'\n"
+ " time.sleep(0.1)\n"
+ + self._driver_source(flags_helper, str(pid_file)),
+ ],
+ cwd=str(_REPO_ROOT),
+ )
+ _assign_to_job(job, driver)
+ (tmp_path / "go.marker").write_text("go", encoding="utf-8")
+ assert _wait_for(pid_file.exists), "driver never wrote the pid file"
+ gw_pid = int(pid_file.read_text(encoding="utf-8"))
+ assert _wait_for(lambda: driver.poll() is not None), (
+ "driver did not exit"
+ )
+ assert _pid_alive(gw_pid), "grandchild died before job teardown"
+ # THE teardown: closing the last job handle fires
+ # KILL_ON_JOB_CLOSE against every process still in the job.
+ kernel32.CloseHandle(job)
+ job = None
+ time.sleep(2.0)
+ return gw_pid
+ finally:
+ (tmp_path / "go.marker").unlink(missing_ok=True)
+ if job:
+ kernel32.CloseHandle(job)
+
+ def test_breakaway_child_survives_job_teardown(self, tmp_path):
+ pid = self._run_in_job(tmp_path, "windows_detach_flags")
+ try:
+ assert _pid_alive(pid), (
+ "CREATE_BREAKAWAY_FROM_JOB child must survive the parent "
+ "job's kill-on-close teardown"
+ )
+ finally:
+ subprocess.run(
+ ["taskkill", "/PID", str(pid), "/T", "/F"], capture_output=True
+ )
+
+ def test_non_breakaway_child_killed_by_job_teardown(self, tmp_path):
+ """The #48820 kill path, reproduced live: no breakaway → the job's
+ teardown reaps the freshly spawned gateway."""
+ pid = self._run_in_job(tmp_path, "windows_detach_flags_without_breakaway")
+ try:
+ assert not _pid_alive(pid), (
+ "child without breakaway must be killed by kill-on-close "
+ "job teardown — this is the silent gateway death of #48820"
+ )
+ finally:
+ subprocess.run(
+ ["taskkill", "/PID", str(pid), "/T", "/F"], capture_output=True
+ )
+
+
+class TestWatcherRespawnLive:
+ """Drive the real ``_spawn_gateway_restart_watcher`` with real processes."""
+
+ def _run_watcher_cycle(self, tmp_path: Path, monkeypatch) -> tuple[Path, Path]:
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home"))
+ (tmp_path / "home").mkdir(parents=True, exist_ok=True)
+
+ marker = tmp_path / "respawned.marker"
+ # The stub "gateway": records its breakaway stamp env, screams on
+ # stderr (so the stdio sidecar has something to capture), then exits.
+ stub = (
+ "import os, pathlib, sys\n"
+ f"pathlib.Path(r'{marker}').write_text(\n"
+ " os.environ.get('_HERMES_GATEWAY_BREAKAWAY', 'MISSING'),\n"
+ " encoding='utf-8')\n"
+ "print('stub-gateway-stderr-trace', file=sys.stderr)\n"
+ )
+
+ # A real old-pid that exits immediately — the watcher's poll loop
+ # sees it die and respawns.
+ old = subprocess.Popen([sys.executable, "-c", "pass"])
+ old.wait(timeout=30)
+
+ import hermes_cli.gateway as gateway
+
+ assert gateway._spawn_gateway_restart_watcher(
+ old.pid, [sys.executable, "-c", stub]
+ ), "watcher spawn returned False"
+
+ assert _wait_for(marker.exists, timeout_s=60), (
+ "watcher never respawned the stub gateway"
+ )
+ stdio_log = tmp_path / "home" / "logs" / "gateway-stdio.log"
+ return marker, stdio_log
+
+ def test_respawn_stamps_breakaway_and_leaves_stdio_trace(
+ self, tmp_path, monkeypatch
+ ):
+ marker, stdio_log = self._run_watcher_cycle(tmp_path, monkeypatch)
+
+ # (a) Breakaway stamp: on unfixed main the respawn env carried no
+ # stamp, so a job-teardown death was undiagnosable.
+ stamp = marker.read_text(encoding="utf-8").strip()
+ assert stamp in {"1", "0"}, (
+ f"respawned gateway must carry the breakaway stamp, got {stamp!r}"
+ )
+
+ # (b) Stdio trace: on unfixed main stderr went to DEVNULL — a dying
+ # gateway left zero trace (#48820 4th repro).
+ assert _wait_for(
+ lambda: stdio_log.exists()
+ and "stub-gateway-stderr-trace"
+ in stdio_log.read_text(encoding="utf-8", errors="replace"),
+ timeout_s=30,
+ ), "respawned gateway stderr must land in logs/gateway-stdio.log"
+
+
+class TestResumeVerificationLive:
+ """The user-visible lie: '✓ Restarting' printed for a dead gateway."""
+
+ def test_dead_relaunch_is_not_reported_as_success(self, tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / "home"))
+ (tmp_path / "home").mkdir(parents=True, exist_ok=True)
+
+ import hermes_cli.gateway as gateway
+ import hermes_cli.main as hm
+ from hermes_cli.update_cmd import _resume_windows_gateways_after_update
+
+ # Peripheral only: don't regenerate launcher scripts into the temp home.
+ monkeypatch.setattr(hm, "_refresh_windows_gateway_launchers", lambda: None)
+
+ # Real relaunch chain, real watcher, real spawn — but the respawned
+ # "gateway" exits immediately, standing in for the Job-Object
+ # teardown kill. It never registers in the process table as a
+ # gateway, exactly like the dead pid 48452 / 50456 of #48820.
+ def _relaunch(profile, old_pid):
+ dead = subprocess.Popen([sys.executable, "-c", "pass"])
+ dead.wait(timeout=30)
+ return gateway._spawn_gateway_restart_watcher(
+ dead.pid, [sys.executable, "-c", "pass"]
+ )
+
+ monkeypatch.setattr(
+ gateway, "launch_detached_profile_gateway_restart", _relaunch
+ )
+
+ token = {
+ "resume_needed": True,
+ "profiles": {"default": 999999},
+ "unmapped_pids": [],
+ "unmapped": [],
+ }
+
+ printed: list = []
+ real_print = print
+ monkeypatch.setattr(
+ "builtins.print", lambda *a, **k: printed.append(" ".join(map(str, a)))
+ )
+ try:
+ with pytest.raises(RuntimeError, match="not verified alive"):
+ _resume_windows_gateways_after_update(token)
+ finally:
+ monkeypatch.setattr("builtins.print", real_print)
+
+ text = "\n".join(printed)
+ assert "✓ Restarting" not in text, (
+ "the updater must not vouch for a gateway that is not alive "
+ f"(#48820). Printed:\n{text}"
+ )
+ assert "could not be verified" in text
diff --git a/tests/hermes_cli/test_gateway_multiplex_status.py b/tests/hermes_cli/test_gateway_multiplex_status.py
new file mode 100644
index 0000000000..0c516db590
--- /dev/null
+++ b/tests/hermes_cli/test_gateway_multiplex_status.py
@@ -0,0 +1,61 @@
+"""PR #69118: a named profile served by the default multiplexer reports as running.
+
+``hermes gateway status`` / ``gateway list`` / ``profile list`` keyed liveness
+off the profile's own gateway.pid, so a satellite profile served by the default
+multiplexer showed "not running" even though the multiplexer was its live
+inbound process. All three now consult the same
+``named_profile_served_by_running_multiplexer()`` lookup the start guard and
+cron liveness use.
+"""
+
+from __future__ import annotations
+
+import io
+import os
+from contextlib import redirect_stdout
+from types import SimpleNamespace
+
+
+def _fake_multiplexer(monkeypatch, tmp_path, *, multiplex: bool):
+ import hermes_constants
+ import gateway.status as status
+
+ (tmp_path / "profiles" / "beta").mkdir(parents=True)
+ (tmp_path / "config.yaml").write_text(
+ f"gateway:\n multiplex_profiles: {'true' if multiplex else 'false'}\n"
+ )
+ (tmp_path / "gateway.pid").write_text(str(os.getpid()))
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / "profiles" / "beta"))
+ monkeypatch.setattr(hermes_constants, "_default_hermes_root_memo", None)
+ monkeypatch.setattr(status, "_pid_exists", lambda pid: True)
+
+
+def _run_status():
+ from hermes_cli import gateway as gw
+
+ buf = io.StringIO()
+ with redirect_stdout(buf):
+ gw._gateway_command_inner(
+ SimpleNamespace(gateway_command="status", deep=False, full=False, system=False)
+ )
+ return buf.getvalue().splitlines()[0]
+
+
+def test_served_named_profile_reports_running(monkeypatch, tmp_path):
+ from hermes_cli.profiles import list_profiles
+
+ _fake_multiplexer(monkeypatch, tmp_path, multiplex=True)
+
+ beta = next(p for p in list_profiles() if p.name == "beta")
+ assert beta.gateway_running is True
+ assert _run_status().startswith("✓ Gateway is running via the default-profile multiplexer")
+
+
+def test_unserved_named_profile_still_reports_stopped(monkeypatch, tmp_path):
+ from hermes_cli.profiles import list_profiles
+
+ _fake_multiplexer(monkeypatch, tmp_path, multiplex=False)
+
+ beta = next(p for p in list_profiles() if p.name == "beta")
+ assert beta.gateway_running is False
+ assert _run_status().startswith("✗ Gateway is not running")
diff --git a/tests/hermes_cli/test_goals.py b/tests/hermes_cli/test_goals.py
index 625ccbe111..413a9330ed 100644
--- a/tests/hermes_cli/test_goals.py
+++ b/tests/hermes_cli/test_goals.py
@@ -798,3 +798,38 @@ class TestContractAndBackgroundCompose:
assert verdict == "wait"
assert wait_directive and wait_directive.get("pid") == 4242
+
+class TestBlockedVerdict:
+ """#100954: a genuinely unachievable goal must be refused, not completed."""
+
+ def test_parse_judge_response_accepts_blocked(self):
+ from hermes_cli.goals import _parse_judge_response
+
+ verdict, reason, parse_failed, _wd = _parse_judge_response(
+ '{"verdict": "blocked", "reason": "the repo was deleted"}'
+ )
+ assert verdict == "blocked"
+ assert reason == "the repo was deleted"
+ assert parse_failed is False
+
+ def test_blocked_verdict_pauses_goal_instead_of_done(self, hermes_home):
+ from unittest.mock import patch
+ from hermes_cli.goals import GoalManager
+
+ mgr = GoalManager(session_id="blocked-sid")
+ mgr.set("delete a repository that does not exist")
+ with patch(
+ "hermes_cli.goals.judge_goal",
+ return_value=("blocked", "the repo does not exist", False, None, False),
+ ):
+ decision = mgr.evaluate_after_turn(
+ "The repo cannot be deleted: it does not exist."
+ )
+
+ assert decision["verdict"] == "blocked"
+ assert decision["status"] == "paused"
+ assert decision["should_continue"] is False
+ assert "unachievable" in decision["message"].lower()
+ assert mgr.state is not None
+ assert mgr.state.status == "paused"
+ assert "unachievable" in (mgr.state.paused_reason or "").lower()
diff --git a/tests/hermes_cli/test_gui_command.py b/tests/hermes_cli/test_gui_command.py
index 85348f9a6e..c9f6e483e2 100644
--- a/tests/hermes_cli/test_gui_command.py
+++ b/tests/hermes_cli/test_gui_command.py
@@ -99,6 +99,41 @@ def _make_packaged_executable(root: Path, monkeypatch) -> Path:
return exe
+def _staging_dir_from(cmd) -> Path:
+ """Extract the ``-c.directories.output=`` electron-builder override
+ ``cmd_gui`` appends to ``npm run pack`` (stage-and-swap, #86443)."""
+ for arg in cmd:
+ if isinstance(arg, str) and arg.startswith("-c.directories.output="):
+ return Path(arg.split("=", 1)[1])
+ raise AssertionError(f"no staging output override in {cmd!r}")
+
+
+def _packaged_exe_rel() -> Path:
+ """Packaged-exe path relative to electron-builder's output dir on THIS host."""
+ if sys.platform == "darwin":
+ return Path("mac-arm64") / "Hermes.app" / "Contents" / "MacOS" / "Hermes"
+ if sys.platform == "win32":
+ return Path("win-unpacked") / "Hermes.exe"
+ return Path("linux-unpacked") / "hermes"
+
+
+def _pack_into_staging(root: Path, content: str = "", returncode: int = 0):
+ """``subprocess.run`` side effect mimicking a real ``npm run pack``: lays
+ the packaged app down inside the STAGING dir named on the command line
+ (never in release/), then returns *returncode*. Non-pack commands (the
+ launch) return success."""
+ def _run(cmd, **kwargs):
+ if len(cmd) >= 3 and cmd[1:3] == ["run", "pack"]:
+ exe = _staging_dir_from(cmd) / _packaged_exe_rel()
+ exe.parent.mkdir(parents=True, exist_ok=True)
+ exe.write_text(content, encoding="utf-8")
+ if sys.platform not in ("darwin", "win32"):
+ (exe.parent / "chrome-sandbox").write_text("", encoding="utf-8")
+ return subprocess.CompletedProcess(cmd, returncode)
+ return subprocess.CompletedProcess(cmd, 0)
+ return _run
+
+
def test_gui_installs_packages_and_launches_desktop_app(tmp_path, monkeypatch):
root = _make_desktop_tree(tmp_path)
desktop_dir = root / "apps" / "desktop"
@@ -116,7 +151,7 @@ def test_gui_installs_packages_and_launches_desktop_app(tmp_path, monkeypatch):
patch("hermes_cli.main._desktop_macos_relaunchable_fixup"), \
patch("hermes_cli.main._desktop_linux_sandbox_fixup", return_value=True), \
patch("hermes_cli.main._register_linux_desktop_entry"), \
- patch("hermes_cli.main.subprocess.run", side_effect=[pack_ok, launch_ok]) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run, \
pytest.raises(SystemExit) as exc:
cli_main.cmd_gui(_ns())
@@ -128,7 +163,13 @@ def test_gui_installs_packages_and_launches_desktop_app(tmp_path, monkeypatch):
assert mock_install.call_args.kwargs["capture_output"] is False
install_env = mock_install.call_args.kwargs["env"]
assert install_env is not None and "PATH" in install_env
- assert mock_run.call_args_list[0].args[0] == ["/usr/bin/npm", "run", "pack"]
+ pack_cmd = mock_run.call_args_list[0].args[0]
+ assert pack_cmd[:4] == ["/usr/bin/npm", "run", "pack", "--"]
+ # Stage-and-swap (#86443): the pack targets a staging dir beside release/,
+ # never release/ itself.
+ staging = _staging_dir_from(pack_cmd)
+ assert staging.parent == desktop_dir and staging.name.startswith(".staging-")
+ assert not staging.exists() # swapped into release/ and cleaned up
assert mock_run.call_args_list[0].kwargs["cwd"] == desktop_dir
launched = mock_run.call_args_list[1].args[0]
if sys.platform.startswith("linux"):
@@ -258,23 +299,29 @@ def test_gui_does_not_retry_after_packaged_executable_exists(tmp_path, monkeypat
"""
root = _make_desktop_tree(tmp_path)
monkeypatch.setattr(cli_main, "PROJECT_ROOT", root)
- # Executable EXISTS at failure time → late failure, not a corrupt download.
- _make_packaged_executable(root, monkeypatch)
+ live_exe = _make_packaged_executable(root, monkeypatch)
+ live_exe.write_text("good build", encoding="utf-8")
monkeypatch.delenv("ELECTRON_MIRROR", raising=False)
install_ok = subprocess.CompletedProcess(["npm", "ci"], 0)
- pack_fail = subprocess.CompletedProcess(["npm", "run", "pack"], 1)
+ # Executable EXISTS in the STAGING output at failure time → late failure
+ # (e.g. signing), not a corrupt download. With stage-and-swap (#86443) the
+ # discriminator reads the staging dir, so the fake pack lays it down there.
+ pack_fail = _pack_into_staging(root, content="half-signed", returncode=1)
with patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"), \
patch("hermes_cli.main._run_npm_install_deterministic", return_value=install_ok), \
patch("hermes_cli.main._desktop_macos_relaunchable_fixup"), \
patch("hermes_cli.main._purge_electron_build_cache", return_value=[Path("/c/electron.zip")]) as mock_purge, \
patch("hermes_cli.main._redownload_electron_dist", return_value=True) as mock_dl, \
- patch("hermes_cli.main.subprocess.run", return_value=pack_fail) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=pack_fail) as mock_run, \
pytest.raises(SystemExit) as exc:
cli_main.cmd_gui(_ns())
assert exc.value.code == 1
+ # The live app was never touched by the failed pack (#86443).
+ assert live_exe.read_text(encoding="utf-8") == "good build"
+ assert not list((root / "apps" / "desktop").glob(".staging-*"))
# Neither destructive recovery runs, and there is exactly ONE pack attempt.
mock_purge.assert_not_called()
mock_dl.assert_not_called()
@@ -1062,7 +1109,7 @@ def test_gui_bridges_ozone_hint_to_launch_env(tmp_path, monkeypatch):
patch("hermes_cli.main._desktop_linux_sandbox_fixup", return_value=True), \
patch("hermes_cli.config.load_config", return_value=cfg), \
patch("hermes_cli.linux_desktop_entry.install_desktop_entry", return_value=None), \
- patch("hermes_cli.main.subprocess.run", side_effect=[ok, ok]) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run, \
pytest.raises(SystemExit):
cli_main.cmd_gui(_ns())
@@ -1078,7 +1125,7 @@ def test_gui_bridges_ozone_hint_to_launch_env(tmp_path, monkeypatch):
patch("hermes_cli.main._desktop_linux_sandbox_fixup", return_value=True), \
patch("hermes_cli.config.load_config", return_value=cfg), \
patch("hermes_cli.linux_desktop_entry.install_desktop_entry", return_value=None), \
- patch("hermes_cli.main.subprocess.run", side_effect=[ok, ok]) as mock_run2, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run2, \
pytest.raises(SystemExit):
cli_main.cmd_gui(_ns())
@@ -1160,7 +1207,7 @@ def test_gui_linux_packaged_launch_bridges_detected_password_store(tmp_path, mon
patch("hermes_cli.config.load_config", return_value={}), \
patch("hermes_cli.linux_desktop_entry.install_desktop_entry", return_value=None), \
patch("hermes_cli.main._detect_linux_password_store", return_value="gnome-libsecret"), \
- patch("hermes_cli.main.subprocess.run", side_effect=[ok, ok]) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run, \
pytest.raises(SystemExit):
cli_main.cmd_gui(_ns())
@@ -1183,7 +1230,7 @@ def test_gui_linux_source_launch_bridges_detected_password_store(tmp_path, monke
patch("hermes_cli.config.load_config", return_value={}), \
patch("hermes_cli.linux_desktop_entry.install_desktop_entry", return_value=None), \
patch("hermes_cli.main._detect_linux_password_store", return_value="kwallet6"), \
- patch("hermes_cli.main.subprocess.run", side_effect=[ok, ok]) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run, \
pytest.raises(SystemExit):
cli_main.cmd_gui(_ns(source=True))
@@ -1211,7 +1258,7 @@ def test_gui_config_password_store_skips_detection(tmp_path, monkeypatch):
patch("hermes_cli.config.load_config", return_value=cfg), \
patch("hermes_cli.linux_desktop_entry.install_desktop_entry", return_value=None), \
patch("hermes_cli.main._detect_linux_password_store") as mock_detect, \
- patch("hermes_cli.main.subprocess.run", side_effect=[ok, ok]) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run, \
pytest.raises(SystemExit):
cli_main.cmd_gui(_ns())
@@ -1240,7 +1287,7 @@ def test_gui_explicit_password_store_env_wins_over_config_and_detection(tmp_path
patch("hermes_cli.config.load_config", return_value=cfg), \
patch("hermes_cli.linux_desktop_entry.install_desktop_entry", return_value=None), \
patch("hermes_cli.main._detect_linux_password_store") as mock_detect, \
- patch("hermes_cli.main.subprocess.run", side_effect=[ok, ok]) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run, \
pytest.raises(SystemExit):
cli_main.cmd_gui(_ns())
@@ -1266,10 +1313,178 @@ def test_gui_password_store_bridge_is_linux_only(tmp_path, monkeypatch):
patch("hermes_cli.config.load_config", return_value={}), \
patch("hermes_cli.linux_desktop_entry.install_desktop_entry", return_value=None), \
patch("hermes_cli.main._detect_linux_password_store") as mock_detect, \
- patch("hermes_cli.main.subprocess.run", side_effect=[ok, ok]) as mock_run, \
+ patch("hermes_cli.main.subprocess.run", side_effect=_pack_into_staging(root)) as mock_run, \
pytest.raises(SystemExit):
cli_main.cmd_gui(_ns())
mock_detect.assert_not_called()
launch_env = mock_run.call_args_list[1].kwargs["env"]
assert "HERMES_DESKTOP_PASSWORD_STORE" not in launch_env
+
+
+# ---------------------------------------------------------------------------
+# #86443: stage-and-swap — a failed Desktop rebuild must never remove the
+# working app. electron-builder packs IN PLACE (before-pack.mjs wipes
+# release/ first), so cmd_gui now packs into a staging dir and only
+# renames it over release/ after the staged result verifies.
+# ---------------------------------------------------------------------------
+
+
+def _gui_build_patches(root: Path, run_side_effect):
+ return [
+ patch("hermes_cli.main.shutil.which", return_value="/usr/bin/npm"),
+ patch("hermes_cli.main._run_npm_install_deterministic",
+ return_value=subprocess.CompletedProcess(["npm", "ci"], 0)),
+ patch("hermes_cli.main._desktop_build_needed", return_value=True),
+ patch("hermes_cli.main._write_desktop_build_stamp"),
+ patch("hermes_cli.main._desktop_macos_relaunchable_fixup"),
+ patch("hermes_cli.main._register_linux_desktop_entry"),
+ patch("hermes_cli.main._stop_desktop_processes_locking_build", return_value=[]),
+ patch("hermes_cli.main._purge_electron_build_cache", return_value=[]),
+ patch("hermes_cli.main._redownload_electron_dist", return_value=False),
+ patch("hermes_cli.main.subprocess.run", side_effect=run_side_effect),
+ ]
+
+
+def test_swap_staged_desktop_app_promotes_staged_tree_and_drops_previous(tmp_path):
+ root = _make_desktop_tree(tmp_path)
+ desktop_dir = root / "apps" / "desktop"
+ live_exe = desktop_dir / "release" / _packaged_exe_rel()
+ live_exe.parent.mkdir(parents=True)
+ live_exe.write_text("old", encoding="utf-8")
+ staging = cli_main._desktop_staging_dir(desktop_dir)
+ staged_exe = staging / _packaged_exe_rel()
+ staged_exe.parent.mkdir(parents=True)
+ staged_exe.write_text("new", encoding="utf-8")
+
+ promoted = cli_main._swap_staged_desktop_app(desktop_dir, staging)
+
+ assert promoted == live_exe
+ assert live_exe.read_text(encoding="utf-8") == "new"
+ assert not staging.exists()
+ assert sorted(p.name for p in (desktop_dir / "release").iterdir()) == [_packaged_exe_rel().parts[0]]
+
+
+def test_swap_staged_desktop_app_without_staged_exe_keeps_live_app(tmp_path):
+ """Zero-exit pack that produced nothing: live app untouched, staging gone."""
+ root = _make_desktop_tree(tmp_path)
+ desktop_dir = root / "apps" / "desktop"
+ live_exe = desktop_dir / "release" / _packaged_exe_rel()
+ live_exe.parent.mkdir(parents=True)
+ live_exe.write_text("old", encoding="utf-8")
+ staging = cli_main._desktop_staging_dir(desktop_dir)
+ (staging / "linux-unpacked" / "resources").mkdir(parents=True) # partial tree, no exe
+
+ assert cli_main._swap_staged_desktop_app(desktop_dir, staging) is None
+ assert live_exe.read_text(encoding="utf-8") == "old"
+ assert not staging.exists()
+
+
+def test_swap_staged_desktop_app_rolls_back_when_second_rename_fails(tmp_path, monkeypatch):
+ root = _make_desktop_tree(tmp_path)
+ desktop_dir = root / "apps" / "desktop"
+ live_exe = desktop_dir / "release" / _packaged_exe_rel()
+ live_exe.parent.mkdir(parents=True)
+ live_exe.write_text("old", encoding="utf-8")
+ staging = cli_main._desktop_staging_dir(desktop_dir)
+ staged_exe = staging / _packaged_exe_rel()
+ staged_exe.parent.mkdir(parents=True)
+ staged_exe.write_text("new", encoding="utf-8")
+
+ real_rename = cli_main.os.rename
+ calls = {"n": 0}
+
+ def flaky_rename(src, dst):
+ calls["n"] += 1
+ if calls["n"] == 2: # staged → live
+ raise OSError("EXDEV simulated")
+ return real_rename(src, dst)
+
+ monkeypatch.setattr(cli_main.os, "rename", flaky_rename)
+ assert cli_main._swap_staged_desktop_app(desktop_dir, staging) is None
+ assert live_exe.read_text(encoding="utf-8") == "old"
+ assert not (live_exe.parent.parent / (live_exe.parent.name + ".previous")).exists()
+
+
+def test_gui_failed_pack_leaves_previous_app_untouched(tmp_path, monkeypatch, capsys):
+ """Every pack attempt fails → the pre-existing app is exactly as it was,
+ no staging dir remains, exit is non-zero."""
+ root = _make_desktop_tree(tmp_path)
+ desktop_dir = root / "apps" / "desktop"
+ monkeypatch.setattr(cli_main, "PROJECT_ROOT", root)
+ live_exe = _make_packaged_executable(root, monkeypatch)
+ live_exe.write_text("good build", encoding="utf-8")
+ monkeypatch.setenv("ELECTRON_MIRROR", "https://example.test/electron/")
+
+ def failing_pack(cmd, **kwargs):
+ # Mimic before-pack.mjs wiping appOutDir inside the OUTPUT dir it was
+ # given, then dying (corrupt Electron zip → ENOENT on rename).
+ out = _staging_dir_from(cmd) / _packaged_exe_rel().parts[0]
+ out.mkdir(parents=True, exist_ok=True)
+ (out / "resources").mkdir(exist_ok=True)
+ return subprocess.CompletedProcess(cmd, 1)
+
+ patches = _gui_build_patches(root, failing_pack)
+ for p in patches:
+ p.start()
+ try:
+ with pytest.raises(SystemExit) as exc:
+ cli_main.cmd_gui(_ns(build_only=True))
+ finally:
+ for p in patches:
+ p.stop()
+
+ assert exc.value.code == 1
+ assert live_exe.read_text(encoding="utf-8") == "good build"
+ assert not list(desktop_dir.glob(".staging-*"))
+ assert not list((desktop_dir / "release").glob("*.previous"))
+ out = capsys.readouterr().out
+ assert "previous desktop app was left untouched" in out
+
+
+def test_gui_successful_pack_swaps_new_app_into_release(tmp_path, monkeypatch):
+ root = _make_desktop_tree(tmp_path)
+ desktop_dir = root / "apps" / "desktop"
+ monkeypatch.setattr(cli_main, "PROJECT_ROOT", root)
+ live_exe = _make_packaged_executable(root, monkeypatch)
+ live_exe.write_text("old build", encoding="utf-8")
+
+ patches = _gui_build_patches(root, _pack_into_staging(root, content="new build"))
+ for p in patches:
+ p.start()
+ try:
+ cli_main.cmd_gui(_ns(build_only=True))
+ finally:
+ for p in patches:
+ p.stop()
+
+ assert live_exe.read_text(encoding="utf-8") == "new build"
+ assert not list(desktop_dir.glob(".staging-*"))
+ assert not list((desktop_dir / "release").glob("*.previous"))
+
+
+def test_gui_zero_exit_pack_without_artifact_keeps_previous_app(tmp_path, monkeypatch, capsys):
+ root = _make_desktop_tree(tmp_path)
+ desktop_dir = root / "apps" / "desktop"
+ monkeypatch.setattr(cli_main, "PROJECT_ROOT", root)
+ live_exe = _make_packaged_executable(root, monkeypatch)
+ live_exe.write_text("good build", encoding="utf-8")
+
+ def empty_pack(cmd, **kwargs):
+ _staging_dir_from(cmd).mkdir(parents=True, exist_ok=True)
+ return subprocess.CompletedProcess(cmd, 0)
+
+ patches = _gui_build_patches(root, empty_pack)
+ for p in patches:
+ p.start()
+ try:
+ with pytest.raises(SystemExit) as exc:
+ cli_main.cmd_gui(_ns(build_only=True))
+ finally:
+ for p in patches:
+ p.stop()
+
+ assert exc.value.code == 1
+ assert live_exe.read_text(encoding="utf-8") == "good build"
+ assert not list(desktop_dir.glob(".staging-*"))
+ assert "produced no launchable app" in capsys.readouterr().out
diff --git a/tests/hermes_cli/test_hf_browse.py b/tests/hermes_cli/test_hf_browse.py
new file mode 100644
index 0000000000..7e7e689a90
--- /dev/null
+++ b/tests/hermes_cli/test_hf_browse.py
@@ -0,0 +1,186 @@
+"""The HF browser: search the firehose, price it roughly, and let any
+GGUF become a normal staged model.
+
+Parsing contracts run against canned HF API shapes (no network); route
+contracts run against the real FastAPI app with the HF client stubbed."""
+
+from __future__ import annotations
+
+import pytest
+from fastapi.testclient import TestClient
+
+from hermes_cli.local_runtime.estimator import HardwareBudget
+from hermes_cli.local_runtime.hf_browse import (
+ HFFileGroup,
+ HFModelHit,
+ repo_files,
+ rough_fit,
+ search_models,
+)
+
+GIB = 1 << 30
+
+
+@pytest.fixture
+def client(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ (tmp_path / ".hermes").mkdir()
+ from hermes_cli import web_server
+
+ test_client = TestClient(web_server.app)
+ test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
+ return test_client
+
+
+def _budget(vram_gib, ram_gib=64):
+ return HardwareBudget(usable_vram_bytes=int(vram_gib * GIB),
+ total_device_bytes=int(vram_gib * GIB),
+ ram_available_bytes=int(ram_gib * GIB))
+
+
+def test_search_parses_hf_hits(monkeypatch):
+ canned = [
+ {"id": "unsloth/Qwen3.8-27B-GGUF", "downloads": 872724, "likes": 47,
+ "lastModified": "2026-08-18", "gated": False},
+ {"id": "bartowski/whatever-GGUF", "downloads": 5, "likes": 0,
+ "lastModified": "2026-01-01", "gated": "auto"},
+ ]
+ monkeypatch.setattr("hermes_cli.local_runtime.hf_browse._get_json",
+ lambda url: canned)
+ hits = search_models("qwen")
+ assert hits[0].repo == "unsloth/Qwen3.8-27B-GGUF"
+ assert hits[0].downloads == 872724
+ assert hits[1].gated is True # HF 'auto'-gated counts as gated
+
+
+def test_repo_files_groups_splits_and_excludes_companions(monkeypatch):
+ canned = [
+ {"path": "Qwen3.8-27B-Q4_K_M.gguf", "size": 17 * GIB},
+ {"path": "mmproj-BF16.gguf", "size": 1 * GIB},
+ {"path": "UD-Q8/model-00001-of-00002.gguf", "size": 30 * GIB},
+ {"path": "UD-Q8/model-00002-of-00002.gguf", "size": 12 * GIB},
+ {"path": "README.md", "size": 1000},
+ {"path": "dspark-draft-Q8_0.gguf", "size": 9 * GIB},
+ ]
+ monkeypatch.setattr("hermes_cli.local_runtime.hf_browse._get_json",
+ lambda url: canned)
+ groups = repo_files("any/repo")
+ labels = {g.label: g for g in groups}
+ assert "Q4_K_M" in labels and labels["Q4_K_M"].total_bytes == 17 * GIB
+ # Split parts collapse into one group, ordered, summed.
+ split = next(g for g in groups if len(g.paths) == 2)
+ assert split.total_bytes == 42 * GIB
+ assert split.paths[0].endswith("00001-of-00002.gguf")
+ # Companions (mmproj, draft) are not standalone models.
+ assert not any("mmproj" in p or "dspark" in p
+ for g in groups for p in g.paths)
+ # Largest first.
+ assert groups[0].total_bytes >= groups[-1].total_bytes
+
+
+def test_rough_fit_bands():
+ b = _budget(29.6, ram_gib=64)
+ assert rough_fit(20 * GIB, b) == "fits-gpu" # + fill-ins under 29.6
+ assert rough_fit(28 * GIB, b) == "needs-ram" # weights spill
+ assert rough_fit(120 * GIB, b) == "too-big"
+
+
+def test_search_route_requires_query_and_maps_errors(client, monkeypatch):
+ r = client.get("/api/local-models/search", params={"q": " "})
+ assert r.status_code == 200 and r.json() == {"hits": []}
+
+ def boom(q, limit):
+ raise RuntimeError("HF down")
+
+ monkeypatch.setattr("hermes_cli.local_runtime.hf_browse.search_models", boom)
+ r = client.get("/api/local-models/search", params={"q": "qwen"})
+ assert r.status_code == 502
+
+
+def test_browsed_download_stages_and_bounces(client, tmp_path, monkeypatch):
+ """A browsed download must land in the machine-scoped models dir and
+ bounce the router — the seam that makes it a NORMAL model."""
+ body = b"GGUF" + b"\x00" * 60
+
+ class FakeResponse:
+ headers = {"Content-Length": str(len(body))}
+
+ def __init__(self):
+ self._data = body
+
+ def read(self, n=-1):
+ out, self._data = self._data, b""
+ return out
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ monkeypatch.setattr("urllib.request.urlopen",
+ lambda *a, **k: FakeResponse())
+ bounced = {}
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
+ lambda: bounced.setdefault("yes", True))
+
+ r = client.post("/api/local-models/download-browsed",
+ json={"repo": "someone/Some-GGUF",
+ "paths": ["Some-Model-Q4_K_M.gguf"]})
+ assert r.status_code == 200
+ job_id = r.json()["job_id"]
+
+ import time as _time
+
+ deadline = _time.time() + 10
+ status = None
+ while _time.time() < deadline:
+ status = client.get(f"/api/local-models/jobs/{job_id}").json()
+ if status["status"] in ("done", "error"):
+ break
+ _time.sleep(0.05)
+ assert status["status"] == "done", status.get("error")
+
+ from hermes_cli.local_runtime.bootstrap import models_dir
+
+ assert (models_dir() / "Some-Model-Q4_K_M.gguf").exists()
+ assert bounced.get("yes") is True
+
+
+def test_browsed_download_rejects_non_gguf(client):
+ r = client.post("/api/local-models/download-browsed",
+ json={"repo": "a/b", "paths": ["model.safetensors"]})
+ assert r.status_code == 422
+
+
+def test_sideload_links_and_bounces(client, tmp_path, monkeypatch):
+ src = tmp_path / "My-Local-Model-Q5_K_M.gguf"
+ src.write_bytes(b"GGUF" + b"\x00" * 32)
+ bounced = {}
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
+ lambda: bounced.setdefault("yes", True))
+
+ r = client.post("/api/local-models/sideload", json={"path": str(src)})
+ assert r.status_code == 200
+ assert r.json()["model_id"] == "My-Local-Model-Q5_K_M"
+
+ from hermes_cli.local_runtime.bootstrap import models_dir
+
+ dest = models_dir() / src.name
+ assert dest.exists()
+ assert bounced.get("yes") is True
+ # The original must be untouched.
+ assert src.exists()
+
+ # Idempotent: sideloading again short-circuits.
+ r = client.post("/api/local-models/sideload", json={"path": str(src)})
+ assert r.json().get("already_present") is True
+
+
+def test_sideload_rejects_non_gguf(client, tmp_path):
+ src = tmp_path / "model.bin"
+ src.write_bytes(b"nope")
+ r = client.post("/api/local-models/sideload", json={"path": str(src)})
+ assert r.status_code == 422
diff --git a/tests/hermes_cli/test_kanban_goal_mode.py b/tests/hermes_cli/test_kanban_goal_mode.py
index 61ece645ff..00616bfd7d 100644
--- a/tests/hermes_cli/test_kanban_goal_mode.py
+++ b/tests/hermes_cli/test_kanban_goal_mode.py
@@ -205,3 +205,21 @@ class TestCLIJudgeGate:
rc, complete_calls = self._run(monkeypatch, goal_mode=False)
assert rc == 0
assert complete_calls == ["t1"]
+
+ def test_judge_blocked_verdict_rejects_completion(self, monkeypatch, capsys):
+ """#100954: an unachievable goal must not complete silently.
+
+ The judge's ``blocked`` verdict is a refusal, not a completion —
+ ``complete_task`` must never run and stderr must steer the user
+ toward re-scoping / recording the block.
+ """
+ rc, complete_calls = self._run(
+ monkeypatch,
+ verdict="blocked",
+ reason="the target repository does not exist",
+ )
+ err = capsys.readouterr().err
+ assert rc != 0, "blocked verdict must reject the completion"
+ assert complete_calls == [], "an unachievable goal must never reach complete_task"
+ assert "unachievable" in err.lower()
+ assert "kanban block" in err.lower()
diff --git a/tests/hermes_cli/test_kanban_init_lock_bounded.py b/tests/hermes_cli/test_kanban_init_lock_bounded.py
index d7730712c6..38c5782713 100644
--- a/tests/hermes_cli/test_kanban_init_lock_bounded.py
+++ b/tests/hermes_cli/test_kanban_init_lock_bounded.py
@@ -66,7 +66,7 @@ def test_initialized_path_connect_skips_init_lock(kanban_home):
start = time.monotonic()
kb.connect().close()
elapsed = time.monotonic() - start
- assert elapsed < 1.0, f"fast-path connect blocked on the init lock ({elapsed:.2f}s)"
+ assert elapsed < 5.0, f"fast-path connect blocked on the init lock ({elapsed:.2f}s)"
finally:
release.set()
t.join(timeout=5)
@@ -85,7 +85,7 @@ def test_first_init_connect_is_bounded_when_lock_held(kanban_home, monkeypatch):
conn.close()
elapsed = time.monotonic() - start
# Proceeded within roughly the timeout window (not unbounded).
- assert 0.4 <= elapsed < 3.0, f"expected bounded ~0.6s acquire, got {elapsed:.2f}s"
+ assert 0.4 <= elapsed < 8.0, f"expected bounded ~0.6s acquire, got {elapsed:.2f}s"
assert str(db_path.resolve()) in kb._INITIALIZED_PATHS
finally:
release.set()
diff --git a/tests/hermes_cli/test_kanban_notify.py b/tests/hermes_cli/test_kanban_notify.py
index ec01f5a5d3..3eea0a1920 100644
--- a/tests/hermes_cli/test_kanban_notify.py
+++ b/tests/hermes_cli/test_kanban_notify.py
@@ -1113,6 +1113,34 @@ def test_gc_spares_reopened_task_even_when_old(kanban_home):
conn.close()
+def _set_task_status(kb, conn, tid, status):
+ """Force a task into ``status`` with a matching status event."""
+ with kb.write_txn(conn):
+ conn.execute("UPDATE tasks SET status = ? WHERE id = ?", (status, tid))
+ kb._append_event(conn, tid, "status", {"status": status})
+
+
+def test_gc_purges_blocked_task_that_never_done(kanban_home):
+ import hermes_cli.kanban_db as kb
+
+ conn = kb.connect()
+ try:
+ tid = kb.create_task(conn, title="stuck blocked", assignee="worker1")
+ kb.add_notify_sub(
+ conn, task_id=tid, platform="telegram", chat_id="c-blocked",
+ notifier_profile="default",
+ )
+ _set_task_status(kb, conn, tid, "blocked")
+ _backdate_task(kb, conn, tid, days=45)
+
+ purged = kb.purge_stale_done_notify_subs(conn, max_age_days=30)
+
+ assert purged == 1
+ assert kb.list_notify_subs(conn, tid) == []
+ finally:
+ conn.close()
+
+
def test_gc_archived_rows_already_removed_by_unsub(kanban_home):
import hermes_cli.kanban_db as kb
diff --git a/tests/hermes_cli/test_linux_desktop_entry.py b/tests/hermes_cli/test_linux_desktop_entry.py
index 8d8c246af5..acb372b2f8 100644
--- a/tests/hermes_cli/test_linux_desktop_entry.py
+++ b/tests/hermes_cli/test_linux_desktop_entry.py
@@ -2,8 +2,10 @@
from __future__ import annotations
+import io
import os
import stat
+import struct
import sys
from pathlib import Path
@@ -31,6 +33,26 @@ def _make_project(tmp_path: Path) -> Path:
return root
+def _png_ihdr(width: int, height: int) -> bytes:
+ """Minimal PNG prefix whose IHDR the installer can parse (no pixels)."""
+ return (
+ b"\x89PNG\r\n\x1a\n"
+ + b"\x00\x00\x00\r"
+ + b"IHDR"
+ + struct.pack(">II", width, height)
+ )
+
+
+def _stub_install(tmp_path, monkeypatch) -> None:
+ hermes_bin = tmp_path / "bin" / "hermes"
+ hermes_bin.parent.mkdir(exist_ok=True)
+ hermes_bin.write_text("", encoding="utf-8")
+ monkeypatch.setattr(
+ "hermes_cli.relaunch.resolve_hermes_bin", lambda: str(hermes_bin)
+ )
+ monkeypatch.setattr(lde, "refresh_desktop_databases", lambda _dir: [])
+
+
def _parse(entry_text: str) -> dict:
values = {}
for line in entry_text.splitlines():
@@ -94,8 +116,8 @@ def test_install_prefers_themed_icon_from_hicolor(tmp_path, xdg_home, monkeypatc
# And the icon really landed in the hicolor tree: the fixture icon is
# a fake PNG (no valid IHDR), so the size is unknown and the icon
- # lands under scalable/.
- dest = xdg_home / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png"
+ # lands under 256x256/ (indexed; never scalable, which is SVG-only).
+ dest = xdg_home / "icons" / "hicolor" / "256x256" / "apps" / "hermes.png"
assert dest.is_file()
assert dest.read_bytes() == lde.icon_path(root).read_bytes()
@@ -965,8 +987,8 @@ def test_probe_accepts_shell_launcher_wrapper(tmp_path, xdg_home, monkeypatch):
def test_install_icon_handles_truncated_png_header(tmp_path, xdg_home, monkeypatch):
"""A truncated PNG (valid signature + IHDR tag, <24 bytes) must not
- raise struct.error out of the fail-safe: it lands in scalable/ like
- any other unknown-size image."""
+ raise struct.error out of the fail-safe: it lands in 256x256/ like
+ any other unknown-size raster."""
root = _make_project(tmp_path)
icon = lde.icon_path(root)
icon.write_bytes(
@@ -984,5 +1006,96 @@ def test_install_icon_handles_truncated_png_header(tmp_path, xdg_home, monkeypat
values = _parse(entry.read_text(encoding="utf-8"))
assert values["Icon"] == "hermes"
- dest = xdg_home / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png"
+ dest = xdg_home / "icons" / "hicolor" / "256x256" / "apps" / "hermes.png"
assert dest.is_file()
+
+
+def test_hicolor_subdir_puts_rasters_in_indexed_dirs_never_scalable():
+ """Panel lookup uses fixed sizes. scalable/ is SVG-only."""
+ assert lde._hicolor_subdir(None) == "256x256"
+ assert lde._hicolor_subdir((1024, 1024)) == "256x256"
+ assert lde._hicolor_subdir((512, 512)) == "512x512"
+ assert lde._hicolor_subdir((256, 256)) == "256x256"
+ assert lde._hicolor_subdir((48, 48)) == "48x48"
+ assert lde._hicolor_subdir((24, 24)) == "24x24"
+ assert lde._hicolor_subdir((64, 32)) == "256x256"
+
+
+def test_install_places_1024_png_in_256x256_not_scalable(
+ tmp_path, xdg_home, monkeypatch
+):
+ """The shipped desktop asset is 1024×1024. A PNG in scalable/ is what
+ Cinnamon's panel rasterizes as a mangled low-res icon."""
+ root = _make_project(tmp_path)
+ lde.icon_path(root).write_bytes(_png_ihdr(1024, 1024))
+ _stub_install(tmp_path, monkeypatch)
+
+ entry = lde.install_desktop_entry(root)
+ values = _parse(entry.read_text(encoding="utf-8"))
+
+ dest = xdg_home / "icons" / "hicolor" / "256x256" / "apps" / "hermes.png"
+ stale = xdg_home / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png"
+ assert values["Icon"] == "hermes"
+ assert dest.is_file()
+ assert dest.read_bytes() == lde.icon_path(root).read_bytes()
+ assert not stale.exists()
+
+
+def test_install_removes_stale_scalable_png(tmp_path, xdg_home, monkeypatch):
+ """v2026.8.31 wrote the PNG into scalable/. A later hermes desktop
+ must delete that leftover so Cinnamon does not keep using it."""
+ root = _make_project(tmp_path)
+ lde.icon_path(root).write_bytes(_png_ihdr(1024, 1024))
+ _stub_install(tmp_path, monkeypatch)
+
+ stale = xdg_home / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png"
+ stale.parent.mkdir(parents=True)
+ stale.write_bytes(b"old scalable png")
+
+ lde.install_desktop_entry(root)
+
+ dest = xdg_home / "icons" / "hicolor" / "256x256" / "apps" / "hermes.png"
+ assert dest.is_file()
+ assert not stale.exists()
+
+
+def test_install_exact_48_png_uses_48x48_dir(tmp_path, xdg_home, monkeypatch):
+ root = _make_project(tmp_path)
+ lde.icon_path(root).write_bytes(_png_ihdr(48, 48))
+ _stub_install(tmp_path, monkeypatch)
+
+ lde.install_desktop_entry(root)
+
+ dest = xdg_home / "icons" / "hicolor" / "48x48" / "apps" / "hermes.png"
+ assert dest.is_file()
+ assert not (
+ xdg_home / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png"
+ ).exists()
+
+
+def test_install_resizes_decodable_png_to_panel_sizes(
+ tmp_path, xdg_home, monkeypatch
+):
+ """A decodeable PNG is Lanczos-resized so the 24px slot is actually 24px."""
+ from PIL import Image
+
+ root = _make_project(tmp_path)
+ im = Image.new("RGBA", (64, 64), (255, 255, 255, 255))
+ for x in range(16, 48):
+ for y in range(16, 48):
+ im.putpixel((x, y), (0, 0, 0, 255))
+ buf = io.BytesIO()
+ im.save(buf, format="PNG")
+ lde.icon_path(root).write_bytes(buf.getvalue())
+ _stub_install(tmp_path, monkeypatch)
+
+ lde.install_desktop_entry(root)
+
+ dest_24 = xdg_home / "icons" / "hicolor" / "24x24" / "apps" / "hermes.png"
+ dest_256 = xdg_home / "icons" / "hicolor" / "256x256" / "apps" / "hermes.png"
+ stale = xdg_home / "icons" / "hicolor" / "scalable" / "apps" / "hermes.png"
+ assert dest_24.is_file()
+ assert dest_256.is_file()
+ assert not stale.exists()
+ assert struct.unpack(">II", dest_24.read_bytes()[16:24]) == (24, 24)
+ assert struct.unpack(">II", dest_256.read_bytes()[16:24]) == (256, 256)
diff --git a/tests/hermes_cli/test_load_progress.py b/tests/hermes_cli/test_load_progress.py
new file mode 100644
index 0000000000..3da34a8a44
--- /dev/null
+++ b/tests/hermes_cli/test_load_progress.py
@@ -0,0 +1,260 @@
+"""Model-load progress: SSE events -> composite percent -> wait notices.
+
+The 40-second problem: a cold local model streams 16-21 GB of weights
+before the first token, and the chat rendered that as the generic
+"provider may be slow or overloaded" stall warning. llama-server's child
+emits real per-tensor progress which the router relays over /models/sse
+ONLY — these tests pin the consumer that turns that stream into the
+status route's `loading` field and the chat's load notice."""
+
+from __future__ import annotations
+
+import json
+import time
+
+import hermes_cli.local_runtime.load_progress as lp
+
+
+def setup_function(_fn):
+ with lp._lock:
+ lp._snapshot.clear()
+
+
+# ── composite percent ────────────────────────────────────────
+
+
+def test_composite_percent_text_stage_dominates():
+ stages = ["text_model", "spec_model", "mmproj_model"]
+ # Text model owns [0, 85): halfway through it reads ~42%.
+ assert lp._composite_percent(stages, "text_model", 0.5) == 42 # 0.5*85
+ # Extras start where text ends and never regress below it.
+ assert lp._composite_percent(stages, "spec_model", 0.0) == 85
+ assert lp._composite_percent(stages, "mmproj_model", 1.0) == 100
+
+
+def test_composite_percent_monotone_across_stage_walk():
+ """Walking the stages in llama-server's real order never moves the
+ bar backwards — the property that makes the bar trustworthy."""
+ stages = ["text_model", "spec_model", "mmproj_model"]
+ walk = [("text_model", v / 10) for v in range(11)] + \
+ [("spec_model", v / 10) for v in range(11)] + \
+ [("mmproj_model", v / 10) for v in range(11)]
+ seen = [lp._composite_percent(stages, s, v) for s, v in walk]
+ assert seen == sorted(seen)
+ assert seen[0] == 0 and seen[-1] == 100
+
+
+def test_composite_percent_single_stage_is_plain():
+ assert lp._composite_percent(["text_model"], "text_model", 0.4) == 40
+
+
+# ── event application ────────────────────────────────────────
+
+
+def _loading_event(value: float, current: str = "text_model") -> dict:
+ return {"status": "loading",
+ "progress": {"stages": ["text_model", "mmproj_model"],
+ "current": current, "value": value}}
+
+
+def test_loading_events_build_snapshot_and_terminal_clears():
+ lp._apply_event("m1", "status_change", _loading_event(0.5))
+ snap = lp.get_loading_progress()
+ assert "m1" in snap
+ assert snap["m1"]["percent"] == 42 # 0.5 * 85 within text stage
+ assert snap["m1"]["stage"] == "text_model"
+
+ lp._apply_event("m1", "status_change", {"status": "loaded", "info": {}})
+ assert lp.get_loading_progress() == {}
+
+
+def test_unload_and_failure_clear_too():
+ lp._apply_event("m1", "status_change", _loading_event(0.2))
+ lp._apply_event("m1", "status_change", {"status": "unloaded", "exit_code": 1})
+ assert lp.get_loading_progress() == {}
+
+ lp._apply_event("m2", "status_change", _loading_event(0.9))
+ lp._apply_event("m2", "model_remove", {})
+ assert lp.get_loading_progress() == {}
+
+
+def test_progressless_loading_event_keeps_entry_alive():
+ """The router's first model_status event says just {status: loading} —
+ it must register the load (indeterminate) without inventing a percent."""
+ lp._apply_event("m1", "model_status", {"status": "loading"})
+ snap = lp.get_loading_progress()
+ assert snap["m1"]["percent"] == 0
+
+
+def test_stale_entries_expire():
+ lp._apply_event("m1", "status_change", _loading_event(0.5))
+ with lp._lock:
+ lp._snapshot["m1"]["ts"] -= lp._STALE_ENTRY_TTL_S + 1
+ assert lp.get_loading_progress() == {}
+
+
+# ── chat wait-notice ─────────────────────────────────────────
+
+
+def test_load_notice_for_managed_model(tmp_path, monkeypatch):
+ from agent.chat_completion_helpers import _managed_local_load_notice
+
+ state = tmp_path / "server.json"
+ state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
+ "api_key": "k"}), encoding="utf-8")
+ monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
+ lambda: state)
+ lp._apply_event("Qwen-Test", "status_change", _loading_event(0.5))
+ monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
+
+ class _Agent:
+ base_url = "http://127.0.0.1:18434/v1"
+
+ notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
+ assert notice is not None
+ assert notice.startswith("⏳ loading Qwen-Test into memory — 42%")
+
+ # Different endpoint (user's own server): never claim its loads.
+ class _Other:
+ base_url = "http://127.0.0.1:9999/v1"
+
+ assert _managed_local_load_notice(_Other(), {"model": "Qwen-Test"}) is None
+ # Managed endpoint but a model that isn't loading: no notice.
+ assert _managed_local_load_notice(_Agent(), {"model": "Elsewhere"}) is None
+
+
+def test_load_notice_matches_desktop_wait_filter():
+ """The notices must pass the desktop's providerWaitText regex and parse
+ under parseModelLoadWait's shapes — pinned here as plain string
+ contracts so the two sides can't drift silently."""
+ import re
+
+ accept = r"^(?:⏳|⚠|↻|⚙)\s*(?:waiting on|loading|processing prompt|no (?:output|response)|model returned)"
+
+ load = "⏳ loading Qwen3.6-35B-A3B-UD-Q4_K_M into memory — 43% (responses start once the model is loaded)"
+ assert re.match(accept, load)
+ m = re.match(r"^⏳\s*loading\s+(.+?)\s+into memory\s+—\s+(\d{1,3})%", load)
+ assert m and m.group(1) == "Qwen3.6-35B-A3B-UD-Q4_K_M" and m.group(2) == "43"
+
+ prefill = "⚙ processing prompt — 31%"
+ assert re.match(accept, prefill)
+ p = re.match(r"^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?", prefill)
+ assert p and p.group(1) == "31"
+
+ bare = "⚙ processing prompt"
+ assert re.match(accept, bare)
+ b = re.match(r"^⚙\s*processing prompt(?:\s+—\s+(\d{1,3})%)?", bare)
+ assert b and b.group(1) is None
+
+
+# ── prefill progress ─────────────────────────────────────────
+
+
+def test_prefill_notice_for_managed_model(tmp_path, monkeypatch):
+ from agent.chat_completion_helpers import _managed_local_load_notice
+
+ state = tmp_path / "server.json"
+ state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
+ "api_key": "k"}), encoding="utf-8")
+ monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
+ lambda: state)
+ monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
+ # No load in flight; a prefill counter is live.
+ monkeypatch.setattr(lp, "get_prefill_progress",
+ lambda model: {"processed": 12288})
+ import agent.chat_completion_helpers as cch
+
+ monkeypatch.setattr(cch, "estimate_request_context_tokens",
+ lambda kw: 39551)
+
+ class _Agent:
+ base_url = "http://127.0.0.1:18434/v1"
+
+ notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
+ assert notice == "⚙ processing prompt — 31%"
+
+ # Counter past the estimate (estimator undercounted): no honest
+ # denominator, so no percent — never >100%.
+ monkeypatch.setattr(cch, "estimate_request_context_tokens", lambda kw: 100)
+ notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
+ assert notice == "⚙ processing prompt"
+
+
+def test_load_notice_outranks_prefill(tmp_path, monkeypatch):
+ """While a load entry exists the load notice wins — prefill can't start
+ before the model is resident, so a simultaneous claim means the load
+ snapshot is authoritative."""
+ from agent.chat_completion_helpers import _managed_local_load_notice
+
+ state = tmp_path / "server.json"
+ state.write_text(json.dumps({"base_url": "http://127.0.0.1:18434/v1",
+ "api_key": "k"}), encoding="utf-8")
+ monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
+ lambda: state)
+ monkeypatch.setattr(lp, "_ensure_watcher", lambda: None)
+ lp._apply_event("Qwen-Test", "status_change", _loading_event(0.5))
+ monkeypatch.setattr(lp, "get_prefill_progress",
+ lambda model: {"processed": 999})
+
+ class _Agent:
+ base_url = "http://127.0.0.1:18434/v1"
+
+ notice = _managed_local_load_notice(_Agent(), {"model": "Qwen-Test"})
+ assert notice is not None and notice.startswith("⏳ loading")
+
+
+def test_prefill_progress_reads_busiest_processing_slot(monkeypatch):
+ monkeypatch.setattr(lp, "_endpoint", lambda: ("http://127.0.0.1:1", "k"))
+
+ class _Resp:
+ def __init__(self, payload):
+ self._payload = payload
+
+ def read(self):
+ return json.dumps(self._payload).encode()
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ slots = [
+ {"id": 0, "is_processing": False, "n_prompt_tokens_processed": 500},
+ {"id": 1, "is_processing": True, "n_prompt_tokens_processed": 42},
+ {"id": 2, "is_processing": True, "n_prompt_tokens_processed": 32768},
+ ]
+ monkeypatch.setattr(lp.urllib.request, "urlopen",
+ lambda req, timeout=0: _Resp(slots))
+ assert lp.get_prefill_progress("m") == {"processed": 32768}
+
+ # Nothing processing -> None (idle slots' counters are leftovers).
+ idle = [{"id": 0, "is_processing": False, "n_prompt_tokens_processed": 500}]
+ monkeypatch.setattr(lp.urllib.request, "urlopen",
+ lambda req, timeout=0: _Resp(idle))
+ assert lp.get_prefill_progress("m") is None
+
+ # Unreachable server -> None, never an exception.
+ def _boom(req, timeout=0):
+ raise OSError("refused")
+
+ monkeypatch.setattr(lp.urllib.request, "urlopen", _boom)
+ assert lp.get_prefill_progress("m") is None
+
+
+def test_endpoint_respects_ownership_guard(monkeypatch):
+ """The watcher's endpoint MUST come from the ownership-guarded reader.
+ Regression: a raw state-file read attached the SSE watcher to a
+ foreign install's server on the shared stable port (health answers
+ for anyone; only the dead-pid check proves ownership)."""
+ import hermes_cli.local_runtime.load_progress as lp
+
+ # Guard says "not ours": no endpoint, regardless of state on disk.
+ monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint",
+ lambda: None)
+ assert lp._endpoint() is None
+
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.endpoint._state_endpoint",
+ lambda: {"base_url": "http://127.0.0.1:18434/v1", "api_key": "k"})
+ assert lp._endpoint() == ("http://127.0.0.1:18434", "k")
diff --git a/tests/hermes_cli/test_local_abandoned_requests.py b/tests/hermes_cli/test_local_abandoned_requests.py
new file mode 100644
index 0000000000..0f9d05260f
--- /dev/null
+++ b/tests/hermes_cli/test_local_abandoned_requests.py
@@ -0,0 +1,213 @@
+"""Abandoned-request lifecycle: work sent to the managed local server must
+die when its caller goes away, and teardown must never orphan VRAM.
+
+The incident this guards: auxiliary calls (title generation + retries)
+queued at the router behind a cold model load, their clients timed out
+and hung up, and the router then dispatched them anyway. Non-streamed
+responses write the socket only after the FULL generation, so nothing
+noticed the dead clients — two uncapped decodes ran at full GPU for the
+better part of an hour with nobody listening.
+
+Three contracts, one per failure link:
+1. Auxiliary requests to the managed local endpoint are always streamed
+ (a dead client then cancels decode at the first chunk write).
+2. Explicit caller max_tokens caps reach the managed local endpoint
+ (title generation's 64-token cap must not be silently dropped).
+3. Supervisor teardown terminates the whole process tree, and a router
+ respawn reaps orphaned model children first (each holds GiB of VRAM).
+"""
+
+from __future__ import annotations
+
+import json
+import subprocess
+import types
+
+import pytest
+
+import agent.auxiliary_client as aux
+from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
+
+
+MANAGED_URL = "http://127.0.0.1:18434/v1"
+
+
+@pytest.fixture
+def managed_state(tmp_path, monkeypatch):
+ """A supervisor state file declaring the managed endpoint, cache reset."""
+ state = tmp_path / "server.json"
+ state.write_text(json.dumps({"base_url": MANAGED_URL, "api_key": "k",
+ "pid": 4242}), encoding="utf-8")
+ monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
+ lambda: state)
+ monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
+ return state
+
+
+# ── 1. managed endpoint is always streamed ───────────────────
+
+
+def test_managed_endpoint_requires_stream(managed_state):
+ assert aux._provider_requires_stream("custom", MANAGED_URL) is True
+
+
+def test_managed_detection_matches_netloc_not_substring(managed_state):
+ # Same host, different port: a user's own external server — untouched.
+ assert aux._provider_requires_stream("custom", "http://127.0.0.1:8080/v1") is False
+
+
+def test_no_state_file_means_no_managed_endpoint(tmp_path, monkeypatch):
+ monkeypatch.setattr("hermes_cli.local_runtime.supervisor.state_path",
+ lambda: tmp_path / "absent.json")
+ monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
+ assert aux._is_managed_local_endpoint(MANAGED_URL) is False
+
+
+def test_remote_providers_unaffected(managed_state):
+ assert aux._provider_requires_stream("nous",
+ "https://inference-api.nousresearch.com/v1/") is False
+
+
+# ── 2. explicit caps reach the managed endpoint ──────────────
+
+
+def test_explicit_max_tokens_forwarded_to_managed_local(managed_state, monkeypatch):
+ monkeypatch.setattr(aux, "_current_custom_base_url", lambda: MANAGED_URL)
+ kwargs = aux._build_call_kwargs(
+ "custom", "Qwen-Local", [{"role": "user", "content": "hi"}],
+ max_tokens=64, timeout=30.0, task="title_generation")
+ assert kwargs.get("max_tokens") == 64 or kwargs.get("max_completion_tokens") == 64, (
+ "explicit caller cap dropped on the managed local endpoint — an "
+ "EOS-less generation then runs to the full context window")
+
+
+def test_no_default_cap_policy_unchanged_for_remote(monkeypatch):
+ # A generic remote provider still drops the cap (the forwarding gate
+ # is an allow-list). openrouter no longer qualifies as the example
+ # here: main forwards its caps deliberately (#41035, 402 affordability).
+ monkeypatch.setattr(aux, "_managed_local_cache", (0.0, ""))
+ kwargs = aux._build_call_kwargs(
+ "openai", "some/model", [{"role": "user", "content": "hi"}],
+ max_tokens=64, timeout=30.0)
+ assert "max_tokens" not in kwargs and "max_completion_tokens" not in kwargs
+
+
+# ── 3. teardown kills the tree; respawn reaps orphans ────────
+
+
+class _FakeChild:
+ def __init__(self, pid):
+ self.pid = pid
+ self.terminated = False
+ self.killed = False
+
+ def terminate(self):
+ self.terminated = True
+
+ def is_running(self):
+ return not self.terminated and not self.killed
+
+ def kill(self):
+ self.killed = True
+
+
+def test_terminate_tree_terminates_children_too(monkeypatch):
+ children = [_FakeChild(101), _FakeChild(102)]
+
+ class _FakeParentProc:
+ def __init__(self, pid):
+ self.pid = pid
+
+ def children(self, recursive=False):
+ assert recursive is True
+ return children
+
+ fake_psutil = types.SimpleNamespace(Process=_FakeParentProc)
+ monkeypatch.setitem(__import__("sys").modules, "psutil", fake_psutil)
+
+ class _FakeRouter:
+ pid = 4242
+ terminated = False
+
+ def terminate(self):
+ _FakeRouter.terminated = True
+
+ def wait(self, timeout=None):
+ return 0
+
+ def poll(self):
+ return None
+
+ LlamaServerSupervisor._terminate_tree(_FakeRouter())
+ assert _FakeRouter.terminated
+ assert all(c.terminated for c in children), (
+ "router children orphaned on stop — each holds GiB of VRAM")
+
+
+def test_terminate_tree_survives_missing_psutil(monkeypatch):
+ import builtins
+
+ real_import = builtins.__import__
+
+ def _no_psutil(name, *a, **k):
+ if name == "psutil":
+ raise ImportError("nope")
+ return real_import(name, *a, **k)
+
+ monkeypatch.setattr(builtins, "__import__", _no_psutil)
+
+ class _FakeRouter:
+ pid = 4242
+ terminated = False
+
+ def terminate(self):
+ _FakeRouter.terminated = True
+
+ def wait(self, timeout=None):
+ return 0
+
+ LlamaServerSupervisor._terminate_tree(_FakeRouter())
+ assert _FakeRouter.terminated # router still stopped without psutil
+
+
+def test_reap_orphans_kills_only_our_parentless_binaries(tmp_path, monkeypatch):
+ exe = tmp_path / "llama-server.exe"
+ exe.write_text("")
+
+ orphan = _FakeChild(300)
+ adopted = _FakeChild(301) # parent alive -> not an orphan
+ foreign = _FakeChild(302) # different binary -> never touched
+
+ def _info(pid, exe_path, ppid):
+ p = _FakeChild(pid)
+ p.info = {"exe": exe_path, "ppid": ppid}
+ return p
+
+ procs = [
+ _info(300, str(exe), 9999), # dead parent -> reap
+ _info(301, str(exe), 1), # live parent -> keep
+ _info(302, str(tmp_path / "other.exe"), 9999), # foreign -> keep
+ ]
+ reaped = []
+ for p in procs:
+ p.kill = lambda p=p: reaped.append(p.info and p.pid)
+
+ class _NoSuch(Exception):
+ pass
+
+ fake_psutil = types.SimpleNamespace(
+ process_iter=lambda attrs: procs,
+ pid_exists=lambda pid: pid == 1,
+ NoSuchProcess=_NoSuch,
+ AccessDenied=_NoSuch,
+ )
+ monkeypatch.setitem(__import__("sys").modules, "psutil", fake_psutil)
+ monkeypatch.setattr("hermes_cli.local_runtime.supervisor.server_binary",
+ lambda install_dir: exe)
+
+ sup = LlamaServerSupervisor.__new__(LlamaServerSupervisor)
+ sup.install_dir = tmp_path
+ sup.proc = None
+ sup._reap_orphaned_children()
+
+ assert reaped == [300], f"reaped {reaped}; wanted only the orphan (300)"
diff --git a/tests/hermes_cli/test_local_context_resolution.py b/tests/hermes_cli/test_local_context_resolution.py
new file mode 100644
index 0000000000..26e3c28e91
--- /dev/null
+++ b/tests/hermes_cli/test_local_context_resolution.py
@@ -0,0 +1,120 @@
+"""Context-length resolution for the managed llama.cpp router.
+
+The incident: the statusbar showed 131K for a local model the server had
+granted 262144 tokens. The router reports ``meta: null`` on /v1/models
+for a model that is not currently LOADED (models autoload on first chat,
+so at session start the model is routinely unloaded), and /v1/models/{id}
+404s — every metadata probe missed, resolution fell through to the
+name-pattern defaults, and the "qwen" family catch-all (131072) shipped
+as the compressor's budget and the statusbar's denominator.
+
+Contract: for a llama.cpp server, /props default_generation_settings.n_ctx
+(the preset-backed RUNTIME window, served even for unloaded models) is
+the authority, probed before the /v1/models fallbacks.
+"""
+
+from __future__ import annotations
+
+import http.server
+import json
+import threading
+
+import pytest
+
+import agent.model_metadata as mm
+
+
+GRANTED = 262144
+
+
+@pytest.fixture
+def router():
+ """Stub of the llama-server router with the model UNLOADED:
+ /v1/models carries meta=null; /props answers from the preset."""
+
+ class _Router(http.server.BaseHTTPRequestHandler):
+ def do_GET(self):
+ if self.path.startswith("/props"):
+ body = {"default_generation_settings": {"n_ctx": GRANTED}}
+ elif self.path == "/v1/models":
+ body = {"data": [{
+ "id": "Qwen-Test-UD-Q4_K_M",
+ "owned_by": "llamacpp",
+ "meta": None,
+ "status": {"value": "unloaded"},
+ }]}
+ else: # /v1/models/{id} -> 404, as the real router answers
+ self.send_response(404)
+ self.send_header("Content-Length", "0")
+ self.end_headers()
+ return
+ raw = json.dumps(body).encode()
+ self.send_response(200)
+ self.send_header("Content-Type", "application/json")
+ self.send_header("Content-Length", str(len(raw)))
+ self.end_headers()
+ self.wfile.write(raw)
+
+ def log_message(self, *a):
+ pass
+
+ server = http.server.HTTPServer(("127.0.0.1", 0), _Router)
+ threading.Thread(target=server.serve_forever, daemon=True).start()
+ yield f"http://127.0.0.1:{server.server_address[1]}/v1"
+ server.shutdown()
+
+
+def test_unloaded_llamacpp_model_resolves_granted_window(router, monkeypatch):
+ monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "llamacpp")
+ monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
+ ctx = mm._query_local_context_length_uncached("Qwen-Test-UD-Q4_K_M", router)
+ assert ctx == GRANTED, (
+ f"resolved {ctx}; an unloaded model must resolve the preset window "
+ "from /props, not fall through to name-pattern catch-alls")
+
+
+def test_props_beats_meta_when_model_loaded(router, monkeypatch):
+ """/props is probed first even when /v1/models would answer: n_ctx from
+ /props is the same runtime value, and probing it first keeps loaded and
+ unloaded models on one code path."""
+ monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "llamacpp")
+ monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
+ ctx = mm._query_local_context_length_uncached("Qwen-Test-UD-Q4_K_M", router)
+ assert ctx == GRANTED
+
+
+def test_non_llamacpp_servers_skip_props(monkeypatch):
+ """Ollama/LM Studio/vLLM keep their existing probe order — /props is
+ llama.cpp-shaped and must not be consulted for other server types."""
+ calls = []
+
+ class _FakeResp:
+ status_code = 404
+
+ def json(self):
+ return {}
+
+ class _FakeClient:
+ def __init__(self, *a, **k):
+ pass
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ def get(self, url):
+ calls.append(url)
+ return _FakeResp()
+
+ def post(self, url, **k):
+ calls.append(url)
+ return _FakeResp()
+
+ import httpx
+ monkeypatch.setattr(httpx, "Client", _FakeClient)
+ monkeypatch.setattr(mm, "detect_local_server_type", lambda *a, **k: "vllm")
+ monkeypatch.setattr(mm, "_endpoint_blackholed", lambda *a, **k: False)
+ mm._query_local_context_length_uncached("m", "http://127.0.0.1:9999/v1")
+ assert not any("/props" in u for u in calls)
diff --git a/tests/hermes_cli/test_local_growth.py b/tests/hermes_cli/test_local_growth.py
new file mode 100644
index 0000000000..6367668aee
--- /dev/null
+++ b/tests/hermes_cli/test_local_growth.py
@@ -0,0 +1,280 @@
+"""In-session growth contracts (growth.py + the presets override seam).
+
+The live half of the window ladder: grow before compress, overrides
+persist across boots, physics re-checked every boot, growth state dies
+with the model."""
+
+from __future__ import annotations
+
+import pytest
+
+
+@pytest.fixture
+def hermes_home(tmp_path, monkeypatch):
+ home = tmp_path / ".hermes"
+ home.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ return home
+
+
+def test_overrides_roundtrip_and_clear(hermes_home):
+ from hermes_cli.local_runtime.growth import (
+ clear_window_override,
+ load_window_overrides,
+ save_window_override,
+ )
+
+ assert load_window_overrides() == {}
+ save_window_override("model-a", 98304)
+ save_window_override("model-b", 262144)
+ assert load_window_overrides() == {"model-a": 98304, "model-b": 262144}
+ clear_window_override("model-a")
+ assert load_window_overrides() == {"model-b": 262144}
+ # Clearing a missing key is a no-op, not an error.
+ clear_window_override("never-existed")
+
+
+def test_corrupt_overrides_read_as_empty(hermes_home):
+ from hermes_cli.local_runtime.growth import (
+ load_window_overrides,
+ window_overrides_path,
+ )
+
+ path = window_overrides_path()
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_text("{not json", encoding="utf-8")
+ assert load_window_overrides() == {}
+
+
+def test_growth_declines_foreign_endpoints(hermes_home):
+ """Only the server THIS process supervises grows — a detected external
+ server or another process's endpoint returns None untouched."""
+ from hermes_cli.local_runtime.growth import maybe_grow_window
+
+ grown = maybe_grow_window(
+ "some-model", base_url="http://127.0.0.1:9999/v1",
+ session_tokens=100_000, current_window=65536)
+ assert grown is None
+
+
+def test_occupancy_confirmed_skips_gate_one():
+ """The agent's compression gate IS the occupancy signal: when it fired,
+ growth must not re-derive its own edge and hold. Decision-table check
+ with a synthetic profile."""
+ from hermes_cli.local_runtime.context_policy import growth_decision
+ from hermes_cli.local_runtime.estimator import (
+ HardwareBudget,
+ LayerKind,
+ ModelProfile,
+ )
+
+ gib = 1 << 30
+ profile = ModelProfile(
+ name="m", weights_bytes=2 * gib, embd_table_bytes=0,
+ n_ctx_train=262144,
+ layers=[(LayerKind.FULL, 4096)] * 16 + [(LayerKind.RECURRENT, 0)] * 48)
+ budget = HardwareBudget(usable_vram_bytes=26 * gib,
+ total_device_bytes=32 * gib,
+ ram_available_bytes=64 * gib)
+
+ # Hermes' threshold (e.g. 80% of window) can sit BELOW the ladder's 85%
+ # occupancy gate: 78K of a 96K window is 81%.
+ kwargs = dict(current_window=98304, session_tokens=78_000,
+ measured_decode_tok_s=None, server_idle=True)
+ ungated = growth_decision(profile, budget, **kwargs)
+ assert ungated.action == "hold", "sanity: below the ladder's own gate"
+
+ confirmed = growth_decision(profile, budget, occupancy_confirmed=True, **kwargs)
+ assert confirmed.action == "grow"
+ assert confirmed.next_window and confirmed.next_window > 98304
+
+
+def _stage_fake_gguf(mdir, name):
+ mdir.mkdir(parents=True, exist_ok=True)
+ (mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 64)
+
+
+def _header_stub(sampling: dict | None = None):
+ """A read_gguf_header stand-in for tests that monkeypatch the reader:
+ just enough surface for preset generation (sampling ladder included)."""
+
+ class _Stub:
+ sampling_defaults = dict(sampling or {})
+
+ return _Stub()
+
+
+def _tiny_profile(model_id: str):
+ from hermes_cli.local_runtime.estimator import LayerKind, ModelProfile
+
+ gib = 1 << 30
+ return ModelProfile(
+ name=model_id, weights_bytes=2 * gib, embd_table_bytes=0,
+ n_ctx_train=131072,
+ layers=[(LayerKind.FULL, 512)] * 4)
+
+
+def test_preset_generation_for_catalog_model_with_mmproj(hermes_home, tmp_path, monkeypatch):
+ """generate_presets must survive a model that IS in the catalog and
+ carries a vision projector — this executes the find_entry_for_model +
+ mmproj overhead branch that synthetic test models skip. Regression:
+ the branch once treated the (entry, variant) tuple as the entry and
+ crashed every real boot into the stock-fit fallback."""
+ import hermes_cli.local_runtime.presets as presets_mod
+
+ from hermes_cli.local_runtime.catalog import CATALOG
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+
+ # A real catalog id with an mmproj (the recommended row has one).
+ entry = next(e for e in CATALOG if e.mmproj is not None)
+ variant = entry.variants[-1]
+ mdir = tmp_path / "models"
+ _stage_fake_gguf(mdir, variant.model_id)
+
+ monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
+ monkeypatch.setattr(presets_mod, "profile_from_gguf",
+ lambda h: _tiny_profile(variant.model_id))
+
+ gib = 1 << 30
+ budget = HardwareBudget(usable_vram_bytes=24 * gib,
+ total_device_bytes=24 * gib,
+ ram_available_bytes=64 * gib)
+ entries = presets_mod.generate_presets(mdir, budget, tmp_path / "p.ini")
+ assert len(entries) == 1
+ assert entries[0].refusal is None
+ assert entries[0].window > 0
+
+
+def test_preset_restores_grown_window_capped_at_native(hermes_home, tmp_path, monkeypatch):
+ """A persisted override lifts the preset window; an absurd override is
+ capped at native. GGUF parsing is stubbed — the contract under test is
+ the override plumbing, not the reader."""
+ import hermes_cli.local_runtime.presets as presets_mod
+
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+ from hermes_cli.local_runtime.growth import save_window_override
+
+ mdir = tmp_path / "models"
+ _stage_fake_gguf(mdir, "tiny-dense")
+ monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
+ monkeypatch.setattr(presets_mod, "profile_from_gguf",
+ lambda h: _tiny_profile("tiny-dense"))
+
+ gib = 1 << 30
+ budget = HardwareBudget(usable_vram_bytes=24 * gib,
+ total_device_bytes=24 * gib,
+ ram_available_bytes=64 * gib)
+ preset = tmp_path / "presets.ini"
+
+ baseline = presets_mod.generate_presets(mdir, budget, preset)[0]
+ assert baseline.window == 131072 # tiny model: native from the start
+
+ # Override above native must cap at native, not exceed it.
+ save_window_override("tiny-dense", 10_000_000)
+ capped = presets_mod.generate_presets(mdir, budget, preset)[0]
+ assert capped.window == 131072
+
+
+def test_preset_ignores_override_below_launch_window(hermes_home, tmp_path, monkeypatch):
+ """Overrides only ever RAISE the window (growth is monotone); a stale
+ smaller override never shrinks a launch decision."""
+ import hermes_cli.local_runtime.presets as presets_mod
+
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+ from hermes_cli.local_runtime.growth import save_window_override
+
+ mdir = tmp_path / "models"
+ _stage_fake_gguf(mdir, "tiny-dense")
+ monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
+ monkeypatch.setattr(presets_mod, "profile_from_gguf",
+ lambda h: _tiny_profile("tiny-dense"))
+ save_window_override("tiny-dense", 65536)
+
+ gib = 1 << 30
+ budget = HardwareBudget(usable_vram_bytes=24 * gib,
+ total_device_bytes=24 * gib,
+ ram_available_bytes=64 * gib)
+ entry = presets_mod.generate_presets(mdir, budget, tmp_path / "p.ini")[0]
+ assert entry.window == 131072
+
+
+def test_preset_restores_grown_window_midladder(hermes_home, tmp_path, monkeypatch):
+ """The real growth shape: launch at a lower rung, override to a middle
+ rung -> the preset window follows the override."""
+ import hermes_cli.local_runtime.presets as presets_mod
+
+ from hermes_cli.local_runtime.estimator import HardwareBudget, LayerKind, ModelProfile
+ from hermes_cli.local_runtime.growth import save_window_override
+
+ gib = 1 << 30
+ # Expensive dense KV so the launch decision lands BELOW native on this
+ # budget: 60 layers x 4 KiB/tok f16 -> q8 ~= 120 KiB/tok.
+ profile = ModelProfile(
+ name="big-dense", weights_bytes=20 * gib, embd_table_bytes=0,
+ n_ctx_train=262144,
+ layers=[(LayerKind.FULL, 4096)] * 60)
+ mdir = tmp_path / "models"
+ _stage_fake_gguf(mdir, "big-dense")
+ monkeypatch.setattr(presets_mod, "read_gguf_header", lambda p: _header_stub())
+ monkeypatch.setattr(presets_mod, "profile_from_gguf", lambda h: profile)
+
+ budget = HardwareBudget(usable_vram_bytes=28 * gib,
+ total_device_bytes=32 * gib,
+ ram_available_bytes=128 * gib)
+ baseline = presets_mod.generate_presets(mdir, budget, tmp_path / "a.ini")[0]
+ assert baseline.window < 262144, "sanity: launch below native"
+
+ grown = baseline.window * 2
+ save_window_override("big-dense", grown)
+ restored = presets_mod.generate_presets(mdir, budget, tmp_path / "b.ini")[0]
+ assert restored.window >= grown, "override must lift the launch window"
+
+
+def test_sampling_ladder_file_beats_catalog_beats_nothing(hermes_home, tmp_path, monkeypatch):
+ """The sampling deference ladder: the GGUF's own general.sampling.*
+ wins per key, catalog fills only what the file left silent, and a
+ model with neither gets no sampling keys at all (llama.cpp defaults).
+ Policy keys (ctx-size, cache types) must never be displaced."""
+ import configparser
+
+ import hermes_cli.local_runtime.presets as presets_mod
+ from hermes_cli.local_runtime.catalog import CATALOG
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+
+ # A real catalog entry WITH catalog sampling, staged on disk.
+ entry = next(e for e in CATALOG if e.sampling)
+ variant = entry.variants[-1]
+ mdir = tmp_path / "models"
+ _stage_fake_gguf(mdir, variant.model_id)
+ _stage_fake_gguf(mdir, "off-catalog-model")
+
+ gib = 1 << 30
+ budget = HardwareBudget(usable_vram_bytes=64 * gib,
+ total_device_bytes=64 * gib,
+ ram_available_bytes=64 * gib)
+ # The catalog model's file carries temp; catalog must fill the rest
+ # but NOT displace the file's value. The off-catalog file carries none.
+ def fake_header(path):
+ if variant.model_id in str(path):
+ return _header_stub({"temp": "0.42"})
+ return _header_stub()
+
+ monkeypatch.setattr(presets_mod, "read_gguf_header", fake_header)
+ monkeypatch.setattr(presets_mod, "profile_from_gguf",
+ lambda h: _tiny_profile("x"))
+
+ out = tmp_path / "presets.ini"
+ presets_mod.generate_presets(mdir, budget, out)
+ ini = configparser.ConfigParser()
+ ini.read(out)
+
+ sec = ini[variant.model_id]
+ assert sec["temp"] == "0.42", "file's own sampling must win per key"
+ for k, v in entry.sampling.items():
+ if k != "temp":
+ assert sec[k] == v, f"catalog must fill the silent key {k}"
+ assert "ctx-size" in sec, "policy keys survive the ladder"
+
+ off = ini["off-catalog-model"]
+ assert "temp" not in off and "top-p" not in off, (
+ "no file keys + no catalog entry = llama.cpp defaults, not ours")
diff --git a/tests/hermes_cli/test_local_models_routes.py b/tests/hermes_cli/test_local_models_routes.py
new file mode 100644
index 0000000000..6ee98aae1e
--- /dev/null
+++ b/tests/hermes_cli/test_local_models_routes.py
@@ -0,0 +1,293 @@
+"""Contract tests for the local-models dashboard routes (Rollout 4).
+
+Real FastAPI TestClient against the real router; the runtime pieces
+underneath are exercised against temp HERMES_HOME (autouse fixture). Network
+downloads are stubbed at the urllib boundary — never live."""
+
+from __future__ import annotations
+
+import io
+import json
+import time
+from pathlib import Path
+
+import pytest
+from fastapi.testclient import TestClient
+
+
+@pytest.fixture
+def client(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli import web_server
+
+ test_client = TestClient(web_server.app)
+ # Same auth pattern as the git-route tests: present the session token.
+ test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
+ return test_client
+
+
+def test_local_models_routes_require_auth(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli import web_server
+
+ unauth = TestClient(web_server.app)
+ assert unauth.get("/api/local-models/status").status_code == 401
+
+
+def _write_fake_gguf(path: Path, size: int = 1024) -> None:
+ path.parent.mkdir(parents=True, exist_ok=True)
+ path.write_bytes(b"GGUF" + b"\x00" * size)
+
+
+# ── status ───────────────────────────────────────────────────
+
+
+def test_status_shape_and_defaults(client):
+ r = client.get("/api/local-models/status")
+ assert r.status_code == 200
+ data = r.json()
+ # Contract: every key the pane's first paint needs, present and typed.
+ assert isinstance(data["enabled"], bool)
+ assert isinstance(data["tag"], str) and data["tag"].startswith("b")
+ assert isinstance(data["runtime_installed"], bool)
+ assert isinstance(data["server_running"], bool)
+ assert isinstance(data["models"], list)
+
+
+def test_status_lists_staged_models_with_labels(client, tmp_path):
+ from hermes_cli.local_runtime.bootstrap import models_dir
+
+ _write_fake_gguf(models_dir() / "Some-Model.gguf", size=2048)
+ data = client.get("/api/local-models/status").json()
+ ids = [m["id"] for m in data["models"]]
+ assert "Some-Model" in ids
+ row = data["models"][ids.index("Some-Model")]
+ assert row["size_bytes"] > 0
+ assert row["size_label"].endswith("GB")
+
+
+# ── hardware ─────────────────────────────────────────────────
+
+
+def test_hardware_plain_facts(client):
+ data = client.get("/api/local-models/hardware").json()
+ assert isinstance(data["uma"], bool)
+ assert data["ram_total_bytes"] > 0
+ assert data["vram_total_bytes"] >= 0
+ # GPU fields are None-able (non-NVIDIA machines) but must exist.
+ assert "gpu_name" in data and "gpu_util_percent" in data and "vram_used_bytes" in data
+
+
+# ── catalog ──────────────────────────────────────────────────
+
+
+def test_catalog_prices_every_entry_for_this_machine(client):
+ data = client.get("/api/local-models/catalog").json()
+ assert len(data["models"]) >= 3
+ for row in data["models"]:
+ # The three user questions, answered on every row:
+ assert row["size_label"].endswith("GB") # how big
+ assert isinstance(row["fits"], bool) # will it fit
+ assert row["fit_summary"] # what shape
+ if row["fits"]:
+ assert row["start_window"] >= 1
+ assert row["start_window_label"].endswith("K")
+ else:
+ assert "memory" in row["fit_summary"].lower()
+ assert isinstance(row["downloaded"], bool)
+
+
+def test_catalog_never_hides_unaffordable_models(client, monkeypatch):
+ """Unaffordable entries stay visible with a plain reason — hiding them
+ is how users conclude the feature is broken."""
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+
+ tiny = HardwareBudget(usable_vram_bytes=1 << 30, total_device_bytes=1 << 30,
+ ram_available_bytes=1 << 30)
+ monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
+ lambda **kw: tiny)
+ data = client.get("/api/local-models/catalog").json()
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ assert len(data["models"]) == len(CATALOG)
+ refused = [m for m in data["models"] if not m["fits"]]
+ assert refused, "a 1 GiB machine must refuse the 20 GB models"
+ for row in refused:
+ assert row["fit_detail"] or row["fit_summary"]
+
+
+# ── downloads ────────────────────────────────────────────────
+
+
+def test_download_unknown_model_404s(client):
+ r = client.post("/api/local-models/download", json={"model_id": "nope"})
+ assert r.status_code == 404
+
+
+def test_download_short_of_server_length_errors_and_cleans_up(client, monkeypatch):
+ """Catalog sizes are advisory (upstream re-uploads may make them
+ stale — a mismatch against the CATALOG must not fail a download).
+ The server's own declared length is the only completeness check:
+ fewer bytes than the server promised means a dropped connection, so
+ the job errors and nothing is staged."""
+
+ class FakeResponse(io.BytesIO):
+ # Body is 17 bytes; the server promises 32 — a truncated stream.
+ headers = {"Content-Length": "32"}
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ monkeypatch.setattr("urllib.request.urlopen",
+ lambda *a, **k: FakeResponse(b"not the real body"))
+
+ # Pin a generous budget: variant selection prices against the machine
+ # running the test, and a GPU-less CI runner honestly refuses every
+ # build (409) — this test is about the download path, not selection.
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+
+ budget = HardwareBudget(usable_vram_bytes=64 << 30,
+ total_device_bytes=64 << 30,
+ ram_available_bytes=64 << 30)
+ monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
+ lambda **kw: budget)
+
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ entry_id = CATALOG[0].id
+ r = client.post("/api/local-models/download", json={"model_id": entry_id})
+ assert r.status_code == 200
+ job_id = r.json()["job_id"]
+ assert job_id
+
+ deadline = time.time() + 10
+ status = None
+ while time.time() < deadline:
+ status = client.get(f"/api/local-models/jobs/{job_id}").json()
+ if status["status"] in ("done", "error"):
+ break
+ time.sleep(0.05)
+ assert status is not None and status["status"] == "error"
+ assert "bytes" in status["error"].lower()
+
+ from hermes_cli.local_runtime.bootstrap import models_dir
+
+ assert not (models_dir() / f"{entry_id}.gguf").exists()
+ assert not (models_dir() / f"{entry_id}.part").exists()
+
+
+def test_download_already_downloaded_short_circuits(client, monkeypatch):
+ from hermes_cli.local_runtime.bootstrap import models_dir
+ from hermes_cli.local_runtime.catalog import CATALOG, select_variant
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+
+ # Pin the budget so the selected variant is deterministic in the test.
+ budget = HardwareBudget(usable_vram_bytes=64 << 30, total_device_bytes=64 << 30,
+ ram_available_bytes=64 << 30)
+ monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
+ lambda **kw: budget)
+ choice = select_variant(CATALOG[0], budget)
+ assert choice is not None
+ _write_fake_gguf(models_dir() / choice.variant.files[0].local_name)
+ r = client.post("/api/local-models/download", json={"model_id": CATALOG[0].id})
+ assert r.status_code == 200
+ assert r.json()["already_downloaded"] is True
+
+
+def test_delete_model(client):
+ from hermes_cli.local_runtime.bootstrap import models_dir
+
+ _write_fake_gguf(models_dir() / "Doomed.gguf")
+ assert client.delete("/api/local-models/models/Doomed").status_code == 200
+ assert not (models_dir() / "Doomed.gguf").exists()
+ assert client.delete("/api/local-models/models/Doomed").status_code == 404
+
+
+# ── runtime install ──────────────────────────────────────────
+
+
+def test_runtime_install_rejects_impossible_combo(client, monkeypatch):
+ """Impossible platform/backend combos fail the POST itself with the
+ resolver's honest message — not a background job that dies silently.
+ (win-arm64-vulkan; the old cuda case became real upstream at ~b1036x.)"""
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries._host_os_arch", lambda: ("win", "arm64"))
+ r = client.post("/api/local-models/runtime/install", json={"backend": "vulkan"})
+ assert r.status_code == 400
+ assert "arm64" in r.json()["detail"]
+
+
+def test_job_poll_unknown_404s(client):
+ assert client.get("/api/local-models/jobs/deadbeef").status_code == 404
+
+
+def test_eject_without_supervisor_is_not_a_500(client, monkeypatch):
+ """Eject on an ADOPTED server (no in-process supervisor — the shape
+ every backend restart produces, since boot adopts the running server
+ via the state file) must route through the persisted endpoint, not
+ crash. Regression: _state_endpoint was only imported inside the
+ status route, so eject raised NameError -> 500 for every adopted-
+ server session."""
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.bootstrap.get_supervisor", lambda: None)
+ # No running server either: the route must answer 409 (no server),
+ # never a NameError 500.
+ monkeypatch.setattr(
+ "hermes_cli.web_routers.local_models._state_endpoint", lambda: None)
+ r = client.post("/api/local-models/eject", json={"model_id": "anything"})
+ assert r.status_code == 409, (r.status_code, r.text)
+
+
+def test_download_tolerates_stale_catalog_size(client, monkeypatch):
+ """Upstream re-uploads make catalog sizes stale; a download whose
+ delivered bytes are self-consistent with the SERVER's declared length
+ must succeed even when the catalog said something else. (This is the
+ tolerance the sha removal was for — being out of date must not break
+ downloads.)"""
+
+ body = b"x" * 48 # server-consistent: Content-Length == body length
+
+ class FakeResponse(io.BytesIO):
+ headers = {"Content-Length": str(len(body))}
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ monkeypatch.setattr("urllib.request.urlopen",
+ lambda *a, **k: FakeResponse(body))
+
+ from hermes_cli.local_runtime.estimator import HardwareBudget
+
+ budget = HardwareBudget(usable_vram_bytes=64 << 30,
+ total_device_bytes=64 << 30,
+ ram_available_bytes=64 << 30)
+ monkeypatch.setattr("hermes_cli.local_runtime.hardware.probe_budget",
+ lambda **kw: budget)
+ # Keep the post-download server bounce out of this unit.
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.bootstrap.refresh_local_runtime",
+ lambda: False)
+
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ # Catalog size for this entry is in the tens of GB — wildly stale
+ # versus our 48-byte body. The download must still land.
+ entry_id = CATALOG[0].id
+ r = client.post("/api/local-models/download", json={"model_id": entry_id})
+ assert r.status_code == 200
+ job_id = r.json()["job_id"]
+
+ deadline = time.time() + 10
+ status = None
+ while time.time() < deadline:
+ status = client.get(f"/api/local-models/jobs/{job_id}").json()
+ if status["status"] in ("done", "error"):
+ break
+ time.sleep(0.05)
+ assert status is not None and status["status"] == "done", status.get("error")
diff --git a/tests/hermes_cli/test_local_picker_identity.py b/tests/hermes_cli/test_local_picker_identity.py
new file mode 100644
index 0000000000..c83067a7cb
--- /dev/null
+++ b/tests/hermes_cli/test_local_picker_identity.py
@@ -0,0 +1,74 @@
+"""The managed local server owns its picker identity.
+
+A live session on the managed llama-server reports provider "custom"
+(the resolution seam's generic label for a raw base_url). The picker
+payload used to materialize that as a duplicate "Custom endpoint" group
+above the Local row — same staged models listed twice, checkmark on the
+wrong group. Contract: when the current session points at the managed
+endpoint, the Local row is current and no custom-endpoint duplicate
+exists; a user's own external endpoint keeps its row untouched."""
+
+from __future__ import annotations
+
+import dataclasses
+
+import pytest
+
+
+MANAGED = {"base_url": "http://127.0.0.1:18434/v1", "api_key": "k"}
+STAGED = {"Qwen-A-UD-Q4_K_M", "Qwen-B-UD-Q4_K_M"}
+
+
+@pytest.fixture
+def ctx(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ import hermes_cli.inventory as inv
+
+ monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.staged_model_ids",
+ lambda: set(STAGED))
+ monkeypatch.setattr("hermes_cli.local_runtime.endpoint._state_endpoint",
+ lambda: dict(MANAGED))
+ context = inv.load_picker_context()
+ return inv, context
+
+
+def _rows(inv, context, **overrides):
+ context = dataclasses.replace(context, **overrides)
+ return inv.build_models_payload(context, explicit_only=True)["providers"]
+
+
+def test_managed_custom_session_shows_only_the_local_row(ctx):
+ inv, context = ctx
+ rows = _rows(inv, context,
+ current_provider="custom",
+ current_model="Qwen-A-UD-Q4_K_M",
+ current_base_url=MANAGED["base_url"])
+ slugs = [r["slug"] for r in rows]
+ assert "llamacpp" in slugs
+ assert "custom" not in slugs, (
+ "managed endpoint leaked a duplicate 'Custom endpoint' group")
+ local = next(r for r in rows if r["slug"] == "llamacpp")
+ assert local["is_current"] is True
+ assert local["name"] == "Local"
+
+
+def test_external_custom_endpoint_keeps_its_row(ctx):
+ inv, context = ctx
+ rows = _rows(inv, context,
+ current_provider="custom",
+ current_model="some-model",
+ current_base_url="http://my-vllm-box:8000/v1")
+ slugs = [r["slug"] for r in rows]
+ assert "custom" in slugs, "a real external endpoint must keep its row"
+ custom = next(r for r in rows if r["slug"] == "custom")
+ assert custom["is_current"] is True
+ local = next(r for r in rows if r["slug"] == "llamacpp")
+ assert local["is_current"] is False
+
+
+def test_remote_provider_session_unaffected(ctx):
+ inv, context = ctx
+ rows = _rows(inv, context)
+ local = next(r for r in rows if r["slug"] == "llamacpp")
+ assert local["is_current"] is False
+ assert "custom" not in [r["slug"] for r in rows if r.get("is_current")]
diff --git a/tests/hermes_cli/test_local_quickstart.py b/tests/hermes_cli/test_local_quickstart.py
new file mode 100644
index 0000000000..f6a8f56193
--- /dev/null
+++ b/tests/hermes_cli/test_local_quickstart.py
@@ -0,0 +1,187 @@
+"""Quickstart route: one POST from nothing to a working local default.
+
+Contract, not implementation: the route must (a) preflight-fail
+synchronously when nothing fits, (b) report which legs the job will run
+(runtime install / model download), skipping legs already satisfied,
+and (c) run install -> download -> activate through the same code paths
+the individual routes use. The slow legs are stubbed at their module
+boundaries; the sequencing and job bookkeeping are real.
+"""
+
+from __future__ import annotations
+
+import time
+from pathlib import Path
+
+import pytest
+from fastapi.testclient import TestClient
+
+
+@pytest.fixture
+def client(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli import web_server
+
+ test_client = TestClient(web_server.app)
+ test_client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
+ return test_client
+
+
+def _wait_job(client, job_id: str, timeout: float = 10.0) -> dict:
+ deadline = time.monotonic() + timeout
+ while time.monotonic() < deadline:
+ job = client.get(f"/api/local-models/jobs/{job_id}").json()
+ if job["status"] != "running":
+ return job
+ time.sleep(0.05)
+ raise AssertionError(f"job {job_id} still running after {timeout}s")
+
+
+def test_quickstart_unknown_model_404s(client):
+ r = client.post("/api/local-models/quickstart", json={"model_id": "no-such"})
+ assert r.status_code == 404
+
+
+def test_quickstart_refuses_when_nothing_fits(client, monkeypatch):
+ """Preflight is synchronous: a machine no catalog entry fits gets a 409
+ with guidance, not a doomed background job."""
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.catalog.select_variant", lambda *a, **k: None)
+ r = client.post("/api/local-models/quickstart", json={})
+ assert r.status_code == 409
+ assert "Local Models" in r.json()["detail"]
+
+
+def test_quickstart_runs_all_three_legs(client, monkeypatch, tmp_path):
+ """Fresh machine: install runtime -> download recommended -> activate.
+ Each leg is asserted by its observable call, in order."""
+ calls: list[str] = []
+
+ # Leg 1: no runtime installed yet; install is the stubbed binaries call.
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.installed_tags", lambda: [])
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.ensure_runtime_installed",
+ lambda tag, backend, progress=None: calls.append("install"))
+
+ # Leg 2: nothing staged; the download writes the files the plan names.
+ def _fake_download(url, dest, job, *, base_done=0, keep_totals=False):
+ Path(dest).parent.mkdir(parents=True, exist_ok=True)
+ Path(dest).write_bytes(b"GGUF\x00")
+ calls.append("download")
+
+ monkeypatch.setattr(
+ "hermes_cli.web_routers.local_models.download_file", _fake_download)
+
+ # Leg 3: activation — stub the server start and the model assignment.
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
+ lambda config, force=False: calls.append("server") or None)
+ monkeypatch.setattr(
+ "hermes_cli.web_routers.local_models._state_endpoint",
+ lambda: {"base_url": "http://127.0.0.1:1/v1", "api_key": "k"})
+ from hermes_cli import web_deps
+
+ monkeypatch.setattr(
+ web_deps, "late",
+ lambda name: (lambda *a, **k: calls.append("assign")))
+
+ r = client.post("/api/local-models/quickstart", json={})
+ assert r.status_code == 200
+ body = r.json()
+ assert body["needs_runtime"] is True
+ assert body["needs_download"] is True
+ assert body["download_bytes"] > 0
+
+ job = _wait_job(client, body["job_id"])
+ assert job["status"] == "done", job["error"]
+ assert job["kind"] == "quickstart"
+ # Order is the contract: engine, weights, server, default.
+ assert calls[0] == "install"
+ assert "download" in calls
+ assert calls.index("install") < calls.index("download") < calls.index("assign")
+
+ # Durable effect: the runtime is enabled in config.
+ from hermes_cli.config import load_config
+
+ assert load_config()["local_runtime"]["enabled"] is True
+
+
+def test_quickstart_skips_satisfied_legs(client, monkeypatch):
+ """Runtime present and model already staged: the response says so and
+ the job goes straight to activation."""
+ calls: list[str] = []
+
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.installed_tags", lambda: ["b10362"])
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.ensure_runtime_installed",
+ lambda tag, backend, progress=None: calls.append("install"))
+
+ # Every catalog variant reads as staged.
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ all_ids = {v.model_id for e in CATALOG for v in e.variants}
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.bootstrap.staged_model_ids", lambda: all_ids)
+ monkeypatch.setattr(
+ "hermes_cli.web_routers.local_models.download_file",
+ lambda *a, **k: calls.append("download"))
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
+ lambda config, force=False: None)
+ monkeypatch.setattr(
+ "hermes_cli.web_routers.local_models._state_endpoint",
+ lambda: {"base_url": "http://127.0.0.1:1/v1", "api_key": "k"})
+ from hermes_cli import web_deps
+
+ monkeypatch.setattr(
+ web_deps, "late",
+ lambda name: (lambda *a, **k: calls.append("assign")))
+
+ r = client.post("/api/local-models/quickstart", json={})
+ assert r.status_code == 200
+ body = r.json()
+ assert body["needs_runtime"] is False
+ assert body["needs_download"] is False
+ assert body["download_bytes"] == 0
+
+ job = _wait_job(client, body["job_id"])
+ assert job["status"] == "done", job["error"]
+ assert "install" not in calls and "download" not in calls
+ assert calls == ["assign"] or calls[-1] == "assign"
+
+
+@pytest.fixture
+def quickstart_ready(monkeypatch):
+ """Preflight passes without hardware or network: the runtime reads as
+ installed and every entry's first variant is servable, so the POST
+ reaches the single-flight lock instead of 409ing at fit/engine
+ preflight on machines where nothing fits."""
+ from hermes_cli.local_runtime.catalog import VariantChoice
+
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.installed_tags", lambda: ["b10362"])
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.catalog.select_variant",
+ lambda entry, budget: VariantChoice(variant=entry.variants[0],
+ zero_spill=True,
+ reason_key="best-fits"))
+ monkeypatch.setattr(
+ "hermes_cli.web_routers.local_models._engine_too_old",
+ lambda min_engine: False)
+
+
+def test_quickstart_is_single_flight(client, quickstart_ready, monkeypatch):
+ """A second quickstart while one runs must 409, not start a twin job
+ (the job sequences installs, downloads, a server bounce, and a config
+ write — two interleaved runs corrupt all four)."""
+ import hermes_cli.web_routers.local_models as lm
+
+ lm._QUICKSTART_LOCK.acquire()
+ try:
+ r = client.post("/api/local-models/quickstart", json={})
+ assert r.status_code == 409
+ assert "already running" in r.json()["detail"].lower()
+ finally:
+ lm._QUICKSTART_LOCK.release()
diff --git a/tests/hermes_cli/test_local_recommendation.py b/tests/hermes_cli/test_local_recommendation.py
new file mode 100644
index 0000000000..4aa3aabd58
--- /dev/null
+++ b/tests/hermes_cli/test_local_recommendation.py
@@ -0,0 +1,170 @@
+"""The recommendation decision table — the reviewable matrix.
+
+The recommendation itself is DERIVED (catalog.recommended_entry: best
+quality among resident entries clearing the pleasant speed floor, else
+fastest resident, else least-painful spilled), so nobody hand-maintains
+per-hardware-class picks. This table is the editorial control on that
+derivation: it enumerates the real memory size classes x {discrete,
+unified} and pins every cell. A catalog change (new model, quality
+re-rank, quant swap) flips cells HERE, and the diff of this file in
+review IS the sign-off on what each machine class gets.
+
+These are decision pins, not change-detectors: each cell is a choice a
+human approved, exactly like a golden file. When a cell flips on
+purpose, update it in the same commit and say why. When one flips by
+surprise, that is the test doing its job.
+
+Budgets mirror hardware.probe_budget's planning-mode shapes (margins,
+UMA headroom) so the cells match what a real machine of that class
+resolves.
+"""
+
+from __future__ import annotations
+
+import pytest
+
+from hermes_cli.local_runtime.catalog import (
+ CATALOG,
+ PLEASANT_FLOOR_TOK_S,
+ predicted_decode_tok_s,
+ recommended_entry,
+ recommended_id,
+ select_variant,
+)
+from hermes_cli.local_runtime.estimator import HardwareBudget
+
+_GIB = 1 << 30
+
+
+def _discrete(size_gb: int) -> HardwareBudget:
+ total = size_gb * _GIB
+ margin = max(2 * _GIB, int(total * 0.09))
+ return HardwareBudget(usable_vram_bytes=max(0, total - margin),
+ total_device_bytes=total,
+ ram_available_bytes=64 * _GIB, uma=False)
+
+
+def _unified(size_gb: int) -> HardwareBudget:
+ total = size_gb * _GIB
+ return HardwareBudget(usable_vram_bytes=int(total * 0.80),
+ total_device_bytes=total,
+ ram_available_bytes=0, uma=True)
+
+
+# The decision table. Cells were generated by the resolver and then
+# reviewed as editorial decisions:
+#
+# VRAM | discrete | unified
+# -----+-------------------------+------------------------
+# 8 | qwen3.6-35b-a3b spilled | (none fits)
+# 16 | qwen3.6-35b-a3b spilled | (none fits)
+# 24 | qwen3.8-27b | (none fits)
+# 32 | qwen3.8-27b | qwen3.6-35b-a3b
+# 48 | qwen3.8-27b | qwen3.6-35b-a3b
+# 96 | qwen3.8-27b | qwen3.6-35b-a3b
+# 128 | qwen3.8-flash-next | qwen3.6-35b-a3b
+# 256 | qwen3.8-flash-next | qwen3.8-flash-next
+# 512 | qwen3.8-flash-next | qwen3.8-flash-next
+#
+# Reading guide for reviewers:
+# - Discrete <=16 GB: nothing runs resident; the 35B MoE is the least
+# painful spill (active slice streams from host; a dense spill reads
+# every weight over the bus).
+# - Discrete 24-96 GB: the 27B is the flagship experience — dense reads
+# at ~1 TB/s clear the floor easily, so quality decides.
+# - Discrete/unified where Flash Next fits resident (128 GB discrete,
+# 256+ GB unified): the frontier model is the pick — highest quality,
+# and its sparse decode clears the floor even at UMA bandwidth
+# (~24 tok/s predicted at 210 GB/s).
+# - Unified 32-128 GB — the Spark class, the reason this resolver
+# exists: the dense 27B predicts ~13 tok/s at UMA bandwidth (below
+# the pleasant floor), so the 35B-A3B (~60 tok/s) wins.
+# - Unified <=24 GB: no entry passes the physics check inside the UMA
+# budget (spilling is impossible on UMA by construction — the pool IS
+# the RAM). The pane's browse flow is the path for those machines
+# until a small catalog entry lands (revisit when one does).
+DECISION_TABLE = [
+ (8, "discrete", "qwen3.6-35b-a3b", "least-painful-spilled"),
+ (8, "unified", None, None),
+ (16, "discrete", "qwen3.6-35b-a3b", "least-painful-spilled"),
+ (16, "unified", None, None),
+ (24, "discrete", "qwen3.8-27b", "best-quality-resident"),
+ (24, "unified", None, None),
+ (32, "discrete", "qwen3.8-27b", "best-quality-resident"),
+ (32, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
+ (48, "discrete", "qwen3.8-27b", "best-quality-resident"),
+ (48, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
+ (96, "discrete", "qwen3.8-27b", "best-quality-resident"),
+ (96, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
+ (128, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
+ (128, "unified", "qwen3.6-35b-a3b", "speed-gated-quality"),
+ (256, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
+ (256, "unified", "qwen3.8-flash-next", "best-quality-resident"),
+ (512, "discrete", "qwen3.8-flash-next", "best-quality-resident"),
+ (512, "unified", "qwen3.8-flash-next", "best-quality-resident"),
+]
+
+
+@pytest.mark.parametrize(
+ ("size_gb", "kind", "expected", "expected_reason"),
+ DECISION_TABLE,
+ ids=[f"{s}GB-{k}" for s, k, _, _ in DECISION_TABLE])
+def test_recommendation_decision_table(size_gb, kind, expected, expected_reason):
+ """Pins the pick AND its reason per cell: the reason is user-facing
+ (the Recommended badge's tooltip), so a cell whose rationale flips
+ without the pick flipping is still a review-worthy change."""
+ budget = _discrete(size_gb) if kind == "discrete" else _unified(size_gb)
+ picked = recommended_entry(budget)
+ if expected is None:
+ assert picked is None
+ else:
+ assert picked is not None
+ assert (picked[0].id, picked[1]) == (expected, expected_reason)
+
+
+# ── invariants behind the table (survive catalog changes) ──
+
+
+def test_every_entry_carries_the_recommendation_axes():
+ """quality and decode_fraction are authoring requirements: an entry
+ without them silently loses every quality comparison (quality=0) or
+ prices as dense (decode_fraction=1.0)."""
+ for entry in CATALOG:
+ assert entry.quality > 0, f"{entry.id} has no quality ordering"
+ assert 0.0 < entry.decode_fraction <= 1.0, entry.id
+ if not entry.moe:
+ assert entry.decode_fraction == 1.0, (
+ f"{entry.id} is dense — it reads every weight per token")
+
+
+def test_unified_never_recommends_a_below_floor_dense_model():
+ """The Spark rule, as an invariant: whatever the catalog holds, a
+ unified-memory machine must not be told to run a model whose
+ predicted decode is below the pleasant floor while a resident
+ alternative clears it."""
+ budget = _unified(128)
+ pick = recommended_id(budget)
+ assert pick is not None
+ entry = next(e for e in CATALOG if e.id == pick)
+ choice = select_variant(entry, budget)
+ assert choice is not None and choice.zero_spill
+ clears = [
+ e for e in CATALOG
+ if (c := select_variant(e, budget)) is not None and c.zero_spill
+ and predicted_decode_tok_s(e, c.variant, budget) >= PLEASANT_FLOOR_TOK_S
+ ]
+ if clears:
+ assert predicted_decode_tok_s(entry, choice.variant, budget) >= PLEASANT_FLOOR_TOK_S
+
+
+def test_quality_decides_where_speed_permits():
+ """On big discrete hardware every resident entry clears the floor, so
+ the pick must be the highest-quality fitting entry — the axis that
+ justifies carrying an editorial field at all."""
+ budget = _discrete(512)
+ pick = recommended_id(budget)
+ resident = [
+ e for e in CATALOG
+ if (c := select_variant(e, budget)) is not None and c.zero_spill
+ ]
+ assert pick == max(resident, key=lambda e: e.quality).id
diff --git a/tests/hermes_cli/test_local_runtime.py b/tests/hermes_cli/test_local_runtime.py
new file mode 100644
index 0000000000..0603e9ec29
--- /dev/null
+++ b/tests/hermes_cli/test_local_runtime.py
@@ -0,0 +1,825 @@
+"""Contract tests for hermes_cli.local_runtime — Rollouts 1+2.
+
+Per the design's verification plan: relationships and contracts, no
+change-detector tests, real imports against temp HERMES_HOME (the autouse
+fixture isolates it). The stub HTTP server speaks just enough llama-server
+(/props, /health, /models, /v1/chat/completions, /metrics, /slots) to
+exercise detection fingerprinting and supervisor logic without a GPU.
+"""
+
+from __future__ import annotations
+
+import json
+import os
+import threading
+from http.server import BaseHTTPRequestHandler, HTTPServer
+from pathlib import Path
+
+import pytest
+
+from hermes_cli.local_runtime.binaries import (
+ AssetPlan,
+ BinaryResolutionError,
+ resolve_assets,
+ select_backend,
+)
+from hermes_cli.local_runtime.detect import DetectedServer, probe_port
+
+
+# ── stub llama-server ────────────────────────────────────────
+
+
+class _StubHandler(BaseHTTPRequestHandler):
+ """Minimal llama-server imitation; behavior driven by class attrs."""
+
+ props: dict = {}
+ models: dict | None = None
+ require_auth = False
+ chat_answer = "Paris"
+ requests_processing = 0
+ slots: list = []
+
+ def _send(self, code: int, body: dict | str | None = None) -> None:
+ raw = (json.dumps(body) if isinstance(body, dict) else (body or "")).encode()
+ self.send_response(code)
+ self.send_header("Content-Type", "application/json")
+ self.send_header("Content-Length", str(len(raw)))
+ self.end_headers()
+ self.wfile.write(raw)
+
+ def do_GET(self): # noqa: N802
+ if self.require_auth and "Authorization" not in self.headers:
+ self._send(401, {})
+ return
+ path = self.path.split("?")[0] # router telemetry uses ?model=
+ if path == "/props":
+ self._send(200, self.props)
+ elif path == "/health":
+ self._send(200, {"status": "ok"})
+ elif path == "/models":
+ if self.models is None:
+ self._send(404, {})
+ else:
+ self._send(200, self.models)
+ elif path == "/metrics":
+ self._send(200, f"llamacpp:requests_processing {self.requests_processing}\n")
+ elif path == "/slots":
+ raw = json.dumps(self.slots).encode()
+ self.send_response(200)
+ self.send_header("Content-Type", "application/json")
+ self.send_header("Content-Length", str(len(raw)))
+ self.end_headers()
+ self.wfile.write(raw)
+ else:
+ self._send(404, {})
+
+ def do_POST(self): # noqa: N802
+ if self.path == "/v1/chat/completions":
+ self._send(200, {"choices": [{"message": {
+ "role": "assistant", "content": self.chat_answer}}]})
+ elif self.path == "/models/load":
+ self._send(200, {"success": True})
+ elif self.path == "/models/unload":
+ type(self).unloaded = getattr(type(self), "unloaded", [])
+ length = int(self.headers.get("Content-Length", 0))
+ body = json.loads(self.rfile.read(length)) if length else {}
+ type(self).unloaded.append(body.get("model"))
+ self._send(200, {"success": True})
+ else:
+ self._send(404, {})
+
+ def log_message(self, *args): # silence
+ pass
+
+
+@pytest.fixture
+def stub_server():
+ """Yields (port, handler_class); handler attrs are per-test mutable."""
+
+ class Handler(_StubHandler):
+ props = {}
+ models = None
+ require_auth = False
+ slots = []
+
+ server = HTTPServer(("127.0.0.1", 0), Handler)
+ thread = threading.Thread(target=server.serve_forever, daemon=True)
+ thread.start()
+ yield server.server_address[1], Handler
+ server.shutdown()
+
+
+# ── detection (Rollout 1) ────────────────────────────────────
+
+
+def test_probe_fingerprints_real_llama_server(stub_server):
+ port, handler = stub_server
+ handler.props = {
+ "build_info": "b10290-c8e03ce81",
+ "model_path": "C:/models/some model with spaces.gguf",
+ "default_generation_settings": {"n_ctx": 65536},
+ }
+ handler.models = {"data": [{"id": "m", "status": {"value": "unloaded"}}]}
+ hit = probe_port(port)
+ assert isinstance(hit, DetectedServer)
+ assert hit.base_url == f"http://127.0.0.1:{port}/v1"
+ assert hit.build_info.startswith("b10290")
+ assert hit.n_ctx == 65536
+ assert hit.router_mode is True
+ assert hit.auth_required is False
+
+
+def test_probe_rejects_non_llama_openai_server(stub_server):
+ # Answers /props with no build_info (e.g. some other local service).
+ port, handler = stub_server
+ handler.props = {"something": "else"}
+ assert probe_port(port) is None
+
+
+def test_probe_single_model_mode_is_not_router(stub_server):
+ port, handler = stub_server
+ handler.props = {"build_info": "b10290-x", "model_path": "m.gguf"}
+ handler.models = None # /models 404s in plain (non-router) mode
+ hit = probe_port(port)
+ assert hit is not None
+ assert hit.router_mode is False
+
+
+def test_probe_auth_required_still_detected(stub_server):
+ port, handler = stub_server
+ handler.require_auth = True
+ hit = probe_port(port)
+ assert hit is not None
+ assert hit.auth_required is True
+
+
+def test_probe_dead_port_returns_none():
+ # Bind-then-close to get a port that is definitely closed.
+ import socket
+ with socket.socket() as s:
+ s.bind(("127.0.0.1", 0))
+ dead_port = s.getsockname()[1]
+ assert probe_port(dead_port) is None
+
+
+# ── binary resolver (Rollout 2) ──────────────────────────────
+
+
+@pytest.mark.parametrize("os_name,arch,backend,ok", [
+ ("win", "x64", "cuda", True),
+ ("win", "x64", "vulkan", True),
+ ("win", "x64", "cpu", True),
+ ("win", "arm64", "cpu", True),
+ ("win", "arm64", "cuda", True), # upstream ships these since ~b1036x (CUDA 13.4)
+ ("win", "arm64", "vulkan", False),
+ ("macos", "arm64", "metal", True),
+ ("ubuntu", "x64", "vulkan", True),
+ ("ubuntu", "x64", "cpu", True),
+ ("ubuntu", "x64", "cuda", False), # no prebuilt linux CUDA
+])
+def test_resolver_platform_matrix(os_name, arch, backend, ok):
+ if ok:
+ plan = resolve_assets("b10290", backend, os_name=os_name, arch=arch)
+ assert plan.assets, "resolvable combination must yield assets"
+ # Invariant: every asset names the tag or is a paired runtime zip.
+ for asset in plan.assets:
+ assert "b10290" in asset or asset.startswith("cudart-")
+ else:
+ with pytest.raises(BinaryResolutionError):
+ resolve_assets("b10290", backend, os_name=os_name, arch=arch)
+
+
+def test_windows_cuda_pairs_cudart():
+ """Windows CUDA must ship the runtime zip — users have no toolkit."""
+ plan = resolve_assets("b10290", "cuda", os_name="win", arch="x64")
+ assert any(a.startswith("cudart-") for a in plan.assets)
+
+
+def test_windows_cuda_arm64_pairs_cudart_on_its_own_version():
+ """arm64 CUDA rides its own CUDA line (13.4 at b10362, verified live):
+ both zips must agree on version and name the arch."""
+ plan = resolve_assets("b10362", "cuda", os_name="win", arch="arm64")
+ assert len(plan.assets) == 2
+ assert all("arm64" in a for a in plan.assets)
+ versions = {a.split("cuda-")[1].split("-")[0] for a in plan.assets}
+ assert len(versions) == 1, f"paired zips disagree on CUDA version: {plan.assets}"
+ assert any(a.startswith("cudart-") for a in plan.assets)
+ assert any(a.startswith("llama-") for a in plan.assets)
+
+
+def test_install_dir_is_profile_scoped(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ plan = AssetPlan(tag="b10290", backend="cuda")
+ assert str(tmp_path) in str(plan.install_dir)
+ assert "runtimes" in plan.install_dir.parts
+
+
+@pytest.mark.parametrize("vendor,os_name,expected", [
+ ("NVIDIA GeForce RTX 5090", "win", "cuda"),
+ ("nvidia", "ubuntu", "cuda"),
+ ("AMD Radeon RX 7900", "win", "vulkan"),
+ ("intel", "win", "vulkan"),
+ (None, "win", "cpu"),
+ ("", "ubuntu", "cpu"),
+ ("nvidia", "macos", "metal"), # macOS is Metal regardless
+ (None, "macos", "metal"),
+])
+def test_backend_selection(vendor, os_name, expected):
+ assert select_backend(vendor, os_name=os_name) == expected
+
+
+def test_sha256_mismatch_rejects(tmp_path, monkeypatch):
+ """A pinned hash that doesn't match the download must hard-fail."""
+ from hermes_cli.local_runtime import binaries
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ # Pre-place a wrong-content "download" so no network is touched. The
+ # asset name is host-dependent (win/.zip, ubuntu/.tar.gz, macos/.zip)
+ # — resolve it the way the installer will, so the poisoned file is the
+ # one it verifies on every CI platform.
+ plan = binaries.resolve_assets("b10290", "cpu")
+ asset = plan.assets[0]
+ downloads = binaries.runtimes_root() / "downloads"
+ downloads.mkdir(parents=True)
+ (downloads / asset).write_bytes(b"not the real archive")
+ with pytest.raises(BinaryResolutionError, match="sha256 mismatch"):
+ binaries.ensure_runtime_installed(
+ "b10290", "cpu",
+ expected_sha256={asset: "0" * 64})
+ # The poisoned download must not survive for a retry to trust.
+ assert not (downloads / asset).exists()
+
+
+# ── supervisor contracts (stubbed; no GPU) ───────────────────
+
+
+def _make_supervisor(tmp_path, port):
+ """Supervisor pointed at the stub: skip spawn, drive HTTP logic only."""
+ from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
+
+ sup = LlamaServerSupervisor(
+ install_dir=tmp_path, models_dir=tmp_path, port=port)
+ return sup
+
+
+def test_touch_generate_is_the_readiness_proof(stub_server, tmp_path):
+ port, handler = stub_server
+ sup = _make_supervisor(tmp_path, port)
+ handler.chat_answer = "Paris"
+ assert sup.touch_generate("m") is True
+ handler.chat_answer = "I cannot answer that."
+ assert sup.touch_generate("m") is False
+
+
+def test_touch_generate_scans_reasoning_content(stub_server, tmp_path):
+ """Reasoning models answer inside reasoning_content (receipted pitfall)."""
+ port, handler = stub_server
+ sup = _make_supervisor(tmp_path, port)
+
+ class ReasoningHandler(handler): # type: ignore[valid-type]
+ def do_POST(self): # noqa: N802
+ if self.path == "/v1/chat/completions":
+ self._send(200, {"choices": [{"message": {
+ "role": "assistant", "content": "",
+ "reasoning_content": "The capital of France is Paris."}}]})
+ else:
+ self._send(404, {})
+
+ # Swap handler class on the live stub server socket is overkill; just
+ # verify the scan logic path via the normal handler with empty content.
+ handler.chat_answer = ""
+ assert sup.touch_generate("m") is False # empty content, no reasoning field
+
+
+def test_ensure_model_ready_unknown_model_raises(stub_server, tmp_path):
+ port, handler = stub_server
+ handler.models = {"data": [{"id": "present", "status": {"value": "unloaded"}}]}
+ sup = _make_supervisor(tmp_path, port)
+ with pytest.raises(KeyError):
+ sup.ensure_model_ready("absent")
+
+
+def test_model_failures_surface_exit_code_not_retry(stub_server, tmp_path):
+ """Design: child failures surface, never auto-retry."""
+ port, handler = stub_server
+ handler.models = {"data": [
+ {"id": "ok", "status": {"value": "loaded"}},
+ {"id": "dead", "status": {"value": "failed", "exit_code": -1073741819}},
+ ]}
+ sup = _make_supervisor(tmp_path, port)
+ failures = sup.model_failures()
+ assert failures == {"dead": -1073741819}
+
+
+def test_is_idle_requires_no_busy_slots_and_zero_processing(stub_server, tmp_path):
+ port, handler = stub_server
+ sup = _make_supervisor(tmp_path, port)
+ # Router telemetry is per-child (?model=); a loaded model must exist for
+ # is_idle to have anything to check.
+ handler.models = {"data": [{"id": "m", "status": {"value": "loaded"}}]}
+ handler.slots = [{"id": 0, "is_processing": False}]
+ handler.requests_processing = 0
+ assert sup.is_idle() is True
+ handler.slots = [{"id": 0, "is_processing": True}]
+ assert sup.is_idle() is False
+ handler.slots = [{"id": 0, "is_processing": False}]
+ handler.requests_processing = 2
+ assert sup.is_idle() is False
+
+
+def test_base_url_dials_loopback_ip_never_localhost(tmp_path):
+ """C12: localhost costs ~2s/request on Windows."""
+ sup = _make_supervisor(tmp_path, 9999)
+ assert "127.0.0.1" in sup.base_url
+ assert "localhost" not in sup.base_url
+
+
+# ── provider integration (existing alias mechanism, no new plugin) ──
+
+
+def test_llamacpp_aliases_route_to_custom_profile():
+ """Design + maintainer direction: llamacpp fits the EXISTING provider
+ mechanism — the aliases already resolve to the keyless custom profile;
+ no parallel provider plugin exists."""
+ from providers import get_provider_profile
+
+ for alias in ("llamacpp", "llama.cpp", "llama-cpp"):
+ profile = get_provider_profile(alias)
+ assert profile is not None, alias
+ assert profile.name == "custom"
+ assert profile.env_vars == () # credential is reachability
+
+
+def test_llamacpp_endpoint_resolution_prefers_managed(tmp_path, monkeypatch, stub_server):
+ """provider: llamacpp with a live managed server resolves to it,
+ api-key included."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ port, handler = stub_server
+ from hermes_cli.local_runtime import endpoint as ep
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ # A LIVE pid: the ownership guard treats health-200 + dead recorded
+ # pid as a foreign server on our stable port (scratch-profile
+ # collision), so claiming this test process models "our server".
+ "base_url": f"http://127.0.0.1:{port}/v1", "api_key": "sk-managed", "pid": os.getpid(),
+ }), encoding="utf-8")
+ resolved = ep.resolve_llamacpp_endpoint()
+ assert resolved == {"base_url": f"http://127.0.0.1:{port}/v1", "api_key": "sk-managed"}
+
+
+def test_llamacpp_endpoint_stale_state_falls_through(tmp_path, monkeypatch):
+ """A crashed-without-cleanup state file (dead pid, dead endpoint) must
+ not blackhole requests: state ignored -> detection (none here) -> None."""
+ import socket
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ with socket.socket() as s:
+ s.bind(("127.0.0.1", 0))
+ dead_port = s.getsockname()[1]
+ from hermes_cli.local_runtime import endpoint as ep
+ from hermes_cli.local_runtime.detect import DEFAULT_PROBE_PORTS
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ "base_url": f"http://127.0.0.1:{dead_port}/v1", "api_key": "sk-x", "pid": 1,
+ }), encoding="utf-8")
+ monkeypatch.setattr(ep, "_pid_alive", lambda pid: False)
+ # Keep detection away from any real server on 8080 during the test.
+ monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS",
+ (dead_port,))
+ assert ep.resolve_llamacpp_endpoint() is None
+ assert DEFAULT_PROBE_PORTS # (import kept honest)
+
+
+def test_llamacpp_dead_server_raises_friendly_error(tmp_path, monkeypatch):
+ """A llamacpp send with no server must say WHY in user terms, not fall
+ through to the generic custom path (which lands on a cloud provider
+ with a placeholder key and surfaces as a baffling '401 Invalid API
+ key'). Message tracks the off switch: enabled = probably starting;
+ disabled = the user turned it off."""
+ import pytest
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+
+ from hermes_cli import runtime_provider as rp
+
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.endpoint.resolve_llamacpp_endpoint",
+ lambda *a, **k: None)
+
+ monkeypatch.setattr(
+ "hermes_cli.config.load_config",
+ lambda: {"local_runtime": {"enabled": False}})
+ with pytest.raises(ValueError, match="turned off"):
+ rp._resolve_named_custom_runtime(requested_provider="llamacpp")
+
+ monkeypatch.setattr(
+ "hermes_cli.config.load_config",
+ lambda: {"local_runtime": {"enabled": True}})
+ with pytest.raises(ValueError, match="isn't running"):
+ rp._resolve_named_custom_runtime(requested_provider="llamacpp")
+
+ # An explicit base_url is the user pointing at a specific server —
+ # that path keeps its own error reporting, never this one.
+ result = rp._resolve_named_custom_runtime(
+ requested_provider="llamacpp",
+ explicit_base_url="http://127.0.0.1:9999/v1")
+ assert result is None or result.get("base_url", "").startswith("http://127.0.0.1:9999")
+
+
+def test_llamacpp_endpoint_starting_server_resolves(tmp_path, monkeypatch):
+ """The restart race: state written at spawn, server not yet healthy,
+ supervisor child alive — resolution must return the endpoint (a
+ STARTING server is configured, not missing credentials; this exact
+ race threw the app back to onboarding on the first restart test)."""
+ import socket
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ with socket.socket() as s:
+ s.bind(("127.0.0.1", 0))
+ not_listening = s.getsockname()[1]
+ from hermes_cli.local_runtime import endpoint as ep
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ "base_url": f"http://127.0.0.1:{not_listening}/v1",
+ "api_key": "sk-starting", "pid": 4242,
+ }), encoding="utf-8")
+ monkeypatch.setattr(ep, "_pid_alive", lambda pid: True)
+ resolved = ep.resolve_llamacpp_endpoint()
+ assert resolved is not None
+ assert resolved["api_key"] == "sk-starting"
+
+
+def test_llamacpp_endpoint_waits_for_boot_in_flight(tmp_path, monkeypatch):
+ """The SECOND restart race (no state file at all yet): a fresh backend's
+ readiness probe resolves before the lifespan boot thread has even
+ spawned the server. With the runtime enabled+installed, resolution must
+ poll briefly and pick up the state file when the boot thread writes it
+ — not report unconfigured."""
+ import threading
+ import time as _time
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime import endpoint as ep
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ # Boot is in flight: runtime enabled + binary installed.
+ monkeypatch.setattr(ep, "_boot_in_flight", lambda config: True)
+ monkeypatch.setattr(ep, "_pid_alive", lambda pid: True)
+ # Nothing detected externally.
+ monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
+
+ def _late_writer():
+ _time.sleep(0.6)
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ "base_url": "http://127.0.0.1:59999/v1",
+ "api_key": "sk-boot", "pid": 777,
+ }), encoding="utf-8")
+
+ t = threading.Thread(target=_late_writer)
+ t.start()
+ try:
+ resolved = ep.resolve_llamacpp_endpoint(wait_for_boot_s=5.0)
+ finally:
+ t.join()
+ assert resolved is not None
+ assert resolved["api_key"] == "sk-boot"
+
+
+def test_resolution_kicks_boot_when_no_thread_is_booting(tmp_path, monkeypatch):
+ """The dead-router-mid-flight case: runtime enabled+installed, but no
+ state file and NO lifespan boot thread running (the router died after
+ backend start — tree-killed with a stale backend, or the stable port
+ was owned by another install and the ownership guard refused it).
+ Resolution must not just wait for a boot that nobody is doing — it
+ kicks ensure_local_runtime itself and picks up the state file that
+ boot writes."""
+ import time as _time
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime import bootstrap as bs
+ from hermes_cli.local_runtime import endpoint as ep
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ monkeypatch.setattr(ep, "_boot_in_flight", lambda config: True)
+ monkeypatch.setattr(ep, "_pid_alive", lambda pid: True)
+ monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
+
+ def _fake_ensure(config, force=False):
+ _time.sleep(0.3) # a real spawn takes a moment
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ "base_url": "http://127.0.0.1:59998/v1",
+ "api_key": "sk-kicked", "pid": 778,
+ }), encoding="utf-8")
+
+ monkeypatch.setattr(bs, "ensure_local_runtime", _fake_ensure)
+
+ resolved = ep.resolve_llamacpp_endpoint(config={}, wait_for_boot_s=5.0)
+ assert resolved is not None
+ assert resolved["api_key"] == "sk-kicked"
+
+
+def test_boot_in_flight_real_gate(tmp_path, monkeypatch):
+ """_boot_in_flight exercised FOR REAL (the previous regression test
+ monkeypatched it — and the real one threw TypeError on every call,
+ silently disabling the boot wait). Enabled + verified manifest on
+ disk -> True; either missing -> False."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime import endpoint as ep
+ from hermes_cli.local_runtime.binaries import runtimes_root
+
+ enabled = {"local_runtime": {"enabled": True}}
+ # Not installed yet -> False.
+ assert ep._boot_in_flight(enabled) is False
+ # Verified install manifest -> True.
+ install = runtimes_root() / "b10290" / "cuda"
+ install.mkdir(parents=True)
+ (install / "manifest.json").write_text(
+ json.dumps({"tag": "b10290", "verified_version": "5015 (abc)"}),
+ encoding="utf-8")
+ assert ep._boot_in_flight(enabled) is True
+ # Disabled -> False even when installed.
+ assert ep._boot_in_flight({"local_runtime": {"enabled": False}}) is False
+
+
+def test_idle_sweep_unloads_idle_models(tmp_path, monkeypatch, stub_server):
+ """Residency v2 contract: after the idle threshold, idle loaded models
+ unload — no exemptions; demand reloads anything the user returns to.
+ Idleness is the C5 contract (no busy slots)."""
+ port, handler = stub_server
+ handler.models = {"data": [
+ {"id": "model-a", "status": {"value": "loaded"}},
+ {"id": "model-b", "status": {"value": "loaded"}},
+ ]}
+ handler.slots = [] # everyone idle per C5
+ handler.requests_processing = 0
+ handler.unloaded = []
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
+
+ sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
+
+ t0 = 1000.0
+ # First sweep: starts the idle clocks, nothing unloads yet.
+ assert sup.sweep_idle(now=t0) == []
+ # Before the threshold: still nothing.
+ assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S - 1) == []
+ # Past the threshold: both idle models unload.
+ assert sorted(sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 1)) == ["model-a", "model-b"]
+ assert sorted(handler.unloaded) == ["model-a", "model-b"]
+
+
+def test_idle_sweep_busy_model_resets_clock(tmp_path, monkeypatch, stub_server):
+ """A model seen busy (C5: busy slot) restarts its idle clock — an
+ active conversation never trips the sweep."""
+ port, handler = stub_server
+ handler.models = {"data": [{"id": "side-m", "status": {"value": "loaded"}}]}
+ handler.unloaded = []
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
+
+ sup = LlamaServerSupervisor(tmp_path / "i", tmp_path / "m", port=port)
+
+ t0 = 1000.0
+ handler.slots = [] # idle: clock starts
+ assert sup.sweep_idle(now=t0) == []
+ handler.slots = [{"is_processing": True}] # busy mid-window
+ assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S) == []
+ handler.slots = [] # idle again: clock restarts, not expired
+ assert sup.sweep_idle(now=t0 + sup.IDLE_UNLOAD_S + 10) == []
+ assert handler.unloaded == []
+
+
+def test_staged_models_requires_every_split_part(tmp_path, monkeypatch):
+ """A split GGUF mid-download must NOT count as staged: the picker, the
+ catalog's 'downloaded' flag, and the router's model list all read
+ staged_models(), and a first part with missing continuations is not
+ servable. Single files and complete splits count; continuation parts
+ never count as their own model."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ import hermes_cli.local_runtime.bootstrap as bs
+
+ mdir = bs.models_dir()
+ mdir.mkdir(parents=True, exist_ok=True)
+
+ (mdir / "Single-Q4_K_M.gguf").touch()
+ # Complete split: both parts present.
+ (mdir / "Whole-Q4-00001-of-00002.gguf").touch()
+ (mdir / "Whole-Q4-00002-of-00002.gguf").touch()
+ # Mid-download split: first part only, of three.
+ (mdir / "Partial-Q4-00001-of-00003.gguf").touch()
+
+ assert bs.staged_model_ids() == ["Single-Q4_K_M", "Whole-Q4"]
+
+
+def test_bootstrap_skips_boot_with_no_staged_models(tmp_path, monkeypatch):
+ """Residency: enabled + installed but zero staged models -> no server
+ boot (nothing to serve; the walked-away story)."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ import hermes_cli.local_runtime.bootstrap as bs
+
+ monkeypatch.setattr(bs, "_SUPERVISOR", None)
+ called = {"spawn": False}
+
+ def _boom(*a, **k):
+ called["spawn"] = True
+ raise AssertionError("must not reach install/spawn")
+
+ monkeypatch.setattr("hermes_cli.local_runtime.binaries.ensure_runtime_installed", _boom)
+ result = bs.ensure_local_runtime({"local_runtime": {"enabled": True}})
+ assert result is None
+ assert called["spawn"] is False
+
+
+def test_endpoint_identity_stable_across_supervisor_instances(tmp_path, monkeypatch):
+ """Round-7 contract: base_url AND api_key survive a restart as a unit.
+ Two supervisor constructions (= two backend boots) must agree on both —
+ sessions persist the resolved pair, so either piece rotating strands
+ every resumed session (connection error / HTTP 401)."""
+ import socket as _socket
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime import supervisor as sup_mod
+ from hermes_cli.local_runtime.supervisor import LlamaServerSupervisor
+
+ # A test-owned default port: the production default may legitimately be
+ # held by a live managed server on the dev machine.
+ with _socket.socket() as s:
+ s.bind(("127.0.0.1", 0))
+ test_port = s.getsockname()[1]
+ monkeypatch.setattr(sup_mod, "_DEFAULT_PORT", test_port)
+
+ first = LlamaServerSupervisor(tmp_path / "install", tmp_path / "models")
+ second = LlamaServerSupervisor(tmp_path / "install", tmp_path / "models")
+ assert first.api_key == second.api_key
+ assert len(first.api_key) >= 16
+ assert first.port == second.port == test_port
+ # The key is persisted, not per-process state.
+ key_file = tmp_path / ".hermes" / "runtimes" / "llamacpp" / ".api_key"
+ assert key_file.exists()
+ assert key_file.read_text(encoding="utf-8").strip() == first.api_key
+
+
+def test_llamacpp_endpoint_no_wait_when_not_enabled(tmp_path, monkeypatch):
+ """No boot in flight (runtime disabled/uninstalled): resolution returns
+ None promptly instead of burning the wait budget."""
+ import time as _time
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime import endpoint as ep
+
+ monkeypatch.setattr(ep, "_boot_in_flight", lambda config: False)
+ monkeypatch.setattr("hermes_cli.local_runtime.detect.DEFAULT_PROBE_PORTS", ())
+ t0 = _time.monotonic()
+ assert ep.resolve_llamacpp_endpoint(wait_for_boot_s=8.0) is None
+ assert _time.monotonic() - t0 < 3.0
+
+
+def test_switch_model_explicit_llamacpp_provider(tmp_path, monkeypatch, stub_server):
+ """The desktop dropdown path: switch_model(explicit_provider='llamacpp')
+ must resolve the managed provider — not 'Unknown provider' (the
+ desktop-review symptom). E2E through the real pipeline against a stub server."""
+ port, handler = stub_server
+ handler.models = {"data": [{"id": "stub-model-a", "owned_by": "llamacpp"}]}
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ "base_url": f"http://127.0.0.1:{port}/v1",
+ # Live pid: ownership guard rejects health-200 + dead recorded pid
+ # (foreign server on our stable port).
+ "api_key": "sk-managed", "pid": os.getpid(),
+ }), encoding="utf-8")
+
+ from hermes_cli.model_switch import switch_model
+
+ result = switch_model(
+ "stub-model-a",
+ current_provider="nous",
+ current_model="Hermes-4.5",
+ current_base_url="",
+ explicit_provider="llamacpp",
+ )
+ assert result.success, result.error_message
+ assert f"127.0.0.1:{port}" in (result.base_url or "")
+ assert result.api_key == "sk-managed"
+
+
+def test_runtime_provider_seam_llamacpp_alias(tmp_path, monkeypatch, stub_server):
+ """End to end through the REAL resolver: provider='llamacpp' with no
+ base_url lands on the managed endpoint with source='local-runtime'."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ port, handler = stub_server
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ # A LIVE pid: the ownership guard treats health-200 + dead recorded
+ # pid as a foreign server on our stable port (scratch-profile
+ # collision), so claiming this test process models "our server".
+ "base_url": f"http://127.0.0.1:{port}/v1", "api_key": "sk-managed", "pid": os.getpid(),
+ }), encoding="utf-8")
+
+ from hermes_cli.runtime_provider import _resolve_named_custom_runtime
+
+ runtime = _resolve_named_custom_runtime(requested_provider="llamacpp")
+ assert runtime is not None
+ assert runtime["source"] == "local-runtime"
+ assert runtime["base_url"] == f"http://127.0.0.1:{port}/v1"
+ assert runtime["api_key"] == "sk-managed"
+ assert runtime["provider"] == "custom"
+
+
+def test_runtime_provider_seam_explicit_base_url_wins(tmp_path, monkeypatch):
+ """A user-specified base_url must never be overridden by the managed
+ endpoint — pointing at a specific server means that server."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ "base_url": "http://127.0.0.1:1/v1", "api_key": "sk-managed", "pid": 1,
+ }), encoding="utf-8")
+
+ from hermes_cli.runtime_provider import _resolve_named_custom_runtime
+
+ runtime = _resolve_named_custom_runtime(
+ requested_provider="llamacpp",
+ explicit_base_url="http://127.0.0.1:9999/v1")
+ assert runtime is not None
+ assert runtime["base_url"] == "http://127.0.0.1:9999/v1"
+ assert runtime["source"] != "local-runtime"
+
+
+def test_local_runtime_config_defaults_shape():
+ """Contract: the section exists, is off by default, and carries no
+ context/VRAM knobs (design: constants, not knobs)."""
+ from hermes_cli.config_defaults import DEFAULT_CONFIG
+
+ cfg = DEFAULT_CONFIG["local_runtime"]
+ assert cfg["enabled"] is False
+ assert isinstance(cfg["tag"], str) and cfg["tag"].startswith("b")
+ forbidden = [k for k in cfg if "context" in k or "ctx" in k or "vram" in k or "kv" in k]
+ assert forbidden == [], f"policy constants leaked into config: {forbidden}"
+
+
+# ── bootstrap contracts ──────────────────────────────────────
+
+
+def test_bootstrap_disabled_is_noop(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime import bootstrap
+
+ monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
+ assert bootstrap.ensure_local_runtime({"local_runtime": {"enabled": False}}) is None
+ assert bootstrap.ensure_local_runtime({}) is None
+ assert bootstrap.ensure_local_runtime(None) is None
+
+
+def test_bootstrap_reuses_running_server(tmp_path, monkeypatch, stub_server):
+ """A live state file (another process supervising) short-circuits the
+ install/spawn path entirely."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ port, handler = stub_server
+ from hermes_cli.local_runtime import bootstrap
+ from hermes_cli.local_runtime.supervisor import state_path
+
+ monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
+ state_path().parent.mkdir(parents=True, exist_ok=True)
+ state_path().write_text(json.dumps({
+ "base_url": f"http://127.0.0.1:{port}/v1", "api_key": "k", "pid": os.getpid(),
+ }), encoding="utf-8")
+
+ called = []
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.ensure_runtime_installed",
+ lambda *a, **k: called.append(1))
+ assert bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}}) is None
+ assert called == []
+
+
+def test_bootstrap_failure_never_raises(tmp_path, monkeypatch):
+ """Session start must survive a broken runtime: failures log + return
+ None, chat falls back to configured providers."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli.local_runtime import bootstrap
+
+ monkeypatch.setattr(bootstrap, "_SUPERVISOR", None)
+ monkeypatch.setattr(bootstrap, "_detect_gpu_vendor", lambda: None)
+
+ def boom(*a, **k):
+ raise RuntimeError("no network")
+
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.ensure_runtime_installed", boom)
+ result = bootstrap.ensure_local_runtime({"local_runtime": {"enabled": True}})
+ assert result is None # no exception escaped
diff --git a/tests/hermes_cli/test_local_runtime_picker_row.py b/tests/hermes_cli/test_local_runtime_picker_row.py
new file mode 100644
index 0000000000..0ba2f1d880
--- /dev/null
+++ b/tests/hermes_cli/test_local_runtime_picker_row.py
@@ -0,0 +1,90 @@
+"""The llamacpp provider row in the model picker payload.
+
+Contract: staged local GGUFs appear as a selectable provider row in
+build_models_payload — the same payload /api/model/options and the desktop
+picker consume — whenever models are staged, without any credential."""
+
+from __future__ import annotations
+
+import pytest
+
+
+@pytest.fixture
+def hermes_home(tmp_path, monkeypatch):
+ home = tmp_path / ".hermes"
+ home.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ return home
+
+
+def _stage(home, *names):
+ mdir = home / "models"
+ mdir.mkdir(exist_ok=True)
+ for name in names:
+ (mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 64)
+
+
+def test_no_staged_models_no_row(hermes_home):
+ from hermes_cli.inventory import _local_runtime_row, load_picker_context
+
+ assert _local_runtime_row(load_picker_context()) is None
+
+
+def test_staged_models_make_a_selectable_row(hermes_home):
+ from hermes_cli.inventory import _local_runtime_row, load_picker_context
+
+ _stage(hermes_home, "Qwen3-4B-Instruct-2507-UD-Q8_K_XL", "Some-Other-Model")
+ row = _local_runtime_row(load_picker_context())
+ assert row is not None
+ assert row["slug"] == "llamacpp"
+ assert row["authenticated"] is True
+ assert "Qwen3-4B-Instruct-2507-UD-Q8_K_XL" in row["models"]
+ assert row["total_models"] == 2
+
+
+def test_row_marks_current_when_config_points_at_llamacpp(hermes_home):
+ from hermes_cli.inventory import _local_runtime_row, load_picker_context
+
+ _stage(hermes_home, "M")
+ ctx = load_picker_context().with_overrides(current_provider="llamacpp")
+ row = _local_runtime_row(ctx)
+ assert row is not None and row["is_current"] is True
+
+
+def test_full_payload_includes_local_row(hermes_home):
+ """Through the REAL payload builder — the shape the desktop picker eats."""
+ from hermes_cli.inventory import build_models_payload, load_picker_context
+
+ _stage(hermes_home, "Local-Model-X")
+ payload = build_models_payload(
+ load_picker_context(),
+ probe_custom_providers=False,
+ probe_current_custom_provider=False,
+ )
+ slugs = [p["slug"] for p in payload["providers"]]
+ assert "llamacpp" in slugs
+ row = payload["providers"][slugs.index("llamacpp")]
+ assert row["models"] == ["Local-Model-X"]
+
+
+def test_explicit_only_filter_keeps_local_row_on_any_profile(hermes_home):
+ """The desktop dropdown requests explicit_only=True, and the local row
+ has no config credential by design (credential is reachability). The
+ filter must treat staged models as explicit configuration — otherwise
+ the row only survives on the profile whose config points at llamacpp,
+ and every other profile's dropdown silently loses local models."""
+ from hermes_cli.inventory import _filter_explicit_provider_rows, _local_runtime_row, load_picker_context
+
+ _stage(hermes_home, "Qwen3.8-27B-UD-Q5_K_XL")
+ ctx = load_picker_context()
+ row = _local_runtime_row(ctx)
+ assert row is not None
+
+ # Simulate a profile whose current provider is a cloud one (the normal
+ # profile's shape): explicit-only filtering must keep the local row.
+ import dataclasses
+
+ ctx = dataclasses.replace(ctx, current_provider="anthropic")
+ kept = _filter_explicit_provider_rows([row], ctx)
+ assert kept, "explicit-only filter dropped the local-runtime row"
+ assert kept[0]["slug"] == "llamacpp"
diff --git a/tests/hermes_cli/test_local_runtime_updates.py b/tests/hermes_cli/test_local_runtime_updates.py
new file mode 100644
index 0000000000..f6f4d43ea2
--- /dev/null
+++ b/tests/hermes_cli/test_local_runtime_updates.py
@@ -0,0 +1,123 @@
+"""Engine-update contracts (Rollout 4 follow-up):
+
+- default tag flows from DEFAULT_CONFIG unless the user pinned;
+- boot serves what is INSTALLED, never downloads (the ladder);
+- update_available only when the local engine is enabled AND installed
+ AND the configured tag is missing on disk;
+- the update itself is a button-driven job, and prune keeps N-1.
+"""
+
+from __future__ import annotations
+
+import json
+
+import pytest
+
+
+@pytest.fixture
+def hermes_home(tmp_path, monkeypatch):
+ home = tmp_path / ".hermes"
+ home.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ return home
+
+
+def _install_fake_tag(home, tag: str, backend: str = "cuda") -> None:
+ d = home / "runtimes" / "llamacpp" / tag / backend
+ d.mkdir(parents=True)
+ (d / "manifest.json").write_text(json.dumps({
+ "tag": tag, "backend": backend, "assets": {},
+ "verified_version": f"version: {tag.lstrip('b')}",
+ }), encoding="utf-8")
+ # server_binary() looks for the executable name per-OS; give it both.
+ (d / "llama-server.exe").write_bytes(b"MZ fake")
+ (d / "llama-server").write_bytes(b"\x7fELF fake")
+
+
+def test_installed_tags_newest_first(hermes_home):
+ from hermes_cli.local_runtime.binaries import installed_tags
+
+ assert installed_tags() == []
+ _install_fake_tag(hermes_home, "b10290")
+ _install_fake_tag(hermes_home, "b10412")
+ assert installed_tags() == ["b10412", "b10290"]
+
+
+def test_default_tag_flows_from_default_config(hermes_home):
+ """Unpinned users inherit the Hermes-release default (deep-merge);
+ the shipped default must be a plausible rolling tag."""
+ from hermes_cli.config import load_config
+ from hermes_cli.config_defaults import DEFAULT_CONFIG
+
+ default_tag = DEFAULT_CONFIG["local_runtime"]["tag"]
+ assert default_tag.startswith("b") and default_tag.lstrip("b").isdigit()
+ assert load_config()["local_runtime"]["tag"] == default_tag
+
+
+def test_update_available_requires_enabled_and_installed(hermes_home, monkeypatch):
+ """The flag's truth table: enabled+installed+configured-missing only."""
+ from fastapi.testclient import TestClient
+
+ from hermes_cli import web_server
+
+ client = TestClient(web_server.app)
+ # Same auth pattern as the other local-models route tests.
+ client.headers[web_server._SESSION_HEADER_NAME] = web_server._SESSION_TOKEN
+
+ def status():
+ r = client.get("/api/local-models/status")
+ assert r.status_code == 200, r.text
+ return r.json()
+
+ import hermes_cli.web_routers.local_models as lm
+
+ # Case 1: enabled, configured newer than installed -> update available.
+ monkeypatch.setattr(lm, "_runtime_section",
+ lambda: {"enabled": True, "tag": "b10412"})
+ _install_fake_tag(hermes_home, "b10290")
+ s = status()
+ assert s["update_available"] is True
+ assert s["configured_tag"] == "b10412"
+ assert s["tag"] == "b10290" # serving what's installed
+
+ # Case 2: configured tag installed -> no update.
+ _install_fake_tag(hermes_home, "b10412")
+ s = status()
+ assert s["update_available"] is False
+ assert s["tag"] == "b10412"
+
+ # Case 3: disabled -> never flagged, even with a mismatch.
+ monkeypatch.setattr(lm, "_runtime_section",
+ lambda: {"enabled": False, "tag": "b10999"})
+ assert status()["update_available"] is False
+
+
+def test_boot_never_downloads_missing_tag(hermes_home, monkeypatch):
+ """The ladder: configured-but-not-installed serves the newest installed
+ tag; nothing installed means no boot (and NO download either way)."""
+ from hermes_cli.local_runtime import bootstrap
+
+ calls = []
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.binaries.ensure_runtime_installed",
+ lambda tag, backend, **kw: calls.append(tag) or (_ for _ in ()).throw(
+ AssertionError("boot must not reach install for missing tags")))
+
+ # Nothing installed: returns None before any install attempt.
+ cfg = {"local_runtime": {"enabled": True, "tag": "b10412"}}
+ assert bootstrap.ensure_local_runtime(cfg) is None
+ assert calls == []
+
+
+def test_prune_keeps_n_minus_one(hermes_home):
+ from hermes_cli.local_runtime.binaries import installed_tags, prune_old_tags
+
+ for tag in ("b10100", "b10200", "b10290"):
+ _install_fake_tag(hermes_home, tag)
+ prune_old_tags(["b10290", "b10200"])
+ assert installed_tags() == ["b10290", "b10200"]
+ # downloads/ cache dir must survive pruning when present.
+ downloads = hermes_home / "runtimes" / "llamacpp" / "downloads"
+ downloads.mkdir(exist_ok=True)
+ prune_old_tags(["b10290"])
+ assert downloads.exists()
diff --git a/tests/hermes_cli/test_local_server_lifecycle.py b/tests/hermes_cli/test_local_server_lifecycle.py
new file mode 100644
index 0000000000..7a2c5300ef
--- /dev/null
+++ b/tests/hermes_cli/test_local_server_lifecycle.py
@@ -0,0 +1,114 @@
+"""Server on/off lifecycle route (round-9 feedback: 'we should be able to
+completely turn off the local engine'). Contract: stop tears the server
+down AND persists enabled=false (durable, unlike eject); start persists
+enabled=true and boots."""
+
+from __future__ import annotations
+
+import pytest
+from fastapi.testclient import TestClient
+
+
+@pytest.fixture
+def client(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path / ".hermes"))
+ from hermes_cli import web_server
+
+ test_client = TestClient(web_server.app)
+ token = getattr(web_server, "_SESSION_TOKEN", "")
+ if token:
+ test_client.headers["Authorization"] = f"Bearer {token}"
+ return test_client
+
+
+def test_stop_disables_and_tears_down(client, monkeypatch):
+ stopped = {"called": False}
+
+ def _shutdown():
+ stopped["called"] = True
+
+ class _FakeSup:
+ pass
+
+ monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.get_supervisor",
+ lambda: _FakeSup())
+ monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.shutdown_local_runtime",
+ _shutdown)
+
+ r = client.post("/api/local-models/server", json={"action": "stop"})
+ assert r.status_code == 200
+ assert stopped["called"] is True
+
+ from hermes_cli.config import load_config
+
+ assert load_config()["local_runtime"]["enabled"] is False
+
+
+def test_start_enables_and_boots(client, monkeypatch):
+ booted = {"called": False}
+
+ class _FakeSup:
+ base_url = "http://127.0.0.1:18434/v1"
+
+ def _ensure(config, force=False):
+ booted["called"] = True
+ assert force is True
+ return _FakeSup()
+
+ monkeypatch.setattr("hermes_cli.local_runtime.bootstrap.ensure_local_runtime",
+ _ensure)
+
+ r = client.post("/api/local-models/server", json={"action": "start"})
+ assert r.status_code == 200
+ assert booted["called"] is True
+
+ from hermes_cli.config import load_config
+
+ assert load_config()["local_runtime"]["enabled"] is True
+
+
+def test_bogus_action_rejected(client):
+ r = client.post("/api/local-models/server", json={"action": "reboot"})
+ assert r.status_code == 400
+
+
+def test_status_reports_loaded_models_from_live_router(client, monkeypatch):
+ """Round-11 regression: the loaded-models read inside the status route
+ raised NameError (missing json import), the blanket except swallowed it,
+ and {} shipped as truth — 'Not in memory' on a machine with 30 GB of
+ VRAM in use. This test exercises the REAL route against a stub router
+ and demands the loaded set comes through."""
+ import http.server
+ import json as _json
+ import threading
+
+ class _Router(http.server.BaseHTTPRequestHandler):
+ def do_GET(self):
+ body = _json.dumps({"data": [
+ {"id": "m-loaded", "status": {"value": "loaded"}},
+ {"id": "m-loading", "status": {"value": "loading"}},
+ {"id": "m-cold", "status": {"value": "unloaded"}},
+ ]}).encode()
+ self.send_response(200)
+ self.send_header("Content-Length", str(len(body)))
+ self.end_headers()
+ self.wfile.write(body)
+
+ def log_message(self, *a):
+ pass
+
+ server = http.server.HTTPServer(("127.0.0.1", 0), _Router)
+ threading.Thread(target=server.serve_forever, daemon=True).start()
+ try:
+ port = server.server_address[1]
+ # Patch the ROUTE's binding: local_models binds _state_endpoint via
+ # from-import at module load, so patching the endpoint module's
+ # attribute never reaches the name the route actually calls.
+ monkeypatch.setattr(
+ "hermes_cli.web_routers.local_models._state_endpoint",
+ lambda: {"base_url": f"http://127.0.0.1:{port}/v1", "api_key": "k"})
+ payload = client.get("/api/local-models/status").json()
+ assert payload["server_running"] is True
+ assert payload["loaded_models"] == {"m-loaded": "loaded", "m-loading": "loading"}
+ finally:
+ server.shutdown()
diff --git a/tests/hermes_cli/test_loops.py b/tests/hermes_cli/test_loops.py
index 3a630c4206..a99f9f6a0b 100644
--- a/tests/hermes_cli/test_loops.py
+++ b/tests/hermes_cli/test_loops.py
@@ -443,6 +443,20 @@ class TestTickLifecycle:
decision = mgr.complete_tick("3 tests still failing")
assert decision["stopped"] is False
+ def test_until_judge_blocked_pauses(self, hermes_home):
+ """An unachievable stop condition pauses the loop instead of spinning to the tick budget."""
+ from hermes_cli.loops import LoopManager
+
+ mgr = LoopManager(session_id="t11b")
+ state = mgr.set("poll", interval_seconds=300, until="the deleted repo's CI is green")
+ state.next_due_at = time.time() - 1
+ mgr.fire_tick()
+ with patch("hermes_cli.goals.judge_goal", return_value=("blocked", "repo no longer exists", False, None, False)):
+ decision = mgr.complete_tick("The repository was deleted; there is no CI to watch.")
+ assert decision["stopped"] is True
+ assert decision["status"] == "paused"
+ assert "unachievable" in decision["message"]
+
def test_until_judge_error_fails_open(self, hermes_home):
from hermes_cli.loops import LoopManager
diff --git a/tests/hermes_cli/test_managed_vision_capability.py b/tests/hermes_cli/test_managed_vision_capability.py
new file mode 100644
index 0000000000..661976b88d
--- /dev/null
+++ b/tests/hermes_cli/test_managed_vision_capability.py
@@ -0,0 +1,185 @@
+"""Vision capability for managed local models answers from ground truth.
+
+Cloud capability catalogs have never heard of a local GGUF, so without a
+managed-runtime answer every local model reads as text-only: pasted images
+detour to a cloud auxiliary (a screenshot leaving a local-first machine)
+or fail outright. The lookup chain must consult the managed runtime
+between the user's config override and the cloud catalog."""
+
+from __future__ import annotations
+
+import pytest
+
+
+@pytest.fixture
+def hermes_home(tmp_path, monkeypatch):
+ home = tmp_path / ".hermes"
+ home.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ import importlib
+
+ import hermes_constants
+
+ importlib.reload(hermes_constants)
+ yield home
+ importlib.reload(hermes_constants)
+
+
+def _stage(home_root, name):
+ # Machine-scoped models dir (the shared root — tmp HERMES_HOME IS the root here).
+ from hermes_cli.local_runtime.bootstrap import models_dir
+
+ mdir = models_dir()
+ mdir.mkdir(parents=True, exist_ok=True)
+ (mdir / f"{name}.gguf").write_bytes(b"GGUF" + b"\x00" * 32)
+
+
+def test_not_ours_returns_none(hermes_home):
+ from hermes_cli.local_runtime.capabilities import managed_model_supports_vision
+
+ assert managed_model_supports_vision("gpt-4o") is None
+
+
+def test_catalog_vision_model_with_projector_on_disk(hermes_home):
+ """Staged catalog model with an mmproj present: True (server down —
+ the catalog + on-disk projector answer)."""
+ from hermes_cli.local_runtime.bootstrap import assets_dir
+ from hermes_cli.local_runtime.capabilities import managed_model_supports_vision
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ entry = next(e for e in CATALOG if e.mmproj is not None)
+ variant = entry.variants[-1]
+ _stage(hermes_home, variant.model_id)
+ adir = assets_dir()
+ adir.mkdir(parents=True, exist_ok=True)
+ (adir / entry.mmproj.local_name).write_bytes(b"GGUF mmproj")
+
+ assert managed_model_supports_vision(variant.model_id) is True
+
+
+def test_catalog_vision_model_missing_projector_is_blind(hermes_home):
+ """Same model, projector NOT on disk: False — it genuinely cannot see,
+ and claiming otherwise sends an image to a model that errors on it."""
+ from hermes_cli.local_runtime.capabilities import managed_model_supports_vision
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ entry = next(e for e in CATALOG if e.mmproj is not None)
+ variant = entry.variants[-1]
+ _stage(hermes_home, variant.model_id)
+
+ assert managed_model_supports_vision(variant.model_id) is False
+
+
+def test_live_props_beats_catalog(hermes_home, monkeypatch):
+ """A running child's modalities report wins over the catalog: the
+ server that will receive the image is the authority."""
+ import hermes_cli.local_runtime.capabilities as caps
+
+ from hermes_cli.local_runtime.catalog import CATALOG
+
+ entry = next(e for e in CATALOG if e.mmproj is not None)
+ variant = entry.variants[-1]
+ _stage(hermes_home, variant.model_id)
+ # Catalog would say False (no projector staged) — live props says True.
+ monkeypatch.setattr(caps, "_props_modalities", lambda mid: True)
+ assert caps.managed_model_supports_vision(variant.model_id) is True
+
+
+def test_lookup_chain_consults_managed_runtime(hermes_home, monkeypatch):
+ """_lookup_supports_vision: user override wins, then the managed
+ answer, and the cloud catalog is never reached for a managed model."""
+ import agent.image_routing as ir
+
+ monkeypatch.setattr(
+ "hermes_cli.local_runtime.capabilities.managed_model_supports_vision",
+ lambda mid: True)
+
+ def catalog_must_not_run(*a, **k):
+ raise AssertionError("cloud catalog consulted for a managed model")
+
+ monkeypatch.setattr("agent.models_dev.get_model_capabilities",
+ catalog_must_not_run)
+
+ got = ir._lookup_supports_vision("llamacpp", "Some-Local-Model", {})
+ assert got is True
+
+ # Explicit user override still outranks the managed answer.
+ cfg = {"model": {"provider": "llamacpp", "name": "Some-Local-Model",
+ "supports_vision": False}}
+ got = ir._lookup_supports_vision("llamacpp", "Some-Local-Model", cfg)
+ assert got is False
+
+
+def test_webp_transcodes_to_png_for_managed_provider(hermes_home, monkeypatch, tmp_path):
+ """A .webp attachment bound for the managed server must arrive as PNG:
+ llama.cpp's stb_image decoder has no WebP support and drops the part
+ SILENTLY — the model confabulates a description of an image it never
+ saw. Measured live: the same red square answered 'Red' as PNG and
+ 'Unseen' as WebP."""
+ pytest.importorskip("PIL")
+ import io
+
+ from PIL import Image
+
+ import agent.image_routing as ir
+
+ webp_path = tmp_path / "shot.webp"
+ img = Image.new("RGB", (32, 32), (255, 0, 0))
+ img.save(webp_path, format="WEBP")
+
+ monkeypatch.setattr("agent.auxiliary_client._runtime_main_value",
+ lambda k: {"provider": "llamacpp",
+ "base_url": ""}.get(k, ""))
+
+ data_url = ir._file_to_data_url(webp_path)
+ assert data_url is not None
+ assert data_url.startswith("data:image/png;base64,"), (
+ "webp must transcode to png for the managed server")
+
+
+def test_webp_passes_through_for_cloud_providers(hermes_home, monkeypatch, tmp_path):
+ """Cloud providers accept WebP natively — no transcode tax for them."""
+ pytest.importorskip("PIL")
+ from PIL import Image
+
+ import agent.image_routing as ir
+
+ webp_path = tmp_path / "shot.webp"
+ Image.new("RGB", (32, 32), (255, 0, 0)).save(webp_path, format="WEBP")
+
+ monkeypatch.setattr("agent.auxiliary_client._runtime_main_value",
+ lambda k: {"provider": "anthropic",
+ "base_url": ""}.get(k, ""))
+
+ data_url = ir._file_to_data_url(webp_path)
+ assert data_url is not None
+ assert data_url.startswith("data:image/webp;base64,")
+
+
+def test_vision_analyze_normalization_narrows_for_managed(hermes_home, monkeypatch, tmp_path):
+ """vision_analyze's native fast path embeds the image into conversation
+ history via _normalize_to_supported_image — for the managed server a
+ WebP must convert to PNG THERE too, or the tool path re-introduces the
+ silent-drop confabulation the attachment path just fixed."""
+ pytest.importorskip("PIL")
+ from PIL import Image
+
+ import tools.vision_tools as vt
+
+ webp_path = tmp_path / "img.webp"
+ Image.new("RGB", (32, 32), (255, 0, 0)).save(webp_path, format="WEBP")
+
+ monkeypatch.setattr("agent.auxiliary_client._runtime_main_value",
+ lambda k: {"provider": "llamacpp",
+ "base_url": ""}.get(k, ""))
+ out_path, mime, err = vt._normalize_to_supported_image(webp_path, "image/webp")
+ assert err is None
+ assert mime == "image/png", "managed server: webp must normalize to png"
+
+ # Cloud providers keep webp untouched.
+ monkeypatch.setattr("agent.auxiliary_client._runtime_main_value",
+ lambda k: {"provider": "anthropic",
+ "base_url": ""}.get(k, ""))
+ out_path, mime, err = vt._normalize_to_supported_image(webp_path, "image/webp")
+ assert err is None
+ assert mime == "image/webp"
diff --git a/tests/hermes_cli/test_model_catalog.py b/tests/hermes_cli/test_model_catalog.py
index b4d8e8a40a..3e9c1844ff 100644
--- a/tests/hermes_cli/test_model_catalog.py
+++ b/tests/hermes_cli/test_model_catalog.py
@@ -307,6 +307,30 @@ class TestProviderOverride:
assert result == [("override/model", "custom")]
+class TestRefreshCadence:
+ def test_default_ttl_is_twenty_minutes_and_legacy_hours_honoured(self):
+ from hermes_cli import model_catalog
+
+ with patch("hermes_cli.config.load_config", return_value={"model_catalog": {"ttl_minutes": 20}}):
+ assert model_catalog.refresh_interval_seconds() == 20 * 60
+ # A user-set legacy ttl_hours still wins while ttl_minutes sits at its default.
+ with patch("hermes_cli.config.load_config", return_value={"model_catalog": {"ttl_minutes": 20, "ttl_hours": 3}}):
+ assert model_catalog.refresh_interval_seconds() == 3 * 3600
+
+ def test_refresh_catalogs_forces_every_source(self):
+ from hermes_cli import model_catalog
+
+ with patch.object(model_catalog, "_load_catalog_config", return_value={
+ "enabled": True, "url": "http://master", "ttl_hours": 1.0, "providers": {},
+ }), patch.object(model_catalog, "get_catalog", return_value=_valid_manifest()) as gc, \
+ patch("hermes_cli.models.fetch_openrouter_models") as orm, \
+ patch("hermes_cli.models.fetch_nous_recommended_models") as nous:
+ assert model_catalog.refresh_catalogs() is True
+ gc.assert_called_once_with(force_refresh=True)
+ orm.assert_called_once_with(force_refresh=True)
+ nous.assert_called_once_with(force_refresh=True)
+
+
class TestIntegrationWithModelsModule:
"""Exercise the fallback paths via the real callers in hermes_cli.models."""
diff --git a/tests/hermes_cli/test_model_prefix_routing_87189.py b/tests/hermes_cli/test_model_prefix_routing_87189.py
new file mode 100644
index 0000000000..c4802d1910
--- /dev/null
+++ b/tests/hermes_cli/test_model_prefix_routing_87189.py
@@ -0,0 +1,148 @@
+"""Regression tests for vendor-prefix model routing and dict model.aliases (#87189).
+
+``--model nous/deepseek-v4-pro`` / ``--model ollama/qwen3.5:4b`` used to fall
+through provider auto-detection and be sent to the configured default provider
+(api.anthropic.com) with the prefixed name intact, producing HTTP 404. Dict
+entries under ``model.aliases`` (``localqwen: {model: ..., provider: ...}``)
+were silently dropped because only string values were parsed.
+"""
+
+import hermes_cli.models as models
+import hermes_cli.model_switch as model_switch
+
+
+class TestVendorPrefixRouting:
+ """detect_provider_for_model honors a ``vendor/model`` prefix for
+ providers the user actually configured in their ``providers:`` block."""
+
+ def test_configured_provider_prefix_routes_to_provider(self, monkeypatch):
+ monkeypatch.setattr(models, "_find_openrouter_slug", lambda _name: None)
+ monkeypatch.setattr(models, "_configured_provider_ids", lambda: {"nous"})
+ detected = models.detect_provider_for_model("nous/deepseek-v4-pro", "anthropic")
+ assert detected == ("nous", "deepseek-v4-pro")
+
+ def test_local_provider_prefix_routes_to_provider(self, monkeypatch):
+ monkeypatch.setattr(models, "_find_openrouter_slug", lambda _name: None)
+ monkeypatch.setattr(models, "_configured_provider_ids", lambda: {"ollama"})
+ detected = models.detect_provider_for_model("ollama/qwen3.5:4b", "anthropic")
+ assert detected == ("ollama", "qwen3.5:4b")
+
+ def test_unconfigured_builtin_vendor_prefix_not_rerouted(self, monkeypatch):
+ """Built-in vendor slugs keep catalog/default routing.
+
+ ``google/gemini-2.5-flash`` is aggregator-native: the web config
+ field expects it to switch to OpenRouter, not to the Gemini provider
+ (``TestDenormalizeProviderSwitch`` in test_web_server.py). With no
+ user-configured provider for the vendor, prefix routing must stay
+ out of the way even when the models.dev catalog is unavailable.
+ """
+ monkeypatch.setattr(models, "_find_openrouter_slug", lambda _name: None)
+ monkeypatch.setattr(models, "_configured_provider_ids", lambda: set())
+ detected = models.detect_provider_for_model(
+ "google/gemini-2.5-flash", "ollama-local"
+ )
+ assert detected is None
+
+ def test_configured_provider_wins_over_alias_canonicalization(self, monkeypatch):
+ """A user-named ``ollama`` block must not be rewritten to ``custom``."""
+ monkeypatch.setattr(models, "_find_openrouter_slug", lambda _name: None)
+ monkeypatch.setattr(models, "_configured_provider_ids", lambda: {"ollama"})
+ assert models._PROVIDER_ALIASES.get("ollama") == "custom" # precondition
+ detected = models.detect_provider_for_model("ollama/qwen3.5:4b", "anthropic")
+ assert detected == ("ollama", "qwen3.5:4b")
+
+ def test_provider_alias_prefix_canonicalized_when_configured(self, monkeypatch):
+ monkeypatch.setattr(models, "_find_openrouter_slug", lambda _name: None)
+ monkeypatch.setattr(models, "_configured_provider_ids", lambda: {"zai"})
+ detected = models.detect_provider_for_model("glm/glm-4.7", "anthropic")
+ assert detected == ("zai", "glm-4.7")
+
+ def test_unknown_vendor_prefix_still_unmatched(self, monkeypatch):
+ monkeypatch.setattr(models, "_find_openrouter_slug", lambda _name: None)
+ monkeypatch.setattr(models, "_configured_provider_ids", lambda: {"ollama"})
+ assert models.detect_provider_for_model("notaprovider/foo-model", "anthropic") is None
+
+ def test_openrouter_slug_still_wins_over_prefix_routing(self, monkeypatch):
+ """Aggregator-native slugs keep their existing OpenRouter routing."""
+ monkeypatch.setattr(
+ models, "_find_openrouter_slug", lambda _name: "deepseek/deepseek-chat"
+ )
+ monkeypatch.setattr(models, "_configured_provider_ids", lambda: {"deepseek"})
+ detected = models.detect_provider_for_model("deepseek/deepseek-chat", "anthropic")
+ assert detected == ("openrouter", "deepseek/deepseek-chat")
+
+ def test_bare_model_detection_unchanged(self, monkeypatch):
+ monkeypatch.setattr(models, "_find_openrouter_slug", lambda _name: None)
+ detected = models.detect_provider_for_model("deepseek-chat", "anthropic")
+ assert detected == ("deepseek", "deepseek-chat")
+
+
+class TestDictModelAliases:
+ """``model.aliases`` accepts dict entries with an explicit provider."""
+
+ def _load_with(self, monkeypatch, cfg):
+ monkeypatch.setattr("hermes_cli.config.load_config", lambda: cfg)
+ return model_switch._load_direct_aliases()
+
+ def test_dict_entry_with_explicit_provider(self, monkeypatch):
+ cfg = {
+ "model": {
+ "aliases": {
+ "localqwen": {"model": "qwen3.5:4b", "provider": "custom"},
+ },
+ },
+ }
+ aliases = self._load_with(monkeypatch, cfg)
+ da = aliases["localqwen"]
+ assert (da.model, da.provider) == ("qwen3.5:4b", "custom")
+
+ def test_dict_entry_with_base_url(self, monkeypatch):
+ cfg = {
+ "model": {
+ "aliases": {
+ "qwen": {
+ "model": "qwen3.5:4b",
+ "provider": "ollama",
+ "base_url": "http://localhost:11434/v1",
+ },
+ },
+ },
+ }
+ aliases = self._load_with(monkeypatch, cfg)
+ da = aliases["qwen"]
+ assert (da.model, da.provider, da.base_url) == (
+ "qwen3.5:4b", "ollama", "http://localhost:11434/v1",
+ )
+
+ def test_dict_entry_without_provider_uses_model_provider(self, monkeypatch):
+ cfg = {
+ "model": {
+ "provider": "openrouter",
+ "aliases": {"bare": {"model": "some-model"}},
+ },
+ }
+ aliases = self._load_with(monkeypatch, cfg)
+ da = aliases["bare"]
+ assert (da.model, da.provider) == ("some-model", "openrouter")
+
+ def test_string_entries_still_parse(self, monkeypatch):
+ cfg = {
+ "model": {
+ "aliases": {"ds-flash": "deepseek/deepseek-v4-flash"},
+ },
+ }
+ aliases = self._load_with(monkeypatch, cfg)
+ da = aliases["ds-flash"]
+ assert (da.model, da.provider) == ("deepseek-v4-flash", "deepseek")
+
+ def test_model_aliases_block_keeps_priority_over_model_aliases(self, monkeypatch):
+ cfg = {
+ "model_aliases": {
+ "shared": {"model": "from-top-block", "provider": "custom"},
+ },
+ "model": {
+ "aliases": {"shared": {"model": "from-nested", "provider": "ollama"}},
+ },
+ }
+ aliases = self._load_with(monkeypatch, cfg)
+ assert aliases["shared"].model == "from-top-block"
diff --git a/tests/hermes_cli/test_model_switch_persist_default.py b/tests/hermes_cli/test_model_switch_persist_default.py
index 11394c4222..b53c8913e1 100644
--- a/tests/hermes_cli/test_model_switch_persist_default.py
+++ b/tests/hermes_cli/test_model_switch_persist_default.py
@@ -52,6 +52,24 @@ class TestResolvePersistBehavior:
with _config({"model": {"persist_switch_by_default": True}}):
assert resolve_persist_behavior(False, False, explicit_provider="") is True
+ def test_first_pick_persists_then_session_only(self):
+ # #90235 / #86414: the ONE policy every surface (CLI, gateway, Desktop
+ # picker) defers to. With no default ever configured, the first pick
+ # persists (even with --provider, which is how the Desktop picker
+ # always sends it) so resolve_provider never falls through to a stray
+ # env key on restart. Once a default exists, a plain pick is
+ # session-only unless --global / persist_switch_by_default.
+ with _config({"model": {}}):
+ assert resolve_persist_behavior(False, False, explicit_provider="anthropic") is True
+ with _config({"model": ""}):
+ assert resolve_persist_behavior(False, False) is True
+ with _config({"model": {"default": "gpt-5.6", "provider": "openai-codex"}}):
+ assert resolve_persist_behavior(False, False, explicit_provider="openai-api") is False
+ assert resolve_persist_behavior(False, False) is False
+ assert resolve_persist_behavior(True, False, explicit_provider="openai-api") is True
+ with _config({"model": "gpt-5.6"}):
+ assert resolve_persist_behavior(False, False) is False
+
# ---------------------------------------------------------------------------
# helper
diff --git a/tests/hermes_cli/test_nous_policy_filter.py b/tests/hermes_cli/test_nous_policy_filter.py
new file mode 100644
index 0000000000..948885603f
--- /dev/null
+++ b/tests/hermes_cli/test_nous_policy_filter.py
@@ -0,0 +1,246 @@
+"""Narrowing the Nous model lists to an org's policy.
+
+The gateway omits policy-blocked rows from an authenticated ``GET /v1/models``,
+so that response's keys are the reachable set.
+"""
+
+from __future__ import annotations
+
+import base64
+import json
+
+import pytest
+
+import hermes_cli.models as models_mod
+import hermes_cli.nous_account as account_mod
+from hermes_cli.models import (
+ _NOUS_POLICY_APPEND_MAX,
+ nous_policy_allowed_ids,
+ restrict_to_nous_policy,
+)
+from hermes_cli.nous_account import nous_policy_present
+
+
+def _jwt(claims: dict) -> str:
+ def seg(obj):
+ raw = json.dumps(obj).encode()
+ return base64.urlsafe_b64encode(raw).rstrip(b"=").decode()
+
+ return f"{seg({'alg': 'RS256'})}.{seg(claims)}.sig"
+
+
+class TestRestrictToNousPolicy:
+ def test_none_leaves_the_list_untouched(self):
+ ids = ["a/one", "b/two"]
+ assert restrict_to_nous_policy(ids, None) == ids
+
+ def test_empty_set_leaves_the_list_untouched(self):
+ """Empty is a failed read, not an org that may reach nothing."""
+ ids = ["a/one", "b/two"]
+ assert restrict_to_nous_policy(ids, set()) == ids
+
+ def test_drops_ids_outside_the_policy(self):
+ assert restrict_to_nous_policy(
+ ["a/one", "b/two", "c/three"], {"a/one", "c/three"}
+ ) == ["a/one", "c/three"]
+
+ def test_preserves_curated_order(self):
+ curated = ["z/last", "a/first", "m/middle"]
+ allowed = {"a/first", "m/middle", "z/last"}
+ assert restrict_to_nous_policy(curated, allowed) == curated
+
+ def test_keeps_a_free_sibling_when_its_base_is_reachable(self):
+ """Portal free recommendations are ``:free`` ids."""
+ assert restrict_to_nous_policy(["vendor/model:free"], {"vendor/model"}) == [
+ "vendor/model:free"
+ ]
+
+ def test_keeps_a_free_id_listed_in_its_own_right(self):
+ assert restrict_to_nous_policy(
+ ["vendor/model:free"], {"vendor/model:free"}
+ ) == ["vendor/model:free"]
+
+ def test_drops_a_free_sibling_whose_base_is_blocked(self):
+ assert restrict_to_nous_policy(["vendor/model:free"], {"other/model"}) == []
+
+
+class TestNousPolicyAllowedIds:
+ @pytest.fixture(autouse=True)
+ def _clear_cache(self):
+ models_mod._pricing_cache.clear()
+ models_mod._pricing_cache_retry_after.clear()
+ yield
+ models_mod._pricing_cache.clear()
+ models_mod._pricing_cache_retry_after.clear()
+
+ def _patch(self, monkeypatch, *, policy_present, api_key="sk-test", pricing=None):
+ calls = []
+ monkeypatch.setattr(
+ account_mod, "nous_policy_present", lambda: policy_present
+ )
+ monkeypatch.setattr(
+ models_mod,
+ "_resolve_nous_pricing_credentials",
+ lambda: (api_key, "https://inference.example.com"),
+ )
+
+ def _fake_fetch(**kwargs):
+ calls.append(kwargs)
+ return pricing if pricing is not None else {}
+
+ monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
+ return calls
+
+ def test_returns_the_authenticated_catalog_keys(self, monkeypatch):
+ calls = self._patch(
+ monkeypatch,
+ policy_present=True,
+ pricing={"a/one": {}, "b/two": {}},
+ )
+ assert nous_policy_allowed_ids() == {"a/one", "b/two"}
+ assert len(calls) == 1
+ assert calls[0]["api_key"] == "sk-test"
+
+ def test_declines_to_filter_an_unrestricted_org(self, monkeypatch):
+ calls = self._patch(monkeypatch, policy_present=False, pricing={"a/one": {}})
+ assert nous_policy_allowed_ids() is None
+ assert calls == [], "an unrestricted org should not pay for the read"
+
+ def test_declines_to_filter_when_the_claim_is_unknown(self, monkeypatch):
+ """Absent is an older mint, not an unrestricted org."""
+ calls = self._patch(monkeypatch, policy_present=None, pricing={"a/one": {}})
+ assert nous_policy_allowed_ids() is None
+ assert calls == []
+
+ def test_declines_to_filter_on_an_anonymous_read(self, monkeypatch):
+ """An anonymous read returns the full, unfiltered catalog."""
+ self._patch(monkeypatch, policy_present=True, api_key="", pricing={"a/one": {}})
+ assert nous_policy_allowed_ids() is None
+
+ def test_declines_to_filter_on_an_empty_read(self, monkeypatch):
+ """A failed fetch must not read as an org that may reach nothing."""
+ self._patch(monkeypatch, policy_present=True, pricing={})
+ assert nous_policy_allowed_ids() is None
+
+
+class TestNousPolicyPresent:
+ def _patch_token(self, monkeypatch, token):
+ import hermes_cli.auth as auth_mod
+
+ monkeypatch.setattr(
+ auth_mod,
+ "get_provider_auth_state",
+ lambda _p: {"access_token": token} if token is not None else {},
+ )
+
+ @pytest.mark.parametrize("claim,expected", [(True, True), (False, False)])
+ def test_reads_the_claim(self, monkeypatch, claim, expected):
+ self._patch_token(monkeypatch, _jwt({"policy_present": claim}))
+ assert nous_policy_present() is expected
+
+ def test_absent_claim_is_unknown_not_false(self, monkeypatch):
+ self._patch_token(monkeypatch, _jwt({"org_id": "org_1"}))
+ assert nous_policy_present() is None
+
+ def test_non_boolean_claim_is_unknown(self, monkeypatch):
+ self._patch_token(monkeypatch, _jwt({"policy_present": "yes"}))
+ assert nous_policy_present() is None
+
+ def test_no_token_is_unknown(self, monkeypatch):
+ self._patch_token(monkeypatch, None)
+ assert nous_policy_present() is None
+
+ def test_undecodable_token_is_unknown(self, monkeypatch):
+ self._patch_token(monkeypatch, "not-a-jwt")
+ assert nous_policy_present() is None
+
+
+class TestNousPolicyNotice:
+
+ def _patch(self, monkeypatch, present):
+ monkeypatch.setattr(account_mod, "nous_policy_present", lambda: present)
+
+ def test_shows_a_line_for_a_governed_org(self, monkeypatch):
+ self._patch(monkeypatch, True)
+ assert "restricts which models" in account_mod.nous_policy_notice(removed=True)
+
+ @pytest.mark.parametrize("present", [False, None])
+ def test_silent_otherwise(self, monkeypatch, present):
+ """Absent is an older mint, not an unrestricted org."""
+ self._patch(monkeypatch, present)
+ assert account_mod.nous_policy_notice(removed=True) == ""
+
+ def test_silent_when_the_filter_removed_nothing(self, monkeypatch):
+ """The catalog read fails open, so a governed org can still end up with
+ a full list — saying it was filtered would be false."""
+ self._patch(monkeypatch, True)
+ assert account_mod.nous_policy_notice(removed=False) == ""
+
+ def test_names_no_models(self, monkeypatch):
+ """The blocked set is most of the catalog under an allowlist."""
+ self._patch(monkeypatch, True)
+ notice = account_mod.nous_policy_notice(removed=True)
+ assert "/" not in notice, f"looks like it names a model: {notice}"
+ assert len(notice.splitlines()) == 1
+
+
+class TestAllowlistOutsideTheCuratedList:
+ """An allowlist can name only models the curated manifest lacks, which
+ intersecting alone turns into an empty picker."""
+
+ def test_surfaces_an_allowed_model_the_curated_list_lacks(self):
+ assert restrict_to_nous_policy(
+ ["vendor/a", "vendor/b"], {"amazon/nova-2-lite-v1"}, rescue_empty=True
+ ) == ["amazon/nova-2-lite-v1"]
+
+ def test_does_not_append_when_the_curated_overlap_is_non_empty(self):
+ kept = restrict_to_nous_policy(
+ ["z/curated", "a/curated"],
+ {"z/curated", "a/curated", "new/model"},
+ rescue_empty=True,
+ )
+ assert kept == ["z/curated", "a/curated"]
+
+ def test_does_not_rescue_a_catalog_sized_allowed_set(self):
+ """Past the cap the set reads as a whole catalog, and dumping it would
+ bury the curated order the pickers show on purpose."""
+ oversized = {f"cn/model-{i}" for i in range(_NOUS_POLICY_APPEND_MAX + 1)}
+ assert restrict_to_nous_policy(["vendor/one"], oversized, rescue_empty=True) == []
+
+
+class TestRescueIsOptIn:
+ """The rescue is meaningful only for the list a user picks from."""
+
+ def test_rescue_only_when_asked(self):
+ assert restrict_to_nous_policy([], {"a/one"}, rescue_empty=True) == ["a/one"]
+
+ def test_an_already_empty_unavailable_list_is_never_filled(self):
+ """A paid tier has no gated models, so this list is legitimately
+ empty — not a filter result to rescue."""
+ reachable = {f"cn/model-{i}" for i in range(42)}
+ assert restrict_to_nous_policy([], reachable) == []
+
+
+class TestPolicyRunsBeforeTierSplit:
+ """A rescued id must still pass the free/paid predicate.
+
+ Rescuing after the tier split put paid models back into a free-tier user's
+ selectable list, and the same id into both lists at once.
+ """
+
+ def test_a_rescued_paid_model_stays_unavailable_for_a_free_tier_user(self):
+ from hermes_cli.models import partition_nous_models_by_tier
+
+ pricing = {
+ "vendor/free": {"prompt": "0", "completion": "0"},
+ "vendor/paid": {"prompt": "0.000002", "completion": "0.00001"},
+ }
+ narrowed = restrict_to_nous_policy(
+ list(pricing), {"vendor/paid"}, rescue_empty=True
+ )
+ selectable, unavailable = partition_nous_models_by_tier(
+ narrowed, pricing, free_tier=True
+ )
+
+ assert selectable == []
+ assert unavailable == ["vendor/paid"]
diff --git a/tests/hermes_cli/test_nous_policy_surfaces.py b/tests/hermes_cli/test_nous_policy_surfaces.py
new file mode 100644
index 0000000000..04146a617f
--- /dev/null
+++ b/tests/hermes_cli/test_nous_policy_surfaces.py
@@ -0,0 +1,309 @@
+"""Every Nous model list is narrowed to the org's policy before it is shown.
+
+Four surfaces build their list from the curated manifest unioned with the
+Portal's ``recommended-models`` endpoint; neither source is authenticated.
+"""
+
+from __future__ import annotations
+
+import argparse
+
+import pytest
+
+import hermes_cli.models as models_mod
+
+CURATED = ["vendor/allowed", "vendor/blocked"]
+ALLOWED = {"vendor/allowed"}
+
+
+@pytest.fixture
+def policy(monkeypatch):
+ """An org whose policy admits only ``vendor/allowed``."""
+ monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: ALLOWED)
+ return ALLOWED
+
+
+@pytest.fixture
+def no_policy(monkeypatch):
+ """An unrestricted org — lists must come through untouched."""
+ monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: None)
+
+
+class TestLoginNous:
+
+ def _run(self, monkeypatch, tmp_path):
+ import hermes_cli.auth as auth_mod
+ import hermes_cli.nous_subscription as ns
+
+ seen: dict = {}
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ monkeypatch.setattr(
+ auth_mod,
+ "_nous_device_code_login",
+ lambda **_k: {
+ "access_token": "tok",
+ "agent_key": "key",
+ "inference_base_url": "https://inference.example.com",
+ "portal_base_url": "https://portal.example.com",
+ "refresh_token": "r",
+ "token_expires_at": 9999999999,
+ },
+ )
+ monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED))
+ monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
+ monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
+ monkeypatch.setattr(
+ models_mod,
+ "union_with_portal_paid_recommendations",
+ lambda ids, pricing, _portal: (list(ids), pricing),
+ )
+ monkeypatch.setattr(ns, "prompt_enable_tool_gateway", lambda _c: None)
+
+ def _capture(model_ids, **kwargs):
+ seen["model_ids"] = list(model_ids)
+ return None
+
+ monkeypatch.setattr(auth_mod, "_prompt_model_selection", _capture)
+
+ args = argparse.Namespace(
+ portal_url=None, inference_url=None, client_id=None, scope=None,
+ no_browser=True, timeout=15.0, ca_bundle=None, insecure=False,
+ )
+ auth_mod._login_nous(args, auth_mod.PROVIDER_REGISTRY["nous"])
+ return seen
+
+ def test_hidden_model_is_not_offered(self, monkeypatch, tmp_path, policy):
+ assert self._run(monkeypatch, tmp_path).get("model_ids") == ["vendor/allowed"]
+
+ def test_unrestricted_org_sees_the_full_curated_list(
+ self, monkeypatch, tmp_path, no_policy
+ ):
+ assert self._run(monkeypatch, tmp_path).get("model_ids") == CURATED
+
+
+class TestModelSwitchPicker:
+ """The ``/model`` picker's nous branch (``list_authenticated_providers``)."""
+
+ def _rows(self, monkeypatch):
+ import hermes_cli.auth as auth_mod
+ import hermes_cli.model_switch as ms
+
+ monkeypatch.setattr(
+ auth_mod,
+ "_load_auth_store",
+ lambda *a, **k: {"providers": {"nous": {"access_token": "tok"}}},
+ )
+ monkeypatch.setattr(models_mod, "get_curated_nous_model_ids", lambda: list(CURATED))
+ monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
+ monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
+ monkeypatch.setattr(
+ models_mod,
+ "union_with_portal_paid_recommendations",
+ lambda ids, pricing, _portal: (list(ids), pricing),
+ )
+ rows = ms.list_authenticated_providers(max_models=10)
+ return next((r for r in rows if r["slug"] == "nous"), None)
+
+ def test_hidden_model_is_filtered(self, monkeypatch, policy):
+ row = self._rows(monkeypatch)
+ assert row is not None, "nous row should be listed"
+ assert "vendor/blocked" not in row["models"]
+ assert "vendor/allowed" in row["models"]
+
+ def test_unrestricted_org_keeps_both(self, monkeypatch, no_policy):
+ row = self._rows(monkeypatch)
+ assert row is not None
+ assert set(CURATED) <= set(row["models"])
+
+ def test_filter_survives_a_failed_recommendation_fetch(self, monkeypatch, policy):
+ """The filter sits outside the try wrapping the Portal union."""
+
+ def _boom(_p):
+ raise RuntimeError("portal down")
+
+ monkeypatch.setattr(models_mod, "get_pricing_for_provider", _boom)
+ row = self._rows(monkeypatch)
+ assert row is not None
+ assert "vendor/blocked" not in row["models"]
+
+
+class TestRecommendedDefaultEndpoint:
+ """This endpoint picks a model the user never sees chosen."""
+
+ def _call(self, monkeypatch):
+ import hermes_cli.auth as auth_mod
+ from hermes_cli.web_server import get_recommended_default_model
+
+ # Blocked first, so an unfiltered list would make it the silent
+ # default — otherwise this passes whether or not the filter runs.
+ monkeypatch.setattr(
+ models_mod, "get_curated_nous_model_ids",
+ lambda: ["vendor/blocked", "vendor/allowed"],
+ )
+ monkeypatch.setattr(models_mod, "get_pricing_for_provider", lambda _p: {})
+ monkeypatch.setattr(models_mod, "check_nous_free_tier", lambda **_k: None)
+ monkeypatch.setattr(
+ models_mod,
+ "union_with_portal_paid_recommendations",
+ lambda ids, pricing, _portal: (list(ids), pricing),
+ )
+ monkeypatch.setattr(auth_mod, "get_provider_auth_state", lambda _p: {})
+ return get_recommended_default_model(provider="nous")
+
+ def test_hidden_model_is_never_the_silent_default(self, monkeypatch, policy):
+ assert self._call(monkeypatch)["model"] == "vendor/allowed"
+
+ def test_unrestricted_org_is_unaffected(self, monkeypatch, no_policy):
+ assert self._call(monkeypatch)["model"] == "vendor/blocked"
+
+
+class TestAuxiliaryFastModel:
+ """``_fast_model_from_catalog`` uses the catalog's keys as a source of ids."""
+
+ def _pick(self, monkeypatch, *, catalog):
+ import agent.auxiliary_client as aux
+
+ seen: dict = {}
+
+ def _fake_fetch(*, api_key=None, base_url="", timeout=8.0, **_k):
+ seen["api_key"] = api_key
+ return {mid: {} for mid in catalog}
+
+ monkeypatch.setattr(
+ models_mod, "_resolve_nous_pricing_credentials",
+ lambda: ("sk-nous", "https://inference.example.com"),
+ )
+ monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
+ picked = aux._fast_model_from_catalog("nous")
+ return picked, seen
+
+ def test_reads_the_catalog_with_nous_oauth_credentials(self, monkeypatch, no_policy):
+ """The api-key resolver raises for OAuth providers."""
+ _, seen = self._pick(monkeypatch, catalog=["vendor/haiku-fast"])
+ assert seen["api_key"] == "sk-nous"
+
+ def test_hidden_model_is_not_selected(self, monkeypatch, policy):
+ import agent.auxiliary_client as aux
+
+ monkeypatch.setattr(
+ models_mod, "nous_policy_allowed_ids", lambda **_k: {"vendor/allowed"}
+ )
+ monkeypatch.setattr(aux, "_FAST_MODEL_FAMILIES", ("vendor/",))
+ monkeypatch.setattr(aux, "_FAST_MODEL_EXCLUDE", ())
+ picked, _ = self._pick(
+ monkeypatch, catalog=["vendor/blocked", "vendor/allowed"]
+ )
+ assert picked == "vendor/allowed"
+
+
+class TestNousPrefetch:
+ """The nous disk-cache entry is write-only, so prefetching it is a round
+ trip for nothing."""
+
+ def test_nous_is_not_collected_for_prefetch(self, monkeypatch):
+ import hermes_cli.auth as auth_mod
+ import hermes_cli.model_switch as ms
+
+ monkeypatch.setattr(
+ auth_mod, "_load_auth_store",
+ lambda *a, **k: {"providers": {"nous": {"access_token": "tok"}}},
+ )
+ slugs = ms._collect_authed_provider_slugs({}, {"nous": list(CURATED)}, [])
+ assert "nous" not in slugs
+
+
+class TestPolicyNoticeIsShown:
+
+ def test_login_prints_it(self, monkeypatch, tmp_path, policy, capsys):
+ import hermes_cli.nous_account as account_mod
+
+ monkeypatch.setattr(account_mod, "nous_policy_present", lambda: True)
+ TestLoginNous()._run(monkeypatch, tmp_path)
+ assert "restricts which models" in capsys.readouterr().out
+
+ def test_login_silent_for_an_ungoverned_org(
+ self, monkeypatch, tmp_path, no_policy, capsys
+ ):
+ import hermes_cli.nous_account as account_mod
+
+ monkeypatch.setattr(account_mod, "nous_policy_present", lambda: False)
+ TestLoginNous()._run(monkeypatch, tmp_path)
+ assert "restricts which models" not in capsys.readouterr().out
+
+
+class TestAuxFallbackRespectsPolicy:
+ """Steps 2-4 of the aux ladder are policy-blind: `resolve_aux_model` queries
+ a public recommendation and the rest are hardcoded."""
+
+ def _patch(self, monkeypatch, *, allowed, recommended):
+ import agent.auxiliary_client as aux
+ import providers
+
+ monkeypatch.setattr(models_mod, "nous_policy_allowed_ids", lambda **_k: allowed)
+ monkeypatch.setattr(
+ models_mod, "_resolve_nous_pricing_credentials",
+ lambda: ("sk", "https://inference.example.com"),
+ )
+ # No fast-family match, so the catalog step yields nothing.
+ monkeypatch.setattr(
+ models_mod, "fetch_models_with_pricing",
+ lambda **_k: {"vendor/allowed-large": {}},
+ )
+
+ class _Profile:
+ default_aux_model = ""
+
+ def resolve_aux_model(self, **_k):
+ return recommended
+
+ monkeypatch.setattr(providers, "get_provider_profile", lambda _p: _Profile())
+ return aux
+
+ def test_blocked_recommendation_is_not_used(self, monkeypatch):
+ aux = self._patch(
+ monkeypatch, allowed={"vendor/allowed-large"},
+ recommended="vendor/blocked-haiku",
+ )
+ assert aux._get_aux_model_for_provider("nous", prefer_fast=True) == ""
+
+ def test_allowed_recommendation_still_used(self, monkeypatch):
+ aux = self._patch(
+ monkeypatch, allowed={"vendor/allowed-large", "vendor/ok-haiku"},
+ recommended="vendor/ok-haiku",
+ )
+ assert (
+ aux._get_aux_model_for_provider("nous", prefer_fast=True)
+ == "vendor/ok-haiku"
+ )
+
+ def test_ungoverned_org_is_unaffected(self, monkeypatch):
+ aux = self._patch(
+ monkeypatch, allowed=None, recommended="vendor/anything"
+ )
+ assert (
+ aux._get_aux_model_for_provider("nous", prefer_fast=True)
+ == "vendor/anything"
+ )
+
+
+def test_titling_seeds_the_shared_catalog_entry_like_the_pickers(monkeypatch):
+ """The aux catalog read shares the pickers' cache entry, so seeding it
+ without the Nous-only arguments costs the picker its sale chrome and leaves
+ the policy catalog with no expiry."""
+ import agent.auxiliary_client as aux
+
+ monkeypatch.setattr(
+ models_mod, "_resolve_nous_pricing_credentials",
+ lambda: ("tok", "https://inference.example.com"),
+ )
+ seen: dict = {}
+
+ def _fake_fetch(**kwargs):
+ seen.update(kwargs)
+ return {"vendor/haiku": {}}
+
+ monkeypatch.setattr(models_mod, "fetch_models_with_pricing", _fake_fetch)
+ aux._fast_model_from_catalog("nous")
+
+ assert seen.get("include_sale_original") is True
+ assert seen.get("cache_ttl_seconds") == models_mod._NOUS_CATALOG_TTL_SECONDS
diff --git a/tests/hermes_cli/test_nous_subscription.py b/tests/hermes_cli/test_nous_subscription.py
index d9f71a9af1..c9ffaa931a 100644
--- a/tests/hermes_cli/test_nous_subscription.py
+++ b/tests/hermes_cli/test_nous_subscription.py
@@ -58,6 +58,52 @@ def test_get_nous_subscription_features_recognizes_direct_exa_backend(monkeypatc
assert features.web.current_provider == "exa"
+def test_get_nous_subscription_features_recognizes_keyless_tavily_backend(monkeypatch):
+ """Selecting Tavily in setup/tools counts as available with no API key.
+
+ Mirrors tools.web_tools._is_backend_available('tavily'): keyless is
+ opt-in via web.backend / search_backend / extract_backend, not a
+ silent empty-install default. The setup summary previously required
+ TAVILY_API_KEY and printed a false 'missing' after a skipped key prompt.
+ """
+ monkeypatch.setattr(ns, "get_env_value", lambda name: "")
+ monkeypatch.setattr(
+ ns, "get_nous_portal_account_info", lambda: _account(logged_in=False)
+ )
+ monkeypatch.setattr(ns, "_toolset_enabled", lambda config, key: key == "web")
+ monkeypatch.setattr(ns, "_has_agent_browser", lambda: False)
+ monkeypatch.setattr(ns, "resolve_openai_audio_api_key", lambda: "")
+ monkeypatch.setattr(ns, "has_direct_modal_credentials", lambda: False)
+
+ features = ns.get_nous_subscription_features({"web": {"backend": "tavily"}})
+
+ assert features.web.available is True
+ assert features.web.active is True
+ assert features.web.managed_by_nous is False
+ assert features.web.direct_override is True
+ assert features.web.current_provider == "tavily"
+ assert features.web.explicit_configured is True
+
+
+def test_keyless_tavily_search_backend_without_shared_backend(monkeypatch):
+ monkeypatch.setattr(ns, "get_env_value", lambda name: "")
+ monkeypatch.setattr(
+ ns, "get_nous_portal_account_info", lambda: _account(logged_in=False)
+ )
+ monkeypatch.setattr(ns, "_toolset_enabled", lambda config, key: key == "web")
+ monkeypatch.setattr(ns, "_has_agent_browser", lambda: False)
+ monkeypatch.setattr(ns, "resolve_openai_audio_api_key", lambda: "")
+ monkeypatch.setattr(ns, "has_direct_modal_credentials", lambda: False)
+
+ features = ns.get_nous_subscription_features(
+ {"web": {"search_backend": "tavily"}}
+ )
+
+ assert features.web.available is True
+ assert features.web.active is True
+ assert features.web.current_provider == "tavily"
+
+
def test_unconfigured_web_without_keys_is_unavailable(monkeypatch):
monkeypatch.setattr(ns, "get_env_value", lambda name: "")
monkeypatch.setattr(
diff --git a/tests/hermes_cli/test_platform_actions.py b/tests/hermes_cli/test_platform_actions.py
index c32bdad9cd..8a7d66e05a 100644
--- a/tests/hermes_cli/test_platform_actions.py
+++ b/tests/hermes_cli/test_platform_actions.py
@@ -39,6 +39,17 @@ def _runner_with(adapters: dict):
return patch("gateway.run._gateway_runner_ref", lambda: runner)
+def _multiplex_runner_with(*, default: dict, profiles: dict, active_profile: str = "default"):
+ """A runner using the REAL GatewayAuthorizationMixin resolution ladder."""
+ from gateway.authz_mixin import GatewayAuthorizationMixin
+
+ runner = GatewayAuthorizationMixin.__new__(GatewayAuthorizationMixin)
+ runner.adapters = default
+ runner._profile_adapters = profiles
+ runner._active_profile_name = lambda: active_profile
+ return patch("gateway.run._gateway_runner_ref", lambda: runner)
+
+
def _telegram_adapter(connected=True):
a = MagicMock()
a.platform = Platform.TELEGRAM
@@ -264,6 +275,49 @@ class TestVerbRouting:
assert result["error"] == "invalid_argument"
+class TestMultiplexProfileRouting:
+ """A plugin acting during a secondary profile's turn must act through THAT
+ profile's adapter, never the default profile's — the fail-closed contract
+ of GatewayAuthorizationMixin._authorization_adapter (#85245)."""
+
+ def test_secondary_profile_routes_to_its_own_adapter_not_default(self):
+ actions = PlatformActions("p")
+ default_adapter = _telegram_adapter()
+ team_b_adapter = _telegram_adapter()
+ with (
+ _grant(True),
+ _multiplex_runner_with(
+ default={Platform.TELEGRAM: default_adapter},
+ profiles={"team-b": {Platform.TELEGRAM: team_b_adapter}},
+ ),
+ patch("hermes_cli.profiles.get_active_profile_name", return_value="team-b"),
+ ):
+ result = asyncio.run(actions.add_reaction("telegram", "1", "2", "x"))
+ assert result["ok"] is True
+ team_b_adapter._set_reaction.assert_awaited_once()
+ default_adapter._set_reaction.assert_not_awaited()
+
+ @pytest.mark.parametrize(
+ "resolver",
+ [
+ {"return_value": "team-b"}, # stamped profile, no registry entry
+ {"side_effect": RuntimeError("boom")}, # profile resolution itself fails
+ ],
+ ids=["no-registry-entry", "resolution-error"],
+ )
+ def test_unresolvable_profile_fails_closed_never_default_bot(self, resolver):
+ actions = PlatformActions("p")
+ default_adapter = _telegram_adapter()
+ with (
+ _grant(True),
+ _multiplex_runner_with(default={Platform.TELEGRAM: default_adapter}, profiles={}),
+ patch("hermes_cli.profiles.get_active_profile_name", **resolver),
+ ):
+ result = asyncio.run(actions.add_reaction("telegram", "1", "2", "x"))
+ assert result["error"] == "adapter_not_registered"
+ default_adapter._set_reaction.assert_not_awaited()
+
+
class TestPluginContextWiring:
def test_ctx_platform_actions_bound_to_plugin_id(self):
from hermes_cli.plugins import PluginContext, PluginManager, PluginManifest
diff --git a/tests/hermes_cli/test_plugins.py b/tests/hermes_cli/test_plugins.py
index 4f4f67be0c..d44abafb4f 100644
--- a/tests/hermes_cli/test_plugins.py
+++ b/tests/hermes_cli/test_plugins.py
@@ -933,6 +933,13 @@ class TestDeliveryParity:
class TestForceReloadSymmetry:
"""Force rediscovery restores non-plugin state it wiped (#64178)."""
+ @pytest.fixture(autouse=True)
+ def _cleanup_shell_hook_registry(self):
+ yield
+ import agent.shell_hooks as shell_hooks_mod
+
+ shell_hooks_mod.reset_for_tests()
+
def test_force_reload_re_registers_shell_hooks(self, monkeypatch):
"""config.yaml shell hooks are re-wired after force=True (#60036)."""
calls = []
@@ -992,6 +999,7 @@ class TestForceReloadSymmetry:
def test_re_register_config_hooks_clears_idempotence_set(self, monkeypatch):
import agent.shell_hooks as shell_hooks_mod
+ from hermes_constants import get_hermes_home
recorded = {}
monkeypatch.setattr(
@@ -1002,8 +1010,9 @@ class TestForceReloadSymmetry:
monkeypatch.setattr(
"hermes_cli.config.load_config", lambda: {"hooks": {}}
)
+ home_key = str(get_hermes_home().expanduser().resolve())
with shell_hooks_mod._registered_lock:
- shell_hooks_mod._registered.add(("post_llm_call", None, "echo hi"))
+ shell_hooks_mod._registered.add((home_key, "post_llm_call", None, "echo hi"))
shell_hooks_mod.re_register_config_hooks()
@@ -1048,7 +1057,7 @@ class TestForceReloadSymmetry:
assert started.wait(timeout=1.0)
assert results == [{"ok": True}]
- assert elapsed < 1.0, f"caller blocked for {elapsed:.2f}s after timeout"
+ assert elapsed < 5.0, f"caller blocked for {elapsed:.2f}s after timeout"
hold.set()
def test_hook_callback_within_timeout_returns_value(self, monkeypatch):
@@ -1132,7 +1141,7 @@ class TestForceReloadSymmetry:
elapsed = time.monotonic() - t0
assert len(starts) == 1
- assert elapsed < 1.0
+ assert elapsed < 5.0
hold.set()
def test_pre_tool_call_timeout_fail_closed(self, monkeypatch):
@@ -1166,7 +1175,7 @@ class TestForceReloadSymmetry:
elapsed = time.monotonic() - t0
assert msg == _PRE_TOOL_CALL_TIMEOUT_BLOCK_MESSAGE
- assert elapsed < 1.0
+ assert elapsed < 5.0
# Still-running / suppression window must also fail closed.
msg2 = resolve_pre_tool_block("web_search", {"query": "y"})
@@ -1219,6 +1228,50 @@ class TestForceReloadSymmetry:
assert _PRE_TOOL_CALL_TIMEOUT_BLOCK_MESSAGE in result
hold.set()
+ def test_force_reload_of_one_profile_does_not_orphan_another(self, monkeypatch):
+ """Real two-manager regression: force-reloading profile A's plugin
+ manager must leave profile B's shell hook registered exactly once —
+ not duplicated, not dropped (#92682 review).
+ """
+ import hermes_cli.plugins as plugins_mod
+ import agent.shell_hooks as shell_hooks_mod
+
+ cfg = {"hooks": {"on_session_start": [{"command": "/bin/true"}]}}
+ monkeypatch.setenv("HERMES_ACCEPT_HOOKS", "1")
+ monkeypatch.setattr("hermes_cli.config.load_config", lambda: cfg)
+ monkeypatch.setattr(
+ PluginManager, "_discover_and_load_inner", lambda self_inner: None,
+ )
+
+ monkeypatch.setenv("HERMES_HOME", "/tmp/profile-a")
+ mgr_a = PluginManager()
+ plugins_mod._plugin_manager = mgr_a
+ shell_hooks_mod.register_from_config(cfg, accept_hooks=True)
+
+ monkeypatch.setenv("HERMES_HOME", "/tmp/profile-b")
+ mgr_b = PluginManager()
+ plugins_mod._plugin_manager = mgr_b
+ shell_hooks_mod.register_from_config(cfg, accept_hooks=True)
+
+ assert len(mgr_a._hooks.get("on_session_start", [])) == 1
+ assert len(mgr_b._hooks.get("on_session_start", [])) == 1
+
+ # Force-reload A. Its own manager's hook is wiped and restored;
+ # B's manager (and idempotence key) must be untouched.
+ mgr_a.discover_and_load(force=True)
+
+ assert len(mgr_a._hooks.get("on_session_start", [])) == 1
+ assert len(mgr_b._hooks.get("on_session_start", [])) == 1
+
+ # B's later adapter reconnect re-runs register_from_config(); its
+ # idempotence key must still be intact, so this must be a no-op
+ # rather than appending a second callback to B's live manager.
+ monkeypatch.setenv("HERMES_HOME", "/tmp/profile-b")
+ second = shell_hooks_mod.register_from_config(cfg, accept_hooks=True)
+
+ assert second == []
+ assert len(mgr_b._hooks.get("on_session_start", [])) == 1
+
class TestPreToolCallBlocking:
"""Tests for the pre_tool_call block directive helper."""
diff --git a/tests/hermes_cli/test_pricing_cache_auth_key.py b/tests/hermes_cli/test_pricing_cache_auth_key.py
new file mode 100644
index 0000000000..d9846097fd
--- /dev/null
+++ b/tests/hermes_cli/test_pricing_cache_auth_key.py
@@ -0,0 +1,210 @@
+"""``_pricing_cache`` keys on the credential, not just the base URL.
+
+Nous ``/v1/models`` answers each caller with the catalog their org may reach,
+so an anonymous read, and two different tokens, must not share a cache entry.
+"""
+
+from __future__ import annotations
+
+import json
+from unittest.mock import MagicMock
+
+import pytest
+
+import hermes_cli.models as models_mod
+from hermes_cli.models import fetch_models_with_pricing, peek_cached_pricing
+
+BASE = "https://inference-api.example.com"
+
+# What the endpoint serves anonymously vs. to a policy-restricted caller.
+_FULL = ["vendor/allowed", "vendor/blocked"]
+_FILTERED = ["vendor/allowed"]
+
+
+@pytest.fixture(autouse=True)
+def _clear_pricing_cache():
+ models_mod._pricing_cache.clear()
+ models_mod._pricing_cache_retry_after.clear()
+ yield
+ models_mod._pricing_cache.clear()
+ models_mod._pricing_cache_retry_after.clear()
+
+
+@pytest.fixture
+def catalog(monkeypatch):
+ """Serve the filtered catalog to an authenticated read, the full one to an
+ anonymous read, and record every request."""
+ requests: list[str | None] = []
+
+ def _fake_urlopen(req, timeout=8.0):
+ auth = req.get_header("Authorization")
+ requests.append(auth)
+ ids = _FILTERED if auth else _FULL
+ payload = {
+ "data": [
+ {"id": mid, "pricing": {"prompt": "0.000002", "completion": "0.00001"}}
+ for mid in ids
+ ]
+ }
+ resp = MagicMock()
+ resp.read.return_value = json.dumps(payload).encode()
+ resp.__enter__ = lambda self: self
+ resp.__exit__ = lambda *a: False
+ return resp
+
+ monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", _fake_urlopen)
+ return requests
+
+
+@pytest.fixture
+def per_org_catalog(monkeypatch):
+ """Serve each token the catalog its own org may reach."""
+ requests: list[str | None] = []
+
+ def _fake_urlopen(req, timeout=8.0):
+ auth = req.get_header("Authorization")
+ requests.append(auth)
+ org = "a" if auth == "Bearer tok-a" else "b"
+ payload = {
+ "data": [
+ {
+ "id": f"org-{org}/only",
+ "pricing": {"prompt": "0.000002", "completion": "0.00001"},
+ }
+ ]
+ }
+ resp = MagicMock()
+ resp.read.return_value = json.dumps(payload).encode()
+ resp.__enter__ = lambda self: self
+ resp.__exit__ = lambda *a: False
+ return resp
+
+ monkeypatch.setattr(models_mod, "_urlopen_model_catalog_request", _fake_urlopen)
+ return requests
+
+
+def test_one_token_does_not_receive_another_tokens_catalog(per_org_catalog):
+ """Two orgs in one process — a long-lived gateway or desktop backend after
+ a profile switch or re-login."""
+ a = fetch_models_with_pricing(api_key="tok-a", base_url=BASE)
+ b = fetch_models_with_pricing(api_key="tok-b", base_url=BASE)
+
+ assert list(a) == ["org-a/only"]
+ assert list(b) == ["org-b/only"], "token B was handed token A's catalog"
+ assert len(per_org_catalog) == 2, "token B must reach the network"
+
+
+def test_credential_value_does_not_appear_in_the_cache_key():
+ """Guards against keying on the raw token."""
+ assert "sk-super-secret" not in models_mod._pricing_auth_fingerprint("sk-super-secret")
+
+
+def test_anonymous_and_authenticated_reads_are_separate(catalog):
+ """Also pins the header: anonymous must send none."""
+ anon = fetch_models_with_pricing(api_key="", base_url=BASE)
+ authed = fetch_models_with_pricing(api_key="sk-test", base_url=BASE)
+
+ assert sorted(anon) == sorted(_FULL)
+ assert sorted(authed) == sorted(_FILTERED)
+ assert catalog == [None, "Bearer sk-test"]
+
+
+@pytest.mark.parametrize("api_key", ["sk-test", ""])
+def test_repeated_read_still_hits_the_cache(catalog, api_key):
+ """Widening the key must not cost the caching it was there for."""
+ first = fetch_models_with_pricing(api_key=api_key, base_url=BASE)
+ second = fetch_models_with_pricing(api_key=api_key, base_url=BASE)
+
+ assert first == second
+ assert len(catalog) == 1, "second read should be served from cache"
+
+
+def test_force_refresh_replaces_only_its_own_entry(catalog):
+ """A forced authenticated re-read must leave the anonymous entry intact."""
+ fetch_models_with_pricing(api_key="", base_url=BASE)
+ fetch_models_with_pricing(api_key="sk-test", base_url=BASE)
+ fetch_models_with_pricing(api_key="sk-test", base_url=BASE, force_refresh=True)
+
+ assert len(catalog) == 3
+ anon = fetch_models_with_pricing(api_key="", base_url=BASE)
+ assert sorted(anon) == sorted(_FULL)
+ assert len(catalog) == 3, "the anonymous entry should have survived"
+
+
+class TestPeekCachedPricing:
+ def test_returns_empty_when_nothing_cached(self):
+ assert peek_cached_pricing(BASE) == {}
+
+ def test_accepts_a_v1_suffixed_url(self, catalog):
+ """The agent holds a /v1-suffixed base URL; fetchers key on the root."""
+ fetch_models_with_pricing(api_key="sk-test", base_url=BASE)
+ assert sorted(peek_cached_pricing(BASE + "/v1")) == sorted(_FILTERED)
+
+ def test_prefers_the_authenticated_catalog(self, catalog):
+ fetch_models_with_pricing(api_key="", base_url=BASE)
+ fetch_models_with_pricing(api_key="sk-test", base_url=BASE)
+ assert sorted(peek_cached_pricing(BASE)) == sorted(_FILTERED)
+
+ def test_falls_back_to_the_anonymous_catalog(self, catalog):
+ fetch_models_with_pricing(api_key="", base_url=BASE)
+ assert sorted(peek_cached_pricing(BASE)) == sorted(_FULL)
+
+ def test_never_fetches(self, catalog):
+ peek_cached_pricing(BASE)
+ assert catalog == []
+
+
+class TestNousCatalogExpiry:
+ """A Nous catalog reflects the org's policy, which an admin can change while
+ a long-lived process holds the entry."""
+
+ def test_entry_expires_so_a_policy_change_is_picked_up(self, catalog, monkeypatch):
+ from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
+
+ fetch_models_with_pricing(
+ api_key="sk-test", base_url=BASE,
+ cache_ttl_seconds=_NOUS_CATALOG_TTL_SECONDS,
+ )
+ assert len(catalog) == 1
+
+ now = models_mod.time.monotonic()
+ monkeypatch.setattr(
+ models_mod.time, "monotonic",
+ lambda: now + _NOUS_CATALOG_TTL_SECONDS + 1,
+ )
+ fetch_models_with_pricing(
+ api_key="sk-test", base_url=BASE,
+ cache_ttl_seconds=_NOUS_CATALOG_TTL_SECONDS,
+ )
+ assert len(catalog) == 2, "expired entry should be re-read"
+
+ def test_no_ttl_keeps_the_entry_indefinitely(self, catalog, monkeypatch):
+ """Other providers' catalogs carry no policy and must not start
+ re-fetching."""
+ fetch_models_with_pricing(api_key="sk-test", base_url=BASE)
+ now = models_mod.time.monotonic()
+ monkeypatch.setattr(models_mod.time, "monotonic", lambda: now + 86_400)
+ fetch_models_with_pricing(api_key="sk-test", base_url=BASE)
+ assert len(catalog) == 1
+
+ def test_peek_prefers_the_newest_credential(self, per_org_catalog):
+ """After a rotation the older entry is still resident and, being
+ insertion-ordered, comes first."""
+ fetch_models_with_pricing(api_key="tok-a", base_url=BASE, cache_ttl_seconds=300)
+ fetch_models_with_pricing(api_key="tok-b", base_url=BASE, cache_ttl_seconds=300)
+ assert list(peek_cached_pricing(BASE)) == ["org-b/only"]
+
+ def test_peek_skips_an_expired_entry(self, catalog, monkeypatch):
+ """Reading _pricing_cache directly walked straight past the TTL."""
+ from hermes_cli.models import _NOUS_CATALOG_TTL_SECONDS
+
+ fetch_models_with_pricing(
+ api_key="sk-test", base_url=BASE,
+ cache_ttl_seconds=_NOUS_CATALOG_TTL_SECONDS,
+ )
+ now = models_mod.time.monotonic()
+ monkeypatch.setattr(
+ models_mod.time, "monotonic",
+ lambda: now + _NOUS_CATALOG_TTL_SECONDS + 1,
+ )
+ assert peek_cached_pricing(BASE) == {}
diff --git a/tests/hermes_cli/test_profiles_sidebar_cache.py b/tests/hermes_cli/test_profiles_sidebar_cache.py
index 5bb113a028..ead0c6b08b 100644
--- a/tests/hermes_cli/test_profiles_sidebar_cache.py
+++ b/tests/hermes_cli/test_profiles_sidebar_cache.py
@@ -128,6 +128,30 @@ class SidebarCacheTests(unittest.TestCase):
self.assertEqual(scan(), {"ok": True})
self.assertEqual(calls, 2)
+ def test_does_not_cache_payloads_that_carry_profile_errors(self):
+ # A 200 with a non-empty errors[] is how a failed profile scan is
+ # reported. Caching it for the TTL keeps the empty recents page in
+ # front of a store that has already recovered.
+ calls = 0
+
+ @profiles._sidebar_singleflight_cache
+ def scan():
+ nonlocal calls
+ calls += 1
+ if calls == 1:
+ return {
+ "errors": [{"profile": "default", "error": "disk I/O error"}],
+ "recents": {"sessions": []},
+ }
+ return {"errors": [], "recents": {"sessions": [{"id": "yesterday"}]}}
+
+ first = scan()
+ second = scan()
+
+ self.assertEqual(first["errors"][0]["error"], "disk I/O error")
+ self.assertEqual(second["recents"]["sessions"], [{"id": "yesterday"}])
+ self.assertEqual(calls, 2)
+
def test_can_be_disabled(self):
calls = 0
diff --git a/tests/hermes_cli/test_restart_plan_reconciliation.py b/tests/hermes_cli/test_restart_plan_reconciliation.py
index eebe427728..b87cc98f7a 100644
--- a/tests/hermes_cli/test_restart_plan_reconciliation.py
+++ b/tests/hermes_cli/test_restart_plan_reconciliation.py
@@ -159,6 +159,124 @@ def test_external_supervisor_counts_as_restarted():
assert outcomes[0]["outcome"] == "restarted"
+def test_unmanaged_serve_runtime_under_default_profile_is_unaccounted():
+ """#100479: an sshd-spawned `serve --isolated` has no systemd unit and
+ shares the default profile with the gateway. A gateway-only restart
+ must not be read as covering it — it must trip the tripwire instead."""
+ serve_runtime = RuntimeRecord(
+ kind="serve",
+ profile="default",
+ pid=900,
+ supervisor="manual-serve",
+ restart_via=_restart_mechanism("manual-serve", "default"),
+ )
+ outcomes = match_runtime_outcomes(
+ _plan(_rt("default", 100, supervisor="systemd"), serve_runtime),
+ restarted_services=["hermes-gateway"], relaunched_profiles=[],
+ externally_supervised_profiles=[], killed_pids=set(), failed_units=[],
+ )
+ by_pid = {o["pid"]: o["outcome"] for o in outcomes}
+ assert by_pid[100] == "restarted"
+ assert by_pid[900] == "unaccounted"
+ assert report_unaccounted_runtimes(outcomes) is True
+
+
+def _serve(profile: str, pid: int, kind: str = "serve") -> RuntimeRecord:
+ return RuntimeRecord(
+ kind=kind,
+ profile=profile,
+ pid=pid,
+ supervisor="manual-serve",
+ restart_via=_restart_mechanism("manual-serve", profile),
+ )
+
+
+def test_serve_never_borrows_relaunched_or_external_gateway_profile():
+ """Sibling site of #100479: the relaunched_profiles / external-supervisor
+ bookkeeping is gateway vocabulary too. A manual gateway relaunch under
+ ``default`` (or a named profile) says nothing about a serve that shares
+ the profile name."""
+ outcomes = match_runtime_outcomes(
+ _plan(_rt("default", 100), _serve("default", 900),
+ _rt("work", 101), _serve("work", 901, kind="dashboard")),
+ restarted_services=[], relaunched_profiles=["default"],
+ externally_supervised_profiles=["work"], killed_pids=set(), failed_units=[],
+ )
+ by_pid = {o["pid"]: o["outcome"] for o in outcomes}
+ assert by_pid == {
+ 100: "restarted", 900: "unaccounted", 101: "restarted", 901: "unaccounted"
+ }
+
+
+def test_named_profile_serve_does_not_match_gateway_profile_unit():
+ """``hermes-gateway-work.service`` restarted must not credit the ``work``
+ serve — the old substring match (``"work" in unit``) did exactly that."""
+ outcomes = match_runtime_outcomes(
+ _plan(_rt("work", 101, supervisor="systemd"), _serve("work", 901)),
+ restarted_services=["hermes-gateway-work.service"], relaunched_profiles=[],
+ externally_supervised_profiles=[], killed_pids=set(), failed_units=[],
+ )
+ by_pid = {o["pid"]: o["outcome"] for o in outcomes}
+ assert by_pid == {101: "restarted", 901: "unaccounted"}
+
+
+def test_serve_reconciles_against_its_own_unit_vocabulary():
+ """A serve IS covered when a ``hermes-serve*`` unit for its profile was
+ restarted (or failed) — scope-qualified identities included."""
+ outcomes = match_runtime_outcomes(
+ _plan(_serve("default", 900), _serve("work", 901),
+ _serve("ops", 902, kind="dashboard"), _serve("qa", 903)),
+ restarted_services=["hermes-gateway", "user/hermes-serve",
+ "hermes-serve-work.service", "hermes-dashboard-ops"],
+ relaunched_profiles=[], externally_supervised_profiles=[],
+ killed_pids=set(), failed_units=["hermes-serve-qa.service"],
+ )
+ by_pid = {o["pid"]: o["outcome"] for o in outcomes}
+ assert by_pid == {900: "restarted", 901: "restarted", 902: "restarted", 903: "failed"}
+ # exact names: ``work`` must not claim ``hermes-serve-workbench``
+ outcomes = match_runtime_outcomes(
+ _plan(_serve("work", 901)),
+ restarted_services=["hermes-serve-workbench.service"], relaunched_profiles=[],
+ externally_supervised_profiles=[], killed_pids=set(), failed_units=[],
+ )
+ assert outcomes[0]["outcome"] == "unaccounted"
+
+
+def test_serve_outcome_follows_incarnation_probe_when_provided():
+ """With the (pid, create_time) survivor probe result, liveness decides:
+ a pre-update serve that is gone was replaced (restarted); one still
+ alive is unaccounted — even when a hermes-serve unit was restarted."""
+ plan = _plan(_serve("default", 900), _serve("default", 901, kind="dashboard"))
+ outcomes = match_runtime_outcomes(
+ plan, restarted_services=["hermes-serve.service"], relaunched_profiles=[],
+ externally_supervised_profiles=[], killed_pids=set(), failed_units=[],
+ stale_serve_pids={900},
+ )
+ by_pid = {o["pid"]: o["outcome"] for o in outcomes}
+ assert by_pid == {900: "unaccounted", 901: "restarted"}
+ # killed pid still wins as "stopped"; probe None => fail closed
+ outcomes = match_runtime_outcomes(
+ plan, restarted_services=[], relaunched_profiles=[],
+ externally_supervised_profiles=[], killed_pids={901}, failed_units=[],
+ stale_serve_pids=None,
+ )
+ by_pid = {o["pid"]: o["outcome"] for o in outcomes}
+ assert by_pid == {900: "unaccounted", 901: "stopped"}
+
+
+def test_unaccounted_serve_report_names_serve_remedy_not_gateway_restart(capsys):
+ outcomes = match_runtime_outcomes(
+ _plan(_serve("default", 900)),
+ restarted_services=["hermes-gateway"], relaunched_profiles=[],
+ externally_supervised_profiles=[], killed_pids=set(), failed_units=[],
+ )
+ assert report_unaccounted_runtimes(outcomes) is True
+ out = capsys.readouterr().out
+ assert "serve [default] pid 900" in out
+ assert "hermes-serve.service" in out
+ assert "hermes gateway restart" not in out
+
+
def test_mixed_fleet_only_the_missed_one_escalates(capsys):
outcomes = match_runtime_outcomes(
_plan(
diff --git a/tests/hermes_cli/test_runtime_install_progress.py b/tests/hermes_cli/test_runtime_install_progress.py
new file mode 100644
index 0000000000..1ca5636f93
--- /dev/null
+++ b/tests/hermes_cli/test_runtime_install_progress.py
@@ -0,0 +1,166 @@
+"""Runtime install progress: the slow legs must tick, never hang silently.
+
+The incident: 'Installing runtime…' sat frozen for minutes on a slow
+line — ensure_runtime_installed downloaded and extracted with no
+progress stream, so both the quickstart hero and the pane's install row
+showed a dead bar. Contract: _download and _extract tick per chunk /
+per member, ensure_runtime_installed forwards a staged stream, and the
+router's hook translates it into live job fields."""
+
+from __future__ import annotations
+
+import io
+import zipfile
+from pathlib import Path
+
+import hermes_cli.local_runtime.binaries as binaries
+from hermes_cli.web_routers.local_models import _job, _runtime_progress_hook
+
+
+def _make_zip(path: Path, names_sizes: dict[str, int]) -> None:
+ with zipfile.ZipFile(path, "w") as z:
+ for name, size in names_sizes.items():
+ z.writestr(name, b"x" * size)
+
+
+def test_extract_ticks_per_member(tmp_path):
+ archive = tmp_path / "runtime.zip"
+ _make_zip(archive, {"a.bin": 1000, "b.bin": 3000, "c.bin": 500})
+ ticks: list[tuple[int, int]] = []
+ binaries._extract(archive, tmp_path / "out",
+ progress=lambda d, t: ticks.append((d, t)))
+ assert len(ticks) == 3
+ total = 4500
+ assert all(t == total for _, t in ticks)
+ assert [d for d, _ in ticks] == sorted(d for d, _ in ticks)
+ assert ticks[-1][0] == total
+ assert (tmp_path / "out" / "b.bin").stat().st_size == 3000
+
+
+def test_download_ticks_with_content_length(tmp_path, monkeypatch):
+ payload = b"y" * (3 << 20) # 3 MiB -> several 1 MiB chunks
+
+ class _Resp(io.BytesIO):
+ headers = {"Content-Length": str(len(payload))}
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ monkeypatch.setattr(binaries.urllib.request, "urlopen",
+ lambda url, timeout=0: _Resp(payload))
+ ticks: list[tuple[int, int]] = []
+ dest = tmp_path / "asset.zip"
+ binaries._download("http://x/asset.zip", dest,
+ progress=lambda d, t: ticks.append((d, t)))
+ assert dest.read_bytes() == payload
+ assert len(ticks) >= 3
+ assert ticks[-1] == (len(payload), len(payload))
+
+
+def test_ensure_runtime_installed_forwards_staged_progress(tmp_path, monkeypatch):
+ """The full install path emits download -> verify -> extract stages
+ (per asset) and a final verify, all through one callback."""
+ monkeypatch.setattr(binaries, "runtimes_root", lambda: tmp_path)
+
+ class _Plan:
+ assets = ["a.zip", "b.zip"]
+ backend = "cuda"
+ install_dir = tmp_path / "b1" / "cuda"
+
+ monkeypatch.setattr(binaries, "resolve_assets", lambda tag, backend: _Plan())
+ monkeypatch.setattr(binaries, "verify_install", lambda d, t: "ok")
+
+ def _fake_download(url, dest, progress=None):
+ _make_zip(dest, {"f.bin": 2048})
+ if progress is not None:
+ progress(1024, 2048)
+ progress(2048, 2048)
+
+ monkeypatch.setattr(binaries, "_download", _fake_download)
+
+ events: list[tuple[str, str]] = []
+ binaries.ensure_runtime_installed(
+ "b1", "cuda",
+ progress=lambda stage, d, t, label: events.append((stage, label)))
+
+ stages = [s for s, _ in events]
+ assert "download" in stages and "extract" in stages and "verify" in stages
+ # Two assets -> per-asset labels on the slow stages.
+ assert ("download", "1/2") in events and ("download", "2/2") in events
+ assert ("extract", "1/2") in events and ("extract", "2/2") in events
+ # Stage order per asset: download before extract.
+ assert stages.index("download") < stages.index("extract")
+
+
+def test_progress_hook_translates_stages_to_job_fields():
+ job = _job("quickstart", "Test Model")
+ hook = _runtime_progress_hook(job)
+
+ hook("download", 5 << 20, 100 << 20, "1/2")
+ assert job["phase"] == "downloading-runtime"
+ assert "1/2" in job["detail"]
+ assert job["done_bytes"] == 5 << 20
+ assert job["total_bytes"] == 100 << 20
+
+ # Rapid second tick inside the throttle window is dropped...
+ hook("download", 6 << 20, 100 << 20, "1/2")
+ assert job["done_bytes"] == 5 << 20
+ # ...but a terminal tick (done == total) always lands.
+ hook("download", 100 << 20, 100 << 20, "1/2")
+ assert job["done_bytes"] == 100 << 20
+
+ hook("extract", 10, 100, "")
+ assert job["phase"] in ("downloading-runtime", "unpacking-runtime")
+
+ job2 = _job("runtime-install", "x")
+ hook2 = _runtime_progress_hook(job2)
+ hook2("extract", 100, 100, "")
+ assert job2["phase"] == "unpacking-runtime"
+ hook2("verify", 0, 0, "")
+ assert job2["phase"] == "verifying-runtime"
+ assert job2["total_bytes"] is None # indeterminate bar, not a stuck 0%
+
+
+def test_progress_hook_accumulates_across_assets(monkeypatch):
+ """A two-asset engine reads as ONE growing download: the second asset's
+ bytes stack on the first's instead of restarting the bar at zero, and
+ unpack/verify leave the finished download's counters standing."""
+ # Drive the throttle's clock so every tick lands (the real hook drops
+ # sub-250ms non-terminal ticks; this test is about arithmetic, not
+ # pacing — pacing has its own assertions above).
+ from hermes_cli.web_routers import local_models as lm
+
+ clock = {"now": 0.0}
+
+ def fake_monotonic():
+ clock["now"] += 1.0
+ return clock["now"]
+
+ monkeypatch.setattr(lm.time, "monotonic", fake_monotonic)
+
+ job = _job("runtime-install", "engine")
+ hook = _runtime_progress_hook(job)
+
+ hook("download", 40 << 20, 40 << 20, "1/2")
+ assert job["done_bytes"] == 40 << 20
+ assert job["total_bytes"] == 40 << 20
+
+ # Second asset starts: counters continue from the first asset's total.
+ hook("download", 0, 60 << 20, "2/2")
+ assert job["done_bytes"] == 40 << 20
+ assert job["total_bytes"] == 100 << 20
+
+ hook("download", 60 << 20, 60 << 20, "2/2")
+ assert job["done_bytes"] == 100 << 20
+ assert job["total_bytes"] == 100 << 20
+
+ # Unpack and verify narrate without rewinding the finished bar.
+ hook("extract", 1, 100, "2/2")
+ assert job["phase"] == "unpacking-runtime"
+ assert job["done_bytes"] == 100 << 20
+ hook("verify", 0, 0, "")
+ assert job["phase"] == "verifying-runtime"
+ assert job["done_bytes"] == 100 << 20
diff --git a/tests/hermes_cli/test_runtime_machine_scope.py b/tests/hermes_cli/test_runtime_machine_scope.py
new file mode 100644
index 0000000000..d58edebaec
--- /dev/null
+++ b/tests/hermes_cli/test_runtime_machine_scope.py
@@ -0,0 +1,82 @@
+"""The managed runtime is machine-scoped, not profile-scoped.
+
+Engine binaries, models, presets, and server state are machine assets: a
+second profile must reuse them, never re-download 20 GB of GGUFs or fight
+the running server for its port. Profile-scoped decisions (default model,
+enabled flag) stay in each profile's config.yaml."""
+
+from __future__ import annotations
+
+import importlib
+
+import pytest
+
+
+@pytest.fixture
+def profile_home(tmp_path, monkeypatch):
+ """A NAMED-profile HERMES_HOME under /profiles/."""
+ root = tmp_path / ".hermes"
+ profile = root / "profiles" / "coder"
+ profile.mkdir(parents=True)
+ monkeypatch.setenv("HERMES_HOME", str(profile))
+ # hermes_constants memoizes root resolution per (native, env) pair;
+ # reload to make the new env authoritative for this test.
+ import hermes_constants
+
+ importlib.reload(hermes_constants)
+ yield root, profile
+ importlib.reload(hermes_constants)
+
+
+def test_models_and_runtimes_resolve_to_the_shared_root(profile_home):
+ root, profile = profile_home
+ import hermes_cli.local_runtime.binaries as binaries
+ import hermes_cli.local_runtime.bootstrap as bootstrap
+
+ models = bootstrap.models_dir()
+ runtimes = binaries.runtimes_root()
+
+ assert models == root / "models", (
+ f"models dir leaked into the profile: {models}")
+ assert runtimes == root / "runtimes" / "llamacpp", (
+ f"runtimes dir leaked into the profile: {runtimes}")
+ assert "profiles" not in models.parts
+ assert "profiles" not in runtimes.parts
+
+
+def test_all_runtime_state_follows_runtimes_root(profile_home):
+ """Presets, window overrides, server state, and the api key all live
+ under runtimes_root() — one resolver, so profile-scoping bugs cannot
+ come back one file at a time."""
+ root, profile = profile_home
+ from hermes_cli.local_runtime.growth import window_overrides_path
+ from hermes_cli.local_runtime.presets import read_preset_decisions
+ import hermes_cli.local_runtime.binaries as binaries
+
+ shared = root / "runtimes" / "llamacpp"
+ assert window_overrides_path() == shared / "window_overrides.json"
+ # read_preset_decisions' default path must be the shared INI: write a
+ # section there and read it back through the default-path branch.
+ shared.mkdir(parents=True, exist_ok=True)
+ (shared / "presets.ini").write_text("[m1]\nctx-size = 65536\n",
+ encoding="utf-8")
+ assert "m1" in read_preset_decisions()
+
+
+def test_default_profile_paths_unchanged(tmp_path, monkeypatch):
+ """HERMES_HOME at the root itself (default profile) resolves exactly
+ as before the scoping change — no migration for existing installs."""
+ root = tmp_path / ".hermes"
+ root.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(root))
+ import hermes_constants
+
+ importlib.reload(hermes_constants)
+ try:
+ import hermes_cli.local_runtime.binaries as binaries
+ import hermes_cli.local_runtime.bootstrap as bootstrap
+
+ assert bootstrap.models_dir() == root / "models"
+ assert binaries.runtimes_root() == root / "runtimes" / "llamacpp"
+ finally:
+ importlib.reload(hermes_constants)
diff --git a/tests/hermes_cli/test_serve_mcp_discovery_after_bind.py b/tests/hermes_cli/test_serve_mcp_discovery_after_bind.py
new file mode 100644
index 0000000000..0ae23f019e
--- /dev/null
+++ b/tests/hermes_cli/test_serve_mcp_discovery_after_bind.py
@@ -0,0 +1,70 @@
+"""Desktop `serve` starts background MCP discovery only after the socket binds.
+
+The MCP SDK import (~350ms) used to run on a thread started BEFORE
+web_server was imported, holding the GIL against the main thread's own
+import path and delaying the READY sentinel the Desktop waits on.
+"""
+
+from __future__ import annotations
+
+import logging
+import threading
+
+import hermes_cli.mcp_startup as mcp_startup
+import hermes_cli.web_server as web_server
+from tests.hermes_cli.test_dashboard_auth_gate import _stub_uvicorn_run
+
+
+def _reset_discovery_state(monkeypatch):
+ monkeypatch.setattr(mcp_startup, "_mcp_discovery_started", False)
+ monkeypatch.setattr(mcp_startup, "_mcp_discovery_thread", None)
+ monkeypatch.setattr(mcp_startup, "_mcp_discovery_deferred", None)
+
+
+def test_desktop_serve_arms_mcp_discovery_only_after_ready_sentinel(monkeypatch):
+ _reset_discovery_state(monkeypatch)
+ order: list[str] = []
+ monkeypatch.setattr(
+ mcp_startup,
+ "start_background_mcp_discovery",
+ lambda *, logger, thread_name: order.append("discovery:" + thread_name),
+ )
+ monkeypatch.setattr(web_server, "_write_machine_sentinel_line", lambda line: order.append("sentinel"))
+ _stub_uvicorn_run(monkeypatch)
+
+ web_server.start_server(
+ host="127.0.0.1", port=0, open_browser=False, headless=True,
+ start_mcp_discovery_after_bind=True,
+ )
+ timer = mcp_startup._mcp_discovery_deferred
+ assert order == ["sentinel"] and isinstance(timer, threading.Timer)
+ timer.cancel()
+ # An agent build inside the delay window pulls discovery forward itself.
+ mcp_startup.wait_for_mcp_discovery(timeout=0)
+ assert order == ["sentinel", "discovery:dashboard-mcp-discovery"]
+ assert mcp_startup._mcp_discovery_deferred is None
+
+ # Without the flag (dashboard / non-Desktop serve) start_server does not
+ # start discovery itself — cmd_dashboard's pre-import path still owns it.
+ order.clear()
+ _reset_discovery_state(monkeypatch)
+ web_server.start_server(host="127.0.0.1", port=0, open_browser=False, headless=True)
+ assert order == ["sentinel"] and mcp_startup._mcp_discovery_deferred is None
+
+
+def test_deferred_discovery_fires_once_and_is_idempotent(monkeypatch):
+ _reset_discovery_state(monkeypatch)
+ calls: list[str] = []
+ monkeypatch.setattr(
+ mcp_startup,
+ "start_background_mcp_discovery",
+ lambda *, logger, thread_name: calls.append(thread_name),
+ )
+ log = logging.getLogger("test")
+ mcp_startup.defer_background_mcp_discovery(logger=log, thread_name="t", delay=60)
+ mcp_startup.defer_background_mcp_discovery(logger=log, thread_name="t", delay=60) # second arm is a no-op
+ first = mcp_startup._mcp_discovery_deferred
+ mcp_startup._start_deferred_mcp_discovery_now()
+ mcp_startup._start_deferred_mcp_discovery_now()
+ assert calls == ["t"]
+ assert first is not None and mcp_startup._mcp_discovery_deferred is None
diff --git a/tests/hermes_cli/test_session_list_reader_disposable.py b/tests/hermes_cli/test_session_list_reader_disposable.py
new file mode 100644
index 0000000000..1109c1b5d8
--- /dev/null
+++ b/tests/hermes_cli/test_session_list_reader_disposable.py
@@ -0,0 +1,96 @@
+"""Read-only session-list opens must stay disposable.
+
+Two properties that a "keep the read-only handle for the process lifetime"
+optimisation silently destroys. Both are asserted against the real
+``_open_session_db_at_path`` read path the sidebar poll uses, because both
+failures are invisible in a unit test that mocks the store.
+
+1. **The store on disk is the truth.** Recovering a corrupt ``state.db``
+ is a file swap (``mv state.db state.db.corrupt-…; cp -a recovered.db
+ state.db``) performed while the backend is stopped, but a poll can also
+ race a restore. A reader pinned to the old inode keeps serving
+ pre-recovery rows forever, so the user "recovers" and still sees the
+ broken list.
+
+2. **Forensic backup must stay reachable.** ``offline_file_access`` refuses
+ raw byte access while ANY tracked connection is registered for the path,
+ because a raw ``close()`` would cancel this process's POSIX advisory locks
+ (howtocorrupt §2.2). ``_backup_db_file`` (the copy taken BEFORE a malformed
+ store is repaired) and ``_db_fingerprint`` (the repair-attempt ledger key)
+ both go through it. A never-closed list reader makes both fail for the rest
+ of the process, so a repair runs without its forensic backup and the
+ ledger degrades to a size-only key.
+"""
+
+from __future__ import annotations
+
+import shutil
+
+from hermes_cli.sqlite_safe_read import LiveConnectionError, offline_file_access
+from hermes_cli.web_server import _open_session_db_at_path
+from hermes_state import SessionDB, _db_fingerprint
+
+
+def _ids(db) -> list:
+ return [row["id"] for row in db.list_sessions_rich(limit=10, compact_rows=True)]
+
+
+def test_poll_observes_a_replaced_state_db(tmp_path):
+ db_path = tmp_path / "state.db"
+ old = SessionDB(db_path=db_path)
+ old.create_session("before-recovery", source="cli")
+ old.close()
+
+ first = _open_session_db_at_path(db_path, read_only=True)
+ try:
+ assert _ids(first) == ["before-recovery"]
+ finally:
+ first.close()
+
+ # `hermes sessions recover` writes a clean database, which the operator
+ # then installs over the corrupt one.
+ recovered = tmp_path / "recovered-state.db"
+ rebuilt = SessionDB(db_path=recovered)
+ rebuilt.create_session("after-recovery", source="cli")
+ rebuilt.close()
+
+ for suffix in ("-wal", "-shm"):
+ sidecar = db_path.with_name(db_path.name + suffix)
+ if sidecar.exists():
+ sidecar.unlink()
+ db_path.unlink()
+ shutil.copy2(recovered, db_path)
+
+ second = _open_session_db_at_path(db_path, read_only=True)
+ try:
+ assert _ids(second) == ["after-recovery"]
+ finally:
+ second.close()
+
+
+def test_poll_leaves_forensic_backup_reachable(tmp_path):
+ db_path = tmp_path / "state.db"
+ writer = SessionDB(db_path=db_path)
+ writer.create_session("s1", source="cli")
+ writer.close()
+
+ baseline = _db_fingerprint(db_path)
+ assert baseline is not None
+
+ poll = _open_session_db_at_path(db_path, read_only=True)
+ try:
+ assert _ids(poll) == ["s1"]
+ finally:
+ poll.close()
+
+ # The raw-copy path a malformed-store repair takes before it touches
+ # anything must still be permitted after the poll.
+ try:
+ with offline_file_access(db_path, what="forensic-backup"):
+ pass
+ except LiveConnectionError as exc: # pragma: no cover - failure detail
+ raise AssertionError(
+ f"a session-list poll left a tracked connection open: {exc}"
+ ) from exc
+
+ assert _db_fingerprint(db_path) == baseline
diff --git a/tests/hermes_cli/test_session_recovery.py b/tests/hermes_cli/test_session_recovery.py
index 3cabe5a750..42b5db314b 100644
--- a/tests/hermes_cli/test_session_recovery.py
+++ b/tests/hermes_cli/test_session_recovery.py
@@ -648,4 +648,188 @@ def test_partial_recovery_clears_only_unreadable_system_prompt_refs(
conn.close()
+def _insert_delivery_obligations(path: Path, rows: list[tuple[object, ...]]) -> None:
+ from gateway.delivery_ledger import _initialize_schema
+ conn = sqlite3.connect(str(path), isolation_level=None)
+ try:
+ _initialize_schema(conn)
+ conn.executemany(
+ """INSERT INTO delivery_obligations (
+ obligation_id, session_key, platform, chat_id, thread_id,
+ content, state, attempts, created_at, updated_at,
+ owner_pid, owner_started_at, last_error, adapter_profile
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)""",
+ rows,
+ )
+ finally:
+ conn.close()
+
+
+def test_recovery_copies_delivery_obligations(tmp_path: Path) -> None:
+ """Owed replies must survive salvage — #100313 lost 6 obligation rows."""
+
+ source = tmp_path / "state.db"
+ output = tmp_path / "recovered.db"
+ _make_source(source)
+ now = 1_720_000_000.0
+ _insert_delivery_obligations(
+ source,
+ [
+ (
+ "ob-pending",
+ "telegram:1:chat-1",
+ "telegram",
+ "chat-1",
+ None,
+ "owed reply",
+ "pending",
+ 0,
+ now,
+ now,
+ 4242,
+ 99,
+ None,
+ "default",
+ ),
+ (
+ "ob-delivered",
+ "telegram:1:chat-1",
+ "telegram",
+ "chat-1",
+ None,
+ "already sent",
+ "delivered",
+ 1,
+ now,
+ now + 1,
+ None,
+ None,
+ None,
+ "default",
+ ),
+ ],
+ )
+
+ inspection = inspect_session_database(source, work_dir=tmp_path)
+ assert inspection["tables"]["delivery_obligations"]["available"] is True
+ assert inspection["tables"]["delivery_obligations"]["rows"] == 2
+
+ report = recover_session_database(source, output, work_dir=tmp_path)
+ copied = report["copy"]["delivery_obligations"]
+ assert copied["status"] == "complete"
+ assert copied["copied_rows"] == 2
+ assert report["verification"]["table_counts"]["delivery_obligations"] == 2
+ assert report["complete"] is True
+ assert report["verified"] is True
+ assert report["installed"] is False
+
+ conn = sqlite3.connect(str(output))
+ try:
+ recovered = conn.execute(
+ """SELECT obligation_id, state, content, owner_pid, adapter_profile
+ FROM delivery_obligations ORDER BY obligation_id"""
+ ).fetchall()
+ finally:
+ conn.close()
+ assert recovered == [
+ ("ob-delivered", "delivered", "already sent", None, "default"),
+ ("ob-pending", "pending", "owed reply", 4242, "default"),
+ ]
+
+
+def test_recovery_without_delivery_ledger_is_not_lossy(tmp_path: Path) -> None:
+ """CLI-only stores never created the lazy table; that is not data loss."""
+
+ source = tmp_path / "state.db"
+ output = tmp_path / "recovered.db"
+ _make_source(source)
+
+ report = recover_session_database(source, output, work_dir=tmp_path)
+ assert report["copy"]["delivery_obligations"]["status"] == "missing"
+ assert "delivery_obligations" not in report["verification"]["table_counts"]
+ assert report["complete"] is True
+ assert report["verified"] is True
+
+
+
+
+
+def test_recovery_flags_delivery_obligation_count_mismatch_as_loss(
+ tmp_path: Path, monkeypatch: pytest.MonkeyPatch
+) -> None:
+ """A source-vs-destination ledger count mismatch must not verify as complete.
+
+ The destination table is created through the registered initializer; a
+ real SQL trigger that silently drops one row stands in for the "rows went
+ missing on the way over" failure the verifier has to catch.
+ """
+
+ from hermes_cli import session_recovery
+
+ source = tmp_path / "state.db"
+ output = tmp_path / "recovered.db"
+ _make_source(source)
+ now = 1_720_000_000.0
+ _insert_delivery_obligations(
+ source,
+ [
+ ("ob-a", "k", "telegram", "chat-1", None, "a", "pending", 0, now, now, None, None, None, "default"),
+ ("ob-b", "k", "telegram", "chat-1", None, "b", "pending", 0, now, now, None, None, None, "default"),
+ ],
+ )
+
+ real_init = session_recovery._AUXILIARY_TABLE_SCHEMAS["delivery_obligations"]
+
+ def lossy_init(conn: sqlite3.Connection) -> None:
+ real_init(conn)
+ conn.execute(
+ """CREATE TRIGGER drop_ob_b BEFORE INSERT ON delivery_obligations
+ WHEN NEW.obligation_id = 'ob-b' BEGIN SELECT RAISE(IGNORE); END"""
+ )
+
+ monkeypatch.setitem(
+ session_recovery._AUXILIARY_TABLE_SCHEMAS, "delivery_obligations", lossy_init
+ )
+
+ report = recover_session_database(source, output, work_dir=tmp_path)
+ assert report["verification"]["table_counts"]["delivery_obligations"] == 1
+ assert report["complete"] is False
+ assert any(
+ "delivery_obligations count is 1, expected 2" in error
+ for error in report["verification"]["errors"]
+ )
+
+
+def test_lost_and_found_direct_copy_creates_lazy_delivery_ledger(tmp_path: Path) -> None:
+ """The .recover lane copies the ledger even though SessionDB never made it."""
+
+ from hermes_cli.session_lost_and_found import _copy_direct_tables
+
+ recovered_source = tmp_path / "lost_and_found.db"
+ now = 1_720_000_000.0
+ _insert_delivery_obligations(
+ recovered_source,
+ [
+ ("ob-1", "k", "telegram", "chat-1", None, "one", "pending", 0, now, now, None, None, None, "default"),
+ ("ob-2", "k", "telegram", "chat-1", None, "two", "failed", 3, now, now, None, None, "boom", "default"),
+ ],
+ )
+ output = tmp_path / "rebuilt.db"
+ SessionDB(db_path=output).close()
+
+ lf_conn = sqlite3.connect(str(recovered_source), isolation_level=None)
+ dest = sqlite3.connect(str(output), isolation_level=None)
+ try:
+ assert not dest.execute(
+ "SELECT 1 FROM sqlite_master WHERE type='table' AND name='delivery_obligations'"
+ ).fetchall()
+ copied = _copy_direct_tables(lf_conn, dest)
+ assert copied["delivery_obligations"] == 2
+ rows = dest.execute(
+ "SELECT obligation_id, state, last_error FROM delivery_obligations ORDER BY obligation_id"
+ ).fetchall()
+ finally:
+ lf_conn.close()
+ dest.close()
+ assert rows == [("ob-1", "pending", None), ("ob-2", "failed", "boom")]
diff --git a/tests/hermes_cli/test_set_config_value.py b/tests/hermes_cli/test_set_config_value.py
index e4c5f8ca15..d83a2af3ca 100644
--- a/tests/hermes_cli/test_set_config_value.py
+++ b/tests/hermes_cli/test_set_config_value.py
@@ -53,6 +53,7 @@ class TestExplicitAllowlist:
"DISCORD_BOT_TOKEN",
"SLACK_BOT_TOKEN",
"SLACK_APP_TOKEN",
+ "API_SERVER_KEY",
])
def test_explicit_key_routes_to_env(self, key, _isolated_hermes_home):
set_config_value(key, "test-value-123")
diff --git a/tests/hermes_cli/test_setup_blank_slate.py b/tests/hermes_cli/test_setup_blank_slate.py
index b401a2069e..e08d67c8e3 100644
--- a/tests/hermes_cli/test_setup_blank_slate.py
+++ b/tests/hermes_cli/test_setup_blank_slate.py
@@ -53,6 +53,14 @@ class TestBlankSlateMinimalToolsets:
from tools.registry import registry as _tool_registry
_entry = _tool_registry.get_entry("vision_analyze")
monkeypatch.setattr(_entry, "check_fn", lambda: True)
+ # This test pins disabled_toolsets SUBTRACTION, not deferral policy —
+ # assemble with the legacy everything-eager override so the expected
+ # list stays deferral-independent (#97979 defers process_manage by
+ # default, which would swap it for the three bridge tools here).
+ from tools.tool_search import ToolSearchConfig
+ _legacy = ToolSearchConfig.from_raw({"enabled": "on", "defer": []})
+ monkeypatch.setattr("tools.tool_search.load_config", lambda: _legacy)
+ monkeypatch.setattr("tools.tool_search.load_config_readonly", lambda: _legacy)
from hermes_cli.tools_config import _get_platform_tools
cfg = {}
_blank_slate_minimal_toolsets(cfg)
@@ -67,7 +75,7 @@ class TestBlankSlateMinimalToolsets:
names = sorted(
{(d.get("function") or {}).get("name") or d.get("name") for d in defs}
)
- assert names == ["patch", "process", "read_file", "search_files",
+ assert names == ["patch", "process_manage", "read_file", "search_files",
"skill_manage", "skill_view", "skills_list",
"terminal", "vision_analyze", "write_file"]
diff --git a/tests/hermes_cli/test_setup_telemetry.py b/tests/hermes_cli/test_setup_telemetry.py
index e6ebcb428c..2397524343 100644
--- a/tests/hermes_cli/test_setup_telemetry.py
+++ b/tests/hermes_cli/test_setup_telemetry.py
@@ -25,6 +25,51 @@ def test_setup_telemetry_enables_shared_metrics(monkeypatch):
assert config["telemetry"]["shared_metrics"]["enabled"] is True
+def test_disabling_collection_closes_the_send_consent_window(monkeypatch, tmp_path):
+ """`hermes tools` -> disable shared metrics must withdraw send consent.
+
+ The not-enabled branch returned early without recording anything, so the
+ consent window stayed open and re-enabling later would release every
+ package collected in between.
+ """
+ from hermes_cli.observability.shared_metrics import SharedMetricsStore
+ from hermes_cli.observability.shared_metrics_sender import (
+ reconcile_send_consent,
+ )
+ from hermes_cli.sqlite_util import write_txn
+
+ store = SharedMetricsStore(
+ database_path=tmp_path / "m.db", outbox_directory=tmp_path / "o"
+ )
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics.SharedMetricsStore",
+ lambda *a, **k: store,
+ )
+
+ # The user had consented; now they turn collection off entirely.
+ monkeypatch.setattr(
+ "hermes_cli.setup.prompt_yes_no", lambda _question, default: False
+ )
+ config = {"telemetry": {"shared_metrics": {"enabled": True, "send": True}}}
+ # Consent was granted earlier, so a window is open — that is precisely
+ # the state whose closure must be recorded.
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, True)
+
+ setup_telemetry(config)
+
+ assert config["telemetry"]["shared_metrics"]["enabled"] is False
+ assert config["telemetry"]["shared_metrics"]["send"] is False
+ with store._connection() as connection:
+ open_windows = connection.execute(
+ "SELECT COUNT(*) FROM send_consent_windows WHERE closed_at IS NULL"
+ ).fetchone()[0]
+ assert open_windows == 0, (
+ "disabling collection left the send consent window open"
+ )
+
+
def test_setup_parser_accepts_telemetry_section():
parser = argparse.ArgumentParser()
subparsers = parser.add_subparsers(dest="command")
diff --git a/tests/hermes_cli/test_shared_metrics_consent_windows.py b/tests/hermes_cli/test_shared_metrics_consent_windows.py
new file mode 100644
index 0000000000..66b58d5dd3
--- /dev/null
+++ b/tests/hermes_cli/test_shared_metrics_consent_windows.py
@@ -0,0 +1,310 @@
+"""Property tests for the consent-interval model.
+
+Ported from the /tmp validation harness that gated the redesign: every
+scenario here is a defect that actually occurred (rounds 3-5) or a clock
+adversary the day-stamp model could not survive. The v1 and v2 drafts of the
+redesign each FAILED scenarios in this file before shipping — that is the
+harness working, and why these run against the real store and the real
+reconciler rather than a model of them.
+"""
+
+from __future__ import annotations
+
+import json
+from datetime import datetime, timedelta, timezone
+
+import pytest
+
+from hermes_cli.observability.shared_metrics import SharedMetricsStore
+from hermes_cli.observability.shared_metrics_sender import (
+ CONSENT_GATE_SQL,
+ reconcile_send_consent,
+)
+from hermes_cli.sqlite_util import write_txn
+
+T0 = datetime(2026, 8, 1, tzinfo=timezone.utc)
+
+
+def ts(days=0, hours=0):
+ return (T0 + timedelta(days=days, hours=hours)).isoformat().replace(
+ "+00:00", "Z"
+ )
+
+
+def dt(days=0, hours=0):
+ return T0 + timedelta(days=days, hours=hours)
+
+
+@pytest.fixture
+def store(tmp_path):
+ return SharedMetricsStore(
+ database_path=tmp_path / "m.db", outbox_directory=tmp_path / "o"
+ )
+
+
+def _add(store, pid, start, end):
+ """Store a package the way the generator does: at period end."""
+ with store._connection() as connection:
+ with write_txn(connection):
+ connection.execute(
+ "INSERT INTO package_outbox(package_id, period_start, period_end,"
+ " payload_json, created_at, exported_at) VALUES (?, ?, ?, ?, ?, ?)",
+ (pid, start, end, json.dumps({"package_id": pid}), end, end),
+ )
+ connection.execute(
+ """INSERT INTO consent_marks(name, stamp) VALUES ('data', ?)
+ ON CONFLICT(name) DO UPDATE SET stamp = MAX(stamp, excluded.stamp)""",
+ (end,),
+ )
+
+
+def _observe(store, send_enabled, when):
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, send_enabled, now=when)
+
+
+def _eligible(store):
+ with store._connection() as connection:
+ return sorted(
+ row[0]
+ for row in connection.execute(
+ f"SELECT package_id FROM package_outbox WHERE {CONSENT_GATE_SQL}"
+ )
+ )
+
+
+def _windows(store):
+ with store._connection() as connection:
+ return [
+ tuple(row)
+ for row in connection.execute(
+ "SELECT opened_at, last_confirmed_at, closed_at"
+ " FROM send_consent_windows ORDER BY opened_at"
+ )
+ ]
+
+
+class TestRefusedWindowIsNeverReleased:
+ def test_on_off_on_with_realistic_interleaving(self, store):
+ """Rounds 3 and 5: the refused middle must never transmit, and
+ neither consented era may be lost."""
+ _observe(store, True, dt(0))
+ for n in range(5):
+ _add(store, f"d{n:02d}", ts(days=n), ts(days=n + 1))
+ _observe(store, True, dt(days=n + 1))
+ _observe(store, False, dt(5))
+ for n in range(5, 10):
+ _add(store, f"d{n:02d}", ts(days=n), ts(days=n + 1))
+ _observe(store, True, dt(10))
+ for n in range(10, 15):
+ _add(store, f"d{n:02d}", ts(days=n), ts(days=n + 1))
+ _observe(store, True, dt(days=n + 1))
+
+ eligible = _eligible(store)
+ assert not [p for p in eligible if 5 <= int(p[1:]) < 10], eligible
+ assert [f"d{n:02d}" for n in range(5)] == eligible[:5], (
+ "pre-revocation consented backlog was destroyed"
+ )
+ assert [f"d{n:02d}" for n in range(10, 15)] == eligible[5:], eligible
+
+ def test_hand_edit_with_a_90_day_silent_gap(self, store):
+ """Round 5 D1, strongest form: NOTHING observes the off window.
+
+ The close back-dates to the last confirmed moment, so the unobserved
+ gap is outside every window and fails closed.
+ """
+ _observe(store, True, dt(0))
+ _add(store, "consented", ts(0, 1), ts(0, 2))
+ _observe(store, True, dt(0, 6))
+ for n in range(1, 90, 10):
+ _add(store, f"REFUSED-d{n}", ts(days=n), ts(days=n, hours=1))
+ _observe(store, False, dt(90)) # first observation: boot on day 90
+ _observe(store, True, dt(91))
+ _observe(store, True, dt(92))
+
+ eligible = _eligible(store)
+ assert not [p for p in eligible if p.startswith("REFUSED")], eligible
+ assert "consented" in eligible, (
+ "the confirmed-morning package must survive the reconciliation"
+ )
+
+
+class TestClockAdversaries:
+ def test_forward_poison_then_revoke_releases_nothing(self, store):
+ """Round 6 D1: one glitched-forward sample must not defeat a close.
+
+ Unfixed, the poisoned obs mark dragged last_confirmed_at to 2099, a
+ later revoke stamped closed_at = 2099, and the closed window then
+ CONTAINED every refused period that followed — all 8 refused
+ packages became eligible. The close now clamps to the closing
+ observation's own raw stamp, so an honest clock at revoke time pulls
+ the window back to the true revoke moment.
+ """
+ _observe(store, True, dt(0))
+ _observe(store, True, datetime(2099, 1, 1, tzinfo=timezone.utc))
+ _observe(store, False, dt(1)) # honest clock at revoke
+ for n in range(2, 10):
+ _add(store, f"REFUSED-{n}", ts(days=n), ts(days=n, hours=2))
+
+ leaked = [p for p in _eligible(store) if p.startswith("REFUSED")]
+ assert not leaked, f"poisoned horizon released refused data: {leaked}"
+
+ def test_forward_poison_cannot_wedge_consent_forever(self, store):
+ """The obs-advance cap bounds the damage of one insane sample.
+
+ Uncapped, a 2099 sample would clamp every future window open at
+ 2099, suppressing consented data for decades (fail-closed but
+ permanent). Capped, the mark moves at most MAX_OBS_ADVANCE_SECONDS
+ past its previous value, so honest time overtakes it.
+ """
+ from hermes_cli.observability.shared_metrics_sender import (
+ MAX_OBS_ADVANCE_SECONDS,
+ )
+
+ _observe(store, True, dt(0))
+ _observe(store, True, datetime(2099, 1, 1, tzinfo=timezone.utc))
+ with store._connection() as connection:
+ stamp = connection.execute(
+ "SELECT stamp FROM consent_marks WHERE name = 'obs'"
+ ).fetchone()[0]
+ ceiling = ts(days=MAX_OBS_ADVANCE_SECONDS // 86_400)
+ assert stamp <= ceiling, (
+ f"one glitched sample advanced the mark unboundedly: {stamp}"
+ )
+
+ # Consented data from shortly after the cap horizon still flows once
+ # honest observations catch the marks up.
+ horizon_days = MAX_OBS_ADVANCE_SECONDS // 86_400
+ _add(
+ store,
+ "post-glitch",
+ ts(days=horizon_days + 1),
+ ts(days=horizon_days + 1, hours=4),
+ )
+ _observe(store, True, dt(days=horizon_days + 2))
+ assert "post-glitch" in _eligible(store), (
+ "consent wedged after a forward glitch"
+ )
+
+ def test_rollback_at_re_enable_releases_nothing(self, store):
+ """Round 5 D2: the data mark clamps opens above existing packages."""
+ _observe(store, True, dt(0))
+ _observe(store, True, dt(5))
+ _observe(store, False, dt(5))
+ for n in range(1, 4):
+ _add(store, f"REFUSED-{n}", ts(days=5, hours=n), ts(days=5, hours=n + 1))
+ _observe(store, True, dt(-12)) # 12-day rollback at re-enable
+ _observe(store, True, dt(-11))
+
+ during = [p for p in _eligible(store) if p.startswith("REFUSED")]
+ assert not during, f"rollback released refused packages: {during}"
+
+ _observe(store, True, dt(20)) # clock recovers
+ _observe(store, True, dt(21))
+ after = [p for p in _eligible(store) if p.startswith("REFUSED")]
+ assert not after, f"recovery released refused packages: {after}"
+
+ def test_recovery_does_not_wedge_future_sending(self, store):
+ _observe(store, True, dt(0))
+ _observe(store, False, dt(5))
+ _observe(store, True, dt(-12))
+ _observe(store, True, dt(20))
+ _add(store, "post-recovery", ts(21), ts(21, 4))
+ _observe(store, True, dt(22))
+ assert "post-recovery" in _eligible(store)
+
+
+class TestSubDayGranularity:
+ def test_intra_day_refusal_holds_back_the_whole_day_package(self, store):
+ """Round 5 D3: a day package spanning a refused stretch must wait."""
+ _observe(store, True, dt(0))
+ _observe(store, True, dt(10, 9))
+ _observe(store, False, dt(10, 9))
+ _observe(store, True, dt(10, 18))
+ _observe(store, True, dt(11, 2))
+ _add(store, "halfday", ts(10), ts(11))
+ assert "halfday" not in _eligible(store)
+
+
+class TestReconcilerProperties:
+ def test_idempotent_under_replay(self, store):
+ for _ in range(4):
+ _observe(store, True, dt(0))
+ _observe(store, False, dt(2))
+ for _ in range(5):
+ _observe(store, False, dt(3))
+ _observe(store, True, dt(4))
+ for _ in range(3):
+ _observe(store, True, dt(5))
+ assert len(_windows(store)) == 2
+
+ def test_the_observation_mark_is_monotonic(self, store):
+ """A rolled-back clock must never lower the observation high-water.
+
+ Every downstream guarantee leans on this: closes clamp to it via
+ last_confirmed_at, and opens clamp to max(obs, data). Found as a
+ surviving mutant (obs upsert rewritten from MAX to overwrite) —
+ the leak scenarios happen to be covered by the data mark whenever a
+ leakable package exists, but the property itself must hold on its
+ own, not by coincidence of the sibling mark.
+ """
+ _observe(store, True, dt(5))
+ _observe(store, True, dt(0)) # rollback
+ with store._connection() as connection:
+ stamp = connection.execute(
+ "SELECT stamp FROM consent_marks WHERE name = 'obs'"
+ ).fetchone()[0]
+ assert stamp == ts(5), f"obs mark moved backwards: {stamp}"
+
+ def test_the_real_package_writer_advances_the_data_mark(self, store):
+ """Round 6 D2: the harness's _add re-implements the data-mark insert,
+ so deleting the advance from the REAL writer survived 314 tests.
+ This drives the production exporter instead.
+ """
+ from datetime import date, timedelta as _td
+
+ yesterday = (date.today() - _td(days=1)).isoformat()
+ with store._connection() as connection:
+ with write_txn(connection):
+ connection.execute(
+ "INSERT INTO counter_aggregates("
+ " period_start, metric_name, hermes_version, os_family,"
+ " architecture, install_method, dimensions_json, value,"
+ " packaged_value"
+ ") VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
+ (
+ yesterday, "hermes.client.active", "0.0.0-test",
+ "macos", "arm64", "git", "{}", 1, 0,
+ ),
+ )
+
+ exported = store.create_and_export_package_if_due()
+ assert exported, "the generator was expected to export yesterday's period"
+
+ with store._connection() as connection:
+ row = connection.execute(
+ "SELECT stamp FROM consent_marks WHERE name = 'data'"
+ ).fetchone()
+ assert row is not None and row[0] >= yesterday, (
+ "the production package writer did not advance the data mark"
+ )
+
+ def test_the_gate_is_read_only(self, store):
+ _observe(store, True, dt(0))
+ before = _windows(store)
+ for _ in range(10):
+ _eligible(store)
+ assert _windows(store) == before
+
+ def test_no_window_fails_closed(self, store):
+ _add(store, "orphan", ts(0), ts(1))
+ assert _eligible(store) == []
+
+ def test_fresh_package_waits_one_heartbeat_then_releases(self, store):
+ """The documented latency cost of confirmation-based windows."""
+ _observe(store, True, dt(0))
+ _add(store, "fresh", ts(0, 1), ts(0, 2))
+ assert _eligible(store) == []
+ _observe(store, True, dt(0, 3))
+ assert _eligible(store) == ["fresh"]
diff --git a/tests/hermes_cli/test_shared_metrics_send_config.py b/tests/hermes_cli/test_shared_metrics_send_config.py
new file mode 100644
index 0000000000..2af8958a2b
--- /dev/null
+++ b/tests/hermes_cli/test_shared_metrics_send_config.py
@@ -0,0 +1,165 @@
+"""Tests for shared-metrics send configuration resolution."""
+
+from __future__ import annotations
+
+import logging
+
+import pytest
+
+from hermes_cli.config import DEFAULT_CONFIG
+from hermes_cli.observability.shared_metrics_send_config import (
+ DEFAULT_ENDPOINT,
+ resolve_send_config,
+ reset_warning_latch_for_tests,
+)
+
+
+@pytest.fixture(autouse=True)
+def _reset_latch():
+ reset_warning_latch_for_tests()
+ yield
+ reset_warning_latch_for_tests()
+
+
+def _config(**shared):
+ return {"telemetry": {"shared_metrics": shared}}
+
+
+class TestDefaults:
+ def test_send_is_registered_disabled_by_default(self):
+ shared = DEFAULT_CONFIG["telemetry"]["shared_metrics"]
+ assert shared["enabled"] is False
+ assert shared["send"] is False
+
+ def test_default_endpoint_is_production(self):
+ shared = DEFAULT_CONFIG["telemetry"]["shared_metrics"]
+ assert shared["endpoint"] == DEFAULT_ENDPOINT
+ assert DEFAULT_ENDPOINT.startswith("https://")
+
+ def test_empty_config_sends_nothing(self):
+ resolved = resolve_send_config({})
+ assert resolved.enabled is False
+ assert resolved.send is False
+
+ def test_none_config_is_tolerated(self):
+ assert resolve_send_config(None).send is False
+
+
+class TestSendRequiresCollection:
+ def test_collection_alone_does_not_send(self):
+ resolved = resolve_send_config(_config(enabled=True))
+ assert resolved.enabled is True
+ assert resolved.send is False
+
+ def test_send_with_collection_sends(self):
+ resolved = resolve_send_config(_config(enabled=True, send=True))
+ assert resolved.send is True
+
+ def test_send_without_collection_is_refused(self):
+ resolved = resolve_send_config(_config(enabled=False, send=True))
+ assert resolved.send is False
+ # send must never imply enabled
+ assert resolved.enabled is False
+
+ def test_send_without_collection_logs_an_error(self, caplog):
+ with caplog.at_level(logging.ERROR):
+ resolve_send_config(_config(enabled=False, send=True))
+ errors = [r for r in caplog.records if r.levelno >= logging.ERROR]
+ assert len(errors) == 1
+ assert "enabled is false" in errors[0].getMessage()
+
+ def test_the_error_is_logged_once_per_process(self, caplog):
+ with caplog.at_level(logging.ERROR):
+ for _ in range(5):
+ resolve_send_config(_config(enabled=False, send=True))
+ errors = [r for r in caplog.records if r.levelno >= logging.ERROR]
+ assert len(errors) == 1, "misconfiguration must not spam every hook fire"
+
+
+class TestEndpointPrecedence:
+ def test_config_endpoint_overrides_default(self):
+ resolved = resolve_send_config(
+ _config(enabled=True, send=True, endpoint="https://example.test/v1")
+ )
+ assert resolved.endpoint == "https://example.test/v1"
+
+ def test_no_environment_variable_can_redirect_telemetry(self, monkeypatch):
+ """A consent hazard: an inherited env var must not silently retarget.
+
+ AGENTS.md also reserves HERMES_* for secrets, not behaviour.
+ """
+ for name in (
+ "HERMES_TELEMETRY_ENDPOINT",
+ "TELEMETRY_ENDPOINT",
+ "HERMES_SHARED_METRICS_ENDPOINT",
+ ):
+ monkeypatch.setenv(name, "https://attacker.test/v1")
+ resolved = resolve_send_config(_config(enabled=True, send=True))
+ assert resolved.endpoint == DEFAULT_ENDPOINT
+
+ def test_blank_endpoint_falls_back_to_production(self):
+ resolved = resolve_send_config(_config(enabled=True, send=True, endpoint=" "))
+ assert resolved.endpoint == DEFAULT_ENDPOINT
+
+ def test_endpoint_is_stripped(self):
+ resolved = resolve_send_config(
+ _config(enabled=True, send=True, endpoint=" https://staging.test/v1 ")
+ )
+ assert resolved.endpoint == "https://staging.test/v1"
+
+
+class TestTransportSafety:
+ def test_plaintext_endpoint_is_refused(self, caplog):
+ with caplog.at_level(logging.ERROR):
+ resolved = resolve_send_config(
+ _config(enabled=True, send=True, endpoint="http://example.test/v1")
+ )
+ assert resolved.send is False, "telemetry must not go out in clear text"
+ assert any("https" in r.getMessage() for r in caplog.records)
+
+ @pytest.mark.parametrize(
+ "endpoint",
+ [
+ "http://localhost:8099/v1/telemetry",
+ "http://127.0.0.1:8099/v1/telemetry",
+ ],
+ )
+ def test_loopback_http_is_allowed_for_testing(self, endpoint):
+ resolved = resolve_send_config(
+ _config(enabled=True, send=True, endpoint=endpoint)
+ )
+ assert resolved.send is True
+
+ def test_nonsense_scheme_is_refused(self):
+ resolved = resolve_send_config(
+ _config(enabled=True, send=True, endpoint="ftp://example.test/v1")
+ )
+ assert resolved.send is False
+
+ @pytest.mark.parametrize(
+ "endpoint",
+ [
+ "ftp://localhost/v1/telemetry",
+ "gopher://localhost/v1/telemetry",
+ "ws://127.0.0.1/v1/telemetry",
+ ],
+ )
+ def test_a_non_http_scheme_on_loopback_is_still_refused(self, endpoint):
+ """The scheme is allowlisted, not merely checked for plaintext http.
+
+ Gap found by mutation testing: replacing the `http` scheme test with
+ `if True` survived the whole suite, because every non-http scheme case
+ pointed at a REMOTE host, where the loopback branch rejects it anyway.
+ Only a non-http scheme aimed at loopback distinguishes an allowlist
+ from a plaintext-only check.
+ """
+ resolved = resolve_send_config(
+ _config(enabled=True, send=True, endpoint=endpoint)
+ )
+ assert resolved.send is False
+
+ def test_unsafe_endpoint_does_not_block_collection(self):
+ resolved = resolve_send_config(
+ _config(enabled=True, send=True, endpoint="http://example.test/v1")
+ )
+ assert resolved.enabled is True
diff --git a/tests/hermes_cli/test_shared_metrics_send_migration.py b/tests/hermes_cli/test_shared_metrics_send_migration.py
new file mode 100644
index 0000000000..54518c644a
--- /dev/null
+++ b/tests/hermes_cli/test_shared_metrics_send_migration.py
@@ -0,0 +1,224 @@
+"""Tests for the additive send-state migration on ``package_outbox``.
+
+The store schema version must NOT move when these columns are added: the
+existing loader raises on any version it does not recognise, so bumping it
+would hard-fail an older Hermes (a second profile on an older build, or a
+rollback) against the same database file.
+"""
+
+from __future__ import annotations
+
+import json
+import sqlite3
+
+import pytest
+
+from hermes_cli.observability.shared_metrics import SharedMetricsStore
+
+SEND_COLUMNS = {
+ "sent_at",
+ "send_state",
+ "send_attempts",
+ "next_attempt_at",
+ "last_error",
+ "sent_install_id",
+}
+
+
+def _columns(db_path):
+ connection = sqlite3.connect(db_path)
+ try:
+ return {row[1] for row in connection.execute("PRAGMA table_info(package_outbox)")}
+ finally:
+ connection.close()
+
+
+def _schema_version(db_path):
+ connection = sqlite3.connect(db_path)
+ try:
+ row = connection.execute(
+ "SELECT value FROM telemetry_state WHERE key = 'schema_version'"
+ ).fetchone()
+ return row[0] if row else None
+ finally:
+ connection.close()
+
+
+@pytest.fixture
+def store(tmp_path):
+ return SharedMetricsStore(
+ database_path=tmp_path / "metrics.sqlite3",
+ outbox_directory=tmp_path / "outbox",
+ )
+
+
+class TestFreshDatabase:
+ def test_send_columns_exist(self, store):
+ assert SEND_COLUMNS <= _columns(store.database_path)
+
+ def test_original_columns_survive(self, store):
+ assert {
+ "package_id",
+ "period_start",
+ "period_end",
+ "payload_json",
+ "created_at",
+ "exported_at",
+ } <= _columns(store.database_path)
+
+ def test_send_attempts_defaults_to_zero(self, store):
+ connection = sqlite3.connect(store.database_path)
+ try:
+ connection.execute(
+ """
+ INSERT INTO package_outbox(
+ package_id, period_start, period_end, payload_json, created_at
+ ) VALUES ('p', '2026-01-01', '2026-01-02', '{}', '2026-01-01T00:00:00Z')
+ """
+ )
+ connection.commit()
+ row = connection.execute(
+ "SELECT send_attempts, send_state, sent_install_id FROM package_outbox"
+ ).fetchone()
+ finally:
+ connection.close()
+ assert row[0] == 0
+ assert row[1] is None
+ assert row[2] is None
+
+
+class TestUpgradeFromPreSendDatabase:
+ """The real-world case: a database written before this feature existed."""
+
+ @pytest.fixture
+ def legacy_db(self, tmp_path):
+ path = tmp_path / "metrics.sqlite3"
+ connection = sqlite3.connect(path)
+ try:
+ connection.execute(
+ """
+ CREATE TABLE telemetry_state (
+ key TEXT PRIMARY KEY,
+ value TEXT NOT NULL
+ )
+ """
+ )
+ connection.execute(
+ "INSERT INTO telemetry_state(key, value) VALUES ('schema_version', '2')"
+ )
+ connection.execute(
+ """
+ CREATE TABLE package_outbox (
+ package_id TEXT PRIMARY KEY,
+ period_start TEXT NOT NULL,
+ period_end TEXT NOT NULL,
+ payload_json TEXT NOT NULL,
+ created_at TEXT NOT NULL,
+ exported_at TEXT
+ )
+ """
+ )
+ connection.execute(
+ """
+ CREATE TABLE counter_aggregates (
+ period_start TEXT NOT NULL,
+ metric_name TEXT NOT NULL,
+ hermes_version TEXT NOT NULL,
+ os_family TEXT NOT NULL,
+ architecture TEXT NOT NULL,
+ install_method TEXT NOT NULL,
+ dimensions_json TEXT NOT NULL,
+ value INTEGER NOT NULL,
+ packaged_value INTEGER NOT NULL,
+ PRIMARY KEY (
+ period_start, metric_name, hermes_version, os_family,
+ architecture, install_method, dimensions_json
+ )
+ )
+ """
+ )
+ for i in range(3):
+ connection.execute(
+ """
+ INSERT INTO package_outbox(
+ package_id, period_start, period_end, payload_json,
+ created_at, exported_at
+ ) VALUES (?, ?, ?, ?, ?, ?)
+ """,
+ (
+ f"pkg-{i}",
+ "2026-08-2%d" % i,
+ "2026-08-2%d" % (i + 1),
+ json.dumps({"package_id": f"pkg-{i}"}),
+ "2026-08-2%dT00:00:00Z" % i,
+ "2026-08-2%dT01:00:00Z" % i,
+ ),
+ )
+ connection.commit()
+ finally:
+ connection.close()
+ return path
+
+ def test_upgrade_preserves_every_row(self, legacy_db, tmp_path):
+ SharedMetricsStore(
+ database_path=legacy_db, outbox_directory=tmp_path / "outbox"
+ )
+ connection = sqlite3.connect(legacy_db)
+ try:
+ count = connection.execute("SELECT COUNT(*) FROM package_outbox").fetchone()[0]
+ payloads = connection.execute(
+ "SELECT package_id, payload_json FROM package_outbox ORDER BY package_id"
+ ).fetchall()
+ finally:
+ connection.close()
+ assert count == 3
+ assert payloads == [
+ ("pkg-0", '{"package_id": "pkg-0"}'),
+ ("pkg-1", '{"package_id": "pkg-1"}'),
+ ("pkg-2", '{"package_id": "pkg-2"}'),
+ ]
+
+ def test_upgrade_adds_the_send_columns(self, legacy_db, tmp_path):
+ SharedMetricsStore(
+ database_path=legacy_db, outbox_directory=tmp_path / "outbox"
+ )
+ assert SEND_COLUMNS <= _columns(legacy_db)
+
+ def test_upgrade_does_not_move_the_schema_version(self, legacy_db, tmp_path):
+ """Bumping would make older builds refuse the same file."""
+ SharedMetricsStore(
+ database_path=legacy_db, outbox_directory=tmp_path / "outbox"
+ )
+ assert _schema_version(legacy_db) == "2"
+
+ def test_migration_is_idempotent(self, legacy_db, tmp_path):
+ for _ in range(3):
+ SharedMetricsStore(
+ database_path=legacy_db, outbox_directory=tmp_path / "outbox"
+ )
+ columns = [
+ row[1]
+ for row in sqlite3.connect(legacy_db).execute(
+ "PRAGMA table_info(package_outbox)"
+ )
+ ]
+ assert len(columns) == len(set(columns)), "columns were added more than once"
+
+ def test_queries_written_before_this_change_still_work(self, legacy_db, tmp_path):
+ """The shipped export query selects named columns; it must be unaffected."""
+ SharedMetricsStore(
+ database_path=legacy_db, outbox_directory=tmp_path / "outbox"
+ )
+ connection = sqlite3.connect(legacy_db)
+ try:
+ rows = connection.execute(
+ """
+ SELECT package_id, payload_json
+ FROM package_outbox
+ WHERE exported_at IS NULL
+ ORDER BY created_at, package_id
+ """
+ ).fetchall()
+ finally:
+ connection.close()
+ assert rows == []
diff --git a/tests/hermes_cli/test_shared_metrics_send_wiring.py b/tests/hermes_cli/test_shared_metrics_send_wiring.py
new file mode 100644
index 0000000000..29f572c697
--- /dev/null
+++ b/tests/hermes_cli/test_shared_metrics_send_wiring.py
@@ -0,0 +1,446 @@
+"""Tests for wiring the sender into the shared-metrics export hook.
+
+The properties that matter here are negative ones: the interactive path must
+not block, and nothing must leave the machine unless the user opted in.
+"""
+
+from __future__ import annotations
+
+import threading
+import time
+
+import pytest
+
+from hermes_cli.observability import relay_shared_metrics as mod
+
+
+class FakeStore:
+ def __init__(self):
+ self.exported = 0
+
+ def create_and_export_package_if_due(self):
+ self.exported += 1
+ return []
+
+
+class RealBackedStore:
+ """A store with a genuine SQLite connection, for consent-state tests.
+
+ The consent edge detector writes to telemetry_state, and it is wrapped in
+ a broad except. Against a stub without _connection it would swallow an
+ AttributeError and silently do nothing — which is exactly the failure this
+ file needs to be able to catch.
+ """
+
+ def __init__(self, tmp_path):
+ from hermes_cli.observability.shared_metrics import SharedMetricsStore
+
+ self._real = SharedMetricsStore(
+ database_path=tmp_path / "m.db", outbox_directory=tmp_path / "o"
+ )
+ self.exported = 0
+
+ def _connection(self):
+ return self._real._connection()
+
+ def create_and_export_package_if_due(self):
+ self.exported += 1
+ return []
+
+
+class FakeSubscriber:
+ def __init__(self):
+ self.store = FakeStore()
+
+
+class Runtime(mod._Runtime):
+ """A _Runtime with the relay host stubbed out."""
+
+ def __init__(self):
+ self._sessions_lock = threading.RLock()
+ self._sessions = {}
+ self._task_creation_lock = threading.RLock()
+ self._task_sessions_lock = threading.RLock()
+ self._send_lock = threading.RLock()
+ self._send_thread = None
+ self._task_sessions = {}
+ self._turn_sessions = {}
+ self.subscriber = FakeSubscriber()
+
+
+@pytest.fixture
+def runtime():
+ return Runtime()
+
+
+def _config(**shared):
+ return {"telemetry": {"shared_metrics": shared}}
+
+
+@pytest.fixture
+def capture_sender(monkeypatch):
+ """Replace the sender with a recorder and return the record."""
+ record = {"passes": [], "endpoints": []}
+
+ class FakeSender:
+ def __init__(self, store, endpoint, **kwargs):
+ record["endpoints"].append(endpoint)
+
+ def send_pending(self):
+ record["passes"].append(time.time())
+
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics_sender.SharedMetricsSender",
+ FakeSender,
+ )
+ return record
+
+
+def _set_config(monkeypatch, config):
+ monkeypatch.setattr(
+ "hermes_cli.config.read_raw_config_readonly", lambda: config, raising=False
+ )
+
+
+class TestOptIn:
+ def test_no_send_when_nothing_is_configured(self, runtime, monkeypatch, capture_sender):
+ _set_config(monkeypatch, {})
+ runtime._export()
+ runtime._join_send_thread(timeout=1)
+ assert capture_sender["passes"] == []
+
+ def test_no_send_when_only_collection_is_on(self, runtime, monkeypatch, capture_sender):
+ _set_config(monkeypatch, _config(enabled=True))
+ runtime._export()
+ runtime._join_send_thread(timeout=1)
+ assert capture_sender["passes"] == []
+
+ def test_no_send_when_send_is_on_without_collection(
+ self, runtime, monkeypatch, capture_sender
+ ):
+ _set_config(monkeypatch, _config(enabled=False, send=True))
+ runtime._export()
+ runtime._join_send_thread(timeout=1)
+ assert capture_sender["passes"] == []
+
+ def test_sends_when_both_are_on(self, runtime, monkeypatch, capture_sender):
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ runtime._export()
+ runtime._join_send_thread(timeout=2)
+ assert len(capture_sender["passes"]) == 1
+
+ def test_uses_the_resolved_endpoint(self, runtime, monkeypatch, capture_sender):
+ _set_config(
+ monkeypatch,
+ _config(enabled=True, send=True, endpoint="https://staging.test/v1"),
+ )
+ runtime._export()
+ runtime._join_send_thread(timeout=2)
+ assert capture_sender["endpoints"] == ["https://staging.test/v1"]
+
+ def test_export_still_runs_when_sending_is_off(self, runtime, monkeypatch, capture_sender):
+ _set_config(monkeypatch, _config(enabled=True))
+ runtime._export()
+ assert runtime.subscriber.store.exported == 1
+
+
+class TestInteractivePathIsNotBlocked:
+ def test_export_returns_before_the_send_finishes(
+ self, runtime, monkeypatch
+ ):
+ started = threading.Event()
+ release = threading.Event()
+
+ class SlowSender:
+ def __init__(self, store, endpoint, **kwargs):
+ pass
+
+ def send_pending(self):
+ started.set()
+ release.wait(5)
+
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics_sender.SharedMetricsSender",
+ SlowSender,
+ )
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+
+ began = time.monotonic()
+ runtime._export()
+ elapsed = time.monotonic() - began
+
+ assert started.wait(2), "the send should have started"
+ assert elapsed < 1.0, "finish_task must not wait on the network"
+ release.set()
+ runtime._join_send_thread(timeout=5)
+
+ def test_the_send_thread_is_a_daemon(self, runtime, monkeypatch, capture_sender):
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ runtime._export()
+ with runtime._send_lock:
+ thread = runtime._send_thread
+ assert thread is not None
+ assert thread.daemon, "an unfinished send must not hold the process open"
+ runtime._join_send_thread(timeout=2)
+
+ def test_only_one_pass_runs_at_a_time(self, runtime, monkeypatch):
+ release = threading.Event()
+ starts = []
+
+ class SlowSender:
+ def __init__(self, store, endpoint, **kwargs):
+ pass
+
+ def send_pending(self):
+ starts.append(1)
+ release.wait(5)
+
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics_sender.SharedMetricsSender",
+ SlowSender,
+ )
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+
+ for _ in range(5):
+ runtime._export()
+ time.sleep(0.2)
+ assert len(starts) == 1, "hook fires must not pile up send passes"
+ release.set()
+ runtime._join_send_thread(timeout=5)
+
+
+class TestConsentWindows:
+ """Consent reconciliation must work from the relay, in any order.
+
+ Round 4's edge detector missed the idle-revocation path; round 5 found it
+ was also dead code whenever collection was off (handles_hook gated it).
+ These tests drive the relay entry points against the single reconciler
+ and assert on the interval table — the only consent state that exists.
+ """
+
+ def _runtime(self, tmp_path):
+ runtime = Runtime()
+ runtime.subscriber.store = RealBackedStore(tmp_path)
+ return runtime
+
+ def _windows(self, runtime):
+ with runtime.subscriber.store._connection() as connection:
+ return [
+ tuple(row)
+ for row in connection.execute(
+ "SELECT opened_at, last_confirmed_at, closed_at"
+ " FROM send_consent_windows ORDER BY opened_at"
+ )
+ ]
+
+ def test_revoking_while_idle_closes_the_window(
+ self, monkeypatch, tmp_path, capture_sender
+ ):
+ runtime = self._runtime(tmp_path)
+
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ runtime._send_exported_packages()
+
+ # User edits config.yaml: send: false. Hooks keep firing normally.
+ _set_config(monkeypatch, _config(enabled=True, send=False))
+ for _ in range(6):
+ runtime._send_exported_packages()
+
+ windows = self._windows(runtime)
+ assert windows and all(w[2] is not None for w in windows), (
+ f"revoking while idle left a window open: {windows}"
+ )
+
+ def test_replayed_observations_create_no_junk_windows(
+ self, monkeypatch, tmp_path, capture_sender
+ ):
+ """Reconciliation is idempotent — there is no edge to double-count."""
+ runtime = self._runtime(tmp_path)
+
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ for _ in range(4):
+ runtime._send_exported_packages()
+ _set_config(monkeypatch, _config(enabled=True, send=False))
+ for _ in range(4):
+ runtime._send_exported_packages()
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ for _ in range(4):
+ runtime._send_exported_packages()
+
+ assert len(self._windows(runtime)) == 2
+
+ def test_a_never_consented_user_gets_no_window(
+ self, monkeypatch, tmp_path, capture_sender
+ ):
+ runtime = self._runtime(tmp_path)
+ _set_config(monkeypatch, _config(enabled=True, send=False))
+ for _ in range(5):
+ runtime._send_exported_packages()
+
+ assert self._windows(runtime) == []
+
+ def test_re_enabling_opens_a_new_window_after_the_refusal(
+ self, monkeypatch, tmp_path, capture_sender
+ ):
+ """The refused gap must fall BETWEEN the two windows."""
+ runtime = self._runtime(tmp_path)
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ runtime._send_exported_packages()
+ _set_config(monkeypatch, _config(enabled=True, send=False))
+ runtime._send_exported_packages()
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ runtime._send_exported_packages()
+
+ windows = self._windows(runtime)
+ assert len(windows) == 2
+ first, second = windows
+ assert first[2] is not None, "first window must be closed"
+ assert second[2] is None, "second window must be open"
+ assert second[0] >= first[2], (
+ f"new window may not overlap the refused gap: {windows}"
+ )
+
+ def test_reconcile_runs_even_when_collection_is_disabled(
+ self, monkeypatch, tmp_path
+ ):
+ """Round-5 D1: enabled:false must not make consent handling dead code.
+
+ The module-level once-per-process reconciler must close the window
+ regardless of handles_hook(). Drives the real observe_lifecycle gate
+ path: handles_hook is False throughout.
+ """
+ from hermes_cli.observability.shared_metrics import SharedMetricsStore
+ from hermes_cli.observability.shared_metrics_sender import (
+ reconcile_send_consent,
+ )
+ from hermes_cli.sqlite_util import write_txn
+
+ # Lay the store out exactly as production does, under a redirected
+ # HERMES_HOME: the boot reconciler probes the default path (without
+ # constructing the store — the constructor creates directories), so
+ # the probe and the store must agree the way they do in production.
+ home = tmp_path / "home"
+ monkeypatch.setattr(
+ "hermes_constants.get_hermes_home", lambda: home
+ )
+ root = home / "telemetry" / "shared_metrics"
+ store = SharedMetricsStore(
+ database_path=root / "metrics.sqlite3",
+ outbox_directory=root / "outbox",
+ )
+ # A consent window is open from an earlier consented era.
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, True)
+
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics.SharedMetricsStore",
+ lambda *a, **k: store,
+ )
+ _set_config(monkeypatch, _config(enabled=False, send=False))
+ monkeypatch.setattr(mod, "_consent_reconcile_done", False)
+
+ # The full lifecycle entry point, with collection OFF.
+ mod.observe_lifecycle("finish_task")
+
+ with store._connection() as connection:
+ open_windows = connection.execute(
+ "SELECT COUNT(*) FROM send_consent_windows WHERE closed_at IS NULL"
+ ).fetchone()[0]
+ assert open_windows == 0, (
+ "enabled:false made the consent reconciler unreachable (D1)"
+ )
+
+
+
+class TestFailureIsolation:
+ def test_a_sender_crash_does_not_propagate(self, runtime, monkeypatch):
+ class Exploding:
+ def __init__(self, store, endpoint, **kwargs):
+ pass
+
+ def send_pending(self):
+ raise RuntimeError("boom")
+
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics_sender.SharedMetricsSender",
+ Exploding,
+ )
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+ runtime._export() # must not raise
+ runtime._join_send_thread(timeout=2)
+
+ def test_an_unreadable_config_does_not_break_export(self, runtime, monkeypatch, capture_sender):
+ def explode():
+ raise OSError("config unreadable")
+
+ monkeypatch.setattr(
+ "hermes_cli.config.read_raw_config_readonly", explode, raising=False
+ )
+ runtime._export()
+ assert runtime.subscriber.store.exported == 1
+ assert capture_sender["passes"] == []
+
+ def test_join_is_safe_with_no_thread(self, runtime):
+ runtime._join_send_thread(timeout=0.1)
+
+ def test_join_waits_for_an_in_flight_send(self, runtime, monkeypatch):
+ """shutdown() must give a started send a chance to finish.
+
+ A short-lived CLI exits straight after its final export; without the
+ join the daemon thread is killed mid-request, and the hook path is the
+ only delivery cadence this feature has.
+ """
+ finished = []
+ release = threading.Event()
+
+ class SlowSender:
+ def __init__(self, store, endpoint, **kwargs):
+ pass
+
+ def send_pending(self):
+ release.wait(3)
+ finished.append(True)
+
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics_sender.SharedMetricsSender",
+ SlowSender,
+ )
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+
+ runtime._export()
+ release.set()
+ runtime._join_send_thread(timeout=3)
+ assert finished == [True]
+
+ def test_shutdown_joins_the_send_thread(self, monkeypatch):
+ """shutdown() must actually wait, not merely mention the join.
+
+ Behavioural, not a source grep: an earlier version of this test
+ inspected getsource for a method name, which AGENTS.md rejects as a
+ change-detector and which a no-op rename would have passed.
+ """
+ runtime = Runtime()
+ released = threading.Event()
+ finished = []
+
+ class SlowSender:
+ def __init__(self, store, endpoint, **kwargs):
+ pass
+
+ def send_pending(self):
+ released.wait(3)
+ finished.append(True)
+
+ monkeypatch.setattr(
+ "hermes_cli.observability.shared_metrics_sender.SharedMetricsSender",
+ SlowSender,
+ )
+ _set_config(monkeypatch, _config(enabled=True, send=True))
+
+ # Stand in for the parts of shutdown() that need a live relay.
+ runtime._export()
+ assert runtime._send_thread is not None
+ released.set()
+ runtime._join_send_thread()
+ assert finished == [True], "shutdown returned while a send was in flight"
diff --git a/tests/hermes_cli/test_shared_metrics_sender.py b/tests/hermes_cli/test_shared_metrics_sender.py
new file mode 100644
index 0000000000..cbaff4c9ea
--- /dev/null
+++ b/tests/hermes_cli/test_shared_metrics_sender.py
@@ -0,0 +1,1029 @@
+"""Tests for the shared-metrics sender.
+
+Covers the four contract responses, the period-based consent gate, frozen
+identity across rotation, transactional claiming, and the invariant that
+matters most: a package file is never deleted, because the outbox is the
+user's local history rather than a send queue.
+"""
+
+from __future__ import annotations
+
+import json
+import sqlite3
+from datetime import datetime, timedelta, timezone
+
+import pytest
+
+from hermes_cli.observability.shared_metrics import SharedMetricsStore
+from hermes_cli.observability.shared_metrics_sender import (
+ MAX_ATTEMPTS,
+ MAX_PACKAGES_PER_PASS,
+ MAX_SEND_ATTEMPTS,
+ REQUEST_TIMEOUT_SECONDS,
+ SharedMetricsSender,
+ reconcile_send_consent,
+)
+from hermes_cli.sqlite_util import write_txn
+
+INSTALL_ID = "12a73e97-4de9-4766-830d-9ca1192c0420"
+NOW = datetime(2026, 8, 26, 12, 0, tzinfo=timezone.utc)
+ENDPOINT = "https://telemetry.test/v1/telemetry"
+
+
+class FakeResponse:
+ def __init__(self, status, retry_after=None, body=""):
+ self.status = status
+ self.retry_after = retry_after
+ self.body = body
+
+
+class FakeTransport:
+ """Records every POST and replays a scripted sequence of responses."""
+
+ def __init__(self, *responses):
+ self._responses = list(responses)
+ self.calls = []
+
+ def __call__(self, endpoint, payload, *, timeout):
+ self.calls.append({"endpoint": endpoint, "payload": payload, "timeout": timeout})
+ if not self._responses:
+ return FakeResponse(202)
+ item = self._responses.pop(0)
+ if isinstance(item, Exception):
+ raise item
+ return item
+
+ @property
+ def bodies(self):
+ return [json.loads(c["payload"].decode("utf-8")) for c in self.calls]
+
+
+@pytest.fixture
+def store(tmp_path):
+ """A store with a broad consent window already open.
+
+ Most tests exercise claiming/retry/transport, not the consent gate, and
+ the interval gate fails closed with no window. One window opened before
+ every test package and confirmed well past NOW keeps those tests about
+ what they are about. Gate tests clear it via _clear_consent.
+ """
+ built = SharedMetricsStore(
+ database_path=tmp_path / "metrics.sqlite3",
+ outbox_directory=tmp_path / "outbox",
+ )
+ _grant_consent(built)
+ return built
+
+
+def _grant_consent(
+ store,
+ opened=datetime(2026, 8, 20, tzinfo=timezone.utc),
+ confirmed_through=datetime(2026, 10, 1, tzinfo=timezone.utc),
+):
+ """Open a consent window and heartbeat it forward, via the real writer."""
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, True, now=opened)
+ reconcile_send_consent(connection, True, now=confirmed_through)
+
+
+def _revoke_consent(store, at):
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, False, now=at)
+
+
+def _clear_consent(store):
+ """Remove all consent state, for tests of the fail-closed default."""
+ with store._connection() as connection:
+ with write_txn(connection):
+ connection.execute("DELETE FROM send_consent_windows")
+ connection.execute("DELETE FROM consent_marks")
+
+
+def _add_package(store, package_id, period_day, *, exported=True, install_id=INSTALL_ID):
+ payload = {
+ "schema_version": "hermes.shared_metrics.v2",
+ "package_id": package_id,
+ "install_id": install_id,
+ "period_start": f"{period_day}T00:00:00Z",
+ "period_end": f"{period_day}T23:59:59Z",
+ "metrics": [{"name": "hermes.client.active", "type": "counter", "value": 1}],
+ }
+ with store._connection() as connection:
+ connection.execute(
+ """
+ INSERT INTO package_outbox(
+ package_id, period_start, period_end, payload_json,
+ created_at, exported_at
+ ) VALUES (?, ?, ?, ?, ?, ?)
+ """,
+ (
+ package_id,
+ f"{period_day}T00:00:00Z",
+ f"{period_day}T23:59:59Z",
+ json.dumps(payload),
+ f"{period_day}T01:00:00Z",
+ f"{period_day}T01:00:01Z" if exported else None,
+ ),
+ )
+ path = store.outbox_directory / f"{package_id}.json"
+ path.write_text(json.dumps(payload, indent=2, sort_keys=True))
+ return path
+
+
+def _row(store, package_id):
+ with store._connection() as connection:
+ row = connection.execute(
+ """
+ SELECT send_state, sent_at, send_attempts, next_attempt_at,
+ last_error, sent_install_id
+ FROM package_outbox WHERE package_id = ?
+ """,
+ (package_id,),
+ ).fetchone()
+ return dict(
+ send_state=row[0],
+ sent_at=row[1],
+ send_attempts=row[2],
+ next_attempt_at=row[3],
+ last_error=row[4],
+ sent_install_id=row[5],
+ )
+
+
+def _iso(moment):
+ return moment.astimezone(timezone.utc).isoformat().replace("+00:00", "Z")
+
+
+def _sender(store, transport, **kwargs):
+ return SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=transport,
+ sleep=lambda _s: None,
+ now=lambda: NOW,
+ **kwargs,
+ )
+
+
+class TestContractResponses:
+ def test_202_marks_sent(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(202))
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.sent == 1
+ row = _row(store, "pkg-1")
+ assert row["send_state"] == "sent"
+ assert row["sent_at"] is not None
+
+ def test_400_is_permanent_and_never_retried(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(400, body='{"error":"invalid_envelope"}'))
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.rejected == 1
+ assert len(transport.calls) == 1, "a 400 must not be retried"
+ assert _row(store, "pkg-1")["send_state"] == "rejected"
+
+ # A later pass must not pick it up again.
+ transport2 = FakeTransport(FakeResponse(202))
+ _sender(store, transport2).send_pending()
+ assert transport2.calls == []
+
+ @pytest.mark.parametrize("status", [401, 403, 404, 422, 500, 503])
+ def test_unspecified_statuses_are_retried_not_discarded(self, store, status):
+ """403 is the ingest origin guard; a bad edge config must not lose data."""
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(*[FakeResponse(status)] * 3)
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.deferred == 1
+ assert _row(store, "pkg-1")["send_state"] == "pending"
+
+ def test_413_is_permanent(self, store):
+ """A package over the 1 MiB cap cannot shrink by being retried."""
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(413))
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.rejected == 1
+ assert len(transport.calls) == 1
+
+ def test_429_defers_using_retry_after(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(429, retry_after="120"))
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.deferred == 1
+ assert len(transport.calls) == 1, "429 waits rather than burning attempts"
+ row = _row(store, "pkg-1")
+ assert row["send_state"] == "pending"
+ assert row["next_attempt_at"] == "2026-08-26T12:02:00Z"
+
+ def test_429_without_retry_after_still_defers(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(429))
+ _sender(store, transport).send_pending()
+ assert _row(store, "pkg-1")["next_attempt_at"] > "2026-08-26T12:00:00Z"
+
+ def test_absurd_retry_after_is_clamped(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(429, retry_after="99999999"))
+ _sender(store, transport).send_pending()
+ # clamped to 24h, not years
+ assert _row(store, "pkg-1")["next_attempt_at"] <= "2026-08-27T12:00:00Z"
+
+ def test_5xx_retries_then_defers(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(
+ FakeResponse(503), FakeResponse(503), FakeResponse(503)
+ )
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.deferred == 1
+ assert len(transport.calls) == 3, "three in-process attempts"
+ assert _row(store, "pkg-1")["send_state"] == "pending"
+
+ def test_5xx_then_success_within_the_same_pass(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(503), FakeResponse(202))
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.sent == 1
+ assert len(transport.calls) == 2
+
+ def test_transport_failure_is_retryable(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(
+ OSError("offline"), OSError("offline"), FakeResponse(202)
+ )
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.sent == 1
+
+ def test_persistent_offline_defers_without_raising(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(*[OSError("offline")] * 3)
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.deferred == 1
+ assert "OSError" in _row(store, "pkg-1")["last_error"]
+
+
+class TestConsentGate:
+ def test_packages_from_before_opt_in_are_never_sent(self, store):
+ # Consent opens on Aug 24; the "old" package's period predates it.
+ _clear_consent(store)
+ _grant_consent(store, opened=datetime(2026, 8, 24, tzinfo=timezone.utc))
+ _add_package(store, "old", "2026-08-20")
+ _add_package(store, "new", "2026-08-26")
+ transport = FakeTransport(FakeResponse(202))
+ _sender(store, transport).send_pending()
+ assert [b["package_id"] for b in transport.bodies] == ["new"]
+
+ def test_a_period_straddling_opt_in_day_is_sent_whole(self, store):
+ """The head/tail bug: both packages for the opt-in period must go."""
+ _add_package(store, "head", "2026-08-26")
+ _add_package(store, "tail", "2026-08-26") # created later, same period
+ transport = FakeTransport(FakeResponse(202), FakeResponse(202))
+ _sender(store, transport).send_pending()
+ assert sorted(b["package_id"] for b in transport.bodies) == ["head", "tail"]
+
+ def test_opt_in_is_immortalised_as_a_window_not_a_day(self, store):
+ """The window survives replayed observations without moving."""
+ with store._connection() as connection:
+ rows = connection.execute(
+ "SELECT opened_at, closed_at FROM send_consent_windows"
+ ).fetchall()
+ assert len(rows) == 1 and rows[0][1] is None
+ _grant_consent(store) # replay: must not create a second window
+ with store._connection() as connection:
+ count = connection.execute(
+ "SELECT COUNT(*) FROM send_consent_windows"
+ ).fetchone()[0]
+ assert count == 1
+
+ def test_no_consent_window_means_nothing_is_sent(self, store):
+ """The gate fails closed: absence of a window is absence of consent."""
+ _clear_consent(store)
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(202))
+ _sender(store, transport).send_pending()
+ assert transport.calls == []
+
+ def test_unexported_packages_are_skipped(self, store):
+ _add_package(store, "pending-export", "2026-08-26", exported=False)
+ transport = FakeTransport(FakeResponse(202))
+ _sender(store, transport).send_pending()
+ assert transport.calls == []
+
+ def test_revoking_then_re_enabling_never_releases_the_off_window(self, store):
+ """The R3/R5 leak: re-opt-in must not release the refused interval.
+
+ Under the interval model the refused days fall BETWEEN two windows;
+ no later observation can place them inside one, so the property holds
+ for any number of on/off cycles — not just the single cycle the old
+ moving day-stamp was patched to survive.
+ """
+ _clear_consent(store)
+ _grant_consent(store, opened=NOW - timedelta(days=2), confirmed_through=NOW)
+ _add_package(store, "consented", "2026-08-25")
+
+ # User turns sending off; packages keep being collected for 3 days.
+ _revoke_consent(store, at=NOW)
+ for day in ("2026-08-27", "2026-08-28", "2026-08-29"):
+ _add_package(store, f"refused-{day}", day)
+
+ # User re-enables 5 days later; heartbeat confirms past the horizon.
+ later = NOW + timedelta(days=5)
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, True, now=later)
+ reconcile_send_consent(
+ connection, True, now=later + timedelta(days=30)
+ )
+
+ transport = FakeTransport(*[FakeResponse(202)] * 10)
+ SharedMetricsSender(
+ store, ENDPOINT, post=transport, sleep=lambda _s: None, now=lambda: later
+ ).send_pending()
+
+ sent = [json.loads(c["payload"])["package_id"] for c in transport.calls]
+ assert not any("refused" in pid for pid in sent), (
+ f"transmitted packages collected while sending was off: {sent}"
+ )
+ # And the interval model's improvement over the day-stamp: the
+ # pre-revocation consented package is NOT collateral damage.
+ assert "consented" in sent, (
+ "the consented backlog was destroyed by the revoke/re-enable cycle"
+ )
+
+ def test_a_package_from_after_re_enabling_is_sent(self, store):
+ """The revocation handling must not wedge sending off permanently."""
+ _clear_consent(store)
+ _grant_consent(store, opened=NOW - timedelta(days=2), confirmed_through=NOW)
+ _revoke_consent(store, at=NOW)
+
+ later = NOW + timedelta(days=5)
+ with store._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(connection, True, now=later)
+ reconcile_send_consent(
+ connection, True, now=later + timedelta(days=10)
+ )
+ _add_package(store, "after-re-optin", (later + timedelta(days=1)).date().isoformat())
+ transport = FakeTransport(FakeResponse(202))
+ SharedMetricsSender(
+ store, ENDPOINT, post=transport, sleep=lambda _s: None,
+ now=lambda: later + timedelta(days=2),
+ ).send_pending()
+ assert len(transport.calls) == 1
+
+
+class TestIdentity:
+ def test_the_stable_install_id_is_transmitted_as_is(self, store):
+ """Product decision 2026-08-27: no pseudonymization.
+
+ The wire body carries the profile-scoped install_id verbatim. This
+ test is the deliberate inversion of the pre-decision assertion that
+ the raw id never crossed the wire.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(202))
+ _sender(store, transport).send_pending()
+ assert transport.bodies[0]["install_id"] == INSTALL_ID
+
+ def test_transmitted_id_is_frozen_on_the_row(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(503), FakeResponse(202))
+ _sender(store, transport).send_pending()
+ assert _row(store, "pkg-1")["sent_install_id"] == transport.bodies[0]["install_id"]
+
+ def test_retries_send_identical_bytes(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(503), FakeResponse(503), FakeResponse(202))
+ _sender(store, transport).send_pending()
+ payloads = {c["payload"] for c in transport.calls}
+ assert len(payloads) == 1, "a resend must be byte-identical per the contract"
+
+ def test_only_install_id_differs_from_the_stored_package(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ transport = FakeTransport(FakeResponse(202))
+ _sender(store, transport).send_pending()
+ sent = transport.bodies[0]
+ with store._connection() as connection:
+ stored = json.loads(
+ connection.execute(
+ "SELECT payload_json FROM package_outbox WHERE package_id = 'pkg-1'"
+ ).fetchone()[0]
+ )
+ assert set(sent) == set(stored)
+ for key in stored:
+ if key != "install_id":
+ assert sent[key] == stored[key]
+
+
+class TestOutboxIsNotAQueue:
+ def test_a_sent_package_file_is_not_deleted(self, store):
+ path = _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(202))).send_pending()
+ assert path.exists(), "the outbox is the user's history, not a send queue"
+
+ def test_a_rejected_package_file_is_not_deleted(self, store):
+ path = _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(400))).send_pending()
+ assert path.exists()
+
+ def test_the_package_row_survives_sending(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(202))).send_pending()
+ with store._connection() as connection:
+ assert connection.execute(
+ "SELECT COUNT(*) FROM package_outbox WHERE package_id = 'pkg-1'"
+ ).fetchone()[0] == 1
+
+
+class TestClaimingAndBounds:
+ def test_a_sent_package_is_not_resent(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(202))).send_pending()
+ second = FakeTransport(FakeResponse(202))
+ _sender(store, second).send_pending()
+ assert second.calls == []
+
+ def test_a_deferred_package_is_skipped_until_due(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(429, retry_after="600"))).send_pending()
+ second = FakeTransport(FakeResponse(202))
+ _sender(store, second).send_pending()
+ assert second.calls == [], "backoff must survive within the same process"
+
+ def test_a_deferred_package_is_retried_once_due(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(429, retry_after="60"))).send_pending()
+
+ later = SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=(transport := FakeTransport(FakeResponse(202))),
+ sleep=lambda _s: None,
+ now=lambda: NOW + timedelta(minutes=5),
+ )
+ later.send_pending()
+ assert len(transport.calls) == 1
+
+ def test_attempts_are_counted(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(429))).send_pending()
+ assert _row(store, "pkg-1")["send_attempts"] == 1
+
+ def test_a_pass_is_bounded(self, store):
+ for i in range(MAX_PACKAGES_PER_PASS + 5):
+ _add_package(store, f"pkg-{i:02d}", "2026-08-26")
+ transport = FakeTransport(*[FakeResponse(202)] * 40)
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.sent == MAX_PACKAGES_PER_PASS
+
+ def test_two_concurrent_passes_do_not_double_send(self, store):
+ """Claiming is what stops two Hermes processes duplicating work.
+
+ The second pass must RECORD what it saw rather than raise: _send_one
+ catches every exception as a retryable transport failure, so an
+ assertion thrown inside a transport would be swallowed and this test
+ would pass no matter what the claim did.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+
+ first_calls = []
+ second_calls = []
+
+ def second_transport(endpoint, payload, *, timeout):
+ second_calls.append(payload)
+ return FakeResponse(202)
+
+ def transport(endpoint, payload, *, timeout):
+ first_calls.append(payload)
+ # A second sender runs while the first is mid-flight.
+ SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=second_transport,
+ sleep=lambda _s: None,
+ now=lambda: NOW,
+ ).send_pending()
+ return FakeResponse(202)
+
+ _sender(store, transport).send_pending()
+ assert len(first_calls) == 1
+ assert second_calls == [], (
+ "a concurrent pass claimed a package already in flight"
+ )
+
+ def test_a_claim_leases_the_row_long_enough_to_cover_a_worst_case_send(
+ self, store
+ ):
+ """The lease must outlast one package's worst legal duration.
+
+ Asserting merely "in the future" passed for a 1-second lease, which is
+ useless: a package can legally take three 30s timeouts plus backoff.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+ claimed = _sender(store, FakeTransport())._claim_next(NOW, set())
+ assert claimed is not None
+
+ worst_case = REQUEST_TIMEOUT_SECONDS * MAX_ATTEMPTS + 1 + 5 + 25
+ deadline = NOW + timedelta(seconds=worst_case)
+ assert _row(store, "pkg-1")["next_attempt_at"] >= _iso(deadline), (
+ "lease expires before a single package can legally finish"
+ )
+
+ def test_a_slow_multi_package_pass_does_not_lose_its_lease(self, store):
+ """Regression: a batch-wide lease expired while later rows were sent.
+
+ One package can legally take ~96s (three 30s timeouts plus backoff).
+ With 20 rows claimed under one shared lease, the later rows' leases
+ expired mid-pass and a second process re-sent them. Packages are now
+ claimed one at a time, immediately before transmission.
+ """
+ for i in range(3):
+ _add_package(store, f"pkg-{i}", "2026-08-26")
+
+ clock = {"t": NOW}
+ first_posts, second_posts = [], []
+
+
+ def transport(endpoint, payload, *, timeout):
+ pid = json.loads(payload)["package_id"]
+ first_posts.append(pid)
+ # Burn the worst-case time budget for a single package.
+ clock["t"] += timedelta(seconds=96)
+ # A concurrent process probes for work while this package is still
+ # in flight. It must not be able to claim the package we hold.
+ # Restricted to that package so the probe cannot legitimately pick
+ # up the OTHER pending rows and make the assertion ambiguous.
+ held = _row(store, pid)
+ if held["next_attempt_at"] is not None:
+ eligible = held["next_attempt_at"] <= _iso(clock["t"])
+ if eligible and held["send_state"] != "sent":
+ second_posts.append(pid)
+ return FakeResponse(202)
+
+ SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=transport,
+ sleep=lambda _s: None,
+ now=lambda: clock["t"],
+ ).send_pending()
+
+ assert sorted(first_posts) == ["pkg-0", "pkg-1", "pkg-2"]
+ assert second_posts == [], (
+ f"a concurrent pass re-sent {second_posts} after a lease expired"
+ )
+
+ def test_a_re_eligible_head_row_does_not_starve_the_tail(self, store):
+ """Regression: `seen` terminated the pass instead of skipping a row.
+
+ The claim query is LIMIT 1. When the oldest row was already handled
+ this pass but had become eligible again (short Retry-After, or a pass
+ outliving the 15-minute failure backoff), _claim_next returned None
+ and send_pending read that as "queue empty", abandoning every healthy
+ package behind it. Measured: 10 of 19 delivered.
+ """
+ _add_package(store, "aaa-head", "2026-08-26")
+ for i in range(5):
+ _add_package(store, f"zzz-{i}", "2026-08-26")
+ # Order by created_at puts the head first.
+ with store._connection() as connection:
+ connection.execute(
+ "UPDATE package_outbox SET created_at = '2026-08-26T00:00:00Z'"
+ " WHERE package_id = 'aaa-head'"
+ )
+
+ posts = []
+
+ def transport(endpoint, payload, *, timeout):
+ pid = json.loads(payload)["package_id"]
+ posts.append(pid)
+ if pid == "aaa-head":
+ # Well-behaved service: retry in one second, so the head is
+ # eligible again immediately.
+ return FakeResponse(429, retry_after="1")
+ return FakeResponse(202)
+
+ clock = {"t": NOW}
+ SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=transport,
+ sleep=lambda _s: None,
+ now=lambda: clock["t"] + timedelta(seconds=30 * len(posts)),
+ ).send_pending()
+
+ delivered = {p for p in posts if p.startswith("zzz")}
+ assert delivered == {f"zzz-{i}" for i in range(5)}, (
+ f"tail starved by a re-eligible head row; delivered {delivered}"
+ )
+
+ def test_a_poisoned_package_is_abandoned_eventually(self, store):
+ """Without a ceiling a doomed row is retried ~160 times over 30 days.
+
+ Drives the real loop rather than pre-setting a counter: a row seeded
+ at exactly the limit is also excluded by other predicates, so that
+ version of this test passed even with the ceiling removed.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+
+ clock = {"t": NOW}
+ attempts = []
+
+ def transport(endpoint, payload, *, timeout):
+ attempts.append(1)
+ return FakeResponse(503)
+
+ # Run many passes, always well past any backoff, as a month of hook
+ # fires against a permanently failing package would.
+ for i in range(60):
+ SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=transport,
+ sleep=lambda _s: None,
+ now=lambda: clock["t"] + timedelta(hours=i),
+ ).send_pending()
+
+ row = _row(store, "pkg-1")
+ assert row["send_attempts"] <= MAX_SEND_ATTEMPTS, (
+ f"package retried {row['send_attempts']} times with no ceiling"
+ )
+ assert len(attempts) < 100, (
+ f"{len(attempts)} requests burned on one doomed package"
+ )
+
+ def test_a_lapsed_claimant_yields_even_before_anyone_reclaims(self, store):
+ """Seventh review: the check-to-POST expiry race.
+
+ A claims, sleeps past its own lease, and wakes BEFORE any other
+ process reclaims. Its token is still in the row, so a read-only
+ ownership check passes — and then B reclaims while A's POST is in
+ flight: both send. The pre-POST renewal must instead REJECT a
+ claimant whose lease already expired, whether or not anyone has
+ reclaimed yet, because expiry alone means another process may claim
+ at any moment.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+
+ posts = []
+ sender_a = SharedMetricsSender(
+ store, ENDPOINT,
+ post=lambda e, p, *, timeout: (posts.append("A"), FakeResponse(202))[1],
+ sleep=lambda _s: None,
+ now=lambda: clock["t"],
+ )
+ clock = {"t": NOW}
+ claimed = sender_a._claim_next(NOW, set())
+ assert claimed is not None and not claimed["skip"]
+
+ # Suspended past the 300s lease; wakes with the row NOT yet reclaimed.
+ clock["t"] = NOW + timedelta(seconds=400)
+ result = sender_a._send_one(claimed)
+
+ assert posts == [], (
+ "a claimant with an expired lease transmitted before renewal"
+ )
+ assert result == "deferred"
+ # The row must remain claimable by the next process.
+ row = _row(store, "pkg-1")
+ assert row["send_state"] == "pending"
+
+ def test_renewal_extends_the_lease_across_the_post(self, store):
+ """A healthy in-lease claimant renews and its POST is covered.
+
+ Round-8 review: the original assertion was `>=` under a frozen
+ clock, which a renewal that matches the row but never extends the
+ lease also satisfies — the exact mutant that double-POSTs (the
+ un-extended lease expires mid-POST and a second process reclaims).
+ The renewal must move the deadline STRICTLY forward to now + lease,
+ so renew from a later clock and require the exact new deadline.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+ clock = {"t": NOW}
+ sender = SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=lambda e, p, *, timeout: FakeResponse(202),
+ sleep=lambda _s: None,
+ now=lambda: clock["t"],
+ )
+ claimed = sender._claim_next(NOW, set())
+ assert claimed is not None
+ lease_before = _row(store, "pkg-1")["next_attempt_at"]
+
+ # 100s into the (300s) lease: still healthy, renews mid-flight.
+ clock["t"] = NOW + timedelta(seconds=100)
+ assert sender._renew_claim("pkg-1", claimed["claim_token"]) is True
+ lease_after = _row(store, "pkg-1")["next_attempt_at"]
+ assert lease_after > lease_before, (
+ "renewal granted authority without extending the lease"
+ )
+ # And not just 'later': the full fresh lease from the renewal clock.
+ expected = (NOW + timedelta(seconds=100 + 300)).strftime(
+ "%Y-%m-%dT%H:%M:%SZ"
+ )
+ assert lease_after == expected
+
+ def test_a_lapsed_claimant_resuming_after_reclaim_cannot_double_post(
+ self, store
+ ):
+ """PR-review P1: expiry -> reclaim -> old claimant resumes.
+
+ A claims, then is suspended (laptop lid) BEFORE its POST. The lease
+ expires; B reclaims and POSTs; A wakes and proceeds. The pre-POST
+ ownership check must make A yield without transmitting.
+
+ Scope note: the check closes the claim->POST gap. A suspension that
+ lands mid-POST (bytes already leaving) is not client-fixable — that
+ residual needs server-side dedupe and is documented on _send_one.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+
+ posts = []
+
+ def post_a(endpoint, payload, *, timeout):
+ posts.append("A")
+ return FakeResponse(202)
+
+ def post_b(endpoint, payload, *, timeout):
+ posts.append("B")
+ return FakeResponse(202)
+
+ sender_a = SharedMetricsSender(
+ store, ENDPOINT, post=post_a, sleep=lambda _s: None, now=lambda: NOW
+ )
+ # A claims, then the process is suspended before _send_one runs.
+ claimed_a = sender_a._claim_next(NOW, set())
+ assert claimed_a is not None and not claimed_a["skip"]
+
+ # 400s later (past the 300s lease) B claims and completes the send.
+ later = NOW + timedelta(seconds=400)
+ sender_b = SharedMetricsSender(
+ store, ENDPOINT, post=post_b, sleep=lambda _s: None, now=lambda: later
+ )
+ outcome_b = sender_b.send_pending()
+ assert outcome_b.sent == 1
+
+ # A resumes exactly where it left off.
+ result_a = sender_a._send_one(claimed_a)
+
+ row = _row(store, "pkg-1")
+ assert posts == ["B"], (
+ f"a lapsed claimant transmitted after reclaim: {posts}"
+ )
+ assert result_a == "deferred"
+ assert row["send_state"] == "sent", "B's settlement must stand"
+
+ def test_a_lapsed_claimants_backoff_cannot_clobber_the_new_claim(self, store):
+ """The token must fence DEFERS too, not just the 202 settlement.
+
+ A's transport fails after B has reclaimed; A's backoff write must
+ not move next_attempt_at under B's live lease.
+ """
+ _add_package(store, "pkg-1", "2026-08-26")
+ sender_a = SharedMetricsSender(
+ store, ENDPOINT,
+ post=FakeTransport(OSError("net"), OSError("net"), OSError("net")),
+ sleep=lambda _s: None, now=lambda: NOW,
+ )
+ claimed_a = sender_a._claim_next(NOW, set())
+ assert claimed_a is not None and not claimed_a["skip"]
+
+ later = NOW + timedelta(seconds=400)
+ sender_b = SharedMetricsSender(
+ store, ENDPOINT, post=FakeTransport(),
+ sleep=lambda _s: None, now=lambda: later,
+ )
+ claimed_b = sender_b._claim_next(later, set())
+ assert claimed_b is not None and not claimed_b["skip"]
+ lease_b = _row(store, "pkg-1")["next_attempt_at"]
+
+ # A's exhausted retries try to write a 15-minute backoff.
+ result = sender_a._send_one(claimed_a)
+ assert result == "deferred"
+ assert _row(store, "pkg-1")["next_attempt_at"] == lease_b, (
+ "a lapsed claimant's backoff overwrote the live claim's lease"
+ )
+
+ def test_an_expired_lease_is_reclaimed(self, store):
+ """A process killed mid-pass must not strand its packages."""
+ _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(OSError("killed"), OSError(""), OSError(""))).send_pending()
+
+ later = SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=(transport := FakeTransport(FakeResponse(202))),
+ sleep=lambda _s: None,
+ now=lambda: NOW + timedelta(hours=2),
+ )
+ later.send_pending()
+ assert len(transport.calls) == 1
+
+ def test_a_lapsed_sender_cannot_resurrect_a_sent_package(self, store):
+ """Terminal state must win over a straggler's write."""
+ _add_package(store, "pkg-1", "2026-08-26")
+ _sender(store, FakeTransport(FakeResponse(202))).send_pending()
+ assert _row(store, "pkg-1")["send_state"] == "sent"
+
+ # A straggler from an earlier pass tries to defer the same row.
+ _sender(store, FakeTransport())._defer("pkg-1", 600, "stale")
+ assert _row(store, "pkg-1")["send_state"] == "sent", (
+ "a lapsed pass overwrote a completed send"
+ )
+
+
+class TestResilience:
+ def test_a_corrupt_row_does_not_stop_the_pass(self, store):
+ _add_package(store, "good", "2026-08-26")
+ with store._connection() as connection:
+ connection.execute(
+ """
+ INSERT INTO package_outbox(
+ package_id, period_start, period_end, payload_json,
+ created_at, exported_at
+ ) VALUES ('bad', '2026-08-26T00:00:00Z', '2026-08-26T23:59:59Z',
+ 'not json', '2026-08-26T00:00:00Z', '2026-08-26T01:00:00Z')
+ """
+ )
+ transport = FakeTransport(*[FakeResponse(202)] * 5)
+ outcome = _sender(store, transport).send_pending()
+ assert outcome.sent >= 1
+
+ @pytest.mark.parametrize(
+ "payload_json",
+ [
+ '["a", "list"]',
+ "null",
+ '"a string"',
+ "42",
+ '{"no_install_id": true}',
+ '{"install_id": ""}',
+ '{"install_id": null}',
+ ],
+ )
+ def test_valid_json_that_is_not_a_usable_package_is_skipped(
+ self, store, payload_json
+ ):
+ """Regression: a top-level array parsed fine, then .get() raised.
+
+ The AttributeError escaped the claim transaction and blocked every
+ healthy package behind it.
+ """
+ with store._connection() as connection:
+ connection.execute(
+ """
+ INSERT INTO package_outbox(
+ package_id, period_start, period_end, payload_json,
+ created_at, exported_at
+ ) VALUES ('bad', '2026-08-26T00:00:00Z', '2026-08-26T23:59:59Z',
+ ?, '2026-08-26T00:00:00Z', '2026-08-26T01:00:00Z')
+ """,
+ (payload_json,),
+ )
+ _add_package(store, "good", "2026-08-26")
+
+ transport = FakeTransport(*[FakeResponse(202)] * 5)
+ outcome = _sender(store, transport).send_pending()
+
+ assert outcome.sent == 1, "the healthy package must still go out"
+ assert [json.loads(c["payload"])["package_id"] for c in transport.calls] == [
+ "good"
+ ]
+ assert _row(store, "bad")["send_state"] == "rejected"
+
+ def test_send_pending_never_raises_on_a_broken_database(self, store, tmp_path):
+ store.database_path.write_text("this is not a database")
+ outcome = _sender(store, FakeTransport(FakeResponse(202))).send_pending()
+ assert outcome.sent == 0
+
+
+class TestConsentRevocation:
+ """`send: false` must stop an in-flight pass, not just the next one."""
+
+ def test_revoking_consent_mid_pass_stops_further_sends(self, store):
+ for i in range(4):
+ _add_package(store, f"pkg-{i}", "2026-08-26")
+
+ consented = {"value": True}
+ posts = []
+
+ def transport(endpoint, payload, *, timeout):
+ posts.append(json.loads(payload)["package_id"])
+ consented["value"] = False # user flips send off during the pass
+ return FakeResponse(202)
+
+ outcome = SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=transport,
+ sleep=lambda _s: None,
+ now=lambda: NOW,
+ consent_check=lambda: consented["value"],
+ ).send_pending()
+
+ assert len(posts) == 1, f"kept sending after consent was revoked: {posts}"
+ assert outcome.sent == 1
+
+ def test_no_send_at_all_when_consent_is_already_false(self, store):
+ _add_package(store, "pkg-1", "2026-08-26")
+ posts = []
+ SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=lambda *a, **k: posts.append(1) or FakeResponse(202),
+ sleep=lambda _s: None,
+ now=lambda: NOW,
+ consent_check=lambda: False,
+ ).send_pending()
+ assert posts == []
+
+ def test_an_unreadable_consent_check_fails_closed(self, store):
+ """If consent cannot be established, do not transmit."""
+ _add_package(store, "pkg-1", "2026-08-26")
+ posts = []
+
+ def explode():
+ raise OSError("config unreadable")
+
+ SharedMetricsSender(
+ store,
+ ENDPOINT,
+ post=lambda *a, **k: posts.append(1) or FakeResponse(202),
+ sleep=lambda _s: None,
+ now=lambda: NOW,
+ consent_check=explode,
+ ).send_pending()
+ assert posts == []
+
+
+class TestCompression:
+ """Compression lives in the real transport, so exercise _post directly."""
+
+ def _captured_request(self, payload: bytes):
+ import urllib.request
+
+ from hermes_cli.observability import shared_metrics_sender as mod
+
+ captured = {}
+
+ class FakeConn:
+ status = 202
+ headers = {}
+
+ def read(self, _n=None):
+ return b"{}"
+
+ def __enter__(self):
+ return self
+
+ def __exit__(self, *a):
+ return False
+
+ def fake_urlopen(request, timeout=None):
+ captured["data"] = request.data
+ captured["headers"] = {k.lower(): v for k, v in request.headers.items()}
+ return FakeConn()
+
+ original = urllib.request.urlopen
+ urllib.request.urlopen = fake_urlopen
+ try:
+ mod._post(ENDPOINT, payload, timeout=5)
+ finally:
+ urllib.request.urlopen = original
+ return captured
+
+ def test_large_payloads_are_gzipped(self):
+ payload = json.dumps({"filler": "x" * 20000}).encode("utf-8")
+ captured = self._captured_request(payload)
+ assert captured["data"][:2] == b"\x1f\x8b", "gzip magic bytes"
+ assert captured["headers"].get("Content-encoding".lower()) == "gzip"
+
+ def test_gzip_actually_shrinks_the_body(self):
+ payload = json.dumps({"filler": "x" * 20000}).encode("utf-8")
+ captured = self._captured_request(payload)
+ assert len(captured["data"]) < len(payload)
+
+ def test_gzip_is_deterministic_across_time(self):
+ """Kills the mtime footgun: gzip embeds a timestamp by default.
+
+ The in-pass retry test cannot catch this — both attempts compress
+ within the same second. Compressing the same bytes at two different
+ wall-clock seconds is what actually exercises mtime=0.
+ """
+ import time as _time
+
+ payload = json.dumps({"filler": "x" * 20000}).encode("utf-8")
+ first = self._captured_request(payload)["data"]
+ _time.sleep(1.1)
+ second = self._captured_request(payload)["data"]
+ assert first == second, (
+ "gzip output changed between seconds — mtime is being embedded"
+ )
+
+ def test_small_payloads_are_sent_plain(self):
+ payload = b'{"small": true}'
+ captured = self._captured_request(payload)
+ assert captured["data"] == payload
+ assert "content-encoding" not in captured["headers"]
diff --git a/tests/hermes_cli/test_shared_metrics_sender_e2e.py b/tests/hermes_cli/test_shared_metrics_sender_e2e.py
new file mode 100644
index 0000000000..85be9b2388
--- /dev/null
+++ b/tests/hermes_cli/test_shared_metrics_sender_e2e.py
@@ -0,0 +1,282 @@
+"""End-to-end test: the real sender against a real HTTP server.
+
+Everything else stubs the transport. This exercises the actual code path —
+urllib, gzip, headers, socket — against a live server on loopback, so a
+transport-level mistake that a fake would hide fails here instead.
+"""
+
+from __future__ import annotations
+
+import gzip
+import json
+import sqlite3
+import threading
+from datetime import datetime, timezone
+from http.server import BaseHTTPRequestHandler, HTTPServer
+
+import pytest
+
+from hermes_cli.observability.shared_metrics import SharedMetricsStore
+from hermes_cli.observability.shared_metrics_sender import SharedMetricsSender
+
+INSTALL_ID = "12a73e97-4de9-4766-830d-9ca1192c0420"
+NOW = datetime(2026, 8, 26, 12, 0, tzinfo=timezone.utc)
+
+
+class Ingest(BaseHTTPRequestHandler):
+ """A stand-in for the ingest service that records what it receives."""
+
+ received: list = []
+ script: list = []
+
+ def do_POST(self): # noqa: N802 - stdlib naming
+ length = int(self.headers.get("Content-Length") or 0)
+ raw = self.rfile.read(length)
+ if self.headers.get("Content-Encoding") == "gzip":
+ body = gzip.decompress(raw)
+ else:
+ body = raw
+ type(self).received.append(
+ {
+ "headers": {k.lower(): v for k, v in self.headers.items()},
+ "body": json.loads(body.decode("utf-8")),
+ # Keep the RAW request bytes: comparing only the parsed body
+ # would not notice a non-deterministic transport encoding.
+ "raw": raw,
+ "raw_len": len(raw),
+ "decoded_len": len(body),
+ }
+ )
+ status, payload, extra = (
+ type(self).script.pop(0) if type(self).script else (202, {}, {})
+ )
+ encoded = json.dumps(payload).encode("utf-8")
+ self.send_response(status)
+ self.send_header("Content-Type", "application/json")
+ self.send_header("Content-Length", str(len(encoded)))
+ for key, value in extra.items():
+ self.send_header(key, value)
+ self.end_headers()
+ self.wfile.write(encoded)
+
+ def log_message(self, format, *args): # noqa: A002 - stdlib signature
+ pass
+
+
+@pytest.fixture
+def server():
+ Ingest.received = []
+ Ingest.script = []
+ httpd = HTTPServer(("127.0.0.1", 0), Ingest)
+ thread = threading.Thread(target=httpd.serve_forever, daemon=True)
+ thread.start()
+ yield httpd
+ httpd.shutdown()
+ httpd.server_close()
+
+
+@pytest.fixture
+def store(tmp_path):
+ built = SharedMetricsStore(
+ database_path=tmp_path / "metrics.sqlite3",
+ outbox_directory=tmp_path / "outbox",
+ )
+ # Open a consent window covering the fixture packages; the interval gate
+ # fails closed without one, and this file tests transport, not consent.
+ from datetime import datetime, timezone
+
+ from hermes_cli.observability.shared_metrics_sender import (
+ reconcile_send_consent,
+ )
+ from hermes_cli.sqlite_util import write_txn
+
+ with built._connection() as connection:
+ with write_txn(connection):
+ reconcile_send_consent(
+ connection, True, now=datetime(2026, 8, 20, tzinfo=timezone.utc)
+ )
+ reconcile_send_consent(
+ connection, True, now=datetime(2026, 10, 1, tzinfo=timezone.utc)
+ )
+ return built
+
+
+def _endpoint(server):
+ host, port = server.server_address
+ return f"http://{host}:{port}/v1/telemetry"
+
+
+def _add(store, package_id, day="2026-08-26", metrics=1):
+ payload = {
+ "schema_version": "hermes.shared_metrics.v2",
+ "package_id": package_id,
+ "install_id": INSTALL_ID,
+ "generated_at": f"{day}T01:00:00Z",
+ "period_start": f"{day}T00:00:00Z",
+ "period_end": f"{day}T23:59:59Z",
+ "resource": {
+ "hermes_version": "0.20.5",
+ "os_family": "macos",
+ "architecture": "arm64",
+ "install_method": "git",
+ },
+ "metrics": [
+ {
+ "name": f"hermes.metric.{i}",
+ "type": "counter",
+ "dimensions": {"outcome": "ok"},
+ "value": i,
+ }
+ for i in range(metrics)
+ ],
+ }
+ with store._connection() as connection:
+ connection.execute(
+ """
+ INSERT INTO package_outbox(
+ package_id, period_start, period_end, payload_json,
+ created_at, exported_at
+ ) VALUES (?, ?, ?, ?, ?, ?)
+ """,
+ (
+ package_id,
+ f"{day}T00:00:00Z",
+ f"{day}T23:59:59Z",
+ json.dumps(payload),
+ f"{day}T01:00:00Z",
+ f"{day}T01:00:01Z",
+ ),
+ )
+ return payload
+
+
+def _sender(store, server):
+ return SharedMetricsSender(
+ store, _endpoint(server), sleep=lambda _s: None, now=lambda: NOW
+ )
+
+
+class TestRealTransport:
+ def test_a_package_is_delivered_and_marked_sent(self, store, server):
+ _add(store, "pkg-1")
+ outcome = _sender(store, server).send_pending()
+
+ assert outcome.sent == 1
+ assert len(Ingest.received) == 1
+ assert Ingest.received[0]["body"]["package_id"] == "pkg-1"
+
+ with store._connection() as connection:
+ state = connection.execute(
+ "SELECT send_state FROM package_outbox WHERE package_id = 'pkg-1'"
+ ).fetchone()[0]
+ assert state == "sent"
+
+ def test_the_stable_install_id_crosses_the_wire_as_is(self, store, server):
+ """Product decision 2026-08-27: the raw install_id is transmitted."""
+ _add(store, "pkg-1", metrics=40)
+ _sender(store, server).send_pending()
+ assert Ingest.received[0]["body"]["install_id"] == INSTALL_ID
+
+ def test_content_type_is_json(self, store, server):
+ _add(store, "pkg-1")
+ _sender(store, server).send_pending()
+ assert Ingest.received[0]["headers"]["content-type"] == "application/json"
+
+ def test_a_realistic_package_is_gzipped_over_the_wire(self, store, server):
+ # ~40 metrics matches the real outbox's larger packages.
+ _add(store, "pkg-1", metrics=120)
+ _sender(store, server).send_pending()
+ record = Ingest.received[0]
+ assert record["headers"].get("content-encoding") == "gzip"
+ assert record["raw_len"] < record["decoded_len"]
+
+ def test_the_server_can_parse_what_we_send(self, store, server):
+ """Proves the bytes are valid JSON after transport and decompression."""
+ original = _add(store, "pkg-1", metrics=120)
+ _sender(store, server).send_pending()
+ received = Ingest.received[0]["body"]
+ assert received["metrics"] == original["metrics"]
+ assert received["resource"] == original["resource"]
+
+ def test_400_is_permanent(self, store, server):
+ _add(store, "pkg-1")
+ Ingest.script = [(400, {"error": "invalid_envelope"}, {})]
+ outcome = _sender(store, server).send_pending()
+ assert outcome.rejected == 1
+ assert len(Ingest.received) == 1
+
+ def test_429_is_honoured(self, store, server):
+ _add(store, "pkg-1")
+ Ingest.script = [(429, {"error": "rate_limited"}, {"Retry-After": "90"})]
+ outcome = _sender(store, server).send_pending()
+ assert outcome.deferred == 1
+ with store._connection() as connection:
+ retry_at = connection.execute(
+ "SELECT next_attempt_at FROM package_outbox WHERE package_id = 'pkg-1'"
+ ).fetchone()[0]
+ assert retry_at == "2026-08-26T12:01:30Z"
+
+ def test_5xx_retries_then_succeeds(self, store, server):
+ _add(store, "pkg-1")
+ Ingest.script = [
+ (503, {"error": "storage_unavailable"}, {}),
+ (202, {"package_id": "pkg-1"}, {}),
+ ]
+ outcome = _sender(store, server).send_pending()
+ assert outcome.sent == 1
+ assert len(Ingest.received) == 2
+
+ def test_a_retry_sends_identical_bytes(self, store, server):
+ _add(store, "pkg-1", metrics=5)
+ Ingest.script = [(503, {}, {}), (202, {}, {})]
+ _sender(store, server).send_pending()
+ first, second = Ingest.received
+ assert first["body"] == second["body"]
+ assert first["raw"] == second["raw"], (
+ "the raw request bytes must match, not just the parsed body"
+ )
+
+ def test_a_gzipped_retry_is_byte_identical_on_the_wire(self, store, server):
+ """gzip embeds an mtime by default, which would break this."""
+ _add(store, "pkg-1", metrics=200)
+ Ingest.script = [(503, {}, {}), (202, {}, {})]
+ _sender(store, server).send_pending()
+ first, second = Ingest.received
+ assert first["headers"].get("content-encoding") == "gzip"
+ assert first["raw"] == second["raw"]
+
+ def test_several_packages_in_one_pass(self, store, server):
+ for i in range(5):
+ _add(store, f"pkg-{i}")
+ outcome = _sender(store, server).send_pending()
+ assert outcome.sent == 5
+ assert len(Ingest.received) == 5
+
+ def test_the_outbox_directory_is_untouched(self, store, server, tmp_path):
+ _add(store, "pkg-1")
+ marker = store.outbox_directory / "pkg-1.json"
+ marker.write_text('{"kept": true}')
+ _sender(store, server).send_pending()
+ assert marker.exists()
+ assert json.loads(marker.read_text()) == {"kept": True}
+
+ def test_a_dead_server_defers_without_raising(self, store, server):
+ _add(store, "pkg-1")
+ host, port = server.server_address
+ server.shutdown()
+ server.server_close()
+ sender = SharedMetricsSender(
+ store,
+ f"http://{host}:{port}/v1/telemetry",
+ sleep=lambda _s: None,
+ now=lambda: NOW,
+ )
+ outcome = sender.send_pending()
+ assert outcome.deferred == 1
+ with store._connection() as connection:
+ state, error = connection.execute(
+ "SELECT send_state, last_error FROM package_outbox"
+ " WHERE package_id = 'pkg-1'"
+ ).fetchone()
+ assert state == "pending"
+ assert error
diff --git a/tests/hermes_cli/test_shared_metrics_tools_toggle.py b/tests/hermes_cli/test_shared_metrics_tools_toggle.py
new file mode 100644
index 0000000000..462bfcbd90
--- /dev/null
+++ b/tests/hermes_cli/test_shared_metrics_tools_toggle.py
@@ -0,0 +1,89 @@
+"""Tests for the `hermes tools` shared-metrics consent toggle.
+
+AGENTS.md requires outbound telemetry to be reachable from a config gate, the
+setup prompt, AND `hermes tools`. These cover the third surface.
+"""
+
+from __future__ import annotations
+
+import pytest
+
+from hermes_cli.tools_config import (
+ _configure_shared_metrics_interactive,
+ _shared_metrics_menu_label,
+ _shared_metrics_state,
+)
+
+
+def _config(**shared):
+ return {"telemetry": {"shared_metrics": shared}}
+
+
+class TestState:
+ def test_missing_telemetry_section_is_off(self):
+ assert _shared_metrics_state({}) == (False, False)
+
+ def test_malformed_section_does_not_raise(self):
+ assert _shared_metrics_state({"telemetry": "nonsense"}) == (False, False)
+
+ def test_reads_both_flags(self):
+ assert _shared_metrics_state(_config(enabled=True, send=True)) == (True, True)
+
+
+class TestMenuLabel:
+ def test_off_state(self):
+ assert "off" in _shared_metrics_menu_label({})
+
+ def test_local_only_state(self):
+ label = _shared_metrics_menu_label(_config(enabled=True))
+ assert "collecting locally" in label
+ assert "Nous" not in label
+
+ def test_sending_state_names_the_destination(self):
+ label = _shared_metrics_menu_label(_config(enabled=True, send=True))
+ assert "sending to Nous" in label
+
+
+class TestToggle:
+ def test_enabling_send_persists(self, monkeypatch):
+ config = _config(enabled=True)
+ saved = {}
+ monkeypatch.setattr(
+ "hermes_cli.setup.prompt_yes_no", lambda *_a, **_k: True
+ )
+ monkeypatch.setattr(
+ "hermes_cli.setup._record_send_consent_change", lambda **_k: None
+ )
+ monkeypatch.setattr(
+ "hermes_cli.tools_config.save_config",
+ lambda cfg: saved.update({"cfg": cfg}),
+ )
+ _configure_shared_metrics_interactive(config)
+ assert config["telemetry"]["shared_metrics"]["send"] is True
+ assert saved, "a consent change must be written to disk"
+
+ def test_no_write_when_nothing_changed(self, monkeypatch):
+ config = _config(enabled=False, send=False)
+ saved = []
+ monkeypatch.setattr(
+ "hermes_cli.setup.prompt_yes_no", lambda *_a, **_k: False
+ )
+ monkeypatch.setattr(
+ "hermes_cli.tools_config.save_config", lambda cfg: saved.append(cfg)
+ )
+ _configure_shared_metrics_interactive(config)
+ assert saved == []
+
+ def test_disabling_collection_also_disables_sending(self, monkeypatch):
+ """The toggle must not leave send=true with nothing to send."""
+ config = _config(enabled=True, send=True)
+ monkeypatch.setattr(
+ "hermes_cli.setup.prompt_yes_no", lambda *_a, **_k: False
+ )
+ monkeypatch.setattr(
+ "hermes_cli.tools_config.save_config", lambda cfg: None
+ )
+ _configure_shared_metrics_interactive(config)
+ shared = config["telemetry"]["shared_metrics"]
+ assert shared["enabled"] is False
+ assert shared["send"] is False
diff --git a/tests/hermes_cli/test_startup_model_routing_87189.py b/tests/hermes_cli/test_startup_model_routing_87189.py
new file mode 100644
index 0000000000..ffa3b902a9
--- /dev/null
+++ b/tests/hermes_cli/test_startup_model_routing_87189.py
@@ -0,0 +1,151 @@
+"""Regression tests for startup model/provider routing (#87189)."""
+
+from hermes_cli import model_switch
+
+
+def test_startup_route_uses_configured_nous_provider(monkeypatch):
+ monkeypatch.setattr(model_switch, "DIRECT_ALIASES", {})
+ route = model_switch.resolve_startup_model_route(
+ "nous/deepseek-v4-pro",
+ user_providers={"nous": {"base_url": "https://inference.example/v1"}},
+ )
+ assert route == model_switch.StartupModelRoute("deepseek-v4-pro", "nous", "")
+
+
+def test_startup_route_keeps_configured_custom_provider_name(monkeypatch):
+ monkeypatch.setattr(model_switch, "DIRECT_ALIASES", {})
+ route = model_switch.resolve_startup_model_route(
+ "ollama/qwen3.5:4b",
+ user_providers={"ollama": {"base_url": "http://localhost:11434/v1"}},
+ )
+ assert route == model_switch.StartupModelRoute("qwen3.5:4b", "ollama", "")
+
+
+def test_startup_route_does_not_consume_aggregator_namespace(monkeypatch):
+ monkeypatch.setattr(model_switch, "DIRECT_ALIASES", {})
+ route = model_switch.resolve_startup_model_route(
+ "openrouter/anthropic/claude-sonnet",
+ user_providers={"openrouter": {"base_url": "https://openrouter.ai/api/v1"}},
+ )
+ assert route is None
+
+
+def test_startup_route_aggregator_native_slug_stays_on_aggregator(monkeypatch):
+ """On OpenRouter, ``anthropic/claude-...`` is an aggregator-native slug.
+
+ A ``providers.anthropic`` block in the same config must NOT steal the
+ route — bare vendor slugs resolve WITHIN the aggregator first
+ (aggregator-aware resolution contract).
+ """
+ monkeypatch.setattr(model_switch, "DIRECT_ALIASES", {})
+ monkeypatch.setattr(
+ "hermes_cli.models._find_openrouter_slug",
+ lambda name: "anthropic/claude-opus-4.6",
+ )
+ route = model_switch.resolve_startup_model_route(
+ "anthropic/claude-opus-4.6",
+ current_provider="openrouter",
+ user_providers={"anthropic": {"apiKey": "sk-test"}},
+ )
+ assert route is None
+
+
+def test_startup_route_non_aggregator_current_provider_still_routes(monkeypatch):
+ monkeypatch.setattr(model_switch, "DIRECT_ALIASES", {})
+ route = model_switch.resolve_startup_model_route(
+ "nous/deepseek-v4-pro",
+ current_provider="anthropic",
+ user_providers={"nous": {"base_url": "https://inference.example/v1"}},
+ )
+ assert route == model_switch.StartupModelRoute("deepseek-v4-pro", "nous", "")
+
+
+def test_startup_route_resolves_dict_alias_and_preserves_endpoint(monkeypatch):
+ monkeypatch.setattr(
+ model_switch,
+ "DIRECT_ALIASES",
+ {
+ "localqwen": model_switch.DirectAlias(
+ "qwen3.5:4b", "custom", "http://localhost:11434/v1"
+ )
+ },
+ )
+ route = model_switch.resolve_startup_model_route("localqwen")
+ assert route == model_switch.StartupModelRoute(
+ "qwen3.5:4b", "custom", "http://localhost:11434/v1"
+ )
+
+
+def test_startup_route_url_alias_never_keeps_foreign_provider_label(monkeypatch):
+ """A URL-bearing alias labelled ``anthropic`` must resolve as ``custom``.
+
+ Keeping the label would let the alias reach the anthropic
+ explicit-runtime branch with a foreign base_url and put the live vendor
+ token on the alias host's wire (#28660 / #83612).
+ """
+ monkeypatch.setattr(
+ model_switch,
+ "DIRECT_ALIASES",
+ {
+ "urlalias": model_switch.DirectAlias(
+ "qwen3.5:4b", "anthropic", "http://localhost:11434/v1"
+ )
+ },
+ )
+ route = model_switch.resolve_startup_model_route("urlalias")
+ assert route is not None
+ assert route.provider == "custom"
+ assert route.base_url == "http://localhost:11434/v1"
+
+
+def test_startup_route_alias_carries_own_api_key(monkeypatch):
+ monkeypatch.setattr(
+ model_switch,
+ "DIRECT_ALIASES",
+ {
+ "keyed": model_switch.DirectAlias(
+ "some-model",
+ "custom",
+ "https://proxy.example/v1",
+ api_key="sk-alias-key",
+ )
+ },
+ )
+ route = model_switch.resolve_startup_model_route("keyed")
+ assert route is not None
+ assert route.api_key == "sk-alias-key"
+
+
+def test_startup_route_explicit_provider_wins_over_alias_label(monkeypatch):
+ monkeypatch.setattr(
+ model_switch,
+ "DIRECT_ALIASES",
+ {"ds": model_switch.DirectAlias("deepseek-chat", "deepseek", "")},
+ )
+ route = model_switch.resolve_startup_model_route(
+ "ds", explicit_provider="openrouter"
+ )
+ assert route is not None
+ assert route.provider == "openrouter"
+ assert route.model == "deepseek-chat"
+
+
+def test_model_aliases_dict_entries_are_loaded(monkeypatch):
+ monkeypatch.setattr(
+ "hermes_cli.config.load_config",
+ lambda: {
+ "model": {
+ "aliases": {
+ "localqwen": {
+ "model": "qwen3.5:4b",
+ "provider": "custom",
+ "base_url": "http://localhost:11434/v1",
+ }
+ }
+ }
+ },
+ )
+ aliases = model_switch._load_direct_aliases()
+ assert aliases["localqwen"] == model_switch.DirectAlias(
+ "qwen3.5:4b", "custom", "http://localhost:11434/v1"
+ )
diff --git a/tests/hermes_cli/test_status.py b/tests/hermes_cli/test_status.py
index 4a5746b94d..37d0c0bb24 100644
--- a/tests/hermes_cli/test_status.py
+++ b/tests/hermes_cli/test_status.py
@@ -15,6 +15,18 @@ def test_show_status_all_does_not_print_keenable_key_value(monkeypatch, capsys,
assert sentinel not in output
+def test_show_status_all_does_not_print_tavily_key_value(monkeypatch, capsys, tmp_path):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ sentinel = "NONSECRET_SENTINEL_VALUE_DO_NOT_PRINT_TAVILY_123456"
+ monkeypatch.setenv("TAVILY_API_KEY", sentinel)
+
+ show_status(SimpleNamespace(all=True, deep=False))
+
+ output = capsys.readouterr().out
+ assert "Tavily" in output
+ assert sentinel not in output
+
+
def test_show_status_termux_gateway_section_skips_systemctl(monkeypatch, capsys, tmp_path):
from hermes_cli import status as status_mod
import hermes_cli.auth as auth_mod
diff --git a/tests/hermes_cli/test_terminal_notify.py b/tests/hermes_cli/test_terminal_notify.py
new file mode 100644
index 0000000000..fce8b90d5e
--- /dev/null
+++ b/tests/hermes_cli/test_terminal_notify.py
@@ -0,0 +1,50 @@
+"""display.bell_on_prompt / bell_on_complete also drive OSC 9 + Warp OSC 777 via _ring_bell."""
+
+import json
+
+from cli import HermesCLI
+from hermes_cli import terminal_notify
+
+_WARP_OK = {
+ "TERM_PROGRAM": "WarpTerminal",
+ "WARP_CLI_AGENT_PROTOCOL_VERSION": "1",
+ "WARP_CLIENT_VERSION": "v0.2026.08.01.00.00.stable_01",
+}
+
+
+def _ring(monkeypatch, *, flag_on, env, **kwargs):
+ for key in _WARP_OK:
+ monkeypatch.delenv(key, raising=False)
+ for key, value in env.items():
+ monkeypatch.setenv(key, value)
+ written = []
+ monkeypatch.setattr(terminal_notify, "_write_tty", written.append)
+ cli = HermesCLI.__new__(HermesCLI)
+ cli.bell_on_prompt = flag_on
+ cli.session_id = "sess-1"
+ cli._ring_bell(prompt=True, **kwargs)
+ return "".join(written)
+
+
+def test_osc9_body_emitted_and_sanitized_only_when_flag_on(monkeypatch):
+ out = _ring(monkeypatch, flag_on=True, env={}, context="approval\x1b\x07\x00\x7f!")
+ assert out == "\x1b]9;Hermes: approval!\x07"
+ assert _ring(monkeypatch, flag_on=False, env={}, context="approval") == ""
+
+
+def test_warp_osc777_only_under_supported_warp_build(monkeypatch):
+ out = _ring(monkeypatch, flag_on=True, env=_WARP_OK, context="approval", detail="rm -rf build")
+ prefix = "\x1b]777;notify;warp://cli-agent;"
+ assert out.count(prefix) == 1
+ payload = json.loads(out.split(prefix, 1)[1].rstrip("\x07"))
+ assert payload["agent"] == "hermes"
+ assert payload["event"] == "permission_request"
+ assert payload["summary"] == "rm -rf build"
+ assert payload["session_id"] == "sess-1"
+ assert payload["v"] == 1
+ # Broken build (advertises the protocol var but can't render) → OSC 9 only.
+ broken = dict(_WARP_OK, WARP_CLIENT_VERSION="v0.2026.03.25.08.24.stable_05")
+ assert prefix not in _ring(monkeypatch, flag_on=True, env=broken, context="approval")
+ # Not Warp at all → OSC 9 only.
+ not_warp = dict(_WARP_OK, TERM_PROGRAM="ghostty")
+ assert prefix not in _ring(monkeypatch, flag_on=True, env=not_warp, context="approval")
diff --git a/tests/hermes_cli/test_tools_config.py b/tests/hermes_cli/test_tools_config.py
index 8523bf3f94..ce19955cc8 100644
--- a/tests/hermes_cli/test_tools_config.py
+++ b/tests/hermes_cli/test_tools_config.py
@@ -245,6 +245,7 @@ def test_first_install_nous_auto_configures_video_gen(monkeypatch):
"FIRECRAWL_API_KEY",
"FIRECRAWL_API_URL",
"KEENABLE_API_KEY",
+ "TAVILY_API_KEY",
"PARALLEL_API_KEY",
"BROWSERBASE_API_KEY",
"BROWSERBASE_PROJECT_ID",
diff --git a/tests/hermes_cli/test_unified_pool_quirk.py b/tests/hermes_cli/test_unified_pool_quirk.py
new file mode 100644
index 0000000000..39c7c23ed7
--- /dev/null
+++ b/tests/hermes_cli/test_unified_pool_quirk.py
@@ -0,0 +1,233 @@
+"""The unified-pool vendor quirk must never misclassify a discrete card.
+
+On unified-memory NVIDIA devices under Windows, nvidia-smi answers from
+a WDDM carve-out while the CUDA allocator addresses the whole pool at
+full bandwidth. Budgeting from smi there produces false "larger than
+your GPU memory" rows and -ot CPU pinning that measures far slower than
+letting the allocator place everything.
+
+The other direction is the regression this file guards: a workstation
+card must never be budgeted as UMA. The driver's INTEGRATED attribute
+decides when readable (both directions); the attribute-less engine
+fallback needs two independent numeric gates, each alone unmeetable by
+any discrete card."""
+
+from __future__ import annotations
+
+import hermes_cli.local_runtime.hardware as hw
+
+GIB = 1 << 30
+
+# Representative unified-memory device shape: a 48 GiB box whose smi
+# reports only the 16 GiB carve-out while the allocator sees the pool.
+UMA_SMI_TOTAL = 16320 << 20
+UMA_POOL = 46464 << 20
+UMA_RAM = 48 * GIB
+
+
+def _no_cache(monkeypatch):
+ monkeypatch.setattr(hw, "_pool_probe_cache", None)
+
+
+# ── _unified_pool_bytes: the classification gate ─────────────
+
+
+def test_integrated_attribute_wins_positive(monkeypatch):
+ """Driver says integrated=True -> unified, no numeric gates needed."""
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: (UMA_POOL, True))
+ assert hw._unified_pool_bytes(UMA_SMI_TOTAL, UMA_RAM) == UMA_POOL
+
+
+def test_integrated_attribute_wins_negative(monkeypatch):
+ """Driver says integrated=False -> discrete, even when the numbers
+ would pass both fallback gates (attribute outranks arithmetic)."""
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: (UMA_POOL, False))
+ assert hw._unified_pool_bytes(UMA_SMI_TOTAL, UMA_RAM) is None
+
+
+def test_engine_fallback_uma_shape_passes(monkeypatch):
+ """Attribute unreadable (engine fallback): the carve-out shape passes
+ both gates — pool several times the smi report, pool ~= RAM."""
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: (UMA_POOL, None))
+ assert hw._unified_pool_bytes(UMA_SMI_TOTAL, UMA_RAM) == UMA_POOL
+
+
+def test_discrete_card_agreeing_within_rounding_stays_discrete(monkeypatch):
+ """A healthy discrete card: allocator and smi agree within rounding.
+ Fails the disagreement gate regardless of box RAM."""
+ smi = 24 * GIB
+ monkeypatch.setattr(hw, "_device_pool_view",
+ lambda: (smi + (200 << 20), None))
+ assert hw._unified_pool_bytes(smi, 24 * GIB) is None
+ assert hw._unified_pool_bytes(smi, 256 * GIB) is None
+
+
+def test_ram_matched_workstation_card_stays_discrete(monkeypatch):
+ """The nastiest lookalike: a 48 GB card in a 48 GB box. smi and the
+ allocator AGREE (both say 48), so the disagreement gate fails even
+ though pool == RAM would pass the size gate."""
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: (48 * GIB, None))
+ assert hw._unified_pool_bytes(48 * GIB, 48 * GIB) is None
+
+
+def test_pool_smaller_than_ram_fraction_stays_discrete(monkeypatch):
+ """Disagreement without the RAM-sized-pool signature stays discrete:
+ a hypothetical card whose allocator over-reports 2x in a huge-RAM box
+ is a driver bug to distrust, not a unified pool."""
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: (32 * GIB, None))
+ assert hw._unified_pool_bytes(16 * GIB, 128 * GIB) is None
+
+
+def test_no_probe_available_stays_discrete(monkeypatch):
+ """No driver API, no engine binary -> exactly today's behavior."""
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: None)
+ assert hw._unified_pool_bytes(UMA_SMI_TOTAL, UMA_RAM) is None
+
+
+# ── probe_budget wiring ──────────────────────────────────────
+
+
+def _uma_machine(monkeypatch, *, view):
+ _no_cache(monkeypatch)
+ monkeypatch.setattr(hw, "_nvidia_vram",
+ lambda: (UMA_SMI_TOTAL, 14848 << 20))
+ monkeypatch.setattr(hw, "_ram_bytes",
+ lambda: (UMA_RAM, 32 * GIB))
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: view)
+
+
+def test_budget_unified_pool_planning(monkeypatch):
+ """Planning budget on unified memory: the allocator pool itself minus
+ headroom, uma=True, ram_available=0 — host memory must not
+ double-count as spill room. NO OS-RAM clamp: carved-out memory is
+ invisible to GlobalMemoryStatusEx (a larger carve shrinks reported
+ OS RAM while the pool stays constant), so clamping to OS RAM would
+ throw away exactly the carved capacity."""
+ _uma_machine(monkeypatch, view=(UMA_POOL, True))
+ b = hw.probe_budget(planning=True)
+ assert b.uma is True
+ assert b.ram_available_bytes == 0
+ assert b.total_device_bytes == UMA_POOL
+ assert b.usable_vram_bytes == int(UMA_POOL * (1 - hw._UMA_HEADROOM_FRACTION))
+ # The whole point: the budget must dwarf the carve-out.
+ assert b.usable_vram_bytes > 2 * UMA_SMI_TOTAL
+
+
+def test_budget_unified_pool_live_counts_dedicated_free_plus_os_available(monkeypatch):
+ """Live budget = smi-free + OS-available (each side alone under-counts:
+ smi free saturates at the carve-out, OS-available can't see it)."""
+ _uma_machine(monkeypatch, view=(UMA_POOL, True))
+ b = hw.probe_budget(planning=False)
+ assert b.uma is True
+ live = (14848 << 20) + 32 * GIB
+ assert b.usable_vram_bytes == int(min(UMA_POOL, live)
+ * (1 - hw._UMA_HEADROOM_FRACTION))
+
+
+def test_budget_unified_no_smi_still_classifies(monkeypatch):
+ """nvidia-smi off PATH must not change the verdict: the driver API
+ (system loader, PATH-independent) still classifies unified, planning
+ still budgets the full pool, and live falls back to OS-available."""
+ _no_cache(monkeypatch)
+ monkeypatch.setattr(hw, "_nvidia_vram", lambda: None)
+ monkeypatch.setattr(hw, "_ram_bytes", lambda: (UMA_RAM, 32 * GIB))
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: (UMA_POOL, True))
+ planning = hw.probe_budget(planning=True)
+ assert planning.uma is True
+ assert planning.total_device_bytes == UMA_POOL
+ assert planning.usable_vram_bytes == int(
+ UMA_POOL * (1 - hw._UMA_HEADROOM_FRACTION))
+ live = hw.probe_budget(planning=False)
+ assert live.usable_vram_bytes == int(32 * GIB * (1 - hw._UMA_HEADROOM_FRACTION))
+
+
+def test_budget_discrete_unchanged_when_probe_says_discrete(monkeypatch):
+ """integrated=False keeps the existing discrete path bit-for-bit."""
+ _uma_machine(monkeypatch, view=(UMA_POOL, False))
+ b = hw.probe_budget(planning=True)
+ assert b.uma is False
+ assert b.total_device_bytes == UMA_SMI_TOTAL
+ margin = max(hw._MARGIN_FLOOR, int(UMA_SMI_TOTAL * hw._MARGIN_FRACTION))
+ assert b.usable_vram_bytes == UMA_SMI_TOTAL - margin
+ assert b.ram_available_bytes == UMA_RAM
+
+
+def test_budget_discrete_unchanged_when_probe_unavailable(monkeypatch):
+ _uma_machine(monkeypatch, view=None)
+ b = hw.probe_budget(planning=True)
+ assert b.uma is False
+ assert b.ram_available_bytes == UMA_RAM
+
+
+def test_engine_fallback_without_smi_stays_conservative(monkeypatch):
+ """Engine-fallback view (no INTEGRATED verdict) + no smi numbers: the
+ disagreement gate has nothing to compare against, so the quirk stays
+ off and budgeting falls to the conservative RAM-as-UMA path — an
+ attribute-less pool claim alone must never flip the verdict."""
+ _no_cache(monkeypatch)
+ monkeypatch.setattr(hw, "_nvidia_vram", lambda: None)
+ monkeypatch.setattr(hw, "_ram_bytes", lambda: (UMA_RAM, 32 * GIB))
+ monkeypatch.setattr(hw, "_device_pool_view", lambda: (UMA_POOL, None))
+ b = hw.probe_budget(planning=True)
+ assert b.uma is True
+ assert b.total_device_bytes == UMA_RAM # RAM path, not the pool
+
+
+def test_smi_resolver_caches_and_survives_empty_path(monkeypatch):
+ """The resolver consults PATH first, and a resolution (hit or miss) is
+ cached for the process."""
+ calls = []
+ monkeypatch.setattr(hw, "_smi_path_cache", None)
+ monkeypatch.setattr(hw.shutil, "which",
+ lambda name: calls.append(name) or "/usr/bin/nvidia-smi")
+ assert hw._nvidia_smi_path() == "/usr/bin/nvidia-smi"
+ assert hw._nvidia_smi_path() == "/usr/bin/nvidia-smi"
+ assert len(calls) == 1
+
+
+# ── probe cache ──────────────────────────────────────────────
+
+
+def test_pool_probe_hit_is_cached_for_process(monkeypatch):
+ _no_cache(monkeypatch)
+ calls = []
+
+ def probe():
+ calls.append(1)
+ return (UMA_POOL, True)
+
+ monkeypatch.setattr(hw, "_cuda_driver_pool", probe)
+ monkeypatch.setattr(hw, "_engine_device_pool", lambda: None)
+ assert hw._device_pool_view() == (UMA_POOL, True)
+ assert hw._device_pool_view() == (UMA_POOL, True)
+ assert len(calls) == 1
+
+
+def test_pool_probe_miss_retries_after_ttl(monkeypatch):
+ """A miss must not be permanent: the engine binary can appear
+ mid-session via the pane's runtime install."""
+ _no_cache(monkeypatch)
+ monkeypatch.setattr(hw, "_cuda_driver_pool", lambda: None)
+ answers = [None, (UMA_POOL, None)]
+ monkeypatch.setattr(hw, "_engine_device_pool", lambda: answers.pop(0))
+
+ now = [1000.0]
+ monkeypatch.setattr(hw.time, "monotonic", lambda: now[0])
+ assert hw._device_pool_view() is None
+ now[0] += 1.0 # inside TTL: cached miss, no re-probe
+ assert hw._device_pool_view() is None
+ now[0] += hw._POOL_NEGATIVE_TTL_S + 1.0
+ assert hw._device_pool_view() == (UMA_POOL, None)
+ assert not answers
+
+
+# ── device-line parsing (engine fallback) ────────────────────
+
+
+def test_device_line_regex_handles_parenthesized_names():
+ """Device names may contain their own parentheses — the LAST
+ parenthesized group must win."""
+ line = (" CUDA0: NVIDIA Example Device (1234-core Example GPU) "
+ "(46464 MiB, 46284 MiB free)")
+ m = hw._DEVICE_LINE_RE.search(line)
+ assert m and int(m.group(1)) == 46464
diff --git a/tests/hermes_cli/test_update_autostash_orphan_warning.py b/tests/hermes_cli/test_update_autostash_orphan_warning.py
new file mode 100644
index 0000000000..99b932560b
--- /dev/null
+++ b/tests/hermes_cli/test_update_autostash_orphan_warning.py
@@ -0,0 +1,104 @@
+"""Orphaned update-autostash surfacing (#63717 problem 6).
+
+``hermes update`` can legitimately leave an autostash behind (--keep-stash
+parks it; a conflicted restore preserves it), but nothing ever mentioned those
+entries again — they persisted invisibly for weeks. ``hermes update`` now
+warns about ``hermes-update-autostash-*`` entries older than the threshold.
+Behavioral tests use real git repos; no production mocking of the code under
+test.
+"""
+
+import subprocess
+from datetime import datetime, timedelta, timezone
+
+import pytest
+
+from hermes_cli import update_cmd
+
+
+def _git(cwd, *args, check=True):
+ return subprocess.run(
+ ["git", *args], cwd=cwd, capture_output=True, text=True, check=check
+ )
+
+
+def _make_repo_with_autostash(tmp_path, age_days: float):
+ """Real repo with one hermes-update-autostash entry aged ``age_days``."""
+ import shutil
+
+ if shutil.which("git") is None:
+ pytest.skip("git not available")
+ _git(tmp_path, "init", "-q", "-b", "main")
+ _git(tmp_path, "config", "user.email", "t@example.com")
+ _git(tmp_path, "config", "user.name", "t")
+ (tmp_path / "tracked.txt").write_text("v1\n")
+ _git(tmp_path, "add", "-A")
+ _git(tmp_path, "commit", "-qm", "init")
+
+ (tmp_path / "tracked.txt").write_text("local change\n")
+ stamp = (
+ datetime.now(timezone.utc) - timedelta(days=age_days)
+ ).strftime("%Y%m%d-%H%M%S")
+ name = f"hermes-update-autostash-{stamp}"
+ _git(tmp_path, "stash", "push", "--include-untracked", "-m", name)
+ return name
+
+
+def test_old_autostash_is_surfaced(tmp_path, capsys):
+ name = _make_repo_with_autostash(tmp_path, age_days=9)
+ count = update_cmd._warn_orphaned_update_autostashes(["git"], tmp_path)
+ out = capsys.readouterr().out
+ assert count == 1
+ assert "leftover update autostash" in out
+ assert name in out
+ assert "git stash apply" in out
+ # Never a GC: the entry must still exist.
+ listed = _git(tmp_path, "stash", "list").stdout
+ assert name in listed
+
+
+def test_fresh_autostash_is_not_flagged(tmp_path, capsys):
+ _make_repo_with_autostash(tmp_path, age_days=1)
+ count = update_cmd._warn_orphaned_update_autostashes(["git"], tmp_path)
+ assert count == 0
+ assert "leftover update autostash" not in capsys.readouterr().out
+
+
+def test_non_hermes_stash_is_ignored(tmp_path, capsys):
+ import shutil
+
+ if shutil.which("git") is None:
+ pytest.skip("git not available")
+ _git(tmp_path, "init", "-q", "-b", "main")
+ _git(tmp_path, "config", "user.email", "t@example.com")
+ _git(tmp_path, "config", "user.name", "t")
+ (tmp_path / "tracked.txt").write_text("v1\n")
+ _git(tmp_path, "add", "-A")
+ _git(tmp_path, "commit", "-qm", "init")
+ (tmp_path / "tracked.txt").write_text("user's own WIP\n")
+ _git(tmp_path, "stash", "push", "-m", "my own stash from 20200101-000000")
+ count = update_cmd._warn_orphaned_update_autostashes(["git"], tmp_path)
+ assert count == 0
+ assert "leftover update autostash" not in capsys.readouterr().out
+
+
+def test_unparseable_autostash_timestamp_left_alone(tmp_path, capsys):
+ import shutil
+
+ if shutil.which("git") is None:
+ pytest.skip("git not available")
+ _git(tmp_path, "init", "-q", "-b", "main")
+ _git(tmp_path, "config", "user.email", "t@example.com")
+ _git(tmp_path, "config", "user.name", "t")
+ (tmp_path / "tracked.txt").write_text("v1\n")
+ _git(tmp_path, "add", "-A")
+ _git(tmp_path, "commit", "-qm", "init")
+ (tmp_path / "tracked.txt").write_text("change\n")
+ _git(tmp_path, "stash", "push", "-m", "hermes-update-autostash-notadate")
+ count = update_cmd._warn_orphaned_update_autostashes(["git"], tmp_path)
+ assert count == 0
+
+
+def test_git_failure_is_nonfatal(tmp_path):
+ # Not a git repo at all — must return 0, not raise.
+ assert update_cmd._warn_orphaned_update_autostashes(["git"], tmp_path) == 0
diff --git a/tests/hermes_cli/test_update_fleet_restart_pending.py b/tests/hermes_cli/test_update_fleet_restart_pending.py
index f8390047a2..d5da6456e9 100644
--- a/tests/hermes_cli/test_update_fleet_restart_pending.py
+++ b/tests/hermes_cli/test_update_fleet_restart_pending.py
@@ -102,16 +102,21 @@ def _patch_update_deps(monkeypatch, tmp_path, run_side_effect):
monkeypatch.setattr(
hermes_main, "_finish_dashboard_update_cleanup", lambda *a, **k: None
)
+ monkeypatch.setattr(
+ update_cmd, "_finish_dashboard_update_cleanup", lambda *a, **k: None
+ )
monkeypatch.setattr(hermes_main, "_build_web_ui", lambda *a, **k: None)
monkeypatch.setattr(
update_cmd, "_venv_core_imports_healthy", lambda: (True, "")
)
monkeypatch.setattr(update_cmd, "_update_node_dependencies", lambda: [])
+ monkeypatch.setattr(update_cmd, "_purge_stale_hermes_modules", lambda: None)
+ monkeypatch.setattr(hermes_main, "_purge_stale_hermes_modules", lambda: None)
import hermes_cli.gateway as hermes_gateway
monkeypatch.setattr(
- hermes_gateway, "find_gateway_pids", lambda all_profiles=False: []
+ hermes_gateway, "find_gateway_pids", lambda **_kwargs: []
)
monkeypatch.setattr(hermes_gateway, "supports_systemd_services", lambda: False)
monkeypatch.setattr(
@@ -302,6 +307,97 @@ def test_marker_written_after_pull_cleared_after_successful_restart(
assert "✓ Code updated!" in out
+def test_clean_update_warns_about_surviving_pre_update_serve_runtime(
+ monkeypatch, tmp_path, capsys
+):
+ """The successful update path must surface an inventoried stale serve."""
+ args = _update_args()
+ _patch_update_deps(monkeypatch, tmp_path, _make_head_moved_side_effect())
+ monkeypatch.setattr(
+ update_cmd,
+ "_surviving_pre_update_serve_runtimes",
+ lambda _plan: [
+ {
+ "pid": 5555,
+ "kind": "serve",
+ "profile": "default",
+ "supervisor": "manual-serve",
+ }
+ ],
+ )
+
+ hermes_main.cmd_update(args)
+
+ out = capsys.readouterr().out
+ assert "pid 5555" in out
+ assert "serve" in out
+ assert "pre-update code" in out
+
+
+def test_clean_update_escalates_surviving_serve_as_unaccounted(
+ monkeypatch, tmp_path, capsys
+):
+ """#100479 end to end: the plan inventoried a gateway (restarted through
+ ``hermes-gateway.service``) and an unmanaged ``serve`` on the same
+ default profile. The serve survives the update as the SAME process, so
+ the update must (1) warn, (2) reconcile it as ``unaccounted`` instead of
+ borrowing the gateway's restart, and (3) exit 1 with a ``partial``
+ receipt — not print a clean success."""
+ from hermes_cli.update_inventory import (
+ RuntimeRecord, UpdatePlan, _restart_mechanism,
+ )
+ import hermes_cli.update_inventory as ui
+
+ args = _update_args()
+ _patch_update_deps(monkeypatch, tmp_path, _make_head_moved_side_effect())
+
+ plan = UpdatePlan()
+ plan.runtimes = [
+ RuntimeRecord(kind="gateway", profile="default", pid=4444,
+ supervisor="systemd",
+ restart_via=_restart_mechanism("systemd", "default")),
+ RuntimeRecord(kind="serve", profile="default", pid=5555,
+ supervisor="manual-serve",
+ restart_via=_restart_mechanism("manual-serve", "default"),
+ detail={"create_time": 1000.0}),
+ ]
+ monkeypatch.setattr(ui, "collect_runtime_inventory", lambda: plan)
+ # The restart phase's own bookkeeping says the gateway unit restarted
+ # (systemd branch is stubbed off in _patch_update_deps, so feed it here).
+ real_match = ui.match_runtime_outcomes
+
+ def _match(p, **kw):
+ kw["restarted_services"] = list(kw.get("restarted_services") or []) + [
+ "hermes-gateway.service"
+ ]
+ return real_match(p, **kw)
+
+ monkeypatch.setattr(ui, "match_runtime_outcomes", _match)
+ # Real survivor probe semantics against a fake ledger: pid 5555 is still
+ # the same incarnation the plan recorded.
+ import hermes_cli.process_identity as pi
+
+ monkeypatch.setattr(
+ pi, "ledger_entries",
+ lambda **_k: [{"pid": 5555, "purpose": "serve", "create_time": 1000.0}],
+ )
+
+ with pytest.raises(SystemExit) as excinfo:
+ hermes_main.cmd_update(args)
+ assert excinfo.value.code == 1
+
+ out = capsys.readouterr().out
+ assert "pid 5555" in out and "pre-update code" in out
+ assert "Planned runtimes the restart phase never touched" in out
+ assert "serve [default] pid 5555" in out
+
+ latest = get_hermes_home() / "logs" / "update_receipts" / "latest.json"
+ receipt = json.loads(latest.read_text(encoding="utf-8"))
+ assert receipt["outcome"] == "partial"
+ by_pid = {o["pid"]: o["outcome"] for o in receipt["runtime_outcomes"]}
+ assert by_pid == {4444: "restarted", 5555: "unaccounted"}
+
+
def test_interrupt_between_pull_and_restart_leaves_marker(
monkeypatch, tmp_path
):
diff --git a/tests/hermes_cli/test_update_state_autorestore.py b/tests/hermes_cli/test_update_state_autorestore.py
index 75ab2c3552..7b54340835 100644
--- a/tests/hermes_cli/test_update_state_autorestore.py
+++ b/tests/hermes_cli/test_update_state_autorestore.py
@@ -211,3 +211,116 @@ def test_restore_helper_propagates_copy_errors(tmp_path):
with pytest.raises(OSError):
_restore_state_db_from_snapshot(state_path, tmp_path / "does-not-exist.db")
+
+
+# ── Multi-profile coverage (#97994) ─────────────────────────────────────
+
+
+def _make_valid_db(path: Path, rows: int) -> None:
+ conn = sqlite3.connect(path)
+ conn.execute("CREATE TABLE sessions (id INTEGER PRIMARY KEY, name TEXT)")
+ conn.executemany(
+ "INSERT INTO sessions (name) VALUES (?)",
+ [(str(i),) for i in range(rows)],
+ )
+ conn.commit()
+ conn.close()
+
+
+def _make_valid_snapshot(home: Path, snap_id: str, rows: int) -> None:
+ snap_dir = home / "state-snapshots" / snap_id
+ snap_dir.mkdir(parents=True)
+ _make_valid_db(snap_dir / "state.db", rows)
+
+
+def test_post_update_guard_covers_sibling_profiles(tmp_path, monkeypatch, capsys):
+ """#97994: the guard must verify + auto-restore EVERY profile's state.db,
+ not just the root home's. Pre-update snapshots already cover siblings
+ (#66140); the guard was the missing half."""
+ from hermes_cli import update_cmd
+ from hermes_cli.backup import _sibling_profile_homes
+
+ root_home = tmp_path / "default-home"
+ root_home.mkdir()
+ sibling_home = tmp_path / "profiles" / "work"
+ sibling_home.mkdir(parents=True)
+
+ # Root DB: valid, with its own snapshot — must be left untouched.
+ _make_valid_db(root_home / "state.db", 10)
+ root_before = (root_home / "state.db").read_bytes()
+
+ # Sibling: live DB corrupted post-update (the #68474 zeroed signature),
+ # with its own VALID pre-update snapshot under its own snapshots dir.
+ _make_valid_snapshot(sibling_home, "20260901-pre-update", 25)
+ (sibling_home / "state.db").write_bytes(b"\x00" * 4096)
+
+ monkeypatch.setattr(update_cmd, "get_hermes_home", lambda: root_home)
+ monkeypatch.setattr(
+ "hermes_cli.backup._sibling_profile_homes",
+ lambda invoking_home: [("work", sibling_home)],
+ )
+
+ update_cmd._verify_and_restore_state_dbs_post_update()
+
+ # Sibling restored from ITS snapshot (snap rows, not zeroed bytes).
+ assert _row_count(sibling_home / "state.db") == 25
+ # Root DB byte-identical — untouched.
+ assert (root_home / "state.db").read_bytes() == root_before
+ # Operator-visible restore message mentions the profile.
+ out = capsys.readouterr().out
+ assert "profile work" in out
+
+
+def test_post_update_guard_leaves_valid_sibling_dbs_alone(tmp_path, monkeypatch, capsys):
+ """A healthy sibling profile must not be touched — the guard only acts
+ on corruption."""
+ from hermes_cli import update_cmd
+
+ root_home = tmp_path / "default-home"
+ root_home.mkdir()
+ sibling_home = tmp_path / "profiles" / "work"
+ sibling_home.mkdir(parents=True)
+
+ _make_valid_db(root_home / "state.db", 10)
+ _make_valid_db(sibling_home / "state.db", 7)
+ sibling_before = (sibling_home / "state.db").read_bytes()
+
+ monkeypatch.setattr(update_cmd, "get_hermes_home", lambda: root_home)
+ monkeypatch.setattr(
+ "hermes_cli.backup._sibling_profile_homes",
+ lambda invoking_home: [("work", sibling_home)],
+ )
+
+ update_cmd._verify_and_restore_state_dbs_post_update()
+
+ assert (sibling_home / "state.db").read_bytes() == sibling_before
+ out = capsys.readouterr().out
+ assert "corrupted" not in out
+
+
+def test_post_update_guard_survives_missing_sibling_snapshot(tmp_path, monkeypatch, capsys):
+ """Corrupt sibling with NO snapshot: guard must report and continue,
+ never raise into the update tail."""
+ from hermes_cli import update_cmd
+
+ root_home = tmp_path / "default-home"
+ root_home.mkdir()
+ sibling_home = tmp_path / "profiles" / "work"
+ sibling_home.mkdir(parents=True)
+
+ _make_valid_db(root_home / "state.db", 10)
+ (sibling_home / "state.db").write_bytes(b"\x00" * 4096)
+
+ monkeypatch.setattr(update_cmd, "get_hermes_home", lambda: root_home)
+ monkeypatch.setattr(
+ "hermes_cli.backup._sibling_profile_homes",
+ lambda invoking_home: [("work", sibling_home)],
+ )
+
+ # Must not raise even though no snapshot exists to restore from.
+ update_cmd._verify_and_restore_state_dbs_post_update()
+
+ out = capsys.readouterr().out
+ assert "corrupted" in out
+ # Still corrupt (no snapshot) — but the guard completed cleanly.
+ assert (sibling_home / "state.db").read_bytes() == b"\x00" * 4096
diff --git a/tests/hermes_cli/test_update_wedged_gateway.py b/tests/hermes_cli/test_update_wedged_gateway.py
index 71349e3e7a..ece8ec3a66 100644
--- a/tests/hermes_cli/test_update_wedged_gateway.py
+++ b/tests/hermes_cli/test_update_wedged_gateway.py
@@ -14,6 +14,7 @@ import json
import os
import shutil
import socket
+import sys
import tempfile
import threading
import time
@@ -30,6 +31,19 @@ from gateway.shutdown_watchdog import (
write_loop_heartbeat,
)
+# Native Windows exposes neither ``socket.AF_UNIX`` nor an asyncio UNIX
+# server, so the witness cases that create real socket nodes
+# (``_silent_socket_node``) or run the real producer
+# (``loop_heartbeat_forever``) cannot execute there. Only those cases are
+# skipped: the witness-absent contracts (mocked probes, file-only
+# heartbeats) are platform-independent and keep running on Windows, per
+# the Windows behavior pinned alongside the product-side guarantee.
+_NEEDS_UNIX_SOCKETS = pytest.mark.skipif(
+ sys.platform == "win32",
+ reason="requires real UNIX-domain sockets "
+ "(socket.AF_UNIX / asyncio.start_unix_server), unavailable on native Windows",
+)
+
@pytest.fixture()
def tmp_path():
@@ -482,6 +496,7 @@ class TestLoopTickWitness:
witnesses agree the loop stopped scheduling.
"""
+ @_NEEDS_UNIX_SOCKETS
def test_stalled_heartbeat_write_never_escalates_a_running_loop(
self, tmp_path, monkeypatch
):
@@ -631,6 +646,7 @@ class TestLoopTickWitness:
thread.join(timeout=5.0)
assert not errors, errors
+ @_NEEDS_UNIX_SOCKETS
def test_off_loop_completion_cannot_manufacture_fresh_liveness(self, tmp_path):
"""A write landing after the loop froze must not look alive.
@@ -650,6 +666,7 @@ class TestLoopTickWitness:
== gateway_cli.GATEWAY_LOOP_UNKNOWN
)
+ @_NEEDS_UNIX_SOCKETS
def test_true_wedge_requires_sustained_witness_silence(self, tmp_path):
"""Stale file + armed socket silent across the whole window: WEDGED.
@@ -808,10 +825,16 @@ class TestLoopTickWitness:
gateway_cli.probe_gateway_loop_liveness(pid, home=tmp_path)
== gateway_cli.GATEWAY_LOOP_WEDGED
)
- # And a fresh legacy file stays safe even if a dead-listener node
- # exists for the PID (leftover from a newer process): the silent
- # socket denies ALIVE, and UNKNOWN never escalates — the drain path
- # keeps the full budget either way.
+
+ @_NEEDS_UNIX_SOCKETS
+ def test_legacy_fresh_file_with_dead_node_is_unknown(self, tmp_path):
+ """A fresh legacy file stays safe under a dead-listener node.
+
+ A dead-listener node for the PID (leftover from a newer process):
+ the silent socket denies ALIVE, and UNKNOWN never escalates — the
+ drain path keeps the full budget either way.
+ """
+ pid = 4242
_write_heartbeat(tmp_path, pid, age_s=5.0)
_silent_socket_node(get_loop_tick_socket_path(tmp_path, pid))
assert (
@@ -821,6 +844,7 @@ class TestLoopTickWitness:
== gateway_cli.GATEWAY_LOOP_UNKNOWN
)
+ @_NEEDS_UNIX_SOCKETS
@pytest.mark.asyncio
async def test_producer_rebinds_over_stale_socket_node(self, tmp_path):
"""A leftover node from a dead process must not disarm the witness.
@@ -862,6 +886,7 @@ class TestLoopTickWitness:
except asyncio.CancelledError:
pass
+ @_NEEDS_UNIX_SOCKETS
def test_transient_stall_below_wedge_budget_never_escalates(
self, tmp_path, monkeypatch
):
@@ -914,6 +939,7 @@ class TestLoopTickWitness:
state["thread"].join(timeout=5.0)
assert not errors, errors
+ @_NEEDS_UNIX_SOCKETS
def test_sustained_stop_above_wedge_budget_still_escalates(
self, tmp_path
):
@@ -960,6 +986,113 @@ class TestLoopTickWitness:
assert not errors, errors
+class TestLoopTickTcpWitness:
+ """Non-POSIX arm: the producer publishes ``loop_tick_tcp_port`` and the
+ consumer probes 127.0.0.1: instead of the AF_UNIX node. The
+ two-witness contract must hold identically over TCP."""
+
+ @staticmethod
+ def _tcp_answerer():
+ """A loopback listener that answers b"1" — the armed, dispatching loop."""
+ srv = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
+ srv.bind(("127.0.0.1", 0))
+ srv.listen(8)
+ stop = threading.Event()
+
+ def serve():
+ srv.settimeout(0.1)
+ while not stop.is_set():
+ try:
+ conn, _ = srv.accept()
+ except socket.timeout:
+ continue
+ try:
+ conn.sendall(b"1")
+ finally:
+ conn.close()
+
+ thread = threading.Thread(target=serve, daemon=True)
+ thread.start()
+ return srv.getsockname()[1], stop, srv
+
+ @staticmethod
+ def _write_tcp_heartbeat(home, pid, port, age_s=0.0):
+ write_loop_heartbeat(
+ pid=pid,
+ home=home,
+ extra={"loop_tick_socket": True, "loop_tick_tcp_port": port},
+ )
+ if age_s:
+ path = get_loop_heartbeat_path(home)
+ stamp = time.time() - age_s
+ os.utime(path, (stamp, stamp))
+
+ def test_stale_file_with_answering_tcp_witness_is_alive(self, tmp_path):
+ """#90502 shape over TCP: a stalled write must not kill a live loop."""
+ port, stop, srv = self._tcp_answerer()
+ try:
+ self._write_tcp_heartbeat(tmp_path, 4343, port, age_s=600.0)
+ assert (
+ gateway_cli.probe_gateway_loop_liveness(4343, home=tmp_path)
+ == gateway_cli.GATEWAY_LOOP_ALIVE
+ )
+ finally:
+ stop.set()
+ srv.close()
+
+ def test_stale_file_with_silent_tcp_witness_is_wedged(self, tmp_path):
+ """Armed TCP witness that never answers across the window: WEDGED."""
+ silent = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
+ silent.bind(("127.0.0.1", 0))
+ silent.listen(1) # accepts but never sends
+ try:
+ port = silent.getsockname()[1]
+ self._write_tcp_heartbeat(tmp_path, 4344, port, age_s=600.0)
+ assert (
+ gateway_cli.probe_gateway_loop_liveness(
+ 4344, home=tmp_path, tick_timeout=0.2, tick_gap_s=0.05
+ )
+ == gateway_cli.GATEWAY_LOOP_WEDGED
+ )
+ finally:
+ silent.close()
+
+ def test_fresh_file_with_silent_tcp_witness_is_unknown(self, tmp_path):
+ """Fresh file + silent TCP witness: an off-loop write landed after a
+ freeze — not proof of liveness, never destructive authority."""
+ silent = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
+ silent.bind(("127.0.0.1", 0))
+ silent.listen(1)
+ try:
+ port = silent.getsockname()[1]
+ self._write_tcp_heartbeat(tmp_path, 4345, port)
+ assert (
+ gateway_cli.probe_gateway_loop_liveness(
+ 4345, home=tmp_path, tick_timeout=0.2
+ )
+ == gateway_cli.GATEWAY_LOOP_UNKNOWN
+ )
+ finally:
+ silent.close()
+
+ def test_garbage_tcp_port_falls_back_to_socket_contract(self, tmp_path):
+ """A non-numeric port must not be treated as an armed witness."""
+ write_loop_heartbeat(
+ pid=4346,
+ home=tmp_path,
+ extra={"loop_tick_socket": False, "loop_tick_tcp_port": "nope"},
+ )
+ path = get_loop_heartbeat_path(tmp_path)
+ stamp = time.time() - 600.0
+ os.utime(path, (stamp, stamp))
+ # loop_tick_socket=False + no usable TCP port: witness could not be
+ # armed, staleness is not proof -> UNKNOWN, never WEDGED.
+ assert (
+ gateway_cli.probe_gateway_loop_liveness(4346, home=tmp_path)
+ == gateway_cli.GATEWAY_LOOP_UNKNOWN
+ )
+
+
def test_default_probe_budget_stays_inside_query_tier():
"""The module doc pins the worst-case wedge-suspected probe at ~3.4s,
'far inside the 10s query tier'. Assert the strike-count math so
diff --git a/tests/hermes_cli/test_web_server.py b/tests/hermes_cli/test_web_server.py
index e00fbd4031..9f7a6d810a 100644
--- a/tests/hermes_cli/test_web_server.py
+++ b/tests/hermes_cli/test_web_server.py
@@ -319,6 +319,29 @@ class TestWebServerEndpoints:
monitor.close()
writer.close()
+ def test_get_sessions_transient_ioerr_is_503(self, monkeypatch):
+ """Busy store, not a gone store: the desktop keeps the list it has."""
+ import sqlite3
+
+ from hermes_cli import web_server
+
+ def boom(*_args, **_kwargs):
+ raise sqlite3.OperationalError("disk I/O error")
+
+ monkeypatch.setattr(web_server, "_open_session_db_for_profile", boom)
+ assert self.client.get("/api/sessions?limit=1&offset=0").status_code == 503
+
+ def test_get_sessions_non_transient_operational_error_is_500(self, monkeypatch):
+ import sqlite3
+
+ from hermes_cli import web_server
+
+ def boom(*_args, **_kwargs):
+ raise sqlite3.OperationalError("no such table: sessions")
+
+ monkeypatch.setattr(web_server, "_open_session_db_for_profile", boom)
+ assert self.client.get("/api/sessions?limit=1&offset=0").status_code == 500
+
def test_get_status_loads_gateway_config_off_event_loop(self, monkeypatch):
"""Cold gateway config loading must not block the WebSocket loop.
diff --git a/tests/hermes_cli/test_web_server_profile_unification.py b/tests/hermes_cli/test_web_server_profile_unification.py
index f715af1f69..75a3f368f2 100644
--- a/tests/hermes_cli/test_web_server_profile_unification.py
+++ b/tests/hermes_cli/test_web_server_profile_unification.py
@@ -7,6 +7,7 @@ reads/writes land in the REQUESTED profile, the dashboard's own profile
stays untouched, and the chat PTY env is scoped via HERMES_HOME.
"""
import json
+from contextlib import contextmanager
import pytest
import yaml
@@ -353,6 +354,47 @@ class TestProfileScopedModel:
+ def test_model_options_uses_config_only_scope_for_selected_profile(
+ self, client, monkeypatch
+ ):
+ """Regression (#58576): _profile_scope holds _SKILLS_PROFILE_LOCK
+ across its body, and the payload build can block up to 15s on a
+ models.dev cache miss — a cold request would starve concurrent
+ /api/config on the same lock. The handler must scope the worker
+ through _config_profile_scope (contextvar only, no lock) for the
+ selected profile."""
+ import hermes_cli.web_server as web_server
+
+ scopes = []
+
+ @contextmanager
+ def _recording_config_scope(profile):
+ scopes.append(("config", profile))
+ yield object()
+
+ @contextmanager
+ def _recording_profile_scope(profile):
+ scopes.append(("full", profile))
+ yield object()
+
+ monkeypatch.setattr(
+ web_server, "_config_profile_scope", _recording_config_scope
+ )
+ monkeypatch.setattr(web_server, "_profile_scope", _recording_profile_scope)
+ monkeypatch.setattr(
+ "hermes_cli.inventory.load_picker_context", lambda: object()
+ )
+ monkeypatch.setattr(
+ "hermes_cli.inventory.build_model_options_payload",
+ lambda _ctx, **kwargs: {"providers": [], "model": "", "provider": ""},
+ )
+
+ resp = client.get("/api/model/options", params={"profile": "worker_beta"})
+ assert resp.status_code == 200
+ # Only the config-only scope may wrap the payload build; entering
+ # _profile_scope would hold _SKILLS_PROFILE_LOCK across it (#58576).
+ assert scopes == [("config", "worker_beta")]
+
def test_model_info_unknown_profile_404(self, client, isolated_profiles):
"""Regression: the broad except used to convert the 404 into a 200
with empty model info ("no model set" — silently wrong)."""
diff --git a/tests/hermes_cli/test_web_server_tts_lease.py b/tests/hermes_cli/test_web_server_tts_lease.py
new file mode 100644
index 0000000000..be32621997
--- /dev/null
+++ b/tests/hermes_cli/test_web_server_tts_lease.py
@@ -0,0 +1,157 @@
+"""``POST /api/audio/tts-lease`` — desktop speech toggles as TTS warm-up/release.
+
+The desktop's "Read replies aloud" and voice-conversation toggles call this so
+the backend can pre-load the configured TTS engine when speech is about to be
+needed and unload resident local models once no surface holds a lease.
+"""
+
+from __future__ import annotations
+
+import pytest
+
+
+@pytest.fixture
+def isolated_profiles(tmp_path, monkeypatch, _isolate_hermes_home):
+ from hermes_constants import get_hermes_home
+ from hermes_cli import profiles
+
+ default_home = get_hermes_home()
+ profiles_root = default_home / "profiles"
+ worker_home = profiles_root / "worker_beta"
+ for home in (default_home, worker_home):
+ home.mkdir(parents=True, exist_ok=True)
+ (home / "config.yaml").write_text("{}\n", encoding="utf-8")
+ (worker_home / ".env").write_text("", encoding="utf-8")
+
+ monkeypatch.setattr(profiles, "_get_default_hermes_home", lambda: default_home)
+ monkeypatch.setattr(profiles, "_get_profiles_root", lambda: profiles_root)
+ return {"default": default_home, "worker_beta": worker_home}
+
+
+@pytest.fixture
+def client(monkeypatch, isolated_profiles):
+ try:
+ from starlette.testclient import TestClient
+ except ImportError:
+ pytest.skip("fastapi/starlette not installed")
+
+ import hermes_state
+ from hermes_constants import get_hermes_home
+ from hermes_cli.web_server import app, _SESSION_HEADER_NAME, _SESSION_TOKEN
+
+ monkeypatch.setattr(hermes_state, "DEFAULT_DB_PATH", get_hermes_home() / "state.db")
+ c = TestClient(app)
+ c.headers[_SESSION_HEADER_NAME] = _SESSION_TOKEN
+ return c
+
+
+@pytest.fixture(autouse=True)
+def _clean_leases():
+ from tools import tts_tool
+
+ tts_tool._reset_tts_leases_for_tests()
+ for cache in tts_tool._LOCAL_TTS_MODEL_CACHES.values():
+ cache.clear()
+ yield
+ tts_tool._reset_tts_leases_for_tests()
+ for cache in tts_tool._LOCAL_TTS_MODEL_CACHES.values():
+ cache.clear()
+
+
+def test_active_acquires_and_warms(client, monkeypatch):
+ from tools import tts_tool
+
+ warmed = []
+ monkeypatch.setattr(
+ tts_tool,
+ "warm_tts_provider",
+ lambda cfg=None, provider=None: warmed.append(1) or {"provider": "piper", "warmed": True, "action": "loaded"},
+ )
+
+ resp = client.post("/api/audio/tts-lease", json={"lease": "desktop:read-aloud", "active": True})
+ assert resp.status_code == 200
+ body = resp.json()
+ assert body["ok"] is True
+ assert body["lease"] == "desktop:read-aloud"
+ assert body["active"] is True
+ assert body["leases"] == 1
+ assert body["action"] == "loaded"
+ assert warmed == [1]
+ assert tts_tool.tts_lease_holders() == ["desktop:read-aloud"]
+
+
+def test_inactive_releases_and_unloads_when_last(client, monkeypatch):
+ from tools import tts_tool
+
+ monkeypatch.setattr(tts_tool, "warm_tts_provider", lambda cfg=None, provider=None: {"action": "noop", "warmed": False, "provider": "piper"})
+ client.post("/api/audio/tts-lease", json={"lease": "desktop:read-aloud", "active": True})
+ client.post("/api/audio/tts-lease", json={"lease": "desktop:conversation:abc", "active": True})
+ tts_tool._piper_voice_cache["voice"] = object()
+
+ first = client.post("/api/audio/tts-lease", json={"lease": "desktop:read-aloud", "active": False}).json()
+ assert first["leases"] == 1
+ assert first["released"] == 0
+ assert len(tts_tool._piper_voice_cache) == 1
+
+ last = client.post("/api/audio/tts-lease", json={"lease": "desktop:conversation:abc", "active": False}).json()
+ assert last["leases"] == 0
+ assert last["released"] == 1
+ assert tts_tool._piper_voice_cache == {}
+
+
+def test_warm_failure_is_reported_not_an_http_error(client, monkeypatch):
+ from tools import tts_tool
+
+ def _boom(cfg=None, provider=None):
+ raise RuntimeError("engine exploded")
+
+ monkeypatch.setattr(tts_tool, "warm_tts_provider", _boom)
+ resp = client.post("/api/audio/tts-lease", json={"lease": "desktop:read-aloud", "active": True})
+ assert resp.status_code == 200
+ body = resp.json()
+ assert body["ok"] is True
+ assert body["action"] == "error"
+ assert "engine exploded" in body["error"]
+
+
+def test_blank_lease_rejected(client):
+ resp = client.post("/api/audio/tts-lease", json={"lease": " ", "active": True})
+ assert resp.status_code == 400
+
+
+def test_active_default_true(client, monkeypatch):
+ from tools import tts_tool
+
+ monkeypatch.setattr(tts_tool, "warm_tts_provider", lambda cfg=None, provider=None: {"action": "noop", "warmed": False, "provider": "x"})
+ resp = client.post("/api/audio/tts-lease", json={"lease": "tui:x"})
+ assert resp.json()["active"] is True
+ assert tts_tool.tts_lease_holders() == ["tui:x"]
+
+
+def test_acquire_resolves_provider_inside_target_profile(client, isolated_profiles, monkeypatch):
+ """Warm-up must read the REQUESTING profile's tts config, like /api/audio/speak."""
+ import yaml
+ from tools import tts_tool
+
+ (isolated_profiles["worker_beta"] / "config.yaml").write_text(
+ yaml.safe_dump({"tts": {"provider": "kittentts"}}), encoding="utf-8"
+ )
+ seen = {}
+
+ def _fake_warm(cfg=None, provider=None):
+ from hermes_constants import get_hermes_home
+
+ seen["home"] = str(get_hermes_home())
+ seen["provider"] = tts_tool._get_provider(tts_tool._load_tts_config())
+ return {"action": "noop", "warmed": False, "provider": seen["provider"]}
+
+ monkeypatch.setattr(tts_tool, "warm_tts_provider", _fake_warm)
+ resp = client.post("/api/audio/tts-lease?profile=worker_beta", json={"lease": "desktop:read-aloud", "active": True})
+ assert resp.status_code == 200
+ assert seen["home"] == str(isolated_profiles["worker_beta"])
+ assert seen["provider"] == "kittentts"
+
+
+def test_unknown_profile_404(client):
+ resp = client.post("/api/audio/tts-lease?profile=ghost", json={"lease": "desktop:read-aloud", "active": True})
+ assert resp.status_code == 404
diff --git a/tests/hermes_cli/test_windows_gateway_job_teardown_48820.py b/tests/hermes_cli/test_windows_gateway_job_teardown_48820.py
new file mode 100644
index 0000000000..285f0b1b6c
--- /dev/null
+++ b/tests/hermes_cli/test_windows_gateway_job_teardown_48820.py
@@ -0,0 +1,189 @@
+"""Regression tests for #48820 (4th repro): job-object teardown killed the
+post-update respawned gateway silently, and the updater printed
+"✓ Restarting Windows gateway profile(s)" anyway.
+
+Two fixes under test:
+
+1. ``_spawn_gateway_restart_watcher``'s inlined watcher source must
+ (a) route the respawned gateway's stray stdout/stderr to
+ ``logs/gateway-stdio.log`` (it was ``DEVNULL`` — a gateway killed by
+ parent Job Object teardown left ZERO trace anywhere), and
+ (b) stamp ``_HERMES_GATEWAY_BREAKAWAY`` =1/0 on the respawn env exactly
+ like the canonical ``gateway_windows._spawn_detached``, so the
+ lifecycle/exit-diag records show whether the gateway escaped the
+ parent's Job Object.
+
+2. ``_resume_windows_gateways_after_update`` must verify a stable gateway
+ process actually exists (via ``gateway_windows._wait_for_gateway_ready``)
+ before printing the ✓ — a truthy launch return only proves the watcher
+ process was created, not that the respawned gateway survived the
+ updater's Job Object teardown.
+"""
+
+from unittest.mock import patch
+
+import pytest
+
+import hermes_cli.gateway as gateway
+import hermes_cli.gateway_windows as gateway_windows
+import hermes_cli.main as hm
+from hermes_cli._subprocess_compat import _WINDOWS_GATEWAY_BREAKAWAY_ENV
+from hermes_cli.update_cmd import _resume_windows_gateways_after_update
+
+
+# ---------------------------------------------------------------------------
+# 1. Watcher template contract
+# ---------------------------------------------------------------------------
+
+
+def _captured_watcher_source(monkeypatch) -> str:
+ """Spawn the watcher with a mocked Popen and return the inlined -c source."""
+ captured = {}
+
+ def fake_popen(argv, **kwargs):
+ captured["argv"] = argv
+ captured["kwargs"] = kwargs
+
+ class _P:
+ pid = 12345
+
+ return _P()
+
+ monkeypatch.setattr(gateway.subprocess, "Popen", fake_popen)
+ assert gateway._spawn_gateway_restart_watcher(
+ 999999, ["python", "-m", "hermes_cli.main", "gateway", "run"]
+ )
+ argv = captured["argv"]
+ assert argv[1] == "-c"
+ return argv[2]
+
+
+class TestWatcherRespawnTemplate:
+ def test_respawn_stdio_routed_to_sidecar_log_not_devnull(self, monkeypatch):
+ """DEVNULL swallowed the dying gateway's last words (#48820 4th
+ repro: 'Zero trace anywhere ... because the watcher respawns with
+ stdout=DEVNULL, stderr=DEVNULL')."""
+ src = _captured_watcher_source(monkeypatch)
+ assert "gateway-stdio.log" in src, (
+ "watcher respawn must route stray stdout/stderr to the same "
+ "sidecar log _spawn_detached uses, so a gateway killed moments "
+ "after respawn leaves a trace"
+ )
+ # DEVNULL remains only as the fallback when the log dir is
+ # unavailable — the popen kwargs must not be hardwired to it.
+ assert '"stdout": _stdio_target' in src
+ assert '"stderr": _stdio_target' in src
+
+ def test_respawn_stamps_breakaway_state_like_spawn_detached(
+ self, monkeypatch
+ ):
+ """The respawned gateway must carry _HERMES_GATEWAY_BREAKAWAY=1 on
+ the primary (breakaway) spawn and =0 on the no-breakaway fallback,
+ mirroring gateway_windows._spawn_detached — without the stamp, a
+ job-teardown kill is indistinguishable from any other silent death
+ in the exit diagnostics."""
+ src = _captured_watcher_source(monkeypatch)
+ assert "_WINDOWS_GATEWAY_BREAKAWAY_ENV" in src
+ assert _WINDOWS_GATEWAY_BREAKAWAY_ENV == "_HERMES_GATEWAY_BREAKAWAY"
+ # Primary stamps "1", the OSError fallback stamps "0".
+ assert '_WINDOWS_GATEWAY_BREAKAWAY_ENV: "1"' in src
+ assert '_WINDOWS_GATEWAY_BREAKAWAY_ENV: "0"' in src
+
+ def test_respawn_source_compiles(self, monkeypatch):
+ """The inlined -c template is built via str.format over a
+ dedented literal — guard against brace/indentation regressions."""
+ src = _captured_watcher_source(monkeypatch)
+ compile(src, "", "exec")
+
+ def test_watcher_fallback_retry_preserved(self, monkeypatch):
+ """The ERROR_ACCESS_DENIED retry without breakaway must survive."""
+ src = _captured_watcher_source(monkeypatch)
+ assert "windows_detach_flags_without_breakaway" in src
+
+
+# ---------------------------------------------------------------------------
+# 2. Post-update resume liveness gate
+# ---------------------------------------------------------------------------
+
+
+def _token(profiles: dict) -> dict:
+ return {
+ "resume_needed": True,
+ "profiles": profiles,
+ "unmapped_pids": [],
+ "unmapped": [],
+ }
+
+
+class TestResumeLivenessGate:
+ @pytest.fixture(autouse=True)
+ def _windows(self, monkeypatch):
+ monkeypatch.setattr(hm, "_is_windows", lambda: True)
+ monkeypatch.setattr(hm, "_refresh_windows_gateway_launchers", lambda: None)
+ monkeypatch.setattr(
+ gateway, "launch_detached_profile_gateway_restart", lambda *_a: True
+ )
+ monkeypatch.setattr(
+ gateway, "launch_detached_gateway_restart_by_cmdline", lambda *_a: True
+ )
+
+ def test_dead_respawn_fails_the_resume_instead_of_printing_check(
+ self, monkeypatch
+ ):
+ """No stable gateway after the relaunch → the resume raises (update
+ marked incomplete) instead of printing '✓ Restarting'. This is the
+ exact #48820 3rd/4th-repro hole: spawn succeeded, gateway died
+ within seconds, success was reported, platforms were offline for
+ 12.5 hours."""
+ monkeypatch.setattr(
+ gateway_windows, "_wait_for_gateway_ready", lambda **_kw: []
+ )
+ token = _token({"default": 1111})
+ printed = []
+ with patch("builtins.print", side_effect=lambda *a, **k: printed.append(a)):
+ with pytest.raises(RuntimeError, match="not verified alive"):
+ _resume_windows_gateways_after_update(token)
+
+ text = " ".join(str(a) for a in printed)
+ assert "✓ Restarting" not in text
+ assert "could not be verified" in text
+ # The profile stays on the token so retry/reporting still sees it.
+ assert token["profiles"] == {"default": 1111}
+ assert token["resume_needed"] is True
+
+ def test_live_respawn_prints_check_and_writes_attestation(self, monkeypatch):
+ monkeypatch.setattr(
+ gateway_windows, "_wait_for_gateway_ready", lambda **_kw: [777]
+ )
+ attested = {}
+ monkeypatch.setattr(
+ gateway_windows,
+ "_write_start_attestation",
+ lambda pids, via: attested.update(pids=pids, via=via),
+ )
+ token = _token({"default": 1111})
+ printed = []
+ with patch("builtins.print", side_effect=lambda *a, **k: printed.append(a)):
+ _resume_windows_gateways_after_update(token)
+
+ text = " ".join(str(a) for a in printed)
+ assert "✓ Restarting" in text
+ assert attested == {"pids": [777], "via": "post-update relaunch"}
+ assert token["resume_needed"] is False
+
+ def test_liveness_poll_scans_all_profiles(self, monkeypatch):
+ """The resume relaunches the whole fleet; the verification must not
+ be scoped to the active profile."""
+ seen = {}
+
+ def fake_wait(**kwargs):
+ seen.update(kwargs)
+ return [777]
+
+ monkeypatch.setattr(gateway_windows, "_wait_for_gateway_ready", fake_wait)
+ monkeypatch.setattr(
+ gateway_windows, "_write_start_attestation", lambda *_a, **_kw: None
+ )
+ with patch("builtins.print"):
+ _resume_windows_gateways_after_update(_token({"work": 2222}))
+ assert seen.get("all_profiles") is True
diff --git a/tests/hermes_cli/test_windows_update_restart_reconciliation.py b/tests/hermes_cli/test_windows_update_restart_reconciliation.py
index 0de2d6bbd9..4b3279ab56 100644
--- a/tests/hermes_cli/test_windows_update_restart_reconciliation.py
+++ b/tests/hermes_cli/test_windows_update_restart_reconciliation.py
@@ -24,6 +24,7 @@ from unittest.mock import patch
import pytest
import hermes_cli.gateway as gateway
+import hermes_cli.gateway_windows as gateway_windows
import hermes_cli.main as hm
from hermes_cli.update_cmd import _resume_windows_gateways_after_update
from hermes_cli.update_inventory import (
@@ -43,6 +44,21 @@ def _token(profiles: dict) -> dict:
}
+@pytest.fixture(autouse=True)
+def _stub_post_relaunch_liveness(monkeypatch):
+ """The resume path now verifies a stable gateway process actually exists
+ before vouching for the relaunch (#48820 3rd/4th repro — a parent Job
+ Object killing the respawned gateway made '✓ Restarting' a lie). These
+ reconciliation tests exercise the token bookkeeping, not the liveness
+ poll, so stub it as 'gateway came up'."""
+ monkeypatch.setattr(
+ gateway_windows, "_wait_for_gateway_ready", lambda **_kw: [4242]
+ )
+ monkeypatch.setattr(
+ gateway_windows, "_write_start_attestation", lambda *_a, **_kw: None
+ )
+
+
def test_resume_records_successfully_relaunched_profiles_on_the_token(monkeypatch):
monkeypatch.setattr(hm, "_is_windows", lambda: True)
monkeypatch.setattr(hm, "_refresh_windows_gateway_launchers", lambda: None)
diff --git a/tests/hermes_state/test_canonical_title_guard.py b/tests/hermes_state/test_canonical_title_guard.py
index b315ff785d..784825c25e 100644
--- a/tests/hermes_state/test_canonical_title_guard.py
+++ b/tests/hermes_state/test_canonical_title_guard.py
@@ -69,3 +69,43 @@ def test_auto_titler_still_cannot_touch_the_canonical_row(db):
assert not db.set_auto_title(sid, "Chat about groceries", source=SessionDB.TITLE_SOURCE_LLM)
row = db.get_session_by_title(SessionDB.CANONICAL_BOT_CHAT_TITLE)
assert row and row["id"] == sid
+
+
+def test_auto_titler_cannot_rename_derived_canonical_bot_chat(db):
+ # #99517: the guard must be provenance-blind. A derived (rank 0) canonical
+ # title loses to an llm (rank 1) auto-title on precedence alone, so the
+ # identity check — not precedence — has to stop the write.
+ db.create_session("derived", source="desktop")
+ assert db._set_session_title(
+ "derived",
+ SessionDB.CANONICAL_BOT_CHAT_TITLE,
+ source=SessionDB.TITLE_SOURCE_DERIVED,
+ )
+ assert db.set_session_hidden("derived", True)
+
+ assert not db.set_auto_title(
+ "derived",
+ "Renamed by titler",
+ source=SessionDB.TITLE_SOURCE_LLM,
+ )
+ row = db.get_session("derived")
+ assert row["title"] == SessionDB.CANONICAL_BOT_CHAT_TITLE
+ assert row["title_source"] == SessionDB.TITLE_SOURCE_DERIVED
+
+
+def test_auto_titler_can_rename_visible_derived_bot_chat(db):
+ # Control: hidden is still the discriminator — a visible session that
+ # merely carries the text "Bot Chat" upgrades derived -> llm as usual.
+ db.create_session("visible", source="desktop")
+ assert db._set_session_title(
+ "visible",
+ SessionDB.CANONICAL_BOT_CHAT_TITLE,
+ source=SessionDB.TITLE_SOURCE_DERIVED,
+ )
+
+ assert db.set_auto_title(
+ "visible",
+ "Renamed by titler",
+ source=SessionDB.TITLE_SOURCE_LLM,
+ )
+ assert db.get_session("visible")["title"] == "Renamed by titler"
diff --git a/tests/hermes_state/test_deleted_wal_generation_guard.py b/tests/hermes_state/test_deleted_wal_generation_guard.py
new file mode 100644
index 0000000000..cdbd201220
--- /dev/null
+++ b/tests/hermes_state/test_deleted_wal_generation_guard.py
@@ -0,0 +1,198 @@
+"""Refuse SessionDB open/write when a deleted WAL generation is still held.
+
+A live writer that keeps the unlinked ``state.db-wal`` inode while a second
+opener would mint a fresh WAL is the split-brain that produces intermittent
+``database disk image is malformed`` / ``disk I/O error``. The store must
+fail closed on both the open and write paths instead of creating the second
+generation.
+"""
+
+import os
+import sqlite3
+import sys
+from pathlib import Path
+
+import pytest
+
+import hermes_state
+from hermes_state import (
+ DeletedWalGenerationError,
+ SessionDB,
+ classify_persistence_error,
+ iter_deleted_sqlite_sidecar_holders,
+ refuse_deleted_wal_generation,
+)
+
+
+@pytest.fixture
+def force_wal(monkeypatch):
+ """Pin WAL so this host's vulnerable SQLite still matches production topology."""
+ monkeypatch.setattr(
+ hermes_state, "is_sqlite_wal_reset_vulnerable", lambda version_info=None: False
+ )
+ monkeypatch.setattr(hermes_state, "resolve_journal_mode", lambda: "wal")
+
+
+def _make_db(path: Path, session_id: str, content: str) -> SessionDB:
+ db = SessionDB(db_path=path)
+ db.create_session(session_id, "cli")
+ db.append_message(session_id, role="user", content=content)
+ return db
+
+
+def _require_wal(db: SessionDB) -> Path:
+ if not db._wal_active:
+ db.close()
+ pytest.skip("WAL not active on this filesystem")
+ wal = Path(os.fspath(db.db_path) + "-wal")
+ if not wal.exists():
+ db.close()
+ pytest.skip("WAL sidecar missing after first write")
+ return wal
+
+
+def _unlink_sidecars(db_path: Path) -> None:
+ for suffix in ("-wal", "-shm"):
+ sidecar = Path(os.fspath(db_path) + suffix)
+ if sidecar.exists():
+ os.unlink(sidecar)
+
+
+def test_classify_deleted_wal_is_replaced_not_disk():
+ err = DeletedWalGenerationError(
+ "FATAL: a live process holds a deleted state.db-wal or state.db-shm "
+ "inode while the path names a different (or missing) generation."
+ )
+ assert classify_persistence_error(err) == "replaced"
+ assert classify_persistence_error(str(err)) == "replaced"
+
+
+def test_iter_holders_empty_on_non_linux(monkeypatch, tmp_path):
+ monkeypatch.setattr(hermes_state.sys, "platform", "win32")
+ assert iter_deleted_sqlite_sidecar_holders(tmp_path / "state.db") == []
+
+
+def test_clean_open_and_second_open_still_work(tmp_path, force_wal):
+ path = tmp_path / "state.db"
+ db = _make_db(path, "s1", "hello")
+ _require_wal(db)
+ db.close()
+ reopened = SessionDB(db_path=path)
+ try:
+ reopened.append_message("s1", role="user", content="second-open")
+ rows = reopened.get_messages("s1")
+ assert any(m["content"] == "second-open" for m in rows)
+ finally:
+ reopened.close()
+
+
+def test_delete_journal_two_writers_still_work(tmp_path, monkeypatch):
+ monkeypatch.setattr(hermes_state, "resolve_journal_mode", lambda: "delete")
+ monkeypatch.setattr(
+ hermes_state, "is_sqlite_wal_reset_vulnerable", lambda version_info=None: False
+ )
+ path = tmp_path / "state.db"
+ a = _make_db(path, "s", "from-a")
+ try:
+ assert not Path(os.fspath(path) + "-wal").exists()
+ b = SessionDB(db_path=path)
+ try:
+ b.append_message("s", role="user", content="from-b")
+ contents = [m["content"] for m in b.get_messages("s")]
+ assert "from-a" in contents
+ assert "from-b" in contents
+ finally:
+ b.close()
+ finally:
+ a.close()
+
+
+@pytest.mark.skipif(
+ not sys.platform.startswith("linux"),
+ reason="deleted-WAL /proc scan is Linux-only",
+)
+def test_iter_finds_self_after_wal_unlink(tmp_path, force_wal):
+ path = tmp_path / "state.db"
+ db = _make_db(path, "s", "held")
+ wal = _require_wal(db)
+ inode_before = wal.stat().st_ino
+ _unlink_sidecars(path)
+ holders = iter_deleted_sqlite_sidecar_holders(path)
+ try:
+ assert holders, "expected this process to still hold the deleted WAL inode"
+ assert any("(deleted)" in target for _pid, target in holders)
+ assert any(
+ target.removesuffix(" (deleted)").endswith(("-wal", "-shm"))
+ for _pid, target in holders
+ )
+ assert not wal.exists() or wal.stat().st_ino != inode_before
+ finally:
+ db.close()
+
+
+@pytest.mark.skipif(
+ not sys.platform.startswith("linux"),
+ reason="deleted-WAL /proc scan is Linux-only",
+)
+def test_second_sessiondb_open_refuses_and_does_not_mint_wal(tmp_path, force_wal):
+ path = tmp_path / "state.db"
+ writer = _make_db(path, "s", "before-unlink")
+ wal = _require_wal(writer)
+ inode_before = wal.stat().st_ino
+ _unlink_sidecars(path)
+ assert not wal.exists()
+
+ with pytest.raises(DeletedWalGenerationError, match="deleted state.db-wal"):
+ SessionDB(db_path=path)
+
+ assert not wal.exists(), "open must refuse before sqlite3.connect mints a WAL"
+ # If a WAL somehow reappeared it must not be a new generation.
+ if wal.exists():
+ assert wal.stat().st_ino == inode_before
+ writer.close()
+
+
+@pytest.mark.skipif(
+ not sys.platform.startswith("linux"),
+ reason="deleted-WAL write halt uses Linux unlink semantics",
+)
+def test_writer_halts_after_own_wal_unlinked(tmp_path, force_wal):
+ path = tmp_path / "state.db"
+ db = _make_db(path, "s", "before")
+ _require_wal(db)
+ recorded = db._db_sidecar_identity.get("-wal")
+ assert recorded is not None
+ _unlink_sidecars(path)
+
+ with pytest.raises(DeletedWalGenerationError, match="deleted state.db-wal"):
+ db.append_message("s", role="user", content="after-unlink")
+ assert db._db_wal_generation_lost is True
+
+ with pytest.raises(DeletedWalGenerationError):
+ db.append_message("s", role="user", content="second-after-halt")
+ db.close()
+
+
+@pytest.mark.skipif(
+ not sys.platform.startswith("linux"),
+ reason="deleted-WAL /proc scan is Linux-only",
+)
+def test_refuse_helper_raises_while_deleted_wal_held(tmp_path, force_wal):
+ path = tmp_path / "state.db"
+ raw = sqlite3.connect(str(path))
+ try:
+ raw.execute("PRAGMA journal_mode=WAL")
+ raw.execute("CREATE TABLE t (id INTEGER PRIMARY KEY, v TEXT)")
+ raw.execute("INSERT INTO t VALUES (1, 'held')")
+ raw.commit()
+ wal = Path(str(path) + "-wal")
+ assert wal.exists()
+ os.unlink(wal)
+ shm = Path(str(path) + "-shm")
+ if shm.exists():
+ os.unlink(shm)
+ with pytest.raises(DeletedWalGenerationError):
+ refuse_deleted_wal_generation(path)
+ assert not wal.exists()
+ finally:
+ raw.close()
diff --git a/tests/hermes_state/test_shared_session_db_registry.py b/tests/hermes_state/test_shared_session_db_registry.py
index ea279c9647..58d4837625 100644
--- a/tests/hermes_state/test_shared_session_db_registry.py
+++ b/tests/hermes_state/test_shared_session_db_registry.py
@@ -35,10 +35,12 @@ def _clean_registry():
registry.close_all()
registry._generations.clear()
registry._retired.clear()
+ registry._opening.clear()
yield
registry.close_all()
registry._generations.clear()
registry._retired.clear()
+ registry._opening.clear()
def _replace_file_preserving_schema(src: Path, dst: Path) -> None:
@@ -178,6 +180,127 @@ def stats_live_for(path: Path):
class TestTeardownOutsideLock:
+ def test_concurrent_cold_acquire_opens_one_writer(self, tmp_path, monkeypatch):
+ """Concurrent first callers must not construct redundant writers.
+
+ Returning one winning object is not enough: every losing constructor
+ has already opened its own writable SQLite connection by then. Hold
+ the first construction so peer callers overlap deterministically and
+ assert the registry single-flights the open itself.
+ """
+ db_path = tmp_path / "state.db"
+ callers = 6
+ ready = threading.Barrier(callers + 1)
+ release_open = threading.Event()
+ count_lock = threading.Lock()
+ open_calls = 0
+ results = []
+ errors = []
+
+ class _FakeDB:
+ def __init__(self, path):
+ self.db_path = path
+ self._shared_registry_owned = False
+ self.closed = False
+
+ def close(self):
+ self.closed = True
+
+ def _blocked_open(path):
+ nonlocal open_calls
+ with count_lock:
+ open_calls += 1
+ assert release_open.wait(5.0)
+ return _FakeDB(path)
+
+ monkeypatch.setattr(registry, "_open_session_db", _blocked_open)
+
+ def _acquire():
+ try:
+ ready.wait()
+ results.append(registry.acquire(db_path))
+ except BaseException as exc: # pragma: no cover - failure path
+ errors.append(exc)
+
+ threads = [threading.Thread(target=_acquire) for _ in range(callers)]
+ for thread in threads:
+ thread.start()
+ ready.wait()
+ time.sleep(0.1)
+ release_open.set()
+ for thread in threads:
+ thread.join(10.0)
+ assert not thread.is_alive(), "concurrent acquire deadlocked"
+
+ assert errors == []
+ assert open_calls == 1
+ assert len({id(db) for db in results}) == 1
+ for db in results:
+ assert registry.release(db) is True
+
+ def test_waiter_retries_after_cold_open_failure(self, tmp_path, monkeypatch):
+ """A failed elected opener must wake a peer to retry the path."""
+ db_path = tmp_path / "state.db"
+ first_entered = threading.Event()
+ release_failure = threading.Event()
+ open_calls = 0
+ results = []
+ errors = []
+
+ class _FakeDB:
+ def __init__(self, path):
+ self.db_path = path
+ self._shared_registry_owned = False
+
+ def close(self):
+ pass
+
+ def _fail_then_open(path):
+ nonlocal open_calls
+ open_calls += 1
+ if open_calls == 1:
+ first_entered.set()
+ assert release_failure.wait(5.0)
+ raise OSError("transient open failure")
+ return _FakeDB(path)
+
+ monkeypatch.setattr(registry, "_open_session_db", _fail_then_open)
+
+ def _acquire():
+ try:
+ results.append(registry.acquire(db_path))
+ except BaseException as exc:
+ errors.append(exc)
+
+ first = threading.Thread(target=_acquire)
+ second = threading.Thread(target=_acquire)
+ first.start()
+ assert first_entered.wait(5.0)
+ second.start()
+ time.sleep(0.1)
+ release_failure.set()
+ first.join(10.0)
+ second.join(10.0)
+
+ assert not first.is_alive()
+ assert not second.is_alive()
+ assert open_calls == 2
+ assert len(errors) == 1
+ assert isinstance(errors[0], OSError)
+ assert len(results) == 1
+ assert registry.release(results[0]) is True
+
+ def test_equivalent_path_spellings_share_generation(self, tmp_path):
+ """Registry identity is the resolved file, not caller spelling."""
+ db_path = tmp_path / "nested" / "state.db"
+ equivalent = tmp_path / "nested" / ".." / "nested" / "state.db"
+
+ first = registry.acquire(db_path)
+ second = registry.acquire(equivalent)
+ assert first is second
+ assert registry.release(first) is True
+ assert registry.release(second) is True
+
def test_final_release_does_not_hold_registry_lock_during_close(self, tmp_path, monkeypatch):
"""A final release's teardown (token-writer stop, WAL checkpoint,
read-pool drain) must run OUTSIDE the registry lock — otherwise
@@ -231,10 +354,16 @@ class TestTeardownOutsideLock:
def _worker(n):
try:
- for _ in range(20):
+ for index in range(20):
db = registry.acquire(db_path)
try:
- db.get_session("nonexistent")
+ db.create_session(
+ session_id=f"worker-{n}-{index}",
+ source="test",
+ model="test-model",
+ model_config={},
+ system_prompt=None,
+ )
finally:
registry.release(db)
except Exception as exc: # pragma: no cover - failure path
@@ -248,6 +377,12 @@ class TestTeardownOutsideLock:
assert not t.is_alive(), "worker deadlocked"
assert errors == []
+ verifier = registry.acquire(db_path)
+ try:
+ with verifier._lock:
+ assert verifier._conn.execute("PRAGMA integrity_check").fetchone()[0] == "ok"
+ finally:
+ registry.release(verifier)
stats = registry.stats()
assert stats["live_generations"] == 0
assert stats["retired_generations"] == 0
diff --git a/tests/hermes_state/test_state_db_corrupt_quarantine.py b/tests/hermes_state/test_state_db_corrupt_quarantine.py
new file mode 100644
index 0000000000..2acac22939
--- /dev/null
+++ b/tests/hermes_state/test_state_db_corrupt_quarantine.py
@@ -0,0 +1,236 @@
+"""Quarantine of a live SessionDB handle after structural (non-FTS) corruption.
+
+Field evidence (the #90837 lost/reordered-page-write class): a gateway kept
+retrying writes for ~50 minutes after ``gateway_routing`` reported
+``database disk image is malformed``; on SIGTERM the close-time
+``PRAGMA wal_checkpoint(PASSIVE)`` then wrote 15 pages to the wrong page
+numbers (page 1 received a ``messages_fts_trigram_data`` leaf) and the file
+stopped opening at all. Once structural corruption is observed on a handle
+the only safe policy is to stop touching the file.
+"""
+
+import sqlite3
+
+import pytest
+
+from hermes_state import SessionDB, StateDbCorruptError
+
+
+class _MalformedConn:
+ """Connection proxy whose every execute reports bare SQLITE_CORRUPT."""
+
+ def __init__(self, real_conn):
+ self._real = real_conn
+
+ def execute(self, *args, **kwargs):
+ raise sqlite3.DatabaseError("database disk image is malformed")
+
+ def __getattr__(self, name):
+ return getattr(self._real, name)
+
+
+class TestQuarantineAfterStructuralCorruption:
+ def test_structural_corruption_sets_sticky_flag_and_raises_typed(self, tmp_path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ real_conn = db._conn
+ try:
+ db.create_session(session_id="s1", source="cli", model="test")
+ db._conn = _MalformedConn(real_conn)
+ with pytest.raises(StateDbCorruptError, match="malformed") as excinfo:
+ db.create_session(session_id="s2", source="cli", model="test")
+ assert isinstance(excinfo.value.__cause__, sqlite3.DatabaseError)
+ assert db._db_corrupt is True
+ # Structural damage must never be mistaken for FTS-scoped damage.
+ assert db._fts_stale is False
+ finally:
+ db._conn = real_conn
+ db.close()
+
+
+class _RecordingConn:
+ """Connection proxy that records every SQL text and delegates."""
+
+ def __init__(self, real_conn):
+ self._real = real_conn
+ self.recorded = []
+
+ def execute(self, sql, *args, **kwargs):
+ self.recorded.append(str(sql))
+ return self._real.execute(sql, *args, **kwargs)
+
+ def __getattr__(self, name):
+ return getattr(self._real, name)
+
+
+def _quarantined_db(tmp_path):
+ """A SessionDB whose first corrupt write already tripped the quarantine."""
+ db = SessionDB(db_path=tmp_path / "state.db")
+ real_conn = db._conn
+ db.create_session(session_id="s1", source="cli", model="test")
+ db._conn = _MalformedConn(real_conn)
+ with pytest.raises(StateDbCorruptError):
+ db.create_session(session_id="s2", source="cli", model="test")
+ db._conn = real_conn
+ assert db._db_corrupt is True
+ return db, real_conn
+
+
+class TestQuarantinedHandleStopsTouchingTheFile:
+ def test_subsequent_writes_fail_fast_without_touching_connection(self, tmp_path):
+ db, real_conn = _quarantined_db(tmp_path)
+ recorder = _RecordingConn(real_conn)
+ db._conn = recorder
+ try:
+ with pytest.raises(StateDbCorruptError):
+ db.create_session(session_id="s3", source="cli", model="test")
+ assert recorder.recorded == []
+ finally:
+ db._conn = real_conn
+ db.close()
+
+ def test_close_skips_wal_checkpoint_when_quarantined(self, tmp_path, caplog):
+ db, real_conn = _quarantined_db(tmp_path)
+ recorder = _RecordingConn(real_conn)
+ db._conn = recorder
+ with caplog.at_level("WARNING", logger="hermes_state"):
+ db.close()
+ assert not any("wal_checkpoint" in sql for sql in recorder.recorded)
+ assert db._conn is None
+ assert any(
+ "Skipping the close-time WAL checkpoint" in rec.getMessage()
+ and "hermes sessions recover" in rec.getMessage()
+ for rec in caplog.records
+ )
+
+ def test_close_disables_sqlite_internal_checkpoint_on_py312(self, tmp_path):
+ """Quarantine must also stop SQLite's own last-connection checkpoint.
+
+ Skipping the explicit PRAGMA is not enough: sqlite3.Connection.close()
+ runs an internal PASSIVE checkpoint and unlinks -wal/-shm unless
+ SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE is set (Connection.setconfig,
+ Python 3.12+). On 3.11 the switch is unavailable — skip there.
+ """
+ flag = getattr(sqlite3, "SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE", None)
+ db = SessionDB(db_path=tmp_path / "state.db")
+ if flag is None or not hasattr(db._conn, "setconfig"):
+ db.close()
+ pytest.skip("SQLITE_DBCONFIG_NO_CKPT_ON_CLOSE needs Python 3.12+")
+ real_conn = db._conn
+ db.create_session(session_id="s1", source="cli", model="test")
+ assert real_conn.getconfig(flag) is False
+ db._conn = _MalformedConn(real_conn)
+ with pytest.raises(StateDbCorruptError):
+ db.create_session(session_id="s2", source="cli", model="test")
+ db._conn = real_conn
+ # _halt_db_corrupt armed the no-checkpoint-on-close switch.
+ assert real_conn.getconfig(flag) is True
+ db.close()
+
+ def test_reopen_after_close_refused_when_quarantined(self, tmp_path, monkeypatch):
+ from unittest.mock import MagicMock
+
+ db, real_conn = _quarantined_db(tmp_path)
+ db.close()
+ reopen = MagicMock()
+ monkeypatch.setattr("hermes_state._connect_tracked_db", reopen)
+ with pytest.raises(StateDbCorruptError, match="structural corruption"):
+ db.create_session(session_id="s4", source="cli", model="test")
+ reopen.assert_not_called()
+ # The read fallback after close() goes through the same reopen path.
+ with pytest.raises(StateDbCorruptError, match="refusing to reopen"):
+ db.get_session("s1")
+ reopen.assert_not_called()
+
+
+class TestQuarantineScope:
+ def test_fts_scoped_corruption_does_not_trip_flag(self, tmp_path):
+ """Corrupt FTS shadow tables keep the existing fail-open detach path."""
+ path = tmp_path / "state.db"
+ db = SessionDB(db_path=path)
+ db.create_session(session_id="s1", source="cli", model="test")
+ db.append_message("s1", role="user", content="hello world")
+ raw = sqlite3.connect(str(path))
+ raw.execute(
+ "UPDATE messages_fts_data SET block = X'DEADBEEFDEADBEEFDEADBEEFDEADBEEF'"
+ )
+ raw.commit()
+ raw.close()
+ try:
+ db.append_message("s1", role="user", content="healed append")
+ assert db._db_corrupt is False
+ assert db._fts_stale is True
+ assert db._fts_enabled is False
+ finally:
+ db.close()
+
+ def test_replaced_file_takes_precedence_over_corrupt(self, tmp_path):
+ import os
+
+ from hermes_state import StateDbReplacedError
+
+ live = tmp_path / "state.db"
+ other = tmp_path / "other.db"
+ db = SessionDB(db_path=live)
+ real_conn = db._conn
+ try:
+ db.create_session(session_id="s1", source="cli", model="test")
+ if db._db_file_identity is None:
+ pytest.skip("filesystem does not expose st_dev/st_ino")
+ alt = SessionDB(db_path=other)
+ alt.create_session("other", "cli")
+ alt.close()
+ os.replace(other, live)
+ db._conn = _MalformedConn(real_conn)
+ with pytest.raises(StateDbReplacedError):
+ db.create_session(session_id="s2", source="cli", model="test")
+ assert db._db_replaced is True
+ assert db._db_corrupt is False
+ finally:
+ db._conn = real_conn
+ db.close()
+
+ def test_classify_persistence_error_maps_quarantine_to_corrupt(self):
+ from hermes_state import _STATE_DB_CORRUPT_MSG, classify_persistence_error
+
+ assert classify_persistence_error(StateDbCorruptError("x")) == "corrupt"
+ # The stringified form (RPC boundaries) must classify the same way.
+ assert classify_persistence_error(_STATE_DB_CORRUPT_MSG) == "corrupt"
+
+
+@pytest.fixture
+def _clean_registry():
+ import hermes_state_registry as registry
+
+ registry.close_all()
+ registry._generations.clear()
+ registry._retired.clear()
+ yield registry
+ registry.close_all()
+ registry._generations.clear()
+ registry._retired.clear()
+
+
+class TestSharedRegistry:
+ def test_holders_share_quarantine_and_close_all_skips_checkpoint(
+ self, tmp_path, _clean_registry
+ ):
+ registry = _clean_registry
+ path = tmp_path / "state.db"
+ holder_a = registry.acquire(path)
+ holder_b = registry.acquire(path)
+ assert holder_a is holder_b
+ real_conn = holder_a._conn
+ holder_a.create_session(session_id="s1", source="cli", model="test")
+
+ holder_a._conn = _MalformedConn(real_conn)
+ with pytest.raises(StateDbCorruptError):
+ holder_a.create_session(session_id="s2", source="cli", model="test")
+ recorder = _RecordingConn(real_conn)
+ holder_b._conn = recorder
+
+ with pytest.raises(StateDbCorruptError):
+ holder_b.create_session(session_id="s3", source="cli", model="test")
+
+ registry.close_all()
+ assert not any("wal_checkpoint" in sql for sql in recorder.recorded)
+ assert holder_a._conn is None
diff --git a/tests/hermes_state/test_state_db_file_identity.py b/tests/hermes_state/test_state_db_file_identity.py
index 1877cf857a..3cc1ca1272 100644
--- a/tests/hermes_state/test_state_db_file_identity.py
+++ b/tests/hermes_state/test_state_db_file_identity.py
@@ -190,3 +190,110 @@ def test_divert_session_transcript_jsonl_appends(tmp_path, monkeypatch):
def _stat_changed(path: Path, recorded) -> bool:
st = os.stat(path)
return (st.st_dev, st.st_ino) != recorded
+
+
+# ---------------------------------------------------------------------------
+# Lock safety of the identity probe itself (#100368 / howtocorrupt §2.2).
+#
+# _read_sqlite_application_id runs on EVERY write against the LIVE state.db.
+# Before the _pread_db_header fix it did open("rb")/read/close, and that
+# close() cancelled every POSIX advisory lock this process held on the file
+# — including the WAL-mode DMS shared lock of the writer connection. These
+# tests measure the actual kernel lock table (/proc/locks), so they are
+# Linux-only; the hazard itself is POSIX-only.
+# ---------------------------------------------------------------------------
+
+def _posix_locks_on(paths):
+ """Set of (inode, type, mode, start, end) locks held by this pid."""
+ import sys as _sys
+ if not _sys.platform.startswith("linux"):
+ pytest.skip("lock-table probe requires /proc/locks (Linux)")
+ inodes = {}
+ for p in paths:
+ try:
+ inodes[os.stat(p).st_ino] = str(p)
+ except OSError:
+ continue
+ pid = os.getpid()
+ held = set()
+ for line in Path("/proc/locks").read_text().splitlines():
+ parts = line.split()
+ try:
+ lpid = int(parts[4])
+ ino = int(parts[5].split(":")[2])
+ except (IndexError, ValueError):
+ continue
+ if lpid == pid and ino in inodes:
+ held.add((ino, parts[1], parts[3], parts[6], parts[7]))
+ return held
+
+
+def test_identity_probe_does_not_cancel_live_posix_locks(tmp_path):
+ """The on-write header probe must not drop the writer's DMS lock."""
+ from hermes_state import _read_sqlite_application_id
+
+ live = tmp_path / "state.db"
+ db = _make_db(live, "probe-sess", "seed")
+ try:
+ sidecars = [live, Path(str(live) + "-shm")]
+ # Hold an open write transaction: that is when the connection holds
+ # POSIX range locks on the main db file, and exactly the state a
+ # concurrent _raise_if_db_replaced probe (another thread, same
+ # process) can destroy.
+ db._conn.execute("BEGIN IMMEDIATE")
+ db._conn.execute(
+ "UPDATE sessions SET source = source WHERE id = 'probe-sess'"
+ )
+ before = _posix_locks_on(sidecars)
+ assert before, "expected in-transaction WAL connection to hold POSIX locks"
+
+ for _ in range(3):
+ _read_sqlite_application_id(live)
+
+ after = _posix_locks_on(sidecars)
+ db._conn.rollback()
+ lost = before - after
+ assert not lost, (
+ "identity probe cancelled POSIX locks held by the live "
+ f"connection (howtocorrupt §2.2): {lost}"
+ )
+ # The decisive check: the WAL DMS shared lock on the MAIN db file
+ # must survive. With the pre-fix open/read/close probe the close()
+ # cancels it (it is already gone by the time the connection has run
+ # its first identity check in __init__), leaving other processes
+ # free to treat this writer as dead and rerun WAL-index recovery
+ # underneath it.
+ db_ino = os.stat(live).st_ino
+ main_db_locks = {lk for lk in after if lk[0] == db_ino}
+ assert main_db_locks, (
+ "live writer connection holds no POSIX lock on state.db itself — "
+ "the WAL DMS lock was cancelled by a raw open/close probe "
+ "(howtocorrupt §2.2)"
+ )
+ # The connection must still be able to commit.
+ db.append_message("probe-sess", role="user", content="post-probe")
+ finally:
+ db.close()
+
+
+def test_identity_probe_still_detects_replacement_after_fd_cache(tmp_path):
+ """The cached-fd probe rebinds when the path names a new inode."""
+ from hermes_state import _read_sqlite_application_id
+
+ live = tmp_path / "state.db"
+ other = tmp_path / "other.db"
+ db = _make_db(live, "live-sess", "original")
+ _require_identity(db)
+ first = _read_sqlite_application_id(live) # populates the fd cache
+ db.close()
+
+ alt = _make_db(other, "other-sess", "replacement")
+ alt.close()
+ os.replace(other, live)
+
+ second = _read_sqlite_application_id(live)
+ assert second is not None
+ assert second != first, (
+ "probe kept reading the retired inode instead of rebinding to the "
+ "replacement file"
+ )
diff --git a/tests/hermes_state/test_sweep_orphaned_sessions.py b/tests/hermes_state/test_sweep_orphaned_sessions.py
index 4554513f2f..8d56985d61 100644
--- a/tests/hermes_state/test_sweep_orphaned_sessions.py
+++ b/tests/hermes_state/test_sweep_orphaned_sessions.py
@@ -16,6 +16,7 @@ to be older than the cutoff:
actively producing messages.
"""
+import threading
import time
import pytest
@@ -44,6 +45,15 @@ def _set_message_timestamps(db: SessionDB, session_id: str, ts: float) -> None:
db._conn.commit()
+def _set_last_activity(db: SessionDB, session_id: str, ts: float) -> None:
+ conn = db._conn
+ assert conn is not None
+ conn.execute(
+ "UPDATE sessions SET last_activity_at = ? WHERE id = ?", (ts, session_id)
+ )
+ conn.commit()
+
+
def _make_session(
db: SessionDB,
session_id: str,
@@ -99,6 +109,17 @@ class TestSweepOrphanedSessions:
assert db.sweep_orphaned_sessions(max_idle_seconds=IDLE_S) == []
assert db.get_session("active")["ended_at"] is None
+ def test_recent_heartbeat_spares_old_session(self, db):
+ """A turn heartbeat is activity even before its next message lands."""
+ stale = time.time() - 48 * 3600
+ _make_session(
+ db, "active-heartbeat", source="tui", started_at=stale, message_at=stale
+ )
+ _set_last_activity(db, "active-heartbeat", time.time())
+
+ assert db.sweep_orphaned_sessions(max_idle_seconds=IDLE_S) == []
+ assert db.get_session("active-heartbeat")["ended_at"] is None
+
def test_fresh_session_with_old_copied_messages_spared(self, db):
"""Compression/branch children copy history — old message timestamps
on a just-created row must not get it swept."""
@@ -163,9 +184,398 @@ class TestSweepOrphanedSessions:
assert db.get_session("stale-cli")["end_reason"] == "startup_orphan_reap"
assert db.get_session("stale-tui")["ended_at"] is None
+ def test_explicit_source_scope_spares_gateway_sessions(self, db):
+ stale = time.time() - 8 * 3600
+ _make_session(
+ db, "stale-cron", source="cron", started_at=stale, message_at=stale
+ )
+ for sid, session_key in (
+ ("keyed-telegram", "telegram:chat:1"),
+ ("unkeyed-telegram", None),
+ ):
+ db.create_session(sid, source="telegram", session_key=session_key)
+ db.append_message(sid, role="user", content="hello")
+ _set_message_timestamps(db, sid, stale)
+ _backdate_session(db, sid, stale)
+
+ assert db.sweep_orphaned_sessions(
+ max_idle_seconds=IDLE_S, sources=("cron",)
+ ) == ["stale-cron"]
+ assert db.get_session("stale-cron")["end_reason"] == "startup_orphan_reap"
+ assert db.get_session("keyed-telegram")["ended_at"] is None
+ assert db.get_session("unkeyed-telegram")["ended_at"] is None
+
+ def test_automatic_source_scope_spares_pinned_session(self, db):
+ stale = time.time() - 8 * 3600
+ _make_session(
+ db, "pinned", source="cli", started_at=stale, message_at=stale
+ )
+ db.set_session_pinned("pinned", True)
+
+ assert db.sweep_orphaned_sessions(
+ max_idle_seconds=IDLE_S,
+ sources=("cli",),
+ exclude_pinned=True,
+ ) == []
+ assert db.get_session("pinned")["ended_at"] is None
+
+ def test_live_turn_lease_on_compression_lineage_spares_session(self, db):
+ stale = time.time() - 8 * 3600
+ _make_session(db, "root", source="cli", started_at=stale, message_at=stale)
+ db.end_session("root", "compression")
+ db.create_session("tip", source="cli", parent_session_id="root")
+ db.append_message("tip", role="user", content="continued")
+ _set_message_timestamps(db, "tip", stale)
+ _backdate_session(db, "tip", stale)
+ assert db.try_acquire_session_turn_lease(
+ "tip", "external-turn", ttl_seconds=300
+ )
+
+ assert db.sweep_orphaned_sessions(
+ max_idle_seconds=IDLE_S, sources=("cli",)
+ ) == []
+ assert db.get_session("tip")["ended_at"] is None
+
+ def test_active_compression_lock_spares_and_expiry_fences_owner(self, db):
+ stale = time.time() - 8 * 3600
+ _make_session(
+ db, "compressing", source="cli", started_at=stale, message_at=stale
+ )
+ assert db.try_acquire_compression_lock(
+ "compressing", "compressor", ttl_seconds=300
+ )
+
+ assert db.sweep_orphaned_sessions(
+ max_idle_seconds=IDLE_S, sources=("cli",)
+ ) == []
+
+ conn = db._conn
+ assert conn is not None
+ conn.execute(
+ "UPDATE compression_locks SET expires_at = ? WHERE session_id = ?",
+ (time.time() - 1, "compressing"),
+ )
+ conn.commit()
+
+ assert db.sweep_orphaned_sessions(
+ max_idle_seconds=IDLE_S, sources=("cli",)
+ ) == ["compressing"]
+ assert db.get_compression_lock_holder("compressing") is None
+ assert db.refresh_compression_lock("compressing", "compressor") is False
+
+ def test_expired_turn_lease_does_not_block_sweep(self, db):
+ stale = time.time() - 8 * 3600
+ _make_session(
+ db, "expired", source="cli", started_at=stale, message_at=stale
+ )
+ assert db.try_acquire_session_turn_lease(
+ "expired", "expired-turn", ttl_seconds=300
+ )
+ db._conn.execute(
+ "UPDATE session_turn_leases SET expires_at = ? WHERE conversation_id = ?",
+ (time.time() - 1, "expired"),
+ )
+ db._conn.commit()
+
+ assert db.sweep_orphaned_sessions(
+ max_idle_seconds=IDLE_S, sources=("cli",)
+ ) == ["expired"]
+ assert db.get_session("expired")["end_reason"] == "startup_orphan_reap"
+ assert db.refresh_session_turn_lease("expired", "expired-turn") is False
+
+ def test_auto_prune_closes_stale_state_owned_rows_but_spares_live_turns(self, db):
+ stale = time.time() - 100 * 86400
+ recent = time.time() - 86400
+ for sid, source in (
+ ("orphan", "cli"),
+ ("live-turn", "cli"),
+ ("stale-cron", "cron"),
+ ("runtime-owned-ui", "tui"),
+ ):
+ _make_session(db, sid, source=source, started_at=stale, message_at=stale)
+ _set_last_activity(db, sid, stale)
+ _make_session(
+ db,
+ "recent-orphan",
+ source="cli",
+ started_at=recent,
+ message_at=recent,
+ )
+ _set_last_activity(db, "recent-orphan", recent)
+ db.create_session(
+ "keyed", source="telegram", session_key="telegram:chat:1"
+ )
+ _backdate_session(db, "keyed", stale)
+ db.create_session("unkeyed-gateway", source="telegram")
+ _backdate_session(db, "unkeyed-gateway", stale)
+ _set_last_activity(db, "unkeyed-gateway", stale)
+ assert db.try_acquire_session_turn_lease(
+ "live-turn", "external-turn", ttl_seconds=300
+ )
+ db.register_backend_heartbeat(
+ backend_id="unrelated-dashboard",
+ pid=12345,
+ started_at=time.time(),
+ last_heartbeat=time.time(),
+ )
+
+ first = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+
+ assert first["pruned"] == 0
+ assert db.get_session("orphan")["end_reason"] == "startup_orphan_reap"
+ assert db.get_session("stale-cron")["end_reason"] == "startup_orphan_reap"
+ assert db.get_session("live-turn")["ended_at"] is None
+ assert db.get_session("recent-orphan")["ended_at"] is None
+ assert db.get_session("runtime-owned-ui")["ended_at"] is None
+ assert db.get_session("keyed")["ended_at"] is None
+ assert db.get_session("unkeyed-gateway")["ended_at"] is None
+
+ second = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+
+ assert second["pruned"] == 0
+ assert db.get_session("orphan") is not None
+ assert db.get_session("stale-cron") is not None
+
+ db._conn.execute(
+ "UPDATE sessions SET ended_at = ? WHERE id IN (?, ?)",
+ (stale, "orphan", "stale-cron"),
+ )
+ db._conn.commit()
+ third = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+
+ assert third["pruned"] == 2
+ assert db.get_session("orphan") is None
+ assert db.get_session("stale-cron") is None
+
+ def test_failed_maintenance_marker_keeps_newly_swept_row_recoverable(
+ self, db, monkeypatch
+ ):
+ stale = time.time() - 100 * 86400
+ _make_session(
+ db,
+ "recoverable",
+ source="cli",
+ started_at=stale,
+ message_at=stale,
+ )
+ _set_last_activity(db, "recoverable", stale)
+ set_meta = db.set_meta
+ fail_once = True
+
+ def flaky_set_meta(key, value):
+ nonlocal fail_once
+ if key == "last_auto_prune" and fail_once:
+ fail_once = False
+ raise RuntimeError("injected marker failure")
+ return set_meta(key, value)
+
+ monkeypatch.setattr(db, "set_meta", flaky_set_meta)
+
+ first = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+ retry = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+
+ assert first["error"] == "injected marker failure"
+ assert retry["pruned"] == 0
+ assert db.get_session("recoverable")["end_reason"] == "startup_orphan_reap"
+
+ def test_concurrent_auto_maintenance_preserves_the_recovery_window(
+ self, db, monkeypatch
+ ):
+ stale = time.time() - 100 * 86400
+ _make_session(db, "concurrent", source="cli", started_at=stale, message_at=stale)
+ _set_last_activity(db, "concurrent", stale)
+ peer = SessionDB(db.db_path)
+ read_barrier = threading.Barrier(2)
+ second_done = threading.Event()
+ release_first_prune = threading.Event()
+ errors = []
+ results = {}
+
+ for instance in (db, peer):
+ get_meta = instance.get_meta
+
+ def synchronized_get_meta(key, *, _get_meta=get_meta):
+ value = _get_meta(key)
+ if key == "last_auto_prune":
+ try:
+ read_barrier.wait(timeout=1)
+ except threading.BrokenBarrierError:
+ pass
+ return value
+
+ monkeypatch.setattr(instance, "get_meta", synchronized_get_meta)
+
+ prune_sessions = db.prune_sessions
+
+ def delayed_prune(*args, **kwargs):
+ assert release_first_prune.wait(timeout=5)
+ return prune_sessions(*args, **kwargs)
+
+ monkeypatch.setattr(db, "prune_sessions", delayed_prune)
+
+ def run(name, instance, *, done=None):
+ try:
+ results[name] = instance.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=24,
+ vacuum=False,
+ )
+ except BaseException as exc: # pragma: no cover - asserted below
+ errors.append(exc)
+ finally:
+ if done is not None:
+ done.set()
+
+ first = threading.Thread(target=run, args=("first", db))
+ second = threading.Thread(
+ target=run, args=("second", peer), kwargs={"done": second_done}
+ )
+ try:
+ first.start()
+ second.start()
+ assert second_done.wait(timeout=5)
+ release_first_prune.set()
+ finally:
+ release_first_prune.set()
+ first.join(timeout=5)
+ second.join(timeout=5)
+ peer.close()
+
+ assert not first.is_alive()
+ assert not second.is_alive()
+ assert errors == []
+ assert sum(bool(result["skipped"]) for result in results.values()) == 1
+ assert sum(int(result["pruned"]) for result in results.values()) == 0
+ assert db.get_session("concurrent")["end_reason"] == "startup_orphan_reap"
+
+ def test_auto_prune_spares_compression_root_of_live_turn(self, db):
+ stale = time.time() - 100 * 86400
+ _make_session(db, "root", source="cli", started_at=stale, message_at=stale)
+ db.end_session("root", "compression")
+ db.create_session("tip", source="cli", parent_session_id="root")
+ db.append_message("tip", role="user", content="continued")
+ _set_message_timestamps(db, "tip", stale)
+ _backdate_session(db, "tip", stale)
+ _set_last_activity(db, "tip", stale)
+ assert db.try_acquire_session_turn_lease(
+ "tip", "external-turn", ttl_seconds=300
+ )
+
+ result = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+
+ assert result["pruned"] == 0
+ assert db.get_session("root") is not None
+ assert db.get_session("tip")["ended_at"] is None
+
+ def test_auto_prune_spares_prior_sweep_row_with_new_turn_lease(self, db):
+ stale = time.time() - 100 * 86400
+ _make_session(db, "racy", source="cli", started_at=stale, message_at=stale)
+ _set_last_activity(db, "racy", stale)
+ db.end_session("racy", "startup_orphan_reap")
+ assert db.try_acquire_session_turn_lease(
+ "racy", "arriving-turn", ttl_seconds=300
+ )
+
+ result = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+
+ assert result["pruned"] == 0
+ assert db.get_session("racy") is not None
+
+ def test_auto_prune_spares_prior_sweep_row_with_new_compression_lock(self, db):
+ stale = time.time() - 100 * 86400
+ _make_session(
+ db,
+ "racy-compression",
+ source="cli",
+ started_at=stale,
+ message_at=stale,
+ )
+ _set_last_activity(db, "racy-compression", stale)
+ db.end_session("racy-compression", "startup_orphan_reap")
+ assert db.try_acquire_compression_lock(
+ "racy-compression", "arriving-compressor", ttl_seconds=300
+ )
+
+ result = db.maybe_auto_prune_and_vacuum(
+ retention_days=90,
+ min_interval_hours=0,
+ vacuum=False,
+ )
+
+ assert result["pruned"] == 0
+ assert db.get_session("racy-compression") is not None
+
def test_returns_empty_on_empty_db(self, db):
assert db.sweep_orphaned_sessions(max_idle_seconds=IDLE_S) == []
+ def test_auto_prune_reports_closed_count_and_deletes_after_second_window(
+ self, db
+ ):
+ """#54189 end-to-end: leaky producers (cron/kanban/subagent) never set
+ ``ended_at``; pass 1 closes them (reported via ``closed``), pass 2 —
+ after a further retention window — deletes them, and a messaging row
+ is never touched by either pass."""
+ stale = time.time() - 200 * 86400
+ for sid, source in (
+ ("cron-0", "cron"),
+ ("kanban-1", "kanban"),
+ ("subagent-2", "subagent"),
+ ("telegram-3", "telegram"),
+ ):
+ _make_session(db, sid, source=source, started_at=stale, message_at=stale)
+ _set_last_activity(db, sid, stale)
+
+ first = db.maybe_auto_prune_and_vacuum(
+ retention_days=90, min_interval_hours=0, vacuum=False
+ )
+ assert first["closed"] == 3
+ assert first["pruned"] == 0
+ for sid in ("cron-0", "kanban-1", "subagent-2"):
+ assert db.get_session(sid)["end_reason"] == "startup_orphan_reap"
+ assert db.get_session("telegram-3")["ended_at"] is None
+
+ # Simulate the next maintenance pass after another retention window.
+ db._conn.execute(
+ "UPDATE sessions SET ended_at = ended_at - 91 * 86400 "
+ "WHERE end_reason = 'startup_orphan_reap'"
+ )
+ db._conn.commit()
+ second = db.maybe_auto_prune_and_vacuum(
+ retention_days=90, min_interval_hours=0, vacuum=False
+ )
+ assert second["closed"] == 0
+ assert second["pruned"] == 3
+ remaining = [r["id"] for r in db._conn.execute("SELECT id FROM sessions")]
+ assert remaining == ["telegram-3"]
+
def test_zero_ttl_is_noop(self, db):
stale = time.time() - 8 * 3600
_make_session(db, "stale-tui", source="tui", started_at=stale, message_at=stale)
diff --git a/tests/plugins/dashboard_auth/test_opaque_bearer_not_unreachable.py b/tests/plugins/dashboard_auth/test_opaque_bearer_not_unreachable.py
new file mode 100644
index 0000000000..8b62421484
--- /dev/null
+++ b/tests/plugins/dashboard_auth/test_opaque_bearer_not_unreachable.py
@@ -0,0 +1,156 @@
+"""#94558 — a non-JWT bearer must not be reported as "Auth provider unreachable".
+
+Hosted agents answered every opaque/peer bearer on the gated API with a fast
+HTTP 503 ``{"detail": "Auth provider 'nous' unreachable"}`` while Portal was
+perfectly healthy: ``NousDashboardAuthProvider._verify_jwt`` folded *every*
+``PyJWKClient`` failure — including ``DecodeError('Not enough segments')`` for
+a token that is not a JWT at all — into ``ProviderError``. Only a transport
+failure fetching the JWKS is "unreachable"; anything else means "not my
+token" (``verify_session`` -> None -> 401 / next provider).
+
+Real ``NousDashboardAuthProvider`` + real ``SelfHostedOIDCProvider`` JWKS path,
+a real local HTTP JWKS server (reachable case) or a closed port (unreachable),
+and the real gated web_server app for the HTTP-level assertion.
+"""
+from __future__ import annotations
+
+import json
+import threading
+from http.server import BaseHTTPRequestHandler, HTTPServer
+
+import jwt
+import pytest
+from starlette.testclient import TestClient
+
+from hermes_cli import web_server
+from hermes_cli.dashboard_auth import (
+ InvalidCodeError,
+ ProviderError,
+ classify_jwks_lookup_error,
+ clear_providers,
+ register_provider,
+)
+from hermes_cli.dashboard_auth.cookies import SESSION_AT_COOKIE
+import plugins.dashboard_auth.nous as nous_plugin
+
+OPAQUE_PEER_KEY = "hk_live_opaque_peer_key_0123456789abcdef"
+# Well-formed RS256 JWT header with an unknown kid, bogus payload/signature.
+FOREIGN_KID_JWT = "eyJhbGciOiJSUzI1NiIsImtpZCI6Inp6eiJ9.e30.sig"
+
+
+@pytest.fixture(scope="module")
+def empty_jwks_server():
+ """A reachable JWKS endpoint that knows no keys."""
+
+ class _H(BaseHTTPRequestHandler):
+ def do_GET(self): # noqa: N802
+ self.send_response(200)
+ self.send_header("content-type", "application/json")
+ self.end_headers()
+ self.wfile.write(json.dumps({"keys": []}).encode())
+
+ def log_message(self, *a): # silence
+ pass
+
+ srv = HTTPServer(("127.0.0.1", 0), _H)
+ t = threading.Thread(target=srv.serve_forever, daemon=True)
+ t.start()
+ yield f"http://127.0.0.1:{srv.server_address[1]}"
+ srv.shutdown()
+
+
+def _nous(portal_url: str) -> nous_plugin.NousDashboardAuthProvider:
+ return nous_plugin.NousDashboardAuthProvider(client_id="agent:test-instance", portal_url=portal_url)
+
+
+# ── classifier ────────────────────────────────────────────────────────────
+
+def test_classifier_maps_transport_failure_to_provider_error():
+ exc = jwt.PyJWKClientConnectionError("Fail to fetch data from the url")
+ assert isinstance(classify_jwks_lookup_error(exc), ProviderError)
+
+
+@pytest.mark.parametrize(
+ "exc",
+ [
+ jwt.DecodeError("Not enough segments"),
+ jwt.PyJWKSetError("The JWK Set did not contain any keys"),
+ jwt.InvalidTokenError("bad"),
+ ],
+)
+def test_classifier_maps_unverifiable_token_to_invalid_code(exc):
+ assert isinstance(classify_jwks_lookup_error(exc), InvalidCodeError)
+
+
+def test_classifier_keeps_bare_jwk_client_error_as_provider_fault():
+ assert isinstance(classify_jwks_lookup_error(jwt.PyJWKClientError("weird JWKS shape")), ProviderError)
+
+
+# ── Nous provider ─────────────────────────────────────────────────────────
+
+def test_opaque_bearer_with_healthy_portal_is_not_unreachable(empty_jwks_server):
+ provider = _nous(empty_jwks_server)
+ assert provider.verify_session(access_token=OPAQUE_PEER_KEY) is None
+
+
+def test_foreign_kid_jwt_with_healthy_portal_is_not_unreachable(empty_jwks_server):
+ provider = _nous(empty_jwks_server)
+ assert provider.verify_session(access_token=FOREIGN_KID_JWT) is None
+
+
+def test_real_jwt_with_unreachable_portal_still_raises_provider_error():
+ provider = _nous("http://127.0.0.1:9") # discard port: connection refused
+ with pytest.raises(ProviderError):
+ provider.verify_session(access_token=FOREIGN_KID_JWT)
+
+
+def test_opaque_bearer_with_unreachable_portal_is_still_just_not_ours():
+ """No network call is even needed to know an opaque string is not our JWT."""
+ provider = _nous("http://127.0.0.1:9")
+ assert provider.verify_session(access_token=OPAQUE_PEER_KEY) is None
+
+
+# ── self-hosted OIDC provider (sibling site of the same hunk) ──────────────
+
+def test_self_hosted_provider_shares_the_classification(empty_jwks_server, monkeypatch):
+ import plugins.dashboard_auth.self_hosted as sh
+
+ provider = object.__new__(sh.SelfHostedOIDCProvider)
+ provider._jwks_client = None
+ provider._client_id = "hermes"
+ monkeypatch.setattr(
+ provider, "_get_discovery",
+ lambda: {"jwks_uri": f"{empty_jwks_server}/jwks", "issuer": empty_jwks_server},
+ )
+ with pytest.raises(InvalidCodeError):
+ provider._verify_id_token(OPAQUE_PEER_KEY)
+
+
+# ── HTTP level: the gated API answers 401, not 503 ────────────────────────
+
+@pytest.fixture
+def _gated_nous(empty_jwks_server):
+ clear_providers()
+ prev = {k: getattr(web_server.app.state, k, None) for k in ("bound_host", "bound_port", "auth_required")}
+ web_server.app.state.bound_host = "agent.example.test"
+ web_server.app.state.bound_port = 443
+ web_server.app.state.auth_required = True
+ register_provider(_nous(empty_jwks_server))
+ yield TestClient(web_server.app, base_url="https://agent.example.test")
+ clear_providers()
+ for k, v in prev.items():
+ setattr(web_server.app.state, k, v)
+
+
+def test_gated_api_rejects_opaque_bearer_with_401_not_503(_gated_nous):
+ r = _gated_nous.get("/api/auth/me", headers={"Authorization": f"Bearer {OPAQUE_PEER_KEY}"})
+ assert r.status_code != 503, r.text
+ assert r.status_code == 401
+ assert "unreachable" not in r.text.lower()
+
+
+def test_gated_api_rejects_opaque_cookie_with_401_not_503(_gated_nous):
+ _gated_nous.cookies.set(SESSION_AT_COOKIE, OPAQUE_PEER_KEY)
+ r = _gated_nous.get("/api/auth/me")
+ assert r.status_code != 503, r.text
+ assert "unreachable" not in r.text.lower()
diff --git a/tests/plugins/image_gen/test_meta_ai_provider.py b/tests/plugins/image_gen/test_meta_ai_provider.py
new file mode 100644
index 0000000000..3ff129a539
--- /dev/null
+++ b/tests/plugins/image_gen/test_meta_ai_provider.py
@@ -0,0 +1,316 @@
+"""Tests for the bundled Meta Model API image_gen plugin (muse-image)."""
+
+from __future__ import annotations
+
+import importlib
+from pathlib import Path
+from types import SimpleNamespace
+from unittest.mock import MagicMock, patch
+
+import pytest
+
+# The plugin directory uses a hyphen, which is not a valid Python identifier
+# for the dotted-import form. Load it via importlib so tests don't need to
+# touch sys.path or rename the directory.
+meta_plugin = importlib.import_module("plugins.image_gen.meta-ai")
+
+
+# 1×1 transparent PNG — valid bytes for save_b64_image()
+_PNG_HEX = (
+ "89504e470d0a1a0a0000000d49484452000000010000000108060000001f15c4"
+ "890000000d49444154789c6300010000000500010d0a2db40000000049454e44"
+ "ae426082"
+)
+
+
+def _b64_png() -> str:
+ import base64
+
+ return base64.b64encode(bytes.fromhex(_PNG_HEX)).decode()
+
+
+def _fake_response(*, b64=None, url=None, revised_prompt=None):
+ item = SimpleNamespace(b64_json=b64, url=url, revised_prompt=revised_prompt)
+ return SimpleNamespace(data=[item])
+
+
+@pytest.fixture(autouse=True)
+def _tmp_hermes_home(tmp_path, monkeypatch):
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ # Clear every auth + override env var so tests start from a clean slate.
+ for env in (
+ "MODEL_API_KEY",
+ "META_API_KEY",
+ "META_MODEL_API_KEY",
+ "META_BASE_URL",
+ "META_IMAGE_MODEL",
+ ):
+ monkeypatch.delenv(env, raising=False)
+ yield tmp_path
+
+
+@pytest.fixture
+def provider(monkeypatch):
+ monkeypatch.setenv("META_MODEL_API_KEY", "test-key")
+ return meta_plugin.MetaImageGenProvider()
+
+
+def _patched_openai(fake_client: MagicMock):
+ fake_openai = MagicMock()
+ fake_openai.OpenAI.return_value = fake_client
+ return patch.dict("sys.modules", {"openai": fake_openai})
+
+
+# ── Metadata ────────────────────────────────────────────────────────────────
+
+
+class TestMetadata:
+ def test_name(self, provider):
+ assert provider.name == "meta-ai"
+
+ def test_display_name(self, provider):
+ assert provider.display_name == "Meta Model API"
+
+ def test_default_model(self, provider):
+ assert provider.default_model() == "muse-image-1.0"
+
+ def test_list_models(self, provider):
+ ids = [m["id"] for m in provider.list_models()]
+ assert ids == ["muse-image-1.0"]
+
+ def test_catalog_entries_have_display_speed_strengths_price(self, provider):
+ for entry in provider.list_models():
+ assert entry["display"]
+ assert entry["speed"]
+ assert entry["strengths"]
+ assert entry["price"]
+
+ def test_text_only_capabilities(self, provider):
+ caps = provider.capabilities()
+ assert caps["modalities"] == ["text"]
+ assert caps["max_reference_images"] == 0
+
+
+# ── Availability ────────────────────────────────────────────────────────────
+
+
+class TestAvailability:
+ def test_no_api_key_unavailable(self):
+ assert meta_plugin.MetaImageGenProvider().is_available() is False
+
+ @pytest.mark.parametrize(
+ "env", ["MODEL_API_KEY", "META_API_KEY", "META_MODEL_API_KEY"]
+ )
+ def test_each_auth_alias_makes_available(self, monkeypatch, env):
+ monkeypatch.setenv(env, "test")
+ assert meta_plugin.MetaImageGenProvider().is_available() is True
+
+
+# ── Auth / base-url resolution ────────────────────────────────────────────────
+
+
+class TestResolution:
+ def test_api_key_priority_order(self, monkeypatch):
+ # MODEL_API_KEY wins over the aliases.
+ monkeypatch.setenv("META_MODEL_API_KEY", "third")
+ monkeypatch.setenv("META_API_KEY", "second")
+ monkeypatch.setenv("MODEL_API_KEY", "first")
+ assert meta_plugin._resolve_api_key() == "first"
+
+ def test_default_base_url(self):
+ assert meta_plugin._resolve_base_url() == "https://api.meta.ai/v1"
+
+ def test_base_url_override(self, monkeypatch):
+ monkeypatch.setenv("META_BASE_URL", "https://proxy.internal/v1")
+ assert meta_plugin._resolve_base_url() == "https://proxy.internal/v1"
+
+
+# ── Model resolution ──────────────────────────────────────────────────────────
+
+
+class TestModelResolution:
+ def test_default(self):
+ model_id, _meta = meta_plugin._resolve_model()
+ assert model_id == "muse-image-1.0"
+
+ def test_env_var_override_ignores_unknown(self, monkeypatch):
+ monkeypatch.setenv("META_IMAGE_MODEL", "not-a-real-model")
+ model_id, _meta = meta_plugin._resolve_model()
+ # Unknown id is ignored; falls through to the default.
+ assert model_id == "muse-image-1.0"
+
+ def test_caller_model_kwarg_wins(self, monkeypatch):
+ # The dispatcher forwards top-level image_gen.model as the `model`
+ # kwarg; it must beat the env override (#55893 bug class).
+ monkeypatch.setitem(
+ meta_plugin._MODELS,
+ "muse-image-test",
+ dict(meta_plugin._MODELS["muse-image-1.0"]),
+ )
+ monkeypatch.setenv("META_IMAGE_MODEL", "muse-image-1.0")
+ model_id, _meta = meta_plugin._resolve_model("muse-image-test")
+ assert model_id == "muse-image-test"
+
+ def test_caller_model_unknown_falls_through(self):
+ model_id, _meta = meta_plugin._resolve_model("not-a-real-model")
+ assert model_id == "muse-image-1.0"
+
+
+# ── Generate ──────────────────────────────────────────────────────────────────
+
+
+class TestGenerate:
+ def test_model_kwarg_reaches_payload(self, provider, monkeypatch):
+ monkeypatch.setitem(
+ meta_plugin._MODELS,
+ "muse-image-test",
+ dict(meta_plugin._MODELS["muse-image-1.0"]),
+ )
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(b64=_b64_png())
+ with _patched_openai(fake_client):
+ result = provider.generate("a cat", model="muse-image-test")
+ assert result["success"] is True
+ assert (
+ fake_client.images.generate.call_args.kwargs["model"] == "muse-image-test"
+ )
+
+ def test_badge_is_standard_paid(self, provider):
+ assert provider.get_setup_schema()["badge"] == "paid"
+
+ def test_empty_prompt_rejected(self, provider):
+ result = provider.generate("", aspect_ratio="square")
+ assert result["success"] is False
+ assert result["error_type"] == "invalid_argument"
+ assert result["provider"] == "meta-ai"
+
+ def test_missing_api_key(self):
+ result = meta_plugin.MetaImageGenProvider().generate("a cat")
+ assert result["success"] is False
+ assert result["error_type"] == "auth_required"
+
+ def test_b64_saves_to_cache(self, provider, tmp_path):
+ png_bytes = bytes.fromhex(_PNG_HEX)
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(b64=_b64_png())
+
+ with _patched_openai(fake_client):
+ result = provider.generate("a cat", aspect_ratio="landscape")
+
+ assert result["success"] is True
+ assert result["model"] == "muse-image-1.0"
+ assert result["aspect_ratio"] == "landscape"
+ assert result["provider"] == "meta-ai"
+ assert result["modality"] == "text"
+
+ saved = Path(result["image"])
+ assert saved.exists()
+ assert saved.parent == tmp_path / "cache" / "images"
+ assert saved.read_bytes() == png_bytes
+
+ call_kwargs = fake_client.images.generate.call_args.kwargs
+ assert call_kwargs["model"] == "muse-image-1.0"
+ assert call_kwargs["size"] == "1536x1024"
+ assert call_kwargs["n"] == 1
+
+ def test_client_uses_meta_base_url(self, provider):
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(b64=_b64_png())
+ fake_openai = MagicMock()
+ fake_openai.OpenAI.return_value = fake_client
+
+ with patch.dict("sys.modules", {"openai": fake_openai}):
+ provider.generate("a cat")
+
+ assert (
+ fake_openai.OpenAI.call_args.kwargs["base_url"] == "https://api.meta.ai/v1"
+ )
+
+ def test_base_url_override_reaches_client(self, provider, monkeypatch):
+ monkeypatch.setenv("META_BASE_URL", "https://proxy.internal/v1")
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(b64=_b64_png())
+ fake_openai = MagicMock()
+ fake_openai.OpenAI.return_value = fake_client
+
+ with patch.dict("sys.modules", {"openai": fake_openai}):
+ provider.generate("a cat")
+
+ assert (
+ fake_openai.OpenAI.call_args.kwargs["base_url"]
+ == "https://proxy.internal/v1"
+ )
+
+ @pytest.mark.parametrize(
+ "aspect,expected_size",
+ [
+ ("landscape", "1536x1024"),
+ ("square", "1024x1024"),
+ ("portrait", "1024x1536"),
+ ],
+ )
+ def test_aspect_ratio_mapping(self, provider, aspect, expected_size):
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(b64=_b64_png())
+
+ with _patched_openai(fake_client):
+ provider.generate("a cat", aspect_ratio=aspect)
+
+ assert fake_client.images.generate.call_args.kwargs["size"] == expected_size
+
+ def test_revised_prompt_passed_through(self, provider):
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(
+ b64=_b64_png(),
+ revised_prompt="A photo of a cat",
+ )
+
+ with _patched_openai(fake_client):
+ result = provider.generate("a cat")
+
+ assert result["revised_prompt"] == "A photo of a cat"
+
+ def test_url_response_is_cached_locally(self, provider):
+ """A URL response is materialized locally (symmetric to the openai/xai
+ providers) so ephemeral signed URLs can't expire mid-flight."""
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(
+ b64=None,
+ url="https://example.com/img.webp",
+ )
+
+ with (
+ _patched_openai(fake_client),
+ patch.object(
+ meta_plugin,
+ "save_url_image",
+ return_value=Path("/tmp/meta_20260524_000000_deadbeef.webp"),
+ ) as mock_save_url,
+ ):
+ result = provider.generate("a cat")
+
+ assert result["success"] is True
+ assert result["image"].startswith("/")
+ assert "example.com" not in result["image"]
+ mock_save_url.assert_called_once()
+
+ def test_empty_response_errors(self, provider):
+ fake_client = MagicMock()
+ fake_client.images.generate.return_value = _fake_response(b64=None, url=None)
+
+ with _patched_openai(fake_client):
+ result = provider.generate("a cat")
+
+ assert result["success"] is False
+ assert result["error_type"] == "empty_response"
+
+ def test_api_error_surfaced(self, provider):
+ fake_client = MagicMock()
+ fake_client.images.generate.side_effect = RuntimeError("boom")
+
+ with _patched_openai(fake_client):
+ result = provider.generate("a cat")
+
+ assert result["success"] is False
+ assert result["error_type"] == "api_error"
+ assert "boom" in result["error"]
diff --git a/tests/plugins/memory/test_hindsight_provider.py b/tests/plugins/memory/test_hindsight_provider.py
index 475e3adb37..b48e2eb383 100644
--- a/tests/plugins/memory/test_hindsight_provider.py
+++ b/tests/plugins/memory/test_hindsight_provider.py
@@ -10,6 +10,7 @@ import os
import re
import stat
import sys
+import threading
import time
from datetime import datetime
from pathlib import Path
@@ -32,6 +33,7 @@ from plugins.memory.hindsight import (
_normalize_retain_tags,
_resolve_bank_id_template,
_sanitize_bank_segment,
+ _WRITER_SENTINEL,
)
@@ -1643,3 +1645,68 @@ class TestClientAutoUpgradeRoutesThroughLazyDeps:
assert len(calls) == 1 # attempted exactly once, init still completed
assert any("runtime installs are disabled" in r.getMessage()
for r in caplog.records)
+
+
+
+class TestMultiplexBackgroundScope:
+ """Under multiplex_profiles get_secret fails closed on an unscoped thread;
+ the writer / daemon-start threads are spawned from a scoped context and
+ must carry it along (#92608, #94933)."""
+
+ @pytest.fixture()
+ def scoped_embedded(self, tmp_path, monkeypatch):
+ from agent.secret_scope import (
+ build_profile_secret_scope, reset_secret_scope, set_multiplex_active, set_secret_scope,
+ )
+ from hermes_constants import reset_hermes_home_override, set_hermes_home_override
+
+ created = []
+
+ class FakeHindsightEmbedded:
+ def __init__(self, **kwargs):
+ created.append(kwargs["llm_api_key"])
+ self._manager = SimpleNamespace(is_running=lambda profile: False, stop=lambda profile: None)
+ self._ensure_started = lambda: None
+
+ dem = SimpleNamespace(console=None)
+ monkeypatch.setitem(sys.modules, "hindsight", SimpleNamespace(HindsightEmbedded=FakeHindsightEmbedded))
+ monkeypatch.setitem(sys.modules, "hindsight_embed", SimpleNamespace(daemon_embed_manager=dem))
+ monkeypatch.setitem(sys.modules, "hindsight_embed.daemon_embed_manager", dem)
+ monkeypatch.setattr("plugins.memory.hindsight._check_local_runtime", lambda: (True, ""))
+
+ home = tmp_path / "profiles" / "p1"
+ (home / "hindsight").mkdir(parents=True)
+ (home / ".env").write_text("HINDSIGHT_LLM_API_KEY=p1-secret\n")
+ (home / "hindsight" / "config.json").write_text(json.dumps(
+ {"mode": "local_embedded", "llm_provider": "openai", "llm_model": "m", "memory_mode": "hybrid"}
+ ))
+ # Enter the profile scope the way gateway _profile_runtime_scope does.
+ set_multiplex_active(True)
+ monkeypatch.setattr("plugins.memory.hindsight.get_hermes_home", lambda: home)
+ home_tok = set_hermes_home_override(str(home))
+ scope_tok = set_secret_scope(build_profile_secret_scope(home))
+ yield created, home
+ set_multiplex_active(False)
+ reset_secret_scope(scope_tok)
+ reset_hermes_home_override(home_tok)
+
+ def test_writer_thread_resolves_profile_secret(self, scoped_embedded):
+ created, home = scoped_embedded
+ p = HindsightMemoryProvider()
+ p._mode = "local_embedded"
+ p._config = {"profile": "hermes", "llm_provider": "openai", "llm_model": "m"}
+ p._ensure_writer()
+ p._retain_queue.put(p._get_client) # real body: get_secret(HINDSIGHT_LLM_API_KEY)
+ p._retain_queue.put(_WRITER_SENTINEL)
+ p._writer_thread.join(timeout=5)
+ assert created == ["p1-secret"]
+
+ def test_daemon_start_thread_resolves_profile_secret(self, scoped_embedded):
+ created, home = scoped_embedded
+ p = HindsightMemoryProvider()
+ p.initialize(session_id="s1", hermes_home=str(home), platform="cli")
+ for t in threading.enumerate():
+ if t.name == "hindsight-daemon-start":
+ t.join(timeout=5)
+ assert created == ["p1-secret"]
+ assert "Daemon started successfully" in (home / "logs" / "hindsight-embed.log").read_text()
diff --git a/tests/plugins/platforms/photon/test_multiplex_profile_scope.py b/tests/plugins/platforms/photon/test_multiplex_profile_scope.py
new file mode 100644
index 0000000000..5b32327b1e
--- /dev/null
+++ b/tests/plugins/platforms/photon/test_multiplex_profile_scope.py
@@ -0,0 +1,122 @@
+"""Multiplex secondary-profile scope tests for the Photon adapter + auth module.
+
+__init__'s project_id, check_requirements'/validate_config's node_bin/
+project_id, _env_enablement's home_channel, _reactions_enabled's
+PHOTON_REACTIONS, __init__'s require_mention, and _standalone_send's
+sidecar_port, plus auth.py's load_project_credentials/
+load_dashboard_project_id, all previously read raw os.getenv
+unconditionally (only PHOTON_PROJECT_SECRET/PHOTON_SIDECAR_TOKEN were
+already scoped via _get_scoped_secret). Under gateway.multiplex_profiles,
+os.environ holds the DEFAULT profile's YAML-to-env bridge output -- a
+secondary profile with its own (different or absent) Photon config could
+silently authenticate against the default profile's Spectrum project, or
+have its mention-gating/reaction behavior driven by the default profile's
+settings.
+
+Notably project_id was a stronger variant of the bug (like the IRC fix in
+this series): __init__'s original
+`os.getenv("PHOTON_PROJECT_ID") or extra.get("project_id") or stored_id`
+ordering let a raw env read override even an explicitly configured
+config.yaml extra.
+
+Mirrors the LINE/DingTalk/IRC/Mattermost fix for #98738.
+"""
+from __future__ import annotations
+
+import os
+from pathlib import Path
+
+import pytest
+
+from gateway.config import PlatformConfig
+from plugins.platforms.photon import auth as photon_auth
+from plugins.platforms.photon.adapter import PhotonAdapter
+
+_PHOTON_ENV = (
+ "PHOTON_PROJECT_ID",
+ "PHOTON_PROJECT_SECRET",
+ "PHOTON_DASHBOARD_PROJECT_ID",
+ "PHOTON_REQUIRE_MENTION",
+ "PHOTON_REACTIONS",
+ "PHOTON_HOME_CHANNEL",
+ "PHOTON_HOME_CHANNEL_NAME",
+ "PHOTON_SIDECAR_PORT",
+)
+
+
+@pytest.fixture
+def tmp_hermes_home(tmp_path: Path, monkeypatch: pytest.MonkeyPatch):
+ """Isolate from the real ~/.hermes/auth.json fallback in load_project_credentials()."""
+ home = tmp_path / "hermes"
+ home.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(home))
+ for key in _PHOTON_ENV:
+ monkeypatch.delenv(key, raising=False)
+ yield home
+ for key in _PHOTON_ENV:
+ os.environ.pop(key, None)
+
+
+@pytest.fixture
+def multiplex_scope():
+ """Install multiplex + a secondary-profile secret scope; restore after."""
+ tokens = []
+
+ def install(scope=None):
+ from agent.secret_scope import set_multiplex_active, set_secret_scope
+
+ set_multiplex_active(True)
+ tokens.append(set_secret_scope(scope or {}))
+ return tokens[-1]
+
+ yield install
+
+ from agent.secret_scope import reset_secret_scope, set_multiplex_active
+
+ for token in reversed(tokens):
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+
+@pytest.fixture
+def default_profile_env(monkeypatch):
+ """The default profile's YAML-to-env bridge output in os.environ."""
+ monkeypatch.setenv("PHOTON_PROJECT_ID", "default-project-id")
+ monkeypatch.setenv("PHOTON_PROJECT_SECRET", "default-project-secret")
+ monkeypatch.setenv("PHOTON_REQUIRE_MENTION", "true")
+ monkeypatch.setenv("PHOTON_REACTIONS", "true")
+
+
+class TestAuthMultiplexProfileScope:
+ """load_project_credentials / load_dashboard_project_id (auth.py)."""
+
+ def test_scoped_miss_does_not_leak_default_project_id(
+ self, tmp_hermes_home, multiplex_scope, default_profile_env
+ ):
+ multiplex_scope({"SOMETHING_ELSE": "x"})
+ sid, secret = photon_auth.load_project_credentials()
+ assert sid is None
+ assert secret is None
+ adapter = PhotonAdapter(PlatformConfig(enabled=True, extra={}))
+ assert adapter._project_id == ""
+ assert adapter.require_mention is False
+ assert adapter._reactions_enabled() is False
+
+class TestAdapterMultiplexProfileScope:
+ """PhotonAdapter.__init__ / _env_enablement / _reactions_enabled (adapter.py)."""
+
+ def test_secondary_extra_wins_over_default_profile_env(
+ self, tmp_hermes_home, multiplex_scope, default_profile_env
+ ):
+ """A secondary profile's own config.yaml extra project_id must be
+ authoritative -- not the default profile's bridged env value. The
+ pre-fix ordering (raw os.getenv checked BEFORE extra) meant even an
+ explicit extra config was silently overridden."""
+ multiplex_scope({"PHOTON_PROJECT_SECRET": "profile-secret"})
+ cfg = PlatformConfig(
+ enabled=True,
+ extra={"project_id": "profile-project-id"},
+ )
+ adapter = PhotonAdapter(cfg)
+ assert adapter._project_id == "profile-project-id"
+
diff --git a/tests/plugins/test_a2a_plugin.py b/tests/plugins/test_a2a_plugin.py
index 346284b277..e5d2668884 100644
--- a/tests/plugins/test_a2a_plugin.py
+++ b/tests/plugins/test_a2a_plugin.py
@@ -910,7 +910,9 @@ def _make_live_adapter(monkeypatch, reply_fn=None):
port = _free_port()
monkeypatch.setenv("A2A_PORT", str(port))
- adapter = A2AAdapter(PlatformConfig(enabled=True))
+ # A scoped secondary profile ignores the process env (#100382); pass the
+ # port through config.extra so both construction paths bind the same port.
+ adapter = A2AAdapter(PlatformConfig(enabled=True, extra={"port": port}))
async def fake_handle_message(event):
if reply_fn is None:
@@ -1223,6 +1225,54 @@ class TestInboundRoundTrip:
asyncio.run(run())
+ def test_multiplex_adapter_keeps_profile_scoped_peer_tokens(self, monkeypatch):
+ """A secondary listener must not authenticate with the default profile's tokens."""
+ from agent.secret_scope import (
+ reset_secret_scope,
+ set_multiplex_active,
+ set_secret_scope,
+ )
+
+ monkeypatch.setenv("A2A_PEER_TOKENS", "default:default-token")
+ monkeypatch.delenv("A2A_BEARER_TOKEN", raising=False)
+ monkeypatch.setenv("A2A_HOST", "127.0.0.1")
+
+ set_multiplex_active(True)
+ scope_token = set_secret_scope(
+ {"A2A_PEER_TOKENS": "secondary:secondary-token"}
+ )
+ try:
+ adapter, base = _make_live_adapter(monkeypatch)
+ finally:
+ reset_secret_scope(scope_token)
+
+ async def run():
+ try:
+ assert await adapter.connect() is True
+ response = await asyncio.to_thread(
+ _post_json,
+ base + "/",
+ _send_body("profile-scoped auth"),
+ {"Authorization": "Bearer secondary-token"},
+ )
+ assert response["result"]["status"]["state"] == "TASK_STATE_COMPLETED"
+
+ with pytest.raises(urllib.error.HTTPError) as exc_info:
+ await asyncio.to_thread(
+ _post_json,
+ base + "/",
+ _send_body("wrong profile"),
+ {"Authorization": "Bearer default-token"},
+ )
+ assert exc_info.value.code == 401
+ finally:
+ await adapter.disconnect()
+
+ try:
+ asyncio.run(run())
+ finally:
+ set_multiplex_active(False)
+
# --------------------------------------------------------------------------
# Push notifications end-to-end (inline config in message/send)
@@ -1618,3 +1668,101 @@ print('fake reply')
title = con.execute("SELECT title FROM sessions WHERE id='sess-1'").fetchone()[0]
con.close()
assert title == "a2a-dev-ctx-unsafe-value"
+
+
+# --------------------------------------------------------------------------
+# Multiplex secondary-profile scope (construction-time config leak)
+# --------------------------------------------------------------------------
+#
+# __init__'s port/advertised-toolsets reads and _load_served_agents's
+# description default all previously read raw A2A_* env vars unconditionally.
+# Under a multiplexed secondary profile, os.environ holds the DEFAULT
+# profile's YAML-to-env bridge output — a secondary profile with its own
+# (different, or absent) A2A config would silently borrow the default
+# profile's port, toolset advertisement, agent name, or Agent Card
+# description. Mirrors the Buzz/SimpleX fix for #98738.
+
+_A2A_ENV_VARS = (
+ "A2A_PORT",
+ "A2A_AGENT_NAME",
+ "A2A_ADVERTISED_TOOLSETS",
+ "A2A_AGENT_DESCRIPTION",
+)
+
+
+@pytest.fixture(autouse=True)
+def _clean_a2a_construction_env(monkeypatch):
+ """Keep the new multiplex tests hermetic regardless of ambient env."""
+ for var in _A2A_ENV_VARS:
+ monkeypatch.delenv(var, raising=False)
+ yield
+
+
+@pytest.fixture
+def multiplex_scope():
+ """Install multiplex + a secondary-profile secret scope; restore after."""
+ tokens = []
+
+ def install(scope=None):
+ from agent.secret_scope import set_multiplex_active, set_secret_scope
+
+ set_multiplex_active(True)
+ tokens.append(set_secret_scope(scope or {}))
+ return tokens[-1]
+
+ yield install
+
+ from agent.secret_scope import reset_secret_scope, set_multiplex_active
+
+ for token in reversed(tokens):
+ reset_secret_scope(token)
+ set_multiplex_active(False)
+
+
+@pytest.fixture
+def default_profile_env(monkeypatch):
+ """The default profile's YAML-to-env bridge output in os.environ."""
+ monkeypatch.setenv("A2A_PORT", "9111")
+ monkeypatch.setenv("A2A_AGENT_NAME", "default-profile-agent")
+ monkeypatch.setenv("A2A_ADVERTISED_TOOLSETS", "default-only-toolset")
+ monkeypatch.setenv("A2A_AGENT_DESCRIPTION", "Default profile's own agent.")
+
+
+class TestMultiplexConstructionScope:
+
+ def test_secondary_profile_never_borrows_default_profile_env(
+ self, multiplex_scope, default_profile_env
+ ):
+ """The secondary profile's own config is authoritative; keys absent
+ from it fall to the module defaults, never to the default profile's
+ bridged A2A_* env values."""
+ from plugins.platforms.a2a.adapter import A2AAdapter, _DEFAULT_PORT
+ from gateway.config import PlatformConfig
+
+ multiplex_scope()
+ assert A2AAdapter(PlatformConfig(enabled=True, extra={"port": 9222})).port == 9222
+
+ adapter = A2AAdapter(PlatformConfig(enabled=True, extra={}))
+ assert adapter.port == _DEFAULT_PORT
+ assert adapter.agent_name != "default-profile-agent"
+ assert adapter._agents[""]["description"] == (
+ "Hermes Agent — a general-purpose agent reachable over A2A."
+ )
+
+ def test_default_profile_unscoped_keeps_env_precedence(
+ self, monkeypatch, default_profile_env
+ ):
+ """Multiplex ON but no scope (the DEFAULT profile constructs
+ unscoped): env is its own bridge output and still wins."""
+ from agent.secret_scope import set_multiplex_active
+ from plugins.platforms.a2a.adapter import A2AAdapter
+ from gateway.config import PlatformConfig
+
+ set_multiplex_active(True)
+ try:
+ adapter = A2AAdapter(PlatformConfig(enabled=True, extra={}))
+ finally:
+ set_multiplex_active(False)
+ assert adapter.port == 9111
+ assert adapter.agent_name == "default-profile-agent"
+ assert adapter._agents[""]["description"] == "Default profile's own agent."
diff --git a/tests/plugins/test_a2a_schema_registration.py b/tests/plugins/test_a2a_schema_registration.py
index 6068b62fc7..76f9a3819d 100644
--- a/tests/plugins/test_a2a_schema_registration.py
+++ b/tests/plugins/test_a2a_schema_registration.py
@@ -35,7 +35,8 @@ def test_a2a_call_schema_round_trips_through_tool_describe(monkeypatch):
monkeypatch.setattr(
tool_search,
"is_deferrable_tool_name",
- lambda name: name == "a2a_call",
+ # #97979 added the defer_tools positional (curated-set override).
+ lambda name, defer_tools=None: name == "a2a_call",
)
described = json.loads(
diff --git a/tests/plugins/web/test_web_search_provider_plugins.py b/tests/plugins/web/test_web_search_provider_plugins.py
index 117733e045..9a0f253147 100644
--- a/tests/plugins/web/test_web_search_provider_plugins.py
+++ b/tests/plugins/web/test_web_search_provider_plugins.py
@@ -3,7 +3,7 @@
Covers:
- All bundled plugins (brave-free, ddgs, searxng, exa, parallel,
- firecrawl, keenable, xai) instantiate and self-report the expected
+ tavily, firecrawl, keenable, xai) instantiate and self-report the expected
capabilities + ABC-derived defaults.
- Each plugin's ``is_available()`` correctly reflects env-var presence.
- The web_search_registry resolves an active provider in the documented
@@ -35,6 +35,8 @@ def _clear_web_env(monkeypatch: pytest.MonkeyPatch) -> None:
"BRAVE_SEARCH_API_KEY",
"SEARXNG_URL",
"KEENABLE_API_KEY",
+ "TAVILY_API_KEY",
+ "TAVILY_BASE_URL",
"EXA_API_KEY",
"PARALLEL_API_KEY",
"PARALLEL_SEARCH_MODE",
@@ -82,6 +84,7 @@ class TestBundledPluginsRegister:
"keenable",
"parallel",
"searxng",
+ "tavily",
"xai",
]
@@ -94,6 +97,7 @@ class TestBundledPluginsRegister:
("exa", True, True),
("parallel", True, True),
("keenable", True, True),
+ ("tavily", True, True),
("firecrawl", True, True),
# xai: search-only via Grok's agentic web_search tool.
("xai", True, False),
@@ -115,7 +119,7 @@ class TestBundledPluginsRegister:
@pytest.mark.parametrize(
"plugin_name",
- ["brave-free", "ddgs", "searxng", "exa", "parallel", "firecrawl", "keenable", "xai"],
+ ["brave-free", "ddgs", "searxng", "exa", "parallel", "tavily", "firecrawl", "keenable", "xai"],
)
def test_each_plugin_has_name_and_display_name(self, plugin_name: str) -> None:
_ensure_plugins_loaded()
@@ -165,6 +169,16 @@ class TestIsAvailable:
monkeypatch.setenv("KEENABLE_API_KEY", "real")
assert p.is_available() is True
+ def test_tavily_requires_api_key(self, monkeypatch: pytest.MonkeyPatch) -> None:
+ _ensure_plugins_loaded()
+ from agent.web_search_registry import get_provider
+
+ p = get_provider("tavily")
+ assert p is not None
+ assert p.is_available() is False
+ monkeypatch.setenv("TAVILY_API_KEY", "real")
+ assert p.is_available() is True
+
def test_exa_requires_api_key(self, monkeypatch: pytest.MonkeyPatch) -> None:
_ensure_plugins_loaded()
from agent.web_search_registry import get_provider
diff --git a/tests/run_agent/test_413_compression.py b/tests/run_agent/test_413_compression.py
index a91a585823..2c99ff03a0 100644
--- a/tests/run_agent/test_413_compression.py
+++ b/tests/run_agent/test_413_compression.py
@@ -979,6 +979,156 @@ class TestPreflightCompression:
assert result["final_response"] == "Recovered after overflow"
assert mock_compress.call_count == 2
+ def test_provider_overflow_rechecks_complete_request_before_retry(self, agent):
+ """Provider-proven overflow bypasses post-compaction estimate deferral.
+
+ The first recovery pass drops message rows but rebuilds a larger
+ request. The compressor then awaits real usage, so the old path sent
+ that oversized request back to llama.cpp, which may silently truncate
+ instead of returning another overflow error. Recovery must run another
+ bounded preflight pass first.
+ """
+ agent.compression_enabled = True
+ agent.max_compression_attempts = 2
+ agent.context_compressor.context_length = 65_536
+ agent.context_compressor.threshold_tokens = 34_078
+
+ overflow = Exception(
+ "request (70000 tokens) exceeds the available context size "
+ "(65536 tokens)"
+ )
+ overflow.status_code = 400
+ agent.client.chat.completions.create.side_effect = [overflow]
+
+ history = [
+ {"role": "user", "content": "earlier question"},
+ {"role": "assistant", "content": "earlier answer"},
+ ]
+ compress_calls = 0
+
+ def _request_pressure(*_args, **_kwargs):
+ if agent.client.chat.completions.create.call_count == 0:
+ return 30_000
+ return 70_000
+
+ def _compress(_messages, *_args, **_kwargs):
+ nonlocal compress_calls
+ compress_calls += 1
+ return (
+ [
+ {"role": "user", "content": f"summary {compress_calls}"},
+ {"role": "assistant", "content": "summary acknowledged"},
+ ],
+ "rebuilt prompt remains oversized",
+ )
+
+ with (
+ patch(
+ "agent.turn_context.estimate_request_tokens_rough",
+ return_value=30_000,
+ ),
+ patch(
+ "agent.conversation_loop._midturn_request_pressure_tokens",
+ side_effect=_request_pressure,
+ ),
+ patch.object(
+ agent.context_compressor,
+ "should_defer_preflight_to_real_usage",
+ return_value=True,
+ ),
+ patch.object(agent, "_compress_context", side_effect=_compress) as mock_compress,
+ patch.object(agent, "_persist_session"),
+ patch.object(agent, "_save_trajectory"),
+ patch.object(agent, "_cleanup_task_resources"),
+ ):
+ result = agent.run_conversation(
+ "continue",
+ conversation_history=history,
+ )
+
+ assert result["completed"] is False
+ assert result["compression_exhausted"] is True
+ assert mock_compress.call_count == 2
+ assert agent.client.chat.completions.create.call_count == 1
+
+ def test_long_context_tier_recovery_rechecks_complete_request_before_retry(self, agent):
+ """The Anthropic long-context 429 handler is the same recovery class.
+
+ It compacts and restarts on row count alone, exactly like the generic
+ overflow handler. The rebuilt request must be measured against the
+ (now-reduced) window before the provider is retried, so a compaction
+ that drops rows but stays oversized fails closed instead of being
+ sent again.
+ """
+ agent.compression_enabled = True
+ agent.max_compression_attempts = 2
+ agent.context_compressor.context_length = 1_000_000
+ agent.context_compressor.threshold_tokens = 500_000
+
+ tier_error = Exception(
+ "Extra usage is required for long context requests."
+ )
+ tier_error.status_code = 429
+ agent.client.chat.completions.create.side_effect = [tier_error]
+
+ history = [
+ {"role": "user", "content": "earlier question"},
+ {"role": "assistant", "content": "earlier answer"},
+ ]
+ compress_calls = 0
+
+ def _request_pressure(*_args, **_kwargs):
+ if agent.client.chat.completions.create.call_count == 0:
+ return 30_000
+ return 250_000
+
+ def _compress(_messages, *_args, **_kwargs):
+ nonlocal compress_calls
+ compress_calls += 1
+ return (
+ [
+ {"role": "user", "content": f"summary {compress_calls}"},
+ {"role": "assistant", "content": "summary acknowledged"},
+ ],
+ "rebuilt prompt remains oversized",
+ )
+
+ def _update_model(*, context_length, **_kwargs):
+ agent.context_compressor.context_length = context_length
+ agent.context_compressor.threshold_tokens = context_length // 2
+
+ with (
+ patch(
+ "agent.turn_context.estimate_request_tokens_rough",
+ return_value=30_000,
+ ),
+ patch(
+ "agent.conversation_loop._midturn_request_pressure_tokens",
+ side_effect=_request_pressure,
+ ),
+ patch.object(
+ agent.context_compressor,
+ "should_defer_preflight_to_real_usage",
+ return_value=True,
+ ),
+ patch.object(
+ agent.context_compressor, "update_model", side_effect=_update_model
+ ),
+ patch.object(agent, "_compress_context", side_effect=_compress) as mock_compress,
+ patch.object(agent, "_persist_session"),
+ patch.object(agent, "_save_trajectory"),
+ patch.object(agent, "_cleanup_task_resources"),
+ ):
+ result = agent.run_conversation(
+ "continue",
+ conversation_history=history,
+ )
+
+ assert result["completed"] is False
+ assert result["compression_exhausted"] is True
+ assert mock_compress.call_count == 2
+ assert agent.client.chat.completions.create.call_count == 1
+
def test_interrupt_before_first_provider_call_restores_preflight_display_seed(self, agent):
"""Interrupted turns must not keep a speculative preflight display seed.
diff --git a/tests/run_agent/test_background_review.py b/tests/run_agent/test_background_review.py
index ca2ebea949..7a10ef16e3 100644
--- a/tests/run_agent/test_background_review.py
+++ b/tests/run_agent/test_background_review.py
@@ -461,6 +461,27 @@ def test_background_review_registers_before_start_runs_and_cleans_up(monkeypatch
assert agent._active_children == []
+def test_background_review_snapshot_isolated_from_live_nested_messages():
+ """A review must not mutate the persisted/live transcript through aliases."""
+ original = [{
+ "role": "assistant",
+ "content": [{"type": "text", "text": "answer"}],
+ "tool_calls": [{
+ "id": "call-1",
+ "function": {"name": "read_file", "arguments": '{"path":"x"}'},
+ }],
+ }]
+
+ from agent.turn_finalizer import _clone_background_review_messages
+
+ snapshot = _clone_background_review_messages(original)
+ snapshot[0]["content"][0]["text"] = "review mutation"
+ snapshot[0]["tool_calls"][0]["function"]["arguments"] = "{}"
+
+ assert original[0]["content"][0]["text"] == "answer"
+ assert original[0]["tool_calls"][0]["function"]["arguments"] == '{"path":"x"}'
+
+
def test_live_turn_waits_for_review_exit_before_relay_and_turn_context(monkeypatch):
"""The outer production wrapper waits before same-session instrumentation."""
review_entered = threading.Event()
diff --git a/tests/run_agent/test_direct_contexts_stream_inline.py b/tests/run_agent/test_direct_contexts_stream_inline.py
new file mode 100644
index 0000000000..22720f09f0
--- /dev/null
+++ b/tests/run_agent/test_direct_contexts_stream_inline.py
@@ -0,0 +1,243 @@
+"""Delegated children and cron turns stream on the wire (#90202, #100260).
+
+``should_use_direct_api_call`` contexts (gateway cron turns, delegate_task
+children) must not spawn the interrupt worker — it wedges inside their nested
+thread pools (#62151, #60203). The original fix short-circuited them onto the
+NON-streaming wire, which silently dropped every liveness property streaming
+provides: edge proxies killed the silent POST (z.ai HTTP 524, #90202), and the
+non-stream stale watchdog could not tell a reasoning model's thinking phase
+from a hung provider (#100260 — children died at exactly ``stale_timeout``).
+
+These tests pin the replacement contract: those contexts stay on the streaming
+path, issue ``stream=True`` on the calling thread (no worker), and keep the
+stale detector + cross-thread interrupt abort working from the monitor thread.
+"""
+
+import json
+import threading
+import time
+from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
+from types import SimpleNamespace
+
+import pytest
+
+import run_agent
+from agent import chat_completion_helpers as helpers
+from agent.chat_completion_helpers import (
+ interruptible_streaming_api_call,
+ should_use_direct_api_call,
+)
+
+
+# ---------------------------------------------------------------------------
+# Real OpenAI-wire SSE server: records the wire ``stream`` flag per request.
+# ---------------------------------------------------------------------------
+
+
+class _Wire:
+ def __init__(self, *, stall_after_first_chunk: bool = False):
+ self.requests: list[dict] = []
+ self.stall = stall_after_first_chunk
+ self.hits = threading.Semaphore(0)
+ wire = self
+
+ class Handler(BaseHTTPRequestHandler):
+ def log_message(self, *_a):
+ pass
+
+ def do_POST(self):
+ n = int(self.headers.get("content-length", 0))
+ body = json.loads(self.rfile.read(n) or b"{}")
+ if not self.path.endswith("/chat/completions"):
+ # Local-endpoint capability probes (/api/show etc.) —
+ # answer fast so agent construction never waits on the
+ # stalling stream below.
+ self.send_response(404)
+ self.end_headers()
+ return
+ wire.requests.append(body)
+ wire.hits.release()
+ self.send_response(200)
+ self.send_header("content-type", "text/event-stream")
+ self.end_headers()
+ first = {
+ "id": "c1", "object": "chat.completion.chunk", "created": 1, "model": "m",
+ "choices": [{"index": 0, "delta": {"role": "assistant", "content": "hello"},
+ "finish_reason": None}],
+ }
+ self.wfile.write(f"data: {json.dumps(first)}\n\n".encode())
+ self.wfile.flush()
+ if wire.stall:
+ try:
+ for _ in range(400):
+ time.sleep(0.05)
+ self.wfile.write(b": keepalive\n\n")
+ self.wfile.flush()
+ except Exception:
+ pass
+ return
+ second = {
+ "id": "c1", "object": "chat.completion.chunk", "created": 1, "model": "m",
+ "choices": [{"index": 0, "delta": {"content": " world"}, "finish_reason": None}],
+ }
+ fin = {
+ "id": "c1", "object": "chat.completion.chunk", "created": 1, "model": "m",
+ "choices": [{"index": 0, "delta": {}, "finish_reason": "stop"}],
+ "usage": {"prompt_tokens": 3, "completion_tokens": 2, "total_tokens": 5},
+ }
+ for c in (second, fin):
+ self.wfile.write(f"data: {json.dumps(c)}\n\n".encode())
+ self.wfile.write(b"data: [DONE]\n\n")
+ self.wfile.flush()
+
+ self.server = ThreadingHTTPServer(("127.0.0.1", 0), Handler)
+ threading.Thread(target=self.server.serve_forever, daemon=True).start()
+ self.base_url = f"http://127.0.0.1:{self.server.server_address[1]}/v1"
+
+ def close(self):
+ self.server.shutdown()
+ self.server.server_close()
+
+
+@pytest.fixture
+def wire():
+ w = _Wire()
+ yield w
+ w.close()
+
+
+@pytest.fixture
+def stalling_wire():
+ w = _Wire(stall_after_first_chunk=True)
+ yield w
+ w.close()
+
+
+def _make_agent(base_url: str, *, platform: str):
+ return run_agent.AIAgent(
+ api_key="test-key",
+ base_url=base_url,
+ model="m",
+ provider="custom",
+ platform=platform,
+ quiet_mode=True,
+ skip_context_files=True,
+ skip_memory=True,
+ enabled_toolsets=[],
+ max_iterations=1,
+ )
+
+
+_KW = {"model": "m", "messages": [{"role": "user", "content": "hi"}]}
+
+
+@pytest.mark.parametrize("platform", ["subagent", "cron"])
+def test_direct_contexts_stream_on_the_wire_and_on_the_calling_thread(wire, platform):
+ agent = _make_agent(wire.base_url, platform=platform)
+ assert should_use_direct_api_call(agent) is True
+
+ issued_on = {}
+ real_create = agent._create_request_openai_client
+
+ def spy(*a, **k):
+ issued_on["tid"] = threading.get_ident()
+ return real_create(*a, **k)
+
+ agent._create_request_openai_client = spy
+
+ response = interruptible_streaming_api_call(agent, dict(_KW))
+
+ completions = [r for r in wire.requests if "messages" in r]
+ assert completions, "no chat completion reached the wire"
+ assert completions[-1].get("stream") is True, (
+ f"{platform} turn went out non-streaming: stream={completions[-1].get('stream')!r}"
+ )
+ # No interrupt worker: the request was dispatched from the caller's thread
+ # (the #62151 / #60203 deadlock class needs the request on a spawned worker).
+ assert issued_on["tid"] == threading.get_ident()
+ assert response.choices[0].message.content == "hello world"
+ assert response.choices[0].finish_reason == "stop"
+
+
+def test_interactive_platform_still_uses_the_worker_thread(wire):
+ """Regression guard for the refactor: non-direct contexts keep the
+ interrupt worker (interactive /stop responsiveness relies on it)."""
+ agent = _make_agent(wire.base_url, platform="cli")
+ assert should_use_direct_api_call(agent) is False
+
+ issued_on = {}
+ real_create = agent._create_request_openai_client
+
+ def spy(*a, **k):
+ issued_on["tid"] = threading.get_ident()
+ return real_create(*a, **k)
+
+ agent._create_request_openai_client = spy
+ response = interruptible_streaming_api_call(agent, dict(_KW))
+
+ assert issued_on["tid"] != threading.get_ident()
+ assert response.choices[0].message.content == "hello world"
+
+
+def test_inline_stream_stale_detector_still_fires_from_monitor_thread(
+ stalling_wire, monkeypatch
+):
+ """The stale-stream detector moved onto a monitor thread for inline
+ mode; a stream that sends one chunk then only keep-alives must still be
+ killed at the stale budget instead of hanging until the socket dies."""
+ monkeypatch.setenv("HERMES_STREAM_STALE_TIMEOUT", "1.0")
+ monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
+ agent = _make_agent(stalling_wire.base_url, platform="subagent")
+
+ started = time.time()
+ response = interruptible_streaming_api_call(agent, dict(_KW))
+ elapsed = time.time() - started
+
+ assert elapsed < 6.0, f"inline stream was not bounded by the stale detector ({elapsed:.1f}s)"
+ # A partial delta was delivered → the loop gets the length-truncated
+ # partial-stream stub (same contract as the worker path).
+ assert getattr(response, "id", None) == helpers.PARTIAL_STREAM_STUB_ID
+ assert response.choices[0].finish_reason == helpers.FINISH_REASON_LENGTH
+
+
+def test_inline_stream_cross_thread_interrupt_aborts_promptly(stalling_wire, monkeypatch):
+ """``AIAgent.interrupt()`` from another thread (cron watchdog, delegation
+ stall monitor) must abort the inline stream and surface InterruptedError
+ — the property the direct_api_call path guaranteed via
+ ``_active_request_abort``."""
+ monkeypatch.setenv("HERMES_STREAM_STALE_TIMEOUT", "60")
+ monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
+ agent = _make_agent(stalling_wire.base_url, platform="cron")
+ box: dict = {}
+
+ def _run():
+ t0 = time.time()
+ try:
+ interruptible_streaming_api_call(agent, dict(_KW))
+ box["outcome"] = "returned"
+ except BaseException as exc: # noqa: BLE001 — record whatever surfaces
+ box["outcome"] = type(exc).__name__
+ box["elapsed"] = time.time() - t0
+
+ worker = threading.Thread(target=_run, daemon=True)
+ worker.start()
+ assert stalling_wire.hits.acquire(timeout=5.0), "request never reached the wire"
+ time.sleep(0.3) # let the first chunk land
+ agent.interrupt("test interrupt")
+ worker.join(timeout=10.0)
+
+ assert not worker.is_alive(), "inline stream did not unwind after interrupt"
+ assert box["outcome"] == "InterruptedError"
+ assert box["elapsed"] < 5.0
+
+
+def test_should_use_direct_api_call_gate_is_unchanged():
+ """The routing predicate itself is untouched — only what it routes to."""
+ def mk(platform, api_mode="chat_completions", provider="openrouter"):
+ return SimpleNamespace(platform=platform, api_mode=api_mode, provider=provider)
+
+ assert should_use_direct_api_call(mk("cron")) is True
+ assert should_use_direct_api_call(mk("subagent")) is True
+ assert should_use_direct_api_call(mk("cli")) is False
+ assert should_use_direct_api_call(mk("cron", api_mode="anthropic_messages")) is False
+ assert should_use_direct_api_call(mk("cron", provider="moa")) is False
diff --git a/tests/run_agent/test_flush_diverts_on_corrupt_state_db.py b/tests/run_agent/test_flush_diverts_on_corrupt_state_db.py
new file mode 100644
index 0000000000..26e941c495
--- /dev/null
+++ b/tests/run_agent/test_flush_diverts_on_corrupt_state_db.py
@@ -0,0 +1,70 @@
+"""Agent flush path: a quarantined (structurally corrupt) SessionDB diverts to JSONL.
+
+Mirrors the replaced-file contract: the batch that SQLite will never take
+again is kept on disk under ``sessions/.jsonl`` instead of only in RAM,
+the flush fails closed (no retry loop), and the turn-end explanation gets the
+``corrupt`` cause.
+"""
+
+from __future__ import annotations
+
+from pathlib import Path
+from types import SimpleNamespace
+
+from hermes_state import SessionDB, StateDbCorruptError
+from run_agent import AIAgent
+
+
+def _flush_agent(db, session_id):
+ agent = SimpleNamespace(
+ _session_db=db,
+ _session_db_created=True,
+ _persist_disabled=False,
+ session_id=session_id,
+ _session_persist_lock=None,
+ _flushed_db_message_ids=set(),
+ _flushed_db_message_session_id=None,
+ _last_flushed_db_idx=0,
+ _db_flush_scan_prefix=None,
+ _persist_user_message_idx=None,
+ _persist_user_message_override=None,
+ _persist_user_message_timestamp=None,
+ _pending_cli_user_message=None,
+ _active_session_turn_lease_holder=None,
+ _last_persistence_error_cause=None,
+ _compression_adoption_failed=False,
+ )
+ agent._ensure_db_session = lambda: None
+ agent._flush_messages_to_session_db = (
+ AIAgent._flush_messages_to_session_db.__get__(agent, AIAgent)
+ )
+ agent._flush_messages_to_session_db_unlocked = (
+ AIAgent._flush_messages_to_session_db_unlocked.__get__(agent, AIAgent)
+ )
+ return agent
+
+
+def test_flush_diverts_batch_to_jsonl_when_handle_is_quarantined(
+ tmp_path: Path, monkeypatch
+) -> None:
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ db = SessionDB(db_path=tmp_path / "state.db")
+ try:
+ db.create_session("live", source="cli")
+ agent = _flush_agent(db, "live")
+
+ def _quarantined(self, *, session_id, messages, **kwargs):
+ raise StateDbCorruptError("database disk image is malformed (quarantined)")
+
+ monkeypatch.setattr(SessionDB, "append_messages_batch", _quarantined)
+
+ messages = [{"role": "user", "content": "kept-on-disk-after-corruption"}]
+ result = agent._flush_messages_to_session_db(messages, [])
+
+ assert result is False
+ assert agent._last_persistence_error_cause == "corrupt"
+ jsonl = tmp_path / "sessions" / "live.jsonl"
+ assert jsonl.is_file()
+ assert "kept-on-disk-after-corruption" in jsonl.read_text(encoding="utf-8")
+ finally:
+ db.close()
diff --git a/tests/run_agent/test_length_continuation_thinking_exhaustion.py b/tests/run_agent/test_length_continuation_thinking_exhaustion.py
new file mode 100644
index 0000000000..a70e71fdf7
--- /dev/null
+++ b/tests/run_agent/test_length_continuation_thinking_exhaustion.py
@@ -0,0 +1,311 @@
+"""Regression tests for thinking-only length truncations.
+
+GLM-5.3-flash on ollama-cloud with reasoning_effort=high can burn the ENTIRE
+output cap on reasoning delivered in a separate field and return
+finish_reason="length" with NO visible content (verified live: max_tokens=4096
+→ completion_tokens=4096, reasoning ~18.5KB, content empty).
+
+The old continuation flow handled this badly:
+ 1. the empty response was appended as an interim assistant fragment,
+ poisoning the transcript until the pre-call sanitizer "healed" it
+ (observed 3+ healings per turn);
+ 2. every continuation re-ran with thinking ON, re-deriving — and re-burning
+ — the whole thinking budget against a growing context, so 4 attempts
+ still produced nothing and the turn died with
+ "Response remained truncated after 4 continuation attempts".
+
+The fix: skip empty interim fragments, and issue the continuation with a
+one-shot reasoning-off override so the budget goes to writing the answer.
+"""
+
+from __future__ import annotations
+
+from types import SimpleNamespace
+from unittest.mock import MagicMock, patch
+
+import pytest
+
+from hermes_constants import FINISH_REASON_LENGTH
+
+
+class _AgentStandIn:
+ """Minimal agent surface _reasoning_config_for_wire needs."""
+
+ def __init__(self, reasoning_config):
+ self.reasoning_config = reasoning_config
+
+
+class TestReasoningOffOneShotOverride:
+ def test_flag_consumed_exactly_once(self):
+ from agent.chat_completion_helpers import _reasoning_config_for_wire
+
+ agent = _AgentStandIn({"enabled": True, "effort": "high"})
+ # Without the flag the reasoning config passes through untouched.
+ assert _reasoning_config_for_wire(agent) == {
+ "enabled": True,
+ "effort": "high",
+ }
+
+ agent._ephemeral_reasoning_off = True
+ cfg = _reasoning_config_for_wire(agent)
+ assert cfg["enabled"] is False
+ assert cfg["effort"] == "none"
+ assert agent._ephemeral_reasoning_off is False, (
+ "The one-shot override must be consumed by the first call."
+ )
+
+ # Subsequent calls keep the user's own reasoning config.
+ assert _reasoning_config_for_wire(agent) == {
+ "enabled": True,
+ "effort": "high",
+ }
+
+ def test_flag_with_no_user_reasoning_config(self):
+ from agent.chat_completion_helpers import _reasoning_config_for_wire
+
+ agent = _AgentStandIn(None)
+ agent._ephemeral_reasoning_off = True
+ cfg = _reasoning_config_for_wire(agent)
+ assert cfg == {"enabled": False, "effort": "none"}
+
+
+@pytest.fixture()
+def loop_agent():
+ from run_agent import AIAgent
+
+ with (
+ patch("run_agent.get_tool_definitions", return_value=[]),
+ patch("run_agent.check_toolset_requirements", return_value={}),
+ patch("run_agent.OpenAI"),
+ ):
+ a = AIAgent(
+ api_key="test-key-1234567890",
+ base_url="https://openrouter.ai/api/v1",
+ quiet_mode=True,
+ skip_context_files=True,
+ skip_memory=True,
+ )
+ a.client = MagicMock()
+ a._cached_system_prompt = "You are helpful."
+ a._use_prompt_caching = False
+ a.compression_enabled = False
+ a.save_trajectories = False
+ return a
+
+
+def _thinking_only_length_response():
+ """finish_reason='length' with reasoning but zero visible content — the
+ live GLM-5.3-flash-on-ollama-cloud shape (normal response id, NOT the
+ partial-stream stub)."""
+ from tests.run_agent.test_run_agent import _mock_assistant_msg
+
+ return SimpleNamespace(
+ id="chatcmpl-thinking-exhausted",
+ model="test/model",
+ choices=[SimpleNamespace(
+ index=0,
+ message=_mock_assistant_msg(content=""),
+ finish_reason=FINISH_REASON_LENGTH,
+ )],
+ usage=None,
+ )
+
+
+def _full_response(content):
+ from tests.run_agent.test_run_agent import _mock_response
+
+ return _mock_response(content=content, finish_reason="stop")
+
+
+def _truncated_text_response(content):
+ from tests.run_agent.test_run_agent import _mock_response
+
+ return _mock_response(content=content, finish_reason=FINISH_REASON_LENGTH)
+
+
+def _run(agent, message, history=None):
+ with (
+ patch.object(agent, "_persist_session"),
+ patch.object(agent, "_save_trajectory"),
+ patch.object(agent, "_cleanup_task_resources"),
+ ):
+ return agent.run_conversation(message, conversation_history=history)
+
+
+def _no_empty_assistant_rows(messages):
+ return [
+ m for m in messages
+ if m.get("role") == "assistant"
+ and not (m.get("content") or "").strip()
+ and not m.get("tool_calls")
+ ]
+
+
+class TestThinkingOnlyTruncation:
+ def test_retry_after_thinking_only_truncation_completes(self, loop_agent):
+ """One thinking-only truncation, then a normal answer: the retry must
+ drop thinking (one-shot), boost the output cap, and finish the turn."""
+ loop_agent.client.chat.completions.create.side_effect = [
+ _thinking_only_length_response(),
+ _full_response("Here is the full answer."),
+ ]
+ result = _run(loop_agent, "write me a long report")
+
+ assert result["completed"] is True
+ assert "full answer" in (result["final_response"] or "")
+ assert _no_empty_assistant_rows(result["messages"]) == [], (
+ "An empty (thinking-only) truncated response must never be "
+ "appended to the transcript."
+ )
+
+ calls = loop_agent.client.chat.completions.create.call_args_list
+ assert len(calls) == 2
+ # Continuation retry boosts the output cap (2^1 × 4096 base floor).
+ assert calls[1].kwargs.get("max_tokens") == 8192, (
+ "The continuation retry must request a larger output budget than "
+ "the request that truncated."
+ )
+ assert loop_agent._ephemeral_reasoning_off is False, (
+ "The one-shot reasoning-off override must be consumed by the "
+ "continuation call."
+ )
+
+ def test_thinking_only_truncation_sets_reasoning_off(self, loop_agent):
+ from tests.run_agent.test_run_agent import _mock_response
+
+ loop_agent.client.chat.completions.create.side_effect = [
+ _thinking_only_length_response(),
+ _mock_response(
+ content="done", finish_reason=FINISH_REASON_LENGTH
+ ),
+ _full_response("finally complete."),
+ ]
+ _run(loop_agent, "write me a long report")
+
+ calls = loop_agent.client.chat.completions.create.call_args_list
+ assert len(calls) == 3
+ # The thinking-only fragment set the flag; it was consumed by the
+ # next call, and the SECOND truncated fragment (which had visible
+ # text) does not set it again — so the third call sees thinking ON.
+ assert loop_agent._ephemeral_reasoning_off is False
+
+ def test_full_ceiling_with_empty_fragments_still_settles(self, loop_agent):
+ """All four attempts thinking-only: the turn must exit through the
+ ceiling with an actionable final_response, no poisoned transcript,
+ and no leaked reasoning-off flag."""
+ loop_agent.client.chat.completions.create.side_effect = [
+ _thinking_only_length_response() for _ in range(4)
+ ]
+ result = _run(loop_agent, "write me a long report")
+
+ assert result["completed"] is False
+ assert result["partial"] is True
+ assert "truncated after 4 continuation attempts" in (result.get("error") or "")
+ assert result["final_response"], (
+ "An all-empty ceiling exit must still surface a user-facing "
+ "message instead of an invisible None."
+ )
+ assert "reasoning" in (result["final_response"] or "").lower()
+ assert _no_empty_assistant_rows(result["messages"]) == []
+ assert loop_agent._ephemeral_reasoning_off is False, (
+ "The ceiling exit must clear the pending one-shot override so the "
+ "next turn does not silently lose thinking."
+ )
+
+ def test_mixed_fragments_keep_visible_text(self, loop_agent):
+ """A visible fragment followed by a thinking-only one: the visible
+ text must be stitched, the empty one skipped."""
+ loop_agent.client.chat.completions.create.side_effect = [
+ _truncated_text_response("visible part one. "),
+ _thinking_only_length_response(),
+ _full_response("and the ending."),
+ ]
+ result = _run(loop_agent, "write me a long report")
+
+ assert result["completed"] is True
+ assert "visible part one." in (result["final_response"] or "")
+ assert "and the ending." in (result["final_response"] or "")
+ assert _no_empty_assistant_rows(result["messages"]) == []
+
+class TestReasoningOffReachesTheWire:
+ def test_continuation_request_carries_reasoning_off_on_the_wire(self, loop_agent):
+ """The flag is only useful if the continuation REQUEST goes out with
+ thinking disabled — assert the OpenRouter extra_body, not the flag."""
+ loop_agent.reasoning_config = {"enabled": True, "effort": "high"}
+ loop_agent._supports_reasoning_extra_body = lambda: True
+ loop_agent.client.chat.completions.create.side_effect = [
+ _thinking_only_length_response(),
+ _full_response("Here is the full answer."),
+ ]
+ result = _run(loop_agent, "write me a long report")
+ assert result["completed"] is True
+
+ calls = loop_agent.client.chat.completions.create.call_args_list
+ assert len(calls) == 2
+ first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning")
+ second = (calls[1].kwargs.get("extra_body") or {}).get("reasoning")
+ assert first == {"enabled": True, "effort": "high"}, first
+ assert second is not None and second.get("enabled") is False, (
+ f"continuation must be sent with thinking off, got {second!r}"
+ )
+
+ def test_reasoning_off_is_exactly_one_request_and_prefix_stays_stable(self, loop_agent):
+ """Prompt-cache invariant for the override.
+
+ The reasoning parameter is part of the provider's cache key on
+ config-sensitive providers (Anthropic renders thinking/effort into
+ the prompt; OpenAI lists reasoning.effort as a prefix-affecting
+ setting), so the reasoning-off request is a deliberate one-request
+ cache miss. It must stay exactly one request: the request AFTER it
+ (a second, visible-text continuation) must go out with the
+ configured reasoning again, and the system prompt must be
+ byte-identical on every request so the miss never compounds into a
+ rebuilt prefix.
+ """
+ loop_agent.reasoning_config = {"enabled": True, "effort": "high"}
+ loop_agent._supports_reasoning_extra_body = lambda: True
+ loop_agent.client.chat.completions.create.side_effect = [
+ _thinking_only_length_response(),
+ _truncated_text_response("PART ONE of the answer"),
+ _full_response(" and PART TWO, done."),
+ ]
+ result = _run(loop_agent, "write me a long report")
+ assert result["completed"] is True
+ assert "PART ONE" in result["final_response"]
+ assert "PART TWO" in result["final_response"]
+
+ calls = loop_agent.client.chat.completions.create.call_args_list
+ assert len(calls) == 3
+ wire = [
+ (c.kwargs.get("extra_body") or {}).get("reasoning") for c in calls
+ ]
+ assert wire[0] == {"enabled": True, "effort": "high"}, wire
+ assert wire[1] == {"enabled": False, "effort": "none"}, wire
+ assert wire[2] == {"enabled": True, "effort": "high"}, (
+ f"reasoning must be restored on the very next request; got {wire!r}"
+ )
+ system_prompts = {
+ c.kwargs["messages"][0]["content"] for c in calls
+ if c.kwargs["messages"][0].get("role") == "system"
+ }
+ assert len(system_prompts) == 1, (
+ "system prompt must be byte-identical across the retry sequence "
+ "(the override may only change request parameters, never the prefix)"
+ )
+ assert loop_agent._ephemeral_reasoning_off is False
+
+ def test_stale_flag_does_not_leak_into_next_turn(self, loop_agent):
+ """A flag armed by a previous turn that never reached build_api_kwargs
+ (interrupt/error between arm and consume) must not silently strip
+ thinking from the next turn's first request."""
+ loop_agent.reasoning_config = {"enabled": True, "effort": "high"}
+ loop_agent._supports_reasoning_extra_body = lambda: True
+ loop_agent._ephemeral_reasoning_off = True # stale from a prior turn
+ loop_agent.client.chat.completions.create.side_effect = [
+ _full_response("fresh turn answer."),
+ ]
+ result = _run(loop_agent, "hello")
+ assert result["completed"] is True
+ calls = loop_agent.client.chat.completions.create.call_args_list
+ first = (calls[0].kwargs.get("extra_body") or {}).get("reasoning")
+ assert first == {"enabled": True, "effort": "high"}, first
diff --git a/tests/run_agent/test_model_streaming_config.py b/tests/run_agent/test_model_streaming_config.py
new file mode 100644
index 0000000000..1ed5073f12
--- /dev/null
+++ b/tests/run_agent/test_model_streaming_config.py
@@ -0,0 +1,151 @@
+"""``model.streaming`` config seeds the session's streaming decision (#72901).
+
+The conversation loop prefers ``stream=True`` for every turn — subagents
+included — for liveness health-checking (#3120). Self-hosted OpenAI-compatible
+backends with broken streaming tool-call paths (e.g. vLLM
+``--tool-call-parser qwen3_xml`` + reasoning parser) can leak tool-call markup
+into plain text and return zero ``tool_calls``, silently no-oping delegated
+tasks. ``model.streaming: false`` must seed ``_disable_streaming`` at agent
+init so the whole session (parent and subagents) uses the non-streaming path.
+"""
+import os
+from pathlib import Path
+from unittest.mock import MagicMock, patch
+
+from run_agent import AIAgent
+
+_BASE = {
+ "model": {
+ "default": "test/model",
+ "provider": "custom",
+ "base_url": "http://127.0.0.1:9999/v1",
+ "api_key": "x",
+ }
+}
+
+
+def _build_agent(config):
+ with patch("hermes_cli.config.load_config_readonly", return_value=config):
+ return AIAgent(
+ api_key="x",
+ base_url="http://127.0.0.1:9999/v1",
+ model="test/model",
+ provider="custom",
+ quiet_mode=True,
+ skip_context_files=True,
+ skip_memory=True,
+ )
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_false_seeds_disable_streaming(mock_openai):
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent({"model": {**_BASE["model"], "streaming": False}})
+
+ assert agent._disable_streaming is True
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_absent_keeps_streaming_enabled(mock_openai):
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent(_BASE)
+
+ assert agent._disable_streaming is False
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_true_keeps_streaming_enabled(mock_openai):
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent({"model": {**_BASE["model"], "streaming": True}})
+
+ assert agent._disable_streaming is False
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_string_false_seeds_disable_streaming(mock_openai):
+ """String falsy values ('false', '0') must also disable streaming —
+ YAML users commonly quote booleans."""
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent({"model": {**_BASE["model"], "streaming": "false"}})
+
+ assert agent._disable_streaming is True
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_zero_seeds_disable_streaming(mock_openai):
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent({"model": {**_BASE["model"], "streaming": 0}})
+
+ assert agent._disable_streaming is True
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_invalid_value_keeps_streaming_enabled(mock_openai):
+ """Unrecognized values warn and keep the safe default (streaming on),
+ rather than silently disabling or crashing init."""
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent({"model": {**_BASE["model"], "streaming": "flase"}})
+
+ assert agent._disable_streaming is False
+
+
+@patch("run_agent.OpenAI")
+def test_missing_model_section_keeps_streaming_enabled(mock_openai):
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent({})
+
+ assert agent._disable_streaming is False
+
+
+@patch("run_agent.OpenAI")
+def test_legacy_string_model_section_does_not_crash(mock_openai):
+ """The top-level ``model`` key is a legacy string; init must not crash."""
+ mock_openai.return_value = MagicMock()
+ agent = _build_agent({"model": "test/model"})
+
+ assert agent._disable_streaming is False
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_false_applies_to_every_agent_built_from_config(mock_openai):
+ """Delegate children are constructed through the same init, so any agent
+ (parent or subagent) built under this config gets the escape hatch —
+ covering the reported failure surface."""
+ mock_openai.return_value = MagicMock()
+ cfg = {"model": {**_BASE["model"], "streaming": False}}
+
+ first = _build_agent(cfg)
+ second = _build_agent(cfg)
+
+ assert first._disable_streaming is True
+ assert second._disable_streaming is True
+
+
+@patch("run_agent.OpenAI")
+def test_streaming_false_read_from_real_config_file(mock_openai):
+ """End-to-end: a real config.yaml in HERMES_HOME (sandboxed per-test by
+ conftest) with ``model.streaming: false`` must seed the flag through the
+ actual config loader — not just the patched function."""
+ mock_openai.return_value = MagicMock()
+ home = Path(os.environ["HERMES_HOME"])
+ (home / "config.yaml").write_text(
+ "model:\n"
+ " default: \"test/model\"\n"
+ " provider: \"custom\"\n"
+ " base_url: \"http://127.0.0.1:9999/v1\"\n"
+ " api_key: \"x\"\n"
+ " streaming: false\n",
+ encoding="utf-8",
+ )
+
+ agent = AIAgent(
+ api_key="x",
+ base_url="http://127.0.0.1:9999/v1",
+ model="test/model",
+ provider="custom",
+ quiet_mode=True,
+ skip_context_files=True,
+ skip_memory=True,
+ )
+
+ assert agent._disable_streaming is True
diff --git a/tests/run_agent/test_partial_stream_finish_reason.py b/tests/run_agent/test_partial_stream_finish_reason.py
index 2fd563a5f0..975cfacc54 100644
--- a/tests/run_agent/test_partial_stream_finish_reason.py
+++ b/tests/run_agent/test_partial_stream_finish_reason.py
@@ -91,6 +91,95 @@ class TestPartialStreamStubFinishReason:
assert response.choices[0].message.tool_calls is None
+class TestTerminalChunkFenceException:
+ """A superseded writer must still accept the provider's terminal
+ finish_reason chunk. Fending that chunk leaves finish_reason None
+ after real text was delivered, which the drop-guard mislabels as a
+ mid-stream drop even though the provider completed the stream.
+ """
+
+ @patch("run_agent.AIAgent._create_request_openai_client")
+ @patch("run_agent.AIAgent._close_request_openai_client")
+ def test_superseded_writer_accepts_finish_reason_chunk(
+ self, _mock_close, mock_create, monkeypatch,
+ ):
+ monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
+ agent_box = {}
+
+ class SupersedeBeforeFinish:
+ response = SimpleNamespace(headers={})
+
+ def __iter__(self):
+ yield _make_stream_chunk(content="Long prose that is complete.")
+ agent_box["agent"]._claim_stream_writer()
+ # Marker-only terminal chunk (empty delta), as vLLM emits.
+ yield _make_stream_chunk(finish_reason="stop")
+
+ mock_client = MagicMock()
+ mock_client.chat.completions.create.return_value = SupersedeBeforeFinish()
+ mock_create.return_value = mock_client
+
+ agent = _make_agent()
+ agent_box["agent"] = agent
+ response = agent._interruptible_streaming_api_call({})
+
+ assert response.id != PARTIAL_STREAM_STUB_ID
+ assert response.choices[0].finish_reason == "stop"
+ assert response.choices[0].message.content == "Long prose that is complete."
+
+ @patch("run_agent.AIAgent._create_request_openai_client")
+ @patch("run_agent.AIAgent._close_request_openai_client")
+ def test_superseded_writer_still_fences_further_content(
+ self, _mock_close, mock_create, monkeypatch,
+ ):
+ monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
+ agent_box = {}
+
+ class SupersedeBeforeMoreText:
+ response = SimpleNamespace(headers={})
+
+ def __iter__(self):
+ yield _make_stream_chunk(content="kept ")
+ agent_box["agent"]._claim_stream_writer()
+ # A False accept_chunk ends consumption; this text must
+ # never reach the accumulator, and the later finish chunk
+ # is never seen (the fence still stops *further* content).
+ yield _make_stream_chunk(content="must-not-append")
+ yield _make_stream_chunk(finish_reason="stop")
+
+ mock_client = MagicMock()
+ mock_client.chat.completions.create.return_value = SupersedeBeforeMoreText()
+ mock_create.return_value = mock_client
+
+ agent = _make_agent()
+ agent_box["agent"] = agent
+ response = agent._interruptible_streaming_api_call({})
+
+ content = response.choices[0].message.content or ""
+ assert "must-not-append" not in content
+ assert "kept" in content
+ assert response.id == PARTIAL_STREAM_STUB_ID
+
+ @patch("run_agent.AIAgent._create_request_openai_client")
+ @patch("run_agent.AIAgent._close_request_openai_client")
+ def test_genuine_truncation_without_finish_still_drops(
+ self, _mock_close, mock_create, monkeypatch,
+ ):
+ monkeypatch.setenv("HERMES_STREAM_RETRIES", "0")
+
+ def _truncated():
+ yield _make_stream_chunk(content="cut off with no terminal chunk")
+
+ mock_client = MagicMock()
+ mock_client.chat.completions.create.side_effect = lambda *a, **kw: _truncated()
+ mock_create.return_value = mock_client
+
+ agent = _make_agent()
+ response = agent._interruptible_streaming_api_call({})
+
+ assert response.id == PARTIAL_STREAM_STUB_ID
+ assert response.choices[0].finish_reason == FINISH_REASON_LENGTH
+
# ── Clean stream-end mid-tool-call (no exception, no finish_reason) ─────────
diff --git a/tests/run_agent/test_run_agent.py b/tests/run_agent/test_run_agent.py
index 5a8f067a28..05af3b8d1d 100644
--- a/tests/run_agent/test_run_agent.py
+++ b/tests/run_agent/test_run_agent.py
@@ -2253,7 +2253,7 @@ class TestConcurrentToolExecution:
def test_invoke_tool_handles_agent_level_tools(self, agent):
"""_invoke_tool should handle todo tool directly."""
with patch("tools.todo_tool.todo_tool", return_value='{"ok":true}') as mock_todo:
- result = agent._invoke_tool("todo", {"todos": []}, "task-1")
+ result = agent._invoke_tool("todo_list", {"todos": []}, "task-1")
mock_todo.assert_called_once()
assert "ok" in result
@@ -2345,7 +2345,7 @@ class TestConcurrentToolExecution:
"""Sequential and concurrent agent-level paths share post-hook ownership."""
from agent.agent_runtime_helpers import agent_runtime_owns_post_tool_hook
- for tool_name in ("todo", "session_search", "memory", "clarify", "delegate_task"):
+ for tool_name in ("todo_list", "session_search", "memory", "clarify", "delegate_task"):
assert agent_runtime_owns_post_tool_hook(agent, tool_name) is True
agent._context_engine_tool_names = {"context_query"}
@@ -2491,7 +2491,7 @@ class TestAgentRuntimePostHookOwnershipSync:
"""Exercise post-hook ownership through both agent-runtime tool paths."""
_CASES = (
- ("todo", {"todos": []}),
+ ("todo_list", {"todos": []}),
("session_search", {"query": "needle"}),
("memory", {"action": "view", "target": "memory"}),
("clarify", {"question": "Continue?"}),
@@ -2501,7 +2501,7 @@ class TestAgentRuntimePostHookOwnershipSync:
("annotate_preview", {"action": "clear"}),
("read_window_below", {}),
("setup_mcp", {"server": "linear", "action": "install"}),
- ("tour", {"action": "stop"}),
+ ("gui_tour", {"action": "stop"}),
("delegate_task", {"goal": "Check the child path"}),
)
@@ -4340,11 +4340,11 @@ class TestRunConversation:
assert requested_caps == [65536, 65536]
def test_ollama_glm_stop_after_tools_without_terminal_boundary_requests_continuation(self, agent):
- """Ollama-hosted GLM responses can misreport truncated output as stop."""
+ """Local Ollama-hosted GLM (no :cloud suffix) misreports truncated output as stop."""
self._setup_agent(agent)
agent.base_url = "http://localhost:11434/v1"
agent._base_url_lower = agent.base_url.lower()
- agent.model = "glm-5.1:cloud"
+ agent.model = "glm-4-9b" # local GLM — no :cloud suffix
tool_turn = _mock_response(
content="",
@@ -4384,11 +4384,18 @@ class TestRunConversation:
assert third_call_messages[-1]["role"] == "user"
assert "truncated by the output length limit" in third_call_messages[-1]["content"]
-
-
-
-
-
+ @pytest.mark.parametrize("base_url, model", [
+ ("https://ollama.com/v1", "glm-5.3-flash"), # Ollama Cloud host (#72316)
+ ("http://localhost:11434/v1", "glm-5.1:cloud"), # :cloud via local proxy (#98406)
+ ])
+ def test_ollama_cloud_glm_stop_is_never_rewritten(self, agent, base_url, model):
+ """Ollama Cloud reports finish_reason faithfully — an unpunctuated stop stays stop."""
+ self._setup_agent(agent)
+ agent.base_url = base_url
+ agent._base_url_lower = base_url.lower()
+ agent.model = model
+ unpunctuated = SimpleNamespace(content="Based on the results the best next step is to update the config", tool_calls=None)
+ assert agent._should_treat_stop_as_truncated("stop", unpunctuated, [{"role": "tool", "content": "r"}]) is False
def test_length_thinking_exhausted_skips_continuation(self, agent):
"""When finish_reason='length' but content is only thinking, skip retries."""
@@ -4414,7 +4421,7 @@ class TestRunConversation:
# Should have a user-friendly response (not None)
assert result["final_response"] is not None
assert "Thinking Budget Exhausted" in result["final_response"]
- assert "/thinkon" in result["final_response"]
+ assert "/reasoning" in result["final_response"]
def test_length_with_tool_calls_returns_partial_without_executing_tools(self, agent):
diff --git a/tests/run_agent/test_run_agent_codex_responses.py b/tests/run_agent/test_run_agent_codex_responses.py
index 9c0165e066..26228f0b21 100644
--- a/tests/run_agent/test_run_agent_codex_responses.py
+++ b/tests/run_agent/test_run_agent_codex_responses.py
@@ -2502,3 +2502,126 @@ def test_codex_first_compaction_continuation_is_still_a_bare_retry(monkeypatch):
if m.get("role") == "user"
and m.get("content") == _CODEX_INCOMPLETE_NUDGE
]
+
+
+class _LazyCreateStream:
+ """Lazy iterable fake — events are produced during consumption, not upfront.
+
+ ``_FakeCreateStream`` materializes its events with ``list(events)`` in
+ __init__, which would run any side effect a generator encodes (such as
+ retiring the request token) before consumption starts. Retirement tests
+ need the side effect to land *between* two consumed frames.
+ """
+
+ def __init__(self, event_factory):
+ self._event_factory = event_factory
+ self.closed = False
+
+ def __iter__(self):
+ return iter(self._event_factory())
+
+ def close(self):
+ self.closed = True
+
+
+def _retiring_stream(agent, deltas, *, retire_after):
+ """Yield ``deltas`` lazily, clearing the request token mid-stream.
+
+ The token is cleared just before yielding delta index ``retire_after``,
+ mimicking a watchdog (TTFB / stream-idle / stale-call) retiring the
+ in-flight request while the worker thread is still draining SSE frames.
+ """
+
+ def _events():
+ yield SimpleNamespace(type="response.created")
+ for index, delta in enumerate(deltas):
+ if index == retire_after:
+ agent._active_codex_stream_request_token = None
+ yield SimpleNamespace(type="response.output_text.delta", delta=delta)
+ # A retired stream never reaches a terminal frame on the wire; the
+ # connection is force-closed under it.
+
+ return _LazyCreateStream(_events)
+
+
+def test_run_codex_stream_retired_request_raises_instead_of_partial_final(monkeypatch):
+ """A retired request must not be normalized into a completed response.
+
+ ``_consume_codex_event_stream`` returns ``status=terminal_status`` which
+ defaults to ``"completed"``, and its only guard is
+ ``if not saw_terminal and not output``. A watchdog kill mid-stream leaves
+ ``saw_terminal=False`` but ``output``/text non-empty, so the partial text
+ used to come back as a ``finish_reason=stop`` response and get persisted as
+ a complete assistant turn (a long reply would just stop mid-sentence).
+
+ Retirement must surface as a retryable ``TimeoutError`` instead.
+ """
+ agent = _build_agent(monkeypatch)
+ token = object()
+ agent._active_codex_stream_request_token = token
+
+ def _fake_create(**kwargs):
+ assert kwargs.get("stream") is True
+ return _retiring_stream(
+ agent, ["1. Create ", "(6/6)", " [END-BILLING"], retire_after=2
+ )
+
+ agent.client = SimpleNamespace(responses=SimpleNamespace(create=_fake_create))
+
+ with pytest.raises(TimeoutError, match="retired"):
+ agent._run_codex_stream(_codex_request_kwargs())
+
+
+def test_run_codex_stream_without_token_keeps_partial_tolerance(monkeypatch):
+ """No token installed (non-watchdog callers) keeps the existing behavior.
+
+ ``_active_codex_stream_request_token`` is only set by
+ ``interruptible_api_call``. Auxiliary callers (compression summaries,
+ title generation) drive ``_run_codex_stream`` directly with no token and
+ must keep tolerating a stream that ends without a terminal frame.
+ """
+ agent = _build_agent(monkeypatch)
+ agent._active_codex_stream_request_token = None
+ output_item = SimpleNamespace(
+ type="message",
+ status="completed",
+ content=[SimpleNamespace(type="output_text", text="no terminal frame")],
+ )
+
+ def _fake_create(**kwargs):
+ return _FakeCreateStream([
+ SimpleNamespace(type="response.created"),
+ SimpleNamespace(type="response.output_item.done", item=output_item),
+ ])
+
+ agent.client = SimpleNamespace(responses=SimpleNamespace(create=_fake_create))
+
+ response = agent._run_codex_stream(_codex_request_kwargs())
+ assert response.status == "completed"
+ assert response.output == [output_item]
+
+
+def test_run_codex_stream_retired_request_stops_firing_callbacks(monkeypatch):
+ """Deltas that arrive after retirement must not reach the UI callbacks.
+
+ The gateway caches AIAgent instances per session, so a retired worker that
+ keeps draining frames would otherwise stream tokens from an abandoned
+ attempt into the live turn's bubble alongside the retry's output.
+ """
+ agent = _build_agent(monkeypatch)
+ token = object()
+ agent._active_codex_stream_request_token = token
+
+ streamed: list[str] = []
+ monkeypatch.setattr(agent, "_fire_stream_delta", streamed.append)
+
+ def _fake_create(**kwargs):
+ return _retiring_stream(agent, ["keep", "DROPPED"], retire_after=1)
+
+ agent.client = SimpleNamespace(responses=SimpleNamespace(create=_fake_create))
+
+ with pytest.raises(TimeoutError):
+ agent._run_codex_stream(_codex_request_kwargs())
+
+ assert streamed == ["keep"]
+ assert "DROPPED" not in streamed
diff --git a/tests/run_agent/test_tool_arg_coercion.py b/tests/run_agent/test_tool_arg_coercion.py
index 4390c3e9a1..00dcb289b0 100644
--- a/tests/run_agent/test_tool_arg_coercion.py
+++ b/tests/run_agent/test_tool_arg_coercion.py
@@ -244,5 +244,5 @@ class TestCoerceToolArgsNested:
"""Against the real todo schema from the registry."""
import json as _json
args = {"todos": [_json.dumps({"id": "1", "content": "x", "status": "pending"})]}
- result = coerce_tool_args("todo", args)
+ result = coerce_tool_args("todo_list", args)
assert result["todos"][0] == {"id": "1", "content": "x", "status": "pending"}
diff --git a/tests/run_agent/test_tool_call_guardrail_runtime.py b/tests/run_agent/test_tool_call_guardrail_runtime.py
index ca6e80aac4..fbc0a51460 100644
--- a/tests/run_agent/test_tool_call_guardrail_runtime.py
+++ b/tests/run_agent/test_tool_call_guardrail_runtime.py
@@ -5,6 +5,8 @@ import uuid
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
+import pytest
+
from run_agent import AIAgent
@@ -36,7 +38,12 @@ def _mock_response(content="Hello", finish_reason="stop", tool_calls=None):
return SimpleNamespace(choices=[choice], model="test/model", usage=None)
-def _make_agent(*tool_names: str, max_iterations: int = 10, config: dict | None = None) -> AIAgent:
+def _make_agent(
+ *tool_names: str,
+ max_iterations: int = 10,
+ config: dict | None = None,
+ platform: str | None = None,
+) -> AIAgent:
with (
patch("run_agent.get_tool_definitions", return_value=_make_tool_defs(*tool_names)),
patch("run_agent.check_toolset_requirements", return_value={}),
@@ -51,6 +58,7 @@ def _make_agent(*tool_names: str, max_iterations: int = 10, config: dict | None
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
+ platform=platform or "cli",
)
agent.client = MagicMock()
agent._cached_system_prompt = "You are helpful."
@@ -86,6 +94,29 @@ def _hard_stop_config(**overrides) -> dict:
return cfg
+def test_gateway_platform_uses_hard_stop_default_without_cli_opt_in():
+ agent = _make_agent("web_search", platform="telegram")
+ args = {"query": "same"}
+
+ _seed_exact_failures(agent, "web_search", args, count=5)
+
+ decision = getattr(agent, "_tool_guardrails").before_call("web_search", args)
+ assert decision.action == "block"
+ assert decision.code == "repeated_exact_failure_block"
+
+
+@pytest.mark.parametrize("platform", ["desktop", "acp"])
+def test_interactive_platforms_keep_warning_only_default(platform):
+ agent = _make_agent("web_search", platform=platform)
+ args = {"query": "same"}
+
+ _seed_exact_failures(agent, "web_search", args, count=5)
+
+ decision = getattr(agent, "_tool_guardrails").before_call("web_search", args)
+ assert decision.action == "allow"
+ assert decision.code == "allow"
+
+
def test_default_sequential_path_warns_repeated_exact_failure_without_blocking_execution():
agent = _make_agent("web_search")
args = {"query": "same"}
diff --git a/tests/run_agent/test_turn_completion_explainer.py b/tests/run_agent/test_turn_completion_explainer.py
index 99b052c7f9..3a7f6b62e0 100644
--- a/tests/run_agent/test_turn_completion_explainer.py
+++ b/tests/run_agent/test_turn_completion_explainer.py
@@ -438,3 +438,8 @@ def test_run_conversation_partial_stream_recovery_surfaces_explanation():
assert result["response_previewed"] is False
+def test_classify_persistence_error_quarantined_handle_is_corrupt() -> None:
+ """A quarantined SessionDB raises the typed error; it stays in the corrupt bucket."""
+ from hermes_state import StateDbCorruptError, classify_persistence_error
+
+ assert classify_persistence_error(StateDbCorruptError("quarantined")) == "corrupt"
diff --git a/tests/scripts/test_case_collision_check.py b/tests/scripts/test_case_collision_check.py
new file mode 100644
index 0000000000..4546919b9e
--- /dev/null
+++ b/tests/scripts/test_case_collision_check.py
@@ -0,0 +1,118 @@
+"""Wrappers for scripts/check-case-collisions.py.
+
+Same pattern as tests/scripts/test_windows_footguns_full_repo_scan.py: run
+the real checker and assert its outcomes, so a normal pytest run catches a
+regression — someone committing a case-colliding pair — without anyone
+having to remember to run the script by hand.
+
+The collision cases are built with ``git update-index --cacheinfo`` (index
+only, never touching the working tree), so they exercise the same index the
+checker reads and work even on a case-insensitive filesystem, where the two
+spellings cannot coexist on disk.
+"""
+
+from __future__ import annotations
+
+import hashlib
+import subprocess
+import sys
+from pathlib import Path
+
+REPO_ROOT = Path(__file__).resolve().parents[2]
+SCRIPT = REPO_ROOT / "scripts" / "check-case-collisions.py"
+
+
+def _git_blob_sha(data: bytes) -> str:
+ """The git object hash for a blob with ``data`` as its content."""
+ header = f"blob {len(data)}\0".encode("ascii")
+ return hashlib.sha1(header + data).hexdigest()
+
+
+def _run_check(*args, root=None):
+ cmd = [sys.executable, str(SCRIPT)] + list(args)
+ if root is not None:
+ cmd.append(str(root))
+ return subprocess.run(
+ cmd,
+ capture_output=True,
+ text=True,
+ timeout=60,
+ stdin=subprocess.DEVNULL,
+ cwd=REPO_ROOT,
+ )
+
+
+def _git_init(tmp_path) -> Path:
+ repo = tmp_path / "repo"
+ repo.mkdir()
+ subprocess.run(["git", "init", "-q"], cwd=repo, check=True)
+ return repo
+
+
+def test_full_repo_has_no_case_colliding_paths():
+ """The real checker against the whole tracked tree must exit clean."""
+ result = _run_check()
+ assert result.returncode == 0, (
+ f"Case-collision check failed:\n{result.stdout}\n{result.stderr}"
+ )
+
+
+def test_detects_case_colliding_paths(tmp_path):
+ """Same-directory Foo.txt + foo.txt must fail, naming both paths."""
+ repo = _git_init(tmp_path)
+ subprocess.run(
+ [
+ "git", "update-index", "--add", "--cacheinfo",
+ f"100644,{_git_blob_sha(b'a')},Foo.txt",
+ ],
+ cwd=repo, check=True,
+ )
+ subprocess.run(
+ [
+ "git", "update-index", "--add", "--cacheinfo",
+ f"100644,{_git_blob_sha(b'b')},foo.txt",
+ ],
+ cwd=repo, check=True,
+ )
+
+ result = _run_check(root=repo)
+ assert result.returncode == 1, f"expected failure, got:\n{result.stdout}"
+ assert "Foo.txt" in result.stdout
+ assert "foo.txt" in result.stdout
+
+
+def test_detects_directory_case_collisions(tmp_path):
+ """The comparison is on the FULL path — dir/Foo.txt vs DIR/foo.txt too."""
+ repo = _git_init(tmp_path)
+ subprocess.run(
+ [
+ "git", "update-index", "--add", "--cacheinfo",
+ f"100644,{_git_blob_sha(b'a')},src/Helper.py",
+ ],
+ cwd=repo, check=True,
+ )
+ subprocess.run(
+ [
+ "git", "update-index", "--add", "--cacheinfo",
+ f"100644,{_git_blob_sha(b'b')},SRC/helper.py",
+ ],
+ cwd=repo, check=True,
+ )
+
+ result = _run_check(root=repo)
+ assert result.returncode == 1, f"expected failure, got:\n{result.stdout}"
+ assert "src/Helper.py" in result.stdout
+ assert "SRC/helper.py" in result.stdout
+
+
+def test_same_name_in_different_dirs_is_not_a_collision(tmp_path):
+ """a/Readme.txt and b/readme.txt share a basename but not a path."""
+ repo = _git_init(tmp_path)
+ (repo / "a").mkdir()
+ (repo / "b").mkdir()
+ (repo / "a" / "Readme.txt").write_text("a", encoding="utf-8")
+ (repo / "b" / "readme.txt").write_text("b", encoding="utf-8")
+ subprocess.run(["git", "add", "-A"], cwd=repo, check=True)
+
+ result = _run_check(root=repo)
+ assert result.returncode == 0, f"expected clean, got:\n{result.stdout}"
diff --git a/tests/scripts/test_contributor_map.py b/tests/scripts/test_contributor_map.py
index 40fd3567a2..b6082ccfe5 100644
--- a/tests/scripts/test_contributor_map.py
+++ b/tests/scripts/test_contributor_map.py
@@ -116,3 +116,66 @@ def test_cli_entrypoint_end_to_end(tmp_path):
assert proc.returncode == 0, proc.stderr
out = (tmp_path / "contributors" / "emails" / "cli@example.com").read_text(encoding="utf-8")
assert out.splitlines()[0] == "cliperson"
+
+
+# ── case-insensitive filename collisions ──────────────────────────────
+#
+# The mapping key IS the filename, so two emails differing only in case are the
+# same file on Windows and on default macOS. When both exist, git writes one and
+# then reports the other as modified in a FRESH clone, permanently: the repo can
+# never be checked out clean on those platforms.
+#
+# The historical agent@Agents-Mac-mini.local / agent@agents-Mac-mini.local pair
+# was removed from the tree (fcdae2cf0b), so there is no allowlist: any pair
+# is a regression. scripts/check-case-collisions.py enforces the same
+# invariant repo-wide in CI; this test keeps it visible next to the writer.
+EMAILS_DIR = REPO_ROOT / "contributors" / "emails"
+
+
+def test_no_case_insensitive_mapping_collisions():
+ groups: dict[str, set[str]] = {}
+ for entry in EMAILS_DIR.iterdir():
+ if entry.is_file():
+ groups.setdefault(entry.name.casefold(), set()).add(entry.name)
+
+ collisions = {frozenset(names) for names in groups.values() if len(names) > 1}
+
+ assert not collisions, (
+ "contributor mappings differing only in case cannot coexist on "
+ "case-insensitive filesystems (Windows, default macOS) — a fresh clone "
+ f"there is permanently dirty: {sorted(sorted(c) for c in collisions)}"
+ )
+
+
+def test_add_contributor_refuses_a_case_collision(tmp_path, monkeypatch):
+ d = tmp_path / "emails"
+ d.mkdir()
+ (d / "agent@Example-Host.local").write_text("someone\n")
+
+ import add_contributor as mod
+
+ monkeypatch.setattr(mod, "EMAILS_DIR", d)
+
+ assert mod.add_contributor("agent@example-host.local", "otherperson") == 1
+ assert not (d / "agent@example-host.local").exists()
+
+
+def test_add_contributor_refuses_case_collision_even_for_same_login(emails_dir, capsys):
+ # Same login, different spelling: still refused — the problem is the
+ # filename pair, not the login. The exact spelling is what's "present".
+ emails_dir.mkdir(parents=True)
+ (emails_dir / "Foo@Example.com").write_text("foouser\n")
+
+ assert add_contributor("foo@example.com", "foouser") == 1
+ assert "Foo@Example.com" in capsys.readouterr().err
+ assert sorted(p.name for p in emails_dir.iterdir()) == ["Foo@Example.com"]
+ # Exact-case re-add is the ordinary idempotent path.
+ assert add_contributor("Foo@Example.com", "foouser") == 0
+
+
+def test_case_collision_uses_casefold(emails_dir):
+ # casefold, not lower: matches how macOS/Windows fold non-ASCII (ß ~ ss).
+ emails_dir.mkdir(parents=True)
+ (emails_dir / "strasse@example.com").write_text("someone\n")
+ assert add_contributor("STRASSE@example.com", "someone") == 1
+ assert add_contributor("straße@example.com", "someone") == 1
diff --git a/tests/state/test_fts_rebuild_admission.py b/tests/state/test_fts_rebuild_admission.py
index b6baf087a0..ac923c6a6c 100644
--- a/tests/state/test_fts_rebuild_admission.py
+++ b/tests/state/test_fts_rebuild_admission.py
@@ -17,9 +17,11 @@ prove nothing.
"""
import contextlib
+import errno
import subprocess
import sqlite3
import sys
+import time
from pathlib import Path
import pytest
@@ -224,3 +226,399 @@ class TestSchemaPathAdmission:
# Recovered: breadcrumb cleared, triggers restored.
assert _meta_value(db_path, FTS_STALE_KEY) is None
assert _base_fts_triggers(db_path) == set(_FTS_TRIGGERS)
+
+
+# ---------------------------------------------------------------------------
+# Orphaned-fd staleness break (issue #100108).
+#
+# flock belongs to the open file DESCRIPTION, which fork() duplicates into
+# children. A holder that forks (multiprocessing worker, daemonized helper)
+# and then crashes leaves the flock held by the child forever — the kernel's
+# holder-death release never fires, and every contender deferred forever
+# ("FTS rebuild lock ... held by another process for more than 120s").
+# The fix records the acquirer's pid + start time under the lock; a contender
+# that times out breaks the lock ONLY when that recorded holder is provably
+# dead, and fails closed on any indeterminate state.
+# ---------------------------------------------------------------------------
+
+_ORPHANING_HOLDER_SCRIPT = """
+import os, sys, time
+sys.path.insert(0, {repo!r})
+import hermes_state_common
+
+admission = hermes_state_common.fts_rebuild_admission({db!r})
+admitted = admission.__enter__()
+assert admitted is True
+pid = os.fork()
+if pid == 0:
+ # Forked child: shares the lock fd's open file description. Sleep far
+ # beyond the test, never releasing.
+ time.sleep(600)
+ os._exit(0)
+print("child", pid, flush=True)
+# Crash WITHOUT releasing (no __exit__): simulates the production holder
+# dying mid-rebuild after having forked.
+os._exit(1)
+"""
+
+
+@contextlib.contextmanager
+def _orphaned_fork_holder(db_path: Path):
+ """Real #100108 shape: acquirer records itself, forks, dies."""
+ import os
+ import signal
+
+ script = _ORPHANING_HOLDER_SCRIPT.format(
+ repo=str(Path(hermes_state_common.__file__).parent), db=str(db_path)
+ )
+ proc = subprocess.Popen(
+ [sys.executable, "-c", script], stdout=subprocess.PIPE, text=True
+ )
+ line = proc.stdout.readline().strip()
+ assert line.startswith("child ")
+ grandchild = int(line.split()[1])
+ proc.wait(timeout=10) # the acquirer is now dead; grandchild holds the fd
+ try:
+ yield grandchild
+ finally:
+ with contextlib.suppress(OSError):
+ os.kill(grandchild, signal.SIGKILL)
+
+
+class TestOrphanedHolderStalenessBreak:
+ @pytest.mark.live_system_guard_bypass
+ def test_rebuild_breaks_lock_of_dead_forker(self, db, fast_timeout):
+ """The #100108 repro: recorded holder dead, forked child holds the
+ flock. The contender must break the orphaned lock and rebuild."""
+ with _orphaned_fork_holder(db.db_path):
+ assert db.rebuild_fts() >= 1
+
+ def test_admission_still_fails_closed_for_live_unrecorded_holder(
+ self, db, fast_timeout
+ ):
+ """A live holder that wrote no record (pre-fix build, non-Hermes
+ tool) is indeterminate — must defer, never break."""
+ with _rebuild_lock_held_by_other_process(db.db_path):
+ assert db.rebuild_fts() == 0
+
+ def test_admission_fails_closed_for_live_recorded_holder(
+ self, db, fast_timeout, monkeypatch
+ ):
+ """A record naming a live pid must defer even after timeout."""
+ import json
+ import os
+
+ lock = _lock_file(db.db_path)
+ with _rebuild_lock_held_by_other_process(db.db_path) as proc:
+ record = {
+ "pid": proc.pid,
+ "start_ticks": hermes_state_common._proc_start_ticks(proc.pid),
+ "acquired_at": 0,
+ }
+ lock.write_bytes(json.dumps(record).encode())
+ assert db.rebuild_fts() == 0
+
+ def test_holder_record_cleared_on_normal_release(self, tmp_path):
+ lock = tmp_path / "x.db.fts_rebuild.lock"
+ with hermes_state_common.fts_rebuild_admission(tmp_path / "x.db") as ok:
+ assert ok is True
+ assert b"pid" in lock.read_bytes()
+ assert lock.read_bytes() == b""
+
+ @pytest.mark.live_system_guard_bypass
+ def test_repair_lock_breaks_orphaned_holder(self, tmp_path, monkeypatch):
+ """_cross_process_repair_lock shares the same staleness break."""
+ import hermes_state
+
+ monkeypatch.setattr(hermes_state, "_REPAIR_LOCK_TIMEOUT_SECONDS", 0.5)
+ db_path = tmp_path / "state.db"
+ db_path.touch()
+
+ script = """
+import os, sys, time
+sys.path.insert(0, {repo!r})
+from pathlib import Path
+import hermes_state
+
+lock_cm = hermes_state._cross_process_repair_lock(Path({db!r}))
+assert lock_cm.__enter__() is True
+pid = os.fork()
+if pid == 0:
+ time.sleep(600)
+ os._exit(0)
+print("child", pid, flush=True)
+os._exit(1)
+""".format(repo=str(Path(hermes_state_common.__file__).parent), db=str(db_path))
+ import os
+ import signal
+
+ proc = subprocess.Popen(
+ [sys.executable, "-c", script], stdout=subprocess.PIPE, text=True
+ )
+ grandchild = int(proc.stdout.readline().strip().split()[1])
+ proc.wait(timeout=10)
+ try:
+ import hermes_state as hs
+
+ with hs._cross_process_repair_lock(db_path) as holding:
+ assert holding is True
+ finally:
+ with contextlib.suppress(OSError):
+ os.kill(grandchild, signal.SIGKILL)
+
+
+class TestNonContentionErrnoFailsFast:
+ def test_non_contention_oserror_does_not_wait_out_timeout(
+ self, tmp_path, monkeypatch
+ ):
+ import fcntl
+
+ monkeypatch.setattr(
+ hermes_state_common, "_FTS_REBUILD_LOCK_TIMEOUT_SECONDS", 30.0
+ )
+
+ def _flock(*_args, **_kwargs):
+ raise OSError(getattr(errno, "ESTALE", errno.EIO), "stale handle")
+
+ monkeypatch.setattr(fcntl, "flock", _flock)
+ db_path = tmp_path / "state.db"
+ t0 = time.monotonic()
+ with hermes_state_common.fts_rebuild_admission(db_path) as admitted:
+ assert admitted is False
+ assert time.monotonic() - t0 < 2.0
+
+ def test_retry_deferred_fts_recovery_rebuilds_same_instance(
+ self, tmp_path, monkeypatch
+ ):
+ """Gateway-shaped: same SessionDB stays open and retries after deferral."""
+ import hermes_state_schema
+
+ monkeypatch.setattr(hermes_state_schema, "_FTS_STALE_RETRY_SECONDS", 0.0)
+ db_path = tmp_path / "state.db"
+ d = SessionDB(db_path=db_path)
+ if not d._fts_enabled:
+ d.close()
+ pytest.skip("FTS5 unavailable in this build")
+ d.create_session("s1", source="test")
+ d.append_message("s1", "user", "hello recovery path")
+ d.close()
+
+ raw = sqlite3.connect(str(db_path))
+ raw.execute(
+ "INSERT OR REPLACE INTO state_meta(key, value) VALUES (?, '1')",
+ (FTS_STALE_KEY,),
+ )
+ for trig in _FTS_TRIGGERS:
+ raw.execute(f"DROP TRIGGER IF EXISTS {trig}")
+ raw.commit()
+ raw.close()
+
+ holders = [(4242, str(db_path))]
+ monkeypatch.setattr(
+ SessionDB, "_foreign_state_db_holders", lambda self: list(holders)
+ )
+ d2 = SessionDB(db_path=db_path)
+ try:
+ assert d2._fts_stale is True
+ d2._fts_stale_retry_after = 0.0
+ assert d2.retry_deferred_fts_recovery() is False
+ holders.clear()
+ d2._fts_stale_retry_after = 0.0
+ assert d2.retry_deferred_fts_recovery() is True
+ assert d2._fts_stale is False
+ finally:
+ d2.close()
+ assert _meta_value(db_path, FTS_STALE_KEY) is None
+ assert _base_fts_triggers(db_path) == set(_FTS_TRIGGERS)
+
+ def test_non_contention_errno_skips_holder_warning(
+ self, tmp_path, monkeypatch, caplog
+ ):
+ """The fast-fail must not ALSO log the misleading 'held by another
+ process for more than Ns' line — there is no holder."""
+ import fcntl
+ import logging
+
+ monkeypatch.setattr(
+ hermes_state_common, "_FTS_REBUILD_LOCK_TIMEOUT_SECONDS", 30.0
+ )
+
+ def _flock(*_args, **_kwargs):
+ raise OSError(errno.ENOTSUP, "no locks on this fs")
+
+ monkeypatch.setattr(fcntl, "flock", _flock)
+ with caplog.at_level(logging.INFO, logger="hermes_state"):
+ with hermes_state_common.fts_rebuild_admission(
+ tmp_path / "state.db"
+ ) as admitted:
+ assert admitted is False
+ messages = [r.getMessage() for r in caplog.records]
+ assert any("non-contention error" in m for m in messages)
+ assert not any("held by another process" in m for m in messages)
+
+ def test_repair_lock_non_contention_errno_fails_fast(
+ self, tmp_path, monkeypatch
+ ):
+ """Sibling site: the state.db repair lock shares the errno filter."""
+ import fcntl
+
+ import hermes_state
+
+ monkeypatch.setattr(hermes_state, "_REPAIR_LOCK_TIMEOUT_SECONDS", 30.0)
+
+ def _flock(*_args, **_kwargs):
+ raise OSError(errno.EIO, "i/o error")
+
+ monkeypatch.setattr(fcntl, "flock", _flock)
+ t0 = time.monotonic()
+ with hermes_state._cross_process_repair_lock(tmp_path / "state.db") as ok:
+ assert ok is False
+ assert time.monotonic() - t0 < 2.0
+
+ @pytest.mark.parametrize(
+ "exc, expected",
+ [
+ (BlockingIOError(errno.EAGAIN, "x"), True),
+ (OSError(errno.EWOULDBLOCK, "x"), True),
+ (OSError(errno.EACCES, "x"), True),
+ (OSError(errno.ESTALE, "x"), False),
+ (OSError(errno.ENOTSUP, "x"), False),
+ (OSError(errno.ENOLCK, "x"), False),
+ (OSError(errno.EIO, "x"), False),
+ (ValueError("not an oserror"), False),
+ ],
+ )
+ def test_is_advisory_lock_contention_table(self, exc, expected):
+ assert hermes_state_common.is_advisory_lock_contention(exc) is expected
+
+
+class TestDeferredFtsRetryInProcess:
+ """Gateway shape (#100108): one SessionDB stays open for days. A deferral
+ at open must be recoverable from an in-process periodic tick, with the
+ REAL rebuild lock held by a REAL child process at open time."""
+
+ @staticmethod
+ def _mark_stale(db_path: Path) -> None:
+ raw = sqlite3.connect(str(db_path))
+ raw.execute(
+ "INSERT OR REPLACE INTO state_meta(key, value) VALUES (?, '1')",
+ (FTS_STALE_KEY,),
+ )
+ for trig in _FTS_TRIGGERS:
+ raw.execute(f"DROP TRIGGER IF EXISTS {trig}")
+ raw.commit()
+ raw.close()
+
+ def test_retry_is_non_blocking_while_live_holder_and_backs_off(
+ self, tmp_path, fast_timeout, monkeypatch
+ ):
+ import hermes_state_schema
+
+ db_path = tmp_path / "state.db"
+ d = SessionDB(db_path=db_path)
+ if not d._fts_enabled:
+ d.close()
+ pytest.skip("FTS5 unavailable in this build")
+ d.create_session("s1", source="test")
+ d.append_message("s1", "user", "hello gateway retry")
+ d.close()
+ self._mark_stale(db_path)
+
+ with _rebuild_lock_held_by_other_process(db_path):
+ gw = SessionDB(db_path=db_path) # long-lived "gateway" open
+ try:
+ assert gw._fts_stale is True
+ # Live holder: the retry must return quickly (timeout=0),
+ # not wait out any admission budget.
+ monkeypatch.setattr(
+ hermes_state_common, "_FTS_REBUILD_LOCK_TIMEOUT_SECONDS", 30.0
+ )
+ t0 = time.monotonic()
+ assert gw.retry_deferred_fts_recovery() is False
+ assert time.monotonic() - t0 < 2.0
+ assert gw._fts_stale is True
+ # Rate limit engaged: an immediate second call is a no-op.
+ assert gw.retry_deferred_fts_recovery() is False
+ # Backoff doubled (60s -> 120s) but capped at the max.
+ assert gw._fts_stale_retry_interval == min(
+ 2 * hermes_state_schema._FTS_STALE_RETRY_SECONDS,
+ hermes_state_schema._FTS_STALE_RETRY_MAX_SECONDS,
+ )
+ assert gw._fts_stale_retry_after > time.monotonic()
+ except BaseException:
+ gw.close()
+ raise
+ # Holder gone. Same instance recovers on the next eligible tick.
+ try:
+ gw._fts_stale_retry_after = 0.0
+ assert gw.retry_deferred_fts_recovery() is True
+ assert gw._fts_stale is False
+ assert gw._fts_enabled is True
+ # Search actually works again on this very instance.
+ gw.append_message("s1", "user", "needle-after-holder-gone")
+ assert gw.retry_deferred_fts_recovery() is False # nothing stale
+ finally:
+ gw.close()
+ assert _meta_value(db_path, FTS_STALE_KEY) is None
+ assert _base_fts_triggers(db_path) == set(_FTS_TRIGGERS)
+
+ def test_gateway_housekeeping_tick_drives_the_retry(
+ self, tmp_path, fast_timeout, monkeypatch
+ ):
+ """The retry hangs off the EXISTING housekeeping loop (no new thread)
+ and reaches shared-registry instances."""
+ import threading
+
+ import hermes_state_registry
+ import hermes_state_schema
+ import gateway.run as grun
+
+ monkeypatch.setattr(hermes_state_schema, "_FTS_STALE_RETRY_SECONDS", 0.0)
+ db_path = tmp_path / "state.db"
+ d = SessionDB(db_path=db_path)
+ if not d._fts_enabled:
+ d.close()
+ pytest.skip("FTS5 unavailable in this build")
+ d.create_session("s1", source="test")
+ d.append_message("s1", "user", "hello housekeeping")
+ d.close()
+ self._mark_stale(db_path)
+
+ with _rebuild_lock_held_by_other_process(db_path):
+ gw = hermes_state_registry.acquire(db_path)
+ try:
+ assert gw._fts_stale is True
+ assert gw in hermes_state_registry.live_shared_session_dbs()
+ stop = threading.Event()
+ th = threading.Thread(
+ target=grun._start_gateway_housekeeping,
+ args=(stop,),
+ kwargs={"interval": 0.05},
+ daemon=True,
+ )
+ th.start()
+ deadline = time.monotonic() + 10.0
+ while gw._fts_stale and time.monotonic() < deadline:
+ time.sleep(0.05)
+ stop.set()
+ th.join(timeout=5)
+ assert gw._fts_stale is False
+ assert gw._fts_enabled is True
+ finally:
+ hermes_state_registry.release_or_close(gw)
+ assert _meta_value(db_path, FTS_STALE_KEY) is None
+
+ def test_retry_noop_when_not_stale_or_read_only(self, tmp_path):
+ db_path = tmp_path / "state.db"
+ d = SessionDB(db_path=db_path)
+ try:
+ assert d._fts_stale is False
+ assert d.retry_deferred_fts_recovery() is False
+ finally:
+ d.close()
+ ro = SessionDB(db_path=db_path, read_only=True)
+ try:
+ ro._fts_stale = True
+ assert ro.retry_deferred_fts_recovery() is False
+ finally:
+ ro.close()
diff --git a/tests/state/test_fts_runtime_rebuild.py b/tests/state/test_fts_runtime_rebuild.py
index 8779186e72..ac030f4cb6 100644
--- a/tests/state/test_fts_runtime_rebuild.py
+++ b/tests/state/test_fts_runtime_rebuild.py
@@ -332,7 +332,15 @@ class TestRuntimeFtsRebuild:
with pytest.raises(sqlite3.DatabaseError) as caught:
db._execute_write(lambda _conn: (_ for _ in ()).throw(structural))
- assert caught.value is structural
+ # Structural corruption quarantines the handle: the typed error wraps
+ # the original (cause preserved, SQLite result code copied) and the
+ # sticky flag is set, so later writes fail fast.
+ from hermes_state import StateDbCorruptError
+
+ assert isinstance(caught.value, StateDbCorruptError)
+ assert caught.value.__cause__ is structural
+ assert caught.value.sqlite_errorcode == sqlite3.SQLITE_CORRUPT
+ assert db._db_corrupt is True
assert rebuild_called is False
assert db._fts_stale is False
assert _meta_value(tmp_path / "state.db", FTS_STALE_KEY) is None
@@ -831,6 +839,19 @@ class TestPhysicalCorruptionAcceptance:
# The misdiagnosis message from the field incident must be gone.
assert "canonical message rows are preserved" not in caplog.text
assert "attempting one-shot in-place FTS rebuild" not in caplog.text
+ # Structural damage quarantines the handle: typed error, sticky
+ # flag, later writes fail fast, and close() must not checkpoint
+ # the WAL over a damaged page image (the #90950 page-1 clobber).
+ from hermes_state import StateDbCorruptError
+
+ assert isinstance(caught.value, StateDbCorruptError)
+ assert db._db_corrupt is True
+ with pytest.raises(StateDbCorruptError):
+ db.append_message("s1", "user", "second write after corruption")
+ caplog.clear()
+ with caplog.at_level("WARNING", logger="hermes_state"):
+ db.close()
+ assert "Skipping the close-time WAL checkpoint" in caplog.text
finally:
db.close()
diff --git a/tests/state/test_session_git_metadata_generation.py b/tests/state/test_session_git_metadata_generation.py
index 2e1cb30988..b97727725c 100644
--- a/tests/state/test_session_git_metadata_generation.py
+++ b/tests/state/test_session_git_metadata_generation.py
@@ -237,7 +237,7 @@ def test_legacy_sessions_table_reconciles_generation_column(tmp_path):
assert "git_metadata_generation" in columns
assert reopened._conn.execute(
"SELECT version FROM schema_version"
- ).fetchone()[0] == SCHEMA_VERSION == 26
+ ).fetchone()[0] == SCHEMA_VERSION
reopened.create_session("session", "desktop", cwd="/repo")
assert reopened.update_session_cwd("session", "/repo") == 1
finally:
diff --git a/tests/state/test_state_db_lock_fail_closed.py b/tests/state/test_state_db_lock_fail_closed.py
new file mode 100644
index 0000000000..a98a23f7cf
--- /dev/null
+++ b/tests/state/test_state_db_lock_fail_closed.py
@@ -0,0 +1,160 @@
+"""Unopenable admission lock files must fail CLOSED (#100368).
+
+`state.db` has two cross-process admission authorities that gate destructive
+work on a file several Hermes processes share (gateway service, the Desktop
+app's `hermes serve` backend, CLI sessions, the TUI slash worker):
+
+* `hermes_state_common.fts_rebuild_admission` — full structural FTS rebuilds
+* `hermes_state._cross_process_repair_lock` — writable_schema surgery / VACUUM
+
+Both document themselves as fail-closed, and both honoured that only for a
+*timed-out* acquire. When the lock file could not be `open()`ed at all they
+yielded True and proceeded "with in-process serialisation only" — which is no
+cross-process authority whatsoever.
+
+That inversion is reachable exactly when it does the most damage. Creating the
+lock file needs a directory entry and an inode, so on a full disk `open()`
+raises ENOSPC — while a sibling process that opened ITS handle before the disk
+filled is still mid-rebuild or mid-surgery. Every process then ran concurrent
+destructive work on the same live DB, i.e. the precise interleaving PR #93200
+added these locks to prevent. #100368 reports that shape: a disk-full trigger,
+then a fresh corruption on every boot with other writers alive, and no
+re-corruption on a boot with zero other writers.
+
+These tests drive a real unopenable lock path (a directory where the code
+expects a file, so `open()` raises a genuine OSError from the kernel) rather
+than monkeypatching the helpers, and assert the deferral is honoured at both
+the primitive and the behavior level.
+"""
+
+import sqlite3
+import sys
+from pathlib import Path
+
+import pytest
+
+import hermes_state
+import hermes_state_common
+from hermes_state import SessionDB, repair_state_db_schema
+
+
+def _make_unopenable(lock_path: Path) -> None:
+ """Make ``open(lock_path, "a+b")`` raise a real OSError.
+
+ A directory standing where the code expects a regular file yields
+ IsADirectoryError on POSIX and PermissionError on Windows — both OSError,
+ both raised by the kernel. This stands in for the ENOSPC/EMFILE the field
+ reports hit, without needing to fill a real disk.
+ """
+ lock_path.unlink(missing_ok=True)
+ lock_path.mkdir(parents=True, exist_ok=True)
+ with pytest.raises(OSError):
+ open(lock_path, "a+b").close()
+
+
+# ── FTS rebuild authority ───────────────────────────────────────────────────
+
+
+def test_fts_admission_fails_closed_when_lock_file_is_unopenable(tmp_path):
+ """The primitive must refuse admission, not fall back to no authority."""
+ db_path = tmp_path / "state.db"
+ _make_unopenable(db_path.with_name(db_path.name + ".fts_rebuild.lock"))
+
+ with hermes_state_common.fts_rebuild_admission(db_path) as admitted:
+ assert admitted is False
+
+
+def test_fts_admission_still_admits_a_pathless_db(tmp_path):
+ """Guardrail: an in-memory store has no cross-process surface at all.
+
+ The fix must not turn the legitimate no-op case into a permanent deferral.
+ """
+ with hermes_state_common.fts_rebuild_admission(None) as admitted:
+ assert admitted is True
+
+
+def test_rebuild_fts_defers_when_lock_file_is_unopenable(tmp_path):
+ """Behavior: the rebuild entry point reports no progress and rebuilds nothing."""
+ db = SessionDB(db_path=tmp_path / "state.db")
+ if not db._fts_enabled:
+ db.close()
+ pytest.skip("FTS5 unavailable in this build")
+ try:
+ db.create_session("s1", source="test")
+ db.append_message("s1", "user", "hello world")
+
+ # Sanity: with an openable lock the rebuild really runs, so a 0 below
+ # is the deferral and not an unrelated no-op.
+ assert db.rebuild_fts() >= 1
+
+ _make_unopenable(
+ db.db_path.with_name(db.db_path.name + ".fts_rebuild.lock")
+ )
+ assert db.rebuild_fts() == 0
+ finally:
+ try:
+ db.close()
+ except Exception:
+ pass
+
+
+# ── Schema-surgery authority ────────────────────────────────────────────────
+
+
+def _build_healthy_db(db_path: Path) -> None:
+ db = SessionDB(db_path=db_path)
+ db.create_session("s1", source="test")
+ db.append_message("s1", "user", "hello world")
+ db.close()
+
+
+def _corrupt_duplicate_fts(db_path: Path) -> None:
+ """Inject a duplicate messages_fts row into sqlite_master.
+
+ Reproduces 'malformed database schema (messages_fts) - table
+ messages_fts already exists'.
+ """
+ conn = sqlite3.connect(str(db_path))
+ conn.execute("PRAGMA writable_schema=ON")
+ conn.execute(
+ "INSERT INTO sqlite_master (type, name, tbl_name, rootpage, sql) "
+ "SELECT type, name, tbl_name, rootpage, sql FROM sqlite_master "
+ "WHERE name='messages_fts'"
+ )
+ conn.commit()
+ conn.close()
+
+
+def test_repair_lock_fails_closed_when_lock_file_is_unopenable(tmp_path):
+ """The primitive must refuse the repair authority."""
+ db_path = tmp_path / "state.db"
+ _make_unopenable(db_path.with_name(db_path.name + ".repair.lock"))
+
+ with hermes_state._cross_process_repair_lock(db_path) as holding:
+ assert holding is False
+
+
+@pytest.mark.skipif(sys.platform == "win32", reason="writable_schema corruption harness")
+def test_repair_skips_surgery_when_lock_file_is_unopenable(tmp_path):
+ """Behavior: no writable_schema surgery, no forensic backup, DB untouched.
+
+ A full disk is the worst possible moment to start an unsynchronised
+ VACUUM on a live shared DB, and it is exactly when the lock file cannot
+ be created.
+ """
+ db_path = tmp_path / "state.db"
+ _build_healthy_db(db_path)
+ _corrupt_duplicate_fts(db_path)
+ assert hermes_state._db_opens_cleanly(db_path) is not None
+ before = db_path.read_bytes()
+
+ _make_unopenable(db_path.with_name(db_path.name + ".repair.lock"))
+
+ report = repair_state_db_schema(db_path)
+
+ assert report["repaired"] is False
+ assert "repair lock" in (report["error"] or "")
+ assert report["backup_path"] is None
+ assert not list(tmp_path.glob("state.db.malformed-backup-*"))
+ # The damaged image is left byte-identical for the next (authorised) pass.
+ assert db_path.read_bytes() == before
diff --git a/tests/test_desktop_update_windows_progress.py b/tests/test_desktop_update_windows_progress.py
index 031971570e..70708d754b 100644
--- a/tests/test_desktop_update_windows_progress.py
+++ b/tests/test_desktop_update_windows_progress.py
@@ -35,17 +35,24 @@ def _read_progress(url: str, deadline: float) -> dict[str, object]:
``urlopen(timeout=5)`` propagating TimeoutError was exactly the Aug 2026
flake (run 32440286339). Only a listener that stays unresponsive until
the deadline fails the test.
+
+ Per-attempt timeout is 1s, not 5s: a connection the kernel accepted into
+ the backlog before the runspace was serving never gets answered, and a 5s
+ wait on it burned half the readiness budget per attempt (two stale
+ attempts = red, run 33591547099). The script's own readiness handshake
+ now keeps that gap from reaching us, but the probe should not be able to
+ lose the whole budget to one dead socket either way.
"""
last_exc: Exception | None = None
attempted = False
while not attempted or time.monotonic() < deadline:
attempted = True
try:
- with urlopen(f"{url}progress", timeout=5) as response:
+ with urlopen(f"{url}progress", timeout=1) as response:
return json.loads(response.read().decode("utf-8"))
except (TimeoutError, OSError) as exc: # transient stall — retry
last_exc = exc
- time.sleep(0.2)
+ time.sleep(0.1)
raise AssertionError(
f"/progress unresponsive until deadline (last error: {last_exc!r})"
)
diff --git a/tests/test_env_loader_secret_sources.py b/tests/test_env_loader_secret_sources.py
index 303ed92268..c2959144c9 100644
--- a/tests/test_env_loader_secret_sources.py
+++ b/tests/test_env_loader_secret_sources.py
@@ -174,6 +174,85 @@ def test_cold_profile_bitwarden_uses_profile_bootstrap_without_global_env(
assert os.environ.get("ANTHROPIC_API_KEY") is None
+def test_single_profile_scoped_load_keeps_override_behavior(tmp_path, monkeypatch):
+ """Without multiplex, a scoped load keeps its historical override behaviour.
+
+ Ported from #77970 (@DonShelly): the guard must key on the multiplex flag,
+ not on the home override alone -- single-profile ``-p`` runs still load.
+ """
+ from agent import secret_scope
+ from hermes_constants import reset_hermes_home_override, set_hermes_home_override
+
+ monkeypatch.delenv("HERMES_TEST_SHARED_ADAPTER_CONFIG", raising=False)
+ other_home = tmp_path / "other"
+ other_home.mkdir()
+ (other_home / ".env").write_text("HERMES_TEST_SHARED_ADAPTER_CONFIG=second\n")
+
+ was_active = secret_scope.is_multiplex_active()
+ secret_scope.set_multiplex_active(False)
+ home_token = set_hermes_home_override(other_home)
+ try:
+ loaded = env_loader.load_hermes_dotenv(hermes_home=other_home)
+ finally:
+ secret_scope.set_multiplex_active(was_active)
+ reset_hermes_home_override(home_token)
+
+ try:
+ assert os.environ.get("HERMES_TEST_SHARED_ADAPTER_CONFIG") == "second"
+ assert (other_home / ".env") in loaded
+ finally:
+ os.environ.pop("HERMES_TEST_SHARED_ADAPTER_CONFIG", None)
+
+
+def test_multiplex_dotenv_load_hydrates_sources_without_global_env(
+ tmp_path, monkeypatch
+):
+ """The safe multiplex path must still refresh profile secret sources."""
+ from agent import secret_scope
+ import agent.secret_sources.bitwarden as bw_module
+ from agent.secret_sources import registry as reg_module
+ from hermes_constants import (
+ reset_hermes_home_override,
+ set_hermes_home_override,
+ )
+
+ monkeypatch.delenv("BWS_ACCESS_TOKEN", raising=False)
+ monkeypatch.delenv("ANTHROPIC_API_KEY", raising=False)
+ (tmp_path / ".env").write_text(
+ "BWS_ACCESS_TOKEN=profile-bootstrap\n", encoding="utf-8"
+ )
+ (tmp_path / "config.yaml").write_text(
+ "secrets:\n"
+ " bitwarden:\n"
+ " enabled: true\n"
+ " project_id: test-project\n"
+ " access_token_env: BWS_ACCESS_TOKEN\n",
+ encoding="utf-8",
+ )
+ monkeypatch.setattr(bw_module, "find_bws", lambda **_kw: Path("/fake/bws"))
+ monkeypatch.setattr(
+ bw_module,
+ "fetch_bitwarden_secrets",
+ lambda **_kw: ({"ANTHROPIC_API_KEY": "profile-provider-key"}, []),
+ )
+ reg_module._reset_registry_for_tests()
+
+ was_active = secret_scope.is_multiplex_active()
+ home_token = set_hermes_home_override(tmp_path)
+ secret_scope.set_multiplex_active(True)
+ try:
+ assert env_loader.load_hermes_dotenv(hermes_home=tmp_path) == []
+ finally:
+ secret_scope.set_multiplex_active(was_active)
+ reset_hermes_home_override(home_token)
+
+ assert env_loader.get_secret_source_values(tmp_path) == {
+ "ANTHROPIC_API_KEY": "profile-provider-key"
+ }
+ assert os.environ.get("BWS_ACCESS_TOKEN") is None
+ assert os.environ.get("ANTHROPIC_API_KEY") is None
+
+
def test_cold_profile_hydration_seeds_op_env_bootstrap(tmp_path, monkeypatch):
"""The .op.env bootstrap file must feed cold-profile hydration.
diff --git a/tests/test_hermes_state.py b/tests/test_hermes_state.py
index 28480c1d15..a52a90aa65 100644
--- a/tests/test_hermes_state.py
+++ b/tests/test_hermes_state.py
@@ -281,6 +281,83 @@ class TestConnectionLifecycle:
healed.close()
assert list(tmp_path.glob("*malformed-backup*"))
+ def test_read_only_open_retries_transient_wal_ioerr(self, tmp_path, monkeypatch):
+ """A transient SQLITE_IOERR on a read-only open must retry, not raise.
+
+ A ``mode=ro`` connection cannot perform WAL recovery (recovery would
+ need to write the -shm index, which read-only mode refuses), so a
+ concurrent checkpoint / WAL reset / frame-flush on the writer side can
+ surface "disk I/O error" to a reader on a perfectly healthy database
+ (#100436). The transition window is millisecond-scale; a bounded retry
+ must let the open succeed instead of 500-ing the /api/sessions poll
+ and every other read-only opener.
+ """
+ import sqlite3
+
+ from hermes_cli.sqlite_safe_read import has_live_connection
+
+ db_path = tmp_path / "state.db"
+ writable = SessionDB(db_path=db_path)
+ writable.create_session("wal-race", source="cli")
+ writable.close()
+
+ real_connect = hermes_state._connect_tracked_db
+ attempts = []
+
+ def flaky_connect(*args, **kwargs):
+ attempts.append(kwargs.get("uri"))
+ if len(attempts) == 1:
+ # First open lands inside the writer's WAL transition window.
+ raise sqlite3.OperationalError("disk I/O error")
+ return real_connect(*args, **kwargs)
+
+ monkeypatch.setattr(hermes_state, "_connect_tracked_db", flaky_connect)
+ # Keep the test fast: one backoff tick is enough; the retry budget
+ # itself is exercised by the attempt count below.
+ monkeypatch.setattr(hermes_state, "_READ_ONLY_IOERR_RETRY_BACKOFF_S", 0.0)
+
+ read_only = SessionDB(db_path=db_path, read_only=True)
+ try:
+ assert read_only._fts_enabled is True
+ matches = read_only.search_messages("wal-race")
+ finally:
+ read_only.close()
+
+ assert len(attempts) >= 2, "the transient IOERR must be retried"
+ assert has_live_connection(db_path) is False # no leaked connections
+
+ def test_read_only_open_exhausts_retry_budget_for_persistent_ioerr(
+ self, tmp_path, monkeypatch
+ ):
+ """A persistent SQLITE_IOERR must exhaust the budget and raise.
+
+ The retry exists to ride out a millisecond WAL transition — a
+ storage layer that keeps failing after the full budget is genuinely
+ broken and must surface the error (and not loop forever).
+ """
+ import sqlite3
+
+ db_path = tmp_path / "state.db"
+ writable = SessionDB(db_path=db_path)
+ writable.create_session("broken-disk", source="cli")
+ writable.close()
+
+ attempts = []
+
+ def bad_connect(*args, **kwargs):
+ attempts.append(1)
+ raise sqlite3.OperationalError("disk I/O error")
+
+ monkeypatch.setattr(hermes_state, "_connect_tracked_db", bad_connect)
+ monkeypatch.setattr(hermes_state, "_READ_ONLY_IOERR_RETRY_BACKOFF_S", 0.0)
+ budget = hermes_state._READ_ONLY_IOERR_RETRY_ATTEMPTS
+
+ with pytest.raises(sqlite3.OperationalError, match="disk I/O error"):
+ SessionDB(db_path=db_path, read_only=True)
+
+ # budget + 1 = the initial attempt plus `budget` retries.
+ assert len(attempts) == budget + 1
+
# =========================================================================
# Session lifecycle
@@ -1698,7 +1775,7 @@ class TestSchemaInit:
assert binding["user_id"] == "208214988"
assert binding["session_key"] == "telegram:dm:208214988:thread:17585"
assert binding["session_id"] == "topic-session"
- assert db.get_meta("telegram_dm_topic_schema_version") == "2"
+ assert db.get_meta("telegram_dm_topic_schema_version") == "3"
db.close()
@@ -2675,6 +2752,24 @@ class TestCompressionChainProjection:
assert db.get_compression_tip("mid1") == "tip1"
assert db.get_compression_tip("tip1") == "tip1"
+ def test_list_serves_full_lineage_ids_for_projected_rows(self, db):
+ """The projected tip row must carry every chain id. Root and tip
+ alone are not enough client-side: a persisted tile or route can hold
+ a MIDDLE segment's id (it was the tip when opened), and without the
+ intermediates that surface cannot prove it names this conversation —
+ which is how one chat ends up open twice after a compaction."""
+ import time as _time
+ self._build_compression_chain(db, _time.time() - 3600)
+ db.create_session("solo", "cli")
+ db.append_message("solo", "user", "standalone")
+ db._conn.commit()
+
+ sessions = db.list_sessions_rich(source="cli", limit=20)
+ tip_row = next(s for s in sessions if s["id"] == "tip1")
+ assert tip_row["_lineage_ids"] == ["root1", "mid1", "tip1"]
+ solo_row = next(s for s in sessions if s["id"] == "solo")
+ assert solo_row.get("_lineage_ids") is None
+
def test_list_surfaces_tip_for_compressed_root(self, db):
@@ -2912,6 +3007,7 @@ class TestVacuum:
def test_auto_maintenance_records_successful_vacuum(self, db, monkeypatch):
monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 3)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: 0.5) # ratio gate open
vacuum_calls = []
monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
@@ -2923,6 +3019,7 @@ class TestVacuum:
def test_auto_maintenance_skips_recent_vacuum(self, db, monkeypatch):
monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 3)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: 0.5) # ratio gate open
db.set_meta("last_vacuum", str(time.time()))
vacuum_calls = []
monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
@@ -2937,6 +3034,7 @@ class TestVacuum:
def test_auto_maintenance_retries_after_vacuum_interval(self, db, monkeypatch):
monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 3)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: 0.5) # ratio gate open
db.set_meta("last_vacuum", str(time.time() - 31 * 86400))
vacuum_calls = []
monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
@@ -2951,6 +3049,7 @@ class TestVacuum:
def test_auto_maintenance_retries_after_failed_vacuum(self, db, monkeypatch):
monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 3)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: 0.5) # ratio gate open
vacuum_calls = []
def fail_first_vacuum():
@@ -2971,6 +3070,97 @@ class TestVacuum:
assert vacuum_calls == [True, True]
assert db.get_meta("last_vacuum") is not None
+ # ── freelist-ratio gate (#54189) ─────────────────────────────────────
+ def test_auto_maintenance_skips_vacuum_below_freelist_ratio(self, db, monkeypatch):
+ """A prune that frees few pages on a dense DB must NOT trigger VACUUM."""
+ monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 1)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: 0.05)
+ vacuum_calls = []
+ monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
+
+ result = db.maybe_auto_prune_and_vacuum(min_interval_hours=0)
+
+ assert result["pruned"] == 1
+ assert result["vacuumed"] is False
+ assert result["freelist_ratio"] == 0.05
+ assert vacuum_calls == []
+ assert db.get_meta("last_vacuum") is None
+ # The prune itself still counts as a maintenance run.
+ assert db.get_meta("last_auto_prune") is not None
+
+ def test_auto_maintenance_vacuums_above_freelist_ratio(self, db, monkeypatch):
+ monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 1)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: 0.40)
+ vacuum_calls = []
+ monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
+
+ result = db.maybe_auto_prune_and_vacuum(min_interval_hours=0)
+
+ assert result["vacuumed"] is True
+ assert result["freelist_ratio"] == 0.40
+ assert vacuum_calls == [True]
+
+ def test_auto_maintenance_freelist_ratio_exactly_at_threshold_skips(self, db, monkeypatch):
+ """Gate is strictly greater-than: 25.0% reclaimable does not VACUUM."""
+ from hermes_state import AUTO_VACUUM_MIN_FREELIST_RATIO
+
+ monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 1)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: AUTO_VACUUM_MIN_FREELIST_RATIO)
+ vacuum_calls = []
+ monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
+
+ result = db.maybe_auto_prune_and_vacuum(min_interval_hours=0)
+
+ assert result["vacuumed"] is False
+ assert vacuum_calls == []
+
+ def test_auto_maintenance_unknown_freelist_ratio_falls_back_to_time_throttle(self, db, monkeypatch):
+ """If the pragmas cannot be read, don't silently disable VACUUM forever."""
+ monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 1)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: None)
+ vacuum_calls = []
+ monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
+
+ result = db.maybe_auto_prune_and_vacuum(min_interval_hours=0)
+
+ assert result["vacuumed"] is True
+ assert result["freelist_ratio"] is None
+ assert vacuum_calls == [True]
+
+ def test_auto_maintenance_ratio_gate_threshold_is_overridable(self, db, monkeypatch):
+ monkeypatch.setattr(db, "prune_sessions", lambda **_kwargs: 1)
+ monkeypatch.setattr(db, "_freelist_ratio", lambda: 0.10)
+ vacuum_calls = []
+ monkeypatch.setattr(db, "vacuum", lambda: vacuum_calls.append(True))
+
+ result = db.maybe_auto_prune_and_vacuum(
+ min_interval_hours=0, min_vacuum_freelist_ratio=0.05
+ )
+
+ assert result["vacuumed"] is True
+ assert vacuum_calls == [True]
+
+ def test_freelist_ratio_reads_real_pragmas(self, db):
+ """Real-DB check: freeing most of the file pushes the ratio past the gate."""
+ from hermes_state import AUTO_VACUUM_MIN_FREELIST_RATIO
+
+ db.create_session(session_id="keep", source="cli")
+ db.append_message(session_id="keep", role="user", content="hi")
+ for i in range(6):
+ sid = f"bulk{i}"
+ db.create_session(session_id=sid, source="cli")
+ for _ in range(20):
+ db.append_message(session_id=sid, role="assistant", content="z" * 4000)
+ db._conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
+ dense = db._freelist_ratio()
+ assert dense is not None and dense < AUTO_VACUUM_MIN_FREELIST_RATIO
+
+ for i in range(6):
+ db.delete_session(f"bulk{i}")
+ db._conn.execute("PRAGMA wal_checkpoint(TRUNCATE)")
+ sparse = db._freelist_ratio()
+ assert sparse is not None and sparse > AUTO_VACUUM_MIN_FREELIST_RATIO
+
def test_wal_size_limit_is_bounded(self, db):
"""journal_size_limit must be a finite bound, not SQLite's -1 default.
@@ -3101,7 +3291,8 @@ class TestAutoMaintenance:
)
db._conn.commit()
- def test_first_run_prunes_and_vacuums(self, db):
+ def test_first_run_prunes_and_skips_vacuum_when_little_reclaimable(self, db):
+ """Pruning two empty sessions frees almost nothing → prune yes, VACUUM no."""
self._make_old_ended(db, "old1", days_old=100)
self._make_old_ended(db, "old2", days_old=100)
db.create_session(session_id="new", source="cli") # active, must survive
@@ -3109,12 +3300,40 @@ class TestAutoMaintenance:
result = db.maybe_auto_prune_and_vacuum(retention_days=90)
assert result["skipped"] is False
assert result["pruned"] == 2
- assert result["vacuumed"] is True
+ assert result["vacuumed"] is False # freelist ratio gate (#54189)
+ assert result["freelist_ratio"] is not None
+ assert result["freelist_ratio"] <= 0.25
assert result.get("error") is None
assert db.get_session("old1") is None
assert db.get_session("old2") is None
assert db.get_session("new") is not None
+ def test_first_run_prunes_and_vacuums_when_mostly_reclaimable(self, db):
+ """Pruning the bulk of the file's pages crosses the 25% gate → VACUUM runs."""
+ db.create_session(session_id="new", source="cli") # active, must survive
+ db.append_message(session_id="new", role="user", content="hi")
+ for i in range(6):
+ sid = f"old{i}"
+ self._make_old_ended(db, sid, days_old=100)
+ for _ in range(20):
+ db.append_message(session_id=sid, role="assistant", content="z" * 4000)
+ # Keep the row aged: append_message bumps activity, prune ages by
+ # latest message, so push the message timestamps back too.
+ db._conn.execute(
+ "UPDATE messages SET timestamp = ? WHERE session_id = ?",
+ (time.time() - 100 * 86400, sid),
+ )
+ db._conn.commit()
+
+ result = db.maybe_auto_prune_and_vacuum(retention_days=90)
+ assert result["skipped"] is False
+ assert result["pruned"] == 6
+ assert result["freelist_ratio"] > 0.25
+ assert result["vacuumed"] is True
+ assert result.get("error") is None
+ assert db.get_session("new") is not None
+ assert db.get_meta("last_vacuum") is not None
+
def test_second_call_within_interval_skips(self, db):
self._make_old_ended(db, "old", days_old=100)
first = db.maybe_auto_prune_and_vacuum(
@@ -4260,6 +4479,63 @@ def test_gateway_session_recovery_does_not_cross_newer_reset_boundary(
) is None
+def test_peer_fallback_never_adopts_a_sibling_profiles_row(tmp_path, monkeypatch):
+ """#74285: the peer-tuple fallback is fenced by the store's own profile.
+
+ A Telegram DM peer tuple (chat_id == user_id, no thread) is identical for
+ every bot, so a legacy sibling-profile row sitting in this store — written
+ before the per-profile partition — must lose to the older own row, and
+ with no own row recovery must return nothing rather than the sibling's.
+ """
+ import hermes_state
+
+ root = tmp_path / "hermes"
+ root.mkdir()
+ monkeypatch.setenv("HERMES_HOME", str(root))
+ monkeypatch.setattr(hermes_state, "DEFAULT_DB_PATH", hermes_state._IMPORT_DEFAULT_DB_PATH)
+ store = SessionDB(db_path=root / "state.db") # owner: default
+ try:
+ peer = {"user_id": "42", "chat_id": "42", "chat_type": "dm"}
+ store.create_session("sibling", "telegram", session_key="agent:bot2:telegram:dm:42",
+ profile_name="bot2", **peer)
+ store.append_message("sibling", "user", "bot2's conversation")
+
+ def recover():
+ return store.find_latest_gateway_session_for_peer(
+ source="telegram", session_key="agent:main:telegram:dm:42", **peer
+ )
+
+ assert recover() is None # only the sibling exists: fail closed
+
+ store.create_session("own", "telegram", session_key="agent:main:telegram:dm:42:old", **peer)
+ store.append_message("own", "user", "default's conversation")
+ store._execute_write(
+ lambda c: c.execute("UPDATE sessions SET last_activity_at = 1 WHERE id = 'own'")
+ )
+ assert recover()["id"] == "own" # older own row beats newer sibling row
+ finally:
+ store.close()
+
+
+def test_child_inherits_parent_profile_only_within_its_key_namespace(db):
+ """#88381: parent→child ``profile_name`` COALESCE is fenced by ``agent::``.
+
+ A default child (``agent:main:``) forked from a sibling profile's row must
+ not be durably mislabelled as that profile's; same-namespace and keyless
+ (CLI/subagent) children keep inheriting.
+ """
+ db.create_session("parent", "telegram", session_key="agent:bot2:telegram:dm:42",
+ profile_name="bot2")
+ db.create_session("cross", "telegram", parent_session_id="parent",
+ session_key="agent:main:telegram:dm:42")
+ db.create_session("same", "telegram", parent_session_id="parent",
+ session_key="agent:bot2:telegram:dm:42:r2")
+ db.create_session("keyless", "cli", parent_session_id="parent")
+ assert db.get_session("cross")["profile_name"] is None
+ assert db.get_session("same")["profile_name"] == "bot2"
+ assert db.get_session("keyless")["profile_name"] == "bot2"
+
+
@@ -4574,6 +4850,66 @@ class TestGetMessagesPagination:
assert exc_info.value.message_count == 5
assert exc_info.value.limit == 4
+ def test_resume_safety_tip_only_counts_the_tip_segment(self, db):
+ """A deep compression lineage behind a small tip resumes tip-only.
+
+ The Desktop Bot Chat shape: many compaction segments (~29k rows of
+ lineage) and a small live tip. Callers that never materialize the
+ ancestors (deferred / omit_messages / lazy resume, tip-only model
+ restore) must be bounded by the tip alone, and the message must name
+ the scope it counted.
+ """
+ prev = None
+ for i in range(6):
+ sid = f"seg-{i}"
+ kwargs = {"parent_session_id": prev} if prev else {}
+ db.create_session(session_id=sid, source="tui", **kwargs)
+ db.append_messages_batch(
+ sid,
+ [{"role": "user", "content": f"{sid}-{j}"} for j in range(4)],
+ )
+ if i < 5:
+ db.end_session(sid, "compression")
+ prev = sid
+
+ assert db.get_resume_message_count("seg-5") == 24
+ assert db.get_resume_message_count("seg-5", tip_only=True) == 4
+ with pytest.raises(hermes_state.SessionResumeTooLargeError) as full:
+ db.assert_resume_safe("seg-5", max_messages=10)
+ assert "across its lineage" in str(full.value)
+ assert db.assert_resume_safe("seg-5", max_messages=10, tip_only=True) == 4
+ with pytest.raises(hermes_state.SessionResumeTooLargeError) as tip:
+ db.assert_resume_safe("seg-5", max_messages=3, tip_only=True)
+ assert tip.value.message_count == 4
+ assert "in its tip segment" in str(tip.value)
+
+ def test_resume_guard_counts_exactly_what_a_branch_resume_loads(self, db):
+ """An explicit /branch copy owns its transcript: the guard and the
+ resume readers must agree that its lineage is itself alone."""
+ db.create_session(session_id="parent", source="tui")
+ db.append_messages_batch(
+ "parent",
+ [{"role": "user", "content": f"parent-{i}"} for i in range(6)],
+ )
+ db.create_session(
+ session_id="branch",
+ source="tui",
+ parent_session_id="parent",
+ model_config={"_branched_from": "parent"},
+ )
+ db.append_messages_batch(
+ "branch",
+ [{"role": "user", "content": f"branch-{i}"} for i in range(2)],
+ )
+
+ _, display = db.get_resume_conversations("branch")
+ assert len(display) == 2
+ assert db.get_ancestor_display_prefix("branch") == []
+ # Before: the guard walked parent_session_id and counted 8, so a branch
+ # could be refused for rows a resume would never load.
+ assert db.get_resume_message_count("branch") == 2
+ assert db.assert_resume_safe("branch", max_messages=5) == 2
+
def test_export_safety_is_bounded_to_the_requested_active_segment(self, db):
db.create_session(session_id="root", source="cli")
db.append_messages_batch(
diff --git a/tests/test_install_sh_termux_python_bounds.py b/tests/test_install_sh_termux_python_bounds.py
new file mode 100644
index 0000000000..443873bf64
--- /dev/null
+++ b/tests/test_install_sh_termux_python_bounds.py
@@ -0,0 +1,220 @@
+"""Behavioral regression tests for Termux Python selection."""
+
+from __future__ import annotations
+
+import os
+import shutil
+import stat
+import subprocess
+import sys
+from pathlib import Path
+
+
+REPO_ROOT = Path(__file__).resolve().parent.parent
+INSTALL_SH = REPO_ROOT / "scripts" / "install.sh"
+SETUP_HERMES_SH = REPO_ROOT / "setup-hermes.sh"
+
+
+def _write_executable(path: Path, content: str) -> Path:
+ path.write_text(content)
+ path.chmod(path.stat().st_mode | stat.S_IXUSR)
+ return path
+
+
+def _write_fake_python(bin_dir: Path, name: str, version: str) -> Path:
+ return _write_executable(
+ bin_dir / name,
+ f"""#!{sys.executable}
+import os
+import sys
+
+VERSION = {version!r}
+VERSION_INFO = tuple(int(part) for part in VERSION.split('.')[:3]) + ('final', 0)
+
+if len(sys.argv) >= 2 and sys.argv[1] == '--version':
+ print(f'Python {{VERSION}}')
+ raise SystemExit(0)
+
+if len(sys.argv) >= 3 and sys.argv[1] == '-c':
+ sys.version = f'{{VERSION}} (fake)'
+ sys.version_info = VERSION_INFO
+ exec(sys.argv[2], {{'__name__': '__main__'}})
+ raise SystemExit(0)
+
+if len(sys.argv) >= 3 and sys.argv[1:3] == ['-m', 'venv']:
+ target = sys.argv[3] if len(sys.argv) >= 4 else 'venv'
+ bin_path = os.path.join(target, 'bin')
+ os.makedirs(bin_path, exist_ok=True)
+ python_path = os.path.join(bin_path, 'python')
+ with open(python_path, 'w', encoding='utf-8') as handle:
+ handle.write('''#!/bin/sh\nif [ "${{1:-}}" = '-m' ] && [ "${{2:-}}" = 'pip' ]; then\n exit 0\nfi\nexit 0\n''')
+ os.chmod(python_path, 0o755)
+ raise SystemExit(0)
+
+raise SystemExit(0)
+""",
+ )
+
+
+def _write_unsupported_explicit_pythons(bin_dir: Path, *except_names: str) -> None:
+ for name in ("python3.11", "python3.12", "python3.13"):
+ if name not in except_names and not (bin_dir / name).exists():
+ _write_fake_python(bin_dir, name, "3.14.6")
+
+
+def _write_termux_command_stubs(bin_dir: Path) -> None:
+ _write_executable(
+ bin_dir / "uname",
+ "#!/bin/sh\n[ \"${1:-}\" = '-s' ] && echo Linux || echo Linux\n",
+ )
+ if not (bin_dir / "pkg").exists():
+ _write_executable(bin_dir / "pkg", "#!/bin/sh\nexit 0\n")
+ _write_executable(bin_dir / "git", "#!/bin/sh\necho 'git version 2.50.0'\n")
+ _write_executable(bin_dir / "node", "#!/bin/sh\necho 'v22.12.0'\n")
+ _write_executable(bin_dir / "npm", "#!/bin/sh\nexit 0\n")
+ _write_executable(bin_dir / "curl", "#!/bin/sh\nexit 0\n")
+ _write_executable(bin_dir / "rg", "#!/bin/sh\nexit 0\n")
+
+
+def _termux_env(tmp_path: Path, bin_dir: Path) -> dict[str, str]:
+ prefix = tmp_path / "com.termux" / "files" / "usr"
+ (prefix / "bin").mkdir(parents=True)
+ env = os.environ.copy()
+ env.update({
+ "ANDROID_API_LEVEL": "35",
+ "HOME": str(tmp_path / "home"),
+ "HERMES_HOME": str(tmp_path / "home" / ".hermes"),
+ "PATH": f"{bin_dir}{os.pathsep}{env.get('PATH', os.defpath)}",
+ "PREFIX": str(prefix),
+ "TERMUX_VERSION": "0.118.0",
+ })
+ return env
+
+
+def _run_install_prerequisites(tmp_path: Path) -> subprocess.CompletedProcess[str]:
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir(exist_ok=True)
+ _write_termux_command_stubs(bin_dir)
+ env = _termux_env(tmp_path, bin_dir)
+ bash = shutil.which("bash") or "/bin/bash"
+ return subprocess.run(
+ [bash, str(INSTALL_SH), "--stage", "prerequisites", "--non-interactive"],
+ env=env,
+ text=True,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.STDOUT,
+ check=False,
+ )
+
+
+def _copy_setup_checkout(tmp_path: Path) -> Path:
+ checkout = tmp_path / "checkout"
+ checkout.mkdir()
+ shutil.copy2(SETUP_HERMES_SH, checkout / "setup-hermes.sh")
+ return checkout
+
+
+def _run_setup(tmp_path: Path) -> subprocess.CompletedProcess[str]:
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir(exist_ok=True)
+ _write_termux_command_stubs(bin_dir)
+ env = _termux_env(tmp_path, bin_dir)
+ checkout = _copy_setup_checkout(tmp_path)
+ bash = shutil.which("bash") or "/bin/bash"
+ return subprocess.run(
+ [bash, str(checkout / "setup-hermes.sh")],
+ env=env,
+ input="n\n",
+ text=True,
+ stdout=subprocess.PIPE,
+ stderr=subprocess.STDOUT,
+ check=False,
+ )
+
+
+def test_install_stage_prefers_compatible_minor_over_unsupported_default(
+ tmp_path: Path,
+) -> None:
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir()
+ _write_fake_python(bin_dir, "python3.11", "3.11.15")
+ _write_fake_python(bin_dir, "python", "3.14.6")
+
+ result = _run_install_prerequisites(tmp_path)
+
+ assert result.returncode == 0, result.stdout
+ assert "Python found: Python 3.11.15" in result.stdout
+
+
+def test_install_stage_rejects_post_install_unsupported_default(tmp_path: Path) -> None:
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir()
+ _write_fake_python(bin_dir, "python", "3.14.6")
+ _write_unsupported_explicit_pythons(bin_dir)
+
+ result = _run_install_prerequisites(tmp_path)
+
+ assert result.returncode == 1
+ assert "Termux Python Python 3.14.6 is not supported" in result.stdout
+ assert "Hermes requires Python >=3.11,<3.14" in result.stdout
+ assert "pkg install tur-repo && pkg install python3.13" in result.stdout
+
+
+def test_install_stage_provisions_supported_python_from_tur(tmp_path: Path) -> None:
+ """When the default Termux python is too new, the installer falls back to
+ the Termux User Repository (TUR) and picks up a supported interpreter that
+ `pkg install python3.13` provides."""
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir()
+ _write_fake_python(bin_dir, "python", "3.14.6")
+ # Shadow any host python3.11/3.12/3.13 so the candidate scan can't find a
+ # supported interpreter before the TUR fallback runs.
+ _write_unsupported_explicit_pythons(bin_dir)
+
+ # Stateful pkg stub: `pkg install -y python3.13` drops a supported fake
+ # interpreter into PATH, mimicking a successful TUR package install.
+ staged = tmp_path / "staged"
+ staged.mkdir()
+ _write_fake_python(staged, "python3.13", "3.13.7")
+ _write_executable(
+ bin_dir / "pkg",
+ "#!/bin/sh\n"
+ "for arg in \"$@\"; do\n"
+ f" if [ \"$arg\" = 'python3.13' ]; then cp {staged}/python3.13 {bin_dir}/python3.13; fi\n"
+ "done\n"
+ "exit 0\n",
+ )
+
+ result = _run_install_prerequisites(tmp_path)
+
+ assert result.returncode == 0, result.stdout
+ assert "Python installed from TUR: Python 3.13.7" in result.stdout
+
+
+def test_setup_script_prefers_compatible_minor_over_unsupported_default(
+ tmp_path: Path,
+) -> None:
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir()
+ _write_fake_python(bin_dir, "python3.11", "3.14.6")
+ _write_fake_python(bin_dir, "python3.12", "3.12.11")
+ _write_fake_python(bin_dir, "python", "3.14.6")
+
+ result = _run_setup(tmp_path)
+
+ assert result.returncode == 0, result.stdout
+ assert "Python 3.12.11 found" in result.stdout
+ assert (tmp_path / "checkout" / "venv" / "bin" / "python").exists()
+
+
+def test_setup_script_rejects_unsupported_default(tmp_path: Path) -> None:
+ bin_dir = tmp_path / "bin"
+ bin_dir.mkdir()
+ _write_fake_python(bin_dir, "python", "3.14.6")
+ _write_unsupported_explicit_pythons(bin_dir)
+
+ result = _run_setup(tmp_path)
+
+ assert result.returncode == 1
+ assert "Termux Python Python 3.14.6 is not supported" in result.stdout
+ assert "Hermes requires Python >=3.11,<3.14" in result.stdout
diff --git a/tests/test_managed_runtime_resolution.py b/tests/test_managed_runtime_resolution.py
index e52a4ddcfb..e3943e351a 100644
--- a/tests/test_managed_runtime_resolution.py
+++ b/tests/test_managed_runtime_resolution.py
@@ -27,6 +27,7 @@ from __future__ import annotations
import ast
import functools
+import os
from pathlib import Path
import pytest
@@ -122,21 +123,23 @@ def _iter_which_calls(tree: ast.AST):
def _source_files() -> list[Path]:
files: list[Path] = []
- for path in REPO_ROOT.rglob("*.py"):
- rel = path.relative_to(REPO_ROOT)
- if rel.parts and rel.parts[0] in _EXEMPT_DIRS:
- continue
- # Skip packaging copies of the source tree (sdist extractions like
- # hermes_agent-0.20.5/, build/ and *.egg-info dirs). CI jobs that
- # build the wheel leave one in the workspace; scanning it re-finds
- # every already-exempted call site under a versioned path prefix
- # that can never match an _ALLOWED key, failing the guard on code
- # that was never touched. A dir is a packaging copy iff its top
- # level carries PKG-INFO (sdist/egg metadata) or it is a build/
- # dist output directory.
- if rel.parts and _is_packaging_copy(rel.parts[0]):
- continue
- files.append(path)
+ # os.walk instead of Path.rglob: rglob raises FileNotFoundError when a
+ # directory vanishes mid-scan — a sibling CI job's sdist extraction
+ # (hermes_agent-/) gets created and deleted concurrently, and
+ # that TOCTOU failed this guard on runs 33531869442/33455779041-era
+ # workspaces. os.walk tolerates vanishing dirs (onerror=None), and
+ # pruning exempt/packaging dirs at the top level also skips their
+ # subtrees entirely.
+ for dirpath, dirnames, filenames in os.walk(REPO_ROOT):
+ rel_dir = Path(dirpath).relative_to(REPO_ROOT)
+ if rel_dir == Path("."):
+ dirnames[:] = [
+ d for d in dirnames
+ if d not in _EXEMPT_DIRS and not _is_packaging_copy(d)
+ ]
+ for fname in filenames:
+ if fname.endswith(".py"):
+ files.append(Path(dirpath) / fname)
return files
diff --git a/tests/test_model_tools.py b/tests/test_model_tools.py
index a967f61575..9e1fa9886e 100644
--- a/tests/test_model_tools.py
+++ b/tests/test_model_tools.py
@@ -227,7 +227,7 @@ class TestHandleFunctionCall:
class TestAgentLoopTools:
def test_expected_tools_in_set(self):
- assert "todo" in _AGENT_LOOP_TOOLS
+ assert "todo_list" in _AGENT_LOOP_TOOLS
assert "memory" in _AGENT_LOOP_TOOLS
assert "session_search" in _AGENT_LOOP_TOOLS
assert "delegate_task" in _AGENT_LOOP_TOOLS
diff --git a/tests/test_session_vacuum_config.py b/tests/test_session_vacuum_config.py
index d231996b59..43adfdacba 100644
--- a/tests/test_session_vacuum_config.py
+++ b/tests/test_session_vacuum_config.py
@@ -8,6 +8,87 @@ def test_default_config_exposes_vacuum_interval():
assert DEFAULT_CONFIG["sessions"]["min_vacuum_interval_days"] == 30
+def test_default_config_auto_prune_on_with_90_day_retention():
+ """#54189: state.db retention is ON by default (ended sessions, 90 days)."""
+ from hermes_cli.config import DEFAULT_CONFIG
+
+ sessions = DEFAULT_CONFIG["sessions"]
+ assert sessions["auto_prune"] is True
+ assert sessions["retention_days"] == 90
+ assert sessions["vacuum_after_prune"] is True
+
+
+def test_fresh_config_runs_auto_prune_at_startup(monkeypatch, tmp_path: Path):
+ """A config.yaml with NO ``sessions:`` keys must reach the prune call with the
+ new defaults (the loader deep-merges DEFAULT_CONFIG)."""
+ import cli
+ import hermes_cli.config
+ import hermes_constants
+ from hermes_cli.config import DEFAULT_CONFIG
+
+ session_db = MagicMock()
+ session_db.get_meta.return_value = "already-done"
+ # Simulate load_config() on a fresh home: only defaults for the section.
+ monkeypatch.setattr(
+ hermes_cli.config,
+ "load_config",
+ lambda: {"sessions": dict(DEFAULT_CONFIG["sessions"])},
+ )
+ monkeypatch.setattr(hermes_constants, "get_hermes_home", lambda: tmp_path)
+
+ cli._run_state_db_auto_maintenance(session_db)
+
+ session_db.maybe_auto_prune_and_vacuum.assert_called_once_with(
+ retention_days=90,
+ min_interval_hours=24,
+ min_vacuum_interval_days=30,
+ vacuum=True,
+ sessions_dir=tmp_path / "sessions",
+ )
+
+
+def test_explicit_auto_prune_false_is_respected(monkeypatch, tmp_path: Path):
+ """Migration guard: an install that explicitly opted out keeps its choice."""
+ import cli
+ import hermes_cli.config
+ import hermes_constants
+
+ session_db = MagicMock()
+ session_db.get_meta.return_value = "already-done"
+ monkeypatch.setattr(
+ hermes_cli.config,
+ "load_config",
+ lambda: {"sessions": {"auto_prune": False, "retention_days": 90}},
+ )
+ monkeypatch.setattr(hermes_constants, "get_hermes_home", lambda: tmp_path)
+
+ cli._run_state_db_auto_maintenance(session_db)
+
+ session_db.maybe_auto_prune_and_vacuum.assert_not_called()
+
+
+def test_shipped_template_does_not_pin_sessions_keys():
+ """Installers copy cli-config.yaml.example verbatim into config.yaml, so any
+ uncommented ``sessions:`` value there becomes an EXPLICIT user setting that
+ would freeze the retention defaults. The template must leave them commented
+ so code defaults (and future flips) apply."""
+ import yaml
+
+ template = Path(__file__).resolve().parents[1] / "cli-config.yaml.example"
+ data = yaml.safe_load(template.read_text(encoding="utf-8")) or {}
+ assert "sessions" not in data
+
+
+def test_loader_yields_new_defaults_for_fresh_home(monkeypatch, tmp_path: Path):
+ """Real load_config() against an empty HERMES_HOME → auto_prune on, 90 days."""
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ from hermes_cli.config import load_config
+
+ sessions = load_config().get("sessions") or {}
+ assert sessions.get("auto_prune") is True
+ assert sessions.get("retention_days") == 90
+
+
def test_cli_auto_maintenance_forwards_vacuum_interval(monkeypatch, tmp_path: Path):
import cli
import hermes_cli.config
diff --git a/tests/test_sqlite_wal_reset_gate.py b/tests/test_sqlite_wal_reset_gate.py
index a717b6bb5e..1dafce8115 100644
--- a/tests/test_sqlite_wal_reset_gate.py
+++ b/tests/test_sqlite_wal_reset_gate.py
@@ -69,6 +69,7 @@ class TestApplyWalWalResetGate:
assert mode == "delete"
assert conn.execute("PRAGMA journal_mode").fetchone()[0].lower() == "delete"
assert any("instead of enabling WAL" in r.getMessage() for r in caplog.records)
+ assert any(sys.executable in r.getMessage() for r in caplog.records)
conn.close()
def test_existing_wal_left_alone_when_vulnerable(
diff --git a/tests/test_state_db_notadb_fail_closed.py b/tests/test_state_db_notadb_fail_closed.py
index fbdf376fef..9fce19e3e1 100644
--- a/tests/test_state_db_notadb_fail_closed.py
+++ b/tests/test_state_db_notadb_fail_closed.py
@@ -1,9 +1,10 @@
"""Tests for fail-closed state.db NOTADB handling and journal-mode EIO retries.
-Covers the two independently-valuable pieces salvaged from the state.db
-hardening rollup:
+Covers:
* fail closed when a live write connection reports ``file is not a database``;
+* the write-path SQLITE_IOERR retry boundary: admitted only when the callback
+ has provably not run, never by closing and replaying;
* transient ``disk i/o error`` retry in ``_on_disk_journal_mode`` so a
one-shot EIO doesn't push callers onto the fail-closed unknown-mode branch.
"""
@@ -13,7 +14,7 @@ from unittest.mock import MagicMock
import pytest
-from hermes_state import SessionDB, _on_disk_journal_mode
+from hermes_state import SessionDB, StateDbCorruptError, _on_disk_journal_mode
class _NotADbOnce:
@@ -41,14 +42,100 @@ class TestFailClosedAfterNotADb:
reopen = MagicMock()
monkeypatch.setattr("hermes_state._connect_tracked_db", reopen)
db._conn = _NotADbOnce(real_conn)
- with pytest.raises(sqlite3.DatabaseError, match="not a database"):
+ with pytest.raises(sqlite3.DatabaseError, match="not a database") as excinfo:
db.create_session(session_id="s2", source="cli", model="test")
reopen.assert_not_called()
+ # NOTADB on a live write is structural: the handle is quarantined.
+ assert isinstance(excinfo.value, StateDbCorruptError)
+ assert db._db_corrupt is True
finally:
db._conn = real_conn
db.close()
+class TestWriteIoerrRetryBoundary:
+ """IOERR retry is admitted by EFFECT POSITION, not error spelling.
+
+ ``_execute_write`` owns non-idempotent transcript/counter mutations, so
+ replaying its callback is only safe when the first attempt provably did
+ nothing. SQLite does not define ``SQLITE_IOERR`` as pre-effect-only (an
+ IOERR at fsync/commit may or may not have landed), so the admission gate
+ is "did the callback start", not "does the message say disk I/O".
+ """
+
+ def test_ioerr_on_begin_retries_because_the_callback_never_ran(self, tmp_path):
+ db = SessionDB(db_path=tmp_path / "state.db")
+ real_conn = db._conn
+ try:
+
+ class _BeginIoerrOnce:
+ def __init__(self, conn):
+ self._real = conn
+ self.begins = 0
+
+ def execute(self, sql, *args, **kwargs):
+ if str(sql).strip().upper().startswith("BEGIN") and self.begins == 0:
+ self.begins += 1
+ raise sqlite3.OperationalError("disk I/O error")
+ return self._real.execute(sql, *args, **kwargs)
+
+ def __getattr__(self, name):
+ return getattr(self._real, name)
+
+ proxy = _BeginIoerrOnce(real_conn)
+ db._conn = proxy
+ db.create_session(session_id="s1", source="cli", model="test")
+ assert proxy.begins == 1
+ finally:
+ db._conn = real_conn
+
+ rows = db.list_sessions_rich(limit=10, compact_rows=True)
+ assert [row["id"] for row in rows] == ["s1"]
+ db.close()
+
+ def test_ioerr_after_the_callback_mutates_does_not_replay(self, tmp_path):
+ """Settlement is unknown once the callback has run — surface, don't rerun."""
+ db = SessionDB(db_path=tmp_path / "state.db")
+ try:
+ calls = []
+
+ def mutate_then_fail(conn):
+ calls.append(1)
+ conn.execute(
+ "INSERT INTO sessions (id, started_at, source) VALUES (?, ?, ?)",
+ (f"row-{len(calls)}", 1.0, "cli"),
+ )
+ raise sqlite3.OperationalError("disk I/O error")
+
+ with pytest.raises(sqlite3.OperationalError, match="disk I/O error"):
+ db._execute_write(mutate_then_fail)
+
+ assert calls == [1], "a started write must not be replayed"
+ assert db.list_sessions_rich(limit=10, compact_rows=True) == []
+ finally:
+ db.close()
+
+ def test_write_ioerr_never_closes_the_connection(self, tmp_path, monkeypatch):
+ """close() cancels this process's POSIX locks for every sibling fd."""
+ db = SessionDB(db_path=tmp_path / "state.db")
+ try:
+ closed = []
+ monkeypatch.setattr(
+ type(db._conn), "close", lambda self: closed.append(1), raising=False
+ )
+
+ def always_ioerr(conn):
+ raise sqlite3.OperationalError("disk I/O error")
+
+ with pytest.raises(sqlite3.OperationalError):
+ db._execute_write(always_ioerr)
+
+ assert closed == []
+ assert db._conn is not None
+ finally:
+ db.close()
+
+
class TestOnDiskJournalModeEioRetry:
def _conn_raising_then(self, failures, result_rows):
conn = MagicMock()
diff --git a/tests/test_toolsets.py b/tests/test_toolsets.py
index 73ce7a9eab..8499335d6a 100644
--- a/tests/test_toolsets.py
+++ b/tests/test_toolsets.py
@@ -265,7 +265,7 @@ class TestResolveToolsetIncludeRegistry:
finally:
registry.deregister("__probe_registry_only_tool__")
- assert static == {"terminal", "process"}, static
+ assert static == {"terminal", "process_manage"}, static
# Registered into 'terminal' but not part of the static definition — it
# must only appear in the merged view.
assert "__probe_registry_only_tool__" in merged
@@ -275,7 +275,7 @@ class TestResolveToolsetIncludeRegistry:
def test_static_view_threads_through_includes(self):
# 'debugging' has direct tools [terminal, process] and includes [web, file]
static = set(resolve_toolset("debugging", include_registry=False))
- assert {"terminal", "process"} <= static
+ assert {"terminal", "process_manage"} <= static
assert "web_search" in static
assert "read_file" in static
diff --git a/tests/test_tui_gateway_server.py b/tests/test_tui_gateway_server.py
index 08ecdcfd6e..b5ea0fccdc 100644
--- a/tests/test_tui_gateway_server.py
+++ b/tests/test_tui_gateway_server.py
@@ -633,7 +633,7 @@ def test_slash_exec_compress_flag_on_applies_host_control_mirror(monkeypatch):
def __init__(self):
self.controls = []
- def control(self, sid, *, route_name, payload=None, wait=True, timeout=30.0):
+ def control(self, sid, *, route_name, payload=None, wait=True, timeout=30.0, on_late_ack=None):
self.controls.append((sid, route_name, dict(payload or {}), wait))
return {
"type": "control.ack",
@@ -5534,7 +5534,7 @@ def test_superseded_runtime_finalized_without_reclaimed_broadcast(monkeypatch):
# mark it finalized-for-lookup via a different stored key is wrong —
# instead simulate the mint race by removing it from lookup).
old["_finalized"] = False
- monkeypatch.setattr(server, "_find_live_session_by_key", lambda _k: None)
+ monkeypatch.setattr(server, "_find_live_session_by_key", lambda _k, *_a: None)
result = server._claim_or_reuse_live("new-sid", "stored-super", fresh, None)
@@ -5765,6 +5765,147 @@ def test_ws_orphan_reap_disabled_when_grace_zero(monkeypatch):
assert fired["timer"] is False
+def test_ws_orphan_reap_defers_running_turn_with_fresh_activity(monkeypatch):
+ """#98028/#100325: a client-absent turn whose activity clock is fresh is
+ NOT interrupted — it keeps running detached and the reaper re-polls at the
+ grace interval. Once the clock goes stale the wedged-turn interrupt fires,
+ and after the turn settles the session is reaped as before."""
+ callbacks = []
+ delays = []
+ interrupted = []
+ torn_down = []
+
+ class _Timer:
+ def __init__(self, delay, callback):
+ delays.append(delay)
+ callbacks.append(callback)
+ self.daemon = False
+
+ def start(self):
+ return None
+
+ activity = {"seconds_since_activity": 1.0}
+ agent = types.SimpleNamespace(
+ get_activity_summary=lambda: dict(activity),
+ interrupt=lambda message=None: interrupted.append("interrupted"),
+ )
+
+ class _DeadThread:
+ def is_alive(self):
+ return False
+
+ session = _session(
+ agent=agent,
+ transport=server._detached_ws_transport,
+ running=True,
+ _run_thread=_DeadThread(),
+ )
+ server._sessions["fresh-sid"] = session
+ monkeypatch.setattr(server, "_WS_ORPHAN_REAP_GRACE_S", 0.01)
+ monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 300.0)
+ monkeypatch.setattr(server.threading, "Timer", _Timer)
+ monkeypatch.setattr(server, "_load_cfg", lambda: {})
+ monkeypatch.setattr(
+ server,
+ "_teardown_popped_session",
+ lambda claimed, *, end_reason: torn_down.append((claimed, end_reason)) or True,
+ )
+
+ try:
+ server._schedule_ws_orphan_reap("fresh-sid")
+
+ # Two grace cycles with fresh activity: no interrupt, reschedule at
+ # the GRACE interval (not the 1s interrupt-settle poll).
+ for _ in range(2):
+ callbacks.pop(0)()
+ assert interrupted == []
+ assert not session.get("_client_gone_interrupt_requested")
+ assert delays[-1] == server._WS_ORPHAN_REAP_GRACE_S
+ assert "fresh-sid" in server._sessions
+
+ # Activity goes stale (turn wedged) -> interrupt fires on next poll.
+ activity["seconds_since_activity"] = 301.0
+ callbacks.pop(0)()
+ assert interrupted == ["interrupted"]
+ assert session["_client_gone_interrupt_requested"] is True
+
+ # Turn settles -> reap proceeds exactly as today.
+ session["running"] = False
+ callbacks.pop(0)()
+ assert "fresh-sid" not in server._sessions
+ assert torn_down == [(session, "ws_orphan_reap")]
+ finally:
+ server._sessions.pop("fresh-sid", None)
+
+
+def test_ws_orphan_activity_gate_zero_restores_interrupt_at_grace(monkeypatch):
+ """ws_orphan_activity_stale_s=0 opts out: fresh activity no longer defers
+ the client-gone interrupt (pre-#98028 behaviour)."""
+ callbacks = []
+ interrupted = []
+
+ class _Timer:
+ def __init__(self, _delay, callback):
+ callbacks.append(callback)
+ self.daemon = False
+
+ def start(self):
+ return None
+
+ class _LiveThread:
+ def is_alive(self):
+ return True
+
+ agent = types.SimpleNamespace(
+ get_activity_summary=lambda: {"seconds_since_activity": 0.5},
+ interrupt=lambda message=None: interrupted.append("interrupted"),
+ )
+ session = _session(
+ agent=agent,
+ transport=server._detached_ws_transport,
+ running=True,
+ _run_thread=_LiveThread(),
+ )
+ server._sessions["optout-sid"] = session
+ monkeypatch.setattr(server, "_WS_ORPHAN_REAP_GRACE_S", 0.01)
+ monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 0.0)
+ monkeypatch.setattr(server.threading, "Timer", _Timer)
+ monkeypatch.setattr(server, "_load_cfg", lambda: {})
+
+ try:
+ server._schedule_ws_orphan_reap("optout-sid")
+ callbacks.pop(0)()
+ assert interrupted == ["interrupted"]
+ assert session["_client_gone_interrupt_requested"] is True
+ finally:
+ server._sessions.pop("optout-sid", None)
+
+
+def test_ws_orphan_activity_gate_unreadable_summary_stays_eligible(monkeypatch):
+ """A broken/opaque activity summary must fail CLOSED (not fresh): the
+ wedged-turn interrupt-at-grace safety net is preserved."""
+
+ def _boom():
+ raise RuntimeError("summary unavailable")
+
+ agent = types.SimpleNamespace(get_activity_summary=_boom)
+ monkeypatch.setattr(server, "_WS_ORPHAN_ACTIVITY_STALE_S", 300.0)
+ assert server._ws_orphan_turn_activity_is_fresh({"agent": agent}) is False
+ # No agent / no summary method: same conservative answer.
+ assert server._ws_orphan_turn_activity_is_fresh({"agent": None}) is False
+ assert (
+ server._ws_orphan_turn_activity_is_fresh(
+ {"agent": types.SimpleNamespace()}
+ )
+ is False
+ )
+ # Never-stamped clock (None) is not fresh either.
+ agent2 = types.SimpleNamespace(
+ get_activity_summary=lambda: {"seconds_since_activity": None}
+ )
+ assert server._ws_orphan_turn_activity_is_fresh({"agent": agent2}) is False
+
+
def test_init_session_fires_reset_hook(monkeypatch):
hooks = []
@@ -8363,7 +8504,7 @@ def test_config_set_fast_updates_live_agent_session_scoped(monkeypatch):
monkeypatch.setattr(server, "_emit", lambda *args: emits.append(args))
monkeypatch.setattr(
"hermes_cli.models.resolve_fast_mode_overrides",
- lambda _model_id: {"service_tier": "priority"},
+ lambda _model_id, **_route: {"service_tier": "priority"},
)
try:
@@ -8442,7 +8583,7 @@ def test_config_set_fast_rejects_unsupported_model(monkeypatch):
)
monkeypatch.setattr(
"hermes_cli.models.resolve_fast_mode_overrides",
- lambda _model_id: None,
+ lambda _model_id, **_route: None,
)
try:
@@ -8762,7 +8903,7 @@ def test_enable_gateway_prompts_sets_gateway_env(monkeypatch):
def test_setup_status_reports_provider_config(monkeypatch):
- monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda: False)
+ monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda **_kw: False)
resp = server.handle_request({"id": "1", "method": "setup.status", "params": {}})
@@ -8784,7 +8925,7 @@ def test_probe_credentials_allows_keyless_custom_runtime():
def test_setup_runtime_check_rejects_empty_runtime_key(monkeypatch):
- monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda: True)
+ monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda **_kw: True)
monkeypatch.setattr(
"hermes_cli.runtime_provider.resolve_runtime_provider",
lambda requested=None: {
@@ -8806,7 +8947,7 @@ def test_setup_runtime_check_rejects_empty_runtime_key(monkeypatch):
def test_setup_runtime_check_allows_no_key_custom_runtime(monkeypatch):
- monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda: True)
+ monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda **_kw: True)
monkeypatch.setattr(
"hermes_cli.runtime_provider.resolve_runtime_provider",
lambda requested=None: {
@@ -8823,7 +8964,7 @@ def test_setup_runtime_check_allows_no_key_custom_runtime(monkeypatch):
def test_setup_runtime_check_rejects_implicit_bedrock_when_unconfigured(monkeypatch):
- monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda: False)
+ monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda **_kw: False)
monkeypatch.setattr(
"hermes_cli.runtime_provider.resolve_runtime_provider",
lambda requested=None: {
@@ -8841,7 +8982,7 @@ def test_setup_runtime_check_rejects_implicit_bedrock_when_unconfigured(monkeypa
def test_setup_runtime_check_honors_requested_provider(monkeypatch):
"""Onboarding must be able to validate the provider the user just connected."""
- monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda: True)
+ monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda **_kw: True)
def fake_resolve(requested=None, **kwargs):
if requested == "nous":
@@ -8872,6 +9013,67 @@ def test_setup_runtime_check_honors_requested_provider(monkeypatch):
assert default["result"]["provider"] == "anthropic"
+def test_setup_readiness_scopes_to_requested_profile(monkeypatch, tmp_path):
+ """#94071: the Desktop preflights a freshly created bot on its target
+ backend. ``profile`` binds THAT profile's home + .env — launch-process
+ credentials must not make an unconfigured bot look ready, and the bot's
+ own .env must be what the strict check sees."""
+ from agent import secret_scope
+ from hermes_constants import get_hermes_home
+
+ bot_home = tmp_path / "profiles" / "bot"
+ bot_home.mkdir(parents=True)
+ monkeypatch.setenv("OPENROUTER_API_KEY", "sk-or-launch-profile-secret-0000")
+ monkeypatch.setattr("hermes_cli.profiles.profile_exists", lambda name: name == "bot")
+ monkeypatch.setattr(server, "_profile_home", lambda profile: bot_home if profile == "bot" else None)
+ seen = {}
+
+ def fake_resolve(requested=None, **kwargs):
+ seen["home"] = Path(str(get_hermes_home())).resolve()
+ seen["secret"] = secret_scope.get_secret("OPENROUTER_API_KEY")
+ return {"provider": "openrouter", "api_key": seen["secret"] or "", "source": "env"}
+
+ monkeypatch.setattr("hermes_cli.runtime_provider.resolve_runtime_provider", fake_resolve)
+
+ secret_scope.set_multiplex_active(True)
+ try:
+ status = server.handle_request(
+ {"id": "1", "method": "setup.status", "params": {"profile": "bot"}}
+ )
+ assert status["result"] == {"provider_configured": False, "profile": "bot"}
+
+ (bot_home / ".env").write_text("OPENROUTER_API_KEY=sk-or-bot-profile-secret-00001\n")
+ status = server.handle_request(
+ {"id": "2", "method": "setup.status", "params": {"profile": "bot"}}
+ )
+ runtime = server.handle_request(
+ {"id": "3", "method": "setup.runtime_check", "params": {"profile": "bot"}}
+ )
+ finally:
+ secret_scope.set_multiplex_active(False)
+
+ assert status["result"] == {"provider_configured": True, "profile": "bot"}
+ assert runtime["result"]["ok"] is True
+ assert runtime["result"]["profile"] == "bot"
+ assert seen == {"home": bot_home.resolve(), "secret": "sk-or-bot-profile-secret-00001"}
+ assert Path(str(get_hermes_home())).resolve() != bot_home.resolve()
+
+
+def test_setup_readiness_unknown_profile_never_answers_for_launch_profile(monkeypatch):
+ monkeypatch.setattr("hermes_cli.main._has_any_provider_configured", lambda **_kw: True)
+ monkeypatch.setattr(
+ "hermes_cli.runtime_provider.resolve_runtime_provider",
+ lambda requested=None, **kw: {"provider": "openrouter", "api_key": "sk-or-launch-0000000000", "source": "env"},
+ )
+ monkeypatch.setattr("hermes_cli.profiles.profile_exists", lambda name: False)
+
+ for method in ("setup.status", "setup.runtime_check"):
+ resp = server.handle_request({"id": "1", "method": method, "params": {"profile": "ghost"}})
+ assert resp["result"]["ok"] is False
+ assert resp["result"]["profile"] == "ghost"
+ assert "does not exist" in resp["result"]["error"]
+
+
def test_complete_slash_drops_removed_provider_alias():
# `/provider` was folded into a single `/model` command, so autocomplete
# must no longer offer the dead alias...
@@ -10359,7 +10561,7 @@ def test_session_compress_returns_compute_host_history(monkeypatch):
}
-def test_session_compress_forwards_120_second_budget_to_compute_host(monkeypatch):
+def test_session_compress_forwards_config_ceiling_budget_to_compute_host(monkeypatch):
session = _session(agent=None, _compute_host_active=True)
server._sessions["sid"] = session
calls = []
@@ -10378,6 +10580,9 @@ def test_session_compress_forwards_120_second_budget_to_compute_host(monkeypatch
monkeypatch.setattr(server, "_session_uses_compute_host", lambda _session: True)
monkeypatch.setattr(server, "_send_compute_host_control", send_control)
+ monkeypatch.setattr(
+ server, "_load_cfg", lambda: {"compression": {"context_total_ceiling_seconds": 300}}
+ )
try:
resp = server.handle_request(
@@ -10387,17 +10592,17 @@ def test_session_compress_forwards_120_second_budget_to_compute_host(monkeypatch
server._sessions.pop("sid", None)
assert resp["result"]["status"] == "compressed"
- assert calls == [
- (
- ("sid",),
- {
- "route_name": "session.compress",
- "command": "/compress",
- "wait": True,
- "timeout": 120.0,
- },
- )
- ]
+ assert len(calls) == 1
+ (sid_arg,), kwargs = calls[0]
+ assert sid_arg == "sid"
+ assert kwargs["route_name"] == "session.compress"
+ assert kwargs["command"] == "/compress"
+ assert kwargs["wait"] is True
+ # #97948: the waiter follows compression.context_total_ceiling_seconds
+ # (+30s slack) instead of a hard-coded 120s, and registers a late-ack
+ # handler so a compress that outlives it is still adopted.
+ assert kwargs["timeout"] == 330.0
+ assert callable(kwargs["on_late_ack"])
def test_session_compress_preserves_compute_host_aborted_summary(monkeypatch):
@@ -13547,7 +13752,13 @@ def test_slow_agent_build_emits_keyed_progress_notice(monkeypatch):
def test_agent_build_failure_surfaces_error_and_drops_turn(monkeypatch):
"""When the build itself FAILS (agent_error set when ready fires), the
prompt must not run and the failure must reach the client as a visible
- error event — never a silent drop."""
+ error event — never a silent drop.
+
+ prompt.submit retries a completed failed build once (fresh provider
+ resolution un-wedges sessions whose failure cause was fixed), so the
+ build stub here is a faithful failing build: it sets agent_error and
+ fires the session's CURRENT ready event (the retry installs a new one).
+ A no-op stub would leave that event unset and hang the patient wait."""
threads = []
emitted = []
calls = {"run_prompt": 0}
@@ -13570,12 +13781,16 @@ def test_agent_build_failure_surfaces_error_and_drops_turn(monkeypatch):
session["agent_error"] = "No LLM provider configured" # ...but failed
server._sessions["sid"] = session
+ def _failing_build(sid, session):
+ session["agent_error"] = "No LLM provider configured"
+ session["agent_ready"].set()
+
try:
monkeypatch.setattr(server.threading, "Thread", _FakeThread)
monkeypatch.setattr(server, "_emit", lambda *args, **kwargs: emitted.append(args))
monkeypatch.setattr(server, "_ensure_session_db_row", lambda session: None)
monkeypatch.setattr(server, "_persist_branch_seed", lambda session: None)
- monkeypatch.setattr(server, "_start_agent_build", lambda sid, session: None)
+ monkeypatch.setattr(server, "_start_agent_build", _failing_build)
monkeypatch.setattr(
server,
"_run_prompt_submit",
@@ -14835,6 +15050,73 @@ def test_session_most_recent_honors_params_profile(monkeypatch, tmp_path):
assert resp["result"]["session_id"] == "ml-tip"
+def test_handoff_request_uses_session_profile_home(monkeypatch, tmp_path):
+ """Handoff validation must read the owning session's gateway config."""
+ import contextlib
+
+ from gateway.config import GatewayConfig, HomeChannel, Platform, PlatformConfig
+ from hermes_cli.config import get_hermes_home
+ from tui_gateway import methods_session
+
+ methods_session.register(server)
+ profile_home = tmp_path / "profiles" / "coder"
+ profile_home.mkdir(parents=True)
+ seen_homes = []
+
+ def load_config():
+ home = get_hermes_home()
+ seen_homes.append(home)
+ config = GatewayConfig()
+ if home == profile_home:
+ config.platforms[Platform.DISCORD] = PlatformConfig(
+ enabled=True,
+ home_channel=HomeChannel(
+ platform=Platform.DISCORD,
+ chat_id="discord-home",
+ name="Hermes / #chat-coding",
+ ),
+ )
+ return config
+
+ class ProfileDB:
+ def get_session(self, _key):
+ return {"id": _key}
+
+ def request_handoff(self, _key, platform):
+ return platform == "discord"
+
+ @contextlib.contextmanager
+ def profile_db(_session):
+ yield ProfileDB()
+
+ monkeypatch.setattr("gateway.config.load_gateway_config", load_config)
+ monkeypatch.setattr(server, "_ensure_session_db_row", lambda _session: None)
+ monkeypatch.setattr(server, "_session_db", profile_db)
+ server._sessions["handoff-profile"] = {
+ "running": False,
+ "session_key": "desktop-coder-session",
+ "profile_home": str(profile_home),
+ }
+ try:
+ resp = server.handle_request(
+ {
+ "id": "1",
+ "method": "handoff.request",
+ "params": {
+ "session_id": "handoff-profile",
+ "platform": "discord",
+ },
+ }
+ )
+ finally:
+ server._sessions.pop("handoff-profile", None)
+
+ assert "result" in resp, resp
+ assert resp["result"]["queued"] is True
+ assert seen_homes == [profile_home]
+ assert get_hermes_home() != profile_home
+
+
def test_session_create_reports_requested_profile_name(monkeypatch, tmp_path):
"""Issue #62503: session.create info.profile_name must not always be launch."""
profile_home = tmp_path / "profiles" / "mlperf"
@@ -15272,6 +15554,312 @@ def test_session_branch_writes_to_parent_profile_db(monkeypatch, tmp_path):
server._sessions.pop(k, None)
+def test_session_create_persists_seeded_branch_child(monkeypatch):
+ """A desktop branch (session.create with parent_session_id + seeded
+ messages) must persist its row + transcript immediately (#93959).
+
+ The renderer re-fetches the fresh child via REST and defer_history
+ hydration right after create; both read the DB. An unpersisted child
+ 404s/hydrates empty, the client fail-latch refuses to bind it, and the
+ user gets an infinite spinner whose optimistic row vanishes on restart.
+ """
+
+ class _FakeAgent:
+ def __init__(self):
+ self.model = "test-model"
+
+ seen: dict = {}
+
+ class _FakeDB:
+ def get_session_title(self, key):
+ seen["parent_title"] = key
+ return "My Parent Session"
+
+ def get_next_title_in_lineage(self, current):
+ return f"{current} #2"
+
+ def create_session(self, key, **kwargs):
+ seen["created"] = key
+ seen["parent"] = kwargs.get("parent_session_id")
+ seen["branched_from"] = (kwargs.get("model_config") or {}).get("_branched_from")
+
+ def append_messages_batch(self, session_id, messages, **kwargs):
+ seen["messages"] = list(messages)
+
+ def set_session_title(self, key, title):
+ seen["title"] = title
+ return True
+
+ monkeypatch.setattr(server, "_get_db", lambda: _FakeDB())
+ monkeypatch.setattr(server, "_make_agent", lambda sid, key, session_db=None, **_kw: _FakeAgent())
+ monkeypatch.setattr(server, "_SlashWorker", lambda *a, **k: None)
+ monkeypatch.setattr(server, "_session_info", lambda _a, *a2: {"model": "x"})
+ monkeypatch.setattr(server, "_probe_credentials", lambda _a: None)
+ monkeypatch.setattr(server, "_wire_callbacks", lambda _sid: None)
+ monkeypatch.setattr(server, "_emit", lambda *a, **kw: None)
+
+ import tools.approval as _approval
+
+ monkeypatch.setattr(_approval, "register_gateway_notify", lambda key, cb: None)
+ monkeypatch.setattr(_approval, "load_permanent_allowlist", lambda: None)
+
+ seeded = [
+ {"role": "user", "content": "hello from parent"},
+ {"role": "assistant", "content": "parent reply"},
+ ]
+
+ resp = server.handle_request(
+ {
+ "id": "1",
+ "method": "session.create",
+ "params": {
+ "cols": 96,
+ "source": "desktop",
+ "parent_session_id": "20260823_084113_6de211",
+ "messages": seeded,
+ },
+ }
+ )
+
+ assert "result" in resp, resp
+ key = resp["result"]["stored_session_id"]
+
+ # Row persisted up front with lineage linkage and a lineage title —
+ # not deferred to the first prompt.
+ assert seen.get("created") == key
+ assert seen.get("parent") == "20260823_084113_6de211"
+ assert seen.get("branched_from") == "20260823_084113_6de211"
+ assert seen.get("title") == "My Parent Session #2"
+
+ # Seeded transcript copied into the durable row so REST prefetch and
+ # defer_history hydration both find it immediately.
+ assert len(seen.get("messages") or []) == 2
+ assert seen["messages"][0]["content"] == "hello from parent"
+
+ # The live record no longer queues the title — the DB already holds it.
+ runtime_sid = resp["result"]["session_id"]
+ assert server._sessions[runtime_sid]["pending_title"] is None
+
+ server._sessions.pop(runtime_sid, None)
+
+
+def test_session_create_branch_seed_failure_does_not_break_create(monkeypatch):
+ """Best-effort persistence: a broken DB must not fail session.create."""
+
+ class _FakeAgent:
+ def __init__(self):
+ self.model = "test-model"
+
+ class _BrokenDB:
+ def get_session_title(self, key):
+ raise RuntimeError("db down")
+
+ monkeypatch.setattr(server, "_get_db", lambda: _BrokenDB())
+ monkeypatch.setattr(server, "_make_agent", lambda sid, key, session_db=None, **_kw: _FakeAgent())
+ monkeypatch.setattr(server, "_SlashWorker", lambda *a, **k: None)
+ monkeypatch.setattr(server, "_session_info", lambda _a, *a2: {"model": "x"})
+ monkeypatch.setattr(server, "_probe_credentials", lambda _a: None)
+ monkeypatch.setattr(server, "_wire_callbacks", lambda _sid: None)
+ monkeypatch.setattr(server, "_emit", lambda *a, **kw: None)
+
+ import tools.approval as _approval
+
+ monkeypatch.setattr(_approval, "register_gateway_notify", lambda key, cb: None)
+ monkeypatch.setattr(_approval, "load_permanent_allowlist", lambda: None)
+
+ resp = server.handle_request(
+ {
+ "id": "1",
+ "method": "session.create",
+ "params": {
+ "source": "desktop",
+ "parent_session_id": "parent-1",
+ "messages": [{"role": "user", "content": "seed"}],
+ },
+ }
+ )
+
+ # Create itself still succeeds — the lazy first-prompt path remains as
+ # the fallback for the seed.
+ assert "result" in resp
+
+ server._sessions.pop(resp["result"]["stored_session_id"], None)
+
+
+def test_session_create_seed_failure_after_row_compensates(monkeypatch):
+ """Partial-failure compensation (#93959 review): if the row commits but
+ the transcript copy fails, the just-created child is DELETED so the lazy
+ first-prompt fallback can retry cleanly. Without this, a durable empty
+ row defeats _ensure_session_db_row's INSERT OR IGNORE and the renderer
+ fail-latches on a transcript-less session again."""
+
+ class _FakeAgent:
+ def __init__(self):
+ self.model = "test-model"
+
+ seen: dict = {}
+
+ class _FakeDB:
+ def get_session_title(self, key):
+ return "Parent"
+
+ def get_next_title_in_lineage(self, current):
+ return f"{current} #2"
+
+ def create_session(self, key, **kwargs):
+ seen["created"] = key
+
+ def append_messages_batch(self, session_id, messages, **kwargs):
+ raise RuntimeError("transcript write failed")
+
+ def delete_session(self, session_id):
+ seen["deleted"] = session_id
+ return True
+
+ monkeypatch.setattr(server, "_get_db", lambda: _FakeDB())
+ monkeypatch.setattr(server, "_make_agent", lambda sid, key, session_db=None, **_kw: _FakeAgent())
+ monkeypatch.setattr(server, "_SlashWorker", lambda *a, **k: None)
+ monkeypatch.setattr(server, "_session_info", lambda _a, *a2: {"model": "x"})
+ monkeypatch.setattr(server, "_probe_credentials", lambda _a: None)
+ monkeypatch.setattr(server, "_wire_callbacks", lambda _sid: None)
+ monkeypatch.setattr(server, "_emit", lambda *a, **kw: None)
+
+ import tools.approval as _approval
+
+ monkeypatch.setattr(_approval, "register_gateway_notify", lambda key, cb: None)
+ monkeypatch.setattr(_approval, "load_permanent_allowlist", lambda: None)
+
+ resp = server.handle_request(
+ {
+ "id": "1",
+ "method": "session.create",
+ "params": {
+ "source": "desktop",
+ "parent_session_id": "parent-1",
+ "title": "My Branch",
+ "messages": [{"role": "user", "content": "seed"}],
+ },
+ }
+ )
+
+ assert "result" in resp
+ key = resp["result"]["stored_session_id"]
+ # The half-written child was rolled back — no durable empty row left to
+ # shadow the lazy seed path.
+ assert seen.get("deleted") == key
+ # pending_title survived: it still lands via the lazy post-turn apply.
+ runtime_sid = resp["result"]["session_id"]
+ assert server._sessions[runtime_sid]["pending_title"] == "My Branch"
+
+ server._sessions.pop(runtime_sid, None)
+
+
+def test_session_create_seed_disk_full_keeps_row_for_retry(monkeypatch):
+ """Disk-full is NOT compensated: the row stays (deleting data on a full
+ disk can make things worse), create still succeeds, and the failure is
+ observable at warning level (#93959 review)."""
+
+ import logging as _logging
+
+ class _FakeAgent:
+ def __init__(self):
+ self.model = "test-model"
+
+ class _FakeDB:
+ def get_session_title(self, key):
+ return "Parent"
+
+ def get_next_title_in_lineage(self, current):
+ return f"{current} #2"
+
+ def create_session(self, key, **kwargs):
+ pass
+
+ def append_messages_batch(self, session_id, messages, **kwargs):
+ raise OSError(28, "No space left on device")
+
+ monkeypatch.setattr(server, "_get_db", lambda: _FakeDB())
+ monkeypatch.setattr(server, "_make_agent", lambda sid, key, session_db=None, **_kw: _FakeAgent())
+ monkeypatch.setattr(server, "_SlashWorker", lambda *a, **k: None)
+ monkeypatch.setattr(server, "_session_info", lambda _a, *a2: {"model": "x"})
+ monkeypatch.setattr(server, "_probe_credentials", lambda _a: None)
+ monkeypatch.setattr(server, "_wire_callbacks", lambda _sid: None)
+ monkeypatch.setattr(server, "_emit", lambda *a, **kw: None)
+
+ import tools.approval as _approval
+
+ monkeypatch.setattr(_approval, "register_gateway_notify", lambda key, cb: None)
+ monkeypatch.setattr(_approval, "load_permanent_allowlist", lambda: None)
+
+ records: list = []
+
+ class _Capture(_logging.Handler):
+ def emit(self, record):
+ records.append(record)
+
+ handler = _Capture(level=_logging.WARNING)
+ root = _logging.getLogger()
+ root.addHandler(handler)
+ try:
+ resp = server.handle_request(
+ {
+ "id": "1",
+ "method": "session.create",
+ "params": {
+ "source": "desktop",
+ "parent_session_id": "parent-1",
+ "messages": [{"role": "user", "content": "seed"}],
+ },
+ }
+ )
+ finally:
+ root.removeHandler(handler)
+
+ assert "result" in resp
+ # The failure surfaced at WARNING (observable), not buried at debug.
+ warnings = [r for r in records if r.levelno >= _logging.WARNING]
+ assert any("seeded-branch persistence failed" in r.getMessage() for r in warnings)
+
+ server._sessions.pop(resp["result"]["stored_session_id"], None)
+
+
+def test_session_create_without_parent_still_defers_row(monkeypatch):
+ """Plain drafts keep the lazy-row contract: no parent + no explicit branch
+ intent means no eager persistence (the original draft-hygiene invariant)."""
+
+ class _FakeAgent:
+ def __init__(self):
+ self.model = "test-model"
+
+ calls: dict = {"create": 0}
+
+ class _FakeDB:
+ def create_session(self, *a, **k):
+ calls["create"] += 1
+
+ monkeypatch.setattr(server, "_get_db", lambda: _FakeDB())
+ monkeypatch.setattr(server, "_make_agent", lambda sid, key, session_db=None, **_kw: _FakeAgent())
+ monkeypatch.setattr(server, "_SlashWorker", lambda *a, **k: None)
+ monkeypatch.setattr(server, "_session_info", lambda _a, *a2: {"model": "x"})
+ monkeypatch.setattr(server, "_probe_credentials", lambda _a: None)
+ monkeypatch.setattr(server, "_wire_callbacks", lambda _sid: None)
+ monkeypatch.setattr(server, "_emit", lambda *a, **kw: None)
+
+ import tools.approval as _approval
+
+ monkeypatch.setattr(_approval, "register_gateway_notify", lambda key, cb: None)
+ monkeypatch.setattr(_approval, "load_permanent_allowlist", lambda: None)
+
+ resp = server.handle_request(
+ {"id": "1", "method": "session.create", "params": {"cols": 80}}
+ )
+ sid = resp["result"]["session_id"]
+ server._sessions[sid]["agent_ready"].wait(timeout=2.0)
+
+ assert calls["create"] == 0, "plain drafts must not persist eagerly"
+
+ server._sessions.pop(sid, None)
+
def test_session_branch_installs_parent_profile_secret_scope(monkeypatch, tmp_path):
"""The branched agent must be built under the parent profile's secrets.
diff --git a/tests/tools/conftest.py b/tests/tools/conftest.py
index cefe4584dc..e8fec5cf5c 100644
--- a/tests/tools/conftest.py
+++ b/tests/tools/conftest.py
@@ -83,6 +83,7 @@ def register_all_web_providers():
from plugins.web.firecrawl.provider import FirecrawlWebSearchProvider
from plugins.web.parallel.provider import ParallelWebSearchProvider
from plugins.web.keenable.provider import KeenableWebSearchProvider
+ from plugins.web.tavily.provider import TavilyWebSearchProvider
from plugins.web.searxng.provider import SearXNGWebSearchProvider
from plugins.web.xai.provider import XAIWebSearchProvider
@@ -94,6 +95,7 @@ def register_all_web_providers():
FirecrawlWebSearchProvider,
ParallelWebSearchProvider,
KeenableWebSearchProvider,
+ TavilyWebSearchProvider,
SearXNGWebSearchProvider,
XAIWebSearchProvider,
):
diff --git a/tests/tools/test_94248_timeout_transport_drain.py b/tests/tools/test_94248_timeout_transport_drain.py
new file mode 100644
index 0000000000..2347ad2915
--- /dev/null
+++ b/tests/tools/test_94248_timeout_transport_drain.py
@@ -0,0 +1,197 @@
+"""#94248 (native half): delegation timeout must drain transports FD-safely.
+
+A timed-out child's daemon worker is typically parked inside an in-flight
+OpenSSL read. The timeout thread must (1) never hard-close the child while the
+worker future is running (deferred close, #90889), and (2) drain the child's
+transports with socket ``shutdown()`` only — never ``client.close()`` — so the
+blocked read settles with EOF/EPIPE and the worker can unwind (bounded drain).
+Cross-thread FD release under a live SSL BIO is the #29507/#67142/#70773
+native-corruption family.
+"""
+from __future__ import annotations
+
+import threading
+import time
+from types import SimpleNamespace
+
+from tools import delegate_tool
+
+
+class _SslBlockedChild:
+ """Worker blocks (modelling an in-flight SSL read) until drained."""
+
+ def __init__(self) -> None:
+ self.tool_progress_callback = None
+ self._credential_pool = None
+ self._delegate_saved_tool_names = []
+ self._delegate_role = "leaf"
+ self._delegate_depth = 1
+ self._subagent_id = None
+ self.model = "test-model"
+ self.session_prompt_tokens = 0
+ self.session_completion_tokens = 0
+ self.session_estimated_cost_usd = 0.0
+ self.session_cost_status = "unknown"
+ self.read_settled = threading.Event() # drain "EOF" signal
+ self.unwound = threading.Event()
+ self.closed = threading.Event()
+ self.close_while_blocked = False
+ self.drain_calls: list[str] = []
+ self.drain_threads: list[str] = []
+
+ def run_conversation(self, **_kwargs):
+ # Models the worker blocked in ssl.read: only the FD-safe drain
+ # (socket shutdown -> EOF) settles it; interrupts alone do not.
+ assert self.read_settled.wait(timeout=10), "drain never settled the read"
+ time.sleep(0.05) # post-read unwind work (turn-finally flush)
+ self.unwound.set()
+ return {
+ "final_response": "",
+ "completed": False,
+ "interrupted": True,
+ "api_calls": 1,
+ "messages": [],
+ }
+
+ def hard_interrupt(self, *_a, **_k):
+ # Cooperative interrupt cannot unblock a thread inside OpenSSL read.
+ pass
+
+ def get_activity_summary(self):
+ return {"api_call_count": 1}
+
+ def _drain_transports_after_abandonment(self, *, reason: str) -> int:
+ self.drain_calls.append(reason)
+ self.drain_threads.append(threading.current_thread().name)
+ self.read_settled.set()
+ return 1
+
+ def close(self):
+ if not self.unwound.is_set():
+ self.close_while_blocked = True
+ self.closed.set()
+
+
+def _run(child, monkeypatch, timeout=0.4):
+ parent = SimpleNamespace(
+ session_id="parent-94248-drain",
+ _current_task_id=None,
+ _active_children=[child],
+ _active_children_lock=threading.Lock(),
+ )
+ monkeypatch.setattr(delegate_tool, "_get_child_timeout", lambda: timeout)
+ if hasattr(delegate_tool, "_get_worktree_isolation"):
+ monkeypatch.setattr(delegate_tool, "_get_worktree_isolation", lambda: False)
+ return delegate_tool._run_single_child(
+ task_index=0,
+ goal="exercise timeout transport drain",
+ child=child,
+ parent_agent=parent,
+ )
+
+
+def test_timeout_drains_transports_so_blocked_worker_can_unwind(monkeypatch):
+ child = _SslBlockedChild()
+
+ result = _run(child, monkeypatch)
+
+ assert result["status"] == "timeout"
+ # The drain ran from the timeout path (immediate sweep) and settled the
+ # blocked read; without it the worker would still be parked in ssl.read.
+ assert any(r.startswith("delegate_timeout") for r in child.drain_calls), (
+ "timeout path never drained the abandoned child's transports"
+ )
+ assert child.unwound.wait(timeout=5), (
+ "worker never unwound — the drain did not settle its blocked read"
+ )
+ assert child.closed.wait(timeout=5)
+ assert not child.close_while_blocked, (
+ "child.close() ran while the worker was still inside its blocked read"
+ )
+
+
+def test_timeout_drain_failure_does_not_break_timeout_result(monkeypatch):
+ child = _SslBlockedChild()
+
+ def _raising_drain(*, reason: str) -> int:
+ child.drain_calls.append(reason)
+ raise RuntimeError("transport sweep exploded")
+
+ child._drain_transports_after_abandonment = _raising_drain
+
+ result = _run(child, monkeypatch)
+
+ assert result["status"] == "timeout"
+ assert child.drain_calls, "drain hook was never attempted"
+ # Unblock the worker manually so the deferred close can run.
+ child.read_settled.set()
+ assert child.unwound.wait(timeout=5)
+ assert child.closed.wait(timeout=5)
+
+
+def test_timeout_without_drain_hook_still_defers_close(monkeypatch):
+ """Children lacking the hook (test doubles, third-party agents) keep the
+ plain deferred-close behavior."""
+ child = _SslBlockedChild()
+ # Shadow the hook with a non-callable: the timeout path must skip it.
+ child.__dict__["_drain_transports_after_abandonment"] = None
+
+ result = _run(child, monkeypatch)
+
+ assert result["status"] == "timeout"
+ assert not child.closed.is_set(), (
+ "close must stay deferred while the worker future is running"
+ )
+ child.read_settled.set()
+ assert child.unwound.wait(timeout=5)
+ assert child.closed.wait(timeout=5)
+ assert not child.close_while_blocked
+
+
+class _FakeSocket:
+ def __init__(self):
+ self.shutdown_calls = 0
+ self.closed = False
+
+ def settimeout(self, _v):
+ pass
+
+ def shutdown(self, _how):
+ self.shutdown_calls += 1
+
+ def close(self):
+ self.closed = True
+
+
+def test_agent_drain_shuts_sockets_down_without_fd_release(monkeypatch):
+ """AIAgent._drain_transports_after_abandonment must shutdown(), not close()."""
+ import threading as _threading
+ from unittest.mock import patch
+
+ with patch("run_agent.AIAgent.__init__", return_value=None):
+ from run_agent import AIAgent
+
+ agent = AIAgent.__new__(AIAgent)
+
+ sock = _FakeSocket()
+ close_calls = {"n": 0}
+
+ class _FakeClient:
+ def close(self):
+ close_calls["n"] += 1
+
+ agent.client = _FakeClient()
+ agent._client_lock = _threading.RLock()
+ agent._codex_session = None
+ agent._active_request_abort = None
+
+ import agent.agent_runtime_helpers as arh
+
+ monkeypatch.setattr(arh, "_iter_pool_sockets", lambda _c: iter([sock]))
+
+ drained = agent._drain_transports_after_abandonment(reason="delegate_timeout_test")
+
+ assert drained == 1
+ assert sock.shutdown_calls == 1
+ assert not sock.closed, "drain must never release socket FDs"
+ assert close_calls["n"] == 0, "drain must never call client.close()"
diff --git a/tests/tools/test_bot_mode_dm.py b/tests/tools/test_bot_mode_dm.py
index e83ec30ae0..ed0936f4c9 100644
--- a/tests/tools/test_bot_mode_dm.py
+++ b/tests/tools/test_bot_mode_dm.py
@@ -422,6 +422,29 @@ def test_delivery_runner_preserves_child_failure_and_unlinks(tmp_path):
assert not dm_file.exists()
+def test_delivery_runner_surfaces_live_owner_refusal(tmp_path, capsys):
+ """#100523: the CLI's single-owner lease refusal is a delivery FAILURE the
+ sender can read, not a raw exit-1 with the payload silently gone."""
+ dm_file = tmp_path / "message.txt"
+ dm_file.write_text("hi", encoding="utf-8")
+ child = tmp_path / "owned.py"
+ child.write_text(
+ "import sys\n"
+ "print('Session abc already has a live owner (desktop, pid 1).', file=sys.stderr)\n"
+ "raise SystemExit(1)\n",
+ encoding="utf-8",
+ )
+
+ returncode = bot_mode_dm._run_delivery(
+ [sys.executable, str(child), "-p", "ops"], str(dm_file), stdin_file=False
+ )
+
+ assert returncode == 1
+ payload = json.loads(capsys.readouterr().out)
+ assert payload["reason"] == "target_busy"
+ assert "NOT delivered" in payload["error"]
+
+
def test_query_file_delivery_closes_stdin_for_initial_attempt_and_retry(
tmp_path, monkeypatch
):
diff --git a/tests/tools/test_bot_relay.py b/tests/tools/test_bot_relay.py
index b7b9f4a37f..7b9fb528f1 100644
--- a/tests/tools/test_bot_relay.py
+++ b/tests/tools/test_bot_relay.py
@@ -151,6 +151,31 @@ def test_waiter_command_quotes_and_targets_reply_file(root):
assert "rm -rf" not in cmd # sanity: single quoted -c payload
+def test_waiter_picks_up_reply_within_a_sub_second_cadence(root):
+ """The reply file is written once; the waiter must notice it fast, not
+ on a multi-second sleep (dead air the sender's completion notification
+ inherits on every cross-machine reply)."""
+ import shlex
+ import subprocess
+ import threading
+ import time
+
+ env = {"id": "c" * 32, "target_handle": "researcher", "target_connection": "ssh-vps"}
+ reply_path = bot_relay.relay_root(root) / bot_relay.REPLIES_DIR / f"{env['id']}.json"
+ reply_path.parent.mkdir(parents=True, exist_ok=True)
+
+ def write_reply():
+ time.sleep(0.3)
+ reply_path.write_text(json.dumps({"reply": "pong"}), encoding="utf-8")
+
+ threading.Thread(target=write_reply, daemon=True).start()
+ started = time.monotonic()
+ proc = subprocess.run(shlex.split(bot_relay.waiter_command(root, env)), capture_output=True, text=True, timeout=10)
+ elapsed = time.monotonic() - started
+ assert proc.returncode == 0 and "pong" in proc.stdout
+ assert elapsed < 1.5, f"waiter took {elapsed:.2f}s to notice a reply written at 0.3s"
+
+
def test_roster_rejects_connection_id_outside_handle_charset(root):
bad = [
{"profile": "researcher", "handle": "researcher", "connection_id": "vps'); print(1)"},
diff --git a/tests/tools/test_browser_cleanup.py b/tests/tools/test_browser_cleanup.py
index 6c929da628..b1f89b3c84 100644
--- a/tests/tools/test_browser_cleanup.py
+++ b/tests/tools/test_browser_cleanup.py
@@ -83,3 +83,106 @@ class TestBrowserCleanup:
assert browser_tool._session_last_activity == {}
assert browser_tool._recording_sessions == set()
assert browser_tool._cleanup_done is True
+
+
+class TestInactivityJanitorMultiplex:
+ """#86402 / #100738: the process-global janitor thread has no profile scope."""
+
+ def setup_method(self):
+ from agent import secret_scope
+ from tools import browser_tool
+
+ self.bt = browser_tool
+ self.saved = {
+ name: getattr(browser_tool, name).copy()
+ for name in (
+ "_active_sessions", "_session_last_activity",
+ "_session_owner_homes", "_cleanup_failures", "_recording_sessions",
+ )
+ }
+ self.orig_timeout = browser_tool.BROWSER_SESSION_INACTIVITY_TIMEOUT
+ browser_tool.BROWSER_SESSION_INACTIVITY_TIMEOUT = 0
+ for name in self.saved:
+ getattr(browser_tool, name).clear()
+ secret_scope.set_multiplex_active(True)
+
+ def teardown_method(self):
+ from agent import secret_scope
+
+ secret_scope.set_multiplex_active(False)
+ self.bt.BROWSER_SESSION_INACTIVITY_TIMEOUT = self.orig_timeout
+ for name, saved in self.saved.items():
+ live = getattr(self.bt, name)
+ live.clear()
+ live.update(saved)
+
+ def test_janitor_tears_down_under_owner_profile_scope(self, tmp_path, monkeypatch):
+ from agent import secret_scope
+ from hermes_constants import (
+ get_hermes_home, reset_hermes_home_override, set_hermes_home_override,
+ )
+
+ monkeypatch.setenv("HERMES_HOME", str(tmp_path))
+ monkeypatch.delenv("CAMOFOX_URL", raising=False)
+ monkeypatch.delenv("BROWSER_CDP_URL", raising=False)
+ p1 = tmp_path / "profiles" / "p1"
+ p1.mkdir(parents=True)
+ (p1 / ".env").write_text("CAMOFOX_URL=http://127.0.0.1:1\n")
+
+ # Profile p1's turn opens the session; the janitor later runs unscoped.
+ home_tok = set_hermes_home_override(str(p1))
+ scope_tok = secret_scope.set_secret_scope(secret_scope.build_profile_secret_scope(p1))
+ try:
+ self.bt._update_session_activity("t1")
+ self.bt._active_sessions["t1"] = {"session_name": "s1", "bb_session_id": None}
+ finally:
+ secret_scope.reset_secret_scope(scope_tok)
+ reset_hermes_home_override(home_tok)
+ self.bt._session_last_activity["t1"] -= 10
+
+ seen = {}
+
+ def fake_close(task_id, cmd, args, timeout=None):
+ seen["home"] = str(get_hermes_home())
+ seen["url"] = secret_scope.get_secret("CAMOFOX_URL")
+ return {"success": True}
+
+ with (
+ patch("tools.browser_tool._run_browser_command", side_effect=fake_close),
+ patch("tools.browser_camofox._delete", return_value={}),
+ patch("tools.browser_tool.os.path.exists", return_value=False),
+ ):
+ self.bt._cleanup_inactive_browser_sessions()
+
+ assert seen == {"home": str(p1), "url": "http://127.0.0.1:1"}
+ assert "t1" not in self.bt._session_last_activity
+ assert "t1" not in self.bt._active_sessions
+ assert "t1" not in self.bt._session_owner_homes
+
+ def test_repeated_failures_force_reap_and_close_cloud_session(self):
+ from unittest.mock import MagicMock
+
+ self.bt._active_sessions["t1"] = {"session_name": "s1", "bb_session_id": "bb-1"}
+ self.bt._session_last_activity["t1"] = 1.0
+ provider = MagicMock()
+
+ with (
+ patch("tools.browser_tool.cleanup_browser", side_effect=RuntimeError("boom")),
+ patch("tools.browser_tool._get_cloud_provider", return_value=provider),
+ patch("tools.browser_tool.os.path.exists", return_value=False),
+ ):
+ for _ in range(self.bt.MAX_INACTIVITY_CLEANUP_FAILURES - 1):
+ self.bt._cleanup_inactive_browser_sessions()
+ # An activity touch must NOT reset the failure budget.
+ self.bt._update_session_activity("t1")
+ self.bt._session_last_activity["t1"] = 1.0
+ assert self.bt._cleanup_failures["t1"] == self.bt.MAX_INACTIVITY_CLEANUP_FAILURES - 1
+ assert "t1" in self.bt._active_sessions
+ provider.close_session.assert_not_called()
+
+ self.bt._cleanup_inactive_browser_sessions()
+
+ provider.close_session.assert_called_once_with("bb-1")
+ assert "t1" not in self.bt._active_sessions
+ assert "t1" not in self.bt._session_last_activity
+ assert "t1" not in self.bt._cleanup_failures
diff --git a/tests/tools/test_cronjob_run_delivery_notice.py b/tests/tools/test_cronjob_run_delivery_notice.py
new file mode 100644
index 0000000000..e82b2ad090
--- /dev/null
+++ b/tests/tools/test_cronjob_run_delivery_notice.py
@@ -0,0 +1,274 @@
+"""Honesty of the manual-run delivery notice (issue #83993).
+
+A manual ``cronjob(action='run')`` finishes with a completion summary line
+
+ Delivery target: (output was delivered there by the job itself)
+
+that was appended UNCONDITIONALLY for non-local targets — even when
+``run_one_job`` had just written ``last_delivery_error`` onto the refreshed
+job record because the post-run delivery (telegram/discord/…) failed. The
+calling agent then relayed "all good" over a failed delivery.
+
+The note must follow the refreshed job record: a set ``last_delivery_error``
+means delivery FAILED with the error text surfaced; an empty/missing error
+keeps the legacy wording byte-for-byte (zero regression), and local jobs
+always say saved-locally.
+"""
+
+import contextlib
+import time
+from unittest.mock import patch
+
+import pytest
+
+from tools.cronjob_tools import _manual_run_delivery_note
+
+
+@pytest.fixture(autouse=True)
+def _clean_state():
+ """Reset the shared async-delegation world around each test.
+
+ The dispatch tests below submit real workers onto the process-wide
+ daemon executor in ``tools.async_delegation``. A finished worker parks
+ idle holding an ``_idle_semaphore`` token, so the NEXT dispatch in this
+ process REUSES that thread instead of spawning a fresh one — and only
+ the fresh-spawn path keeps upstream's dispatch-and-return test winning
+ its patch-visibility race: ``Thread.start()`` blocks the dispatching
+ thread until the worker has bootstrapped, so the worker performs
+ ``_run_claimed_job``'s lazy ``from cron.scheduler import run_one_job``
+ while the test's patches are still active. On the idle-reuse path
+ ``submit`` returns with the GIL still held, the patch block unwinds
+ first, and the worker binds the REAL ``run_one_job`` — which then runs
+ the fake job for real ("no model configured") and the mock never fires.
+ Without this reset, test_cronjob_run_background.py's
+ ``test_dispatches_and_returns_handle_immediately`` fails
+ deterministically whenever this file runs before it. Mirrors
+ ``tests/tools/test_async_delegation.py::_clean_state``.
+ """
+ from tools import async_delegation as ad
+ from tools.process_registry import process_registry
+
+ ad._reset_for_tests()
+ while not process_registry.completion_queue.empty():
+ process_registry.completion_queue.get_nowait()
+ yield
+ # Give just-drained workers a beat to finalize BEFORE resetting, so
+ # their completion events land now instead of leaking into the next
+ # test's queue (mirrors test_async_delegation.py).
+ deadline = time.monotonic() + 2.0
+ while ad.active_count() and time.monotonic() < deadline:
+ time.sleep(0.02)
+ ad._reset_for_tests()
+ while not process_registry.completion_queue.empty():
+ process_registry.completion_queue.get_nowait()
+
+
+def _job(job_id, deliver):
+ """Per-test job dict with a UNIQUE id.
+
+ Background workers outlive their test (daemon executor) and hold the id
+ in the scheduler's shared running set until the run finishes; reusing an
+ id across tests trips the in-flight dedupe guard on a straggler.
+ """
+ return {
+ "id": job_id,
+ "name": f"dn run {job_id}",
+ "prompt": "hi",
+ "schedule": {"kind": "cron", "expr": "0 9 * * *"},
+ "deliver": deliver,
+ }
+
+
+@contextlib.contextmanager
+def _bound_session_key(key):
+ """Bind the approval session key contextvar (background dispatch gate)."""
+ from tools.approval import _approval_session_key
+
+ token = _approval_session_key.set(key)
+ try:
+ yield
+ finally:
+ _approval_session_key.reset(token)
+
+
+def _dispatch_diag(res) -> str:
+ """Failure renderer for the wiring tests' dispatch asserts: the result
+ dict plus the scheduler running set, so a broken assert names the return
+ path that was actually taken instead of a bare KeyError."""
+ try:
+ from cron.scheduler import get_running_job_ids
+
+ running = sorted(get_running_job_ids())
+ except Exception as e: # pragma: no cover - diagnostic only
+ running = f"