refactor(hermes_bootstrap,hermes_startup_watchdog,hermes_logging): collapse duplicated dump/handle plumbing, compact docs

This commit is contained in:
Teknium
2026-09-02 18:14:53 -07:00
parent 113f04616b
commit 857b43ef92
3 changed files with 272 additions and 557 deletions
+43 -163
View File
@@ -1,50 +1,12 @@
"""Windows UTF-8 bootstrap for Hermes entry points. """Windows UTF-8 bootstrap for Hermes entry points (no-op on POSIX).
Python on Windows has two long-standing text-encoding footguns: Windows binds stdio to the console code page (cp1252), so ``print("café")`` raises
``UnicodeEncodeError``, and Python children inherit the same default unless
1. ``sys.stdout`` / ``sys.stderr`` are bound to the console code page ``PYTHONUTF8``/``PYTHONIOENCODING`` are set. Import this module first in every entry
(``cp1252`` on US-locale installs), so ``print("café")`` crashes with point (``hermes``, ``hermes-agent``, ``hermes-acp``, ``gateway.run``, ``batch_runner``,
``UnicodeEncodeError: 'charmap' codec can't encode character``. ``cron/scheduler``). It does NOT re-exec with ``-X utf8``: ``open()`` in the current
process still needs an explicit ``encoding="utf-8"`` (ruff ``PLW1514``). POSIX is left
2. Child processes spawned via ``subprocess`` don't know to use UTF-8 alone deliberately — users' ``LANG``/``LC_*`` choices are respected.
unless ``PYTHONUTF8`` and/or ``PYTHONIOENCODING`` are set in their
environment — so any Python subprocess (the execute_code sandbox,
delegation children, linter subprocesses, etc.) inherits the same
cp1252 defaults and hits the same UnicodeEncodeError.
This module fixes both on Windows *only* — POSIX is untouched. It
should be imported at the very top of every Hermes entry point
(``hermes``, ``hermes-agent``, ``hermes-acp``, ``python -m gateway.run``,
``batch_runner.py``, ``cron/scheduler.py``) before any other imports
that might do file I/O or print to stdout.
What this module does on Windows:
- Sets ``os.environ["PYTHONUTF8"] = "1"`` (PEP 540 UTF-8 mode) so
every child process we spawn uses UTF-8 for ``open()`` and stdio.
- Sets ``os.environ["PYTHONIOENCODING"] = "utf-8"`` for belt-and-
suspenders — some tools read this instead of / in addition to
``PYTHONUTF8``.
- Reconfigures ``sys.stdout`` / ``sys.stderr`` to UTF-8 in the current
process, using the ``reconfigure()`` API (Python 3.7+). This fixes
``print("café")`` in the parent without a re-exec.
What this module does NOT do:
- It does not re-exec Python with ``-X utf8``, so ``open()`` calls in
the *current* process still default to locale encoding. Those need
an explicit ``encoding="utf-8"`` at the call site (lint rule
``PLW1514`` / ``PYI058``). Ruff is the right tool for that sweep.
What this module does on POSIX:
- Nothing. POSIX systems are already UTF-8 by default in 99% of cases,
and we don't want to touch ``LANG``/``LC_*`` behavior that users may
have configured intentionally. If someone hits a C/POSIX locale on
Linux, they can export ``PYTHONUTF8=1`` themselves — we won't override.
Idempotent: safe to call multiple times. ``_bootstrap_once`` guards
against double-reconfigure.
""" """
from __future__ import annotations from __future__ import annotations
@@ -57,97 +19,45 @@ _bootstrap_applied = False
def apply_windows_utf8_bootstrap() -> bool: def apply_windows_utf8_bootstrap() -> bool:
"""Apply the Windows UTF-8 bootstrap if we're on Windows. """Apply the Windows UTF-8 bootstrap once; True only when it was applied this call."""
Returns True if bootstrap was applied (i.e. we're on Windows and
haven't already done this), False otherwise. The return value is
advisory — callers normally don't need it, but tests may want to
assert the path was taken.
Idempotent: subsequent calls after the first are a no-op.
"""
global _bootstrap_applied global _bootstrap_applied
if not _IS_WINDOWS: if not _IS_WINDOWS or _bootstrap_applied:
return False
if _bootstrap_applied:
return False return False
# 1. Child processes inherit these and run in UTF-8 mode. # setdefault() so a user can opt out with PYTHONUTF8=0 / PYTHONIOENCODING=...
# We use setdefault() rather than overwriting so the user can
# explicitly opt out by setting PYTHONUTF8=0 in their environment
# (or PYTHONIOENCODING=something-else) if they really want to.
os.environ.setdefault("PYTHONUTF8", "1") os.environ.setdefault("PYTHONUTF8", "1")
os.environ.setdefault("PYTHONIOENCODING", "utf-8") os.environ.setdefault("PYTHONIOENCODING", "utf-8")
# 2. Reconfigure the current process's stdio to UTF-8. Needed # os.environ changes don't rebind streams bound at interpreter startup, so
# because os.environ changes don't retroactively rebind sys.stdout # reconfigure them in-process. errors="replace" keeps a non-UTF-8 legacy
# — those were bound at interpreter startup based on the console # pipe on stdin from crashing us (U+FFFD instead of an exception).
# code page. ``reconfigure`` is a TextIOWrapper method since 3.7. # Non-TextIOWrapper streams (BytesIO in tests, embedded hosts) have no
# # reconfigure(): skip — the env-var fix for children is the bigger win.
# errors="replace" means that if we ever *read* something from for stream_name in ("stdout", "stderr", "stdin"):
# stdin that isn't UTF-8 (unlikely but possible with piped input reconfigure = getattr(getattr(sys, stream_name, None), "reconfigure", None)
# from legacy tools), we'll get U+FFFD replacement chars rather
# than a crash. Output is pure UTF-8.
for stream_name in ("stdout", "stderr"):
stream = getattr(sys, stream_name, None)
if stream is None:
continue
reconfigure = getattr(stream, "reconfigure", None)
if reconfigure is None: if reconfigure is None:
# Not a TextIOWrapper (could be redirected to a BytesIO in
# tests, or a non-standard stream in some embedded cases).
# Skip silently — the env-var fix is still in effect for
# child processes, which is the bigger win.
continue continue
try: try:
reconfigure(encoding="utf-8", errors="replace") reconfigure(encoding="utf-8", errors="replace")
except (OSError, ValueError): except (OSError, ValueError):
# Already closed, or someone replaced it with something pass # closed, or replaced with something non-reconfigurable
# non-reconfigurable. Non-fatal.
pass
# stdin is reconfigured separately with errors="replace" too — input
# from a legacy pipe shouldn't crash the process.
stdin = getattr(sys, "stdin", None)
if stdin is not None:
reconfigure = getattr(stdin, "reconfigure", None)
if reconfigure is not None:
try:
reconfigure(encoding="utf-8", errors="replace")
except (OSError, ValueError):
pass
_bootstrap_applied = True _bootstrap_applied = True
return True return True
def suppress_platform_ver_console() -> None: def suppress_platform_ver_console() -> None:
"""Stub ``platform._syscmd_ver`` on Windows — decode-crash + flash guard. """Stub ``platform._syscmd_ver`` on Windows — decode-crash + console-flash guard.
CPython's ``platform.win32_ver()`` (reached via ``platform.uname()`` / ``platform.win32_ver()`` (reached via ``platform.platform()``, which the OpenAI SDK
``platform.platform()``, which the OpenAI SDK touches for its calls) shells out ``cmd /c ver`` with ``shell=True`` and no ``CREATE_NO_WINDOW``: a
platform headers) shells out ``cmd /c ver``. Two failure modes: windowless parent (pythonw gateway, slash/kanban workers) flashes a console per call,
and Python 3.11.0/3.11.1 (no ``encoding="locale"`` fix) strict-utf-8-decodes the OEM
- **Console flash**: the ``check_output(..., shell=True)`` call has no code page output under PEP 540 mode and raises (#69413). Returning the inputs makes
``CREATE_NO_WINDOW``, so a windowless parent (pythonw gateway, slash ``win32_ver()`` fall back to ``sys.getwindowsversion()`` — same data, no subprocess.
workers, kanban workers) flashes a visible console per call. Mirrors ``hermes_cli._subprocess_compat.suppress_platform_ver_console`` for callers
- **UnicodeDecodeError on Python 3.11.0/3.11.1**: those micros lack that never import ``hermes_cli.main``; double application is harmless.
CPython's ``encoding="locale"`` fix (added 3.11.2), so under PEP 540
UTF-8 mode (which we enable above) the ``ver`` output — OEM code page
bytes on localized Windows — is strict-utf-8 decoded and raises,
crashing ``platform.platform()`` in any process that inherits
``PYTHONUTF8=1`` (issue #69413).
Stubbing ``_syscmd_ver`` to return its inputs makes ``win32_ver()`` hit
its documented fallback and read the version from
``sys.getwindowsversion()`` — same data, in-process, no subprocess.
Mirrors ``hermes_cli._subprocess_compat.suppress_platform_ver_console``
(kept there for callers that don't import bootstrap); double
application is harmless. Lives here so EVERY entry point gets it —
``tui_gateway/slash_worker.py``, ``tui_gateway/entry.py``,
``run_agent.py``, ``batch_runner.py``, and ``cli.py`` import only
``hermes_bootstrap``, never ``hermes_cli.main``.
""" """
if not _IS_WINDOWS: if not _IS_WINDOWS:
return return
@@ -161,34 +71,19 @@ def suppress_platform_ver_console() -> None:
platform._syscmd_ver = _quiet_syscmd_ver platform._syscmd_ver = _quiet_syscmd_ver
except Exception: except Exception:
# Hardening only — never let it break an entry point. pass # hardening only — never break an entry point
pass
def harden_import_path(src_root: str | None = None) -> None: def harden_import_path(src_root: str | None = None) -> None:
"""Stop a package in the current directory from shadowing Hermes modules. """Stop a package in the current directory from shadowing Hermes modules.
Hermes ships top-level modules with common names (``utils``, ``proxy``, Hermes ships top-level modules with common names (``utils``, ``proxy``, ``ui``); a
``ui``). Python always seeds ``sys.path`` with the current directory, so project with its own ``utils/`` launched from its directory would win the import.
launching an entry point from a project that has its own ``utils/`` package The cwd reaches ``sys.path`` as ``""``/``"."`` (script/``-m`` launches) AND as an
makes ``from utils import ...`` resolve to the *user's* package and crash absolute path (venv activation, PYTHONPATH), so both are handled: relative forms are
with an ImportError before the gateway can even start. dropped and the Hermes root is *relocated* to the front, not merely inserted when
absent. ``src_root`` defaults to this module's directory (the repo root for every
The current directory reaches ``sys.path`` two ways, and a complete guard shipped entry point), so no spawner env var is required.
has to handle both:
- As the empty string ``""`` (or ``"."``) that Python inserts at
``sys.path[0]`` for ``-m`` / script launches.
- As its own *absolute* path, when a venv activation or a project that
adds itself to ``PYTHONPATH`` puts the directory there explicitly.
We drop the relative forms outright, then force the real Hermes source root
to the front — relocating it ahead of any absolute cwd entry rather than
only inserting when absent, so an absolute cwd path can't keep winning.
``src_root`` defaults to the directory this module lives in, which is the
repository root for every shipped entry point, so the guard is
self-sufficient and does not depend on the spawner exporting an env var.
""" """
root = src_root or os.environ.get("HERMES_PYTHON_SRC_ROOT") or os.path.dirname( root = src_root or os.environ.get("HERMES_PYTHON_SRC_ROOT") or os.path.dirname(
os.path.abspath(__file__) os.path.abspath(__file__)
@@ -202,18 +97,12 @@ def harden_import_path(src_root: str | None = None) -> None:
def activate_durable_lazy_target() -> None: def activate_durable_lazy_target() -> None:
"""Put the durable lazy-install dir on ``sys.path`` if one is configured. """Put the durable lazy-install dir (``HERMES_LAZY_INSTALL_TARGET``) on ``sys.path``.
On immutable Docker images the agent venv is sealed and lazy installs Immutable Docker images seal the venv and redirect lazy installs to the data volume;
are redirected to a writable dir on the data volume packages installed there on a previous run must be importable before any backend
(``HERMES_LAZY_INSTALL_TARGET``, e.g. ``/opt/data/lazy-packages``). imports its SDK. Appends to the END of ``sys.path`` so the core venv always wins name
Packages installed there on a previous run must be importable on this collisions (see ``tools.lazy_deps``). Never raises; unset target is a no-op.
run, so we activate the dir here — at the very first import, before any
backend module imports its SDK.
The activation appends to the END of ``sys.path`` so the core venv
always wins name collisions (see ``tools.lazy_deps`` for the full
security rationale). Never raises; a missing/empty target is a no-op.
""" """
if not os.environ.get("HERMES_LAZY_INSTALL_TARGET", "").strip(): if not os.environ.get("HERMES_LAZY_INSTALL_TARGET", "").strip():
return return
@@ -221,19 +110,10 @@ def activate_durable_lazy_target() -> None:
from tools import lazy_deps from tools import lazy_deps
lazy_deps.activate_durable_lazy_target() lazy_deps.activate_durable_lazy_target()
except Exception: except Exception:
# Bootstrap must never crash an entry point. If activation fails the pass # a failed activation just leaves the backend reporting itself unavailable
# backend simply reports itself unavailable, exactly as before.
pass
# Apply on import — entry points just need ``import hermes_bootstrap`` # Apply on import — entry points only need ``import hermes_bootstrap`` first.
# (or ``from hermes_bootstrap import apply_windows_utf8_bootstrap``) at
# the very top of their module, before importing anything else. The
# import side effect does the right thing.
apply_windows_utf8_bootstrap() apply_windows_utf8_bootstrap()
suppress_platform_ver_console() suppress_platform_ver_console()
# Activate the durable lazy-install target (immutable Docker images) so
# packages installed into the data volume on a previous run are importable
# this run, before any backend module imports its SDK. No-op when unset.
activate_durable_lazy_target() activate_durable_lazy_target()
+100 -171
View File
@@ -1,12 +1,9 @@
"""Centralized logging setup for Hermes Agent. """Centralized logging setup for Hermes Agent.
Log files produced: agent.log — INFO+, all agent/tool/session activity (the main log) errors.log — Log files: agent.log (INFO+, everything), errors.log (WARNING+), gateway.log (INFO+,
WARNING+, errors and warnings only (quick triage) gateway.log — INFO+, gateway-only events (created gateway components; ``mode="gateway"``), gui.log (INFO+, dashboard/TUI-gateway;
when mode="gateway") gui.log — INFO+, dashboard/websocket/TUI-gateway events (created when ``mode="gui"``). All are rotating files driven through one async queue and formatted
mode="gui") with ``RedactingFormatter`` so secrets never reach disk.
All files use ``RotatingFileHandler`` with ``RedactingFormatter`` so secrets are never written to
disk.
""" """
import atexit import atexit
@@ -21,28 +18,15 @@ from logging.handlers import QueueHandler, QueueListener
from pathlib import Path from pathlib import Path
from typing import Optional, Sequence from typing import Optional, Sequence
# On Windows, stdlib ``RotatingFileHandler`` calls ``os.rename()`` in # Windows-ONLY swap (#44873): stdlib ``RotatingFileHandler.doRollover()`` calls
# ``doRollover()`` and fails with ``PermissionError [WinError 32]`` whenever # ``os.rename()``, which fails with ``PermissionError [WinError 32]`` whenever
# another process holds an append-mode handle on ``agent.log`` — which is # another process holds an append handle on ``agent.log`` — essentially always
# essentially always in Hermes (TUI, gateway, ``hy_memory`` server, MCP # in Hermes (TUI, gateway, hy_memory, MCP servers, CLI commands all log) —
# servers, and on-demand CLI commands all log from separate processes), # pinning the file at the size threshold and spamming stderr on every emit.
# pinning ``agent.log`` at the 5 MiB threshold and spamming stderr with # ``concurrent-log-handler`` serializes rollover with a cross-process lock.
# a traceback on every emit. ``concurrent-log-handler`` wraps the rename in a # POSIX keeps stdlib: renames of open files work, and managed mode (NixOS)
# cross-process file lock (via ``portalocker``: pywin32 on Windows) so only # relies on stdlib's exact ``_open()``/``doRollover()`` lifecycle for the
# one process rotates at a time and the others wait their turn. # 0660 chmod and eager file creation; CLH opens lazily and rotates differently.
#
# This swap is Windows-ONLY and deliberately so:
# * The bug (WinError 32 on rename-while-open) is specific to Windows file
# locking semantics — POSIX renames an open file fine, so stdlib already
# works correctly on Linux/macOS.
# * On POSIX, managed-mode (NixOS) relies on the exact ``_open()`` /
# ``doRollover()`` lifecycle of stdlib ``RotatingFileHandler`` (the
# ``_ManagedRotatingFileHandler`` subclass chmods 0660 after each). CLH
# opens lazily and rotates differently, which breaks the group-writable
# guarantee and the eager file-creation those paths depend on.
# Aliasing keeps every existing ``RotatingFileHandler`` reference in this
# module (class declaration, ``isinstance`` checks, docstring) working
# unchanged. See #44873.
if sys.platform == "win32": if sys.platform == "win32":
from concurrent_log_handler import ( # noqa: E402 from concurrent_log_handler import ( # noqa: E402
ConcurrentRotatingFileHandler as RotatingFileHandler, ConcurrentRotatingFileHandler as RotatingFileHandler,
@@ -53,17 +37,13 @@ else:
from hermes_constants import get_config_path, get_hermes_home, mkdir_under_hermes_home from hermes_constants import get_config_path, get_hermes_home, mkdir_under_hermes_home
# Sentinel to track whether setup_logging() has already run. The function # setup_logging() is idempotent: a second call is a no-op unless ``force=True``.
# is idempotent — calling it twice is safe but the second call is a no-op
# unless ``force=True``.
_logging_initialized = False _logging_initialized = False
# Thread-local storage for per-conversation session context. # Thread-local per-conversation session context.
_session_context = threading.local() _session_context = threading.local()
# Default log format — includes timestamp, level, optional session tag, # ``%(session_tag)s`` exists on every LogRecord via _install_session_record_factory().
# logger name, and message. The ``%(session_tag)s`` field is guaranteed to
# exist on every LogRecord via _install_session_record_factory() below.
_LOG_FORMAT = "%(asctime)s %(levelname)s%(session_tag)s %(name)s: %(message)s" _LOG_FORMAT = "%(asctime)s %(levelname)s%(session_tag)s %(name)s: %(message)s"
_LOG_FORMAT_VERBOSE = "%(asctime)s - %(name)s - %(levelname)s%(session_tag)s - %(message)s" _LOG_FORMAT_VERBOSE = "%(asctime)s - %(name)s - %(levelname)s%(session_tag)s - %(message)s"
@@ -71,17 +51,16 @@ _LOG_FORMAT_VERBOSE = "%(asctime)s - %(name)s - %(levelname)s%(session_tag)s - %
def _safe_stderr(): # type: ignore[return] def _safe_stderr(): # type: ignore[return]
"""Return a stderr stream that tolerates Unicode on all platforms. """Return a stderr stream that tolerates Unicode on all platforms.
We wrap ``sys.stderr`` in a ``TextIOWrapper`` with ``errors='replace'`` so log lines are never Wraps ``sys.stderr`` with ``errors='replace'`` so un-encodable characters become
lost — un-encodable characters are replaced with ``?`` instead of crashing the process. ``?`` instead of crashing the process.
""" """
stream = sys.stderr stream = sys.stderr
encoding = getattr(stream, "encoding", None) or "utf-8" encoding = getattr(stream, "encoding", None) or "utf-8"
# Already UTF-8 or surrogate-aware — no wrapping needed.
if encoding.lower().replace("-", "") in ("utf8", "utf8surrogateescape"): if encoding.lower().replace("-", "") in ("utf8", "utf8surrogateescape"):
return stream return stream
try: try:
wrapped = io.TextIOWrapper(stream.buffer, encoding="utf-8", errors="replace", line_buffering=True) wrapped = io.TextIOWrapper(stream.buffer, encoding="utf-8", errors="replace", line_buffering=True)
# Prevent the wrapper from closing the underlying buffer when it is garbage-collected. # Prevent the wrapper from closing the underlying buffer when garbage-collected.
wrapped.close = lambda: None # type: ignore[assignment] wrapped.close = lambda: None # type: ignore[assignment]
return wrapped return wrapped
except Exception: except Exception:
@@ -89,12 +68,11 @@ def _safe_stderr(): # type: ignore[return]
def _is_windows_concurrent_log_lock_timeout(exc: BaseException | None) -> bool: def _is_windows_concurrent_log_lock_timeout(exc: BaseException | None) -> bool:
"""Return True for concurrent-log-handler's Windows lock timeout. """True for concurrent-log-handler's Windows lock timeout.
On Windows Desktop, slash-command workers and the gateway can all write to the same rotating log Slash-command workers and the gateway share rotating files on Windows Desktop;
files. ``concurrent-log-handler`` serializes rollover with a cross-process lock, but when when another process holds the rollover lock too long CLH raises this
another process holds that lock too long it raises this RuntimeError. Logging failures should RuntimeError, which must not escape into Desktop chat output.
not escape into Desktop chat output.
""" """
return ( return (
sys.platform == "win32" sys.platform == "win32"
@@ -117,8 +95,6 @@ def _quiet_noisy_loggers() -> None:
logging.getLogger(name).setLevel(logging.WARNING) logging.getLogger(name).setLevel(logging.WARNING)
# Public session context API
def set_session_context(session_id: str) -> None: def set_session_context(session_id: str) -> None:
"""Set the session ID for the current thread.""" """Set the session ID for the current thread."""
_session_context.session_id = session_id _session_context.session_id = session_id
@@ -129,27 +105,24 @@ def clear_session_context() -> None:
_session_context.session_id = None _session_context.session_id = None
# Record factory — injects session_tag into every LogRecord at creation
def _install_session_record_factory() -> None: def _install_session_record_factory() -> None:
"""Replace the global LogRecord factory with one that adds ``session_tag``. """Replace the global LogRecord factory with one that adds ``session_tag``.
Unlike a handler/logger ``Filter``, the record factory runs for EVERY record in the process, Unlike a Filter, the record factory runs for EVERY record in the process (propagated
including propagated and third-party-handled ones, so ``%(session_tag)s`` is always available and third-party-handled ones included), so ``%(session_tag)s`` never KeyErrors.
and never KeyErrors. Idempotent: a marker attribute prevents double-wrapping on reload. Idempotent via a marker attribute.
""" """
current_factory = logging.getLogRecordFactory() current_factory = logging.getLogRecordFactory()
if getattr(current_factory, "_hermes_session_injector", False): if getattr(current_factory, "_hermes_session_injector", False):
return # already installed return
def _session_record_factory(*args, **kwargs): def _session_record_factory(*args, **kwargs):
record = current_factory(*args, **kwargs) record = current_factory(*args, **kwargs)
sid = getattr(_session_context, "session_id", None) sid = getattr(_session_context, "session_id", None)
record.session_tag = f" [{sid}]" if sid else "" # type: ignore[attr-defined] record.session_tag = f" [{sid}]" if sid else "" # type: ignore[attr-defined]
# QueueListener formats records on its own thread, after the # QueueListener formats on its own thread, after the profile-scoped
# profile-scoped ContextVar has gone out of scope. Keep the resolved # ContextVar is gone; keep the resolved home on the record so a
# home on the record so a multiplex desktop ticker can route the log # multiplex desktop ticker can route to the job owner's files (#97489).
# to the job owner's files (#97489).
try: try:
record.hermes_home = str(get_hermes_home().resolve()) # type: ignore[attr-defined] record.hermes_home = str(get_hermes_home().resolve()) # type: ignore[attr-defined]
except Exception: except Exception:
@@ -160,13 +133,10 @@ def _install_session_record_factory() -> None:
logging.setLogRecordFactory(_session_record_factory) logging.setLogRecordFactory(_session_record_factory)
# Install immediately on import — session_tag is available on all records # Install on import so session_tag exists on all records even before setup_logging().
# from this point forward, even before setup_logging() is called.
_install_session_record_factory() _install_session_record_factory()
# Filters
class _ComponentFilter(logging.Filter): class _ComponentFilter(logging.Filter):
"""Only pass records whose logger name starts with one of *prefixes*.""" """Only pass records whose logger name starts with one of *prefixes*."""
@@ -178,13 +148,11 @@ class _ComponentFilter(logging.Filter):
return record.name.startswith(self._prefixes) return record.name.startswith(self._prefixes)
# Logger name prefixes that belong to each component. # Logger name prefixes per component; used by _ComponentFilter and ``hermes logs --component``.
# Used by _ComponentFilter and exposed for ``hermes logs --component``.
COMPONENT_PREFIXES = { COMPONENT_PREFIXES = {
# ``plugins.platforms`` covers messaging-platform adapters that migrated # ``plugins.platforms``: messaging adapters that migrated out of
# out of ``gateway/platforms/`` into bundled plugins (#41112) — they are # ``gateway/platforms/`` into bundled plugins (#41112) are still gateway
# still gateway components and their logs belong in gateway.log / match # components and belong in gateway.log.
# ``hermes logs --component gateway``.
"gateway": ("gateway", "hermes_plugins", "plugins.platforms"), "gateway": ("gateway", "hermes_plugins", "plugins.platforms"),
"agent": ("agent", "run_agent", "model_tools", "batch_runner"), "agent": ("agent", "run_agent", "model_tools", "batch_runner"),
"tools": ("tools",), "tools": ("tools",),
@@ -194,8 +162,6 @@ COMPONENT_PREFIXES = {
} }
# Main setup
def setup_logging( def setup_logging(
*, *,
hermes_home: Optional[Path] = None, hermes_home: Optional[Path] = None,
@@ -205,18 +171,16 @@ def setup_logging(
mode: Optional[str] = None, mode: Optional[str] = None,
force: bool = False, force: bool = False,
) -> Path: ) -> Path:
"""Configure the Hermes logging subsystem. """Configure the Hermes logging subsystem; returns the ``logs/`` directory.
Safe to call multiple times; the second call is a no-op unless *force* is ``True``. Level and Safe to call multiple times; the second call is a no-op unless *force*. Level and
rotation defaults come from config.yaml ``logging.*``. ``mode="gateway"`` adds ``gateway.log`` rotation defaults come from config.yaml ``logging.*``. ``mode="gateway"`` adds
(gateway components only) and ``mode="gui"`` adds ``gui.log`` (dashboard / TUI-gateway). ``gateway.log`` and ``mode="gui"`` adds ``gui.log``.
Returns the ``logs/`` directory.
""" """
global _logging_initialized global _logging_initialized
home = hermes_home or get_hermes_home() home = hermes_home or get_hermes_home()
log_dir = mkdir_under_hermes_home(home / "logs") log_dir = mkdir_under_hermes_home(home / "logs")
# Read config defaults (best-effort — config may not be loaded yet).
cfg_level, cfg_max_size, cfg_backup = _read_logging_config() cfg_level, cfg_max_size, cfg_backup = _read_logging_config()
level_name = (log_level or cfg_level or "INFO").upper() level_name = (log_level or cfg_level or "INFO").upper()
@@ -224,13 +188,12 @@ def setup_logging(
max_bytes = (max_size_mb or cfg_max_size or 5) * 1024 * 1024 max_bytes = (max_size_mb or cfg_max_size or 5) * 1024 * 1024
backups = backup_count or cfg_backup or 3 backups = backup_count or cfg_backup or 3
# Lazy import to avoid circular dependency at module load time. from agent.redact import RedactingFormatter # lazy: circular at module load
from agent.redact import RedactingFormatter
root = logging.getLogger() root = logging.getLogger()
# (filename, level, max_bytes, backup_count, component) — component gates on ``mode`` and # (filename, level, max_bytes, backup_count, component) — a component gates
# restricts the file to that component's logger prefixes. # the file on ``mode`` and restricts it to that component's logger prefixes.
handler_specs = ( handler_specs = (
("agent.log", level, max_bytes, backups, None), ("agent.log", level, max_bytes, backups, None),
("errors.log", logging.WARNING, 2 * 1024 * 1024, 2, None), ("errors.log", logging.WARNING, 2 * 1024 * 1024, 2, None),
@@ -249,7 +212,7 @@ def setup_logging(
if _logging_initialized and not force: if _logging_initialized and not force:
return log_dir return log_dir
# Ensure root logger level is low enough for the handlers to fire. # Root level must be low enough for the handlers to fire.
if root.level == logging.NOTSET or root.level > level: if root.level == logging.NOTSET or root.level > level:
root.setLevel(level) root.setLevel(level)
@@ -265,7 +228,6 @@ def setup_verbose_logging() -> None:
root = logging.getLogger() root = logging.getLogger()
# Avoid adding duplicate stream handlers.
if any(getattr(h, "_hermes_verbose", False) for h in root.handlers): if any(getattr(h, "_hermes_verbose", False) for h in root.handlers):
return return
@@ -275,7 +237,6 @@ def setup_verbose_logging() -> None:
handler._hermes_verbose = True # type: ignore[attr-defined] handler._hermes_verbose = True # type: ignore[attr-defined]
root.addHandler(handler) root.addHandler(handler)
# Lower root logger level so DEBUG records reach all handlers.
if root.level > logging.DEBUG: if root.level > logging.DEBUG:
root.setLevel(logging.DEBUG) root.setLevel(logging.DEBUG)
@@ -284,8 +245,6 @@ def setup_verbose_logging() -> None:
logging.getLogger("rex-deploy").setLevel(logging.INFO) logging.getLogger("rex-deploy").setLevel(logging.INFO)
# Internal helpers
def _quietly(fn) -> None: def _quietly(fn) -> None:
"""Call *fn* (a ``close``/``stop`` bound method) swallowing errors — teardown must never raise.""" """Call *fn* (a ``close``/``stop`` bound method) swallowing errors — teardown must never raise."""
try: try:
@@ -295,21 +254,19 @@ def _quietly(fn) -> None:
class _ManagedRotatingFileHandler(RotatingFileHandler): class _ManagedRotatingFileHandler(RotatingFileHandler):
"""RotatingFileHandler that ensures group-writable perms in managed mode """RotatingFileHandler with managed-mode perms and external-rotation detection.
In managed mode (NixOS) the setgid stateDir needs group-readable files, but ``_open()`` and In managed mode (NixOS) the setgid stateDir needs group-readable files, but
``doRollover()`` honor the umask (0644), so ``chmod 0660`` is applied after both. Also, a ``_open()``/``doRollover()`` honor the umask (0644), so ``chmod 0660`` follows both.
rotating handler holds an fd: if the file is rotated externally (logrotate, ``mv``) writes A rotating handler also holds an fd: if the file is rotated externally (logrotate,
silently go to the old inode, so before each emit the path's inode is compared to the open ``mv``) writes silently go to the old inode, so each emit compares the path's inode
stream's and the file reopened on mismatch (the ``WatchedFileHandler`` pattern). to the open stream's and reopens on mismatch (the ``WatchedFileHandler`` pattern).
""" """
def __init__(self, *args, **kwargs): def __init__(self, *args, **kwargs):
from hermes_cli.config import is_managed from hermes_cli.config import is_managed
self._managed = is_managed() self._managed = is_managed()
super().__init__(*args, **kwargs) super().__init__(*args, **kwargs)
# Snapshot the inode of the currently open stream so emit() can
# detect external rotation without an extra fstat per write.
self._record_stream_stat() self._record_stream_stat()
def _chmod_if_managed(self): def _chmod_if_managed(self):
@@ -320,7 +277,7 @@ class _ManagedRotatingFileHandler(RotatingFileHandler):
pass pass
def _record_stream_stat(self, st: Optional[os.stat_result] = None) -> None: def _record_stream_stat(self, st: Optional[os.stat_result] = None) -> None:
"""Snapshot dev/ino of ``baseFilename`` so we can detect external rotation.""" """Snapshot dev/ino of ``baseFilename`` so emit() can detect external rotation."""
try: try:
st = st or os.stat(self.baseFilename) st = st or os.stat(self.baseFilename)
self._stat_dev, self._stat_ino = st.st_dev, st.st_ino self._stat_dev, self._stat_ino = st.st_dev, st.st_ino
@@ -328,10 +285,10 @@ class _ManagedRotatingFileHandler(RotatingFileHandler):
self._stat_dev, self._stat_ino = None, None self._stat_dev, self._stat_ino = None, None
def _reopen_stream(self, stat_result=None) -> None: def _reopen_stream(self, stat_result=None) -> None:
"""Close the current stream and open ``baseFilename`` afresh (best-effort). """Close and reopen ``baseFilename`` (best-effort).
On failure the stream is left ``None`` so the next emit bails rather than writing to a On failure the stream is left ``None`` so the next emit bails rather than
stale inode. writing to a stale inode.
""" """
if self.stream is not None: if self.stream is not None:
_quietly(self.stream.close) _quietly(self.stream.close)
@@ -343,18 +300,15 @@ class _ManagedRotatingFileHandler(RotatingFileHandler):
self._record_stream_stat(stat_result) self._record_stream_stat(stat_result)
def _reopen_if_externally_rotated(self) -> None: def _reopen_if_externally_rotated(self) -> None:
"""Reopen the stream when ``baseFilename`` no longer matches our fd. """Reopen when ``baseFilename`` was renamed, unlinked, or replaced by another inode.
Triggered when ``baseFilename`` was renamed (logrotate), unlinked, or replaced by a Silent + best-effort: any error falls back to the existing (possibly stale)
different inode. Silent + best-effort: any error falls back to the existing (possibly stale)
stream so logging keeps working instead of dying on a stat failure. stream so logging keeps working instead of dying on a stat failure.
""" """
try: try:
st = os.stat(self.baseFilename) st = os.stat(self.baseFilename)
except FileNotFoundError: except FileNotFoundError:
# File was rotated/unlinked underneath us: reopen so a fresh inode self._reopen_stream() # rotated/unlinked underneath us: recreate at the path
# is created at the expected path.
self._reopen_stream()
return return
except OSError: except OSError:
return # transient — try again on the next emit return # transient — try again on the next emit
@@ -362,22 +316,20 @@ class _ManagedRotatingFileHandler(RotatingFileHandler):
if self._stat_dev is None or self._stat_ino is None: if self._stat_dev is None or self._stat_ino is None:
self._record_stream_stat(st) self._record_stream_stat(st)
elif (st.st_dev, st.st_ino) != (self._stat_dev, self._stat_ino): elif (st.st_dev, st.st_ino) != (self._stat_dev, self._stat_ino):
# baseFilename now points at a DIFFERENT inode than the one we hold open.
self._reopen_stream(st) self._reopen_stream(st)
def emit(self, record: logging.LogRecord) -> None: def emit(self, record: logging.LogRecord) -> None:
# Cheap-ish stat-per-record check; the kernel caches inode metadata # The kernel caches inode metadata, so this stat is sub-microsecond on a hot file.
# so the syscall is sub-microsecond on a hot file.
if self.stream is not None or os.path.exists(self.baseFilename): if self.stream is not None or os.path.exists(self.baseFilename):
self._reopen_if_externally_rotated() self._reopen_if_externally_rotated()
super().emit(record) super().emit(record)
def handleError(self, record: logging.LogRecord) -> None: def handleError(self, record: logging.LogRecord) -> None:
"""Suppress the known Windows ``concurrent-log-handler`` lock timeout """Suppress the known Windows ``concurrent-log-handler`` lock timeout.
CLH's ``emit()`` catches the ``"Cannot acquire lock after N attempts"`` RuntimeError and CLH's ``emit()`` routes that RuntimeError here, so this is the single point to
routes it here, so this override is the single point to silence it before stdlib prints to silence it before stdlib prints to stderr (which the Desktop slash-worker
stderr (which the Desktop slash-worker captures and surfaces into chat output). captures into chat output).
""" """
if not _is_windows_concurrent_log_lock_timeout(sys.exc_info()[1]): if not _is_windows_concurrent_log_lock_timeout(sys.exc_info()[1]):
super().handleError(record) super().handleError(record)
@@ -390,8 +342,8 @@ class _ManagedRotatingFileHandler(RotatingFileHandler):
def doRollover(self): def doRollover(self):
super().doRollover() super().doRollover()
self._chmod_if_managed() self._chmod_if_managed()
# Our own rollover writes a new baseFilename; refresh the snapshot # Our own rollover writes a new baseFilename; refresh the snapshot so
# so the next emit doesn't mistake it for external rotation. # the next emit doesn't mistake it for external rotation.
self._record_stream_stat() self._record_stream_stat()
@@ -411,9 +363,8 @@ def _new_file_handler(
class _ProfileRoutingFileHandler(logging.Handler): class _ProfileRoutingFileHandler(logging.Handler):
"""Route queued records to the log file for their Hermes home. """Route queued records to the log file for their Hermes home.
The handler itself is used only behind the existing QueueListener, so its small routing lock Used only behind the QueueListener, so its small routing lock never blocks an agent
never blocks an agent or dashboard event loop. The underlying handlers retain the existing or dashboard event loop. Per-home handlers keep rotation, redaction and managed perms.
rotation, redaction, and managed permission behavior.
""" """
def __init__(self, existing: RotatingFileHandler, profile_homes: Sequence[Path]) -> None: def __init__(self, existing: RotatingFileHandler, profile_homes: Sequence[Path]) -> None:
@@ -465,15 +416,10 @@ class _ProfileRoutingFileHandler(logging.Handler):
super().close() super().close()
# Asynchronous file logging — keep the cross-process rotation lock off the loop # Asynchronous file logging: an ``emit`` can block on the cross-process
# # rotation lock (module header); on an asyncio thread that stalls the loop and
# The rotating file handlers serialize rollover with a cross-process lock (see # drops WebSocket clients. Every file handler is therefore driven by a single
# the module header): when several Hermes processes log to the same file, an # QueueListener thread; loggers only do a non-blocking enqueue.
# ``emit`` can block while another process holds that lock. When the emitting
# thread is an asyncio event loop, that block stalls the loop and drops
# WebSocket clients. To keep file I/O off the hot path, every file handler is
# driven by a single ``QueueListener`` on a dedicated thread; loggers only touch
# an in-memory queue (a non-blocking enqueue).
_log_queue: "Optional[queue.SimpleQueue]" = None _log_queue: "Optional[queue.SimpleQueue]" = None
_queue_listener: Optional[QueueListener] = None _queue_listener: Optional[QueueListener] = None
@@ -481,20 +427,19 @@ _queued_file_handlers: list = []
_queue_atexit_registered = False _queue_atexit_registered = False
# Guards every read-modify-write of the four globals above. setup_logging() # Guards every read-modify-write of the four globals above. setup_logging()
# holds no lock and its _logging_initialized guard runs AFTER handler # holds no lock and its _logging_initialized guard runs AFTER handler
# registration, so _register_queued_handler() can run concurrently with a # registration, so _register_queued_handler() can race a flush/reset from
# flush/reset from another thread (gateway init racing a plugin/CLI path). # another thread (gateway init vs a plugin/CLI path); without this, two
# Without this, two threads can interleave listener.stop()/reassign/start() # threads can interleave stop()/reassign/start() and leave two live listeners.
# and leave the queue with two live listeners or an orphaned worker thread.
_queue_state_lock = threading.Lock() _queue_state_lock = threading.Lock()
class _NonFormattingQueueHandler(QueueHandler): class _NonFormattingQueueHandler(QueueHandler):
"""``QueueHandler`` for an in-process queue. """``QueueHandler`` for an in-process queue.
Stdlib ``prepare()`` formats and strips ``args``/``exc_info`` for pickling across processes; Stdlib ``prepare()`` formats and strips ``args``/``exc_info`` for cross-process
our queue is in-process, so the target handlers get an unformatted record and apply their own pickling; ours is in-process, so targets get the unformatted record and apply their
``RedactingFormatter`` on the listener thread. A shallow copy is returned because the emitting own ``RedactingFormatter`` on the listener thread. A shallow copy is returned because
thread's synchronous handlers may mutate ``record.message`` while the listener reads it. the emitting thread's synchronous handlers may mutate ``record.message`` meanwhile.
""" """
def prepare(self, record: logging.LogRecord) -> logging.LogRecord: def prepare(self, record: logging.LogRecord) -> logging.LogRecord:
@@ -502,10 +447,7 @@ class _NonFormattingQueueHandler(QueueHandler):
def _stop_queue_listener() -> None: def _stop_queue_listener() -> None:
"""Flush and stop the background log listener (idempotent, thread-safe). """Flush and stop the background log listener (idempotent; atexit hook, so it takes the lock)."""
This is the atexit hook, so it must acquire the state lock itself.
"""
global _queue_listener global _queue_listener
with _queue_state_lock: with _queue_state_lock:
listener, _queue_listener = _queue_listener, None listener, _queue_listener = _queue_listener, None
@@ -516,8 +458,8 @@ def _stop_queue_listener() -> None:
def _start_queue_listener_locked() -> None: def _start_queue_listener_locked() -> None:
"""(Re)build + start a listener over the current handler set (``_queue_state_lock`` held). """(Re)build + start a listener over the current handler set (``_queue_state_lock`` held).
A running listener is stopped first; this only happens while handlers are being added A running listener is stopped first; this only happens while handlers are being
(queue empty), so ``stop()`` returns immediately. added (queue empty), so ``stop()`` returns immediately.
""" """
global _queue_listener global _queue_listener
if _queue_listener is not None: if _queue_listener is not None:
@@ -527,9 +469,10 @@ def _start_queue_listener_locked() -> None:
def _register_queued_handler(handler: logging.Handler) -> None: def _register_queued_handler(handler: logging.Handler) -> None:
"""Route *handler* through the shared async queue instead of attaching it to *root* directly, so """Route *handler* through the shared async queue instead of attaching it to root.
emitting threads never block on file I/O or the cross-process rotation lock. The
``QueueListener`` applies each handler's own level and filters on its worker thread. Emitting threads never block on file I/O or the rotation lock; the ``QueueListener``
applies each handler's own level and filters on its worker thread.
""" """
global _log_queue, _queue_atexit_registered global _log_queue, _queue_atexit_registered
with _queue_state_lock: with _queue_state_lock:
@@ -537,9 +480,7 @@ def _register_queued_handler(handler: logging.Handler) -> None:
_log_queue = queue.SimpleQueue() _log_queue = queue.SimpleQueue()
qh = _NonFormattingQueueHandler(_log_queue) qh = _NonFormattingQueueHandler(_log_queue)
qh._hermes_queue = True # type: ignore[attr-defined] qh._hermes_queue = True # type: ignore[attr-defined]
# Always funnel through the root logger so records from any logger # Always on the root logger so records from any logger reach the queue.
# (production passes root here; callers may pass a child) reach the
# queue via propagation.
logging.getLogger().addHandler(qh) logging.getLogger().addHandler(qh)
_queued_file_handlers.append(handler) _queued_file_handlers.append(handler)
_start_queue_listener_locked() _start_queue_listener_locked()
@@ -553,12 +494,10 @@ def _register_queued_handler(handler: logging.Handler) -> None:
def flush_log_queue() -> None: def flush_log_queue() -> None:
"""Block until all queued records have been written, then resume. """Block until all queued records have been written, then resume.
Draining is done by stopping the listener (which processes every pending record before joining) Stops the listener (which processes every pending record before joining) and
and restarting it. Used by tests that read a log file right after emitting to it. restarts it. ``stop()`` joins the worker thread — do NOT call this on a hard-exit
path where the listener may be wedged on the rotation lock; use
NOTE: ``stop()`` joins the worker thread, so this blocks until the queue is empty. Do NOT call ``drain_log_queue()`` there, which bounds the wait.
this on a hard-exit path where the listener may be wedged on the rotation lock — use
``drain_log_queue()`` there instead, which bounds the wait.
""" """
with _queue_state_lock: with _queue_state_lock:
listener = _queue_listener listener = _queue_listener
@@ -570,10 +509,8 @@ def flush_log_queue() -> None:
def drain_log_queue(timeout: float = 1.0) -> None: def drain_log_queue(timeout: float = 1.0) -> None:
"""Best-effort, time-bounded drain for hard-exit paths (no restart). """Best-effort, time-bounded drain for hard-exit paths (no restart).
Unlike ``flush_log_queue()``, this stops the listener WITHOUT restarting it (the process is If the listener's worker is wedged on the cross-process rotation lock — the very
about to exit) and bounds the drain: if the listener's worker thread is wedged on the cross- failure async logging exists to survive — an unbounded join would re-freeze shutdown.
process rotation lock — the very failure this async-logging change exists to survive — an
unbounded ``stop()``/join would re-freeze the shutdown path.
""" """
listener = _queue_listener listener = _queue_listener
if listener is None: if listener is None:
@@ -586,12 +523,10 @@ def drain_log_queue(timeout: float = 1.0) -> None:
def enable_profile_log_routing(profile_homes: Sequence[str | Path]) -> bool: def enable_profile_log_routing(profile_homes: Sequence[str | Path]) -> bool:
"""Make the queued file logs follow a desktop profile context. """Make the queued file logs follow a desktop profile context.
``setup_logging`` normally binds handlers to one process home. The desktop dashboard is the ``setup_logging`` binds handlers to one process home; the desktop dashboard's
exception: its embedded cron ticker may run jobs for every profile. Replace the existing static embedded cron ticker may run jobs for every profile, so its static file handlers
file handlers with profile routers after that profile list is known. are replaced with profile routers once the profile list is known. Returns ``True``
when routing is (or already was) enabled; a single-profile caller is left untouched.
Returns ``True`` when routing is enabled or was already enabled. A single-profile caller is left
untouched because its existing handlers are already correctly scoped.
""" """
global _queue_listener global _queue_listener
homes: list[Path] = [] homes: list[Path] = []
@@ -654,9 +589,7 @@ def _add_rotating_handler(
formatter: logging.Formatter, formatter: logging.Formatter,
log_filter: Optional[logging.Filter] = None, log_filter: Optional[logging.Filter] = None,
) -> None: ) -> None:
"""Register a queued ``RotatingFileHandler`` for *path*, skipping if one already exists for the """Register a queued ``RotatingFileHandler`` for *path*; idempotent per resolved path."""
same resolved file path (idempotent).
"""
resolved = path.resolve() resolved = path.resolve()
for existing in _queued_file_handlers: for existing in _queued_file_handlers:
# Already attached directly, or already covered by the profile router. # Already attached directly, or already covered by the profile router.
@@ -671,19 +604,16 @@ def _add_rotating_handler(
) )
if log_filter is not None: if log_filter is not None:
handler.addFilter(log_filter) handler.addFilter(log_filter)
# Route through the async queue instead of ``logger.addHandler(handler)`` so # Queue, not ``addHandler``: the rotation-lock wait never runs on the caller's thread.
# the rotation-lock wait never runs on the caller's (often event-loop) thread.
_register_queued_handler(handler) _register_queued_handler(handler)
def _read_logging_config(): def _read_logging_config():
"""Best-effort read of ``logging.*`` from config.yaml.""" """Best-effort read of ``logging.*`` from config.yaml."""
try: try:
# Prefer the shared (mtime, size)-keyed raw-config cache so this read # Prefer the shared (mtime, size)-keyed raw-config cache so this reuses
# reuses the parse hermes_cli.main's early bridge already did (one # hermes_cli.main's early parse (one config.yaml parse per process);
# config.yaml parse per process instead of 3-4). Fall back to a # fall back to a direct parse for bare hermes_logging consumers.
# direct parse when hermes_cli.config isn't importable (bare
# hermes_logging consumers).
try: try:
from hermes_cli.config import read_raw_config as _rrc from hermes_cli.config import read_raw_config as _rrc
cfg = _rrc() or {} cfg = _rrc() or {}
@@ -696,8 +626,7 @@ def _read_logging_config():
cfg = fast_safe_load(f) or {} cfg = fast_safe_load(f) or {}
if not cfg: if not cfg:
return (None, None, None) return (None, None, None)
# Managed scope: an administrator can pin logging.* too. Overlay via # Managed scope: an administrator can pin logging.* too (fail-open overlay).
# the shared helper (fail-open) since this reads config.yaml directly.
try: try:
from hermes_cli import managed_scope from hermes_cli import managed_scope
cfg = managed_scope.apply_managed_overlay(cfg) cfg = managed_scope.apply_managed_overlay(cfg)
+129 -223
View File
@@ -1,82 +1,29 @@
"""Startup-liveness watchdog — respawn a gateway that wedges before its loop runs (OOF-298). """Startup-liveness watchdog — respawn a gateway that wedges before its loop runs (OOF-298).
The existing liveness backstops all assume startup succeeded: Every other liveness backstop (loop-liveness watchdog, shutdown watchdog, heartbeat file)
assumes startup succeeded; none can fire if the process deadlocks before the event loop
is alive (OOF-298: ~30h with every thread in ``futex_wait_queue``, zero logs, s6 saw a
live PID). A daemon thread armed at process entry and disarmed once the loop is confirmed
live dumps all-thread stacks (``faulthandler``), records the exit in the lifecycle ledger
(NS-608) and ``os._exit``\\ s with the service-restart code so s6/systemd respawn.
* the loop-liveness watchdog (:mod:`gateway.shutdown_watchdog`) is armed by Slow-but-alive startups are not killed. Order of authority: (1) phase-owned progress
``GatewayRunner._start_loop_liveness_guards`` — *inside* the running event leases (:func:`report_startup_progress`) — authoritative, prove the *startup path itself*
loop's startup path; is alive, work for I/O-bound phases (schema migration, corruption repair in
* the shutdown watchdog is armed at ``stop()``; ``SessionDB.__init__``) with ~zero CPU; (2) process-wide CPU progress, capped at
* the loop heartbeat file is written by an asyncio task. ``_MAX_CPU_EXTENSIONS`` since an unrelated thread burning CPU must not hide a parked
startup thread forever. Known limitation: a *spinning* startup deadlock earns the capped
extensions before firing; the observed class is parked-thread deadlocks. Idle-by-design
waits call :func:`kick_startup_watchdog` (respawn-storm backoff, up to 300s); MCP
discovery's 120s wait sits inside the 300s default.
None of them can fire if the process deadlocks **before the event loop comes IMPORT-LIGHTNESS IS A CORRECTNESS PROPERTY: top-level module, stdlib only. Arming must
alive**. That failure mode is real: OOF-298 documents a hosted gateway whose precede importing ``gateway`` (hundreds of modules; an import-time deadlock is in scope),
process sat for ~30 hours with every thread parked in ``futex_wait_queue``, and at fire time the wedged main thread may hold the import lock — so the fire path does
zero log lines written, ``/health`` unreachable — while s6 saw a live PID and no imports on its own thread; the ledger write runs on a helper thread joined with a
therefore never respawned it, and a stale ``gateway_state.json`` from the timeout. Config is env-only (``HERMES_STARTUP_WATCHDOG=0``,
*previous* life told every status surface the gateway was "draining". ``HERMES_STARTUP_WATCHDOG_TIMEOUT_S``) because config.yaml parsing is itself in scope.
Everything is best-effort: a watchdog failure must never affect the startup it observes.
This module closes that gap with a plain daemon OS thread armed at process
entry, disarmed the moment the event loop is confirmed live (the point where
the existing loop-liveness watchdog takes over). If startup neither reaches
that milestone nor exits within the deadline, the watchdog dumps all-thread
stacks via ``faulthandler``, records the exit in the lifecycle ledger
(NS-608) so the next boot classifies it correctly, and ``os._exit``\\ s with
the service-restart code so s6/systemd revive the process instead of
babysitting a zombie.
Slow-but-alive startups are NOT killed. Two mechanisms, in order of
authority:
1. **Phase-owned progress leases** (:func:`report_startup_progress`): a
startup phase that is about to do legitimately long synchronous work
(large ``state.db`` schema migrations, corruption repair/backup — both
run inside ``SessionDB.__init__`` well before the loop starts, and both
can be I/O-bound with near-zero CPU) declares a lease for its honest
worst case. The lease is the authoritative signal: it proves the
*startup path itself* is alive, not merely that the process is warm.
2. **CPU progress, as a bounded fallback only**: if the deadline expires
but the process consumed meaningful CPU during the window
(``time.process_time()`` is process-wide), the deadline is extended —
at most ``_MAX_CPU_EXTENSIONS`` times. Process-wide CPU proves activity,
not startup progress (an unrelated daemon thread burning CPU must not
hide a parked startup thread forever), hence the cap. Phases that hold
a current lease are never subject to the cap.
The OOF-298 deadlock class parks every thread in futex waits, accrues ~zero
CPU, and owns no lease — it fires on schedule. Known limitation, documented
deliberately: a *spinning* (busy-wait) startup deadlock reads as CPU
progress and gets the capped extensions before firing; the observed
incident class is parked-thread deadlocks, which fire immediately.
Waits that are idle-by-design get explicit handling instead:
* the respawn-storm breaker's intentional backoff sleep (up to 300s) calls
:func:`kick_startup_watchdog` with the sleep budget before sleeping;
* MCP tool discovery's internal wait is bounded at 120s, comfortably inside
the 300s default deadline.
IMPORT-LIGHTNESS IS A CORRECTNESS PROPERTY of this module, not a style
preference. It lives at the repository top level (not inside the ``gateway``
package) and imports **only stdlib** because:
1. ``gateway/__init__`` eagerly imports the config/session/delivery graph —
hundreds of modules, DB-adjacent code included. Arming must happen
*before* that graph is imported, or an import-time deadlock (a plausible
shape of "wedged before the loop, no logs") sits outside the watchdog's
coverage.
2. At fire time the main thread may be wedged **holding the import lock**;
any import attempted on the watchdog thread could then block forever.
The fire path therefore performs no imports at all on its own thread —
the lifecycle-ledger write (which does import) runs on a short-lived
helper thread joined with a timeout, and ``os._exit`` happens regardless.
Config surface is deliberately env-only (``HERMES_STARTUP_WATCHDOG=0`` to
disable, ``HERMES_STARTUP_WATCHDOG_TIMEOUT_S`` to tune): the watchdog must be
armed before config.yaml is loaded — a wedge during config parsing is exactly
in scope — so it cannot depend on config for its own enablement.
Everything here is best-effort: a watchdog failure must never affect the
startup it is observing.
""" """
from __future__ import annotations from __future__ import annotations
@@ -97,9 +44,8 @@ logger = logging.getLogger(__name__)
DEFAULT_STARTUP_WATCHDOG_TIMEOUT_S = 300.0 DEFAULT_STARTUP_WATCHDOG_TIMEOUT_S = 300.0
_MIN_TIMEOUT_S = 30.0 _MIN_TIMEOUT_S = 30.0
# Mirrors gateway.restart.GATEWAY_SERVICE_RESTART_EXIT_CODE. Duplicated here # Mirrors gateway.restart.GATEWAY_SERVICE_RESTART_EXIT_CODE (parity test in
# (with a parity test in tests/gateway/test_startup_watchdog.py) because this # tests/gateway/test_startup_watchdog.py) — this module must not import gateway.
# module must not import the gateway package — see module docstring.
SERVICE_RESTART_EXIT_CODE = 75 SERVICE_RESTART_EXIT_CODE = 75
ENV_STARTUP_WATCHDOG = "HERMES_STARTUP_WATCHDOG" ENV_STARTUP_WATCHDOG = "HERMES_STARTUP_WATCHDOG"
@@ -109,62 +55,50 @@ _DUMP_RELATIVE = ("logs", "gateway-startup-watchdog.log")
_FALSEY = frozenset({"0", "false", "no", "off"}) _FALSEY = frozenset({"0", "false", "no", "off"})
# The waiter re-reads its deadline at most this often, so kick_/deadline # The waiter re-reads its deadline at most this often so kicks/extensions
# extensions take effect promptly without busy-waiting. # take effect promptly without busy-waiting.
_POLL_SLICE_S = 5.0 _POLL_SLICE_S = 5.0
# Minimum process CPU-time delta (seconds) within one expired deadline window # Minimum process CPU-time delta within one expired window to count as
# for startup to count as "making progress" and earn a fallback extension. A # progress. A parked futex deadlock accrues microseconds; a schema migration
# parked futex deadlock accrues microseconds; a schema migration accrues # accrues orders of magnitude more per window even on slow disks.
# orders of magnitude more than this per window even on slow disks.
_CPU_PROGRESS_MIN_S = 1.0 _CPU_PROGRESS_MIN_S = 1.0
# Hard cap on CPU-fallback extensions. CPU is process-wide evidence and can # Hard cap on CPU-fallback extensions: CPU is process-wide evidence, so it may
# be produced by threads unrelated to startup, so it may only stretch the # only stretch the runway to (1 + cap) x timeout (3 x 300s = 20min); anything
# runway to (1 + cap) x timeout; anything longer must hold an explicit # longer must hold an explicit phase lease.
# phase lease (report_startup_progress). 3 x 300s default = 20min total.
_MAX_CPU_EXTENSIONS = 3 _MAX_CPU_EXTENSIONS = 3
# Per-call clamp on progress leases (report_startup_progress). A phase that # Per-call clamp on progress leases; a phase that needs longer renews (the
# genuinely needs longer renews its lease — the renewal is itself the # renewal is the liveness evidence). 15min covers the observed worst-case
# liveness evidence. 15 minutes covers the observed worst-case single # single migration step on multi-GB state.db files with margin.
# migration step on multi-GB state.db files with generous margin.
_MAX_LEASE_S = 900.0 _MAX_LEASE_S = 900.0
# How long the fire path waits for the lifecycle-ledger helper thread before # Bounded wait for the lifecycle-ledger helper thread (import lock may be
# exiting anyway (the import lock may be held by the wedged main thread). # held by the wedged main thread).
_LEDGER_JOIN_TIMEOUT_S = 5.0 _LEDGER_JOIN_TIMEOUT_S = 5.0
# Upper bound on the ENTIRE forensic fire path (logging, dump record, # Upper bound on the ENTIRE forensic fire path. A sibling escort thread that
# faulthandler, ledger). A sibling escort thread — which touches no logging, # touches no logging/filesystem/locks hard-exits if forensics wedge (e.g. the
# no filesystem, and no application locks — hard-exits the process if the # wedged main thread holds the logging handler lock, disk full/hung). Must
# forensics wedge (e.g. the wedged main thread holds the logging handler # exceed _LEDGER_JOIN_TIMEOUT_S.
# lock, or the disk is full/hung). Must exceed _LEDGER_JOIN_TIMEOUT_S.
_FIRE_EXIT_BOUND_S = 10.0 _FIRE_EXIT_BOUND_S = 10.0
# Handle lifecycle states. Transitions are guarded by the handle's state # Handle states. Transitions are guarded by the handle's state lock so a
# lock so a disarm and a fire can never both "win" (P2 race, PR #89750 # disarm and a fire can never both "win": armed -> disarmed or armed -> firing.
# review): armed -> disarmed (startup reached a live loop) or
# armed -> firing (deadline expired with no CPU progress) — never both.
_ARMED = "armed" _ARMED = "armed"
_DISARMED = "disarmed" _DISARMED = "disarmed"
_FIRING = "firing" _FIRING = "firing"
# Module-level singleton: the arm sites (hermes_cli.main / hermes_cli.gateway # Module singleton: the arm sites (hermes_cli.main / hermes_cli.gateway /
# / gateway.run.main / cli.py --gateway / scripts/hermes-gateway) and the # gateway.run.main / cli.py --gateway) and the disarm site (GatewayRunner)
# disarm site (GatewayRunner, once the loop is live) have no shared object to # share no object, and only one gateway startup ever runs per process.
# hand a handle through, and only one gateway startup ever runs per process.
_handle_lock = threading.Lock() _handle_lock = threading.Lock()
_handle: Optional["StartupWatchdogHandle"] = None _handle: Optional["StartupWatchdogHandle"] = None
def _process_hermes_home() -> Path: def _process_hermes_home() -> Path:
"""HERMES_HOME for process-level diagnostic files. """HERMES_HOME for diagnostic files — stdlib-only replica of the hermes_constants default."""
Stdlib-only replica of ``hermes_constants``' platform default — this
module must not import application code (see module docstring). Hosted
images always set ``HERMES_HOME`` explicitly.
"""
val = os.environ.get("HERMES_HOME", "").strip() val = os.environ.get("HERMES_HOME", "").strip()
if val: if val:
return Path(val) return Path(val)
@@ -183,8 +117,7 @@ def get_startup_watchdog_dump_path(home: Optional[Path] = None) -> Path:
def startup_watchdog_disabled() -> bool: def startup_watchdog_disabled() -> bool:
"""True when ``HERMES_STARTUP_WATCHDOG`` opts out explicitly.""" """True when ``HERMES_STARTUP_WATCHDOG`` opts out explicitly."""
raw = os.environ.get(ENV_STARTUP_WATCHDOG, "").strip().lower() return os.environ.get(ENV_STARTUP_WATCHDOG, "").strip().lower() in _FALSEY
return raw in _FALSEY
def resolve_startup_watchdog_timeout() -> float: def resolve_startup_watchdog_timeout() -> float:
@@ -207,23 +140,30 @@ def resolve_startup_watchdog_timeout() -> float:
return max(value, _MIN_TIMEOUT_S) return max(value, _MIN_TIMEOUT_S)
def _write_dump_record(record: Dict[str, Any]) -> None: def _append_dump(write, failure_msg: str) -> None:
"""Append a one-line JSON metadata record beside the faulthandler dump.""" """Open the dump file for append and hand it to *write*; failures only log at DEBUG."""
try: try:
path = get_startup_watchdog_dump_path() path = get_startup_watchdog_dump_path()
path.parent.mkdir(parents=True, exist_ok=True) path.parent.mkdir(parents=True, exist_ok=True)
with open(path, "a", encoding="utf-8") as fh: with open(path, "a", encoding="utf-8") as fh:
fh.write(json.dumps(record, default=str) + "\n") write(fh)
except Exception: except Exception:
logger.debug("Failed to write startup watchdog dump record", exc_info=True) logger.debug(failure_msg, exc_info=True)
def _write_dump_record(record: Dict[str, Any]) -> None:
"""Append a one-line JSON metadata record beside the faulthandler dump."""
_append_dump(
lambda fh: fh.write(json.dumps(record, default=str) + "\n"),
"Failed to write startup watchdog dump record",
)
def _mark_lifecycle_exit(exit_code: int) -> None: def _mark_lifecycle_exit(exit_code: int) -> None:
"""Record the watchdog exit in the NS-608 lifecycle sentinel. """Record the watchdog exit in the NS-608 lifecycle sentinel.
Runs on a dedicated helper thread (see ``_fire``): the ``import`` below Runs on a helper thread (see ``_fire``): the import can block on the
can block indefinitely on the interpreter import lock if the wedged main interpreter import lock, and the fire path must reach ``os._exit`` regardless.
thread holds it, and the fire path must reach ``os._exit`` regardless.
""" """
try: try:
from gateway.lifecycle_ledger import mark_exited from gateway.lifecycle_ledger import mark_exited
@@ -246,22 +186,19 @@ class StartupWatchdogHandle:
self._disarmed_event = threading.Event() self._disarmed_event = threading.Event()
self._thread: Optional[threading.Thread] = None self._thread: Optional[threading.Thread] = None
self._extensions = 0 self._extensions = 0
# Phase-owned progress lease (see lease()). monotonic deadline the # Phase-owned progress lease (see lease()): monotonic deadline the
# current startup phase has claimed for legitimately long sync work. # current startup phase has claimed for legitimately long sync work.
self._lease_until = 0.0 self._lease_until = 0.0
self._lease_phase: Optional[str] = None self._lease_phase: Optional[str] = None
self._lease_count = 0 self._lease_count = 0
# Set by _fire() once forensics complete; the exit escort thread # Set by _fire() once forensics complete so the exit escort stands down.
# uses it to stand down when the normal exit path won the race.
self._fire_done = threading.Event() self._fire_done = threading.Event()
def disarm(self) -> None: def disarm(self) -> None:
"""Startup reached a live event loop — stand down. Idempotent. """Startup reached a live event loop — stand down. Idempotent.
Atomic with respect to firing: whichever of disarm/fire takes the Atomic with respect to firing: whichever of disarm/fire takes the state
state lock first wins, so a disarm that lands before the fire lock first wins, so a disarm landing before the fire sequence is never lost.
sequence begins is always honored (never lost to a deadline that
expired concurrently).
""" """
with self._state_lock: with self._state_lock:
if self._state == _ARMED: if self._state == _ARMED:
@@ -271,9 +208,8 @@ class StartupWatchdogHandle:
def kick(self, extra_s: float = 0.0) -> None: def kick(self, extra_s: float = 0.0) -> None:
"""Push the deadline out to ``now + timeout + extra_s``. """Push the deadline out to ``now + timeout + extra_s``.
For call sites that are about to block intentionally with ~zero CPU For call sites about to block intentionally with ~zero CPU (the
activity (the respawn-storm breaker's backoff sleep), which would respawn-storm backoff sleep), otherwise indistinguishable from a parked deadlock.
otherwise be indistinguishable from a parked deadlock.
""" """
try: try:
extra = max(0.0, float(extra_s)) extra = max(0.0, float(extra_s))
@@ -283,18 +219,13 @@ class StartupWatchdogHandle:
self._deadline = time.monotonic() + self.timeout_s + extra self._deadline = time.monotonic() + self.timeout_s + extra
def lease(self, expected_s: float, phase: str = "") -> None: def lease(self, expected_s: float, phase: str = "") -> None:
"""Claim a progress lease: this startup phase is alive and expects """Claim a progress lease: this phase expects up to ``expected_s`` more seconds of work.
up to ``expected_s`` more seconds of legitimate synchronous work.
This is the authoritative "still making progress" signal — unlike The authoritative "still making progress" signal — owned by the startup path,
process-wide CPU time it is owned by the startup path itself, so it so it works for I/O-bound phases and cannot be counterfeited by unrelated
works for I/O-bound phases (corruption repair, backups) that accrue threads. Clamped to ``_MAX_LEASE_S`` per call so one buggy caller cannot
almost no CPU, and it cannot be counterfeited by unrelated threads. silence the watchdog indefinitely; long phases renew. Never raises.
"""
Leases are clamped to ``_MAX_LEASE_S`` per call so a single buggy
caller cannot silence the watchdog indefinitely; genuinely long
phases renew periodically (renewal proves continued liveness).
Never raises."""
try: try:
expected = float(expected_s) expected = float(expected_s)
except (TypeError, ValueError): except (TypeError, ValueError):
@@ -332,14 +263,11 @@ class StartupWatchdogHandle:
def _fire(self) -> None: def _fire(self) -> None:
"""Forensics, then exit — with the exit itself independently bounded. """Forensics, then exit — with the exit itself independently bounded.
Everything in here that produces forensics (logging, the JSON dump Everything producing forensics (logging, dump record, faulthandler, ledger)
record, faulthandler, the lifecycle ledger) can in principle block: can block: the wedged main thread may hold the logging handler lock, the disk
the wedged main thread may hold the logging handler lock, the disk may be hung. None of that may stop the respawn, so the escort thread starts
may be full or hung. None of that may stop the respawn. An escort FIRST. ``os._exit`` is async-signal-safe and lock-free by design.
thread is started FIRST; it touches no logging, no filesystem and """
no application locks — it sleeps, checks whether the normal exit
happened, and otherwise calls the exit seam itself. ``os._exit``
is async-signal-safe and lock-free by design."""
try: try:
escort = threading.Thread( escort = threading.Thread(
target=self._exit_escort, target=self._exit_escort,
@@ -381,22 +309,15 @@ class StartupWatchdogHandle:
faulthandler.dump_traceback(all_threads=True) faulthandler.dump_traceback(all_threads=True)
except Exception: except Exception:
logger.debug("Startup watchdog faulthandler dump failed", exc_info=True) logger.debug("Startup watchdog faulthandler dump failed", exc_info=True)
# Also dump stacks into the log file: on detached/windowless runs # Also dump into the log file: detached/windowless runs (pythonw, some
# (pythonw, some service managers) stderr may be absent, and the # service managers) may have no stderr, and forensics are the point.
# whole point of firing is to leave forensics behind. _append_dump(
try: lambda fh: faulthandler.dump_traceback(file=fh, all_threads=True),
path = get_startup_watchdog_dump_path() "Startup watchdog file-based faulthandler dump failed",
path.parent.mkdir(parents=True, exist_ok=True) )
with open(path, "a", encoding="utf-8") as fh: # Ledger write on a helper thread (it imports application code; the
faulthandler.dump_traceback(file=fh, all_threads=True) # wedged main thread may hold the import lock). Bounded join, then exit
except Exception: # regardless — NS-608 classification is best-effort; the respawn is not.
logger.debug(
"Startup watchdog file-based faulthandler dump failed", exc_info=True
)
# Lifecycle-ledger write on a helper thread: it imports application
# code, and the wedged main thread may hold the import lock. Bounded
# join, then exit regardless (NS-608 classification is best-effort;
# the respawn is not).
try: try:
ledger_thread = threading.Thread( ledger_thread = threading.Thread(
target=_mark_lifecycle_exit, target=_mark_lifecycle_exit,
@@ -414,9 +335,9 @@ class StartupWatchdogHandle:
def _exit_escort(self) -> None: def _exit_escort(self) -> None:
"""Hard-exit if the forensic fire path wedges (bounded-exit seam). """Hard-exit if the forensic fire path wedges (bounded-exit seam).
Deliberately free of log handlers, filesystem access, module loads Deliberately free of log handlers, filesystem access, module loads and any
and any lock shared with application code: its only dependencies lock shared with application code: only a sleep, an Event check and the exit seam.
are a monotonic sleep, an Event check, and the exit seam.""" """
self._sleep(_FIRE_EXIT_BOUND_S) self._sleep(_FIRE_EXIT_BOUND_S)
if self._fire_done.is_set(): if self._fire_done.is_set():
return return
@@ -432,6 +353,14 @@ class StartupWatchdogHandle:
"""Seam for tests; production is a bare ``os._exit``.""" """Seam for tests; production is a bare ``os._exit``."""
os._exit(code) os._exit(code)
def _extend_if_armed(self, deadline: float) -> bool:
"""Set a new deadline under the state lock; False when no longer armed."""
with self._state_lock:
if self._state != _ARMED:
return False
self._deadline = deadline
return True
def _run(self) -> None: def _run(self) -> None:
last_cpu = self._process_cpu_seconds() last_cpu = self._process_cpu_seconds()
while True: while True:
@@ -444,28 +373,16 @@ class StartupWatchdogHandle:
if self._disarmed_event.wait(timeout=min(remaining, _POLL_SLICE_S)): if self._disarmed_event.wait(timeout=min(remaining, _POLL_SLICE_S)):
return return
continue continue
# Deadline expired. Order of authority: # Deadline expired. Order of authority: (1) a phase lease is honored
# # outright; (2) CPU progress extends at most _MAX_CPU_EXTENSIONS
# 1. Phase lease (report_startup_progress): the startup path # times (process-wide CPU proves activity, not startup progress).
# itself declared long legitimate work — honor it outright.
# Works for I/O-bound phases with ~zero CPU (corruption
# repair, backups) and cannot be faked by unrelated threads.
# 2. CPU progress, bounded: process-wide CPU proves the process
# is doing *something*, not that startup is progressing (an
# unrelated daemon thread could burn CPU while the startup
# thread sits parked forever). Extend at most
# _MAX_CPU_EXTENSIONS times, then fire regardless.
now = time.monotonic() now = time.monotonic()
with self._state_lock: with self._state_lock:
lease_until = self._lease_until lease_until = self._lease_until
lease_phase = self._lease_phase lease_phase = self._lease_phase
if lease_until > now: if lease_until > now:
with self._state_lock: if not self._extend_if_armed(max(lease_until, now + min(_POLL_SLICE_S, self.timeout_s))):
if self._state != _ARMED: return
return
self._deadline = max(
lease_until, now + min(_POLL_SLICE_S, self.timeout_s)
)
try: try:
logger.warning( logger.warning(
"Gateway startup exceeded %.0fs but phase %r holds a " "Gateway startup exceeded %.0fs but phase %r holds a "
@@ -476,7 +393,7 @@ class StartupWatchdogHandle:
) )
except Exception: except Exception:
pass pass
# Leased work may be I/O-bound; reset the CPU baseline so a # Leased work may be I/O-bound; reset the CPU baseline so the
# post-lease window is judged on its own activity. # post-lease window is judged on its own activity.
last_cpu = self._process_cpu_seconds() last_cpu = self._process_cpu_seconds()
continue continue
@@ -490,10 +407,8 @@ class StartupWatchdogHandle:
window_delta = cpu - last_cpu window_delta = cpu - last_cpu
last_cpu = cpu last_cpu = cpu
self._extensions += 1 self._extensions += 1
with self._state_lock: if not self._extend_if_armed(time.monotonic() + self.timeout_s):
if self._state != _ARMED: return
return
self._deadline = time.monotonic() + self.timeout_s
try: try:
logger.warning( logger.warning(
"Gateway startup exceeded %.0fs but is consuming CPU " "Gateway startup exceeded %.0fs but is consuming CPU "
@@ -568,10 +483,9 @@ def arm_startup_watchdog(
def disarm_startup_watchdog() -> None: def disarm_startup_watchdog() -> None:
"""Disarm the process-wide startup watchdog, if armed. Never raises. """Disarm the process-wide startup watchdog, if armed. Never raises.
The handle's ``disarm()`` is called while still holding the singleton ``disarm()`` runs while still holding the singleton lock — it is non-blocking,
lock — it is non-blocking, and holding the lock closes the window where and this closes the window where a concurrent re-arm could swap in a new
a concurrent re-arm could swap in a new handle that the disarm then handle that the disarm then misses.
misses.
""" """
global _handle global _handle
try: try:
@@ -584,44 +498,36 @@ def disarm_startup_watchdog() -> None:
logger.debug("Failed to disarm gateway startup watchdog", exc_info=True) logger.debug("Failed to disarm gateway startup watchdog", exc_info=True)
def kick_startup_watchdog(extra_s: float = 0.0) -> None: def _with_armed_handle(method: str, failure_msg: str, *args) -> None:
"""Extend the armed watchdog's deadline. No-op when not armed; never raises. """Call ``handle.<method>(*args)`` on the armed handle; no-op when unarmed, never raises."""
Call before intentionally blocking with ~zero CPU activity (e.g. the
respawn-storm breaker's backoff sleep) so the idle wait is not mistaken
for a parked deadlock.
"""
try: try:
with _handle_lock: with _handle_lock:
handle = _handle handle = _handle
if handle is not None: if handle is not None:
handle.kick(extra_s) getattr(handle, method)(*args)
except Exception: except Exception:
logger.debug("Failed to kick gateway startup watchdog", exc_info=True) logger.debug(failure_msg, exc_info=True)
def kick_startup_watchdog(extra_s: float = 0.0) -> None:
"""Extend the armed watchdog's deadline. No-op when not armed; never raises.
Call before intentionally blocking with ~zero CPU activity (e.g. the
respawn-storm backoff sleep) so the idle wait is not mistaken for a parked deadlock.
"""
_with_armed_handle("kick", "Failed to kick gateway startup watchdog", extra_s)
def report_startup_progress(expected_s: float, phase: str = "") -> None: def report_startup_progress(expected_s: float, phase: str = "") -> None:
"""Declare a phase-owned progress lease on the armed startup watchdog. """Declare a phase-owned progress lease on the armed startup watchdog.
Call from startup phases about to perform legitimately long synchronous Call from startup phases about to do legitimately long synchronous work — most
work — most importantly ``state.db`` schema migrations and corruption importantly ``state.db`` schema migrations and corruption repair/backup inside
repair/backup inside ``SessionDB.__init__`` — passing an honest worst ``SessionDB.__init__`` — with an honest worst case, renewing for multi-step phases.
case for the work about to be done, and renew periodically for Per-call duration is clamped to ``_MAX_LEASE_S``. No-op when not armed; never
multi-step phases. Unlike CPU-time inference, a lease is owned by the raises — safe to call unconditionally from application code.
startup path itself: it works for I/O-bound work that accrues ~zero CPU
and cannot be counterfeited by unrelated busy threads.
Per-call lease duration is clamped to ``_MAX_LEASE_S``; renewals prove
continued liveness. No-op when the watchdog is not armed; never raises —
safe to call unconditionally from application code.
""" """
try: _with_armed_handle("lease", "Failed to report startup progress", expected_s, phase)
with _handle_lock:
handle = _handle
if handle is not None:
handle.lease(expected_s, phase)
except Exception:
logger.debug("Failed to report startup progress", exc_info=True)
def _reset_for_tests() -> None: def _reset_for_tests() -> None: