From 13fb87af92ef0e321e44d1982ec39b6b9129cc2a Mon Sep 17 00:00:00 2001
From: Teknium <127238744+teknium1@users.noreply.github.com>
Date: Wed, 2 Sep 2026 12:15:38 -0700
Subject: [PATCH] refactor(cron): decompose scheduler run_job/_deliver_result
and dedupe delivery lanes
cron/scheduler.py (8535 -> 6353):
- _deliver_result split into per-target helpers: _resolve_target_transport (live/relay/standalone
transport + enablement), _deliver_via_live_adapter (_live_route_metadata for Telegram DM-topic
vs forum routing, _live_send_text with the cancel()-based timeout disambiguation,
_live_send_media, _seed_live_delivery_sessions), _standalone_send/_deliver_standalone. The
three interpreter-shutdown skip branches and the repeated log+append+continue pattern collapse
into one _note_target_error / one shutdown message; _TargetDelivery carries per-target state.
- Thread/channel session seeding unified into _seed_cron_session (was two near-identical
functions).
- run_job decomposed into _run_no_agent_job, _apply_monitor_gate, _load_cron_job_config
(_CronJobConfig), _resolve_job_runtime, _check_model_drift, _open_cron_session_db,
_run_agent_with_watchdog, _finalize_cron_session, plus one _run_doc_header and one _audit
closure for the success/failure paths.
- _run_one_job_body: ownership-lost bookkeeping, delivery composition and outcome classification
extracted; tick: _acquire_tick_lock/_release_tick_lock, _maybe_reap_dead_owners,
_sweep_stale_inflight_for_tick, _process_due_job, _submit_with_guard.
- _build_job_prompt: context_from injection and skill loading extracted; one
_prepend_context_block for the four fenced-data blocks.
- One _start_heartbeat_thread for the script- and fire-claim heartbeat threads.
- Dropped unreachable return in SharedRouteAdapters.get; boolean-return and nested-if shapes
collapsed.
- Comments/docstrings compacted by hand, rationale kept (fd-leak reason for the late SessionDB
close callback, title-persistence rules, no_agent classification gate, inactivity-vs-provider
timeout ordering, stale-claim force-release, interruption token keying).
---
cron/scheduler.py | 7509 +++++++------------
tests/cron/test_scheduler_shutdown_guard.py | 1 -
2 files changed, 2655 insertions(+), 4855 deletions(-)
diff --git a/cron/scheduler.py b/cron/scheduler.py
index ba2d0bde4b..e94ff86858 100644
--- a/cron/scheduler.py
+++ b/cron/scheduler.py
@@ -1,11 +1,5 @@
-"""
-Cron job scheduler - executes due jobs.
-
-Provides tick() which checks for due jobs and runs them. The gateway
-calls this every 60 seconds from a background thread.
-
-Uses a file-based lock (~/.hermes/cron/.tick.lock) so only one tick
-runs at a time if multiple processes overlap.
+"""Cron job scheduler: tick() runs due jobs (gateway calls it every 60s from a background thread).
+A file lock (~/.hermes/cron/.tick.lock) keeps overlapping processes to one tick at a time.
"""
import asyncio
@@ -25,9 +19,10 @@ import sys
import threading
import time
import uuid
+from dataclasses import dataclass
from datetime import datetime, timezone
-# fcntl is Unix-only; on Windows use msvcrt for file locking
+# fcntl is Unix-only; Windows uses msvcrt
try:
import fcntl
except ImportError:
@@ -39,9 +34,8 @@ except ImportError:
from pathlib import Path
from typing import Any, Callable, List, Optional, Protocol
-# Add parent directory to path for imports BEFORE repo-level imports.
-# Without this, standalone invocations (e.g. after `hermes update` reloads
-# the module) fail with ModuleNotFoundError for hermes_time et al.
+# Must precede repo-level imports: standalone invocations (e.g. module reload after
+# `hermes update`) otherwise fail with ModuleNotFoundError for hermes_time et al.
sys.path.insert(0, str(Path(__file__).parent.parent))
from hermes_constants import get_hermes_home
@@ -65,41 +59,22 @@ logger = logging.getLogger(__name__)
def _close_late_session_db_result(future: "concurrent.futures.Future") -> None:
- """Done-callback: close a SessionDB whose constructor finished after run_job's timeout.
-
- When ``run_job``'s SessionDB init times out, the worker thread is abandoned
- (``shutdown(wait=False)``) so the job can proceed without a session store.
- If the constructor later completes inside that abandoned worker, the
- Future's result — an open SessionDB holding .db / WAL / SHM file handles —
- would be orphaned and never closed, leaking descriptors until EMFILE
- (#72782). This callback retrieves and closes that eventual late result.
+ """Done-callback: close a SessionDB whose constructor finished after run_job's init timeout
+ (worker abandoned via ``shutdown(wait=False)``), else its .db/WAL/SHM handles leak to EMFILE.
"""
- try:
+ with contextlib.suppress(Exception):
db = future.result()
if db is not None:
from hermes_state import release_or_close
release_or_close(db)
- except Exception:
- pass
def _set_cron_session_title(session_db, session_id, base_title):
- """Robustly title a finished cron session before it is closed.
+ """Persist a non-blank, unique title for a finished cron session; returns it (None if unset).
- Centralizes the title write so the cron finally block can guarantee a
- non-blank, unique title is persisted before end_session()/close() tear
- the connection down (issues #50535, #50536, #50537):
-
- - #50535: never leaves the session blank. base_title already carries a
- cron-id fallback for nameless jobs; this also guards a failed write.
- - #50537: a duplicate title makes set_session_title raise ValueError (the
- unique-title index). Recover by appending a #N suffix via
- get_next_title_in_lineage() when supported, instead of swallowing the
- error and ending up untitled. If lineage dedup is unavailable, raise.
- - #50536: this runs synchronously in the cron finally block ahead of the
- session close, so no in-flight title write can race the close.
-
- Returns the title actually persisted, or None if nothing could be set.
+ Runs synchronously in the cron finally block BEFORE end_session()/close() so no write races the
+ close. Duplicate (unique-title index ValueError) -> get_next_title_in_lineage(); if unavailable,
+ raise rather than end up untitled.
"""
if not session_db or not session_id:
return None
@@ -110,8 +85,7 @@ def _set_cron_session_title(session_db, session_id, base_title):
session_db.set_session_title(session_id, title)
return title
except ValueError:
- # Title collision against the unique-title index. Fall back to the
- # next title in the lineage (base #2, base #3, ...) when supported.
+ # Unique-title collision: fall back to the next lineage title (base #2, #3, ...).
next_title_fn = getattr(session_db, "get_next_title_in_lineage", None)
if next_title_fn is None:
raise
@@ -123,18 +97,8 @@ def _set_cron_session_title(session_db, session_id, base_title):
def _fallback_chain_phrase() -> str:
- """Wording for the fallback-chain clause of a provider-failure message.
-
- "Fallback chain was exhausted or unavailable." used to fire
- unconditionally on every provider failure, which implies a fallback was
- attempted and failed. Most installs have fallback_providers: [] (no
- chain configured at all), so that wording was actively misleading: it
- sent the operator looking for why a fallback "failed" when none was
- ever attempted. Distinguish the two cases explicitly.
-
- Fails open to the original ambiguous-but-safe wording if config can't be
- read (e.g. mid-shutdown, permissions) -- never let a lookup error crash
- failure-message generation itself.
+ """Fallback-chain clause for a provider-failure message: "exhausted" vs "none configured" (most
+ installs). Fails open to the ambiguous wording if config can't be read — never crash delivery.
"""
try:
cfg = load_config() or {}
@@ -153,19 +117,9 @@ def _fallback_chain_phrase() -> str:
def _failure_streak_nudge(job: dict) -> str:
"""Return a review nudge when a recurring job keeps failing, else "".
- Inspired by Poke (poke.com), which "encourages users to review recurring
- automations that haven't been acted upon": once a recurring job has failed
- several runs in a row, the per-run failure ping stops being information and
- starts being noise — the useful message is "this automation needs your
- attention (fix, pause, or remove it)".
-
- The streak counter (``failure_streak``) is persisted by
- ``cron.jobs.mark_job_run`` and reset on any successful run. Because the
- failure message is delivered BEFORE ``mark_job_run`` records this run, the
- prospective streak for the current failure is stored+1.
-
- Threshold config: ``cron.failure_nudge_threshold`` (default 3, ``0``
- disables the nudge). One-shot jobs never nudge — they don't recur.
+ ``failure_streak`` is persisted by ``cron.jobs.mark_job_run`` (reset on success); the failure
+ message is delivered BEFORE mark_job_run records this run, hence stored+1.
+ Threshold: ``cron.failure_nudge_threshold`` (default 3, 0 disables).
"""
schedule_kind = (job.get("schedule") or {}).get("kind")
if schedule_kind not in {"cron", "interval"}:
@@ -193,12 +147,8 @@ def _failure_streak_nudge(job: dict) -> str:
def _detect_gateway_code_skew() -> tuple[str, str] | None:
- """Boot-vs-disk revision skew for THIS process, or None.
-
- Thin wrapper over ``gateway.code_skew.detect_code_skew`` so the failure
- summarizer stays a pure function under test (monkeypatch this seam) and
- a broken import can never take the delivery path down with it.
- """
+ """Boot-vs-disk revision skew for THIS process, or None. Test seam over
+ ``gateway.code_skew.detect_code_skew``; a broken import must never take delivery down."""
try:
from gateway.code_skew import detect_code_skew
@@ -210,24 +160,11 @@ def _detect_gateway_code_skew() -> tuple[str, str] | None:
class CronTickYielded(RuntimeError):
"""A stale-code ticker yielded this tick to a fresh gateway.
- Raised by ``tick()`` BEFORE the tick lock is acquired when the process is
- provably running stale code (boot fingerprint ≠ disk), it does NOT own the
- gateway runtime lock, and that lock is held by another (fresh) process.
- Fresh code picks the job up within one tick interval, so the stale process
- must stay out of the dispatch race entirely — including lock contention,
- which would otherwise starve the fresh ticker on a busy minute.
-
- Raised instead of returned so the provider loops
- (``cron/scheduler_provider.py``) record it via ``record_ticker_error`` and
- mark the heartbeat ``success=False``: a yielded tick is NOT a healthy tick
- (``hermes cron status`` must not show green while jobs only fire from the
- other process). Liveness stays visible — the loop keeps beating and keeps
- yielding; if the fresh gateway dies, its lock releases and the stale
- ticker's next tick proceeds normally (self-healing, no restart needed).
-
- Skew detection returning ``None`` (non-git install, no boot fingerprint —
- e.g. a one-shot CLI tick, or any probe failure) never yields: yield only
- on certainty, fail open otherwise.
+ Raised by ``tick()`` BEFORE the tick lock is acquired when boot fingerprint ≠ disk, this process
+ does NOT own the gateway runtime lock, and a fresh process holds it; the stale process must stay
+ out of the dispatch race entirely (lock contention would starve the fresh ticker). Skew ``None``
+ (non-git, no fingerprint, probe failure) never yields: fail open. Raised, not returned, so
+ provider loops record it via ``record_ticker_error`` and ``hermes cron status`` isn't green.
"""
def __init__(self, boot_rev: str, disk_rev: str) -> None:
@@ -239,25 +176,16 @@ class CronTickYielded(RuntimeError):
)
-# Log the yield at most once per episode: a stale ticker that keeps yielding
-# for hours must not spam the error log every interval. Reset when the
-# condition clears (proceeds without yielding) or the skew changes.
+# Log the yield at most once per episode (reset when the skew changes) to avoid per-interval spam.
_YIELD_LOG_INTERVAL_SECONDS = 3600.0
_last_yield_log: dict[str, object] = {}
def _should_yield_tick_to_fresh_gateway() -> tuple[str, str] | None:
- """Decide whether this tick must yield to a fresher gateway process.
+ """Return ``(boot_rev, disk_rev)`` when this tick must yield to a fresher gateway, else None.
- Returns the ``(boot_rev, disk_rev)`` skew labels when ALL of: this process
- has a boot fingerprint that differs from the checkout on disk (code
- skew), it does not own the gateway runtime lock, and some other process
- currently holds that lock — i.e. a fresh gateway is alive and will
- dispatch due jobs itself. Returns ``None`` otherwise.
-
- Every probe failure returns ``None``: the gateway-status import, the lock
- probe, and skew detection are each individually fail-open. Yielding is a
- certainty claim, never a guess.
+ Yields only when ALL hold: code skew, we don't own the runtime lock, another process holds it.
+ Every probe failure returns None — yielding is a certainty claim, never a guess.
"""
skew = _detect_gateway_code_skew()
if skew is None:
@@ -293,12 +221,7 @@ def _log_tick_yield_once(reason: str) -> None:
def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
- """Return a compact one-line failure message for chat delivery.
-
- Full details stay in the cron output directory and the logs. Chat should
- show the operator what broke without dumping provider JSON, retry noise, or
- stack traces into the delivery channel.
- """
+ """Compact one-line failure message for chat delivery (full details stay in cron output)."""
job_name = job.get("name") or job.get("id") or "cron job"
text = (error or "unknown error").strip()
lower = text.lower()
@@ -321,43 +244,20 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
f"unintended spend. {remediation}"
)
- # A no_agent job IS its script — run_job short-circuits it before any model
- # is reached ("no LLM involvement", see the no_agent branch in run_job). So
- # provider timeouts, rate limits, auth errors and fallback chains are not
- # merely unlikely for these jobs, they are structurally impossible. Classify
- # on the job's MODE before pattern-matching its prose.
- #
- # Without this gate the branches below classify by substring, so a script's
- # own wording decides which subsystem gets blamed. _run_job_script reports a
- # timeout as "Script timed out after {n}s: {path}" — that contains "timed
- # out", so it matched the provider branch and the operator was told
- # "provider timeout. Fallback chain was exhausted or unavailable." for a job
- # that never opened a socket. "429" or "authentication" appearing anywhere
- # in a script's output misfires the same way.
- #
- # A delivery line that names the wrong subsystem is worse than no line at
- # all: it does not merely fail to inform, it sends the reader to the wrong
- # place.
- #
- # Falling through leaves the generic cleaner below to report what actually
- # happened, naming the script. No new message text is needed.
+ # no_agent jobs never reach a model, so provider errors are structurally impossible for them.
+ # Gate on job MODE before substring matching, or a script's own wording ("timed out", "429")
+ # would blame the wrong subsystem; the generic cleaner below reports what actually happened.
provider_reachable = not job.get("no_agent")
- # Script execution happens outside the LLM/provider path (also for
- # agent-backed jobs that run a context script). Check the script runner's
- # explicit error contract ("Script timed out after {n}s: {path}") before
- # generic timeout matching so a script timeout never claims a provider
- # fallback was attempted (#82460 @jbagdonas, #78503 @daxro).
+ # Script runner contract ("Script timed out after {n}s: {path}") — also for agent jobs with a
+ # context script. Must precede generic timeout matching so it never claims a provider fallback.
if lower.startswith("script timed out"):
return (
f"⚠️ Cron '{job_name}' failed: script timed out. "
"No model was invoked. Full details saved in cron output."
)
- # Provider/API failures are the common noisy path. Keep these short.
- # Match 429 as a whole token (#83188 @cation98): bare substring matching
- # let identifiers containing those digits (job ids, ports, hashes) trip
- # a false "provider rate limit" alert.
+ # Whole-token 429: substrings in job ids/ports/hashes tripped false rate-limit alerts.
if provider_reachable and (
re.search(r"\b429\b", text) or "rate limit" in lower or "usage limit" in lower
):
@@ -372,22 +272,8 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
"Full details saved in cron output."
)
- # The scheduler's own inactivity watchdog (see the TimeoutError raised
- # above at "Cron job '{job_name}' idle for {secs}s (limit {limit}s) —
- # last activity: {desc}") produces a message that contains the substring
- # "timed out"/"timeout" nowhere, but DOES contain "idle for ... (limit
- # ...)" — however older/other call sites can still phrase an inactivity
- # abort using "timed out" wording, so match on the "idle for Ns (limit"
- # shape specifically (case-insensitive) BEFORE the generic provider-
- # timeout branch below. Without this, an inactivity timeout — the job's
- # OWN tool call/turn going quiet, no provider or fallback chain ever
- # involved — gets rewritten into a misleading "provider timeout /
- # fallback chain exhausted" message, sending the operator to debug the
- # wrong system entirely (field-reported: a stuck `terminal` tool call
- # tripped the 600s inactivity limit and was reported as a
- # provider/fallback failure). Mirrors the same reordering fix
- # upstream issue #59549 applied for script timeouts vs provider timeouts
- # — check the more specific, deterministic signature first.
+ # Scheduler inactivity watchdog shape ("idle for {n}s (limit {m}s)"). Must precede the generic
+ # provider-timeout branch: the job's own tool going quiet involves no provider/fallback chain.
if re.search(r"idle for \d+s\s*\(limit \d+s\)", lower):
return (
f"⚠️ Cron '{job_name}' failed: the job itself stalled — no tool/API "
@@ -405,9 +291,7 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
"Full details saved in cron output."
)
- # Match authentication/authorization wording at a word boundary and the
- # 401/403 status codes as whole tokens, so "oauth", "4015" and similar do
- # not trip a misleading auth message.
+ # Whole-token 401/403 and auth wording so "oauth", "4015" etc. don't trip a false auth message.
if provider_reachable and (
re.search(r"authenticat|authoriz", lower) or re.search(r"\b(401|403)\b", text)
):
@@ -416,37 +300,18 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
"Full details saved in cron output."
)
- # Strip common exception wrappers and collapse provider payloads. Bound
- # the input first so a multi-KB provider blob cannot slow the
- # substitutions.
- cleaned = re.sub(
- r"^(RuntimeError|Exception|ValueError|HTTPStatusError):\s*",
- "", text[:2000],
- )
+ # Strip exception wrappers; bound input first so a multi-KB blob can't slow the regexes.
+ cleaned = re.sub(r"^(RuntimeError|Exception|ValueError|HTTPStatusError):\s*", "", text[:2000])
cleaned = re.sub(r"\s+", " ", cleaned).strip()
if len(cleaned) > 180:
cleaned = cleaned[:177].rstrip() + "..."
message = f"⚠️ Cron '{job_name}' failed: {cleaned}"
- # Import-class failures (#95294 part 3): a long-lived gateway whose
- # checkout was updated underneath it (interrupted `hermes update`, manual
- # git pull) serves MIXED modules — old entries frozen in sys.modules,
- # new files loaded by lazy imports — and every agent cron job then dies
- # with `cannot import name X` / ModuleNotFoundError. The error itself
- # reads like a code bug, so operators debug the wrong thing (2 days on
- # the reporting incident, 15 missed jobs). This process knows its own
- # boot fingerprint: when boot SHA differs from disk HEAD, APPEND the
- # cause and the one-command fix — never replacing the raw error text,
- # which carries the failing symbol name.
- #
- # Fail-safe by construction: skew detection returns None on non-git
- # installs and in processes without a boot fingerprint (no false
- # accusations — message delivered unchanged), the probe seam swallows
- # every exception, and no_agent script jobs are excluded via the same
- # mode-gate as the provider branches (a fresh subprocess resolves
- # imports consistently against disk; its ImportError is the script's
- # own problem, and blaming gateway skew would send the reader to the
- # wrong place).
+ # Import-class failures in a gateway whose checkout changed underneath it (mixed sys.modules)
+ # read like code bugs. When boot SHA ≠ disk HEAD, APPEND cause + fix — never replace the raw
+ # error, which carries the failing symbol. Fail-safe: skew is None on non-git/no-fingerprint
+ # (message unchanged); no_agent jobs excluded via the same mode gate (a fresh subprocess
+ # resolves imports against disk, so its ImportError is the script's own problem).
if provider_reachable and re.search(
r"cannot import name|modulenotfounderror|importerror", lower
):
@@ -468,20 +333,11 @@ def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
def _upsert_incident_for_failure(
job: dict, error: str, *, output_file: Optional[Any] = None
) -> tuple[bool, Optional[str]]:
- """Record a durable failure incident for this run.
+ """Record a durable failure incident (grouped by job + error signature).
- The incident store groups "same job + same error signature" across runs so
- an operator-acked failure stops re-pinging every run. Returns
- ``(acked, incident_id)``: ``acked`` is True when the incident for this
- exact signature is already ``closed`` (acked) — the per-run failure ping
- should be suppressed. ``incident_id`` lets the caller mark the incident
- ``alerted`` after the ping actually goes out. The streak nudge and
- ``_summarize_cron_failure_for_delivery`` text stay intact for un-acked
- failures.
-
- Best-effort: an incident-store error must never break the cron run or the
- delivery path — failures are logged at debug and the caller delivers as if
- no incident existed.
+ Returns ``(acked, incident_id)``; acked=True when the signature's incident is already
+ ``closed`` -> suppress the per-run ping.
+ Best-effort: store errors log at debug and the caller delivers as if no incident existed.
"""
try:
from cron.incidents import get_incident, upsert_incident
@@ -504,12 +360,7 @@ def _upsert_incident_for_failure(
def _mark_incident_alerted(incident_id: Optional[str]) -> None:
- """Record that a failure ping for this incident reached delivery.
-
- Best-effort like the upsert: bookkeeping never breaks the cron run.
- ``set_incident_state`` is a no-op for closed incidents, so this can
- never resurrect an acked signature.
- """
+ """Best-effort: mark incident ``alerted`` (no-op for closed; never resurrects an acked one)."""
if not incident_id:
return
try:
@@ -521,35 +372,18 @@ def _mark_incident_alerted(incident_id: Optional[str]) -> None:
class CronPromptInjectionBlocked(Exception):
- """Raised by _build_job_prompt when the fully-assembled prompt trips the
- injection scanner. Caught in run_job so the operator sees a clean
- "job blocked" delivery instead of the scheduler crashing.
-
- Assembled-prompt scanning (including loaded skill content) plugs the
- gap from #3968: create-time scanning only covers the user-supplied
- prompt field; skill content loaded at runtime was never scanned, so a
- malicious skill could carry an injection payload that reached the
- non-interactive (auto-approve) cron agent.
- """
+ """Raised by _build_job_prompt when the assembled prompt (incl. runtime-loaded skill content,
+ unseen by create-time scanning) trips the injection scanner; run_job turns it into a clean
+ "job blocked" delivery."""
def _resolve_cron_disabled_toolsets(cfg: dict) -> list[str]:
"""Toolsets a cron-spawned agent must never receive.
- Two toolsets are always disabled in cron context regardless of config:
- - ``messaging`` — interactive, needs a live gateway session
- - ``clarify`` — interactive, blocks waiting for user input
-
- ``cronjob`` is policy-denied by default (loop prevention, not a security
- boundary) and config-gated: setting ``cron.allow_agent_scheduling: true``
- in config.yaml drops it from the base denylist so cron-spawned agents may
- manage the user's cron table. The gate only removes the built-in policy
- denial — it never overrides the user denylist below.
-
- User-level ``agent.disabled_toolsets`` from config.yaml is layered on top
- so per-job ``enabled_toolsets`` cannot bypass policy that applies to
- ordinary agent runs (#25752 — LLM-supplied enabled_toolsets was widening
- past config.yaml's denylist).
+ ``messaging``/``clarify`` always (interactive). ``cronjob`` by default (loop prevention, not a
+ security boundary); ``cron.allow_agent_scheduling: true`` lifts only that, never the user
+ denylist. ``agent.disabled_toolsets`` is layered on top so per-job ``enabled_toolsets`` cannot
+ widen past config.yaml's denylist.
"""
cron_cfg = (cfg or {}).get("cron") or {}
if cron_cfg.get("allow_agent_scheduling"):
@@ -570,24 +404,14 @@ def _resolve_cron_disabled_toolsets(cfg: dict) -> list[str]:
def _merge_mcp_into_per_job_toolsets(per_job: list[str], cfg: dict) -> list[str]:
"""Layer enabled MCP servers onto a per-job ``enabled_toolsets`` allowlist.
- A per-job list scopes the *native* toolsets, but on its own it silently
- drops every MCP server: ``discover_mcp_tools()`` registers the tools into
- the global registry, yet ``get_tool_definitions(enabled_toolsets=...)``
- only keeps toolsets named in the list. The agent then rejects every
- ``mcp_*`` call with "Unknown tool". This restores parity with
- ``_get_platform_tools`` MCP semantics:
-
- * ``no_mcp`` sentinel present -> no MCP servers (sentinel stripped)
- * one or more MCP server names already listed -> treat as an allowlist,
- add nothing further (the user named exactly the servers they want)
- * otherwise -> union in every globally-enabled MCP server
+ Without this a per-job list silently drops every MCP server ("Unknown tool" on mcp_* calls).
+ Mirrors ``_get_platform_tools``: ``no_mcp`` sentinel -> none (sentinel stripped); any MCP server
+ already listed -> treat as allowlist, add nothing; otherwise union in all globally-enabled.
"""
result = [t for t in per_job if t != "no_mcp"]
if "no_mcp" in per_job:
return result
- # lazy import: avoid heavy hermes_cli import at cron module load (matches
- # _resolve_cron_enabled_toolsets' fallback) and share one MCP-membership
- # computation with the gateway/CLI platform resolver.
+ # lazy: avoid heavy hermes_cli import at module load; shares MCP-membership with gateway/CLI
from hermes_cli.tools_config import enabled_mcp_server_names
enabled_mcp = enabled_mcp_server_names(cfg)
if set(result) & enabled_mcp:
@@ -601,21 +425,10 @@ def _merge_mcp_into_per_job_toolsets(per_job: list[str], cfg: dict) -> list[str]
def _resolve_cron_enabled_toolsets(job: dict, cfg: dict) -> list[str] | None:
"""Resolve the toolset list for a cron job.
- Precedence:
- 1. Per-job ``enabled_toolsets`` (set via ``cronjob`` tool on create/update).
- Keeps the agent's job-scoped toolset override intact — #6130. Enabled
- MCP servers are layered on per ``_merge_mcp_into_per_job_toolsets`` so a
- native-toolset allowlist does not silently strip MCP tools.
- 2. Per-platform ``hermes tools`` config for the ``cron`` platform.
- Mirrors gateway behavior (``_get_platform_tools(cfg, platform_key)``)
- so users can gate cron toolsets globally without recreating every job.
- 3. ``None`` on any lookup failure — AIAgent loads the full default set
- (legacy behavior before this change, preserved as the safety net).
-
- _DEFAULT_OFF_TOOLSETS ({moa, homeassistant, rl}) are removed by
- ``_get_platform_tools`` for unconfigured platforms, so fresh installs
- get cron WITHOUT ``moa`` by default (issue reported by Norbert —
- surprise $4.63 run).
+ Precedence: per-job ``enabled_toolsets`` (+ ``_merge_mcp_into_per_job_toolsets``) > ``cron``
+ platform config (``_get_platform_tools``) > ``None`` on any failure (full default set).
+ ``_get_platform_tools`` strips _DEFAULT_OFF_TOOLSETS ({moa, homeassistant, rl}) for unconfigured
+ platforms, so fresh installs run cron without ``moa``.
"""
per_job = job.get("enabled_toolsets")
if per_job:
@@ -634,20 +447,9 @@ def _resolve_cron_enabled_toolsets(job: dict, cfg: dict) -> list[str] | None:
def _resolve_job_reasoning_config(job: dict, cfg: dict, model: str) -> dict | None:
"""Resolve the effective reasoning config for a cron run.
- Precedence: per-job ``reasoning_effort`` pin (validated at the store
- choke point, ``cron/jobs.py::_normalize_reasoning_effort``) wins outright
- over config resolution — both the global ``agent.reasoning_effort`` and
- per-model ``agent.reasoning_overrides``. The pin is model-independent by
- design: it also governs an auth-fallback model swap, and capability
- clamping for the model that actually runs stays owned by the provider
- transports at send time (exactly like config-set effort).
-
- A value that no longer parses (hand-edited jobs.json) logs a warning and
- falls back to config resolution — a bad pin must degrade the run's
- thinking level, never kill the tick.
-
- Absent/None pin returns ``resolve_reasoning_config(cfg, model)``
- byte-identical, preserving pre-feature behavior.
+ Per-job ``reasoning_effort`` pin beats global and per-model config; it is model-independent by
+ design (also governs an auth-fallback swap) — clamping stays with provider transports at send
+ time. An unparseable pin warns and falls back, never kills the tick. No pin -> config.
"""
from hermes_constants import parse_reasoning_effort, resolve_reasoning_config
@@ -655,11 +457,7 @@ def _resolve_job_reasoning_config(job: dict, cfg: dict, model: str) -> dict | No
if pinned is not None:
parsed = parse_reasoning_effort(pinned)
if parsed is not None:
- logger.info(
- "Job '%s': using per-job reasoning_effort '%s'",
- job.get("id", "?"),
- pinned,
- )
+ logger.info("Job '%s': using per-job reasoning_effort '%s'", job.get("id", "?"), pinned)
return parsed
logger.warning(
"Job '%s': invalid stored reasoning_effort %r — ignoring the pin "
@@ -673,8 +471,7 @@ def _resolve_job_reasoning_config(job: dict, cfg: dict, model: str) -> dict | No
return resolve_reasoning_config(cfg if isinstance(cfg, dict) else {}, str(model))
-# Valid delivery platforms — used to validate user-supplied platform names
-# in cron delivery targets, preventing env var enumeration via crafted names.
+# Validates user-supplied delivery platform names, preventing env-var enumeration via crafted names.
_KNOWN_DELIVERY_PLATFORMS = frozenset({
"telegram", "discord", "slack", "whatsapp", "signal",
"matrix", "mattermost", "homeassistant", "dingtalk", "feishu",
@@ -682,8 +479,7 @@ _KNOWN_DELIVERY_PLATFORMS = frozenset({
"qqbot", "yuanbao",
})
-# Platforms that support a configured cron/notification home target, mapped to
-# the environment variable used by gateway setup/runtime config.
+# Platforms supporting a cron/notification home target -> env var used by gateway config.
_HOME_TARGET_ENV_VARS = {
"matrix": "MATRIX_HOME_ROOM",
"telegram": "TELEGRAM_HOME_CHANNEL",
@@ -703,13 +499,9 @@ _HOME_TARGET_ENV_VARS = {
"whatsapp_cloud": "WHATSAPP_CLOUD_HOME_CHANNEL",
}
-# Legacy env var names kept for back-compat. Each entry is the current
-# primary env var → the previous name. _get_home_target_chat_id falls
-# back to the legacy name if the primary is unset, so users who set the
-# old name before the rename keep working until they migrate.
-_LEGACY_HOME_TARGET_ENV_VARS = {
- "QQBOT_HOME_CHANNEL": "QQ_HOME_CHANNEL",
-}
+# Back-compat: primary env var -> previous name; _get_home_target_chat_id falls back to the legacy
+# name when the primary is unset.
+_LEGACY_HOME_TARGET_ENV_VARS = {"QQBOT_HOME_CHANNEL": "QQ_HOME_CHANNEL"}
from cron.jobs import (
_ensure_cron_dir,
@@ -727,113 +519,64 @@ from cron.jobs import (
)
from cron.executions import create_execution, finish_execution, mark_execution_running
-# Sentinel: when a cron agent has nothing new to report, it can start its
-# response with this marker to suppress delivery. Output is still saved
-# locally for audit.
+# Response marker that suppresses delivery (output is still saved locally for audit).
SILENT_MARKER = "[SILENT]"
-# Canonical silence tokens recognized in cron output. Cron's contract is
-# intentionally looser than the gateway's exact-whole-response rule: the cron
-# system prompt *instructs* the agent to emit "[SILENT]", and real agents often
-# bracket it with a short note or trailing newline. We therefore suppress when
-# a marker is the entire response OR appears as its own first/last line — but
-# NOT when a token merely appears mid-sentence in a genuine report (e.g.
-# "I considered staying [SILENT] but here is the summary…" must deliver).
-# The actual matcher is shared with the webhook lane —
-# gateway.response_filters.is_autonomous_silence_response — so the two
-# autonomous lanes cannot drift apart.
-
def _is_cron_silence_response(text: str) -> bool:
"""Return True when a cron final response should suppress delivery.
- Recognizes the bracketed ``[SILENT]`` sentinel (whole-response, first line,
- or last line) plus the bracketless ``SILENT`` / ``NO_REPLY`` / ``NO REPLY``
- variants the model emits when it drops the brackets (#51438, #46917).
- Whitespace-trimmed and case-insensitive. A token buried mid-sentence is
- treated as real content and delivered.
-
- Delegates to the shared autonomous-lane matcher in
- :mod:`gateway.response_filters` (also used by the webhook adapter).
+ Looser than the gateway's exact-whole-response rule: ``[SILENT]`` (or SILENT / NO_REPLY /
+ NO REPLY) counts as the whole response OR its own first/last line — NOT mid-sentence. Shares the
+ webhook-lane matcher in :mod:`gateway.response_filters` so the two cannot drift.
"""
from gateway.response_filters import is_autonomous_silence_response
return is_autonomous_silence_response(text)
-# ---------------------------------------------------------------------------
-# Persistent thread pool for parallel cron jobs.
-# The tick function submits jobs here and returns immediately so the ticker
-# thread is never blocked by long-running jobs (e.g. the fixer running 15+ min).
-# ---------------------------------------------------------------------------
+# Persistent pool for parallel cron jobs: tick() submits and returns; long jobs never block it.
_parallel_pool: Optional[concurrent.futures.ThreadPoolExecutor] = None
_parallel_pool_max_workers: Optional[int] = None
_running_job_ids: set = set()
_running_fire_owners: dict[str, dict[object, tuple[Optional[str], Path]]] = {}
_running_lock = threading.Lock()
-# Wall-clock (time.time()) instant each in-flight job id was claimed by
-# ``_submit_with_guard``, plus the future that owns its release (a pending
-# sentinel until ``pool.submit`` returns). Together these bound the
-# in-flight set: an id whose claim is older than its allowance AND has no
-# live future can only be a leak — the release path never ran — so the
-# stale-sweep force-releases it instead of letting every later tick
-# short-circuit on "already running" until the whole gateway process
-# restarts (incident: jarvis board-pm-triage-* jobs, 2026-08-02; recurring
-# router/watchdog no_agent jobs, 2026-08-14 t_20e23f84).
+# Per in-flight id: time.time() claim instant + the future owning its release (``_FUTURE_PENDING``
+# until pool.submit returns). Past-allowance with no live future = leak; the sweep force-releases.
_running_since: dict = {}
_running_futures: dict = {}
-# Sentinel installed in ``_running_futures`` at claim time, before
-# ``pool.submit`` has returned a real future. This closes the race the
-# stale sweep previously had: a sweep landing between the claim critical
-# section and the future-record section saw ``missing`` and could (in
-# principle) release a claim that was about to get its future. With the
-# sentinel there is never a window where a claim has neither an age nor a
-# future marker — it is ``_FUTURE_PENDING`` until the real future lands.
+# Installed in ``_running_futures`` at claim time so a sweep landing before ``pool.submit`` returns
+# never sees ``missing`` and releases a claim about to get its future.
_FUTURE_PENDING = object()
-# Countable signal for unified-health: how many stale claims this process has
-# force-released, and the most recent ones. Exposed via
-# ``get_inflight_guard_stats()`` and mirrored to a JSONL under the cron dir so
-# an out-of-process probe can catch a wedge in-cycle.
+# Forced-release count/history for ``get_inflight_guard_stats()``; mirrored to JSONL for probes.
_forced_release_count: int = 0
_forced_releases: list = []
_FORCED_RELEASE_HISTORY = 20
-# Floor for the stale allowance, in minutes. Effective allowance per job is
-# max(2 * interval, this) so a slow-but-healthy hourly job is never clipped.
+# Stale-allowance floor (minutes); per-job allowance is max(2 * interval, this).
_INFLIGHT_MIN_ALLOWANCE_MINUTES = 30.0
-# Execution tokens (``object()`` identity keys from ``_running_fire_owners``)
-# of runs the shutdown path force-interrupted — see
-# ``mark_running_jobs_interrupted`` below. ``run_one_job``'s own completion
-# path checks its OWN token before writing ``last_status`` so a cron agent
-# thread that keeps running in-process after its tool was killed out from
-# under it — and produces a plausible-looking final response from truncated
-# output — can never overwrite the interrupted status with a false "ok"
-# (#60432). Token keying keeps an interruption scoped to that exact
-# execution: a later run of the same job ID (recurring jobs reuse the ID
-# every fire) must not inherit the stale flag. Legacy dispatch paths without
-# a registered fire owner fall back to storing the bare job ID.
+# Execution tokens (``_running_fire_owners`` identity keys) force-interrupted at shutdown; see
+# ``mark_running_jobs_interrupted``. ``run_one_job`` checks its OWN token before writing
+# ``last_status`` so a still-running agent thread can't overwrite "interrupted" with a false "ok".
+# Token keying scopes the flag to one execution (recurring jobs reuse IDs); legacy paths without a
+# fire owner fall back to the bare job ID.
_interrupted_job_ids: set = set()
class _CancelEventLike(Protocol):
- """Structural type for cancellation sources (``threading.Event`` and
- ``_CombinedCancelEvent`` both satisfy it)."""
+ """Structural type for cancellation sources (``threading.Event``, ``_CombinedCancelEvent``)."""
def is_set(self) -> bool: ...
def set(self) -> None: ...
class _CombinedCancelEvent:
- """Duck-typed ``threading.Event`` that ORs several cancellation sources.
-
- ``run_one_job`` already derives a ``lost_ownership`` event from the
- fire-claim heartbeat; transports (dashboard webhook drain, API server
- shutdown) contribute their own per-task event. The worker only ever
- calls ``is_set()`` / ``set()``, so a tiny wrapper beats a pump thread.
+ """Duck-typed ``threading.Event`` ORing several cancellation sources (fire-claim heartbeat
+ ``lost_ownership`` + per-transport events). Workers only call is_set()/set(), so no pump thread.
"""
def __init__(self, *events: Optional["_CancelEventLike"]) -> None:
@@ -848,49 +591,24 @@ class _CombinedCancelEvent:
def get_running_job_ids() -> "frozenset[str]":
- """Thread-safe snapshot of cron job IDs currently executing.
-
- A job ID is a member from the moment ``_submit_with_guard`` dispatches
- it onto the parallel/sequential pool until ``_process_job`` returns —
- i.e. for the job's *entire* run, tool calls included, not just the
- ticker's dispatch instant.
-
- The gateway shutdown path (``gateway/run.py::GatewayRunner.
- _drain_active_agents``) reads this to treat in-flight cron work as
- active the same way it already treats in-flight chat sessions via
- ``_running_agents`` — cron jobs run through their own thread pool here,
- entirely outside that dict, so without this the drain is structurally
- blind to them (#60432).
- """
+ """Thread-safe snapshot of executing job IDs (dispatch until ``_process_job`` returns). Read by
+ the gateway shutdown drain, otherwise blind to cron work (runs outside ``_running_agents``)."""
with _running_lock:
return frozenset(_running_job_ids | _running_fire_owners.keys())
def try_register_running_job(job_id: str) -> bool:
- """Atomically add ``job_id`` to the in-flight running set.
+ """Atomically add ``job_id`` to the in-flight set; False (caller must skip) if already mid-run.
- Returns False (without registering) when the job is already mid-run —
- the caller must skip the fire. This is the single dedupe owner shared by
- the ticker's ``_submit_with_guard`` and manual runs
- (``tools/cronjob_tools``): the fire claim alone cannot prevent a
- double-fire because its TTL (300s) is routinely outlived by real jobs,
- after which a manual ``cronjob(action='run')`` would claim successfully
- and run the same job concurrently (idea from #53395 by @izumi0uu).
-
- Registration also makes the run visible to ``get_running_job_ids`` (the
- gateway shutdown drain, #60432) and ``mark_running_jobs_interrupted``.
- Callers MUST pair a successful registration with
- ``release_running_job`` in a ``finally`` block.
+ Single dedupe owner for ticker + manual runs (the fire claim's 300s TTL is outlived by real
+ jobs). Callers MUST pair success with ``release_running_job`` in a ``finally``.
"""
with _running_lock:
if job_id in _running_job_ids:
return False
_running_job_ids.add(job_id)
- # Claim timestamp + pending-future sentinel are recorded in the SAME
- # critical section as the add, so there is never a window where an
- # id is in-flight without an age the stale sweep can bound it by
- # (t_3778a491). The sentinel is replaced by the real owning future
- # once ``pool.submit`` returns.
+ # Same critical section as the add: no window where an in-flight id lacks an age the sweep
+ # can bound. Sentinel is replaced by the real future once ``pool.submit`` returns.
_running_since[job_id] = time.time()
_running_futures[job_id] = _FUTURE_PENDING
return True
@@ -905,15 +623,8 @@ def release_running_job(job_id: str) -> None:
def _inflight_min_allowance_minutes() -> float:
- """Floor for the stale in-flight allowance, in minutes.
-
- Effective allowance per job is ``max(2 * interval, this)``, so a
- slow-but-healthy long-interval job is never clipped by the sweep.
- Reads ``cron.inflight_max_minutes`` from config.yaml; the
- ``HERMES_CRON_INFLIGHT_MAX_MINUTES`` env var is kept as an internal
- escape hatch only.
- """
- try:
+ """Stale allowance floor (min): ``cron.inflight_max_minutes``, else env escape hatch/default."""
+ with contextlib.suppress(Exception):
_ucfg = load_config() or {}
_cfg_val = (
_ucfg.get("cron", {}) if isinstance(_ucfg, dict) else {}
@@ -922,8 +633,6 @@ def _inflight_min_allowance_minutes() -> float:
val = float(_cfg_val)
if val > 0:
return val
- except Exception:
- pass
raw = os.getenv("HERMES_CRON_INFLIGHT_MAX_MINUTES", "").strip()
if raw:
try:
@@ -939,28 +648,16 @@ def _inflight_min_allowance_minutes() -> float:
return _INFLIGHT_MIN_ALLOWANCE_MINUTES
-# Cache for cron expression interval computation (expression → minutes).
-# A cron expression's cadence never changes, so computing it once per expr
-# avoids repeated croniter evaluation on every 60s tick.
+# expr -> minutes; cadence never changes, so avoid re-evaluating croniter every tick.
_cron_interval_cache: dict = {}
def _cron_interval_minutes(expr: str) -> Optional[float]:
- """Approximate the natural interval of a cron expression, in minutes.
-
- The persisted job store keeps ``schedule`` as an already-parsed dict
- (``{"kind": "cron", "expr": "0 9 * * 1"}``), so the stale allowance for
- a cron job cannot be derived from a schedule *string* — it must come
- from the expression itself. We measure the gap between the next two
- fire times with croniter; that is the job's cadence, and the sweep's
- allowance becomes ``max(2 * cadence, floor)`` exactly like interval
- jobs. Falls back to ``None`` (→ floor allowance) if croniter is
- missing or the expression cannot be evaluated.
- """
+ """Cron expression cadence (gap between next two fires) in minutes; None -> floor allowance."""
if expr in _cron_interval_cache:
return _cron_interval_cache[expr]
result = None
- try:
+ with contextlib.suppress(Exception):
from cron.jobs import _ensure_croniter
if _ensure_croniter():
@@ -973,26 +670,14 @@ def _cron_interval_minutes(expr: str) -> Optional[float]:
second = it.get_next(datetime)
gap = (second - first).total_seconds() / 60.0
result = gap if gap > 0 else None
- except Exception:
- pass
_cron_interval_cache[expr] = result
return result
def _job_interval_minutes(job: dict) -> Optional[float]:
- """Best-effort interval length for a job, in minutes (None if unknown).
-
- Reads the PERSISTED schedule shape first: the job store keeps
- ``schedule`` as an already-parsed dict (``{"kind": "interval",
- "minutes": N}`` or ``{"kind": "cron", "expr": "..."}``), NOT the string
- form that ``parse_schedule`` consumes. The string path is kept only as
- a defensive fallback for programmatic callers that still build string
- schedules (and for tests that exercise that shape).
-
- ``kind == "once"`` (one-shot) has no recurring interval — returns None,
- so the sweep uses the documented floor allowance.
- """
- try:
+ """Best-effort job interval in minutes (None if unknown / one-shot -> floor). ``schedule`` is
+ persisted as a parsed dict; the string path is only a fallback for programmatic callers."""
+ with contextlib.suppress(Exception):
schedule = job.get("schedule")
if isinstance(schedule, str) and schedule.strip():
from cron.jobs import parse_schedule
@@ -1005,18 +690,11 @@ def _job_interval_minutes(job: dict) -> Optional[float]:
return float(minutes) if minutes else None
if kind == "cron":
return _cron_interval_minutes(str(schedule.get("expr") or ""))
- except Exception:
- pass
return None
def get_inflight_guard_stats() -> dict:
- """Probe-visible snapshot of the in-flight guard.
-
- ``forced_releases`` is a monotonic counter of stale claims this process
- has force-released; any non-zero value means a cron job wedged and was
- recovered without a gateway restart.
- """
+ """Probe-visible snapshot; non-zero ``forced_releases`` means a job wedged and was recovered."""
now = time.time()
with _running_lock:
return {
@@ -1052,22 +730,11 @@ def _record_forced_release(job_id: str, name: str, age_seconds: float, allowance
def sweep_stale_inflight(due_jobs: Optional[list] = None) -> list:
- """Force-release in-flight claims that can no longer be making progress.
+ """Force-release in-flight claims that can no longer be making progress; returns released ids.
- A claim is stale when it is older than ``max(2 * interval, floor)`` AND
- either has no live future at all (the wedge class: the claim was taken
- but the release path was never installed — e.g. a hang in the submit
- path before ``pool.submit`` returned) or has a future that already
- finished without discarding the id.
-
- Every release logs a WARNING with the countable ``event=forced_release``
- signal, bumps a probe-visible counter (``get_inflight_guard_stats()``),
- mirrors a JSONL row under the cron dir, and writes ``last_error`` on the
- job so the wedge surfaces on the job row instead of being invisible
- until a downstream liveness key goes dead hours later. A forced release
- never consumes a finite-repeat job's budget (see below).
-
- Returns the list of released job ids.
+ Stale = older than ``max(2 * interval, floor)`` AND (no live future — submit path hung before
+ ``pool.submit`` returned — or finished without discarding the id). Each release logs WARNING
+ ``event=forced_release``, bumps the probe counter, mirrors JSONL, and writes ``last_error``.
"""
global _forced_release_count
@@ -1076,25 +743,14 @@ def sweep_stale_inflight(due_jobs: Optional[list] = None) -> list:
now = time.time()
stale: list = []
- # Latest durable execution per RELEASABLE-LOOKING in-flight job id, loaded
- # in one indexed query. Used for the persisted-state reconciliation below
- # (t_8b5480b3): an in-memory claim whose OWN run's execution row is
- # terminal cannot represent a live run — the durable ledger proves that
- # run already ended — so the claim is stale by construction, regardless of
- # its in-memory age. A leaked claim is then recoverable without
- # force-run/resume. Two-phase so the healthy steady state pays no DB
- # work: a claim with a live future is never released, so the query only
- # covers claims whose future is missing/pending/done (the snapshot is
- # taken under _running_lock; iterating a set concurrently mutated by
- # try_register/release_running_job can raise RuntimeError). A claim that
- # becomes releasable between the snapshot and the sweep loop simply waits
- # for the next tick's query.
+ # Latest durable execution per releasable-looking claim, one indexed query. A claim whose OWN
+ # run's row is terminal is stale regardless of age. Two-phase so the healthy path pays no DB
+ # work: only claims with a missing/pending/done future are queried. Snapshot under
+ # _running_lock — iterating the set while try_register/release mutate it raises RuntimeError.
from cron.executions import _TERMINAL_STATES as _terminal_states
with _running_lock:
- _claim_futures = {
- job_id: _running_futures.get(job_id) for job_id in _running_job_ids
- }
+ _claim_futures = {job_id: _running_futures.get(job_id) for job_id in _running_job_ids}
_ledger_candidates = [
job_id
for job_id, fut in _claim_futures.items()
@@ -1111,14 +767,9 @@ def sweep_stale_inflight(due_jobs: Optional[list] = None) -> list:
def _row_belongs_to_claim(row: dict, claim_started: float) -> bool:
"""True when the ledger row was claimed at/after this in-memory claim.
- The latest terminal row proves THIS claim's run ended only if it was
- created by this claim's dispatch (create_execution runs moments AFTER
- try_register_running_job). A terminal row older than the in-memory
- claim is the PREVIOUS run's outcome — for a recurring job that is the
- common case in the window between try_register and create_execution,
- and releasing on it would double-dispatch a healthy fresh claim.
- Unparseable timestamps fail closed (row treated as previous-run; the
- age-based path below still bounds the claim).
+ A terminal row older than the claim is the PREVIOUS run's (common for recurring jobs in the
+ try_register->create_execution window); releasing on it would double-dispatch. Unparseable
+ timestamps fail closed (treated as previous-run; the age path still bounds the claim).
"""
claimed_at = row.get("claimed_at")
if not claimed_at:
@@ -1130,16 +781,14 @@ def sweep_stale_inflight(due_jobs: Optional[list] = None) -> list:
except (ValueError, TypeError, OSError):
return False
- # Precompute job intervals OUTSIDE _running_lock so croniter evaluation
- # does not block try_register/release_running_job for other jobs.
+ # Compute intervals OUTSIDE _running_lock so croniter doesn't block try_register/release.
_intervals = {jid: _job_interval_minutes(j) for jid, j in by_id.items()}
with _running_lock:
for job_id in list(_running_job_ids):
started = _running_since.get(job_id)
if started is None:
- # Claim predates this guard (or was injected directly) — adopt
- # it now so it becomes sweepable one allowance from here.
+ # Claim predates this guard — adopt it; sweepable one allowance from now.
_running_since[job_id] = now
continue
age = now - started
@@ -1149,29 +798,13 @@ def sweep_stale_inflight(due_jobs: Optional[list] = None) -> list:
allowance = max(allowance, 2.0 * interval_minutes * 60.0)
fut = _running_futures.get(job_id)
if fut is _FUTURE_PENDING:
- # The claim is past its allowance and the owning future still
- # has not been installed — the submit path itself (SessionDB
- # init, agent import, config load) hung before ``pool.submit``
- # returned. That is exactly the wedge class; release it.
+ # Submit path hung before ``pool.submit`` returned — the wedge class; release it.
pass
elif fut is not None and not fut.done():
continue # genuinely still executing
- # Persisted-state reconciliation: if the durable executions ledger
- # shows THIS claim's run reached a terminal state, the claim is
- # provably stale even if it is still inside its in-memory age
- # allowance (or was adopted fresh this tick). Release it now so
- # the job re-dispatches on the next tick without force-run/resume
- # (t_8b5480b3 — the 2026-08-14 recurring-router wedge where the
- # in-memory age bound alone could not see a run the ledger had
- # already finished). The row must belong to THIS claim
- # (claimed_at >= claim registration): for a recurring job the
- # latest terminal row is usually the PREVIOUS run's outcome —
- # a fresh claim in the try_register→create_execution window, or a
- # finished run whose worker finally hasn't released yet, would
- # otherwise be force-released and double-dispatched. Reaching
- # here implies the future is missing/pending/done (the live-future
- # case continued above), so every claim in this branch was a
- # ledger-query candidate.
+ # Ledger reconciliation: a terminal row belonging to THIS claim proves it stale even
+ # inside its age allowance. Row must be this claim's, else a recurring job's previous
+ # run would double-dispatch a fresh claim.
latest = _latest.get(job_id)
if (
latest is not None
@@ -1210,23 +843,12 @@ def sweep_stale_inflight(due_jobs: Optional[list] = None) -> list:
future_state,
)
_record_forced_release(job_id, name, age, allowance)
- # A ledger-terminal release is authoritative: the durable executions
- # ledger ALREADY records how the last run ended (completed/failed/
- # unknown), so we must NOT call mark_job_run here — doing so would
- # clobber an honest completed/ok status with a synthetic failure, or
- # double-write an already-recorded failure. We only release the claim
- # so the job re-dispatches on its next due tick; the ledger is the
- # record of record for the outcome. The age-based release below keeps
- # the original wedge-surfacing mark_job_run behaviour (an age-release
- # may have no ledger row at all, so surfacing last_error is the only
- # way the wedge becomes visible).
+ # Ledger already records how the run ended: mark_job_run here would clobber an honest
+ # ok status with a synthetic failure or double-write a failure.
if _reason == "ledger-terminal":
continue
- # Finite-repeat guard: a forced release is NOT a real run, so it must
- # not consume a finite one-shot's repeat budget or let mark_job_run
- # auto-delete the row (completed >= times). The claim is released and
- # the row is left untouched, so the job re-fires normally on its next
- # due tick (self-heal) instead of being deleted.
+ # Age release may lack a ledger row, so last_error is how it surfaces. But a forced release
+ # is NOT a real run: never consume a finite repeat budget or let mark_job_run auto-delete.
repeat = job.get("repeat") or {}
if isinstance(repeat, dict) and repeat.get("times") is not None:
logger.warning(
@@ -1256,35 +878,12 @@ def mark_running_jobs_interrupted(
*,
only_owners: Optional[set] = None,
) -> list:
- """Best-effort: mark every currently in-flight cron job interrupted.
+ """Best-effort: mark every in-flight cron job interrupted; returns the job IDs marked.
- Called by the gateway shutdown path immediately after it force-kills
- tool subprocesses (``process_registry.kill_all()``). A job whose tool
- subprocess was just killed out from under it must never be allowed to
- report success — even though its agent thread is still alive in this
- same process and may go on to produce a plausible-looking final
- response from the now-truncated tool output.
-
- Records the job IDs in ``_interrupted_job_ids`` BEFORE writing
- ``last_status`` so ``run_one_job``'s own eventual completion for the
- same job (racing in its own thread) sees the flag and skips its normal
- write instead of clobbering this one — see the check near the end of
- ``run_one_job``. This does not attempt to correlate the killed
- subprocess PID to a specific job ID (the process registry tracks PIDs,
- not cron job IDs); any job still dispatched at the moment of a forced
- kill is treated as interrupted, matching the coarser precedent already
- set by ``GatewayRunner._interrupt_running_agents``, which interrupts
- every entry in ``_running_agents`` on a drain timeout without
- per-agent correlation either.
-
- ``only_owners``: optional set of ``(job_id, fire_owner)`` pairs. When
- given (dashboard webhook drain), ONLY those exact executions are
- marked — unrelated runs sharing the process (e.g. the desktop ticker's
- own jobs) are left untouched. Interruption flags are recorded per
- execution token, so a later run of the same job ID never consumes a
- stale flag that targeted its dead predecessor.
-
- Returns the list of job IDs marked, for the caller to log.
+ Called by gateway shutdown right after ``process_registry.kill_all()``: a job whose tool was
+ killed must never report success even if its agent thread produces a plausible response.
+ ``only_owners`` (``(job_id, fire_owner)`` pairs) restricts marking to those executions. Tokens
+ go into ``_interrupted_job_ids`` BEFORE ``last_status`` is written so ``run_one_job`` sees them.
"""
with _running_lock:
active_fires = [
@@ -1293,10 +892,7 @@ def mark_running_jobs_interrupted(
for token, (owner, profile_home) in executions.items()
]
if only_owners is not None:
- active_fires = [
- fire for fire in active_fires
- if (fire[1], fire[2]) in only_owners
- ]
+ active_fires = [fire for fire in active_fires if (fire[1], fire[2]) in only_owners]
registered_ids = {job_id for _t, job_id, _o, _p in active_fires}
if only_owners is None:
active_fires.extend(
@@ -1315,11 +911,8 @@ def mark_running_jobs_interrupted(
"leaving persisted state untouched",
job_id,
)
- # Still report the interruption to the caller: the gateway
- # shutdown path uses the returned IDs to send the
- # interrupted-cron notice while adapters are still connected
- # (#82232). The in-memory interrupt flag WAS recorded above —
- # only the persisted last_status write is skipped here.
+ # Still report it: shutdown uses the returned IDs for the interrupted-cron notice. The
+ # in-memory flag WAS recorded above; only the persisted last_status write is skipped.
marked.append(job_id)
continue
try:
@@ -1337,21 +930,9 @@ def mark_running_jobs_interrupted(
def _is_interrupted(job_id: str, token: Optional[object] = None) -> bool:
- """Non-destructive peek at whether the shutdown path has marked THIS
- execution interrupted (see ``mark_running_jobs_interrupted``).
-
- Called by ``run_one_job`` BEFORE it decides what to deliver — a job
- whose tool subprocess was killed mid-flight may still produce a
- plausible-looking ``final_response`` from the truncated output, and
- that must not go out to the user as if it were a normal result.
- Unlike ``_consume_interrupted_flag`` below, this does not clear the
- flag: the later, authoritative check (right before ``last_status`` is
- written) still needs to see it. ``token`` scopes the check to one
- exact execution: owner-registered runs are matched by token, so a
- fresh run reusing the same job ID is not poisoned by a flag that
- targeted its dead predecessor. The bare job ID is only ever stored
- for legacy dispatch paths with no registered fire owner.
- """
+ """Non-destructive peek: has shutdown marked THIS execution interrupted? Used before deciding
+ what to deliver; does not clear the flag (the authoritative pre-``last_status`` check needs it).
+ ``token`` scopes to one execution so a fresh run reusing the job ID isn't poisoned."""
with _running_lock:
if token is not None and token in _interrupted_job_ids:
return True
@@ -1359,13 +940,8 @@ def _is_interrupted(job_id: str, token: Optional[object] = None) -> bool:
def _consume_interrupted_flag(job_id: str, token: Optional[object] = None) -> bool:
- """Return True and clear the flag if the shutdown path already marked
- THIS execution interrupted (see ``mark_running_jobs_interrupted``).
-
- Called by ``run_one_job`` right before it would otherwise write its own
- ``last_status``. Consuming (discarding) rather than just checking keeps
- the flag from leaking across a later, unrelated run of the same job ID
- (recurring jobs reuse their ID every fire)."""
+ """Return True and clear the flag if shutdown marked THIS execution interrupted. Called right
+ before ``last_status`` is written; consuming stops the flag leaking into a later run."""
with _running_lock:
hit = False
if token is not None and token in _interrupted_job_ids:
@@ -1385,14 +961,8 @@ def _inactivity_watchdog_loop(
stop: threading.Event,
future_done: Callable[[], bool],
) -> bool:
- """Poll job idle time until the limit, stop, or the watched future completes.
-
- Driven by ``threading.Event.wait`` (a kernel timeout), not asyncio, so a
- blocked event-loop / ``run_job`` thread cannot disable this watchdog the
- way ``asyncio.sleep`` / ``wait_for`` would (family A of #94285 — the
- 4118s-idle-on-a-600s-limit cron hang). Returns True when *limit_s* of
- inactivity was observed.
- """
+ """Poll idle time until limit (-> True), stop, or the future completes (-> False). Uses
+ ``threading.Event.wait``, not asyncio, so a blocked event loop cannot disable the watchdog."""
while not stop.wait(poll_s):
if future_done():
return False
@@ -1405,14 +975,11 @@ def _inactivity_watchdog_loop(
return False
-
def _cron_inactivity_seconds() -> float:
"""Parse HERMES_CRON_TIMEOUT (seconds). 0 = unlimited; bad input = 600.
- Shared by run_job's inactivity monitor (which maps 0 to "no limit") and
- the cwd-lock bound below (which keeps the wait bounded regardless) so
- the two sites cannot drift apart — the lock bound must stay at or above
- the inactivity limit or waiters would fail while a healthy holder runs.
+ Shared by run_job's inactivity monitor and the cwd-lock bound so they can't drift: the lock
+ bound must stay >= the inactivity limit or waiters fail while a healthy holder runs.
"""
raw = os.getenv("HERMES_CRON_TIMEOUT", "").strip()
if not raw:
@@ -1447,10 +1014,8 @@ def _shutdown_parallel_pool() -> None:
_parallel_pool_max_workers = None
-
atexit.register(_shutdown_parallel_pool)
-# Per-fire usage audit log for cron token spend instrumentation.
-# Resolves through _get_hermes_home() so profile-scoped paths work correctly.
+# Per-fire usage audit log; resolves via _get_hermes_home() so profile-scoped paths work.
def _usage_audit_path() -> Path:
return _get_hermes_home() / "cron" / "usage_audit.jsonl"
@@ -1458,16 +1023,12 @@ def _usage_audit_path() -> Path:
def _utcnow_iso_ms() -> str:
"""RFC3339 UTC timestamp with millisecond precision and 'Z' suffix."""
now = datetime.now(timezone.utc)
- # %f gives microseconds; trim to milliseconds.
return now.strftime("%Y-%m-%dT%H:%M:%S.") + f"{now.microsecond // 1000:03d}Z"
def _write_usage_audit(record: dict) -> None:
- """Append a single JSONL line to ~/.hermes/cron/usage_audit.jsonl.
-
- NEVER raises — a logger bug must not break cron jobs. Wraps the entire
- write (path resolve, mkdir, json.dumps, file append) in a single try.
- """
+ """Append one JSONL line to cron/usage_audit.jsonl. NEVER raises — a logger bug must not
+ break cron jobs (the whole write is inside one try)."""
try:
path = _usage_audit_path()
_ensure_cron_dir(path.parent)
@@ -1479,46 +1040,27 @@ def _write_usage_audit(record: dict) -> None:
def _interpreter_shutting_down(exc: Optional[BaseException] = None) -> bool:
- """True when the Python interpreter is finalizing.
+ """True when the interpreter is finalizing (tick fired during gateway teardown).
- A cron tick can fire while the gateway is tearing down — SIGTERM from
- ``hermes update`` / ``hermes gateway stop`` / systemd restart, or an
- OOM-kill. Once finalization starts, ``concurrent.futures`` refuses new
- work with ``RuntimeError: cannot schedule new futures after interpreter
- shutdown`` and asyncio's default executor is gone, so *any* attempt to
- schedule delivery (live-adapter, ``asyncio.run``, or a fresh pool) is
- doomed and only pollutes ``errors.log`` with a traceback. Callers use
- this to skip gracefully with a warning instead of crashing (#58720,
- #55924).
-
- ``exc`` lets a caller also treat an already-raised scheduling error as a
- shutdown signal: the ``concurrent.futures`` module-global flag can be set
- a hair before ``sys.is_finalizing()`` flips, so matching the error text is
- a safe fallback for that race.
-
- Thin wrapper — the predicate itself lives in
- ``tools.interpreter_shutdown.interpreter_shutting_down`` (shared with the
- conversation loop and the concurrent tool executor) so the shutdown-race
- bug class is fixed in one place. Kept as a module symbol because tests
- and callers throughout this file reference it by this name.
+ Once finalization starts, concurrent.futures/asyncio refuse new work, so any delivery attempt
+ (live adapter, asyncio.run, fresh pool) only pollutes errors.log — callers skip with a warning.
+ ``exc`` lets an already-raised scheduling error count as a shutdown signal. Thin wrapper over
+ ``tools.interpreter_shutdown`` (shared with the gateway).
"""
from tools.interpreter_shutdown import interpreter_shutting_down
return interpreter_shutting_down(exc)
-# Backward-compatible module override used by tests and emergency monkeypatches.
+# Module override hook for tests / emergency monkeypatches.
_hermes_home: Path | None = None
def _get_hermes_home() -> Path:
- """Resolve Hermes home dynamically while preserving test monkeypatch hooks.
+ """Resolve Hermes home at call time (honouring the test override).
- Cron is per-profile by design (#4707): the in-process ticker runs inside a
- profile-scoped gateway, so resolving the active HERMES_HOME at call time
- means a profile's jobs are stored AND executed under that profile's home
- (its .env, config.yaml, scripts, skills). Do not freeze this at import or
- anchor it at the shared default root — either re-breaks profile isolation.
+ Cron is per-profile: jobs must be stored AND executed under the active profile's home. Do not
+ freeze this at import or anchor it at the shared default root — either breaks profile isolation.
"""
return _hermes_home or get_hermes_home()
@@ -1530,16 +1072,12 @@ def _get_lock_paths() -> tuple[Path, Path]:
return lock_dir, lock_dir / ".tick.lock"
-# Errnos that mean "another ticker (or manual tick) holds the tick lock",
-# as opposed to a real failure opening/locking the file. Everything else —
-# most importantly EMFILE/ENFILE (fd exhaustion, #87644) and EACCES on
-# open() — must be surfaced, never swallowed as lock contention.
def _is_lock_contention_errno(err: OSError) -> bool:
- """Return True when *err* from the lock syscall means the lock is held.
+ """True when *err* from the lock syscall means another ticker holds the lock.
- - POSIX: ``flock(LOCK_EX|LOCK_NB)`` reports EWOULDBLOCK/EAGAIN when
- another process holds the lock (EACCES on some NFS implementations).
- - Windows: ``msvcrt.locking(LK_NBLCK)`` reports EACCES/EDEADLK.
+ POSIX flock: EWOULDBLOCK/EAGAIN (EACCES on some NFS); Windows msvcrt.locking: EACCES/EDEADLK.
+ Everything else — notably EMFILE/ENFILE (fd exhaustion) and EACCES on open() — must be
+ surfaced, never swallowed as contention.
"""
if err.errno is None:
return False
@@ -1551,81 +1089,44 @@ def _is_lock_contention_errno(err: OSError) -> bool:
def _is_fd_exhaustion_text(text: str) -> bool:
- """Text-level half of :func:`_is_fd_exhaustion` (shared with the CLI hint)."""
+ """Text half of _is_fd_exhaustion (shared with the CLI hint)."""
lowered = text.lower()
return "too many open files" in lowered or "emfile" in lowered
def _is_fd_exhaustion(exc: BaseException) -> bool:
- """Return True when *exc* indicates file-descriptor exhaustion.
-
- Recognizes EMFILE/ENFILE by errno, and the "Too many open files" wording
- for wrapped exceptions (``load_jobs`` wraps the raw OSError in a
- RuntimeError with that message, #87644).
- """
+ """True when *exc* indicates fd exhaustion: EMFILE/ENFILE errno, or the "Too many open files"
+ wording for wrapped exceptions (load_jobs wraps the OSError in a RuntimeError)."""
if isinstance(exc, OSError) and exc.errno in (errno.EMFILE, errno.ENFILE):
return True
return _is_fd_exhaustion_text(str(exc))
def _reclaim_fds_best_effort() -> None:
- """Best-effort attempt to free leaked file descriptors.
-
- The cron FD-leak family (#60859, #79742, #80792) leaks descriptors from
- abandoned workers/sessions. Two safe, idempotent levers:
-
- 1. ``gc.collect()`` — closes file-like objects held only in reference
- cycles (the classic unclosed-file leak shape), which CPython would
- otherwise never finalize.
- 2. ``apply_nofile_soft_limit()`` — raise RLIMIT_NOFILE's soft limit
- toward the configured target when the hard limit allows, giving the
- process headroom to keep serving even before every leak is freed.
-
- Never raises: a reclamation attempt must not make the ticker worse.
- """
- try:
+ """Best-effort fd reclamation: gc.collect() closes file objects stuck in reference cycles;
+ apply_nofile_soft_limit() raises the RLIMIT_NOFILE soft limit for headroom. Never raises."""
+ with contextlib.suppress(Exception):
import gc
gc.collect()
- except Exception:
- pass
- try:
+ with contextlib.suppress(Exception):
from hermes_cli.resource_limits import apply_nofile_soft_limit
apply_nofile_soft_limit(None)
- except Exception:
- pass
def _resolve_cron_surface_mode(pconfig, logical_platform_name: str) -> str:
- """Resolve the continuable-cron delivery surface for a platform config.
+ """Return ``"in_channel"`` or ``"thread"`` (default) for a platform config.
- Returns ``"in_channel"`` or ``"thread"`` (default). Two config shapes:
-
- - Native adapter: the flat key ``platforms.
.extra.cron_continuable_surface``
- (shipped shape, unchanged).
- - Relay-fronted: ``platforms.relay.extra..cron_continuable_surface``
- — the same per-logical-platform sub-block the relay's documented Slack
- knobs use (``reply_in_thread``, ``dm_top_level_threads_as_sessions``;
- see RelayAdapter._relay_slack_extra). The sub-block wins over a flat
- key when both exist, matching _relay_slack_extra precedence, and is
- scoped to its logical platform so a ``slack:`` block cannot leak onto
- another fronted platform.
-
- Precedence nuance vs _relay_slack_extra: that helper is all-or-nothing
- (a sub-dict REPLACES the flat extra entirely), while this one falls back
- to the flat key when the sub-block exists but omits the knob. The
- difference is deliberate — the flat key is the legacy staging shape and
- must keep working — but note a flat ``cron_continuable_surface`` then
- applies to EVERY platform this relay fronts; only the per-platform D6
- capability gate contains it. Scope the knob under the sub-block on
- multi-platform relays.
-
- Field gap (2026-08-18): the scheduler read only the flat key, so on the
- relay lane — where pconfig is platforms.relay — operators had NO working
- location for the knob and briefs always threaded.
+ Native: flat ``platforms..extra.cron_continuable_surface``. Relay-fronted:
+ ``platforms.relay.extra..cron_continuable_surface`` (same sub-block as the relay's
+ Slack knobs); the sub-block wins over the flat key and is scoped to its logical platform.
+ Unlike _relay_slack_extra (all-or-nothing), this falls back to the flat key when the sub-block
+ omits the knob — deliberate, the flat key must keep working — so a flat value applies to EVERY
+ platform the relay fronts (only the D6 capability gate contains it). Scope it on multi-platform
+ relays.
"""
- try:
+ with contextlib.suppress(Exception):
extra = getattr(pconfig, "extra", None) or {}
sub = extra.get(str(logical_platform_name or "").lower())
if isinstance(sub, dict) and sub.get("cron_continuable_surface") is not None:
@@ -1634,61 +1135,27 @@ def _resolve_cron_surface_mode(pconfig, logical_platform_name: str) -> str:
raw = extra.get("cron_continuable_surface")
if raw is not None and str(raw).strip().lower() == "in_channel":
return "in_channel"
- except Exception:
- pass
return "thread"
def _resolve_origin(job: dict) -> Optional[dict]:
- """Extract origin info from a job, preserving any extra routing metadata.
-
- Treats non-dict origins (free-form provenance strings, ints, lists from
- migration scripts or hand-edited jobs.json) as missing instead of
- crashing with ``AttributeError`` on ``origin.get(...)``. Without this
- guard, a job tagged with e.g. ``"combined-digest-replaces-x-and-y"``
- crashed every fire attempt with
- ``'str' object has no attribute 'get'`` — ``mark_job_run`` recorded the
- failure, but the next tick re-loaded the same poisoned origin and
- crashed identically until the field was patched manually (#18722).
- """
+ """Extract origin info from a job. Non-dict origins (provenance strings, hand-edited
+ jobs.json) are treated as missing — otherwise every fire crashed on ``origin.get``."""
origin = job.get("origin")
- if not isinstance(origin, dict):
- return None
- platform = origin.get("platform")
- chat_id = origin.get("chat_id")
- if platform and chat_id:
+ if isinstance(origin, dict) and origin.get("platform") and origin.get("chat_id"):
return origin
return None
def _cron_mirror_delivery_enabled(job: dict, cfg: Optional[dict] = None) -> bool:
- """Whether a cron delivery should also be mirrored into the target chat's
- gateway session transcript.
+ """Whether a cron delivery is also mirrored into the target chat's session transcript.
- Default OFF — preserves the historical isolation guarantee (cron deliveries
- live only in the cron job's own session, never the target chat's history)
- byte-for-byte for everyone who does not opt in.
-
- CARVE-OUT: the ``in_channel`` continuable surface seeds its target
- session independently of this knob (see ``_deliver_result`` /
- ``_seed_cron_channel_session``). in_channel is itself opt-in
- (``cron_continuable_surface: in_channel`` + the adapter capability bit),
- and the seed IS the feature — a continuable flat brief without its seed
- is a brief the next reply can't see. This knob keeps governing the
- SEPARATE default/thread-surface transcript mirror only.
-
- Precedence (first decisive value wins):
- 1. Per-job ``attach_to_session`` (bool) — set via the ``cronjob`` tool,
- lets one briefing job opt in without flipping global behaviour.
- 2. Global ``cron.mirror_delivery`` (bool) in config.yaml.
- 3. False.
-
- When enabled, the cron's final output is appended to the target session as
- an assistant turn via the existing ``gateway.mirror.mirror_to_session`` —
- the same primitive ``send_message`` uses — so the next user reply in that
- chat sees the brief in context (no "what is Task #2?" amnesia). This is
- alternation- and cache-safe: the append lands at a turn boundary between
- user turns, never mid-loop, and never mutates the cached system prompt.
+ Default OFF (cron deliveries live only in the job's own session unless opted in). Precedence:
+ per-job ``attach_to_session`` (bool) → global ``cron.mirror_delivery`` → False.
+ CARVE-OUT: the ``in_channel`` surface seeds its target session independently of this knob
+ (the seed IS that feature; in_channel is itself opt-in) — this knob governs only the
+ default/thread-surface mirror. The mirror uses ``mirror_to_session`` at a turn boundary, so it
+ is alternation- and cache-safe.
"""
per_job = job.get("attach_to_session")
if isinstance(per_job, bool):
@@ -1705,20 +1172,8 @@ def _target_matches_origin(origin: dict, platform_name: str, chat_id: str,
thread_id: Optional[str]) -> bool:
"""True when a delivery target is the job's own origin conversation.
- Mirroring is scoped to the origin session by design (see
- ``_maybe_mirror_cron_delivery``). A job created from a live gateway chat
- stamps that chat as ``origin`` (``cronjob_tools._origin_from_env``), and
- that session is guaranteed to exist — it is the very conversation the user
- was in when they scheduled the job. Fan-out targets (``deliver=all``,
- explicit ``platform:chat_id`` to some *other* chat, or a home-channel
- fallback for an origin-less API/script job) are deliberately NOT mirrored:
- they are broadcasts, not a continuation of a conversation, and may point at
- a chat the user never opened an agent session in.
-
- This makes the historical "cold-start" worry a non-case: when the mirror
- semantically applies (target == origin) the session always exists; when no
- session exists, the target was never the origin conversation, so we simply
- do not mirror.
+ Mirroring is scoped to the origin session (guaranteed to exist — the job was created there).
+ Fan-out targets (``all``, explicit other chats) are broadcasts and deliberately NOT mirrored.
"""
if not origin:
return False
@@ -1726,23 +1181,14 @@ def _target_matches_origin(origin: dict, platform_name: str, chat_id: str,
return False
if str(origin.get("chat_id", "")) != str(chat_id):
return False
- # thread_id must match when the origin pins one (topic-scoped chats); a
- # target that lost the thread_id is not the same conversation lane.
+ # A pinned origin thread_id must match — a target without it is a different lane.
origin_thread = origin.get("thread_id")
- if origin_thread is not None and str(origin_thread) != str(thread_id or ""):
- return False
- return True
+ return origin_thread is None or str(origin_thread) == str(thread_id or "")
-# Resolution-provenance ranking for the dedup OR-merge in
-# _resolve_delivery_targets: higher rank = stronger mirror claim. Broadcast
-# expansions rank 0 so "origin,all"/"all,origin" hitting the same chat keeps
-# the origin(-fallback) tag regardless of token order.
-_MIRROR_PROVENANCE_RANK = {
- "origin": 3,
- "origin_fallback": 2,
- "explicit": 1,
-}
+# Provenance rank for the dedup OR-merge in _resolve_delivery_targets (higher = stronger mirror
+# claim). Broadcasts rank 0 so "origin,all"/"all,origin" keep the origin tag regardless of order.
+_MIRROR_PROVENANCE_RANK = {"origin": 3, "origin_fallback": 2, "explicit": 1}
def _target_mirror_eligible(
@@ -1754,31 +1200,12 @@ def _target_mirror_eligible(
) -> bool:
"""Whether a resolved delivery target may receive the transcript mirror.
- The June origin-scoping refactor gated mirroring on target == origin,
- which correctly excluded broadcasts but also silenced two legitimate
- conversation shapes — both hit by script-provisioned ("managed") crons,
- which never capture an origin (``_origin_from_env`` only fires for jobs
- created from a live gateway chat):
-
- - ``origin_fallback``: ``deliver=origin`` with no captured origin resolves
- to the home channel — the user's primary conversation standing in for
- the origin, not a broadcast. Eligible under the same flags as a true
- origin target. (Field report 2026-08-17: brief delivered to the Slack
- DM, mirror silently skipped, reply hit a context-less session.)
- - ``explicit``: a ``platform:chat_id`` target is eligible ONLY when the
- job itself opts in via ``attach_to_session: true`` — the job author
- declaring this target a conversation (managed per-user DM briefings).
- The global ``cron.mirror_delivery`` flag never activates explicit
- targets: it must not start writing transcript entries into arbitrary
- explicitly-addressed chats (shared channels, other users' DMs).
-
- Broadcast expansions (``all``, bare-platform home targets) carry no
- provenance tag and are never eligible — unchanged invariant.
-
- ``origin_match`` lets the caller pass a precomputed
- ``_target_matches_origin`` result (``_deliver_result`` already computes it
- for the same target); when ``None`` it is computed here so tests and
- future callers stay self-contained.
+ Origin targets: always. ``origin_fallback`` (deliver=origin with no captured origin → home
+ channel, standing in for the user's primary conversation): same flags as a true origin.
+ ``explicit`` ``platform:chat_id``: ONLY with per-job ``attach_to_session: true`` — the global
+ flag must never write transcripts into arbitrary explicitly-addressed chats (shared channels,
+ other users' DMs). Untagged broadcasts (``all``, bare-platform home) are never eligible.
+ ``origin_match`` may be precomputed by the caller; computed here when ``None``.
"""
if origin_match is None:
origin = _resolve_origin(job) or {}
@@ -1790,12 +1217,8 @@ def _target_mirror_eligible(
return True
resolved_from = target.get("_resolved_from")
if resolved_from == "origin_fallback":
- # Same activation rules as an origin target: per-job attach wins,
- # else the global flag. This deliberately restates the precedence
- # _cron_mirror_delivery_enabled encodes (keep the two in sync): the
- # sole production caller pre-merges it into `global_mirror`, but the
- # helper must stay correct standalone — a per-job False must beat a
- # raw global True for any caller that does not pre-merge.
+ # Same precedence as _cron_mirror_delivery_enabled (keep in sync): a per-job False must
+ # beat a global True even for callers that don't pre-merge `global_mirror`.
per_job = job.get("attach_to_session")
if isinstance(per_job, bool):
return per_job
@@ -1806,16 +1229,11 @@ def _target_mirror_eligible(
def _inchannel_seed_allowed(*, is_dm: bool, user_id: Optional[str]) -> bool:
- """Whether the flat in_channel session seed may run for a target.
+ """Whether the flat in_channel seed may run.
- Group-channel session keys are user-isolated
- (``…:group::`` — see _seed_cron_channel_session); a
- seed without a real user_id would create an orphan session that no
- inbound reply ever resolves to, which is worse than no seed (the plain
- mirror can still land if a session exists). DM keys don't embed
- user_id, so DM targets are always seedable. Origin-captured jobs carry
- the scheduler's user_id; origin-less managed jobs typically don't, and
- their group-channel targets must fall back to the plain mirror.
+ Group keys are user-isolated (``…:group::``): seeding without a real user_id
+ creates an orphan session no reply resolves to — worse than no seed. DM keys omit user_id, so
+ DMs are always seedable; origin-less group targets fall back to the plain mirror.
"""
return bool(is_dm or user_id)
@@ -1832,20 +1250,10 @@ def _maybe_mirror_cron_delivery(
) -> None:
"""Best-effort mirror of a cron delivery into the origin chat's session.
- No-op unless ``enabled`` (resolved once by the caller, and already scoped to
- the origin target — see ``_target_matches_origin``). Reuses the shipped
- ``mirror_to_session`` so cron rides exactly the same path that interactive
- ``send_message`` mirroring already uses, including passing ``user_id`` so a
- per-user-isolated group chat resolves to the exact member who scheduled the
- job (parity with ``send_message``). All failures are swallowed — a delivery
- that succeeded must never be reported as failed because the transcript
- mirror hit a problem.
-
- Because the caller only enables this for the target that equals the job's
- origin conversation, the session is expected to exist (the job was born in
- that session). A missing session therefore indicates an origin-less /
- fan-out delivery that should not have been mirrored anyway, and is treated
- as a silent no-op — never a synthetic session is created.
+ No-op unless ``enabled`` (caller resolves it, scoped to the origin target). Rides the same
+ ``mirror_to_session`` path as ``send_message``, passing ``user_id`` so user-isolated group
+ chats resolve to the scheduling member. All failures swallowed — a successful delivery must
+ never be reported failed because the mirror broke.
"""
if not enabled:
return
@@ -1855,14 +1263,8 @@ def _maybe_mirror_cron_delivery(
try:
from gateway.mirror import mirror_to_session
- # Mirror as a USER turn with a labelled prefix, NOT an assistant turn.
- # The brief is not the agent speaking; an assistant-role mirror lands as
- # assistant→assistant after the agent's last turn and breaks strict
- # alternation (issue #2221, the exact failure #2313 removed). A
- # user-role turn collapses safely via repair_message_sequence's
- # consecutive-user merge on every provider, and the prefix preserves the
- # "this came from cron" context that the dropped SQLite mirror metadata
- # would otherwise lose on replay.
+ # USER role + labelled prefix, NOT assistant: an assistant-role mirror lands
+ # assistant→assistant and breaks strict alternation; consecutive user turns merge safely.
ok = mirror_to_session(
platform_name,
str(chat_id),
@@ -1896,14 +1298,8 @@ def _open_continuable_cron_thread(
chat_id: str,
loop,
) -> Optional[str]:
- """Open a dedicated thread for a continuable cron job (thread-preferred).
-
- Returns the new ``thread_id`` on success, or ``None`` when the platform has
- no thread primitive (WhatsApp/Signal/SMS) or creation failed — the ``None``
- return is the caller's signal to fall back to the origin-DM mirror, the same
- open-thread-or-fallback shape as ``GatewayRunner._process_handoff``. Reuses
- the shipped ``adapter.create_handoff_thread``; no new adapter surface.
- """
+ """Open a thread for a continuable cron job via ``adapter.create_handoff_thread``. Returns the
+ thread_id, or ``None`` (no thread primitive / failed) = caller falls back to the DM mirror."""
create_thread = getattr(adapter, "create_handoff_thread", None)
if not callable(create_thread) or loop is None:
return None
@@ -1927,6 +1323,71 @@ def _open_continuable_cron_thread(
return None
+def _seed_cron_session(
+ job: dict,
+ adapter,
+ platform_name: str,
+ chat_id: str,
+ text: str,
+ *,
+ thread_id: Optional[str],
+ chat_type: str,
+ user_id: Optional[str],
+ user_name: Optional[str] = None,
+ chat_name: Optional[str],
+ scope_id: Optional[str],
+ discord_keys_on_thread: bool = False,
+) -> bool:
+ """Create the session row (so the mirror has a target) and mirror the brief as a USER turn.
+ The seeded key must equal the reply's ``build_session_key``: chat_type, user_id, thread_id and
+ scope_id (Slack team id) are all part of it, so callers pass exactly what the reply carries."""
+ from gateway.config import Platform
+ from gateway.session import SessionSource
+
+ seeded_session_id: Optional[str] = None
+ session_store = getattr(adapter, "_session_store", None)
+ if session_store is not None:
+ try:
+ platform_enum = Platform(platform_name.lower())
+ except (ValueError, KeyError):
+ platform_enum = None
+ if platform_enum is not None:
+ # Discord keys in-thread messages with chat_id == thread_id; Slack/Telegram use the
+ # parent channel.
+ seed_chat_id = (
+ str(thread_id)
+ if discord_keys_on_thread and platform_enum == Platform.DISCORD
+ else str(chat_id)
+ )
+ dest_source = SessionSource(
+ platform=platform_enum,
+ chat_id=seed_chat_id,
+ chat_name=chat_name,
+ chat_type=chat_type,
+ user_id=user_id,
+ user_name=user_name,
+ thread_id=thread_id,
+ scope_id=str(scope_id) if scope_id else None,
+ )
+ # Create the row and pass its exact id to the mirror — origin-heuristic rediscovery
+ # bails on populated chats.
+ _entry = session_store.get_or_create_session(dest_source)
+ seeded_session_id = getattr(_entry, "session_id", None)
+
+ from gateway.mirror import mirror_to_session
+
+ return mirror_to_session(
+ platform_name,
+ str(chat_id),
+ f"[Cron delivery: {job.get('name') or job.get('id', 'cron')}]\n{text}",
+ source_label="cron",
+ thread_id=thread_id,
+ user_id=user_id,
+ role="user",
+ session_id=seeded_session_id,
+ )
+
+
def _seed_cron_thread_session(
job: dict,
adapter,
@@ -1938,97 +1399,23 @@ def _seed_cron_thread_session(
is_dm: bool = False,
scope_id: Optional[str] = None,
) -> None:
- """Seed the freshly-opened cron thread's session with the brief.
-
- Without this the brief is *visible* in the new thread but absent from any
- transcript, so the user's first reply in-thread would hit a session with no
- record of it ("what is Task #2?"). We create the thread-keyed session (the
- same key the user's reply will resolve to — ``build_session_key`` keys
- threads as participant-shared, so no ``user_id`` is needed) and append the
- brief as an assistant turn via the shipped ``mirror_to_session``.
-
- ``scope_id`` is the workspace/server scope (Slack team id).
- ``build_session_key`` embeds it in every Slack key, so a scoped reply's
- key carries it — the seed must reproduce it or the seeded row is
- unreachable (the scope-less flat-seed sibling of the is_dm keying bug).
- Best-effort None for platforms without scope.
-
- ``is_dm`` selects the seeded ``chat_type``: a thread under a DM must seed
- ``chat_type="dm"`` because the user's in-thread DM reply arrives with
- chat_type="dm" and ``build_session_key`` routes DM threads through the DM
- arm (``...:dm::``) — a "thread"-typed seed lands in
- ``...:thread::``, a row no DM reply ever resolves to
- (continuation amnesia, Alice live 2026-08-20, job 8e21a957b77b). Channel
- threads keep ``chat_type="thread"`` (their replies really do arrive as
- threads). Same sibling-lane class as the flat seed's ``is_dm``
- (dcca9d8cfe).
-
- Mirrors ``GatewayRunner._process_handoff``'s seed step, but standalone:
- cron reaches the live ``SessionStore`` through the adapter's
- ``_session_store`` handle rather than the gateway object. Best-effort — a
- delivery that already succeeded is never failed by a seeding problem.
- """
+ """Seed the freshly-opened cron thread's session with the brief (never raises), else the
+ user's in-thread reply resolves to a transcript without it. Threads are participant-shared (no
+ real user_id); a DM thread must seed ``chat_type="dm"`` — DM-thread replies route through the DM
+ arm (``…:dm::``), so a "thread"-typed seed is a row no DM reply ever hits."""
text = (mirror_text or "").strip()
if not text:
return
try:
- from gateway.config import Platform
- from gateway.session import SessionSource
-
- seeded_session_id: Optional[str] = None
- session_store = getattr(adapter, "_session_store", None)
- if session_store is not None:
- try:
- platform_enum = Platform(platform_name.lower())
- except (ValueError, KeyError):
- platform_enum = None
- if platform_enum is not None:
- # Discord thread destinations must key on the thread's OWN id
- # to match how the Discord adapter keys organic in-thread
- # messages (chat_id == thread_id). Other platforms (Slack,
- # Telegram) use chat_id == parent_channel for thread messages,
- # so the parent chat_id is correct for them. See the matching
- # guard in GatewayRunner._process_handoff.
- if platform_enum == Platform.DISCORD:
- seed_chat_id = str(thread_id)
- else:
- seed_chat_id = str(chat_id)
- dest_source = SessionSource(
- platform=platform_enum,
- chat_id=seed_chat_id,
- chat_name=chat_name,
- # DM threads key through the DM arm (see docstring); the
- # reply's chat_type is what the seed must reproduce.
- chat_type="dm" if is_dm else "thread",
- user_id="system:cron",
- user_name="Cron",
- thread_id=str(thread_id),
- scope_id=str(scope_id) if scope_id else None,
- )
- # Ensure the thread-keyed session row exists so the mirror has
- # a target and the user's later reply joins the same session.
- # Capture the exact id — the mirror writes into THIS row, not
- # an origin-heuristic rediscovery (which bails on populated
- # chats; same class as the flat-seed live failure 2026-08-19).
- _entry = session_store.get_or_create_session(dest_source)
- seeded_session_id = getattr(_entry, "session_id", None)
-
- from gateway.mirror import mirror_to_session
-
- # User-role + labelled prefix (see _maybe_mirror_cron_delivery): the
- # seeded brief must not read as an assistant turn, or the user's first
- # in-thread reply produces assistant→user→... off a phantom assistant
- # message. Pass the seed user_id so the mirror resolves the exact
- # thread-keyed session row we just created.
- ok = mirror_to_session(
- platform_name,
- str(chat_id),
- f"[Cron delivery: {job.get('name') or job.get('id', 'cron')}]\n{text}",
- source_label="cron",
+ ok = _seed_cron_session(
+ job, adapter, platform_name, chat_id, text,
thread_id=str(thread_id),
+ chat_type="dm" if is_dm else "thread",
user_id="system:cron",
- role="user",
- session_id=seeded_session_id,
+ user_name="Cron",
+ chat_name=chat_name,
+ scope_id=scope_id,
+ discord_keys_on_thread=True,
)
if ok:
logger.info(
@@ -2042,8 +1429,7 @@ def _seed_cron_thread_session(
job.get("id", "?"), platform_name, chat_id, thread_id,
)
except Exception as e:
- # WARNING, not debug: a silent seed failure IS the continuation-
- # amnesia bug (Alice 2026-08-19) — it must be visible in production.
+ # WARNING, not debug: a silent seed failure IS the continuation-amnesia bug.
logger.warning(
"Job '%s': seeding cron thread session failed for %s:%s:%s: %s",
job.get("id", "?"), platform_name, chat_id, thread_id, e,
@@ -2062,87 +1448,23 @@ def _seed_cron_channel_session(
chat_name: Optional[str] = None,
scope_id: Optional[str] = None,
) -> bool:
- """Seed the FLAT (thread_id=None) session for an ``in_channel`` cron delivery.
-
- The ``in_channel`` surface (D1/D2) delivers the brief flat into the channel
- with no thread, so the continuation surface is the whole-channel /
- whole-DM session keyed ``thread_id=None`` — the same bucket
- ``reply_in_thread: false`` routes an inbound plain reply to.
-
- Unlike the thread path, the shipped delivery-mirror alone is NOT sufficient
- here: ``mirror_to_session`` only APPENDS to a session that already EXISTS
- (``_find_session_id`` → no-op when none matches), and a flat channel
- ``(…, None)`` row is only created when a human posts a top-level message the
- bot processes — a ``chat_postMessage`` cron delivery never goes through the
- inbound handler, so the row is usually absent and the mirror silently drops
- the brief (verified live: the brief never landed, the reply had no context).
- So we CREATE the flat session row first, exactly like
- ``_seed_cron_thread_session`` does for threads, then mirror into it.
-
- The session KEY must match what the user's later inbound reply resolves to
- (``build_session_key``):
- - **Channel** (``chat_type="group"``): key is
- ``…:group::`` — user-isolated — so the seed MUST carry
- the **origin's real ``user_id``** (the member who scheduled the job), NOT
- a synthetic ``system:cron`` id, or the reply keys to a different session.
- - **1:1 DM** (``chat_type="dm"``): the key is ``…:dm:`` and does
- NOT embed ``user_id``, so any ``user_id`` resolves to the same session.
- ``chat_type`` mirrors the inbound handler's own choice
- (``"dm" if is_dm else "group"``, ``adapter.py``), so the seeded key is
- byte-identical to the reply's key.
-
- Returns True if a seed row was created and the brief mirrored, else False
- (caller falls back to the plain mirror). Best-effort — a delivery that
- already succeeded is never failed by a seeding problem.
- """
+ """Seed the FLAT (thread_id=None) session for an ``in_channel`` delivery; True on success.
+ ``mirror_to_session`` only APPENDS to an existing session and the flat row is only created by an
+ inbound human message, so the row must be created first or the brief is silently dropped. Group
+ keys are user-isolated (``…:group::``): the seed MUST carry the origin's real
+ user_id, not ``system:cron``; DM keys omit user_id. chat_type mirrors the inbound handler."""
text = (mirror_text or "").strip()
if not text:
return False
try:
- from gateway.config import Platform
- from gateway.session import SessionSource
-
chat_type = "dm" if is_dm else "group"
- session_store = getattr(adapter, "_session_store", None)
- seeded_session_id: Optional[str] = None
- if session_store is not None:
- try:
- platform_enum = Platform(platform_name.lower())
- except (ValueError, KeyError):
- platform_enum = None
- if platform_enum is not None:
- dest_source = SessionSource(
- platform=platform_enum,
- chat_id=str(chat_id),
- chat_name=chat_name,
- chat_type=chat_type,
- user_id=str(user_id) if user_id else None,
- thread_id=None, # flat — the whole-channel/DM session
- # Workspace scope: build_session_key embeds it in every
- # Slack key, so a scoped reply only resolves to this row
- # when the seed carries it too (see thread-seed docstring).
- scope_id=str(scope_id) if scope_id else None,
- )
- # Create the flat session row so the mirror has a target and the
- # user's later plain reply joins the SAME session. Capture the
- # exact session id: the mirror must write into THIS row, not
- # re-discover it via origin heuristics (which bail out on
- # populated chats where the flat session coexists with
- # per-message thread sessions — live failure, Alice 2026-08-19).
- _entry = session_store.get_or_create_session(dest_source)
- seeded_session_id = getattr(_entry, "session_id", None)
-
- from gateway.mirror import mirror_to_session
-
- ok = mirror_to_session(
- platform_name,
- str(chat_id),
- f"[Cron delivery: {job.get('name') or job.get('id', 'cron')}]\n{text}",
- source_label="cron",
- thread_id=None,
+ ok = _seed_cron_session(
+ job, adapter, platform_name, chat_id, text,
+ thread_id=None, # flat — the whole-channel/DM session
+ chat_type=chat_type,
user_id=str(user_id) if user_id else None,
- session_id=seeded_session_id,
- role="user",
+ chat_name=chat_name,
+ scope_id=scope_id,
)
if ok:
logger.info(
@@ -2151,9 +1473,7 @@ def _seed_cron_channel_session(
)
return bool(ok)
except Exception as e:
- # WARNING, not debug: a silent seed failure IS the "agent has no idea
- # about its own brief" bug (Alice 2026-08-19) — it must be visible in
- # production logs.
+ # WARNING, not debug: a silent seed failure IS the continuation-amnesia bug.
logger.warning(
"Job '%s': seeding in_channel session failed for %s:%s: %s",
job.get("id", "?"), platform_name, chat_id, e,
@@ -2162,14 +1482,8 @@ def _seed_cron_channel_session(
def _cron_job_origin_log_suffix(job: dict) -> str:
- """Return safe provenance details for security warnings about a cron job.
-
- The scheduler normally has no live HTTP request object when it detects a
- bad stored ``context_from`` reference. Including the job's saved origin
- makes future probe logs actionable without exposing secrets: platform/chat
- metadata for gateway-created jobs, and optional source-IP fields for API
- surfaces that persist them in origin metadata.
- """
+ """Secret-free provenance suffix (origin platform/chat/source-IP fields) for security warnings
+ about a bad stored ``context_from`` reference, where no live request object exists."""
origin = job.get("origin")
if not isinstance(origin, dict):
return ""
@@ -2186,31 +1500,19 @@ def _cron_job_origin_log_suffix(job: dict) -> str:
def _plugin_cron_env_var(platform_name: str) -> str:
- """Return the cron home-channel env var registered by a plugin platform.
-
- Falls through the platform registry so plugins that set
- ``cron_deliver_env_var`` on their ``PlatformEntry`` get cron delivery
- support without editing this module.
- """
- try:
+ """Cron home-channel env var registered by a plugin ``PlatformEntry.cron_deliver_env_var``."""
+ with contextlib.suppress(Exception):
from hermes_cli.plugins import discover_plugins
discover_plugins() # idempotent
from gateway.platform_registry import platform_registry
entry = platform_registry.get(platform_name.lower())
if entry and entry.cron_deliver_env_var:
return entry.cron_deliver_env_var
- except Exception:
- pass
return ""
def _is_known_delivery_platform(platform_name: str) -> bool:
- """Whether ``platform_name`` is a valid cron delivery target.
-
- Hardcoded built-ins in ``_KNOWN_DELIVERY_PLATFORMS`` are checked first;
- plugin platforms registered via ``PlatformEntry`` are accepted if they
- provide a ``cron_deliver_env_var``.
- """
+ """Valid cron delivery platform: built-in, or plugin with a ``cron_deliver_env_var``."""
name = platform_name.lower()
if name in _KNOWN_DELIVERY_PLATFORMS:
return True
@@ -2218,11 +1520,7 @@ def _is_known_delivery_platform(platform_name: str) -> bool:
def _resolve_home_env_var(platform_name: str) -> str:
- """Return the env var name for a platform's cron home channel.
-
- Built-in platforms are in ``_HOME_TARGET_ENV_VARS``; plugin platforms are
- resolved from the platform registry.
- """
+ """Env var name for a platform's cron home channel (built-in table, then plugin registry)."""
name = platform_name.lower()
env_var = _HOME_TARGET_ENV_VARS.get(name)
if env_var:
@@ -2231,17 +1529,10 @@ def _resolve_home_env_var(platform_name: str) -> str:
def _get_config_home_channel(platform_name: str):
- """Return the persisted ``HomeChannel`` for a platform from gateway config.
+ """Persisted ``HomeChannel`` from gateway config — the canonical store ``/sethome`` writes.
- ``/sethome`` declares ``config.yaml`` canonical (it is the only store that
- survives for relay-fronted logical platforms, whose adapters are not
- natively enabled) and mirrors the value into the legacy
- ``_HOME_CHANNEL`` env var only as a best-effort compatibility
- shim. Cron historically read ONLY the env mirror, so a home channel that
- existed solely in config.yaml — e.g. Discord fronted by the relay
- connector, where no ``DISCORD_HOME_CHANNEL`` was ever exported — was
- invisible and jobs silently fell back to local-only. Reading the
- canonical store here fixes that for every relay-fronted platform at once.
+ The ``_HOME_CHANNEL`` env var is only a best-effort mirror; relay-fronted platforms
+ may exist solely in config.yaml, so reading only the env mirror silently drops their delivery.
"""
try:
from gateway.config import load_gateway_config, Platform
@@ -2258,15 +1549,11 @@ def _get_config_home_channel(platform_name: str):
def _env_home_target_chat_id(platform_name: str) -> str:
- """Return the home chat id from the legacy env mirror only (no config).
+ """Home chat id from the env mirror only (no config).
- Reads through ``get_secret`` (not raw ``os.getenv``) so a profile-scoped
- secret scope wins in a multiplex gateway. ``DISCORD_HOME_CHANNEL`` lives in
- each profile's ``.env``; in a multiplex process the winning cron tick runs
- with the job-owning profile's scope installed (run_one_job sets it), so
- reading via ``get_secret`` resolves the OWNING profile's chat id rather
- than the host process's ``os.environ`` (#83182, chat-id leg — the token
- leg was fixed earlier; chat id / thread id resolve through the same leak).
+ Reads via ``get_secret``, not ``os.getenv``: in a multiplex gateway the tick runs with the
+ job-owning profile's secret scope (run_one_job sets it), so this resolves the OWNING profile's
+ chat id rather than the host process's environ.
"""
env_var = _resolve_home_env_var(platform_name)
if not env_var:
@@ -2291,12 +1578,8 @@ def _env_home_target_chat_id(platform_name: str) -> str:
def _get_home_target_chat_id(platform_name: str) -> str:
- """Return the configured home target chat/room ID for a delivery platform.
-
- Resolution order: platform env var (legacy mirror, kept first so an
- operator override keeps winning) → legacy env var name → the canonical
- ``home_channel`` block persisted in config.yaml by ``/sethome``.
- """
+ """Home target chat id: env var (first, so operator overrides win) → legacy env var →
+ config.yaml ``home_channel``."""
value = _env_home_target_chat_id(platform_name)
if value:
return value
@@ -2307,15 +1590,10 @@ def _get_home_target_chat_id(platform_name: str) -> str:
def _get_home_target_thread_id(platform_name: str) -> Optional[str]:
- """Return the optional thread/topic ID for a platform home target.
+ """Optional thread/topic id for a platform home target.
- Telegram-only override: ``TELEGRAM_CRON_THREAD_ID`` takes precedence over
- ``TELEGRAM_HOME_CHANNEL_THREAD_ID`` for cron delivery. When topic mode is
- enabled, deliveries that land in the root DM (thread_id unset) end up in
- the system-only lobby where the user cannot reply — the gateway returns
- the lobby reminder and drops ``reply_to_message_id`` (#24409). Pointing
- cron at a dedicated topic via this env var lets replies work as expected
- without changing the lobby invariant.
+ Telegram: ``TELEGRAM_CRON_THREAD_ID`` overrides ``TELEGRAM_HOME_CHANNEL_THREAD_ID`` — in topic
+ mode a root-DM delivery lands in the system-only lobby where the user cannot reply.
"""
env_var = _resolve_home_env_var(platform_name)
try:
@@ -2347,10 +1625,8 @@ def _get_home_target_thread_id(platform_name: str) -> Optional[str]:
value = os.getenv(f"{legacy}_THREAD_ID", "").strip()
if value:
return value
- # Canonical config.yaml fallback — same rationale as
- # _get_home_target_chat_id, and thread affinity only applies when the
- # chat itself resolved from the same config block (an env-provided chat
- # id keeps its env-provided thread semantics).
+ # config.yaml fallback only when the chat id also came from config (an env-provided chat id
+ # keeps its env-provided thread semantics).
if not _env_home_target_chat_id(platform_name):
home = _get_config_home_channel(platform_name)
if home is not None and home.thread_id:
@@ -2359,36 +1635,22 @@ def _get_home_target_thread_id(platform_name: str) -> Optional[str]:
def _iter_home_target_platforms():
- """Iterate built-in + plugin platform names that expose a home channel.
-
- Used by the ``deliver=origin`` fallback when the job has no origin.
- """
+ """Iterate built-in + plugin platform names that expose a home channel."""
for name in _HOME_TARGET_ENV_VARS:
yield name
- try:
+ with contextlib.suppress(Exception):
from hermes_cli.plugins import discover_plugins
discover_plugins() # idempotent
from gateway.platform_registry import platform_registry
for entry in platform_registry.plugin_entries():
if entry.cron_deliver_env_var and entry.name not in _HOME_TARGET_ENV_VARS:
yield entry.name
- except Exception:
- pass
def _relay_fronted_delivery_platforms(connected: set) -> set:
- """Logical platforms deliverable through a connected relay connector.
-
- ``get_connected_platforms()`` only sees NATIVELY configured platforms.
- On a relay-fronted deployment (relay in ``config.platforms``, the real
- platform credential living in the connector) the fronted platforms are
- absent from that set although fire-time routing delivers to them via
- ``resolve_delivery_transport`` + ``RelayAdapter.fronts_platform``. This
- keeps validation symmetric with routing by consulting the same
- env-derived deploy stamp (``GATEWAY_RELAY_PLATFORMS``) the live
- adapter's identity set is seeded from. No relay connected -> empty set,
- so native topologies keep the strict credential check unchanged.
- """
+ """Logical platforms deliverable through a connected relay. ``get_connected_platforms()`` only
+ sees native platforms; fronted ones come from the same ``GATEWAY_RELAY_PLATFORMS`` stamp
+ fire-time routing uses (validation symmetric with routing). No relay -> empty set."""
if "relay" not in connected:
return set()
try:
@@ -2401,18 +1663,11 @@ def _relay_fronted_delivery_platforms(connected: set) -> set:
def cron_delivery_targets() -> list[dict]:
- """Return the platforms a cron job can auto-deliver to.
+ """Platforms a cron job can auto-deliver to (single source of truth for UIs).
- Single source of truth for any UI (dashboard dropdown, etc.) that lets a
- user pick a cron delivery target. A platform is included when it is a valid
- cron delivery platform AND its gateway is configured (enabled + credentials
- present). Each entry reports whether the platform's home target (the
- room/channel cron posts to) is set — a platform can be configured for
- interactive use but still lack the home target an unattended cron job needs.
-
- Returns a list of dicts: ``{"id", "name", "home_target_set", "home_env_var"}``
- ordered by the gateway's canonical platform order. Callers should always
- prepend the implicit ``local`` option themselves — it needs no config.
+ Included when a valid delivery platform AND gateway-configured; ``home_target_set`` flags
+ whether the home channel exists. Returns ``{"id", "name", "home_target_set", "home_env_var"}``
+ dicts in canonical order; callers prepend the implicit ``local`` option themselves.
"""
targets: list[dict] = []
try:
@@ -2440,10 +1695,7 @@ def cron_delivery_targets() -> list[dict]:
}
)
- # Bot Chat targets: one per local profile. Machine-local by design (the
- # scheduler delivers via a local chat subprocess), so the names listed
- # here are exactly the names that resolve at fire time — no gateway
- # config, no home channel needed.
+ # Bot Chat targets: one per local profile (machine-local; no gateway config or home channel).
try:
from hermes_cli.profiles import list_profile_names
@@ -2464,20 +1716,12 @@ def cron_delivery_targets() -> list[dict]:
def _origin_thread_is_stale(origin: dict) -> bool:
"""True when a Slack origin's thread is a stale creation-turn artifact.
- Relay-fronted Slack in thread-per-message mode stamps each top-level
- message's own id as the session thread (a session KEY, not a durable
- location). Jobs persisted before origin capture learned to drop that
- stamp carry it as ``origin.thread_id`` forever. Heuristic that repairs
- them at fire time without touching genuine threads: when the origin
- chat IS the configured Slack home chat (the ``/sethome`` conversation),
- a pinned origin thread is the creation-message artifact — the user's
- delivery expectation for their home conversation is top-level (or the
- home target's own configured thread). Non-home chats keep their
- threads: a job deliberately created inside a working thread stays there.
+ Thread-per-message Slack stamps each top-level message id as the session thread (a KEY, not a
+ location); old jobs carry it as ``origin.thread_id``. Heuristic: if the origin chat IS the Slack
+ home chat, the pinned thread is that artifact and delivery goes top-level (or to the home
+ target's thread). Non-home chats keep their threads.
"""
- if str(origin.get("platform") or "").lower() != "slack":
- return False
- if not origin.get("thread_id"):
+ if str(origin.get("platform") or "").lower() != "slack" or not origin.get("thread_id"):
return False
home_chat = _get_home_target_chat_id("slack")
return bool(home_chat) and str(origin.get("chat_id")) == str(home_chat)
@@ -2486,11 +1730,22 @@ def _origin_thread_is_stale(origin: dict) -> bool:
def _origin_delivery_thread(origin: dict):
"""The thread a deliver=origin job should use, stale stamps dropped."""
if _origin_thread_is_stale(origin):
- home_thread = _get_home_target_thread_id("slack")
- return home_thread if home_thread else None
+ return _get_home_target_thread_id("slack") or None
return origin.get("thread_id")
+def _home_target(platform_name: str, chat_id: str, resolved_from: Optional[str] = None) -> dict:
+ """Target dict for a platform's configured home channel (+ optional mirror provenance)."""
+ target = {
+ "platform": platform_name,
+ "chat_id": chat_id,
+ "thread_id": _get_home_target_thread_id(platform_name),
+ }
+ if resolved_from:
+ target["_resolved_from"] = resolved_from
+ return target
+
+
def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[dict]:
"""Resolve one concrete auto-delivery target for a cron job."""
@@ -2499,9 +1754,7 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d
if deliver_value == "local":
return None
- # bot-chat[:] — checked before the generic platform:chat_id
- # split below so the profile-name argument is never misparsed as a
- # chat_id on an unknown platform.
+ # Must precede the generic platform:chat_id split so the profile name isn't parsed as chat_id.
bot_chat_profile = parse_bot_chat_deliver_token(deliver_value)
if bot_chat_profile is not None:
return _resolve_bot_chat_target(job, bot_chat_profile)
@@ -2512,12 +1765,10 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d
"platform": origin["platform"],
"chat_id": str(origin["chat_id"]),
"thread_id": _origin_delivery_thread(origin),
- # Resolution provenance for mirror eligibility (see
- # _target_mirror_eligible): this IS the origin conversation.
+ # Provenance for _target_mirror_eligible.
"_resolved_from": "origin",
}
- # Origin missing (e.g. job created via API/script) — try each
- # platform's home channel as a fallback instead of silently dropping.
+ # No origin (API/script job): fall back to a home channel instead of silently dropping.
for platform_name in _iter_home_target_platforms():
chat_id = _get_home_target_chat_id(platform_name)
if chat_id:
@@ -2526,16 +1777,8 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d
job.get("name", job.get("id", "?")),
platform_name,
)
- return {
- "platform": platform_name,
- "chat_id": chat_id,
- "thread_id": _get_home_target_thread_id(platform_name),
- # The fallback stands in for the user's primary
- # conversation (NOT a broadcast) — mirror-eligible so
- # continuable crons work for script-provisioned jobs
- # that never captured an origin.
- "_resolved_from": "origin_fallback",
- }
+ # Stands in for the primary conversation (NOT a broadcast): mirror-eligible.
+ return _home_target(platform_name, chat_id, "origin_fallback")
return None
if ":" in deliver_value:
@@ -2548,20 +1791,13 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d
)
prepare_send_message_platforms()
- # pass_unresolved_references: stored jobs have no model in the loop to react
- # to a resolution error, and a target the directory doesn't know
- # (fresh install, platform-native id) used to be handed to the
- # adapter as written. Dropping it here silently loses the job's
- # output.
+ # pass_unresolved_references: no model in the loop to react; an unknown-to-directory target
+ # must reach the adapter as written or the job's output is silently lost.
chat_id, thread_id, resolution_error = resolve_send_target(
platform_key, rest, pass_unresolved_references=True
)
if resolution_error:
- logger.warning(
- "Invalid cron delivery target '%s': %s",
- deliver_value,
- resolution_error,
- )
+ logger.warning("Invalid cron delivery target '%s': %s", deliver_value, resolution_error)
return None
if (
@@ -2579,8 +1815,7 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d
"platform": platform_name,
"chat_id": chat_id,
"thread_id": thread_id,
- # Explicit platform:chat target — mirror-eligible only under the
- # job's own attach_to_session opt-in (see _target_mirror_eligible).
+ # Mirror-eligible only under the job's own attach_to_session opt-in.
"_resolved_from": "explicit",
}
@@ -2588,11 +1823,7 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d
if origin and origin.get("platform") == platform_name:
chat_id = _get_home_target_chat_id(platform_name)
if chat_id:
- return {
- "platform": platform_name,
- "chat_id": chat_id,
- "thread_id": _get_home_target_thread_id(platform_name),
- }
+ return _home_target(platform_name, chat_id)
return {
"platform": platform_name,
"chat_id": str(origin["chat_id"]),
@@ -2602,22 +1833,12 @@ def _resolve_single_delivery_target(job: dict, deliver_value: str) -> Optional[d
if not _is_known_delivery_platform(platform_name):
return None
chat_id = _get_home_target_chat_id(platform_name)
- if not chat_id:
- return None
-
- return {
- "platform": platform_name,
- "chat_id": chat_id,
- "thread_id": _get_home_target_thread_id(platform_name),
- }
+ return _home_target(platform_name, chat_id) if chat_id else None
def _get_bot_chat_delivery_timeout() -> int:
- """Timeout for one bot-chat delivery turn (the target bot runs a full
- agent turn on the injected output, so this is minutes, not seconds).
-
- ``cron.bot_chat_delivery_timeout_seconds`` in config.yaml; default 600.
- """
+ """Timeout for one bot-chat delivery turn (a full agent turn — minutes, not seconds).
+ ``cron.bot_chat_delivery_timeout_seconds``; default 600."""
try:
cfg = load_config()
value = int(cfg.get("cron", {}).get("bot_chat_delivery_timeout_seconds", 600))
@@ -2627,18 +1848,12 @@ def _get_bot_chat_delivery_timeout() -> int:
def _deliver_to_bot_chat(job: dict, content: str, profile: str) -> Optional[str]:
- """Deliver job output into a profile's canonical Bot Chat as an inbound turn.
+ """Deliver job output into a profile's canonical Bot Chat as a real inbound user turn.
- Runs ``hermes [-p ] chat --in ~ -c "Bot Chat" --create-if-missing
- -Q --query-file `` — the exact lane Bot Mode agent-to-agent messages
- use, so the adopt-before-mint canonical-session rules apply and the target
- bot receives the output as a real user-role message it can act on.
- Alternation-safe by construction: this is an inbound turn on the chat
- command lane, not a transcript splice.
-
- ``profile`` is ``""`` for the job's own profile (subprocess inherits this
- scheduler's HERMES_HOME) or a validated local profile name. Returns None
- on success or an error string for ``last_delivery_error``.
+ Runs ``hermes [-p ] chat --in ~ -c "Bot Chat" --create-if-missing -Q --query-file`` —
+ the same lane Bot Mode agent-to-agent messages use, so canonical-session rules apply and it is
+ alternation-safe (inbound turn, not a transcript splice). ``profile`` is ``""`` for the job's
+ own profile. Returns None on success or an error string for ``last_delivery_error``.
"""
import shutil as _shutil
import tempfile
@@ -2663,12 +1878,10 @@ def _deliver_to_bot_chat(job: dict, content: str, profile: str) -> Optional[str]
env = os.environ.copy()
if profile:
argv += ["-p", profile]
- # -p owns profile resolution in the child; a leftover HERMES_HOME
- # from THIS scheduler's profile must not shadow it.
+ # -p owns profile resolution; this scheduler's HERMES_HOME must not shadow it.
env.pop("HERMES_HOME", None)
- # The prefix tells the receiving bot this is scheduled output, not the
- # human typing — mirrors the Bot Mode sender-attribution convention.
+ # Prefix marks this as scheduled output, not the human (Bot Mode sender-attribution).
message = (
f'[Cronjob "{job_name}" output — scheduled job, not the user. '
f"Review it, act on anything that needs action, and summarize "
@@ -2706,10 +1919,7 @@ def _deliver_to_bot_chat(job: dict, content: str, profile: str) -> Optional[str]
)
logger.warning("Job '%s': %s", job_id, msg)
return msg
- logger.info(
- "Job '%s': delivered to Bot Chat of profile '%s'",
- job_id, profile or "(own)",
- )
+ logger.info("Job '%s': delivered to Bot Chat of profile '%s'", job_id, profile or "(own)")
return None
except subprocess.TimeoutExpired:
msg = (
@@ -2726,23 +1936,15 @@ def _deliver_to_bot_chat(job: dict, content: str, profile: str) -> Optional[str]
return msg
finally:
if query_file:
- try:
+ with contextlib.suppress(OSError):
os.unlink(query_file)
- except OSError:
- pass
def _normalize_deliver_value(deliver) -> str:
- """Normalize a stored/submitted ``deliver`` value to its canonical string form.
+ """Normalize ``deliver`` to its canonical comma-separated string; ``"local"`` when falsy.
- The contract is that ``deliver`` is a string (``"local"``, ``"origin"``,
- ``"telegram"``, ``"telegram:-1001:17"``, or comma-separated combinations).
- Historically some callers — MCP clients passing an array, direct edits of
- ``jobs.json``, or stale code paths — have stored a list/tuple like
- ``["telegram"]``. ``str(["telegram"])`` would serialize to the literal
- string ``"['telegram']"``, which is not a known platform and fails
- resolution silently. Flatten lists/tuples into a comma-separated string
- so both forms work. Returns ``"local"`` for anything falsy.
+ Lists/tuples (MCP clients, hand-edited jobs.json) are flattened — ``str(["telegram"])`` would
+ yield ``"['telegram']"`` and fail resolution silently.
"""
if deliver is None or deliver == "":
return "local"
@@ -2752,30 +1954,18 @@ def _normalize_deliver_value(deliver) -> str:
return str(deliver)
-# Routing intent tokens — resolved at fire time, not create time, so a
-# job created before Telegram was wired up will pick up Telegram once it
-# comes online. ``all`` expands into the set of connected platforms
-# (those with a configured home chat_id) in _expand_routing_tokens.
+# Routing tokens resolve at fire time (a job outlives platform wiring). ``all`` = platforms with a
+# configured home chat_id (_expand_routing_tokens); ``bot-chat`` is NOT in ``all`` (costs a turn).
_ROUTING_TOKENS = frozenset({"all"})
-# Pseudo-platform for delivering job output INTO a profile's canonical
-# "Bot Chat" session as a real inbound turn (the bot sees it, runs a turn,
-# and can respond — Bot Mode's agent-to-agent lane, not a transcript
-# mirror). ``bot-chat`` targets the job's own profile; ``bot-chat:``
-# targets a named profile on THIS machine. Deliberately excluded from the
-# ``all`` routing token: ``all`` fans out to messaging home channels, and a
-# bot-chat delivery costs a full agent turn.
+# Pseudo-platform: deliver output as a real inbound turn into a profile's "Bot Chat" (not a mirror).
+# ``bot-chat`` = own profile; ``bot-chat:`` = named profile on THIS machine.
BOT_CHAT_PLATFORM = "bot-chat"
def parse_bot_chat_deliver_token(part: str) -> Optional[str]:
- """Return the target profile for a ``bot-chat[:]`` deliver token.
-
- Returns ``""`` for the bare token (the job's own profile), the profile
- name for the explicit form, or ``None`` when ``part`` is not a bot-chat
- token at all. Case-insensitive on the token; the profile name is
- normalized by the profile layer at resolve time.
- """
+ """``bot-chat[:]`` → ``""`` (own profile), the name, or ``None`` if not a bot-chat token.
+ Token is case-insensitive; the name is normalized later by the profile layer."""
raw = (part or "").strip()
lowered = raw.lower()
if lowered == BOT_CHAT_PLATFORM:
@@ -2787,18 +1977,10 @@ def parse_bot_chat_deliver_token(part: str) -> Optional[str]:
def _resolve_bot_chat_target(job: dict, profile_arg: str) -> Optional[dict]:
- """Resolve a bot-chat deliver token to a concrete delivery target.
-
- ``profile_arg`` is ``""`` for the job's own profile (the HERMES_HOME
- this scheduler runs under — machine-local and self-referential, so no
- ``-p`` flag is needed at send time) or an explicit profile name that
- must exist in THIS machine's profile root. Cross-machine delivery is
- intentionally unsupported: names resolve only against the local
- ``~/.hermes/profiles/`` tree, so same-named profiles on other gateways
- can never be targeted by accident.
- """
+ """Resolve a bot-chat token to a delivery target. ``""`` = own profile (no ``-p`` needed);
+ otherwise the profile must exist locally — cross-machine delivery is intentionally unsupported
+ so same-named profiles on other gateways can never be targeted by accident."""
if not profile_arg:
- # Own profile: chat subprocess inherits HERMES_HOME, no name needed.
return {"platform": BOT_CHAT_PLATFORM, "chat_id": "", "thread_id": None}
try:
from hermes_cli.profiles import normalize_profile_name, profile_exists
@@ -2821,13 +2003,8 @@ def _resolve_bot_chat_target(job: dict, profile_arg: str) -> Optional[dict]:
def _expand_routing_tokens(part: str) -> List[str]:
- """Expand a routing-intent token to concrete platform names.
-
- ``all`` expands to every platform in ``_iter_home_target_platforms()``
- that has a configured home chat_id right now. Unknown / non-token
- values pass through unchanged as a single-element list, so the caller
- can treat every token uniformly.
- """
+ """Expand ``all`` to every home-target platform with a configured chat_id; non-tokens pass
+ through as a single-element list."""
token = part.lower()
if token not in _ROUTING_TOKENS:
return [part]
@@ -2839,11 +2016,9 @@ def _expand_routing_tokens(part: str) -> List[str]:
def _delivery_lane_value(job: dict, *, for_failure: bool = False):
- """Raw deliver-lane value for a run outcome: the failure lane when
- ``for_failure`` and the job overrides it, else ``deliver``. Keeps
- delivery bookkeeping (outcome classification, unresolved-origin,
- incident 'alerted' marking) reading the SAME lane the notice was
- actually routed through (NS-788 review finding B1)."""
+ """Raw deliver-lane value for a run outcome: the failure lane when ``for_failure`` and the job
+ overrides it, else ``deliver``. Bookkeeping (outcome classification, unresolved-origin, incident
+ 'alerted' marking) must read the SAME lane the notice was routed through (NS-788)."""
if for_failure:
failure_deliver = job.get("failure_deliver")
if failure_deliver is not None and str(failure_deliver).strip():
@@ -2852,31 +2027,17 @@ def _delivery_lane_value(job: dict, *, for_failure: bool = False):
def _resolve_delivery_targets(job: dict, *, for_failure: bool = False) -> List[dict]:
- """Resolve all concrete auto-delivery targets for a cron job.
-
- Accepts the legacy comma-separated ``deliver`` string plus the
- ``all`` routing-intent token, which expands to every platform with
- a configured home channel. Tokens may be combined with explicit
- targets: ``origin,all`` and ``all,telegram:-100:17`` both work.
- Duplicate (platform, chat_id, thread_id) tuples are collapsed by the
- existing dedup pass.
-
- ``for_failure=True`` resolves failure-category engine notices
- (failure summaries, interrupted-run notices, drift/preflight
- alerts): when the job carries a ``failure_deliver`` value, targets
- resolve from it INSTEAD of ``deliver`` — ``failure_deliver: local``
- is the structural opt-out for shared channels (NS-788, Coatue).
- Absent ``failure_deliver``, failure delivery follows ``deliver``
- exactly as before.
- """
- deliver_raw = _delivery_lane_value(job, for_failure=for_failure)
- deliver = _normalize_deliver_value(deliver_raw)
+ """Resolve auto-delivery targets from comma-separated ``deliver``; ``all`` expands to every
+ platform with a home channel and combines with explicit targets. Dedup by (platform, chat_id,
+ thread_id). ``for_failure=True`` (failure summaries, interrupted-run notices, drift/preflight
+ alerts) resolves from ``failure_deliver`` INSTEAD when the job carries one — ``failure_deliver:
+ local`` is the structural opt-out for shared channels; absent, failures follow ``deliver``."""
+ deliver = _normalize_deliver_value(_delivery_lane_value(job, for_failure=for_failure))
if deliver == "local":
return []
raw_parts = [p.strip() for p in deliver.split(",") if p.strip()]
- # Expand routing intents.
parts: List[str] = []
for raw in raw_parts:
parts.extend(_expand_routing_tokens(raw))
@@ -2891,10 +2052,8 @@ def _resolve_delivery_targets(job: dict, *, for_failure: bool = False) -> List[d
seen[key] = target
targets.append(target)
else:
- # OR-merge resolution provenance on dedup: "origin,all" (either
- # order) resolving to the same chat must keep the
- # origin/origin_fallback tag — a mirror-eligible token must not
- # lose eligibility to token order (see _target_mirror_eligible).
+ # OR-merge provenance on dedup: "origin,all" in either order must keep the
+ # origin/origin_fallback tag or mirror eligibility would depend on token order.
kept = seen[key]
if _MIRROR_PROVENANCE_RANK.get(str(target.get("_resolved_from") or ""), 0) > \
_MIRROR_PROVENANCE_RANK.get(str(kept.get("_resolved_from") or ""), 0):
@@ -2908,8 +2067,7 @@ def _resolve_delivery_target(job: dict) -> Optional[dict]:
return targets[0] if targets else None
-# Media extension sets — audio routing is centralized in gateway.platforms.base
-# via should_send_media_as_audio() so Telegram-specific rules stay in one place.
+# Audio routing is centralized in gateway.platforms.base.should_send_media_as_audio().
_VIDEO_EXTS = frozenset({'.mp4', '.mov', '.avi', '.mkv', '.webm', '.3gp'})
_IMAGE_EXTS = frozenset({'.jpg', '.jpeg', '.png', '.webp', '.gif'})
@@ -2923,44 +2081,32 @@ def _send_media_via_adapter(
job: dict,
platform=None,
) -> list:
- """Send extracted MEDIA files as native platform attachments via a live adapter.
-
- Routes each file to the appropriate adapter method (send_voice, send_image_file,
- send_video, send_document) based on file extension — mirroring the routing logic
- in ``BasePlatformAdapter._process_message_background``.
-
- Returns a list of per-file error strings (empty when every attachment
- delivered). Callers surface these into the job's delivery errors so a
- dropped attachment is visible in ``last_error``/run status instead of
- only in the gateway log (the silent-drop half of the manual-run
- attachment bug: text delivered, file vanished, job marked ok).
- """
- from pathlib import Path
-
- from gateway.platforms.base import BasePlatformAdapter, should_send_media_as_audio
+ """Send MEDIA files as native attachments (routed by extension, as in
+ _process_message_background). Returns per-file error strings so a dropped attachment surfaces
+ in run status, not just the gateway log."""
+ from gateway.platforms.base import (
+ BasePlatformAdapter, should_send_media_as_audio, validate_media_delivery_path,
+ )
+ from agent.async_utils import safe_schedule_threadsafe
+ job_ref = {"id": job.get("id", "?")}
errors: list = []
requested = [(str(p), v) for p, v in (media_files or [])]
media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files)
- # Report paths the safety filter dropped: the model referenced them in
- # MEDIA: tags but they will never be sent (missing file, denied prefix,
- # or strict-mode policy miss).
+ # Report paths the safety filter dropped (missing file, denied prefix, strict-mode miss).
kept = {p for p, _ in media_files}
for raw_path, _v in requested:
try:
- from gateway.platforms.base import validate_media_delivery_path
-
- if validate_media_delivery_path(raw_path) not in kept:
- errors.append(
- f"attachment dropped by media path policy: {raw_path}"
- )
+ dropped = validate_media_delivery_path(raw_path) not in kept
except Exception:
+ dropped = True
+ if dropped:
errors.append(f"attachment dropped by media path policy: {raw_path}")
+ route_platform = platform if platform is not None else getattr(adapter, "platform", None)
for media_path, _is_voice in media_files:
try:
ext = Path(media_path).suffix.lower()
- route_platform = platform if platform is not None else getattr(adapter, "platform", None)
if should_send_media_as_audio(route_platform, ext, is_voice=_is_voice):
coro = adapter.send_voice(chat_id=chat_id, audio_path=media_path, metadata=metadata)
elif ext in _VIDEO_EXTS:
@@ -2970,70 +2116,39 @@ def _send_media_via_adapter(
else:
coro = adapter.send_document(chat_id=chat_id, file_path=media_path, metadata=metadata)
- from agent.async_utils import safe_schedule_threadsafe
future = safe_schedule_threadsafe(coro, loop)
if future is None:
- msg = f"cannot send media {media_path}: gateway loop unavailable"
- logger.warning("Job '%s': %s", job.get("id", "?"), msg)
- errors.append(msg)
+ _note_target_error(
+ job_ref, f"cannot send media {media_path}: gateway loop unavailable", errors,
+ )
return errors
try:
- # Large attachments (long TTS audio, concatenated recordings,
- # big exports) can legitimately exceed a fixed 30s upload
- # window. Configurable, matching the other cron timeouts
- # (cron.media_send_timeout_seconds in config.yaml, or the
- # HERMES_CRON_MEDIA_SEND_TIMEOUT env override).
+ # Large attachments can exceed 30s; configurable via _get_media_send_timeout().
result = future.result(timeout=_get_media_send_timeout())
except TimeoutError:
future.cancel()
raise
if result and not getattr(result, "success", True):
- msg = (
- f"media send failed for {media_path}: "
- f"{getattr(result, 'error', 'unknown')}"
+ _note_target_error(
+ job_ref,
+ f"media send failed for {media_path}: {getattr(result, 'error', 'unknown')}",
+ errors,
)
- logger.warning("Job '%s': %s", job.get("id", "?"), msg)
- errors.append(msg)
except Exception as e:
- # Argument-less exceptions (notably TimeoutError, the most likely
- # failure on this path) have an empty str(), which would render
- # the reason as nothing at all. Fall back to the class name.
- msg = (
- f"failed to send media {media_path}: {str(e) or type(e).__name__}"
+ # TimeoutError etc. have an empty str(); fall back to the class name.
+ _note_target_error(
+ job_ref, f"failed to send media {media_path}: {str(e) or type(e).__name__}", errors,
)
- logger.warning("Job '%s': %s", job.get("id", "?"), msg)
- errors.append(msg)
return errors
def _confirm_adapter_delivery(send_result, job_id: str = "?", unverified: Optional[list] = None) -> bool:
"""Return True only if ``send_result`` unambiguously confirms delivery.
- A live adapter that returns ``None`` (e.g. a swallowed exception, a busy
- platform, or a code path that returns early without producing a
- ``SendResult``) must NOT be treated as success — doing so causes the
- scheduler to log ``"delivered to via live adapter"`` while the
- gateway never actually sees the message (#47056).
-
- Likewise, a result carrying no ``success`` at all (a partial mock, or a
- ``dict`` from a code path that never reached the adapter) is a contract
- violation: it does not actually tell us whether the send succeeded.
- Require an explicit, truthy ``success`` to count as confirmed.
-
- Both shapes are inspected the same way, because ``_deliver_to_platform``
- returns either a ``SendResult`` object or a plain ``dict``:
-
- * ``delivered is False`` is a REJECTION even when ``success`` is truthy.
- The silence-narration filter returns
- ``{"success": True, "delivered": False}`` — a successfully *dropped*
- message, not a delivered one. Reading only ``success`` there is how a
- cron brief was logged as delivered while the user got nothing (#77763).
- * No ``message_id`` and no ``raw_response`` means we have no positive
- evidence of a send. That is not proof of failure either (some adapters
- legitimately return a bare success), so it is still accepted — but
- logged at WARNING so an UNVERIFIED delivery is visible in the log
- instead of masquerading as a confirmed one. Telegram ``SendResult``
- objects carry ``message_id``; the dict-filter shape does not.
+ ``None`` or no ``success`` attr/key is NOT success (would log "delivered" while nothing was
+ sent). ``delivered is False`` REJECTS even with truthy ``success``: the silence-narration filter
+ returns ``{"success": True, "delivered": False}`` (dropped). No ``message_id``/``raw_response``
+ is still accepted (some adapters return a bare success) but logged at WARNING as UNVERIFIED.
"""
if send_result is None:
return False
@@ -3071,26 +2186,12 @@ def _is_channel_dm_topic(
loop: Any,
job_id: str,
) -> bool:
- """Decide whether an (already-ambiguous) Telegram topic target is a genuine
- Bot API *channel* Direct-Messages topic (route via
- ``direct_messages_topic_id``) rather than a forum-style topic in a private
- chat (route via ``message_thread_id``).
-
- Callers gate this on the ambiguous shape first
- (``telegram::``) — that shape is
- identical for both cases, so shape alone cannot decide (this was the #52060
- regression). The real signal is the chat *type*: a genuine channel DM topic
- lives on a ``channel`` chat. Probe the live adapter's ``get_chat_info`` once
- and only return True when the chat is a channel.
-
- Fails SAFE to ``message_thread_id`` (returns False) for adapters without a
- probe, or any probe error/timeout — that is the pre-#22773 behaviour and the
- correct default for the common forum-topic case.
- """
- # Resolve on the CLASS, not the instance (general pitfall #11): a MagicMock
- # instance auto-creates a truthy ``get_chat_info`` attribute, so an
- # instance-level probe would misclassify test doubles. Real adapters expose
- # the coroutine on the class regardless.
+ """Is an ambiguous ``telegram::`` target a channel
+ Direct-Messages topic (``direct_messages_topic_id``) rather than a private-chat forum topic
+ (``message_thread_id``)? Shape cannot decide; signal is ``get_chat_info`` type == ``channel``.
+ Fails SAFE to False (thread routing) without a probe or on any probe error/timeout."""
+ # Resolve on the CLASS, not the instance: a MagicMock instance auto-creates a truthy
+ # ``get_chat_info``, so an instance-level probe would misclassify test doubles.
get_chat_info = getattr(type(runtime_adapter), "get_chat_info", None)
if not callable(get_chat_info):
return False
@@ -3102,8 +2203,7 @@ def _is_channel_dm_topic(
)
if future is None:
return False
- # Lighter than a send (metadata-only Bot API call), so a shorter bound
- # than the 30s/60s send waits elsewhere in this file is intentional.
+ # Metadata-only call, so a shorter bound than the send waits is intentional.
info = future.result(timeout=10)
except Exception:
logger.debug(
@@ -3122,12 +2222,8 @@ def _is_channel_dm_topic(
def _cron_delivery_notify_enabled(cfg: Optional[dict]) -> bool:
- """Resolve ``cron.delivery.notify`` (config.yaml). Default True.
-
- Only an explicit boolean ``False`` (or a YAML ``false``/``off`` that parses
- to it) disables the push notification; a missing/malformed section keeps
- the default so a typo can never silently make cron briefs silent.
- """
+ """Resolve ``cron.delivery.notify`` (default True). Only an explicit ``False`` disables; a
+ missing/malformed section keeps the default so a typo cannot silently mute briefs."""
try:
cron_cfg = (cfg or {}).get("cron")
if not isinstance(cron_cfg, dict):
@@ -3141,15 +2237,9 @@ def _cron_delivery_notify_enabled(cfg: Optional[dict]) -> bool:
def _record_delivery_verification(job: dict, unverified_targets: list) -> None:
- """Persist the UNVERIFIED-delivery marker on the job record.
-
- ``last_delivery_unverified`` is a list of ``platform:chat_id`` targets
- whose live adapter acked the send with no message_id/raw_response, or
- ``None`` once a run delivered with positive evidence (or to no live
- target). Skips the write when nothing changed so the common verified
- path costs no jobs.json save. Never raises — status bookkeeping must not
- fail a delivery.
- """
+ """Persist ``last_delivery_unverified``: list of ``platform:chat_id`` targets acked with no
+ evidence, or None. Skips the write when unchanged; never raises (bookkeeping must not fail a
+ delivery)."""
new_value = list(unverified_targets) or None
if (job.get("last_delivery_unverified") or None) == new_value:
return
@@ -3158,27 +2248,513 @@ def _record_delivery_verification(job: dict, unverified_targets: list) -> None:
update_job(job["id"], {"last_delivery_unverified": new_value})
except Exception as exc: # pragma: no cover - defensive
- logger.debug(
- "Job '%s': could not record delivery verification: %s", job.get("id"), exc,
+ logger.debug("Job '%s': could not record delivery verification: %s", job.get("id"), exc)
+
+
+@dataclass
+class _TargetDelivery:
+ """Per-target delivery state shared by the live-adapter and standalone lanes."""
+
+ job: dict
+ platform: Any
+ platform_name: str
+ chat_id: str
+ thread_id: Optional[str]
+ transport: Any
+ pconfig: Any
+ runtime_adapter: Any
+ target_adapters: Any
+ config: Any
+ loop: Any
+ notify_delivery: bool
+ origin: dict
+ origin_target: bool
+ origin_user_id: Optional[str]
+ is_dm_target: bool
+ mirror_text: str
+ mirror_this_target: bool
+ in_channel_surface: bool
+ inchannel_continuable: bool
+ opened_thread_id: Optional[str]
+
+ @property
+ def is_relay(self) -> bool:
+ return self.transport is not None and self.transport.is_relay
+
+ @property
+ def where(self) -> str:
+ return f"{self.platform_name}:{self.chat_id}"
+
+
+def _note_target_error(job: dict, msg: str, errors: list) -> None:
+ """Log a per-target delivery failure as a WARNING and record it in ``errors``."""
+ logger.warning("Job '%s': %s", job["id"], msg)
+ errors.append(msg)
+
+
+def _warn_live_lane_failure(job: dict, msg: str, is_relay: bool) -> None:
+ """Relay targets have no standalone fallback, so the log line must not promise one."""
+ if is_relay:
+ logger.warning("Job '%s': %s", job["id"], msg)
+ else:
+ logger.warning("Job '%s': %s, falling back to standalone", job["id"], msg)
+
+
+def _resolve_target_transport(job: dict, platform, platform_name: str, target: dict, adapters, config):
+ """Resolve ``(transport, pconfig, runtime_adapter, target_adapters)`` for one target, or
+ ``(None, error)`` when it cannot be served (relay-fronted with no live transport, or not
+ configured/enabled)."""
+ from gateway.delivery import resolve_delivery_transport
+
+ target_adapters = adapters
+ if isinstance(adapters, SharedRouteAdapters):
+ # Credentialless satellite: the primary adapter serves THIS target only when an exact
+ # primary route maps it to this profile; a miss fails closed below.
+ shared = adapters.get(platform, target)
+ target_adapters = {platform: shared} if shared is not None else {}
+ transport = resolve_delivery_transport(platform, config, target_adapters)
+ if transport is not None:
+ pconfig = transport.config
+ runtime_adapter = transport.adapter
+ else:
+ # Relay-fronted platforms have NO standalone fallback (the connector owns the credential),
+ # so surface that instead of the native configured/enabled gate, which misdiagnoses them.
+ from gateway.relay import relay_fronted_platforms
+
+ if platform_name in relay_fronted_platforms():
+ return None, (
+ f"platform '{platform_name}' is relay-fronted and has no "
+ "live gateway transport; start the gateway (its ticker "
+ "owns relay-fronted delivery and will fire the job on "
+ "schedule)"
+ )
+ pconfig = config.platforms.get(platform)
+ runtime_adapter = None
+
+ if transport is not None and transport.is_relay:
+ # Relay transport carries the RELAY adapter's config (enablement already checked). The
+ # logical platform is deliberately NOT natively enabled, so the native gate must not apply.
+ if pconfig is None:
+ from gateway.config import PlatformConfig
+ pconfig = PlatformConfig(enabled=True)
+ elif not pconfig or not pconfig.enabled:
+ return None, f"platform '{platform_name}' not configured/enabled"
+ return (transport, pconfig, runtime_adapter, target_adapters), None
+
+
+def _inchannel_surface_supported(runtime_adapter, platform_name: str) -> bool:
+ """D6 probe: can this adapter deliver a continuable in_channel brief on ``platform_name``?
+ Per-platform check first (one RelayAdapter fronts N platforms; the scalar attr only carries
+ the PRIMARY identity's bit); native adapters use the class attribute."""
+ per_platform_check = getattr(runtime_adapter, "supports_inchannel_continuable_for_platform", None)
+ if callable(per_platform_check):
+ try:
+ return bool(per_platform_check(platform_name))
+ except Exception:
+ return False
+ return bool(getattr(runtime_adapter, "supports_inchannel_continuable", False))
+
+
+def _live_route_metadata(t: _TargetDelivery) -> tuple[Optional[str], dict, dict]:
+ """Compute ``(route_thread_id, route_metadata, media_metadata)`` for a live send, ONCE so text
+ and media agree. ``telegram::`` is ambiguous (private
+ forum topic vs channel DM topic need OPPOSITE routing) — see ``_is_channel_dm_topic``.
+ ``thread_id`` rides in ``route_metadata`` to bypass the DeliveryRouter's private-chat
+ reply-anchor requirement for anchorless cron sends."""
+ from gateway.config import Platform
+ from gateway.delivery import _looks_like_int, looks_like_telegram_private_chat_id
+
+ job = t.job
+ thread_id = t.thread_id
+ is_ambiguous_telegram_topic = (
+ t.platform == Platform.TELEGRAM
+ and thread_id is not None
+ and looks_like_telegram_private_chat_id(str(t.chat_id))
+ and _looks_like_int(str(thread_id))
+ )
+ if is_ambiguous_telegram_topic and _is_channel_dm_topic(
+ t.runtime_adapter, t.chat_id, t.loop, job["id"],
+ ):
+ # Channel DM topic: direct_messages_topic_id, no bare thread_id; media mirrors text.
+ route_thread_id = None
+ route_metadata = {
+ "direct_messages_topic_id": str(thread_id),
+ "job_id": job["id"],
+ "notify": t.notify_delivery,
+ }
+ media_metadata = {"direct_messages_topic_id": str(thread_id), "notify": t.notify_delivery}
+ else:
+ # Forum-style topic or non-topic target: message_thread_id.
+ route_thread_id = str(thread_id) if thread_id is not None else None
+ route_metadata = {"job_id": job["id"], "notify": t.notify_delivery}
+ if route_thread_id:
+ route_metadata["thread_id"] = route_thread_id
+ media_metadata = {"notify": t.notify_delivery}
+ if thread_id:
+ media_metadata["thread_id"] = thread_id
+
+ # Relay egress needs metadata.scope_id (fail-closed tenant guard; scope cache is COLD after a
+ # restart; router stamps HOME only). Origin targets only: a wrong fan-out scope is worse than
+ # none.
+ if t.origin_target and t.origin.get("scope_id"):
+ route_metadata.setdefault("scope_id", str(t.origin["scope_id"]))
+ media_metadata.setdefault("scope_id", str(t.origin["scope_id"]))
+ return route_thread_id, route_metadata, media_metadata
+
+
+def _live_send_text(
+ t: _TargetDelivery,
+ text_to_send: str,
+ route_thread_id: Optional[str],
+ route_metadata: dict,
+ *,
+ target_errors: list,
+ delivery_errors: list,
+ unverified_targets: list,
+) -> tuple[bool, bool, Any]:
+ """Schedule the text send on the gateway loop; returns ``(adapter_ok, timed_out, message_id)``.
+ Re-raises a real send error so the caller falls through to standalone."""
+ from agent.async_utils import safe_schedule_threadsafe
+ from gateway.delivery import DeliveryRouter, DeliveryTarget
+
+ job = t.job
+ router = DeliveryRouter(t.config, t.target_adapters)
+ route_target = DeliveryTarget(
+ platform=t.platform,
+ chat_id=str(t.chat_id),
+ thread_id=route_thread_id,
+ is_explicit=True,
+ )
+ # Thread routing goes via the target, not a bare metadata "thread_id": the router only applies
+ # its Telegram DM-topic detection when thread_id/message_thread_id are absent from metadata.
+ future = safe_schedule_threadsafe(
+ router._deliver_to_platform(route_target, text_to_send, route_metadata),
+ t.loop,
+ )
+ if future is None:
+ target_errors.append("live adapter event loop scheduling failed")
+ return False, False, None
+ try:
+ send_result = future.result(timeout=60)
+ except TimeoutError:
+ # Slow confirmation != failure; future.cancel() disambiguates. False -> already in flight,
+ # cannot be un-sent, standalone resend would DUPLICATE: assume delivered. True -> never
+ # started (loop wedged): MUST fall through to standalone or it is silently dropped.
+ if future.cancel():
+ msg = (
+ f"live adapter send to {t.where} "
+ "timed out before the coroutine was dispatched"
+ )
+ logger.warning("Job '%s': %s, falling back to standalone", job["id"], msg)
+ target_errors.append(msg)
+ return False, False, None
+ logger.warning(
+ "Job '%s': live adapter send to %s:%s timed out "
+ "after 60s; already dispatched (in flight), "
+ "assuming delivered (skipping standalone fallback "
+ "to avoid duplicate)",
+ job["id"], t.platform_name, t.chat_id,
)
+ return True, True, None
+ except Exception as ex:
+ # Real send error (not a slow confirmation): fall through to standalone.
+ target_errors.append(f"live adapter send failed: {ex}")
+ raise
+
+ # _deliver_to_platform returns a SendResult, or a plain dict {"success": True, "delivered":
+ # False, ...} when the silence-narration filter drops the message.
+ if isinstance(send_result, dict):
+ send_raw_response = send_result.get("raw_response")
+ delivered_message_id = send_result.get("message_id")
+ else:
+ send_raw_response = getattr(send_result, "raw_response", None)
+ delivered_message_id = getattr(send_result, "message_id", None)
+ _evidence_gap: list = []
+ send_success = _confirm_adapter_delivery(send_result, job["id"], _evidence_gap)
+ if send_success and _evidence_gap:
+ unverified_targets.append(t.where)
+
+ if not send_success:
+ if isinstance(send_result, dict):
+ # A filtered drop carries no "error" — name the filter instead of reporting "unknown".
+ err = send_result.get("error") or send_result.get("filtered") or "unknown"
+ shape = "dict"
+ elif send_result is not None:
+ err = getattr(send_result, "error", None)
+ shape = type(send_result).__name__
+ else:
+ err = "no response from adapter"
+ shape = "None"
+ msg = f"live adapter send to {t.where} returned unconfirmed result ({shape}, error={err})"
+ _warn_live_lane_failure(job, msg, t.is_relay)
+ target_errors.append(msg)
+ return False, False, None
+ if send_raw_response and t.thread_id and send_raw_response.get("thread_fallback"):
+ requested_thread_id = send_raw_response.get("requested_thread_id") or t.thread_id
+ _note_target_error(
+ job,
+ f"configured thread_id {requested_thread_id} for "
+ f"{t.where} was not found; delivered without thread_id",
+ delivery_errors,
+ )
+ return True, False, delivered_message_id
+
+
+def _live_send_media(t: _TargetDelivery, media_metadata: dict, media_files: list, delivery_errors: list) -> None:
+ """Send extracted media as native attachments with the same routing as the text send."""
+ routed_media_metadata = dict(media_metadata or {})
+ if t.is_relay:
+ routed_media_metadata["_relay_logical_platform"] = t.platform.value
+ logical_home = t.config.get_home_channel(t.platform)
+ if logical_home is not None and logical_home.chat_id == t.chat_id:
+ if logical_home.user_id:
+ routed_media_metadata["user_id"] = logical_home.user_id
+ if logical_home.scope_id:
+ routed_media_metadata["scope_id"] = logical_home.scope_id
+ _media_errors = _send_media_via_adapter(
+ t.runtime_adapter,
+ t.chat_id,
+ media_files,
+ routed_media_metadata or None,
+ t.loop,
+ t.job,
+ platform=t.platform,
+ )
+ # Surface per-file failures into run status: text delivered but attachment lost is not ok.
+ for _me in _media_errors:
+ delivery_errors.append(f"{_me} (target {t.where})")
+
+
+def _seed_live_delivery_sessions(t: _TargetDelivery, delivered_message_id) -> None:
+ """After a confirmed live send, seed continuation session(s) and run the generic mirror.
+ Thread seeding is deferred here so open-succeeds/deliver-fails never seeds an unseen brief."""
+ job = t.job
+ origin = t.origin
+ thread_seeded = False
+ inchannel_seeded = False
+ if t.opened_thread_id:
+ _seed_cron_thread_session(
+ job, t.runtime_adapter, t.platform_name, t.chat_id,
+ t.opened_thread_id, t.mirror_text,
+ chat_name=origin.get("chat_name"),
+ is_dm=t.is_dm_target,
+ scope_id=origin.get("scope_id"),
+ )
+ thread_seeded = True
+ # in_channel: CREATE + seed the flat session (the mirror only APPENDS to an existing one). Same
+ # `inchannel_continuable` gate as the flatten in _deliver_result (must not drift). Origin
+ # seed without mirror opt-in; others only via _inchannel_seed_allowed (user-less seed = orphan).
+ if t.in_channel_surface and t.inchannel_continuable and not thread_seeded:
+ inchannel_seeded = _seed_cron_channel_session(
+ job, t.runtime_adapter, t.platform_name, t.chat_id,
+ t.mirror_text, is_dm=t.is_dm_target,
+ user_id=t.origin_user_id,
+ chat_name=origin.get("chat_name"),
+ scope_id=origin.get("scope_id"),
+ )
+ if not inchannel_seeded:
+ logger.warning(
+ "Job '%s': in_channel seed did NOT land on %s:%s "
+ "— a plain reply will not see this brief",
+ job["id"], t.platform_name, t.chat_id,
+ )
+ # Companion THREAD seed: a reply in the brief's own thread keys to (chat, thread=),
+ # which the flat seed never touches. Seed it too so BOTH reply surfaces continue the job.
+ if delivered_message_id:
+ _seed_cron_thread_session(
+ job, t.runtime_adapter, t.platform_name, t.chat_id,
+ str(delivered_message_id), t.mirror_text,
+ chat_name=origin.get("chat_name"),
+ is_dm=t.is_dm_target,
+ scope_id=origin.get("scope_id"),
+ )
+ elif t.in_channel_surface and not t.inchannel_continuable:
+ logger.warning(
+ "Job '%s': in_channel delivery to %s:%s is not a "
+ "continuable target (origin=%s:%s thread=%s; not the "
+ "origin conversation, and not a mirror-eligible "
+ "fallback/opted-in target the seed can key) — seed "
+ "skipped; the plain mirror below may still apply",
+ job["id"], t.platform_name, t.chat_id,
+ origin.get("platform"), origin.get("chat_id"),
+ origin.get("thread_id"),
+ )
+ _maybe_mirror_cron_delivery(
+ job, t.platform_name, t.chat_id, t.mirror_text,
+ thread_id=t.thread_id, user_id=t.origin_user_id,
+ enabled=t.mirror_this_target and not thread_seeded and not inchannel_seeded,
+ )
+
+
+def _deliver_via_live_adapter(
+ t: _TargetDelivery,
+ cleaned_text: str,
+ media_files: list,
+ *,
+ target_errors: list,
+ delivery_errors: list,
+ unverified_targets: list,
+) -> bool:
+ """Deliver one target via the live gateway adapter; True once delivered. ``target_errors`` =
+ this lane's soft failures (surfaced only if standalone also fails); ``delivery_errors`` =
+ partial failures (media, thread fallback) that surface even on success."""
+ job = t.job
+ route_thread_id, route_metadata, media_metadata = _live_route_metadata(t)
+ delivered = False
+ try:
+ # Send cleaned text (MEDIA tags stripped) through the gateway's DeliveryRouter so it gets
+ # the same platform routing as live messages (Telegram's three-mode topic routing).
+ text_to_send = cleaned_text.strip()
+ adapter_ok = True
+ timed_out = False
+ delivered_message_id = None
+ if not text_to_send and not media_files:
+ # Fail closed so the run reports the empty payload.
+ _note_target_error(
+ job, f"live adapter send skipped (empty text and no media) for {t.where}", target_errors,
+ )
+ adapter_ok = False
+ elif text_to_send:
+ adapter_ok, timed_out, delivered_message_id = _live_send_text(
+ t, text_to_send, route_thread_id, route_metadata,
+ target_errors=target_errors,
+ delivery_errors=delivery_errors,
+ unverified_targets=unverified_targets,
+ )
+
+ # Media rides the same DM-topic-aware routing as text. Skipped after a confirmation
+ # timeout (loop contended, text already assumed delivered) — record the drop instead.
+ if adapter_ok and not timed_out and media_files:
+ _live_send_media(t, media_metadata, media_files, delivery_errors)
+ elif timed_out and media_files:
+ _note_target_error(
+ job,
+ f"{len(media_files)} media attachment(s) not delivered to "
+ f"{t.where} (live adapter confirmation timed out)",
+ delivery_errors,
+ )
+
+ if adapter_ok:
+ # Log WHERE it went: a ghost delivery in the wrong lane is otherwise indistinguishable.
+ logger.info(
+ "Job '%s': delivered to %s:%s via live adapter thread=%s message_id=%s",
+ job["id"], t.platform_name, t.chat_id,
+ route_thread_id if route_thread_id is not None else "-",
+ delivered_message_id if delivered_message_id is not None else "-",
+ )
+ delivered = True
+ _seed_live_delivery_sessions(t, delivered_message_id)
+ except Exception as e:
+ err_msg = f"live adapter delivery to {t.where} failed: {e}"
+ if not any(err_msg in err for err in target_errors):
+ target_errors.append(err_msg)
+ _warn_live_lane_failure(job, err_msg, t.is_relay)
+ return delivered
+
+
+def _standalone_send(t: _TargetDelivery, content: str, media_files: list) -> tuple[Any, Optional[str]]:
+ """Run the standalone sender for one target: ``(result, None)`` or ``(None, error)`` (already
+ logged — WARNING for a shutdown race, ERROR with traceback otherwise)."""
+ from tools.send_message_tool import _send_to_platform
+
+ job = t.job
+ shutdown_msg = f"delivery to {t.where} skipped — interpreter is shutting down"
+
+ def _send():
+ return _send_to_platform(
+ t.platform, t.pconfig, t.chat_id, content, thread_id=t.thread_id, media_files=media_files,
+ )
+
+ def _failed(e) -> tuple[None, str]:
+ msg = f"delivery to {t.where} failed: {e}"
+ logger.error("Job '%s': %s", job["id"], msg, exc_info=True)
+ return None, msg
+
+ # Interpreter finalizing (SIGTERM/restart/OOM): asyncio.run and a fresh ThreadPoolExecutor both
+ # raise "cannot schedule new futures after interpreter shutdown" — warn, not ERROR traceback.
+ if _interpreter_shutting_down():
+ logger.warning("Job '%s': %s", job["id"], shutdown_msg)
+ return None, shutdown_msg
+ # The live lane failed closed on an empty payload; standalone senders don't (Telegram returns
+ # success=True for empty content WITHOUT an API call) — a phantom delivery would result.
+ if not content.strip() and not media_files:
+ msg = f"standalone send skipped (empty text and no media) for {t.where}"
+ logger.warning("Job '%s': %s", job["id"], msg)
+ return None, msg
+ coro = _send()
+ try:
+ return asyncio.run(coro), None
+ except RuntimeError as run_err:
+ # asyncio.run() refuses inside a running loop; close the unstarted coro, retry in a thread.
+ coro.close()
+ if _interpreter_shutting_down(run_err):
+ logger.warning("Job '%s': %s", job["id"], shutdown_msg)
+ return None, shutdown_msg
+ # The fallback can itself raise (SMTP, result timeout); catch it or remaining targets skip.
+ try:
+ pool = concurrent.futures.ThreadPoolExecutor(max_workers=1)
+ try:
+ # A fresh thread does NOT inherit the profile ContextVars (home override + secret
+ # scope); run in the active context or the sender reads the default bot token.
+ _fallback_context = contextvars.copy_context()
+ future = pool.submit(_fallback_context.run, asyncio.run, _send())
+ return future.result(timeout=30), None
+ finally:
+ pool.shutdown(wait=False)
+ except Exception as e:
+ if _interpreter_shutting_down(e):
+ logger.warning("Job '%s': %s", job["id"], shutdown_msg)
+ return None, shutdown_msg
+ return _failed(e)
+ except Exception as e:
+ return _failed(e)
+
+
+def _deliver_standalone(
+ t: _TargetDelivery, content: str, media_files: list, target_errors: list, delivery_errors: list,
+) -> None:
+ """Standalone fallback for a target the live lane did not deliver."""
+ job = t.job
+ if t.is_relay:
+ # Relay owns the destination and credential; a native retry could duplicate — fail closed.
+ if not target_errors:
+ target_errors.append(f"relay delivery to {t.where} failed")
+ delivery_errors.extend(target_errors)
+ return
+ result, err = _standalone_send(t, content, media_files)
+ if err is None and result and result.get("error"):
+ # Not inside an except block — the error comes from the result dict, no traceback.
+ err = f"delivery error: {result['error']} (target {t.where})"
+ logger.error("Job '%s': %s", job["id"], err)
+ if err is not None:
+ target_errors.append(err)
+ delivery_errors.extend(target_errors)
+ return
+
+ # Standalone senders report per-file attachment failures in ``warnings`` while returning
+ # success; surface them so a vanished attachment doesn't mark the run ok.
+ _sender_warnings = (result.get("warnings") if isinstance(result, dict) else None) or []
+ for _w in _sender_warnings:
+ msg = f"delivery warning: {_w} (target {t.where})"
+ logger.error("Job '%s': %s", job["id"], msg)
+ delivery_errors.append(msg)
+
+ logger.info("Job '%s': delivered to %s:%s", job["id"], t.platform_name, t.chat_id)
+ # Thread seeding only happens on the live lane, so no thread_seeded gate applies here.
+ _maybe_mirror_cron_delivery(
+ job, t.platform_name, t.chat_id, t.mirror_text,
+ thread_id=t.thread_id, user_id=t.origin_user_id,
+ enabled=t.mirror_this_target,
+ )
def _deliver_result(
job: dict, content: str, adapters=None, loop=None, *, for_failure: bool = False
) -> Optional[str]:
- """
- Deliver job output to the configured target(s) (origin chat, specific platform, etc.).
-
- When ``adapters`` and ``loop`` are provided (gateway is running), tries to
- use the live adapter first — this supports E2EE rooms (e.g. Matrix) where
- the standalone HTTP path cannot encrypt. Falls back to standalone send if
- the adapter path fails or is unavailable.
-
- ``for_failure=True`` routes failure-category engine notices through the
- job's ``failure_deliver`` override when present (NS-788).
-
- Returns None on success, or an error string on failure.
- """
+ """Deliver job output to the configured target(s). With ``adapters``/``loop`` (gateway
+ running) the live adapter is tried first (E2EE rooms can't use the standalone HTTP path), then
+ standalone fallback. ``for_failure=True`` routes failure-category notices through the job's
+ ``failure_deliver`` override when present (NS-788). Returns None on success or an error string."""
targets = _resolve_delivery_targets(job, for_failure=for_failure)
if not targets:
deliver_value = _normalize_deliver_value(
@@ -3186,12 +2762,8 @@ def _deliver_result(
)
if deliver_value == "local":
return None # local-only jobs don't deliver — not a failure
- # deliver=origin with no resolvable origin and no configured home
- # channels: treat as local rather than reporting an error. CLI-created
- # jobs never capture a {platform, chat_id} origin, so failing here would
- # make every CLI `deliver=origin` (or auto-detect) job emit a spurious
- # "no delivery target resolved" error on every run (#43014). The output
- # is still persisted in last_output for `cron list`/resume.
+ # deliver=origin with no origin and no home channels: treat as local, not an error — CLI
+ # jobs never capture an origin and would emit a spurious error every run.
if deliver_value == "origin":
logger.info(
"Job '%s': deliver=origin but no origin or home channels — "
@@ -3203,30 +2775,19 @@ def _deliver_result(
logger.warning("Job '%s': %s", job["id"], msg)
return msg
- from tools.send_message_tool import _send_to_platform
from gateway.config import load_gateway_config, Platform
- # Optionally wrap the content with a header/footer so the user knows this
- # is a cron delivery. Wrapping is on by default; set cron.wrap_response: false
- # in config.yaml for clean output.
+ # Wrap with header/footer unless cron.wrap_response: false.
wrap_response = True
user_cfg = None
- try:
+ with contextlib.suppress(Exception):
user_cfg = load_config()
wrap_response = user_cfg.get("cron", {}).get("wrap_response", True)
- except Exception:
- pass
- # cron.delivery.notify (default True): mark live-adapter cron sends as
- # FINAL notifications so the platform pushes them (Telegram's "important"
- # mode otherwise sends with disable_notification=True). Configurable so
- # operators who prefer silent briefs can opt back out.
+ # Mark live sends FINAL so the platform pushes them (Telegram "important" mode mutes otherwise).
notify_delivery = _cron_delivery_notify_enabled(user_cfg)
- # Set when a live adapter acked a send with NO delivery evidence (no
- # message_id / raw_response — the Slack/Matrix/Mattermost bare
- # SendResult(success=True) shape). Persisted on the job as
- # ``last_delivery_unverified`` so `hermes cron list` shows the state
- # instead of it living only in a WARNING log line.
+ # Targets acked with NO evidence (bare SendResult(success=True) — Slack/Matrix/Mattermost);
+ # persisted as ``last_delivery_unverified`` so `hermes cron list` shows it.
unverified_targets: list = []
if wrap_response:
@@ -3242,16 +2803,10 @@ def _deliver_result(
else:
delivery_content = content
- # Extract MEDIA: tags so attachments are forwarded as files, not raw text
from gateway.platforms.base import BasePlatformAdapter
- # Bridge gateway media-policy config (strict / allow_dirs / trust_recent)
- # into the env vars the path validator reads. Gateway startup does this
- # at boot; a standalone process (manual `hermes cron run` from the CLI,
- # a cron tick without the gateway) historically did NOT — so manual runs
- # filtered attachment paths under a DIFFERENT policy than scheduled runs
- # and silently dropped files the gateway would deliver. Idempotent,
- # env-wins, never raises.
+ # Bridge media-policy config into the env vars the path validator reads. The gateway does this
+ # at boot; standalone runs (`hermes cron run`) did not, silently dropping files. Idempotent.
from gateway.media_policy import apply_media_policy_env
apply_media_policy_env(user_cfg)
@@ -3259,9 +2814,7 @@ def _deliver_result(
media_files, cleaned_delivery_content = BasePlatformAdapter.extract_media(delivery_content)
requested_media = [(str(p), v) for p, v in media_files]
media_files = BasePlatformAdapter.filter_media_delivery_paths(media_files)
- # Attachments the policy filter dropped will never be sent on ANY lane —
- # record them up front so the run status says so (previously one
- # stderr WARNING was the only trace: text delivered, file vanished).
+ # Policy-dropped attachments will never be sent on ANY lane — record them in run status.
_policy_dropped = len(requested_media) - len(media_files)
policy_drop_errors = (
[
@@ -3273,20 +2826,14 @@ def _deliver_result(
else []
)
- # Resolve the delivery-mirror gate ONCE (default off). When on, each
- # successful delivery is also appended to the target chat's gateway session
- # transcript so a user reply in that chat sees the cron output in context.
- # Mirror the CLEAN, unwrapped output (not the cron header/footer).
+ # Resolve the mirror gate ONCE (default off): successful deliveries are appended to the target
+ # chat's session transcript. Mirror the CLEAN, unwrapped output (not the header/footer).
try:
mirror_enabled = _cron_mirror_delivery_enabled(job, user_cfg)
except Exception:
mirror_enabled = False
- # Keep the cleaned delivery text available independently of the optional
- # transcript-mirror knob. Continuable surfaces (notably in_channel) must
- # seed their target session even when attach_to_session=false and
- # cron.mirror_delivery=false; gating this value on mirror_enabled makes
- # the seed receive an empty string and return False, which is exactly the
- # live failure reproduced three times on Alice (job ef7bd2869d15).
+ # Independent of the mirror knob: continuable surfaces (in_channel) must seed even when
+ # attach_to_session=false and cron.mirror_delivery=false, else the seed gets "" and fails.
_, mirror_text = BasePlatformAdapter.extract_media(content)
mirror_text = (mirror_text or "").strip()
@@ -3304,18 +2851,14 @@ def _deliver_result(
chat_id = target["chat_id"]
thread_id = target.get("thread_id")
- # bot-chat targets don't ride a gateway adapter: the output becomes a
- # real inbound turn in the target profile's canonical Bot Chat via the
- # chat CLI lane (the same one Bot Mode agent-to-agent sends use). The
- # bot runs a turn and can respond — handled before the Platform enum
- # below, which knows nothing about this pseudo-platform.
+ # bot-chat targets bypass gateway adapters: output becomes an inbound turn in the target
+ # profile's Bot Chat via the chat CLI lane. Must precede the Platform enum, which lacks it.
if platform_name == BOT_CHAT_PLATFORM:
bot_chat_error = _deliver_to_bot_chat(job, content, chat_id)
if bot_chat_error:
delivery_errors.append(bot_chat_error)
continue
- # Diagnostic: log thread_id for topic-aware delivery debugging
origin = _resolve_origin(job) or {}
origin_thread = origin.get("thread_id")
if origin_thread and not thread_id:
@@ -3330,233 +2873,77 @@ def _deliver_result(
job["id"], platform_name, chat_id, thread_id,
)
- # Mirror scope: the origin conversation, the home-channel FALLBACK for
- # an origin-less deliver=origin job (a script-provisioned managed cron
- # standing in for the user's primary conversation — not a broadcast),
- # or an explicit target the job opted into via attach_to_session.
- # Broadcast/fan-out targets are never mirrored (_target_mirror_eligible).
+ # Mirror: origin, home FALLBACK for origin-less deliver=origin, or attach_to_session opt-in.
origin_target = _target_matches_origin(origin, platform_name, chat_id, thread_id)
mirror_this_target = mirror_enabled and _target_mirror_eligible(
job, target, global_mirror=mirror_enabled, origin_match=origin_target,
)
- # Pass the origin's user_id so a per-user-isolated group chat resolves to
- # the exact member who scheduled the job — parity with send_message.
- # Resolved for ANY origin-matching target (not just mirror-enabled):
- # the in_channel seed below needs it too, and it must not depend on
- # the attach_to_session/mirror opt-in.
+ # Resolved for ANY origin match (not just mirror-enabled): the in_channel seed needs it too.
origin_user_id = origin.get("user_id") if origin_target else None
- # DM shape of this target, needed by BOTH the in_channel flatten gate
- # below and the seed/_seed_cron_channel_session chat_type further down:
- # a 1:1 DM keys as ``dm`` (Slack DM channel ids start with "D"; or the
- # origin says so), everything else as ``group``.
+ # DM shape for BOTH the flatten gate and seed chat_type (Slack DM ids start with "D").
origin_chat_type = str(origin.get("chat_type") or "").lower()
is_dm_target = origin_chat_type == "dm" or (
not origin_chat_type and str(chat_id).startswith("D")
)
- # Shared continuable-target gate for the in_channel surface. The
- # thread-flatten and the flat-session seed MUST use the SAME gate —
- # if they drift, the brief and its continuation session land in
- # different places (the split-surface bug the flatten exists to
- # prevent). Origin targets qualify unconditionally (independent of the
- # attach_to_session / mirror opt-in — see 3c52d3589f); non-origin
- # mirror-eligible targets (origin_fallback / opted-in explicit)
- # qualify only when the seed can actually create a resolvable session
- # (_inchannel_seed_allowed: DM-shaped, or a known user_id for
- # user-isolated group keys).
+ # in_channel gate shared by thread-flatten and flat seed — they MUST match or brief and
+ # session land in different places. Origin qualifies unconditionally; others only when the
+ # seed can create a resolvable session (_inchannel_seed_allowed).
inchannel_continuable = origin_target or (
mirror_this_target
and _inchannel_seed_allowed(is_dm=is_dm_target, user_id=origin_user_id)
)
- # Built-in names resolve to their enum member; plugin platform names
- # create dynamic members via Platform._missing_().
+ # Plugin platform names create dynamic members via Platform._missing_().
try:
platform = Platform(platform_name.lower())
except (ValueError, KeyError):
- msg = f"unknown platform '{platform_name}'"
- logger.warning("Job '%s': %s", job["id"], msg)
- delivery_errors.append(msg)
+ _note_target_error(job, f"unknown platform '{platform_name}'", delivery_errors)
continue
- from gateway.delivery import resolve_delivery_transport
-
- target_adapters = adapters
- if isinstance(adapters, SharedRouteAdapters):
- # Credentialless satellite: the primary adapter is a valid
- # transport for THIS target only when an exact primary route maps
- # it to this profile (#101113). Miss → fail closed below.
- shared = adapters.get(platform, target)
- target_adapters = {platform: shared} if shared is not None else {}
- transport = resolve_delivery_transport(platform, config, target_adapters)
- if transport is not None:
- pconfig = transport.config
- runtime_adapter = transport.adapter
- else:
- # No live transport. A relay-fronted platform's ONLY sender is the
- # gateway's live relay adapter — there is no standalone fallback
- # (the connector owns the credential). A manual in-process run
- # (`hermes cron run`) has no live relay adapter, so surface the
- # accurate remediation instead of the native configured/enabled
- # gate, which misdiagnoses relay-fronted deployments.
- from gateway.relay import relay_fronted_platforms
-
- if platform_name in relay_fronted_platforms():
- msg = (
- f"platform '{platform_name}' is relay-fronted and has no "
- "live gateway transport; start the gateway (its ticker "
- "owns relay-fronted delivery and will fire the job on "
- "schedule)"
- )
- logger.warning("Job '%s': %s", job["id"], msg)
- delivery_errors.append(msg)
- continue
- # Preserve the existing standalone delivery path, which uses the
- # logical platform's configured credential.
- pconfig = config.platforms.get(platform)
- runtime_adapter = None
-
- if transport is not None and transport.is_relay:
- # A relay transport carries the RELAY adapter's config, and
- # resolve_delivery_transport already applied relay's enablement
- # rule (config block absent OR enabled). The logical platform is
- # deliberately NOT natively enabled in a relay-fronted deployment
- # (its credential lives in the connector), so the native
- # configured/enabled gate below must not apply — it used to
- # reject exactly the targets the relay was resolved to serve.
- if pconfig is None:
- from gateway.config import PlatformConfig
- pconfig = PlatformConfig(enabled=True)
- elif not pconfig or not pconfig.enabled:
- msg = f"platform '{platform_name}' not configured/enabled"
- logger.warning("Job '%s': %s", job["id"], msg)
- delivery_errors.append(msg)
+ resolved, resolve_err = _resolve_target_transport(
+ job, platform, platform_name, target, adapters, config,
+ )
+ if resolved is None:
+ _note_target_error(job, resolve_err, delivery_errors)
continue
+ transport, pconfig, runtime_adapter, target_adapters = resolved
- # Prefer the resolved live transport when the gateway is running. This
- # supports E2EE native adapters and relay-fronted logical platforms.
- # The live-send path (which SEEDS the flat in_channel continuation
- # session via _seed_cron_channel_session) needs not just a live adapter
- # but a running event loop to schedule the async send onto. Compute that
- # gate ONCE so the in_channel thread_id clear below stays in lockstep
- # with the live-send/seed block further down (they used to drift): an
- # adapter can be present while the loop is absent/not-running, in which
- # case the live-send block is skipped and delivery falls through to the
- # standalone path — which cannot seed the flat session (r3609147550).
+ # Live send needs a RUNNING loop, not just an adapter. Computed ONCE so the in_channel
+ # thread_id clear below stays in lockstep with the seed (standalone cannot seed flat).
live_adapter_ready = (
runtime_adapter is not None
and loop is not None
and getattr(loop, "is_running", lambda: False)()
)
- delivered = False
- target_errors = []
+ target_errors: list = []
- # Continuable cron surface (D1/D2/D6): resolve the delivery surface for
- # this platform generically from its config ``extra``. Default "thread"
- # (today's behaviour, byte-identical). "in_channel" delivers the brief
- # FLAT into the channel (no dedicated thread) so a plain channel reply
- # continues the job in-context via the shared-channel session
- # ``(platform, chat_id, None)`` — the same bucket ``reply_in_thread:
- # false`` routes inbound channel messages to. The key is read
- # generically here (any platform); the ``in_channel`` branch is gated on
- # the adapter capability flag ``supports_inchannel_continuable`` so an
- # unsupported platform fails SAFE to "thread" (Slack is the first
- # consumer; "first consumer ≠ definition").
- surface_mode = _resolve_cron_surface_mode(pconfig, platform_name)
- in_channel_surface = surface_mode == "in_channel"
- if in_channel_surface and runtime_adapter is not None:
- # Per-platform capability first: one RelayAdapter fronts N
- # platforms and the connector advertises the bit per platform at
- # handshake — the scalar attr only carries the PRIMARY identity's
- # bit. Native adapters (no per-platform query) keep the class
- # attribute path unchanged.
- per_platform_check = getattr(
- runtime_adapter, "supports_inchannel_continuable_for_platform",
- None,
+ # Continuable surface (D1/D2/D6) from platform config ``extra``; default "thread".
+ # ``in_channel`` delivers FLAT so a plain channel reply continues via the shared session
+ # ``(platform, chat_id, None)``. Unsupported adapters fail SAFE to thread.
+ in_channel_surface = _resolve_cron_surface_mode(pconfig, platform_name) == "in_channel"
+ if (
+ in_channel_surface
+ and runtime_adapter is not None
+ and not _inchannel_surface_supported(runtime_adapter, platform_name)
+ ):
+ logger.debug(
+ "Job '%s': cron_continuable_surface=in_channel not supported on "
+ "%s, using thread",
+ job.get("id", "?"), platform_name,
)
- if callable(per_platform_check):
- try:
- surface_supported = bool(per_platform_check(platform_name))
- except Exception:
- surface_supported = False
- else:
- surface_supported = bool(getattr(
- runtime_adapter, "supports_inchannel_continuable", False
- ))
- if not surface_supported:
- # Fail safe (D6): platform has no in_channel continuation
- # primitive.
- logger.debug(
- "Job '%s': cron_continuable_surface=in_channel not supported on "
- "%s, using thread",
- job.get("id", "?"), platform_name,
- )
- in_channel_surface = False
+ in_channel_surface = False
if in_channel_surface and inchannel_continuable and live_adapter_ready:
- # Force flat delivery (D2): the continuable-channel target must
- # ignore any inherited origin/target thread_id, or the flat
- # continuable session seeded below (thread_id=None, via
- # _seed_cron_channel_session) never matches where the brief is
- # actually delivered — route_thread_id further down in this loop
- # reads `thread_id` and would otherwise route into the origin
- # thread instead of flat into the channel.
- #
- # Gated on `inchannel_continuable` (the SAME gate as the seed
- # below), NOT `mirror_this_target` alone: for origin targets the
- # seed fires on origin-match alone (in_channel is the
- # continuation surface, independent of the attach_to_session /
- # mirror opt-in), so the flatten must use the SAME gate — with
- # the default knobs off, a mirror-gated flatten kept delivering
- # into the origin thread while the flat session got seeded,
- # leaving the brief and its continuation surface in different
- # places.
- # Gated on `live_adapter_ready` (adapter present AND a running loop)
- # so the clear fires ONLY on the live-send path that actually seeds
- # the flat session — the SAME condition as the live-send block
- # below. `runtime_adapter is not None` alone is broader than that
- # path: an adapter can be present while the event loop is absent or
- # not running, in which case the live-send/seed block is skipped and
- # delivery falls through to the standalone path. Clearing thread_id
- # there would flatten a brief into a channel with NO seeded
- # continuable session behind it (and bypass the D6 capability
- # check), so the standalone fallback must keep the origin thread
- # (review r3609147550).
- #
- # Fan-out / broadcast / explicit-thread targets keep their thread_id
- # (they are not continuable and are never seeded). Placed AFTER
- # mirror_this_target / origin_user_id are computed above — those
- # need the ORIGINAL thread_id to match the origin conversation.
+ # Force flat (D2): an inherited thread_id would never match the flat seed (None). Gated
+ # on `inchannel_continuable` (SAME gate as the seed) AND `live_adapter_ready` (fallback
+ # never seeds). Stay AFTER mirror_this_target/origin_user_id (need ORIGINAL thread_id).
thread_id = None
- # For an in_channel delivery the flat continuation session is created
- # explicitly below (the shipped mirror only APPENDS to an existing
- # session, and the flat channel row is otherwise absent for a
- # chat_postMessage delivery). ``is_dm_target`` (computed above with
- # origin_user_id) selects the session chat_type so the seeded key
- # matches the inbound reply's key. ``inchannel_seeded`` suppresses the
- # generic mirror below so the brief is not double-written.
- inchannel_seeded = False
-
- # Continuable cron (thread-preferred): when mirroring is enabled for the
- # origin target and the gateway is live, try to open a DEDICATED thread
- # for this job and deliver the brief into it. On thread-capable
- # platforms (Telegram/Discord/Slack) the brief + the user's replies live
- # in their own scrollback; the thread-keyed session is seeded so a reply
- # continues with full context. On DM-only platforms (WhatsApp/Signal)
- # create_handoff_thread returns None and we fall back to mirroring into
- # the origin DM session (handled after delivery). Cf. _process_handoff.
- #
- # in_channel surface (D2): SKIP thread creation entirely — leave
- # thread_id=None so the delivery posts flat, then
- # ``_seed_cron_channel_session`` (below) CREATES the shared-channel
- # session and mirrors the brief into it. The shipped mirror alone is
- # NOT enough here: ``mirror_to_session`` only APPENDS to an existing
- # session and a flat ``(platform, chat_id, None)`` row is otherwise
- # absent for a ``chat_postMessage`` delivery, so the seed must create
- # the row first (F5).
- thread_seeded = False
+ # Thread-preferred continuable cron: open a DEDICATED thread; its session is seeded after a
+ # successful send. DM-only platforms return None → mirror the origin DM. in_channel SKIPS
+ # this: it posts flat and _seed_cron_channel_session CREATES the session.
opened_thread_id: Optional[str] = None
if (
mirror_this_target
@@ -3565,548 +2952,44 @@ def _deliver_result(
and loop is not None
and not thread_id # never override an explicit origin thread/topic
):
- new_thread_id = _open_continuable_cron_thread(
+ opened_thread_id = _open_continuable_cron_thread(
job, runtime_adapter, chat_id, loop,
- )
- if new_thread_id:
- # Route THIS delivery into the new thread now (the send needs the
- # thread_id), but defer seeding the thread session until the
- # delivery actually succeeds — otherwise an open-succeeds /
- # deliver-fails case leaves a seeded brief the user never saw,
- # and (worse) suppresses the DM-fallback mirror via thread_seeded.
- thread_id = new_thread_id
- opened_thread_id = new_thread_id
-
- if live_adapter_ready:
- # Telegram topic routing (#22773, regression fixed #52060): a
- # ``telegram::`` cron target is
- # ambiguous — a forum-style topic in a private chat and a genuine
- # Bot API channel Direct-Messages topic share the same shape and
- # need OPPOSITE routing. Disambiguate at delivery time via
- # ``_is_channel_dm_topic`` (see its docstring for the full
- # rationale); ``thread_id`` goes in ``route_metadata`` so the
- # anchorless cron send bypasses the DeliveryRouter's private-chat
- # reply-anchor requirement. Compute the routed metadata ONCE so both
- # the text send (via DeliveryRouter) and the media send agree.
- from gateway.delivery import (
- DeliveryRouter,
- DeliveryTarget,
- _looks_like_int,
- looks_like_telegram_private_chat_id,
- )
-
- is_ambiguous_telegram_topic = (
- platform == Platform.TELEGRAM
- and thread_id is not None
- and looks_like_telegram_private_chat_id(str(chat_id))
- and _looks_like_int(str(thread_id))
- )
- route_via_dm_topic = is_ambiguous_telegram_topic and _is_channel_dm_topic(
- runtime_adapter, chat_id, loop, job["id"],
- )
- if route_via_dm_topic:
- # Genuine Bot API channel Direct-Messages topic (#22773 mode 2):
- # routed via direct_messages_topic_id, no bare thread_id.
- route_thread_id = None
- route_metadata = {
- "direct_messages_topic_id": str(thread_id),
- "job_id": job["id"],
- "notify": notify_delivery,
- }
- # Media metadata mirrors the text routing so attachments land in
- # the same DM topic instead of the General lane (#22773).
- media_metadata = {
- "direct_messages_topic_id": str(thread_id),
- "notify": notify_delivery,
- }
- else:
- # Forum-style topic (private chat / supergroup) or non-topic
- # target: route via message_thread_id (#52060). Put thread_id in
- # *route_metadata* (not just the DeliveryTarget) deliberately —
- # the DeliveryRouter's private-chat topic detection
- # (gateway/delivery.py) demands a reply anchor when thread_id is
- # absent from metadata; cron deliveries have no inbound reply
- # anchor, so the metadata key bypasses that check and lets the
- # adapter route via a plain message_thread_id.
- route_thread_id = str(thread_id) if thread_id is not None else None
- route_metadata = {"job_id": job["id"], "notify": notify_delivery}
- if route_thread_id:
- route_metadata["thread_id"] = route_thread_id
- media_metadata = {"notify": notify_delivery}
- if thread_id:
- media_metadata["thread_id"] = thread_id
-
- # Relay egress needs a tenant discriminator on the frame: the
- # connector's fail-closed guard resolves the workspace/guild from
- # metadata.scope_id, and after a gateway restart the RelayAdapter's
- # per-chat scope cache is COLD (learned only from inbound), while
- # DeliveryRouter stamps scope only for the configured HOME channel
- # (gateway/delivery.py). A scoped origin that is not the home chat
- # therefore egressed with no scope_id at all and could be rejected
- # before delivery — the delivery-leg sibling of the seed-key scope
- # fix. Origin-matching targets only: a fan-out/broadcast target's
- # tenant is NOT the origin's, and stamping the wrong scope is worse
- # than none (the router/home path handles fan-out home targets).
- if origin_target and origin.get("scope_id"):
- route_metadata.setdefault("scope_id", str(origin["scope_id"]))
- media_metadata = dict(media_metadata or {})
- media_metadata.setdefault("scope_id", str(origin["scope_id"]))
-
- try:
- # Send cleaned text (MEDIA tags stripped) — not the raw content.
- # Route through the gateway's DeliveryRouter so the live send
- # gets the same platform-specific routing as live messages —
- # in particular Telegram's three-mode topic routing. The
- # standalone cron path lacked this, so DM-topic cron deliveries
- # landed in the General topic or were rejected by Bot API 10.0
- # (#22773).
- text_to_send = cleaned_delivery_content.strip()
- adapter_ok = True
- timed_out = False
- delivered_message_id = None
- if not text_to_send and not media_files:
- # Nothing to hand the adapter at all. This used to fall
- # straight through to the `if adapter_ok:` branch below and
- # log "delivered to via live adapter" for a send that
- # never happened (#77763). Fail closed so the run reports
- # the empty payload instead.
- msg = (
- f"live adapter send skipped (empty text and no media) "
- f"for {platform_name}:{chat_id}"
- )
- logger.warning("Job '%s': %s", job["id"], msg)
- target_errors.append(msg)
- adapter_ok = False
- elif text_to_send:
- from agent.async_utils import safe_schedule_threadsafe
-
- router = DeliveryRouter(config, target_adapters)
- route_target = DeliveryTarget(
- platform=platform,
- chat_id=str(chat_id),
- thread_id=route_thread_id,
- is_explicit=True,
- )
- # Pass thread routing via the target (not a bare metadata
- # "thread_id"): the router only applies its Telegram DM-topic
- # detection when "thread_id"/"message_thread_id" are absent
- # from metadata, deriving the routing from target.thread_id
- # or the explicit direct_messages_topic_id above.
- future = safe_schedule_threadsafe(
- router._deliver_to_platform(
- route_target,
- text_to_send,
- route_metadata,
- ),
- loop,
- )
- if future is None:
- adapter_ok = False
- target_errors.append("live adapter event loop scheduling failed")
- else:
- send_result = None
- timeout_handled = False
- try:
- send_result = future.result(timeout=60)
- except TimeoutError:
- # #38922: a slow confirmation does NOT necessarily
- # mean the send failed — but we must distinguish two
- # cases via future.cancel()'s return value:
- #
- # cancel() == False -> the coroutine was already
- # running on the gateway loop when the timeout
- # fired; the request is in flight on the wire and
- # cannot be un-sent. Re-sending via standalone
- # would be a guaranteed DUPLICATE, so treat it as
- # delivered (assume-delivered).
- #
- # cancel() == True -> the scheduled callback never
- # started executing (loop wedged/backlogged for
- # the full 60s), so nothing was sent. We MUST
- # fall through to the standalone path or the
- # message is silently dropped (worse than a
- # duplicate).
- cancelled = future.cancel()
- if cancelled:
- msg = (
- f"live adapter send to {platform_name}:{chat_id} "
- "timed out before the coroutine was dispatched"
- )
- logger.warning(
- "Job '%s': %s, falling back to standalone",
- job["id"], msg,
- )
- target_errors.append(msg)
- adapter_ok = False # fall through to standalone path
- timeout_handled = True
- else:
- timed_out = True
- timeout_handled = True
- logger.warning(
- "Job '%s': live adapter send to %s:%s timed out "
- "after 60s; already dispatched (in flight), "
- "assuming delivered (skipping standalone fallback "
- "to avoid duplicate)",
- job["id"], platform_name, chat_id,
- )
- except Exception as ex:
- # A real send error (not a slow confirmation) — fall
- # through to the standalone path so the message is
- # still delivered.
- target_errors.append(f"live adapter send failed: {ex}")
- raise
-
- if timeout_handled:
- # The timeout branch above already decided the
- # outcome (assume-delivered if in flight, or
- # adapter_ok=False to fall through if never
- # dispatched). send_result is None, so skip the
- # confirmation/thread-fallback inspection below.
- pass
- else:
- # _deliver_to_platform returns either a SendResult
- # (.success attr) or, when the silence-narration
- # filter drops the message, a plain dict
- # {"success": True, "delivered": False, ...}.
- # Normalize both shapes so a getattr default doesn't
- # misread a dict, and so a None / success-less object
- # is NOT counted as delivered (#47056). The
- # confirmation itself handles both shapes: a truthy
- # `success` with `delivered: False` is a drop, not a
- # delivery (#77763).
- if isinstance(send_result, dict):
- send_raw_response = send_result.get("raw_response")
- delivered_message_id = send_result.get("message_id")
- else:
- send_raw_response = getattr(send_result, "raw_response", None)
- delivered_message_id = getattr(send_result, "message_id", None)
- _evidence_gap: list = []
- send_success = _confirm_adapter_delivery(
- send_result, job["id"], _evidence_gap,
- )
- if send_success and _evidence_gap:
- unverified_targets.append(f"{platform_name}:{chat_id}")
-
- if not send_success:
- if isinstance(send_result, dict):
- # A filtered drop carries no "error" — name
- # the filter instead of reporting "unknown".
- err = (
- send_result.get("error")
- or send_result.get("filtered")
- or "unknown"
- )
- shape = "dict"
- elif send_result is not None:
- err = getattr(send_result, "error", None)
- shape = type(send_result).__name__
- else:
- err = "no response from adapter"
- shape = "None"
- msg = (
- f"live adapter send to {platform_name}:{chat_id} "
- f"returned unconfirmed result ({shape}, error={err})"
- )
- if transport is not None and transport.is_relay:
- logger.warning("Job '%s': %s", job["id"], msg)
- else:
- logger.warning(
- "Job '%s': %s, falling back to standalone",
- job["id"], msg,
- )
- target_errors.append(msg)
- adapter_ok = False # fall through to standalone path
- elif (
- send_raw_response
- and thread_id
- and send_raw_response.get("thread_fallback")
- ):
- requested_thread_id = send_raw_response.get("requested_thread_id") or thread_id
- msg = (
- f"configured thread_id {requested_thread_id} for "
- f"{platform_name}:{chat_id} was not found; delivered without thread_id"
- )
- logger.warning("Job '%s': %s", job["id"], msg)
- delivery_errors.append(msg)
-
- # Send extracted media files as native attachments via the live
- # adapter, using the same DM-topic-aware routing as the text send
- # (#22773 — media previously used a bare thread_id and landed in
- # the General lane for private DM topics). Skip on an in-flight
- # confirmation timeout: the gateway loop is contended, so each
- # media send would also block its 30s budget, and the text
- # payload is already assumed delivered (#38922). Record the
- # skipped attachments so the drop is visible rather than silently
- # lost.
- if adapter_ok and not timed_out and media_files:
- routed_media_metadata = dict(media_metadata or {})
- if transport is not None and transport.is_relay:
- routed_media_metadata["_relay_logical_platform"] = platform.value
- logical_home = config.get_home_channel(platform)
- if logical_home is not None and logical_home.chat_id == chat_id:
- if logical_home.user_id:
- routed_media_metadata["user_id"] = logical_home.user_id
- if logical_home.scope_id:
- routed_media_metadata["scope_id"] = logical_home.scope_id
- _media_errors = _send_media_via_adapter(
- runtime_adapter,
- chat_id,
- media_files,
- routed_media_metadata or None,
- loop,
- job,
- platform=platform,
- )
- # Surface per-file failures into the run status (parity
- # with the standalone lane): text delivered but an
- # attachment didn't is a visible partial failure, not ok.
- for _me in _media_errors:
- _msg = f"{_me} (target {platform_name}:{chat_id})"
- delivery_errors.append(_msg)
- elif timed_out and media_files:
- msg = (
- f"{len(media_files)} media attachment(s) not delivered to "
- f"{platform_name}:{chat_id} (live adapter confirmation timed out)"
- )
- logger.warning("Job '%s': %s", job["id"], msg)
- delivery_errors.append(msg)
-
- if adapter_ok:
- # Log WHERE it went, not just that it went: a ghost delivery
- # that landed in the wrong lane (General topic instead of the
- # routed thread) is indistinguishable from a real one without
- # the routing identity (#77763).
- logger.info(
- "Job '%s': delivered to %s:%s via live adapter thread=%s message_id=%s",
- job["id"], platform_name, chat_id,
- route_thread_id if route_thread_id is not None else "-",
- delivered_message_id if delivered_message_id is not None else "-",
- )
- delivered = True
- # Seed the thread session only now that delivery into it
- # succeeded (deferred from thread-open above).
- if opened_thread_id and not thread_seeded:
- _seed_cron_thread_session(
- job, runtime_adapter, platform_name, chat_id,
- opened_thread_id, mirror_text,
- chat_name=origin.get("chat_name"),
- is_dm=is_dm_target,
- scope_id=origin.get("scope_id"),
- )
- thread_seeded = True
- # in_channel surface: CREATE + seed the flat channel/DM
- # session (the shipped mirror only appends to an existing
- # session — the flat row is otherwise absent for a
- # chat_postMessage delivery, so the brief would be lost).
- # Gated on `inchannel_continuable` — the SHARED gate with
- # the thread-flatten above (they must not drift, or the
- # brief and its continuation session land in different
- # places). Origin targets seed without requiring the
- # mirror opt-in: in_channel IS the continuation surface —
- # a continuable flat cron without its seed is a brief the
- # next reply can't see (the bug Victor hit live
- # 2026-08-19: agent had "no idea about the delivery
- # message"). Mirror-eligible NON-origin targets
- # (origin_fallback / opted-in explicit — see
- # _target_mirror_eligible) also seed, guarded by
- # _inchannel_seed_allowed inside the gate: group-channel
- # keys are user-isolated, so a seed without a user_id
- # (origin-less managed cron into a shared channel) would
- # create an orphan session no reply resolves to — those
- # fall back to the plain mirror instead.
- if in_channel_surface and inchannel_continuable and not thread_seeded:
- inchannel_seeded = _seed_cron_channel_session(
- job, runtime_adapter, platform_name, chat_id,
- mirror_text, is_dm=is_dm_target,
- user_id=origin_user_id,
- chat_name=origin.get("chat_name"),
- scope_id=origin.get("scope_id"),
- )
- if not inchannel_seeded:
- logger.warning(
- "Job '%s': in_channel seed did NOT land on %s:%s "
- "— a plain reply will not see this brief",
- job["id"], platform_name, chat_id,
- )
- # Companion THREAD-surface seed (live gap, Alice
- # 2026-08-19): a flat brief is still a Slack message
- # the user can reply to IN ITS THREAD — the natural
- # mobile/desktop affordance — and that reply keys to
- # (chat, thread=), a session the flat seed
- # never touches. Seed it too so BOTH reply surfaces
- # continue the job. Uses the delivered message id as
- # the thread anchor; best-effort like every seed.
- if delivered_message_id:
- _seed_cron_thread_session(
- job, runtime_adapter, platform_name, chat_id,
- str(delivered_message_id), mirror_text,
- chat_name=origin.get("chat_name"),
- is_dm=is_dm_target,
- scope_id=origin.get("scope_id"),
- )
- elif in_channel_surface and not inchannel_continuable:
- logger.warning(
- "Job '%s': in_channel delivery to %s:%s is not a "
- "continuable target (origin=%s:%s thread=%s; not the "
- "origin conversation, and not a mirror-eligible "
- "fallback/opted-in target the seed can key) — seed "
- "skipped; the plain mirror below may still apply",
- job["id"], platform_name, chat_id,
- origin.get("platform"), origin.get("chat_id"),
- origin.get("thread_id"),
- )
- _maybe_mirror_cron_delivery(
- job, platform_name, chat_id, mirror_text,
- thread_id=thread_id, user_id=origin_user_id,
- enabled=mirror_this_target and not thread_seeded and not inchannel_seeded,
- )
- except Exception as e:
- err_msg = f"live adapter delivery to {platform_name}:{chat_id} failed: {e}"
- if not any(err_msg in err for err in target_errors):
- target_errors.append(err_msg)
- if transport is not None and transport.is_relay:
- logger.warning("Job '%s': %s", job["id"], err_msg)
- else:
- logger.warning(
- "Job '%s': %s, falling back to standalone",
- job["id"], err_msg,
- )
+ ) or None
+ if opened_thread_id:
+ thread_id = opened_thread_id
+ t = _TargetDelivery(
+ job=job,
+ platform=platform,
+ platform_name=platform_name,
+ chat_id=chat_id,
+ thread_id=thread_id,
+ transport=transport,
+ pconfig=pconfig,
+ runtime_adapter=runtime_adapter,
+ target_adapters=target_adapters,
+ config=config,
+ loop=loop,
+ notify_delivery=notify_delivery,
+ origin=origin,
+ origin_target=origin_target,
+ origin_user_id=origin_user_id,
+ is_dm_target=is_dm_target,
+ mirror_text=mirror_text,
+ mirror_this_target=mirror_this_target,
+ in_channel_surface=in_channel_surface,
+ inchannel_continuable=inchannel_continuable,
+ opened_thread_id=opened_thread_id,
+ )
+ delivered = live_adapter_ready and _deliver_via_live_adapter(
+ t, cleaned_delivery_content, media_files,
+ target_errors=target_errors,
+ delivery_errors=delivery_errors,
+ unverified_targets=unverified_targets,
+ )
if not delivered:
- if transport is not None and transport.is_relay:
- # Relay owns the logical destination and its connector owns the
- # platform credential. A native retry could duplicate delivery
- # and cannot be authenticated correctly, so fail closed.
- if not target_errors:
- target_errors.append(
- f"relay delivery to {platform_name}:{chat_id} failed"
- )
- delivery_errors.extend(target_errors)
- continue
- # If the interpreter is finalizing (gateway SIGTERM / restart /
- # OOM), scheduling any new delivery is futile — asyncio.run and a
- # fresh ThreadPoolExecutor both raise "cannot schedule new futures
- # after interpreter shutdown". Skip gracefully with a warning
- # rather than emitting an ERROR traceback on every restart-race
- # (#58720, #55924).
- if _interpreter_shutting_down():
- msg = f"delivery to {platform_name}:{chat_id} skipped — interpreter is shutting down"
- logger.warning("Job '%s': %s", job["id"], msg)
- target_errors.append(msg)
- delivery_errors.extend(target_errors)
- continue
- # The live lane already failed closed on an empty payload; the
- # standalone senders do not. The Telegram adapter returns
- # SendResult(success=True) for empty content WITHOUT an API call,
- # so falling through here turns a phantom live delivery into a
- # phantom standalone one and logs it as delivered (#77763). Both
- # _send_to_platform call sites below are reached through this
- # point, so one guard closes the lane.
- if not cleaned_delivery_content.strip() and not media_files:
- msg = (
- f"standalone send skipped (empty text and no media) "
- f"for {platform_name}:{chat_id}"
- )
- logger.warning("Job '%s': %s", job["id"], msg)
- target_errors.append(msg)
- delivery_errors.extend(target_errors)
- continue
- # Standalone path: run the async send in a fresh event loop (safe from any thread)
- coro = _send_to_platform(platform, pconfig, chat_id, cleaned_delivery_content, thread_id=thread_id, media_files=media_files)
- try:
- result = asyncio.run(coro)
- except RuntimeError as run_err:
- # asyncio.run() checks for a running loop before awaiting the coroutine;
- # when it raises, the original coro was never started — close it to
- # prevent "coroutine was never awaited" RuntimeWarning, then retry in a
- # fresh thread that has no running loop.
- coro.close()
- # If the RuntimeError is the interpreter-finalization signal,
- # the fresh-thread fallback would fail identically — skip
- # gracefully instead of logging a shutdown-race traceback.
- if _interpreter_shutting_down(run_err):
- msg = f"delivery to {platform_name}:{chat_id} skipped — interpreter is shutting down"
- logger.warning("Job '%s': %s", job["id"], msg)
- target_errors.append(msg)
- delivery_errors.extend(target_errors)
- continue
- # The thread-pool fallback can itself raise (SMTP ConnectionError,
- # future.result timeout, etc.). An exception raised inside this
- # `except RuntimeError` block is NOT caught by the sibling
- # `except Exception` below — it would escape _deliver_result()
- # and crash the whole delivery loop, silently skipping every
- # remaining target (#47163). Wrap the fallback in its own
- # try/except so a per-target failure is logged and the loop
- # continues to the next target.
- try:
- pool = concurrent.futures.ThreadPoolExecutor(max_workers=1)
- try:
- # The fallback worker is a fresh thread: it does NOT
- # inherit the multiplexed profile ContextVars (home
- # override + secret scope). Run inside a copy of the
- # active context so the standalone sender reads THIS
- # profile's bot token, not the process default's
- # (#100489) — same pattern as the session-db and
- # heartbeat workers in this module.
- _fallback_context = contextvars.copy_context()
- future = pool.submit(
- _fallback_context.run,
- asyncio.run,
- _send_to_platform(platform, pconfig, chat_id, cleaned_delivery_content, thread_id=thread_id, media_files=media_files),
- )
- result = future.result(timeout=30)
- finally:
- pool.shutdown(wait=False)
- except Exception as e:
- # A shutdown-race here is expected during teardown; downgrade
- # to a warning so it doesn't read as a genuine failure.
- if _interpreter_shutting_down(e):
- msg = f"delivery to {platform_name}:{chat_id} skipped — interpreter is shutting down"
- logger.warning("Job '%s': %s", job["id"], msg)
- target_errors.append(msg)
- delivery_errors.extend(target_errors)
- continue
- msg = f"delivery to {platform_name}:{chat_id} failed: {e}"
- logger.error("Job '%s': %s", job["id"], msg, exc_info=True)
- target_errors.extend([msg])
- delivery_errors.extend(target_errors)
- continue
- except Exception as e:
- msg = f"delivery to {platform_name}:{chat_id} failed: {e}"
- logger.error("Job '%s': %s", job["id"], msg, exc_info=True)
- target_errors.extend([msg])
- delivery_errors.extend(target_errors)
- continue
-
- if result and result.get("error"):
- # Include target context (platform/chat) so a bare error string
- # like "Discord send failed: TimeoutError: " is attributable.
- # Not inside an except block — the error comes from the send
- # result dict, so there is no traceback to attach.
- msg = f"delivery error: {result['error']} (target {platform_name}:{chat_id})"
- logger.error("Job '%s': %s", job["id"], msg)
- target_errors.extend([msg])
- delivery_errors.extend(target_errors)
- continue
-
- # Standalone senders report per-file attachment failures in
- # ``warnings`` while still returning success (the text leg
- # delivered). Surface them: a cron whose PDF/image silently
- # vanished used to mark the run ok with no trace — the exact
- # "manual run delivers text but no attachment" field report.
- _sender_warnings = (
- result.get("warnings") if isinstance(result, dict) else None
- ) or []
- for _w in _sender_warnings:
- msg = f"delivery warning: {_w} (target {platform_name}:{chat_id})"
- logger.error("Job '%s': %s", job["id"], msg)
- delivery_errors.append(msg)
-
- logger.info("Job '%s': delivered to %s:%s", job["id"], platform_name, chat_id)
- _maybe_mirror_cron_delivery(
- job, platform_name, chat_id, mirror_text,
- thread_id=thread_id, user_id=origin_user_id,
- enabled=mirror_this_target and not thread_seeded,
+ _deliver_standalone(
+ t, cleaned_delivery_content, media_files, target_errors, delivery_errors,
)
if policy_drop_errors:
@@ -4162,14 +3045,8 @@ _DEFAULT_MEDIA_SEND_TIMEOUT = 300
def _get_media_send_timeout() -> int:
- """Resolve the per-attachment media-send timeout from env/config.
-
- Mirrors the ``script_timeout_seconds`` resolution pattern: the
- HERMES_CRON_MEDIA_SEND_TIMEOUT env var wins, then
- ``cron.media_send_timeout_seconds`` in config.yaml, then the default
- (300s — large attachments like long TTS audio can legitimately exceed
- the old fixed 30s upload window).
- """
+ """Per-attachment media-send timeout: HERMES_CRON_MEDIA_SEND_TIMEOUT env, then
+ ``cron.media_send_timeout_seconds``, then 300s (long TTS audio can exceed a 30s window)."""
env_value = os.getenv("HERMES_CRON_MEDIA_SEND_TIMEOUT", "").strip()
if env_value:
try:
@@ -4197,15 +3074,9 @@ def _get_media_send_timeout() -> int:
def _get_session_db_timeout() -> float:
- """Resolve the bound on run_job's SessionDB init from env/config.
-
- Mirrors the ``script_timeout_seconds`` resolution pattern: the
- HERMES_CRON_SESSION_DB_TIMEOUT env var wins, then
- ``cron.session_db_timeout_seconds`` in config.yaml (present in
- DEFAULT_CONFIG, so ``load_config()``'s deep-merge supplies it), then
- 10s. Unlike the sibling timeouts, 0 is meaningful (unlimited — legacy
- behavior, opt-in for debugging), so values are passed through untouched.
- """
+ """Bound on run_job's SessionDB init: HERMES_CRON_SESSION_DB_TIMEOUT env, then
+ ``cron.session_db_timeout_seconds`` (in DEFAULT_CONFIG), then 10s. Unlike sibling timeouts,
+ 0 is meaningful (unlimited, debugging opt-in), so values pass through untouched."""
env_value = os.getenv("HERMES_CRON_SESSION_DB_TIMEOUT", "").strip()
if env_value:
try:
@@ -4223,9 +3094,7 @@ def _get_session_db_timeout() -> float:
if configured is not None:
return float(configured)
except Exception as exc:
- logger.debug(
- "Failed to load cron.session_db_timeout_seconds from config: %s", exc
- )
+ logger.debug("Failed to load cron.session_db_timeout_seconds from config: %s", exc)
return 10.0
@@ -4247,15 +3116,9 @@ def _read_windows_pyvenv_cfg(venv_dir: Path) -> dict[str, str]:
def _windows_cron_python_invocation(python_exe: str) -> tuple[str, dict[str, str]]:
- """Return an output-capable hidden Python invocation for Windows scripts.
-
- Cron scripts capture stdout/stderr, so using ``pythonw.exe`` directly can
- lose script output. uv-created venv ``python.exe`` launchers are also a
- problem: even with CREATE_NO_WINDOW, the launcher can re-exec the base
- console interpreter and flash a visible window. For uv venvs, bypass the
- launcher and run the base ``python.exe`` directly with the venv paths
- overlaid in the environment.
- """
+ """Hidden, output-capable Python invocation for Windows cron scripts. ``pythonw.exe`` loses
+ captured output; uv venv launchers can re-exec the base console python and flash a window even
+ with CREATE_NO_WINDOW, so run the base python directly with venv paths overlaid in env."""
if sys.platform != "win32":
return python_exe, {}
@@ -4276,10 +3139,7 @@ def _windows_cron_python_invocation(python_exe: str) -> tuple[str, dict[str, str
if base_python.exists() and site_packages.exists():
interpreter = base_python
env_overlay["VIRTUAL_ENV"] = str(venv_dir)
- pythonpath_entries = [
- str(Path(__file__).resolve().parents[1]),
- str(site_packages),
- ]
+ pythonpath_entries = [str(Path(__file__).resolve().parents[1]), str(site_packages)]
existing_pythonpath = os.environ.get("PYTHONPATH", "")
if existing_pythonpath:
pythonpath_entries.append(existing_pythonpath)
@@ -4314,23 +3174,17 @@ def _terminate_cron_script_process(proc: subprocess.Popen) -> None:
except (ProcessLookupError, PermissionError, OSError):
process_group = None
if process_group is not None:
- try:
+ with contextlib.suppress(subprocess.TimeoutExpired):
proc.wait(timeout=1.0)
- except subprocess.TimeoutExpired:
- pass
- # Escalate whenever ANY group member survived the TERM: a
- # TERM-ignoring descendant keeps the stdio pipe write ends
- # open, and the caller's communicate() would then block on
- # EOF forever. killpg(pgid, 0) probes group liveness.
+ # Escalate if ANY group member survived TERM: a survivor holds the pipe write ends
+ # open and the caller's communicate() would block on EOF forever.
try:
os.killpg(process_group, 0) # windows-footgun: ok — POSIX-only branch
except (ProcessLookupError, OSError):
process_group = None
if process_group is not None:
- try:
+ with contextlib.suppress((ProcessLookupError, PermissionError, OSError)):
os.killpg(process_group, getattr(signal, "SIGKILL", signal.SIGTERM))
- except (ProcessLookupError, PermissionError, OSError):
- pass
try:
proc.wait(timeout=1.0)
except subprocess.TimeoutExpired:
@@ -4341,9 +3195,7 @@ def _terminate_cron_script_process(proc: subprocess.Popen) -> None:
def _terminate_cron_script_tree(proc: subprocess.Popen) -> None:
"""Terminate a script tree, then fall back to the local process-group path."""
if proc.poll() is not None:
- # Already exited (e.g. finished right at the deadline): nothing to
- # signal, and calling kill_process_tree on a reaped pid would log a
- # spurious "no signal" warning. Mirrors _terminate_cron_script_process.
+ # Already reaped: kill_process_tree would log a spurious "no signal" warning.
return
pid = getattr(proc, "pid", None)
if not isinstance(pid, int) or pid <= 0:
@@ -4355,9 +3207,8 @@ def _terminate_cron_script_tree(proc: subprocess.Popen) -> None:
_terminate_cron_script_process(proc)
return
try:
- # Function-local so tests can monkeypatch agent.deadline.kill_process_tree;
- # separate from the kill try below so a packaging/import problem
- # surfaces as what it is instead of masquerading as a kill failure.
+ # Function-local (monkeypatchable); separate try so an import problem is not
+ # misreported as a kill failure.
from agent.deadline import kill_process_tree
except Exception:
logger.warning(
@@ -4386,34 +3237,19 @@ def _terminate_cron_script_tree(proc: subprocess.Popen) -> None:
def _drain_script_pipes(proc: subprocess.Popen) -> None:
- """Reap a terminated script process without ever blocking indefinitely.
-
- A descendant that survived the tree kill can hold the pipe write ends
- open, so a bare ``communicate()`` would wait for EOF forever. Bound the
- drain, then abandon the pipes — the caller only needs the process reaped
- and the worker thread unblocked, not the output.
- """
- try:
+ """Reap a terminated script without blocking forever: a surviving descendant can hold the pipe
+ write ends open, so bound the drain and abandon the pipes (output is not needed)."""
+ with contextlib.suppress(subprocess.TimeoutExpired):
proc.communicate(timeout=5.0)
return
- except subprocess.TimeoutExpired:
- pass
- try:
+ with contextlib.suppress(OSError):
proc.kill()
- except OSError:
- pass
for stream in (proc.stdout, proc.stderr):
- try:
+ with contextlib.suppress(OSError):
if stream is not None:
stream.close()
- except OSError:
- pass
- try:
+ with contextlib.suppress(subprocess.TimeoutExpired):
proc.wait(timeout=5.0)
- except subprocess.TimeoutExpired:
- # Truly wedged — leave the zombie to the OS reaper rather than
- # blocking the cron worker thread forever.
- pass
def _windows_cron_bootstrap_argv(
@@ -4423,28 +3259,14 @@ def _windows_cron_bootstrap_argv(
) -> list[str]:
"""Bootstrap a cron script under the base interpreter with ``.pth`` support.
- The uv-venv overlay mode runs the base ``python.exe`` (to avoid the
- launcher re-execing a console interpreter and flashing a window) and
- re-attaches the venv via ``PYTHONPATH``. But ``PYTHONPATH`` entries are
- plain ``sys.path`` additions — Python's site initialization never
- processes ``.pth`` files for them (only ``site.addsitedir()`` does) — so
- editable installs (``pip install -e``, ``__editable__*.pth`` links) are
- invisible to cron script jobs.
-
- Bootstrap with ``site.addsitedir()`` on the venv ``site-packages``, then
- exec the script as ``__main__``. ``runpy.run_path`` keeps ``__file__``
- correct; ``sys.path[0]`` is set to the script's directory to preserve the
- ``python script.py`` import semantics. Note: ``runpy`` does not set
- ``__package__``/``__spec__`` the way a direct invocation does, so
- package-relative imports (``from . import x``) may behave differently.
- Falls back to a plain invocation if the venv layout is unresolvable —
- the pre-existing PYTHONPATH behaviour is strictly better than failing
- to run at all.
+ Overlay mode runs base ``python.exe`` (avoids the launcher flashing a console window) with the
+ venv on ``PYTHONPATH`` — but ``.pth`` files are only processed by ``site.addsitedir()``, so
+ editable installs would be invisible. Bootstrap via addsitedir + ``runpy.run_path`` (keeps
+ ``__file__`` and ``sys.path[0]`` semantics); plain invocation if the venv is unresolvable.
"""
site_packages = Path(env_overlay.get("VIRTUAL_ENV", "")) / "Lib" / "site-packages"
if not site_packages.is_dir():
- # Silent here would make the "editable installs invisible" failure
- # undiagnosable; the pre-existing PYTHONPATH-only behaviour applies.
+ # Warn: silent fallback would make "editable installs invisible" undiagnosable.
logger.warning(
"Windows cron script: venv site-packages %s not found; running "
"without .pth processing (editable installs may be unimportable)",
@@ -4467,75 +3289,34 @@ def _run_job_script(
workdir: Optional[str] = None,
cancel_event: Optional[_CancelEventLike] = None,
) -> tuple[bool, str]:
- """Execute a cron job's data-collection script and capture its output.
+ """Execute a cron job's script and return ``(success, output)``; on failure *output* is the
+ error message for the LLM to report.
- Scripts must reside within HERMES_HOME/scripts/. Both relative and
- absolute paths are resolved and validated against this directory to
- prevent arbitrary script execution via path traversal or absolute
- path injection.
-
- Supported interpreters (chosen by file extension):
-
- * ``.sh`` / ``.bash`` — run with ``/bin/bash``
- * anything else — run with the current Python interpreter
- (``sys.executable``), preserving the original behaviour for
- Python-based pre-check and data-collection scripts.
-
- Shell support lets ``no_agent=True`` jobs ship classic bash watchdogs
- (the `memory-watchdog.sh` pattern) without wrapping them in Python.
-
- Subprocess environment is passed through ``_sanitize_subprocess_env`` so
- provider credentials and other Hermes-managed secrets are not inherited
- (SECURITY.md §2.3), matching terminal and MCP child processes.
-
- Args:
- script_path: Path to the script. Relative paths are resolved
- against HERMES_HOME/scripts/. Absolute and ~-prefixed paths
- are also validated to ensure they stay within the scripts dir.
- workdir: Optional absolute path to use as the script's cwd.
- When set, the subprocess runs in this directory instead of
- the scripts-dir parent. The Python process cwd is NEVER
- mutated, avoiding the global-side-effect bug where a cron
- job's ``os.chdir()`` leaks into concurrent gateway sessions
- (#69396).
-
- Returns:
- (success, output) — on failure *output* contains the error message so the
- LLM can report the problem to the user.
+ Scripts MUST resolve inside HERMES_HOME/scripts/ (relative, absolute and ``~`` paths are all
+ validated — path traversal / absolute-path injection). Interpreter by extension:
+ ``.sh``/``.bash`` → bash, else ``sys.executable``. Env goes through ``build_subprocess_env``
+ (SECURITY.md §2.3).
+ ``workdir`` sets the subprocess cwd only; the Python process cwd is NEVER mutated (an
+ ``os.chdir()`` would leak into concurrent gateway sessions).
"""
scripts_dir = _get_hermes_home() / "scripts"
_ensure_cron_dir(scripts_dir)
scripts_dir_resolved = scripts_dir.resolve()
- # Same ingestion contract as cron.lifecycle_guard._expand_candidate_path:
- # a NUL-bearing value can never name a real script, and on Windows the
- # Path operations raise ValueError *after* expanduser (expanduser never
- # expands "~user" there, so the try below never fires) — reject eagerly
- # so both platforms fail cleanly instead of crashing the scheduler.
- # str() first so the guard itself can never raise TypeError on a
- # non-str script_path (e.g. a Path passed by a future caller) — the
- # guard must be crash-proof even though every current call site
- # passes a plain str (#86832 review).
+ # Same contract as cron.lifecycle_guard._expand_candidate_path. Reject NUL eagerly: on Windows
+ # Path ops raise ValueError *after* expanduser so the try below would not catch it. str() first
+ # so the guard itself cannot raise on a non-str script_path.
if "\x00" in str(script_path):
return False, f"Blocked: script path contains a NUL byte: {script_path!r}"
try:
raw = Path(script_path).expanduser()
except (ValueError, RuntimeError, OSError):
- # Same ingestion contract as cron.lifecycle_guard: a NUL-bearing
- # value (ValueError) or an unexpandable ``~`` (RuntimeError with no
- # resolvable HOME) can never name a real script. The creation-time
- # guard tolerates such values as "nothing to scan", so they can
- # reach fire time — fail the run with a report instead of crashing
- # the scheduler with an unhandled exception.
+ # RuntimeError: unexpandable ``~`` (no resolvable HOME).
return False, f"Blocked: script path is not a valid filesystem path: {script_path!r}"
- if raw.is_absolute():
- path = raw.resolve()
- else:
- path = (scripts_dir / raw).resolve()
+ path = raw.resolve() if raw.is_absolute() else (scripts_dir / raw).resolve()
- # Guard against path traversal, absolute path injection, and symlink
- # escape — scripts MUST reside within HERMES_HOME/scripts/.
+ # Traversal / absolute-path / symlink escape guard — MUST stay inside HERMES_HOME/scripts/.
try:
path.relative_to(scripts_dir_resolved)
except ValueError:
@@ -4551,20 +3332,11 @@ def _run_job_script(
script_timeout = _get_script_timeout()
- # Pick an interpreter by extension. Bash for .sh/.bash, Python for
- # everything else. We deliberately do NOT honour the file's own
- # shebang: the scripts dir is trusted, but keeping the interpreter
- # choice explicit here keeps the allowed surface small and auditable.
+ # Interpreter by extension; the shebang is deliberately NOT honoured (small, auditable surface).
suffix = path.suffix.lower()
if suffix in {".sh", ".bash"}:
- # Resolve bash dynamically so Windows (Git Bash) and Linux/macOS
- # all work. On native Windows without Git for Windows installed
- # shutil.which returns None — fall back to a clear error rather
- # than a FileNotFoundError with a confusing "[WinError 2]"
- # traceback.
- _bash = shutil.which("bash") or (
- "/bin/bash" if os.path.isfile("/bin/bash") else None
- )
+ # which() finds Git Bash on Windows; None there → clear error instead of a "[WinError 2]".
+ _bash = shutil.which("bash") or ("/bin/bash" if os.path.isfile("/bin/bash") else None)
if _bash is None:
return False, (
f"Cannot run .sh/.bash script {path.name!r}: bash not found on PATH. "
@@ -4576,9 +3348,7 @@ def _run_job_script(
else:
python_exe, env_overlay = _windows_cron_python_invocation(sys.executable)
if env_overlay:
- # Overlay mode (Windows uv venv): PYTHONPATH alone cannot make
- # editable installs importable — .pth processing needs
- # site.addsitedir() (see _windows_cron_bootstrap_argv).
+ # Windows uv-venv overlay: needs the .pth bootstrap for editable installs.
argv = _windows_cron_bootstrap_argv(python_exe, env_overlay, str(path))
else:
argv = [python_exe, str(path)]
@@ -4596,10 +3366,7 @@ def _run_job_script(
}
env = build_subprocess_env()
env.update(env_overlay)
- # Use the job's workdir as the subprocess cwd when configured,
- # otherwise default to the scripts-dir parent (back-compat).
- # NEVER mutate the Python process cwd — that would leak into
- # concurrent gateway sessions (#69396).
+ # Subprocess cwd only (default: scripts-dir parent). NEVER os.chdir() the process.
_script_cwd = workdir or str(path.parent)
proc = subprocess.Popen(
argv,
@@ -4613,22 +3380,15 @@ def _run_job_script(
deadline = time.monotonic() + script_timeout
while True:
if cancel_event is not None and cancel_event.is_set():
- # Same bug class as the timeout site below: a cancelled fire
- # must not orphan own-session grandchildren either.
+ # Tree-kill here too: a cancelled fire must not orphan own-session grandchildren.
_terminate_cron_script_tree(proc)
_drain_script_pipes(proc)
return False, "Script cancelled because cron fire ownership was lost"
remaining = deadline - time.monotonic()
if remaining <= 0:
- # Phase 4a (#85125): a script timeout must leave ZERO living
- # descendants. killpg only reaches the script's own process
- # group — a grandchild that called setsid (backgrounded
- # shell jobs, watchdogs) escapes it and keeps running after
- # the job reports failure (#71148 / #59549).
- # agent.deadline.kill_process_tree snapshots the descendant
- # set via psutil BEFORE signalling, so own-session
- # grandchildren are reached too — the unified deadline
- # layer's tree-kill (#85147, d6a5cb9725).
+ # Timeout must leave ZERO descendants: killpg misses setsid grandchildren
+ # (watchdogs, backgrounded shell jobs); kill_process_tree snapshots descendants
+ # BEFORE signalling.
_terminate_cron_script_tree(proc)
_drain_script_pipes(proc)
return False, f"Script timed out after {script_timeout}s: {path}"
@@ -4641,7 +3401,7 @@ def _run_job_script(
stdout = (stdout_raw or "").strip()
stderr = (stderr_raw or "").strip()
- # Redact secrets from both stdout and stderr before any return path.
+ # Redact secrets before ANY return path.
try:
from agent.redact import redact_sensitive_text
stdout = redact_sensitive_text(stdout)
@@ -4665,23 +3425,33 @@ def _run_job_script(
return False, f"Script execution failed: {exc}"
+def _start_heartbeat_thread(loop_fn, name: str, fail_log) -> Optional[threading.Thread]:
+ """Start ``loop_fn`` on a daemon thread inside a copy of the current context (multiplexed
+ profile ContextVars). On failure calls ``fail_log()`` inside the except (traceback intact) and
+ returns None."""
+ thread = threading.Thread(
+ target=contextvars.copy_context().run, args=(loop_fn,), name=name, daemon=True,
+ )
+ try:
+ thread.start()
+ except Exception:
+ fail_log()
+ return None
+ return thread
+
+
def _run_job_script_with_claim_heartbeat(
job: dict,
script_path: str,
workdir: Optional[str] = None,
cancel_event: Optional[_CancelEventLike] = None,
) -> tuple[bool, str]:
- """Run a cron script while keeping its owned one-shot claim fresh.
+ """Run a cron script while heartbeating its owned one-shot claim.
- Script execution is synchronous and may legitimately outlive the stale
- claim TTL. Without a concurrent heartbeat, another scheduler process can
- mistake the live run for a dead owner and dispatch the same one-shot again.
- Recurring jobs and unclaimed/manual runs have no durable one-shot claim and
- therefore use the ordinary script path without starting a thread.
-
- The claim owner is captured from the dispatched job and never re-read from
- storage. ``heartbeat_run_claim`` compares that stable owner before every
- refresh, so a stale runner cannot extend a replacement owner's claim.
+ A long script can outlive the stale-claim TTL; without a heartbeat another scheduler would
+ re-dispatch the one-shot. Recurring/unclaimed runs have no durable claim → no thread. The owner
+ is captured from the dispatched job, never re-read, so a stale runner cannot extend a
+ replacement owner's claim.
"""
schedule = job.get("schedule")
claim = job.get("run_claim")
@@ -4695,55 +3465,34 @@ def _run_job_script_with_claim_heartbeat(
job_id = str(job.get("id") or "")
stop = threading.Event()
- heartbeat_context = contextvars.copy_context()
def _heartbeat_loop() -> None:
while not stop.wait(_RUN_CLAIM_HEARTBEAT_SECONDS):
try:
heartbeat_run_claim(job_id, expected_owner=owner)
except Exception:
- logger.debug(
- "Job '%s': script run_claim heartbeat failed",
- job_id,
- exc_info=True,
- )
+ logger.debug("Job '%s': script run_claim heartbeat failed", job_id, exc_info=True)
- heartbeat_thread = threading.Thread(
- target=heartbeat_context.run,
- args=(_heartbeat_loop,),
- name="cron-script-claim-heartbeat",
- daemon=True,
+ heartbeat_thread = _start_heartbeat_thread(
+ _heartbeat_loop, "cron-script-claim-heartbeat",
+ lambda: logger.debug(
+ "Job '%s': could not start script run_claim heartbeat", job_id, exc_info=True,
+ ),
)
- try:
- heartbeat_thread.start()
- except Exception:
- logger.debug(
- "Job '%s': could not start script run_claim heartbeat",
- job_id,
- exc_info=True,
- )
+ if heartbeat_thread is None:
return _run_job_script(script_path, workdir=workdir, cancel_event=cancel_event)
try:
return _run_job_script(script_path, workdir=workdir, cancel_event=cancel_event)
finally:
stop.set()
- # Event.wait() wakes immediately. Keep completion bounded if the
- # heartbeat is already waiting on another process's jobs-file lock.
+ # Bounded join: the heartbeat may be blocked on another process's jobs-file lock.
heartbeat_thread.join(timeout=1.0)
def _parse_wake_gate(script_output: str) -> bool:
- """Parse the last non-empty stdout line of a cron job's pre-check script
- as a wake gate.
-
- The convention (ported from nanoclaw #1232): if the last stdout line is
- JSON like ``{"wakeAgent": false}``, the agent is skipped entirely — no
- LLM run, no delivery. Any other output (non-JSON, missing flag, gate
- absent, or ``wakeAgent: true``) means wake the agent normally.
-
- Returns True if the agent should wake, False to skip.
- """
+ """Wake gate: False only if the last non-empty stdout line is JSON ``{"wakeAgent": false}``
+ (agent skipped entirely — no LLM run, no delivery); anything else wakes normally."""
if not script_output:
return True
stripped_lines = [line for line in script_output.splitlines() if line.strip()]
@@ -4759,139 +3508,183 @@ def _parse_wake_gate(script_output: str) -> bool:
return gate.get("wakeAgent", True) is not False
+def _prepend_context_block(prompt: str, heading: str, intro: str, body: str) -> str:
+ """Prefix ``prompt`` with a fenced ``## heading`` data block."""
+ return f"## {heading}\n{intro}\n\n```\n{body}\n```\n\n{prompt}"
+
+
+_MAX_CONTEXT_CHARS = 8000
+
+
+def _inject_context_from(job: dict, prompt: str) -> tuple[str, bool]:
+ """Prepend the latest output of each ``context_from`` job; returns ``(prompt, injected)``."""
+ context_from = job.get("context_from")
+ if not context_from:
+ return prompt, False
+ from cron.jobs import get_cron_output_dir
+ output_dir = get_cron_output_dir()
+ if isinstance(context_from, str):
+ context_from = [context_from]
+ injected = False
+ for source_job_id in context_from:
+ # "self" = the job's own id: continuity across runs without touching session history.
+ if isinstance(source_job_id, str) and source_job_id.strip().lower() == "self":
+ source_job_id = str(job.get("id") or "")
+ is_self = source_job_id == job.get("id")
+ # Traversal guard — valid job IDs are hex strings.
+ if not source_job_id or not all(c in "0123456789abcdef" for c in source_job_id):
+ logger.warning(
+ "context_from: skipping invalid job_id %r for job_id=%r name=%r%s",
+ source_job_id, job.get("id"), job.get("name"), _cron_job_origin_log_suffix(job),
+ )
+ continue
+ try:
+ output_files = sorted(
+ (output_dir / source_job_id).glob("*.md"),
+ key=lambda f: f.stat().st_mtime,
+ reverse=True,
+ )
+ if not output_files:
+ continue # silent skip — no output yet
+ latest_output = output_files[0].read_text(encoding="utf-8").strip()
+ if len(latest_output) > _MAX_CONTEXT_CHARS:
+ latest_output = latest_output[:_MAX_CONTEXT_CHARS] + "\n\n[... output truncated ...]"
+ if not latest_output:
+ continue # silent skip — empty output
+ if is_self:
+ prompt = _prepend_context_block(
+ prompt, "Your previous run's output",
+ "The following is this job's most recent output from its "
+ "previous run. Use it for continuity: avoid repeating what "
+ "was already reported, and continue where the last run "
+ "left off.",
+ latest_output,
+ )
+ else:
+ prompt = _prepend_context_block(
+ prompt, f"Output from job '{source_job_id}'",
+ "The following is the most recent output from a preceding "
+ "cron job. Use it as context for your analysis.",
+ latest_output,
+ )
+ injected = True
+ except (OSError, PermissionError) as e:
+ # silent skip — never put error text into the prompt
+ logger.warning("context_from: failed to read output for job %r: %s", source_job_id, e)
+ return prompt, injected
+
+
+def _load_cron_skill_parts(job: dict, skill_names: list[str]) -> list[str]:
+ """Load each named skill/bundle into prompt parts; unknown ones are skipped with a user notice."""
+ from tools.skills_tool import skill_view
+ from tools.skill_usage import bump_use
+ from agent.skill_bundles import build_bundle_invocation_message, resolve_bundle_command_key
+ from agent.skill_utils import normalize_skill_lookup_name
+
+ job_label = job.get("name", job.get("id"))
+ task_id = str(job.get("id") or "") or None
+ parts: list[str] = []
+ skipped: list[str] = []
+ for skill_name in skill_names:
+ # Bundles shadow same-slug skills, mirroring the CLI/gateway slash-command path.
+ bundle_key = resolve_bundle_command_key(skill_name.lstrip("/"))
+ if bundle_key:
+ bundle_payload = build_bundle_invocation_message(
+ bundle_key, user_instruction="", task_id=task_id,
+ )
+ if bundle_payload:
+ if parts:
+ parts.append("")
+ parts.append(bundle_payload[0])
+ continue
+ logger.warning(
+ "Cron job '%s': bundle '%s' could not load any skills, skipping", job_label, skill_name,
+ )
+ skipped.append(skill_name)
+ continue
+
+ try:
+ loaded = json.loads(skill_view(normalize_skill_lookup_name(skill_name)))
+ except (json.JSONDecodeError, TypeError):
+ logger.warning("Cron job '%s': skill '%s' returned invalid JSON, skipping", job_label, skill_name)
+ skipped.append(skill_name)
+ continue
+ if not loaded.get("success"):
+ error = loaded.get("error") or f"Failed to load skill '{skill_name}'"
+ logger.warning("Cron job '%s': skill not found, skipping — %s", job_label, error)
+ skipped.append(skill_name)
+ continue
+
+ try:
+ bump_use(skill_name, task_id=task_id)
+ except Exception:
+ logger.debug("Cron job: failed to bump skill usage for '%s'", skill_name, exc_info=True)
+
+ if parts:
+ parts.append("")
+ parts.extend([
+ f'[IMPORTANT: The user has invoked the "{skill_name}" skill, indicating they want you to follow its instructions. The full skill content is loaded below.]',
+ "",
+ str(loaded.get("content") or "").strip(),
+ ])
+
+ if skipped:
+ parts.insert(0, (
+ f"[IMPORTANT: The following skill(s) were listed for this job but could not be found "
+ f"and were skipped: {', '.join(skipped)}. "
+ f"Start your response with a brief notice so the user is aware, e.g.: "
+ f"'⚠️ Skill(s) not found and skipped: {', '.join(skipped)}']"
+ ))
+ return parts
+
+
def _build_job_prompt(
job: dict,
prerun_script: Optional[tuple] = None,
extra_prompt: Optional[str] = None,
) -> str:
- """Build the effective prompt for a cron job, optionally loading one or more skills first.
+ """Build the effective prompt for a cron job, optionally loading skills first.
- Args:
- job: The cron job dict.
- prerun_script: Optional ``(success, stdout)`` from a script that has
- already been executed by the caller (e.g. for a wake-gate check).
- When provided, the script is not re-executed and the cached
- result is used for prompt injection. When omitted, the script
- (if any) runs inline as before.
- extra_prompt: Optional per-run context (from ``cronjob(action='run')``,
- #57331 — salvaged from #57342 by @liuhao1024). Appended to the
- stored prompt under a ``## Run Context`` header for this single
- fire only — never persisted to the job definition.
+ ``prerun_script``: cached ``(success, stdout)`` from a script the caller already ran (wake-gate
+ check) — skips re-execution. ``extra_prompt``: per-run ``## Run Context`` for this fire only,
+ never persisted to the job.
"""
user_prompt = str(job.get("prompt") or "")
if extra_prompt:
user_prompt = f"{user_prompt}\n\n## Run Context\n{extra_prompt}"
prompt = user_prompt
skills = job.get("skills")
- # True when runtime-collected DATA (script stdout, upstream-job output)
- # has been injected into the prompt. Data content legitimately quotes
- # command-shape strings (a triage feed ingesting a bug report that
- # pastes `rm -rf /`), so it must not be scanned with the strict
- # user-prompt pattern set — see _scan_assembled_cron_prompt.
+ # Runtime DATA (script stdout, upstream output) legitimately quotes command-shape strings, so it
+ # must not be scanned with the strict user-prompt set — see _scan_assembled_cron_prompt.
has_injected_data = False
- # Run data-collection script if configured, inject output as context.
script_path = job.get("script")
if script_path:
if prerun_script is not None:
success, script_output = prerun_script
else:
success, script_output = _run_job_script(script_path)
+ if success and not script_output:
+ return None # no output → nothing to report, skip the AI call
if success:
- if script_output:
- prompt = (
- "## Script Output\n"
- "The following data was collected by a pre-run script. "
- "Use it as context for your analysis.\n\n"
- f"```\n{script_output}\n```\n\n"
- f"{prompt}"
- )
- has_injected_data = True
- else:
- # Script produced no output — nothing to report, skip AI call.
- return None
- else:
- prompt = (
- "## Script Error\n"
- "The data-collection script failed. Report this to the user.\n\n"
- f"```\n{script_output}\n```\n\n"
- f"{prompt}"
+ prompt = _prepend_context_block(
+ prompt, "Script Output",
+ "The following data was collected by a pre-run script. "
+ "Use it as context for your analysis.",
+ script_output,
)
- has_injected_data = True
+ else:
+ prompt = _prepend_context_block(
+ prompt, "Script Error",
+ "The data-collection script failed. Report this to the user.",
+ script_output,
+ )
+ has_injected_data = True
- # Inject output from referenced cron jobs as context.
- context_from = job.get("context_from")
- if context_from:
- from cron.jobs import get_cron_output_dir
- output_dir = get_cron_output_dir()
- if isinstance(context_from, str):
- context_from = [context_from]
- for source_job_id in context_from:
- # "self" resolves to the job's own id: the job wakes up with its
- # most recent output injected, giving recurring jobs continuity
- # across runs (dedupe against what was already reported, continue
- # where the last run left off) without touching session history.
- is_self = False
- if isinstance(source_job_id, str) and source_job_id.strip().lower() == "self":
- source_job_id = str(job.get("id") or "")
- is_self = True
- elif source_job_id == job.get("id"):
- is_self = True
- # Guard against path traversal — valid job IDs are 12-char hex strings
- if not source_job_id or not all(c in "0123456789abcdef" for c in source_job_id):
- logger.warning(
- "context_from: skipping invalid job_id %r for job_id=%r name=%r%s",
- source_job_id,
- job.get("id"),
- job.get("name"),
- _cron_job_origin_log_suffix(job),
- )
- continue
- try:
- job_output_dir = output_dir / source_job_id
- if not job_output_dir.exists():
- continue # silent skip — no output yet
- output_files = sorted(
- job_output_dir.glob("*.md"),
- key=lambda f: f.stat().st_mtime,
- reverse=True,
- )
- if not output_files:
- continue # silent skip — no output yet
- latest_output = output_files[0].read_text(encoding="utf-8").strip()
- # Truncate to 8K characters to avoid prompt bloat
- _MAX_CONTEXT_CHARS = 8000
- if len(latest_output) > _MAX_CONTEXT_CHARS:
- latest_output = latest_output[:_MAX_CONTEXT_CHARS] + "\n\n[... output truncated ...]"
- if latest_output:
- if is_self:
- prompt = (
- "## Your previous run's output\n"
- "The following is this job's most recent output from its "
- "previous run. Use it for continuity: avoid repeating what "
- "was already reported, and continue where the last run "
- "left off.\n\n"
- f"```\n{latest_output}\n```\n\n"
- f"{prompt}"
- )
- else:
- prompt = (
- f"## Output from job '{source_job_id}'\n"
- "The following is the most recent output from a preceding "
- "cron job. Use it as context for your analysis.\n\n"
- f"```\n{latest_output}\n```\n\n"
- f"{prompt}"
- )
- has_injected_data = True
- else:
- continue # silent skip — empty output
- except (OSError, PermissionError) as e:
- logger.warning("context_from: failed to read output for job %r: %s", source_job_id, e)
- # silent skip — do not pollute the prompt with error messages
+ prompt, _ctx_injected = _inject_context_from(job, prompt)
+ has_injected_data = has_injected_data or _ctx_injected
- # Inject the job's durable notepad (per-job KV scratchpad surviving
- # scheduled wake-ups). Empty notepad renders as "" so jobs that never
- # use the feature get a byte-identical prompt.
+ # Durable per-job notepad; empty renders as "" so unused → byte-identical prompt.
from cron import notepad as cron_notepad
notepad_section = cron_notepad.render_notepad_section(str(job.get("id") or ""))
@@ -4899,8 +3692,6 @@ def _build_job_prompt(
prompt = f"{notepad_section}{prompt}"
has_injected_data = True
- # Always prepend cron execution guidance so the agent knows how
- # delivery works and can suppress delivery when appropriate.
cron_hint = (
"[IMPORTANT: You are running as a scheduled cron job. "
"DELIVERY: Your final response will be automatically delivered "
@@ -4929,92 +3720,18 @@ def _build_job_prompt(
user_prompt=user_prompt,
)
- from tools.skills_tool import skill_view
- from tools.skill_usage import bump_use
- from agent.skill_bundles import build_bundle_invocation_message, resolve_bundle_command_key
- from agent.skill_utils import normalize_skill_lookup_name
-
- parts = []
- skipped: list[str] = []
- for skill_name in skill_names:
- # Cron jobs historically accepted only skill names here, but the CLI/gateway
- # slash-command path lets bundles shadow skills with the same slug. Mirror
- # that behavior so `skills: ["my-bundle"]` expands bundle members instead
- # of being treated as a missing skill.
- bundle_key = resolve_bundle_command_key(skill_name.lstrip("/"))
- if bundle_key:
- bundle_payload = build_bundle_invocation_message(
- bundle_key,
- user_instruction="",
- task_id=str(job.get("id") or "") or None,
- )
- if bundle_payload:
- bundle_message, _loaded_bundle_skills, _missing_bundle_skills = bundle_payload
- if parts:
- parts.append("")
- parts.append(bundle_message)
- continue
- logger.warning(
- "Cron job '%s': bundle '%s' could not load any skills, skipping",
- job.get("name", job.get("id")),
- skill_name,
- )
- skipped.append(skill_name)
- continue
-
- try:
- loaded = json.loads(skill_view(normalize_skill_lookup_name(skill_name)))
- except (json.JSONDecodeError, TypeError):
- logger.warning("Cron job '%s': skill '%s' returned invalid JSON, skipping", job.get("name", job.get("id")), skill_name)
- skipped.append(skill_name)
- continue
- if not loaded.get("success"):
- error = loaded.get("error") or f"Failed to load skill '{skill_name}'"
- logger.warning("Cron job '%s': skill not found, skipping — %s", job.get("name", job.get("id")), error)
- skipped.append(skill_name)
- continue
-
- # Bump usage so the curator sees this skill as actively used.
- try:
- bump_use(skill_name, task_id=str(job.get("id") or "") or None)
- except Exception:
- logger.debug("Cron job: failed to bump skill usage for '%s'", skill_name, exc_info=True)
-
- content = str(loaded.get("content") or "").strip()
- if parts:
- parts.append("")
- parts.extend(
- [
- f'[IMPORTANT: The user has invoked the "{skill_name}" skill, indicating they want you to follow its instructions. The full skill content is loaded below.]',
- "",
- content,
- ]
- )
-
- if skipped:
- notice = (
- f"[IMPORTANT: The following skill(s) were listed for this job but could not be found "
- f"and were skipped: {', '.join(skipped)}. "
- f"Start your response with a brief notice so the user is aware, e.g.: "
- f"'⚠️ Skill(s) not found and skipped: {', '.join(skipped)}']"
- )
- parts.insert(0, notice)
-
+ parts = _load_cron_skill_parts(job, skill_names)
stable_prefix = None
if prompt:
from agent.skill_commands import append_user_instruction
parts.append("")
- # The skill blocks (and any skipped-skill notice) above are stable per
- # job config; the appended instruction carries the volatile per-run
- # data (cron hint + prompt + script output + run context). Declare
- # that boundary for the Anthropic cache planner (#81867).
+ # Skill blocks are stable per job config; the appended instruction is volatile per-run.
+ # Declare that boundary for the Anthropic cache planner.
stable_prefix = append_user_instruction(parts, prompt)
assembled = _scan_assembled_cron_prompt("\n".join(parts), job, has_skills=True)
if stable_prefix and len(assembled) > len(stable_prefix) and assembled.startswith(stable_prefix):
- # Guarded because the injection scanner may sanitize (mutate) the
- # assembled bytes; a mismatch simply falls back to whole-message
- # caching.
+ # Guarded: the scanner may mutate the bytes; mismatch → whole-message caching.
from agent.prompt_cache_boundary import register_stable_prefix
register_stable_prefix(stable_prefix)
@@ -5029,54 +3746,22 @@ def _scan_assembled_cron_prompt(
has_injected_data: bool = False,
user_prompt: Optional[str] = None,
) -> str:
- """Scan the fully-assembled cron prompt for injection patterns. Raises
- ``CronPromptInjectionBlocked`` when a match fires so ``run_job`` can
- surface a clear refusal to the operator.
+ """Scan the assembled cron prompt for injection; raise ``CronPromptInjectionBlocked`` on a hit.
- Plugs the #3968 gap: ``_scan_cron_prompt`` runs on the user-supplied
- prompt at create/update, but skill content is loaded from disk at
- runtime and was never scanned. Since cron runs non-interactively
- (auto-approves tool calls), a malicious skill carrying an injection
- payload bypassed every gate.
-
- Two pattern tiers, selected by what the assembled prompt CONTAINS,
- not just whether skills are attached:
-
- - When the assembled prompt is essentially the user prompt + the cron
- hint (no skills, no injected data), the STRICT ``_scan_cron_prompt``
- patterns apply: a bare ``rm -rf /`` in a small directive prompt is a
- smoking gun, not prose.
- - When the assembled prompt includes runtime-loaded content — skill
- markdown (``has_skills=True``) or DATA injected from a job script's
- stdout / an upstream job's output (``has_injected_data=True``) — the
- LOOSER ``_scan_cron_skill_assembled`` pattern set is used: only
- unambiguous prompt-injection directives block; command-shape
- patterns are dropped and invisible unicode is sanitized (stripped +
- logged) rather than blocked, to avoid false-positives that
- permanently kill a job. Skill bodies are vetted at install time by
- ``skills_guard.py``; script output is produced by operator-authored
- code, the same trust class — and data feeds (e.g. a triage bot
- ingesting bug reports) legitimately quote dangerous commands.
-
- When the looser tier is selected because of injected data only,
- ``user_prompt`` (the raw, pre-assembly prompt) is additionally scanned
- with the STRICT set so the user-authored surface keeps the full
- create/update-time guarantee at runtime (defense-in-depth for legacy
- jobs that predate the create-time scanner).
+ Needed because skill content is loaded from disk at runtime (never scanned at create/update)
+ and cron auto-approves tool calls. Tier is chosen by what the prompt CONTAINS: user prompt +
+ hint only → STRICT ``_scan_cron_prompt``; skills or injected data → LOOSER
+ ``_scan_cron_skill_assembled`` (command-shape patterns dropped, invisible unicode sanitized not
+ blocked, so a false positive cannot permanently kill a job). With injected data but no skills,
+ ``user_prompt`` is additionally scanned STRICT (defense-in-depth for legacy jobs).
"""
from tools.cronjob_tools import _scan_cron_prompt, _scan_cron_skill_assembled
if has_skills or has_injected_data:
- # Runtime-loaded content (vetted skill markdown and/or data from
- # operator-authored scripts) legitimately contains command-shape
- # strings. Invisible unicode is sanitized (not blocked) so a stray
- # zero-width space can't permanently kill the job; the cleaned
- # prompt is what actually runs.
+ # The cleaned (sanitized) prompt is what actually runs.
cleaned, scan_error = _scan_cron_skill_assembled(assembled)
assembled = cleaned
if not scan_error and not has_skills and user_prompt:
- # Data-injection path: keep the strict guarantee on the
- # user-authored prompt itself.
scan_error = _scan_cron_prompt(user_prompt)
else:
scan_error = _scan_cron_prompt(assembled)
@@ -5092,33 +3777,18 @@ def _scan_assembled_cron_prompt(
def _guard_job_credential_exfil(job: dict) -> None:
- """Fail closed if a job's stored provider/base_url pair would exfiltrate a
- credential (F8 runtime backstop; CWE-200/CWE-522).
+ """Fail closed (RuntimeError) if the stored provider/base_url pair could exfiltrate a key.
- The model-callable cron tool validates this on create/update, but a job
- persisted before that guard — or written directly to the jobs store —
- reaches the scheduler's provider-resolution sink unchecked. Re-validate the
- EFFECTIVE stored pair with the same guard the tool uses, so a named
- provider's stored key is never paired with an off-host base_url at fire
- time. Raises ``RuntimeError`` (caught by the run_job failure path → the run
- is aborted and reported) when the pair is unsafe; returns ``None`` otherwise.
-
- Fallback providers come from operator config, not the model-callable job, so
- they are trusted and validated by the caller, not here.
+ Runtime backstop: jobs persisted before the create/update guard, or written directly to the
+ store, reach provider resolution unchecked. Fallback providers come from operator config and
+ are validated by the caller, not here.
"""
try:
from tools.cronjob_tools import _validate_cron_base_url
err = _validate_cron_base_url(job.get("provider"), job.get("base_url"))
except Exception as exc:
- # Fail CLOSED: this is the last guard before provider resolution, so an
- # unexpected validator/import error must not silently allow an unvetted
- # pair through. A job that carries no base_url override cannot exfiltrate
- # a stored credential via this path (there is nothing to validate, and
- # the validator would return None), so it still runs — that keeps the
- # overwhelmingly-common no-override jobs from wedging on an unrelated
- # error. But any job that DID set a base_url is refused until the
- # validator can actually vet the pair. Operator fallback providers come
- # from config, not the job, so they are unaffected.
+ # Fail CLOSED on validator/import errors — but only for jobs WITH a base_url override; a job
+ # without one cannot exfiltrate via this path, so it still runs.
if job.get("base_url"):
err = (
f"could not validate provider/base_url pair "
@@ -5140,13 +3810,8 @@ def _guard_job_credential_exfil(job: dict) -> None:
def _block_and_pause_job(
job_id: str, job_name: str, reason: str
) -> tuple[bool, str, str, Optional[str]]:
- """Fail a run closed and pause the job so it stops being scheduled.
-
- Used for job shapes that can never run (a5e29e688dc0). Returning an error
- alone is not enough — an unrunnable job that stays enabled re-fires on
- every tick forever. Pausing writes ``paused_at``/``paused_reason``, giving
- an auditable record of why the scheduler stopped it.
- """
+ """Fail a run closed and pause the job: an unrunnable job left enabled re-fires every tick
+ forever; ``paused_at``/``paused_reason`` give an auditable record."""
from cron.jobs import pause_job
logger.error("Job '%s': %s", job_id, reason)
@@ -5167,124 +3832,76 @@ def _block_and_pause_job(
return False, doc, alert, reason
-# Marker prefix stamped into the error string returned by ``run_job`` when the
-# pre-dispatch configuration validation (T1-26) refuses to run the agent.
-# ``run_one_job`` keys off it to record ``last_status='blocked_config'`` and to
-# apply the alert-once dedup. The ``:silent`` variant means "already alerted on
-# a previous tick — do not deliver again".
+# Error-string prefixes from ``run_job``; ``run_one_job`` keys off them for last_status and the
+# alert-once dedup. ``:silent`` = already alerted on a previous tick — do not deliver again.
BLOCKED_CONFIG_MARKER = "[blocked_config]"
BLOCKED_CONFIG_SILENT_MARKER = "[blocked_config:silent]"
-# Marker prefix for a #44585 drift-guard skip. Same alert-once contract as
-# blocked_config: run_one_job keys off it to record last_status and the
-# ``:silent`` variant means "already alerted on a previous tick — do not
-# deliver again" (the drift_alerted bit on the job record, #73506 shape).
+# Drift-guard skip: same contract (drift_alerted bit on the job record).
DRIFT_SKIP_MARKER = "[drift_skip]"
DRIFT_SKIP_SILENT_MARKER = "[drift_skip:silent]"
+_TRANSIENT_NET_EXC_NAMES = frozenset({
+ "ConnectError", "ConnectTimeout", "ReadTimeout", "WriteTimeout", "PoolTimeout", "NetworkError",
+ "TimeoutException", "ClientConnectorError", "ClientConnectorDNSError", "ServerTimeoutError",
+ "ClientOSError",
+})
+_DNS_FAILURE_NEEDLES = ("nodename nor servname", "name or service not known")
+_TRANSIENT_OSERROR_NEEDLES = _DNS_FAILURE_NEEDLES + (
+ "temporary failure in name resolution", "network is unreachable",
+)
+_TRANSIENT_HTTP_NEEDLES = _TRANSIENT_OSERROR_NEEDLES + (
+ "failed to resolve", "connection refused", "timed out", "timeout",
+)
+_TRANSIENT_ERRNOS = frozenset({
+ errno.ECONNREFUSED, errno.ECONNRESET, errno.EHOSTUNREACH, errno.ENETUNREACH, errno.ENETDOWN,
+ errno.ETIMEDOUT, errno.EAGAIN,
+})
+
def _is_transient_provider_resolve_error(exc: BaseException) -> bool:
- """True when primary provider resolution failed for a transient network reason.
+ """True when primary provider resolution failed for a transient network reason (DNS blip,
+ ConnectError...). Must be eligible for ``fallback_providers`` like AuthError, else a healthy
+ fallback rung is never tried and the job dies before the first model call."""
+ import socket
- Agent crons resolve OAuth credentials (token refresh / discovery) before the
- agent loop starts. A short DNS outage (Cloudflare WARP / macOS resolver blip)
- surfaces as httpx/httpcore ConnectError or raw OSError errno 8 ("nodename nor
- servname provided") and must be eligible for ``fallback_providers`` the same
- way AuthError already is — otherwise a healthy XAI_API_KEY / Anthropic rung
- never gets tried and the whole job dies before the first model call.
- """
- # Walk the cause chain; scheduler wraps raw transport errors.
+ # gaierror carries EAI_* codes, plain OSError carries errno — never mix the namespaces (raw
+ # literals like {8, 7, 11} are macOS-only and wrong on Linux).
+ eai_transient = {
+ getattr(socket, n) for n in ("EAI_NONAME", "EAI_AGAIN", "EAI_FAIL", "EAI_NODATA")
+ if hasattr(socket, n)
+ }
+ # Walk the cause chain; the scheduler wraps raw transport errors.
seen: set[int] = set()
cur: Optional[BaseException] = exc
while cur is not None and id(cur) not in seen:
seen.add(id(cur))
- name = type(cur).__name__
module = type(cur).__module__ or ""
msg = str(cur).lower()
- # Explicit transport classes from httpx/httpcore/aiohttp.
- if name in {
- "ConnectError",
- "ConnectTimeout",
- "ReadTimeout",
- "WriteTimeout",
- "PoolTimeout",
- "NetworkError",
- "TimeoutException",
- "ClientConnectorError",
- "ClientConnectorDNSError",
- "ServerTimeoutError",
- "ClientOSError",
- }:
+ if type(cur).__name__ in _TRANSIENT_NET_EXC_NAMES:
+ return True
+ if any(m in module for m in ("httpx", "httpcore", "aiohttp")) and any(
+ needle in msg for needle in _TRANSIENT_HTTP_NEEDLES
+ ):
return True
- if "httpx" in module or "httpcore" in module or "aiohttp" in module:
- if any(
- needle in msg
- for needle in (
- "nodename nor servname",
- "name or service not known",
- "temporary failure in name resolution",
- "failed to resolve",
- "connection refused",
- "network is unreachable",
- "timed out",
- "timeout",
- )
- ):
- return True
if isinstance(cur, OSError):
- # Platform-safe classification (the raw-literal set {8, 7, 11, ...}
- # from the first revision mixed macOS getaddrinfo constants with
- # errno values and does not hold on Linux — see PR review).
- # socket.gaierror carries getaddrinfo codes (EAI_*), plain OSError
- # carries errno; compare each against its own constant namespace.
- import errno as _errno
- import socket as _socket
-
- if isinstance(cur, _socket.gaierror):
- _eai_transient = {
- getattr(_socket, _n)
- for _n in ("EAI_NONAME", "EAI_AGAIN", "EAI_FAIL", "EAI_NODATA")
- if hasattr(_socket, _n)
- }
- if cur.errno in _eai_transient:
+ if isinstance(cur, socket.gaierror):
+ if cur.errno in eai_transient:
return True
- else:
- err_no = getattr(cur, "errno", None)
- if err_no in {
- _errno.ECONNREFUSED,
- _errno.ECONNRESET,
- _errno.EHOSTUNREACH,
- _errno.ENETUNREACH,
- _errno.ENETDOWN,
- _errno.ETIMEDOUT,
- _errno.EAGAIN,
- }:
- return True
- if any(
- needle in msg
- for needle in (
- "nodename nor servname",
- "name or service not known",
- "temporary failure in name resolution",
- "network is unreachable",
- )
- ):
+ elif getattr(cur, "errno", None) in _TRANSIENT_ERRNOS:
return True
- # Bare RuntimeError/Exception that already carries the DNS text
- # (format_runtime_provider_error sometimes surfaces the raw message).
- if "nodename nor servname" in msg or "name or service not known" in msg:
+ if any(needle in msg for needle in _TRANSIENT_OSERROR_NEEDLES):
+ return True
+ # Bare exceptions that carry the raw DNS text (format_runtime_provider_error).
+ if any(needle in msg for needle in _DNS_FAILURE_NEEDLES):
return True
cur = cur.__cause__ or cur.__context__
return False
def _cron_preflight_enabled(cfg: dict) -> bool:
- """Whether cron pre-dispatch configuration validation is enabled.
-
- Default ON; only the literal boolean ``false`` under ``cron.preflight``
- opts out (mirrors ``cron_model_drift_guard_enabled`` semantics).
- """
+ """Preflight is ON unless ``cron.preflight`` is literally ``false``."""
cron_cfg = (cfg or {}).get("cron")
if not isinstance(cron_cfg, dict):
return True
@@ -5292,15 +3909,9 @@ def _cron_preflight_enabled(cfg: dict) -> bool:
def _preflight_check_provider_key(job: dict, cfg: dict) -> Optional[str]:
- """READ-ONLY probe: would provider resolution fail for lack of a key?
-
- Mirrors the effective requested-provider computation from run_job's
- resolution block without any side effects on the run. When a fallback
- chain is configured the check is skipped entirely — the existing
- auth-fallback path may legitimately rescue a missing primary key, so
- blocking here would break that contract (and burning zero LLM calls is
- already guaranteed by the fallback resolution being config-local).
- """
+ """READ-ONLY probe: would provider resolution fail for lack of a key? Mirrors run_job's
+ requested-provider computation. Skipped when a fallback chain exists — auth-fallback may
+ legitimately rescue a missing primary key, so blocking here would break that contract."""
try:
if get_fallback_chain(cfg):
return None
@@ -5332,27 +3943,18 @@ def _preflight_check_provider_key(job: dict, cfg: dict) -> Optional[str]:
f"{job.get('id')} --provider `."
)
except Exception:
- # Non-auth resolution errors (bad config shapes, network probes,
- # import issues) are NOT a missing-credential condition — let the
- # real resolution path handle and report them as before.
- return None
+ return None # non-auth errors are not a missing-credential verdict; real path reports them
return None
def _primary_profile_routes_for_current_home() -> list:
- """Primary gateway ``profile_routes`` that target the profile currently
- being served, or ``[]`` (also when this IS the primary home).
+ """Primary gateway ``profile_routes`` targeting the profile being served; ``[]`` if this IS the
+ primary home.
- Under ``gateway.multiplex_profiles`` a satellite profile's cron jobs are
- ticked by the primary gateway's in-process ticker (#69377) and delivered
- through the primary gateway's live adapters — the satellite home never
- holds the platform credentials itself (giving it a token of its own is a
- ``duplicate_credential`` fatal). Reads the primary config.yaml directly
- (both the top-level and nested ``gateway.`` forms) instead of
- ``load_gateway_config()`` so no primary platform config leaks into this
- process's environment. Shared by the preflight rescue (#97476) and the
- delivery-time shared-transport resolver (#101113) so route semantics
- cannot drift between the two halves.
+ Satellite crons are ticked and delivered by the primary gateway (a satellite holding its own
+ token is a ``duplicate_credential`` fatal). Reads the primary config.yaml directly (top-level or
+ nested ``gateway.``) instead of ``load_gateway_config()`` so no primary platform config leaks
+ into this process. Shared by preflight rescue and delivery-time resolution so they cannot drift.
"""
try:
from hermes_constants import get_default_hermes_root, get_hermes_home
@@ -5387,16 +3989,12 @@ def _primary_profile_routes_for_current_home() -> list:
if route.enabled and profile_matches_home(route.profile)
]
except Exception:
- logger.debug(
- "primary-gateway profile-route lookup unavailable",
- exc_info=True,
- )
+ logger.debug("primary-gateway profile-route lookup unavailable", exc_info=True)
return []
def _delivery_platform_routed_from_primary_gateway(platform_name: str) -> bool:
- """True when the primary gateway routes this platform to the profile the
- scheduler is currently serving (preflight rescue, #97476)."""
+ """True when the primary gateway routes this platform to the profile being served."""
platform_key = platform_name.lower()
return any(
str(route.platform).lower() == platform_key
@@ -5405,19 +4003,11 @@ def _delivery_platform_routed_from_primary_gateway(platform_name: str) -> bool:
class SharedRouteAdapters:
- """Read-only adapter map for a credentialless satellite profile (#101113).
+ """Read-only adapter map for a credentialless satellite profile.
- A satellite under ``gateway.profile_routes`` owns no bot credential and so
- has no adapter map of its own; its inbound traffic arrives on the PRIMARY
- adapter and is routed to it by an exact route. Its cron output must go
- back out the same transport — but ONLY for targets an enabled primary
- route maps to this profile. ``get(platform, target)`` resolves the primary
- adapter iff the route matcher used by inbound routing
- (``ProfileRoute.matches``) accepts the target's ``chat_id``/``thread_id``;
- every other lookup is a miss, so an unmatched target, a disabled route, or
- a route naming another profile still fails closed (never the default bot).
- A plain ``get(platform)`` (no target) is always a miss: routing is
- per-target, not per-platform.
+ ``get(platform, target)`` resolves the PRIMARY adapter iff the inbound route matcher
+ (``ProfileRoute.matches``) accepts the target; anything else (unmatched target, disabled route,
+ other profile, or target-less ``get(platform)``) is a miss — fail closed, never the default bot.
"""
def __init__(self, primary_adapters, routes) -> None:
@@ -5445,24 +4035,15 @@ class SharedRouteAdapters:
if route.matches(str(route.platform), chat_id=chat_id, thread_id=thread_id):
return adapter
return default
- return False
def _preflight_check_delivery(job: dict) -> Optional[str]:
- """Check the job's delivery target(s) resolve to configured platforms.
+ """Check delivery targets resolve to configured platforms.
- ``local``/``origin`` (and the ``all`` routing token) need no gateway
- credentials and are never checked — a deliver=local job must not pay a
- gateway-config load. For concrete platform targets, an unknown platform
- always blocks; a known platform additionally blocks when the gateway
- config is loadable and reports it unconnected (enabled + credentials —
- the same source `cron_delivery_targets` uses). Gateway-config load
- failures fail OPEN so a transient config hiccup never wedges delivery
- that would have worked.
-
- ``failure_deliver`` is checked with the same rules: a typo'd failure
- platform would otherwise only surface when a failure occurs — exactly
- when the notice must not be lost (NS-788 follow-up).
+ ``local``/``origin``/``all`` are never checked (no gateway-config load). Unknown platform always
+ blocks; known platform blocks only if the gateway config loads AND reports it unconnected.
+ Config load failures fail OPEN. ``failure_deliver`` is checked with the same rules: a typo'd
+ failure platform would otherwise only surface when a failure occurs (NS-788).
"""
deliver_value = _normalize_deliver_value(job.get("deliver", "local"))
failure_deliver_value = _normalize_deliver_value(
@@ -5477,9 +4058,7 @@ def _preflight_check_delivery(job: dict) -> Optional[str]:
part = part.strip()
if not part or part.lower() in {"local", "origin", "all"}:
continue
- # bot-chat targets need no gateway credentials — they deliver via a
- # local chat subprocess. Unknown-profile failures surface per run in
- # last_delivery_error (and are validated at create time).
+ # bot-chat targets deliver via a local subprocess; failures surface in last_delivery_error.
if parse_bot_chat_deliver_token(part) is not None:
continue
platform_parts.append(part.split(":", 1)[0].strip())
@@ -5499,9 +4078,7 @@ def _preflight_check_delivery(job: dict) -> Optional[str]:
from gateway.config import load_gateway_config
gateway_config = load_gateway_config()
- connected = {
- p.value for p in gateway_config.get_connected_platforms()
- }
+ connected = {p.value for p in gateway_config.get_connected_platforms()}
connected |= _relay_fronted_delivery_platforms(connected)
except Exception:
logger.debug(
@@ -5510,10 +4087,7 @@ def _preflight_check_delivery(job: dict) -> Optional[str]:
)
return None # fail-open
if platform_name.lower() not in connected:
- # Multiplex escape hatch: a satellite profile whose deliveries
- # are routed by the primary gateway's profile_routes is served
- # by the primary's adapters, so its own unconnected reading is
- # a false block (#97476).
+ # Multiplex: a satellite served by the primary's adapters reads unconnected — no block.
if _delivery_platform_routed_from_primary_gateway(platform_name):
continue
return (
@@ -5525,15 +4099,8 @@ def _preflight_check_delivery(job: dict) -> Optional[str]:
def _preflight_check_skills(job: dict) -> Optional[str]:
- """Check attached skills report ready (no missing required env/commands).
-
- Consults the same ``readiness_status`` payload ``skill_view`` computes
- for interactive use. Skills that fail to load at all are left to the
- existing skipped-skill handling in ``_build_job_prompt`` (fail-open):
- this check only blocks on an affirmative "setup needed" verdict, i.e.
- the skill exists but its required environment is missing — a run that
- is guaranteed to misfire.
- """
+ """Block only on an affirmative ``setup_needed`` verdict from ``skill_view``; skills that fail
+ to load fall through to ``_build_job_prompt``'s skipped-skill handling (fail-open)."""
skills = job.get("skills")
if skills is None:
legacy = job.get("skill")
@@ -5581,21 +4148,9 @@ def _preflight_check_skills(job: dict) -> Optional[str]:
def _preflight_job_config(job: dict, cfg: dict) -> Optional[str]:
- """Pre-dispatch configuration validation (T1-26).
-
- Returns a human-readable reason when the job's configuration cannot
- produce a successful run — missing provider API key, unconfigured
- delivery platform, or an attached skill with missing required env —
- so the caller can refuse the run BEFORE any agent machinery is
- constructed and no LLM call is burned. Returns ``None`` when the
- configuration validates (or when a check cannot be evaluated: every
- check fails open, so preflight can only ever block on an affirmative
- misconfiguration verdict).
-
- Same fail-before-spend spirit as the #44585 drift guard and the
- fail-loud-on-hidden-tools direction in #27948; alert dedup follows the
- alert-once pattern from the dead-pin auto-pause (#73506).
- """
+ """Pre-dispatch validation: return a reason (missing key, unconfigured delivery, unready skill)
+ so the caller refuses BEFORE building agent machinery or burning an LLM call. Every check fails
+ open — preflight blocks only on an affirmative misconfiguration verdict."""
for name, check in (
("provider_key", lambda: _preflight_check_provider_key(job, cfg)),
("skills", lambda: _preflight_check_skills(job)),
@@ -5604,9 +4159,7 @@ def _preflight_job_config(job: dict, cfg: dict) -> Optional[str]:
try:
reason = check()
except Exception:
- logger.debug(
- "preflight check %s raised — failing open", name, exc_info=True
- )
+ logger.debug("preflight check %s raised — failing open", name, exc_info=True)
continue
if reason:
return reason
@@ -5639,11 +4192,7 @@ def _run_cron_cleanup_with_timeout(
timeout_seconds: Optional[float] = None,
) -> bool:
"""Run fallible post-run cleanup without permanently wedging a cron ID."""
- timeout = (
- _cron_cleanup_timeout_seconds()
- if timeout_seconds is None
- else float(timeout_seconds)
- )
+ timeout = (_cron_cleanup_timeout_seconds() if timeout_seconds is None else float(timeout_seconds))
if timeout <= 0:
try:
cleanup()
@@ -5663,10 +4212,8 @@ def _run_cron_cleanup_with_timeout(
finally:
done.set()
- # A daemon thread is deliberate: unlike ThreadPoolExecutor workers it is
- # not joined by Python's interpreter-exit hook if the cleanup target never
- # returns. The scheduler can release its dispatch guard and the gateway can
- # still shut down normally.
+ # Daemon thread is deliberate: unlike ThreadPoolExecutor workers it is not joined at interpreter
+ # exit if cleanup never returns, so the gateway can still shut down.
worker = threading.Thread(
target=_runner,
name=f"cron-cleanup-{job_id}",
@@ -5688,12 +4235,8 @@ def _run_cron_cleanup_with_timeout(
class _BoundedCronSessionDB:
- """Proxy SessionDB cleanup calls through the cron cleanup timeout.
-
- After the first failed or timed-out operation the proxy fails subsequent
- calls immediately. A damaged SQLite connection should leak at most one
- abandoned cleanup worker, not one worker per finalization step.
- """
+ """Proxy SessionDB cleanup calls through the cron cleanup timeout; after the first failure or
+ timeout all later calls fail immediately (a damaged connection leaks at most one worker)."""
def __init__(self, session_db, job_id: str):
self._session_db = session_db
@@ -5727,9 +4270,7 @@ class _BoundedCronSessionDB:
error = result.get("error")
if error is not None:
raise error
- # No exception reached the caller and the operation still did
- # not complete: this is the timeout path. Disable the damaged
- # connection so later finalization steps fail immediately.
+ # No error yet not complete == timeout: disable so later steps fail fast.
self._disabled = True
raise TimeoutError(f"session finalization method {name} timed out")
return result.get("value")
@@ -5737,6 +4278,753 @@ class _BoundedCronSessionDB:
return _bounded
+def _job_doc_header(job_name: str, job_id: str, now_iso: str, mode: str) -> str:
+ """Common markdown header for the short-circuit run docs (no_agent / monitor)."""
+ return (
+ f"# Cron Job: {job_name}\n\n"
+ f"**Job ID:** {job_id}\n"
+ f"**Run Time:** {now_iso}\n"
+ f"**Mode:** {mode}\n"
+ )
+
+
+def _resolve_job_workdir(job: dict, job_id: str) -> Optional[str]:
+ """Configured job workdir, or None when unset / no longer a directory (logged)."""
+ workdir = (job.get("workdir") or "").strip() or None
+ if workdir and not Path(workdir).is_dir():
+ logger.warning(
+ "Job '%s': configured workdir %r no longer exists — running without it",
+ job_id, workdir,
+ )
+ return None
+ return workdir
+
+
+def _run_no_agent_job(
+ job: dict, job_id: str, job_name: str, cancel_event,
+) -> tuple[bool, str, str, Optional[str]]:
+ """no_agent short-circuit — the script IS the job (no AIAgent, no tokens). stdout → delivered
+ verbatim; empty stdout or wakeAgent=false → silent success; non-zero exit/timeout → error alert.
+ """
+ # Load .env first so auto-delivery can resolve *_HOME_CHANNEL: the agent path's per-run dotenv
+ # reload never runs for no_agent jobs. Does not override existing values.
+ try:
+ from hermes_cli.env_loader import load_hermes_dotenv
+
+ load_hermes_dotenv(hermes_home=_get_hermes_home())
+ except Exception:
+ logger.debug("Job '%s': no_agent .env reload failed", job_id, exc_info=True)
+
+ script_path = job.get("script")
+ # Legacy/hand-edited no_agent job without a script: pause it, or it re-fires every tick.
+ if not str(script_path or "").strip():
+ from cron.jobs import NO_AGENT_WITHOUT_SCRIPT_ERROR
+
+ return _block_and_pause_job(job_id, job_name, NO_AGENT_WITHOUT_SCRIPT_ERROR)
+
+ # Pass workdir as subprocess cwd; never os.chdir() (leaks into concurrent gateway sessions).
+ _job_workdir = _resolve_job_workdir(job, job_id)
+ try:
+ ok, output = _run_job_script_with_claim_heartbeat(
+ job, script_path, workdir=_job_workdir, cancel_event=cancel_event,
+ )
+ except Exception as exc:
+ logger.exception("Job '%s': script execution raised unexpectedly", job_id)
+ ok, output = False, f"Script execution failed: {exc}"
+
+ now_iso = _hermes_now().strftime("%Y-%m-%d %H:%M:%S")
+ header = _job_doc_header(job_name, job_id, now_iso, "no_agent (script)")
+
+ if not ok:
+ # Deliver the error: a silently broken watchdog is the worst-case outcome.
+ alert = (
+ f"⚠ Cron watchdog '{job_name}' script failed\n\n"
+ f"{output}\n\n"
+ f"Time: {now_iso}"
+ )
+ return False, f"{header}**Status:** script failed\n\n{output}\n", alert, output
+
+ # wakeAgent=false is a silent signal, same as empty stdout.
+ if not _parse_wake_gate(output):
+ logger.info("Job '%s' (no_agent): wakeAgent=false gate — silent run", job_id)
+ return True, f"{header}**Status:** silent (wakeAgent=false)\n", SILENT_MARKER, None
+
+ if not output.strip():
+ logger.info("Job '%s' (no_agent): empty stdout — silent run", job_id)
+ return True, f"{header}**Status:** silent (empty output)\n", SILENT_MARKER, None
+
+ return True, f"{header}\n---\n\n{output}\n", output, None
+
+
+def _apply_monitor_gate(
+ job: dict, job_id: str, job_name: str, extra_prompt: Optional[str],
+) -> tuple[Optional[tuple], Optional[str]]:
+ """Monitor gate (hash-suppressed change detection). Must run BEFORE any agent machinery so an
+ unchanged tick costs no LLM/delivery. Returns ``(early_result | None, extra_prompt)``; when
+ early_result is None, extra_prompt may carry the injected monitor context.
+ """
+ from cron.monitor import check_monitor, job_has_monitor
+
+ if not job_has_monitor(job):
+ return None, extra_prompt
+ _mon = check_monitor(job)
+ _mon_now = _hermes_now().strftime("%Y-%m-%d %H:%M:%S")
+ header = _job_doc_header(job_name, job_id, _mon_now, "monitor")
+ if not _mon.ok:
+ # Source failure is an ERROR, never a change: alert so a broken monitor can't silently
+ # stop watching. Stored hash untouched.
+ logger.error("Job '%s': monitor source failed: %s", job_id, _mon.error)
+ _mon_alert = (
+ f"⚠ Cron monitor '{job_name}' source failed\n\n"
+ f"{_mon.error}\n\n"
+ f"Time: {_mon_now}"
+ )
+ return (
+ False, f"{header}**Status:** monitor source failed\n\n{_mon.error}\n", _mon_alert, _mon.error,
+ ), extra_prompt
+ if not _mon.changed:
+ # Unchanged: silent no_change tick (ledger doc kept; SILENT_MARKER blocks delivery).
+ logger.info("Job '%s': monitor output unchanged — suppressing agent run", job_id)
+ return (
+ True, f"{header}**Status:** no_change (agent run suppressed)\n", SILENT_MARKER, None,
+ ), extra_prompt
+ # Changed (or first run): inject monitor context via the per-run seam, then normal agent run.
+ if _mon.context_block:
+ extra_prompt = (
+ f"{_mon.context_block}\n\n{extra_prompt}" if extra_prompt else _mon.context_block
+ )
+ return None, extra_prompt
+
+
+@dataclass
+class _CronJobConfig:
+ """Config-derived inputs for one agent-backed cron run."""
+
+ cfg: dict
+ model: str
+ model_cfg: Any
+ cron_default_provider: str
+
+
+def _load_cron_job_config(job: dict, job_id: str, job_name: str) -> _CronJobConfig:
+ """Load config.yaml and resolve the run's model.
+
+ Precedence: per-job override > cron.model (fleet default) > HERMES_MODEL > config ``model:``.
+ Re-read every tick (no cache) so ``hermes cron edit --model`` takes effect next tick. An axis
+ resolved from cron.model / cron.model_provider is explicit, so the drift guard skips it.
+ """
+ model = job.get("model") or os.getenv("HERMES_MODEL") or ""
+ _cron_default_provider = ""
+ _cfg: dict = {}
+ _model_cfg: Any = {}
+ try:
+ from hermes_cli.config import read_user_config_raw
+ _cfg_path = str(_get_hermes_home() / "config.yaml")
+ if os.path.exists(_cfg_path):
+ _cfg = read_user_config_raw(Path(_cfg_path))
+ # Honor administrator-pinned managed scope (fail-open; no-op without managed scope).
+ with contextlib.suppress(Exception):
+ from hermes_cli import managed_scope
+ _cfg = managed_scope.apply_managed_overlay(_cfg)
+ _cfg = _expand_env_vars(_cfg)
+ # Coerce null to {} so a falsy default never clobbers a resolved env value.
+ _model_cfg = _cfg.get("model") or {}
+ _cron_cfg_for_model = _cfg.get("cron") or {}
+ _cron_default_model = ""
+ if isinstance(_cron_cfg_for_model, dict):
+ _cron_default_model = str(_cron_cfg_for_model.get("model") or "").strip()
+ _cron_default_provider = str(_cron_cfg_for_model.get("model_provider") or "").strip()
+ if not job.get("model"):
+ if _cron_default_model:
+ model = _cron_default_model
+ else:
+ # Shared with Desktop's impact summary so both compare against the same model.
+ _, _global_model = resolve_cron_model_drift_defaults(_cfg)
+ if _global_model:
+ model = _global_model
+ except Exception as e:
+ logger.warning("Job '%s': failed to load config.yaml, using defaults: %s", job_id, e)
+
+ # Fail fast: an empty model otherwise reaches the provider as an opaque 400.
+ if not (isinstance(model, str) and model.strip()):
+ raise RuntimeError(
+ f"Cron job '{job_name}' has no model configured "
+ f"(job.model={job.get('model')!r}, "
+ f"HERMES_MODEL={os.getenv('HERMES_MODEL', '')!r}, "
+ "config.yaml model.default missing or empty). "
+ f"Set a per-job model via "
+ f"`hermes cron edit {job_id} --model ` or set a "
+ "default with `hermes model `."
+ )
+
+ with contextlib.suppress(Exception):
+ from hermes_constants import apply_ipv4_preference
+ _net_cfg = _cfg.get("network", {})
+ if isinstance(_net_cfg, dict) and _net_cfg.get("force_ipv4"):
+ apply_ipv4_preference(force=True)
+ return _CronJobConfig(_cfg, model, _model_cfg, _cron_default_provider)
+
+
+def _load_prefill_messages(cfg: dict, job_id: str) -> Optional[list]:
+ """Prefill messages from env or config.yaml (top-level key canonical; agent.* is legacy)."""
+ agent_cfg = cfg.get("agent", {}) if isinstance(cfg.get("agent", {}), dict) else {}
+ prefill_file = (
+ os.getenv("HERMES_PREFILL_MESSAGES_FILE", "")
+ or cfg.get("prefill_messages_file", "")
+ or agent_cfg.get("prefill_messages_file", "")
+ )
+ if not prefill_file:
+ return None
+ pfpath = Path(prefill_file).expanduser()
+ if not pfpath.is_absolute():
+ pfpath = _get_hermes_home() / pfpath
+ if not pfpath.exists():
+ return None
+ try:
+ with open(pfpath, "r", encoding="utf-8") as _pf:
+ prefill_messages = json.load(_pf)
+ return prefill_messages if isinstance(prefill_messages, list) else None
+ except Exception as e:
+ logger.warning("Job '%s': failed to parse prefill messages file '%s': %s", job_id, pfpath, e)
+ return None
+
+
+def _preflight_or_block(job: dict, job_id: str, job_name: str, cfg: dict) -> Optional[tuple]:
+ """Pre-dispatch config validation: refuse unrunnable jobs (missing key, unready skill,
+ unconfigured delivery) BEFORE AIAgent is built. run_one_job keys off BLOCKED_CONFIG_MARKER to
+ record blocked_config and alert once (`preflight_alerted` bit). Must run after the wake gate so
+ silent ticks stay silent. Opt-out: `cron.preflight: false`. Returns failure tuple or None.
+ """
+ _pf_reason = None
+ try:
+ if _cron_preflight_enabled(cfg):
+ _pf_reason = _preflight_job_config(job, cfg)
+ if not _pf_reason and job.get("preflight_alerted"):
+ # Config healthy again: clear alert-once marker so a future break re-alerts.
+ with contextlib.suppress(Exception):
+ from cron.jobs import clear_preflight_alerted
+ clear_preflight_alerted(job_id)
+ except Exception:
+ # Fail open: the validator must never take down a runnable job.
+ logger.debug("Job '%s': preflight validation errored — failing open", job_id, exc_info=True)
+ _pf_reason = None
+ if not _pf_reason:
+ return None
+
+ logger.warning(
+ "Job '%s' (ID: %s): BLOCKED by pre-dispatch config "
+ "validation — %s (no LLM call was made)",
+ job_name, job_id, _pf_reason,
+ )
+ already_alerted = False
+ try:
+ from cron.jobs import mark_preflight_alerted
+ already_alerted = mark_preflight_alerted(job_id)
+ except Exception:
+ logger.debug("Job '%s': could not persist preflight alert marker", job_id, exc_info=True)
+ marker = BLOCKED_CONFIG_SILENT_MARKER if already_alerted else BLOCKED_CONFIG_MARKER
+ blocked_doc = (
+ f"# Cron Job: {job_name}\n\n"
+ f"**Job ID:** {job_id}\n"
+ f"**Run Time:** {_hermes_now().strftime('%Y-%m-%d %H:%M:%S')}\n"
+ f"**Status:** BLOCKED (configuration)\n\n"
+ "Pre-dispatch validation found a configuration problem and "
+ "the agent was NOT run (no tokens spent).\n\n"
+ f"**Reason:** {_pf_reason}\n\n"
+ "The job will stay blocked (without re-alerting) until the "
+ "configuration is fixed; the next healthy run clears this "
+ "state. Set `cron.preflight: false` in config.yaml to "
+ "disable this validation."
+ )
+ return False, blocked_doc, "", f"{marker} {_pf_reason}"
+
+
+def _resolve_job_runtime(
+ job: dict, job_id: str, jc: _CronJobConfig,
+) -> tuple[dict, str, Optional[str]]:
+ """Resolve the runtime, walking the fallback chain on auth/transient-network errors.
+
+ Returns ``(runtime, model, primary_provider_for_drift)``; provider+model swap atomically (never
+ swap only the provider while keeping a paid primary model).
+ """
+ from hermes_cli.runtime_provider import (
+ resolve_runtime_provider,
+ format_runtime_provider_error,
+ )
+ from hermes_cli.auth import AuthError
+
+ model = jc.model
+ configured_provider_for_drift = (
+ str(jc.model_cfg.get("provider") or "").strip().lower()
+ if isinstance(jc.model_cfg, dict)
+ else ""
+ )
+ primary_provider_for_drift = (
+ str(job.get("provider") or "").strip().lower()
+ or configured_provider_for_drift
+ or None
+ )
+ try:
+ # Do NOT pass HERMES_INFERENCE_PROVIDER as `requested`: it would override persisted config
+ # and resurrect stale providers for unpinned jobs.
+ runtime_kwargs = {
+ "requested": job.get("provider") or jc.cron_default_provider or None,
+ # api_mode must derive from the model actually run, not the stale persisted default.
+ "target_model": model,
+ }
+ if job.get("base_url"):
+ runtime_kwargs["explicit_base_url"] = job.get("base_url")
+ runtime = resolve_runtime_provider(**runtime_kwargs)
+ primary_provider_for_drift = (
+ str(runtime.get("provider") or "").strip().lower() or primary_provider_for_drift
+ )
+ return runtime, model, primary_provider_for_drift
+ except Exception as resolve_exc:
+ # Walk the fallback chain on AuthError AND transient network/DNS failures (e.g. during
+ # OAuth refresh); anything else re-raises.
+ is_auth = isinstance(resolve_exc, AuthError)
+ is_transient_net = _is_transient_provider_resolve_error(resolve_exc)
+ if not (is_auth or is_transient_net):
+ raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
+
+ primary_provider_for_drift = (
+ str(getattr(resolve_exc, "provider", "") or "").strip().lower()
+ or primary_provider_for_drift
+ )
+ logger.warning(
+ "Job '%s': primary provider resolve failed (%s: %s), trying fallback",
+ job_id, "auth" if is_auth else "transient network", resolve_exc,
+ )
+ for entry in get_fallback_chain(jc.cfg):
+ if not isinstance(entry, dict):
+ continue
+ fb_provider = str(entry.get("provider") or "").strip()
+ fb_model = str(entry.get("model") or "").strip()
+ if not fb_provider or not fb_model:
+ continue
+ try:
+ from hermes_cli.fallback_config import resolve_entry_api_key
+
+ fb_kwargs = {"requested": fb_provider, "target_model": fb_model}
+ if entry.get("base_url"):
+ fb_kwargs["explicit_base_url"] = entry["base_url"]
+ fb_api_key = resolve_entry_api_key(entry)
+ if fb_api_key:
+ fb_kwargs["explicit_api_key"] = fb_api_key
+ runtime = resolve_runtime_provider(**fb_kwargs)
+ logger.info(
+ "Job '%s': fallback resolved to %s model %s",
+ job_id, runtime.get("provider"), fb_model,
+ )
+ return runtime, fb_model, primary_provider_for_drift
+ except Exception as fb_exc:
+ logger.debug("Job '%s': fallback %s failed: %s", job_id, fb_provider, fb_exc)
+ raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
+
+
+def _check_model_drift(
+ job: dict, job_id: str, cfg: dict, runtime: dict,
+ primary_provider_for_drift: Optional[str], primary_model_for_drift: str,
+) -> None:
+ """Fail-closed provider/model drift guard; raises RuntimeError (with drift marker) on drift.
+
+ An unpinned job follows the global default, which may have switched to a paid provider/model
+ since creation. For each unpinned axis with a creation snapshot (job["_snapshot"]) that
+ now resolves differently: skip the run, no paid call, alert to pin. No snapshot, pinned axes,
+ or resolution from the cron.model fleet default never count as drift.
+ """
+ if not cron_model_drift_guard_enabled(cfg):
+ return
+ _current_provider = str(
+ primary_provider_for_drift or runtime.get("provider") or ""
+ ).strip().lower()
+ _current_model = str(primary_model_for_drift or "").strip().lower()
+ _drift: list[str] = []
+ for _axis in cron_model_drift_axes(
+ job, current_provider=_current_provider, current_model=_current_model, config=cfg,
+ ):
+ _snapshot = str(job.get(f"{_axis}_snapshot") or "").strip().lower()
+ _current = _current_provider if _axis == "provider" else _current_model
+ _drift.append(f"{_axis} '{_snapshot}' -> '{_current}'")
+ if not _drift:
+ return
+ _changes = "; ".join(_drift)
+ # A finite one-shot is consumed by this attempt, so "edit the job" is a dead end for it.
+ _repeat = job.get("repeat") if isinstance(job.get("repeat"), dict) else {}
+ _finite_oneshot = (
+ isinstance(job.get("schedule"), dict)
+ and job["schedule"].get("kind") == "once"
+ and _repeat.get("times") == 1
+ )
+ if _finite_oneshot:
+ _remediation = (
+ "This finite one-shot job is consumed by this attempted run; "
+ "create a new one-shot job at a future time with an explicit "
+ "provider and model."
+ )
+ else:
+ _remediation = (
+ "To run on the new config, on the host running Hermes "
+ "pin it explicitly: "
+ f"`hermes cron edit {job_id} --provider "
+ "--model ` (or pin the original values to keep "
+ "them)."
+ )
+ logger.warning(
+ "Job '%s': SKIPPED — global inference config drifted since "
+ "creation (%s) and this job is unpinned. Skipped to prevent "
+ "unintended spend. %s",
+ job_id, _changes, _remediation,
+ )
+ # Alert-once via drift_alerted bit (silent marker suppresses delivery); a successful run
+ # clears it and re-arms the alert.
+ _drift_already_alerted = False
+ with contextlib.suppress(Exception):
+ from cron.jobs import mark_drift_alerted
+
+ _drift_already_alerted = mark_drift_alerted(job_id)
+ _drift_marker = DRIFT_SKIP_SILENT_MARKER if _drift_already_alerted else DRIFT_SKIP_MARKER
+ raise RuntimeError(
+ f"{_drift_marker} Skipped to prevent unintended spend: global "
+ f"inference config drifted since this job was created "
+ f"({_changes}), and this job is unpinned. No inference call "
+ f"was made. {_remediation} "
+ f"This alert is sent once; the job stays skipped until the "
+ f"config is pinned or restored. See #44585."
+ )
+
+
+def _load_credential_pool(runtime: dict, job_id: str):
+ runtime_provider = str(runtime.get("provider") or "").strip().lower()
+ if not runtime_provider:
+ return None
+ try:
+ from agent.credential_pool import load_pool
+ pool = load_pool(runtime_provider)
+ if pool.has_credentials():
+ logger.info(
+ "Job '%s': loaded credential pool for provider %s with %d entries",
+ job_id, runtime_provider, len(pool.entries()),
+ )
+ return pool
+ except Exception as e:
+ logger.debug("Job '%s': failed to load credential pool for %s: %s", job_id, runtime_provider, e)
+ return None
+
+
+def _init_cron_mcp_tools(job_id: str) -> None:
+ """Register MCP servers for the agent's tool registry. Idempotent across ticks; non-fatal so a
+ broken MCP server never kills a working job."""
+ try:
+ from tools.mcp_tool import discover_mcp_tools
+ _mcp_tools = discover_mcp_tools()
+ if _mcp_tools:
+ logger.info("Job '%s': %d MCP tool(s) available", job_id, len(_mcp_tools))
+ except Exception as _mcp_exc:
+ logger.warning("Job '%s': MCP initialization failed (non-fatal): %s", job_id, _mcp_exc)
+
+
+def _open_cron_session_db(job: dict):
+ """Open the SQLite session store under its own timeout (HERMES_CRON_TIMEOUT only watches
+ run_conversation). A wedged sqlite3.connect returns None (no session store) instead of
+ wedging the worker thread."""
+ _session_db_timeout = _get_session_db_timeout()
+ try:
+ from hermes_state import get_shared_session_db
+
+ if _session_db_timeout <= 0:
+ return get_shared_session_db()
+ _session_db_pool = concurrent.futures.ThreadPoolExecutor(max_workers=1)
+ # Copy the context so a profile run resolves ITS OWN home/state.db on the worker thread
+ # instead of the process-global default.
+ _session_db_context = contextvars.copy_context()
+ _session_db_future = _session_db_pool.submit(_session_db_context.run, get_shared_session_db)
+ try:
+ return _session_db_future.result(timeout=_session_db_timeout)
+ except concurrent.futures.TimeoutError:
+ # The abandoned worker may still finish; close its late result or its SQLite FDs leak.
+ _session_db_future.add_done_callback(_close_late_session_db_result)
+ raise
+ finally:
+ # Abandon a wedged connect() rather than blocking shutdown on it.
+ _session_db_pool.shutdown(wait=False)
+ except concurrent.futures.TimeoutError:
+ logger.error(
+ "Job '%s': SessionDB init did not return within %.0fs — proceeding "
+ "without a session store for this run instead of blocking it "
+ "forever",
+ job.get("id", "?"), _session_db_timeout,
+ )
+ except Exception as e:
+ logger.debug("Job '%s': SQLite session store not available: %s", job.get("id", "?"), e)
+ return None
+
+
+def _run_agent_with_watchdog(
+ agent, prompt: str, job: dict, job_id: str, job_name: str, task_id: str, cancel_event,
+) -> dict:
+ """Run ``agent.run_conversation`` on a worker thread under the inactivity watchdog.
+
+ Inactivity (not wall-clock) limit from the agent's activity tracker; default 600s, override
+ HERMES_CRON_TIMEOUT, 0 = unlimited.
+ """
+ _cron_timeout = _cron_inactivity_seconds()
+ _cron_inactivity_limit = _cron_timeout if _cron_timeout > 0 else None
+ _POLL_INTERVAL = 5.0
+ # Heartbeat the one-shot run_claim while alive: without it a long run looks like a dead owner
+ # and gets re-dispatched / stale-removed out from under the live run.
+ _job_schedule = job.get("schedule")
+ _is_oneshot = isinstance(_job_schedule, dict) and _job_schedule.get("kind") == "once"
+ _run_claim = job.get("run_claim")
+ _run_claim_owner = str(_run_claim.get("by") or "") if isinstance(_run_claim, dict) else ""
+ _last_claim_heartbeat = time.monotonic()
+
+ def _abort_if_fire_claim_lost() -> None:
+ if cancel_event is None or not cancel_event.is_set():
+ return
+ if agent is not None and hasattr(agent, "interrupt"):
+ agent.interrupt("Cron fire claim ownership was lost")
+ raise RuntimeError(f"Cron job '{job_name}' lost its durable fire claim ownership")
+
+ def _heartbeat_run_claim_if_due():
+ nonlocal _last_claim_heartbeat
+ if not _is_oneshot or not _run_claim_owner:
+ return
+ _mono = time.monotonic()
+ if _mono - _last_claim_heartbeat < _RUN_CLAIM_HEARTBEAT_SECONDS:
+ return
+ _last_claim_heartbeat = _mono
+ try:
+ heartbeat_run_claim(job_id, expected_owner=_run_claim_owner)
+ except Exception:
+ logger.debug("Job '%s': run_claim heartbeat failed", job_name, exc_info=True)
+
+ _cron_pool = concurrent.futures.ThreadPoolExecutor(max_workers=1)
+ # Carry scheduler-scoped ContextVar state (e.g. env passthrough) into the worker thread.
+ _cron_context = contextvars.copy_context()
+ _cron_future = _cron_pool.submit(
+ _cron_context.run, agent.run_conversation, prompt, task_id=task_id,
+ )
+ _inactivity_timeout = False
+ _watch_stop = threading.Event()
+
+ def _idle_seconds() -> float:
+ if not hasattr(agent, "get_activity_summary"):
+ return 0.0
+ try:
+ _act = agent.get_activity_summary()
+ return float(_act.get("seconds_since_activity", 0.0) or 0.0)
+ except Exception:
+ return 0.0
+
+ def _watch_inactivity() -> None:
+ nonlocal _inactivity_timeout
+ if _cron_inactivity_limit is None:
+ return
+ if _inactivity_watchdog_loop(
+ get_idle_seconds=_idle_seconds,
+ limit_s=_cron_inactivity_limit,
+ poll_s=_POLL_INTERVAL,
+ stop=_watch_stop,
+ future_done=_cron_future.done,
+ ):
+ _inactivity_timeout = True
+
+ _watch_thread = threading.Thread(
+ target=_watch_inactivity,
+ name=f"cron-inactivity-{str(job_id)[:8]}",
+ daemon=True,
+ )
+ try:
+ if _cron_inactivity_limit is not None:
+ # Separate daemon thread so a hung get_activity_summary can't stop the limit firing.
+ _watch_thread.start()
+ if _cron_inactivity_limit is None and not _is_oneshot and cancel_event is None:
+ result = _cron_future.result()
+ else:
+ result = None
+ while True:
+ done, _ = concurrent.futures.wait({_cron_future}, timeout=_POLL_INTERVAL)
+ if done:
+ _abort_if_fire_claim_lost()
+ result = _cron_future.result()
+ break
+ if _inactivity_timeout:
+ break
+ _abort_if_fire_claim_lost()
+ _heartbeat_run_claim_if_due()
+ except Exception:
+ _cron_pool.shutdown(wait=False, cancel_futures=True)
+ raise
+ finally:
+ _watch_stop.set()
+ _cron_pool.shutdown(wait=False, cancel_futures=True)
+
+ if _inactivity_timeout:
+ _activity = {}
+ if hasattr(agent, "get_activity_summary"):
+ with contextlib.suppress(Exception):
+ _activity = agent.get_activity_summary()
+ _last_desc = _activity.get("last_activity_desc", "unknown")
+ _secs_ago = _activity.get("seconds_since_activity", 0)
+ logger.error(
+ "Job '%s' idle for %.0fs (inactivity limit %.0fs) "
+ "| last_activity=%s | iteration=%s/%s | tool=%s",
+ job_name, _secs_ago, _cron_inactivity_limit,
+ _last_desc, _activity.get("api_call_count", 0), _activity.get("max_iterations", 0),
+ _activity.get("current_tool") or "none",
+ )
+ request_hard_interrupt(agent, "Cron job timed out (inactivity)")
+ raise TimeoutError(
+ f"Cron job '{job_name}' idle for "
+ f"{int(_secs_ago)}s (limit {int(_cron_inactivity_limit)}s) "
+ f"— last activity: {_last_desc}"
+ )
+
+ if not isinstance(result, dict):
+ raise RuntimeError(
+ f"agent.run_conversation returned {type(result).__name__} instead of dict: {result!r}"
+ )
+ return result
+
+
+def _final_response_from_result(result: dict, job_id: str, job_name: str, AIAgent) -> str:
+ """Turn a ``run_conversation`` result into the deliverable final response.
+
+ Raises RuntimeError on `failed=True`/`completed=False`: the error text may sit in
+ `final_response` and would otherwise be delivered as the reply with the job marked ok.
+ """
+ turn_exit_reason = str(result.get("turn_exit_reason") or "")
+ final_response_text = (result.get("final_response") or "").strip()
+ max_iteration_summary = (
+ result.get("failed") is not True
+ and result.get("completed") is False
+ and turn_exit_reason.startswith("max_iterations_reached(")
+ and bool(final_response_text)
+ )
+ if result.get("failed") is True or (result.get("completed") is False and not max_iteration_summary):
+ raise RuntimeError(result.get("error") or final_response_text or "agent reported failure")
+ if max_iteration_summary:
+ logger.warning(
+ "Job '%s' reached the iteration limit but produced a final fallback response; "
+ "delivering the response instead of failing the cron run",
+ job_name,
+ )
+
+ final_response = result.get("final_response", "") or ""
+ # Repair model-mangled computer_use media paths before delivery (fail-open, as in gateway).
+ if final_response:
+ from gateway.media_repair import repair_explicit_computer_use_media_paths
+
+ final_response = repair_explicit_computer_use_media_paths(
+ final_response, result.get("messages", []),
+ )
+ if final_response.strip() == "(No response generated)":
+ final_response = ""
+ # The "⚠️ No reply" turn-completion explainer would be delivered as a cron warning; detect it
+ # via the same formatter and treat as empty so cron stays silent on abnormal empty turns.
+ if final_response.strip() and turn_exit_reason:
+ # Render every persistence-cause variant or cause-refined text slips through.
+ _explainer_variants = []
+ try:
+ from hermes_state import PERSISTENCE_ERROR_CAUSES as _causes
+ except Exception:
+ _causes = ("locked", "disk", "unknown")
+ for _cause in (None, *_causes):
+ try:
+ _variant = AIAgent._format_turn_completion_explanation(turn_exit_reason, _cause)
+ except TypeError:
+ try:
+ _variant = AIAgent._format_turn_completion_explanation(turn_exit_reason)
+ except Exception:
+ _variant = ""
+ except Exception:
+ _variant = ""
+ if _variant:
+ _explainer_variants.append(_variant.strip())
+ if final_response.strip() in _explainer_variants:
+ logger.info(
+ "Job '%s': abnormal empty turn (%s) — suppressing explainer for cron delivery",
+ job_id, turn_exit_reason,
+ )
+ final_response = ""
+ return final_response
+
+
+def _finalize_cron_session(session_db, agent, job_id: str, job_name: str, cron_session_id: str) -> None:
+ """Title, classify, end and release the cron session after the agent turn has returned."""
+ # Bound every DB op so storage failure cannot hold the dispatch guard.
+ _session_db = _BoundedCronSessionDB(session_db, job_id)
+ # Compression may have rotated the run onto a continuation: finalize that, not the stale cron
+ # id. SessionDB lineage is authoritative; agent.session_id is only a fail-safe.
+ _final_cron_session_id = cron_session_id
+ try:
+ _compression_tip = _session_db.get_compression_tip(cron_session_id)
+ if _compression_tip:
+ _final_cron_session_id = _compression_tip
+ except (Exception, KeyboardInterrupt) as e:
+ with contextlib.suppress((Exception, KeyboardInterrupt)):
+ _agent_session_id = getattr(agent, "session_id", None)
+ if _agent_session_id:
+ _final_cron_session_id = _agent_session_id
+ logger.debug("Job '%s': failed to resolve cron compression tip: %s", job_id, e)
+ # Title must persist BEFORE end_session()/close(). Run-time suffix keeps it unique against the
+ # sessions.title index; the fallbacks below guarantee a non-blank title.
+ try:
+ _title_base = " ".join(job_name.split())[:60].strip() or f"cron {job_id}"
+ _cron_title = f"{_title_base} · {_hermes_now().strftime('%b %d %H:%M')}"
+ if not _set_cron_session_title(_session_db, _final_cron_session_id, _cron_title):
+ _set_cron_session_title(_session_db, _final_cron_session_id, f"cron {job_id}")
+ except (Exception, KeyboardInterrupt) as e:
+ logger.debug("Job '%s': failed to set cron session title: %s", job_id, e)
+ # Never leave the session untitled.
+ for _fallback in (
+ getattr(_session_db, "get_next_title_in_lineage", lambda b: b)(f"cron {job_id}"),
+ f"cron {job_id} {_final_cron_session_id[-6:]}",
+ ):
+ try:
+ if _set_cron_session_title(_session_db, _final_cron_session_id, _fallback):
+ break
+ except (Exception, KeyboardInterrupt):
+ continue
+ # Book cron_complete only when the last row is a real assistant reply ([SILENT] counts). Only a
+ # POSITIVELY recognized bad status downgrades (keep tuple in sync with
+ # session_lifecycle_statuses); unknown values / probe failures fail OPEN.
+ _end_reason = "cron_complete"
+ try:
+ _statuses = _session_db.session_lifecycle_statuses([_final_cron_session_id])
+ _lifecycle = _statuses.get(_final_cron_session_id)
+ if _lifecycle in ("interrupted", "error", "empty"):
+ _end_reason = "cron_incomplete_no_output"
+ logger.warning(
+ "Job '%s': session ended without a final assistant "
+ "message (lifecycle=%s) — booking run as %s",
+ job_id, _lifecycle, _end_reason,
+ )
+ except (Exception, KeyboardInterrupt) as e:
+ logger.debug("Job '%s': session lifecycle classification failed: %s", job_id, e)
+ try:
+ _session_db.end_session(_final_cron_session_id, _end_reason)
+ except (Exception, KeyboardInterrupt) as e:
+ logger.debug("Job '%s': failed to end session: %s", job_id, e)
+ try:
+ from hermes_state import release_or_close
+ release_or_close(_session_db)
+ except (Exception, KeyboardInterrupt) as e:
+ logger.debug("Job '%s': failed to close SQLite session store: %s", job_id, e)
+
+
+def _run_doc_header(job: dict, title: str, job_id: str, prompt: str) -> str:
+ """Header of the persisted run document (title, ids, schedule, prompt)."""
+ return (
+ f"# Cron Job: {title}\n\n"
+ f"**Job ID:** {job_id}\n"
+ f"**Run Time:** {_hermes_now().strftime('%Y-%m-%d %H:%M:%S')}\n"
+ f"**Schedule:** {job.get('schedule_display', 'N/A')}\n\n"
+ f"## Prompt\n\n{prompt}\n\n"
+ )
+
+
def run_job(
job: dict,
*,
@@ -5745,35 +5033,17 @@ def run_job(
cancel_event: Optional[_CancelEventLike] = None,
execution_id: Optional[str] = None,
) -> tuple[bool, str, str, Optional[str]]:
- """
- Execute a single cron job.
+ """Execute a single cron job. Returns (success, full_output_doc, final_response, error).
- ``defer_agent_teardown``: when a caller passes a list, ``run_job`` skips
- the agent's async-resource teardown (``agent.close()`` +
- ``cleanup_stale_async_clients()``) in its ``finally`` block and instead
- appends the live agent to that list. The caller is then responsible for
- calling ``_teardown_cron_agent(agent)`` AFTER it has delivered the result.
- This closes the ordering window in #58720 where delivery ran against a
- torn-down async client (defense-in-depth alongside the interpreter-shutdown
- guard). When ``None`` (the default) teardown happens inline as before, so
- every existing caller is unchanged.
-
- ``extra_prompt``: optional per-run context from ``cronjob(action='run',
- prompt=...)`` (#57331). Appended to the stored prompt for this fire only —
- never persisted to the job definition.
-
- Returns:
- Tuple of (success, full_output_doc, final_response, error_message)
+ ``defer_agent_teardown``: if a list, the live agent is appended instead of torn down in
+ ``finally``; the caller MUST call ``_teardown_cron_agent(agent)`` AFTER delivery (delivery
+ against a torn-down async client fails). ``extra_prompt``: per-fire context, never persisted.
"""
job_id = job["id"]
job_name = str(job.get("name") or job.get("prompt") or job_id or "cron job")
- # Fail closed on a corrupt config.yaml before any agent-driven work
- # (issue #81952): a cron fire is fully non-interactive, and continuing
- # on built-in defaults lets provider auto-detection adopt .env
- # credentials the config never named, billing a provider the user did
- # not choose. no_agent script jobs are exempt — they never construct an
- # AIAgent or spend tokens. Escape hatch: HERMES_IGNORE_USER_CONFIG=1.
+ # Fail closed on a corrupt config.yaml: defaults would let auto-detection bill a provider the
+ # user never chose. no_agent jobs are exempt. Escape hatch: HERMES_IGNORE_USER_CONFIG=1.
if not job.get("no_agent"):
from hermes_cli.config import (
InvalidUserConfigError,
@@ -5786,224 +5056,24 @@ def run_job(
logger.error("Job '%s': refusing to run — %s", job_id, exc)
return (False, f"# Cron Job: {job_name}\n\nError: {exc}\n", "", str(exc))
- # ---------------------------------------------------------------
- # no_agent short-circuit — the script IS the job, no LLM involvement.
- # ---------------------------------------------------------------
- # This mirrors the classic "run a bash script on a timer, send its
- # stdout to telegram" watchdog pattern. The agent path is skipped
- # entirely: no AIAgent, no prompt, no tool loop, no token spend.
- #
- # We check this BEFORE importing run_agent / constructing SessionDB so
- # a pure-script tick never pays for the agent machinery it isn't going
- # to use. Keep this block self-contained.
- #
- # Semantics:
- # - script stdout (trimmed) → delivered verbatim as the final message
- # - empty stdout → silent run (no delivery, success=True)
- # - non-zero exit / timeout → delivered as an error alert, success=False
- # - wakeAgent=false gate → treated like empty stdout (silent), since
- # the whole point of no_agent is that there
- # is no agent to wake
+ # no_agent short-circuits BEFORE importing run_agent / opening SessionDB.
if job.get("no_agent"):
- # Load .env before the script runs so auto-delivery can resolve home
- # channels. A standalone cron tick process typically starts WITHOUT
- # TELEGRAM_HOME_CHANNEL/DISCORD_HOME_CHANNEL in its environment, and
- # the agent path's per-run dotenv reload below never executes for
- # no_agent jobs — every deliver=telegram/all script job failed with
- # "no delivery target resolved". load_hermes_dotenv does not override
- # already-set vars, so the gateway's in-process tick is unaffected.
- try:
- from hermes_cli.env_loader import load_hermes_dotenv
+ return _run_no_agent_job(job, job_id, job_name, cancel_event)
- load_hermes_dotenv(hermes_home=_get_hermes_home())
- except Exception:
- logger.debug(
- "Job '%s': no_agent .env reload failed", job_id, exc_info=True
- )
-
- script_path = job.get("script")
- # Legacy/hand-edited records can still carry no_agent with a missing or
- # whitespace-only script. Erroring alone left the job enabled, so it
- # re-fired every tick — pause it instead (a5e29e688dc0).
- if not str(script_path or "").strip():
- from cron.jobs import NO_AGENT_WITHOUT_SCRIPT_ERROR
-
- return _block_and_pause_job(
- job_id,
- job_name,
- NO_AGENT_WITHOUT_SCRIPT_ERROR,
- )
-
- # Apply workdir if configured — lets scripts use predictable relative
- # paths. For no_agent jobs this is passed as the subprocess cwd so the
- # Python process cwd is NEVER mutated — avoiding the global-side-effect
- # bug where os.chdir() leaks into concurrent gateway sessions (#69396).
- _job_workdir = (job.get("workdir") or "").strip() or None
- if _job_workdir and not Path(_job_workdir).is_dir():
- logger.warning(
- "Job '%s': configured workdir %r no longer exists — running without it",
- job_id, _job_workdir,
- )
- _job_workdir = None
-
- try:
- ok, output = _run_job_script_with_claim_heartbeat(
- job, script_path, workdir=_job_workdir, cancel_event=cancel_event,
- )
- except Exception as exc:
- logger.exception(
- "Job '%s': script execution raised unexpectedly", job_id,
- )
- ok, output = False, f"Script execution failed: {exc}"
-
- now_iso = _hermes_now().strftime("%Y-%m-%d %H:%M:%S")
-
- if not ok:
- # Script crashed / timed out / exited non-zero. Deliver the
- # error so the user knows the watchdog itself broke — silent
- # failure for an alerting job is the worst-case outcome.
- alert = (
- f"⚠ Cron watchdog '{job_name}' script failed\n\n"
- f"{output}\n\n"
- f"Time: {now_iso}"
- )
- doc = (
- f"# Cron Job: {job_name}\n\n"
- f"**Job ID:** {job_id}\n"
- f"**Run Time:** {now_iso}\n"
- f"**Mode:** no_agent (script)\n"
- f"**Status:** script failed\n\n"
- f"{output}\n"
- )
- return False, doc, alert, output
-
- # Honour the wakeAgent gate as a silent signal — `wakeAgent: false`
- # means "nothing to report this tick", same as empty stdout.
- if not _parse_wake_gate(output):
- logger.info(
- "Job '%s' (no_agent): wakeAgent=false gate — silent run", job_id
- )
- silent_doc = (
- f"# Cron Job: {job_name}\n\n"
- f"**Job ID:** {job_id}\n"
- f"**Run Time:** {now_iso}\n"
- f"**Mode:** no_agent (script)\n"
- f"**Status:** silent (wakeAgent=false)\n"
- )
- return True, silent_doc, SILENT_MARKER, None
-
- if not output.strip():
- logger.info("Job '%s' (no_agent): empty stdout — silent run", job_id)
- silent_doc = (
- f"# Cron Job: {job_name}\n\n"
- f"**Job ID:** {job_id}\n"
- f"**Run Time:** {now_iso}\n"
- f"**Mode:** no_agent (script)\n"
- f"**Status:** silent (empty output)\n"
- )
- return True, silent_doc, SILENT_MARKER, None
-
- doc = (
- f"# Cron Job: {job_name}\n\n"
- f"**Job ID:** {job_id}\n"
- f"**Run Time:** {now_iso}\n"
- f"**Mode:** no_agent (script)\n\n"
- f"---\n\n"
- f"{output}\n"
- )
- return True, doc, output, None
-
- # ---------------------------------------------------------------
- # Fail-closed guard for legacy / hand-edited agent jobs that have nothing
- # to run: blank prompt, no script, no skills (a5e29e688dc0). create_job /
- # update_job now reject this shape, but jobs.json records written before
- # that guard — or edited by hand since — can still reach here and would
- # otherwise wake the LLM with an empty instruction on every fire. Pause
- # the job so it stops being scheduled, and never construct the agent.
- # ---------------------------------------------------------------
+ # Legacy / hand-edited job with nothing to run: pause it instead of waking the LLM every fire.
from cron.jobs import EMPTY_PAYLOAD_ERROR, job_payload_is_empty
if job_payload_is_empty(job):
- return _block_and_pause_job(
- job_id,
- job_name,
- EMPTY_PAYLOAD_ERROR,
- )
+ return _block_and_pause_job(job_id, job_name, EMPTY_PAYLOAD_ERROR)
- # ---------------------------------------------------------------
- # Monitor gate — hash-suppressed change detection (see cron/monitor.py).
- # Runs BEFORE any agent machinery is constructed so an unchanged tick
- # costs one cheap source run + one hash, no LLM, no delivery.
- # ---------------------------------------------------------------
- from cron.monitor import check_monitor, job_has_monitor
+ _early, extra_prompt = _apply_monitor_gate(job, job_id, job_name, extra_prompt)
+ if _early is not None:
+ return _early
- _monitor_context: Optional[str] = None
- if job_has_monitor(job):
- _mon = check_monitor(job)
- _mon_now = _hermes_now().strftime("%Y-%m-%d %H:%M:%S")
- if not _mon.ok:
- # Source failure is an ERROR, never a change: alert the user so
- # a broken monitor can't silently stop watching. Stored hash is
- # untouched (check_monitor persists nothing on failure).
- logger.error("Job '%s': monitor source failed: %s", job_id, _mon.error)
- _mon_doc = (
- f"# Cron Job: {job_name}\n\n"
- f"**Job ID:** {job_id}\n"
- f"**Run Time:** {_mon_now}\n"
- f"**Mode:** monitor\n"
- f"**Status:** monitor source failed\n\n"
- f"{_mon.error}\n"
- )
- _mon_alert = (
- f"⚠ Cron monitor '{job_name}' source failed\n\n"
- f"{_mon.error}\n\n"
- f"Time: {_mon_now}"
- )
- return False, _mon_doc, _mon_alert, _mon.error
- if not _mon.changed:
- # Unchanged output — suppress the agent run entirely. Recorded
- # as a silent no_change tick (visible in the executions ledger
- # via this doc; SILENT_MARKER blocks delivery).
- logger.info(
- "Job '%s': monitor output unchanged — suppressing agent run",
- job_id,
- )
- _mon_doc = (
- f"# Cron Job: {job_name}\n\n"
- f"**Job ID:** {job_id}\n"
- f"**Run Time:** {_mon_now}\n"
- f"**Mode:** monitor\n"
- f"**Status:** no_change (agent run suppressed)\n"
- )
- return True, _mon_doc, SILENT_MARKER, None
- # Changed (or first run): inject the monitor context into the prompt
- # through the existing per-run context seam and fall through to a
- # normal agent run.
- _monitor_context = _mon.context_block
- if _monitor_context:
- extra_prompt = (
- f"{_monitor_context}\n\n{extra_prompt}" if extra_prompt else _monitor_context
- )
-
- # ---------------------------------------------------------------
- # Default (LLM) path — import and construct the agent machinery now
- # that we know we actually need it. Doing these imports here instead of
- # at module top keeps no_agent ticks from paying for AIAgent / SessionDB
- # construction costs.
- # ---------------------------------------------------------------
from run_agent import AIAgent
- # NOTE: the SQLite session store used to be initialized here, BEFORE the
- # wake-gate and prompt-validation early returns below. Every gated run
- # (``wakeAgent: false``, blocked prompt) opened state.db and returned
- # without reaching the finally that closes it, relying on GC to release
- # the handle. Init now happens inside the main try, right before the
- # agent is constructed — after every early-return path (#96290).
-
- # Wake-gate: if this job has a pre-check script, run it BEFORE building
- # the prompt so a ``{"wakeAgent": false}`` response can short-circuit
- # the whole agent run. We pass the result into _build_job_prompt so
- # the script is only executed once.
+ # Wake-gate: run the pre-check script BEFORE building the prompt; its result is passed into
+ # _build_job_prompt so the script runs only once.
prerun_script = None
script_path = job.get("script")
if script_path:
@@ -6012,10 +5082,7 @@ def run_job(
)
_ran_ok, _script_output = prerun_script
if _ran_ok and not _parse_wake_gate(_script_output):
- logger.info(
- "Job '%s' (ID: %s): wakeAgent=false, skipping agent run",
- job_name, job_id,
- )
+ logger.info("Job '%s' (ID: %s): wakeAgent=false, skipping agent run", job_name, job_id)
silent_doc = (
f"# Cron Job: {job_name}\n\n"
f"**Job ID:** {job_id}\n"
@@ -6025,14 +5092,9 @@ def run_job(
return True, silent_doc, SILENT_MARKER, None
try:
- prompt = _build_job_prompt(
- job, prerun_script=prerun_script, extra_prompt=extra_prompt
- )
+ prompt = _build_job_prompt(job, prerun_script=prerun_script, extra_prompt=extra_prompt)
except CronPromptInjectionBlocked as block_exc:
- # Assembled prompt (user prompt + loaded skill content) tripped the
- # injection scanner. Refuse to run the agent this tick and surface
- # a clear failure to the operator so they see WHY the scheduled job
- # didn't run and can audit the offending skill.
+ # Injection scanner tripped: refuse this tick and tell the operator WHY.
logger.warning(
"Job '%s' (ID: %s): blocked by prompt-injection scanner — %s",
job_name, job_id, block_exc,
@@ -6060,61 +5122,23 @@ def run_job(
logger.info("Prompt: %s", prompt[:100])
agent = None
+ model = ""
- # Use ContextVars for per-job session/delivery state so parallel jobs
- # don't clobber each other's targets (os.environ is process-global).
+ # ContextVars, not os.environ (process-global), so parallel jobs don't clobber each other.
from gateway.session_context import set_session_vars, clear_session_vars, _VAR_MAP
- # Cron execution is an internal scheduler context, not a live inbound
- # gateway message. Do not seed HERMES_SESSION_* contextvars from the
- # stored ``origin`` (which is delivery routing metadata, not a sender
- # identity). Several tool consumers branch on these vars during job
- # execution and would otherwise behave as if a real user from the
- # origin chat was driving the agent:
- # - tools/terminal_tool.py: background-process notification routing
- # (notify_on_complete / watch_patterns) reads HERMES_SESSION_PLATFORM
- # and HERMES_SESSION_CHAT_ID to populate watcher_platform / chat_id,
- # which would route completion notifications to the origin chat
- # instead of via HERMES_CRON_AUTO_DELIVER_* below.
- # - tools/tts_tool.py: picks Opus vs MP3 based on
- # HERMES_SESSION_PLATFORM == "telegram".
- # - tools/skills_tool.py + agent/prompt_builder.py: per-platform
- # skill-disable lists and the system-prompt cache key both consume
- # HERMES_SESSION_PLATFORM.
- # - tools/send_message_tool.py: mirror source labelling and the
- # send_message gate read HERMES_SESSION_PLATFORM.
- # Cron output delivery itself reads job["origin"] directly via
- # _resolve_origin(job) and the HERMES_CRON_AUTO_DELIVER_* vars set
- # below, so clearing HERMES_SESSION_* here does not affect delivery.
- # Resolve workdir BEFORE set_session_vars so we can pass it as cwd=,
- # letting set_session_vars handle the _SESSION_CWD ContextVar set/clear
- # via its existing machinery (clear_session_vars calls clear_session_cwd
- # internally). This avoids a separate import/set/clear dance (#69396).
- _job_workdir = (job.get("workdir") or "").strip() or None
- if _job_workdir and not Path(_job_workdir).is_dir():
- logger.warning(
- "Job '%s': configured workdir %r no longer exists — running without it",
- job_id, _job_workdir,
- )
- _job_workdir = None
+ # Do NOT seed HERMES_SESSION_* from job["origin"]: it is delivery metadata, not a sender, and
+ # terminal/tts/skills/send_message tools would act as if the origin user were driving the
+ # agent. Delivery reads job["origin"] and HERMES_CRON_AUTO_DELIVER_* directly, so blanking is
+ # safe. Resolve workdir BEFORE set_session_vars so it owns the _SESSION_CWD set/clear.
+ _job_workdir = _resolve_job_workdir(job, job_id)
_ctx_tokens = set_session_vars(
platform="",
chat_id="",
chat_name="",
- # A cron job cannot receive a completion after its turn ends. We clear the
- # HERMES_SESSION_* routing keys just below, so an async delegation's
- # completion event carries session_key="" — _enrich_async_delegation_routing
- # cannot resolve it and _inject_watch_notification drops it ("no routing
- # metadata"). And by the time a child finishes, run_job has already shipped
- # the job's final response via _deliver_result; there is no turn left to
- # re-enter. (Worse, get_current_session_key() can fall back to the ambient
- # os.environ HERMES_SESSION_KEY, which risks routing a cron subagent's output
- # into an unrelated user chat.)
- #
- # Declaring the channel stateless routes delegate_task to its existing
- # inline/synchronous path, so results return within the job's own turn.
- # See declare_stateless_channel(). Upstream: #53027, #63142.
+ # Cron can't receive completions after its turn; async delegation output could otherwise
+ # route to an unrelated chat via the ambient session key. Stateless => inline delegation.
async_delivery=False,
cwd=_job_workdir or "",
)
@@ -6126,11 +5150,8 @@ def run_job(
for _var_name in _cron_delivery_vars:
_VAR_MAP[_var_name].set("")
- # Tool calls are keyed by a per-run task id. Bind the cron workdir to that
- # identity instead of mutating process-global TERMINAL_CWD. The session CWD
- # record is the tool-layer authority for terminal/file/code-exec/delegation,
- # while the _SESSION_CWD ContextVar above remains the prompt/context-file
- # authority. Both are isolated across concurrent cron runs.
+ # Bind workdir to the per-run task id (tool-layer cwd authority) instead of mutating global
+ # TERMINAL_CWD; _SESSION_CWD above remains the prompt/context-file authority.
_cron_task_id = (
f"cron:{job_id}:"
f"{execution_id or job.get('execution_id') or uuid.uuid4().hex}"
@@ -6145,46 +5166,20 @@ def run_job(
_session_db = None
try:
- # Scope cron approval policy to this job. Keep the token so the finally
- # restores the pre-job state instead of pinning an explicit empty value,
- # which would suppress the legacy os.environ fallback used by standalone
- # cron entrypoints and tests.
+ # Scope cron approval policy; the finally RESETS via token (pinning "" would suppress the
+ # legacy os.environ fallback used by standalone entrypoints/tests).
_cron_session_token = _cron_session_var.set("1")
- # Mark this job as NOT the dispatcher-owned kanban worker.
- #
- # A kanban worker is a normal `hermes chat -q` CLI agent whose default
- # toolset includes `cronjob`, running with HERMES_KANBAN_TASK
- # legitimately in its own env; `cronjob(action="run")` calls
- # run_one_job() -> run_job() right here in that process. Without this
- # marker the cron agent is misread as that worker: the kanban toolset is
- # force-added, the worker protocol is injected into its system prompt,
- # and kanban_complete defaults task_id to $HERMES_KANBAN_TASK -- letting
- # an unrelated cron job close the worker's task and overwrite real
- # results.
- #
- # A ContextVar, NOT an os.environ clear: the env is process-global and
- # shared with the worker's own claim heartbeat (run_agent._touch_activity
- # -> heartbeat_current_worker_from_env, which would starve and let the
- # dispatcher reclaim a live task), the gateway's kanban watchers, and
- # concurrent cron jobs on the parallel pool. contextvars.copy_context()
- # at the run_conversation hop carries this into the agent thread.
+ # Mark NOT the kanban worker: a worker's cronjob(action="run") lands here with
+ # HERMES_KANBAN_TASK in env, and an unrelated job could close the worker's task. Must be a
+ # ContextVar, NOT an os.environ clear (env is shared with the worker heartbeat and
+ # concurrent jobs); copy_context() carries it into the agent thread.
_non_dispatcher_token = enter_non_dispatcher_owned_context()
if _job_workdir:
logger.info("Job '%s': using task-scoped workdir %s", job_id, _job_workdir)
- # Re-read .env and config.yaml fresh every run so provider/key
- # changes take effect without a gateway restart. Route through
- # load_hermes_dotenv (not a bare load_dotenv) and reset the secret-
- # source cache first: startup already applied external secrets and
- # recorded this HERMES_HOME in _APPLIED_HOMES, so a naive reload would
- # re-apply only the .env placeholder and never re-resolve a Bitwarden/
- # BSM-backed secret — leaving cron jobs 401'ing on the placeholder
- # (#33465). Clearing the cache forces the re-pull; the resolved secret
- # overrides the placeholder only when secrets.bitwarden.override_existing
- # is set (mirrors startup), and the Bitwarden value-cache keeps the
- # forced re-pull off the network. load_hermes_dotenv also handles the
- # utf-8/latin-1 encoding fallback internally.
+ # Re-read .env every run; reset the secret-source cache FIRST or a Bitwarden/BSM-backed
+ # secret is never re-resolved (only the placeholder reloads -> 401s).
from hermes_cli.env_loader import (
load_hermes_dotenv,
reset_secret_source_cache,
@@ -6202,504 +5197,46 @@ def run_job(
else str(delivery_target["thread_id"])
)
- # Model resolution precedence: per-job override > cron.model (the
- # cron-fleet default) > HERMES_MODEL env > config.yaml ``model:``
- # (string or ``{default: ...}``). The per-job value is intentionally
- # re-read from storage every tick so a ``hermes cron edit --model``
- # after a failed run takes effect on the next tick — there is no
- # in-memory cache.
- model = job.get("model") or os.getenv("HERMES_MODEL") or ""
+ jc = _load_cron_job_config(job, job_id, job_name)
+ _cfg = jc.cfg
+ model = jc.model
- # cron.model / cron.model_provider: a deliberate cron-fleet default
- # so unattended jobs stop shadowing chat `/model` switches. When an
- # axis resolves from here, the #44585 drift guard is skipped for that
- # axis — following cron.model is explicit, not drift.
- _cron_default_model = ""
- _cron_default_provider = ""
+ prefill_messages = _load_prefill_messages(_cfg, job_id)
- # Load config.yaml for model, reasoning, prefill, toolsets, provider routing
- _cfg = {}
- _model_cfg = {}
- try:
- from hermes_cli.config import read_user_config_raw
- _cfg_path = str(_get_hermes_home() / "config.yaml")
- if os.path.exists(_cfg_path):
- _cfg = read_user_config_raw(Path(_cfg_path))
- # Managed scope: a scheduled job must honor administrator-pinned
- # model / reasoning / toolsets / provider_routing too. This loader
- # builds its own dict, so overlay managed values via the shared
- # helper (fail-open, no-op when no managed scope).
- try:
- from hermes_cli import managed_scope
- _cfg = managed_scope.apply_managed_overlay(_cfg)
- except Exception:
- pass
- _cfg = _expand_env_vars(_cfg)
- # Coerce null/missing to {} so a falsy default never
- # clobbers an already-resolved env value with ``None``.
- _model_cfg = _cfg.get("model") or {}
- _cron_cfg_for_model = _cfg.get("cron") or {}
- if isinstance(_cron_cfg_for_model, dict):
- _cron_default_model = str(
- _cron_cfg_for_model.get("model") or ""
- ).strip()
- _cron_default_provider = str(
- _cron_cfg_for_model.get("model_provider") or ""
- ).strip()
- if not job.get("model"):
- if _cron_default_model:
- # Cron-fleet default beats the global chat model: it is
- # the user's explicit "cron runs on this" setting.
- model = _cron_default_model
- else:
- # Shared with Desktop's post-save impact summary so both
- # paths compare snapshots against the same global model.
- _, _global_model = resolve_cron_model_drift_defaults(_cfg)
- if _global_model:
- model = _global_model
- except Exception as e:
- logger.warning("Job '%s': failed to load config.yaml, using defaults: %s", job_id, e)
-
- # Fail fast if no model resolved from job / env / config.yaml: an empty
- # model otherwise reaches the provider as an opaque 400 (#23979).
- if not (isinstance(model, str) and model.strip()):
- raise RuntimeError(
- f"Cron job '{job_name}' has no model configured "
- f"(job.model={job.get('model')!r}, "
- f"HERMES_MODEL={os.getenv('HERMES_MODEL', '')!r}, "
- "config.yaml model.default missing or empty). "
- f"Set a per-job model via "
- f"`hermes cron edit {job_id} --model ` or set a "
- "default with `hermes model `."
- )
-
- # Apply IPv4 preference if configured.
- try:
- from hermes_constants import apply_ipv4_preference
- _net_cfg = _cfg.get("network", {})
- if isinstance(_net_cfg, dict) and _net_cfg.get("force_ipv4"):
- apply_ipv4_preference(force=True)
- except Exception:
- pass
-
- # Reasoning config is resolved after provider authentication so an auth
- # fallback can first replace the primary model with its configured model.
- # Resolution itself happens via _resolve_job_reasoning_config below
- # (per-job pin > agent.reasoning_overrides > agent.reasoning_effort).
-
- # Prefill messages from env or config.yaml. The top-level
- # prefill_messages_file key is canonical; agent.prefill_messages_file is
- # retained as a legacy fallback for older CLI/godmode configs.
- prefill_messages = None
- agent_cfg = _cfg.get("agent", {}) if isinstance(_cfg.get("agent", {}), dict) else {}
- prefill_file = (
- os.getenv("HERMES_PREFILL_MESSAGES_FILE", "")
- or _cfg.get("prefill_messages_file", "")
- or agent_cfg.get("prefill_messages_file", "")
- )
- if prefill_file:
- pfpath = Path(prefill_file).expanduser()
- if not pfpath.is_absolute():
- pfpath = _get_hermes_home() / pfpath
- if pfpath.exists():
- try:
- with open(pfpath, "r", encoding="utf-8") as _pf:
- prefill_messages = json.load(_pf)
- if not isinstance(prefill_messages, list):
- prefill_messages = None
- except Exception as e:
- logger.warning("Job '%s': failed to parse prefill messages file '%s': %s", job_id, pfpath, e)
- prefill_messages = None
-
- # Max iterations — resolved through resolve_turn_limit() so that
- # agent.max_turns: none / unlimited → sys.maxsize sentinel, and
- # explicit 0 / null / "none" are honored instead of skipped by `or`.
+ # resolve_turn_limit() honors none/unlimited (sys.maxsize) and explicit 0 / null.
from hermes_cli.config import resolve_turn_limit as _resolve_turn_limit
_mt = _cfg.get("agent", {}).get("max_turns")
if _mt is None:
_mt = _cfg.get("max_turns")
max_iterations = _resolve_turn_limit(_mt)
- # Provider routing
pr = _cfg.get("provider_routing") or {}
- from hermes_cli.runtime_provider import (
- resolve_runtime_provider,
- format_runtime_provider_error,
- )
- from hermes_cli.auth import AuthError
-
- # F8 runtime backstop: never resolve a stored provider/base_url pair that
- # would ship a named provider's stored credential to an off-host endpoint
- # (CWE-200/CWE-522). The cron tool validates this on create/update, but a
- # job persisted before that guard — or written directly to the jobs store
- # — reaches this sink unchecked. Fail closed before resolution so no
- # off-host call is ever made with a stored key.
+ # Runtime backstop (CWE-200/522): fail closed BEFORE resolution on a provider/base_url pair
+ # that would ship a stored credential off-host; hand-written jobs bypass create-time checks.
_guard_job_credential_exfil(job)
- # ---------------------------------------------------------------
- # Pre-dispatch configuration validation (T1-26).
- #
- # A job whose configuration cannot possibly produce a successful
- # run — missing provider API key (no fallback chain), unready
- # attached skill, unconfigured delivery platform — is refused HERE,
- # before AIAgent is constructed and before the resolution below can
- # feed a doomed runtime into it, so a misconfigured job never burns
- # an LLM call. run_one_job keys off the BLOCKED_CONFIG_MARKER in
- # the returned error to record last_status='blocked_config' and
- # alert exactly once (dedup persisted via the job's
- # `preflight_alerted` bit — the #73506 alert-once shape).
- # Runs after the wake-gate/prompt build so silent script ticks stay
- # silent. Opt-out: `cron.preflight: false` in config.yaml.
- # ---------------------------------------------------------------
- _pf_reason = None
- try:
- if _cron_preflight_enabled(_cfg):
- _pf_reason = _preflight_job_config(job, _cfg)
- if not _pf_reason and job.get("preflight_alerted"):
- # Configuration validates again — clear the alert-once
- # marker so a FUTURE config break re-alerts.
- try:
- from cron.jobs import clear_preflight_alerted
- clear_preflight_alerted(job_id)
- except Exception:
- pass
- except Exception:
- # The validator must never take down a runnable job — fail open.
- logger.debug(
- "Job '%s': preflight validation errored — failing open",
- job_id, exc_info=True,
- )
- _pf_reason = None
-
- if _pf_reason:
- logger.warning(
- "Job '%s' (ID: %s): BLOCKED by pre-dispatch config "
- "validation — %s (no LLM call was made)",
- job_name, job_id, _pf_reason,
- )
- already_alerted = False
- try:
- from cron.jobs import mark_preflight_alerted
- already_alerted = mark_preflight_alerted(job_id)
- except Exception:
- logger.debug(
- "Job '%s': could not persist preflight alert marker",
- job_id, exc_info=True,
- )
- marker = (
- BLOCKED_CONFIG_SILENT_MARKER if already_alerted
- else BLOCKED_CONFIG_MARKER
- )
- blocked_doc = (
- f"# Cron Job: {job_name}\n\n"
- f"**Job ID:** {job_id}\n"
- f"**Run Time:** {_hermes_now().strftime('%Y-%m-%d %H:%M:%S')}\n"
- f"**Status:** BLOCKED (configuration)\n\n"
- "Pre-dispatch validation found a configuration problem and "
- "the agent was NOT run (no tokens spent).\n\n"
- f"**Reason:** {_pf_reason}\n\n"
- "The job will stay blocked (without re-alerting) until the "
- "configuration is fixed; the next healthy run clears this "
- "state. Set `cron.preflight: false` in config.yaml to "
- "disable this validation."
- )
- return False, blocked_doc, "", f"{marker} {_pf_reason}"
+ _blocked = _preflight_or_block(job, job_id, job_name, _cfg)
+ if _blocked is not None:
+ return _blocked
primary_model_for_drift = model
- configured_provider_for_drift = (
- str(_model_cfg.get("provider") or "").strip().lower()
- if isinstance(_model_cfg, dict)
- else ""
- )
- primary_provider_for_drift = (
- str(job.get("provider") or "").strip().lower()
- or configured_provider_for_drift
- or None
- )
- try:
- # Do not inject HERMES_INFERENCE_PROVIDER here. resolve_runtime_provider()
- # already prefers persisted config over stale shell/env overrides when
- # no explicit provider is requested. Passing the env var here short-
- # circuits that precedence and can resurrect old providers (for
- # example DeepSeek) for cron jobs that do not pin provider/model.
- runtime_kwargs = {
- # Per-job user pin wins; otherwise the cron-fleet default
- # provider (cron.model_provider); otherwise resolve from
- # persisted global config.
- "requested": job.get("provider") or _cron_default_provider or None,
- # Derive provider-specific api_mode from the model this job
- # will actually run (per-job pin > env > config default), not
- # the stale persisted default — mirrors the fallback path
- # below, which already passes its fb_model.
- "target_model": model,
- }
- if job.get("base_url"):
- runtime_kwargs["explicit_base_url"] = job.get("base_url")
- runtime = resolve_runtime_provider(**runtime_kwargs)
- primary_provider_for_drift = (
- str(runtime.get("provider") or "").strip().lower()
- or primary_provider_for_drift
- )
- except Exception as resolve_exc:
- # Primary provider resolution failed. Walk fallback_providers for:
- # 1) AuthError (missing/expired credential)
- # 2) Transient network/DNS failures during OAuth refresh or
- # discovery (e.g. macOS morning DNS blip → httpx.ConnectError
- # "[Errno 8] nodename nor servname provided").
- # Previously only AuthError tried the chain; a ConnectError during
- # xai-oauth token refresh killed agent crons even when XAI_API_KEY
- # / Anthropic fallbacks were healthy (Daily Focus Kickoff 2026-08-11).
- # Keeping provider+model atomic still applies — never swap only the
- # provider while retaining a paid primary model.
- is_auth = isinstance(resolve_exc, AuthError)
- is_transient_net = _is_transient_provider_resolve_error(resolve_exc)
- if not (is_auth or is_transient_net):
- raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
-
- primary_provider_for_drift = (
- str(getattr(resolve_exc, "provider", "") or "").strip().lower()
- or primary_provider_for_drift
- )
- reason = "auth" if is_auth else "transient network"
- logger.warning(
- "Job '%s': primary provider resolve failed (%s: %s), trying fallback",
- job_id,
- reason,
- resolve_exc,
- )
- fb_list = get_fallback_chain(_cfg)
- runtime = None
- for entry in fb_list:
- if not isinstance(entry, dict):
- continue
- fb_provider = str(entry.get("provider") or "").strip()
- fb_model = str(entry.get("model") or "").strip()
- if not fb_provider or not fb_model:
- continue
- try:
- from hermes_cli.fallback_config import resolve_entry_api_key
-
- fb_kwargs = {
- "requested": fb_provider,
- "target_model": fb_model,
- }
- if entry.get("base_url"):
- fb_kwargs["explicit_base_url"] = entry["base_url"]
- fb_api_key = resolve_entry_api_key(entry)
- if fb_api_key:
- fb_kwargs["explicit_api_key"] = fb_api_key
- runtime = resolve_runtime_provider(**fb_kwargs)
- model = fb_model
- logger.info(
- "Job '%s': fallback resolved to %s model %s",
- job_id,
- runtime.get("provider"),
- fb_model,
- )
- break
- except Exception as fb_exc:
- logger.debug("Job '%s': fallback %s failed: %s", job_id, fb_provider, fb_exc)
- if runtime is None:
- raise RuntimeError(format_runtime_provider_error(resolve_exc)) from resolve_exc
+ runtime, model, primary_provider_for_drift = _resolve_job_runtime(job, job_id, jc)
reasoning_config = _resolve_job_reasoning_config(
job, _cfg if isinstance(_cfg, dict) else {}, str(model)
)
-
- # Provider/model-drift fail-closed guard (#44585).
- #
- # An UNPINNED job (no explicit job["provider"]/["model"]) follows the
- # global default, which can change after the job was created — a switch
- # to a paid PROVIDER (e.g. nous) OR a paid MODEL on the same provider
- # (e.g. claude-fable-5 on openrouter). Without a guard the job would
- # silently inherit that change and spend real money on every tick — the
- # $7.73 incident named BOTH a provider and a model.
- #
- # create_job() snapshots whatever resolution would have picked at
- # creation for each unpinned axis (job["provider_snapshot"] /
- # job["model_snapshot"]). Here, for each axis that (a) has a snapshot and
- # (b) is unpinned and (c) currently resolves to a DIFFERENT value, we
- # fail closed: skip this run, make NO paid call, and deliver a loud,
- # actionable alert telling the user to pin the axis explicitly.
- #
- # Back-compat: an axis with no snapshot (pre-existing jobs, no_agent, or
- # any axis whose creation-time resolution failed) behaves exactly as
- # before — the guard never engages for it. Pinned axes are unaffected.
- #
- # cron.model / cron.model_provider: an axis resolved from the explicit
- # cron-fleet default is NOT drift — the user deliberately routed
- # unpinned cron jobs there, so the guard is skipped for that axis.
- if cron_model_drift_guard_enabled(_cfg):
- _drift: list[str] = []
- _current_provider = str(
- primary_provider_for_drift or runtime.get("provider") or ""
- ).strip().lower()
- _current_model = str(primary_model_for_drift or "").strip().lower()
- for _axis in cron_model_drift_axes(
- job,
- current_provider=_current_provider,
- current_model=_current_model,
- config=_cfg,
- ):
- _snapshot = str(job.get(f"{_axis}_snapshot") or "").strip().lower()
- _current = _current_provider if _axis == "provider" else _current_model
- _drift.append(f"{_axis} '{_snapshot}' -> '{_current}'")
- if _drift:
- _changes = "; ".join(_drift)
- # Lifecycle-aware remediation (#72056, @sashmatash): a finite
- # one-shot is consumed by this attempted dispatch — telling an
- # operator to edit a spent job is a dead end. Recurring and
- # repeatable jobs get the pin command instead.
- _repeat = job.get("repeat") if isinstance(job.get("repeat"), dict) else {}
- _finite_oneshot = (
- isinstance(job.get("schedule"), dict)
- and job["schedule"].get("kind") == "once"
- and _repeat.get("times") == 1
- )
- if _finite_oneshot:
- _remediation = (
- "This finite one-shot job is consumed by this attempted run; "
- "create a new one-shot job at a future time with an explicit "
- "provider and model."
- )
- else:
- _remediation = (
- "To run on the new config, on the host running Hermes "
- "pin it explicitly: "
- f"`hermes cron edit {job_id} --provider "
- "--model ` (or pin the original values to keep "
- "them)."
- )
- logger.warning(
- "Job '%s': SKIPPED — global inference config drifted since "
- "creation (%s) and this job is unpinned. Skipped to prevent "
- "unintended spend. %s",
- job_id,
- _changes,
- _remediation,
- )
- # Alert-once (#73506 shape): persist the drift_alerted bit so
- # only the FIRST drifted tick delivers; run_one_job suppresses
- # delivery on the silent marker. mark_job_run clears the bit
- # when a run succeeds (drift healed), re-arming the alert.
- _drift_already_alerted = False
- try:
- from cron.jobs import mark_drift_alerted
-
- _drift_already_alerted = mark_drift_alerted(job_id)
- except Exception:
- pass # fail open: better a duplicate alert than none
- _drift_marker = (
- DRIFT_SKIP_SILENT_MARKER if _drift_already_alerted
- else DRIFT_SKIP_MARKER
- )
- raise RuntimeError(
- f"{_drift_marker} Skipped to prevent unintended spend: global "
- f"inference config drifted since this job was created "
- f"({_changes}), and this job is unpinned. No inference call "
- f"was made. {_remediation} "
- f"This alert is sent once; the job stays skipped until the "
- f"config is pinned or restored. See #44585."
- )
+ _check_model_drift(
+ job, job_id, _cfg, runtime, primary_provider_for_drift, primary_model_for_drift,
+ )
fallback_model = get_fallback_chain(_cfg) or None
- credential_pool = None
- runtime_provider = str(runtime.get("provider") or "").strip().lower()
- if runtime_provider:
- try:
- from agent.credential_pool import load_pool
- pool = load_pool(runtime_provider)
- if pool.has_credentials():
- credential_pool = pool
- logger.info(
- "Job '%s': loaded credential pool for provider %s with %d entries",
- job_id,
- runtime_provider,
- len(pool.entries()),
- )
- except Exception as e:
- logger.debug("Job '%s': failed to load credential pool for %s: %s", job_id, runtime_provider, e)
+ credential_pool = _load_credential_pool(runtime, job_id)
+ # MCP servers must be registered before AIAgent is constructed.
+ _init_cron_mcp_tools(job_id)
- # Initialize MCP servers so configured mcp_servers are available to
- # the agent's tool registry before AIAgent is constructed. Without
- # this, cron jobs never saw any MCP tools — only the gateway / CLI
- # paths called discover_mcp_tools() at startup. Idempotent: subsequent
- # ticks short-circuit on already-connected servers inside
- # register_mcp_servers(). Non-fatal on failure: a broken MCP server
- # shouldn't kill an otherwise-working cron job. See #4219.
- try:
- from tools.mcp_tool import discover_mcp_tools
- _mcp_tools = discover_mcp_tools()
- if _mcp_tools:
- logger.info(
- "Job '%s': %d MCP tool(s) available",
- job_id, len(_mcp_tools),
- )
- except Exception as _mcp_exc:
- logger.warning(
- "Job '%s': MCP initialization failed (non-fatal): %s",
- job_id, _mcp_exc,
- )
-
- # Initialize the SQLite session store so cron job messages are
- # persisted and discoverable via session_search (same pattern as
- # gateway/run.py) — only now, after every early-return path
- # (wake-gate, prompt validation, drift skip) has passed, so a gated
- # run never opens state.db just to abandon the handle (#96290).
- #
- # Bounded with its own timeout (separate from HERMES_CRON_TIMEOUT,
- # which only watches the agent's run_conversation below):
- # SessionDB.__init__ opens/migrates state.db synchronously and has no
- # timeout of its own against a wedged sqlite3.connect (e.g. a stale
- # flock left by a crashed sibling process). An unbounded hang here
- # would wedge the job's worker thread, so the init is bounded and a
- # timeout proceeds without a session store instead of blocking the
- # run forever.
- _session_db_timeout = _get_session_db_timeout()
- try:
- from hermes_state import get_shared_session_db
-
- if _session_db_timeout > 0:
- _session_db_pool = concurrent.futures.ThreadPoolExecutor(max_workers=1)
- # The timeout worker is a second thread, so it does not inherit
- # the multiplexed profile ContextVar automatically. Run the
- # constructor inside a copy of the active context so a profile
- # cron run resolves ITS OWN home (and state.db) instead of
- # silently falling back to the process-global default.
- _session_db_context = contextvars.copy_context()
- _session_db_future = _session_db_pool.submit(
- _session_db_context.run, get_shared_session_db
- )
- try:
- _session_db = _session_db_future.result(timeout=_session_db_timeout)
- except concurrent.futures.TimeoutError:
- # The worker is abandoned (shutdown below doesn't wait for
- # it). If SessionDB() later completes inside it, the
- # future's result would be orphaned and its SQLite FDs
- # (.db, WAL, SHM) leak until process exit. Register a
- # done-callback that retrieves and closes any eventual
- # late result (#72782).
- _session_db_future.add_done_callback(_close_late_session_db_result)
- raise
- finally:
- # Don't wait for a wedged connect() to unwind — abandon the
- # worker thread (same pattern as the agent inactivity
- # timeout further down) rather than blocking shutdown on
- # it too.
- _session_db_pool.shutdown(wait=False)
- else:
- # 0 = unlimited (legacy behavior, opt-in for debugging)
- _session_db = get_shared_session_db()
- except concurrent.futures.TimeoutError:
- logger.error(
- "Job '%s': SessionDB init did not return within %.0fs — proceeding "
- "without a session store for this run instead of blocking it "
- "forever",
- job.get("id", "?"), _session_db_timeout,
- )
- except Exception as e:
- logger.debug("Job '%s': SQLite session store not available: %s", job.get("id", "?"), e)
+ # Open state.db only after every early-return gate has passed.
+ _session_db = _open_cron_session_db(job)
agent = AIAgent(
model=model,
@@ -6724,356 +5261,61 @@ def run_job(
enabled_toolsets=_resolve_cron_enabled_toolsets(job, _cfg),
disabled_toolsets=_resolve_cron_disabled_toolsets(_cfg),
quiet_mode=True,
- # Cron jobs should always inherit the user's SOUL.md identity from
- # HERMES_HOME. When a workdir is configured, also inject project
- # context files (AGENTS.md / CLAUDE.md / .cursorrules) from there.
- # Without a workdir, keep cwd context discovery disabled.
+ # Project context files only with a configured workdir; SOUL.md always.
skip_context_files=not bool(_job_workdir),
load_soul_identity=True,
- # Memory is enabled for cron agents like any other agent run:
- # MEMORY.md / USER.md load into the system prompt and the memory
- # tool follows normal toolset resolution, so jobs benefit from
- # (and can update) the user's persistent memory.
skip_memory=False,
skip_background_review=True, # Cron has no human-in-the-loop need for skill/memory review forks (~30K tok/event)
platform="cron",
session_id=_cron_session_id,
session_db=_session_db,
)
-
- # Run the agent with an *inactivity*-based timeout: the job can run
- # for hours if it's actively calling tools / receiving stream tokens,
- # but a hung API call or stuck tool with no activity for the configured
- # duration is caught and killed. Default 600s (10 min inactivity);
- # override via HERMES_CRON_TIMEOUT env var. 0 = unlimited.
- #
- # Uses the agent's built-in activity tracker (updated by
- # _touch_activity() on every tool call, API call, and stream delta).
- _cron_timeout = _cron_inactivity_seconds()
- _cron_inactivity_limit = _cron_timeout if _cron_timeout > 0 else None
- _POLL_INTERVAL = 5.0
- # Keep the one-shot run_claim fresh while the run is alive (#62002):
- # the claim TTL is a dead-owner detector, but without a heartbeat a
- # run that legitimately outlives it (stream stall, laptop asleep
- # mid-run) is indistinguishable from a dead tick — another process
- # re-dispatches it and get_due_jobs stale-removes the job record out
- # from under the live run. Refreshing the claim from this monitor
- # keeps "expired claim" meaning "owner died".
- _job_schedule = job.get("schedule")
- _is_oneshot = (
- isinstance(_job_schedule, dict) and _job_schedule.get("kind") == "once"
- )
- _run_claim = job.get("run_claim")
- _run_claim_owner = (
- str(_run_claim.get("by") or "") if isinstance(_run_claim, dict) else ""
- )
- _last_claim_heartbeat = time.monotonic()
- def _abort_if_fire_claim_lost() -> None:
- if cancel_event is None or not cancel_event.is_set():
- return
- if agent is not None and hasattr(agent, "interrupt"):
- agent.interrupt("Cron fire claim ownership was lost")
- raise RuntimeError(
- f"Cron job '{job_name}' lost its durable fire claim ownership"
- )
-
- def _heartbeat_run_claim_if_due():
- nonlocal _last_claim_heartbeat
- if not _is_oneshot or not _run_claim_owner:
- return
- _mono = time.monotonic()
- if _mono - _last_claim_heartbeat < _RUN_CLAIM_HEARTBEAT_SECONDS:
- return
- _last_claim_heartbeat = _mono
- try:
- heartbeat_run_claim(job_id, expected_owner=_run_claim_owner)
- except Exception:
- logger.debug(
- "Job '%s': run_claim heartbeat failed", job_name, exc_info=True
- )
-
- _cron_pool = concurrent.futures.ThreadPoolExecutor(max_workers=1)
- # Preserve scheduler-scoped ContextVar state (for example skill-declared
- # env passthrough registrations) when the cron run hops into the worker
- # thread used for inactivity timeout monitoring.
- _cron_context = contextvars.copy_context()
- # Tag this fire and time the run_conversation call for the usage_audit.jsonl entry.
_audit_fire_id = uuid.uuid4().hex
_audit_t_start = time.monotonic()
- _cron_future = _cron_pool.submit(
- _cron_context.run,
- agent.run_conversation,
- prompt,
- task_id=_cron_task_id,
+
+ def _audit(result: dict, error: Optional[str]) -> None:
+ """One usage_audit.jsonl line per fire."""
+ _write_usage_audit({
+ "ts": _utcnow_iso_ms(),
+ "job_id": job_id,
+ "fire_id": _audit_fire_id,
+ "prompt_tokens": result.get("prompt_tokens"),
+ "completion_tokens": result.get("completion_tokens"),
+ "total_tokens": result.get("total_tokens"),
+ "response_silent": bool(result.get("response_silent")),
+ "deliver_target": job.get("deliver"),
+ "model": model or None,
+ "duration_ms": int((time.monotonic() - _audit_t_start) * 1000),
+ "error": error,
+ })
+
+ result = _run_agent_with_watchdog(
+ agent, prompt, job, job_id, job_name, _cron_task_id, cancel_event,
)
- _inactivity_timeout = False
- _watch_stop = threading.Event()
-
- def _idle_seconds() -> float:
- if not hasattr(agent, "get_activity_summary"):
- return 0.0
- try:
- _act = agent.get_activity_summary()
- return float(_act.get("seconds_since_activity", 0.0) or 0.0)
- except Exception:
- return 0.0
-
- def _watch_inactivity() -> None:
- nonlocal _inactivity_timeout
- if _cron_inactivity_limit is None:
- return
- if _inactivity_watchdog_loop(
- get_idle_seconds=_idle_seconds,
- limit_s=_cron_inactivity_limit,
- poll_s=_POLL_INTERVAL,
- stop=_watch_stop,
- future_done=_cron_future.done,
- ):
- _inactivity_timeout = True
-
- _watch_thread = threading.Thread(
- target=_watch_inactivity,
- name=f"cron-inactivity-{str(job_id)[:8]}",
- daemon=True,
- )
- try:
- if _cron_inactivity_limit is not None:
- # Daemon thread: kernel ``Event.wait`` timeout, independent of
- # the ``run_job`` thread. A blocked loop / hung
- # ``get_activity_summary`` on this thread can no longer keep
- # the 600s inactivity limit from firing (#94285).
- _watch_thread.start()
- if _cron_inactivity_limit is None and not _is_oneshot and cancel_event is None:
- result = _cron_future.result()
- else:
- result = None
- while True:
- done, _ = concurrent.futures.wait(
- {_cron_future}, timeout=_POLL_INTERVAL,
- )
- if done:
- _abort_if_fire_claim_lost()
- result = _cron_future.result()
- break
- if _inactivity_timeout:
- break
- _abort_if_fire_claim_lost()
- _heartbeat_run_claim_if_due()
- except Exception:
- _cron_pool.shutdown(wait=False, cancel_futures=True)
- raise
- finally:
- _watch_stop.set()
- _cron_pool.shutdown(wait=False, cancel_futures=True)
-
- if _inactivity_timeout:
- # Build diagnostic summary from the agent's activity tracker.
- _activity = {}
- if hasattr(agent, "get_activity_summary"):
- try:
- _activity = agent.get_activity_summary()
- except Exception:
- pass
- _last_desc = _activity.get("last_activity_desc", "unknown")
- _secs_ago = _activity.get("seconds_since_activity", 0)
- _cur_tool = _activity.get("current_tool")
- _iter_n = _activity.get("api_call_count", 0)
- _iter_max = _activity.get("max_iterations", 0)
-
- logger.error(
- "Job '%s' idle for %.0fs (inactivity limit %.0fs) "
- "| last_activity=%s | iteration=%s/%s | tool=%s",
- job_name, _secs_ago, _cron_inactivity_limit,
- _last_desc, _iter_n, _iter_max,
- _cur_tool or "none",
- )
- request_hard_interrupt(agent, "Cron job timed out (inactivity)")
- raise TimeoutError(
- f"Cron job '{job_name}' idle for "
- f"{int(_secs_ago)}s (limit {int(_cron_inactivity_limit)}s) "
- f"— last activity: {_last_desc}"
- )
-
- # Guard against non-dict returns from run_conversation under error conditions
- if not isinstance(result, dict):
- raise RuntimeError(
- f"agent.run_conversation returned {type(result).__name__} instead of dict: {result!r}"
- )
-
- # If the agent itself reported failure (e.g. all retries exhausted on
- # API errors, model abort, mid-run interrupt), do not silently mark the
- # job as successful. run_agent populates `failed=True`/`completed=False`
- # on these paths and may put the error into `final_response`, which
- # would otherwise be delivered as if it were the agent's reply and the
- # job's `last_status` set to "ok". Raise so the except handler below
- # builds the proper failure tuple. (issue #17855)
- turn_exit_reason = str(result.get("turn_exit_reason") or "")
- final_response_text = (result.get("final_response") or "").strip()
- max_iteration_summary = (
- result.get("failed") is not True
- and result.get("completed") is False
- and turn_exit_reason.startswith("max_iterations_reached(")
- and bool(final_response_text)
- )
- if result.get("failed") is True or (result.get("completed") is False and not max_iteration_summary):
- _err_text = (
- result.get("error")
- or final_response_text
- or "agent reported failure"
- )
- raise RuntimeError(_err_text)
- if max_iteration_summary:
- logger.warning(
- "Job '%s' reached the iteration limit but produced a final fallback response; "
- "delivering the response instead of failing the cron run",
- job_name,
- )
-
- final_response = result.get("final_response", "") or ""
- # Recover model-mangled computer_use screenshot paths before delivery
- # media extraction (same repair as the gateway turn/background paths).
- # Cron runs start a fresh conversation, so history_offset=0. The
- # helper is fail-open and no-ops without a MEDIA: directive.
- if final_response:
- from gateway.media_repair import (
- repair_explicit_computer_use_media_paths,
- )
-
- final_response = repair_explicit_computer_use_media_paths(
- final_response,
- result.get("messages", []),
- )
- # Strip leaked placeholder text that upstream may inject on empty completions.
- if final_response.strip() == "(No response generated)":
- final_response = ""
- # Cron silence on abnormal empty turns. The turn-completion explainer
- # (#34452) replaces a blank/empty model turn with a "⚠️ No reply: …"
- # string so interactive surfaces (CLI/gateway) explain why the box is
- # empty. In a cron context that turns a previously-silent empty turn
- # into a delivered warning (Manfredi's Telegram symptom). Detect the
- # explainer text deterministically (via the same formatter that
- # produced it) and treat it as empty so the empty-response suppression
- # and soft-failure marking below apply — restoring pre-#34452 silence
- # for scheduled jobs without disabling the explainer everywhere.
- if final_response.strip() and turn_exit_reason:
- # The formatter's wording varies by persistence cause (locked /
- # disk / unknown), so render every variant — matching only the
- # one-argument render would let cause-refined explainer text slip
- # through and be delivered as a cron warning.
- _explainer_variants = []
- try:
- from hermes_state import PERSISTENCE_ERROR_CAUSES as _causes
- except Exception:
- _causes = ("locked", "disk", "unknown")
- for _cause in (None, *_causes):
- try:
- _variant = AIAgent._format_turn_completion_explanation(
- turn_exit_reason, _cause
- )
- except TypeError:
- # Older single-argument formatter (or a test double).
- try:
- _variant = AIAgent._format_turn_completion_explanation(
- turn_exit_reason
- )
- except Exception:
- _variant = ""
- except Exception:
- _variant = ""
- if _variant:
- _explainer_variants.append(_variant.strip())
- if final_response.strip() in _explainer_variants:
- logger.info(
- "Job '%s': abnormal empty turn (%s) — suppressing explainer for cron delivery",
- job_id,
- turn_exit_reason,
- )
- final_response = ""
- # Use a separate variable for log display; keep final_response clean
- # for delivery logic (empty response = no delivery).
+ final_response = _final_response_from_result(result, job_id, job_name, AIAgent)
+ # Keep final_response clean for delivery logic (empty = no delivery).
logged_response = final_response if final_response else "(No response generated)"
-
- output = f"""# Cron Job: {job_name}
-
-**Job ID:** {job_id}
-**Run Time:** {_hermes_now().strftime('%Y-%m-%d %H:%M:%S')}
-**Schedule:** {job.get('schedule_display', 'N/A')}
-
-## Prompt
-
-{prompt}
-
-## Response
-
-{logged_response}
-"""
-
+ output = _run_doc_header(job, job_name, job_id, prompt) + f"## Response\n\n{logged_response}\n"
logger.info("Job '%s' completed successfully", job_name)
-
- # Emit one JSONL line per fire for usage audit.
- _audit_duration_ms = int((time.monotonic() - _audit_t_start) * 1000)
- _audit_response_silent = _is_cron_silence_response(final_response or "")
- _write_usage_audit({
- "ts": _utcnow_iso_ms(),
- "job_id": job_id,
- "fire_id": _audit_fire_id,
- "prompt_tokens": result.get("prompt_tokens"),
- "completion_tokens": result.get("completion_tokens"),
- "total_tokens": result.get("total_tokens"),
- "response_silent": _audit_response_silent,
- "deliver_target": job.get("deliver"),
- "model": model or None,
- "duration_ms": _audit_duration_ms,
- "error": None,
- })
+ _audit(dict(result, response_silent=_is_cron_silence_response(final_response or "")), None)
return True, output, final_response, None
except Exception as e:
error_msg = f"{type(e).__name__}: {str(e)}"
logger.exception("Job '%s' failed: %s", job_name, error_msg)
- # Best-effort audit write on failure path. _audit_fire_id
- # may be unset if the exception fired before submit() — guard
- # with a None check so the audit write itself never raises.
- if "_audit_fire_id" in locals():
- _audit_duration_ms = int((time.monotonic() - _audit_t_start) * 1000)
- _write_usage_audit({
- "ts": _utcnow_iso_ms(),
- "job_id": job_id,
- "fire_id": _audit_fire_id,
- "prompt_tokens": None,
- "completion_tokens": None,
- "total_tokens": None,
- "response_silent": False,
- "deliver_target": job.get("deliver"),
- "model": model or None,
- "duration_ms": _audit_duration_ms,
- "error": error_msg,
- })
-
- output = f"""# Cron Job: {job_name} (FAILED)
-
-**Job ID:** {job_id}
-**Run Time:** {_hermes_now().strftime('%Y-%m-%d %H:%M:%S')}
-**Schedule:** {job.get('schedule_display', 'N/A')}
-
-## Prompt
-
-{prompt}
-
-## Error
-
-```
-{error_msg}
-```
-"""
+ # _audit is unbound if we failed before the agent ran; the audit write must never raise.
+ if "_audit" in locals():
+ _audit({}, error_msg)
+ output = (
+ _run_doc_header(job, f"{job_name} (FAILED)", job_id, prompt)
+ + f"## Error\n\n```\n{error_msg}\n```\n"
+ )
return False, output, "", error_msg
finally:
_clear_tool_session_cwd(_cron_task_id)
- # Clean up ContextVar session/delivery state for this job.
- # clear_session_vars also clears _SESSION_CWD internally, so no
- # separate clear_session_cwd() call is needed.
+ # clear_session_vars also clears _SESSION_CWD.
clear_session_vars(_ctx_tokens)
if _cron_session_token is not None:
_cron_session_var.reset(_cron_session_token)
@@ -7082,121 +5324,9 @@ def run_job(
for _var_name in _cron_delivery_vars:
_VAR_MAP[_var_name].set("")
if _session_db:
- # The agent turn has already returned. Bound every subsequent DB
- # operation so storage failure cannot hold the dispatch guard.
- _session_db = _BoundedCronSessionDB(_session_db, job_id)
- # Compression can rotate the live agent onto a continuation while
- # this run is in flight. Finalize that continuation, not the stale
- # cron id captured before AIAgent started. SessionDB is the source
- # of truth for the lineage; agent.session_id is only a fail-safe
- # when the lookup itself is unavailable.
- _final_cron_session_id = _cron_session_id
- try:
- _compression_tip = _session_db.get_compression_tip(
- _cron_session_id
- )
- if _compression_tip:
- _final_cron_session_id = _compression_tip
- except (Exception, KeyboardInterrupt) as e:
- try:
- _agent_session_id = getattr(agent, "session_id", None)
- if _agent_session_id:
- _final_cron_session_id = _agent_session_id
- except (Exception, KeyboardInterrupt):
- pass
- logger.debug(
- "Job '%s': failed to resolve cron compression tip: %s",
- job_id,
- e,
- )
- # Title the cron session from the job (name -> id) and PERSIST it
- # BEFORE end_session()/close() tear the connection down, so the
- # close can never run over an in-flight title write (#50536). The
- # run-time suffix keeps it unique against the sessions.title index
- # across runs; _set_cron_session_title dedupes (#50537) and the
- # except-fallback below guarantees a non-blank title (#50535).
- try:
- _title_base = " ".join(job_name.split())[:60].strip() or f"cron {job_id}"
- _cron_title = f"{_title_base} · {_hermes_now().strftime('%b %d %H:%M')}"
- if not _set_cron_session_title(
- _session_db, _final_cron_session_id, _cron_title
- ):
- # Helper returned None (blank base) -> use the id fallback.
- _set_cron_session_title(
- _session_db, _final_cron_session_id, f"cron {job_id}"
- )
- except (Exception, KeyboardInterrupt) as e:
- logger.debug(
- "Job '%s': failed to set cron session title: %s", job_id, e
- )
- # Last-resort: never leave the session blank (#50535). Try the
- # next free title in the lineage, then a bare id-stamped title.
- for _fallback in (
- getattr(_session_db, "get_next_title_in_lineage", lambda b: b)(
- f"cron {job_id}"
- ),
- f"cron {job_id} {_final_cron_session_id[-6:]}",
- ):
- try:
- if _set_cron_session_title(
- _session_db, _final_cron_session_id, _fallback
- ):
- break
- except (Exception, KeyboardInterrupt):
- continue
- # Verified completion booking (#93820): the run may only be
- # recorded as cron_complete when the session's LAST message row is
- # a real assistant reply — a plain answer or the [SILENT] sentinel
- # (both are assistant-text rows, so both classify as 'complete').
- # A turn that died after a tool call, mid-API-wait, or without any
- # assistant text leaves the last row as a tool result / pending
- # call / user prompt and must not surface as a healthy run.
- # session_lifecycle_statuses is the existing cost-bounded
- # classifier for exactly this shape. Only a POSITIVELY recognized
- # pathological status (see the status vocabulary in
- # hermes_state's session_lifecycle_statuses docstring — keep the
- # tuple below in sync when it grows) downgrades the booking: an
- # unknown value (newer classifier shape, test doubles) keeps the
- # historical reason, and so does a failed probe — the booking
- # itself is FAIL-OPEN on probe errors, because classification is
- # best-effort metadata and must not mislabel a healthy run.
- _end_reason = "cron_complete"
- try:
- _statuses = _session_db.session_lifecycle_statuses(
- [_final_cron_session_id]
- )
- _lifecycle = _statuses.get(_final_cron_session_id)
- if _lifecycle in ("interrupted", "error", "empty"):
- _end_reason = "cron_incomplete_no_output"
- logger.warning(
- "Job '%s': session ended without a final assistant "
- "message (lifecycle=%s) — booking run as %s",
- job_id, _lifecycle, _end_reason,
- )
- except (Exception, KeyboardInterrupt) as e:
- logger.debug(
- "Job '%s': session lifecycle classification failed: %s",
- job_id, e,
- )
- try:
- _session_db.end_session(
- _final_cron_session_id, _end_reason
- )
- except (Exception, KeyboardInterrupt) as e:
- logger.debug("Job '%s': failed to end session: %s", job_id, e)
- try:
- from hermes_state import release_or_close
- release_or_close(_session_db)
- except (Exception, KeyboardInterrupt) as e:
- logger.debug("Job '%s': failed to close SQLite session store: %s", job_id, e)
- # Release subprocesses, terminal sandboxes, browser daemons, and the
- # main OpenAI/httpx client held by this ephemeral cron agent. Without
- # this, a gateway that ticks cron every N minutes leaks fds per job
- # until it hits EMFILE (#10200 / "too many open files").
- #
- # When the caller opted to defer teardown (passed a list), hand the live
- # agent back instead of closing it here — delivery must run against a
- # live async client, and the caller tears down afterwards (#58720).
+ _finalize_cron_session(_session_db, agent, job_id, job_name, _cron_session_id)
+ # Tear down the ephemeral agent or the gateway leaks fds per tick (EMFILE). With deferred
+ # teardown, hand the live agent back: delivery needs a live async client.
if defer_agent_teardown is not None:
if agent is not None:
defer_agent_teardown.append(agent)
@@ -7209,10 +5339,8 @@ def _teardown_cron_agent(
) -> None:
"""Release an ephemeral cron agent's async resources within a hard bound.
- Split out of ``run_job``'s ``finally`` so a caller that defers teardown
- (to deliver first — #58720) can invoke the identical cleanup AFTER delivery.
- The timeout matters because this executes after ``run_conversation`` has
- returned, outside the agent inactivity watchdog.
+ Split out of ``run_job``'s ``finally`` so a caller deferring teardown until after delivery runs
+ the identical cleanup. Bounded because this runs outside the agent inactivity watchdog.
"""
def _cleanup_agent() -> None:
try:
@@ -7220,9 +5348,7 @@ def _teardown_cron_agent(
agent.close()
except (Exception, KeyboardInterrupt) as e:
logger.debug("Job '%s': failed to close agent resources: %s", job_id, e)
- # Each cron run spins up a short-lived worker thread whose event loop
- # dies as soon as the ``ThreadPoolExecutor`` shuts down. Any async
- # httpx clients cached under that loop are now unusable — reap them.
+ # Worker-thread event loop dies with the executor; reap httpx clients cached under it.
try:
from agent.auxiliary_client import cleanup_stale_async_clients
cleanup_stale_async_clients()
@@ -7247,7 +5373,6 @@ def _run_with_fire_claim_heartbeat(job: dict, run) -> bool:
job_id = str(job.get("id") or "")
stop = threading.Event()
lost_ownership = threading.Event()
- heartbeat_context = contextvars.copy_context()
def _finish_unstarted(error: str) -> None:
execution_id = job.get("execution_id")
@@ -7265,21 +5390,12 @@ def _run_with_fire_claim_heartbeat(job: dict, run) -> bool:
try:
owns_fire_claim = heartbeat_fire_claim(job_id, expected_owner=owner)
except Exception:
- logger.warning(
- "Job '%s': initial fire_claim validation failed",
- job_id,
- exc_info=True,
- )
- _finish_unstarted(
- "Fire claim ownership could not be validated before execution started."
- )
+ logger.warning("Job '%s': initial fire_claim validation failed", job_id, exc_info=True)
+ _finish_unstarted("Fire claim ownership could not be validated before execution started.")
return True
if owns_fire_claim is False:
- logger.warning(
- "Job '%s': fire claim ownership was already lost before execution",
- job_id,
- )
+ logger.warning("Job '%s': fire claim ownership was already lost before execution", job_id)
_finish_unstarted("Fire claim ownership lost before execution started.")
return True
@@ -7296,11 +5412,7 @@ def _run_with_fire_claim_heartbeat(job: dict, run) -> bool:
return
last_confirmed = time.monotonic()
except Exception:
- logger.debug(
- "Job '%s': fire_claim heartbeat failed",
- job_id,
- exc_info=True,
- )
+ logger.debug("Job '%s': fire_claim heartbeat failed", job_id, exc_info=True)
if (
time.monotonic() - last_confirmed
>= _FIRE_CLAIM_HEARTBEAT_GRACE_SECONDS
@@ -7314,23 +5426,14 @@ def _run_with_fire_claim_heartbeat(job: dict, run) -> bool:
)
return
- heartbeat_thread = threading.Thread(
- target=heartbeat_context.run,
- args=(_heartbeat_loop,),
- name="cron-fire-claim-heartbeat",
- daemon=True,
+ heartbeat_thread = _start_heartbeat_thread(
+ _heartbeat_loop, "cron-fire-claim-heartbeat",
+ lambda: logger.warning(
+ "Job '%s': could not start fire_claim heartbeat", job_id, exc_info=True,
+ ),
)
- try:
- heartbeat_thread.start()
- except Exception:
- logger.warning(
- "Job '%s': could not start fire_claim heartbeat",
- job_id,
- exc_info=True,
- )
- _finish_unstarted(
- "Fire claim heartbeat could not be started; execution was not run."
- )
+ if heartbeat_thread is None:
+ _finish_unstarted("Fire claim heartbeat could not be started; execution was not run.")
return True
try:
@@ -7351,29 +5454,14 @@ def run_one_job(
) -> bool:
"""Run ONE due job end-to-end: execute → save output → deliver → mark.
- This is the shared firing body extracted from ``tick``'s per-job closure so
- that BOTH the built-in ticker and an external provider's ``fire_due`` (e.g.
- Chronos) run the identical sequence — no duplicated correctness.
-
- It does NOT decide whether the job is due or acquire the initial claim —
- both the ticker and external providers use the same store CAS before
- calling it. It does keep an acquired claim alive for the full execution.
-
- Returns True if the job was processed (even if the job itself failed —
- failure is recorded via ``mark_job_run``), False only if processing raised.
-
- ``cancel_event``: optional transport-level cancellation source (dashboard
- webhook drain, API server shutdown). It is OR-combined with the internal
- fire-claim heartbeat's lost-ownership event, so either trigger stops the
- run cooperatively — agent interruption AND script process-tree kill —
- through the single fenced completion path.
+ Shared firing body for BOTH the built-in ticker and external providers' ``fire_due``. Does NOT
+ decide due-ness or acquire the initial claim (callers use the same store CAS); does keep the
+ claim alive. Returns True if processed (job failure is recorded via ``mark_job_run``), False
+ only if processing raised. ``cancel_event``: optional transport-level cancel (dashboard drain).
"""
if extra_prompt is None:
- # A gateway-forwarded manual run (`hermes cron run --prompt` /
- # cronjob(action='run', prompt=...) on a relay-fronted target) stamps
- # its transient context on the job via trigger_job; the ticker/Chronos
- # fire that consumes the manual occurrence carries it here. Single-fire:
- # mark_job_run clears the field after the run.
+ # Gateway-forwarded manual run stamps its prompt on the job via trigger_job; the fire that
+ # consumes the manual occurrence picks it up here. Single-fire: mark_job_run clears it.
_stamped = job.get("manual_run_prompt")
if _stamped and job.get("manual_run_at"):
extra_prompt = str(_stamped)
@@ -7412,6 +5500,91 @@ def run_one_job(
_running_fire_owners.pop(job["id"], None)
+_OWNERSHIP_LOST_INTERRUPTED = "Interrupted by shutdown before terminal completion."
+
+
+def _record_fire_ownership_lost(job_id: str, fire_owner: Optional[str], execution_id: str) -> None:
+ """Bookkeeping after fire-claim ownership loss. A transport-level cancel (dashboard drain) is
+ not a real loss — we still own the claim, so record the interruption via the owner-fenced
+ terminal write instead of leaving fire_claim/last_status stale; otherwise discard."""
+ if fire_owner is not None and heartbeat_fire_claim(job_id, expected_owner=fire_owner):
+ mark_job_run(job_id, False, _OWNERSHIP_LOST_INTERRUPTED, expected_fire_owner=fire_owner)
+ finish_execution(execution_id, success=False, error=_OWNERSHIP_LOST_INTERRUPTED)
+ else:
+ finish_execution(
+ execution_id,
+ success=False,
+ error="Fire claim ownership lost; stale result was discarded.",
+ )
+
+
+def _classify_delivery_outcome(
+ *, delivery_error, should_deliver: bool, unresolved_origin: bool,
+ normalized_deliver: str, incident_acked: bool, success: bool,
+) -> str:
+ if delivery_error:
+ return "failed"
+ if should_deliver and unresolved_origin:
+ return "not_configured"
+ if should_deliver and normalized_deliver != "local":
+ return "delivered"
+ if incident_acked and not success:
+ # Failure ping withheld: operator acked this exact signature (vs. plain "suppressed").
+ return "suppressed_acked"
+ return "suppressed"
+
+
+def _compose_run_delivery(
+ job: dict, *, success: bool, error, final_response: str, output_file,
+) -> tuple[str, bool, bool, bool, Optional[str]]:
+ """Build the text to deliver for a finished run.
+
+ Returns ``(deliver_content, blocked_config, silent_alert, incident_acked, failure_incident_id)``.
+ ``silent_alert``: an alert-once marker says the operator was already told; deliver nothing.
+ """
+ err = str(error) if error else ""
+ # Failed jobs always deliver, except blocked-config / drift-skip runs, which alert exactly ONCE.
+ blocked_config_silent = BLOCKED_CONFIG_SILENT_MARKER in err
+ blocked_config = blocked_config_silent or BLOCKED_CONFIG_MARKER in err
+ drift_skip_silent = DRIFT_SKIP_SILENT_MARKER in err
+ drift_skip = drift_skip_silent or DRIFT_SKIP_MARKER in err
+ incident_acked = False
+ failure_incident_id = None
+ if blocked_config and not success:
+ # Bypass the generic failure summarizer (its auth/timeout heuristics would mislabel this).
+ _pf_text = re.sub(r"\[blocked_config[^\]]*\]\s*", "", err).strip()
+ deliver_content = (
+ f"⛔ Cron '{job.get('name') or job['id']}' blocked by "
+ f"configuration validation (no LLM call was made): "
+ f"{_pf_text} "
+ "This alert is sent once; the job stays blocked until "
+ "the configuration is fixed."
+ )
+ elif success:
+ deliver_content = final_response
+ else:
+ # Record the job+error signature once; if already acked by the operator, suppress the
+ # per-run ping. Best-effort: a ledger failure never breaks delivery.
+ incident_acked, failure_incident_id = _upsert_incident_for_failure(
+ job, error or "", output_file=output_file
+ )
+ if incident_acked and not drift_skip:
+ deliver_content = ""
+ else:
+ deliver_content = (
+ _summarize_cron_failure_for_delivery(job, error) + _failure_streak_nudge(job)
+ )
+ if drift_skip:
+ # Deliver the guard's message intact (summarizer truncation would eat the remediation
+ # command). NOT gated on incident ack: acks silence failure pings, not drift alerts.
+ _drift_text = re.sub(r"\[drift_skip[^\]]*\]\s*", "", err).strip()
+ deliver_content = f"⚠️ Cron '{job.get('name') or job['id']}' skipped: {_drift_text}"
+ return (
+ deliver_content, blocked_config, blocked_config_silent or drift_skip_silent,
+ incident_acked, failure_incident_id,
+ )
+
+
def _run_one_job_body(
job: dict,
*,
@@ -7457,9 +5630,6 @@ def _run_one_job_body(
execution_id = create_execution(job["id"], source="direct")["id"]
delivery_attempted = False
delivery_error = None
- # Durable failure-incident bookkeeping for this run (see cron.incidents):
- # set on the failure paths below; consumed by the delivery_outcome
- # computation and the post-delivery "alerted" transition.
incident_acked = False
failure_incident_id = None
from agent.secret_scope import (
@@ -7469,14 +5639,10 @@ def _run_one_job_body(
)
_scope_token = None
+ _terminal_scope_token = None
try:
- # Pre-run dispatch claim (issue #38758): atomically commit a finite
- # one-shot's dispatch BEFORE its side effect runs, so a tick that dies
- # mid-execution (gateway kill, OOM, segfault, hard-timeout) cannot
- # re-fire the job forever on restart. No-op for recurring jobs (they
- # use advance_next_run) and infinite/no-repeat jobs. This lives here in
- # the shared body so BOTH the built-in ticker and the external provider
- # (Chronos fire_due) get at-most-times semantics.
+ # Commit a finite one-shot's dispatch BEFORE its side effect so a tick dying mid-run cannot
+ # re-fire it forever on restart. No-op for recurring/infinite jobs (at-most-times).
if not claim_dispatch(job["id"]):
logger.info(
"Job '%s': one-shot dispatch limit reached — skipping",
@@ -7489,107 +5655,49 @@ def _run_one_job_body(
)
return True # not an error — already handled/removed
- # The attempt is claimed durably before executor/provider dispatch and
- # becomes running only immediately before the actual run.
mark_execution_running(execution_id)
- # Run and deliver under the profile's secret scope. get_secret() fails
- # closed outside a scope once profile isolation is active, and cron
- # fires from a ticker thread with no per-turn scope. Delivery adapters
- # can also resolve credentials, so resetting after run_job would leave
- # _deliver_result unscoped. Mirrors gateway/run.py's per-turn pattern.
-
- _scope_token = set_secret_scope(
- build_profile_secret_scope(_get_hermes_home())
- )
- # Same isolation for terminal settings (third profile seam; see
- # gateway/run.py _profile_runtime_scope): installs the firing
- # profile's COMPLETE terminal policy for this fire — run, delivery,
- # and bookkeeping — resetting in this function's finally alongside
- # the secret scope. Without it the ticker thread reads the
- # process-global TERMINAL_* env vars a concurrent profile's turn may
- # have pinned (#68559). Resolution failure installs a refusal scope:
- # terminal execution inside the fire raises instead of falling back
- # to the launch process's ambient policy.
+ # get_secret() fails closed outside a scope; the ticker thread has none. Delivery adapters
+ # resolve credentials, so the scope must span delivery too (reset in the outer finally).
+ _scope_token = set_secret_scope(build_profile_secret_scope(_get_hermes_home()))
+ # Same for terminal policy (gateway/run.py _profile_runtime_scope): else the ticker reads
+ # process-global TERMINAL_* env a concurrent profile pinned. Resolution failure installs a
+ # refusal scope — terminal execution raises instead of using the launch process's policy.
from tools.terminal_scope import (
install_profile_terminal_scope,
)
- _terminal_scope_token = install_profile_terminal_scope(
- _get_hermes_home()
- )
- # Defer the cron agent's async-resource teardown until AFTER delivery.
- # run_job normally closes the agent (and reaps stale async clients) in
- # its finally block; doing that before _deliver_result runs means the
- # live send races a torn-down async client (#58720). Passing a holder
- # list makes run_job hand the agent back instead, and we tear it down
- # below once delivery is done. Defense-in-depth alongside the
- # interpreter-shutdown guard in _deliver_result.
+ _terminal_scope_token = install_profile_terminal_scope(_get_hermes_home())
+ # Defer agent teardown until AFTER delivery; closing first races the live send against a
+ # torn-down async client. run_job hands the agent back via this list.
_deferred_agents: list = []
- try:
- if fire_claim_lost is None:
- success, output, final_response, error = run_job(
- job,
- defer_agent_teardown=_deferred_agents,
- extra_prompt=extra_prompt,
- execution_id=execution_id,
- )
- else:
- success, output, final_response, error = run_job(
- job,
- defer_agent_teardown=_deferred_agents,
- extra_prompt=extra_prompt,
- cancel_event=fire_claim_lost,
- execution_id=execution_id,
- )
- except BaseException:
- # run_job's finally still hands back the agent when it raises; tear
- # it down here so a failed run never leaks its async resources
- # (#10200), then re-raise into the outer handler. BaseException
- # (not just Exception) so a KeyboardInterrupt/SystemExit mid-run
- # still triggers teardown before propagating.
+
+ def _teardown_deferred() -> None:
for _deferred_agent in _deferred_agents:
_teardown_cron_agent(_deferred_agent, job["id"])
+
+ _run_kwargs = {
+ "defer_agent_teardown": _deferred_agents,
+ "extra_prompt": extra_prompt,
+ "execution_id": execution_id,
+ }
+ if fire_claim_lost is not None:
+ _run_kwargs["cancel_event"] = fire_claim_lost
+ try:
+ success, output, final_response, error = run_job(job, **_run_kwargs)
+ except BaseException:
+ # run_job hands back the agent even when raising; tear down so a failed run never leaks.
+ # BaseException so KeyboardInterrupt/SystemExit mid-run still trigger teardown.
+ _teardown_deferred()
raise
- # The outer finally resets the scope after delivery and bookkeeping.
if _fire_claim_ownership_lost():
- for _deferred_agent in _deferred_agents:
- _teardown_cron_agent(_deferred_agent, job["id"])
- # Distinguish a real ownership loss (TTL expiry / replacement
- # claim) from a transport-level cancel (dashboard drain): in the
- # latter case WE still own the claim, and silently discarding
- # would leave fire_claim lingering until TTL and last_status
- # stale. Probe ownership once; if still ours, record the
- # interruption through the owner-fenced terminal write.
- if fire_owner is not None and heartbeat_fire_claim(
- job["id"], expected_owner=fire_owner,
- ):
- mark_job_run(
- job["id"],
- False,
- "Interrupted by shutdown before terminal completion.",
- expected_fire_owner=fire_owner,
- )
- finish_execution(
- execution_id,
- success=False,
- error="Interrupted by shutdown before terminal completion.",
- )
- else:
- finish_execution(
- execution_id,
- success=False,
- error="Fire claim ownership lost; stale result was discarded.",
- )
+ _teardown_deferred()
+ _record_fire_ownership_lost(job["id"], fire_owner, execution_id)
return True
- # Everything from here through delivery runs with the agent still live
- # (deferred teardown). Wrap it ALL in a try/finally so that if any step
- # between run_job returning and delivery — save_job_output, the [SILENT]
- # / empty-response computation, or _deliver_result itself — raises, the
- # deferred agent is still torn down. Otherwise the outer `except` would
- # swallow the error and leak the agent's subprocesses/clients (#10200).
+ # Agent is still live through delivery; wrap ALL of save/compose/deliver in try/finally so a
+ # raise anywhere still tears the deferred agent down.
blocked_config = False
side_effect_ownership_lost = False
try:
@@ -7600,13 +5708,8 @@ def _run_one_job_body(
if verbose:
logger.info("Output saved to: %s", output_file)
- # If the gateway shutdown killed this job's tool subprocess
- # mid-flight (#60432), the agent may still have produced a
- # plausible-looking final_response from the truncated output --
- # force the failure path so the delivered message is an honest
- # "this run was interrupted" summary instead of that response.
- # Peek-only: the flag stays set for the authoritative check
- # right before mark_job_run below.
+ # A shutdown-killed tool subprocess can leave a plausible final_response from truncated
+ # output; force the honest "interrupted" failure path. Peek-only (consumed later).
if success and _is_interrupted(job["id"], execution_token):
success = False
error = (
@@ -7614,95 +5717,18 @@ def _run_one_job_body(
"(tool subprocess was killed mid-flight)."
)
- # Deliver the final response to the origin/target chat.
- # If the agent responded with [SILENT], skip delivery (but
- # output is already saved above). Failed jobs always deliver.
- #
- # Exception: a run blocked by pre-dispatch config validation
- # (T1-26) alerts exactly ONCE — the silent marker means the
- # operator was already told on a previous tick, so re-delivering
- # the same alert every tick would be spam (#73506 alert-once
- # shape).
- blocked_config_silent = (
- bool(error) and BLOCKED_CONFIG_SILENT_MARKER in str(error)
+ (
+ deliver_content, blocked_config, _silent_alert,
+ incident_acked, failure_incident_id,
+ ) = _compose_run_delivery(
+ job, success=success, error=error, final_response=final_response,
+ output_file=output_file,
)
- blocked_config = blocked_config_silent or (
- bool(error) and BLOCKED_CONFIG_MARKER in str(error)
- )
- # Drift-guard skip (#44585): same alert-once contract as
- # blocked_config — the silent marker means the operator already
- # got the alert on a previous tick.
- drift_skip_silent = (
- bool(error) and DRIFT_SKIP_SILENT_MARKER in str(error)
- )
- drift_skip = drift_skip_silent or (
- bool(error) and DRIFT_SKIP_MARKER in str(error)
- )
- if blocked_config and not success:
- # Blocked-config alert: bypass the generic failure summarizer
- # (whose auth/timeout heuristics would mislabel this as a
- # provider runtime failure) — say plainly that config
- # validation blocked the run and nothing was spent.
- _pf_text = re.sub(
- r"\[blocked_config[^\]]*\]\s*", "", str(error)
- ).strip()
- deliver_content = (
- f"⛔ Cron '{job.get('name') or job['id']}' blocked by "
- f"configuration validation (no LLM call was made): "
- f"{_pf_text} "
- "This alert is sent once; the job stays blocked until "
- "the configuration is fixed."
- )
- else:
- if success:
- deliver_content = final_response
- else:
- # Durable failure incident: record this job+error
- # signature once and, when the operator already acked it,
- # suppress the per-run failure ping (the streak nudge and
- # the failure summarizer stay intact for un-acked
- # failures). Best-effort — an incident-store error never
- # breaks the delivery path (see _upsert_incident_for_failure).
- incident_acked, failure_incident_id = _upsert_incident_for_failure(
- job, error or "", output_file=output_file
- )
- if incident_acked and not drift_skip:
- deliver_content = ""
- else:
- deliver_content = (
- _summarize_cron_failure_for_delivery(job, error)
- + _failure_streak_nudge(job)
- )
- if drift_skip and not success:
- # Drift-skip alert: bypass the generic summarizer's
- # 180-char truncation (it would eat the remediation
- # command) and strip the internal marker — deliver the
- # guard's own actionable message intact.
- # Deliberately NOT gated on incident ack: a drift skip
- # means the run was never attempted and the message
- # carries the remediation command — acking the failure
- # signature silences failure pings, not drift alerts
- # (which already alert once via the drift_alerted marker).
- _drift_text = re.sub(
- r"\[drift_skip[^\]]*\]\s*", "", str(error)
- ).strip()
- deliver_content = (
- f"⚠️ Cron '{job.get('name') or job['id']}' skipped: "
- f"{_drift_text}"
- )
- # Treat whitespace-only final responses the same as empty
- # responses: do not deliver a blank message, and let the
- # empty-response guard below mark the run as a soft failure.
- should_deliver = bool(deliver_content.strip())
- if blocked_config_silent or drift_skip_silent:
- should_deliver = False
+ # Whitespace-only == empty: skip delivery; the guard below marks it a soft failure.
+ should_deliver = bool(deliver_content.strip()) and not _silent_alert
unresolved_origin = False
- # Cron silence suppression — see _is_cron_silence_response. Replaces the
- # old `SILENT_MARKER in ...upper()` substring check, which both leaked
- # bracketless near-markers ("SILENT" / "NO_REPLY") and wrongly swallowed
- # a real report that merely quoted "[SILENT]" mid-sentence (#51438,
- # #46917). Keeps the intentional bracketed-prefix / trailing-line
- # tolerance the cron contract relies on.
+ # Not a substring check: bare "SILENT"/"NO_REPLY" or a report quoting "[SILENT]" must
+ # not be swallowed; bracketed-prefix / trailing-line tolerance is kept.
if should_deliver and success and _is_cron_silence_response(deliver_content):
logger.info("Job '%s': agent returned %s — skipping delivery", job["id"], SILENT_MARKER)
should_deliver = False
@@ -7743,41 +5769,14 @@ def _run_one_job_body(
except _FireClaimLostDuringSideEffect:
side_effect_ownership_lost = True
finally:
- # Tear down the deferred agent(s) now that save + delivery have run
- # (or raised). Must happen on every path so cron agents never leak
- # their subprocesses/clients (#10200).
- for _deferred_agent in _deferred_agents:
- _teardown_cron_agent(_deferred_agent, job["id"])
+ # Every path must tear down deferred agent(s) so they never leak subprocesses/clients.
+ _teardown_deferred()
if side_effect_ownership_lost or _fire_claim_ownership_lost():
- # Same transport-cancel distinction as the pre-side-effect path:
- # if WE still own the claim, record the interruption instead of
- # discarding silently (lingering claim + stale last_status).
- if fire_owner is not None and heartbeat_fire_claim(
- job["id"], expected_owner=fire_owner,
- ):
- mark_job_run(
- job["id"],
- False,
- "Interrupted by shutdown before terminal completion.",
- expected_fire_owner=fire_owner,
- )
- finish_execution(
- execution_id,
- success=False,
- error="Interrupted by shutdown before terminal completion.",
- )
- else:
- finish_execution(
- execution_id,
- success=False,
- error="Fire claim ownership lost; stale result was discarded.",
- )
+ _record_fire_ownership_lost(job["id"], fire_owner, execution_id)
return True
- # Treat empty final_response as a soft failure so last_status
- # is not "ok" — the agent ran but produced nothing useful.
- # (issue #8585)
+ # Empty final_response is a soft failure so last_status is not "ok".
if success and not final_response.strip():
success = False
error = "Agent completed but produced empty response (model error, timeout, or misconfiguration)"
@@ -7785,14 +5784,8 @@ def _run_one_job_body(
interrupted = _consume_interrupted_flag(job["id"], execution_token)
if interrupted:
if delivery_error:
- # The gateway shutdown already wrote last_status for this run,
- # so mark_job_run is skipped below — but it could not know that
- # the notice we just tried to send never left the process (the
- # adapters were torn down first, #82232). Record the delivery
- # failure on its own via update_job: mark_job_run also advances
- # next_run_at and the repeat counter, and running that a second
- # time for one run would skip a fire or auto-delete the job
- # early.
+ # Shutdown already wrote last_status so mark_job_run is skipped below (a second call
+ # would skip a fire or auto-delete the job); note the unsent notice via update_job.
try:
from cron.jobs import update_job
update_job(job["id"], {"last_delivery_error": delivery_error})
@@ -7821,26 +5814,17 @@ def _run_one_job_body(
error="Fire claim ownership lost before terminal completion.",
)
return True
- normalized_deliver = _normalize_deliver_value(
- _delivery_lane_value(job, for_failure=not success)
+ delivery_outcome = _classify_delivery_outcome(
+ delivery_error=delivery_error,
+ should_deliver=should_deliver,
+ unresolved_origin=unresolved_origin,
+ # Read the lane the notice was actually routed through (failure_deliver on failure).
+ normalized_deliver=_normalize_deliver_value(_delivery_lane_value(job, for_failure=not success)),
+ incident_acked=incident_acked,
+ success=success,
)
- if delivery_error:
- delivery_outcome = "failed"
- elif should_deliver and unresolved_origin:
- delivery_outcome = "not_configured"
- elif should_deliver and normalized_deliver != "local":
- delivery_outcome = "delivered"
- elif incident_acked and not success:
- # Distinct from plain "suppressed" (silence marker / local jobs):
- # the failure ping was withheld because the operator acked this
- # exact signature via `hermes cron incidents ack`.
- delivery_outcome = "suppressed_acked"
- else:
- delivery_outcome = "suppressed"
if delivery_outcome in ("delivered", "not_configured") and not success:
- # The failure ping left the process (or was composed for a
- # configured target) — record it on the incident so the CLI
- # distinguishes "failure seen" from "operator was pinged".
+ # Failure ping left the process (or had a configured target): mark the incident alerted.
_mark_incident_alerted(failure_incident_id)
finish_execution(
execution_id,
@@ -7851,16 +5835,10 @@ def _run_one_job_body(
return True
except BaseException as e: # noqa: BLE001 — deliberate: see below
- # BaseException, not Exception (#73973): the inner run_job handler
- # re-raises CancelledError / KeyboardInterrupt / SystemExit after agent
- # teardown, and none of those are Exception subclasses. If they escape
- # without mark_job_run(False), a finite one-shot is left wedged —
- # claim_dispatch() already consumed repeat.completed, but last_run_at
- # is never written, so the job sits in state "scheduled" until the
- # run-claim TTL expires and the dispatch-limit guard removes it with
- # no output and no error. Record the failure first, then re-raise
- # anything that isn't a plain Exception. Owner fencing still applies:
- # a stale worker must not record over a replacement claim owner.
+ # BaseException, not Exception: CancelledError/KeyboardInterrupt/SystemExit propagate here.
+ # Without mark_job_run(False) a finite one-shot is wedged: claim_dispatch consumed
+ # repeat.completed but last_run_at is never written. Record first, then re-raise
+ # non-Exception. Owner fencing still applies.
_err_text = str(e) or type(e).__name__
logger.error(
"Error processing job %s: %s",
@@ -7869,26 +5847,17 @@ def _run_one_job_body(
exc_info=(type(e), e, e.__traceback__),
)
delivery_outcome = "suppressed"
- # Owner fencing: a stale worker whose fire claim was taken over (or a
- # transport-cancelled worker) must not send a failure alert on top of
- # the replacement run's own delivery — fall through silently and let
- # the fenced bookkeeping below decide what (if anything) to record.
+ # Owner fencing: a stale worker whose claim was taken over (or transport-cancelled) must not
+ # send a failure alert on top of the replacement run's; fall through to fenced bookkeeping.
if (
isinstance(e, Exception)
and not delivery_attempted
and not isinstance(e, _FireClaimLostDuringSideEffect)
and not _fire_claim_ownership_lost()
):
- normalized_deliver = _normalize_deliver_value(
- _delivery_lane_value(job, for_failure=True)
- )
- unresolved_origin = False
- # Durable failure incident: same ack gate as the normal failure
- # delivery above — an acked signature stays silent on this path
- # too, so the retry-path alert cannot re-ping after acknowledgment.
- incident_acked, failure_incident_id = _upsert_incident_for_failure(
- job, _err_text
- )
+ normalized_deliver = _normalize_deliver_value(_delivery_lane_value(job, for_failure=True))
+ # Same ack gate as the normal failure delivery: acked signatures stay silent here too.
+ incident_acked, failure_incident_id = _upsert_incident_for_failure(job, _err_text)
if incident_acked:
delivery_outcome = "suppressed_acked"
else:
@@ -7896,12 +5865,8 @@ def _run_one_job_body(
delivery_attempted = True
delivery_error = _deliver_result(
job,
- # Composed exactly like the normal failure delivery above.
- # mark_job_run below records THIS run in failure_streak
- # whichever layer failed, so a job that fails before the
- # run body every tick builds a streak nobody is ever told
- # about: its alerts only ever leave through here, and the
- # nudge only ever left through there (#88655).
+ # Same text as the normal failure delivery: this run also counts toward
+ # failure_streak, so the nudge must leave through here too.
_summarize_cron_failure_for_delivery(job, _err_text)
+ _failure_streak_nudge(job),
adapters=adapters,
@@ -7910,19 +5875,20 @@ def _run_one_job_body(
)
except Exception as delivery_exc:
delivery_error = str(delivery_exc)
- logger.error(
- "Delivery failed for job %s: %s", job["id"], delivery_exc
- )
- if not delivery_error and normalized_deliver == "origin":
- unresolved_origin = not _resolve_delivery_targets(
- job, for_failure=True
- )
- if delivery_error:
- delivery_outcome = "failed"
- elif unresolved_origin:
- delivery_outcome = "not_configured"
- elif normalized_deliver != "local":
- delivery_outcome = "delivered"
+ logger.error("Delivery failed for job %s: %s", job["id"], delivery_exc)
+ unresolved_origin = bool(
+ not delivery_error
+ and normalized_deliver == "origin"
+ and not _resolve_delivery_targets(job, for_failure=True)
+ )
+ delivery_outcome = _classify_delivery_outcome(
+ delivery_error=delivery_error,
+ should_deliver=True,
+ unresolved_origin=unresolved_origin,
+ normalized_deliver=normalized_deliver,
+ incident_acked=False,
+ success=False,
+ )
if delivery_outcome in ("delivered", "not_configured"):
_mark_incident_alerted(failure_incident_id)
try:
@@ -7935,10 +5901,7 @@ def _run_one_job_body(
mark_job_run(job["id"], False, _err_text, **mark_kwargs)
except Exception as record_err:
# Never let bookkeeping mask the original interruption.
- logger.error(
- "Failed to record interrupted run for job %s: %s",
- job["id"], record_err,
- )
+ logger.error("Failed to record interrupted run for job %s: %s", job["id"], record_err)
try:
finish_execution(
execution_id,
@@ -7947,18 +5910,13 @@ def _run_one_job_body(
delivery_outcome=delivery_outcome,
)
except Exception as record_err:
- logger.error(
- "Failed to finish execution record for job %s: %s",
- job["id"], record_err,
- )
+ logger.error("Failed to finish execution record for job %s: %s", job["id"], record_err)
if not isinstance(e, Exception):
raise
return False
finally:
- # Function-level on purpose: this must scope delivery, deferred-agent
- # teardown, claim-loss handling and bookkeeping, not just run_job.
- # An earlier revision reset inside the run block's finally, which left
- # _deliver_result unscoped — do not move it back in a tidy-up.
+ # Function-level on purpose: must scope delivery, deferred teardown, claim-loss handling and
+ # bookkeeping — not just run_job. Do not move into the run block's finally.
if _scope_token is not None:
reset_secret_scope(_scope_token)
if _terminal_scope_token is not None:
@@ -7968,16 +5926,9 @@ def _run_one_job_body(
def _notify_provider_jobs_changed() -> None:
- """Best-effort: tell the active scheduler provider the job set changed.
-
- Called by the consumer surfaces (model tool / CLI / REST) AFTER a
- successful store mutation (create/update/remove/pause/resume) so an external
- provider (Chronos) can re-provision/cancel the affected one-shot via NAS.
- No-op for the built-in (it re-reads jobs.json each tick), so the default
- path is unchanged. Lives here (not in cron/jobs.py) to keep the store free
- of provider imports — avoids an import cycle and keeps jobs.py low-coupling.
- Never raises into the caller.
- """
+ """Best-effort: tell the active scheduler provider the job set changed. Call AFTER a successful
+ store mutation so an external provider can re-provision/cancel the one-shot; no-op for the
+ built-in. Kept out of cron/jobs.py (import cycle). Never raises."""
try:
from cron.scheduler_provider import resolve_cron_scheduler
resolve_cron_scheduler().on_jobs_changed()
@@ -8030,45 +5981,29 @@ def create_job_with_scheduler_registration(**kwargs) -> dict:
return job
-# Dead-owner claim reclaim throttle (#86721): recover_interrupted_executions
-# opens the executions ledger, so the per-tick reap is rate-limited rather
-# than run on every idle 60s cycle. Tests may reset _last_dead_owner_reap_at
-# to None to force a reap on the next tick.
+# Dead-owner reap is throttled (opens the executions ledger). Tests may reset
+# _last_dead_owner_reap_at to None to force a reap next tick.
_DEAD_OWNER_REAP_INTERVAL_SECONDS = 300.0
_last_dead_owner_reap_at: Optional[float] = None
-# Worktree maintenance throttle: the startup pruner historically ran only on
-# `hermes -w` launches, so on gateway-driven boxes (where sessions arrive via
-# Telegram/Discord and nobody launches the CLI for days) merged scratch trees
-# accumulated into tens of GB. The cron tick is the one reliably periodic
-# process on every install, so it owns a low-frequency sweep too. Tests may
-# reset _last_worktree_maintenance_at to None to force a sweep next tick.
+# Worktree prune throttle: the cron tick is the only reliably periodic process on gateway boxes.
_WORKTREE_MAINTENANCE_INTERVAL_SECONDS = 6 * 3600.0
_last_worktree_maintenance_at: Optional[float] = None
_worktree_maintenance_lock = threading.Lock()
def _worktree_maintenance_repos() -> List[str]:
- """Repos whose ``.worktrees/`` this scheduler should keep pruned.
-
- Candidates: the hermes install checkout itself (where ``hermes -w``
- sessions on dev boxes create trees) and every configured job workdir's
- repo root. Only repos that actually have a ``.worktrees/`` dir survive —
- everything else costs nothing.
- """
+ """Repos whose ``.worktrees/`` to keep pruned: the hermes checkout plus job workdir repo roots,
+ filtered to those that actually have a ``.worktrees/`` dir."""
repos: set = set()
- # The hermes source checkout (editable/git installs). Wheel installs have
- # no .git here and are skipped.
- try:
+ # Hermes source checkout (git installs only; wheel installs have no .git).
+ with contextlib.suppress(Exception):
install_root = Path(__file__).resolve().parent.parent
if (install_root / ".git").exists():
repos.add(str(install_root))
- except Exception:
- pass
- # Job workdirs may point inside other repos the agent works on.
- try:
+ with contextlib.suppress(Exception):
from cron.jobs import load_jobs
for job in load_jobs():
@@ -8085,21 +6020,14 @@ def _worktree_maintenance_repos() -> List[str]:
repos.add(probe.stdout.strip())
except Exception:
continue
- except Exception:
- pass
return [r for r in sorted(repos) if (Path(r) / ".worktrees").is_dir()]
def _maybe_run_worktree_maintenance() -> None:
- """Throttled, threaded worktree prune from the cron tick.
-
- Runs ``cli._prune_stale_worktrees`` (the same conservative pruner the
- ``hermes -w`` startup path uses — dirty/unpushed/live-locked trees are
- never touched) against every candidate repo, on a daemon thread so the
- tick itself never waits on git. Errors never propagate: worktree GC is
- hygiene, not scheduling.
- """
+ """Throttled worktree prune from the cron tick, on a daemon thread so the tick never waits on
+ git. Same conservative pruner as ``hermes -w`` startup (dirty/unpushed/locked trees untouched).
+ Errors never propagate: GC is hygiene, not scheduling."""
global _last_worktree_maintenance_at
now = time.monotonic()
with _worktree_maintenance_lock:
@@ -8122,16 +6050,217 @@ def _maybe_run_worktree_maintenance() -> None:
try:
_prune_stale_worktrees(repo)
except Exception:
- logger.debug(
- "Cron worktree maintenance failed for %s", repo,
- exc_info=True,
- )
+ logger.debug("Cron worktree maintenance failed for %s", repo, exc_info=True)
except Exception:
logger.debug("Cron worktree maintenance skipped", exc_info=True)
- threading.Thread(
- target=_run, name="cron-worktree-prune", daemon=True
- ).start()
+ threading.Thread(target=_run, name="cron-worktree-prune", daemon=True).start()
+
+
+def _acquire_tick_lock(lock_file):
+ """Open + non-blocking lock the tick file. Returns the fd, or None on genuine contention.
+
+ fcntl on Unix, msvcrt on Windows. A real OSError (esp. EMFILE/ENFILE) must NOT pass as
+ contention — the scheduler would look healthy while no job runs — so it is re-raised for the
+ ticker loop to record a FAILED tick.
+ """
+ lock_fd = None
+ try:
+ lock_fd = open(lock_file, "w", encoding="utf-8")
+ if fcntl:
+ fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
+ elif msvcrt:
+ msvcrt.locking(lock_fd.fileno(), msvcrt.LK_NBLCK, 1)
+ return lock_fd
+ except OSError as exc:
+ if lock_fd is not None:
+ with contextlib.suppress(OSError):
+ lock_fd.close()
+ if _is_lock_contention_errno(exc):
+ logger.debug("Tick skipped — another instance holds the lock")
+ return None
+ if _is_fd_exhaustion(exc):
+ # fd reclamation is the ticker loop's job (scheduler_provider.py); here would double it.
+ logger.error(
+ "Cron tick could not acquire tick lock: %s — scheduler will "
+ "attempt fd reclamation and retry with backoff",
+ exc,
+ )
+ else:
+ logger.error("Cron tick could not acquire tick lock: %s", exc)
+ raise
+
+
+def _release_tick_lock(lock_fd) -> None:
+ if fcntl:
+ with contextlib.suppress((OSError, IOError)):
+ fcntl.flock(lock_fd, fcntl.LOCK_UN)
+ elif msvcrt:
+ with contextlib.suppress((OSError, IOError)):
+ msvcrt.locking(lock_fd.fileno(), msvcrt.LK_UNLCK, 1)
+ lock_fd.close()
+
+
+def _maybe_reap_dead_owners() -> None:
+ """Dead-owner reclaim: a run that died mid-flight would leave its row 'claimed' forever. Only
+ rows whose owner process is proved gone are touched (_owner_is_live). Throttled."""
+ global _last_dead_owner_reap_at
+ _reap_now = time.monotonic()
+ if (
+ _last_dead_owner_reap_at is not None
+ and _reap_now - _last_dead_owner_reap_at < _DEAD_OWNER_REAP_INTERVAL_SECONDS
+ ):
+ return
+ _last_dead_owner_reap_at = _reap_now
+ try:
+ from cron.executions import recover_interrupted_executions
+
+ _reclaimed = recover_interrupted_executions()
+ if _reclaimed:
+ logger.warning(
+ "Reclaimed %d cron execution(s) whose owner process died "
+ "before reaching a terminal state (marked unknown)",
+ _reclaimed,
+ )
+ except Exception as _reap_exc:
+ logger.debug("Dead-owner execution reclaim failed: %s", _reap_exc)
+
+
+def _sweep_stale_inflight_for_tick(due_jobs: list) -> None:
+ """Bound the in-flight set BEFORE the dedup guard so a leaked claim is force-released now
+ rather than eating every later fire until restart. Skipped when nothing is in flight."""
+ if not _running_job_ids:
+ return
+ _sweep_jobs = due_jobs
+ with contextlib.suppress(Exception):
+ _inflight_ids = set(_running_job_ids)
+ _due_ids = {j.get("id") for j in due_jobs if isinstance(j, dict)}
+ if not _inflight_ids <= _due_ids:
+ from cron.jobs import load_jobs as _load_all_jobs
+
+ _sweep_jobs = _load_all_jobs()
+ try:
+ sweep_stale_inflight(_sweep_jobs)
+ except Exception as e:
+ logger.warning("Stale in-flight sweep failed: %s", e)
+
+
+def _resolve_max_parallel_workers() -> Optional[int]:
+ """Max workers: env > config.yaml > unbounded (HERMES_CRON_MAX_PARALLEL=1 restores serial)."""
+ try:
+ _env_par = os.getenv("HERMES_CRON_MAX_PARALLEL", "").strip()
+ if _env_par:
+ return int(_env_par) or None
+ except (ValueError, TypeError):
+ logger.warning("Invalid HERMES_CRON_MAX_PARALLEL value; defaulting to unbounded")
+ with contextlib.suppress(Exception):
+ _ucfg = load_config() or {}
+ _cfg_par = (_ucfg.get("cron", {}) if isinstance(_ucfg, dict) else {}).get("max_parallel_jobs")
+ if _cfg_par is not None:
+ return int(_cfg_par) or None
+ return None
+
+
+def _sweep_mcp_orphans() -> None:
+ """Reap MCP stdio orphans (only PIDs flagged by tools.mcp_tool._run_stdio's finally block);
+ run AFTER jobs finish so live sessions are never touched."""
+ try:
+ from tools.mcp_tool import _kill_orphaned_mcp_children
+ _kill_orphaned_mcp_children()
+ except Exception as _e:
+ logger.debug("Post-tick MCP orphan cleanup failed: %s", _e)
+
+
+def _process_due_job(job: dict, adapters, loop, verbose: bool) -> bool:
+ """Run one due job via the shared ``run_one_job`` body."""
+ # Claim only when the worker actually starts, so a queued lease can't expire first.
+ claimed = claim_job_for_fire(job["id"], return_job=True)
+ if not claimed:
+ finish_execution(
+ job["execution_id"],
+ success=False,
+ error="Fire claim lost; execution was not started.",
+ )
+ return True
+ # CAS returns the persisted record; bool fallback only for older test doubles.
+ claimed_job = dict(claimed) if isinstance(claimed, dict) else dict(job)
+ claimed_job["execution_id"] = job["execution_id"]
+ return run_one_job(claimed_job, adapters=adapters, loop=loop, verbose=verbose)
+
+
+def _submit_with_guard(job: dict, pool: concurrent.futures.ThreadPoolExecutor, process_job):
+ """Submit with the in-flight dedup guard; None if a prior tick's run is still in flight.
+ Running-set membership is released in the worker's finally."""
+ job_id = job["id"]
+ job_label = job.get("name", job_id)
+
+ def _clear_run_claim_best_effort() -> None:
+ """Best-effort claim cleanup on dispatch-failure paths. Only one-shots carry a run_claim;
+ clear_run_claim takes _jobs_lock + full load/save and can raise on degraded paths
+ (shutdown, EMFILE) — a claim expiring at TTL beats crashing the tick."""
+ _schedule = job.get("schedule")
+ if not (isinstance(_schedule, dict) and _schedule.get("kind") == "once"):
+ return
+ try:
+ clear_run_claim(job_id)
+ except Exception as claim_err:
+ logger.warning(
+ "Could not clear run_claim for job '%s' after dispatch "
+ "failure: %s (claim will expire at TTL)",
+ job_label, claim_err,
+ )
+
+ def _not_dispatched_shutdown() -> None:
+ logger.warning("Job '%s' not dispatched — interpreter is shutting down", job_label)
+
+ # During interpreter shutdown pool.submit raises; skip — the job fires on the next tick.
+ if _interpreter_shutting_down():
+ _not_dispatched_shutdown()
+ _clear_run_claim_best_effort()
+ return None
+ if not try_register_running_job(job_id):
+ logger.info("Job '%s' already running — skipping", job_label)
+ return None
+ # Record the attempt before dispatch; recovery marks abandoned rows unknown (no retry).
+ try:
+ execution = create_execution(job_id, source="builtin")
+ dispatched_job = dict(job, execution_id=execution["id"])
+ _ctx = contextvars.copy_context()
+ except Exception as execution_err:
+ # Release the claim so the next tick retries instead of wedging "already running".
+ release_running_job(job_id)
+ _clear_run_claim_best_effort()
+ logger.exception(
+ "Job '%s' not dispatched: execution creation failed: %s", job_label, execution_err,
+ )
+ return None
+
+ def _run_and_release(j=dispatched_job, ctx=_ctx):
+ try:
+ return ctx.run(process_job, j)
+ finally:
+ release_running_job(j["id"])
+
+ try:
+ fut = pool.submit(_run_and_release)
+ except Exception as submit_err:
+ release_running_job(job_id)
+ _clear_run_claim_best_effort()
+ finish_execution(
+ execution["id"],
+ success=False,
+ error=f"Executor dispatch failed: {submit_err}",
+ )
+ if isinstance(submit_err, RuntimeError) and _interpreter_shutting_down(submit_err):
+ _not_dispatched_shutdown()
+ else:
+ logger.error("Job '%s' not dispatched: %s", job_label, submit_err)
+ return None
+
+ with _running_lock:
+ if job_id in _running_job_ids:
+ _running_futures[job_id] = fut
+ return fut
def tick(
@@ -8142,33 +6271,12 @@ def tick(
*,
can_dispatch=None,
):
- """
- Check and run all due jobs.
-
- Uses a file lock so only one tick runs at a time, even if the gateway's
- in-process ticker and a standalone daemon or manual tick overlap.
-
- Args:
- verbose: Whether to print status messages
- adapters: Optional dict mapping Platform → live adapter (from gateway)
- loop: Optional asyncio event loop (from gateway) for live adapter sends
- can_dispatch: Optional synchronous gate; false leaves due jobs untouched
- for the next allowed tick
-
- Returns:
- Number of jobs executed (0 if another tick is already running)
- """
- # Stale-code yield gate — BEFORE the lock race (#stale-tick-preemption).
- # A long-lived process whose checkout was updated underneath it (hot
- # ``git pull``, interrupted ``hermes update``) serves MIXED sys.modules:
- # every agent job it dispatches can die on ImportErrors whose real cause
- # is staleness. When this process is provably stale AND a fresher
- # process holds the gateway runtime lock, that process's own ticker
- # dispatches due jobs — this one must not even enter the lock race and
- # preempt dispatch on a busy minute. When no fresh holder exists
- # (desktop-standalone users), yielding would silently kill the user's
- # only ticker, so the tick proceeds and job failures surface through the
- # delivery path's stale-code hint instead.
+ """Check and run all due jobs. File-locked so only one tick runs at a time (gateway ticker vs
+ standalone daemon / manual tick). ``can_dispatch``: optional gate; false leaves due jobs for the
+ next allowed tick. Returns the number of jobs executed (0 if another tick holds the lock)."""
+ # Stale-code yield gate — BEFORE the lock race. A process whose checkout was updated under it
+ # serves mixed sys.modules (jobs die on ImportErrors); if a fresher gateway holds the runtime
+ # lock, ITS ticker dispatches. With no fresh holder (desktop-standalone) the tick proceeds.
_skew = _should_yield_tick_to_fresh_gateway()
if _skew is not None:
_log_tick_yield_once(f"boot={_skew[0]} disk={_skew[1]}")
@@ -8176,181 +6284,47 @@ def tick(
lock_dir, lock_file = _get_lock_paths()
_ensure_cron_dir(lock_dir)
-
- # Cross-platform file locking: fcntl on Unix, msvcrt on Windows.
- # Only genuine lock contention (another ticker holds the lock) skips the
- # tick silently. A real OSError — most importantly EMFILE/ENFILE from fd
- # exhaustion — must NOT be swallowed as "another instance holds the
- # lock": that previously made the scheduler appear healthy (tick returned
- # 0, heartbeat recorded success) while no job ever ran again (#87644).
- lock_fd = None
- try:
- lock_fd = open(lock_file, "w", encoding="utf-8")
- if fcntl:
- fcntl.flock(lock_fd, fcntl.LOCK_EX | fcntl.LOCK_NB)
- elif msvcrt:
- msvcrt.locking(lock_fd.fileno(), msvcrt.LK_NBLCK, 1)
- except OSError as exc:
- if lock_fd is not None and _is_lock_contention_errno(exc):
- logger.debug("Tick skipped — another instance holds the lock")
- try:
- lock_fd.close()
- except OSError:
- pass
- return 0
- # Real failure: log loudly, attempt fd reclamation, and let the
- # caller (ticker loop) see a FAILED tick so liveness degrades
- # instead of reporting healthy-while-stalled.
- if lock_fd is not None:
- try:
- lock_fd.close()
- except OSError:
- pass
- if _is_fd_exhaustion(exc):
- # Reclamation is owned by the ticker loop's except handler
- # (scheduler_provider.py) — it classifies the raised error and
- # runs _reclaim_fds_best_effort exactly once per failed tick.
- # Calling it here too would double the gc.collect() pause.
- logger.error(
- "Cron tick could not acquire tick lock: %s — scheduler will "
- "attempt fd reclamation and retry with backoff",
- exc,
- )
- else:
- logger.error("Cron tick could not acquire tick lock: %s", exc)
- raise
+ lock_fd = _acquire_tick_lock(lock_file)
+ if lock_fd is None:
+ return 0
try:
- # Global emergency stop (`hermes pause`): skip dispatch entirely while
- # the ESTOP sentinel exists. Never touches in-flight runs — due jobs
- # simply wait for the next tick after `hermes resume`. Logged once per
- # engagement (not every tick) by check_paused.
- try:
+ # `hermes pause` ESTOP: skip dispatch, never touch in-flight runs; check_paused logs once.
+ with contextlib.suppress(ImportError):
from agent.estop import check_paused as _estop_check_paused
if _estop_check_paused("cron", logger):
return 0
- except ImportError:
- pass
if can_dispatch is not None and not can_dispatch():
logger.debug("Cron dispatch paused while gateway drains existing work")
return 0
- # Dead-owner claim reclaim (#86721): execution rows carry their owner
- # pid + process start time, but recovery previously ran only at
- # scheduler STARTUP. A one-shot `hermes cron run` that claimed a job
- # and died mid-run (its runner thread lived in the exiting CLI
- # process) left the row 'claimed' forever while the long-lived
- # gateway ticker kept running — blocking every future run of that
- # job. Reap provably-dead owners periodically so stale claims
- # auto-clear without a gateway restart. Only rows whose exact owner
- # process is proved gone are touched (see _owner_is_live), so live
- # runs in other processes are never rewritten. Throttled so idle
- # 60s ticks don't pay a ledger connection every cycle (#33612).
- global _last_dead_owner_reap_at
- _reap_now = time.monotonic()
- if (
- _last_dead_owner_reap_at is None
- or _reap_now - _last_dead_owner_reap_at >= _DEAD_OWNER_REAP_INTERVAL_SECONDS
- ):
- _last_dead_owner_reap_at = _reap_now
- try:
- from cron.executions import recover_interrupted_executions
-
- _reclaimed = recover_interrupted_executions()
- if _reclaimed:
- logger.warning(
- "Reclaimed %d cron execution(s) whose owner process died "
- "before reaching a terminal state (marked unknown)",
- _reclaimed,
- )
- except Exception as _reap_exc:
- logger.debug("Dead-owner execution reclaim failed: %s", _reap_exc)
-
- # Periodic worktree GC (throttled to every 6h, threaded): gateway-only
- # boxes never hit the `hermes -w` startup pruner, so this is the only
- # sweep they get. Same conservative pruner, same guards.
+ _maybe_reap_dead_owners()
+ # Periodic worktree GC (6h, threaded) — the only sweep gateway-only boxes get.
try:
_maybe_run_worktree_maintenance()
except Exception as _wt_exc:
logger.debug("Worktree maintenance dispatch failed: %s", _wt_exc)
due_jobs = get_due_jobs()
-
- # Bound the in-flight set BEFORE the dedup guard is consulted, so a
- # leaked claim is force-released in-cycle rather than silently eating
- # every subsequent fire until the gateway process restarts. Skips the
- # extra load_jobs when there are no in-flight claims (the common idle
- # tick) and reuses due_jobs when they already cover the in-flight set
- # (get_due_jobs calls load_jobs internally, so this avoids a redundant
- # second file read on every active tick).
- if _running_job_ids:
- _sweep_jobs = due_jobs
- try:
- _inflight_ids = set(_running_job_ids)
- _due_ids = {j.get("id") for j in due_jobs if isinstance(j, dict)}
- if not _inflight_ids <= _due_ids:
- from cron.jobs import load_jobs as _load_all_jobs
-
- _sweep_jobs = _load_all_jobs()
- except Exception:
- pass
- try:
- sweep_stale_inflight(_sweep_jobs)
- except Exception as e:
- logger.warning("Stale in-flight sweep failed: %s", e)
+ _sweep_stale_inflight_for_tick(due_jobs)
if not due_jobs:
- # Idle tick: skip config load + pool partitioning entirely
- # (#33612 — the gateway ticker calls tick(verbose=False) every
- # 60s, so idle ticks previously fell through to load_config()).
- # Still run the post-tick MCP orphan sweep: main intentionally
- # sweeps on idle ticks so orphaned stdio children from crashed
- # jobs are reaped even when nothing is due.
+ # Idle tick: skip config load + pool setup, but still reap crashed jobs' MCP orphans.
if verbose:
logger.info("%s - No jobs due", _hermes_now().strftime('%H:%M:%S'))
- try:
- from tools.mcp_tool import _kill_orphaned_mcp_children
- _kill_orphaned_mcp_children()
- except Exception as _e:
- logger.debug("Post-tick MCP orphan cleanup failed: %s", _e)
+ _sweep_mcp_orphans()
return 0
if verbose:
logger.info("%s - %s job(s) due", _hermes_now().strftime('%H:%M:%S'), len(due_jobs))
- # Advance next_run_at for all recurring jobs FIRST, under the file lock,
- # before any execution begins. This preserves at-most-once semantics.
- # For parallel jobs that are already running, the advance keeps
- # bumping next_run_at forward so the grace window never expires.
- # mark_job_run() overwrites next_run_at on completion.
- # Batched: one load + one save for the whole due set, not one per job.
- # Composes with the claim-time advance in claim_job_for_fire: for
- # cron-kind jobs both compute the same next occurrence; interval jobs
- # re-anchor from their own "now" at claim time (harmless for
- # at-most-once — mark_job_run re-anchors at completion regardless).
+ # Advance next_run_at for recurring jobs FIRST, under the lock, before any execution
+ # (at-most-once). Re-advancing running jobs keeps the grace window alive; mark_job_run
+ # overwrites it on completion. Composes with the claim-time advance in claim_job_for_fire.
advance_next_runs([job["id"] for job in due_jobs])
- # Resolve max parallel workers: env var > config.yaml > unbounded.
- # Set HERMES_CRON_MAX_PARALLEL=1 to restore old serial behaviour.
- _max_workers: Optional[int] = None
- try:
- _env_par = os.getenv("HERMES_CRON_MAX_PARALLEL", "").strip()
- if _env_par:
- _max_workers = int(_env_par) or None
- except (ValueError, TypeError):
- logger.warning("Invalid HERMES_CRON_MAX_PARALLEL value; defaulting to unbounded")
- if _max_workers is None:
- try:
- _ucfg = load_config() or {}
- _cfg_par = (
- _ucfg.get("cron", {}) if isinstance(_ucfg, dict) else {}
- ).get("max_parallel_jobs")
- if _cfg_par is not None:
- _max_workers = int(_cfg_par) or None
- except Exception:
- pass
-
+ _max_workers = _resolve_max_parallel_workers()
if verbose:
logger.info(
"Running %d job(s) in parallel (max_workers=%s)",
@@ -8359,181 +6333,22 @@ def tick(
)
def _process_job(job: dict) -> bool:
- """Run one due job end-to-end. Thin wrapper around the shared
- module-level ``run_one_job`` so ``tick`` and external providers
- (Chronos ``fire_due``) use the identical execute→save→deliver→mark
- body."""
- # Acquire the durable claim only when this worker actually starts,
- # not while it may wait behind other work in an executor queue.
- # This prevents a queued lease from expiring before execution.
- claimed = claim_job_for_fire(job["id"], return_job=True)
- if not claimed:
- finish_execution(
- job["execution_id"],
- success=False,
- error="Fire claim lost; execution was not started.",
- )
- return True
- # Production CAS returns the exact persisted record with its unique
- # owner. Bool fallback keeps older test doubles/API overrides
- # compatible; real callers using return_job=True never take it.
- claimed_job = dict(claimed) if isinstance(claimed, dict) else dict(job)
- claimed_job["execution_id"] = job["execution_id"]
- return run_one_job(
- claimed_job,
- adapters=adapters,
- loop=loop,
- verbose=verbose,
- )
-
- # Workdir is task-scoped, so every job uses the normal parallel lane.
- parallel_jobs = due_jobs
+ return _process_due_job(job, adapters, loop, verbose)
+ # Persistent pool, non-blocking dispatch. Already-running jobs are skipped; mark_job_run
+ # re-arms next_run_at on completion, so no catch-up queue is needed.
_results: list = []
_all_futures: list = []
-
- def _submit_with_guard(job: dict, pool: concurrent.futures.ThreadPoolExecutor):
- """Submit a job fire-and-forget with the in-flight dedup guard.
-
- Returns the future, or None if the job was skipped because a prior
- tick's run of the same job is still in flight. The running-set
- membership is released in the worker's finally block.
- """
- job_id = job["id"]
-
- def _clear_run_claim_best_effort() -> None:
- """Best-effort claim cleanup on the dispatch-failure paths.
-
- Only one-shot jobs carry a ``run_claim`` (stamped by
- get_due_jobs, #59229), so recurring jobs skip the call
- entirely — clear_run_claim acquires _jobs_lock (blocking
- cross-process flock) and does a full load_jobs read, and the
- dispatch-failure paths fire exactly when the process can
- least afford N pointless lock/read round-trips (interpreter
- shutdown, EMFILE). clear_run_claim itself does
- load_jobs/save_jobs file I/O; on those degraded paths it can
- raise, and these early-exits exist precisely to skip cleanly
- — a stale claim expiring at the TTL is a better outcome than
- crashing the tick (#86522).
- """
- _schedule = job.get("schedule")
- if not (isinstance(_schedule, dict) and _schedule.get("kind") == "once"):
- return
- try:
- clear_run_claim(job_id)
- except Exception as claim_err:
- logger.warning(
- "Could not clear run_claim for job '%s' after dispatch "
- "failure: %s (claim will expire at TTL)",
- job.get("name", job_id),
- claim_err,
- )
-
- # A tick can race gateway teardown: once the interpreter is
- # finalizing, ``pool.submit`` raises "cannot schedule new futures
- # after interpreter shutdown" and crashes the tick. Skip cleanly —
- # the job stays due and will fire on the next healthy tick
- # (#58720, #55924).
- if _interpreter_shutting_down():
- logger.warning(
- "Job '%s' not dispatched — interpreter is shutting down",
- job.get("name", job_id),
- )
- _clear_run_claim_best_effort()
- return None
- if not try_register_running_job(job_id):
- logger.info("Job '%s' already running — skipping", job.get("name", job_id))
- return None
- # Record the attempt before executor dispatch. Recovery classifies
- # abandoned records as unknown; it never automatically retries them.
- try:
- execution = create_execution(job_id, source="builtin")
- dispatched_job = dict(job, execution_id=execution["id"])
- _ctx = contextvars.copy_context()
- except Exception as execution_err:
- # Init/creation failure between the claim and the submit —
- # release the in-flight claim immediately so the next tick can
- # retry instead of wedging on 'already running' forever (the
- # audit requirement: every add is paired with guaranteed
- # cleanup).
- release_running_job(job_id)
- _clear_run_claim_best_effort()
- logger.exception(
- "Job '%s' not dispatched: execution creation failed: %s",
- job.get("name", job_id),
- execution_err,
- )
- return None
-
- def _run_and_release(j=dispatched_job, ctx=_ctx):
- try:
- return ctx.run(_process_job, j)
- finally:
- release_running_job(j["id"])
-
- try:
- fut = pool.submit(_run_and_release)
- except Exception as submit_err:
- release_running_job(job_id)
- _clear_run_claim_best_effort()
- finish_execution(
- execution["id"],
- success=False,
- error=f"Executor dispatch failed: {submit_err}",
- )
- # Interpreter began finalizing between the guard above and the
- # submit — release the in-flight claim we just took and skip.
- if isinstance(submit_err, RuntimeError) and _interpreter_shutting_down(submit_err):
- logger.warning(
- "Job '%s' not dispatched — interpreter is shutting down",
- job.get("name", job_id),
- )
- return None
- logger.error(
- "Job '%s' not dispatched: %s",
- job.get("name", job_id),
- submit_err,
- )
- return None
-
- # Record the owning future so the stale sweep can distinguish
- # "still executing" from "claim leaked before/after the future".
- with _running_lock:
- if job_id in _running_job_ids:
- _running_futures[job_id] = fut
- return fut
-
-
- # Parallel pass — persistent pool, non-blocking dispatch.
- # Jobs that are already running (from a previous tick) are skipped.
- # mark_job_run() updates next_run_at on completion, so the next tick
- # after completion finds the job due again naturally. No catch-up
- # queue needed.
- if parallel_jobs:
- pool = _get_parallel_pool(_max_workers)
- for job in parallel_jobs:
- fut = _submit_with_guard(job, pool)
- if fut is None:
- continue
- _all_futures.append(fut)
- if not sync:
- _results.append(True) # optimistically counted
-
- # Best-effort sweep of MCP stdio subprocesses that survived their
- # session teardown. Must run AFTER jobs finish so active sessions
- # (including live user chats) are never touched — only PIDs explicitly
- # detected as orphans in tools.mcp_tool._run_stdio's finally block are
- # reaped.
- def _sweep_mcp_orphans() -> None:
- try:
- from tools.mcp_tool import _kill_orphaned_mcp_children
- _kill_orphaned_mcp_children()
- except Exception as _e:
- logger.debug("Post-tick MCP orphan cleanup failed: %s", _e)
+ pool = _get_parallel_pool(_max_workers)
+ for job in due_jobs:
+ fut = _submit_with_guard(job, pool, _process_job)
+ if fut is None:
+ continue
+ _all_futures.append(fut)
+ if not sync:
+ _results.append(True) # optimistically counted
if sync:
- # Sync mode (tests / manual ticks): wait for all dispatched jobs,
- # collect results, then sweep once.
for f in concurrent.futures.as_completed(_all_futures):
try:
_results.append(f.result())
@@ -8543,20 +6358,16 @@ def tick(
_sweep_mcp_orphans()
return sum(_results)
- # Async (gateway ticker) mode: don't block. Sweep orphans via a
- # done-callback fired after the LAST dispatched job completes, so the
- # sweep still happens after jobs finish without stalling the tick.
+ # Async (gateway ticker): sweep via a done-callback after the LAST job completes.
if _all_futures:
_remaining = [len(_all_futures)]
def _on_done(_f: concurrent.futures.Future) -> None:
_remaining[0] -= 1
- try:
+ with contextlib.suppress(Exception):
_exc = _f.exception()
if _exc is not None:
logger.error("Cron job future failed in async mode: %s", _exc, exc_info=(type(_exc), _exc, _exc.__traceback__))
- except Exception:
- pass
if _remaining[0] <= 0:
_sweep_mcp_orphans()
@@ -8568,17 +6379,7 @@ def tick(
return sum(_results)
finally:
- if fcntl:
- try:
- fcntl.flock(lock_fd, fcntl.LOCK_UN)
- except (OSError, IOError):
- pass
- elif msvcrt:
- try:
- msvcrt.locking(lock_fd.fileno(), msvcrt.LK_UNLCK, 1)
- except (OSError, IOError):
- pass
- lock_fd.close()
+ _release_tick_lock(lock_fd)
if __name__ == "__main__":
diff --git a/tests/cron/test_scheduler_shutdown_guard.py b/tests/cron/test_scheduler_shutdown_guard.py
index d6cc7acefb..117b480ace 100644
--- a/tests/cron/test_scheduler_shutdown_guard.py
+++ b/tests/cron/test_scheduler_shutdown_guard.py
@@ -117,6 +117,5 @@ class TestSourceGuardrail:
def test_helper_defined(self, source):
assert "def _interpreter_shutting_down(" in source
- assert "#58720" in source