[verified] fix(cron): preserve active runs across gateway restart

This commit is contained in:
Brooklyn Nicholson
2026-09-03 07:59:37 +07:00
committed by kshitij
parent 68cbe484a4
commit 3373e97693
14 changed files with 1260 additions and 11 deletions
+276 -1
View File
@@ -3203,6 +3203,16 @@ def _deliver_result(
logger.warning("Job '%s': %s", job["id"], msg)
return msg
# Restart-safe workers intentionally have no live gateway adapter objects.
# Hand the send back through a durable queue so the current or replacement
# gateway performs it with relay/E2EE parity. The execution id is the
# idempotency key; the queue never retries an uncertain claimed send.
external_execution = os.environ.get("_HERMES_CRON_EXTERNAL_WORKER", "")
if external_execution and adapters is None:
from cron.delivery_queue import enqueue_and_wait
return enqueue_and_wait(external_execution, job, content)
from tools.send_message_tool import _send_to_platform
from gateway.config import load_gateway_config, Platform
@@ -4118,6 +4128,20 @@ def _deliver_result(
return None
def drain_delivery_queue(adapters, loop) -> int:
"""Send queued worker results through this gateway's live adapters."""
from cron.delivery_queue import drain
return drain(
lambda queued_job, queued_content: _deliver_result(
queued_job,
queued_content,
adapters=adapters,
loop=loop,
)
)
_DEFAULT_SCRIPT_TIMEOUT = 3600 # seconds (1 hour)
# Backward-compatible module override used by tests and emergency monkeypatches.
_SCRIPT_TIMEOUT = _DEFAULT_SCRIPT_TIMEOUT
@@ -7368,6 +7392,34 @@ def run_one_job(
run cooperatively — agent interruption AND script process-tree kill —
through the single fenced completion path.
"""
# Every gateway path (built-in scheduler, external providers, and direct
# API fires) crosses this seam. Ensure the detached worker has a durable
# attempt to adopt before any launch can occur.
if adapters is not None and not job.get("execution_id"):
execution = create_execution(job["id"], source="direct")
job["execution_id"] = execution["id"]
if adapters is not None:
try:
if _launch_external_cron_worker(job):
return True
except Exception as handoff_error:
error = f"Restart-safe cron worker dispatch failed: {handoff_error}"
logger.error("Job '%s': %s", job["id"], error)
claim = job.get("fire_claim")
owner = str(claim.get("by") or "") if isinstance(claim, dict) else ""
try:
mark_job_run(
job["id"],
False,
error,
**({"expected_fire_owner": owner} if owner else {}),
)
finally:
execution_id = job.get("execution_id")
if execution_id:
finish_execution(execution_id, success=False, error=error)
return True
if extra_prompt is None:
# A gateway-forwarded manual run (`hermes cron run --prompt` /
# cronjob(action='run', prompt=...) on a relay-fronted target) stamps
@@ -7491,7 +7543,16 @@ def _run_one_job_body(
# The attempt is claimed durably before executor/provider dispatch and
# becomes running only immediately before the actual run.
mark_execution_running(execution_id)
# Detached workers atomically transition the attempt to running while
# adopting it. In-process paths must win the claimed->running CAS
# here before any user script or agent side effect may begin.
external_owner = os.environ.get("_HERMES_CRON_EXTERNAL_WORKER") == execution_id
if not external_owner and mark_execution_running(execution_id) is None:
logger.warning(
"Cron job %s lost execution ownership before start; skipping",
job["id"],
)
return True
# Run and deliver under the profile's secret scope. get_secret() fails
# closed outside a scope once profile isolation is active, and cron
@@ -7967,6 +8028,210 @@ def _run_one_job_body(
reset_terminal_scope(_terminal_scope_token)
def _launch_external_cron_worker(job: dict) -> bool:
"""Launch *job* outside a managed gateway cgroup when required.
Returns ``False`` when the caller is not a managed systemd gateway and the
existing in-process path should be used. In managed topology, failure to
establish the transient scope raises: falling back would recreate the
restart interruption this handoff exists to prevent.
"""
execution_id = str(job["execution_id"])
job_id = str(job["id"])
handoff_dir = _get_hermes_home() / "cron" / "external-workers"
payload_path = handoff_dir / f"{execution_id}.json"
ack_path = handoff_dir / f"{execution_id}.ready"
command = [
sys.executable,
"-m",
"cron.scheduler",
"--external-worker-file",
str(payload_path),
"--ack-file",
str(ack_path),
]
from agent.secret_scope import is_multiplex_active
from tools.environments.local import build_subprocess_env
from tools.process_registry import restart_safe_gateway_child_argv
multiplex_active = is_multiplex_active()
scoped_command = restart_safe_gateway_child_argv(
command,
unit_suffix=f"cron-{job_id}-exec-{execution_id}",
)
if scoped_command == command:
return False
_ensure_cron_dir(handoff_dir)
try:
handoff_dir.chmod(0o700)
except OSError:
pass
fd = os.open(payload_path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
try:
with os.fdopen(fd, "w", encoding="utf-8") as payload_file:
json.dump(
{
"job": job,
"profile_home": str(_get_hermes_home().resolve()),
"multiplex_active": multiplex_active,
},
payload_file,
)
payload_file.flush()
os.fsync(payload_file.fileno())
except BaseException:
payload_path.unlink(missing_ok=True)
raise
worker_env = build_subprocess_env(
scrub_secrets=multiplex_active,
inherit_profile_home=True,
extra={"HERMES_HOME": str(_get_hermes_home().resolve())},
)
try:
process = subprocess.Popen(
scoped_command,
cwd=str(Path(__file__).resolve().parent.parent),
env=worker_env,
stdin=subprocess.DEVNULL,
stdout=subprocess.DEVNULL,
stderr=subprocess.DEVNULL,
start_new_session=True,
creationflags=windows_hide_flags(),
)
except BaseException:
payload_path.unlink(missing_ok=True)
raise
deadline = time.monotonic() + 5.0
while time.monotonic() < deadline:
if ack_path.exists():
try:
acknowledgement = json.loads(ack_path.read_text(encoding="utf-8"))
except Exception:
logger.exception(
"Cron external worker %s published an unreadable acknowledgement; "
"treating handoff as ownership-uncertain",
execution_id,
)
return True
finally:
ack_path.unlink(missing_ok=True)
if acknowledgement.get("execution_id") != execution_id:
logger.error(
"Cron external worker acknowledgement mismatch for %s; "
"treating handoff as ownership-uncertain",
execution_id,
)
return True
logger.info(
"Cron job '%s' handed to restart-safe worker pid=%s execution=%s",
job_id,
acknowledgement.get("pid"),
execution_id,
)
return True
returncode = process.poll()
if returncode is not None:
payload_path.unlink(missing_ok=True)
raise RuntimeError(
f"cron external worker exited before ownership acknowledgement "
f"(exit {returncode})"
)
time.sleep(0.05)
# The child may have adopted the durable row just before publishing its
# acknowledgement. Never fall back to in-process execution on an uncertain
# handoff: that could duplicate side effects. The execution owner/dead-owner
# recovery ledger remains the authority.
logger.warning(
"Cron external worker for job '%s' did not acknowledge within 5s; "
"leaving the durable execution claim untouched",
job_id,
)
return True
def _run_external_worker_payload(payload_path: Path, ack_path: Path) -> bool:
"""Adopt and execute one gateway-dispatched cron payload.
The execution row is created by the gateway before spawn, then transferred
here before the ready acknowledgement is published. No side effect runs
unless that durable ownership transfer succeeds.
"""
try:
payload = json.loads(payload_path.read_text(encoding="utf-8"))
job = payload["job"]
profile_home = Path(payload["profile_home"]).resolve()
execution_id = str(job["execution_id"])
except Exception:
logger.exception("Cron external worker could not load payload %s", payload_path)
return False
finally:
try:
payload_path.unlink(missing_ok=True)
except OSError:
pass
from agent.secret_scope import (
build_profile_secret_scope,
is_multiplex_active,
reset_secret_scope,
set_multiplex_active,
set_secret_scope,
)
from cron.executions import adopt_claimed_execution
from hermes_cli.env_loader import hydrate_profile_secret_sources
from hermes_constants import (
reset_hermes_home_override,
set_hermes_home_override,
)
home_token = set_hermes_home_override(profile_home)
previous_multiplex = is_multiplex_active()
multiplex_active = bool(payload.get("multiplex_active", False))
set_multiplex_active(multiplex_active)
hydrate_profile_secret_sources(profile_home)
secret_token = set_secret_scope(build_profile_secret_scope(profile_home))
try:
with use_cron_store(profile_home):
if adopt_claimed_execution(execution_id) is None:
logger.error(
"Cron external worker refused execution %s: durable ownership "
"could not be established",
execution_id,
)
return False
try:
ack_path.parent.mkdir(parents=True, exist_ok=True)
fd = os.open(ack_path, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600)
with os.fdopen(fd, "w", encoding="utf-8") as ack_file:
json.dump({"pid": os.getpid(), "execution_id": execution_id}, ack_file)
ack_file.flush()
os.fsync(ack_file.fileno())
except Exception:
logger.exception(
"Cron external worker could not publish ready acknowledgement for %s",
execution_id,
)
return False
old_external_execution = os.environ.get("_HERMES_CRON_EXTERNAL_WORKER")
os.environ["_HERMES_CRON_EXTERNAL_WORKER"] = execution_id
try:
return run_one_job(job, adapters=None, loop=None, verbose=False)
finally:
if old_external_execution is None:
os.environ.pop("_HERMES_CRON_EXTERNAL_WORKER", None)
else:
os.environ["_HERMES_CRON_EXTERNAL_WORKER"] = old_external_execution
finally:
reset_secret_scope(secret_token)
set_multiplex_active(previous_multiplex)
reset_hermes_home_override(home_token)
def _notify_provider_jobs_changed() -> None:
"""Best-effort: tell the active scheduler provider the job set changed.
@@ -8582,4 +8847,14 @@ def tick(
if __name__ == "__main__":
if "--external-worker-file" in sys.argv:
import argparse
parser = argparse.ArgumentParser(add_help=False)
parser.add_argument("--external-worker-file", type=Path, required=True)
parser.add_argument("--ack-file", type=Path, required=True)
args = parser.parse_args()
raise SystemExit(
0 if _run_external_worker_payload(args.external_worker_file, args.ack_file) else 1
)
tick(verbose=True)