feat(cron): let planned downtime skip missed recurring runs

This commit is contained in:
Teknium
2026-09-07 02:41:16 -07:00
parent be82d52cb7
commit df51797e2e
5 changed files with 77 additions and 7 deletions
+3
View File
@@ -359,6 +359,9 @@ kanban:
# Working directory behavior:
# - CLI (`hermes` command): Uses "." (current directory where you run hermes)
# - Gateway/messaging/cron: Uses terminal.cwd here; legacy .env cwd values are deprecated
cron:
catch_up_missed: true # False skips past-grace recurring misses after planned downtime.
terminal:
backend: "local"
cwd: "." # For local backend: "." = current directory. Ignored for remote backends unless a backend documents otherwise.
+14 -7
View File
@@ -2883,8 +2883,8 @@ def _reanchor_stale_cron(d: _DueJob) -> bool:
return False
def _fast_forward_missed_recurring(d: _DueJob, grace: int) -> None:
"""Recurring job past its grace window: skip the accumulated misses, fire once now.
def _fast_forward_missed_recurring(d: _DueJob, grace: int) -> bool:
"""Re-anchor accumulated misses; return whether catch-up was explicitly disabled.
The fast-forward is persisted immediately — NOT redundant with advance_next_run/mark_job_run:
it
@@ -2892,16 +2892,23 @@ def _fast_forward_missed_recurring(d: _DueJob, grace: int) -> None:
calls advance_next_run. mark_job_run re-anchors on completion, so the value is provisional.
"""
if (d.scan.now - d.next_run_dt).total_seconds() <= grace:
return
return False
new_next = d.recompute_next()
if not new_next:
return
return False
d.scan.persist(d.job["id"], next_run_at=new_next)
if not _cron_config_number("catch_up_missed", True, lambda value: value is not False):
logger.info(
"Job '%s' missed its scheduled time (%s, grace=%ds). "
"Skipping missed occurrence because cron.catch_up_missed is false; next run: %s",
d.label, d.next_run, grace, new_next)
return True
logger.info(
"Job '%s' missed its scheduled time (%s, grace=%ds). "
"Running now; next run provisionally set to: %s (re-anchored on completion)",
d.label, d.next_run, grace, new_next)
d.scan.persist(d.job["id"], next_run_at=new_next)
record_catch_up_occurrence()
return False
def _retire_expired_oneshot(d: _DueJob) -> bool:
@@ -3006,8 +3013,8 @@ def _evaluate_due_job(job: Dict[str, Any], scan: _DueScan, run_claim_ttl: float)
if not manual_run and kind == "cron" and _reanchor_stale_cron(d):
return False
grace = _compute_grace_seconds(d.schedule)
if not manual_run and recurring:
_fast_forward_missed_recurring(d, grace)
if not manual_run and recurring and _fast_forward_missed_recurring(d, grace):
return False
if kind == "once":
if _retire_expired_oneshot(d) or _oneshot_dispatch_limit_reached(job, scan):
return False
+1
View File
@@ -1633,6 +1633,7 @@ DEFAULT_CONFIG = {
},
"cron": {
"catch_up_missed": True, # False skips recurring misses beyond the local grace window.
# Let cron-spawned agents use the cronjob toolset (the "cron-librarian" pattern). Off by
# default: policy-denied in cron context to prevent unattended scheduling loops. Jobs
# created this way are user-owned in the same flat jobs table. Interactive toolsets
+41
View File
@@ -0,0 +1,41 @@
"""The local missed-run policy preserves grace and manual triggers."""
from datetime import timedelta
import pytest
from cron import jobs
@pytest.mark.parametrize("catch_up", [True, False])
def test_missed_policy_preserves_grace_and_manual(tmp_path, monkeypatch, catch_up):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
(tmp_path / "config.yaml").write_text(f"cron:\n catch_up_missed: {str(catch_up).lower()}\n", encoding="utf-8")
with jobs.use_cron_store(tmp_path / "cron"):
now = jobs._hermes_now()
for name, lag in [("stale", 14400), ("grace", 30), ("manual", 14400)]:
job = jobs.create_job(prompt=name, schedule="every 1h", model="fixture", deliver="local")
stored = jobs.load_jobs()
row = next(row for row in stored if row["id"] == job["id"])
row["next_run_at"] = (now - timedelta(seconds=lag)).isoformat()
if name == "manual":
row["manual_run_at"] = row["next_run_at"]
jobs.save_jobs(stored)
due = {job["prompt"] for job in jobs.get_due_jobs()}
assert due == ({"stale", "grace", "manual"} if catch_up else {"grace", "manual"})
stale = next(row for row in jobs.load_jobs() if row["prompt"] == "stale")
assert jobs._ensure_aware(jobs.datetime.fromisoformat(stale["next_run_at"])) > now
@pytest.mark.parametrize("uncomputable", [False, True])
def test_default_and_uncomputable_still_catch_up(tmp_path, monkeypatch, uncomputable):
monkeypatch.setenv("HERMES_HOME", str(tmp_path))
if uncomputable:
(tmp_path / "config.yaml").write_text("cron:\n catch_up_missed: false\n", encoding="utf-8")
with jobs.use_cron_store(tmp_path / "cron"):
job = jobs.create_job(prompt="default", schedule="every 1h", model="fixture", deliver="local")
stored = jobs.load_jobs()
stored[0]["next_run_at"] = (jobs._hermes_now() - timedelta(hours=4)).isoformat()
jobs.save_jobs(stored)
if uncomputable:
monkeypatch.setattr(jobs, "compute_next_run", lambda *args: None)
assert [row["id"] for row in jobs.get_due_jobs()] == [job["id"]]
+18
View File
@@ -880,6 +880,24 @@ These misses are stamped on the job record as `last_fire_error` (timestamp + rea
The stamp always reflects **current** auto-fire health: it is overwritten by newer misses and cleared automatically by the next successful run. If you see it, the job and its schedule are fine — the gateway side of the fire path needs attention (most commonly, restart the gateway through its supervisor so it loads the full profile environment: `hermes gateway restart`).
### Local missed-run policy
The local ticker normally runs each overdue recurring job **once**, not once per
missed slot. To avoid catch-up load after a planned gateway stop, set:
```yaml
cron:
catch_up_missed: false # default: true
```
Or run `hermes config set cron.catch_up_missed false`. With this opt-out, a recurring
job later than its existing grace window (half its period, clamped to 120 seconds–2
hours) is re-anchored to its next future occurrence without firing now. The skip is
logged. Jobs inside grace and explicit manual triggers still run normally; if the
next occurrence cannot be computed, the existing run-once fallback is preserved.
This does not change one-shot expiry, resume behavior, or the hosted-provider sweep
below. There is no per-job override.
### Misfire catch-up
When an external scheduler provider is active (managed cron on hosted deployments), the gateway also runs a catch-up sweep: a job whose scheduled time passed with no fire delivered — and whose grace window has elapsed — is claimed and run locally, so an outage in the fire hand-off costs minutes instead of the whole day. The sweep is de-duplicated against late scheduler retries by the same store claim used for normal fires.