diff --git a/cli-config.yaml.example b/cli-config.yaml.example index b63a32cdae..82e3bf7a10 100644 --- a/cli-config.yaml.example +++ b/cli-config.yaml.example @@ -359,6 +359,9 @@ kanban: # Working directory behavior: # - CLI (`hermes` command): Uses "." (current directory where you run hermes) # - Gateway/messaging/cron: Uses terminal.cwd here; legacy .env cwd values are deprecated +cron: + catch_up_missed: true # False skips past-grace recurring misses after planned downtime. + terminal: backend: "local" cwd: "." # For local backend: "." = current directory. Ignored for remote backends unless a backend documents otherwise. diff --git a/cron/jobs.py b/cron/jobs.py index d24acc0b19..51d51a77e0 100644 --- a/cron/jobs.py +++ b/cron/jobs.py @@ -2883,8 +2883,8 @@ def _reanchor_stale_cron(d: _DueJob) -> bool: return False -def _fast_forward_missed_recurring(d: _DueJob, grace: int) -> None: - """Recurring job past its grace window: skip the accumulated misses, fire once now. +def _fast_forward_missed_recurring(d: _DueJob, grace: int) -> bool: + """Re-anchor accumulated misses; return whether catch-up was explicitly disabled. The fast-forward is persisted immediately — NOT redundant with advance_next_run/mark_job_run: it @@ -2892,16 +2892,23 @@ def _fast_forward_missed_recurring(d: _DueJob, grace: int) -> None: calls advance_next_run. mark_job_run re-anchors on completion, so the value is provisional. """ if (d.scan.now - d.next_run_dt).total_seconds() <= grace: - return + return False new_next = d.recompute_next() if not new_next: - return + return False + d.scan.persist(d.job["id"], next_run_at=new_next) + if not _cron_config_number("catch_up_missed", True, lambda value: value is not False): + logger.info( + "Job '%s' missed its scheduled time (%s, grace=%ds). " + "Skipping missed occurrence because cron.catch_up_missed is false; next run: %s", + d.label, d.next_run, grace, new_next) + return True logger.info( "Job '%s' missed its scheduled time (%s, grace=%ds). " "Running now; next run provisionally set to: %s (re-anchored on completion)", d.label, d.next_run, grace, new_next) - d.scan.persist(d.job["id"], next_run_at=new_next) record_catch_up_occurrence() + return False def _retire_expired_oneshot(d: _DueJob) -> bool: @@ -3006,8 +3013,8 @@ def _evaluate_due_job(job: Dict[str, Any], scan: _DueScan, run_claim_ttl: float) if not manual_run and kind == "cron" and _reanchor_stale_cron(d): return False grace = _compute_grace_seconds(d.schedule) - if not manual_run and recurring: - _fast_forward_missed_recurring(d, grace) + if not manual_run and recurring and _fast_forward_missed_recurring(d, grace): + return False if kind == "once": if _retire_expired_oneshot(d) or _oneshot_dispatch_limit_reached(job, scan): return False diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 581a6e9f31..41b846a56f 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -1633,6 +1633,7 @@ DEFAULT_CONFIG = { }, "cron": { + "catch_up_missed": True, # False skips recurring misses beyond the local grace window. # Let cron-spawned agents use the cronjob toolset (the "cron-librarian" pattern). Off by # default: policy-denied in cron context to prevent unattended scheduling loops. Jobs # created this way are user-owned in the same flat jobs table. Interactive toolsets diff --git a/tests/cron/test_catchup_policy.py b/tests/cron/test_catchup_policy.py new file mode 100644 index 0000000000..7f78e1186d --- /dev/null +++ b/tests/cron/test_catchup_policy.py @@ -0,0 +1,41 @@ +"""The local missed-run policy preserves grace and manual triggers.""" +from datetime import timedelta + +import pytest + +from cron import jobs + + +@pytest.mark.parametrize("catch_up", [True, False]) +def test_missed_policy_preserves_grace_and_manual(tmp_path, monkeypatch, catch_up): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + (tmp_path / "config.yaml").write_text(f"cron:\n catch_up_missed: {str(catch_up).lower()}\n", encoding="utf-8") + with jobs.use_cron_store(tmp_path / "cron"): + now = jobs._hermes_now() + for name, lag in [("stale", 14400), ("grace", 30), ("manual", 14400)]: + job = jobs.create_job(prompt=name, schedule="every 1h", model="fixture", deliver="local") + stored = jobs.load_jobs() + row = next(row for row in stored if row["id"] == job["id"]) + row["next_run_at"] = (now - timedelta(seconds=lag)).isoformat() + if name == "manual": + row["manual_run_at"] = row["next_run_at"] + jobs.save_jobs(stored) + due = {job["prompt"] for job in jobs.get_due_jobs()} + assert due == ({"stale", "grace", "manual"} if catch_up else {"grace", "manual"}) + stale = next(row for row in jobs.load_jobs() if row["prompt"] == "stale") + assert jobs._ensure_aware(jobs.datetime.fromisoformat(stale["next_run_at"])) > now + + +@pytest.mark.parametrize("uncomputable", [False, True]) +def test_default_and_uncomputable_still_catch_up(tmp_path, monkeypatch, uncomputable): + monkeypatch.setenv("HERMES_HOME", str(tmp_path)) + if uncomputable: + (tmp_path / "config.yaml").write_text("cron:\n catch_up_missed: false\n", encoding="utf-8") + with jobs.use_cron_store(tmp_path / "cron"): + job = jobs.create_job(prompt="default", schedule="every 1h", model="fixture", deliver="local") + stored = jobs.load_jobs() + stored[0]["next_run_at"] = (jobs._hermes_now() - timedelta(hours=4)).isoformat() + jobs.save_jobs(stored) + if uncomputable: + monkeypatch.setattr(jobs, "compute_next_run", lambda *args: None) + assert [row["id"] for row in jobs.get_due_jobs()] == [job["id"]] diff --git a/website/docs/user-guide/features/cron.md b/website/docs/user-guide/features/cron.md index c20852630d..0b353d5359 100644 --- a/website/docs/user-guide/features/cron.md +++ b/website/docs/user-guide/features/cron.md @@ -880,6 +880,24 @@ These misses are stamped on the job record as `last_fire_error` (timestamp + rea The stamp always reflects **current** auto-fire health: it is overwritten by newer misses and cleared automatically by the next successful run. If you see it, the job and its schedule are fine — the gateway side of the fire path needs attention (most commonly, restart the gateway through its supervisor so it loads the full profile environment: `hermes gateway restart`). +### Local missed-run policy + +The local ticker normally runs each overdue recurring job **once**, not once per +missed slot. To avoid catch-up load after a planned gateway stop, set: + +```yaml +cron: + catch_up_missed: false # default: true +``` + +Or run `hermes config set cron.catch_up_missed false`. With this opt-out, a recurring +job later than its existing grace window (half its period, clamped to 120 seconds–2 +hours) is re-anchored to its next future occurrence without firing now. The skip is +logged. Jobs inside grace and explicit manual triggers still run normally; if the +next occurrence cannot be computed, the existing run-once fallback is preserved. +This does not change one-shot expiry, resume behavior, or the hosted-provider sweep +below. There is no per-job override. + ### Misfire catch-up When an external scheduler provider is active (managed cron on hosted deployments), the gateway also runs a catch-up sweep: a job whose scheduled time passed with no fire delivered — and whose grace window has elapsed — is claimed and run locally, so an outage in the fire hand-off costs minutes instead of the whole day. The sweep is de-duplicated against late scheduler retries by the same store claim used for normal fires.