Inspired by Poke: nudge review of repeatedly-failing recurring cron jobs
Poke (poke.com) 'encourages users to review recurring automations that haven't been acted upon'. Hermes' equivalent pain point is a recurring cron job that fails run after run: each failure delivers the same one-line error with no signal that the automation itself needs attention. - cron/jobs.py: persist a failure_streak counter in mark_job_run — incremented on agent failure, reset on success; delivery failures don't count. Back-compat: missing field reads as 0. - cron/scheduler.py: _failure_streak_nudge() appends a review nudge to the delivered failure summary once a recurring job's streak reaches cron.failure_nudge_threshold (default 3, 0 disables). One-shots never nudge. - hermes_cli/cron.py: 'hermes cron list' shows '(N failures in a row)' on failing jobs with streak >= 2. - docs: new 'Repeated-failure review nudge' section in cron.md. Tests: 17 passed (TestMarkJobRun + TestFailureStreakNudge); E2E verified with real cron store in temp HERMES_HOME.
This commit is contained in:
@@ -1854,6 +1854,7 @@ def create_job(
|
||||
"last_status": None,
|
||||
"last_error": None,
|
||||
"last_delivery_error": None,
|
||||
"failure_streak": 0,
|
||||
# Delivery configuration
|
||||
"deliver": deliver,
|
||||
"origin": origin, # Tracks where job was created for "origin" delivery
|
||||
@@ -2282,6 +2283,15 @@ def _mark_job_run_locked(
|
||||
if success:
|
||||
job.pop("preflight_alerted", None)
|
||||
job.pop("drift_alerted", None)
|
||||
# Consecutive agent-failure streak. Any successful run resets
|
||||
# it; delivery failures alone do NOT count (the agent did its
|
||||
# job). Read by the scheduler's failure-delivery path to nudge
|
||||
# the user to review a repeatedly-failing automation
|
||||
# (Poke-inspired; see cron/scheduler._failure_streak_nudge).
|
||||
if success:
|
||||
job["failure_streak"] = 0
|
||||
else:
|
||||
job["failure_streak"] = int(job.get("failure_streak") or 0) + 1
|
||||
# Track delivery failures separately — cleared on successful delivery
|
||||
job["last_delivery_error"] = delivery_error
|
||||
# Clear any external-fire claim so a re-armed recurring job can
|
||||
|
||||
+46
-1
@@ -148,6 +148,48 @@ def _fallback_chain_phrase() -> str:
|
||||
)
|
||||
|
||||
|
||||
def _failure_streak_nudge(job: dict) -> str:
|
||||
"""Return a review nudge when a recurring job keeps failing, else "".
|
||||
|
||||
Inspired by Poke (poke.com), which "encourages users to review recurring
|
||||
automations that haven't been acted upon": once a recurring job has failed
|
||||
several runs in a row, the per-run failure ping stops being information and
|
||||
starts being noise — the useful message is "this automation needs your
|
||||
attention (fix, pause, or remove it)".
|
||||
|
||||
The streak counter (``failure_streak``) is persisted by
|
||||
``cron.jobs.mark_job_run`` and reset on any successful run. Because the
|
||||
failure message is delivered BEFORE ``mark_job_run`` records this run, the
|
||||
prospective streak for the current failure is stored+1.
|
||||
|
||||
Threshold config: ``cron.failure_nudge_threshold`` (default 3, ``0``
|
||||
disables the nudge). One-shot jobs never nudge — they don't recur.
|
||||
"""
|
||||
schedule_kind = (job.get("schedule") or {}).get("kind")
|
||||
if schedule_kind not in {"cron", "interval"}:
|
||||
return ""
|
||||
try:
|
||||
cfg = load_config() or {}
|
||||
threshold = int(
|
||||
((cfg.get("cron") or {}) if isinstance(cfg, dict) else {}).get(
|
||||
"failure_nudge_threshold", 3
|
||||
)
|
||||
)
|
||||
except Exception:
|
||||
threshold = 3
|
||||
if threshold <= 0:
|
||||
return ""
|
||||
streak = int(job.get("failure_streak") or 0) + 1 # +1 = this run
|
||||
if streak < threshold:
|
||||
return ""
|
||||
job_ref = job.get("name") or job.get("id") or "this job"
|
||||
return (
|
||||
f"\nThis job has failed {streak} runs in a row — worth a review. "
|
||||
f"Fix its prompt/config, or pause it with `hermes cron pause {job_ref}` "
|
||||
"(resume/remove also available) to stop the noise."
|
||||
)
|
||||
|
||||
|
||||
def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
|
||||
"""Return a compact one-line failure message for chat delivery.
|
||||
|
||||
@@ -6070,7 +6112,10 @@ def _run_one_job_body(
|
||||
"the configuration is fixed."
|
||||
)
|
||||
else:
|
||||
deliver_content = final_response if success else _summarize_cron_failure_for_delivery(job, error)
|
||||
deliver_content = final_response if success else (
|
||||
_summarize_cron_failure_for_delivery(job, error)
|
||||
+ _failure_streak_nudge(job)
|
||||
)
|
||||
if drift_skip and not success:
|
||||
# Drift-skip alert: bypass the generic summarizer's
|
||||
# 180-char truncation (it would eat the remediation
|
||||
|
||||
@@ -182,6 +182,9 @@ def cron_list(show_all: bool = False):
|
||||
status_display = color("ok", Colors.GREEN)
|
||||
else:
|
||||
status_display = color(f"{last_status}: {job.get('last_error', '?')}", Colors.RED)
|
||||
streak = int(job.get("failure_streak") or 0)
|
||||
if streak >= 2:
|
||||
status_display += color(f" ({streak} failures in a row)", Colors.RED)
|
||||
print(f" Last run: {last_run} {status_display}")
|
||||
|
||||
latest_execution = job.get("latest_execution")
|
||||
|
||||
@@ -484,6 +484,34 @@ class TestMarkJobRun:
|
||||
assert updated["last_error"] is None
|
||||
assert updated["last_delivery_error"] == "platform 'telegram' not configured"
|
||||
|
||||
def test_failure_streak_increments_and_resets(self, tmp_cron_dir):
|
||||
"""failure_streak counts consecutive agent failures; success resets."""
|
||||
job = create_job(prompt="Flaky", schedule="every 1h")
|
||||
assert get_job(job["id"])["failure_streak"] == 0
|
||||
mark_job_run(job["id"], success=False, error="timeout")
|
||||
mark_job_run(job["id"], success=False, error="timeout")
|
||||
assert get_job(job["id"])["failure_streak"] == 2
|
||||
mark_job_run(job["id"], success=True)
|
||||
assert get_job(job["id"])["failure_streak"] == 0
|
||||
|
||||
def test_failure_streak_ignores_delivery_errors(self, tmp_cron_dir):
|
||||
"""A successful run with a delivery error must not count as a failure."""
|
||||
job = create_job(prompt="Report", schedule="every 1h")
|
||||
mark_job_run(job["id"], success=False, error="timeout")
|
||||
mark_job_run(job["id"], success=True, delivery_error="send failed: 502")
|
||||
assert get_job(job["id"])["failure_streak"] == 0
|
||||
|
||||
def test_failure_streak_backcompat_missing_field(self, tmp_cron_dir):
|
||||
"""Jobs persisted before the field existed increment from 0."""
|
||||
job = create_job(prompt="Old", schedule="every 1h")
|
||||
# Simulate a pre-field record on disk.
|
||||
jobs = load_jobs()
|
||||
for j in jobs:
|
||||
j.pop("failure_streak", None)
|
||||
save_jobs(jobs)
|
||||
mark_job_run(job["id"], success=False, error="boom")
|
||||
assert get_job(job["id"])["failure_streak"] == 1
|
||||
|
||||
|
||||
def test_recurring_cron_not_disabled_when_croniter_missing(self, tmp_cron_dir, monkeypatch):
|
||||
"""Regression test for issue #16265.
|
||||
|
||||
@@ -2556,3 +2556,55 @@ class TestSetCronSessionTitle:
|
||||
assert out == "Nightly Synthesis #2"
|
||||
db.get_next_title_in_lineage.assert_called_once_with("Nightly Synthesis")
|
||||
|
||||
|
||||
|
||||
|
||||
class TestFailureStreakNudge:
|
||||
"""Poke-inspired repeated-failure review nudge (_failure_streak_nudge)."""
|
||||
|
||||
def _job(self, streak, kind="cron", name="scout"):
|
||||
return {
|
||||
"id": "j1",
|
||||
"name": name,
|
||||
"failure_streak": streak,
|
||||
"schedule": {"kind": kind},
|
||||
}
|
||||
|
||||
def test_nudges_at_threshold(self):
|
||||
from cron.scheduler import _failure_streak_nudge
|
||||
# stored streak 2 + this run = 3 >= default threshold 3
|
||||
with patch("cron.scheduler.load_config", return_value={}):
|
||||
out = _failure_streak_nudge(self._job(2))
|
||||
assert "failed 3 runs in a row" in out
|
||||
assert "hermes cron pause scout" in out
|
||||
|
||||
def test_silent_below_threshold(self):
|
||||
from cron.scheduler import _failure_streak_nudge
|
||||
with patch("cron.scheduler.load_config", return_value={}):
|
||||
assert _failure_streak_nudge(self._job(0)) == ""
|
||||
assert _failure_streak_nudge(self._job(1)) == ""
|
||||
|
||||
def test_oneshot_never_nudges(self):
|
||||
from cron.scheduler import _failure_streak_nudge
|
||||
with patch("cron.scheduler.load_config", return_value={}):
|
||||
assert _failure_streak_nudge(self._job(10, kind="once")) == ""
|
||||
|
||||
def test_config_threshold_and_disable(self):
|
||||
from cron.scheduler import _failure_streak_nudge
|
||||
cfg5 = {"cron": {"failure_nudge_threshold": 5}}
|
||||
with patch("cron.scheduler.load_config", return_value=cfg5):
|
||||
assert _failure_streak_nudge(self._job(3)) == ""
|
||||
assert "failed 5 runs" in _failure_streak_nudge(self._job(4))
|
||||
with patch("cron.scheduler.load_config", return_value={"cron": {"failure_nudge_threshold": 0}}):
|
||||
assert _failure_streak_nudge(self._job(50)) == ""
|
||||
|
||||
def test_missing_streak_field_backcompat(self):
|
||||
from cron.scheduler import _failure_streak_nudge
|
||||
job = {"id": "old", "schedule": {"kind": "interval"}} # pre-field job
|
||||
with patch("cron.scheduler.load_config", return_value={}):
|
||||
assert _failure_streak_nudge(job) == ""
|
||||
|
||||
def test_config_load_failure_falls_back(self):
|
||||
from cron.scheduler import _failure_streak_nudge
|
||||
with patch("cron.scheduler.load_config", side_effect=RuntimeError("boom")):
|
||||
assert "failed 3 runs" in _failure_streak_nudge(self._job(2))
|
||||
|
||||
@@ -328,6 +328,21 @@ Inspect recent attempts with `hermes cron runs [job-id] --limit 20` (alias:
|
||||
`history`). Terminal history is bounded; active attempts are never pruned. The
|
||||
ledger is included in quick backups.
|
||||
|
||||
### Repeated-failure review nudge
|
||||
|
||||
Each job tracks a `failure_streak` — consecutive runs where the agent failed
|
||||
(delivery failures don't count). When a *recurring* job's streak reaches the
|
||||
threshold, the failure message delivered to chat gains a review nudge telling
|
||||
you the job has failed N runs in a row and suggesting you fix, pause
|
||||
(`hermes cron pause <job>`), or remove it. Any successful run resets the
|
||||
streak, and `hermes cron list` shows the streak alongside a failing job's last
|
||||
run. One-shot jobs never nudge.
|
||||
|
||||
```yaml
|
||||
cron:
|
||||
failure_nudge_threshold: 3 # default; 0 disables the nudge
|
||||
```
|
||||
|
||||
## Delivery options
|
||||
|
||||
When scheduling jobs, you specify where the output goes:
|
||||
|
||||
Reference in New Issue
Block a user