Inspired by Poke: nudge review of repeatedly-failing recurring cron jobs

Poke (poke.com) 'encourages users to review recurring automations that
haven't been acted upon'. Hermes' equivalent pain point is a recurring
cron job that fails run after run: each failure delivers the same one-line
error with no signal that the automation itself needs attention.

- cron/jobs.py: persist a failure_streak counter in mark_job_run —
  incremented on agent failure, reset on success; delivery failures don't
  count. Back-compat: missing field reads as 0.
- cron/scheduler.py: _failure_streak_nudge() appends a review nudge to the
  delivered failure summary once a recurring job's streak reaches
  cron.failure_nudge_threshold (default 3, 0 disables). One-shots never
  nudge.
- hermes_cli/cron.py: 'hermes cron list' shows '(N failures in a row)' on
  failing jobs with streak >= 2.
- docs: new 'Repeated-failure review nudge' section in cron.md.

Tests: 17 passed (TestMarkJobRun + TestFailureStreakNudge); E2E verified
with real cron store in temp HERMES_HOME.
This commit is contained in:
Teknium
2026-08-06 20:34:10 -07:00
parent bd4b709258
commit 07a5179158
6 changed files with 154 additions and 1 deletions
+10
View File
@@ -1854,6 +1854,7 @@ def create_job(
"last_status": None,
"last_error": None,
"last_delivery_error": None,
"failure_streak": 0,
# Delivery configuration
"deliver": deliver,
"origin": origin, # Tracks where job was created for "origin" delivery
@@ -2282,6 +2283,15 @@ def _mark_job_run_locked(
if success:
job.pop("preflight_alerted", None)
job.pop("drift_alerted", None)
# Consecutive agent-failure streak. Any successful run resets
# it; delivery failures alone do NOT count (the agent did its
# job). Read by the scheduler's failure-delivery path to nudge
# the user to review a repeatedly-failing automation
# (Poke-inspired; see cron/scheduler._failure_streak_nudge).
if success:
job["failure_streak"] = 0
else:
job["failure_streak"] = int(job.get("failure_streak") or 0) + 1
# Track delivery failures separately — cleared on successful delivery
job["last_delivery_error"] = delivery_error
# Clear any external-fire claim so a re-armed recurring job can
+46 -1
View File
@@ -148,6 +148,48 @@ def _fallback_chain_phrase() -> str:
)
def _failure_streak_nudge(job: dict) -> str:
"""Return a review nudge when a recurring job keeps failing, else "".
Inspired by Poke (poke.com), which "encourages users to review recurring
automations that haven't been acted upon": once a recurring job has failed
several runs in a row, the per-run failure ping stops being information and
starts being noise — the useful message is "this automation needs your
attention (fix, pause, or remove it)".
The streak counter (``failure_streak``) is persisted by
``cron.jobs.mark_job_run`` and reset on any successful run. Because the
failure message is delivered BEFORE ``mark_job_run`` records this run, the
prospective streak for the current failure is stored+1.
Threshold config: ``cron.failure_nudge_threshold`` (default 3, ``0``
disables the nudge). One-shot jobs never nudge — they don't recur.
"""
schedule_kind = (job.get("schedule") or {}).get("kind")
if schedule_kind not in {"cron", "interval"}:
return ""
try:
cfg = load_config() or {}
threshold = int(
((cfg.get("cron") or {}) if isinstance(cfg, dict) else {}).get(
"failure_nudge_threshold", 3
)
)
except Exception:
threshold = 3
if threshold <= 0:
return ""
streak = int(job.get("failure_streak") or 0) + 1 # +1 = this run
if streak < threshold:
return ""
job_ref = job.get("name") or job.get("id") or "this job"
return (
f"\nThis job has failed {streak} runs in a row — worth a review. "
f"Fix its prompt/config, or pause it with `hermes cron pause {job_ref}` "
"(resume/remove also available) to stop the noise."
)
def _summarize_cron_failure_for_delivery(job: dict, error: str | None) -> str:
"""Return a compact one-line failure message for chat delivery.
@@ -6070,7 +6112,10 @@ def _run_one_job_body(
"the configuration is fixed."
)
else:
deliver_content = final_response if success else _summarize_cron_failure_for_delivery(job, error)
deliver_content = final_response if success else (
_summarize_cron_failure_for_delivery(job, error)
+ _failure_streak_nudge(job)
)
if drift_skip and not success:
# Drift-skip alert: bypass the generic summarizer's
# 180-char truncation (it would eat the remediation
+3
View File
@@ -182,6 +182,9 @@ def cron_list(show_all: bool = False):
status_display = color("ok", Colors.GREEN)
else:
status_display = color(f"{last_status}: {job.get('last_error', '?')}", Colors.RED)
streak = int(job.get("failure_streak") or 0)
if streak >= 2:
status_display += color(f" ({streak} failures in a row)", Colors.RED)
print(f" Last run: {last_run} {status_display}")
latest_execution = job.get("latest_execution")
+28
View File
@@ -484,6 +484,34 @@ class TestMarkJobRun:
assert updated["last_error"] is None
assert updated["last_delivery_error"] == "platform 'telegram' not configured"
def test_failure_streak_increments_and_resets(self, tmp_cron_dir):
"""failure_streak counts consecutive agent failures; success resets."""
job = create_job(prompt="Flaky", schedule="every 1h")
assert get_job(job["id"])["failure_streak"] == 0
mark_job_run(job["id"], success=False, error="timeout")
mark_job_run(job["id"], success=False, error="timeout")
assert get_job(job["id"])["failure_streak"] == 2
mark_job_run(job["id"], success=True)
assert get_job(job["id"])["failure_streak"] == 0
def test_failure_streak_ignores_delivery_errors(self, tmp_cron_dir):
"""A successful run with a delivery error must not count as a failure."""
job = create_job(prompt="Report", schedule="every 1h")
mark_job_run(job["id"], success=False, error="timeout")
mark_job_run(job["id"], success=True, delivery_error="send failed: 502")
assert get_job(job["id"])["failure_streak"] == 0
def test_failure_streak_backcompat_missing_field(self, tmp_cron_dir):
"""Jobs persisted before the field existed increment from 0."""
job = create_job(prompt="Old", schedule="every 1h")
# Simulate a pre-field record on disk.
jobs = load_jobs()
for j in jobs:
j.pop("failure_streak", None)
save_jobs(jobs)
mark_job_run(job["id"], success=False, error="boom")
assert get_job(job["id"])["failure_streak"] == 1
def test_recurring_cron_not_disabled_when_croniter_missing(self, tmp_cron_dir, monkeypatch):
"""Regression test for issue #16265.
+52
View File
@@ -2556,3 +2556,55 @@ class TestSetCronSessionTitle:
assert out == "Nightly Synthesis #2"
db.get_next_title_in_lineage.assert_called_once_with("Nightly Synthesis")
class TestFailureStreakNudge:
"""Poke-inspired repeated-failure review nudge (_failure_streak_nudge)."""
def _job(self, streak, kind="cron", name="scout"):
return {
"id": "j1",
"name": name,
"failure_streak": streak,
"schedule": {"kind": kind},
}
def test_nudges_at_threshold(self):
from cron.scheduler import _failure_streak_nudge
# stored streak 2 + this run = 3 >= default threshold 3
with patch("cron.scheduler.load_config", return_value={}):
out = _failure_streak_nudge(self._job(2))
assert "failed 3 runs in a row" in out
assert "hermes cron pause scout" in out
def test_silent_below_threshold(self):
from cron.scheduler import _failure_streak_nudge
with patch("cron.scheduler.load_config", return_value={}):
assert _failure_streak_nudge(self._job(0)) == ""
assert _failure_streak_nudge(self._job(1)) == ""
def test_oneshot_never_nudges(self):
from cron.scheduler import _failure_streak_nudge
with patch("cron.scheduler.load_config", return_value={}):
assert _failure_streak_nudge(self._job(10, kind="once")) == ""
def test_config_threshold_and_disable(self):
from cron.scheduler import _failure_streak_nudge
cfg5 = {"cron": {"failure_nudge_threshold": 5}}
with patch("cron.scheduler.load_config", return_value=cfg5):
assert _failure_streak_nudge(self._job(3)) == ""
assert "failed 5 runs" in _failure_streak_nudge(self._job(4))
with patch("cron.scheduler.load_config", return_value={"cron": {"failure_nudge_threshold": 0}}):
assert _failure_streak_nudge(self._job(50)) == ""
def test_missing_streak_field_backcompat(self):
from cron.scheduler import _failure_streak_nudge
job = {"id": "old", "schedule": {"kind": "interval"}} # pre-field job
with patch("cron.scheduler.load_config", return_value={}):
assert _failure_streak_nudge(job) == ""
def test_config_load_failure_falls_back(self):
from cron.scheduler import _failure_streak_nudge
with patch("cron.scheduler.load_config", side_effect=RuntimeError("boom")):
assert "failed 3 runs" in _failure_streak_nudge(self._job(2))
+15
View File
@@ -328,6 +328,21 @@ Inspect recent attempts with `hermes cron runs [job-id] --limit 20` (alias:
`history`). Terminal history is bounded; active attempts are never pruned. The
ledger is included in quick backups.
### Repeated-failure review nudge
Each job tracks a `failure_streak` — consecutive runs where the agent failed
(delivery failures don't count). When a *recurring* job's streak reaches the
threshold, the failure message delivered to chat gains a review nudge telling
you the job has failed N runs in a row and suggesting you fix, pause
(`hermes cron pause <job>`), or remove it. Any successful run resets the
streak, and `hermes cron list` shows the streak alongside a failing job's last
run. One-shot jobs never nudge.
```yaml
cron:
failure_nudge_threshold: 3 # default; 0 disables the nudge
```
## Delivery options
When scheduling jobs, you specify where the output goes: