fix(cron): make failed runs diagnosable without verbose delivery errors
Persist a redacted chained traceback in the private run output and expose redacted last_error in tool and slash listings, including historical errors. Keep the run_job concise error return unchanged for delivery classification. Slim redo of liuhao1024's earliest #104545; adds forced redaction and keeps formatting in a topical sibling. Local SDK/socket A/B verifies diagnosis visibility plus healthy-script, clearing, and private-file controls. Canonical tests queued under the campaign lock at commit time. Co-authored-by: liuhao1024 <sunsky.lau@gmail.com>
This commit is contained in:
+2
-1
@@ -2373,9 +2373,10 @@ def run_job(
|
||||
# No audit row when we failed before the agent existed; the audit write must never raise.
|
||||
if _audit is not None:
|
||||
_audit.write({}, error_msg)
|
||||
from cron.scheduler_diagnostics import format_run_error
|
||||
output = (
|
||||
_run_doc_header(job, f"{job_name} (FAILED)", job_id, prompt)
|
||||
+ f"## Error\n\n```\n{error_msg}\n```\n"
|
||||
+ format_run_error(e)
|
||||
)
|
||||
return False, output, "", error_msg
|
||||
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
"""Private run-document diagnostics, kept separate from delivery summaries."""
|
||||
|
||||
from traceback import format_exception
|
||||
|
||||
from agent.redact import redact_sensitive_text
|
||||
|
||||
|
||||
def format_run_error(exc: BaseException) -> str:
|
||||
"""Retain chained causes without capturing locals or exposing URL credentials."""
|
||||
traceback_text = redact_sensitive_text(
|
||||
"".join(format_exception(exc)), force=True, redact_url_credentials=True,
|
||||
)
|
||||
return f"## Error\n\n```\n{traceback_text}\n```\n"
|
||||
@@ -0,0 +1,78 @@
|
||||
"""Local-I/O A/B for #104538; no inference, external APIs, or deliveries.
|
||||
|
||||
Run with the project Python and pass a checkout to probe (defaults to this tree).
|
||||
Exit 1 on the unfixed checkout: the persisted traceback and listed reason are absent.
|
||||
The injected provider-setup seam performs a real SDK connection to a non-listening
|
||||
loopback socket; this does not diagnose the original macOS/Gmail incident.
|
||||
"""
|
||||
|
||||
import contextlib
|
||||
import io
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import socket
|
||||
import sys
|
||||
import tempfile
|
||||
from unittest.mock import patch
|
||||
|
||||
|
||||
def main():
|
||||
checkout = Path(sys.argv[1]).resolve() if len(sys.argv) > 1 else Path(__file__).resolve().parents[1]
|
||||
for key in list(os.environ):
|
||||
if key.startswith(("HERMES_", "OPENAI_", "OPENROUTER_", "ANTHROPIC_")) or any(
|
||||
part in key for part in ("API_KEY", "TOKEN", "SECRET", "PASSWORD")
|
||||
):
|
||||
os.environ.pop(key, None)
|
||||
with tempfile.TemporaryDirectory(prefix="cron-diagnostics-") as home:
|
||||
os.environ.update(HOME=home, HERMES_HOME=home)
|
||||
sys.path.insert(0, str(checkout))
|
||||
from cron import jobs, scheduler
|
||||
from hermes_cli.cli_commands_mixin import CLICommandsMixin
|
||||
from openai import OpenAI
|
||||
from tools.cronjob_job_args import _format_job
|
||||
|
||||
with socket.socket() as held:
|
||||
held.bind(("127.0.0.1", 0))
|
||||
port = held.getsockname()[1]
|
||||
|
||||
def connection_failure(*args):
|
||||
with OpenAI(api_key="local-probe", base_url=f"http://127.0.0.1:{port}/v1", max_retries=0, timeout=1) as client:
|
||||
try:
|
||||
client.chat.completions.create(model="local-probe", messages=[{"role": "user", "content": "probe"}])
|
||||
except Exception as exc:
|
||||
raise RuntimeError("Connection error.") from exc
|
||||
|
||||
job = jobs.create_job(prompt="Check status", schedule="every 1h", model="local-probe", deliver="local")
|
||||
with patch.object(scheduler, "_resolve_cron_agent_setup", connection_failure):
|
||||
success, output, response, error = scheduler.run_job(job)
|
||||
path = jobs.save_job_output(job["id"], output)
|
||||
saved = path.read_text(encoding="utf-8")
|
||||
jobs.mark_job_run(job["id"], success, error=error)
|
||||
formatted = _format_job(jobs.get_job(job["id"]))
|
||||
console = io.StringIO()
|
||||
with contextlib.redirect_stdout(console):
|
||||
CLICommandsMixin._cron_list(object(), "list", {"all": False})
|
||||
checks = {
|
||||
"persisted_traceback": "Traceback (most recent call last)" in saved,
|
||||
"persisted_connection_cause": "Connection refused" in saved,
|
||||
"listed_run_reason": formatted.get("last_error") == error,
|
||||
"slash_list_reason": f"error: {error}" in console.getvalue(),
|
||||
"concise_return": not success and not response and error == "RuntimeError: Connection error.",
|
||||
"separate_failure_fields": formatted["last_delivery_error"] is None and formatted["last_fire_error"] is None,
|
||||
"private_output": os.name == "nt" or path.stat().st_mode & 0o077 == 0,
|
||||
}
|
||||
script = Path(home) / "scripts" / "healthy.py"
|
||||
script.parent.mkdir(parents=True, exist_ok=True)
|
||||
script.write_text("print('healthy local control')\n", encoding="utf-8")
|
||||
healthy = jobs.create_job(prompt="Local control", schedule="every 1h", script=str(script), no_agent=True, deliver="local")
|
||||
ok, _, final, healthy_error = scheduler.run_job(healthy)
|
||||
checks["healthy_run"] = ok and "healthy local control" in final and healthy_error is None
|
||||
jobs.mark_job_run(job["id"], True)
|
||||
checks["success_clears_error"] = jobs.get_job(job["id"])["last_error"] is None
|
||||
print(json.dumps({"checkout": str(checkout), "checks": checks}, indent=2))
|
||||
assert all(checks.values()), checks
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -1686,6 +1686,8 @@ class CLICommandsMixin:
|
||||
# in last_delivery_error (last_error is None).
|
||||
if status == "delivery_failed" and job.get("last_delivery_error"):
|
||||
status = f"delivery_failed: {job['last_delivery_error']}"
|
||||
elif status == "error" and job.get("last_error"):
|
||||
status = f"error: {job['last_error']}"
|
||||
print(f" Last run: {job['last_run_at']} ({status})")
|
||||
print()
|
||||
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
"""Failure detail belongs in the local audit, not the delivery summary."""
|
||||
from cron import jobs, scheduler
|
||||
|
||||
|
||||
def test_run_error_persists_redacted_cause_but_returns_summary(tmp_path, monkeypatch):
|
||||
def fail_setup(*args):
|
||||
try:
|
||||
raise ConnectionError("https://user:password@localhost/api?token=secret-token")
|
||||
except ConnectionError as cause:
|
||||
raise RuntimeError("Connection error.") from cause
|
||||
|
||||
monkeypatch.setattr(scheduler, "_resolve_cron_agent_setup", fail_setup)
|
||||
with jobs.use_cron_store(tmp_path):
|
||||
job = jobs.create_job(prompt="Check status", schedule="every 1h", model="local-probe", deliver="local")
|
||||
success, output, response, error = scheduler.run_job(job)
|
||||
saved = jobs.save_job_output(job["id"], output).read_text(encoding="utf-8")
|
||||
assert not success and response == "" and error == "RuntimeError: Connection error."
|
||||
assert "Traceback (most recent call last)" in saved and "fail_setup" in saved
|
||||
assert "ConnectionError" in saved and "localhost" in saved
|
||||
assert "password" not in saved and "secret-token" not in saved
|
||||
|
||||
|
||||
def test_list_exposes_run_error_and_clears_it_after_success(tmp_path, capsys):
|
||||
from hermes_cli.cli_commands_mixin import CLICommandsMixin
|
||||
from tools.cronjob_job_args import _format_job
|
||||
|
||||
with jobs.use_cron_store(tmp_path):
|
||||
job = jobs.create_job(prompt="Check status", schedule="every 1h", model="local-probe", deliver="local")
|
||||
reason = "RuntimeError: https://user:password@localhost/api?token=secret-token"
|
||||
jobs.mark_job_run(job["id"], False, error=reason)
|
||||
displayed = _format_job(jobs.get_job(job["id"]))
|
||||
assert displayed["last_error"].startswith("RuntimeError:") and "localhost" in displayed["last_error"]
|
||||
assert "password" not in displayed["last_error"] and "secret-token" not in displayed["last_error"]
|
||||
assert displayed["last_delivery_error"] is None and displayed["last_fire_error"] is None
|
||||
CLICommandsMixin._cron_list(object(), "list", {"all": False})
|
||||
console = capsys.readouterr().out
|
||||
assert f"error: {displayed['last_error']}" in console
|
||||
assert "password" not in console and "secret-token" not in console
|
||||
jobs.mark_job_run(job["id"], True)
|
||||
assert _format_job(jobs.get_job(job["id"]))["last_error"] is None
|
||||
@@ -344,6 +344,8 @@ _FORMAT_JOB_OPTIONAL_KEYS = (
|
||||
|
||||
|
||||
def _format_job(job: Dict[str, Any]) -> Dict[str, Any]:
|
||||
from agent.redact import redact_sensitive_text
|
||||
|
||||
prompt = str(job.get("prompt") or "")
|
||||
skills = _canonical_skills(job.get("skill"), job.get("skills"))
|
||||
job_id = str(job.get("id") or "unknown")
|
||||
@@ -366,6 +368,9 @@ def _format_job(job: Dict[str, Any]) -> Dict[str, Any]:
|
||||
"last_delivery_error": job.get("last_delivery_error"),
|
||||
"last_delivery_unverified": job.get("last_delivery_unverified"),
|
||||
"last_fire_error": job.get("last_fire_error"),
|
||||
"last_error": redact_sensitive_text(
|
||||
job["last_error"], force=True, redact_url_credentials=True,
|
||||
) if job.get("last_error") else job.get("last_error"),
|
||||
"enabled": job.get("enabled", True),
|
||||
# Derive from enabled so half-paused records never render as paused.
|
||||
"state": effective_job_state(job),
|
||||
|
||||
@@ -837,6 +837,20 @@ Cron jobs inherit your configured fallback providers and credential pool rotatio
|
||||
|
||||
This means cron jobs that run at high frequency or during peak hours are more resilient — a single rate-limited key won't fail the entire run.
|
||||
|
||||
## Run failures (`last_error`)
|
||||
|
||||
A failed agent run records a concise `last_error`, visible in job listings and `/cron list`
|
||||
with credential patterns and URL credentials redacted (including previously stored errors).
|
||||
This is separate from `last_fire_error` (scheduler handoff) and `last_delivery_error` (delivery).
|
||||
Those fields can correctly be empty when the agent itself failed.
|
||||
|
||||
For a connection failure, inspect the run document under `cron/output/<job_id>/` in the active
|
||||
Hermes home. Its `## Error` section includes the chained traceback, with credential patterns
|
||||
and URL credentials redacted. The file uses the existing private output-file permissions;
|
||||
traceback locals are not captured. Delivery notices and `last_error` retain the concise error,
|
||||
not the full traceback. Review diagnostics before sharing: redaction is not a guarantee that
|
||||
arbitrary application data is non-sensitive.
|
||||
|
||||
## Missed scheduled fires (`last_fire_error`)
|
||||
|
||||
On hosted (managed-cron) deployments, a scheduled fire travels from the platform scheduler through the dashboard to the gateway's internal API server. If that final hand-off fails — the gateway process is down, or its API-server listener never started — the run never begins, so there is no execution record and no `last_status` to inspect. The tell-tale shape: the job works every time you trigger it manually, but never auto-fires.
|
||||
|
||||
Reference in New Issue
Block a user