feat(scheduler): optional rubric acceptance checklist for scheduled tasks (#451)

Scheduled tasks gain an optional `rubric`: an acceptance checklist graded
after each run by deepagents' RubricMiddleware (LLM-as-a-judge on the
auxiliary model) with one revision retry. No `rubric` key = no-op.

- cron/schedule.py: create_schedule/run_now carry the rubric in the run
  input and cron metadata only when non-blank
- middleware/scheduler.py: schedule_task gains `rubric`; list marks graded rows
- commands/implementation/schedule.py: `/schedule add ... --rubric`,
  `/schedule run` forwards the stored rubric, list gets a Rubric column
- subagents/_factory.py: RubricMiddleware mounted last on the scheduler
  graph so a needs_revision jump skips the memory lifecycle until the
  accepted run; grader is read-only (ls + read_file, eviction off),
  bounded by a 12-call budget, and gets an explicit structured-output
  strategy on OpenRouter (Gemini JSON mode, Anthropic tool calling);
  warns at build for anthropic/claude-fable-5.1 via OpenRouter, which
  grades under neither strategy today
- tests: 17 new cases; fix two pre-existing fixture leaks (callable
  backend stub in test_hitl, import-under-patch in test_async_subagent_factory)
This commit is contained in:
Xi Zhang
2026-09-05 14:04:26 +08:00
committed by GitHub
parent 7dbb68d807
commit a45563ea7f
11 changed files with 889 additions and 26 deletions
+42
View File
@@ -223,3 +223,45 @@ requirements = [
assert manager._kill_owned_stale_process(6174) is False
assert not runtime.pid_file.exists()
assert not runtime.workspace_sidecar.exists()
# ---------------------------------------------------------------------------
# Optional rubric — acceptance criteria graded after each scheduler run
# ---------------------------------------------------------------------------
def test_create_schedule_with_rubric_sends_it_in_input_and_metadata(monkeypatch):
crons, fake = _patch_client(monkeypatch)
rubric = "- scheduled/digest.md exists\n- it contains today's date"
crons.create_schedule(
name="digest",
schedule="0 8 * * 1-5",
prompt="write scheduled/digest.md",
rubric=rubric,
)
kw = fake.crons.create.call_args.kwargs
assert kw["input"] == {
"messages": [{"role": "user", "content": "write scheduled/digest.md"}],
"rubric": rubric,
}
assert kw["metadata"]["rubric"] == rubric
def test_create_schedule_blank_rubric_omits_the_key(monkeypatch):
crons, fake = _patch_client(monkeypatch)
crons.create_schedule(
name="weather", schedule="*/10 * * * *", prompt="search", rubric=" \n"
)
kw = fake.crons.create.call_args.kwargs
assert "rubric" not in kw["input"]
assert "rubric" not in kw["metadata"]
def test_run_now_with_rubric_sends_it_in_input_and_metadata(monkeypatch):
crons, fake = _patch_client(monkeypatch)
fake.threads.create.return_value = {"thread_id": "t-1"}
fake.runs.create.return_value = {"run_id": "r-1"}
crons.run_now("do the thing", rubric="- output.md exists")
run_kw = fake.runs.create.call_args.kwargs
assert run_kw["input"]["rubric"] == "- output.md exists"
assert run_kw["metadata"]["rubric"] == "- output.md exists"