feat(scheduler): optional rubric acceptance checklist for scheduled tasks (#451)

Scheduled tasks gain an optional `rubric`: an acceptance checklist graded
after each run by deepagents' RubricMiddleware (LLM-as-a-judge on the
auxiliary model) with one revision retry. No `rubric` key = no-op.

- cron/schedule.py: create_schedule/run_now carry the rubric in the run
  input and cron metadata only when non-blank
- middleware/scheduler.py: schedule_task gains `rubric`; list marks graded rows
- commands/implementation/schedule.py: `/schedule add ... --rubric`,
  `/schedule run` forwards the stored rubric, list gets a Rubric column
- subagents/_factory.py: RubricMiddleware mounted last on the scheduler
  graph so a needs_revision jump skips the memory lifecycle until the
  accepted run; grader is read-only (ls + read_file, eviction off),
  bounded by a 12-call budget, and gets an explicit structured-output
  strategy on OpenRouter (Gemini JSON mode, Anthropic tool calling);
  warns at build for anthropic/claude-fable-5.1 via OpenRouter, which
  grades under neither strategy today
- tests: 17 new cases; fix two pre-existing fixture leaks (callable
  backend stub in test_hitl, import-under-patch in test_async_subagent_factory)
This commit is contained in:
Xi Zhang
2026-09-05 14:04:26 +08:00
committed by GitHub
parent 7dbb68d807
commit a45563ea7f
11 changed files with 889 additions and 26 deletions
+131 -1
View File
@@ -93,7 +93,7 @@ async def test_run_with_matching_prefix_fires_matched_prompt():
) as rn,
):
await ScheduleCommand().execute(ctx, ["run", "c-123"])
rn.assert_called_once_with("do the thing")
rn.assert_called_once_with("do the thing", rubric=None)
async def test_run_with_no_match_reports():
@@ -230,3 +230,133 @@ async def test_add_name_sanitized_from_nasty_prompt():
assert "/" not in name
# Only safe chars: lowercase alphanumeric and hyphens
assert re.fullmatch(r"[a-z0-9][a-z0-9\-]*", name), f"Unexpected name: {name!r}"
# ---------------------------------------------------------------------------
# Optional --rubric on /schedule add, forwarded by /schedule run, shown in list
# ---------------------------------------------------------------------------
async def test_add_parses_trailing_rubric_flag():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-9"},
) as mk,
):
await ScheduleCommand().execute(
ctx,
[
"add",
"*/10 * * * *",
"write scheduled/digest.md",
"--rubric",
"- scheduled/digest.md has today's date",
],
)
kw = mk.call_args.kwargs
assert kw["prompt"] == "write scheduled/digest.md"
assert kw["rubric"] == "- scheduled/digest.md has today's date"
async def test_add_without_rubric_passes_none():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-9"},
) as mk,
):
await ScheduleCommand().execute(
ctx, ["add", "*/10 * * * *", "search uk weather"]
)
assert mk.call_args.kwargs["rubric"] is None
async def test_run_forwards_stored_rubric():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
rows = [
{
"cron_id": "c-12345",
"metadata": {"prompt": "do the thing", "rubric": "- out.md exists"},
}
]
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch("EvoScientist.cron.schedule.list_schedules", return_value=rows),
patch(
"EvoScientist.cron.schedule.run_now",
return_value={"run_id": "r-1"},
) as rn,
):
await ScheduleCommand().execute(ctx, ["run", "c-123"])
rn.assert_called_once_with("do the thing", rubric="- out.md exists")
async def test_list_table_marks_graded_rows():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, ui = _ctx()
rows = [
{
"cron_id": "c-1",
"schedule": "0 9 * * *",
"enabled": True,
"next_run_date": "2026-06-25T09:00:00+00:00",
"metadata": {"name": "graded", "rubric": "- out.md exists"},
},
{
"cron_id": "c-2",
"schedule": "0 9 * * *",
"enabled": True,
"next_run_date": "2026-06-25T09:00:00+00:00",
"metadata": {"name": "plain"},
},
]
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch("EvoScientist.cron.schedule.list_schedules", return_value=rows),
):
await ScheduleCommand().execute(ctx, ["list"])
table = ui.mount_renderable.call_args.args[0]
rubric_col = next(c for c in table.columns if c.header == "Rubric")
assert list(rubric_col._cells) == ["yes", ""]
async def test_add_treats_last_rubric_flag_as_the_separator():
"""An unquoted prompt may mention the flag; only the final one splits."""
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-9"},
) as mk,
):
await ScheduleCommand().execute(
ctx,
[
"add",
"*/10 * * * *",
"explain",
"the",
"--rubric",
"flag",
"--rubric",
"- notes.md explains the flag",
],
)
kw = mk.call_args.kwargs
assert kw["prompt"] == "explain the --rubric flag"
assert kw["rubric"] == "- notes.md explains the flag"