feat(scheduler): optional rubric acceptance checklist for scheduled tasks (#451)
Scheduled tasks gain an optional `rubric`: an acceptance checklist graded after each run by deepagents' RubricMiddleware (LLM-as-a-judge on the auxiliary model) with one revision retry. No `rubric` key = no-op. - cron/schedule.py: create_schedule/run_now carry the rubric in the run input and cron metadata only when non-blank - middleware/scheduler.py: schedule_task gains `rubric`; list marks graded rows - commands/implementation/schedule.py: `/schedule add ... --rubric`, `/schedule run` forwards the stored rubric, list gets a Rubric column - subagents/_factory.py: RubricMiddleware mounted last on the scheduler graph so a needs_revision jump skips the memory lifecycle until the accepted run; grader is read-only (ls + read_file, eviction off), bounded by a 12-call budget, and gets an explicit structured-output strategy on OpenRouter (Gemini JSON mode, Anthropic tool calling); warns at build for anthropic/claude-fable-5.1 via OpenRouter, which grades under neither strategy today - tests: 17 new cases; fix two pre-existing fixture leaks (callable backend stub in test_hitl, import-under-patch in test_async_subagent_factory)
This commit is contained in:
@@ -223,3 +223,45 @@ requirements = [
|
||||
assert manager._kill_owned_stale_process(6174) is False
|
||||
assert not runtime.pid_file.exists()
|
||||
assert not runtime.workspace_sidecar.exists()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Optional rubric — acceptance criteria graded after each scheduler run
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def test_create_schedule_with_rubric_sends_it_in_input_and_metadata(monkeypatch):
|
||||
crons, fake = _patch_client(monkeypatch)
|
||||
rubric = "- scheduled/digest.md exists\n- it contains today's date"
|
||||
crons.create_schedule(
|
||||
name="digest",
|
||||
schedule="0 8 * * 1-5",
|
||||
prompt="write scheduled/digest.md",
|
||||
rubric=rubric,
|
||||
)
|
||||
kw = fake.crons.create.call_args.kwargs
|
||||
assert kw["input"] == {
|
||||
"messages": [{"role": "user", "content": "write scheduled/digest.md"}],
|
||||
"rubric": rubric,
|
||||
}
|
||||
assert kw["metadata"]["rubric"] == rubric
|
||||
|
||||
|
||||
def test_create_schedule_blank_rubric_omits_the_key(monkeypatch):
|
||||
crons, fake = _patch_client(monkeypatch)
|
||||
crons.create_schedule(
|
||||
name="weather", schedule="*/10 * * * *", prompt="search", rubric=" \n"
|
||||
)
|
||||
kw = fake.crons.create.call_args.kwargs
|
||||
assert "rubric" not in kw["input"]
|
||||
assert "rubric" not in kw["metadata"]
|
||||
|
||||
|
||||
def test_run_now_with_rubric_sends_it_in_input_and_metadata(monkeypatch):
|
||||
crons, fake = _patch_client(monkeypatch)
|
||||
fake.threads.create.return_value = {"thread_id": "t-1"}
|
||||
fake.runs.create.return_value = {"run_id": "r-1"}
|
||||
crons.run_now("do the thing", rubric="- output.md exists")
|
||||
run_kw = fake.runs.create.call_args.kwargs
|
||||
assert run_kw["input"]["rubric"] == "- output.md exists"
|
||||
assert run_kw["metadata"]["rubric"] == "- output.md exists"
|
||||
|
||||
Reference in New Issue
Block a user