feat(scheduler): optional rubric acceptance checklist for scheduled tasks (#451)

Scheduled tasks gain an optional `rubric`: an acceptance checklist graded
after each run by deepagents' RubricMiddleware (LLM-as-a-judge on the
auxiliary model) with one revision retry. No `rubric` key = no-op.

- cron/schedule.py: create_schedule/run_now carry the rubric in the run
  input and cron metadata only when non-blank
- middleware/scheduler.py: schedule_task gains `rubric`; list marks graded rows
- commands/implementation/schedule.py: `/schedule add ... --rubric`,
  `/schedule run` forwards the stored rubric, list gets a Rubric column
- subagents/_factory.py: RubricMiddleware mounted last on the scheduler
  graph so a needs_revision jump skips the memory lifecycle until the
  accepted run; grader is read-only (ls + read_file, eviction off),
  bounded by a 12-call budget, and gets an explicit structured-output
  strategy on OpenRouter (Gemini JSON mode, Anthropic tool calling);
  warns at build for anthropic/claude-fable-5.1 via OpenRouter, which
  grades under neither strategy today
- tests: 17 new cases; fix two pre-existing fixture leaks (callable
  backend stub in test_hitl, import-under-patch in test_async_subagent_factory)
This commit is contained in:
Xi Zhang
2026-09-05 14:04:26 +08:00
committed by GitHub
parent 7dbb68d807
commit a45563ea7f
11 changed files with 889 additions and 26 deletions
@@ -10,13 +10,21 @@ from ..base import Command, CommandContext, SubCommand
from ..manager import manager
def _clean(text: str) -> str:
"""Trim a shlex-joined argument and drop a stray wrapping quote pair."""
return text.strip().strip('"').strip("'")
class ScheduleCommand(Command):
"""Manage scheduled (cron) tasks."""
name = "/schedule"
description = "Manage scheduled (cron) tasks"
subcommands: ClassVar[list[SubCommand]] = [
SubCommand("add", 'Add: /schedule add <m h dom mon dow> "<prompt>"'),
SubCommand(
"add",
'Add: /schedule add <m h dom mon dow> "<prompt>" [--rubric "<checklist>"]',
),
SubCommand("list", "List scheduled tasks"),
SubCommand("remove", "Remove a schedule by id"),
SubCommand("run", "Run a schedule's prompt once now (test)"),
@@ -78,10 +86,19 @@ class ScheduleCommand(Command):
schedule, prompt_tokens = " ".join(rest[:5]), rest[5:]
else:
ctx.ui.append_system(
'Usage: /schedule add "<m h dom mon dow>" "<prompt>"', style="yellow"
'Usage: /schedule add "<m h dom mon dow>" "<prompt>" '
'[--rubric "<checklist>"]',
style="yellow",
)
return
prompt = " ".join(prompt_tokens).strip().strip('"').strip("'")
# Optional trailing acceptance checklist; everything after --rubric is it.
rubric = None
if "--rubric" in prompt_tokens:
# Last occurrence wins so an unquoted prompt may mention the flag.
split_at = len(prompt_tokens) - 1 - prompt_tokens[::-1].index("--rubric")
rubric = _clean(" ".join(prompt_tokens[split_at + 1 :])) or None
prompt_tokens = prompt_tokens[:split_at]
prompt = _clean(" ".join(prompt_tokens))
if not prompt:
ctx.ui.append_system("A task prompt is required.", style="yellow")
return
@@ -90,7 +107,11 @@ class ScheduleCommand(Command):
name = re.sub(r"[^a-z0-9]+", "-", raw).strip("-")[:32] or "task"
try:
rec = await asyncio.to_thread(
crons.create_schedule, name=name, schedule=schedule, prompt=prompt
crons.create_schedule,
name=name,
schedule=schedule,
prompt=prompt,
rubric=rubric,
)
except Exception as exc:
ctx.ui.append_system(f"Error: {exc}", style="red")
@@ -119,6 +140,7 @@ class ScheduleCommand(Command):
table.add_column("Schedule", style="green")
table.add_column("Enabled", style="yellow")
table.add_column("Next run (UTC)", style="white")
table.add_column("Rubric", style="blue")
for r in rows:
meta = r.get("metadata") or {}
table.add_row(
@@ -127,6 +149,7 @@ class ScheduleCommand(Command):
str(r.get("schedule", "")),
"yes" if r.get("enabled", True) else "no",
str(r.get("next_run_date", "")),
"yes" if meta.get("rubric") else "",
)
ctx.ui.mount_renderable(table)
@@ -191,7 +214,8 @@ class ScheduleCommand(Command):
match = await self._resolve_or_report(ctx, crons, prefix)
if match is None:
return
prompt = (match.get("metadata") or {}).get("prompt", "")
meta = match.get("metadata") or {}
prompt = meta.get("prompt", "")
if not str(prompt).strip():
ctx.ui.append_system(
f"Schedule {prefix} has no stored prompt — cannot run it.",
@@ -199,7 +223,9 @@ class ScheduleCommand(Command):
)
return
try:
rec = await asyncio.to_thread(crons.run_now, prompt)
rec = await asyncio.to_thread(
crons.run_now, prompt, rubric=meta.get("rubric") or None
)
except Exception as exc:
ctx.ui.append_system(f"Error: {exc}", style="red")
return
+46 -12
View File
@@ -12,7 +12,7 @@ multiple clients at one hand-started server they will share the same cron store.
from __future__ import annotations
from typing import TYPE_CHECKING
from typing import TYPE_CHECKING, Any
if TYPE_CHECKING:
from langgraph_sdk.schema import Cron, Run
@@ -28,6 +28,33 @@ SCHEDULER_GRAPH_ID = "scheduler"
SCHEDULED_RUN_KIND = "scheduled_task"
def _normalize_rubric(rubric: str | None) -> str | None:
text = (rubric or "").strip()
return text or None
def _scheduled_input(prompt: str, rubric: str | None) -> dict[str, Any]:
"""Run input for the scheduler graph; ``rubric`` rides along only when set.
The key is read by ``RubricMiddleware`` mounted on the scheduler graph — an
absent key means no grading pass at all, so unset stays byte-identical to
the pre-rubric payload.
"""
payload: dict[str, Any] = messages_input(prompt)
if rubric:
payload["rubric"] = rubric
return payload
def _scheduled_metadata(
*, name: str, prompt: str, rubric: str | None
) -> dict[str, str]:
metadata = {"run_kind": SCHEDULED_RUN_KIND, "name": name, "prompt": prompt}
if rubric:
metadata["rubric"] = rubric
return metadata
def _scheduler_url() -> str:
return configured_langgraph_dev_url()
@@ -48,16 +75,26 @@ def is_available() -> bool:
def create_schedule(
*, name: str, schedule: str, prompt: str, timezone: str | None = None
*,
name: str,
schedule: str,
prompt: str,
timezone: str | None = None,
rubric: str | None = None,
) -> Cron:
"""Create a recurring scheduled task on the scheduler graph."""
"""Create a recurring scheduled task on the scheduler graph.
``rubric`` is an optional acceptance checklist graded after each run; blank
means the run is never graded.
"""
rubric = _normalize_rubric(rubric)
# Crons are stored in the langgraph-dev process's .langgraph_api store, not
# tagged by workspace. Isolation is process-level (see module docstring).
return _client().crons.create(
assistant_id=SCHEDULER_GRAPH_ID,
schedule=schedule,
input=messages_input(prompt),
metadata={"run_kind": SCHEDULED_RUN_KIND, "name": name, "prompt": prompt},
input=_scheduled_input(prompt, rubric),
metadata=_scheduled_metadata(name=name, prompt=prompt, rubric=rubric),
timezone=timezone or _default_timezone(),
)
@@ -87,20 +124,17 @@ def set_enabled(cron_id: str, enabled: bool) -> Cron:
return _client().crons.update(cron_id, enabled=enabled)
def run_now(prompt: str) -> Run:
def run_now(prompt: str, *, rubric: str | None = None) -> Run:
"""Fire a one-off scheduler run immediately (for ``/schedule run``).
Output goes wherever the task's prompt specifies; there is no push notification.
"""
rubric = _normalize_rubric(rubric)
client = _client()
thread = client.threads.create(graph_id=SCHEDULER_GRAPH_ID)
return client.runs.create(
thread_id=str(thread["thread_id"]),
assistant_id=SCHEDULER_GRAPH_ID,
input=messages_input(prompt),
metadata={
"run_kind": SCHEDULED_RUN_KIND,
"name": "manual-run",
"prompt": prompt,
},
input=_scheduled_input(prompt, rubric),
metadata=_scheduled_metadata(name="manual-run", prompt=prompt, rubric=rubric),
)
+18 -3
View File
@@ -48,7 +48,9 @@ manage them.
@tool
def schedule_task(name: str, cron: str, prompt: str, timezone: str = "") -> str:
def schedule_task(
name: str, cron: str, prompt: str, timezone: str = "", rubric: str = ""
) -> str:
"""Create a recurring scheduled task that runs unattended in the background.
Translate the user's natural-language timing into a standard 5-field cron
@@ -60,6 +62,12 @@ def schedule_task(name: str, cron: str, prompt: str, timezone: str = "") -> str:
cron: 5-field cron expression.
prompt: the full instruction the background scheduler runs each time.
timezone: optional IANA tz (e.g. "Europe/London"); empty = host local zone.
rubric: optional acceptance checklist, one "- " bullet per line. A
separate reviewer grades each run against it and the task is
re-run once with the reviewer's feedback when a bullet fails.
Fill it only when the request names checkable outputs (a file
that must exist, sections it must contain, a minimum count);
leave empty otherwise.
"""
from ..cron import schedule as crons
@@ -67,7 +75,11 @@ def schedule_task(name: str, cron: str, prompt: str, timezone: str = "") -> str:
return "Scheduler unavailable: the langgraph dev backend is not running."
try:
rec = crons.create_schedule(
name=name, schedule=cron, prompt=prompt, timezone=timezone or None
name=name,
schedule=cron,
prompt=prompt,
timezone=timezone or None,
rubric=rubric or None,
)
except Exception as e:
return f"Error: {e}"
@@ -93,10 +105,13 @@ def list_scheduled_tasks() -> str:
lines = []
for r in rows:
meta = r.get("metadata") or {}
lines.append(
line = (
f"- {str(r.get('cron_id', ''))[:8]} | {meta.get('name', '')} | "
f"{r.get('schedule', '')} | {'on' if r.get('enabled', True) else 'off'}"
)
if meta.get("rubric"):
line += " | rubric"
lines.append(line)
return "\n".join(lines)
+194 -2
View File
@@ -15,13 +15,199 @@ synchronous counterpart: same workspace files, same ``/skills/`` and
from __future__ import annotations
import logging
import os
from typing import Any
from typing import TYPE_CHECKING, Any
from deepagents.middleware.rubric import (
RUBRIC_GRADER_MESSAGE_SOURCE,
GraderResponse,
RubricMiddleware,
)
from langchain.agents import create_agent
from langchain.agents.middleware.types import AgentMiddleware
from langchain.agents.structured_output import ProviderStrategy, ToolStrategy
from langchain_core.messages import AIMessage
if TYPE_CHECKING:
from deepagents.backends.protocol import BackendProtocol
from deepagents.middleware.rubric import RubricEvaluation
from langchain_core.language_models import BaseChatModel
logger = logging.getLogger(__name__)
# Async research agents (no approval path) keep the backend guard forced on;
# internal graphs (scheduler, evomemory, autoskills) run unguarded.
_GUARDED_ASYNC_SUBAGENTS = frozenset({"writing-agent", "data-analysis-agent"})
# Read-only slice of the filesystem tools handed to the scheduler's grader. A
# rubric names its deliverables, so reading them is enough; ``grep``/``glob``
# invited whole-workspace scans (15s timeouts per call on large workspaces).
_SCHEDULER_GRADER_TOOLS = ("ls", "read_file")
# Structured-output strategy per OpenRouter model family. langchain picks the
# grader's strategy from the model profile plus a model-name regex table, and
# OpenRouter breaks each family the other way round: Gemini tool schemas lose
# the criteria ``oneOf`` (every entry comes back null), Anthropic JSON mode
# returns non-JSON. Verified live 2026-09-04. Passed explicitly because a
# profile pin loses to the name regex (``anthropic/claude-fable-5``).
_OPENROUTER_GRADER_STRATEGY: dict[str, type[ProviderStrategy] | type[ToolStrategy]] = {
"google/": ProviderStrategy,
"anthropic/": ToolStrategy,
}
# Model calls one grader attempt may spend before the run fails closed with
# ``grader_error``. Without it a parse-error retry loop inherits the scheduler
# graph's recursion limit and spins for minutes. Sized for a 3-5 bullet rubric
# over a few files: ``ls`` + one ``read_file`` per file + the verdict call.
_SCHEDULER_GRADER_MAX_CALLS = 12
# OpenRouter ids for which neither strategy yields a verdict (probed 2026-09-04):
# the route rejects forced ``tool_choice`` (reasoning on or off) and JSON mode
# drops required fields. Exact ids, not families: ``anthropic/claude-fable-5``
# and the native ``claude-fable-5-1`` grade fine.
_OPENROUTER_UNGRADABLE_IDS = ("anthropic/claude-fable-5.1",)
def _warn_if_grader_unsupported(model: BaseChatModel) -> None:
from EvoScientist.llm.errors import _provider_from_model
if _provider_from_model(model) != "openrouter":
return
model_id = (getattr(model, "model_name", None) or "").lower()
if model_id in _OPENROUTER_UNGRADABLE_IDS:
logger.warning(
"scheduler rubric: grader model %s via OpenRouter cannot return "
"structured verdicts (this route rejects forced tool_choice and its "
"JSON mode drops required fields); rubric runs will end in "
"grader_error. Use the native anthropic provider for this model, or "
"set auxiliary_model to another model (claude-fable-5, Sonnet, Haiku "
"and Gemini all grade through OpenRouter).",
model_id,
)
def _grader_strategy(model: BaseChatModel) -> ProviderStrategy | ToolStrategy | None:
"""Explicit grader strategy on OpenRouter routes; ``None`` defers to langchain."""
from EvoScientist.llm.errors import _provider_from_model
if _provider_from_model(model) != "openrouter":
return None
model_id = (getattr(model, "model_name", None) or "").lower()
for prefix, strategy in _OPENROUTER_GRADER_STRATEGY.items():
if model_id.startswith(prefix):
return strategy(GraderResponse)
return None
class _SchedulerRubricMiddleware(RubricMiddleware):
"""``RubricMiddleware`` whose grader gets an explicit structured-output strategy.
Mirrors upstream ``_ensure_grader`` except for ``response_format``; a bare
``GraderResponse`` there lets langchain choose the strategy, which is wrong
on OpenRouter (see ``_OPENROUTER_GRADER_STRATEGY``).
"""
def _ensure_grader(self) -> Any:
if self._grader is not None:
return self._grader
from deepagents._models import resolve_model
resolved_model = resolve_model(self._model)
self._resolved_model = resolved_model
self._grader = create_agent(
model=resolved_model,
system_prompt=self._system_prompt,
tools=self._tools,
middleware=self._grader_middleware,
name=RUBRIC_GRADER_MESSAGE_SOURCE,
response_format=_grader_strategy(resolved_model) or GraderResponse,
state_schema=self._grader_state_schema,
context_schema=self._grader_context_schema,
)
return self._grader
class _GraderCallBudget(AgentMiddleware):
"""Fail closed once one grader attempt has made ``max_calls`` model calls.
Counts the ``AIMessage``s already in the request, so the budget is per
grader invocation by construction and no per-run state is needed. Raised
on the first attempt it surfaces as ``grader_error``; raised on the
coverage retry, upstream ``_grade`` swallows it and downgrades the first
(unusable) verdict to ``needs_revision`` instead. Both terminate.
"""
name = "scheduler_rubric_grader_budget"
def __init__(self, *, max_calls: int) -> None:
self.max_calls = max_calls
def _check(self, request: Any) -> None:
spent = sum(isinstance(m, AIMessage) for m in request.messages)
if spent >= self.max_calls:
msg = (
f"scheduler rubric grader exceeded {self.max_calls} model calls "
"without a verdict"
)
raise RuntimeError(msg)
def wrap_model_call(self, request, handler):
self._check(request)
return handler(request)
async def awrap_model_call(self, request, handler):
self._check(request)
return await handler(request)
def _log_rubric_evaluation(evaluation: RubricEvaluation) -> None:
logger.info(
"scheduler rubric iteration %s: %s — %s",
evaluation.get("iteration"),
evaluation.get("result"),
evaluation.get("explanation"),
)
def _scheduler_rubric_middleware(*, model: BaseChatModel, backend: BackendProtocol):
"""Acceptance grading for unattended scheduler runs (no-op without a rubric).
Must be mounted LAST: ``after_agent`` hooks run in reverse list order, so
the grader sees the finished run first and a ``needs_revision`` verdict
jumps back to the model before ``EvoMemoryLifecycleMiddleware`` launches
its memory worker — the worker fires once, on the accepted run. The grader
reads the same backend because the deliverables are files the transcript
alone cannot prove exist.
"""
import warnings
from deepagents import FilesystemMiddleware
from langchain_core._api import LangChainBetaWarning
# Eviction thresholds off: both eviction paths write files through the
# backend, which would let the grader touch the shared workspace.
grader_fs = FilesystemMiddleware(
backend=backend,
tools=list(_SCHEDULER_GRADER_TOOLS),
tool_token_limit_before_evict=None,
human_message_token_limit_before_evict=None,
)
_warn_if_grader_unsupported(model)
with warnings.catch_warnings():
# Beta API; graphs build at langgraph dev import, keep the log clean.
warnings.simplefilter("ignore", LangChainBetaWarning)
return _SchedulerRubricMiddleware(
model=model,
grader_middleware=[
grader_fs,
_GraderCallBudget(max_calls=_SCHEDULER_GRADER_MAX_CALLS),
],
max_iterations=2,
on_evaluation=_log_rubric_evaluation,
)
def build_async_subagent_graph(name: str) -> Any:
"""Build a deployable graph for the ``name`` sub-agent defined in yaml.
@@ -118,13 +304,19 @@ def build_async_subagent_graph(name: str) -> Any:
)
guarded = name in _GUARDED_ASYNC_SUBAGENTS
backend = _get_default_backend(guard_dangerous=guarded, refuse_delete=guarded)
if name == "scheduler":
middleware = [
*middleware,
_scheduler_rubric_middleware(model=model, backend=backend),
]
return create_deep_agent(
name=name,
model=model,
system_prompt=spec.get("system_prompt", ""),
tools=spec.get("tools", []) + agent_mcp_tools,
skills=spec.get("skills"),
backend=_get_default_backend(guard_dangerous=guarded, refuse_delete=guarded),
backend=backend,
middleware=middleware,
subagents=subagents,
).with_config({"recursion_limit": cfg.recursion_limit})
+5
View File
@@ -578,6 +578,9 @@ Automate recurring research tasks with cron-style schedules.
/schedule add "0 9 * * 1-5" "Summarise the latest ML papers from arXiv with the paper-navigator skill, and save the summary to /memories/daily-papers.md"
/schedule add "*/10 * * * *" "Check my running experiment's status and append the result to experiment_log.json"
# Optional acceptance checklist: a separate reviewer grades each run against it
/schedule add "0 8 * * 1-5" "Collect yesterday's arXiv diffusion papers into scheduled/digest.md" --rubric "- scheduled/digest.md is updated with today's date; - every entry has a title, an arXiv link and a one-line summary"
# Manage schedules
/schedule list # list active schedules
/schedule remove <id> # delete a schedule
@@ -590,6 +593,8 @@ Note: `/schedule add` requires a cron expression (5 fields, e.g. `*/10 * * * *`)
Output goes wherever the task's prompt tells it to write — there is no enforced output directory, so make the prompt specific about file locations. Run `/schedule list` to review schedules; the agent is also made aware of the active schedules via a `<scheduled_tasks>` context block, so you can just ask it what's scheduled.
`--rubric "<checklist>"` is optional. When set, a separate reviewer (the auxiliary model, with read-only access to the workspace) grades the finished run against the checklist and re-runs the task once with the reviewer's feedback if a bullet fails; without a rubric no grading happens at all. Name concrete deliverables the reviewer can check — files that must exist, sections they must contain, minimum counts. When you schedule in chat, the agent fills the rubric itself if your request names such outputs.
> **Cost note:** each scheduled run consumes LLM tokens. Delete unused schedules with `/schedule remove` to avoid accumulating charges.
<p align="right"><a href="#top">🔝Back to top</a></p>
+3
View File
@@ -11,6 +11,9 @@ from __future__ import annotations
from unittest.mock import MagicMock, patch
# Imported before any patch is active: the module binds ``get_effective_config``
# at import, and a first import under the patch would freeze the mock in place.
import EvoScientist.EvoScientist # noqa: F401
from EvoScientist.config import MemoryObservationWriter
from EvoScientist.memory import MemorySourceType
+42
View File
@@ -223,3 +223,45 @@ requirements = [
assert manager._kill_owned_stale_process(6174) is False
assert not runtime.pid_file.exists()
assert not runtime.workspace_sidecar.exists()
# ---------------------------------------------------------------------------
# Optional rubric — acceptance criteria graded after each scheduler run
# ---------------------------------------------------------------------------
def test_create_schedule_with_rubric_sends_it_in_input_and_metadata(monkeypatch):
crons, fake = _patch_client(monkeypatch)
rubric = "- scheduled/digest.md exists\n- it contains today's date"
crons.create_schedule(
name="digest",
schedule="0 8 * * 1-5",
prompt="write scheduled/digest.md",
rubric=rubric,
)
kw = fake.crons.create.call_args.kwargs
assert kw["input"] == {
"messages": [{"role": "user", "content": "write scheduled/digest.md"}],
"rubric": rubric,
}
assert kw["metadata"]["rubric"] == rubric
def test_create_schedule_blank_rubric_omits_the_key(monkeypatch):
crons, fake = _patch_client(monkeypatch)
crons.create_schedule(
name="weather", schedule="*/10 * * * *", prompt="search", rubric=" \n"
)
kw = fake.crons.create.call_args.kwargs
assert "rubric" not in kw["input"]
assert "rubric" not in kw["metadata"]
def test_run_now_with_rubric_sends_it_in_input_and_metadata(monkeypatch):
crons, fake = _patch_client(monkeypatch)
fake.threads.create.return_value = {"thread_id": "t-1"}
fake.runs.create.return_value = {"run_id": "r-1"}
crons.run_now("do the thing", rubric="- output.md exists")
run_kw = fake.runs.create.call_args.kwargs
assert run_kw["input"]["rubric"] == "- output.md exists"
assert run_kw["metadata"]["rubric"] == "- output.md exists"
+3 -2
View File
@@ -990,7 +990,7 @@ class TestAsyncSubagentGuard:
@staticmethod
def _factory_kwargs_for(name: str) -> dict:
"""Run the async factory for ``name`` and capture the backend kwargs."""
from unittest.mock import MagicMock, patch
from unittest.mock import MagicMock, NonCallableMagicMock, patch
import EvoScientist.EvoScientist as ev
from EvoScientist.subagents import _factory
@@ -999,7 +999,8 @@ class TestAsyncSubagentGuard:
def _spy_backend(**kwargs):
captured.update(kwargs)
return MagicMock()
# Non-callable: deepagents rejects callable backends as removed factories.
return NonCallableMagicMock()
with (
patch.object(ev, "_get_default_backend", _spy_backend),
+131 -1
View File
@@ -93,7 +93,7 @@ async def test_run_with_matching_prefix_fires_matched_prompt():
) as rn,
):
await ScheduleCommand().execute(ctx, ["run", "c-123"])
rn.assert_called_once_with("do the thing")
rn.assert_called_once_with("do the thing", rubric=None)
async def test_run_with_no_match_reports():
@@ -230,3 +230,133 @@ async def test_add_name_sanitized_from_nasty_prompt():
assert "/" not in name
# Only safe chars: lowercase alphanumeric and hyphens
assert re.fullmatch(r"[a-z0-9][a-z0-9\-]*", name), f"Unexpected name: {name!r}"
# ---------------------------------------------------------------------------
# Optional --rubric on /schedule add, forwarded by /schedule run, shown in list
# ---------------------------------------------------------------------------
async def test_add_parses_trailing_rubric_flag():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-9"},
) as mk,
):
await ScheduleCommand().execute(
ctx,
[
"add",
"*/10 * * * *",
"write scheduled/digest.md",
"--rubric",
"- scheduled/digest.md has today's date",
],
)
kw = mk.call_args.kwargs
assert kw["prompt"] == "write scheduled/digest.md"
assert kw["rubric"] == "- scheduled/digest.md has today's date"
async def test_add_without_rubric_passes_none():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-9"},
) as mk,
):
await ScheduleCommand().execute(
ctx, ["add", "*/10 * * * *", "search uk weather"]
)
assert mk.call_args.kwargs["rubric"] is None
async def test_run_forwards_stored_rubric():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
rows = [
{
"cron_id": "c-12345",
"metadata": {"prompt": "do the thing", "rubric": "- out.md exists"},
}
]
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch("EvoScientist.cron.schedule.list_schedules", return_value=rows),
patch(
"EvoScientist.cron.schedule.run_now",
return_value={"run_id": "r-1"},
) as rn,
):
await ScheduleCommand().execute(ctx, ["run", "c-123"])
rn.assert_called_once_with("do the thing", rubric="- out.md exists")
async def test_list_table_marks_graded_rows():
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, ui = _ctx()
rows = [
{
"cron_id": "c-1",
"schedule": "0 9 * * *",
"enabled": True,
"next_run_date": "2026-06-25T09:00:00+00:00",
"metadata": {"name": "graded", "rubric": "- out.md exists"},
},
{
"cron_id": "c-2",
"schedule": "0 9 * * *",
"enabled": True,
"next_run_date": "2026-06-25T09:00:00+00:00",
"metadata": {"name": "plain"},
},
]
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch("EvoScientist.cron.schedule.list_schedules", return_value=rows),
):
await ScheduleCommand().execute(ctx, ["list"])
table = ui.mount_renderable.call_args.args[0]
rubric_col = next(c for c in table.columns if c.header == "Rubric")
assert list(rubric_col._cells) == ["yes", ""]
async def test_add_treats_last_rubric_flag_as_the_separator():
"""An unquoted prompt may mention the flag; only the final one splits."""
from EvoScientist.commands.implementation.schedule import ScheduleCommand
ctx, _ui = _ctx()
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-9"},
) as mk,
):
await ScheduleCommand().execute(
ctx,
[
"add",
"*/10 * * * *",
"explain",
"the",
"--rubric",
"flag",
"--rubric",
"- notes.md explains the flag",
],
)
kw = mk.call_args.kwargs
assert kw["prompt"] == "explain the --rubric flag"
assert kw["rubric"] == "- notes.md explains the flag"
+345
View File
@@ -0,0 +1,345 @@
"""Scheduler graph mounts ``RubricMiddleware`` last, with a read-only grader.
The middleware must be the final entry so its ``after_agent`` runs first in
the reverse-ordered chain and a ``needs_revision`` verdict jumps back to the
model *before* ``EvoMemoryLifecycleMiddleware`` launches a memory worker.
"""
from __future__ import annotations
import logging
import warnings
from contextlib import ExitStack
from unittest.mock import MagicMock, patch
import pytest
from langchain_core.language_models.chat_models import BaseChatModel
from langchain_core.messages import AIMessage, HumanMessage
from langchain_core.outputs import ChatGeneration, ChatResult
# Imported before any patch is active: the module binds ``get_effective_config``
# at import, and a first import under the patch would freeze the mock in place.
import EvoScientist.EvoScientist # noqa: F401
from EvoScientist.config import MemoryObservationWriter
def _build(name: str, workspace, aux_model=None):
"""Build ``name`` through the real factory with heavy deps mocked.
The backend is a real ``FilesystemBackend`` on ``workspace`` because
``FilesystemMiddleware`` rejects callable stand-ins (a ``MagicMock`` looks
like a removed backend factory).
Returns ``(create_deep_agent kwargs, backend, aux_model, lifecycle_stub)``.
"""
from deepagents.backends import FilesystemBackend
backend = FilesystemBackend(root_dir=workspace)
cfg = MagicMock()
cfg.recursion_limit = 1_000_000
cfg.memory_profile_enabled = True
cfg.memory_observations_enabled = True
cfg.memory_observation_writer = MemoryObservationWriter.ALL
cfg.memory_workers_enabled = True
lifecycle_stub = MagicMock(name="EvoMemoryLifecycleMiddleware")
with ExitStack() as stack:
stack.enter_context(
patch("EvoScientist.config.get_effective_config", return_value=cfg)
)
stack.enter_context(patch("EvoScientist.config.apply_config_to_env"))
stack.enter_context(
patch(
"EvoScientist.utils.load_subagents",
return_value=[
{"name": name, "system_prompt": "", "tools": [], "skills": None}
],
)
)
stack.enter_context(patch("EvoScientist.EvoScientist._ensure_chat_model"))
aux = stack.enter_context(
patch(
"EvoScientist.EvoScientist._ensure_auxiliary_chat_model",
**({"return_value": aux_model} if aux_model is not None else {}),
)
)
stack.enter_context(
patch(
"EvoScientist.EvoScientist._get_default_backend", return_value=backend
)
)
stack.enter_context(
patch(
"EvoScientist.EvoScientist._get_default_middleware",
side_effect=lambda **_: [lifecycle_stub],
)
)
stack.enter_context(
patch("EvoScientist.EvoScientist._load_mcp_tools_cached", return_value={})
)
create = stack.enter_context(patch("deepagents.create_deep_agent"))
create.return_value.with_config.return_value = MagicMock()
from EvoScientist.subagents._factory import build_async_subagent_graph
build_async_subagent_graph(name)
return create.call_args.kwargs, backend, aux.return_value, lifecycle_stub
def test_scheduler_graph_mounts_rubric_middleware_last(tmp_path):
from deepagents import RubricMiddleware
kwargs, _backend, _aux, lifecycle_stub = _build("scheduler", tmp_path)
middleware = kwargs["middleware"]
assert middleware[0] is lifecycle_stub
assert isinstance(middleware[-1], RubricMiddleware)
def test_other_async_graphs_do_not_mount_rubric(tmp_path):
from deepagents import RubricMiddleware
kwargs, _backend, _aux, lifecycle_stub = _build("writing-agent", tmp_path)
assert kwargs["middleware"] == [lifecycle_stub]
assert not any(isinstance(m, RubricMiddleware) for m in kwargs["middleware"])
def test_scheduler_grader_gets_read_only_tools_on_the_agent_backend(tmp_path):
from deepagents import FilesystemMiddleware
kwargs, backend, _aux, _stub = _build("scheduler", tmp_path)
rubric = kwargs["middleware"][-1]
grader_fs = rubric._grader_middleware[0]
assert isinstance(grader_fs, FilesystemMiddleware)
assert [t.name for t in grader_fs.tools] == ["ls", "read_file"]
assert grader_fs.backend is backend
assert kwargs["backend"] is backend
def test_scheduler_rubric_uses_scheduler_model_and_allows_one_retry(tmp_path):
kwargs, _backend, aux, _stub = _build("scheduler", tmp_path)
rubric = kwargs["middleware"][-1]
assert rubric._model is aux
assert rubric.max_iterations == 2
def test_scheduler_graph_build_emits_no_beta_warning(tmp_path):
with warnings.catch_warnings(record=True) as caught:
warnings.simplefilter("always")
_build("scheduler", tmp_path)
assert not [w for w in caught if w.category.__name__ == "LangChainBetaWarning"]
def test_rubric_evaluation_is_logged_at_info(caplog):
from EvoScientist.subagents._factory import _log_rubric_evaluation
with caplog.at_level(logging.INFO, logger="EvoScientist.subagents._factory"):
_log_rubric_evaluation(
{
"grading_run_id": "g-1",
"iteration": 0,
"result": "needs_revision",
"explanation": "digest.md is missing today's date",
"criteria": [],
}
)
assert "needs_revision" in caplog.text
assert "missing today's date" in caplog.text
def test_scheduler_grader_never_evicts_to_the_shared_workspace(tmp_path):
"""Both eviction paths write files through the backend; the grader must
stay read-only even on an oversized rubric or transcript."""
kwargs, _backend, _aux, _stub = _build("scheduler", tmp_path)
grader_fs = kwargs["middleware"][-1]._grader_middleware[0]
assert grader_fs._tool_token_limit_before_evict is None
assert grader_fs._human_message_token_limit_before_evict is None
# ---------------------------------------------------------------------------
# Grader structured-output strategy is explicit per OpenRouter model family
# ---------------------------------------------------------------------------
def _openrouter(model_id: str):
from langchain_openrouter import ChatOpenRouter
return ChatOpenRouter(model=model_id, api_key="test-key")
def test_grader_strategy_is_json_mode_for_gemini_on_openrouter():
"""OpenRouter drops the criteria oneOf from Gemini tool schemas; JSON mode works."""
from langchain.agents.structured_output import ProviderStrategy
from EvoScientist.subagents._factory import _grader_strategy
strategy = _grader_strategy(_openrouter("google/gemini-3.8-flash"))
assert isinstance(strategy, ProviderStrategy)
def test_grader_strategy_is_tool_calling_for_anthropic_on_openrouter():
"""OpenRouter's Anthropic JSON mode returns non-JSON; tool calling works.
``claude-fable-5`` matters: its id matches langchain's name-regex fallback,
which would force JSON mode if we only pinned the profile.
"""
from langchain.agents.structured_output import ToolStrategy
from EvoScientist.subagents._factory import _grader_strategy
for model_id in ("anthropic/claude-fable-5", "anthropic/claude-sonnet-4.6"):
assert isinstance(_grader_strategy(_openrouter(model_id)), ToolStrategy)
def test_grader_strategy_defers_to_langchain_elsewhere():
from langchain_anthropic import ChatAnthropic
from EvoScientist.subagents._factory import _grader_strategy
assert (
_grader_strategy(ChatAnthropic(model="claude-haiku-4-5", api_key="k")) is None
)
assert _grader_strategy(_openrouter("qwen/qwen3.8-flash")) is None
def test_scheduler_grader_is_built_with_the_explicit_strategy(tmp_path):
from deepagents.middleware.rubric import GraderResponse
from langchain.agents import create_agent
from langchain.agents.structured_output import ToolStrategy
kwargs, _backend, aux, _stub = _build(
"scheduler", tmp_path, aux_model=_openrouter("anthropic/claude-fable-5")
)
rubric = kwargs["middleware"][-1]
assert rubric._model is aux # no model copy; the strategy is passed explicitly
with patch(
"EvoScientist.subagents._factory.create_agent", wraps=create_agent
) as spy:
rubric._ensure_grader()
response_format = spy.call_args.kwargs["response_format"]
assert isinstance(response_format, ToolStrategy)
assert response_format.schema is GraderResponse
def test_scheduler_grader_builds_against_current_upstream_attributes(tmp_path):
"""Unpatched build: the private deepagents names we mirror still exist."""
kwargs, _backend, _aux, _stub = _build(
"scheduler", tmp_path, aux_model=_openrouter("google/gemini-3.8-flash")
)
rubric = kwargs["middleware"][-1]
grader = rubric._ensure_grader()
assert grader is rubric._ensure_grader() # memoised like upstream
# ---------------------------------------------------------------------------
# Grader call budget: a parse-error ping-pong must fail closed, not spin
# ---------------------------------------------------------------------------
class _BrokenGrader(BaseChatModel):
"""Always answers with a GraderResponse whose criteria are null."""
calls: int = 0
@property
def _llm_type(self) -> str:
return "broken-grader"
def bind_tools(self, tools, **kwargs):
return self
def _generate(self, messages, stop=None, run_manager=None, **kwargs):
self.calls += 1
msg = AIMessage(
content="",
tool_calls=[
{
"name": "GraderResponse",
"args": {
"result": "satisfied",
"explanation": "x",
"criteria": [None],
},
"id": f"call-{self.calls}",
"type": "tool_call",
}
],
)
return ChatResult(generations=[ChatGeneration(message=msg)])
def test_grader_call_budget_stops_a_parse_error_loop():
from deepagents.middleware.rubric import GraderResponse
from langchain.agents import create_agent
from langchain.agents.structured_output import ToolStrategy
from EvoScientist.subagents._factory import _GraderCallBudget
fake = _BrokenGrader()
grader = create_agent(
model=fake,
middleware=[_GraderCallBudget(max_calls=3)],
response_format=ToolStrategy(GraderResponse),
)
with pytest.raises(RuntimeError, match="rubric grader"):
grader.invoke(
{"messages": [HumanMessage("grade this")]},
config={"recursion_limit": 60},
)
assert fake.calls == 3
def test_scheduler_grader_carries_a_call_budget(tmp_path):
from EvoScientist.subagents._factory import _GraderCallBudget
kwargs, _backend, _aux, _stub = _build("scheduler", tmp_path)
grader_mw = kwargs["middleware"][-1]._grader_middleware
assert [type(m).__name__ for m in grader_mw] == [
"FilesystemMiddleware",
"_GraderCallBudget",
]
assert isinstance(grader_mw[1], _GraderCallBudget)
assert grader_mw[1].max_calls == 12
async def test_grader_call_budget_also_guards_the_async_path():
"""langgraph dev grades through ``aafter_agent`` → ``ainvoke``."""
from deepagents.middleware.rubric import GraderResponse
from langchain.agents import create_agent
from langchain.agents.structured_output import ToolStrategy
from EvoScientist.subagents._factory import _GraderCallBudget
fake = _BrokenGrader()
grader = create_agent(
model=fake,
middleware=[_GraderCallBudget(max_calls=2)],
response_format=ToolStrategy(GraderResponse),
)
with pytest.raises(RuntimeError, match="rubric grader"):
await grader.ainvoke(
{"messages": [HumanMessage("grade this")]},
config={"recursion_limit": 60},
)
assert fake.calls == 2
def test_factory_warns_when_openrouter_fable_cannot_grade(tmp_path, caplog):
"""Fable/Mythos via OpenRouter rejects forced tool_choice and returns JSON
missing required fields, so no grader strategy works; say so at build."""
with caplog.at_level(logging.WARNING, logger="EvoScientist.subagents._factory"):
_build(
"scheduler", tmp_path, aux_model=_openrouter("anthropic/claude-fable-5.1")
)
assert "claude-fable-5.1" in caplog.text
assert "auxiliary_model" in caplog.text
def test_factory_stays_quiet_for_supported_openrouter_graders(tmp_path, caplog):
"""Fable 5 (not 5.1) grades fine through OpenRouter, probed 2026-09-04."""
with caplog.at_level(logging.WARNING, logger="EvoScientist.subagents._factory"):
_build(
"scheduler", tmp_path, aux_model=_openrouter("anthropic/claude-sonnet-5")
)
_build("scheduler", tmp_path, aux_model=_openrouter("anthropic/claude-fable-5"))
_build("scheduler", tmp_path, aux_model=_openrouter("google/gemini-3.8-flash"))
assert "rubric" not in caplog.text.lower()
+70
View File
@@ -122,3 +122,73 @@ def test_cancel_empty_cron_id_refuses_without_deleting():
out = cancel_scheduled_task.invoke({"cron_id": " "})
mk.assert_not_called()
assert "Provide" in out
# ---------------------------------------------------------------------------
# Optional rubric on schedule_task / list_scheduled_tasks
# ---------------------------------------------------------------------------
def test_schedule_task_forwards_rubric():
from EvoScientist.middleware.scheduler import schedule_task
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-8"},
) as mk,
):
schedule_task.invoke(
{
"name": "digest",
"cron": "0 8 * * 1-5",
"prompt": "write scheduled/digest.md",
"timezone": "",
"rubric": "- scheduled/digest.md has today's date",
}
)
assert mk.call_args.kwargs["rubric"] == "- scheduled/digest.md has today's date"
def test_schedule_task_without_rubric_forwards_none():
from EvoScientist.middleware.scheduler import schedule_task
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch(
"EvoScientist.cron.schedule.create_schedule",
return_value={"cron_id": "c-8"},
) as mk,
):
schedule_task.invoke(
{"name": "ping", "cron": "0 * * * *", "prompt": "ping", "timezone": ""}
)
assert mk.call_args.kwargs["rubric"] is None
def test_list_scheduled_tasks_marks_graded_rows():
from EvoScientist.middleware.scheduler import list_scheduled_tasks
rows = [
{
"cron_id": "c-1-xyz",
"schedule": "0 9 * * *",
"enabled": True,
"metadata": {"name": "graded", "rubric": "- out.md exists"},
},
{
"cron_id": "c-2-xyz",
"schedule": "0 9 * * *",
"enabled": True,
"metadata": {"name": "plain"},
},
]
with (
patch("EvoScientist.cron.schedule.is_available", return_value=True),
patch("EvoScientist.cron.schedule.list_schedules", return_value=rows),
):
out = list_scheduled_tasks.invoke({})
graded, plain = out.splitlines()
assert "rubric" in graded
assert "rubric" not in plain