Files
EvoScientist-Multi/tests/test_dynamic_review_middleware.py
T
m4 8376f56ab4 feat: native sandbox execution, dynamic review middleware, and workspace files
Adds native sandbox execution runtime, dynamic review middleware, and
workspace file handling, with supporting stream events, prompt, and scope
registry changes plus architecture docs.
2026-08-19 20:00:09 +08:00

375 lines
12 KiB
Python

from __future__ import annotations
from types import SimpleNamespace
from unittest.mock import patch
import pytest
from langchain.agents import create_agent
from langchain.agents.middleware import HumanInTheLoopMiddleware
from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
from langchain_core.tools import tool
import EvoScientist.middleware.dynamic_review as dynamic_review
from EvoScientist.middleware.dynamic_review import (
AutoReviewVerificationError,
DynamicReviewMiddleware,
)
def _config(run_id: str, mode: str, revision: int = 3) -> dict:
return {
"configurable": {
"ai4sci_run_id": run_id,
"ai4sci_review_mode": {
"gateway_url": "http://127.0.0.1:8065",
"requested_mode": mode,
"review_mode_revision": revision,
"run_id": run_id,
"envelope_digest": "d" * 64,
"envelope_signature": "s" * 64,
},
}
}
def _middleware() -> DynamicReviewMiddleware:
return DynamicReviewMiddleware(interrupt_on={"execute": False})
class _ToolCallingFakeModel(FakeMessagesListChatModel):
def bind_tools(self, _tools, *, tool_choice=None, **_kwargs):
return self
def test_manual_before_agent_overwrites_prior_auto_state(monkeypatch):
monkeypatch.setattr(
dynamic_review, "get_config", lambda: _config("run-manual", "manual")
)
middleware = _middleware()
update = middleware.before_agent(
{
"messages": [],
"_verified_review_mode": {
"execution_run_id": "run-old",
"mode": "auto",
"revision": 2,
},
},
SimpleNamespace(),
)
assert update == {
"_verified_review_mode": {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": "run-manual",
"mode": "manual",
"revision": 3,
}
}
@pytest.mark.asyncio
async def test_auto_before_agent_resolves_once_and_after_model_skips_hitl(monkeypatch):
monkeypatch.setattr(
dynamic_review, "get_config", lambda: _config("run-auto", "auto")
)
calls = 0
async def resolve(run_id, review):
nonlocal calls
calls += 1
assert run_id == "run-auto"
assert review["requested_mode"] == "auto"
return {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": run_id,
"mode": "auto",
"revision": 3,
}
monkeypatch.setattr(dynamic_review, "_resolve_async", resolve)
middleware = _middleware()
update = await middleware.abefore_agent({"messages": []}, SimpleNamespace())
with patch.object(HumanInTheLoopMiddleware, "after_model") as parent:
assert (
middleware.after_model({"messages": [], **update}, SimpleNamespace())
is None
)
assert calls == 1
parent.assert_not_called()
@pytest.mark.asyncio
async def test_real_langgraph_auto_mode_executes_tool_without_interrupt(monkeypatch):
calls = []
@tool
def execute(command: str) -> str:
"""Execute a test command."""
calls.append(command)
return "ok"
async def resolve(run_id, _review):
return {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": run_id,
"mode": "auto",
"revision": 3,
}
monkeypatch.setattr(dynamic_review, "_resolve_async", resolve)
agent = create_agent(
model=_ToolCallingFakeModel(
responses=[
AIMessage(
content="",
tool_calls=[
{
"name": "execute",
"args": {"command": "pwd"},
"id": "call-1",
"type": "tool_call",
}
],
),
AIMessage(content="done"),
]
),
tools=[execute],
middleware=[
DynamicReviewMiddleware(
interrupt_on={"execute": {"allowed_decisions": ["approve", "reject"]}}
)
],
)
result = await agent.ainvoke(
{"messages": [HumanMessage(content="run pwd")]},
config=_config("run-auto", "auto"),
)
assert calls == ["pwd"]
assert any(
isinstance(message, ToolMessage) and message.tool_call_id == "call-1"
for message in result["messages"]
)
def test_manual_after_model_uses_existing_hitl_even_for_resume_child(monkeypatch):
monkeypatch.setattr(
dynamic_review, "get_config", lambda: _config("child-run", "manual")
)
middleware = _middleware()
state = {
"messages": [],
"_verified_review_mode": {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": "parent-run",
"mode": "manual",
"revision": 2,
},
}
with patch.object(
HumanInTheLoopMiddleware, "after_model", return_value={"manual": True}
) as parent:
assert middleware.after_model(state, SimpleNamespace()) == {"manual": True}
parent.assert_called_once()
def test_auto_after_model_requires_current_run_context(monkeypatch):
monkeypatch.setattr(dynamic_review, "get_config", lambda: {"configurable": {}})
middleware = _middleware()
with pytest.raises(AutoReviewVerificationError, match="REVIEW_MODE_RUN_MISMATCH"):
middleware.after_model(
{
"messages": [],
"_verified_review_mode": {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": "run-old",
"mode": "auto",
"revision": 3,
},
},
SimpleNamespace(),
)
def test_auto_after_model_resume_reverifies_injected_auto_context(monkeypatch):
monkeypatch.setattr(
dynamic_review, "get_config", lambda: _config("child-run", "auto")
)
calls = []
def resolve(run_id, review):
calls.append(run_id)
assert review["requested_mode"] == "auto"
return {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": run_id,
"mode": "auto",
"revision": review["review_mode_revision"],
}
monkeypatch.setattr(dynamic_review, "_resolve_sync", resolve)
middleware = _middleware()
with patch.object(HumanInTheLoopMiddleware, "after_model") as parent:
assert (
middleware.after_model(
{
"messages": [],
"_verified_review_mode": {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": "parent-run",
"mode": "auto",
"revision": 3,
},
},
SimpleNamespace(),
)
is None
)
parent.assert_not_called()
assert calls == ["child-run"]
def test_auto_after_model_resume_downgrades_to_hitl_when_manual(monkeypatch):
monkeypatch.setattr(
dynamic_review, "get_config", lambda: _config("child-run", "manual")
)
middleware = _middleware()
with patch.object(
HumanInTheLoopMiddleware, "after_model", return_value={"manual": True}
) as parent:
assert middleware.after_model(
{
"messages": [],
"_verified_review_mode": {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": "parent-run",
"mode": "auto",
"revision": 3,
},
},
SimpleNamespace(),
) == {"manual": True}
parent.assert_called_once()
def test_auto_after_model_resume_falls_back_to_hitl_on_verify_failure(monkeypatch):
monkeypatch.setattr(
dynamic_review, "get_config", lambda: _config("child-run", "auto")
)
def resolve(_run_id, _review):
raise AutoReviewVerificationError("AUTO_REVIEW_VERIFICATION_FAILED")
monkeypatch.setattr(dynamic_review, "_resolve_sync", resolve)
middleware = _middleware()
with patch.object(
HumanInTheLoopMiddleware, "after_model", return_value={"manual": True}
) as parent:
assert middleware.after_model(
{
"messages": [],
"_verified_review_mode": {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": "parent-run",
"mode": "auto",
"revision": 3,
},
},
SimpleNamespace(),
) == {"manual": True}
parent.assert_called_once()
def test_auto_after_model_inherits_parent_state_for_resume_child(monkeypatch):
# A legacy resume child has no ai4sci_review_mode injected by the gateway;
# the verified auto state is inherited from the parent run and must be
# reused for the remainder of the turn.
monkeypatch.setattr(
dynamic_review,
"get_config",
lambda: {"configurable": {"ai4sci_run_id": "child-run"}},
)
middleware = _middleware()
with patch.object(HumanInTheLoopMiddleware, "after_model") as parent:
assert (
middleware.after_model(
{
"messages": [],
"_verified_review_mode": {
"protocol": "verified-review-mode-state-v1",
"execution_run_id": "parent-run",
"mode": "auto",
"revision": 3,
},
},
SimpleNamespace(),
)
is None
)
parent.assert_not_called()
def test_auto_response_must_match_signed_request():
review = _config("run-auto", "auto")["configurable"]["ai4sci_review_mode"]
with pytest.raises(
AutoReviewVerificationError, match="AUTO_REVIEW_RESPONSE_INVALID"
):
dynamic_review._validated_auto_state(
"run-auto",
review,
{
"protocol": "resolved-review-mode-v1",
"run_id": "run-auto",
"envelope_digest": "d" * 64,
"mode": "manual",
"revision": 3,
},
)
@pytest.mark.parametrize("auto_approve", [False, True])
def test_default_web_graph_installs_only_dynamic_hitl(monkeypatch, auto_approve):
import EvoScientist.EvoScientist as agent_module
captured = {}
class Agent:
def with_config(self, _config):
return self
cfg = SimpleNamespace(auto_approve=auto_approve, recursion_limit=50)
monkeypatch.setattr(agent_module, "_EvoScientist_agent", None)
monkeypatch.setattr(agent_module, "_ensure_config", lambda: cfg)
monkeypatch.setattr(agent_module, "_get_default_backend", lambda: object())
monkeypatch.setattr(agent_module, "_get_default_middleware", lambda: [])
monkeypatch.setattr(
agent_module,
"load_mcp_and_build_kwargs",
lambda _backend, middleware, **_kwargs: captured.setdefault(
"kwargs", {"middleware": middleware}
),
)
monkeypatch.setattr(
agent_module, "_apply_budgeted_skill_context", lambda kwargs, _backend: kwargs
)
monkeypatch.setattr("deepagents.create_deep_agent", lambda **_kwargs: Agent())
monkeypatch.delenv("EVOSCIENTIST_DEPLOY_MODE", raising=False)
agent_module._get_default_agent()
middleware = captured["kwargs"]["middleware"]
assert sum(isinstance(item, DynamicReviewMiddleware) for item in middleware) == 1
assert not any(type(item) is HumanInTheLoopMiddleware for item in middleware)