8376f56ab4
Adds native sandbox execution runtime, dynamic review middleware, and workspace file handling, with supporting stream events, prompt, and scope registry changes plus architecture docs.
375 lines
12 KiB
Python
375 lines
12 KiB
Python
from __future__ import annotations
|
|
|
|
from types import SimpleNamespace
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
from langchain.agents import create_agent
|
|
from langchain.agents.middleware import HumanInTheLoopMiddleware
|
|
from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel
|
|
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
|
|
from langchain_core.tools import tool
|
|
|
|
import EvoScientist.middleware.dynamic_review as dynamic_review
|
|
from EvoScientist.middleware.dynamic_review import (
|
|
AutoReviewVerificationError,
|
|
DynamicReviewMiddleware,
|
|
)
|
|
|
|
|
|
def _config(run_id: str, mode: str, revision: int = 3) -> dict:
|
|
return {
|
|
"configurable": {
|
|
"ai4sci_run_id": run_id,
|
|
"ai4sci_review_mode": {
|
|
"gateway_url": "http://127.0.0.1:8065",
|
|
"requested_mode": mode,
|
|
"review_mode_revision": revision,
|
|
"run_id": run_id,
|
|
"envelope_digest": "d" * 64,
|
|
"envelope_signature": "s" * 64,
|
|
},
|
|
}
|
|
}
|
|
|
|
|
|
def _middleware() -> DynamicReviewMiddleware:
|
|
return DynamicReviewMiddleware(interrupt_on={"execute": False})
|
|
|
|
|
|
class _ToolCallingFakeModel(FakeMessagesListChatModel):
|
|
def bind_tools(self, _tools, *, tool_choice=None, **_kwargs):
|
|
return self
|
|
|
|
|
|
def test_manual_before_agent_overwrites_prior_auto_state(monkeypatch):
|
|
monkeypatch.setattr(
|
|
dynamic_review, "get_config", lambda: _config("run-manual", "manual")
|
|
)
|
|
middleware = _middleware()
|
|
|
|
update = middleware.before_agent(
|
|
{
|
|
"messages": [],
|
|
"_verified_review_mode": {
|
|
"execution_run_id": "run-old",
|
|
"mode": "auto",
|
|
"revision": 2,
|
|
},
|
|
},
|
|
SimpleNamespace(),
|
|
)
|
|
|
|
assert update == {
|
|
"_verified_review_mode": {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": "run-manual",
|
|
"mode": "manual",
|
|
"revision": 3,
|
|
}
|
|
}
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_auto_before_agent_resolves_once_and_after_model_skips_hitl(monkeypatch):
|
|
monkeypatch.setattr(
|
|
dynamic_review, "get_config", lambda: _config("run-auto", "auto")
|
|
)
|
|
calls = 0
|
|
|
|
async def resolve(run_id, review):
|
|
nonlocal calls
|
|
calls += 1
|
|
assert run_id == "run-auto"
|
|
assert review["requested_mode"] == "auto"
|
|
return {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": run_id,
|
|
"mode": "auto",
|
|
"revision": 3,
|
|
}
|
|
|
|
monkeypatch.setattr(dynamic_review, "_resolve_async", resolve)
|
|
middleware = _middleware()
|
|
update = await middleware.abefore_agent({"messages": []}, SimpleNamespace())
|
|
|
|
with patch.object(HumanInTheLoopMiddleware, "after_model") as parent:
|
|
assert (
|
|
middleware.after_model({"messages": [], **update}, SimpleNamespace())
|
|
is None
|
|
)
|
|
assert calls == 1
|
|
parent.assert_not_called()
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_real_langgraph_auto_mode_executes_tool_without_interrupt(monkeypatch):
|
|
calls = []
|
|
|
|
@tool
|
|
def execute(command: str) -> str:
|
|
"""Execute a test command."""
|
|
calls.append(command)
|
|
return "ok"
|
|
|
|
async def resolve(run_id, _review):
|
|
return {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": run_id,
|
|
"mode": "auto",
|
|
"revision": 3,
|
|
}
|
|
|
|
monkeypatch.setattr(dynamic_review, "_resolve_async", resolve)
|
|
agent = create_agent(
|
|
model=_ToolCallingFakeModel(
|
|
responses=[
|
|
AIMessage(
|
|
content="",
|
|
tool_calls=[
|
|
{
|
|
"name": "execute",
|
|
"args": {"command": "pwd"},
|
|
"id": "call-1",
|
|
"type": "tool_call",
|
|
}
|
|
],
|
|
),
|
|
AIMessage(content="done"),
|
|
]
|
|
),
|
|
tools=[execute],
|
|
middleware=[
|
|
DynamicReviewMiddleware(
|
|
interrupt_on={"execute": {"allowed_decisions": ["approve", "reject"]}}
|
|
)
|
|
],
|
|
)
|
|
|
|
result = await agent.ainvoke(
|
|
{"messages": [HumanMessage(content="run pwd")]},
|
|
config=_config("run-auto", "auto"),
|
|
)
|
|
|
|
assert calls == ["pwd"]
|
|
assert any(
|
|
isinstance(message, ToolMessage) and message.tool_call_id == "call-1"
|
|
for message in result["messages"]
|
|
)
|
|
|
|
|
|
def test_manual_after_model_uses_existing_hitl_even_for_resume_child(monkeypatch):
|
|
monkeypatch.setattr(
|
|
dynamic_review, "get_config", lambda: _config("child-run", "manual")
|
|
)
|
|
middleware = _middleware()
|
|
state = {
|
|
"messages": [],
|
|
"_verified_review_mode": {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": "parent-run",
|
|
"mode": "manual",
|
|
"revision": 2,
|
|
},
|
|
}
|
|
|
|
with patch.object(
|
|
HumanInTheLoopMiddleware, "after_model", return_value={"manual": True}
|
|
) as parent:
|
|
assert middleware.after_model(state, SimpleNamespace()) == {"manual": True}
|
|
parent.assert_called_once()
|
|
|
|
|
|
def test_auto_after_model_requires_current_run_context(monkeypatch):
|
|
monkeypatch.setattr(dynamic_review, "get_config", lambda: {"configurable": {}})
|
|
middleware = _middleware()
|
|
|
|
with pytest.raises(AutoReviewVerificationError, match="REVIEW_MODE_RUN_MISMATCH"):
|
|
middleware.after_model(
|
|
{
|
|
"messages": [],
|
|
"_verified_review_mode": {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": "run-old",
|
|
"mode": "auto",
|
|
"revision": 3,
|
|
},
|
|
},
|
|
SimpleNamespace(),
|
|
)
|
|
|
|
|
|
def test_auto_after_model_resume_reverifies_injected_auto_context(monkeypatch):
|
|
monkeypatch.setattr(
|
|
dynamic_review, "get_config", lambda: _config("child-run", "auto")
|
|
)
|
|
calls = []
|
|
|
|
def resolve(run_id, review):
|
|
calls.append(run_id)
|
|
assert review["requested_mode"] == "auto"
|
|
return {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": run_id,
|
|
"mode": "auto",
|
|
"revision": review["review_mode_revision"],
|
|
}
|
|
|
|
monkeypatch.setattr(dynamic_review, "_resolve_sync", resolve)
|
|
middleware = _middleware()
|
|
|
|
with patch.object(HumanInTheLoopMiddleware, "after_model") as parent:
|
|
assert (
|
|
middleware.after_model(
|
|
{
|
|
"messages": [],
|
|
"_verified_review_mode": {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": "parent-run",
|
|
"mode": "auto",
|
|
"revision": 3,
|
|
},
|
|
},
|
|
SimpleNamespace(),
|
|
)
|
|
is None
|
|
)
|
|
parent.assert_not_called()
|
|
assert calls == ["child-run"]
|
|
|
|
|
|
def test_auto_after_model_resume_downgrades_to_hitl_when_manual(monkeypatch):
|
|
monkeypatch.setattr(
|
|
dynamic_review, "get_config", lambda: _config("child-run", "manual")
|
|
)
|
|
middleware = _middleware()
|
|
|
|
with patch.object(
|
|
HumanInTheLoopMiddleware, "after_model", return_value={"manual": True}
|
|
) as parent:
|
|
assert middleware.after_model(
|
|
{
|
|
"messages": [],
|
|
"_verified_review_mode": {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": "parent-run",
|
|
"mode": "auto",
|
|
"revision": 3,
|
|
},
|
|
},
|
|
SimpleNamespace(),
|
|
) == {"manual": True}
|
|
parent.assert_called_once()
|
|
|
|
|
|
def test_auto_after_model_resume_falls_back_to_hitl_on_verify_failure(monkeypatch):
|
|
monkeypatch.setattr(
|
|
dynamic_review, "get_config", lambda: _config("child-run", "auto")
|
|
)
|
|
|
|
def resolve(_run_id, _review):
|
|
raise AutoReviewVerificationError("AUTO_REVIEW_VERIFICATION_FAILED")
|
|
|
|
monkeypatch.setattr(dynamic_review, "_resolve_sync", resolve)
|
|
middleware = _middleware()
|
|
|
|
with patch.object(
|
|
HumanInTheLoopMiddleware, "after_model", return_value={"manual": True}
|
|
) as parent:
|
|
assert middleware.after_model(
|
|
{
|
|
"messages": [],
|
|
"_verified_review_mode": {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": "parent-run",
|
|
"mode": "auto",
|
|
"revision": 3,
|
|
},
|
|
},
|
|
SimpleNamespace(),
|
|
) == {"manual": True}
|
|
parent.assert_called_once()
|
|
|
|
|
|
def test_auto_after_model_inherits_parent_state_for_resume_child(monkeypatch):
|
|
# A legacy resume child has no ai4sci_review_mode injected by the gateway;
|
|
# the verified auto state is inherited from the parent run and must be
|
|
# reused for the remainder of the turn.
|
|
monkeypatch.setattr(
|
|
dynamic_review,
|
|
"get_config",
|
|
lambda: {"configurable": {"ai4sci_run_id": "child-run"}},
|
|
)
|
|
middleware = _middleware()
|
|
|
|
with patch.object(HumanInTheLoopMiddleware, "after_model") as parent:
|
|
assert (
|
|
middleware.after_model(
|
|
{
|
|
"messages": [],
|
|
"_verified_review_mode": {
|
|
"protocol": "verified-review-mode-state-v1",
|
|
"execution_run_id": "parent-run",
|
|
"mode": "auto",
|
|
"revision": 3,
|
|
},
|
|
},
|
|
SimpleNamespace(),
|
|
)
|
|
is None
|
|
)
|
|
parent.assert_not_called()
|
|
|
|
|
|
def test_auto_response_must_match_signed_request():
|
|
review = _config("run-auto", "auto")["configurable"]["ai4sci_review_mode"]
|
|
with pytest.raises(
|
|
AutoReviewVerificationError, match="AUTO_REVIEW_RESPONSE_INVALID"
|
|
):
|
|
dynamic_review._validated_auto_state(
|
|
"run-auto",
|
|
review,
|
|
{
|
|
"protocol": "resolved-review-mode-v1",
|
|
"run_id": "run-auto",
|
|
"envelope_digest": "d" * 64,
|
|
"mode": "manual",
|
|
"revision": 3,
|
|
},
|
|
)
|
|
|
|
|
|
@pytest.mark.parametrize("auto_approve", [False, True])
|
|
def test_default_web_graph_installs_only_dynamic_hitl(monkeypatch, auto_approve):
|
|
import EvoScientist.EvoScientist as agent_module
|
|
|
|
captured = {}
|
|
|
|
class Agent:
|
|
def with_config(self, _config):
|
|
return self
|
|
|
|
cfg = SimpleNamespace(auto_approve=auto_approve, recursion_limit=50)
|
|
monkeypatch.setattr(agent_module, "_EvoScientist_agent", None)
|
|
monkeypatch.setattr(agent_module, "_ensure_config", lambda: cfg)
|
|
monkeypatch.setattr(agent_module, "_get_default_backend", lambda: object())
|
|
monkeypatch.setattr(agent_module, "_get_default_middleware", lambda: [])
|
|
monkeypatch.setattr(
|
|
agent_module,
|
|
"load_mcp_and_build_kwargs",
|
|
lambda _backend, middleware, **_kwargs: captured.setdefault(
|
|
"kwargs", {"middleware": middleware}
|
|
),
|
|
)
|
|
monkeypatch.setattr(
|
|
agent_module, "_apply_budgeted_skill_context", lambda kwargs, _backend: kwargs
|
|
)
|
|
monkeypatch.setattr("deepagents.create_deep_agent", lambda **_kwargs: Agent())
|
|
monkeypatch.delenv("EVOSCIENTIST_DEPLOY_MODE", raising=False)
|
|
|
|
agent_module._get_default_agent()
|
|
|
|
middleware = captured["kwargs"]["middleware"]
|
|
assert sum(isinstance(item, DynamicReviewMiddleware) for item in middleware) == 1
|
|
assert not any(type(item) is HumanInTheLoopMiddleware for item in middleware)
|