test: cover stop contract, execution adapters, checkpointer race and runtime identity
Docker / build (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Build / build (push) Has been cancelled
Docker / build (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Build / build (push) Has been cancelled
This commit is contained in:
@@ -0,0 +1,325 @@
|
||||
"""B03 auxiliary Graph/SDK contracts; isolated, budgeted nodes only."""
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
if __name__ != "__main__":
|
||||
from tests.test_execution_adapter_stop_contract import (
|
||||
LoopbackSSE, build_run, isolated_network,
|
||||
)
|
||||
else:
|
||||
LoopbackSSE = object
|
||||
|
||||
PLANNED_NODES = (
|
||||
"test_selector_title_graph_stream_and_account",
|
||||
"test_summarizer_graph_stream_and_account",
|
||||
"test_summary_main_profile_thresholds",
|
||||
"test_selector_schema_input_bound",
|
||||
)
|
||||
|
||||
|
||||
def test_summary_main_profile_thresholds():
|
||||
from importlib import import_module
|
||||
from langchain_core.language_models.fake_chat_models import FakeListChatModel
|
||||
from deepagents.backends import CompositeBackend, StateBackend
|
||||
module = import_module("EvoScientist.EvoScientist")
|
||||
main = FakeListChatModel(responses=["main"], profile={"max_input_tokens": 10000})
|
||||
for profile in ({"max_input_tokens": 100000}, None):
|
||||
summary = FakeListChatModel(responses=["summary"], profile=profile)
|
||||
middleware = module._create_run_summarization_middleware(
|
||||
main, CompositeBackend(default=StateBackend, routes={}, artifacts_root="/workspace"), summary)
|
||||
assert middleware._lc_helper.trigger == ("tokens", 8500)
|
||||
assert middleware._lc_helper.keep == ("tokens", 1000)
|
||||
assert middleware._lc_helper.trim_tokens_to_summarize is None
|
||||
assert middleware._get_profile_limits() == (100000 if profile else None)
|
||||
assert middleware._truncate_args_trigger == ("tokens", 8500)
|
||||
assert middleware._truncate_args_keep == ("tokens", 1000)
|
||||
print("main thresholds=8500/1000; summarizer profile independent")
|
||||
|
||||
|
||||
def test_selector_schema_input_bound(monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
from langchain_core.messages import HumanMessage
|
||||
from pydantic import BaseModel, Field
|
||||
from EvoScientist.llm.runtime import _RuntimeAttemptCallback, _provider_input_token_bound, _callback_messages_payload
|
||||
class Selection(BaseModel):
|
||||
tools: list[str] = Field(description="candidate tool description " * 1000)
|
||||
messages = [[HumanMessage(content="select a tool")]]
|
||||
minimum = _provider_input_token_bound({"messages": _callback_messages_payload(messages),
|
||||
"response_format": Selection.model_json_schema()}).total_tokens
|
||||
class BoundaryReached(Exception):
|
||||
pass
|
||||
async def begin(**kwargs):
|
||||
assert kwargs["provider_input_bound_tokens"] >= minimum
|
||||
raise BoundaryReached
|
||||
callback = _RuntimeAttemptCallback(SimpleNamespace(_begin_callback_attempt=begin))
|
||||
monkeypatch.setattr(callback, "_route_for", lambda metadata: ("tool_selector", None))
|
||||
async def scenario():
|
||||
import pytest
|
||||
with pytest.raises(BoundaryReached):
|
||||
await callback.on_chat_model_start({}, messages, run_id="schema-bound",
|
||||
invocation_params={"response_format": Selection})
|
||||
asyncio.run(scenario())
|
||||
print(f"selector response_format schema included in input bound: >= {minimum}")
|
||||
|
||||
|
||||
class CompletingSSE(LoopbackSSE):
|
||||
async def handle(self, reader, writer):
|
||||
task = asyncio.current_task()
|
||||
self.tasks.add(task)
|
||||
self.writers.add(writer)
|
||||
try:
|
||||
header = await asyncio.wait_for(reader.readuntil(b"\r\n\r\n"), 5)
|
||||
headers = dict(line.split(b":", 1) for line in header.split(b"\r\n")[1:] if b":" in line)
|
||||
length = int(next(v for k, v in headers.items() if k.lower() == b"content-length"))
|
||||
body = json.loads(await reader.readexactly(length))
|
||||
self.requests.append(body)
|
||||
assert header.startswith(b"POST /v1/chat/completions ")
|
||||
assert body.get("stream") is True
|
||||
assert body.get("stream_options", {}).get("include_usage") is True
|
||||
tools = body.get("tools", [])
|
||||
selection = next((t for t in tools if t.get("function", {}).get("name") == "ToolSelectionResponse"), None)
|
||||
is_selector = selection is not None or bool(body.get("response_format"))
|
||||
text = '{"tools":["read_file"]}' if is_selector else "B03 streamed completion"
|
||||
writer.write(b"HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nConnection: close\r\n\r\n")
|
||||
for index, part in enumerate((text[:8], text[8:])):
|
||||
delta = {"content": part}
|
||||
if selection:
|
||||
delta = {"tool_calls": [{"index": 0, "function": {"arguments": part}}]}
|
||||
if index == 0:
|
||||
delta["tool_calls"][0].update(id="select-b03", type="function")
|
||||
delta["tool_calls"][0]["function"]["name"] = "ToolSelectionResponse"
|
||||
if index == 0:
|
||||
delta["role"] = "assistant"
|
||||
chunk = {"id": "b03-local", "object": "chat.completion.chunk", "created": 1,
|
||||
"model": body["model"], "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}
|
||||
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\n")
|
||||
await writer.drain()
|
||||
await asyncio.sleep(.02)
|
||||
chunk["choices"] = [{"index": 0, "delta": {}, "finish_reason": "tool_calls" if selection else "stop"}]
|
||||
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\n")
|
||||
chunk["choices"] = []
|
||||
chunk["usage"] = {"prompt_tokens": 13, "completion_tokens": 7, "total_tokens": 20,
|
||||
"prompt_tokens_details": {"cached_tokens": 3}}
|
||||
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\ndata: [DONE]\n\n")
|
||||
await writer.drain()
|
||||
except Exception as exc:
|
||||
self.errors.append(repr(exc))
|
||||
finally:
|
||||
writer.close()
|
||||
await writer.wait_closed()
|
||||
self.writers.discard(writer)
|
||||
self.tasks.discard(task)
|
||||
|
||||
|
||||
def configure_graph(monkeypatch, *, summarize=False):
|
||||
from dataclasses import replace
|
||||
import EvoScientist.web_runtime as web
|
||||
import tests.test_web_model_runtime as helpers
|
||||
from langchain_core.messages import HumanMessage, AIMessage
|
||||
original_factory = web.create_web_agent
|
||||
original_preparation = helpers._preparation
|
||||
|
||||
def factory(**kwargs):
|
||||
kwargs["host"] = replace(kwargs["host"], tool_selector_threshold=10000 if summarize else 0)
|
||||
if summarize:
|
||||
kwargs["model_set"].main_agent.profile = {"max_input_tokens": 10000}
|
||||
kwargs["model_set"].deepagents_summarizer.profile = {"max_input_tokens": 100000}
|
||||
graph = original_factory(**kwargs)
|
||||
if summarize:
|
||||
graph.update_state({"configurable": {"thread_id": "b02:isolated:thread"}}, {
|
||||
"messages": [HumanMessage(content="Earlier request " * 800),
|
||||
AIMessage(content="Earlier answer " * 800),
|
||||
HumanMessage(content="Another request " * 800),
|
||||
AIMessage(content="Another answer " * 800)]})
|
||||
return graph
|
||||
|
||||
def preparation(authority, agent_input, **kwargs):
|
||||
kwargs["title_policy"] = "disabled" if summarize else "best_effort"
|
||||
return original_preparation(authority, agent_input, **kwargs)
|
||||
|
||||
monkeypatch.setattr(web, "create_web_agent", factory)
|
||||
monkeypatch.setattr(helpers, "_preparation", preparation)
|
||||
|
||||
|
||||
|
||||
def assert_accounting(server, sink, expected):
|
||||
attempts = [e.payload for e in sink.events if e.kind == "model_attempt"]
|
||||
completed = [e for e in attempts if e["outcome"] == "succeeded"]
|
||||
print(json.dumps({"evidence": "b03_auxiliary", "requests": [
|
||||
{"stream": r.get("stream"), "stream_options": r.get("stream_options"),
|
||||
"tools": [t.get("function", {}).get("name") for t in r.get("tools", [])],
|
||||
"response_format": r.get("response_format"),
|
||||
"message_prefix": str(r.get("messages", []))[:220]} for r in server.requests],
|
||||
"attempts": attempts}, default=str), flush=True)
|
||||
assert not server.errors, server.errors
|
||||
assert [e["purpose"] for e in completed] == expected
|
||||
assert len(server.requests) == len(completed)
|
||||
assert len({e["attempt_id"] for e in completed}) == len(completed)
|
||||
for event in completed:
|
||||
starts = [e for e in attempts if e["attempt_id"] == event["attempt_id"] and e["outcome"] == "started"]
|
||||
assert len(starts) == 1
|
||||
assert event["attempt_index"] == 1
|
||||
assert event["billing_intent"] == ("user_charge" if event["purpose"] == "main_agent" else "platform_cost")
|
||||
assert event["usage_available"] is True
|
||||
assert event["usage"]["input_tokens"] == 13
|
||||
assert event["usage"]["output_tokens"] == 7
|
||||
assert event["usage"]["cached_input_tokens"] == 3
|
||||
|
||||
|
||||
def test_selector_title_graph_stream_and_account(tmp_path, monkeypatch, isolated_network):
|
||||
configure_graph(monkeypatch)
|
||||
from EvoScientist.llm.runtime import _RuntimeAttemptCallback
|
||||
chunks = {}
|
||||
original_token = _RuntimeAttemptCallback.on_llm_new_token
|
||||
|
||||
async def token_probe(self, token, *, run_id, **kwargs):
|
||||
chunks.setdefault(str(run_id), []).append(token)
|
||||
await original_token(self, token, run_id=run_id, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_RuntimeAttemptCallback, "on_llm_new_token", token_probe)
|
||||
async def scenario():
|
||||
async with CompletingSSE().serve() as server:
|
||||
run, sink = await build_run(tmp_path, monkeypatch, server)
|
||||
try:
|
||||
assert await run.wait_stopped(timeout=20) == "completed"
|
||||
assert_accounting(server, sink, ["tool_selector", "main_agent", "title"])
|
||||
print(json.dumps({"sdk_incremental_callbacks": chunks}), flush=True)
|
||||
assert len(chunks) == 3
|
||||
assert all(len([token for token in values if token]) >= 2 for values in chunks.values())
|
||||
assert any(e.kind == "title" and e.payload.get("title") == "B03 streamed completion" for e in sink.events)
|
||||
finally:
|
||||
await run.cancel("fixture-teardown")
|
||||
asyncio.run(scenario())
|
||||
|
||||
|
||||
def test_summarizer_graph_stream_and_account(tmp_path, monkeypatch, isolated_network):
|
||||
configure_graph(monkeypatch, summarize=True)
|
||||
from importlib import import_module
|
||||
module = import_module("EvoScientist.EvoScientist")
|
||||
original_summary_factory = module._create_run_summarization_middleware
|
||||
diagnostics = []
|
||||
|
||||
def summary_factory(model, backend, summarizer):
|
||||
middleware = original_summary_factory(model, backend, summarizer)
|
||||
original_wrap = middleware.awrap_model_call
|
||||
|
||||
async def wrap_probe(request, handler):
|
||||
history = middleware._get_effective_messages(request)
|
||||
tokens = middleware._count_tokens(history, request.system_message, request.tools)
|
||||
truncated, modified = middleware._truncate_args(history, tokens)
|
||||
cutoff = middleware._determine_cutoff_index(truncated)
|
||||
pending, _ = middleware._partition_messages(truncated, cutoff)
|
||||
trimmed = middleware._lc_helper._trim_messages_for_summary(pending)
|
||||
evidence = {
|
||||
"trigger": middleware._lc_helper.trigger,
|
||||
"keep": middleware._lc_helper.keep,
|
||||
"trim": middleware._lc_helper.trim_tokens_to_summarize,
|
||||
"history_messages": len(history),
|
||||
"history_tokens": middleware._count_tokens(history, None, None),
|
||||
"total_tokens": tokens, "cutoff": cutoff,
|
||||
"summary_messages": len(pending), "trimmed_messages": len(trimmed),
|
||||
"summary_profile_limit": middleware._get_profile_limits(),
|
||||
}
|
||||
diagnostics.append(evidence)
|
||||
print(json.dumps({"middleware_evidence": evidence}), flush=True)
|
||||
assert evidence["trigger"] == ("tokens", 8500)
|
||||
assert evidence["keep"] == ("tokens", 1000)
|
||||
assert evidence["trim"] is None
|
||||
assert evidence["summary_profile_limit"] == 100000
|
||||
assert len(history) >= 4 and evidence["history_tokens"] > 8500
|
||||
assert not modified
|
||||
assert middleware._should_summarize(truncated, tokens)
|
||||
assert cutoff > 0 and pending and trimmed
|
||||
return await original_wrap(request, handler)
|
||||
|
||||
middleware.awrap_model_call = wrap_probe
|
||||
return middleware
|
||||
|
||||
monkeypatch.setattr(module, "_create_run_summarization_middleware", summary_factory)
|
||||
from EvoScientist.llm.runtime import _RuntimeAttemptCallback
|
||||
chunks = {}
|
||||
original_token = _RuntimeAttemptCallback.on_llm_new_token
|
||||
|
||||
async def token_probe(self, token, *, run_id, **kwargs):
|
||||
chunks.setdefault(str(run_id), []).append(token)
|
||||
await original_token(self, token, run_id=run_id, **kwargs)
|
||||
|
||||
monkeypatch.setattr(_RuntimeAttemptCallback, "on_llm_new_token", token_probe)
|
||||
async def scenario():
|
||||
async with CompletingSSE().serve() as server:
|
||||
run, sink = await build_run(tmp_path, monkeypatch, server)
|
||||
try:
|
||||
outcome = await run.wait_stopped(timeout=20)
|
||||
assert_accounting(server, sink, ["deepagents_summarizer", "main_agent"])
|
||||
assert diagnostics
|
||||
print(json.dumps({"sdk_incremental_callbacks": chunks}), flush=True)
|
||||
assert len(chunks) == 2
|
||||
assert all(len([token for token in values if token]) >= 2 for values in chunks.values())
|
||||
assert outcome == "completed"
|
||||
assert "B03 streamed completion" in str(server.requests[-1]["messages"])
|
||||
path = "/workspace/conversation_history/b02:isolated:thread.md"
|
||||
backend = run._host.workspace_backend
|
||||
original = backend.download_files([path])[0]
|
||||
assert original.error is None, original.error
|
||||
assert b"Earlier request " * 800 in original.content
|
||||
readable = backend.read(path)
|
||||
assert readable.error is None, readable.error
|
||||
assert "Earlier request " in str(readable.file_data)
|
||||
assert path in str(server.requests[-1]["messages"])
|
||||
print(json.dumps({"offload_path": path, "raw_bytes": len(original.content),
|
||||
"raw_original_verified": True, "backend_read_verified": True}))
|
||||
finally:
|
||||
await run.cancel("fixture-teardown")
|
||||
asyncio.run(scenario())
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
repo = Path(__file__).resolve().parents[1]
|
||||
root = repo.parent / ".hermes/test-runtime/unified-execution/b03-auxiliary"
|
||||
root.mkdir(parents=True, exist_ok=True)
|
||||
ledger = root / "attempts.jsonl"
|
||||
node = "tests/test_execution_adapter_web_contract.py::" + sys.argv[1]
|
||||
assert sys.argv[1] in PLANNED_NODES
|
||||
records = [json.loads(line) for line in ledger.read_text().splitlines()] if ledger.exists() else []
|
||||
attempt = 1 + sum(r.get("phase") == "start" and r.get("node") == node for r in records)
|
||||
limit = 7 if sys.argv[1] == "test_summarizer_graph_stream_and_account" else 5
|
||||
if limit == 7:
|
||||
assert any(r.get("phase") == "budget_extended" and r.get("node") == node
|
||||
and r.get("limit") == limit for r in records), "budget extension not recorded"
|
||||
assert attempt <= limit, "node budget exhausted"
|
||||
home = root / (sys.argv[1] + f"-{attempt}-home")
|
||||
config = home / "config"
|
||||
config.mkdir(parents=True, exist_ok=True)
|
||||
env = {"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
|
||||
"PYTHONPATH": str(repo), "PYTHON_DOTENV_DISABLED": "1", "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1",
|
||||
"PYTHONDONTWRITEBYTECODE": "1", "EVOSCIENTIST_HOME": str(home),
|
||||
"EVOSCIENTIST_CONFIG_DIR": str(config), "XDG_CONFIG_HOME": str(config)}
|
||||
for key, folder in (("DATA", "data"), ("WORKSPACE", "workspace"), ("SKILLS", "skills"), ("MEMORIES", "memory")):
|
||||
env[f"EVOSCIENTIST_{key}_DIR"] = str(home / folder)
|
||||
command = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest", "-p", "no:cacheprovider",
|
||||
"-q", "-s", "--tb=short", "--disable-warnings", node]
|
||||
with ledger.open("a") as handle:
|
||||
if attempt == 1:
|
||||
handle.write(json.dumps({"phase": "registered", "node": node, "used": 0, "limit": 5,
|
||||
"intent": "RED missing wiring or first-pass compatibility", "command": command}) + "\n")
|
||||
handle.write(json.dumps({"phase": "start", "node": node, "attempt": attempt, "command": command}) + "\n")
|
||||
try:
|
||||
result = subprocess.run(command, cwd=repo, env=env, capture_output=True, text=True, timeout=50)
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
output = exc.stdout or b""
|
||||
if isinstance(output, bytes):
|
||||
output = output.decode(errors="replace")
|
||||
result = subprocess.CompletedProcess(command, 124, output, "runner timeout; child killed and reaped\n")
|
||||
log = root / (sys.argv[1] + f"-{attempt}.log")
|
||||
log.write_text(result.stdout + result.stderr)
|
||||
with ledger.open("a") as handle:
|
||||
handle.write(json.dumps({"phase": "result", "node": node, "attempt": attempt,
|
||||
"exit_code": result.returncode, "log": str(log), "remaining": limit-attempt}) + "\n")
|
||||
print(result.stdout + result.stderr)
|
||||
sys.exit(result.returncode)
|
||||
Reference in New Issue
Block a user