test: cover stop contract, execution adapters, checkpointer race and runtime identity
Docker / build (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Build / build (push) Has been cancelled

This commit is contained in:
m4
2026-09-13 15:12:17 +08:00
parent ea99ce9f7e
commit d4b53bfb08
37 changed files with 4889 additions and 11 deletions
+97
View File
@@ -0,0 +1,97 @@
"""Budgeted real public repository and route contracts; no business database."""
import asyncio
import copy
import json
import os
from pathlib import Path
import subprocess
import sys
ROOT = Path(__file__).resolve().parents[2]
LOG = ROOT / '.hermes/test-runtime/unified-execution/b03-host'
def scenario():
from contextlib import asynccontextmanager
from unittest.mock import patch
from gateway.services import recoverable_runs as service
from gateway.routes import recoverable_runs as route
from gateway.models.thread import RecoverableRunRespond
from gateway.services.pending_display import approval_display
from gateway.services.projection_event_adapter import pending_input_projection_event
sentinel = 'private-canary'
value = {'action_requests': [{'name': 'execute', 'args': {'command': 'printf ' + sentinel}, 'description': sentinel}], 'review_configs': [{'action_name': 'execute', 'allowed_decisions': ['approve', 'edit', 'reject']}], 'hidden': sentinel}
pending = dict(kind='tool_approval', payload=value, safe_payload=value, interrupt_id='i', payload_hash='a' * 64, status='pending', run_id='r')
row = dict(run_id='r', turn_id='t', run_request_id='q', thread_id='t', command_kind='new', status='awaiting_input', message_id='m', pending_input=pending)
item = dict(type='approval_request', item_id='a', display_payload=value, interrupt_id='i', payload_hash='a' * 64)
message = dict(payload={'items': [item]}, message_id='m', projection_version=1, message_index=1)
frozen = copy.deepcopy([pending, row, message])
checks = {}
def safe(obj):
return sentinel not in json.dumps(obj)
checks['projection'] = safe(service.RecoverableRunRepository.projection(row))
checks['event'] = safe(pending_input_projection_event(pending, message_id='m', run_id='r', source_sequence=1).model_dump(mode='json'))
class DB:
async def execute_fetchone(self, *args): return row
async def execute_fetchall(self, query, *args): return [message] if 'SELECT payload,' in query else [row]
async def fetch(self, query, *args):
if 'FROM conversation_messages' in query: return [message]
if 'FROM conversation_run_pending_inputs' in query: return [pending]
return [dict(row, projection_sequence=0)]
async def fetchrow(self, *args): return {'snapshot_version': 1}
async def connection(): return DB()
@asynccontextmanager
async def transaction(**kwargs): yield DB()
async def reads():
repo = service.RecoverableRunRepository()
with patch('gateway.database.get_app_connection', connection), patch.object(service, '_transaction', transaction):
checks['get'] = safe(await repo.get('r', 't', 'u'))
checks['list'] = safe(await repo.list('t', 'u', active_only=False, limit=1))
checks['snapshot'] = safe(await repo.snapshot_thread('t', 'u'))
checks['history'] = safe(await repo.list_authoritative_messages('t', 'u'))
asyncio.run(reads())
for name, args, reviewable in [('execute', {'command': 'printf ' + sentinel}, False), ('mystery', {'path': '/workspace/a'}, False), ('read_file', {'file_path': '/workspace/a'}, True), ('read_file', {'file_path': '/workspace/a', 'secret': sentinel}, False)]:
payload = dict(action_requests=[dict(name=name, args=args)], review_configs=[dict(action_name=name, allowed_decisions=['approve', 'edit', 'reject'])])
for decision in ['approve', 'edit', 'reject']:
for scope in ['current', 'thread']:
body = RecoverableRunRespond.model_construct(decision_request_id='00000000-0000-0000-0000-000000000001', interrupt_id='i', payload_hash='a' * 64, decisions=None, decision=decision, approval_scope=scope)
try:
route._validated_resume_value(dict(kind='tool_approval', payload=payload), body)
accepted = True
except service.RecoverableRunError:
accepted = False
checks[f'{name}-{len(args)}-{decision}-{scope}'] = accepted == (decision == 'reject' or (reviewable and decision == 'approve'))
public = approval_display(payload)
checks[f'display-{name}-{len(args)}'] = public['review_configs'][0]['allowed_decisions'] == (['approve', 'reject'] if reviewable else ['reject'])
checks['unchanged'] = frozen == [pending, row, message]
body = RecoverableRunRespond(decision_request_id='00000000-0000-0000-0000-000000000001', interrupt_id='i', payload_hash='a' * 64, decision='approve')
checks['authorized_auto_independent'] = route._validated_resume_value(pending, body, authorized_auto=True) == {'decisions': [{'type': 'approve'}]}
for reviews in [[], [{'action_name': 'execute', 'allowed_decisions': ['reject']}]]:
try:
route._validated_resume_value(dict(pending, payload=dict(value, review_configs=reviews)), body, authorized_auto=True)
checks['auto-fail-closed-' + str(len(reviews))] = False
except service.RecoverableRunError:
checks['auto-fail-closed-' + str(len(reviews))] = True
from gateway.contracts.conversation_items import ApprovalRequestItem
from gateway.services.pending_display import public_items
typed = ApprovalRequestItem(item_id='item_' + 'a' * 32, item_sequence=1, revision=1, actor={'type': 'system', 'id': 'approval'}, status='in_progress', source={'protocol': 'run-v1', 'first_sequence': 1, 'last_sequence': 1, 'event_count': 1}, parent_run_id='r', interrupt_id='i', payload_hash='a' * 64, pending_input_ref='pending:r:i', display_payload=value)
cleaned = public_items([typed.model_dump(mode='json')])[0]
checks['history_schema_valid'] = ApprovalRequestItem.model_validate(cleaned).item_sequence == 1 and safe(cleaned)
print(json.dumps(checks), flush=True)
return all(checks.values())
if __name__ == '__main__':
if '--child' in sys.argv:
raise SystemExit(0 if scenario() else 1)
LOG.mkdir(parents=True, exist_ok=True)
ledger = LOG / 'pending-public-attempts.jsonl'
records = [json.loads(x) for x in ledger.read_text().splitlines()] if ledger.exists() else []
attempt = 1 + sum(r['phase'] == 'start' for r in records)
assert attempt <= 5
with ledger.open('a') as f: f.write(json.dumps(dict(phase='start', attempt=attempt, limit=5, node='public_reads_and_route_decisions')) + '\n')
env = dict(HOME=str(LOG), PATH='/usr/bin:/bin', PYTHONPATH=str(ROOT / 'Ai4Sci-Web'), PYTHON_DOTENV_DISABLED='1', PYTHONDONTWRITEBYTECODE='1')
result = subprocess.run([str(ROOT / 'Ai4Sci-Web/.venv/bin/python'), __file__, '--child'], env=env, capture_output=True, text=True, timeout=30)
output = result.stdout + result.stderr
(LOG / f'pending-public-{attempt}.log').write_text(output)
with ledger.open('a') as f: f.write(json.dumps(dict(phase='result', attempt=attempt, exit=result.returncode)) + '\n')
print(output)
raise SystemExit(result.returncode)
+69
View File
@@ -0,0 +1,69 @@
"""Budgeted stdlib-only public pending regression; no application imports."""
import ast
import hashlib
import json
import sys
from collections.abc import Mapping
from pathlib import Path
from types import SimpleNamespace
from typing import Any
ROOT = Path(__file__).resolve().parents[2]
sys.path.insert(0, str(ROOT / 'Ai4Sci-Web'))
LOG = ROOT / '.hermes/test-runtime/unified-execution/b03-host'
ledger = LOG / 'pending-pure-attempts.jsonl'
records = [json.loads(x) for x in ledger.read_text().splitlines()] if ledger.exists() else []
attempt = 1 + sum(x['phase'] == 'start' for x in records)
assert attempt <= 5
with ledger.open('a') as f:
f.write(json.dumps(dict(phase='start', node='test_pending_display_allowlist', attempt=attempt, limit=5)) + '\n')
ns: dict[str, Any] = dict(Any=Any, Mapping=Mapping, hashlib=hashlib, json=json,
ProjectionEvent=lambda **kw: SimpleNamespace(**kw),
ProjectionSource=lambda **kw: kw, ItemActor=lambda **kw: kw)
for filename, names in [
('pending_display.py', None),
('recoverable_runs.py', {'canonical_json', 'content_hash', '_value', '_mapping', '_safe_json', 'normalize_interrupt'}),
('projection_event_adapter.py', {'_safe', 'pending_input_projection_event'}),
]:
path = ROOT / 'Ai4Sci-Web/gateway/services' / filename
if not path.exists():
continue
tree = ast.parse(path.read_text())
body = [n for n in tree.body if isinstance(n, ast.FunctionDef) and (names is None or n.name in names)]
if names is None:
body = [n for n in tree.body if isinstance(n, (ast.FunctionDef, ast.Assign, ast.Import, ast.ImportFrom))]
exec(compile(ast.Module(body=body, type_ignores=[]), str(path), 'exec'), ns)
sentinel = 'b03-private-' + 'canary-7f9a'
raw = {'id': 'stable-id', 'value': {
'action_requests': [{'name': 'execute', 'args': {'command': 'printf ' + sentinel, 'checkpoint_details': {'x': sentinel}},
'description': 'Tool: execute Args: ' + sentinel, 'checkpoint_details': {'x': [sentinel]}}],
'review_configs': [{'action_name': 'execute', 'allowed_decisions': ['approve', 'reject'], 'checkpoint_details': {'x': sentinel}}],
'checkpoint_details': {'x': sentinel}}}
before = ns['content_hash'](raw)
expected_payload = ns['_safe_json'](raw['value'])
pending = ns['normalize_interrupt'](raw)
event = ns['pending_input_projection_event'](pending, message_id='m', run_id='r', source_sequence=1)
public = json.dumps([pending['safe_payload'], pending['event'], event.payload])
action = event.payload['display_payload']['action_requests'][0]
checks = {
'public_has_no_sentinel': sentinel not in public,
'no_unknown_nested_fields': 'checkpoint_details' not in public,
'description_regenerated': sentinel not in action['description'] and 'printf' in action['description'],
'command_visible': 'printf' in json.dumps(action['args']),
'internal_payload_unchanged': pending['payload'] == expected_payload,
'decision_hash_unchanged': pending['payload_hash'] == ns['content_hash'](expected_payload),
'raw_hash_unchanged': before == ns['content_hash'](raw),
'identity_compatible': pending['interrupt_id'] == raw['id'] and event.payload['payload_hash'] == pending['payload_hash'],
'review_reject_only': pending['safe_payload']['review_configs'][0]['allowed_decisions'] == ['reject'],
}
# Also test an older stored safe_payload passed directly to the projector.
legacy = dict(pending, safe_payload=raw['value'])
legacy_event = ns['pending_input_projection_event'](legacy, message_id='m', run_id='r', source_sequence=2)
checks['legacy_projection_safe'] = sentinel not in json.dumps(legacy_event.payload)
passed = all(checks.values())
result = json.dumps(checks)
(LOG / f'pending-pure-{attempt}.log').write_text(result + '\n')
with ledger.open('a') as f:
f.write(json.dumps(dict(phase='result', attempt=attempt, passed=passed)) + '\n')
print(result)
raise SystemExit(0 if passed else 1)
+13
View File
@@ -0,0 +1,13 @@
# B04 stop proof attempt ledger
New nodes, approved budget five executions each. Old race and continuation nodes are excluded.
| Node | Initial | Attempt 1 | Attempt 2 | Attempt 3 | Attempt 4 | Attempt 5 |
| --- | --- | --- | --- | --- | --- | --- |
| test_real_v3_stop[cancel] | 0/5 | fixture failure: missing native SubagentTransformer, log /tmp/b04-stop-red1.log | RED: cancellation/queue stop unconfirmed; close adapter absent (/tmp/b04-stop-red2.log) | GREEN 3 passed (/tmp/b04-stop-green3.log) | GREEN 3 passed (/tmp/b04-stop-green4.log); each node now 4/5 | not run |
| test_real_v3_stop[close_error] | 0/5 | fixture failure: missing native SubagentTransformer, log /tmp/b04-stop-red1.log | RED: cancellation/queue stop unconfirmed; close adapter absent (/tmp/b04-stop-red2.log) | GREEN 3 passed (/tmp/b04-stop-green3.log) | GREEN 3 passed (/tmp/b04-stop-green4.log); each node now 4/5 | not run |
| test_real_v3_stop[slow_commit] | 0/5 | fixture failure: missing native SubagentTransformer, log /tmp/b04-stop-red1.log | RED: cancellation/queue stop unconfirmed; close adapter absent (/tmp/b04-stop-red2.log) | GREEN 3 passed (/tmp/b04-stop-green3.log) | GREEN 3 passed (/tmp/b04-stop-green4.log); each node now 4/5 | not run |
Dependency source reviewed: langgraph 1.2.6 AsyncGraphRunStream, AsyncPregelLoop,
AsyncBackgroundExecutor; aiosqlite 0.22.1 Connection._execute and worker queue;
langgraph-checkpoint-sqlite 3.0.3 aput/aput_writes.
+35
View File
@@ -0,0 +1,35 @@
# B04 stop proof evidence
Final run: `/tmp/b04-stop-green4.log`, `3 passed in 0.44s`.
Each new parameter node used 4/5 attempts (ledger alongside this file).
Old writer race and continuation scenarios were not executed.
- cancel: real StateGraph plus native LangChain SubagentTransformer and
controlled FakeListChatModel stream; node finally settled, one terminal,
physical writer rows zero.
- close_error: instance-local graph iterator aclose fails once, then blocks;
repeated wait_stopped returns unknown, stop task remains alive, writer row
remains one; release allows retry, terminal once, writer rows zero.
- slow_commit: real aiosqlite worker operation blocks, its waiting future is
cancelled, operation still executes an INSERT; unknown retains writer;
after release, stop commit barrier makes INSERT visible from a separate
SQLite connection and physical writer rows become zero.
- Default no-registry abort compatibility: `/tmp/b04-stop-default-regression.log`,
`1 passed, 48 deselected in 0.13s`.
No event-stream replacement, global dependency monkeypatch, external network,
business service, PostgreSQL, installation, git commit, or full suite used.
Support is pinned to LangGraph 1.2.6, checkpoint-sqlite 3.0.3, aiosqlite
0.22.1. Adapter retains the graph iterator before dependency abort can discard
it, intercepts cancellation exception args for Pregel exit_task, and waits for
pulls, exits, graph close, mux close and local producers. Database barrier runs
after event generator closure (including exceptional state repair).
Limits: these three runs observed zero *detached* exit_task handles; ordinary
Pregel exit completed inside the cancelled pull. The exit_task capture branch
is source-grounded, not directly fault-triggered by these fixtures. A failed
or cancelled internal exit has no reliable supported recovery interface and
conservatively retains ownership. Unsupported dependency versions do likewise.
No arbitrary detached writers, external update_state/raw SQL writers, external
namespace entry, different registry, file replacement, or saver-level CAS claims.
+28
View File
@@ -0,0 +1,28 @@
"""Register every invocation before launching pytest, including startup errors."""
import json
import os
from pathlib import Path
import subprocess
import sys
root = Path(__file__).resolve().parents[3] / ".hermes/test-runtime/unified-execution/b04-control"
root.mkdir(parents=True, exist_ok=True)
nodes = sys.argv[2:]
for node in nodes:
path = root / (node + ".jsonl")
records = path.read_text().splitlines() if path.exists() else []
assert len(records) < 5, "node budget exhausted"
for node in nodes:
path = root / (node + ".jsonl")
used = len(path.read_text().splitlines()) if path.exists() else 0
with path.open("a") as out:
out.write(json.dumps(dict(node=node, attempt=used + 1, phase=sys.argv[1])) + "\n")
out.flush()
os.fsync(out.fileno())
result = subprocess.run([sys.executable, "-m", "pytest", "-q", *[
"tests/test_host_registry_control.py::test_" + node for node in nodes
]], capture_output=True, text=True)
print(result.stdout, end="")
print(result.stderr, end="", file=sys.stderr)
(root / (sys.argv[1] + "-" + "-".join(nodes) + ".txt")).write_text(result.stdout + result.stderr)
sys.exit(result.returncode)
+203
View File
@@ -0,0 +1,203 @@
"""Isolated B04 process-loss probe. Not a production owner or adapter."""
from __future__ import annotations
import asyncio
import dataclasses
import hashlib
import json
import os
from pathlib import Path
import pickle
import socket
import subprocess
import sys
import time
ROOT = Path(__file__).resolve().parents[3]
REPO = ROOT / "EvoScientist"
BASE = ROOT / ".hermes/test-runtime/unified-execution/b04"
TARGET = "b04_current_host_restart_identity_no_automatic_replay"
def append(path, value):
with path.open("a") as out:
out.write(json.dumps(value, sort_keys=True) + "\n")
out.flush()
os.fsync(out.fileno())
async def child(mode, directory):
started = time.perf_counter()
def blocked(*args, **kwargs):
raise AssertionError("B04 prohibits network")
socket.socket.connect = blocked
socket.socket.connect_ex = blocked
socket.getaddrinfo = blocked
sys.path.insert(0, str(REPO))
import importlib
from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel
from langchain_core.messages import AIMessage
from langgraph.checkpoint.memory import InMemorySaver
from EvoScientist.config.settings import EvoScientistConfig
from EvoScientist.llm.contracts import AgentInputV3, HmacGrantAuthority, WebHostContext
from EvoScientist.llm.model_config import FileEvoModelConfigStore
from EvoScientist.llm.runtime import EvoModelRuntime
from EvoScientist.llm.host_execution_registry import SQLiteHostRegistry
from EvoScientist.web_runtime import create_web_agent, web_tool_registry_manifest
from EvoScientist.workspace_files import ScopedFilesystemBackend
from tests.v3_fixtures import v3_payload, identity_ring, RUNTIME_SECRET, RUNTIME_KEY_ID
from tests.test_web_model_runtime import _preparation, _admission
import resource
imported = time.perf_counter()
module = importlib.import_module("EvoScientist.EvoScientist")
module._load_mcp_tools_cached = lambda **kw: {}
module._load_mcp_config_once = lambda: ("b04-empty", {})
entered = asyncio.Event()
class LocalBlockingModel(FakeMessagesListChatModel):
def bind_tools(self, tools, **kwargs):
return self
async def _agenerate(self, *args, **kwargs):
append(directory / "model-entries.jsonl", {"pid": os.getpid(), "mode": mode})
entered.set()
await asyncio.Event().wait()
class EvidenceSink:
async def commit(self, event):
append(directory / "events.jsonl", {
"event_id": event.event_id, "run_id": event.run_id,
"kind": event.kind, "payload": dict(event.payload),
})
return "committed"
async def confirm(self, *args):
raise AssertionError("unexpected commit ambiguity")
authority = HmacGrantAuthority(RUNTIME_SECRET, RUNTIME_KEY_ID)
store = FileEvoModelConfigStore(directory / "routes.yaml", admin_verifier=authority)
if mode == "start":
payload = v3_payload()
for provider in payload["providers"].values():
for model in provider["models"]:
model["context_window"] = 131072
store.bootstrap_for_development(payload)
config = EvoScientistConfig(enable_async_subagents=False, enable_scheduler=False,
memory_workers_enabled=False, auto_approve=False)
runtime = EvoModelRuntime(store, admission_verifier=authority, quote_authority=authority,
host_registry=SQLiteHostRegistry(directory / "host.sqlite", host_id="b04-host", boot_id=__import__("uuid").uuid4().hex),
identity_key_ring=identity_ring(),
model_factory=lambda **kw: LocalBlockingModel(responses=[AIMessage(content="unused")]),
agent_factory=lambda snapshot, host, models: create_web_agent(
snapshot=snapshot, host=host, model_set=models, config=config))
constructed = time.perf_counter()
metrics = {"mode": mode, "pid": os.getpid(), "runtime_instance_id": runtime.runtime_instance_id,
"import_seconds": imported-started, "runtime_setup_seconds": constructed-imported}
if mode == "start":
for name in ("files", "memory"):
(directory / name).mkdir(exist_ok=True)
tools, revision = web_tool_registry_manifest()
host = WebHostContext(str(directory / "files"), str(directory / "memory"),
ScopedFilesystemBackend(directory / "files"), InMemorySaver(),
tool_registry=tools, tool_registry_revision=revision,
tool_selector_threshold=10000, runtime_event_sink=EvidenceSink())
agent_input = AgentInputV3("B04 controlled host loss", "b04:isolated:thread")
t = time.perf_counter()
quote = await runtime.prepare_model_run(_preparation(authority, agent_input, title_policy="disabled"), agent_input, host)
metrics["prepare_seconds"] = time.perf_counter()-t
admission = _admission(authority, quote)
t = time.perf_counter()
run = await runtime.start_web_run(admission)
metrics["start_graph_seconds"] = time.perf_counter()-t
metrics["run_id"] = run.run_id
metrics["same_grant_same_object"] = await runtime.start_web_run(admission) is run
await asyncio.wait_for(entered.wait(), 10)
metrics["start_to_model_entry_seconds"] = time.perf_counter()-t
metrics["task_done_before_loss"] = run._agent_task.done()
with (directory / "admission.pickle").open("wb") as out:
pickle.dump(admission, out)
out.flush()
os.fsync(out.fileno())
else:
prior = json.loads((directory / "start.json").read_text())
metrics["previous_run_id"] = prior["run_id"]
metrics["incarnation_changed"] = prior["runtime_instance_id"] != runtime.runtime_instance_id
metrics["public_identity_query_methods"] = [name for name in ("inspect", "get_run", "check_run", "get_execution") if callable(getattr(runtime, name, None))]
metrics["prepared_count"] = len(runtime._prepared)
metrics["started_grants_count"] = len(runtime._started_grants)
# Trusted local test artifact, never an untrusted or production pickle.
with (directory / "admission.pickle").open("rb") as inp:
admission = pickle.load(inp)
try:
await runtime.start_web_run(admission)
except Exception as exc:
metrics["old_admission_result"] = str(exc)
else:
raise AssertionError("old admission unexpectedly started")
await asyncio.sleep(0.1)
metrics["model_entries_total"] = len((directory / "model-entries.jsonl").read_text().splitlines())
assert metrics["incarnation_changed"]
metrics["inspection"] = runtime.inspect(prior["run_id"])
assert metrics["old_admission_result"] == "EXECUTION_UNKNOWN"
assert metrics["inspection"]["execution_id"] == prior["run_id"]
assert metrics["inspection"]["status"] == "unknown"
assert metrics["inspection"]["resources_confirmed_exited"] is False
assert metrics["model_entries_total"] == 1
assert metrics["public_identity_query_methods"] == ["inspect"]
metrics["max_rss_bytes_macos"] = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss
(directory / f"{mode}.json").write_text(json.dumps(metrics, indent=2))
print(json.dumps(metrics), flush=True)
if mode == "start":
# Deliberate abrupt loss while Graph is active: no cancellation/teardown.
os._exit(0)
def main():
if len(sys.argv) > 1:
asyncio.run(child(sys.argv[1], Path(sys.argv[2])))
print("B04 restart asyncio.run and interpreter exit", flush=True)
return
os.umask(0o077)
BASE.mkdir(parents=True, exist_ok=True)
ledger = BASE / "attempts.jsonl"
records = [json.loads(line) for line in ledger.read_text().splitlines()] if ledger.exists() else []
used = sum(r.get("phase") == "start" for r in records)
if used >= 5:
raise SystemExit("B04 cumulative budget exhausted")
number = used + 1
directory = BASE / f"attempt-{number}"
directory.mkdir()
home = directory / "home"
home.mkdir()
env = {"PATH": "/usr/bin:/bin", "HOME": str(home), "EVOSCIENTIST_HOME": str(home),
"EVOSCIENTIST_CONFIG_DIR": str(home / "config"), "PYTHON_DOTENV_DISABLED": "1",
"PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1", "WEB_RUNTIME_TEST_KEY": "b04-local-dummy",
"PYTHONPATH": str(REPO), "LANG": "en_US.UTF-8"}
command = [str(REPO / ".venv/bin/python"), str(Path(__file__).resolve())]
snapshot = {name: hashlib.sha256((REPO / name).read_bytes()).hexdigest() for name in (
"EvoScientist/llm/runtime.py", "EvoScientist/llm/contracts.py", "EvoScientist/web_runtime.py",
"tests/support/b04_host_restart_probe.py")}
append(ledger, {"phase": "start", "target": TARGET, "attempt": number, "limit": 5,
"time": time.time(), "command": command, "snapshot_sha256": snapshot})
code = 0
for mode in ("start", "restart"):
t = time.perf_counter()
with (directory / f"{mode}.log").open("w") as out:
try:
result = subprocess.run(command + [mode, str(directory)], cwd=REPO, env=env,
stdout=out, stderr=subprocess.STDOUT, timeout=40)
code = result.returncode
except subprocess.TimeoutExpired:
code = 124
append(ledger, {"phase": "child_exit", "attempt": number, "mode": mode,
"exit_code": code, "wall_seconds": time.perf_counter()-t})
print((directory / f"{mode}.log").read_text())
if code:
break
append(ledger, {"phase": "result", "attempt": number, "exit_code": code,
"time": time.time(), "remaining": 5-number, "directory": str(directory)})
raise SystemExit(code)
if __name__ == "__main__":
main()
@@ -0,0 +1,41 @@
"""Isolated five-attempt budget for the new cross-Turn barrier target only."""
import json
import os
from pathlib import Path
import subprocess
import sys
repo = Path(__file__).resolve().parents[2]
root = repo.parent / ".hermes/test-runtime/unified-execution/b04-checkpoint-writer"
root.mkdir(parents=True, exist_ok=True)
node = "tests/test_checkpoint_writer_race.py::test_cross_turn_checkpoint_writer_barrier"
ledger = root / "attempts.jsonl"
records = [json.loads(line) for line in ledger.read_text().splitlines()] if ledger.exists() else []
attempt = 1 + sum(r.get("phase") == "start" for r in records)
assert attempt <= 5, "target budget exhausted"
home = root / f"home-{attempt}"
home.mkdir(mode=0o700)
env = {"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
"PYTHONPATH": str(repo), "PYTHON_DOTENV_DISABLED": "1", "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1",
"PYTHONDONTWRITEBYTECODE": "1", "EVOSCIENTIST_HOME": str(home),
"EVOSCIENTIST_CONFIG_DIR": str(home / "config"), "XDG_CONFIG_HOME": str(home / "config"),
"EVOSCIENTIST_DATA_DIR": str(home / "data"), "EVOSCIENTIST_WORKSPACE_DIR": str(home / "workspace"),
"EVOSCIENTIST_SKILLS_DIR": str(home / "skills"), "EVOSCIENTIST_MEMORIES_DIR": str(home / "memory")}
command = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest", "-p", "no:cacheprovider",
"-q", "-s", "--tb=short", node]
with ledger.open("a") as out:
if attempt == 1:
out.write(json.dumps(dict(phase="registered", node=node, used=0, limit=5)) + "\n")
out.write(json.dumps(dict(phase="start", node=node, attempt=attempt, stage=sys.argv[1], command=command)) + "\n")
try:
result = subprocess.run(command, cwd=repo, env=env, capture_output=True, text=True, timeout=60)
except subprocess.TimeoutExpired as exc:
output = exc.stdout or b""
result = subprocess.CompletedProcess(command, 124, output.decode() if isinstance(output, bytes) else output, "timeout")
log = root / f"attempt-{attempt}.log"
log.write_text(result.stdout + result.stderr)
os.chmod(log, 0o600)
with ledger.open("a") as out:
out.write(json.dumps(dict(phase="result", node=node, attempt=attempt, exit_code=result.returncode, log=str(log))) + "\n")
print(result.stdout + result.stderr)
sys.exit(result.returncode)
@@ -0,0 +1,41 @@
"""Offline, bounded B04 continuation-only runner."""
import json
import os
from pathlib import Path
import subprocess
import sys
repo = Path(__file__).resolve().parents[2]
root = repo.parent / ".hermes/test-runtime/unified-execution/b04-continuation"
root.mkdir(parents=True, exist_ok=True)
node = "tests/test_continuation_authorization_contract.py::" + sys.argv[1]
ledger = root / "attempts.jsonl"
records = [json.loads(line) for line in ledger.read_text().splitlines()] if ledger.exists() else []
attempt = 1 + sum(r.get("phase") == "start" and r.get("node") == node for r in records)
assert attempt <= 5, "node budget exhausted"
home = root / f"home-{sys.argv[1]}-{attempt}"
home.mkdir(mode=0o700)
env = {"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
"PYTHONPATH": str(repo), "PYTHON_DOTENV_DISABLED": "1", "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1",
"PYTHONDONTWRITEBYTECODE": "1", "EVOSCIENTIST_HOME": str(home),
"EVOSCIENTIST_CONFIG_DIR": str(home / "config"), "XDG_CONFIG_HOME": str(home / "config"),
"EVOSCIENTIST_DATA_DIR": str(home / "data"), "EVOSCIENTIST_WORKSPACE_DIR": str(home / "workspace"),
"EVOSCIENTIST_SKILLS_DIR": str(home / "skills"), "EVOSCIENTIST_MEMORIES_DIR": str(home / "memory")}
command = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest", "-p", "no:cacheprovider",
"-q", "-s", "--tb=short", "--disable-warnings", node]
with ledger.open("a") as out:
if attempt == 1:
out.write(json.dumps(dict(phase="registered", node=node, used=0, limit=5)) + "\n")
out.write(json.dumps(dict(phase="start", node=node, attempt=attempt, command=command)) + "\n")
try:
result = subprocess.run(command, cwd=repo, env=env, capture_output=True, text=True, timeout=60)
except subprocess.TimeoutExpired as exc:
output = exc.stdout or b""
result = subprocess.CompletedProcess(command, 124, output.decode() if isinstance(output, bytes) else output, "timeout")
log = root / f"{sys.argv[1]}-{attempt}.log"
log.write_text(result.stdout + result.stderr)
os.chmod(log, 0o600)
with ledger.open("a") as out:
out.write(json.dumps(dict(phase="result", node=node, attempt=attempt, exit_code=result.returncode, log=str(log))) + "\n")
print(result.stdout + result.stderr)
sys.exit(result.returncode)
+46
View File
@@ -0,0 +1,46 @@
"""Isolated B01 runner with durable per-attempt evidence, maximum three runs."""
import datetime
import json
import os
from pathlib import Path
import subprocess
import sys
repo = Path(__file__).resolve().parents[2]
root = repo.parent / ".hermes/test-runtime/unified-execution/b01"
root.mkdir(parents=True, exist_ok=True)
ledger = root / "attempts.jsonl"
attempts = ledger.read_text().splitlines() if ledger.exists() else []
if len(attempts) >= 3:
raise SystemExit("B01 budget exhausted")
attempt = len(attempts) + 1
home = root / f"attempt-{attempt}"
config = home / "config"
config.mkdir(parents=True, exist_ok=True)
(config / "settings.yaml").write_text("enable_async_subagents: false\nenable_scheduler: false\nmemory_workers_enabled: false\n")
env = {
"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
"PYTHONPATH": str(repo), "PYTHON_DOTENV_DISABLED": "1",
"PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1", "PYTHONDONTWRITEBYTECODE": "1",
"EVOSCIENTIST_HOME": str(home), "EVOSCIENTIST_CONFIG_DIR": str(config),
"XDG_CONFIG_HOME": str(config), "EVOSCIENTIST_DATA_DIR": str(home / "data"),
"EVOSCIENTIST_WORKSPACE_DIR": str(home / "workspace"),
"EVOSCIENTIST_SKILLS_DIR": str(home / "skills"),
"EVOSCIENTIST_MEMORIES_DIR": str(home / "memory"),
}
nodes = [
"test_explicit_web_factory_preserves_environment_and_globals",
"test_two_tenants_real_file_tools_and_sqlite_reopen",
"test_web_manual_execute_interrupts_before_backend",
]
command = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest", "-p", "no:cacheprovider", "-q", "--tb=short"]
command += ["tests/test_execution_adapter_embedding.py::" + n for n in nodes]
record = {"attempt": attempt, "started": datetime.datetime.now().isoformat(), "nodes": nodes, "command": command, "cwd": str(repo)}
with ledger.open("a") as handle:
handle.write(json.dumps(record) + "\n")
result = subprocess.run(command, cwd=repo, env=env, text=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT)
log = root / f"attempt-{attempt}.log"
log.write_text(result.stdout)
print(result.stdout)
print(json.dumps({"attempt": attempt, "exit_code": result.returncode, "log": str(log)}))
sys.exit(result.returncode)
+119
View File
@@ -0,0 +1,119 @@
"""Offline real-v3 stop proofs; each fault has its own five-run budget."""
import asyncio
import sqlite3
import socket
import threading
import pytest
@pytest.mark.parametrize("fault", ["cancel", "close_error", "slow_commit"])
def test_real_v3_stop(tmp_path, monkeypatch, fault):
from langchain_core.language_models.fake_chat_models import FakeListChatModel
from langchain_core.messages import AIMessage
from langgraph.graph import StateGraph, MessagesState, START, END
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
from EvoScientist.llm.contracts import AgentInputV3, WebHostContext
from EvoScientist.llm.host_execution_registry import SQLiteHostRegistry
from tests.test_web_model_runtime import _runtime, _preparation, _admission, _Sink
def forbidden(*args, **kwargs):
raise AssertionError("network forbidden")
monkeypatch.setattr(socket.socket, "connect", forbidden)
monkeypatch.setattr(socket, "getaddrinfo", forbidden)
async def scenario():
entered, settled = asyncio.Event(), asyncio.Event()
close_failed, close_release = asyncio.Event(), asyncio.Event()
release = threading.Event()
runtime, authority = _runtime(tmp_path, monkeypatch)
registry_path = tmp_path / "registry.sqlite"
runtime.host_registry = SQLiteHostRegistry(registry_path, host_id="host", boot_id="stop")
async with AsyncSqliteSaver.from_conn_string(str(tmp_path / "graph.sqlite")) as saver:
model = FakeListChatModel(responses=["controlled streaming response"], sleep=0.01)
async def model_node(state):
try:
async for chunk in model.astream(state["messages"]):
entered.set()
await asyncio.Event().wait()
return {"messages": [AIMessage(content="finished")]}
finally:
settled.set()
builder = StateGraph(MessagesState)
builder.add_node("model", model_node)
builder.add_edge(START, "model")
builder.add_edge("model", END)
from langchain.agents._subagent_transformer import SubagentTransformer
runtime.agent_factory = lambda snapshot, host, models: builder.compile(
checkpointer=saver, transformers=[SubagentTransformer])
sink = _Sink()
host = WebHostContext(str(tmp_path), str(tmp_path), object(), saver, runtime_event_sink=sink)
value = AgentInputV3("stop probe", "user:stop")
quote = await runtime.prepare_model_run(_preparation(authority, value, title_policy="disabled"), value, host)
run = await runtime.start_web_run(_admission(authority, quote))
queued = None
try:
await asyncio.wait_for(entered.wait(), 3)
if fault == "close_error":
# Instance-local dependency fault, not a replacement event pipeline.
raw = run._stream_stop.graph
class FailOnce:
failures = 0
def __aiter__(self): return self
async def __anext__(self): return await raw.__anext__()
async def aclose(self):
if not self.failures:
self.failures += 1
close_failed.set()
raise RuntimeError("controlled close failure")
await close_release.wait()
await raw.aclose()
run._stream_stop.graph = FailOnce()
if fault == "slow_commit":
await saver.conn.execute("CREATE TABLE stop_probe(value TEXT)")
await saver.conn.commit()
busy = threading.Event()
def blocked_write():
busy.set()
release.wait(5)
saver.conn._conn.execute("INSERT INTO stop_probe VALUES ('cancelled future executed')")
queued = asyncio.create_task(saver.conn._execute(blocked_write))
await asyncio.wait_for(asyncio.to_thread(busy.wait), 3)
queued.cancel()
await asyncio.gather(queued, return_exceptions=True)
run._request_cancel("probe")
if fault == "close_error":
await asyncio.wait_for(close_failed.wait(), 3)
assert await run.wait_stopped(timeout=0.02) == "unknown"
assert run._stream_stop.errors
assert not run._stream_stop.task.done()
with sqlite3.connect(registry_path) as db:
assert db.execute("SELECT count(*) FROM checkpoint_writers").fetchone()[0] == 1
close_release.set()
if fault == "slow_commit":
assert await run.wait_stopped(timeout=0.03) == "unknown"
with sqlite3.connect(registry_path) as db:
assert db.execute("SELECT count(*) FROM checkpoint_writers").fetchone()[0] == 1
print("slow_commit: cancelled queued future; unknown retains writer")
release.set()
assert await run.wait_stopped(timeout=3) == "cancelled"
assert settled.is_set()
assert run._checkpoint_writes_stopped
assert sum(e.payload.get("kind") == "run_terminal" for e in run._journal) == 1
with sqlite3.connect(registry_path) as db:
assert db.execute("SELECT count(*) FROM checkpoint_writers").fetchone()[0] == 0
if fault == "slow_commit":
with sqlite3.connect(tmp_path / "graph.sqlite") as db:
assert db.execute("SELECT value FROM stop_probe").fetchone()[0] == "cancelled future executed"
if fault == "close_error":
assert run._stream_stop.errors
print("close_error: retained failure, retried closure, released")
print(f"{fault}: real v3 model settled, terminal=1, writers=0")
print(f"{fault}: exit_tasks_observed={run._stream_stop.exit_tasks_observed}, pulls={len(run._stream_stop.pulls)}")
finally:
release.set()
close_release.set()
if run._terminal_event is None:
run._request_cancel("test cleanup")
await asyncio.gather(run.wait_stopped(timeout=3), return_exceptions=True)
asyncio.run(scenario())
@@ -0,0 +1,13 @@
import inspect
from EvoScientist.langgraph_dev.http import cancel_recoverable_run
def test_cancel_deletes_run_only_after_execution_exit_receipt():
source = inspect.getsource(cancel_recoverable_run)
assert 'initial.get("execution_exited") is not True' in source
assert "await worker_exit.wait_for_exit" in source
exit_check = source.index('receipt.get("execution_exited") is True')
delete = source.index("await Runs.delete")
assert exit_check < delete
assert '"checkpoint_cleanup": "completed"' in source
+158
View File
@@ -0,0 +1,158 @@
"""Distinct cross-Turn writer exclusion scenario, real Graph/SQLite, offline."""
import asyncio
import hashlib
import socket
import uuid
from dataclasses import replace
from typing import TypedDict
import pytest
def test_cross_turn_checkpoint_writer_barrier(tmp_path, monkeypatch):
from EvoScientist.llm.contracts import AgentInputV3, WebHostContext, EvoRuntimeError, canonical_json_v1
from EvoScientist.llm.host_execution_registry import SQLiteHostRegistry
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
from langgraph.graph import StateGraph, START, END
from langgraph.types import Command, interrupt
from tests.test_web_model_runtime import _runtime, _preparation, _admission, _Sink
def forbidden(*args, **kwargs):
raise AssertionError("network forbidden")
monkeypatch.setattr(socket.socket, "connect", forbidden)
monkeypatch.setattr(socket, "getaddrinfo", forbidden)
# Bare StateGraph has no DeepAgents subagent event lane. Adapt only the
# event projection; execution, interrupts and persistence remain real.
async def graph_events(agent, message, thread_id, *, configurable, **kwargs):
from EvoScientist.stream.events import build_agent_stream_input
value = await build_agent_stream_input(message, media=None)
async for update in agent.astream(value, {"configurable": {
**configurable, "thread_id": thread_id}}, stream_mode="updates"):
yield {"type": "graph_update"}
monkeypatch.setattr("EvoScientist.stream.events.stream_agent_events", graph_events)
class State(TypedDict):
messages: list
_verified_review_mode: dict
result: str
consumed = []
async def review(state):
if state["messages"][-1]["content"] == "parallel":
return {"result": "parallel"}
decision = interrupt({"action_requests": [{"name": "probe", "args": {}}]})
consumed.append(decision)
return {"result": "approved"}
async def scenario():
runtime, authority = _runtime(tmp_path, monkeypatch)
peer_root = tmp_path / "peer"
peer_root.mkdir()
peer, _ = _runtime(peer_root, monkeypatch)
registry_path = tmp_path / "registry.sqlite"
runtime.host_registry = SQLiteHostRegistry(registry_path, host_id="host", boot_id="a")
peer.host_registry = SQLiteHostRegistry(registry_path, host_id="host", boot_id="b")
async with AsyncSqliteSaver.from_conn_string(str(tmp_path / "graph.sqlite")) as saver:
async with AsyncSqliteSaver.from_conn_string(str(tmp_path / "graph.sqlite")) as peer_saver:
builder = StateGraph(State)
builder.add_node("mode", lambda state: {"_verified_review_mode": {"mode": "manual"}})
builder.add_node("review", review)
builder.add_edge(START, "mode")
builder.add_edge("mode", "review")
builder.add_edge("review", END)
runtime.agent_factory = lambda snapshot, host, models: builder.compile(checkpointer=host.checkpointer)
peer.agent_factory = runtime.agent_factory
host = WebHostContext(str(tmp_path), str(tmp_path), object(), saver, runtime_event_sink=_Sink())
peer_host = replace(host, checkpointer=peer_saver, runtime_event_sink=_Sink())
def grant(value, turn, **changes):
template = _preparation(authority, value, title_policy="disabled")
fields = dict(template.unsigned_payload())
for key in ("issued_at", "expires_at", "key_id", "schema_version", "contract_type"):
fields.pop(key, None)
fields.update(request_id=str(uuid.uuid4()), turn_id=turn, **changes)
return authority.sign_preparation(**fields, ttl_ms=60000)
async def prepare(rt, h, value, turn, **changes):
return await rt.prepare_model_run(grant(value, turn, **changes), value, h)
value = AgentInputV3("approval", "user:checkpoint")
turn = str(uuid.uuid4())
parent_q = await prepare(runtime, host, value, turn)
parent = await runtime.start_web_run(_admission(authority, parent_q))
assert await parent.wait_stopped(timeout=5) == "awaiting_input"
payload = parent._terminal_event.payload
identity = {k: payload[k] for k in ("checkpoint_thread_id", "checkpoint_id", "checkpoint_ns", "pending_interrupts")}
approved = AgentInputV3(Command(resume={"decisions": [{"type": "approve"}]}), value.checkpoint_thread_id)
child_q = await prepare(runtime, host, approved, turn,
checkpoint_snapshot_id=payload["checkpoint_id"], predecessor_execution_id=parent.run_id,
predecessor_checkpoint_id=payload["checkpoint_id"], predecessor_owner_epoch=1,
continuation_pending_hash=hashlib.sha256(canonical_json_v1(identity)).hexdigest(),
continuation_decision_hash=hashlib.sha256(canonical_json_v1(approved.message.resume)).hexdigest())
other_q = await prepare(runtime, host, value, str(uuid.uuid4()))
peer_q = await prepare(peer, peer_host, value, str(uuid.uuid4()))
parallel_value = AgentInputV3("parallel", "user:other-checkpoint")
parallel_q = await prepare(peer, peer_host, parallel_value, str(uuid.uuid4()))
checked, release = asyncio.Event(), asyncio.Event()
closing, closed = asyncio.Event(), asyncio.Event()
class Client:
async def aclose(self):
closing.set()
await closed.wait()
factory = runtime.agent_factory
def owned_factory(*args):
from EvoScientist.llm.runtime import _construction_owner
owner = _construction_owner.get()
client = Client()
owner._owned_clients[id(client)] = client
return factory(*args)
runtime.agent_factory = owned_factory
validate = runtime._validate_continuation
async def barrier(g, v, h):
await validate(g, v, h)
if g.predecessor_execution_id:
checked.set()
await release.wait()
monkeypatch.setattr(runtime, "_validate_continuation", barrier)
starting = asyncio.create_task(runtime.start_web_run(_admission(authority, child_q)))
late = []
try:
await asyncio.wait_for(checked.wait(), 5)
parallel = await peer.start_web_run(_admission(authority, parallel_q))
assert await parallel.wait_stopped(timeout=5) == "completed"
for rt, quote in ((runtime, other_q), (peer, peer_q)):
try:
late.append(await rt.start_web_run(_admission(authority, quote)))
except EvoRuntimeError as exc:
assert "CHECKPOINT_WRITER_BUSY" in str(exc)
assert not late, "cross-Turn writers admitted after continuation latest validation"
finally:
for run in late:
await run.wait_stopped(timeout=5)
release.set()
child = await starting
await asyncio.wait_for(closing.wait(), 5)
try:
assert await child.wait_stopped(timeout=0.02) == "unknown"
with pytest.raises(EvoRuntimeError, match="CHECKPOINT_WRITER_BUSY"):
await peer.start_web_run(_admission(authority, peer_q))
finally:
closed.set()
await child.wait_stopped(timeout=5)
assert child._terminal_event.payload["outcome"] == "completed"
assert consumed == [{"decisions": [{"type": "approve"}]}]
import sqlite3
with sqlite3.connect(registry_path) as db:
assert db.execute("SELECT count(*) FROM checkpoint_writers").fetchone()[0] == 0
# A stale continuation never consumes a later Turn's pending.
newer = await peer.start_web_run(_admission(authority, peer_q))
assert await newer.wait_stopped(timeout=5) == "awaiting_input"
with pytest.raises(EvoRuntimeError, match="CONTINUATION_CHECKPOINT_STALE"):
await runtime.prepare_model_run(grant(approved, turn,
checkpoint_snapshot_id=payload["checkpoint_id"], predecessor_execution_id=parent.run_id,
predecessor_checkpoint_id=payload["checkpoint_id"], predecessor_owner_epoch=1,
continuation_pending_hash=hashlib.sha256(canonical_json_v1(identity)).hexdigest(),
continuation_decision_hash=hashlib.sha256(canonical_json_v1(approved.message.resume)).hexdigest()), approved, host)
print("cross-Turn same/peer runtime blocked; distinct checkpoint completes; pinned resume consumed once")
asyncio.run(scenario())
@@ -0,0 +1,166 @@
"""B04 durable authorization reaches real Web Graph execution, offline only."""
import asyncio
import hashlib
import importlib
import socket
import sqlite3
import uuid
import pytest
def _authorization_scenario(tmp_path, monkeypatch, *, construction_failure=False):
import faulthandler
faulthandler.dump_traceback_later(15)
from EvoScientist.config.settings import EvoScientistConfig
from EvoScientist.llm.contracts import AgentInputV3, WebHostContext, EvoRuntimeError, canonical_json_v1
from EvoScientist.llm.host_execution_registry import SQLiteHostRegistry
from EvoScientist.web_runtime import create_web_agent, web_tool_registry_manifest
from EvoScientist.workspace_files import ScopedFilesystemBackend
from deepagents.backends.protocol import SandboxBackendProtocol, ExecuteResponse
from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel
from langchain_core.messages import AIMessage, AIMessageChunk
from langchain_core.outputs import ChatGenerationChunk
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
from langgraph.types import Command
from tests.test_web_model_runtime import _runtime, _preparation, _admission, _Sink
def forbidden(*args, **kwargs):
raise AssertionError("network forbidden")
monkeypatch.setattr(socket.socket, "connect", forbidden)
monkeypatch.setattr(socket, "getaddrinfo", forbidden)
module = importlib.import_module("EvoScientist.EvoScientist")
monkeypatch.setattr(module, "_load_mcp_tools_cached", lambda **kw: {})
monkeypatch.setattr(module, "_load_mcp_config_once", lambda: ("b04-offline", {}))
calls = []
class Backend(ScopedFilesystemBackend, SandboxBackendProtocol):
@property
def id(self):
return "b04-no-shell"
def execute(self, command, *, timeout=None):
assert command == "authorized-probe"
calls.append(command)
return ExecuteResponse(output="authorized-result", exit_code=0)
class Model(FakeMessagesListChatModel):
def bind_tools(self, tools, **kwargs):
return self
async def _astream(self, messages, stop=None, run_manager=None, **kwargs):
if any(m.type == "tool" for m in messages):
yield ChatGenerationChunk(message=AIMessageChunk(content="authorized done"))
else:
yield ChatGenerationChunk(message=AIMessageChunk(content="", tool_call_chunks=[{
"name": "execute", "args": '{"command":"authorized-probe"}',
"id": "pending-tool", "index": 0}]))
async def scenario():
import tests.test_web_model_runtime as helpers
payload = helpers.v3_payload()
payload["providers"]["custom-openai"]["models"][0]["context_window"] = 131072
monkeypatch.setattr(helpers, "v3_payload", lambda: payload)
runtime, authority = _runtime(tmp_path, monkeypatch)
registry = SQLiteHostRegistry(tmp_path / "private" / "host.sqlite", host_id="host", boot_id="one")
runtime.host_registry = registry
runtime.model_factory = lambda **kw: Model(responses=[AIMessage(content="unused")])
cfg = EvoScientistConfig(auto_approve=False, enable_async_subagents=False,
enable_scheduler=False, memory_workers_enabled=False)
runtime.agent_factory = lambda snapshot, host, models: create_web_agent(
snapshot=snapshot, host=host, model_set=models, config=cfg)
root = tmp_path / "workspace"
root.mkdir()
tools, revision = web_tool_registry_manifest()
async with AsyncSqliteSaver.from_conn_string(str(tmp_path / "graph.sqlite")) as saver:
host = WebHostContext(str(root), str(root), Backend(root), saver,
tool_selector_threshold=10000, tool_registry=tools,
tool_registry_revision=revision, runtime_event_sink=_Sink())
value = AgentInputV3("request approval", "b04-thread")
quote = await runtime.prepare_model_run(_preparation(authority, value, title_policy="disabled"), value, host)
parent = await runtime.start_web_run(_admission(authority, quote))
assert await parent.wait_stopped(timeout=8) == "awaiting_input"
assert calls == []
state = await parent._agent.aget_state({"configurable": {"thread_id": value.checkpoint_thread_id}})
assert state.values["_verified_review_mode"]["mode"] == "manual"
details = parent._terminal_event.payload
identity = {k: details[k] for k in ("checkpoint_thread_id", "checkpoint_id", "checkpoint_ns", "pending_interrupts")}
pending_hash = hashlib.sha256(canonical_json_v1(identity)).hexdigest()
approved = AgentInputV3(Command(resume={"decisions": [{"type": "approve"}]}), value.checkpoint_thread_id)
decision_hash = hashlib.sha256(canonical_json_v1(approved.message.resume)).hexdigest()
def grant_for(value=approved, **changes):
template = _preparation(authority, value, title_policy="disabled")
fields = ("turn_id", "thread_id", "subject_id", "requested_model_ref", "plan", "roles",
"requires_vision", "reasoning_effort", "title_policy", "gateway_input_digest",
"checkpoint_thread_id", "turn_fencing_token")
return authority.sign_preparation(**{k: getattr(template, k) for k in fields},
**dict(dict(request_id=str(uuid.uuid4()), checkpoint_snapshot_id=identity["checkpoint_id"],
predecessor_execution_id=parent.run_id, predecessor_checkpoint_id=identity["checkpoint_id"],
predecessor_owner_epoch=1, continuation_pending_hash=pending_hash,
continuation_decision_hash=decision_hash), **changes), ttl_ms=60000)
for changes in ({"continuation_decision_hash": "bad"}, {"continuation_pending_hash": "bad"},
{"predecessor_owner_epoch": 2}):
with pytest.raises(EvoRuntimeError):
await runtime.prepare_model_run(grant_for(**changes), approved, host)
with pytest.raises(EvoRuntimeError):
await runtime.prepare_model_run(grant_for(value=value), value, host)
good = grant_for()
prepared = await runtime.prepare_model_run(good, approved, host)
approved.message.resume["decisions"][0]["type"] = "reject"
with pytest.raises(EvoRuntimeError, match="CONTINUATION_DECISION_INVALID"):
await runtime.start_web_run(_admission(authority, prepared))
approved.message.resume["decisions"][0]["type"] = "approve"
if construction_failure:
def fail_factory(*args):
raise RuntimeError("controlled construction failure")
runtime.agent_factory = fail_factory
with pytest.raises(RuntimeError, match="controlled construction failure"):
await runtime.start_web_run(_admission(authority, prepared))
assert registry.inspect(prepared.execution_id)["status"] == "failed"
with sqlite3.connect(registry.path) as db:
row = db.execute("SELECT consumed_by, decision_hash FROM pending_continuations WHERE execution_id=?", (parent.run_id,)).fetchone()
assert row == (prepared.execution_id, decision_hash)
runtime.host_registry = SQLiteHostRegistry(registry.path, host_id="host", boot_id="two")
again = await runtime.prepare_model_run(grant_for(), approved, host)
with pytest.raises(EvoRuntimeError, match="CONTINUATION_CONSUMED_FAILURE_REQUIRES_REAUTHORIZATION"):
await runtime.start_web_run(_admission(authority, again))
assert calls == []
print("B04 consumed construction failure stays consumed across reopen; new admission cannot replay")
return
# Mutating the checkpoint after prepare cannot execute its new pending.
await parent._agent.aupdate_state(state.config, {"_verified_review_mode": {"mode": "auto"}})
with pytest.raises(EvoRuntimeError):
await runtime.start_web_run(_admission(authority, prepared))
assert calls == []
# Restore the original latest checkpoint pointer in this isolated fixture.
with sqlite3.connect(tmp_path / "graph.sqlite") as db:
db.execute("DELETE FROM checkpoints WHERE checkpoint_id > ?", (identity["checkpoint_id"],))
# Reopen registry before bind to prove pending authority is durable.
runtime.host_registry = SQLiteHostRegistry(registry.path, host_id="host", boot_id="two")
child = await runtime.start_web_run(_admission(authority, prepared))
assert child.run_id != parent.run_id
assert await child.wait_stopped(timeout=8) == "completed"
assert calls == ["authorized-probe"]
with sqlite3.connect(registry.path) as db:
row = db.execute("SELECT consumed_by, decision_hash FROM pending_continuations WHERE execution_id=?", (parent.run_id,)).fetchone()
assert row == (child.run_id, decision_hash)
with pytest.raises(EvoRuntimeError):
again = await runtime.prepare_model_run(grant_for(), approved, host)
await runtime.start_web_run(_admission(authority, again))
assert calls == ["authorized-probe"]
assert (tmp_path / "private").stat().st_mode & 0o777 == 0o700
assert (tmp_path / "private" / "host.sqlite").stat().st_mode & 0o777 == 0o600
print("B04 durable pending -> signed decision -> atomic bind -> Graph tool: exactly one; replay rejected")
asyncio.run(scenario())
faulthandler.cancel_dump_traceback_later()
def test_durable_decision_bound_to_graph_execution(tmp_path, monkeypatch):
_authorization_scenario(tmp_path, monkeypatch)
def test_consumed_graph_child_failure_requires_reauthorization(tmp_path, monkeypatch):
_authorization_scenario(tmp_path, monkeypatch, construction_failure=True)
@@ -0,0 +1,202 @@
"""B02 fragmented tool approval and cancellation through real runtime runs."""
import asyncio
import json
import os
from pathlib import Path
import subprocess
import sys
import uuid
import pytest
if __name__ == "__main__":
sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from tests.test_execution_adapter_stop_contract import (
HeartbeatExecutor, LoopbackSSE, build_run, isolated_network,
)
NODE = "test_fragmented_manual_checkpoint_child_run_stops_native_tree"
class FragmentedToolSSE(LoopbackSSE):
async def handle(self, reader, writer):
task = asyncio.current_task()
self.tasks.add(task)
self.writers.add(writer)
try:
header = await asyncio.wait_for(reader.readuntil(b"\r\n\r\n"), 5)
headers = dict(line.split(b":", 1) for line in header.split(b"\r\n")[1:] if b":" in line)
length = int(next(v for k, v in headers.items() if k.lower() == b"content-length"))
assert header.startswith(b"POST /v1/chat/completions ")
body = json.loads(await reader.readexactly(length))
self.requests.append(body)
assert body["stream"] is True
assert any(t["function"]["name"] == "execute" for t in body["tools"])
assert len(self.requests) == 1, "unexpected post-tool model request"
writer.write(b"HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nConnection: close\r\n\r\n")
fragments = ('{"comm', 'and":"b02-', 'heartbeat"}')
for index, fragment in enumerate(fragments):
call = {"index": 0, "function": {"arguments": fragment}}
if index == 0:
call.update(id="b02-fragmented", type="function")
call["function"]["name"] = "execute"
chunk = {"id": "b02-fragmented", "object": "chat.completion.chunk", "created": 1,
"model": body["model"], "choices": [{"index": 0, "delta": {"tool_calls": [call]}, "finish_reason": None}]}
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\n")
await writer.drain()
await asyncio.sleep(.02)
chunk["choices"] = [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}]
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\ndata: [DONE]\n\n")
await writer.drain()
self.requested.set()
except Exception as exc:
self.errors.append(exc)
self.requested.set()
finally:
writer.close()
await writer.wait_closed()
self.writers.discard(writer)
self.tasks.discard(task)
def test_fragmented_manual_checkpoint_child_run_stops_native_tree(tmp_path, monkeypatch, isolated_network):
from EvoScientist import workspace_files
from EvoScientist.native_sandbox import NativeWorkspaceBackend
from EvoScientist.llm.contracts import AgentInputV3
from langgraph.types import Command
from tests.test_web_model_runtime import _preparation, _admission
executor = HeartbeatExecutor(tmp_path)
calls = []
original = executor.execute
def counted(command, **kwargs):
calls.append(command)
return original(command, **kwargs)
executor.execute = counted
scoped = workspace_files.ScopedFilesystemBackend
class LocalBackend(NativeWorkspaceBackend):
def __init__(self, root):
scoped.__init__(self, root)
self._executor = executor
self._sandbox_id = "b02-whitelist"
monkeypatch.setattr(workspace_files, "ScopedFilesystemBackend", LocalBackend)
async def scenario():
child = None
async with FragmentedToolSSE().serve() as server:
parent, sink = await build_run(tmp_path, monkeypatch, server)
try:
await asyncio.wait_for(server.requested.wait(), 8)
assert await parent.wait_stopped(timeout=8) == "awaiting_input"
assert not server.errors
config = {"configurable": {"thread_id": parent._input.checkpoint_thread_id}}
state = await parent._agent.aget_state(config)
assert state.interrupts, (state.next, parent._terminal_event)
payload = parent._terminal_event.payload
assert payload["outcome"] == "awaiting_input"
assert payload["checkpoint_thread_id"] == state.config["configurable"]["thread_id"]
assert payload["checkpoint_id"] == state.config["configurable"]["checkpoint_id"]
assert payload["checkpoint_id"]
assert payload["checkpoint_ns"] == state.config["configurable"]["checkpoint_ns"]
assert payload["pending_interrupts"] == [
{"id": item.id, "value": item.value} for item in state.interrupts
]
assert json.loads(json.dumps(payload))["pending_interrupts"] == payload["pending_interrupts"]
parent_events = [event async for event in parent.stream()]
assert len([e for e in parent_events if e.payload.get("kind") == "run_terminal"]) == 1
assert state.values["_verified_review_mode"]["mode"] == "manual"
assert state.interrupts[0].value["action_requests"][0]["args"] == {"command": "b02-heartbeat"}
await asyncio.sleep(.15)
assert calls == [] and not executor.started.is_set()
assert not executor.child_pid.exists() and not executor.heartbeat.exists()
runtime = parent._runtime
authority = runtime.quote_authority
approved = AgentInputV3(Command(resume={"decisions": [{"type": "approve"}]}), parent._input.checkpoint_thread_id)
template = _preparation(authority, approved, title_policy="disabled")
fields = ("turn_id", "thread_id", "subject_id", "requested_model_ref", "plan", "roles",
"requires_vision", "reasoning_effort", "title_policy", "gateway_input_digest",
"checkpoint_thread_id", "checkpoint_snapshot_id", "turn_fencing_token")
grant = authority.sign_preparation(**{k: getattr(template, k) for k in fields},
request_id=str(uuid.uuid4()), ttl_ms=60_000)
quote = await runtime.prepare_model_run(grant, approved, parent._host)
child = await runtime.start_web_run(_admission(authority, quote))
async with asyncio.timeout(8):
while not executor.heartbeat.exists() or not executor.child_pid.exists():
if child._agent_task.done():
pytest.fail(f"approval child ended before execution: {child._terminal_event}")
await asyncio.sleep(.02)
assert calls == ["b02-heartbeat"]
assert await child.cancel("manual-tool-stop") == "cancelled"
assert await child.wait_stopped(timeout=4) == "cancelled"
await executor.assert_stopped_before_teardown()
assert child._agent_task.done()
events = [event async for event in child.stream()]
await asyncio.sleep(.15)
assert len(server.requests) == 1 and not server.errors
assert calls == ["b02-heartbeat"]
assert len([e for e in events if e.payload.get("kind") == "run_terminal"]) == 1
print(json.dumps({"evidence": "manual_fragmented_runtime_stop", "argument_fragments": 3,
"manual_interrupt": True, "executions": len(calls), "http_requests": len(server.requests),
"worker_finished": executor.finished.is_set(), "child_run_stopped": child._agent_task.done(),
"parent_outcome": payload["outcome"],
"checkpoint_id": payload["checkpoint_id"],
"checkpoint_ns": payload["checkpoint_ns"],
"pending_interrupts": payload["pending_interrupts"]}), flush=True)
finally:
if child is not None:
await child.cancel("teardown")
await parent.cancel("teardown")
executor.emergency_cleanup()
asyncio.run(scenario())
if __name__ == "__main__":
repo = Path(__file__).resolve().parents[1]
root = repo.parent / ".hermes/test-runtime/unified-execution/b02"
root.mkdir(parents=True, exist_ok=True)
ledger = root / "attempts.jsonl"
node = "tests/test_execution_adapter_approval_tools.py::" + NODE
records = [json.loads(line) for line in ledger.read_text().splitlines()] if ledger.exists() else []
attempt = 1 + sum(r.get("phase") == "start" and r.get("node") == node for r in records)
limit = 5
assert attempt <= limit, "node budget exhausted"
home = root / (NODE + f"-{attempt}-home")
config = home / "config"
config.mkdir(parents=True, exist_ok=True)
env = {"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
"PYTHONPATH": str(repo), "PYTHON_DOTENV_DISABLED": "1", "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1",
"PYTHONDONTWRITEBYTECODE": "1", "EVOSCIENTIST_HOME": str(home), "EVOSCIENTIST_CONFIG_DIR": str(config),
"XDG_CONFIG_HOME": str(config), "EVOSCIENTIST_DATA_DIR": str(home / "data"),
"EVOSCIENTIST_WORKSPACE_DIR": str(home / "workspace"), "EVOSCIENTIST_SKILLS_DIR": str(home / "skills"),
"EVOSCIENTIST_MEMORIES_DIR": str(home / "memory")}
command = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest", "-p", "no:cacheprovider",
"-q", "-s", "--tb=short", "--disable-warnings", node]
with ledger.open("a") as handle:
if attempt == 1:
handle.write(json.dumps({"phase": "registered", "node": node, "used": 0, "limit": limit, "command": command}) + "\n")
elif not any(r.get("node") == node and r.get("limit") == limit for r in records):
handle.write(json.dumps({"phase": "authorization_extension", "node": node,
"used": attempt - 1, "limit": limit,
"authority": "explicit user total-five authorization",
"preserve_prior_attempts": True}) + "\n")
handle.write(json.dumps({"phase": "start", "node": node, "attempt": attempt, "command": command}) + "\n")
try:
result = subprocess.run(command, cwd=repo, env=env, capture_output=True, text=True, timeout=60)
except subprocess.TimeoutExpired as exc:
output = exc.stdout or b""
if isinstance(output, bytes):
output = output.decode(errors="replace")
result = subprocess.CompletedProcess(command, 124, output, "runner timeout\n")
log = root / (NODE + f"-{attempt}.log")
log.write_text(result.stdout + result.stderr)
with ledger.open("a") as handle:
handle.write(json.dumps({"phase": "result", "node": node, "attempt": attempt, "exit_code": result.returncode,
"log": str(log), "remaining": limit-attempt}) + "\n")
print(result.stdout + result.stderr)
sys.exit(result.returncode)
+183
View File
@@ -0,0 +1,183 @@
"""B01 real factory contracts. Run with --noconftest in an isolated env."""
import asyncio
import importlib
import json
import os
import pytest
@pytest.fixture
def embedding(tmp_path, monkeypatch):
assert os.environ.get("PYTHON_DOTENV_DISABLED") == "1"
assert "unified-execution" in os.environ["EVOSCIENTIST_HOME"]
from EvoScientist.config.settings import EvoScientistConfig
from EvoScientist.llm.contracts import AgentModelSet, WebHostContext
from EvoScientist.workspace_files import ScopedFilesystemBackend
from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel
runtime = importlib.import_module("EvoScientist.EvoScientist")
web = importlib.import_module("EvoScientist.web_runtime")
monkeypatch.setattr(runtime, "_load_mcp_tools_cached", lambda **kw: {})
monkeypatch.setattr(runtime, "_load_mcp_config_once", lambda: ("b01-empty", {}))
class Model(FakeMessagesListChatModel):
def bind_tools(self, tools, **kwargs):
return self
def build(saver, tenant, responses, backend=None):
root = tmp_path / tenant
root.mkdir(exist_ok=True)
memory = tmp_path / (tenant + "-memory")
memory.mkdir(exist_ok=True)
backend = backend or ScopedFilesystemBackend(root)
model = Model(responses=responses)
cfg = EvoScientistConfig(
openai_api_key="b01-not-a-real-key-" + tenant,
enable_async_subagents=False, enable_scheduler=False,
memory_workers_enabled=False, auto_approve=False,
)
_, revision = web.web_tool_registry_manifest()
graph = web.create_web_agent(
snapshot=None,
host=WebHostContext(
workspace_dir=str(root), memory_dir=str(memory),
workspace_backend=backend, checkpointer=saver,
tool_registry_revision=revision, tool_selector_threshold=10000,
),
model_set=AgentModelSet(model, model, model), config=cfg,
)
return graph, root, backend
return build, runtime
def _run_config(tenant):
# Host-owned identity: SQLite does not isolate by user_id automatically.
identity = json.dumps([tenant, "same-conversation"], separators=(",", ":"))
return {"configurable": {"thread_id": identity, "ai4sci_run_id": tenant}}
def _call(name, args, ident):
from langchain_core.messages import AIMessage
return AIMessage(content="", tool_calls=[
{"name": name, "args": args, "id": ident, "type": "tool_call"}
])
def test_explicit_web_factory_preserves_environment_and_globals(embedding, tmp_path):
from EvoScientist import paths
from langchain_core.messages import AIMessage
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
build, runtime = embedding
before_env = dict(os.environ)
names = ("_config", "_chat_model", "_chat_model_key", "_EvoScientist_agent")
before_globals = {name: vars(runtime).get(name) for name in names}
before_paths = (paths.WORKSPACE_ROOT, paths._active_workspace)
async def scenario():
async with AsyncSqliteSaver.from_conn_string(str(tmp_path / "pure.sqlite")) as saver:
await saver.setup()
for tenant in ("alice", "bob"):
graph, _, _ = build(saver, tenant, [AIMessage(content="ok")])
assert hasattr(graph, "ainvoke")
changed = {key for key in before_env.keys() | os.environ.keys()
if before_env.get(key) != os.environ.get(key)}
assert not changed, f"factory changed environment keys: {sorted(changed)}"
assert all(vars(runtime).get(n) is value for n, value in before_globals.items())
assert (paths.WORKSPACE_ROOT, paths._active_workspace) == before_paths
asyncio.run(scenario())
def test_two_tenants_real_file_tools_and_sqlite_reopen(embedding, tmp_path):
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
build, _ = embedding
configs = {tenant: _run_config(tenant) for tenant in ("alice", "bob")}
assert configs["alice"]["configurable"]["thread_id"] != configs["bob"]["configurable"]["thread_id"]
async def scenario():
db = str(tmp_path / "persistent.sqlite")
async with AsyncSqliteSaver.from_conn_string(db) as saver:
await saver.setup()
graphs = {}
for tenant in configs:
responses = [
_call("write_file", {"file_path": "/workspace/probe.txt", "content": tenant + "_ONLY"}, "write"),
_call("read_file", {"file_path": "/workspace/probe.txt"}, "read"),
_call("read_file", {"file_path": "/workspace/../bob/probe.txt"}, "escape"),
AIMessage(content="done"),
]
graphs[tenant], _, _ = build(saver, tenant, responses)
results = await asyncio.gather(*[
graphs[t].ainvoke({"messages": [HumanMessage(content="file probe")]}, configs[t])
for t in configs
])
for tenant, result in zip(configs, results):
tools = {m.tool_call_id: m for m in result["messages"] if isinstance(m, ToolMessage)}
assert (tmp_path / tenant / "probe.txt").read_text() == tenant + "_ONLY"
assert tenant + "_ONLY" in str(tools["read"].content)
assert "_ONLY" not in str(tools["escape"].content)
assert "error" in str(tools["escape"].content).lower()
other = "bob" if tenant == "alice" else "alice"
assert other + "_ONLY" not in str(result["messages"])
assert (tmp_path / "persistent.sqlite").stat().st_size > 0
async with AsyncSqliteSaver.from_conn_string(db) as reopened:
await reopened.setup()
for tenant in configs:
graph, _, backend = build(reopened, tenant, [AIMessage(content="restored")])
state = await graph.aget_state(configs[tenant])
assert tenant + "_ONLY" in str(state.values["messages"])
other = "bob" if tenant == "alice" else "alice"
assert other + "_ONLY" not in str(state.values["messages"])
history = [item async for item in graph.aget_state_history(configs[tenant])]
assert len(history) > 1
assert all(other + "_ONLY" not in str(s.values) for s in history)
assert backend.write("/workspace/../outside.txt", "forbidden").error
assert not (tmp_path / "outside.txt").exists()
asyncio.run(scenario())
def test_web_manual_execute_interrupts_before_backend(embedding, tmp_path):
from deepagents.backends.protocol import SandboxBackendProtocol, ExecuteResponse
from EvoScientist.workspace_files import ScopedFilesystemBackend
from langchain_core.messages import AIMessage, HumanMessage
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
from langgraph.types import Command
calls = []
class ProbeBackend(ScopedFilesystemBackend, SandboxBackendProtocol):
@property
def id(self):
return "b01-no-shell"
def execute(self, command, *, timeout=None):
# No shell or process is ever launched by this test capability.
assert command == "b01-probe"
calls.append(command)
return ExecuteResponse(output="probe-ok", exit_code=0)
build, _ = embedding
async def scenario():
async with AsyncSqliteSaver.from_conn_string(str(tmp_path / "review.sqlite")) as saver:
await saver.setup()
graph, _, _ = build(saver, "alice", [
_call("execute", {"command": "b01-probe"}, "execute"), AIMessage(content="done")
], ProbeBackend(tmp_path / "alice"))
config = _run_config("alice")
result = await graph.ainvoke({"messages": [HumanMessage(content="probe")]}, config)
assert result.get("__interrupt__")
assert calls == []
state = await graph.aget_state(config)
assert state.values["_verified_review_mode"]["mode"] == "manual"
await graph.ainvoke(Command(resume={"decisions": [{"type": "approve"}]}), config)
assert calls == ["b01-probe"]
asyncio.run(scenario())
@@ -0,0 +1,333 @@
"""B03 isolated history contract, real Graph and SQLite."""
import asyncio
import copy
import importlib.util
import socket
from dataclasses import replace
import pytest
@pytest.mark.parametrize("boundary", ["START", "END"])
def test_initialization_recovers_lost_write_response(tmp_path, boundary):
from EvoScientist.llm import history_rebuild as h
assert hasattr(h, "SqliteInitializationStore"), "durable initialization gate missing"
from langchain_core.messages import HumanMessage, AIMessage
from langgraph.graph import StateGraph, MessagesState, START, END
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
async def scenario():
scope = h.HistoryScope("a", "t", "recover", "w", "g", "tools", 1)
records = [h.HistoryRecord("m", 1, HumanMessage(content="history"))]
history = h.normalize_history(records, source=scope, expected=scope)
original = copy.deepcopy(records)
ran = []
def build(saver):
builder = StateGraph(MessagesState)
def model(state):
ran.append(state["messages"][-1].content)
return {"messages": [AIMessage(content="answer")]}
builder.add_node("model", model)
builder.add_edge(START, "model")
builder.add_edge("model", END)
return builder.compile(checkpointer=saver)
class LostResponse:
def __init__(self, graph):
self.graph = graph
def __getattr__(self, name):
return getattr(self.graph, name)
async def aupdate_state(self, *args, **kwargs):
result = await self.graph.aupdate_state(*args, **kwargs)
if kwargs["as_node"] == (START if boundary == "START" else END):
raise ConnectionError("write committed, response lost")
return result
db = str(tmp_path / "graph.sqlite")
store_path = str(tmp_path / "initialization.sqlite")
store = h.SqliteInitializationStore(store_path)
kwargs = dict(expected=scope, store=store, attempt="attempt-1", owner="host-1", fence=1)
async with AsyncSqliteSaver.from_conn_string(db) as saver:
graph = build(saver)
with pytest.raises(ConnectionError):
await h.create_history_checkpoint(LostResponse(graph), history, **kwargs)
assert store.get(h.history_key(scope))["status"] == "INITIALIZING"
with pytest.raises(ValueError, match="READY"):
await h.invoke_history_checkpoint(graph, scope=scope, store=store,
attempt="attempt-1", owner="host-1", fence=1, input={"messages": []})
assert ran == []
store = h.SqliteInitializationStore(store_path)
kwargs.update(store=store, owner="host-2", fence=2)
async with AsyncSqliteSaver.from_conn_string(db) as saver:
graph = build(saver)
config = await h.create_history_checkpoint(graph, history, **kwargs)
state = await graph.aget_state(config)
assert not state.tasks and not state.next
assert await h.create_history_checkpoint(graph, history, **kwargs) == config
with pytest.raises(ValueError):
await h.create_history_checkpoint(graph, history, **dict(kwargs, owner="host-1", fence=1))
changed = h.normalize_history([h.HistoryRecord("m", 1, HumanMessage(content="changed"))], source=scope, expected=scope)
with pytest.raises(ValueError):
await h.create_history_checkpoint(graph, changed, **kwargs)
with pytest.raises(ValueError):
await h.create_history_checkpoint(graph, history, **dict(kwargs, attempt="other"))
result = await h.invoke_history_checkpoint(graph, scope=scope, store=store,
attempt="attempt-1", owner="host-2", fence=2,
input={"messages": [HumanMessage(content="new")]})
assert result["messages"][-1].content == "answer" and ran == ["new"]
with pytest.raises(ValueError):
await h.create_history_checkpoint(graph, history, **kwargs)
assert records == original
asyncio.run(scenario())
@pytest.mark.parametrize("kind", ["wrong-name", "missing-id", "duplicate-id", "orphan"])
def test_tool_association_conflicts_are_rejected(kind):
from EvoScientist.llm.history_rebuild import HistoryScope, HistoryRecord, normalize_history
from langchain_core.messages import AIMessage, ToolMessage
scope = HistoryScope("a", "t", "turn", "w", "g", "tools", 1)
calls = [{"id": "stable", "name": "A", "args": {}}]
if kind == "duplicate-id":
calls.append({"id": "stable", "name": "B", "args": {}})
messages = [AIMessage(content="", tool_calls=calls), ToolMessage(content="result B",
tool_call_id="" if kind == "missing-id" else "unknown" if kind == "orphan" else "stable",
name="B" if kind in ("wrong-name", "missing-id") else "A")]
records = [HistoryRecord(str(i), 1, m) for i, m in enumerate(messages)]
original = copy.deepcopy(records)
with pytest.raises(ValueError, match="tool"):
normalize_history(records, source=scope, expected=scope)
assert records == original
def test_new_turn_uses_fresh_checkpoint(tmp_path, monkeypatch):
def no_network(*args, **kwargs):
raise AssertionError("network forbidden")
monkeypatch.setattr(socket.socket, "connect", no_network)
assert importlib.util.find_spec("EvoScientist.llm.history_rebuild"), "missing history builder"
from EvoScientist.llm.history_rebuild import (HistoryScope, HistoryRecord, normalize_history,
create_history_checkpoint, invoke_history_checkpoint, SqliteInitializationStore)
from EvoScientist.llm.patches import _validate_openai_tool_history
from langchain_core.messages import AIMessage, HumanMessage, ToolMessage
from langgraph.graph import StateGraph, MessagesState, START, END
from langgraph.types import interrupt
from langgraph.checkpoint.sqlite.aio import AsyncSqliteSaver
scope = HistoryScope("alice", "thread", "new-turn", "workspace", "g1", "t1", 7)
calls = [{"name": "probe", "args": {}, "id": k} for k in ("complete", "missing")]
records = [
HistoryRecord("human", 1, HumanMessage(content="old question")),
HistoryRecord("partial", 2, AIMessage(content="partial answer", tool_calls=calls), partial=True),
HistoryRecord("result", 1, ToolMessage(content="real result", tool_call_id="complete")),
HistoryRecord("media", 1, HumanMessage(content=[
{"type": "text", "text": "attachment"},
{"type": "image_url", "image_url": {"url": "data:image/png;base64,SECRET"}},
]), file_refs=("file:owned-document",)),
]
original = copy.deepcopy(records)
history = normalize_history(records, source=scope, expected=scope)
assert records == original
_validate_openai_tool_history(list(history.messages))
tools = [m for m in history.messages if isinstance(m, ToolMessage)]
assert len(tools) == 1 and tools[0].content == "real result"
ai = [m for m in history.messages if isinstance(m, AIMessage) and m.tool_calls]
assert [c["id"] for c in ai[0].tool_calls] == ["complete"]
text = str(history.messages)
assert "SECRET" not in text and "base64" not in text
assert "file:owned-document" in text
assert "partial" in text and "missing" in text and "incomplete" in text
for field, value in (("tenant_id", "bob"), ("workspace_id", "other"),
("graph_version", "g2"), ("tool_version", "t2"), ("history_revision", 8)):
with pytest.raises(ValueError):
normalize_history(records, source=scope, expected=replace(scope, **{field: value}))
for bad in ([records[0], records[0]], [replace(records[0], revision=-1)]):
with pytest.raises(ValueError):
normalize_history(bad, source=scope, expected=scope)
executed = []
def build(saver):
graph = StateGraph(MessagesState)
def model(state):
last = state["messages"][-1].content
if last == "old pending":
interrupt({"tool": "old-danger", "pending": True})
executed.append("OLD TOOL")
executed.append(last)
return {"messages": [AIMessage(content="new answer")]}
graph.add_node("model", model)
graph.add_edge(START, "model")
graph.add_edge("model", END)
return graph.compile(checkpointer=saver)
async def scenario():
db = str(tmp_path / "history.sqlite")
store = SqliteInitializationStore(str(tmp_path / "initialization.sqlite"))
gate = dict(store=store, attempt="original", owner="host", fence=1)
old = {"configurable": {"thread_id": "legacy-pending"}}
async with AsyncSqliteSaver.from_conn_string(db) as saver:
graph = build(saver)
await graph.ainvoke({"messages": [HumanMessage(content="old pending")]}, old)
before = await graph.aget_state(old)
assert before.next == ("model",) and before.tasks[0].interrupts
old_tuple = await saver.aget_tuple(old)
async with AsyncSqliteSaver.from_conn_string(db) as saver:
graph = build(saver)
config = await create_history_checkpoint(graph, history, expected=scope, **gate)
fresh = await graph.aget_state(config)
assert not fresh.next and not fresh.tasks and executed == []
with pytest.raises(ValueError):
await create_history_checkpoint(graph, history, expected=scope, **dict(gate, attempt="other"))
result = await invoke_history_checkpoint(graph, scope=scope, **gate,
input={"messages": [HumanMessage(content="new input")]})
assert result["messages"][-1].content == "new answer"
bob = replace(scope, tenant_id="bob")
bob_history = normalize_history([], source=bob, expected=bob)
bob_config = await create_history_checkpoint(graph, bob_history, expected=bob, **gate)
assert bob_config["configurable"]["thread_id"] != config["configurable"]["thread_id"]
assert not (await graph.aget_state(bob_config)).values.get("messages")
async with AsyncSqliteSaver.from_conn_string(db) as saver:
graph = build(saver)
assert await graph.aget_state(old) == before
assert await saver.aget_tuple(old) == old_tuple
assert (await graph.aget_state(config)).values["messages"][-1].content == "new answer"
assert executed == ["new input"]
asyncio.run(scenario())
def test_committed_v2_history_replaces_checkpoint_without_replaying_tools():
from typing import Any, cast
from EvoScientist.llm.history_rebuild import committed_history_input
from langchain_core.messages import AIMessage, convert_to_messages
from langgraph.graph.message import add_messages
history = {
"schema": "ai4sci.committed-history.v1", "thread_id": "thread",
"user_uid": "user", "conversation_revision": 2, "excluded_message_id": "new",
"records": [
{"message_id": "u", "message_index": 1, "revision": 1,
"role": "user", "payload": {"content": "old question"}},
{"message_id": "a", "message_index": 2, "revision": 1,
"role": "assistant", "payload": {"items": [
{"item_id": "answer", "item_sequence": 1, "type": "message",
"content": [{"type": "output_text", "text": "old answer"}]},
{"item_id": "tool", "item_sequence": 2, "type": "tool_call",
"name": "search", "input": {"q": "old"}, "status": "failed"},
]}},
],
}
result = committed_history_input(history, {"messages": [{"role": "user", "content": "next"}]},
thread_id="thread", run_id="run")
reduce_messages = cast(Any, add_messages)
messages = reduce_messages([AIMessage(content="stale checkpoint", id="stale")],
convert_to_messages(result["messages"]))
assert [m.id for m in messages] == ["u", "a", "current:run:0"]
assert "historical tool_call" in messages[1].content
assert not messages[1].tool_calls
assert messages[-1].content == "next"
assert result["_summarization_event"] is None
assert len(reduce_messages(messages, convert_to_messages(result["messages"]))) == 3
# A compatible checkpoint is authoritative after the first initialization.
# Its tool protocol and summarization indexes must survive the next turn.
current = {"messages": [{"role": "user", "content": "next"}]}
original = copy.deepcopy(current)
appended = committed_history_input(history, current, thread_id="thread", run_id="run",
checkpoint_exists=True)
checkpoint = [AIMessage(content="checkpoint answer", id="checkpoint",
tool_calls=[{"id": "tool-1", "name": "search", "args": {}}])]
continued = reduce_messages(checkpoint, convert_to_messages(appended["messages"]))
assert [m.id for m in continued] == ["checkpoint", "current:run:0"]
assert continued[0].tool_calls[0]["id"] == "tool-1"
assert "_summarization_event" not in appended
assert current == original
@pytest.mark.parametrize("existing,pending,operation,history,legacy,expected", [
(False, False, "start", True, False, "initialize"),
(True, False, "start", True, False, "append"),
(True, True, "start", True, False, "THREAD_AWAITING_INPUT"),
(True, True, "resume", False, False, "resume"),
(False, False, "resume", False, True, "CHECKPOINT_RESUME_UNAVAILABLE"),
(False, False, "start", False, True, "HISTORY_REQUIRED"),
(False, False, "start", False, False, "new"),
])
def test_runtime_history_admission(existing, pending, operation, history, legacy, expected,
monkeypatch):
from EvoScientist.langgraph_dev import http
async def checkpoint(*args):
return existing, pending
async def old_history(*args):
return legacy
monkeypatch.setattr(http, "_compatible_checkpoint", checkpoint, raising=False)
monkeypatch.setattr(http, "_has_legacy_history", old_history, raising=False)
result = asyncio.run(http._history_admission(None, "thread", "EvoScientist", {},
operation, {} if history else None))
assert result == expected
def test_runtime_reads_checkpoint_through_real_async_factory(monkeypatch):
from contextlib import asynccontextmanager
from langgraph.checkpoint.base import empty_checkpoint
from langgraph.checkpoint.memory import InMemorySaver
from langgraph.graph import StateGraph, MessagesState, START, END
# Installed API factory dispatch, without startup or network services.
monkeypatch.setenv("REDIS_URI", "redis://unused")
monkeypatch.setenv("LANGGRAPH_RUNTIME_VARIANT", "inmem")
from langgraph_api import graph as api_graph, _checkpointer
from langgraph_api import store as api_store
from EvoScientist.langgraph_dev import http
saver = InMemorySaver()
entered = []
@asynccontextmanager
async def factory(config):
entered.append("enter")
builder = StateGraph(MessagesState)
builder.add_node("model", lambda state: {})
builder.add_edge(START, "model")
builder.add_edge("model", END)
try:
yield builder.compile()
finally:
entered.append("exit")
monkeypatch.setitem(api_graph.GRAPHS, "history-contract", factory)
api_graph.classify_factory(factory, "history-contract")
async def get_saver(**kwargs):
return saver
async def get_store():
return None
monkeypatch.setattr(_checkpointer, "get_checkpointer", get_saver)
monkeypatch.setattr(api_store, "get_store", get_store)
async def scenario():
config = {"configurable": {"thread_id": "thread", "checkpoint_ns": ""}}
assert await http._compatible_checkpoint(None, "thread", "history-contract", {}) == (False, False)
cp = empty_checkpoint()
cp["channel_values"] = {"messages": []}
cp["channel_versions"] = {"messages": 1}
await saver.aput(config, cp, {"source": "update", "step": 0,
"parents": {}, "graph_id": "history-contract"}, {"messages": 1})
assert await http._compatible_checkpoint(None, "thread", "history-contract", {}) == (True, False)
assert entered == ["enter", "exit"]
asyncio.run(scenario())
def test_legacy_sqlite_history_is_detected_read_only(tmp_path, monkeypatch):
import sqlite3
import sys
import types
from EvoScientist.langgraph_dev import http
path = tmp_path / "sessions.db"
with sqlite3.connect(path) as conn:
conn.execute("CREATE TABLE checkpoints(thread_id TEXT)")
conn.execute("INSERT INTO checkpoints VALUES ('old')")
before = path.read_bytes()
monkeypatch.setenv("EVOSCIENTIST_LEGACY_CHECKPOINT_PATHS", __import__('json').dumps([str(path)]))
class Threads:
@staticmethod
async def get(*args):
from starlette.exceptions import HTTPException
raise HTTPException(404)
monkeypatch.setitem(sys.modules, "langgraph_runtime.ops", types.SimpleNamespace(Threads=Threads))
async def scenario():
assert await http._has_legacy_history(None, "old")
assert not await http._has_legacy_history(None, "new")
asyncio.run(scenario())
assert before == path.read_bytes()
@@ -0,0 +1,222 @@
"""B03 host boundaries. Only run through the isolated budget runner."""
import asyncio
import base64
import importlib
import json
import os
from pathlib import Path
import subprocess
import sys
import faulthandler
import re
def stage(name):
print('PENDING_STAGE ' + name, flush=True)
if __name__ != '__main__':
faulthandler.enable()
faulthandler.dump_traceback_later(30, repeat=False)
stage('module-import')
if __name__ != "__main__":
from tests.test_execution_adapter_stop_contract import isolated_network
NODES = ("test_main_mcp_registry_and_tenants", "test_media_provider_workspace_boundary",
"test_manual_pending_public_projection")
def factory(tmp_path, responses, tenant="alice", model_class=None, sandbox=False):
stage('factory-import-start')
from EvoScientist.config.settings import EvoScientistConfig
from EvoScientist.llm.contracts import AgentModelSet, WebHostContext
from EvoScientist.web_runtime import create_web_agent, web_tool_registry_manifest
from EvoScientist.workspace_files import ScopedFilesystemBackend
from langgraph.checkpoint.memory import InMemorySaver
from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel
class Model(FakeMessagesListChatModel):
def bind_tools(self, tools, **kwargs):
return self
root = tmp_path / tenant
root.mkdir(exist_ok=True)
backend = ScopedFilesystemBackend(root)
if sandbox:
from deepagents.backends.protocol import SandboxBackendProtocol
class NoExecute(ScopedFilesystemBackend, SandboxBackendProtocol):
@property
def id(self):
return "b03-no-execute"
def execute(self, command, *, timeout=None):
raise AssertionError("manual pending must not execute")
backend = NoExecute(root)
model = (model_class or Model)(responses=responses)
stage('registry-start')
_, revision = web_tool_registry_manifest()
stage('graph-build-start')
graph = create_web_agent(snapshot=None, host=WebHostContext(
workspace_dir=str(root), memory_dir=str(tmp_path / (tenant + "-memory")),
workspace_backend=backend, checkpointer=InMemorySaver(),
tool_registry_revision=revision, tool_selector_threshold=10000),
model_set=AgentModelSet(model, model, model), config=EvoScientistConfig(
auto_approve=False, enable_async_subagents=False, enable_scheduler=False,
memory_workers_enabled=False))
stage('graph-built')
return graph, backend
def call(name, args, ident):
from langchain_core.messages import AIMessage
return AIMessage(content="", tool_calls=[dict(name=name, args=args, id=ident, type="tool_call")])
def config(tenant):
return {"configurable": {"thread_id": tenant + ":same-thread", "ai4sci_run_id": tenant}}
def test_main_mcp_registry_and_tenants(tmp_path, isolated_network):
from EvoScientist.mcp.client import USER_MCP_CONFIG
from langchain_core.messages import AIMessage, ToolMessage
from EvoScientist.llm.contracts import EvoRuntimeError
import pytest
server = tmp_path / "server.py"
server.write_text('from mcp.server.fastmcp import FastMCP\nm = FastMCP("b03")\n@m.tool()\ndef host_echo(value: str) -> str:\n """Return the supplied value without side effects."""\n return "sdk:" + value\nm.run(transport="stdio")\n')
USER_MCP_CONFIG.parent.mkdir(parents=True, exist_ok=True)
settings = {"b03": {"transport": "stdio", "command": sys.executable,
"args": [str(server)], "expose_to": ["main"]}}
USER_MCP_CONFIG.write_text(json.dumps(settings))
async def scenario():
graphs = {t: factory(tmp_path, [call("host_echo", {"value": t}, t), AIMessage(content="done"),
call("host_echo", {"value": "blocked"}, t + "-stale")], t)[0]
for t in ("alice", "bob")}
for tenant, graph in graphs.items():
result = await graph.ainvoke({"messages": [("user", "echo")]}, config(tenant))
tools = [m for m in result["messages"] if isinstance(m, ToolMessage)]
assert len(tools) == 1 and "sdk:" + tenant in str(tools[0].content)
assert (await graph.aget_state(config("other"))).values == {}
settings["b03"]["expose_to"] = ["research-agent"]
USER_MCP_CONFIG.write_text(json.dumps(settings))
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_STALE"):
await graphs["alice"].ainvoke({"messages": [("user", "echo again")]}, config("alice"))
asyncio.run(scenario())
def test_media_provider_workspace_boundary(tmp_path, isolated_network):
from langchain_core.messages import AIMessage
from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel
seen = []
class Model(FakeMessagesListChatModel):
def bind_tools(self, tools, **kwargs):
return self
def _generate(self, messages, stop=None, run_manager=None, **kwargs):
seen.append(messages)
return super()._generate(messages, stop=stop, run_manager=run_manager, **kwargs)
raw = b"b03-local-media"
media = AIMessage(content=[{"type": "image", "base64": base64.b64encode(raw).decode(), "mime_type": "image/png"}])
async def scenario():
graph, alice = factory(tmp_path, [AIMessage(content="done")], model_class=Model)
_, bob = factory(tmp_path, [AIMessage(content="done")], "bob")
result = await graph.ainvoke({"messages": [media, ("user", "describe prior media")]}, config("alice"))
assert seen and "base64" not in str(seen[-1])
stored = next(m for m in result["messages"] if isinstance(m, AIMessage) and isinstance(m.content, list))
path = stored.content[0]["url"]
assert path in str(seen[-1]) and "generated_image" in str(seen[-1])
assert alice.download_files([path])[0].content == raw
assert bob.download_files([path])[0].error
assert bob.download_files(["/../alice/" + path.lstrip("/")])[0].error
asyncio.run(scenario())
def test_manual_pending_public_projection(tmp_path, isolated_network):
stage('real-import-start')
from langchain_core.messages import AIMessage
from gateway.services.recoverable_runs import normalize_interrupt, content_hash, _safe_json
from gateway.services.projection_event_adapter import pending_input_projection_event
stage('real-import-done')
async def scenario():
sentinel = "b03-private-" + "canary-7f9a"
graph, _ = factory(tmp_path, [call("execute", {"command": "printf " + sentinel}, "private-call"), AIMessage(content="done")], sandbox=True)
stage('invoke-start')
await graph.ainvoke({"messages": [("user", "manual approval")]}, config("alice"))
stage('invoke-done')
before = await graph.aget_state(config("alice"))
interrupt = before.tasks[0].interrupts[0]
internal = {"id": interrupt.id, "value": {**interrupt.value, "checkpoint_details": "internal-only"}}
internal["value"]["action_requests"] = json.loads(json.dumps(interrupt.value["action_requests"]))
internal["value"]["review_configs"] = json.loads(json.dumps(interrupt.value["review_configs"]))
internal["value"]["action_requests"][0]["checkpoint_details"] = {"nested": [sentinel]}
internal["value"]["review_configs"][0]["checkpoint_details"] = {"nested": [sentinel]}
frozen = json.loads(json.dumps(internal))
expected_payload = _safe_json(frozen["value"])
expected_hash = content_hash(expected_payload)
raw_hash = content_hash(frozen)
stage('projection-start')
pending = normalize_interrupt(internal)
event = pending_input_projection_event(pending, message_id="message", run_id="run", source_sequence=1)
assert set(event.payload) == {"parent_run_id", "interrupt_id", "payload_hash", "display_payload", "questions", "call_id"}
assert "internal-only" not in str(event.payload)
assert event.payload["interrupt_id"] == interrupt.id
assert event.payload["payload_hash"] == pending["payload_hash"]
assert internal == frozen
assert pending["payload"]["checkpoint_details"] == "internal-only"
assert before == await graph.aget_state(config("alice"))
public = json.dumps({"safe_payload": pending["safe_payload"], "event": pending["event"],
"projection": event.model_dump(mode="json")})
display = event.payload["display_payload"]
checks = {
"public_has_no_sentinel": sentinel not in public,
"no_unknown_nested_fields": "checkpoint_details" not in public,
"description_regenerated": sentinel in frozen["value"]["action_requests"][0]["description"]
and "printf" in display["action_requests"][0]["description"],
"command_visible": "printf" in json.dumps(display["action_requests"][0]["args"]),
"internal_payload_unchanged": pending["payload"] == expected_payload,
"decision_hash_unchanged": pending["payload_hash"] == expected_hash,
"raw_hash_unchanged": content_hash(internal) == raw_hash,
"reviews_compatible": display["review_configs"][0]["allowed_decisions"]
== ["reject"],
}
print(json.dumps(checks), flush=True)
assert all(checks.values()), "public pending confidentiality/identity contract failed"
assert before.values["_verified_review_mode"]["mode"] == "manual"
stage('assertions-done')
asyncio.run(scenario())
stage('asyncio-exit')
if __name__ == "__main__":
repo = Path(__file__).resolve().parents[1]
root = repo.parent / ".hermes/test-runtime/unified-execution/b03-host"
root.mkdir(parents=True, exist_ok=True)
ledger = root / "attempts.jsonl"
name = sys.argv[1]
assert name in NODES
records = [json.loads(x) for x in ledger.read_text().splitlines()] if ledger.exists() else []
attempt = 1 + sum(r.get("phase") == "start" and r["node"] == name for r in records)
assert attempt <= 5
home = root / f"{name}-{attempt}-home"
home.mkdir(parents=True, exist_ok=True)
env = {"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
"PYTHONPATH": str(repo) + os.pathsep + str(repo.parent / "Ai4Sci-Web"),
"PYTHON_DOTENV_DISABLED": "1", "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1",
"EVOSCIENTIST_HOME": str(home), "EVOSCIENTIST_CONFIG_DIR": str(home / "config"),
"XDG_CONFIG_HOME": str(home / "config"), "PYTHONDONTWRITEBYTECODE": "1"}
for key in ("DATA", "WORKSPACE", "SKILLS", "MEMORIES"):
env[f"EVOSCIENTIST_{key}_DIR"] = str(home / key.lower())
cmd = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest", "-p", "no:cacheprovider",
"-q", "-s", "--tb=short", "--disable-warnings", __file__ + "::" + name]
with ledger.open("a") as f:
f.write(json.dumps({"phase": "start", "node": name, "attempt": attempt, "limit": 5,
"command": cmd, "overlap": "B01/B02 fixtures only; new host contract, historical gates not run",
"snapshot": __import__("hashlib").sha256(Path(__file__).read_bytes()).hexdigest()}) + "\n")
try:
result = subprocess.run(cmd, cwd=repo, env=env, capture_output=True, text=True, timeout=180)
output, code = result.stdout + result.stderr, result.returncode
except subprocess.TimeoutExpired as exc:
def text(value):
return value.decode(errors='replace') if isinstance(value, bytes) else (value or '')
output, code = text(exc.stdout) + text(exc.stderr) + '\nstartup/execution timeout\n', 124
output = output.replace('b03-private-canary-7f9a', '[REDACTED]')
log = root / f"{name}-{attempt}.log"
log.write_text(output)
with ledger.open("a") as f:
f.write(json.dumps({"phase": "result", "node": name, "attempt": attempt, "exit": code, "log": str(log)}) + "\n")
print(output)
sys.exit(code)
@@ -0,0 +1,547 @@
"""B02 real resource contracts; run via this file's isolated budget runner."""
from __future__ import annotations
import asyncio
import faulthandler
import json
import os
from pathlib import Path
import signal
import socket
import subprocess
import sys
import threading
from contextlib import asynccontextmanager
import pytest
PLANNED_NODES = (
"test_http_start_without_observer_and_cancel_confirms_eof",
"test_cancel_before_first_observer_never_restarts_graph",
"test_concurrent_cancel_and_cancelled_waiter_preserve_cleanup",
"test_run_timeout_reports_timeout_after_http_exit",
"test_real_execute_cancel_confirms_parent_child_and_worker_exit",
"test_slow_native_worker_retains_run_ownership_until_exit",
"test_owned_clients_close_once_without_closing_borrowed_resources",
"test_parallel_runs_do_not_share_owned_transports",
)
async def build_run(tmp_path, monkeypatch, server):
import importlib
from EvoScientist.config.settings import EvoScientistConfig
from EvoScientist.llm.contracts import AgentInputV3, HmacGrantAuthority, WebHostContext
from EvoScientist.llm.model_config import FileEvoModelConfigStore, EvoModelConfig, endpoint_fingerprint, route_semantics_hash
from EvoScientist.llm.runtime import EvoModelRuntime
from EvoScientist.web_runtime import create_web_agent, web_tool_registry_manifest
from EvoScientist.workspace_files import ScopedFilesystemBackend
from langgraph.checkpoint.memory import InMemorySaver
from tests.v3_fixtures import v3_payload, identity_ring, RUNTIME_SECRET, RUNTIME_KEY_ID
from tests.test_web_model_runtime import _preparation, _admission, _Sink
module = importlib.import_module("EvoScientist.EvoScientist")
monkeypatch.setattr(module, "_load_mcp_tools_cached", lambda **kw: {})
monkeypatch.setattr(module, "_load_mcp_config_once", lambda: ("b02-empty", {}))
monkeypatch.setenv("WEB_RUNTIME_TEST_KEY", "b02-local-dummy")
payload = v3_payload()
payload["providers"]["custom-openai"]["endpoints"][0]["base_url"] = server.url
payload["providers"]["custom-openai"]["models"][0]["context_window"] = 131072
candidate = EvoModelConfig.parse(payload, require_evidence=False)
ring = identity_ring()
for evidence in payload["capability_evidence"]:
route = candidate.concrete_routes("visible-main")[0]
evidence["probe"]["route_semantics_hash"] = route_semantics_hash(candidate, route, ring.derive_current("ai4sci/route-semantics-hash/v3")[1])
evidence["probe"]["endpoint_fingerprint"] = endpoint_fingerprint(candidate, route, ring.derive_current("ai4sci/endpoint-fingerprint/v3")[1])
authority = HmacGrantAuthority(RUNTIME_SECRET, RUNTIME_KEY_ID)
store = FileEvoModelConfigStore(tmp_path / "routes.yaml", admin_verifier=authority)
store.bootstrap_for_development(payload)
config = EvoScientistConfig(enable_async_subagents=False, enable_scheduler=False, memory_workers_enabled=False, auto_approve=False)
def model_factory(**kwargs):
model = EvoModelRuntime._default_model_factory(**kwargs)
server.models.append(model)
assert model.streaming is True and model.disable_streaming is False
return model
runtime = EvoModelRuntime(store, admission_verifier=authority, quote_authority=authority,
model_factory=model_factory,
identity_key_ring=ring, agent_factory=lambda snapshot, host, models:
create_web_agent(snapshot=snapshot, host=host, model_set=models, config=config))
root = tmp_path / "files"
memory = tmp_path / "memory"
root.mkdir()
memory.mkdir()
tools, revision = web_tool_registry_manifest()
sink = _Sink()
host = WebHostContext(str(root), str(memory), ScopedFilesystemBackend(root), InMemorySaver(),
tool_selector_threshold=10000, tool_registry=tools,
tool_registry_revision=revision, runtime_event_sink=sink)
agent_input = AgentInputV3("Wait for a cancellation probe.", "b02:isolated:thread")
quote = await runtime.prepare_model_run(_preparation(authority, agent_input, title_policy="disabled"), agent_input, host)
run = await runtime.start_web_run(_admission(authority, quote))
return run, sink
def test_http_start_without_observer_and_cancel_confirms_eof(tmp_path, monkeypatch, isolated_network):
faulthandler.dump_traceback_later(45, repeat=False)
async def scenario():
async with LoopbackSSE().serve() as server:
run, sink = await build_run(tmp_path, monkeypatch, server)
try:
try:
await asyncio.wait_for(server.requested.wait(), 5)
except TimeoutError:
pytest.fail("accepted start_web_run did not start HTTP without observer")
assert not server.errors
observer = run.stream()
chunks = []
async with asyncio.timeout(5):
async for event in observer:
if event.payload.get("type") == "text":
chunks.append(event.payload.get("content", ""))
if "B02 incremental" in "".join(chunks):
break
assert len(chunks) >= 2, chunks
await observer.aclose()
assert not server.peer_eof.is_set(), "observer owns execution"
assert await run.cancel("b02-http") == "cancelled"
await asyncio.wait_for(run.wait_stopped(), 4)
await asyncio.wait_for(server.peer_eof.wait(), 2)
assert run._agent_task.done()
terminals = [e for e in sink.events if e.payload.get("kind") == "run_terminal"]
assert len(terminals) == 1 and terminals[0].payload["outcome"] == "cancelled"
assert len(server.requests) == 1
print(json.dumps({"evidence": "stream_cancel", "stream": server.requests[0]["stream"],
"chunks": chunks, "peer_eof": server.peer_eof.is_set(),
"agent_done": run._agent_task.done(), "terminals": len(terminals)}), flush=True)
finally:
if run._agent_task is not None and not run._agent_task.done():
run._agent_task.cancel()
await asyncio.gather(run._agent_task, return_exceptions=True)
asyncio.run(scenario())
print("B02 asyncio.run exited; awaiting interpreter teardown", flush=True)
@pytest.fixture
def isolated_network(monkeypatch):
# These must be set BEFORE interpreter startup to prevent import-time leaks.
home = Path(os.environ["HOME"]).resolve()
assert "unified-execution" in str(home)
assert Path(os.environ["EVOSCIENTIST_HOME"]).resolve() == home
assert os.environ.get("PYTHON_DOTENV_DISABLED") == "1"
assert os.environ.get("PYTEST_DISABLE_PLUGIN_AUTOLOAD") == "1"
assert Path(os.environ["EVOSCIENTIST_CONFIG_DIR"]).resolve().is_relative_to(home)
secret_names = [k for k in os.environ if any(
part in k.upper() for part in ("API_KEY", "TOKEN", "SECRET", "PROXY")
)]
assert not secret_names, f"runner inherited credential/proxy keys: {secret_names}"
original_connect = socket.socket.connect
original_connect_ex = socket.socket.connect_ex
original_lookup = socket.getaddrinfo
def require_loopback(address):
assert isinstance(address, tuple) and address[0] == "127.0.0.1", (
f"non-loopback network attempt: {address!r}"
)
def connect(sock, address):
require_loopback(address)
return original_connect(sock, address)
def connect_ex(sock, address):
require_loopback(address)
return original_connect_ex(sock, address)
def lookup(host, *args, **kwargs):
assert host == "127.0.0.1", f"external DNS prohibited: {host!r}"
return original_lookup(host, *args, **kwargs)
monkeypatch.setattr(socket.socket, "connect", connect)
monkeypatch.setattr(socket.socket, "connect_ex", connect_ex)
monkeypatch.setattr(socket, "getaddrinfo", lookup)
class LoopbackSSE:
"""Actual HTTP/1.1 stream, with peer EOF recorded before server teardown."""
def __init__(self, *, tool_call=False):
self.tool_call = tool_call
self.requested = asyncio.Event()
self.peer_eof = asyncio.Event()
self.requests = []
self.tasks = set()
self.writers = set()
self.errors = []
self.models = []
async def handle(self, reader, writer):
task = asyncio.current_task()
self.tasks.add(task)
self.writers.add(writer)
try:
header = await asyncio.wait_for(reader.readuntil(b"\r\n\r\n"), 5)
headers = dict(line.split(b":", 1) for line in header.split(b"\r\n")[1:] if b":" in line)
length = int(next((v for k, v in headers.items() if k.lower() == b"content-length"), b"0"))
assert 0 < length < 2_000_000
body = json.loads(await reader.readexactly(length))
assert header.startswith(b"POST /v1/chat/completions ")
assert body.get("stream") is True, "upstream model HTTP must stream"
assert body.get("tools"), "streaming must retain tools"
self.requests.append(body)
writer.write(b"HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nConnection: close\r\n\r\n")
delta: dict = {"role": "assistant", "content": ""}
if self.tool_call:
assert any(t.get("function", {}).get("name") == "execute" for t in body["tools"])
delta["tool_calls"] = [{"index": 0, "id": "b02-execute", "type": "function", "function": {
"name": "execute", "arguments": json.dumps({"command": "b02-heartbeat"}),
}}]
chunk = {"id": "b02-local", "object": "chat.completion.chunk", "created": 1,
"model": body["model"], "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\n")
await writer.drain()
if not self.tool_call:
for text in ("B02 ", "incremental"):
chunk["choices"][0]["delta"] = {"content": text}
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\n")
await writer.drain()
await asyncio.sleep(.02)
if self.tool_call:
chunk["choices"] = [{"index": 0, "delta": {}, "finish_reason": "tool_calls"}]
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\ndata: [DONE]\n\n")
await writer.drain()
self.requested.set()
if self.tool_call:
return
assert await reader.read() == b""
self.peer_eof.set()
except Exception as exc:
self.errors.append(exc)
self.requested.set()
finally:
writer.close()
await writer.wait_closed()
self.writers.discard(writer)
self.tasks.discard(task)
@asynccontextmanager
async def serve(self):
server = await asyncio.start_server(self.handle, "127.0.0.1", 0)
self.url = f"http://127.0.0.1:{server.sockets[0].getsockname()[1]}/v1"
try:
yield self
finally:
# Emergency teardown is deliberately separate from peer_eof evidence.
closed = set()
emergency_closed = 0
for model in self.models:
for attribute in ("root_async_client", "root_client"):
client = getattr(model, attribute, None)
if client is None or id(client) in closed:
continue
closed.add(id(client))
if client.is_closed():
continue
emergency_closed += 1
result = client.close()
if hasattr(result, "__await__"):
await asyncio.wait_for(result, 3)
assert client.is_closed()
print(json.dumps({"evidence": "fixture_sdk_cleanup", "sdk_wrappers_seen": len(closed),
"emergency_clients_closed": emergency_closed,
"threads": [t.name for t in threading.enumerate()]}), flush=True)
server.close()
await server.wait_closed()
for writer in tuple(self.writers):
writer.close()
tasks = tuple(self.tasks)
for task in tasks:
task.cancel()
await asyncio.gather(*tasks, return_exceptions=True)
class HeartbeatExecutor:
"""Whitelist-only launch fixture; production aexecute/collector stay real."""
def __init__(self, root):
self.root = root
self.process = None
self.started = threading.Event()
self.finished = threading.Event()
self.worker_release = threading.Event()
self.worker_release.set()
self.child_pid = root / "child.pid"
self.heartbeat = root / "heartbeat"
def execute(self, command, *, timeout, cancel_event):
from EvoScientist.native_sandbox import _collect_process
from deepagents.backends.protocol import ExecuteResponse
assert command == "b02-heartbeat"
child = (
"import pathlib,time; p=pathlib.Path('heartbeat'); "
"\nwhile True:\n p.write_text(str(time.monotonic_ns())); time.sleep(.03)\n"
)
parent = (
"import pathlib,signal,subprocess,sys,time\n"
f"p=subprocess.Popen([sys.executable,'-I','-c',{child!r}])\n"
"pathlib.Path('child.pid').write_text(str(p.pid))\n"
"def stop(*args):\n p.wait(timeout=2); sys.exit(0)\n"
"signal.signal(signal.SIGTERM,stop)\n"
"while True: time.sleep(.03)\n"
)
self.process = subprocess.Popen(
[sys.executable, "-I", "-c", parent], cwd=self.root,
env={"HOME": str(self.root), "PATH": "/usr/bin:/bin"},
stdin=subprocess.DEVNULL, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
start_new_session=True, close_fds=True,
)
self.started.set()
try:
raw, code, truncated, _, _ = _collect_process(
self.process, timeout=8, output_limit=4096, cancel_event=cancel_event,
)
# Slow-worker node clears this gate before launch, then releases it
# only after verifying no successful cancellation was committed.
assert self.worker_release.wait(8), "test worker release deadline"
return ExecuteResponse(output=raw.decode(), exit_code=code, truncated=truncated)
finally:
self.finished.set()
async def assert_stopped_before_teardown(self):
assert self.finished.is_set(), "native worker not finished"
assert self.process is not None and self.process.returncode is not None
with pytest.raises(ProcessLookupError):
os.killpg(self.process.pid, 0)
with pytest.raises(ProcessLookupError):
os.kill(int(self.child_pid.read_text()), 0)
before = self.heartbeat.read_bytes()
await asyncio.sleep(.15)
assert self.heartbeat.read_bytes() == before
def emergency_cleanup(self):
self.worker_release.set()
if self.process is not None:
try:
os.killpg(self.process.pid, signal.SIGKILL)
except ProcessLookupError:
pass
self.process.wait(timeout=3)
def test_cancel_before_first_observer_never_restarts_graph(tmp_path, monkeypatch, isolated_network):
async def scenario():
async with LoopbackSSE().serve() as server:
run, sink = await build_run(tmp_path, monkeypatch, server)
assert await run.cancel("before-observer") == "cancelled"
events = [event async for event in run.stream()]
await asyncio.sleep(.1)
assert len([e for e in events if e.payload.get("kind") == "run_terminal"]) == 1
assert not any(e.payload.get("kind") == "run_started" for e in sink.events)
assert not server.requests
assert run._agent_task is None or run._agent_task.done()
asyncio.run(scenario())
def test_slow_native_worker_retains_run_ownership_until_exit(tmp_path, isolated_network):
from EvoScientist.native_sandbox import NativeWorkspaceBackend
async def scenario():
executor = HeartbeatExecutor(tmp_path)
executor.worker_release.clear()
backend = object.__new__(NativeWorkspaceBackend)
backend._executor = executor
task = asyncio.create_task(backend.aexecute("b02-heartbeat"))
try:
async with asyncio.timeout(5):
while not executor.heartbeat.exists() or not executor.child_pid.exists():
await asyncio.sleep(.02)
task.cancel()
await asyncio.sleep(2.2)
assert not task.done(), "cancel abandoned an unconfirmed native worker"
assert not executor.finished.is_set()
executor.worker_release.set()
with pytest.raises(asyncio.CancelledError):
await asyncio.wait_for(task, 3)
await executor.assert_stopped_before_teardown()
finally:
executor.emergency_cleanup()
await asyncio.gather(task, return_exceptions=True)
asyncio.run(scenario())
def test_owned_clients_close_once_without_closing_borrowed_resources(tmp_path, monkeypatch, isolated_network):
async def scenario():
async with LoopbackSSE().serve() as server:
run, sink = await build_run(tmp_path, monkeypatch, server)
await asyncio.wait_for(server.requested.wait(), 5)
assert not server.errors
clients = {}
counts = {}
for model in server.models:
for name in ("root_client", "root_async_client"):
client = getattr(model, name)
transport = client._client
key = id(transport)
if key in clients:
continue
clients[key] = transport
counts[key] = 0
original = transport.aclose if name == "root_async_client" else transport.close
if name == "root_async_client":
async def close(original=original, key=key):
counts[key] += 1
await original()
monkeypatch.setattr(transport, "aclose", close)
else:
def close(original=original, key=key):
counts[key] += 1
original()
monkeypatch.setattr(transport, "close", close)
borrowed_closed = []
for resource in (run._host.checkpointer, run._host.workspace_backend):
monkeypatch.setattr(resource, "close", lambda: borrowed_closed.append(True), raising=False)
# LangChain resource aliases refer to the same SDK transport.
assert await run.cancel("owned-resource-contract") == "cancelled"
assert await run.wait_stopped() == "cancelled"
assert all(client.is_closed for client in clients.values()), "run left SDK transports open"
assert set(counts.values()) == {1}, counts
assert not borrowed_closed
print(json.dumps({"evidence": "production_owned_cleanup", "counts": list(counts.values()),
"borrowed_closed": borrowed_closed}), flush=True)
asyncio.run(scenario())
def test_concurrent_cancel_and_cancelled_waiter_preserve_cleanup(tmp_path, monkeypatch, isolated_network):
async def scenario():
async with LoopbackSSE().serve() as server:
run, sink = await build_run(tmp_path, monkeypatch, server)
await asyncio.wait_for(server.requested.wait(), 5)
transport = server.models[0].root_async_client._client
original = transport.aclose
entered, release = asyncio.Event(), asyncio.Event()
calls = []
async def delayed_close():
calls.append(True)
entered.set()
await release.wait()
await original()
monkeypatch.setattr(transport, "aclose", delayed_close)
cancellations = [asyncio.create_task(run.cancel(str(i))) for i in range(4)]
try:
await asyncio.wait_for(entered.wait(), 4)
waiter = asyncio.create_task(run.wait_stopped(timeout=None))
await asyncio.sleep(0)
waiter.cancel()
cancellations[0].cancel()
await asyncio.gather(waiter, cancellations[0], return_exceptions=True)
assert await run.wait_stopped(timeout=.02) == "unknown"
assert id(transport) in run._owned_clients
assert run._terminal_event is None
assert not run._cleanup_task.done()
assert not transport.is_closed
release.set()
results = await asyncio.gather(*cancellations[1:])
assert results == ["cancelled"] * 3
assert calls == [True]
assert not run._owned_clients
terminals = [e for e in sink.events if e.payload.get("kind") == "run_terminal"]
assert len(terminals) == 1
print(json.dumps({"evidence": "cleanup_race", "bounded_wait": "unknown",
"retained_until_close": True, "close_calls": len(calls),
"terminals": len(terminals), "results": results}), flush=True)
finally:
release.set()
await asyncio.gather(*cancellations, return_exceptions=True)
asyncio.run(scenario())
def test_run_timeout_reports_timeout_after_http_exit(tmp_path, monkeypatch, isolated_network):
from dataclasses import replace
async def scenario():
async with LoopbackSSE().serve() as server:
run, sink = await build_run(tmp_path, monkeypatch, server)
# start_web_run schedules the task; replace before yielding to it.
run._snapshot = replace(run._snapshot, active_run_timeout_seconds=.5)
assert await run.wait_stopped(timeout=5) == "failed"
await asyncio.wait_for(server.peer_eof.wait(), 2)
assert server.requests and not server.errors
assert run._terminal_event.payload["error_code"] == "RUN_TIMEOUT"
assert not run._owned_clients
assert all(m.root_async_client.is_closed() and m.root_client.is_closed() for m in server.models)
print(json.dumps({"evidence": "run_timeout", "error_code": run._terminal_event.payload["error_code"],
"peer_eof": server.peer_eof.is_set(), "resources_remaining": len(run._owned_clients)}), flush=True)
asyncio.run(scenario())
def test_parallel_runs_do_not_share_owned_transports(tmp_path, monkeypatch, isolated_network):
async def scenario():
async with LoopbackSSE().serve() as server:
roots = [tmp_path / str(i) for i in range(2)]
for root in roots:
root.mkdir()
first, _ = await build_run(roots[0], monkeypatch, server)
second, _ = await build_run(roots[1], monkeypatch, server)
try:
async with asyncio.timeout(5):
while len(server.requests) < 2:
await asyncio.sleep(.01)
first_ids = set(first._owned_clients)
second_ids = set(second._owned_clients)
assert first_ids.isdisjoint(second_ids), "independent runs share owned transports"
assert await first.cancel("first-only") == "cancelled"
assert all(not client.is_closed for client in second._owned_clients.values())
assert second._terminal_event is None
print(json.dumps({"evidence": "run_transport_isolation", "first": len(first_ids),
"second": len(second_ids), "disjoint": True}), flush=True)
finally:
await first.cancel("teardown")
await second.cancel("teardown")
asyncio.run(scenario())
if __name__ == "__main__":
repo = Path(__file__).resolve().parents[1]
root = repo.parent / ".hermes/test-runtime/unified-execution/b02"
ledger = root / "attempts.jsonl"
node = "tests/test_execution_adapter_stop_contract.py::" + sys.argv[1]
records = [json.loads(line) for line in ledger.read_text().splitlines()]
attempt = 1 + sum(r.get("phase") == "start" and r.get("node") == node for r in records)
assert sys.argv[1] in PLANNED_NODES, "unregistered node"
limit = 8 if sys.argv[1] == "test_http_start_without_observer_and_cancel_confirms_eof" else 3
assert attempt <= limit, "node budget exhausted"
home = root / (sys.argv[1] + f"-{attempt}-home")
config = home / "config"
config.mkdir(parents=True, exist_ok=True)
env = {
"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
"PYTHONPATH": str(repo), "PYTHON_DOTENV_DISABLED": "1",
"PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1", "PYTHONDONTWRITEBYTECODE": "1",
"EVOSCIENTIST_HOME": str(home), "EVOSCIENTIST_CONFIG_DIR": str(config),
"XDG_CONFIG_HOME": str(config), "EVOSCIENTIST_DATA_DIR": str(home / "data"),
"EVOSCIENTIST_WORKSPACE_DIR": str(home / "workspace"),
"EVOSCIENTIST_SKILLS_DIR": str(home / "skills"),
"EVOSCIENTIST_MEMORIES_DIR": str(home / "memory"),
}
command = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest",
"-p", "no:cacheprovider", "-q", "-s", "--tb=short", "--disable-warnings", node]
with ledger.open("a") as handle:
if attempt == 1:
handle.write(json.dumps({"phase": "registered", "node": node, "used": 0, "limit": limit, "command": command}) + "\n")
handle.write(json.dumps({"phase": "start", "node": node, "attempt": attempt, "command": command}) + "\n")
try:
result = subprocess.run(command, cwd=repo, env=env, capture_output=True, text=True, timeout=60)
except subprocess.TimeoutExpired as exc:
output = exc.stdout or b""
if isinstance(output, bytes):
output = output.decode(errors="replace")
result = subprocess.CompletedProcess(command, 124, output, "runner timeout; child killed and reaped\n")
log = root / (sys.argv[1] + f"-{attempt}.log")
log.write_text(result.stdout + result.stderr)
with ledger.open("a") as handle:
handle.write(json.dumps({"phase": "result", "node": node, "attempt": attempt,
"exit_code": result.returncode, "log": str(log), "remaining": limit-attempt}) + "\n")
print(result.stdout + result.stderr)
sys.exit(result.returncode)
@@ -0,0 +1,325 @@
"""B03 auxiliary Graph/SDK contracts; isolated, budgeted nodes only."""
from __future__ import annotations
import asyncio
import json
import os
from pathlib import Path
import subprocess
import sys
if __name__ != "__main__":
from tests.test_execution_adapter_stop_contract import (
LoopbackSSE, build_run, isolated_network,
)
else:
LoopbackSSE = object
PLANNED_NODES = (
"test_selector_title_graph_stream_and_account",
"test_summarizer_graph_stream_and_account",
"test_summary_main_profile_thresholds",
"test_selector_schema_input_bound",
)
def test_summary_main_profile_thresholds():
from importlib import import_module
from langchain_core.language_models.fake_chat_models import FakeListChatModel
from deepagents.backends import CompositeBackend, StateBackend
module = import_module("EvoScientist.EvoScientist")
main = FakeListChatModel(responses=["main"], profile={"max_input_tokens": 10000})
for profile in ({"max_input_tokens": 100000}, None):
summary = FakeListChatModel(responses=["summary"], profile=profile)
middleware = module._create_run_summarization_middleware(
main, CompositeBackend(default=StateBackend, routes={}, artifacts_root="/workspace"), summary)
assert middleware._lc_helper.trigger == ("tokens", 8500)
assert middleware._lc_helper.keep == ("tokens", 1000)
assert middleware._lc_helper.trim_tokens_to_summarize is None
assert middleware._get_profile_limits() == (100000 if profile else None)
assert middleware._truncate_args_trigger == ("tokens", 8500)
assert middleware._truncate_args_keep == ("tokens", 1000)
print("main thresholds=8500/1000; summarizer profile independent")
def test_selector_schema_input_bound(monkeypatch):
from types import SimpleNamespace
from langchain_core.messages import HumanMessage
from pydantic import BaseModel, Field
from EvoScientist.llm.runtime import _RuntimeAttemptCallback, _provider_input_token_bound, _callback_messages_payload
class Selection(BaseModel):
tools: list[str] = Field(description="candidate tool description " * 1000)
messages = [[HumanMessage(content="select a tool")]]
minimum = _provider_input_token_bound({"messages": _callback_messages_payload(messages),
"response_format": Selection.model_json_schema()}).total_tokens
class BoundaryReached(Exception):
pass
async def begin(**kwargs):
assert kwargs["provider_input_bound_tokens"] >= minimum
raise BoundaryReached
callback = _RuntimeAttemptCallback(SimpleNamespace(_begin_callback_attempt=begin))
monkeypatch.setattr(callback, "_route_for", lambda metadata: ("tool_selector", None))
async def scenario():
import pytest
with pytest.raises(BoundaryReached):
await callback.on_chat_model_start({}, messages, run_id="schema-bound",
invocation_params={"response_format": Selection})
asyncio.run(scenario())
print(f"selector response_format schema included in input bound: >= {minimum}")
class CompletingSSE(LoopbackSSE):
async def handle(self, reader, writer):
task = asyncio.current_task()
self.tasks.add(task)
self.writers.add(writer)
try:
header = await asyncio.wait_for(reader.readuntil(b"\r\n\r\n"), 5)
headers = dict(line.split(b":", 1) for line in header.split(b"\r\n")[1:] if b":" in line)
length = int(next(v for k, v in headers.items() if k.lower() == b"content-length"))
body = json.loads(await reader.readexactly(length))
self.requests.append(body)
assert header.startswith(b"POST /v1/chat/completions ")
assert body.get("stream") is True
assert body.get("stream_options", {}).get("include_usage") is True
tools = body.get("tools", [])
selection = next((t for t in tools if t.get("function", {}).get("name") == "ToolSelectionResponse"), None)
is_selector = selection is not None or bool(body.get("response_format"))
text = '{"tools":["read_file"]}' if is_selector else "B03 streamed completion"
writer.write(b"HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nConnection: close\r\n\r\n")
for index, part in enumerate((text[:8], text[8:])):
delta = {"content": part}
if selection:
delta = {"tool_calls": [{"index": 0, "function": {"arguments": part}}]}
if index == 0:
delta["tool_calls"][0].update(id="select-b03", type="function")
delta["tool_calls"][0]["function"]["name"] = "ToolSelectionResponse"
if index == 0:
delta["role"] = "assistant"
chunk = {"id": "b03-local", "object": "chat.completion.chunk", "created": 1,
"model": body["model"], "choices": [{"index": 0, "delta": delta, "finish_reason": None}]}
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\n")
await writer.drain()
await asyncio.sleep(.02)
chunk["choices"] = [{"index": 0, "delta": {}, "finish_reason": "tool_calls" if selection else "stop"}]
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\n")
chunk["choices"] = []
chunk["usage"] = {"prompt_tokens": 13, "completion_tokens": 7, "total_tokens": 20,
"prompt_tokens_details": {"cached_tokens": 3}}
writer.write(b"data: " + json.dumps(chunk).encode() + b"\n\ndata: [DONE]\n\n")
await writer.drain()
except Exception as exc:
self.errors.append(repr(exc))
finally:
writer.close()
await writer.wait_closed()
self.writers.discard(writer)
self.tasks.discard(task)
def configure_graph(monkeypatch, *, summarize=False):
from dataclasses import replace
import EvoScientist.web_runtime as web
import tests.test_web_model_runtime as helpers
from langchain_core.messages import HumanMessage, AIMessage
original_factory = web.create_web_agent
original_preparation = helpers._preparation
def factory(**kwargs):
kwargs["host"] = replace(kwargs["host"], tool_selector_threshold=10000 if summarize else 0)
if summarize:
kwargs["model_set"].main_agent.profile = {"max_input_tokens": 10000}
kwargs["model_set"].deepagents_summarizer.profile = {"max_input_tokens": 100000}
graph = original_factory(**kwargs)
if summarize:
graph.update_state({"configurable": {"thread_id": "b02:isolated:thread"}}, {
"messages": [HumanMessage(content="Earlier request " * 800),
AIMessage(content="Earlier answer " * 800),
HumanMessage(content="Another request " * 800),
AIMessage(content="Another answer " * 800)]})
return graph
def preparation(authority, agent_input, **kwargs):
kwargs["title_policy"] = "disabled" if summarize else "best_effort"
return original_preparation(authority, agent_input, **kwargs)
monkeypatch.setattr(web, "create_web_agent", factory)
monkeypatch.setattr(helpers, "_preparation", preparation)
def assert_accounting(server, sink, expected):
attempts = [e.payload for e in sink.events if e.kind == "model_attempt"]
completed = [e for e in attempts if e["outcome"] == "succeeded"]
print(json.dumps({"evidence": "b03_auxiliary", "requests": [
{"stream": r.get("stream"), "stream_options": r.get("stream_options"),
"tools": [t.get("function", {}).get("name") for t in r.get("tools", [])],
"response_format": r.get("response_format"),
"message_prefix": str(r.get("messages", []))[:220]} for r in server.requests],
"attempts": attempts}, default=str), flush=True)
assert not server.errors, server.errors
assert [e["purpose"] for e in completed] == expected
assert len(server.requests) == len(completed)
assert len({e["attempt_id"] for e in completed}) == len(completed)
for event in completed:
starts = [e for e in attempts if e["attempt_id"] == event["attempt_id"] and e["outcome"] == "started"]
assert len(starts) == 1
assert event["attempt_index"] == 1
assert event["billing_intent"] == ("user_charge" if event["purpose"] == "main_agent" else "platform_cost")
assert event["usage_available"] is True
assert event["usage"]["input_tokens"] == 13
assert event["usage"]["output_tokens"] == 7
assert event["usage"]["cached_input_tokens"] == 3
def test_selector_title_graph_stream_and_account(tmp_path, monkeypatch, isolated_network):
configure_graph(monkeypatch)
from EvoScientist.llm.runtime import _RuntimeAttemptCallback
chunks = {}
original_token = _RuntimeAttemptCallback.on_llm_new_token
async def token_probe(self, token, *, run_id, **kwargs):
chunks.setdefault(str(run_id), []).append(token)
await original_token(self, token, run_id=run_id, **kwargs)
monkeypatch.setattr(_RuntimeAttemptCallback, "on_llm_new_token", token_probe)
async def scenario():
async with CompletingSSE().serve() as server:
run, sink = await build_run(tmp_path, monkeypatch, server)
try:
assert await run.wait_stopped(timeout=20) == "completed"
assert_accounting(server, sink, ["tool_selector", "main_agent", "title"])
print(json.dumps({"sdk_incremental_callbacks": chunks}), flush=True)
assert len(chunks) == 3
assert all(len([token for token in values if token]) >= 2 for values in chunks.values())
assert any(e.kind == "title" and e.payload.get("title") == "B03 streamed completion" for e in sink.events)
finally:
await run.cancel("fixture-teardown")
asyncio.run(scenario())
def test_summarizer_graph_stream_and_account(tmp_path, monkeypatch, isolated_network):
configure_graph(monkeypatch, summarize=True)
from importlib import import_module
module = import_module("EvoScientist.EvoScientist")
original_summary_factory = module._create_run_summarization_middleware
diagnostics = []
def summary_factory(model, backend, summarizer):
middleware = original_summary_factory(model, backend, summarizer)
original_wrap = middleware.awrap_model_call
async def wrap_probe(request, handler):
history = middleware._get_effective_messages(request)
tokens = middleware._count_tokens(history, request.system_message, request.tools)
truncated, modified = middleware._truncate_args(history, tokens)
cutoff = middleware._determine_cutoff_index(truncated)
pending, _ = middleware._partition_messages(truncated, cutoff)
trimmed = middleware._lc_helper._trim_messages_for_summary(pending)
evidence = {
"trigger": middleware._lc_helper.trigger,
"keep": middleware._lc_helper.keep,
"trim": middleware._lc_helper.trim_tokens_to_summarize,
"history_messages": len(history),
"history_tokens": middleware._count_tokens(history, None, None),
"total_tokens": tokens, "cutoff": cutoff,
"summary_messages": len(pending), "trimmed_messages": len(trimmed),
"summary_profile_limit": middleware._get_profile_limits(),
}
diagnostics.append(evidence)
print(json.dumps({"middleware_evidence": evidence}), flush=True)
assert evidence["trigger"] == ("tokens", 8500)
assert evidence["keep"] == ("tokens", 1000)
assert evidence["trim"] is None
assert evidence["summary_profile_limit"] == 100000
assert len(history) >= 4 and evidence["history_tokens"] > 8500
assert not modified
assert middleware._should_summarize(truncated, tokens)
assert cutoff > 0 and pending and trimmed
return await original_wrap(request, handler)
middleware.awrap_model_call = wrap_probe
return middleware
monkeypatch.setattr(module, "_create_run_summarization_middleware", summary_factory)
from EvoScientist.llm.runtime import _RuntimeAttemptCallback
chunks = {}
original_token = _RuntimeAttemptCallback.on_llm_new_token
async def token_probe(self, token, *, run_id, **kwargs):
chunks.setdefault(str(run_id), []).append(token)
await original_token(self, token, run_id=run_id, **kwargs)
monkeypatch.setattr(_RuntimeAttemptCallback, "on_llm_new_token", token_probe)
async def scenario():
async with CompletingSSE().serve() as server:
run, sink = await build_run(tmp_path, monkeypatch, server)
try:
outcome = await run.wait_stopped(timeout=20)
assert_accounting(server, sink, ["deepagents_summarizer", "main_agent"])
assert diagnostics
print(json.dumps({"sdk_incremental_callbacks": chunks}), flush=True)
assert len(chunks) == 2
assert all(len([token for token in values if token]) >= 2 for values in chunks.values())
assert outcome == "completed"
assert "B03 streamed completion" in str(server.requests[-1]["messages"])
path = "/workspace/conversation_history/b02:isolated:thread.md"
backend = run._host.workspace_backend
original = backend.download_files([path])[0]
assert original.error is None, original.error
assert b"Earlier request " * 800 in original.content
readable = backend.read(path)
assert readable.error is None, readable.error
assert "Earlier request " in str(readable.file_data)
assert path in str(server.requests[-1]["messages"])
print(json.dumps({"offload_path": path, "raw_bytes": len(original.content),
"raw_original_verified": True, "backend_read_verified": True}))
finally:
await run.cancel("fixture-teardown")
asyncio.run(scenario())
if __name__ == "__main__":
repo = Path(__file__).resolve().parents[1]
root = repo.parent / ".hermes/test-runtime/unified-execution/b03-auxiliary"
root.mkdir(parents=True, exist_ok=True)
ledger = root / "attempts.jsonl"
node = "tests/test_execution_adapter_web_contract.py::" + sys.argv[1]
assert sys.argv[1] in PLANNED_NODES
records = [json.loads(line) for line in ledger.read_text().splitlines()] if ledger.exists() else []
attempt = 1 + sum(r.get("phase") == "start" and r.get("node") == node for r in records)
limit = 7 if sys.argv[1] == "test_summarizer_graph_stream_and_account" else 5
if limit == 7:
assert any(r.get("phase") == "budget_extended" and r.get("node") == node
and r.get("limit") == limit for r in records), "budget extension not recorded"
assert attempt <= limit, "node budget exhausted"
home = root / (sys.argv[1] + f"-{attempt}-home")
config = home / "config"
config.mkdir(parents=True, exist_ok=True)
env = {"HOME": str(home), "PATH": str(repo / ".venv/bin") + ":/usr/bin:/bin",
"PYTHONPATH": str(repo), "PYTHON_DOTENV_DISABLED": "1", "PYTEST_DISABLE_PLUGIN_AUTOLOAD": "1",
"PYTHONDONTWRITEBYTECODE": "1", "EVOSCIENTIST_HOME": str(home),
"EVOSCIENTIST_CONFIG_DIR": str(config), "XDG_CONFIG_HOME": str(config)}
for key, folder in (("DATA", "data"), ("WORKSPACE", "workspace"), ("SKILLS", "skills"), ("MEMORIES", "memory")):
env[f"EVOSCIENTIST_{key}_DIR"] = str(home / folder)
command = [str(repo / ".venv/bin/python"), "-m", "pytest", "--noconftest", "-p", "no:cacheprovider",
"-q", "-s", "--tb=short", "--disable-warnings", node]
with ledger.open("a") as handle:
if attempt == 1:
handle.write(json.dumps({"phase": "registered", "node": node, "used": 0, "limit": 5,
"intent": "RED missing wiring or first-pass compatibility", "command": command}) + "\n")
handle.write(json.dumps({"phase": "start", "node": node, "attempt": attempt, "command": command}) + "\n")
try:
result = subprocess.run(command, cwd=repo, env=env, capture_output=True, text=True, timeout=50)
except subprocess.TimeoutExpired as exc:
output = exc.stdout or b""
if isinstance(output, bytes):
output = output.decode(errors="replace")
result = subprocess.CompletedProcess(command, 124, output, "runner timeout; child killed and reaped\n")
log = root / (sys.argv[1] + f"-{attempt}.log")
log.write_text(result.stdout + result.stderr)
with ledger.open("a") as handle:
handle.write(json.dumps({"phase": "result", "node": node, "attempt": attempt,
"exit_code": result.returncode, "log": str(log), "remaining": limit-attempt}) + "\n")
print(result.stdout + result.stderr)
sys.exit(result.returncode)
+376
View File
@@ -0,0 +1,376 @@
"""B02 resource failure nodes.
Budget ledger (test executions, not the runtime's three close attempts):
- Original construction failure [nth_model, agent, get_chat_model]: 4/5.
- Original close failure [2, 99]: 4/5.
- Original borrowed and explicit-transfer nodes: 3/5 (not rerun here).
- Hanging close [sync, async]: 5/5 each (budget exhausted; do not rerun).
- Construction hanging close retirement: 2/5 (RED, GREEN).
Execution log after explicit runtime handoff, all commands used uv run pytest:
1. Hanging close pair -q: exit 1, 2 failed in 6.37s; sync blocked loop,
async starved the other resource. Watchdog/teardown release was not evidence.
2. Hanging close pair -q: exit 0, 2 passed in 1.19s.
3. Construction hanging retirement -q: exit 1, 1 failed in 0.48s;
completed cleanup did not retire the preparation.
4. Construction hanging retirement + original close failure pair + original
construction triple -q: exit 0, 6 passed in 0.66s.
5. Hanging close pair -q: exit 0, 2 passed in 1.19s.
6. Public cancellation regression, pair execution 4/5:
uv run --no-sync pytest tests/test_execution_resource_failures.py::test_hanging_close_isolated_and_recoverable -q
exit 1, 2 failed in 0.70s. After deliberate release and empty owned_clients,
public cancel("retry") awaited the permanently failed stop task and raised
ExceptionGroup containing TimeoutError from _ensure_cleanup (250ms).
This reproduces the supplied deleg_3a10c94b review finding.
7. Same command, pair execution 5/5: exit 0, 2 passed in 4.25s.
Public cancel returns unknown while close hangs; wait_stopped short timeout
returns unknown; caller cancellation preserves the same stop owner. After
deliberate release, cancel and wait_stopped return cancelled; public stream
and committed events contain exactly one run_terminal, and each close ran
once. No internal cleanup helper is used as recovery evidence.
No approval, HTTP, full-suite, DB or external-service tests executed.
"""
import asyncio
import threading
from types import SimpleNamespace as NS
import pytest
from EvoScientist.llm.contracts import AgentModelSet
from EvoScientist.llm.runtime import EvoModelRuntime, _EvoWebRun
class Client:
def __init__(self, failures=0):
self.calls = 0
self.failures = failures
async def aclose(self):
self.calls += 1
if self.calls <= self.failures:
raise RuntimeError("close failed")
def setup_runtime(monkeypatch):
runtime = object.__new__(EvoModelRuntime)
runtime.admission_verifier = NS(require_admission=lambda admission: None)
runtime._registry_lock = asyncio.Lock()
runtime._started_grants = {}
runtime._validate_admission_echo = lambda *args: None
runtime._validate_still_fresh = lambda *args: None
admission = NS(grant_id="g", preparation_id="p", reasoning_effort="", provider_run_reserve_microunits=0, unsigned_payload=lambda: {})
handle = NS(lock=asyncio.Lock(), state="PREPARED", run=None,
quote=NS(expires_at=10**20), snapshot=NS(),
input=NS(metadata={}), host=NS(runtime_event_sink=object()))
runtime._prepared = {"p": handle}
runtime._prepared_tombstones = {}
runtime._prepared_requests = {("s", "r", "t"): "p"}
handle.grant = NS(subject_id="s", request_id="r", turn_id="t")
handle.request_digest = "request-digest"
handle.quote.preparation_id = "p"
monkeypatch.setattr(_EvoWebRun, "_start", lambda self: None)
return runtime, admission, handle
@pytest.mark.parametrize("failure", ["nth_model", "agent", "get_chat_model"])
def test_construction_failure_rolls_back_transports(monkeypatch, failure):
async def scenario():
import httpx
from EvoScientist.llm import models
runtime, admission, handle = setup_runtime(monkeypatch)
clients = []
def client():
result = Client()
clients.append(result)
return result
monkeypatch.setattr(httpx, "Client", client)
monkeypatch.setattr(httpx, "AsyncClient", client)
def chat(**kwargs):
if failure == "get_chat_model":
raise ValueError("construction")
return NS()
monkeypatch.setattr(models, "get_chat_model", chat)
if failure == "nth_model":
route = NS(identity=NS(model_id="test"), runtime_provider="openai", invocation_plan=object())
handle.snapshot.purpose_routes = {"main_agent": (route, route)}
handle.snapshot.purpose_route_call_bounds = {"main_agent": (NS(route_identity=route.identity),)}
runtime._model_factory_kwargs = lambda route: {}
runtime._attach_route_metadata = lambda model, *args: model
calls = 0
def factory(**kwargs):
nonlocal calls
calls += 1
if calls == 2:
raise ValueError("construction")
return runtime._default_model_factory(**kwargs)
runtime.model_factory = factory
def build(*args):
if failure == "nth_model":
return EvoModelRuntime._build_model_set(runtime, *args)
model = runtime._default_model_factory(provider="openai")
return AgentModelSet(model, model, model)
runtime._build_model_set = build
def agent(*args):
raise ValueError("construction")
runtime.agent_factory = agent
with pytest.raises(ValueError, match="construction"):
await runtime.start_web_run(admission)
assert clients and all(item.calls == 1 for item in clients)
assert handle.state != "STARTED"
assert not runtime._started_grants
asyncio.run(scenario())
@pytest.mark.parametrize("close_kind", ["sync", "async"])
def test_hanging_close_isolated_and_recoverable(monkeypatch, close_kind):
async def scenario():
runtime, admission, handle = setup_runtime(monkeypatch)
run = _EvoWebRun(runtime=runtime, admission=admission, prepared=handle,
agent=None, model_set=AgentModelSet(None, None, None))
handle.run = run
handle.state = "STARTED"
runtime.runtime_instance_id = "resource-test"
for name in ("request_id", "turn_id", "admission_snapshot_id",
"admission_id", "hold_id", "turn_fencing_token",
"billing_fencing_token"):
setattr(admission, name, name)
committed = []
class Sink:
async def commit(self, event):
committed.append(event)
return "committed"
handle.host.runtime_event_sink = Sink()
release_sync = threading.Event()
release_async = asyncio.Event()
started = threading.Event()
watchdog_fired = threading.Event()
loop = asyncio.get_running_loop()
class HangingSync:
calls = 0
def close(self):
self.calls += 1
started.set()
release_sync.wait()
class HangingAsync:
calls = 0
async def aclose(self):
self.calls += 1
started.set()
await release_async.wait()
hanging = HangingSync() if close_kind == "sync" else HangingAsync()
good = Client()
run._owned_clients.update({id(c): c for c in (hanging, good)})
pending = []
def emergency_release():
watchdog_fired.set()
release_sync.set()
loop.call_soon_threadsafe(release_async.set)
# A blocked event loop cannot execute an asyncio timeout or finally.
watchdog = threading.Timer(5.0, emergency_release)
watchdog.start()
try:
cleanup = asyncio.create_task(run.cancel("resource-test"))
pending.append(cleanup)
for _ in range(100):
if started.is_set() and good.calls == 1:
break
await asyncio.sleep(0.01)
assert not watchdog_fired.is_set(), "sync close blocked the event loop"
assert started.is_set(), "hanging resource close was not started"
assert good.calls == 1, "hanging close starved another owned resource"
assert id(good) not in run._owned_clients
assert run._owned_clients.get(id(hanging)) is hanging
assert handle.run is run
assert run._terminal_event is None
# Public observation must be bounded; the run owns finalization.
done, _ = await asyncio.wait({cleanup}, timeout=3.0)
assert cleanup in done, "cleanup observation must be bounded"
first_error = cleanup.exception()
owner = run._cancel_task
assert owner is not None and not owner.done()
assert await run.wait_stopped(timeout=0.01) == "unknown"
with pytest.raises(TimeoutError):
await asyncio.wait_for(run.cancel("short-observer"), timeout=0.01)
assert not owner.done(), "caller timeout must not end the stop owner"
assert run._terminal_event is None
assert run._state != "TERMINAL"
assert hanging.calls == 1, "pending close must never be restarted"
assert run._owned_clients.get(id(hanging)) is hanging
assert not watchdog_fired.is_set()
# Only this deliberate release is recovery evidence. Teardown
# releases below cannot turn an earlier failure into a pass.
release_sync.set()
release_async.set()
for _ in range(100):
if not run._owned_clients:
break
await asyncio.sleep(0.01)
assert not run._owned_clients, "released close was not reconciled"
assert await asyncio.wait_for(run.cancel("retry"), timeout=1.0) == "cancelled"
assert await run.wait_stopped(timeout=1.0) == "cancelled"
assert first_error is None
assert cleanup.result() == "unknown"
assert run._cancel_task is owner
assert run._state == "TERMINAL"
events = [event async for event in run.stream()]
assert len(events) == len(committed) == 1
assert events[0].payload["kind"] == "run_terminal"
assert events[0].payload["outcome"] == "cancelled"
assert await run.cancel("duplicate") == "cancelled"
assert len(committed) == 1
assert hanging.calls == good.calls == 1
assert not watchdog_fired.is_set()
finally:
watchdog.cancel()
watchdog.join()
release_sync.set()
release_async.set()
if run._cleanup_task is not None:
pending.append(run._cleanup_task)
if pending:
await asyncio.wait_for(
asyncio.gather(*pending, return_exceptions=True), timeout=2.0
)
asyncio.run(scenario())
def test_construction_hanging_close_retains_owner_until_retirement(monkeypatch):
# New distinct node: registered 0/5 before its first execution.
async def scenario():
from EvoScientist.llm.runtime import _construction_owner
runtime, admission, handle = setup_runtime(monkeypatch)
runtime._prepared_tombstones = {}
runtime._prepared_requests = {("s", "r", "t"): "p"}
handle.grant = NS(subject_id="s", request_id="r", turn_id="t")
handle.request_digest = "request-digest"
handle.quote.preparation_id = "p"
release = asyncio.Event()
good = Client()
class Hanging:
calls = 0
async def aclose(self):
self.calls += 1
await release.wait()
hanging = Hanging()
def build(*args):
owner = _construction_owner.get()
owner._owned_clients.update({id(c): c for c in (hanging, good)})
raise ValueError("construction")
runtime._build_model_set = build
start = asyncio.create_task(runtime.start_web_run(admission))
try:
done, _ = await asyncio.wait({start}, timeout=1.0)
assert start in done, "construction rollback held handle lock indefinitely"
assert start.exception() is not None
assert handle.state == "CONSTRUCTION_FAILED"
assert not handle.lock.locked()
owner = handle.run
assert owner is not None
assert runtime._prepared["p"] is handle
assert owner._owned_clients.get(id(hanging)) is hanging
assert good.calls == hanging.calls == 1
assert owner._terminal_event is None
with pytest.raises(Exception):
await runtime.start_web_run(admission)
release.set()
await asyncio.wait_for(asyncio.shield(owner._cleanup_task), timeout=1.0)
assert not owner._owned_clients
assert "p" not in runtime._prepared
assert ("s", "r", "t") in runtime._prepared_tombstones
with pytest.raises(Exception):
await runtime.start_web_run(admission)
assert hanging.calls == 1
finally:
release.set()
await asyncio.gather(start, return_exceptions=True)
if handle.run is not None and handle.run._cleanup_task is not None:
await asyncio.wait_for(
asyncio.shield(handle.run._cleanup_task), timeout=1.0
)
asyncio.run(scenario())
@pytest.mark.parametrize("failures", [2, 99])
def test_close_failures_are_isolated_and_bounded(monkeypatch, failures):
async def scenario():
runtime, admission, handle = setup_runtime(monkeypatch)
run = _EvoWebRun(runtime=runtime, admission=admission, prepared=handle,
agent=None, model_set=AgentModelSet(None, None, None))
bad, good, other = Client(failures), Client(), Client(failures)
run._owned_clients.update({id(c): c for c in (bad, good, other)})
caught = None
try:
await run._ensure_cleanup()
except BaseException as exc:
caught = exc
assert good.calls == 1, "one failure must not skip remaining resources"
assert bad.calls == other.calls == 3
assert len(run._cleanup_failures) == 2
assert all(len(errors) == min(failures, 3) for errors in run._cleanup_failures.values())
if failures == 2:
assert caught is None
assert not run._owned_clients
await run._ensure_cleanup()
else:
assert isinstance(caught, ExceptionGroup)
assert len(caught.exceptions) == 2
assert len(run._owned_clients) == 2
with pytest.raises(ExceptionGroup):
await run._terminal_locked("completed")
assert run._terminal_event is None
assert bad.calls == other.calls == 3
asyncio.run(scenario())
def test_borrowed_model_clients_are_not_closed(monkeypatch):
async def scenario():
runtime, admission, handle = setup_runtime(monkeypatch)
shared = Client()
model = NS(root_async_client=NS(_client=shared))
run = _EvoWebRun(runtime=runtime, admission=admission, prepared=handle,
agent=None, model_set=AgentModelSet(model, model, model))
await run._ensure_cleanup()
assert shared.calls == 0
asyncio.run(scenario())
def test_custom_factory_explicit_resource_transfer(monkeypatch):
async def scenario():
from EvoScientist.llm import contracts
result_type = getattr(contracts, "ModelFactoryResult", None)
assert result_type is not None, "explicit factory ownership contract missing"
runtime, admission, handle = setup_runtime(monkeypatch)
owned, borrowed = Client(), Client()
model = NS(root_async_client=borrowed)
runtime.model_factory = lambda **kwargs: result_type(model, (owned,))
runtime._model_factory_kwargs = lambda route: {}
runtime._attach_route_metadata = lambda model, *args: model
route = NS(identity=NS(model_id="test"), runtime_provider="test", invocation_plan=object())
handle.snapshot.purpose_route_call_bounds = {"main_agent": (NS(route_identity=route.identity),)}
def build(*args):
built = runtime._build_model(route, "main_agent", handle.snapshot)
assert built is model
return AgentModelSet(built, built, built)
runtime._build_model_set = build
runtime.agent_factory = lambda *args: object()
run = await runtime.start_web_run(admission)
await run._ensure_cleanup()
assert owned.calls == 1
assert borrowed.calls == 0
asyncio.run(scenario())
+115
View File
@@ -60,6 +60,7 @@ def test_runtime_error_repr_preserves_only_stable_code():
@pytest.mark.anyio
async def test_astream_yields_chunks_from_sse(monkeypatch):
monkeypatch.setenv("AI4SCI_EVO_RUNTIME_GRANT_SECRET", "runtime-service-secret")
monkeypatch.delenv("EVOSCIENTIST_BACKEND_SERVICE_TOKEN", raising=False)
model = GatewayProxyChatModel(
gateway_url="http://gw",
run_id="run-1",
@@ -115,6 +116,120 @@ async def test_astream_roundtrips_streaming_tool_call_chunks(monkeypatch):
assert chunks[0].message.tool_call_chunks[0]["id"] == "call-1"
@pytest.mark.anyio
async def test_astream_roundtrips_public_reasoning_summary(monkeypatch):
model = GatewayProxyChatModel(
gateway_url="http://gw",
run_id="run-1",
envelope_signature="sig",
)
msg = {"type": "AIMessageChunk", "data": {"content": []}}
lines = [
f"data: {json.dumps({'delta': {'message': msg, 'reasoning_summary': 'Checked sources.'}})}\n",
"data: [DONE]\n",
]
fake = _FakeClient(lines)
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: fake)
chunks = [c async for c in model._astream([HumanMessage(content="hi")])]
assert chunks[0].message.content == [{
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "Checked sources."}],
}]
assert "reasoning_content" not in chunks[0].message.additional_kwargs
@pytest.mark.anyio
async def test_astream_does_not_duplicate_summary_already_in_message(monkeypatch):
model = GatewayProxyChatModel(
gateway_url="http://gw",
run_id="run-1",
envelope_signature="sig",
)
block = {
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "Checked sources."}],
}
msg = {"type": "AIMessageChunk", "data": {"content": [block]}}
lines = [
f"data: {json.dumps({'delta': {'message': msg, 'reasoning_summary': 'Checked sources.'}})}\n",
"data: [DONE]\n",
]
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: _FakeClient(lines))
chunks = [c async for c in model._astream([HumanMessage(content="hi")])]
assert isinstance(chunks[0].message.content, list)
summaries = [
item for item in chunks[0].message.content
if isinstance(item, dict) and item.get("type") == "reasoning"
]
assert summaries == [block]
@pytest.mark.anyio
async def test_astream_appends_only_cumulative_summary_suffix(monkeypatch):
model = GatewayProxyChatModel(
gateway_url="http://gw",
run_id="run-1",
envelope_signature="sig",
)
prefix = {
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "Checked"}],
}
msg = {"type": "AIMessageChunk", "data": {"content": [prefix]}}
lines = [
f"data: {json.dumps({'delta': {'message': msg, 'reasoning_summary': 'Checked sources.'}})}\n",
"data: [DONE]\n",
]
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: _FakeClient(lines))
chunks = [c async for c in model._astream([HumanMessage(content="hi")])]
assert isinstance(chunks[0].message.content, list)
summary = "".join(
str(part.get("text") or "")
for item in chunks[0].message.content
if isinstance(item, dict) and item.get("type") == "reasoning"
for part in (item.get("summary") or [])
if isinstance(part, dict) and part.get("type") == "summary_text"
)
assert summary == "Checked sources."
@pytest.mark.anyio
async def test_astream_preserves_non_cumulative_summary_delta(monkeypatch):
model = GatewayProxyChatModel(
gateway_url="http://gw",
run_id="run-1",
envelope_signature="sig",
)
first = {
"type": "reasoning",
"summary": [{"type": "summary_text", "text": "Checked source A. "}],
}
msg = {"type": "AIMessageChunk", "data": {"content": [first]}}
lines = [
f"data: {json.dumps({'delta': {'message': msg, 'reasoning_summary': 'Compared source B.'}})}\n",
"data: [DONE]\n",
]
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: _FakeClient(lines))
chunks = [c async for c in model._astream([HumanMessage(content="hi")])]
assert isinstance(chunks[0].message.content, list)
summary = "".join(
str(part.get("text") or "")
for item in chunks[0].message.content
if isinstance(item, dict) and item.get("type") == "reasoning"
for part in (item.get("summary") or [])
if isinstance(part, dict) and part.get("type") == "summary_text"
)
assert summary == "Checked source A. Compared source B."
@pytest.mark.anyio
async def test_astream_raises_on_missing_done(monkeypatch):
model = GatewayProxyChatModel(
+59
View File
@@ -0,0 +1,59 @@
"""Real loopback HTTP cancellation, without a provider or business run."""
import asyncio
import pytest
from langchain_core.messages import HumanMessage
from langgraph.graph import END, START, StateGraph
from EvoScientist.llm.gateway_proxy import GatewayProxyChatModel
@pytest.mark.anyio
async def test_model_cancel_closes_http_sse_reader(monkeypatch):
monkeypatch.setenv("AI4SCI_EVO_RUNTIME_GRANT_SECRET", "test-only-secret")
connected = asyncio.Event()
disconnected = asyncio.Event()
async def serve(reader, writer):
try:
headers = await reader.readuntil(b"\r\n\r\n")
length = next(int(line.split(b":", 1)[1]) for line in headers.splitlines()
if line.lower().startswith(b"content-length:"))
await reader.readexactly(length)
writer.write(b"HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\n"
b"Transfer-Encoding: chunked\r\n\r\n")
await writer.drain()
connected.set()
assert await reader.read() == b""
disconnected.set()
finally:
writer.close()
await writer.wait_closed()
server = await asyncio.start_server(serve, "127.0.0.1", 0)
port = server.sockets[0].getsockname()[1]
model = GatewayProxyChatModel(gateway_url=f"http://127.0.0.1:{port}",
run_id="test", envelope_signature="test")
async def consume(_state):
async for _ in model.astream([HumanMessage(content="test")]):
pytest.fail("silent test stream should produce no model output")
return {}
builder = StateGraph(dict)
builder.add_node("model", consume)
builder.add_edge(START, "model")
builder.add_edge("model", END)
task = asyncio.create_task(builder.compile().ainvoke({}))
try:
await asyncio.wait_for(connected.wait(), 3)
task.cancel()
with pytest.raises(asyncio.CancelledError):
await task
await asyncio.wait_for(disconnected.wait(), 3)
finally:
task.cancel()
await asyncio.gather(task, return_exceptions=True)
server.close()
await server.wait_closed()
@@ -0,0 +1,349 @@
"""Real installed SDK + isolated HTTP transport; SSE is simulated, not Google evidence."""
import asyncio
import json
from types import SimpleNamespace
import httpx
import pytest
from google import genai
from google.genai import types
from langchain_core.messages import HumanMessage
from EvoScientist.llm.gemini_interactions import (
GeminiInteractionsChatModel,
create_gemini_interactions_model,
)
EVENTS = [
{
"event_type": "content.start",
"index": 0,
"content": {"type": "text", "text": ""},
},
{
"event_type": "content.delta",
"index": 0,
"delta": {"type": "text", "text": "hello"},
},
{
"event_type": "content.delta",
"index": 0,
"delta": {"type": "text", "text": " world"},
},
{"event_type": "content.stop", "index": 0},
{
"event_type": "content.start",
"index": 1,
"content": {
"type": "function_call",
"id": "call1",
"name": "lookup",
"arguments": {"q": "test"},
},
},
{"event_type": "content.stop", "index": 1},
{
"event_type": "interaction.complete",
"interaction": {
"id": "fixture1",
"status": "completed",
"model": "gemini-test",
"usage": {
"total_input_tokens": 4,
"total_cached_tokens": 0,
"total_output_tokens": 2,
"total_tokens": 6,
},
},
},
]
class Wire(httpx.SyncByteStream, httpx.AsyncByteStream):
def __init__(self, events, block=False):
self.events = events
self.block = block
self.closed = False
self.delivered = 0
self.waiting = asyncio.Event()
def __iter__(self):
for event in self.events:
self.delivered += 1
yield ("data: " + json.dumps(event) + "\n\n").encode()
async def __aiter__(self):
for chunk in self:
yield chunk
if self.block:
self.waiting.set()
await asyncio.Event().wait()
def close(self):
self.closed = True
async def aclose(self):
self.closed = True
def setup_wire(monkeypatch, events=EVENTS, block=False, status=200):
wire = Wire(events, block)
requests = []
clients = []
def handle(request):
requests.append(json.loads(request.content))
return httpx.Response(
status, headers={"content-type": "text/event-stream"}, stream=wire
)
def client(_self):
transport = httpx.MockTransport(handle)
sdk = genai.Client(
api_key="local-fixture-not-a-credential",
http_options=types.HttpOptions(
base_url="https://isolated.invalid",
client_args={"transport": transport},
async_client_args={"transport": transport},
),
)
clients.append(sdk)
return sdk
monkeypatch.setattr(GeminiInteractionsChatModel, "_client", client)
model = create_gemini_interactions_model(model="gemini-test", api_key="fixture")
return model, wire, requests, clients
async def consume(model, mode):
if mode == "stream":
async for _ in model._astream([HumanMessage(content="hi")]):
pass
elif mode == "sync":
model.invoke("hi")
else:
await model.ainvoke("hi")
async def test_truncated_eof_is_protocol_error(monkeypatch):
for mode in ("stream", "sync", "async"):
model, wire, requests, clients = setup_wire(monkeypatch, EVENTS[:3])
with pytest.raises(RuntimeError, match=r"^MODEL_PROVIDER_PROTOCOL_ERROR$"):
await consume(model, mode)
assert requests[0]["stream"] is True
assert wire.closed
assert clients[0]._api_client._httpx_client.is_closed
async def test_stream_completion_close_and_original_errors(monkeypatch):
from google.genai._interactions import BadRequestError
model, wire, requests, clients = setup_wire(monkeypatch)
chunks = [chunk async for chunk in model._astream([HumanMessage(content="hi")])]
assert [chunk.message.content for chunk in chunks[:2]] == ["hello", " world"]
assert chunks[-1].message.usage_metadata["total_tokens"] == 6
assert chunks[-2].message.tool_calls[0]["args"] == {"q": "test"}
assert requests[0]["stream"] is True
assert wire.closed
assert clients[0]._api_client._async_httpx_client.is_closed
model, wire, _, clients = setup_wire(monkeypatch)
stream = model._astream([HumanMessage(content="hi")])
await anext(stream)
await stream.aclose()
assert wire.closed
assert clients[0]._api_client._async_httpx_client.is_closed
for mode in ("stream", "sync", "async"):
for status, events, error in (
(400, [], BadRequestError),
(
200,
[{"event_type": "error", "error": {"message": "fixture"}}],
RuntimeError,
),
):
model, wire, requests, clients = setup_wire(
monkeypatch, events, status=status
)
with pytest.raises(error) as caught:
await consume(model, mode)
if status == 400:
assert caught.value.status_code == 400
else:
assert str(caught.value) == "MODEL_PROVIDER_PROTOCOL_ERROR"
assert requests[0]["stream"] is True
assert wire.closed
assert clients[0]._api_client._httpx_client.is_closed
if mode != "sync":
assert clients[0]._api_client._async_httpx_client.is_closed
def assert_result(result):
assert result.content[0] == {"type": "text", "text": "hello world"}
assert result.tool_calls[0]["args"] == {"q": "test"}
assert result.usage_metadata["total_tokens"] == 6
assert result.additional_kwargs["gemini_interaction_content"][1]["id"] == "call1"
def test_sync_invoke_uses_sdk_stream(monkeypatch):
model, wire, requests, clients = setup_wire(monkeypatch)
result = model.invoke([HumanMessage(content="hi")])
assert requests[0]["stream"] is True
assert_result(result)
assert wire.closed
assert clients[0]._api_client._httpx_client.is_closed
async def test_async_invoke_aggregates_sdk_stream(monkeypatch):
model, wire, requests, clients = setup_wire(monkeypatch)
result = await model.ainvoke([HumanMessage(content="hi")])
assert requests[0]["stream"] is True
assert_result(result)
assert wire.closed
assert clients[0]._api_client._async_httpx_client.is_closed
assert clients[0]._api_client._httpx_client.is_closed
async def test_incremental_stream_cancellation_closes_owned_clients(monkeypatch):
model, wire, requests, clients = setup_wire(monkeypatch, EVENTS[:3], block=True)
stream = model._astream([HumanMessage(content="hi")])
first = await anext(stream)
assert first.message.content == "hello"
assert wire.delivered == 2
assert requests[0]["stream"] is True
assert (await anext(stream)).message.content == " world"
task = asyncio.ensure_future(anext(stream))
await asyncio.wait_for(wire.waiting.wait(), 2)
task.cancel()
with pytest.raises(asyncio.CancelledError):
await task
assert wire.closed
assert clients[0]._api_client._async_httpx_client.is_closed
assert clients[0]._api_client._httpx_client.is_closed
@pytest.mark.parametrize("mode", ["sync", "async", "stream"])
@pytest.mark.parametrize("failure", ["protocol", "cancel", "success"])
async def test_cleanup_failures_preserve_primary_and_attempt_all(
monkeypatch, caplog, mode, failure
):
"""Independent fault injection, not a replay of transport success tests."""
primary = {
"protocol": RuntimeError("MODEL_PROVIDER_PROTOCOL_ERROR"),
"cancel": asyncio.CancelledError("primary cancellation"),
"success": None,
}[failure]
attempts = []
secret = "cleanup-secret-must-not-be-logged"
def fail_close(resource):
attempts.append(resource)
raise OSError(secret)
class BrokenStream:
def __iter__(self):
yield from EVENTS
if primary is not None:
raise primary
async def __aiter__(self):
for event in self:
yield event
def close(self):
fail_close("stream")
class AsyncBrokenStream:
__aiter__ = BrokenStream.__aiter__
__iter__ = BrokenStream.__iter__
async def close(self):
fail_close("stream")
async def create_async(**request):
assert request["stream"] is True
return AsyncBrokenStream()
def create_sync(**request):
assert request["stream"] is True
return BrokenStream()
async def close_async():
fail_close("async_client")
client = SimpleNamespace(
interactions=SimpleNamespace(create=create_sync),
aio=SimpleNamespace(
interactions=SimpleNamespace(create=create_async), aclose=close_async
),
close=lambda: fail_close("client"),
)
monkeypatch.setattr(GeminiInteractionsChatModel, "_client", lambda self: client)
model = create_gemini_interactions_model(model="gemini-test", api_key="fixture")
expected = type(primary) if primary is not None else RuntimeError
operation = (
model._agenerate([HumanMessage(content="hi")])
if mode == "async" and failure == "cancel"
else consume(model, mode)
)
with pytest.raises(expected) as caught:
await operation
if primary is not None:
assert caught.value is primary
else:
assert str(caught.value) == "MODEL_PROVIDER_CLEANUP_ERROR"
resources = (
["stream", "client"] if mode == "sync" else ["stream", "async_client", "client"]
)
assert attempts == resources
records = [r for r in caplog.records if r.name.endswith("gemini_interactions")]
assert len(records) == len(resources)
assert all(r.exc_info is None for r in records)
assert secret not in caplog.text
for resource, record in zip(resources, records, strict=True):
assert (
record.getMessage() == f"MODEL_PROVIDER_CLEANUP_ERROR resource={resource}"
)
@pytest.mark.parametrize("mode", ["async", "stream"])
async def test_public_task_cancellation_with_cleanup_failure(monkeypatch, caplog, mode):
model, wire, requests, clients = setup_wire(monkeypatch, EVENTS[:3], block=True)
attempts = []
original_sync = genai.Client.close
original_async = genai.client.AsyncClient.aclose
def close_sync(self):
attempts.append("client")
original_sync(self)
raise OSError("synthetic-cleanup-secret")
async def close_async(self):
attempts.append("async_client")
await original_async(self)
raise OSError("synthetic-cleanup-secret")
monkeypatch.setattr(genai.Client, "close", close_sync)
monkeypatch.setattr(genai.client.AsyncClient, "aclose", close_async)
async def public_operation():
if mode == "async":
await model.ainvoke("hi")
else:
async for _ in model.astream("hi"):
pass
task = asyncio.create_task(public_operation())
await asyncio.wait_for(wire.waiting.wait(), 2)
task.cancel("public-cancel-marker")
with pytest.raises(asyncio.CancelledError, match="public-cancel-marker"):
await task
assert task.cancelled()
assert attempts == ["async_client", "client"]
assert requests[0]["stream"] is True
assert wire.closed
assert clients[0]._api_client._async_httpx_client.is_closed
assert clients[0]._api_client._httpx_client.is_closed
assert "synthetic-cleanup-secret" not in caplog.text
+59
View File
@@ -0,0 +1,59 @@
"""Local durable identity only; no resource-exit inference."""
import importlib
import importlib.util
import json
from pathlib import Path
import os
import pytest
def budget(node):
root = Path(__file__).resolve().parents[2] / ".hermes/test-runtime/unified-execution/b04-registry"
root.mkdir(parents=True, exist_ok=True)
ledger = root / (node + ".jsonl")
used = len(ledger.read_text().splitlines()) if ledger.exists() else 0
assert used < 5, "cumulative node budget exhausted"
with ledger.open("a") as out:
out.write(json.dumps({"node": node, "attempt": used + 1, "limit": 5}) + "\n")
out.flush()
os.fsync(out.fileno())
def test_binding_survives_reopen_as_unknown(tmp_path):
name = "EvoScientist.llm.host_execution_registry"
assert importlib.util.find_spec(name) is not None, "durable host registry missing"
Registry = importlib.import_module(name).SQLiteHostRegistry
first = Registry(tmp_path / "host.sqlite", host_id="host", boot_id="boot-a")
first.bind(execution_id="actual-run", grant_id="grant", digest="digest",
thread_id="thread", turn_id="turn")
recovered = Registry(tmp_path / "host.sqlite", host_id="host", boot_id="boot-b")
evidence = recovered.inspect("actual-run")
assert evidence["status"] == "unknown"
assert evidence["boot_id"] == "boot-a"
assert evidence["resources_confirmed_exited"] is False
assert recovered.lookup_grant("grant", "digest")["execution_id"] == "actual-run"
def test_identity_conflicts_and_old_epoch_control(tmp_path):
budget("identity_conflicts_and_old_epoch_control")
from EvoScientist.llm.host_execution_registry import SQLiteHostRegistry
from EvoScientist.llm.contracts import EvoRuntimeError
registry = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="a")
binding = dict(execution_id="run", grant_id="grant", digest="digest",
thread_id="thread", turn_id="turn")
registry.bind(**binding)
with pytest.raises(EvoRuntimeError, match="EXECUTION_IDENTITY_CONFLICT"):
registry.bind(**{**binding, "digest": "different"})
with pytest.raises(EvoRuntimeError, match="TURN_EXECUTION_UNKNOWN"):
registry.bind(**{**binding, "execution_id": "other", "grant_id": "other"})
assert registry.transfer_control("run", expected_epoch=1, new_epoch=2) == 2
with pytest.raises(EvoRuntimeError, match="OWNER_EPOCH_STALE"):
registry.transfer_control("run", expected_epoch=1, new_epoch=3)
with pytest.raises(EvoRuntimeError, match="OWNER_EPOCH_STALE"):
registry.require_control("run", owner_epoch=1)
registry.require_control("run", owner_epoch=2)
recovered = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="b")
assert recovered.inspect("run")["owner_epoch"] == 2
assert recovered.inspect("run")["boot_id"] == "a"
assert recovered.inspect("run")["status"] == "unknown"
assert recovered.lookup_grant("grant", "digest")["execution_id"] == "run"
+328
View File
@@ -0,0 +1,328 @@
"""B04 local control slices; run only explicitly budgeted nodes."""
import asyncio
import threading
import sqlite3
import pytest
@pytest.mark.asyncio
async def test_terminal_release(tmp_path, monkeypatch):
runtime, registry, run = await live_run(tmp_path, monkeypatch)
closing, release = asyncio.Event(), asyncio.Event()
class Client:
async def aclose(self):
closing.set()
await release.wait()
client = Client()
run._owned_clients[id(client)] = client
cancel = asyncio.create_task(runtime.cancel(
run.run_id, reason="done", owner_epoch=1, boot_id="a"))
try:
await asyncio.wait_for(closing.wait(), 2)
assert registry.inspect(run.run_id)["resources_confirmed_exited"] is False
finally:
release.set()
assert await cancel == "cancelled"
reopened = SQLiteHostRegistry(registry.path, host_id="host", boot_id="b")
evidence = reopened.inspect(run.run_id)
assert evidence["status"] == "cancelled"
assert evidence["resources_confirmed_exited"] is True
with sqlite3.connect(registry.path) as db:
assert db.execute("SELECT count(*) FROM active_claims").fetchone()[0] == 0
original = reopened.lookup_grant(run._admission.grant_id,
evidence["digest"])
assert original["execution_id"] == run.run_id
with pytest.raises(EvoRuntimeError, match="EXECUTION_IDENTITY_CONFLICT"):
reopened.bind(execution_id=run.run_id, grant_id="new", digest="new",
thread_id="other", turn_id="other")
from EvoScientist.llm.contracts import EvoRuntimeError, WebHostContext
from EvoScientist.llm.host_execution_registry import SQLiteHostRegistry
from tests.test_web_model_runtime import _runtime, _input, _preparation, _admission, _Sink
@pytest.mark.asyncio
async def test_precommit(tmp_path, monkeypatch):
runtime, authority = _runtime(tmp_path, monkeypatch)
registry = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="a")
runtime.host_registry = registry
value = _input()
quote = await runtime.prepare_model_run(
_preparation(authority, value, title_policy="disabled"), value,
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()))
assert quote.execution_id
assert quote.unsigned_payload()["execution_id"] == quote.execution_id
entered = []
original = runtime.agent_factory
def construct(*args):
assert registry.inspect(quote.execution_id)["execution_id"] == quote.execution_id
assert registry.inspect(quote.execution_id)["source"] == "host_binding_only"
entered.append(quote.execution_id)
return original(*args)
runtime.agent_factory = construct
run = await runtime.start_web_run(_admission(authority, quote))
assert run.run_id == quote.execution_id
assert entered == [quote.execution_id]
await run.cancel("test", owner_epoch=1, boot_id="a")
def test_pending_continuation(tmp_path):
registry = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="a")
registry.bind(execution_id="parent", grant_id="g1", digest="d1",
thread_id="thread", turn_id="turn")
registry.finish("parent", outcome="awaiting_input", checkpoint_id="cp1")
child = dict(execution_id="child", grant_id="g2", digest="d2",
thread_id="thread", turn_id="turn")
with pytest.raises(EvoRuntimeError, match="CONTINUATION_REQUIRED"):
registry.bind(**child)
with pytest.raises(EvoRuntimeError, match="CONTINUATION_INVALID"):
registry.bind(**child, predecessor_execution_id="parent", predecessor_checkpoint_id="wrong")
registry.bind(**child, predecessor_execution_id="parent", predecessor_checkpoint_id="cp1")
assert registry.inspect("parent")["status"] == "awaiting_input"
registry.finish("child", outcome="completed")
with pytest.raises(EvoRuntimeError, match="CONTINUATION_INVALID"):
registry.bind(**{**child, "execution_id": "third", "grant_id": "g3"},
predecessor_execution_id="parent", predecessor_checkpoint_id="cp1")
def test_restart_cancel_intent(tmp_path):
registry = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="a")
registry.bind(execution_id="run", grant_id="grant", digest="digest",
thread_id="thread", turn_id="turn")
registry.accept_cancel("run", owner_epoch=1, boot_id="a", reason="stop")
restarted = SQLiteHostRegistry(registry.path, host_id="host", boot_id="b")
evidence = restarted.inspect("run")
assert evidence["cancel_requested"] is True
assert evidence["recovery_action"] == "inspect_only"
assert evidence["resources_confirmed_exited"] is False
with pytest.raises(EvoRuntimeError, match="EXECUTION_BOOT_MISMATCH"):
restarted.finish("run", outcome="cancelled")
def test_registry_busy(tmp_path):
registry = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="a")
with sqlite3.connect(registry.path) as lock:
lock.execute("BEGIN IMMEDIATE")
with pytest.raises(EvoRuntimeError, match="HOST_REGISTRY_BUSY"):
registry.bind(execution_id="run", grant_id="grant", digest="digest",
thread_id="thread", turn_id="turn")
@pytest.mark.asyncio
async def test_finish_fault(tmp_path, monkeypatch):
from types import SimpleNamespace
for outcome in ("completed", "awaiting_input", "cancelled"):
case = tmp_path / outcome
case.mkdir()
runtime, registry, run = await live_run(case, monkeypatch)
with sqlite3.connect(registry.path) as db:
db.execute("CREATE TRIGGER fail_finish BEFORE INSERT ON terminal_evidence "
"BEGIN SELECT RAISE(ABORT, 'injected finish failure'); END")
run._agent_task.cancel()
await asyncio.gather(run._agent_task, return_exceptions=True)
async def empty(*args, **kwargs):
if False:
yield {}
async def state(*args):
return SimpleNamespace(config={"configurable": {
"thread_id": "thread", "checkpoint_id": "cp"}},
tasks=(), next=(), interrupts=(SimpleNamespace(id="p", value={}),)
if outcome == "awaiting_input" else ())
monkeypatch.setattr("EvoScientist.stream.events.stream_agent_events", empty)
run._agent.aget_state = state
if outcome == "cancelled":
run._request_cancel("fault")
else:
run._agent_task = asyncio.create_task(run._run_agent())
try:
assert await run.wait_stopped(timeout=0.15) == "unknown"
terminals = [e for e in run._host.runtime_event_sink.events
if e.payload.get("kind") == "run_terminal"]
assert len(terminals) == 1
assert terminals[0].payload["outcome"] == outcome
assert registry.inspect(run.run_id)["status"] == "unknown"
finally:
with sqlite3.connect(registry.path) as db:
db.execute("DROP TRIGGER fail_finish")
assert await run.wait_stopped(timeout=3) == outcome
terminals_after = [e for e in run._host.runtime_event_sink.events
if e.payload.get("kind") == "run_terminal"]
assert terminals_after == terminals
assert registry.inspect(run.run_id)["status"] == outcome
with sqlite3.connect(registry.path) as db:
assert db.execute("SELECT count(*) FROM active_claims").fetchone()[0] == 0
@pytest.mark.asyncio
async def test_construction_fault(tmp_path, monkeypatch):
from EvoScientist.llm.runtime import _construction_owner
registry_check = SQLiteHostRegistry(tmp_path / "consumed.sqlite", host_id="host", boot_id="a")
registry_check.bind(execution_id="parent", grant_id="parent", digest="d", thread_id="t", turn_id="t")
registry_check.finish("parent", outcome="awaiting_input", checkpoint_id="cp")
registry_check.bind(execution_id="child", grant_id="child", digest="d", thread_id="t", turn_id="t",
predecessor_execution_id="parent", predecessor_checkpoint_id="cp")
registry_check.finish("child", outcome="failed")
with pytest.raises(EvoRuntimeError, match="CONTINUATION_CONSUMED_FAILURE_REQUIRES_REAUTHORIZATION"):
registry_check.bind(execution_id="retry", grant_id="retry", digest="d", thread_id="t", turn_id="t",
predecessor_execution_id="parent", predecessor_checkpoint_id="cp")
runtime, authority = _runtime(tmp_path, monkeypatch)
registry = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="a")
runtime.host_registry = registry
release = asyncio.Event()
owners = []
class Client:
async def aclose(self):
await release.wait()
def broken(*args):
run = _construction_owner.get()
owners.append(run)
client = Client()
run._owned_clients[id(client)] = client
raise ValueError("construction fault")
runtime.agent_factory = broken
value = _input()
sink = _Sink()
quote = await runtime.prepare_model_run(
_preparation(authority, value, title_policy="disabled"), value,
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink))
admission = _admission(authority, quote)
with pytest.raises((TimeoutError, ValueError)):
await runtime.start_web_run(admission)
run = owners[0]
assert registry.inspect(run.run_id)["status"] == "unknown"
with sqlite3.connect(registry.path) as db:
assert db.execute("SELECT count(*) FROM active_claims").fetchone()[0] == 1
db.execute("CREATE TRIGGER fail_finish BEFORE INSERT ON terminal_evidence "
"BEGIN SELECT RAISE(ABORT, 'injected finish failure'); END")
release.set()
assert await run.wait_stopped(timeout=0.15) == "unknown"
with sqlite3.connect(registry.path) as db:
assert db.execute("SELECT count(*) FROM active_claims").fetchone()[0] == 1
db.execute("DROP TRIGGER fail_finish")
assert await run.wait_stopped(timeout=3) == "failed"
assert registry.inspect(run.run_id)["status"] == "failed"
with sqlite3.connect(registry.path) as db:
assert db.execute("SELECT count(*) FROM active_claims").fetchone()[0] == 0
assert db.execute("SELECT count(*) FROM executions").fetchone()[0] == 1
terminals = [e for e in sink.events if e.payload.get("kind") == "run_terminal"]
assert len(terminals) == 1
assert terminals[0].payload["error_code"] == "RUN_CONSTRUCTION_FAILED"
with pytest.raises(EvoRuntimeError, match="EXECUTION_FAILED_RETRY_REQUIRES_NEW_ADMISSION"):
await runtime.start_web_run(admission)
async def live_run(tmp_path, monkeypatch):
runtime, authority = _runtime(tmp_path, monkeypatch)
registry = SQLiteHostRegistry(tmp_path / "host.sqlite", host_id="host", boot_id="a")
runtime.host_registry = registry
entered = asyncio.Event()
async def events(*args, **kwargs):
entered.set()
await asyncio.Event().wait()
yield {}
monkeypatch.setattr("EvoScientist.stream.events.stream_agent_events", events)
value = _input()
quote = await runtime.prepare_model_run(
_preparation(authority, value, title_policy="disabled"), value,
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
)
run = await runtime.start_web_run(_admission(authority, quote))
await asyncio.wait_for(entered.wait(), 2)
return runtime, registry, run
@pytest.mark.asyncio
async def test_stale_handle(tmp_path, monkeypatch):
runtime, registry, run = await live_run(tmp_path, monkeypatch)
peer = SQLiteHostRegistry(registry.path, host_id="host", boot_id="a")
peer.transfer_control(run.run_id, expected_epoch=1, new_epoch=2)
try:
with pytest.raises(EvoRuntimeError, match="OWNER_EPOCH_REQUIRED"):
await run.cancel("old handle")
with pytest.raises(EvoRuntimeError, match="OWNER_EPOCH_STALE"):
await run.cancel("old owner", owner_epoch=1, boot_id="a")
with pytest.raises(EvoRuntimeError, match="EXECUTION_BOOT_MISMATCH"):
await runtime.cancel(run.run_id, reason="wrong boot", owner_epoch=2, boot_id="b")
with pytest.raises(EvoRuntimeError, match="OWNER_EPOCH_STALE"):
runtime.inspect(run.run_id, owner_epoch=1, boot_id="a")
assert not run._agent_task.done()
assert await runtime.cancel(run.run_id, reason="current", owner_epoch=2, boot_id="a") == "cancelled"
finally:
if not run._agent_task.done():
run._agent_task.cancel()
await asyncio.gather(run._agent_task, return_exceptions=True)
@pytest.mark.asyncio
async def test_transfer_cancel_race(tmp_path, monkeypatch):
runtime, registry, run = await live_run(tmp_path, monkeypatch)
peer = SQLiteHostRegistry(registry.path, host_id="host", boot_id="a")
barrier = threading.Barrier(2)
original = registry.accept_cancel
def racing_accept(*args, **kwargs):
barrier.wait(timeout=2)
return original(*args, **kwargs)
monkeypatch.setattr(registry, "accept_cancel", racing_accept)
def transfer():
barrier.wait(timeout=2)
return peer.transfer_control(run.run_id, expected_epoch=1, new_epoch=2)
worker = asyncio.create_task(asyncio.to_thread(transfer))
await asyncio.sleep(0)
try:
try:
result = await runtime.cancel(run.run_id, reason="racing", owner_epoch=1, boot_id="a")
except EvoRuntimeError as exc:
assert "OWNER_EPOCH_STALE" in str(exc)
result = "stale"
assert await asyncio.wait_for(worker, 3) == 2
monkeypatch.setattr(registry, "accept_cancel", original)
if result == "stale":
assert not run._agent_task.done()
assert await runtime.cancel(run.run_id, reason="current", owner_epoch=2, boot_id="a") == "cancelled"
else:
assert result == "cancelled"
assert run._agent_task.done()
finally:
run._agent_task.cancel()
await asyncio.gather(run._agent_task, worker, return_exceptions=True)
@pytest.mark.asyncio
async def test_accepted_cancel_during_transfer(tmp_path, monkeypatch):
runtime, registry, run = await live_run(tmp_path, monkeypatch)
peer = SQLiteHostRegistry(registry.path, host_id="host", boot_id="a")
closing = asyncio.Event()
release = asyncio.Event()
class Client:
async def aclose(self):
closing.set()
await release.wait()
client = Client()
run._owned_clients[id(client)] = client
cancel = asyncio.create_task(runtime.cancel(
run.run_id, reason="accepted", owner_epoch=1, boot_id="a"))
try:
await asyncio.wait_for(closing.wait(), 2)
assert await asyncio.wait_for(asyncio.to_thread(
peer.transfer_control, run.run_id, expected_epoch=1, new_epoch=2), 1) == 2
with sqlite3.connect(registry.path) as db:
rows = db.execute("SELECT owner_epoch, reason FROM cancel_intents").fetchall()
assert rows == [(1, "accepted")]
with pytest.raises(EvoRuntimeError, match="OWNER_EPOCH_STALE"):
await run.cancel("late", owner_epoch=1, boot_id="a")
assert not cancel.done()
finally:
release.set()
assert await cancel == "cancelled"
+8 -4
View File
@@ -67,20 +67,24 @@ def test_k3_chat_plan_freezes_adapter_compiled_parameters():
plan.sdk_params["max_completion_tokens"] = 1
def test_responses_plan_accepts_native_tools_when_capability_is_enabled():
@pytest.mark.parametrize("purpose", ["main_agent", "title", "tool_selector", "deepagents_summarizer"])
@pytest.mark.parametrize("disable_streaming", [True, "tool_calling", False])
def test_responses_plan_accepts_native_tools_when_capability_is_enabled(purpose, disable_streaming):
plan = compile_invocation_plan(
api_mode="responses",
declared_tool_call_transport="native",
supports_tools=True,
purpose="title",
purpose=purpose,
output_token_limit=1_024,
reasoning_effort="disabled",
runtime_provider="openai",
sdk_params={"max_output_tokens": 1_024, "use_responses_api": True},
sdk_params={"max_output_tokens": 1_024, "use_responses_api": True,
"disable_streaming": disable_streaming, "streaming": False},
)
assert plan.tool_call_transport == "native"
assert plan.streaming is False
assert plan.streaming is True
assert plan.sdk_params["disable_streaming"] is False
def test_plan_rejects_runtime_projection_that_disagrees_with_capabilities():
+225
View File
@@ -0,0 +1,225 @@
"""B03 offline input projection nodes. Each node has a five-run budget.
Run ledger (no historical web-contract nodes are exercised):
malicious_values: 4/5 (prior 3; compatibility regression GREEN)
effective_merge: 4/5 (prior 3; compatibility regression GREEN)
malicious_payload: 3/5 (prior 2; compatibility regression GREEN)
supported_thinking_controls: 4/5 (launcher failure; RED; GREEN; regression GREEN)
legitimate_image_budget: 4/5 (RED; GREEN; regression; image forms GREEN)
Final regression: 5 passed in 1.17s; final image forms: 1 passed in 0.41s.
No selector/summary/history nodes exercised.
Launcher failure: system Python lacks pytest (no node collected); used .venv.
"""
import asyncio
from types import SimpleNamespace
import pytest
from pydantic import BaseModel
from EvoScientist.llm import runtime
def test_supported_thinking_controls():
from EvoScientist.llm.adapter_registry import get_adapter_registry
from langchain_core.messages import HumanMessage
from langchain_openai import ChatOpenAI
from openai._base_client import _merge_mappings
adapter = get_adapter_registry().get("dashscope", "dashscope-v1")
compiled = adapter.compile_runtime_parameters(
"chat_completions", {"reasoning": "high", "reasoning_budget_tokens": 2048}, 4096
)
for controls in (compiled["extra_body"], {"enable_thinking": False},
{"thinking": {"type": "disabled"}}):
model = ChatOpenAI(api_key="offline-not-a-credential", model="offline",
extra_body=controls, use_responses_api=False)
invocation = model._get_invocation_params()
assert runtime._effective_callback_input_parameters(invocation) == {}
payload = model._get_request_payload([HumanMessage(content="local")])
effective = _merge_mappings(payload, payload.pop("extra_body"))
assert all(effective[key] == value for key, value in controls.items())
assert invocation["extra_body"] == controls
for invalid in ({"thinking": {"type": float("nan")}},
{"enable_thinking": object()}, {"unknown_extension": True}):
with pytest.raises(runtime.EvoRuntimeError, match="MODEL_INPUT_PROJECTION_INVALID"):
runtime._effective_callback_input_parameters({"extra_body": invalid})
def test_legitimate_image_budget():
import base64
import io
from PIL import Image
from EvoScientist.document_extract import MAX_IMAGE_BYTES, prepare_image_bytes
from langchain_core.messages import HumanMessage
output = io.BytesIO()
Image.new("RGB", (2048, 2048)).save(output, "PNG", compress_level=0)
raw = output.getvalue()
assert len(raw) > runtime._INPUT_PROJECTION_MAX_BYTES
raw += b"\0" * (MAX_IMAGE_BYTES - len(raw))
assert prepare_image_bytes(raw, "boundary.png") == raw
uri = "data:image/png;base64," + base64.b64encode(raw).decode("ascii")
block = {"type": "image_url", "image_url": {"url": uri}}
payload = {"messages": runtime._callback_messages_payload([
[HumanMessage(content=[block, block])]
])}
bound = runtime._provider_input_token_bound(payload)
assert bound.media_blocks == 2
assert bound.largest_media_bytes == MAX_IMAGE_BYTES
assert bound.media_tokens == 2 * ((len(raw) + 2) // 3 + 512)
assert bound.text_tokens < 4096
small = io.BytesIO()
Image.new("RGB", (1, 1)).save(small, "PNG")
encoded = base64.b64encode(small.getvalue()).decode("ascii")
forms = [
{"type": "image", "mime_type": "image/png", "base64": encoded},
{"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": encoded}},
{"inline_data": {"mime_type": "image/png", "data": encoded}},
{"type": "input_image", "image_url": "data:image/png;base64," + encoded},
]
assert runtime._provider_input_token_bound(forms).media_blocks == len(forms)
assert runtime._provider_input_token_bound([
{"type": "image_url", "image_url": {"url": "https://example.invalid/image.png"}}
]).media_blocks == 0
assert payload["messages"][0][0]["data"]["content"][0]["image_url"]["url"] == uri
invalid = [
{"type": "image_url", "image_url": {"url": "data:image/png;base64,eA=="}},
{"type": "image_url", "image_url": {"url": "data:unknown/fake;base64,eA=="}},
{"type": "text", "base64": uri},
{"type": "image_url", "image_url": {"url": uri + "AAAA"}},
]
for item in invalid:
with pytest.raises(runtime.EvoRuntimeError):
runtime._provider_input_token_bound([item])
for item in ({"schema": block}, {"text": "x" * (8 * 1024 * 1024 + 1)},
{"unknown": "data:image/png;base64,eA=="}):
with pytest.raises(runtime.EvoRuntimeError, match="MODEL_INPUT_PROJECTION_INVALID"):
runtime._provider_input_token_bound(item)
def test_malicious_values_fail_closed():
touched = []
class Duck:
@classmethod
def model_json_schema(cls):
touched.append("schema")
return {}
class Poison:
def __repr__(self):
touched.append("repr")
raise AssertionError("must not format rejected input")
class Selection(BaseModel):
name: str
class BadSchema(BaseModel):
@classmethod
def model_json_schema(cls, *args, **kwargs):
return {"invalid": float("nan")}
cycle = []
cycle.append(cycle)
deep = None
for _ in range(66):
deep = [deep]
cases = [
{1: "non-string key"}, Duck, Poison(), cycle, deep,
float("nan"), float("inf"), -float("inf"),
("tuple",), {"set"}, b"bytes", BadSchema,
[None] * 100_001, "x" * (8 * 1024 * 1024 + 1),
]
failures = []
for index, value in enumerate(cases):
try:
runtime._callback_input_parameters(value)
except runtime.EvoRuntimeError as exc:
if exc.code != "MODEL_INPUT_PROJECTION_INVALID":
failures.append((index, "wrong code"))
except Exception:
failures.append((index, "uncontrolled exception"))
else:
failures.append((index, "accepted"))
assert not failures, failures
assert not touched
assert runtime._callback_input_parameters(Selection) == Selection.model_json_schema()
assert runtime._callback_input_parameters({"safe": [None, True, 1, 1.5, "ok"]}) == {
"safe": [None, True, 1, 1.5, "ok"]
}
shared = {"value": 1}
assert runtime._callback_input_parameters([shared, shared]) == [shared, shared]
def test_effective_merge_callback_bound(monkeypatch):
from langchain_core.messages import HumanMessage
from langchain_openai import ChatOpenAI
from openai._base_client import _merge_mappings
messages = [HumanMessage(content="offline merge conflict")]
defaults = {"tools": [{"type": "function", "function": {"name": "default"}}]}
body = {
"tools": [{"type": "function", "function": {"name": "winner"}}],
"response_format": {"type": "json_object"},
"system": "body system", "instructions": "body instructions",
}
model = ChatOpenAI(api_key="offline-not-a-credential", model_kwargs=defaults,
extra_body=body, use_responses_api=False)
call = {"tools": [{"type": "function", "function": {"name": "call"}}],
"response_format": {"type": "json_schema", "json_schema": {"name": "loser"}},
"system": "call system", "instructions": "call instructions"}
invocation = model._get_invocation_params(**call)
payload = model._get_request_payload(messages, **call)
effective = _merge_mappings(payload, payload.pop("extra_body"))
expected = {key: effective[key] for key in body}
assert expected == body
captured = []
class BoundaryReached(Exception):
pass
async def begin(**kwargs):
captured.append(kwargs)
raise BoundaryReached
callback = runtime._RuntimeAttemptCallback(SimpleNamespace(_begin_callback_attempt=begin))
monkeypatch.setattr(callback, "_route_for", lambda _: ("tool_selector", None))
monkeypatch.setattr(runtime, "_callback_start_failure_details", lambda *args: {})
async def invoke(params):
await callback.on_chat_model_start({}, [messages], run_id="merge", invocation_params=params)
with pytest.raises(BoundaryReached):
asyncio.run(invoke(invocation))
expected_bound = runtime._provider_input_token_bound({
"messages": runtime._callback_messages_payload([messages]), **expected,
}).total_tokens
assert captured[0]["provider_input_bound_tokens"] == expected_bound
assert captured[0]["purpose"] == "tool_selector"
for invalid in (
{"model_kwargs": {"tools": defaults["tools"]}},
{"extra_body": {"input": "unprojected context"}},
{"extra_body": {"unknown_prompt": "unprojected context"}},
{"extra_body": {"messages": []}},
{"tools": {1: "bad key"}},
[],
):
with pytest.raises(runtime.EvoRuntimeError, match="MODEL_INPUT_PROJECTION_INVALID"):
asyncio.run(invoke(invalid))
assert len(captured) == 1
def test_malicious_payload_rejected_before_media_recursion():
cycle = {"content": []}
cycle["content"].append(cycle)
failures = []
for index, value in enumerate((cycle, {"content": float("nan")}, {1: "bad"})):
try:
runtime._provider_input_token_bound(value)
except runtime.EvoRuntimeError as exc:
if exc.code != "MODEL_INPUT_PROJECTION_INVALID":
failures.append((index, "wrong code"))
except Exception:
failures.append((index, "uncontrolled exception"))
else:
failures.append((index, "accepted"))
assert not failures, failures
+110
View File
@@ -0,0 +1,110 @@
"""Cancellation regressions using disposable local process groups only."""
import asyncio
import os
import signal
import subprocess
import sys
import threading
import pytest
import EvoScientist.native_sandbox as sandbox
@pytest.mark.anyio
@pytest.mark.skipif(os.name != "posix", reason="POSIX process groups")
async def test_async_execute_cancellation_waits_for_worker_cleanup(tmp_path, monkeypatch):
files = tmp_path / "files"
runtime = tmp_path / "runtime"
files.mkdir()
runtime.mkdir()
backend = sandbox.NativeWorkspaceBackend(files, runtime, timeout=3)
started = threading.Event()
cleaned = threading.Event()
def execute(_command, *, timeout, cancel_event):
process = subprocess.Popen(
["/bin/sh", "-c", "sleep 30 & wait"],
stdout=subprocess.PIPE, stderr=subprocess.PIPE, start_new_session=True,
)
started.set()
try:
sandbox._collect_process(process, timeout=3, output_limit=1024,
cancel_event=cancel_event)
assert process.poll() is not None
cleaned.set()
finally:
try:
os.killpg(process.pid, signal.SIGKILL)
except ProcessLookupError:
pass
process.wait(timeout=3)
monkeypatch.setattr(backend._executor, "execute", execute)
task = asyncio.create_task(backend.aexecute("sleep 30"))
try:
assert await asyncio.to_thread(started.wait, 3)
task.cancel()
with pytest.raises(asyncio.CancelledError):
await task
assert cleaned.is_set()
finally:
if not task.done():
task.cancel()
await asyncio.gather(task, return_exceptions=True)
@pytest.mark.skipif(os.name != "posix", reason="POSIX process groups")
def test_collector_read_failure_reaps_process(monkeypatch):
process = subprocess.Popen(
[sys.executable, "-c", "import time; print('ready', flush=True); time.sleep(30)"],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
start_new_session=True,
)
def failed_read(*_args):
raise OSError("collector read failed")
try:
with monkeypatch.context() as patcher:
patcher.setattr(sandbox.os, "read", failed_read)
with pytest.raises(OSError, match="collector read failed"):
sandbox._collect_process(process, timeout=3, output_limit=1024)
assert process.poll() is not None, "collector failure left command running"
finally:
try:
os.killpg(process.pid, signal.SIGKILL)
except ProcessLookupError:
pass
process.wait(timeout=3)
@pytest.mark.skipif(os.name != "posix", reason="POSIX process groups")
@pytest.mark.parametrize("cancel", [True, False])
def test_collector_cancel_and_timeout_reap_process_group(cancel):
process = subprocess.Popen(
["/bin/sh", "-c", "sleep 30 & wait"],
stdout=subprocess.PIPE,
stderr=subprocess.PIPE,
start_new_session=True,
)
event = threading.Event()
if cancel:
event.set()
try:
_, code, _, timed_out, cancelled = sandbox._collect_process(
process, timeout=1, output_limit=1024, cancel_event=event
)
assert code == (130 if cancel else 124)
assert (timed_out, cancelled) == (not cancel, cancel)
assert process.poll() is not None
with pytest.raises(ProcessLookupError):
os.killpg(process.pid, 0)
finally:
try:
os.killpg(process.pid, signal.SIGKILL)
except ProcessLookupError:
pass
process.wait(timeout=3)
+128
View File
@@ -0,0 +1,128 @@
import asyncio
import pytest
from tests.test_host_registry_control import _runtime, _input, _preparation, _admission, _Sink
from EvoScientist.llm.contracts import WebHostContext
from EvoScientist.llm.host_execution_registry import SQLiteHostRegistry
from EvoScientist.llm.runtime import _construction_owner
@pytest.mark.asyncio
async def test_owner_observation(tmp_path, monkeypatch, caplog):
runtime, authority = _runtime(tmp_path, monkeypatch)
runtime.host_registry = SQLiteHostRegistry(tmp_path / 'host.db', host_id='h', boot_id='b')
release = asyncio.Event()
owners = []
class Client:
async def aclose(self):
await release.wait()
raise ValueError('SECRET cleanup detail')
def broken(*args):
run = _construction_owner.get()
owners.append(run)
client = Client()
run._owned_clients[id(client)] = client
raise ValueError('construction')
runtime.agent_factory = broken
value = _input()
sink = _Sink()
quote = await runtime.prepare_model_run(
_preparation(authority, value, title_policy='disabled'), value,
WebHostContext('/tmp', '/tmp', object(), object(), runtime_event_sink=sink))
with pytest.raises(TimeoutError):
await runtime.start_web_run(_admission(authority, quote))
run = owners[0]
release.set()
await asyncio.wait({run._terminal_task}, timeout=2)
await asyncio.sleep(0)
try:
state = runtime.inspect(run.run_id, owner_epoch=1, boot_id='b')
assert state['terminal_error_code'] == 'RUN_TERMINAL_CLEANUP_UNCONFIRMED'
assert run._terminal_task._log_traceback is False
assert state['status'] == 'unknown'
assert state['resources_confirmed_exited'] is False
assert 'SECRET' not in caplog.text
assert 'RUN_TERMINAL_CLEANUP_UNCONFIRMED' in caplog.text
assert not sink.events
finally:
run._terminal_task.exception()
@pytest.mark.asyncio
async def test_durable_recovery(tmp_path, monkeypatch):
import hashlib
import json
import sqlite3
from dataclasses import asdict
from EvoScientist.llm.contracts import EvoRuntimeError, canonical_json_v1
from tests.test_host_registry_control import live_run
runtime, registry, run = await live_run(tmp_path, monkeypatch)
run._agent_task.cancel()
await asyncio.gather(run._agent_task, return_exceptions=True)
sink_path = tmp_path / 'sink.db'
class DurableSink:
uncertain = True
def __init__(self):
with sqlite3.connect(sink_path) as db:
db.execute('CREATE TABLE IF NOT EXISTS events (id TEXT PRIMARY KEY, digest TEXT, body TEXT)')
async def commit(self, event):
body = canonical_json_v1(asdict(event))
digest = hashlib.sha256(body).hexdigest()
with sqlite3.connect(sink_path) as db:
prior = db.execute('SELECT digest FROM events WHERE id=?', (event.event_id,)).fetchone()
if prior and prior[0] != digest:
return 'conflict'
db.execute('INSERT OR IGNORE INTO events VALUES (?, ?, ?)',
(event.event_id, digest, body.decode()))
if self.uncertain:
raise OSError('lost commit response')
return 'duplicate' if prior else 'committed'
async def confirm(self, event_id, payload_digest):
if self.uncertain:
raise OSError('confirm unavailable')
with sqlite3.connect(sink_path) as db:
row = db.execute('SELECT digest FROM events WHERE id=?', (event_id,)).fetchone()
return 'absent' if row is None else 'committed' if row[0] == payload_digest else 'conflict'
sink = DurableSink()
object.__setattr__(run._host, 'runtime_event_sink', sink)
with pytest.raises(EvoRuntimeError, match='EVENT_COMMIT_INDETERMINATE'):
await run._terminal_locked('awaiting_input', checkpoint_details={'checkpoint_id': 'cp'})
await asyncio.sleep(0)
assert registry.inspect(run.run_id)['status'] == 'unknown'
intent = registry.terminal_intent(run.run_id)
assert intent['cleanup_confirmed'] is True
assert intent['phase'] == 'prepared'
assert intent['event']['payload']['outcome'] == 'awaiting_input'
restarted = SQLiteHostRegistry(registry.path, host_id='host', boot_id='new')
# A new boot cannot create resource evidence for an old execution.
with pytest.raises(EvoRuntimeError, match='EXECUTION_BOOT_MISMATCH'):
restarted.prepare_terminal(run.run_id, event=intent['event'])
registry.bind(execution_id='unconfirmed', grant_id='unconfirmed', digest='d', thread_id='other', turn_id='t')
runtime.host_registry = restarted
sink = DurableSink()
sink.uncertain = False
assert await runtime.recover_terminal(run.run_id, sink=sink) == 'awaiting_input'
assert await runtime.recover_terminal(run.run_id, sink=sink) == 'awaiting_input'
assert restarted.inspect(run.run_id)['status'] == 'awaiting_input'
assert restarted.terminal_intent(run.run_id)['phase'] == 'registry_finished'
assert await run.wait_stopped(timeout=0.1) == 'awaiting_input'
with pytest.raises(EvoRuntimeError, match='TERMINAL_CLEANUP_EVIDENCE_REQUIRED'):
await runtime.recover_terminal('unconfirmed', sink=sink)
with sqlite3.connect(sink_path) as db:
rows = db.execute('SELECT body FROM events').fetchall()
assert len(rows) == 1
assert json.loads(rows[0][0]) == intent['event']
with sqlite3.connect(registry.path) as db:
assert db.execute('SELECT checkpoint_id FROM pending_continuations WHERE execution_id=?',
(run.run_id,)).fetchone()[0] == 'cp'
assert db.execute('SELECT execution_id FROM active_claims').fetchall() == [('unconfirmed',)]
+62
View File
@@ -770,6 +770,68 @@ def test_openai_gpt_chat_plan_uses_completion_tokens_not_responses_tokens() -> N
assert params == {"max_completion_tokens": 65_000, "use_responses_api": False}
@pytest.mark.parametrize(
("adapter_id", "revision", "model_id", "expected"),
[
(
"openai",
"openai-v1",
"gpt-5.6-sol",
{"effort": "medium", "summary": "auto"},
),
(
"xai",
"xai-v1",
"grok-4.6",
{"effort": "medium", "summary": "auto"},
),
(
"dashscope",
"dashscope-v1",
"qwen3.7-plus",
{"effort": "medium"},
),
],
)
def test_responses_reasoning_uses_adapter_public_summary_contract(
adapter_id: str,
revision: str,
model_id: str,
expected: dict[str, str],
) -> None:
params = get_adapter_registry().get(adapter_id, revision).compile_runtime_parameters(
"responses",
{"reasoning": "medium"},
65_000,
provider_model_id=model_id,
)
assert params["reasoning"] == expected
@pytest.mark.parametrize(
("adapter_id", "revision", "model_id"),
[
("openai", "openai-v1", "gpt-5.6-sol"),
("xai", "xai-v1", "grok-4.6"),
("dashscope", "dashscope-v1", "qwen3.7-plus"),
],
)
def test_responses_reasoning_off_omits_reasoning_parameter(
adapter_id: str,
revision: str,
model_id: str,
) -> None:
params = get_adapter_registry().get(adapter_id, revision).compile_runtime_parameters(
"responses",
{"reasoning": "off"},
65_000,
provider_model_id=model_id,
)
assert "reasoning" not in params
def test_kimi_discovery_descriptor_is_partial_and_has_official_reasoning_policy() -> None:
registration = get_adapter_registry().get("openai", "openai-v1")
+31 -6
View File
@@ -1,11 +1,11 @@
from __future__ import annotations
from typing import Any, cast
import asyncio
from types import SimpleNamespace
from typing import Any, cast
import httpx
import pytest
from types import SimpleNamespace
from langchain_core.messages import ToolMessage
from langgraph.errors import GraphInterrupt
@@ -87,6 +87,7 @@ async def test_tool_effect_gateway_error_preserves_machine_code(monkeypatch):
)
monkeypatch.setenv("AI4SCI_EVO_RUNTIME_GRANT_SECRET", "runtime-service-secret")
monkeypatch.delenv("EVOSCIENTIST_BACKEND_SERVICE_TOKEN", raising=False)
monkeypatch.setattr(recoverable_tools.httpx, "AsyncClient", lambda **_: Client())
with pytest.raises(EvoRuntimeError) as exc_info:
@@ -108,7 +109,20 @@ async def test_tool_effect_gateway_error_preserves_machine_code(monkeypatch):
@pytest.mark.asyncio
async def test_terminal_callback_timeout_returns_tool_error_instead_of_crashing_run(monkeypatch):
@pytest.mark.parametrize(
"callback_error",
[
httpx.ConnectTimeout("terminal callback timed out"),
httpx.HTTPStatusError(
"terminal callback returned 500",
request=httpx.Request("POST", "http://gateway/tool-effect/terminal"),
response=httpx.Response(500),
),
],
)
async def test_terminal_callback_unavailable_returns_tool_error_instead_of_crashing_run(
monkeypatch, callback_error
):
request = SimpleNamespace(
tool_call={"id": "tool-1", "name": "tavily_search", "args": {"query": "x"}}
)
@@ -130,7 +144,7 @@ async def test_terminal_callback_timeout_returns_tool_error_instead_of_crashing_
phases.append(phase)
if phase == "prepare":
return {"action": "execute", "fencing_token": 7}
raise httpx.ConnectTimeout("terminal callback timed out")
raise callback_error
async def handler(_request):
return ToolMessage(
@@ -229,7 +243,18 @@ async def test_terminal_callback_semantic_error_remains_fail_closed(monkeypatch)
@pytest.mark.asyncio
async def test_non_idempotent_terminal_transport_failure_is_fail_closed(monkeypatch):
@pytest.mark.parametrize(
"callback_error",
[
httpx.ConnectTimeout("terminal callback timed out"),
httpx.HTTPStatusError(
"terminal callback returned 500",
request=httpx.Request("POST", "http://gateway/tool-effect/terminal"),
response=httpx.Response(500),
),
],
)
async def test_non_idempotent_terminal_unavailable_is_fail_closed(monkeypatch, callback_error):
request = SimpleNamespace(
tool_call={"id": "tool-4", "name": "send_message", "args": {"text": "hello"}}
)
@@ -249,7 +274,7 @@ async def test_non_idempotent_terminal_transport_failure_is_fail_closed(monkeypa
async def fake_post(_proxy, phase, _payload):
if phase == "prepare":
return {"action": "execute", "fencing_token": 10}
raise httpx.ConnectTimeout("terminal callback timed out")
raise callback_error
async def handler(_request):
return ToolMessage(
@@ -0,0 +1,17 @@
"""Do not advertise restart-safe identity without a durable run backend."""
import asyncio
import json
from starlette.requests import Request
def test_capabilities_do_not_claim_restart_safe_identity(monkeypatch):
monkeypatch.setenv("EVOSCIENTIST_BACKEND_SERVICE_TOKEN", "test-only")
from EvoScientist.langgraph_dev.http import recoverable_run_capabilities
result = asyncio.run(recoverable_run_capabilities(Request({"type": "http"})))
capabilities = json.loads(bytes(result.body))
assert capabilities["deterministic_run_id"] is True
assert capabilities["durable_run_identity"] is False
assert capabilities["run_not_found_proves_absence"] is False
@@ -0,0 +1,95 @@
"""Schema2 tool declarations migrate without silent text-only degradation."""
from copy import deepcopy
import pytest
from EvoScientist.llm.contracts import EvoRuntimeError
from EvoScientist.llm.model_config import EvoModelConfig, convert_v2_to_v3_draft
from tests.v3_fixtures import v3_payload
def test_schema2_tools_require_explicit_native_route_contract():
# Keep this node identity: the historical non_streaming label was the only
# accepted Web tool route contract, not a measured capability or HTTP knob.
payload = v3_payload()
payload["capability_evidence"] = []
models = payload["providers"]["custom-openai"]["models"]
unknown = deepcopy(models[0])
unknown["id"] = "unreferenced-text-model"
models.append(unknown)
for transport in ("native", "non_streaming"):
payload["route_selectors"]["visible-main"]["tool_call_transport"] = transport
config = EvoModelConfig.parse(payload, require_evidence=False)
parsed = config.providers["custom-openai"].models
assert parsed[unknown["id"]].capabilities["tools"] is False
assert parsed[unknown["id"]].capabilities["text"] is True
assert parsed["model-id"].capabilities["tools"] is True
assert parsed["model-id"].capabilities["text"] is True
route = config.concrete_routes("visible-main")[0]
assert config.route_model(route).capabilities["tools"] is True
assert route.tool_call_transport == transport
assert config.capability_evidence == {}
draft, report = convert_v2_to_v3_draft(
payload, target_revision=2, config_identity_key_id="migration-test"
)
assert not report.blocking_issues
migrated = {m["model_key"]: m for m in draft["providers"][0]["models"]}
for key, expected in (("model-id", True), (unknown["id"], False)):
assert migrated[key]["capabilities"]["tools"] is expected
assert migrated[key]["invocation"]["tool_call_transport"] == (
"native" if expected else "disabled"
)
ambiguous = deepcopy(payload)
ambiguous["route_selectors"]["visible-main"]["tool_call_transport"] = "streaming"
with pytest.raises(EvoRuntimeError, match="schema2 tool capability migration"):
EvoModelConfig.parse(ambiguous, require_evidence=False)
with pytest.raises(EvoRuntimeError, match="schema2 tool capability migration"):
convert_v2_to_v3_draft(
ambiguous, target_revision=2, config_identity_key_id="migration-test"
)
# Aggregate declarations across routes, independent of insertion order.
declared = deepcopy(payload["route_selectors"]["visible-main"])
ambiguous["route_selectors"]["declared"] = declared
for reverse in (False, True):
if reverse:
ambiguous["route_selectors"] = dict(
reversed(list(ambiguous["route_selectors"].items()))
)
config = EvoModelConfig.parse(ambiguous, require_evidence=False)
route = config.concrete_routes("visible-main")[0]
assert config.route_model(route).capabilities["tools"] is True
# Same model id on another provider must not inherit the declaration.
isolated = deepcopy(payload)
isolated["providers"]["other"] = deepcopy(payload["providers"]["custom-openai"])
other = {
**declared,
"provider": "other",
"endpoint": "primary",
"tool_call_transport": "streaming",
}
other.pop("endpoint_pool")
isolated["route_selectors"]["other-route"] = other
config = EvoModelConfig.parse(isolated, require_evidence=False)
assert config.providers["other"].models["model-id"].capabilities["tools"] is False
for usage in ("main", "title", "fallback"):
referenced = deepcopy(isolated)
if usage == "main":
selectable = referenced["purpose_routes"]["main_agent"]["selectable"]
selectable["other"] = "other-route"
elif usage == "title":
referenced["purpose_routes"]["title"]["default"] = "other-route"
else:
referenced["tool_protocol_fallbacks"][0]["fallbacks"] = ["other-route"]
with pytest.raises(EvoRuntimeError, match="schema2 tool capability migration"):
EvoModelConfig.parse(referenced, require_evidence=False)
referenced["route_selectors"]["other-route"].update(
tool_call_transport="non_streaming"
)
config = EvoModelConfig.parse(referenced, require_evidence=False)
assert config.providers["other"].models["model-id"].capabilities["tools"] is True
other_models = config.providers["other"].models
assert other_models[unknown["id"]].capabilities["tools"] is False
+21
View File
@@ -409,6 +409,27 @@ class TestV3ProtocolStreaming:
assert len(thinking_events) == 1
assert thinking_events[0]["content"] == "Think once."
async def test_responses_public_reasoning_summary_is_not_private_thinking(self):
message = AIMessage(
content=[{
"type": "reasoning",
"encrypted_content": "opaque",
"summary": [{"type": "summary_text", "text": "Checked official sources."}],
}],
)
agent = FakeV3Agent([protocol_event("messages", (message, {}))])
events = await collect_events(agent)
summaries = [e for e in events if e.get("type") == "reasoning_summary"]
assert summaries == [{
"type": "reasoning_summary",
"summary": "Checked official sources.",
"visibility": "user_visible_summary",
"source_kind": "provider_summary",
}]
assert not any(e.get("type") == "thinking" for e in events)
async def test_tool_selector_reasoning_delta_is_suppressed(self):
"""Selector reasoning must not appear as main-agent thinking."""
import EvoScientist.middleware.tool_selector as selector_mod
@@ -0,0 +1,13 @@
import inspect
from EvoScientist.web_checkpointer import _web_saver_type
def test_web_saver_declares_real_run_cleanup_without_deleting_shared_blobs():
saver_type = _web_saver_type()
source = inspect.getsource(saver_type.adelete_for_runs)
assert saver_type.adelete_for_runs.__qualname__.startswith("_web_saver_type")
assert "checkpoint_writes" in source
assert "DELETE FROM checkpoints" in source
assert "checkpoint_blobs" not in source
assert "metadata->>'run_id'" in source
+51
View File
@@ -0,0 +1,51 @@
import asyncio
from concurrent.futures import Future
import pytest
from EvoScientist.langgraph_dev import worker_exit
@pytest.mark.asyncio
async def test_remote_thread_future_is_drained_before_cancellation_returns():
remote = Future()
entered = asyncio.Event()
async def wait_remote():
entered.set()
return await worker_exit._await_remote_future(remote)
task = asyncio.create_task(wait_remote())
await entered.wait()
task.cancel()
await asyncio.sleep(0)
assert not task.done()
remote.set_result("finished")
with pytest.raises(asyncio.CancelledError):
await task
@pytest.mark.asyncio
async def test_cancellation_listener_restarts_after_idle_timeout(monkeypatch):
calls = 0
done = asyncio.Event()
async def original(queue, run_id, thread_id, event):
nonlocal calls
calls += 1
if calls == 2:
event.set()
await worker_exit._persistent_cancellation_listener(
original, asyncio.Queue(), "run", "thread", done
)
assert calls == 2
@pytest.mark.asyncio
async def test_wait_for_exit_does_not_claim_a_still_active_worker(monkeypatch):
monkeypatch.setattr(worker_exit, "cancel_and_inspect", lambda *_: {
"execution_exited": False,
})
receipt = await worker_exit.wait_for_exit("thread", "run", timeout=0.01)
assert receipt["execution_exited"] is False
+1 -1
View File
@@ -89,7 +89,7 @@ def v3_payload(*, revision: int = 1) -> dict[str, Any]:
"endpoint_pool": "default",
"model": "model-id",
"api_mode": "chat_completions",
"tool_call_transport": "non_streaming",
"tool_call_transport": "native",
}
},
"purpose_routes": {