5a581c78a2
Build / build (push) Has been cancelled
Docker / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Introduce provider, model, and invocation contracts with encrypted configuration persistence. Add web runtime fencing, route fallback, recovery middleware, workspace scoping, and comprehensive tests.
1025 lines
35 KiB
Python
1025 lines
35 KiB
Python
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import base64
|
|
import json
|
|
from pathlib import Path
|
|
from types import SimpleNamespace
|
|
|
|
import pytest
|
|
from langchain_core.messages import HumanMessage
|
|
|
|
from EvoScientist.llm.adapter_registry import get_adapter_registry
|
|
from EvoScientist.llm.contracts import (
|
|
AgentInputV3,
|
|
EvoRuntimeError,
|
|
HmacGrantAuthority,
|
|
WebHostContext,
|
|
)
|
|
from EvoScientist.llm.errors import ModelToolProtocolError
|
|
from EvoScientist.llm.model_config import EvoModelConfig, FileEvoModelConfigStore
|
|
from EvoScientist.llm.runtime import (
|
|
_PROTOCOL_MARGIN_TOKENS,
|
|
EvoModelRuntime,
|
|
_callback_message_schema_debug,
|
|
_callback_payload_debug_summary,
|
|
_invocation_parameters_debug,
|
|
_provider_error_message_debug,
|
|
_provider_failure_details,
|
|
_provider_input_token_bound,
|
|
_run_failure_details,
|
|
_safe_error_code,
|
|
)
|
|
from tests.v3_fixtures import RUNTIME_KEY_ID, RUNTIME_SECRET, identity_ring, v3_payload
|
|
|
|
|
|
class _Sink:
|
|
def __init__(self) -> None:
|
|
self.events = []
|
|
|
|
async def commit(self, event):
|
|
self.events.append(event)
|
|
return "committed"
|
|
|
|
async def confirm(self, _event_id, _payload_digest):
|
|
return "committed"
|
|
|
|
|
|
class _UncertainSink(_Sink):
|
|
def __init__(self, confirmation: str) -> None:
|
|
super().__init__()
|
|
self.confirmation = confirmation
|
|
self.confirmed = []
|
|
|
|
async def commit(self, event):
|
|
self.events.append(event)
|
|
raise ConnectionError("commit result unavailable")
|
|
|
|
async def confirm(self, event_id, payload_digest):
|
|
self.confirmed.append((event_id, payload_digest))
|
|
return self.confirmation
|
|
|
|
|
|
class _TerminalFailingSink(_Sink):
|
|
async def commit(self, event):
|
|
if event.kind == "run" and event.payload.get("kind") == "run_terminal":
|
|
raise EvoRuntimeError("EVO_EVENT_CONFLICT")
|
|
return await super().commit(event)
|
|
|
|
|
|
class _Model:
|
|
def __init__(self) -> None:
|
|
self.metadata = {}
|
|
|
|
def model_copy(self, *, update):
|
|
copy = _Model()
|
|
copy.metadata = update.get("metadata", {})
|
|
return copy
|
|
|
|
async def ainvoke(self, _input, **_kwargs):
|
|
return SimpleNamespace(
|
|
content="Test title", usage_metadata={"input_tokens": 1, "output_tokens": 1}
|
|
)
|
|
|
|
|
|
class _Agent:
|
|
pass
|
|
|
|
|
|
def _runtime(tmp_path: Path, monkeypatch):
|
|
monkeypatch.setenv("WEB_RUNTIME_TEST_KEY", "test-secret")
|
|
authority = HmacGrantAuthority(RUNTIME_SECRET, RUNTIME_KEY_ID)
|
|
store = FileEvoModelConfigStore(
|
|
tmp_path / "model_routes.yaml", admin_verifier=authority
|
|
)
|
|
store.bootstrap_for_development(v3_payload())
|
|
runtime = EvoModelRuntime(
|
|
store,
|
|
admission_verifier=authority,
|
|
quote_authority=authority,
|
|
identity_key_ring=identity_ring(),
|
|
model_factory=lambda **_kwargs: _Model(),
|
|
agent_factory=lambda *_args: _Agent(),
|
|
)
|
|
return runtime, authority
|
|
|
|
|
|
def _input() -> AgentInputV3:
|
|
return AgentInputV3(
|
|
"hello",
|
|
"web:user:thread",
|
|
metadata={"source": "web", "ignored": "not-digested"},
|
|
)
|
|
|
|
|
|
def test_provider_input_bound_counts_decoded_media_separately() -> None:
|
|
raw = b"x" * 30_000
|
|
payload = [
|
|
{
|
|
"type": "image",
|
|
"mime_type": "image/png",
|
|
"base64": base64.b64encode(raw).decode("ascii"),
|
|
}
|
|
]
|
|
|
|
bound = _provider_input_token_bound(payload)
|
|
|
|
assert bound.media_blocks == 1
|
|
assert bound.largest_media_bytes >= len(raw) - 2
|
|
assert bound.media_tokens == (len(raw) + 2) // 3 + 512
|
|
assert bound.total_tokens < len(base64.b64encode(raw))
|
|
|
|
|
|
def test_agent_input_digest_includes_forced_context_repair() -> None:
|
|
normal = AgentInputV3(
|
|
"hello",
|
|
"web:user:thread",
|
|
metadata={"source": "web", "force_context_repair": False},
|
|
)
|
|
repair = AgentInputV3(
|
|
"hello",
|
|
"web:user:thread",
|
|
metadata={"source": "web", "force_context_repair": True},
|
|
)
|
|
|
|
assert normal.canonical_bytes() != repair.canonical_bytes()
|
|
|
|
|
|
def _preparation(
|
|
authority,
|
|
agent_input,
|
|
*,
|
|
reasoning_effort="disabled",
|
|
title_policy="best_effort",
|
|
):
|
|
return authority.sign_preparation(
|
|
request_id="11111111-1111-4111-8111-111111111111",
|
|
turn_id="22222222-2222-4222-8222-222222222222",
|
|
thread_id="thread",
|
|
subject_id="user",
|
|
requested_model_ref="visible-model",
|
|
plan="starter",
|
|
roles=("user",),
|
|
requires_vision=False,
|
|
reasoning_effort=reasoning_effort,
|
|
title_policy=title_policy,
|
|
gateway_input_digest=authority.agent_input_digest(agent_input.projection()),
|
|
checkpoint_thread_id=agent_input.checkpoint_thread_id,
|
|
checkpoint_snapshot_id="sha256:checkpoint",
|
|
turn_fencing_token=1,
|
|
ttl_ms=60_000,
|
|
)
|
|
|
|
|
|
def test_runtime_preserves_model_tool_protocol_error_code() -> None:
|
|
error = ModelToolProtocolError("missing_name", provider="openai")
|
|
|
|
assert _safe_error_code(error) == "MODEL_TOOL_PROTOCOL_INVALID"
|
|
|
|
|
|
def test_provider_failure_details_are_diagnostic_but_do_not_expose_messages() -> None:
|
|
class ProviderBadRequest(RuntimeError):
|
|
status_code = 400
|
|
code = "unsupported_parameter"
|
|
|
|
def __init__(self) -> None:
|
|
super().__init__("request failed with api_key=sk-secret")
|
|
self.body = {
|
|
"error": {
|
|
"code": "unsupported_parameter",
|
|
"message": (
|
|
"invalid temperature: only 1 is allowed; api_key=sk-secret"
|
|
),
|
|
}
|
|
}
|
|
|
|
details = _provider_failure_details(
|
|
ProviderBadRequest(),
|
|
route=None,
|
|
error_code="MODEL_PROVIDER_REQUEST_REJECTED",
|
|
)
|
|
|
|
assert details == {
|
|
"failure_stage": "provider_request",
|
|
"reason": "provider_rejected_request",
|
|
"provider_error_type": "ProviderBadRequest",
|
|
"provider_error_module": __name__,
|
|
"http_status": 400,
|
|
"provider_error_code": "unsupported_parameter",
|
|
"provider_error_parameter": "temperature",
|
|
}
|
|
assert "sk-secret" not in str(details)
|
|
|
|
|
|
def test_request_debug_summary_reports_plan_safe_tool_structure_only() -> None:
|
|
payload = [
|
|
[
|
|
{
|
|
"type": "ai",
|
|
"data": {
|
|
"content": [
|
|
{"type": "text", "text": "secret assistant content"},
|
|
{"type": "tool_call", "name": "read_file"},
|
|
],
|
|
"tool_calls": [{"id": "call_1", "name": "read_file"}],
|
|
},
|
|
},
|
|
{
|
|
"type": "tool",
|
|
"data": {"content": "secret tool result"},
|
|
},
|
|
]
|
|
]
|
|
|
|
summary = _callback_payload_debug_summary(payload, 123)
|
|
|
|
assert summary["tool_calls"] == 1
|
|
assert summary["tool_results"] == 1
|
|
assert summary["content_block_types"] == "text:1,tool_call:1"
|
|
assert "secret" not in str(summary)
|
|
|
|
|
|
def test_request_debug_log_projects_parameter_values_without_secrets() -> None:
|
|
plan = SimpleNamespace(
|
|
sdk_params={
|
|
"max_completion_tokens": 65_000,
|
|
"reasoning_effort": "high",
|
|
"streaming": True,
|
|
"use_responses_api": False,
|
|
"extra_body": {
|
|
"enable_thinking": True,
|
|
"thinking_budget": 8_192,
|
|
"private_extension": "secret extension",
|
|
},
|
|
"api_key": "sk-secret",
|
|
"default_headers": {"Authorization": "Bearer sk-secret"},
|
|
}
|
|
)
|
|
|
|
debug = _invocation_parameters_debug(plan)
|
|
|
|
assert '"max_completion_tokens":65000' in debug
|
|
assert '"reasoning_effort":"high"' in debug
|
|
assert '"enable_thinking":true' in debug
|
|
assert "api_key" not in debug
|
|
assert "Authorization" not in debug
|
|
assert "secret" not in debug
|
|
|
|
|
|
def test_request_debug_message_schema_identifies_empty_and_metadata_without_content() -> (
|
|
None
|
|
):
|
|
payload = [
|
|
[
|
|
{
|
|
"type": "ai",
|
|
"data": {
|
|
"content": "secret assistant content",
|
|
"additional_kwargs": {"reasoning_content": "secret reasoning"},
|
|
"response_metadata": {"model_name": "secret model"},
|
|
},
|
|
},
|
|
{"type": "ai", "data": {"content": ""}},
|
|
]
|
|
]
|
|
|
|
debug = _callback_message_schema_debug(payload)
|
|
|
|
assert '"additional_keys":["reasoning_content"]' in debug
|
|
assert '"response_metadata_keys":["model_name"]' in debug
|
|
assert '"content_chars":24' in debug
|
|
assert '"empty":true' in debug
|
|
assert "secret assistant content" not in debug
|
|
assert "secret reasoning" not in debug
|
|
assert "secret model" not in debug
|
|
|
|
|
|
def test_provider_error_debug_message_is_bounded_and_redacts_route_secrets() -> None:
|
|
class ProviderBadRequest(RuntimeError):
|
|
def __init__(self) -> None:
|
|
super().__init__("fallback includes api_key=sk-route-secret")
|
|
self.body = {
|
|
"error": {
|
|
"message": (
|
|
"Invalid max_completion_tokens; "
|
|
"Authorization: Bearer sk-route-secret"
|
|
)
|
|
}
|
|
}
|
|
|
|
route = SimpleNamespace(
|
|
api_key="sk-route-secret",
|
|
default_headers={"X-Provider-Secret": "header-secret-value"},
|
|
)
|
|
|
|
debug = _provider_error_message_debug(ProviderBadRequest(), route)
|
|
|
|
assert "Invalid max_completion_tokens" in debug
|
|
assert "sk-route-secret" not in debug
|
|
assert "header-secret-value" not in debug
|
|
assert "<redacted>" in debug
|
|
assert len(debug) <= 1_024
|
|
|
|
|
|
def test_json_decode_diagnostics_record_only_structure_hash_and_trace() -> None:
|
|
document = '{"type":"response.output_text.delta","delta":"secret model\noutput"}'
|
|
position = document.index("\n")
|
|
try:
|
|
raise json.JSONDecodeError("Invalid control character", document, position)
|
|
except json.JSONDecodeError as caught:
|
|
error = caught
|
|
|
|
details = _provider_failure_details(
|
|
error,
|
|
route=None,
|
|
error_code="MODEL_PROVIDER_RESPONSE_INVALID",
|
|
)
|
|
|
|
assert details["provider_json_document_bytes"] == len(document)
|
|
assert details["provider_json_line"] == 1
|
|
assert details["provider_json_column"] == position + 1
|
|
assert details["provider_json_position"] == position
|
|
assert details["provider_json_invalid_codepoint"] == "U+000A"
|
|
assert details["provider_json_position_inside_string"] == "true"
|
|
assert details["provider_json_event_type"] == "response.output_text.delta"
|
|
assert details["provider_json_control_codepoints"] == "U+000A:1"
|
|
assert details["provider_json_document_sha256"]
|
|
assert details["provider_json_window_sha256"]
|
|
assert details["provider_json_window_bytes"] == len(document)
|
|
assert details["provider_json_trace"]
|
|
assert "secret model" not in str(details)
|
|
|
|
|
|
def test_stream_timeout_diagnostics_use_structured_exception_attributes() -> None:
|
|
class StreamChunkTimeoutError(TimeoutError):
|
|
chunks_received = 1
|
|
timeout_s = 120.0
|
|
|
|
details = _provider_failure_details(
|
|
StreamChunkTimeoutError("secret upstream message"),
|
|
route=None,
|
|
error_code="MODEL_TIMEOUT",
|
|
)
|
|
|
|
assert details["provider_stream_chunks_received"] == 1
|
|
assert details["provider_stream_idle_timeout_ms"] == 120_000
|
|
assert "secret upstream message" not in str(details)
|
|
|
|
|
|
def test_openai_compatible_adapter_classifies_bad_request_as_contract_rejection() -> (
|
|
None
|
|
):
|
|
class ProviderBadRequest(RuntimeError):
|
|
status_code = 400
|
|
|
|
registration = get_adapter_registry().get(
|
|
"generic-openai-compatible", "generic-openai-compatible-v1"
|
|
)
|
|
|
|
assert registration.classify_error(ProviderBadRequest()).error_code == (
|
|
"MODEL_PROVIDER_REQUEST_REJECTED"
|
|
)
|
|
|
|
|
|
def test_openai_compatible_adapter_classifies_invalid_json_as_provider_response_error() -> (
|
|
None
|
|
):
|
|
registration = get_adapter_registry().get(
|
|
"generic-openai-compatible", "generic-openai-compatible-v1"
|
|
)
|
|
|
|
assert registration.classify_error(
|
|
json.JSONDecodeError("bad", "", 0)
|
|
).error_code == ("MODEL_PROVIDER_RESPONSE_INVALID")
|
|
|
|
|
|
def test_generic_openai_compatible_chat_has_a_protocol_margin() -> None:
|
|
assert (
|
|
_PROTOCOL_MARGIN_TOKENS[("generic-openai-compatible", "chat_completions")] == 64
|
|
)
|
|
|
|
|
|
def test_runtime_rejects_reasoning_duplicated_in_model_options() -> None:
|
|
config = EvoModelConfig.parse(v3_payload(), require_evidence=False)
|
|
model = config.providers["custom-openai"].models["model-id"]
|
|
registration = get_adapter_registry().get(
|
|
"generic-openai-compatible", "generic-openai-compatible-v1"
|
|
)
|
|
|
|
with pytest.raises(EvoRuntimeError, match="AGENT_INPUT_MISMATCH"):
|
|
EvoModelRuntime._validated_user_options(
|
|
model,
|
|
registration,
|
|
{"reasoning": "medium", "reasoning_effort": "high"},
|
|
"main_agent",
|
|
)
|
|
|
|
|
|
def _admission(authority, quote):
|
|
return authority.sign_admission(
|
|
preparation_id=quote.preparation_id,
|
|
request_id=quote.request_id,
|
|
turn_id=quote.turn_id,
|
|
thread_id=quote.thread_id,
|
|
subject_id=quote.subject_id,
|
|
requested_model_ref=quote.requested_model_ref,
|
|
plan=quote.plan,
|
|
roles=quote.roles,
|
|
requires_vision=quote.requires_vision,
|
|
reasoning_effort=quote.reasoning_effort,
|
|
title_policy=quote.title_policy,
|
|
gateway_input_digest=quote.gateway_input_digest,
|
|
prepared_snapshot_digest=quote.prepared_snapshot_digest,
|
|
prepared_input_digest=quote.prepared_input_digest,
|
|
config_revision=quote.config_revision,
|
|
catalog_revision=quote.catalog_revision,
|
|
purpose_attempt_limits=quote.purpose_attempt_limits,
|
|
total_max_attempts=quote.total_max_attempts,
|
|
checkpoint_snapshot_id=quote.checkpoint_snapshot_id,
|
|
tool_registry_snapshot_id=quote.tool_registry_snapshot_id,
|
|
turn_fencing_token=quote.turn_fencing_token,
|
|
admission_snapshot_id="33333333-3333-4333-8333-333333333333",
|
|
admission_id="44444444-4444-4444-8444-444444444444",
|
|
hold_id="55555555-5555-4555-8555-555555555555",
|
|
billing_fencing_token=1,
|
|
provider_run_reserve_microunits=quote.provider_run_reserve_microunits,
|
|
billing_policy_version="test-v3",
|
|
expires_at=quote.expires_at,
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_prepare_freezes_input_and_start_reuses_it(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
sink = _Sink()
|
|
agent_input = _input()
|
|
host = WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
|
|
assert quote.contract_version == 3
|
|
assert quote.prepared_input_digest.startswith("hmac-sha256:")
|
|
assert quote.total_max_attempts == sum(quote.purpose_attempt_limits.values())
|
|
assert quote.provider_run_reserve_microunits == 0
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
assert run.run_id
|
|
with pytest.raises(EvoRuntimeError, match="EVENT_CURSOR_INVALID"):
|
|
await anext(run.stream(agent_input))
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_disabled_title_is_omitted_from_quote_snapshot_and_model_set(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input, title_policy="disabled"),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
|
|
)
|
|
|
|
expected = {"main_agent", "tool_selector", "deepagents_summarizer"}
|
|
assert set(quote.enabled_purposes) == expected
|
|
assert set(quote.purpose_routes) == expected
|
|
assert set(quote.purpose_route_call_bounds) == expected
|
|
assert set(quote.purpose_attempt_limits) == expected
|
|
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
|
|
assert set(run._snapshot.purpose_routes) == expected
|
|
assert set(run._snapshot.purpose_route_call_bounds) == expected
|
|
assert run._model_set.title is None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_best_effort_title_remains_available(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input, title_policy="best_effort"),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
|
|
)
|
|
|
|
assert "title" in quote.enabled_purposes
|
|
assert "title" in quote.purpose_routes
|
|
assert "title" in quote.purpose_route_call_bounds
|
|
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
|
|
assert run._model_set.title is not None
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_forced_context_repair_is_consumed_once_before_provider_start(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = AgentInputV3(
|
|
"hello",
|
|
"web:user:thread",
|
|
metadata={"source": "web", "force_context_repair": True},
|
|
)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
|
|
with pytest.raises(EvoRuntimeError, match="MODEL_CONTEXT_WINDOW_EXCEEDED") as exc:
|
|
await run._attempt_callback.on_chat_model_start(
|
|
{}, [[HumanMessage(content="hello")]], run_id="repair-trigger"
|
|
)
|
|
|
|
assert exc.value.details[0]["reason"] == "forced_context_repair"
|
|
assert exc.value.details[0]["repair_requested"] is True
|
|
assert run._force_context_repair_pending is False
|
|
assert "repair-trigger" not in run._callback_attempts
|
|
|
|
await run._attempt_callback.on_chat_model_start(
|
|
{}, [[HumanMessage(content="hello")]], run_id="repair-retry"
|
|
)
|
|
assert "repair-retry" in run._callback_attempts
|
|
await run._finish_callback_attempt(
|
|
"repair-retry",
|
|
usage={"input_tokens": 1, "output_tokens": 1},
|
|
error_code=None,
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_provider_input_hard_cap_reports_media_aware_safe_breakdown(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
route = run._snapshot.main_routes[0]
|
|
bound = run._snapshot.purpose_route_call_bounds["main_agent"][0]
|
|
|
|
with pytest.raises(EvoRuntimeError, match="MODEL_CONTEXT_WINDOW_EXCEEDED") as exc:
|
|
await run._begin_callback_attempt(
|
|
callback_run_id="too-large",
|
|
purpose="main_agent",
|
|
route=route,
|
|
provider_input_bound_tokens=bound.payload_input_hard_cap + 1,
|
|
provider_input_breakdown={
|
|
"text_input_bound_tokens": 123,
|
|
"media_input_bound_tokens": 456,
|
|
"media_blocks": 2,
|
|
"largest_media_bytes": 1024,
|
|
},
|
|
)
|
|
|
|
assert exc.value.details == (
|
|
{
|
|
"provider_input_bound_tokens": bound.payload_input_hard_cap + 1,
|
|
"payload_input_hard_cap": bound.payload_input_hard_cap,
|
|
"text_input_bound_tokens": 123,
|
|
"media_input_bound_tokens": 456,
|
|
"media_blocks": 2,
|
|
"largest_media_bytes": 1024,
|
|
},
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_route_parameters_compile_once_during_prepare(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
original = EvoModelRuntime._compile_route
|
|
calls = []
|
|
|
|
def compile_once(route, purpose, bound, reasoning_effort):
|
|
calls.append((route.identity.route_key, purpose, bound.max_output_tokens))
|
|
return original(route, purpose, bound, reasoning_effort)
|
|
|
|
monkeypatch.setattr(EvoModelRuntime, "_compile_route", staticmethod(compile_once))
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
|
|
)
|
|
prepare_call_count = len(calls)
|
|
|
|
assert prepare_call_count == 4
|
|
await runtime.start_web_run(_admission(authority, quote))
|
|
assert len(calls) == prepare_call_count
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_prepare_freezes_an_immutable_invocation_plan(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
|
|
)
|
|
|
|
route = runtime._prepared[quote.preparation_id].snapshot.main_routes[0]
|
|
plan = route.invocation_plan
|
|
|
|
assert plan is not None
|
|
assert plan.api_mode == "chat_completions"
|
|
assert plan.output_token_parameter == "max_tokens"
|
|
assert plan.output_token_limit == 512
|
|
assert plan.tool_call_transport == "disabled"
|
|
assert plan.sdk_params["use_responses_api"] is False
|
|
with pytest.raises(TypeError):
|
|
plan.sdk_params["max_tokens"] = 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_boolean_reasoning_adapter_maps_to_provider_extra_body(
|
|
monkeypatch, tmp_path
|
|
):
|
|
monkeypatch.setenv("WEB_RUNTIME_TEST_KEY", "test-secret")
|
|
authority = HmacGrantAuthority(RUNTIME_SECRET, RUNTIME_KEY_ID)
|
|
payload = v3_payload()
|
|
payload["providers"]["custom-openai"]["models"][0]["reasoning"] = {
|
|
"mode": "boolean",
|
|
"enabled_params": {"extra_body": {"enable_thinking": True}},
|
|
"disabled_params": {"extra_body": {"enable_thinking": False}},
|
|
}
|
|
# Rebuild evidence after changing route semantics.
|
|
payload.pop("capability_evidence")
|
|
from tests.v3_fixtures import v3_payload as build_payload
|
|
|
|
rebuilt = build_payload()
|
|
rebuilt["providers"]["custom-openai"]["models"][0]["reasoning"] = payload[
|
|
"providers"
|
|
]["custom-openai"]["models"][0]["reasoning"]
|
|
from EvoScientist.llm.model_config import (
|
|
EvoModelConfig,
|
|
endpoint_fingerprint,
|
|
route_semantics_hash,
|
|
)
|
|
|
|
candidate = EvoModelConfig.parse(rebuilt, require_evidence=False)
|
|
ring = identity_ring()
|
|
semantics_key = ring.derive_current("ai4sci/route-semantics-hash/v3")[1]
|
|
endpoint_key = ring.derive_current("ai4sci/endpoint-fingerprint/v3")[1]
|
|
route = candidate.concrete_routes("visible-main")[0]
|
|
rebuilt["capability_evidence"][0]["probe"]["route_semantics_hash"] = (
|
|
route_semantics_hash(candidate, route, semantics_key)
|
|
)
|
|
rebuilt["capability_evidence"][0]["probe"]["endpoint_fingerprint"] = (
|
|
endpoint_fingerprint(candidate, route, endpoint_key)
|
|
)
|
|
|
|
store = FileEvoModelConfigStore(tmp_path / "routes.yaml", admin_verifier=authority)
|
|
store.bootstrap_for_development(rebuilt)
|
|
calls = []
|
|
runtime = EvoModelRuntime(
|
|
store,
|
|
admission_verifier=authority,
|
|
quote_authority=authority,
|
|
identity_key_ring=identity_ring(),
|
|
model_factory=lambda **kwargs: calls.append(kwargs) or _Model(),
|
|
agent_factory=lambda *_args: _Agent(),
|
|
)
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input, reasoning_effort="high"),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()),
|
|
)
|
|
await runtime.start_web_run(_admission(authority, quote))
|
|
|
|
main = next(item for item in calls if item["max_tokens"] == 512)
|
|
assert main["extra_body"]["enable_thinking"] is True
|
|
assert "reasoning_effort" not in main
|
|
assert all(
|
|
item["extra_body"]["enable_thinking"] is False
|
|
for item in calls
|
|
if item["max_tokens"] != 512
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_prepare_is_idempotent_by_subject_request_and_turn(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
host = WebHostContext(
|
|
"/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()
|
|
)
|
|
|
|
first = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
replay = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
|
|
assert replay == first
|
|
assert len(runtime._prepared) == 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_prepare_rejects_semantic_conflict_for_same_request(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
host = WebHostContext(
|
|
"/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()
|
|
)
|
|
await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
|
|
with pytest.raises(EvoRuntimeError, match="PREPARATION_CONFLICT"):
|
|
await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input, reasoning_effort="high"),
|
|
agent_input,
|
|
host,
|
|
)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_cancelled_prepare_leaves_bounded_idempotency_tombstone(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
host = WebHostContext(
|
|
"/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()
|
|
)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
assert await runtime.cancel_prepared_run(quote.preparation_id, reason="test")
|
|
|
|
with pytest.raises(EvoRuntimeError, match="PREPARATION_STALE"):
|
|
await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
assert not runtime._prepared
|
|
assert len(runtime._prepared_tombstones) == 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_tampered_admission_is_rejected_before_agent_creation(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
host = WebHostContext(
|
|
"/tmp", "/tmp", object(), object(), runtime_event_sink=_Sink()
|
|
)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
admission = _admission(authority, quote)
|
|
tampered = admission.__class__(
|
|
**{
|
|
**admission.unsigned_payload(),
|
|
"plan": "enterprise",
|
|
"signature": admission.signature,
|
|
}
|
|
)
|
|
with pytest.raises(EvoRuntimeError, match="CONTRACT_SIGNATURE_INVALID"):
|
|
await runtime.start_web_run(tampered)
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_start_rejects_changed_tool_registry_revision(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
agent_input = _input()
|
|
revision = {"value": "registry-v1"}
|
|
|
|
def registry_provider():
|
|
return (), revision["value"]
|
|
|
|
host = WebHostContext(
|
|
"/tmp",
|
|
"/tmp",
|
|
object(),
|
|
object(),
|
|
runtime_event_sink=_Sink(),
|
|
tool_registry_provider=registry_provider,
|
|
)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
revision["value"] = "registry-v2"
|
|
|
|
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_STALE"):
|
|
await runtime.start_web_run(_admission(authority, quote))
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_started_event_is_committed_before_provider_boundary(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
sink = _Sink()
|
|
agent_input = _input()
|
|
host = WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
route = run._snapshot.purpose_routes["main_agent"][0]
|
|
await run._begin_callback_attempt(
|
|
callback_run_id="callback",
|
|
purpose="main_agent",
|
|
route=route,
|
|
provider_input_bound_tokens=10,
|
|
)
|
|
assert sink.events[0].payload["outcome"] == "started"
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_normal_agent_model_rounds_are_not_limited_by_quote_attempt_counts(
|
|
monkeypatch, tmp_path
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
sink = _Sink()
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink),
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
route = run._snapshot.purpose_routes["main_agent"][0]
|
|
configured_estimate = quote.purpose_attempt_limits["main_agent"]
|
|
|
|
for index in range(configured_estimate + 2):
|
|
callback_run_id = f"callback-{index}"
|
|
await run._begin_callback_attempt(
|
|
callback_run_id=callback_run_id,
|
|
purpose="main_agent",
|
|
route=route,
|
|
provider_input_bound_tokens=10,
|
|
)
|
|
await run._finish_callback_attempt(
|
|
callback_run_id,
|
|
usage={"input_tokens": 1, "cached_input_tokens": 0, "output_tokens": 1},
|
|
error_code=None,
|
|
)
|
|
|
|
assert run._attempt_counts["main_agent"] == configured_estimate + 2
|
|
starts = [
|
|
event.payload
|
|
for event in sink.events
|
|
if event.kind == "model_attempt" and event.payload["outcome"] == "started"
|
|
]
|
|
assert [event["attempt_index"] for event in starts] == list(
|
|
range(1, configured_estimate + 3)
|
|
)
|
|
assert all(event["provider_reserved_microunits"] == 0 for event in starts)
|
|
|
|
|
|
def test_graph_recursion_error_has_a_stable_runtime_code() -> None:
|
|
GraphRecursionError = type("GraphRecursionError", (Exception,), {})
|
|
|
|
assert _safe_error_code(GraphRecursionError()) == "AGENT_RECURSION_LIMIT_EXCEEDED"
|
|
|
|
|
|
def test_agent_execution_fallback_is_not_misattributed_to_provider() -> None:
|
|
error = TypeError("sensitive graph detail")
|
|
code = _safe_error_code(error, fallback="AGENT_RUNTIME_ERROR")
|
|
|
|
assert code == "AGENT_RUNTIME_ERROR"
|
|
assert _safe_error_code(error) == "MODEL_PROVIDER_ERROR"
|
|
assert _run_failure_details(error, code) == {
|
|
"failure_stage": "agent_execution",
|
|
"reason": "unclassified_agent_exception",
|
|
"agent_error_type": "TypeError",
|
|
"agent_error_module": "builtins",
|
|
}
|
|
assert "sensitive graph detail" not in str(_run_failure_details(error, code))
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_callback_accepts_langchain_messages_for_token_bound(
|
|
monkeypatch, tmp_path, caplog
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
sink = _Sink()
|
|
agent_input = _input()
|
|
host = WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
route = run._snapshot.purpose_routes["main_agent"][0]
|
|
|
|
with caplog.at_level("INFO", logger="EvoScientist.llm.runtime"):
|
|
await run._attempt_callback.on_chat_model_start(
|
|
{},
|
|
[[HumanMessage(content="hello")]],
|
|
run_id="callback",
|
|
metadata={
|
|
"runtime_purpose": "main_agent",
|
|
"route_key": route.identity.route_key,
|
|
},
|
|
)
|
|
await run._attempt_callback.on_llm_new_token(
|
|
"secret streamed text",
|
|
chunk=SimpleNamespace(
|
|
message=SimpleNamespace(
|
|
content="secret streamed text",
|
|
additional_kwargs={},
|
|
tool_call_chunks=[],
|
|
)
|
|
),
|
|
run_id="callback",
|
|
)
|
|
|
|
assert sink.events[0].payload["provider_input_bound_tokens"] > 0
|
|
assert run._attempt_callback.raise_error is True
|
|
assert "phase=request_started" in caplog.text
|
|
assert "phase=first_visible_chunk" in caplog.text
|
|
assert "visible_chunks=1" in caplog.text
|
|
assert "text_chars=20" in caplog.text
|
|
assert "secret streamed text" not in caplog.text
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
@pytest.mark.parametrize(
|
|
("confirmation", "expected_error"),
|
|
[
|
|
("absent", "EVENT_INGRESS_UNAVAILABLE"),
|
|
("conflict", "EVO_EVENT_CONFLICT"),
|
|
("unknown", "EVENT_COMMIT_INDETERMINATE"),
|
|
],
|
|
)
|
|
async def test_uncertain_event_commit_is_confirmed_before_dispatch(
|
|
monkeypatch, tmp_path, confirmation, expected_error
|
|
):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
sink = _UncertainSink(confirmation)
|
|
agent_input = _input()
|
|
host = WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
route = run._snapshot.purpose_routes["main_agent"][0]
|
|
|
|
with pytest.raises(EvoRuntimeError, match=expected_error):
|
|
await run._begin_callback_attempt(
|
|
callback_run_id="callback",
|
|
purpose="main_agent",
|
|
route=route,
|
|
provider_input_bound_tokens=10,
|
|
)
|
|
assert len(sink.confirmed) == 1
|
|
assert run._sequence == 0
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_uncertain_but_committed_event_advances_once(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
sink = _UncertainSink("committed")
|
|
agent_input = _input()
|
|
host = WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink)
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input), agent_input, host
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
route = run._snapshot.purpose_routes["main_agent"][0]
|
|
|
|
await run._begin_callback_attempt(
|
|
callback_run_id="callback",
|
|
purpose="main_agent",
|
|
route=route,
|
|
provider_input_bound_tokens=10,
|
|
)
|
|
|
|
assert run._sequence == 1
|
|
assert len(sink.confirmed) == 1
|
|
|
|
|
|
@pytest.mark.asyncio
|
|
async def test_stream_raises_when_terminal_commit_fails(monkeypatch, tmp_path):
|
|
runtime, authority = _runtime(tmp_path, monkeypatch)
|
|
sink = _TerminalFailingSink()
|
|
agent_input = _input()
|
|
quote = await runtime.prepare_model_run(
|
|
_preparation(authority, agent_input),
|
|
agent_input,
|
|
WebHostContext("/tmp", "/tmp", object(), object(), runtime_event_sink=sink),
|
|
)
|
|
run = await runtime.start_web_run(_admission(authority, quote))
|
|
|
|
async def consume() -> None:
|
|
async for _event in run.stream():
|
|
pass
|
|
|
|
with pytest.raises(EvoRuntimeError, match="EVO_EVENT_CONFLICT"):
|
|
async with asyncio.timeout(1):
|
|
await consume()
|