test: cover stop contract, execution adapters, checkpointer race and runtime identity
Docker / build (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Build / build (push) Has been cancelled
Docker / build (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Build / build (push) Has been cancelled
This commit is contained in:
@@ -0,0 +1,225 @@
|
||||
"""B03 offline input projection nodes. Each node has a five-run budget.
|
||||
|
||||
Run ledger (no historical web-contract nodes are exercised):
|
||||
malicious_values: 4/5 (prior 3; compatibility regression GREEN)
|
||||
effective_merge: 4/5 (prior 3; compatibility regression GREEN)
|
||||
malicious_payload: 3/5 (prior 2; compatibility regression GREEN)
|
||||
supported_thinking_controls: 4/5 (launcher failure; RED; GREEN; regression GREEN)
|
||||
legitimate_image_budget: 4/5 (RED; GREEN; regression; image forms GREEN)
|
||||
Final regression: 5 passed in 1.17s; final image forms: 1 passed in 0.41s.
|
||||
No selector/summary/history nodes exercised.
|
||||
Launcher failure: system Python lacks pytest (no node collected); used .venv.
|
||||
"""
|
||||
import asyncio
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
from pydantic import BaseModel
|
||||
|
||||
from EvoScientist.llm import runtime
|
||||
|
||||
|
||||
def test_supported_thinking_controls():
|
||||
from EvoScientist.llm.adapter_registry import get_adapter_registry
|
||||
from langchain_core.messages import HumanMessage
|
||||
from langchain_openai import ChatOpenAI
|
||||
from openai._base_client import _merge_mappings
|
||||
|
||||
adapter = get_adapter_registry().get("dashscope", "dashscope-v1")
|
||||
compiled = adapter.compile_runtime_parameters(
|
||||
"chat_completions", {"reasoning": "high", "reasoning_budget_tokens": 2048}, 4096
|
||||
)
|
||||
for controls in (compiled["extra_body"], {"enable_thinking": False},
|
||||
{"thinking": {"type": "disabled"}}):
|
||||
model = ChatOpenAI(api_key="offline-not-a-credential", model="offline",
|
||||
extra_body=controls, use_responses_api=False)
|
||||
invocation = model._get_invocation_params()
|
||||
assert runtime._effective_callback_input_parameters(invocation) == {}
|
||||
payload = model._get_request_payload([HumanMessage(content="local")])
|
||||
effective = _merge_mappings(payload, payload.pop("extra_body"))
|
||||
assert all(effective[key] == value for key, value in controls.items())
|
||||
assert invocation["extra_body"] == controls
|
||||
for invalid in ({"thinking": {"type": float("nan")}},
|
||||
{"enable_thinking": object()}, {"unknown_extension": True}):
|
||||
with pytest.raises(runtime.EvoRuntimeError, match="MODEL_INPUT_PROJECTION_INVALID"):
|
||||
runtime._effective_callback_input_parameters({"extra_body": invalid})
|
||||
|
||||
|
||||
def test_legitimate_image_budget():
|
||||
import base64
|
||||
import io
|
||||
from PIL import Image
|
||||
from EvoScientist.document_extract import MAX_IMAGE_BYTES, prepare_image_bytes
|
||||
from langchain_core.messages import HumanMessage
|
||||
|
||||
output = io.BytesIO()
|
||||
Image.new("RGB", (2048, 2048)).save(output, "PNG", compress_level=0)
|
||||
raw = output.getvalue()
|
||||
assert len(raw) > runtime._INPUT_PROJECTION_MAX_BYTES
|
||||
raw += b"\0" * (MAX_IMAGE_BYTES - len(raw))
|
||||
assert prepare_image_bytes(raw, "boundary.png") == raw
|
||||
uri = "data:image/png;base64," + base64.b64encode(raw).decode("ascii")
|
||||
block = {"type": "image_url", "image_url": {"url": uri}}
|
||||
payload = {"messages": runtime._callback_messages_payload([
|
||||
[HumanMessage(content=[block, block])]
|
||||
])}
|
||||
bound = runtime._provider_input_token_bound(payload)
|
||||
assert bound.media_blocks == 2
|
||||
assert bound.largest_media_bytes == MAX_IMAGE_BYTES
|
||||
assert bound.media_tokens == 2 * ((len(raw) + 2) // 3 + 512)
|
||||
assert bound.text_tokens < 4096
|
||||
small = io.BytesIO()
|
||||
Image.new("RGB", (1, 1)).save(small, "PNG")
|
||||
encoded = base64.b64encode(small.getvalue()).decode("ascii")
|
||||
forms = [
|
||||
{"type": "image", "mime_type": "image/png", "base64": encoded},
|
||||
{"type": "image", "source": {"type": "base64", "media_type": "image/png", "data": encoded}},
|
||||
{"inline_data": {"mime_type": "image/png", "data": encoded}},
|
||||
{"type": "input_image", "image_url": "data:image/png;base64," + encoded},
|
||||
]
|
||||
assert runtime._provider_input_token_bound(forms).media_blocks == len(forms)
|
||||
assert runtime._provider_input_token_bound([
|
||||
{"type": "image_url", "image_url": {"url": "https://example.invalid/image.png"}}
|
||||
]).media_blocks == 0
|
||||
assert payload["messages"][0][0]["data"]["content"][0]["image_url"]["url"] == uri
|
||||
invalid = [
|
||||
{"type": "image_url", "image_url": {"url": "data:image/png;base64,eA=="}},
|
||||
{"type": "image_url", "image_url": {"url": "data:unknown/fake;base64,eA=="}},
|
||||
{"type": "text", "base64": uri},
|
||||
{"type": "image_url", "image_url": {"url": uri + "AAAA"}},
|
||||
]
|
||||
for item in invalid:
|
||||
with pytest.raises(runtime.EvoRuntimeError):
|
||||
runtime._provider_input_token_bound([item])
|
||||
for item in ({"schema": block}, {"text": "x" * (8 * 1024 * 1024 + 1)},
|
||||
{"unknown": "data:image/png;base64,eA=="}):
|
||||
with pytest.raises(runtime.EvoRuntimeError, match="MODEL_INPUT_PROJECTION_INVALID"):
|
||||
runtime._provider_input_token_bound(item)
|
||||
|
||||
|
||||
def test_malicious_values_fail_closed():
|
||||
touched = []
|
||||
|
||||
class Duck:
|
||||
@classmethod
|
||||
def model_json_schema(cls):
|
||||
touched.append("schema")
|
||||
return {}
|
||||
|
||||
class Poison:
|
||||
def __repr__(self):
|
||||
touched.append("repr")
|
||||
raise AssertionError("must not format rejected input")
|
||||
|
||||
class Selection(BaseModel):
|
||||
name: str
|
||||
|
||||
class BadSchema(BaseModel):
|
||||
@classmethod
|
||||
def model_json_schema(cls, *args, **kwargs):
|
||||
return {"invalid": float("nan")}
|
||||
|
||||
cycle = []
|
||||
cycle.append(cycle)
|
||||
deep = None
|
||||
for _ in range(66):
|
||||
deep = [deep]
|
||||
cases = [
|
||||
{1: "non-string key"}, Duck, Poison(), cycle, deep,
|
||||
float("nan"), float("inf"), -float("inf"),
|
||||
("tuple",), {"set"}, b"bytes", BadSchema,
|
||||
[None] * 100_001, "x" * (8 * 1024 * 1024 + 1),
|
||||
]
|
||||
failures = []
|
||||
for index, value in enumerate(cases):
|
||||
try:
|
||||
runtime._callback_input_parameters(value)
|
||||
except runtime.EvoRuntimeError as exc:
|
||||
if exc.code != "MODEL_INPUT_PROJECTION_INVALID":
|
||||
failures.append((index, "wrong code"))
|
||||
except Exception:
|
||||
failures.append((index, "uncontrolled exception"))
|
||||
else:
|
||||
failures.append((index, "accepted"))
|
||||
assert not failures, failures
|
||||
assert not touched
|
||||
assert runtime._callback_input_parameters(Selection) == Selection.model_json_schema()
|
||||
assert runtime._callback_input_parameters({"safe": [None, True, 1, 1.5, "ok"]}) == {
|
||||
"safe": [None, True, 1, 1.5, "ok"]
|
||||
}
|
||||
shared = {"value": 1}
|
||||
assert runtime._callback_input_parameters([shared, shared]) == [shared, shared]
|
||||
|
||||
|
||||
def test_effective_merge_callback_bound(monkeypatch):
|
||||
from langchain_core.messages import HumanMessage
|
||||
from langchain_openai import ChatOpenAI
|
||||
from openai._base_client import _merge_mappings
|
||||
|
||||
messages = [HumanMessage(content="offline merge conflict")]
|
||||
defaults = {"tools": [{"type": "function", "function": {"name": "default"}}]}
|
||||
body = {
|
||||
"tools": [{"type": "function", "function": {"name": "winner"}}],
|
||||
"response_format": {"type": "json_object"},
|
||||
"system": "body system", "instructions": "body instructions",
|
||||
}
|
||||
model = ChatOpenAI(api_key="offline-not-a-credential", model_kwargs=defaults,
|
||||
extra_body=body, use_responses_api=False)
|
||||
call = {"tools": [{"type": "function", "function": {"name": "call"}}],
|
||||
"response_format": {"type": "json_schema", "json_schema": {"name": "loser"}},
|
||||
"system": "call system", "instructions": "call instructions"}
|
||||
invocation = model._get_invocation_params(**call)
|
||||
payload = model._get_request_payload(messages, **call)
|
||||
effective = _merge_mappings(payload, payload.pop("extra_body"))
|
||||
expected = {key: effective[key] for key in body}
|
||||
assert expected == body
|
||||
captured = []
|
||||
|
||||
class BoundaryReached(Exception):
|
||||
pass
|
||||
|
||||
async def begin(**kwargs):
|
||||
captured.append(kwargs)
|
||||
raise BoundaryReached
|
||||
|
||||
callback = runtime._RuntimeAttemptCallback(SimpleNamespace(_begin_callback_attempt=begin))
|
||||
monkeypatch.setattr(callback, "_route_for", lambda _: ("tool_selector", None))
|
||||
monkeypatch.setattr(runtime, "_callback_start_failure_details", lambda *args: {})
|
||||
async def invoke(params):
|
||||
await callback.on_chat_model_start({}, [messages], run_id="merge", invocation_params=params)
|
||||
|
||||
with pytest.raises(BoundaryReached):
|
||||
asyncio.run(invoke(invocation))
|
||||
expected_bound = runtime._provider_input_token_bound({
|
||||
"messages": runtime._callback_messages_payload([messages]), **expected,
|
||||
}).total_tokens
|
||||
assert captured[0]["provider_input_bound_tokens"] == expected_bound
|
||||
assert captured[0]["purpose"] == "tool_selector"
|
||||
for invalid in (
|
||||
{"model_kwargs": {"tools": defaults["tools"]}},
|
||||
{"extra_body": {"input": "unprojected context"}},
|
||||
{"extra_body": {"unknown_prompt": "unprojected context"}},
|
||||
{"extra_body": {"messages": []}},
|
||||
{"tools": {1: "bad key"}},
|
||||
[],
|
||||
):
|
||||
with pytest.raises(runtime.EvoRuntimeError, match="MODEL_INPUT_PROJECTION_INVALID"):
|
||||
asyncio.run(invoke(invalid))
|
||||
assert len(captured) == 1
|
||||
|
||||
|
||||
def test_malicious_payload_rejected_before_media_recursion():
|
||||
cycle = {"content": []}
|
||||
cycle["content"].append(cycle)
|
||||
failures = []
|
||||
for index, value in enumerate((cycle, {"content": float("nan")}, {1: "bad"})):
|
||||
try:
|
||||
runtime._provider_input_token_bound(value)
|
||||
except runtime.EvoRuntimeError as exc:
|
||||
if exc.code != "MODEL_INPUT_PROJECTION_INVALID":
|
||||
failures.append((index, "wrong code"))
|
||||
except Exception:
|
||||
failures.append((index, "uncontrolled exception"))
|
||||
else:
|
||||
failures.append((index, "accepted"))
|
||||
assert not failures, failures
|
||||
Reference in New Issue
Block a user