c683f6e739
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
133 lines
4.5 KiB
Python
133 lines
4.5 KiB
Python
from __future__ import annotations
|
|
|
|
import pytest
|
|
|
|
|
|
@pytest.mark.anyio
|
|
async def test_tavily_search_keeps_indexed_summary_when_source_fetch_fails(monkeypatch):
|
|
from EvoScientist.tools import search
|
|
|
|
class _Client:
|
|
def search(self, *_args, **_kwargs):
|
|
return {
|
|
"results": [
|
|
{
|
|
"title": "AIR staff profile",
|
|
"url": "https://air.cas.cn/example",
|
|
"content": "Indexed staff-profile summary.",
|
|
}
|
|
]
|
|
}
|
|
|
|
recorded: list[tuple[str, str]] = []
|
|
|
|
async def fetch_failed(_url: str, timeout: float = 10.0) -> str:
|
|
return "Error fetching content from https://air.cas.cn/example: DNS failed"
|
|
|
|
async def record(service: str, action: str) -> None:
|
|
recorded.append((service, action))
|
|
|
|
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
|
monkeypatch.setattr(search, "fetch_webpage_content", fetch_failed)
|
|
monkeypatch.setattr("EvoScientist.runtime_integrations.record_service_usage", record)
|
|
|
|
result = await search.tavily_search.ainvoke({"query": "高铭 空天院"})
|
|
|
|
assert "Indexed staff-profile summary." in result
|
|
assert "https://air.cas.cn/example" in result
|
|
assert "Tavily-indexed summary" in result
|
|
assert recorded == [("tavily", "search")]
|
|
|
|
|
|
@pytest.mark.anyio
|
|
async def test_tavily_search_bounds_fetched_page_content(monkeypatch):
|
|
from EvoScientist.tools import search
|
|
|
|
class _Client:
|
|
def search(self, *_args, **_kwargs):
|
|
return {
|
|
"results": [
|
|
{
|
|
"title": f"Result {index}",
|
|
"url": f"https://example.com/{index}",
|
|
"content": f"Indexed summary {index}",
|
|
}
|
|
for index in range(3)
|
|
]
|
|
}
|
|
|
|
async def huge_page(_url: str, timeout: float = 10.0) -> str:
|
|
return "page-content " * 10_000
|
|
|
|
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
|
monkeypatch.setattr(search, "fetch_webpage_content", huge_page)
|
|
|
|
result = await search.tavily_search.ainvoke({"query": "bounded search"})
|
|
|
|
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
|
|
for index in range(3):
|
|
assert f"https://example.com/{index}" in result
|
|
assert "[page content truncated]" in result
|
|
|
|
|
|
@pytest.mark.anyio
|
|
async def test_tavily_search_preserves_every_result_url_under_total_budget(monkeypatch):
|
|
from EvoScientist.tools import search
|
|
|
|
class _Client:
|
|
def search(self, *_args, **_kwargs):
|
|
return {
|
|
"results": [
|
|
{
|
|
"title": f"Result {index} " + ("very-long-title " * 800),
|
|
"url": f"https://example.com/result-{index}",
|
|
"content": f"Indexed summary {index}",
|
|
}
|
|
for index in range(3)
|
|
]
|
|
}
|
|
|
|
async def page(_url: str, timeout: float = 10.0) -> str:
|
|
return "page-content " * 1_000
|
|
|
|
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
|
monkeypatch.setattr(search, "fetch_webpage_content", page)
|
|
|
|
result = await search.tavily_search.ainvoke({"query": "preserve urls"})
|
|
|
|
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
|
|
for index in range(3):
|
|
assert f"https://example.com/result-{index}" in result
|
|
assert "[search result content truncated to preserve all titles and URLs]" in result
|
|
|
|
|
|
@pytest.mark.anyio
|
|
async def test_tavily_search_bounds_maliciously_long_url(monkeypatch):
|
|
from EvoScientist.tools import search
|
|
|
|
long_url = "https://example.com/" + ("a" * 20_000)
|
|
|
|
class _Client:
|
|
def search(self, *_args, **_kwargs):
|
|
return {
|
|
"results": [
|
|
{
|
|
"title": "Long URL result",
|
|
"url": long_url,
|
|
"content": "Indexed summary",
|
|
}
|
|
]
|
|
}
|
|
|
|
async def page(_url: str, timeout: float = 10.0) -> str:
|
|
return "page"
|
|
|
|
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
|
monkeypatch.setattr(search, "fetch_webpage_content", page)
|
|
|
|
result = await search.tavily_search.ainvoke({"query": "long url"})
|
|
|
|
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
|
|
assert "https://example.com/" in result
|
|
assert "[URL truncated]" in result
|