Files
EvoScientist-Multi/tests/test_web_search.py
T
m4 c683f6e739
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
feat: prepare EvoScientist 0.3.0
Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
2026-09-03 06:55:56 +08:00

133 lines
4.5 KiB
Python

from __future__ import annotations
import pytest
@pytest.mark.anyio
async def test_tavily_search_keeps_indexed_summary_when_source_fetch_fails(monkeypatch):
from EvoScientist.tools import search
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": "AIR staff profile",
"url": "https://air.cas.cn/example",
"content": "Indexed staff-profile summary.",
}
]
}
recorded: list[tuple[str, str]] = []
async def fetch_failed(_url: str, timeout: float = 10.0) -> str:
return "Error fetching content from https://air.cas.cn/example: DNS failed"
async def record(service: str, action: str) -> None:
recorded.append((service, action))
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", fetch_failed)
monkeypatch.setattr("EvoScientist.runtime_integrations.record_service_usage", record)
result = await search.tavily_search.ainvoke({"query": "高铭 空天院"})
assert "Indexed staff-profile summary." in result
assert "https://air.cas.cn/example" in result
assert "Tavily-indexed summary" in result
assert recorded == [("tavily", "search")]
@pytest.mark.anyio
async def test_tavily_search_bounds_fetched_page_content(monkeypatch):
from EvoScientist.tools import search
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": f"Result {index}",
"url": f"https://example.com/{index}",
"content": f"Indexed summary {index}",
}
for index in range(3)
]
}
async def huge_page(_url: str, timeout: float = 10.0) -> str:
return "page-content " * 10_000
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", huge_page)
result = await search.tavily_search.ainvoke({"query": "bounded search"})
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
for index in range(3):
assert f"https://example.com/{index}" in result
assert "[page content truncated]" in result
@pytest.mark.anyio
async def test_tavily_search_preserves_every_result_url_under_total_budget(monkeypatch):
from EvoScientist.tools import search
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": f"Result {index} " + ("very-long-title " * 800),
"url": f"https://example.com/result-{index}",
"content": f"Indexed summary {index}",
}
for index in range(3)
]
}
async def page(_url: str, timeout: float = 10.0) -> str:
return "page-content " * 1_000
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", page)
result = await search.tavily_search.ainvoke({"query": "preserve urls"})
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
for index in range(3):
assert f"https://example.com/result-{index}" in result
assert "[search result content truncated to preserve all titles and URLs]" in result
@pytest.mark.anyio
async def test_tavily_search_bounds_maliciously_long_url(monkeypatch):
from EvoScientist.tools import search
long_url = "https://example.com/" + ("a" * 20_000)
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": "Long URL result",
"url": long_url,
"content": "Indexed summary",
}
]
}
async def page(_url: str, timeout: float = 10.0) -> str:
return "page"
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", page)
result = await search.tavily_search.ainvoke({"query": "long url"})
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
assert "https://example.com/" in result
assert "[URL truncated]" in result