Files
EvoScientist-Multi/EvoScientist/tools/search.py
T
m4 c683f6e739
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
feat: prepare EvoScientist 0.3.0
Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
2026-09-03 06:55:56 +08:00

159 lines
5.8 KiB
Python

"""Web search tools.
Provides ``tavily_search`` and ``fetch_webpage_content`` for the research agent,
using Tavily for URL discovery and fetching full webpage content.
"""
import asyncio
from typing import Annotated, Literal
import httpx
from langchain_core.tools import InjectedToolArg, tool
from markdownify import markdownify
from tavily import TavilyClient
# Lazy initialization - only create client when needed
_tavily_client = None
MAX_SEARCH_RESULTS = 5
MAX_DISPLAY_QUERY_CHARS = 512
MAX_DISPLAY_TITLE_CHARS = 512
MAX_DISPLAY_URL_CHARS = 2_048
MAX_PAGE_CONTENT_CHARS = 4_000
MAX_SEARCH_RESULT_CHARS = 16_000
_TRUNCATION_MARKER = "\n\n[page content truncated]"
_SEARCH_TRUNCATION_MARKER = (
"\n[search result content truncated to preserve all titles and URLs]"
)
def _get_tavily_client() -> TavilyClient:
"""Get or create the Tavily client (lazy initialization)."""
global _tavily_client
if _tavily_client is None:
_tavily_client = TavilyClient()
return _tavily_client
async def fetch_webpage_content(url: str, timeout: float = 10.0) -> str:
"""Fetch and convert webpage content to markdown.
Args:
url: URL to fetch
timeout: Request timeout in seconds
Returns:
Webpage content as markdown
"""
headers = {
"User-Agent": (
"Mozilla/5.0 (Windows NT 10.0; Win64; x64) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/91.0.4472.124 Safari/537.36"
)
}
try:
async with httpx.AsyncClient() as client:
response = await client.get(url, headers=headers, timeout=timeout)
response.raise_for_status()
content_type = response.headers.get("content-type", "").lower()
if not any(
allowed in content_type
for allowed in ("text/", "application/xhtml+xml")
):
return f"Error fetching content from {url}: unsupported content type {content_type or 'unknown'}"
return markdownify(response.text)
except Exception as e:
return f"Error fetching content from {url}: {e!s}"
@tool(parse_docstring=True)
async def tavily_search(
query: str,
max_results: Annotated[int, InjectedToolArg] = 3,
topic: Annotated[
Literal["general", "news", "finance"], InjectedToolArg
] = "general",
) -> str:
"""Search the web for information on a given query.
Uses Tavily to discover relevant URLs, then fetches and returns
full webpage content as markdown for comprehensive research.
Args:
query: Search query to execute
Returns:
Formatted search results with full webpage content in markdown
"""
def _sync_search() -> dict:
bounded_max_results = max(1, min(int(max_results), MAX_SEARCH_RESULTS))
return _get_tavily_client().search(
query,
max_results=bounded_max_results,
topic=topic,
)
try:
# Run Tavily search asynchronously through the controlled host path.
search_results = await asyncio.to_thread(_sync_search)
from EvoScientist.runtime_integrations import record_service_usage
await record_service_usage("tavily", "search")
results = search_results.get("results", [])
if not results:
return f"No results found for '{query}'"
fetch_tasks = [fetch_webpage_content(r["url"]) for r in results]
contents = await asyncio.gather(*fetch_tasks)
normalized = []
for result, fetched_content in zip(results, contents, strict=False):
title = str(result.get("title") or "Untitled")[:MAX_DISPLAY_TITLE_CHARS]
raw_url = str(result.get("url") or "")
url = (
raw_url
if len(raw_url) <= MAX_DISPLAY_URL_CHARS
else raw_url[: MAX_DISPLAY_URL_CHARS - len("...[URL truncated]")]
+ "...[URL truncated]"
)
tavily_summary = str(result.get("content") or "").strip()
fetch_failed = fetched_content.startswith("Error fetching content from ")
content = tavily_summary if fetch_failed and tavily_summary else fetched_content
fetch_note = (
"\n\n> Source page fetch failed; showing the Tavily-indexed summary."
if fetch_failed and tavily_summary
else ""
)
normalized.append((title, url, content, fetch_note))
display_query = query[:MAX_DISPLAY_QUERY_CHARS]
prefix = f"Found {len(normalized)} live web result(s) for '{display_query}':\n\n"
metadata_blocks = [f"## {title}\n**URL:** {url}\n\n" for title, url, _, _ in normalized]
fixed_chars = len(prefix) + sum(len(block) + len("\n\n---\n") for block in metadata_blocks)
remaining = max(0, MAX_SEARCH_RESULT_CHARS - fixed_chars - len(_SEARCH_TRUNCATION_MARKER))
per_result_budget = remaining // max(1, len(normalized))
result_texts = []
content_truncated = False
for metadata, (_, _, content, fetch_note) in zip(metadata_blocks, normalized, strict=True):
content_budget = max(0, min(MAX_PAGE_CONTENT_CHARS, per_result_budget - len(fetch_note)))
if len(content) > content_budget:
marker_budget = min(len(_TRUNCATION_MARKER), content_budget)
content = (
content[: content_budget - marker_budget]
+ _TRUNCATION_MARKER[:marker_budget]
)
content_truncated = True
result_texts.append(f"{metadata}{content}{fetch_note}\n\n---\n")
formatted = prefix + "".join(result_texts)
if content_truncated:
formatted += _SEARCH_TRUNCATION_MARKER
return formatted
except Exception as e:
return f"Search failed: {e!s}"