diff --git a/EvoScientist/EvoScientist.py b/EvoScientist/EvoScientist.py index ce0d153..0a51282 100644 --- a/EvoScientist/EvoScientist.py +++ b/EvoScientist/EvoScientist.py @@ -555,9 +555,10 @@ def _build_base_kwargs( cfg = cfg if cfg is not None else _ensure_config() tool_registry = {"think_tool": think_tool} + base_tools = [think_tool, skill_manager] if os.environ.get("TAVILY_API_KEY"): tool_registry["tavily_search"] = tavily_search - base_tools = [think_tool, skill_manager] + base_tools.append(tavily_search) subs = load_subagents( SUBAGENTS_CONFIG, @@ -617,15 +618,54 @@ def load_mcp_and_build_kwargs( ) tool_registry = {"think_tool": think_tool} + base_tools = [think_tool, skill_manager] if os.environ.get("TAVILY_API_KEY"): tool_registry["tavily_search"] = tavily_search - base_tools = [think_tool, skill_manager] + base_tools.append(tavily_search) - # Fresh tool registry — start from base tools + MCP tools + # DeepAgents installs these outside ``base_tools`` through middleware. + # MCP tools must never shadow them inside any one agent namespace. + middleware_tool_names = { + "ls", + "read_file", + "write_file", + "edit_file", + "glob", + "grep", + "execute", + "write_todos", + "task", + "start_async_task", + "check_async_task", + "update_async_task", + "cancel_async_task", + "list_async_tasks", + } + + # Fresh tool registry — start from built-ins, then add one representative + # MCP implementation for YAML name resolution. A tool may be exposed to + # several agents, but no agent may contain duplicate names and no MCP tool + # may override a built-in implementation. registry = dict(tool_registry) - for tools in mcp_by_agent.values(): + builtin_names = { + *(str(getattr(tool, "name", "")) for tool in base_tools), + *middleware_tool_names, + } + for agent_name, tools in mcp_by_agent.items(): + seen_for_agent: set[str] = set() for t in tools: - registry[t.name] = t + tool_name = str(t.name) + if tool_name in builtin_names or tool_name in seen_for_agent: + from .llm.contracts import EvoRuntimeError + + raise EvoRuntimeError( + "TOOL_REGISTRY_CONFLICT", + details=( + {"agent_name": str(agent_name), "tool_name": tool_name}, + ), + ) + seen_for_agent.add(tool_name) + registry.setdefault(tool_name, t) mcp_main = mcp_by_agent.pop("main", []) @@ -639,10 +679,31 @@ def load_mcp_and_build_kwargs( subs, workspace_dir=workspace_dir, cfg=cfg, chat_model=chat_model ) - # Inject MCP tools into subagents by name + # Inject MCP tools into subagents by name. YAML-resolved tools already + # belong to that agent namespace, so a second tool with the same name is a + # configuration conflict rather than an item to append silently. for sa in subs: if sa_tools := mcp_by_agent.get(sa["name"], []): - sa.setdefault("tools", []).extend(sa_tools) + target_tools = sa.setdefault("tools", []) + existing_names = { + str(getattr(tool, "name", tool)) for tool in target_tools + } + for tool in sa_tools: + tool_name = str(tool.name) + if tool_name in existing_names: + from .llm.contracts import EvoRuntimeError + + raise EvoRuntimeError( + "TOOL_REGISTRY_CONFLICT", + details=( + { + "agent_name": str(sa["name"]), + "tool_name": tool_name, + }, + ), + ) + existing_names.add(tool_name) + target_tools.append(tool) # Swap selected sub-agents to AsyncSubAgent (must happen AFTER MCP injection # since async sub-agents are remote graphs that load their own tools). diff --git a/EvoScientist/__init__.py b/EvoScientist/__init__.py index 6eaebe2..c566dd1 100644 --- a/EvoScientist/__init__.py +++ b/EvoScientist/__init__.py @@ -9,7 +9,7 @@ from __future__ import annotations from importlib import import_module -__version__ = "0.2.2" +from ._version import __version__ _EXPORTS: dict[str, tuple[str, str]] = { # Agent graph (lazy to avoid expensive initialization at import time) @@ -73,4 +73,5 @@ def __dir__() -> list[str]: return sorted(set(globals()) | set(_EXPORTS)) -__all__ = list(_EXPORTS) +__all__ = ["__version__"] +__all__.extend(_EXPORTS) diff --git a/EvoScientist/_version.py b/EvoScientist/_version.py new file mode 100644 index 0000000..597a009 --- /dev/null +++ b/EvoScientist/_version.py @@ -0,0 +1,3 @@ +"""Package version shared by builds and runtime.""" + +__version__ = "0.3.0" diff --git a/EvoScientist/document_extract.py b/EvoScientist/document_extract.py new file mode 100644 index 0000000..59521b5 --- /dev/null +++ b/EvoScientist/document_extract.py @@ -0,0 +1,466 @@ +"""Bounded document extraction and non-text file policy for workspaces.""" + +from __future__ import annotations + +import os +import subprocess +import sys +import tempfile +import zipfile +from pathlib import Path +from xml.etree import ElementTree as ET + +MAX_DOCUMENT_BYTES = 50 * 1024 * 1024 +MAX_DOCUMENT_RESULT_CHARS = 50_000 +MAX_CONVERTED_DOCUMENT_BYTES = 10 * 1024 * 1024 +DOCUMENT_CONVERSION_TIMEOUT_SECONDS = 60 +MAX_IMAGE_BYTES = 25 * 1024 * 1024 +MAX_IMAGE_EDGE = 2048 +MAX_IMAGE_PIXELS = 40_000_000 +MAX_OOXML_MEMBERS = 10_000 +MAX_OOXML_MEMBER_BYTES = 50 * 1024 * 1024 +MAX_OOXML_EXPANDED_BYTES = 200 * 1024 * 1024 +MAX_OOXML_COMPRESSION_RATIO = 100 + +IMAGE_EXTENSIONS = frozenset( + {".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".png", ".tif", ".tiff", ".webp"} +) +DOCUMENT_EXTENSIONS = frozenset( + { + ".doc", + ".docm", + ".docx", + ".epub", + ".odp", + ".ods", + ".odt", + ".pdf", + ".pot", + ".pps", + ".ppsm", + ".ppsx", + ".ppt", + ".pptm", + ".pptx", + ".rtf", + ".xls", + ".xlsb", + ".xlsm", + ".xlsx", + } +) +ARCHIVE_EXTENSIONS = frozenset( + {".7z", ".bz2", ".gz", ".rar", ".tar", ".tgz", ".xz", ".zip"} +) +DATABASE_EXTENSIONS = frozenset({".db", ".sqlite", ".sqlite3"}) +EXECUTABLE_EXTENSIONS = frozenset( + {".app", ".deb", ".dll", ".dylib", ".elf", ".exe", ".msi", ".rpm", ".so"} +) +DATASET_EXTENSIONS = frozenset( + {".arrow", ".feather", ".h5", ".hdf5", ".npy", ".npz", ".parquet"} +) +MEDIA_EXTENSIONS = frozenset( + {".aac", ".avi", ".flac", ".m4a", ".mkv", ".mov", ".mp3", ".mp4", ".ogg", ".wav", ".webm"} +) + +_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +_A = "http://schemas.openxmlformats.org/drawingml/2006/main" +_S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main" +_R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships" + + +class DocumentExtractionError(RuntimeError): + """A supported document could not be converted to bounded text.""" + + +def classify_file(path: str, head: bytes = b"") -> str: + """Classify a workspace file into one policy category.""" + + extension = Path(path).suffix.lower() + if extension in IMAGE_EXTENSIONS: + return "image" + if extension in DOCUMENT_EXTENSIONS: + return "document" + if extension in ARCHIVE_EXTENSIONS: + return "archive" + if extension in DATABASE_EXTENSIONS or head.startswith(b"SQLite format 3\x00"): + return "database" + if extension in EXECUTABLE_EXTENSIONS or head.startswith((b"MZ", b"\x7fELF")): + return "executable" + if extension in DATASET_EXTENSIONS: + return "dataset" + if extension in MEDIA_EXTENSIONS: + return "media" + return "unknown" + + +def binary_processing_guidance(path: str, kind: str, size_bytes: int) -> str: + """Return bounded, actionable JSON-like guidance for Agent-side programming.""" + + import json + + common = { + "code": "BINARY_PROCESSING_REQUIRED" + if kind != "binary" + else "UNSUPPORTED_BINARY_FILE", + "path": path, + "kind": kind, + "size_bytes": size_bytes, + } + if kind == "archive": + common.update( + action=( + "Use execute with Python to list and validate archive members before " + "selective extraction; never use extractall." + ), + constraints={ + "list_before_extract": True, + "max_members": 2000, + "max_total_uncompressed_bytes": 500 * 1024 * 1024, + "max_member_bytes": 100 * 1024 * 1024, + "max_compression_ratio": 100, + "reject_absolute_or_parent_paths": True, + "do_not_execute_members": True, + }, + ) + elif kind == "database": + common.update( + action=( + "Use execute with Python sqlite3 in read-only mode: " + "file:?mode=ro&immutable=1; set PRAGMA query_only=ON; " + "inspect schema, run bounded SELECT queries with LIMIT, and write " + "large results under artifacts/." + ), + constraints={ + "read_only": True, + "mode": "mode=ro", + "query_only": True, + "max_rows": 1000, + "forbid_attach_database": True, + "forbid_load_extension": True, + }, + ) + elif kind == "executable": + common.update( + action=( + "Use execute only for bounded static metadata inspection (hash, file " + "headers, signature, imports, strings); this file must not be executed." + ), + constraints={"must_not_be_executed": True, "static_analysis_only": True}, + ) + elif kind == "dataset": + common.update( + action=( + "Use execute with the appropriate library to inspect schema, dimensions, " + "statistics, and a bounded sample; do not serialize the whole dataset." + ) + ) + elif kind == "media": + common.update( + action=( + "Use execute with ffprobe/ffmpeg or an available transcription workflow " + "to inspect metadata and selected ranges; do not inline the complete file." + ) + ) + else: + common.update( + kind="binary", + action=( + "This is an unsupported binary file. Use execute only for bounded static " + "inspection; do not execute it or inline its bytes." + ), + ) + return json.dumps(common, ensure_ascii=False, sort_keys=True) + + +def prepare_image_bytes(data: bytes, path: str) -> bytes: + """Validate and downsample an image before it becomes a model media block.""" + + import io + + if len(data) > MAX_IMAGE_BYTES: + raise DocumentExtractionError( + f"IMAGE_TOO_LARGE: {len(data)} bytes exceeds {MAX_IMAGE_BYTES}" + ) + try: + from PIL import Image + + image = Image.open(io.BytesIO(data)) + width, height = image.size + if width * height > MAX_IMAGE_PIXELS: + raise DocumentExtractionError( + f"IMAGE_PIXEL_BUDGET_EXCEEDED: {width}x{height} exceeds " + f"{MAX_IMAGE_PIXELS} pixels" + ) + image.load() + except DocumentExtractionError: + raise + except Exception as exc: + raise DocumentExtractionError( + f"IMAGE_PROCESSING_FAILED: {path}: {type(exc).__name__}: {exc}" + ) from exc + + frame_count = int(getattr(image, "n_frames", 1) or 1) + if frame_count > 1: + image.seek(0) + work = image.convert("RGBA" if image.mode in {"RGBA", "LA"} else "RGB") + else: + work = image.copy() + + if max(work.size) <= MAX_IMAGE_EDGE and frame_count == 1: + return data + + work.thumbnail((MAX_IMAGE_EDGE, MAX_IMAGE_EDGE), Image.Resampling.LANCZOS) + has_alpha = work.mode in {"RGBA", "LA"} or ( + work.mode == "P" and "transparency" in work.info + ) + output = io.BytesIO() + if has_alpha: + if work.mode == "P": + work = work.convert("RGBA") + work.save(output, "PNG", optimize=True) + else: + if work.mode != "RGB": + work = work.convert("RGB") + work.save(output, "JPEG", quality=85, optimize=True) + return output.getvalue() + + +def extract_document_bytes(data: bytes, path: str) -> str: + """Extract readable text from a supported document without exposing bytes.""" + + if len(data) > MAX_DOCUMENT_BYTES: + raise DocumentExtractionError( + f"DOCUMENT_TOO_LARGE: {len(data)} bytes exceeds {MAX_DOCUMENT_BYTES}" + ) + extension = Path(path).suffix.lower() + try: + if extension == ".docx": + return _extract_docx(data) + if extension == ".pptx": + return _extract_pptx(data) + if extension == ".xlsx": + return _extract_xlsx(data) + return _extract_anydoc(data, extension) + except DocumentExtractionError: + raise + except Exception as exc: + raise DocumentExtractionError( + f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}" + ) from exc + + +def paginate_document_text(text: str, *, offset: int, limit: int) -> str: + """Apply line and character budgets to extracted document text.""" + + lines = text.splitlines(keepends=True) + if not lines: + return "(document contains no extractable text)" + if offset >= len(lines): + raise DocumentExtractionError( + f"Line offset {offset} exceeds extracted document length ({len(lines)} lines)" + ) + selected = "".join(lines[offset : offset + limit]) + if len(selected) <= MAX_DOCUMENT_RESULT_CHARS: + return selected + trimmed = selected[:MAX_DOCUMENT_RESULT_CHARS] + boundary = trimmed.rfind("\n") + if boundary > 0: + trimmed = trimmed[: boundary + 1] + consumed = max(1, len(trimmed.splitlines())) + return ( + trimmed + + f"\n[DOCUMENT_OUTPUT_TRUNCATED: use offset={offset + consumed} to continue; " + + f"single-read limit is {MAX_DOCUMENT_RESULT_CHARS} characters]\n" + ) + + +def _validated_ooxml_archive(data: bytes) -> zipfile.ZipFile: + try: + archive = zipfile.ZipFile(_bytes_path(data)) + members = archive.infolist() + except zipfile.BadZipFile as exc: + raise DocumentExtractionError( + "DOCUMENT_EXTRACTION_FAILED: invalid OOXML container" + ) from exc + expanded = 0 + names: set[str] = set() + if len(members) > MAX_OOXML_MEMBERS: + archive.close() + raise DocumentExtractionError( + f"DOCUMENT_RESOURCE_LIMIT: OOXML has {len(members)} members; " + f"limit is {MAX_OOXML_MEMBERS}" + ) + for member in members: + normalized = member.filename.replace("\\", "/") + parts = tuple(part for part in normalized.split("/") if part) + if ( + normalized.startswith("/") + or ".." in parts + or member.filename in names + or bool(member.flag_bits & 0x1) + ): + archive.close() + raise DocumentExtractionError( + "DOCUMENT_RESOURCE_LIMIT: OOXML contains an unsafe, duplicate, " + "or encrypted member" + ) + names.add(member.filename) + expanded += member.file_size + ratio = member.file_size / max(member.compress_size, 1) + if ( + member.file_size > MAX_OOXML_MEMBER_BYTES + or expanded > MAX_OOXML_EXPANDED_BYTES + or ratio > MAX_OOXML_COMPRESSION_RATIO + ): + archive.close() + raise DocumentExtractionError( + "DOCUMENT_RESOURCE_LIMIT: OOXML member expansion exceeds safety limits" + ) + return archive + + +def _zip_xml(data: bytes, member: str) -> ET.Element: + try: + with _validated_ooxml_archive(data) as archive: + raw = archive.read(member) + except KeyError as exc: + raise DocumentExtractionError( + f"DOCUMENT_EXTRACTION_FAILED: missing {member}" + ) from exc + return ET.fromstring(raw) + + +def _bytes_path(data: bytes): + import io + + return io.BytesIO(data) + + +def _ooxml_part_number(name: str) -> int: + stem = Path(name).stem + digits = "".join(character for character in stem if character.isdigit()) + return int(digits) if digits else 0 + + +def _extract_docx(data: bytes) -> str: + root = _zip_xml(data, "word/document.xml") + paragraphs: list[str] = [] + for paragraph in root.iter(f"{{{_W}}}p"): + text = "".join(node.text or "" for node in paragraph.iter(f"{{{_W}}}t")) + if text: + paragraphs.append(text) + if not paragraphs: + raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: DOCX has no text") + return "\n".join(paragraphs) + "\n" + + +def _extract_pptx(data: bytes) -> str: + try: + with _validated_ooxml_archive(data) as archive: + names = sorted( + name + for name in archive.namelist() + if name.startswith("ppt/slides/slide") and name.endswith(".xml") + ) + slides: list[str] = [] + for index, name in enumerate(sorted(names, key=_ooxml_part_number), 1): + root = ET.fromstring(archive.read(name)) + texts = [node.text or "" for node in root.iter(f"{{{_A}}}t")] + slides.append(f"## Slide {index}\n" + "\n".join(t for t in texts if t)) + except zipfile.BadZipFile as exc: + raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid PPTX") from exc + if not slides: + raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: PPTX has no slides") + return "\n\n".join(slides) + "\n" + + +def _extract_xlsx(data: bytes) -> str: + try: + with _validated_ooxml_archive(data) as archive: + shared: list[str] = [] + if "xl/sharedStrings.xml" in archive.namelist(): + root = ET.fromstring(archive.read("xl/sharedStrings.xml")) + shared = [ + "".join(node.text or "" for node in item.iter(f"{{{_S}}}t")) + for item in root.iter(f"{{{_S}}}si") + ] + sheets = sorted( + name + for name in archive.namelist() + if name.startswith("xl/worksheets/sheet") and name.endswith(".xml") + ) + output: list[str] = [] + for index, name in enumerate(sheets, 1): + root = ET.fromstring(archive.read(name)) + output.append(f"## Sheet {index}") + for row in root.iter(f"{{{_S}}}row"): + values: list[str] = [] + for cell in row.iter(f"{{{_S}}}c"): + value_node = cell.find(f"{{{_S}}}v") + value = value_node.text if value_node is not None else "" + if cell.get("t") == "s" and value and value.isdigit(): + shared_index = int(value) + value = shared[shared_index] if shared_index < len(shared) else value + values.append(value or "") + output.append("\t".join(values)) + except zipfile.BadZipFile as exc: + raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid XLSX") from exc + if len(output) <= 1: + raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: XLSX has no sheets") + return "\n".join(output) + "\n" + + +def _extract_anydoc(data: bytes, extension: str) -> str: + source = "" + output = "" + try: + with tempfile.NamedTemporaryFile(suffix=extension, delete=False) as handle: + handle.write(data) + source = handle.name + with tempfile.NamedTemporaryFile(suffix=".md", delete=False) as handle: + output = handle.name + script = ( + "import pathlib,sys; import anydoc; " + "text=anydoc.to_markdown(sys.argv[1]); " + "pathlib.Path(sys.argv[2]).write_text(text, encoding='utf-8')" + ) + subprocess.run( + [sys.executable, "-c", script, source, output], + check=True, + capture_output=True, + timeout=DOCUMENT_CONVERSION_TIMEOUT_SECONDS, + ) + output_path = Path(output) + if output_path.stat().st_size > MAX_CONVERTED_DOCUMENT_BYTES: + raise DocumentExtractionError( + "DOCUMENT_RESOURCE_LIMIT: converted document exceeds output budget" + ) + text = output_path.read_text(encoding="utf-8") + except subprocess.TimeoutExpired as exc: + raise DocumentExtractionError( + f"DOCUMENT_CONVERSION_TIMEOUT: exceeded {DOCUMENT_CONVERSION_TIMEOUT_SECONDS}s" + ) from exc + except subprocess.CalledProcessError as exc: + detail = exc.stderr.decode("utf-8", errors="replace")[-1000:] + raise DocumentExtractionError( + f"DOCUMENT_EXTRACTION_FAILED: converter exited {exc.returncode}: {detail}" + ) from exc + except Exception as exc: + if isinstance(exc, DocumentExtractionError): + raise + raise DocumentExtractionError( + f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}" + ) from exc + finally: + for temporary in (source, output): + if temporary: + try: + os.unlink(temporary) + except OSError: + pass + if not isinstance(text, str) or not text.strip(): + raise DocumentExtractionError( + "DOCUMENT_EXTRACTION_FAILED: document contains no extractable text" + ) + return text.rstrip("\n") + "\n" diff --git a/EvoScientist/llm/contracts.py b/EvoScientist/llm/contracts.py index 7f55ed9..7157227 100644 --- a/EvoScientist/llm/contracts.py +++ b/EvoScientist/llm/contracts.py @@ -55,6 +55,11 @@ class EvoRuntimeError(RuntimeError): self.code = code self.details = tuple(dict(item) for item in details) + def __repr__(self) -> str: + # LangGraph persists task failures using repr(exc). Keep that snapshot + # machine-readable without serializing provider messages or details. + return f"{type(self).__name__}(code={self.code!r})" + def now_ms() -> int: return time.time_ns() // 1_000_000 diff --git a/EvoScientist/llm/gateway_proxy.py b/EvoScientist/llm/gateway_proxy.py index 94a9b35..c28bb62 100644 --- a/EvoScientist/llm/gateway_proxy.py +++ b/EvoScientist/llm/gateway_proxy.py @@ -14,6 +14,8 @@ from langchain_core.tools import BaseTool from langchain_core.utils.function_calling import convert_to_openai_tool from pydantic import Field +from .contracts import EvoRuntimeError + class GatewayProxyChatModel(BaseChatModel): gateway_url: str @@ -78,7 +80,9 @@ class GatewayProxyChatModel(BaseChatModel): ) -> ChatResult: del stop, kwargs attempt_id = self._attempt_id(run_manager) - async with httpx.AsyncClient(timeout=httpx.Timeout(660.0, connect=5.0)) as client: + async with httpx.AsyncClient( + timeout=httpx.Timeout(660.0, connect=5.0) + ) as client: response = await client.post( f"{self.gateway_url.rstrip('/')}/api/internal/recoverable-runs/model/invoke", json={ @@ -115,7 +119,9 @@ class GatewayProxyChatModel(BaseChatModel): "tool_choice": self.bound_tool_choice, "stream": True, } - async with httpx.AsyncClient(timeout=httpx.Timeout(660.0, connect=5.0)) as client: + async with httpx.AsyncClient( + timeout=httpx.Timeout(660.0, connect=5.0) + ) as client: async with client.stream( "POST", f"{self.gateway_url.rstrip('/')}/api/internal/recoverable-runs/model/stream", @@ -126,11 +132,27 @@ class GatewayProxyChatModel(BaseChatModel): async for line in response.aiter_lines(): if not line.startswith("data:"): continue - data = line[len("data:"):].strip() + data = line[len("data:") :].strip() if data == "[DONE]": saw_done = True break chunk = json.loads(data) + if chunk.get("type") == "error": + code = str(chunk.get("code") or "MODEL_PROVIDER_ERROR") + message = str(chunk.get("message") or code) + details = { + key: value + for key, value in { + "http_status": chunk.get("status"), + "retryable": chunk.get("retryable"), + }.items() + if isinstance(value, int | bool) + } + raise EvoRuntimeError( + code, + message, + details=(details,) if details else (), + ) message = _chunk_to_message(chunk) yield ChatGenerationChunk( message=message, diff --git a/EvoScientist/llm/patches.py b/EvoScientist/llm/patches.py index 2b1db19..8822e96 100644 --- a/EvoScientist/llm/patches.py +++ b/EvoScientist/llm/patches.py @@ -29,7 +29,7 @@ from __future__ import annotations import hashlib import os -from typing import Any +from typing import Any, cast # --------------------------------------------------------------------------- @@ -131,6 +131,47 @@ def _patch_openai_empty_sse_keepalive() -> None: _patch_openai_empty_sse_keepalive() +# --------------------------------------------------------------------------- +# Patch: deepagents routes every non-text extension to a media block even when +# a backend has already extracted bounded UTF-8 text from the document. Keep +# real base64 binary data on the media path (the middleware checks encoding +# first), but route extracted Office/PDF results through a normal ToolMessage. +# --------------------------------------------------------------------------- +_deepagents_extracted_document_text_patched = False + + +def _patch_deepagents_extracted_document_text() -> None: + global _deepagents_extracted_document_text_patched + if _deepagents_extracted_document_text_patched: + return + try: + from pathlib import Path as _Path + + import deepagents.middleware.filesystem as _filesystem + + from EvoScientist.document_extract import DOCUMENT_EXTENSIONS + + namespace = cast(dict[str, Any], vars(_filesystem)) + original = namespace["_get_file_type"] + if getattr(original, "_evoscientist_document_text", False): + _deepagents_extracted_document_text_patched = True + return + + def _document_text_type(path: str) -> str: + if _Path(path).suffix.lower() in DOCUMENT_EXTENSIONS: + return "text" + return original(path) + + _document_text_type._evoscientist_document_text = True # type: ignore[attr-defined] + namespace["_get_file_type"] = _document_text_type + _deepagents_extracted_document_text_patched = True + except Exception: + pass + + +_patch_deepagents_extracted_document_text() + + # --------------------------------------------------------------------------- # Patch: ccproxy-api 0.2.7 Codex compatibility. # diff --git a/EvoScientist/middleware/__init__.py b/EvoScientist/middleware/__init__.py index 22b07b4..6b9c23d 100644 --- a/EvoScientist/middleware/__init__.py +++ b/EvoScientist/middleware/__init__.py @@ -51,6 +51,7 @@ from .skill_context import ( DEFAULT_MAX_SKILLS_BYTES, BudgetedSkillsMiddleware, ) +from .subagent_timeout import SubagentTimeoutMiddleware from .tool_error_handler import ToolErrorHandlerMiddleware from .tool_protocol_guard import ToolProtocolGuardMiddleware from .tool_selector import create_tool_selector_middleware @@ -82,6 +83,7 @@ __all__ = [ "RepetitiveToolCallGuardMiddleware", "RuntimeContextMiddleware", "SchedulerMiddleware", + "SubagentTimeoutMiddleware", "ToolErrorHandlerMiddleware", "ToolProtocolGuardMiddleware", "collapse_repetitive_tool_rounds", diff --git a/EvoScientist/middleware/dynamic_review.py b/EvoScientist/middleware/dynamic_review.py index 075b7b1..90f38ff 100644 --- a/EvoScientist/middleware/dynamic_review.py +++ b/EvoScientist/middleware/dynamic_review.py @@ -187,6 +187,17 @@ class DynamicReviewMiddleware(HumanInTheLoopMiddleware): # A LangGraph resume continues at this interrupted node and does not # re-run before_agent. Reusing a manual state is restrictive and is # required for the existing resume child Run to complete. + # EXCEPTION: the gateway snapshots the thread's LIVE review mode + # into each resume child's envelope. When the user flipped the + # thread (or the current interrupt) to auto AFTER the parent run + # started, the injected ai4sci_review_mode context is "auto" — + # verify it and bypass HITL instead of interrupting again. + if review is not None and review.get("requested_mode") == "auto": + try: + _resolve_sync(current_run_id, review) + except AutoReviewVerificationError: + return super().after_model(state, runtime) + return None return super().after_model(state, runtime) raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID") @@ -213,5 +224,14 @@ class DynamicReviewMiddleware(HumanInTheLoopMiddleware): return super().after_model(state, runtime) return None if mode == "manual": + # Mirror the sync path: an injected auto context on a resume child + # (thread flipped to auto after the parent started) bypasses HITL + # after successful re-verification. + if review is not None and review.get("requested_mode") == "auto": + try: + await _resolve_async(current_run_id, review) + except AutoReviewVerificationError: + return super().after_model(state, runtime) + return None return super().after_model(state, runtime) raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID") diff --git a/EvoScientist/middleware/error_normalization.py b/EvoScientist/middleware/error_normalization.py index 8dbe6ab..04dae90 100644 --- a/EvoScientist/middleware/error_normalization.py +++ b/EvoScientist/middleware/error_normalization.py @@ -141,10 +141,13 @@ def _normalize(request: ModelRequest, exc: BaseException) -> ProviderStreamError - ``AgentControlError`` — a platform-owned typed decision. Gateway route fallback and canonical error mapping depend on its concrete type and structured fields, so it must never become a provider incident. + - ``EvoRuntimeError`` — a stable host/runtime error that has already been + classified across the Gateway boundary and must retain its code. - Models we don't recognize as a provider SDK. """ from langchain_core.exceptions import ContextOverflowError + from ..llm.contracts import EvoRuntimeError from ..llm.errors import ( AgentControlError, ProviderStreamError, @@ -166,6 +169,9 @@ def _normalize(request: ModelRequest, exc: BaseException) -> ProviderStreamError if isinstance(exc, AgentControlError): return None + if isinstance(exc, EvoRuntimeError): + return None + # LangGraph control-flow / structural signals must propagate # untouched, regardless of which caller invoked us. if _should_pass_through(exc): diff --git a/EvoScientist/middleware/recoverable_tools.py b/EvoScientist/middleware/recoverable_tools.py index b2be0da..9ff3a8e 100644 --- a/EvoScientist/middleware/recoverable_tools.py +++ b/EvoScientist/middleware/recoverable_tools.py @@ -12,6 +12,8 @@ from langchain.agents.middleware.types import AgentMiddleware from langchain_core.messages import ToolMessage, message_to_dict, messages_from_dict from langgraph.types import Command +from EvoScientist.llm.contracts import EvoRuntimeError + if TYPE_CHECKING: from langchain.agents.middleware.types import ToolCallRequest @@ -91,6 +93,15 @@ async def _post(proxy: Mapping[str, str], phase: str, payload: dict[str, Any]) - "envelope_signature": proxy["envelope_signature"], }, ) + if response.is_error: + try: + error_body = response.json() + except ValueError: + error_body = None + detail = error_body.get("detail") if isinstance(error_body, Mapping) else None + code = detail.get("code") if isinstance(detail, Mapping) else None + if isinstance(code, str) and code.isascii() and code.replace("_", "").isalnum(): + raise EvoRuntimeError(code) response.raise_for_status() return dict(response.json()) diff --git a/EvoScientist/middleware/subagent_timeout.py b/EvoScientist/middleware/subagent_timeout.py new file mode 100644 index 0000000..f91704f --- /dev/null +++ b/EvoScientist/middleware/subagent_timeout.py @@ -0,0 +1,84 @@ +"""Bound synchronous sub-agent calls in hosted Web runs.""" + +from __future__ import annotations + +import asyncio +import json +from collections.abc import Awaitable, Callable +from typing import Any + +from langchain.agents.middleware.types import AgentMiddleware, ToolCallRequest +from langchain_core.messages import ToolMessage +from langgraph.types import Command + + +class SubagentTimeoutMiddleware(AgentMiddleware): + """Cancel a synchronous ``task`` call that exceeds the Web time budget.""" + + @property + def name(self) -> str: + return "subagent_timeout" + + def __init__(self, timeout_seconds: float = 180.0) -> None: + super().__init__() + if timeout_seconds <= 0: + raise ValueError("timeout_seconds must be positive") + self.timeout_seconds = float(timeout_seconds) + + @staticmethod + async def _cancel_task(task: asyncio.Future[Any]) -> None: + task.cancel() + try: + await asyncio.shield(task) + except asyncio.CancelledError: + if not task.done(): + task.add_done_callback(SubagentTimeoutMiddleware._consume_task_result) + raise + except Exception: + pass + + @staticmethod + def _consume_task_result(task: asyncio.Future[Any]) -> None: + try: + task.result() + except (asyncio.CancelledError, Exception): + pass + + async def awrap_tool_call( + self, + request: ToolCallRequest, + handler: Callable[[ToolCallRequest], Awaitable[ToolMessage | Command[Any]]], + ) -> ToolMessage | Command[Any]: + if str(request.tool_call.get("name") or "") != "task": + return await handler(request) + task = asyncio.ensure_future(handler(request)) + try: + done, _pending = await asyncio.wait( + {task}, timeout=self.timeout_seconds + ) + except asyncio.CancelledError: + await self._cancel_task(task) + raise + if done: + return task.result() + + await self._cancel_task(task) + current = asyncio.current_task() + if current is not None and current.cancelling(): + raise asyncio.CancelledError + payload = { + "code": "SUBAGENT_TIMEOUT", + "message": ( + "The delegated sub-agent exceeded the hosted Web time limit. " + "Continue with available evidence or use the controlled web search tool directly." + ), + "retryable": True, + "timeout_seconds": self.timeout_seconds, + } + return ToolMessage( + content=json.dumps(payload, ensure_ascii=False), + tool_call_id=str(request.tool_call.get("id") or "subagent_timeout"), + name="task", + status="error", + additional_kwargs={"error_code": "SUBAGENT_TIMEOUT"}, + ) diff --git a/EvoScientist/native_sandbox.py b/EvoScientist/native_sandbox.py index 6b677c5..5586181 100644 --- a/EvoScientist/native_sandbox.py +++ b/EvoScientist/native_sandbox.py @@ -132,6 +132,20 @@ def _node_version(node: Path) -> tuple[int, int, int]: return parts +def _existing_resolved_paths(candidates: tuple[str, ...]) -> tuple[str, ...]: + resolved: list[str] = [] + seen: set[str] = set() + for candidate in candidates: + try: + value = str(Path(candidate).resolve(strict=True)) + except OSError: + continue + if value not in seen: + seen.add(value) + resolved.append(value) + return tuple(resolved) + + def _system_read_paths() -> tuple[str, ...]: candidates = ( ( @@ -162,7 +176,7 @@ def _system_read_paths() -> tuple[str, ...]: "/dev/urandom", ) ) - return tuple(path for path in candidates if Path(path).exists()) + return _existing_resolved_paths(candidates) def _assert_install_contract() -> NativeSandboxInstallation: @@ -262,6 +276,7 @@ def _sandbox_settings( # srt adds these shared compatibility paths even when callers do # not request them. Explicit deny wins over that built-in allow. "denyWrite": [ + str(files_dir / "uploads"), "/tmp/claude", "/private/tmp/claude", "/dev/tty", @@ -269,7 +284,10 @@ def _sandbox_settings( "/dev/autofs_nowait", ], }, - "enableWeakerNestedSandbox": False, + "enableWeakerNestedSandbox": os.getenv( + "EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", "" + ).strip().lower() + in {"1", "true", "yes", "on"}, "enableWeakerNetworkIsolation": False, "allowAppleEvents": False, "allowPty": False, @@ -622,6 +640,19 @@ _READY_STATE = "unchecked" _READY_ERROR: str | None = None +def _network_preflight_probe(port: int, unix_path: Path) -> str: + return ( + "import os,socket;\n" + "assert 'OPENAI_API_KEY' not in os.environ\n" + f"t=socket.socket(); tcp=t.connect_ex(('127.0.0.1',{port})); t.close()\n" + "try:\n" + f" u=socket.socket(socket.AF_UNIX); unix=u.connect_ex({str(unix_path)!r}); u.close()\n" + "except OSError:\n" + " unix=1\n" + "assert tcp != 0 and unix != 0\n" + ) + + def _run_preflight(installation: NativeSandboxInstallation) -> None: # AF_UNIX paths are limited to roughly 100 bytes on both target platforms; # a deployment workspace path can already exceed that before the filename. @@ -632,6 +663,10 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None: files_dir.mkdir(mode=0o700) runtime_dir.mkdir(mode=0o700) (files_dir / "probe.txt").write_text("allowed", encoding="utf-8") + uploads_dir = files_dir / "uploads" + uploads_dir.mkdir(mode=0o700) + upload_source = uploads_dir / "source.txt" + upload_source.write_text("immutable", encoding="utf-8") outside = root / "outside-secret.txt" outside.write_text("secret", encoding="utf-8") control_secret = runtime_dir / "control-secret.txt" @@ -649,16 +684,13 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None: unix_server.bind(str(unix_path)) unix_server.listen(1) - python_probe = ( - "import os,socket,sys;" - "assert 'OPENAI_API_KEY' not in os.environ;" - f"t=socket.socket(); tcp=t.connect_ex(('127.0.0.1',{port})); t.close();" - f"u=socket.socket(socket.AF_UNIX); unix=u.connect_ex({str(unix_path)!r}); u.close();" - "sys.exit(0 if tcp != 0 and unix != 0 else 9)" - ) + python_probe = _network_preflight_probe(port, unix_path) command = " && ".join( ( 'test "$(cat probe.txt)" = allowed', + 'test "$(cat uploads/source.txt)" = immutable', + "! sh -c 'printf changed > uploads/source.txt' 2>/dev/null", + "! rm uploads/source.txt 2>/dev/null", "printf written > written.txt", 'printf temporary > "$TMPDIR/probe.tmp"', "pandoc --version >/dev/null", @@ -695,6 +727,8 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None: ) if outside.read_text(encoding="utf-8") != "secret": raise NativeSandboxUnavailable("native sandbox preflight escaped workspace") + if upload_source.read_text(encoding="utf-8") != "immutable": + raise NativeSandboxUnavailable("native sandbox preflight modified uploads") def ensure_native_sandbox_ready() -> None: diff --git a/EvoScientist/prompts.py b/EvoScientist/prompts.py index 8efd417..2a2cf21 100644 --- a/EvoScientist/prompts.py +++ b/EvoScientist/prompts.py @@ -267,6 +267,23 @@ echo "PID: $!" # check: ps -p · stop: kill · read This prevents blocking the conversation during long operations.""" +_FILE_PROGRAMMING_GUIDELINES = """# Safe Binary File Programming + +- When `read_file` returns `BINARY_PROCESSING_REQUIRED`, use `execute` to inspect the referenced workspace file with a short, bounded program. Do not report the format as unsupported without trying the indicated safe workflow. +- For ZIP/TAR archives, list and validate members before selective extraction; never use `extractall` on an untrusted archive. Reject absolute paths, `..`, links escaping the destination, excessive member counts, expanded sizes, and compression ratios. Never execute files extracted from an archive. +- For SQLite, open `file:?mode=ro&immutable=1` with `uri=True`, set `PRAGMA query_only=ON`, inspect `sqlite_master`, and use bounded `SELECT ... LIMIT ...` queries. Never use `ATTACH DATABASE`, enable extensions, or modify the source database. +- Do not modify original files under `uploads/`. Write extracted members, query results, converted documents, and other derived artifacts under `artifacts/` or `.ai4sci/`. +- Text extracted by `read_file` from PDF or Office files is a semantic view, not the original container bytes. Never write that text back to the source document with `write_file` or `edit_file`. +""" + +_WEB_RESEARCH_GUIDELINES = """# Controlled Web Research + +- `search_observations` searches local memory only. It does not access the internet and must never be presented as live web search. +- For current facts, public webpages, source discovery, or URL verification, call `tavily_search` first when it is available. Use `web_search` only when that configured search provider is available. +- Do not use `execute`, `curl`, or `httpx` to reach the public internet. The execution sandbox intentionally blocks raw networking; controlled search tools are the only supported network path. +- If controlled web search is unavailable or fails, report that specific limitation once and continue with clearly labeled non-live evidence. Do not repeatedly probe DNS, proxies, direct IPs, or local ports. +""" + # Sandbox (default) header: virtual `/` workspace. _SHELL_GUIDELINES_SANDBOX_HEADER = """# Shell Execution Guidelines @@ -460,6 +477,8 @@ def get_system_prompt( REPORT_TEMPLATE, WRITING_GUIDELINES, shell_guidelines, + _FILE_PROGRAMMING_GUIDELINES, + _WEB_RESEARCH_GUIDELINES, DELEGATION_STRATEGY, ASYNC_NOTIFICATIONS, ] diff --git a/EvoScientist/sessions.py b/EvoScientist/sessions.py index 122d56d..6c0f66c 100644 --- a/EvoScientist/sessions.py +++ b/EvoScientist/sessions.py @@ -1630,6 +1630,210 @@ async def db_stats(top_n: int = 5) -> dict[str, Any]: return out +# --------------------------------------------------------------------------- +# History pruning / VACUUM (gateway timer + admin endpoints) +# --------------------------------------------------------------------------- + + +def _uuid6_unix_ts(checkpoint_id: str) -> float | None: + """Extract the unix timestamp embedded in a UUIDv6 checkpoint id. + + Returns ``None`` for non-UUID or non-v6 ids — legacy ids carry no usable + timestamp, and callers must skip those threads rather than guess an age. + """ + try: + u = uuid.UUID(checkpoint_id) + except (ValueError, AttributeError, TypeError): + return None + if u.version != 6: + return None + ts60 = ((u.int >> 80) << 12) | ((u.int >> 64) & 0x0FFF) + return ts60 / 10_000_000 - 12219292800 + + +async def _count_thread_rows( + conn: aiosqlite.Connection, thread_id: str +) -> tuple[int, int]: + """Return ``(checkpoints, writes)`` row counts for one thread.""" + async with conn.execute( + "SELECT COUNT(*) FROM checkpoints WHERE thread_id = ?", (thread_id,) + ) as cur: + row = await cur.fetchone() + ck = int(row[0]) if row else 0 + wr = 0 + if await _table_exists(conn, "writes"): + async with conn.execute( + "SELECT COUNT(*) FROM writes WHERE thread_id = ?", (thread_id,) + ) as cur: + row = await cur.fetchone() + wr = int(row[0]) if row else 0 + return ck, wr + + +async def _prune_thread_on_conn( + conn: aiosqlite.Connection, thread_id: str, keep_last: int +) -> tuple[int, int]: + """Prune one thread on an open connection; return rows deleted per table. + + Reuses :meth:`PruningCheckpointer._prune_after_put` per + ``(thread_id, checkpoint_ns)`` group so retention is identical to the + live write path, including DeltaChannel snapshot-chain preservation. + Only ``metadata.agent_name == AGENT_NAME`` rows are ever touched. + """ + saver = PruningCheckpointer(conn, keep_per_ns=max(1, int(keep_last))) + before_ck, before_wr = await _count_thread_rows(conn, thread_id) + async with conn.execute( + "SELECT DISTINCT checkpoint_ns FROM checkpoints " + "WHERE thread_id = ? AND json_extract(metadata, '$.agent_name') = ?", + (thread_id, AGENT_NAME), + ) as cur: + namespaces = [r[0] for r in await cur.fetchall()] + for ns in namespaces: + await saver._prune_after_put(thread_id, ns or "") + after_ck, after_wr = await _count_thread_rows(conn, thread_id) + return before_ck - after_ck, before_wr - after_wr + + +async def prune_thread_history( + thread_id: str, keep_last: int = 2, db_path: str | None = None +) -> dict[str, int]: + """Prune one thread's history, keeping its ``keep_last`` most recent rows. + + Returns ``{"deleted_checkpoints": int, "deleted_writes": int}``. + """ + path = str(db_path or get_db_path()) + result = {"deleted_checkpoints": 0, "deleted_writes": 0} + if not Path(path).exists(): + return result + async with aiosqlite.connect(path, timeout=30.0) as conn: + if not await _table_exists(conn, "checkpoints"): + return result + ck, wr = await _prune_thread_on_conn(conn, str(thread_id), keep_last) + result["deleted_checkpoints"] = ck + result["deleted_writes"] = wr + return result + + +async def prune_all_stale_threads( + max_age_hours: float = 72, + keep_last: int = 2, + db_path: str | None = None, +) -> dict[str, int]: + """Prune every EvoScientist thread idle for at least ``max_age_hours``. + + A thread is stale when its newest checkpoint (UUIDv6 timestamp) is older + than the cutoff. Threads whose newest id has no parseable timestamp are + skipped. Returns ``{"databases_processed", "threads_pruned", + "total_deleted_checkpoints", "total_deleted_writes"}``. + """ + path = str(db_path or get_db_path()) + result = { + "databases_processed": 0, + "threads_pruned": 0, + "total_deleted_checkpoints": 0, + "total_deleted_writes": 0, + } + if not Path(path).exists(): + return result + cutoff = time.time() - float(max_age_hours) * 3600.0 + async with aiosqlite.connect(path, timeout=30.0) as conn: + if not await _table_exists(conn, "checkpoints"): + return result + result["databases_processed"] = 1 + async with conn.execute( + "SELECT thread_id, MAX(checkpoint_id) FROM checkpoints " + "WHERE json_extract(metadata, '$.agent_name') = ? " + "GROUP BY thread_id", + (AGENT_NAME,), + ) as cur: + rows = await cur.fetchall() + stale = [ + tid + for tid, newest in rows + if (ts := _uuid6_unix_ts(newest)) is not None and ts < cutoff + ] + for tid in stale: + ck, wr = await _prune_thread_on_conn(conn, tid, keep_last) + if ck or wr: + result["threads_pruned"] += 1 + result["total_deleted_checkpoints"] += ck + result["total_deleted_writes"] += wr + return result + + +async def list_all_thread_ids(db_path: str | None = None) -> list[str]: + """Return all EvoScientist thread ids in the sessions DB.""" + path = str(db_path or get_db_path()) + if not Path(path).exists(): + return [] + async with aiosqlite.connect(path, timeout=30.0) as conn: + if not await _table_exists(conn, "checkpoints"): + return [] + async with conn.execute( + "SELECT DISTINCT thread_id FROM checkpoints " + "WHERE json_extract(metadata, '$.agent_name') = ?", + (AGENT_NAME,), + ) as cur: + return [r[0] for r in await cur.fetchall()] + + +def list_all_session_db_paths() -> list[Path]: + """Return existing session DB paths. + + The current storage layout uses a single shared DB (``get_db_path()``), + so this returns a one-element list when it exists, else ``[]``. + """ + path = get_db_path() + return [path] if path.exists() else [] + + +async def vacuum_db(db_path: str | None = None) -> dict[str, Any]: + """Run ``VACUUM`` on the sessions DB to reclaim freed pages.""" + path = str(db_path or get_db_path()) + p = Path(path) + before = p.stat().st_size if p.exists() else 0 + if p.exists(): + async with aiosqlite.connect(path, timeout=120.0) as conn: + await conn.execute("VACUUM") + await conn.commit() + after = p.stat().st_size if p.exists() else 0 + return { + "db_path": path, + "size_before_bytes": before, + "size_after_bytes": after, + } + + +async def get_aggregated_storage_stats() -> dict[str, Any]: + """Return :func:`db_stats` plus per-thread checkpoint depth stats.""" + stats = await db_stats() + depth: dict[str, Any] = {"min": 0, "max": 0, "avg": 0.0} + path = Path(stats["db_path"]) + if path.exists(): + try: + async with aiosqlite.connect(str(path), timeout=30.0) as conn: + if await _table_exists(conn, "checkpoints"): + async with conn.execute( + "SELECT MIN(n), MAX(n), AVG(n) FROM (" + " SELECT COUNT(*) AS n FROM checkpoints " + " WHERE json_extract(metadata, '$.agent_name') = ? " + " GROUP BY thread_id" + ")", + (AGENT_NAME,), + ) as cur: + row = await cur.fetchone() + if row and row[0] is not None: + depth = { + "min": int(row[0]), + "max": int(row[1]), + "avg": round(float(row[2]), 2), + } + except aiosqlite.Error: + # Read-only diagnostic — mirror db_stats and degrade to zeros. + pass + return {**stats, "thread_depth": depth} + + # --------------------------------------------------------------------------- # langgraph-api / WebUI checkpointer factory # --------------------------------------------------------------------------- diff --git a/EvoScientist/tools/search.py b/EvoScientist/tools/search.py index 3704df1..a980670 100644 --- a/EvoScientist/tools/search.py +++ b/EvoScientist/tools/search.py @@ -14,6 +14,16 @@ from tavily import TavilyClient # Lazy initialization - only create client when needed _tavily_client = None +MAX_SEARCH_RESULTS = 5 +MAX_DISPLAY_QUERY_CHARS = 512 +MAX_DISPLAY_TITLE_CHARS = 512 +MAX_DISPLAY_URL_CHARS = 2_048 +MAX_PAGE_CONTENT_CHARS = 4_000 +MAX_SEARCH_RESULT_CHARS = 16_000 +_TRUNCATION_MARKER = "\n\n[page content truncated]" +_SEARCH_TRUNCATION_MARKER = ( + "\n[search result content truncated to preserve all titles and URLs]" +) def _get_tavily_client() -> TavilyClient: @@ -46,6 +56,12 @@ async def fetch_webpage_content(url: str, timeout: float = 10.0) -> str: async with httpx.AsyncClient() as client: response = await client.get(url, headers=headers, timeout=timeout) response.raise_for_status() + content_type = response.headers.get("content-type", "").lower() + if not any( + allowed in content_type + for allowed in ("text/", "application/xhtml+xml") + ): + return f"Error fetching content from {url}: unsupported content type {content_type or 'unknown'}" return markdownify(response.text) except Exception as e: return f"Error fetching content from {url}: {e!s}" @@ -72,40 +88,71 @@ async def tavily_search( """ def _sync_search() -> dict: + bounded_max_results = max(1, min(int(max_results), MAX_SEARCH_RESULTS)) return _get_tavily_client().search( query, - max_results=max_results, + max_results=bounded_max_results, topic=topic, ) try: - # Run Tavily search asynchronously + # Run Tavily search asynchronously through the controlled host path. search_results = await asyncio.to_thread(_sync_search) + from EvoScientist.runtime_integrations import record_service_usage + + await record_service_usage("tavily", "search") - # Fetch full content for each URL concurrently results = search_results.get("results", []) if not results: return f"No results found for '{query}'" - # Fetch all webpages concurrently fetch_tasks = [fetch_webpage_content(r["url"]) for r in results] contents = await asyncio.gather(*fetch_tasks) - # Format results + normalized = [] + for result, fetched_content in zip(results, contents, strict=False): + title = str(result.get("title") or "Untitled")[:MAX_DISPLAY_TITLE_CHARS] + raw_url = str(result.get("url") or "") + url = ( + raw_url + if len(raw_url) <= MAX_DISPLAY_URL_CHARS + else raw_url[: MAX_DISPLAY_URL_CHARS - len("...[URL truncated]")] + + "...[URL truncated]" + ) + tavily_summary = str(result.get("content") or "").strip() + fetch_failed = fetched_content.startswith("Error fetching content from ") + content = tavily_summary if fetch_failed and tavily_summary else fetched_content + fetch_note = ( + "\n\n> Source page fetch failed; showing the Tavily-indexed summary." + if fetch_failed and tavily_summary + else "" + ) + normalized.append((title, url, content, fetch_note)) + + display_query = query[:MAX_DISPLAY_QUERY_CHARS] + prefix = f"Found {len(normalized)} live web result(s) for '{display_query}':\n\n" + metadata_blocks = [f"## {title}\n**URL:** {url}\n\n" for title, url, _, _ in normalized] + fixed_chars = len(prefix) + sum(len(block) + len("\n\n---\n") for block in metadata_blocks) + remaining = max(0, MAX_SEARCH_RESULT_CHARS - fixed_chars - len(_SEARCH_TRUNCATION_MARKER)) + per_result_budget = remaining // max(1, len(normalized)) + result_texts = [] - for result, content in zip(results, contents, strict=False): - result_text = f"""## {result["title"]} -**URL:** {result["url"]} + content_truncated = False + for metadata, (_, _, content, fetch_note) in zip(metadata_blocks, normalized, strict=True): + content_budget = max(0, min(MAX_PAGE_CONTENT_CHARS, per_result_budget - len(fetch_note))) + if len(content) > content_budget: + marker_budget = min(len(_TRUNCATION_MARKER), content_budget) + content = ( + content[: content_budget - marker_budget] + + _TRUNCATION_MARKER[:marker_budget] + ) + content_truncated = True + result_texts.append(f"{metadata}{content}{fetch_note}\n\n---\n") -{content} - ---- -""" - result_texts.append(result_text) - - return f"""Found {len(result_texts)} result(s) for '{query}': - -{"".join(result_texts)}""" + formatted = prefix + "".join(result_texts) + if content_truncated: + formatted += _SEARCH_TRUNCATION_MARKER + return formatted except Exception as e: return f"Search failed: {e!s}" diff --git a/EvoScientist/web_runtime.py b/EvoScientist/web_runtime.py index 5cf01dd..09dbe38 100644 --- a/EvoScientist/web_runtime.py +++ b/EvoScientist/web_runtime.py @@ -51,10 +51,23 @@ def web_tool_registry_manifest() -> tuple[tuple[dict[str, Any], ...], str]: "additionalProperties": True, "maxProperties": 32, } + tool_descriptions = { + "tavily_search": ( + "Search the live public web through the controlled Tavily service. " + "Use this for current facts, source discovery, and URL verification. " + "Do not use execute, curl, httpx, or raw sandbox networking instead." + ), + "web_search": ( + "Search the live public web through the configured MCP search provider. " + "Use this for current facts and source verification, not local memory recall." + ), + } manifest: tuple[dict[str, Any], ...] = tuple( { "name": name, - "description": "EvoScientist Web runtime tool", + "description": tool_descriptions.get( + name, "EvoScientist Web runtime tool" + ), "schema": schema, } for name in names diff --git a/EvoScientist/workspace_files.py b/EvoScientist/workspace_files.py index 9a9b0dc..abdfb4a 100644 --- a/EvoScientist/workspace_files.py +++ b/EvoScientist/workspace_files.py @@ -375,7 +375,16 @@ def _standard_error(exc: Exception) -> str: def _is_binary_file(path: str, raw: bytes) -> bool: - """Classify common containers explicitly and fall back to content.""" + """Classify common containers explicitly and fall back to content. + + The UTF-8 probe samples the first bytes, so a multi-byte character can be + cut in half at the sample boundary (e.g. an 8192-byte cut splitting a + 3-byte CJK character). ``decode`` with ``errors="ignore"`` would hide real + garbage, so instead we re-probe on failure with the trailing partial + sequence removed: a decode error that vanishes once the (at most 4-byte) + dangling suffix is dropped is a truncation artifact, not binary content. + Files that still fail on the trimmed sample are genuinely not UTF-8. + """ if Path(path).suffix.lower() in _BINARY_EXTENSIONS: return True @@ -384,7 +393,15 @@ def _is_binary_file(path: str, raw: bytes) -> bool: return True try: sample.decode("utf-8") - except UnicodeDecodeError: + except UnicodeDecodeError as exc: + # Only a decode error at the very end of the sample can be a boundary + # cut. Errors positioned mid-sample are real invalid bytes. + if exc.start >= max(0, len(sample) - 4): + try: + sample[: exc.start].decode("utf-8") + except UnicodeDecodeError: + return True + return False return True return False @@ -397,6 +414,13 @@ class ScopedFilesystemBackend(BackendProtocol): root_dir, max_search_file_bytes=max_search_file_bytes ) + @staticmethod + def _uploads_are_read_only(file_path: str) -> bool: + normalized = "/" + file_path.replace("\\", "/").lstrip("/") + return normalized == "/workspace/uploads" or normalized.startswith( + "/workspace/uploads/" + ) + @staticmethod def _search_path(path: str | None) -> str: if path in {None, "/"}: @@ -426,15 +450,72 @@ class ScopedFilesystemBackend(BackendProtocol): return LsResult(error=f"Cannot list '{path}': {_standard_error(exc)}") def read(self, file_path: str, offset: int = 0, limit: int = 2000) -> ReadResult: + from .document_extract import ( + MAX_DOCUMENT_BYTES, + DocumentExtractionError, + binary_processing_guidance, + classify_file, + extract_document_bytes, + paginate_document_text, + prepare_image_bytes, + ) + try: + entry = self.workspace.entry(file_path) with self.workspace.open_binary(file_path) as handle: + head = handle.read(4096) + kind = classify_file(file_path, head) + + if kind == "document": + if entry.size > MAX_DOCUMENT_BYTES: + return ReadResult( + error=( + f"DOCUMENT_TOO_LARGE: '{file_path}' is {entry.size} bytes; " + f"document extraction limit is {MAX_DOCUMENT_BYTES} bytes" + ) + ) + handle.seek(0) + raw = handle.read(MAX_DOCUMENT_BYTES + 1) + try: + extracted = extract_document_bytes(raw, file_path) + content = paginate_document_text( + extracted, offset=max(0, offset), limit=max(1, limit) + ) + except DocumentExtractionError as exc: + return ReadResult(error=str(exc)) + return ReadResult( + file_data={"content": content, "encoding": "utf-8"} + ) + + if kind == "image": + handle.seek(0) + raw = handle.read() + try: + raw = prepare_image_bytes(raw, file_path) + except DocumentExtractionError as exc: + return ReadResult(error=str(exc)) + return ReadResult( + file_data={ + "content": base64.standard_b64encode(raw).decode("ascii"), + "encoding": "base64", + } + ) + + if kind in {"archive", "database", "executable", "dataset", "media"}: + return ReadResult( + error=binary_processing_guidance(file_path, kind, entry.size) + ) + + if _is_binary_file(file_path, head): + return ReadResult( + error=binary_processing_guidance(file_path, "binary", entry.size) + ) + + handle.seek(0) raw = handle.read() if _is_binary_file(file_path, raw): return ReadResult( - file_data={ - "content": base64.standard_b64encode(raw).decode("ascii"), - "encoding": "base64", - } + error=binary_processing_guidance(file_path, "binary", entry.size) ) content = raw.decode("utf-8") empty = check_empty_content(content) @@ -455,6 +536,29 @@ class ScopedFilesystemBackend(BackendProtocol): return ReadResult(error=f"Error reading file '{file_path}': {_standard_error(exc)}") def write(self, file_path: str, content: str) -> WriteResult: + if self._uploads_are_read_only(file_path): + return WriteResult( + error=( + f"Cannot modify original upload '{file_path}'. " + "Write derived content under /workspace/artifacts/ or /workspace/.ai4sci/." + ) + ) + from .document_extract import ( + ARCHIVE_EXTENSIONS, + DATABASE_EXTENSIONS, + DOCUMENT_EXTENSIONS, + ) + + if Path(file_path).suffix.lower() in ( + DOCUMENT_EXTENSIONS | ARCHIVE_EXTENSIONS | DATABASE_EXTENSIONS + ): + return WriteResult( + error=( + f"Cannot write plain text to binary container '{file_path}'. " + "Use execute with an appropriate document, archive, or database " + "library and write a derived file under artifacts/." + ) + ) try: self.workspace.write_new(file_path, content.encode("utf-8")) return WriteResult(path=file_path) @@ -472,6 +576,29 @@ class ScopedFilesystemBackend(BackendProtocol): new_string: str, replace_all: bool = False, ) -> EditResult: + if self._uploads_are_read_only(file_path): + return EditResult( + error=( + f"Cannot modify original upload '{file_path}'. " + "Write derived content under /workspace/artifacts/ or /workspace/.ai4sci/." + ) + ) + from .document_extract import ( + ARCHIVE_EXTENSIONS, + DATABASE_EXTENSIONS, + DOCUMENT_EXTENSIONS, + ) + + if Path(file_path).suffix.lower() in ( + DOCUMENT_EXTENSIONS | ARCHIVE_EXTENSIONS | DATABASE_EXTENSIONS + ): + return EditResult( + error=( + f"Cannot edit binary container '{file_path}' with text replacement. " + "Use execute with an appropriate library and write a derived file " + "under artifacts/." + ) + ) try: with self.workspace.open_binary(file_path) as handle: content = handle.read().decode("utf-8") diff --git a/MANIFEST.in b/MANIFEST.in new file mode 100644 index 0000000..566fa6b --- /dev/null +++ b/MANIFEST.in @@ -0,0 +1,2 @@ +global-exclude *.py[cod] +prune **/__pycache__ diff --git a/docs/architecture/Ai4Sci有界文件读取与Agent自主编程实施方案.md b/docs/architecture/Ai4Sci有界文件读取与Agent自主编程实施方案.md new file mode 100644 index 0000000..3fe65a3 --- /dev/null +++ b/docs/architecture/Ai4Sci有界文件读取与Agent自主编程实施方案.md @@ -0,0 +1,110 @@ +# Ai4Sci 有界文件读取与 Agent 自主编程实施方案 + +> 状态:实施基线 v1.0 +> 范围:EvoScientist Core + Ai4Sci-Web 必要接线 +> 原则:程序提供事实和安全边界,主 Agent 使用现有 read_file/execute 完成渐进处理。 + +## 1. 目标 + +修复 Office/PDF 等二进制完整 Base64 进入模型上下文的问题,同时保留 Agent 对 ZIP、数据库和其他研究文件的自主编程能力。 + +本阶段不新增 inspect_file/process_file,不新增文件路由模型,不新增数据库表,不改变 SSE 事件协议。 + +## 2. 固定处理契约 + +| 类型 | read_file 行为 | 后续处理 | +|---|---|---| +| UTF-8文本/源码 | offset/limit 分页文本 | 主模型直接分析 | +| 图片 | 有界媒体块 | 多模态模型分析 | +| PDF/Office/ODF/RTF/EPUB | 提取为 Markdown/文本并分页 | 需要视觉信息时 Agent 用 execute 渲染指定页 | +| ZIP/TAR/7Z/RAR | 返回结构化事实和安全编程要求,不返回内容 | Agent 用 execute 先列清单,再选择性解压 | +| SQLite/DB | 返回结构化事实和只读查询要求 | Agent 用 sqlite3 mode=ro/query_only 编程查询 | +| 数据集/音视频 | 返回结构化事实和建议命令 | Agent 用现有库/CLI采样、转录或抽帧 | +| EXE/库/未知二进制 | 元数据引用;明确禁止执行 | 仅允许静态检查 | + +## 3. 安全与预算 + +- 文档源文件最大 50 MiB,转换前检查。 +- 单次提取文本最大 50,000 字符,按行截断并返回 next_offset 提示。 +- ZIP禁止 extractall;先检查成员数、总展开量、单成员大小、压缩比、路径穿越、绝对路径和符号链接。 +- ZIP建议上限:2000成员、500 MiB总展开、100 MiB单成员、100:1压缩比、3层嵌套。 +- SQLite必须 `file:?mode=ro&immutable=1`、`PRAGMA query_only=ON`,查询必须有LIMIT,结果写入artifacts/。 +- EXE/DLL/ELF/Mach-O、宏和归档成员不得自动执行。 +- 原始uploads由文件工具和Native Sandbox双层强制只读;派生文件写入artifacts/或.ai4sci/。 +- read_file不得对任何非图片二进制返回完整Base64。 + +## 4. Core改动 + +1. 新增 `EvoScientist/document_extract.py`: + - 文件类型集合和分派; + - 50 MiB输入限制; + - OOXML/可选anydoc文档提取; + - 结构化失败信息。 +2. 修改 `EvoScientist/workspace_files.py`: + - 覆盖 Ai4Sci Web full/scoped 的 `NativeWorkspaceBackend` 读取路径; + - 文件先分类; + - Office/PDF先提取再分页; + - 图片先解码校验,限制源文件、像素数和多帧输入,最大边降采样至 2048px 后再走现有Base64媒体契约; + - OOXML限制成员数、单成员/总展开量、压缩比,拒绝异常路径、重复和加密成员; + - PDF与旧Office转换在隔离子进程中执行,限制60秒和10MiB转换输出; + - ZIP/DB/其他二进制返回结构化错误指引; + - 避免先完整读取大文档再判断类型。 +3. 修改 `EvoScientist/prompts.py`: + - 加入ZIP和数据库安全编程规则; + - 明确文档提取文本不可用普通写工具写回容器。 +4. 显式声明 `firecrawl-anydoc` 依赖;若不可用或格式不支持,返回可操作错误,不回退Base64。 + +## 5. Gateway最小接线 + +- 上传阶段负责扩展名、MIME、配额、路径和Magic校验;拒绝未知 `application/octet-stream` 绕过。 +- 显式允许 `.db/.sqlite/.sqlite3`,并将SQLite MIME别名视为等价;不在Gateway解析数据库。 +- 附件输入至少保留virtual_path;Office/PDF提取、ZIP解包和数据库查询仍在Agent/execute侧。 +- 工具事件归一化保留受限格式的 `error_code`,非终止型文件工具失败显示在现有ToolOutputItem工具卡中。 +- 永久workspace准备失败通过现有ErrorItem和done事件投影为 `failed/incomplete/runtime_error`,并复用 `primary_error_code`;不新增平行终态协议。 + +## 6. 错误码 + +| code | recoverable | 含义 | +|---|---:|---| +| DOCUMENT_EXTRACTION_FAILED | true/视原因 | 文档损坏、加密或转换失败 | +| DOCUMENT_TOO_LARGE | false | 超过50 MiB输入限制 | +| BINARY_PROCESSING_REQUIRED | true | ZIP/DB/媒体等需Agent编程处理 | +| UNSUPPORTED_BINARY_FILE | false | 未知或不允许处理的二进制 | +| MODEL_RATE_LIMITED | true | 文档处理后模型调用受限 | + +## 7. 测试与验收 + +### Core + +- DOCX/PPTX/PDF不返回Base64。 +- ZIP、SQLite、EXE、未知二进制不返回Base64或NUL乱码。 +- 图片仍返回Base64图片媒体契约。 +- 文档提取支持分页和50KB字符上限。 +- 超过50MiB在转换前拒绝。 +- 转换失败不回退Base64。 +- UTF-8边界切分回归保持通过。 +- prompt包含ZIP安全清单、禁止extractall、SQLite只读模板。 + +### 端到端 + +- 重放9.6MB PPTX时,任一ToolMessage/Checkpoint中不存在原文件Base64。 +- Agent可读取提取文本并继续任务。 +- ZIP先列目录后选择性解压;原ZIP哈希不变。 +- SQLite只读查询;原DB哈希不变。 +- 文件处理失败时Run不是正常completed,而是failed/incomplete并带稳定错误码。 + +## 8. 实施阶段 + +1. P0:文件分类、文档提取、阻断Base64、Core单测。 +2. P0:ZIP/SQLite安全编程提示和测试。 +3. P1:Gateway失败投影缺口(若独立评审确认存在)。 +4. P1:真实PPTX/ZIP/SQLite集成重放。 +5. 代码审查、静态扫描、回归修复和最终复验。 + +## 9. 非目标 + +- 不建设独立文件分析微服务。 +- 不实现专用ZIP/数据库Agent工具。 +- 不让隐藏模型生成文件处理计划。 +- 不支持执行上传的可执行文件或宏。 +- 不在本阶段实现全量视频理解或恶意软件动态沙箱。 diff --git a/pyproject.toml b/pyproject.toml index b83f298..d35194e 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "EvoScientist" -version = "0.2.2" +dynamic = ["version"] description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery" readme = "README.md" requires-python = ">=3.11" @@ -17,6 +17,8 @@ classifiers = [ ] dependencies = [ "deepagents[quickjs]~=0.6.12", + "firecrawl-anydoc>=0.1.6,<0.2", + "pillow>=10.0", "langchain>=1.3", "langchain-anthropic>=1.4", "langchain-openai>=1.2", @@ -111,6 +113,9 @@ build-backend = "setuptools.build_meta" [tool.setuptools.packages.find] include = ["EvoScientist*"] +[tool.setuptools.dynamic] +version = { attr = "EvoScientist._version.__version__" } + [tool.setuptools.package-data] EvoScientist = [ "subagents/*.yaml", @@ -118,6 +123,9 @@ EvoScientist = [ "skills/**/*", ] +[tool.setuptools.exclude-package-data] +"*" = ["**/__pycache__/*", "**/*.pyc", "**/*.pyo"] + [tool.pytest.ini_options] testpaths = ["tests"] asyncio_mode = "auto" diff --git a/runtime/native-sandbox/patch_merged_usr.py b/runtime/native-sandbox/patch_merged_usr.py new file mode 100644 index 0000000..cd403da --- /dev/null +++ b/runtime/native-sandbox/patch_merged_usr.py @@ -0,0 +1,79 @@ +from __future__ import annotations + +import argparse +from pathlib import Path + +_ORIGINAL = """ const rootSkip = new Set(['proc', 'dev', 'sys']); + for (const p of readConfig?.denyOnly || []) { + if (normalizePathForSandbox(p) === '/') { + for (const child of fs.readdirSync('/')) { + if (!rootSkip.has(child)) + readDenyPaths.push('/' + child); + } + } +""" + +_REPLACEMENT = """ const rootSkip = new Set(['proc', 'dev', 'sys']); + const rootChildIsAllowedSymlink = (childPath) => { + try { + if (!fs.lstatSync(childPath).isSymbolicLink()) + return false; + const resolved = fs.realpathSync(childPath); + return readAllowPaths.some(allowPath => resolved === allowPath || resolved.startsWith(allowPath + '/')); + } + catch { + return false; + } + }; + for (const p of readConfig?.denyOnly || []) { + if (normalizePathForSandbox(p) === '/') { + for (const child of fs.readdirSync('/')) { + const childPath = '/' + child; + if (!rootSkip.has(child) && !rootChildIsAllowedSymlink(childPath)) + readDenyPaths.push(childPath); + } + } +""" + +_TMPFS_ORIGINAL = """ args.push('--ro-bind', allowPath, allowPath); + logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`); + } + } +} +""" + +_TMPFS_REPLACEMENT = """ args.push('--ro-bind', allowPath, allowPath); + logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`); + } + } + // A denyRead tmpfs must not become an unlisted writable location. Remount + // only the parent mount read-only; explicit writable child binds remain rw. + if (!allowedWritePaths.includes(normalizedPath)) { + args.push('--remount-ro', normalizedPath); + } +} +""" + + +def patch_file(path: Path) -> None: + source = path.read_text(encoding="utf-8") + if _REPLACEMENT in source or _TMPFS_REPLACEMENT in source: + raise RuntimeError("sandbox runtime merged-usr patch is already patched") + if source.count(_ORIGINAL) != 1: + raise RuntimeError("sandbox runtime merged-usr patch target does not match pinned source") + if source.count(_TMPFS_ORIGINAL) != 1: + raise RuntimeError("sandbox runtime read-only tmpfs patch target does not match pinned source") + patched = source.replace(_ORIGINAL, _REPLACEMENT) + patched = patched.replace(_TMPFS_ORIGINAL, _TMPFS_REPLACEMENT) + path.write_text(patched, encoding="utf-8") + + +def main() -> None: + parser = argparse.ArgumentParser() + parser.add_argument("path", type=Path) + args = parser.parse_args() + patch_file(args.path) + + +if __name__ == "__main__": + main() diff --git a/start-langgraph.sh b/start-langgraph.sh index c8455ac..ec48527 100755 --- a/start-langgraph.sh +++ b/start-langgraph.sh @@ -7,6 +7,7 @@ LANGGRAPH_CONFIG="${PROJECT_DIR}/EvoScientist/langgraph_dev/langgraph.json" HOST="${EVOSCIENTIST_LANGGRAPH_HOST:-127.0.0.1}" PORT="${EVOSCIENTIST_LANGGRAPH_DEV_PORT:-3076}" WEB_ENV="${PROJECT_DIR}/../Ai4Sci-Web/.env" +N_JOBS="${EVOSCIENTIST_LANGGRAPH_JOBS_PER_WORKER:-6}" if [[ ! -x "${PROJECT_DIR}/.venv/bin/langgraph" ]]; then echo "LangGraph executable not found: ${PROJECT_DIR}/.venv/bin/langgraph" >&2 @@ -16,6 +17,26 @@ fi cd "${PROJECT_DIR}" +# Ask before stopping the process listening on PORT. Matching every connection +# would also terminate Gateway while it is connected to this Runtime. +PIDS="$(lsof -tiTCP:"${PORT}" -sTCP:LISTEN 2>/dev/null || true)" +if [[ -n "${PIDS}" ]]; then + echo "Port ${PORT} is already in use by PID(s): ${PIDS}" >&2 + ANSWER="" + read -r -p "Kill the process(es) using port ${PORT}? [y/N] " ANSWER || true + case "${ANSWER}" in + [yY]|[yY][eE][sS]) + echo "Killing process(es) on port ${PORT}: ${PIDS}" + kill ${PIDS} + sleep 1 + ;; + *) + echo "LangGraph not started; process(es) on port ${PORT} were left running." >&2 + exit 1 + ;; + esac +fi + # The Web Gateway always supplies a verified conversation workspace scope. export EVOSCIENTIST_DEPLOY_MODE="${EVOSCIENTIST_DEPLOY_MODE:-full}" export EVOSCIENTIST_WORKSPACE_DIR="${EVOSCIENTIST_WORKSPACE_DIR:-${PROJECT_DIR}/../.ai4sci/workspace}" @@ -27,4 +48,4 @@ exec uv run --env-file "${WEB_ENV}" langgraph dev \ --no-browser \ --no-reload \ --allow-blocking \ - --n-jobs-per-worker 1 + --n-jobs-per-worker "${N_JOBS}" diff --git a/tests/test_error_normalization_middleware.py b/tests/test_error_normalization_middleware.py index 1bd014d..537b933 100644 --- a/tests/test_error_normalization_middleware.py +++ b/tests/test_error_normalization_middleware.py @@ -15,6 +15,7 @@ from types import SimpleNamespace import pytest +from EvoScientist.llm.contracts import EvoRuntimeError from EvoScientist.llm.errors import ( AgentControlError, ModelToolProtocolError, @@ -169,6 +170,16 @@ class TestNormalize: assert _normalize(req, error) is None + def test_stable_runtime_error_passes_through(self): + req = _request(_openai_model()) + error = EvoRuntimeError( + "UPSTREAM_RATE_LIMITED", + "模型服务请求频率超限,请稍后重试或切换模型。", + details=({"http_status": 429},), + ) + + assert _normalize(req, error) is None + # --------------------------------------------------------------------------- # _is_provider_error — used by tool selector to distinguish provider @@ -397,6 +408,23 @@ class TestMiddleware: assert excinfo.value.code == "MODEL_TOOL_PROTOCOL_INVALID" assert excinfo.value.fallbackable is True + def test_awrap_preserves_stable_runtime_error_identity(self): + raised = EvoRuntimeError( + "UPSTREAM_RATE_LIMITED", + "模型服务请求频率超限,请稍后重试或切换模型。", + details=({"http_status": 429},), + ) + + async def handler(_req): + raise raised + + req = _request(_openai_model()) + with pytest.raises(EvoRuntimeError) as excinfo: + self._run_awrap(ErrorNormalizationMiddleware(), req, handler) + + assert excinfo.value is raised + assert excinfo.value.code == "UPSTREAM_RATE_LIMITED" + def test_awrap_wraps_any_exception_from_recognized_model(self): """Any exception raised inside a call to a provider-recognized model gets wrapped — including builtins like ``RuntimeError``. diff --git a/tests/test_gateway_proxy.py b/tests/test_gateway_proxy.py index cccf7ad..72961bb 100644 --- a/tests/test_gateway_proxy.py +++ b/tests/test_gateway_proxy.py @@ -4,6 +4,7 @@ import httpx import pytest from langchain_core.messages import HumanMessage +from EvoScientist.llm.contracts import EvoRuntimeError from EvoScientist.llm.gateway_proxy import GatewayProxyChatModel @@ -43,6 +44,18 @@ class _FakeClient: return _FakeStream(self._lines) +def test_runtime_error_repr_preserves_only_stable_code(): + error = EvoRuntimeError( + "UPSTREAM_RATE_LIMITED", + "safe display message", + details=({"provider_request": "must-not-persist"},), + ) + + assert repr(error) == "EvoRuntimeError(code='UPSTREAM_RATE_LIMITED')" + assert "safe display message" not in repr(error) + assert "must-not-persist" not in repr(error) + + @pytest.mark.anyio async def test_astream_yields_chunks_from_sse(monkeypatch): model = GatewayProxyChatModel( @@ -52,7 +65,7 @@ async def test_astream_yields_chunks_from_sse(monkeypatch): ) msg = {"type": "AIMessageChunk", "data": {"content": "hello"}} lines = [ - f'data: {json.dumps({"delta": {"message": msg}})}\n', + f"data: {json.dumps({'delta': {'message': msg}})}\n", 'data: {"delta": {"message": {"type": "AIMessageChunk", "data": {"content": " world"}}}}\n', "data: [DONE]\n", ] @@ -88,7 +101,7 @@ async def test_astream_roundtrips_streaming_tool_call_chunks(monkeypatch): ], }, } - lines = [f'data: {json.dumps({"delta": {"message": msg}})}\n', "data: [DONE]\n"] + lines = [f"data: {json.dumps({'delta': {'message': msg}})}\n", "data: [DONE]\n"] fake = _FakeClient(lines) monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: fake) @@ -114,3 +127,24 @@ async def test_astream_raises_on_missing_done(monkeypatch): with pytest.raises(RuntimeError, match="AI4SCI_MODEL_STREAM_INCOMPLETE"): _ = [c async for c in model._astream([HumanMessage(content="hi")])] + + +@pytest.mark.anyio +async def test_astream_projects_gateway_error_frame(monkeypatch): + model = GatewayProxyChatModel( + gateway_url="http://gw", + run_id="run-1", + envelope_signature="sig", + ) + lines = [ + 'data: {"type":"error","code":"UPSTREAM_RATE_LIMITED",' + '"status":429,"message":"模型服务请求频率超限,请稍后重试或切换模型。"}\n' + ] + fake = _FakeClient(lines) + monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: fake) + + with pytest.raises(EvoRuntimeError) as exc_info: + _ = [c async for c in model._astream([HumanMessage(content="hi")])] + + assert exc_info.value.code == "UPSTREAM_RATE_LIMITED" + assert exc_info.value.details == ({"http_status": 429},) diff --git a/tests/test_native_sandbox.py b/tests/test_native_sandbox.py index f1235f5..1c5cb6c 100644 --- a/tests/test_native_sandbox.py +++ b/tests/test_native_sandbox.py @@ -1,6 +1,8 @@ from __future__ import annotations import json +import sys +import types from pathlib import Path import pytest @@ -21,6 +23,18 @@ def _installation(tmp_path: Path) -> sandbox.NativeSandboxInstallation: ) +def test_existing_read_paths_resolve_and_deduplicate_symlink_aliases(tmp_path: Path): + usr = tmp_path / "usr" + usr.mkdir() + bin_alias = tmp_path / "bin" + bin_alias.symlink_to(usr, target_is_directory=True) + missing = tmp_path / "missing" + + paths = sandbox._existing_resolved_paths((str(usr), str(bin_alias), str(missing))) + + assert paths == (str(usr.resolve()),) + + def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path): files = tmp_path / "files" command_tmp = tmp_path / "runtime" / "tmp" / "run" @@ -38,6 +52,7 @@ def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path "/dev/null", ] assert policy["filesystem"]["denyWrite"] == [ + str(files / "uploads"), "/tmp/claude", "/private/tmp/claude", "/dev/tty", @@ -50,6 +65,48 @@ def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path assert "control" not in json.dumps(policy) +def test_weaker_nested_mode_requires_explicit_environment_opt_in(tmp_path: Path, monkeypatch): + monkeypatch.delenv("EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", raising=False) + files = tmp_path / "files" + command_tmp = tmp_path / "runtime" / "tmp" / "run" + files.mkdir() + command_tmp.mkdir(parents=True) + installation = _installation(tmp_path) + + assert sandbox._sandbox_settings(installation, files, command_tmp)[ + "enableWeakerNestedSandbox" + ] is False + + monkeypatch.setenv("EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", "true") + assert sandbox._sandbox_settings(installation, files, command_tmp)[ + "enableWeakerNestedSandbox" + ] is True + + +def test_network_preflight_probe_accepts_kernel_denied_unix_socket(monkeypatch): + monkeypatch.delenv("OPENAI_API_KEY", raising=False) + + class FakeSocket: + def __init__(self, family=None, *_args): + if family == 1: + raise PermissionError("blocked by seccomp") + + def connect_ex(self, _address): + return 1 + + def close(self): + return None + + fake_socket = types.SimpleNamespace( + AF_UNIX=1, + socket=lambda family=None, *args: FakeSocket(family, *args), + ) + monkeypatch.setitem(sys.modules, "socket", fake_socket) + + namespace: dict[str, object] = {} + exec(sandbox._network_preflight_probe(1234, Path("/blocked.sock")), namespace) + + def test_clean_environment_does_not_inherit_secrets(tmp_path: Path, monkeypatch): command_tmp = tmp_path / "tmp" (command_tmp / "home").mkdir(parents=True) diff --git a/tests/test_native_sandbox_runtime_patch.py b/tests/test_native_sandbox_runtime_patch.py new file mode 100644 index 0000000..ec6a7e9 --- /dev/null +++ b/tests/test_native_sandbox_runtime_patch.py @@ -0,0 +1,62 @@ +from __future__ import annotations + +import runpy +from collections.abc import Callable +from pathlib import Path +from typing import cast + +import pytest + +PATCH_SCRIPT = Path(__file__).parents[1] / "runtime" / "native-sandbox" / "patch_merged_usr.py" + + +def test_patch_skips_only_symlink_aliases_covered_by_read_allow(tmp_path: Path): + namespace = runpy.run_path(str(PATCH_SCRIPT)) + patch_file = cast(Callable[[Path], None], namespace["patch_file"]) + + source = tmp_path / "linux-sandbox-utils.js" + source.write_text( + """function pushReadDenyDirMounts(args, normalizedPath, allowedWritePaths, readAllowPaths) { + const denySep = normalizedPath === '/' ? '/' : normalizedPath + '/'; + args.push('--tmpfs', normalizedPath); + for (const writePath of allowedWritePaths) { + if (writePath.startsWith(denySep) || writePath === normalizedPath) { + args.push('--bind', writePath, writePath); + } + } + for (const allowPath of readAllowPaths) { + if (allowPath.startsWith(denySep) || allowPath === normalizedPath) { + if (!fs.existsSync(allowPath)) { + continue; + } + if (allowedWritePaths.some(w => (w.startsWith(denySep) || w === normalizedPath) && + (allowPath === w || allowPath.startsWith(w + '/')))) { + continue; + } + args.push('--ro-bind', allowPath, allowPath); + logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`); + } + } +} + const rootSkip = new Set(['proc', 'dev', 'sys']); + for (const p of readConfig?.denyOnly || []) { + if (normalizePathForSandbox(p) === '/') { + for (const child of fs.readdirSync('/')) { + if (!rootSkip.has(child)) + readDenyPaths.push('/' + child); + } + } +""", + encoding="utf-8", + ) + + patch_file(source) + patched = source.read_text(encoding="utf-8") + + assert "isSymbolicLink()" in patched + assert "readAllowPaths.some" in patched + assert "resolved.startsWith(allowPath + '/')" in patched + assert "args.push('--remount-ro', normalizedPath)" in patched + assert "!allowedWritePaths.includes(normalizedPath)" in patched + with pytest.raises(RuntimeError, match="already patched"): + patch_file(source) diff --git a/tests/test_prompts.py b/tests/test_prompts.py index 8777339..d944c4a 100644 --- a/tests/test_prompts.py +++ b/tests/test_prompts.py @@ -38,6 +38,23 @@ class TestGetSystemPrompt: result = get_system_prompt() assert "Shell Execution Guidelines" in result + def test_contains_safe_archive_and_sqlite_programming_contracts(self): + result = get_system_prompt(native_web_sandbox=True) + + assert "never use `extractall`" in result + assert "mode=ro&immutable=1" in result + assert "PRAGMA query_only=ON" in result + assert "Never execute files extracted from an archive" in result + assert "Do not modify original files under `uploads/`" in result + + def test_distinguishes_live_web_search_from_local_memory_search(self): + result = get_system_prompt(native_web_sandbox=True) + + assert "tavily_search" in result + assert "search_observations" in result + assert "local memory" in result.lower() + assert "Do not use `execute`, `curl`, or `httpx`" in result + def test_contains_delegation(self): result = get_system_prompt() assert "Sub-Agent Delegation" in result diff --git a/tests/test_recoverable_tools.py b/tests/test_recoverable_tools.py index b42e0d2..d6ca130 100644 --- a/tests/test_recoverable_tools.py +++ b/tests/test_recoverable_tools.py @@ -1,5 +1,9 @@ from __future__ import annotations +import httpx +import pytest + +from EvoScientist.llm.contracts import EvoRuntimeError from EvoScientist.middleware import recoverable_tools @@ -51,3 +55,36 @@ def test_evomemory_never_falls_back_to_parent_model_proxy(monkeypatch): assert proxy is None assert metadata["run_kind"] == "evomemory_turn_worker" + + +@pytest.mark.asyncio +async def test_tool_effect_gateway_error_preserves_machine_code(monkeypatch): + class Client: + async def __aenter__(self): + return self + + async def __aexit__(self, *_): + return None + + async def post(self, url, json): + return httpx.Response( + 409, + json={"detail": {"code": "RUN_FENCE_LOST"}}, + request=httpx.Request("POST", url), + ) + + monkeypatch.setattr(recoverable_tools.httpx, "AsyncClient", lambda **_: Client()) + + with pytest.raises(EvoRuntimeError) as exc_info: + await recoverable_tools._post( + { + "gateway_url": "http://gateway", + "run_id": "run-1", + "envelope_signature": "signature", + }, + "prepare", + {}, + ) + + assert exc_info.value.code == "RUN_FENCE_LOST" + assert repr(exc_info.value) == "EvoRuntimeError(code='RUN_FENCE_LOST')" diff --git a/tests/test_sessions.py b/tests/test_sessions.py index 12b5b9a..2a23070 100644 --- a/tests/test_sessions.py +++ b/tests/test_sessions.py @@ -22,13 +22,19 @@ from EvoScientist.sessions import ( delete_thread, find_similar_threads, generate_thread_id, + get_aggregated_storage_stats, get_db_path, get_most_recent, get_thread_messages, get_thread_metadata, + list_all_session_db_paths, + list_all_thread_ids, list_threads, + prune_all_stale_threads, + prune_thread_history, resolve_thread_id_prefix, thread_exists, + vacuum_db, ) @@ -2919,5 +2925,172 @@ class TestRestoreWebuiThreadsToGlobalStore(unittest.IsolatedAsyncioTestCase): assert restore_called, "_restore_webui_threads_to_global_store must be called" +def _uuid6_from_unix(ts_unix: float) -> str: + """Build a UUIDv6 (time-ordered checkpoint id) from a unix timestamp. + + Production checkpoint ids are UUIDv6, so lexicographic order matches + insertion order and the timestamp is recoverable from the id itself. + """ + greg = int((ts_unix + 12219292800) * 10_000_000) & ((1 << 60) - 1) + high48, low12 = greg >> 12, greg & 0xFFF + rand = uuid.uuid4().int & ((1 << 62) - 1) + value = (high48 << 80) | (0x6 << 76) | (low12 << 64) | (0b10 << 62) | rand + return str(uuid.UUID(int=value)) + + +class TestPruneFunctions(unittest.IsolatedAsyncioTestCase): + """Tests for the prune/vacuum API used by the gateway timer and admin routes.""" + + async def asyncSetUp(self): + import time + + import aiosqlite + + self._tmpdir = tempfile.mkdtemp() + self.db_path = os.path.join(self._tmpdir, "prune_test.db") + now = time.time() + + async with aiosqlite.connect(self.db_path) as conn: + await conn.execute(""" + CREATE TABLE checkpoints ( + thread_id TEXT NOT NULL, + checkpoint_ns TEXT NOT NULL DEFAULT '', + checkpoint_id TEXT NOT NULL, + parent_checkpoint_id TEXT, + type TEXT, + checkpoint BLOB, + metadata TEXT NOT NULL DEFAULT '{}', + PRIMARY KEY (thread_id, checkpoint_ns, checkpoint_id) + ) + """) + await conn.execute(""" + CREATE TABLE writes ( + thread_id TEXT NOT NULL, + checkpoint_ns TEXT NOT NULL DEFAULT '', + checkpoint_id TEXT NOT NULL, + task_id TEXT NOT NULL, + idx INTEGER NOT NULL, + channel TEXT NOT NULL, + type TEXT, + value BLOB, + PRIMARY KEY (thread_id, checkpoint_ns, checkpoint_id, task_id, idx) + ) + """) + await self._insert_thread(conn, "old_thread", 5, now - 10 * 86400) + await self._insert_thread(conn, "new_thread", 3, now - 60) + await self._insert_thread( + conn, "other", 4, now - 10 * 86400, agent="OtherAgent" + ) + await conn.commit() + + async def asyncTearDown(self): + try: + os.unlink(self.db_path) + os.rmdir(self._tmpdir) + except OSError: + pass + + async def _insert_thread(self, conn, tid, count, ts_base, agent=AGENT_NAME): + serde = JsonPlusSerializer() + ctype, cblob = serde.dumps_typed( + {"channel_values": {"messages": [HumanMessage(content=f"seed-{tid}")]}} + ) + prev = None + for i in range(count): + cid = _uuid6_from_unix(ts_base + i) + await conn.execute( + "INSERT INTO checkpoints (thread_id, checkpoint_ns, checkpoint_id," + " parent_checkpoint_id, type, checkpoint, metadata)" + " VALUES (?, '', ?, ?, ?, ?, ?)", + (tid, cid, prev, ctype, cblob, json.dumps({"agent_name": agent})), + ) + await conn.execute( + "INSERT INTO writes (thread_id, checkpoint_ns, checkpoint_id," + " task_id, idx, channel, type, value)" + " VALUES (?, '', ?, 'task', 0, 'ch', 'str', ?)", + (tid, cid, b"x"), + ) + prev = cid + + async def _count(self, tid, table="checkpoints"): + import aiosqlite + + async with aiosqlite.connect(self.db_path) as conn: + async with conn.execute( + f"SELECT COUNT(*) FROM {table} WHERE thread_id = ?", (tid,) + ) as cur: + return (await cur.fetchone())[0] + + async def test_prune_thread_history(self): + result = await prune_thread_history( + "old_thread", keep_last=2, db_path=self.db_path + ) + # keep_last=2 anchors + 1 snapshot-seed ancestor preserved + assert result == {"deleted_checkpoints": 2, "deleted_writes": 2} + assert await self._count("old_thread") == 3 + assert await self._count("old_thread", "writes") == 3 + + async def test_prune_thread_history_other_agent_untouched(self): + result = await prune_thread_history("other", keep_last=1, db_path=self.db_path) + assert result == {"deleted_checkpoints": 0, "deleted_writes": 0} + assert await self._count("other") == 4 + + async def test_prune_all_stale_threads(self): + result = await prune_all_stale_threads( + max_age_hours=24, keep_last=2, db_path=self.db_path + ) + assert result["databases_processed"] == 1 + assert result["threads_pruned"] == 1 + assert result["total_deleted_checkpoints"] == 2 + assert result["total_deleted_writes"] == 2 + # fresh thread and foreign-agent thread untouched + assert await self._count("new_thread") == 3 + assert await self._count("other") == 4 + + async def test_prune_all_stale_threads_none_stale(self): + result = await prune_all_stale_threads( + max_age_hours=24 * 365, keep_last=2, db_path=self.db_path + ) + assert result["threads_pruned"] == 0 + assert result["total_deleted_checkpoints"] == 0 + assert await self._count("old_thread") == 5 + + async def test_list_all_thread_ids(self): + ids = await list_all_thread_ids(db_path=self.db_path) + assert sorted(ids) == ["new_thread", "old_thread"] + + async def test_vacuum_db(self): + result = await vacuum_db(db_path=self.db_path) + assert result["size_after_bytes"] > 0 + assert result["size_before_bytes"] >= result["size_after_bytes"] + + async def test_get_aggregated_storage_stats(self): + with patch( + "EvoScientist.sessions.get_db_path", + return_value=_mock_path(self.db_path), + ): + stats = await get_aggregated_storage_stats() + assert stats["thread_count"] == 2 + assert stats["checkpoint_count"] == 8 + assert stats["thread_depth"]["max"] == 5 + assert stats["thread_depth"]["min"] == 3 + + async def test_list_all_session_db_paths(self): + with patch( + "EvoScientist.sessions.get_db_path", + return_value=_mock_path(self.db_path), + ): + paths = list_all_session_db_paths() + assert len(paths) == 1 + assert str(paths[0]) == self.db_path + + missing = os.path.join(self._tmpdir, "nope.db") + with patch( + "EvoScientist.sessions.get_db_path", + return_value=_mock_path(missing), + ): + assert list_all_session_db_paths() == [] + + if __name__ == "__main__": unittest.main() diff --git a/tests/test_subagent_timeout.py b/tests/test_subagent_timeout.py new file mode 100644 index 0000000..fc71b9b --- /dev/null +++ b/tests/test_subagent_timeout.py @@ -0,0 +1,174 @@ +from __future__ import annotations + +import asyncio +from unittest.mock import MagicMock + +import pytest +from langchain_core.messages import ToolMessage + +from EvoScientist.middleware.subagent_timeout import SubagentTimeoutMiddleware + + +def _request(name: str = "task"): + request = MagicMock() + request.tool_call = {"id": "call-1", "name": name, "args": {}} + return request + + +@pytest.mark.anyio +async def test_subagent_timeout_returns_stable_tool_error(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01) + + async def handler(_request): + await asyncio.sleep(1) + return ToolMessage(content="late", tool_call_id="call-1", name="task") + + result = await middleware.awrap_tool_call(_request(), handler) + + assert isinstance(result, ToolMessage) + assert result.status == "error" + assert result.name == "task" + assert result.additional_kwargs["error_code"] == "SUBAGENT_TIMEOUT" + assert "SUBAGENT_TIMEOUT" in result.content + + +@pytest.mark.anyio +async def test_subagent_timeout_passes_success_through(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=1) + expected = ToolMessage(content="done", tool_call_id="call-1", name="task") + + async def handler(_request): + return expected + + assert await middleware.awrap_tool_call(_request(), handler) is expected + + +@pytest.mark.anyio +async def test_subagent_timeout_does_not_bound_other_tools(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01) + expected = ToolMessage(content="done", tool_call_id="call-1", name="read_file") + + async def handler(_request): + await asyncio.sleep(0.02) + return expected + + assert await middleware.awrap_tool_call(_request("read_file"), handler) is expected + + +@pytest.mark.anyio +async def test_subagent_internal_timeout_error_is_not_reclassified(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=1) + + async def handler(_request): + raise TimeoutError("provider timed out immediately") + + with pytest.raises(TimeoutError, match="provider timed out immediately"): + await middleware.awrap_tool_call(_request(), handler) + + +@pytest.mark.anyio +async def test_parent_cancellation_cancels_subagent_handler(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=10) + handler_cancelled = asyncio.Event() + + async def handler(_request) -> ToolMessage: + try: + await asyncio.Event().wait() + finally: + handler_cancelled.set() + return ToolMessage(content="done", tool_call_id="call-1", name="task") + + invocation = asyncio.create_task( + middleware.awrap_tool_call(_request(), handler) + ) + await asyncio.sleep(0) + invocation.cancel() + + with pytest.raises(asyncio.CancelledError): + await invocation + assert handler_cancelled.is_set() + + +@pytest.mark.anyio +async def test_parent_cancellation_wins_over_handler_cleanup_error(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=10) + + async def handler(_request) -> ToolMessage: + try: + await asyncio.Event().wait() + except asyncio.CancelledError as exc: + raise RuntimeError("cleanup failed") from exc + return ToolMessage(content="done", tool_call_id="call-1", name="task") + + invocation = asyncio.create_task( + middleware.awrap_tool_call(_request(), handler) + ) + await asyncio.sleep(0) + invocation.cancel() + + with pytest.raises(asyncio.CancelledError): + await invocation + + +@pytest.mark.anyio +async def test_deadline_wins_over_handler_cleanup_error(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01) + + async def handler(_request) -> ToolMessage: + try: + await asyncio.Event().wait() + except asyncio.CancelledError as exc: + raise RuntimeError("cleanup failed") from exc + return ToolMessage(content="done", tool_call_id="call-1", name="task") + + result = await middleware.awrap_tool_call(_request(), handler) + + assert isinstance(result, ToolMessage) + assert result.additional_kwargs["error_code"] == "SUBAGENT_TIMEOUT" + + +@pytest.mark.anyio +async def test_parent_cancellation_during_deadline_cleanup_is_not_swallowed(): + middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01) + cleanup_started = asyncio.Event() + release_cleanup = asyncio.Event() + + async def handler(_request) -> ToolMessage: + try: + await asyncio.Event().wait() + except asyncio.CancelledError: + cleanup_started.set() + await release_cleanup.wait() + raise + return ToolMessage(content="done", tool_call_id="call-1", name="task") + + invocation = asyncio.create_task( + middleware.awrap_tool_call(_request(), handler) + ) + await cleanup_started.wait() + invocation.cancel() + release_cleanup.set() + with pytest.raises(asyncio.CancelledError): + await invocation + + +@pytest.mark.anyio +async def test_parent_cancellation_after_cleanup_before_timeout_return_wins(monkeypatch): + middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01) + + async def handler(_request) -> ToolMessage: + await asyncio.Event().wait() + return ToolMessage(content="done", tool_call_id="call-1", name="task") + + async def finish_cleanup_then_cancel_parent(_task): + current = asyncio.current_task() + assert current is not None + current.cancel() + + monkeypatch.setattr(middleware, "_cancel_task", finish_cleanup_then_cancel_parent) + + invocation = asyncio.create_task( + middleware.awrap_tool_call(_request(), handler) + ) + with pytest.raises(asyncio.CancelledError): + await invocation diff --git a/tests/test_web_search.py b/tests/test_web_search.py new file mode 100644 index 0000000..3fd09a6 --- /dev/null +++ b/tests/test_web_search.py @@ -0,0 +1,132 @@ +from __future__ import annotations + +import pytest + + +@pytest.mark.anyio +async def test_tavily_search_keeps_indexed_summary_when_source_fetch_fails(monkeypatch): + from EvoScientist.tools import search + + class _Client: + def search(self, *_args, **_kwargs): + return { + "results": [ + { + "title": "AIR staff profile", + "url": "https://air.cas.cn/example", + "content": "Indexed staff-profile summary.", + } + ] + } + + recorded: list[tuple[str, str]] = [] + + async def fetch_failed(_url: str, timeout: float = 10.0) -> str: + return "Error fetching content from https://air.cas.cn/example: DNS failed" + + async def record(service: str, action: str) -> None: + recorded.append((service, action)) + + monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client()) + monkeypatch.setattr(search, "fetch_webpage_content", fetch_failed) + monkeypatch.setattr("EvoScientist.runtime_integrations.record_service_usage", record) + + result = await search.tavily_search.ainvoke({"query": "高铭 空天院"}) + + assert "Indexed staff-profile summary." in result + assert "https://air.cas.cn/example" in result + assert "Tavily-indexed summary" in result + assert recorded == [("tavily", "search")] + + +@pytest.mark.anyio +async def test_tavily_search_bounds_fetched_page_content(monkeypatch): + from EvoScientist.tools import search + + class _Client: + def search(self, *_args, **_kwargs): + return { + "results": [ + { + "title": f"Result {index}", + "url": f"https://example.com/{index}", + "content": f"Indexed summary {index}", + } + for index in range(3) + ] + } + + async def huge_page(_url: str, timeout: float = 10.0) -> str: + return "page-content " * 10_000 + + monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client()) + monkeypatch.setattr(search, "fetch_webpage_content", huge_page) + + result = await search.tavily_search.ainvoke({"query": "bounded search"}) + + assert len(result) <= search.MAX_SEARCH_RESULT_CHARS + for index in range(3): + assert f"https://example.com/{index}" in result + assert "[page content truncated]" in result + + +@pytest.mark.anyio +async def test_tavily_search_preserves_every_result_url_under_total_budget(monkeypatch): + from EvoScientist.tools import search + + class _Client: + def search(self, *_args, **_kwargs): + return { + "results": [ + { + "title": f"Result {index} " + ("very-long-title " * 800), + "url": f"https://example.com/result-{index}", + "content": f"Indexed summary {index}", + } + for index in range(3) + ] + } + + async def page(_url: str, timeout: float = 10.0) -> str: + return "page-content " * 1_000 + + monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client()) + monkeypatch.setattr(search, "fetch_webpage_content", page) + + result = await search.tavily_search.ainvoke({"query": "preserve urls"}) + + assert len(result) <= search.MAX_SEARCH_RESULT_CHARS + for index in range(3): + assert f"https://example.com/result-{index}" in result + assert "[search result content truncated to preserve all titles and URLs]" in result + + +@pytest.mark.anyio +async def test_tavily_search_bounds_maliciously_long_url(monkeypatch): + from EvoScientist.tools import search + + long_url = "https://example.com/" + ("a" * 20_000) + + class _Client: + def search(self, *_args, **_kwargs): + return { + "results": [ + { + "title": "Long URL result", + "url": long_url, + "content": "Indexed summary", + } + ] + } + + async def page(_url: str, timeout: float = 10.0) -> str: + return "page" + + monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client()) + monkeypatch.setattr(search, "fetch_webpage_content", page) + + result = await search.tavily_search.ainvoke({"query": "long url"}) + + assert len(result) <= search.MAX_SEARCH_RESULT_CHARS + assert "https://example.com/" in result + assert "[URL truncated]" in result diff --git a/tests/test_web_tool_registry.py b/tests/test_web_tool_registry.py index 76ec011..af2b3d2 100644 --- a/tests/test_web_tool_registry.py +++ b/tests/test_web_tool_registry.py @@ -6,6 +6,184 @@ from EvoScientist.llm.contracts import EvoRuntimeError from EvoScientist.web_runtime import _ToolRegistryFenceMiddleware +def test_web_registry_describes_tavily_as_controlled_live_search(monkeypatch): + monkeypatch.setenv("TAVILY_API_KEY", "test-key") + + from EvoScientist.web_runtime import web_tool_registry_manifest + + manifest, _revision = web_tool_registry_manifest() + tavily = next(item for item in manifest if item["name"] == "tavily_search") + + assert "live public web" in tavily["description"].lower() + assert "execute" in tavily["description"].lower() + + +def test_base_kwargs_install_tavily_on_main_agent(monkeypatch): + import EvoScientist.EvoScientist as agent_module + + monkeypatch.setenv("TAVILY_API_KEY", "test-key") + monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None) + monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None) + monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs) + monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt") + monkeypatch.setattr("EvoScientist.utils.load_subagents", lambda *_args, **_kwargs: []) + + kwargs = agent_module._build_base_kwargs( + object(), + [], + cfg=object(), + chat_model=object(), + workspace_dir="/workspace", + ) + + assert "tavily_search" in {getattr(tool, "name", "") for tool in kwargs["tools"]} + + +def _stub_agent_build(monkeypatch, agent_module, subagents=None): + monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None) + monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None) + monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs) + monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt") + monkeypatch.setattr( + "EvoScientist.utils.load_subagents", + lambda *_args, **_kwargs: list(subagents or []), + ) + + +@pytest.mark.parametrize( + "reserved_name", + [ + "skill_manager", + "execute", + "start_async_task", + "check_async_task", + "update_async_task", + "cancel_async_task", + "list_async_tasks", + ], +) +def test_mcp_cannot_override_reserved_tool(monkeypatch, reserved_name): + from types import SimpleNamespace + + import EvoScientist.EvoScientist as agent_module + + monkeypatch.delenv("TAVILY_API_KEY", raising=False) + monkeypatch.setattr( + agent_module, + "_load_mcp_tools_cached", + lambda **_kwargs: {"main": [SimpleNamespace(name=reserved_name)]}, + ) + _stub_agent_build(monkeypatch, agent_module) + + with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"): + agent_module.load_mcp_and_build_kwargs( + object(), + [], + cfg=object(), + chat_model=object(), + workspace_dir="/workspace", + ) + + +def test_mcp_cannot_duplicate_existing_subagent_tool(monkeypatch): + from types import SimpleNamespace + + import EvoScientist.EvoScientist as agent_module + + existing = SimpleNamespace(name="shared_search") + injected = SimpleNamespace(name="shared_search") + monkeypatch.delenv("TAVILY_API_KEY", raising=False) + monkeypatch.setattr( + agent_module, + "_load_mcp_tools_cached", + lambda **_kwargs: {"research-agent": [injected]}, + ) + _stub_agent_build( + monkeypatch, + agent_module, + subagents=[{"name": "research-agent", "tools": [existing]}], + ) + + with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"): + agent_module.load_mcp_and_build_kwargs( + object(), + [], + cfg=object(), + chat_model=object(), + workspace_dir="/workspace", + ) + + +def test_mcp_cannot_override_builtin_tavily(monkeypatch): + from types import SimpleNamespace + + import EvoScientist.EvoScientist as agent_module + + monkeypatch.setenv("TAVILY_API_KEY", "test-key") + monkeypatch.setattr( + agent_module, + "_load_mcp_tools_cached", + lambda **_kwargs: {"main": [SimpleNamespace(name="tavily_search")]}, + ) + monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None) + monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None) + monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs) + monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt") + monkeypatch.setattr("EvoScientist.utils.load_subagents", lambda *_args, **_kwargs: []) + + with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"): + agent_module.load_mcp_and_build_kwargs( + object(), + [], + cfg=object(), + chat_model=object(), + workspace_dir="/workspace", + ) + + +def test_same_mcp_tool_can_be_routed_to_multiple_agents(monkeypatch): + from types import SimpleNamespace + + import EvoScientist.EvoScientist as agent_module + + shared_main = SimpleNamespace(name="shared_search") + shared_research = SimpleNamespace(name="shared_search") + monkeypatch.delenv("TAVILY_API_KEY", raising=False) + monkeypatch.setattr( + agent_module, + "_load_mcp_tools_cached", + lambda **_kwargs: { + "main": [shared_main], + "research-agent": [shared_research], + }, + ) + monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None) + monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None) + monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs) + monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt") + monkeypatch.setattr( + "EvoScientist.utils.load_subagents", + lambda *_args, **_kwargs: [ + {"name": "research-agent", "tools": []} + ], + ) + + kwargs = agent_module.load_mcp_and_build_kwargs( + object(), + [], + cfg=object(), + chat_model=object(), + workspace_dir="/workspace", + ) + + assert shared_main in kwargs["tools"] + research = next( + subagent for subagent in kwargs["subagents"] + if subagent["name"] == "research-agent" + ) + assert shared_research in research["tools"] + + def test_tool_dispatch_fence_rejects_changed_registry(monkeypatch): monkeypatch.setattr( "EvoScientist.web_runtime.web_tool_registry_manifest", diff --git a/tests/test_workspace_files.py b/tests/test_workspace_files.py index 2e1c1fa..f1a357d 100644 --- a/tests/test_workspace_files.py +++ b/tests/test_workspace_files.py @@ -2,10 +2,14 @@ from __future__ import annotations import base64 import os +import sqlite3 +import zipfile +from io import BytesIO from pathlib import Path import pytest from deepagents.backends.protocol import ExecuteResponse +from PIL import Image from EvoScientist.native_sandbox import ( NativeSandboxExecutor, @@ -27,6 +31,19 @@ from EvoScientist.workspace_scope import ( ) +def test_extracted_documents_route_through_deepagents_as_text(): + import deepagents.middleware.filesystem as filesystem_middleware + + from EvoScientist.llm.patches import _patch_deepagents_extracted_document_text + + _patch_deepagents_extracted_document_text() + + get_file_type = vars(filesystem_middleware)["_get_file_type"] + assert get_file_type("/workspace/report.pptx") == "text" + assert get_file_type("/workspace/report.pdf") == "text" + assert get_file_type("/workspace/image.png") == "image" + + def test_normalize_workspace_path_is_strict(): assert normalize_workspace_path("/workspace") == () assert normalize_workspace_path("/workspace/reports/a.txt") == ( @@ -82,25 +99,366 @@ def test_root_lists_workspace_namespace(tmp_path: Path): ] -def test_docx_and_unknown_binary_read_with_base64_contract(tmp_path: Path): +def test_docx_read_extracts_text_instead_of_returning_base64(tmp_path: Path): backend = ScopedFilesystemBackend(tmp_path) - docx = b"PK\x03\x04\x00word/document.xml" - unknown = b"custom\x00binary" - backend.upload_files( - [ - ("/workspace/input.docx", docx), - ("/workspace/payload.custom", unknown), - ] + docx = tmp_path / "input.docx" + with zipfile.ZipFile(docx, "w") as archive: + archive.writestr( + "word/document.xml", + """ + + 有界文档内容 + """, + ) + + result = backend.read("/workspace/input.docx") + + assert result.error is None + assert result.file_data is not None + assert result.file_data["encoding"] == "utf-8" + assert "有界文档内容" in result.file_data["content"] + assert "base64" not in result.file_data["content"] + + +@pytest.mark.parametrize( + ("filename", "kind", "expected"), + [ + ("archive.zip", "archive", "list"), + ("results.sqlite", "database", "mode=ro"), + ("program.exe", "executable", "must not be executed"), + ("payload.custom", "binary", "unsupported"), + ], +) +def test_non_document_binary_returns_bounded_processing_guidance( + tmp_path: Path, filename: str, kind: str, expected: str +): + if filename.endswith(".zip"): + with zipfile.ZipFile(tmp_path / filename, "w") as archive: + archive.writestr("notes.txt", "hello") + elif filename.endswith(".sqlite"): + connection = sqlite3.connect(tmp_path / filename) + connection.execute("CREATE TABLE results(id INTEGER PRIMARY KEY, value TEXT)") + connection.commit() + connection.close() + elif filename.endswith(".exe"): + (tmp_path / filename).write_bytes(b"MZ\x00binary payload") + else: + (tmp_path / filename).write_bytes(b"custom\x00binary payload") + + result = ScopedFilesystemBackend(tmp_path).read(f"/workspace/{filename}") + + assert result.file_data is None + assert result.error is not None + assert "BINARY_PROCESSING_REQUIRED" in result.error or "UNSUPPORTED_BINARY_FILE" in result.error + assert f'"kind": "{kind}"' in result.error + assert expected.lower() in result.error.lower() + assert len(result.error) < 4000 + + +def test_image_read_keeps_base64_media_contract(tmp_path: Path): + buffer = BytesIO() + Image.new("RGB", (32, 24), (1, 2, 3)).save(buffer, "PNG") + raw = buffer.getvalue() + (tmp_path / "image.png").write_bytes(raw) + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/image.png") + + assert result.error is None + assert result.file_data is not None + assert result.file_data["encoding"] == "base64" + + +def test_large_image_is_downsampled_before_base64_delivery(tmp_path: Path): + Image.new("RGB", (3000, 1200), (1, 2, 3)).save(tmp_path / "large.png", "PNG") + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/large.png") + + assert result.error is None + assert result.file_data is not None + decoded = base64.standard_b64decode(result.file_data["content"]) + image = Image.open(BytesIO(decoded)) + image.load() + assert image.size == (2048, 819) + assert image.format == "JPEG" + + +def test_corrupt_image_returns_error_instead_of_base64(tmp_path: Path): + (tmp_path / "broken.png").write_bytes(b"\x89PNG\r\nnot-decodable") + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.png") + + assert result.file_data is None + assert result.error is not None + assert "IMAGE_PROCESSING_FAILED" in result.error + + +def test_image_pixel_budget_is_enforced_before_model_delivery( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +): + import EvoScientist.document_extract as document_extract + + Image.new("RGB", (100, 100), (1, 2, 3)).save(tmp_path / "pixels.png", "PNG") + monkeypatch.setattr(document_extract, "MAX_IMAGE_PIXELS", 9_999) + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/pixels.png") + + assert result.file_data is None + assert result.error is not None + assert "IMAGE_PIXEL_BUDGET_EXCEEDED" in result.error + + +def test_multiframe_image_is_reduced_to_first_frame(tmp_path: Path): + frames = [Image.new("RGB", (20, 10), color) for color in ((255, 0, 0), (0, 255, 0))] + frames[0].save( + tmp_path / "animated.gif", + format="GIF", + save_all=True, + append_images=frames[1:], + duration=100, + loop=0, ) - assert backend.read("/workspace/input.docx").file_data == { - "content": base64.standard_b64encode(docx).decode("ascii"), - "encoding": "base64", - } - assert backend.read("/workspace/payload.custom").file_data == { - "content": base64.standard_b64encode(unknown).decode("ascii"), - "encoding": "base64", - } + result = ScopedFilesystemBackend(tmp_path).read("/workspace/animated.gif") + + assert result.error is None + assert result.file_data is not None + decoded = base64.standard_b64decode(result.file_data["content"]) + image = Image.open(BytesIO(decoded)) + image.load() + assert getattr(image, "n_frames", 1) == 1 + + +def test_pptx_read_extracts_slide_text_without_base64(tmp_path: Path): + with zipfile.ZipFile(tmp_path / "deck.pptx", "w") as archive: + archive.writestr( + "ppt/slides/slide1.xml", + """ + 总体技术架构核心能力说明 + """, + ) + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/deck.pptx") + + assert result.error is None + assert result.file_data is not None + assert result.file_data["encoding"] == "utf-8" + assert "## Slide 1" in result.file_data["content"] + assert "总体技术架构" in result.file_data["content"] + + +def test_document_output_is_bounded_and_returns_continuation_hint(tmp_path: Path): + long_text = "\n".join(f"第{i:05d}行-" + "x" * 80 for i in range(2000)) + with zipfile.ZipFile(tmp_path / "long.docx", "w") as archive: + paragraphs = "".join( + f"{line}" + for line in long_text.splitlines() + ) + archive.writestr( + "word/document.xml", + f""" + {paragraphs}""", + ) + + result = ScopedFilesystemBackend(tmp_path).read( + "/workspace/long.docx", offset=0, limit=2000 + ) + + assert result.error is None + assert result.file_data is not None + content = result.file_data["content"] + assert len(content) < 51_000 + assert "DOCUMENT_OUTPUT_TRUNCATED" in content + assert "use offset=" in content + + +def test_corrupt_document_does_not_fall_back_to_base64(tmp_path: Path): + (tmp_path / "broken.pptx").write_bytes(b"PK\x03\x04not-a-real-presentation") + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.pptx") + + assert result.file_data is None + assert result.error is not None + assert "DOCUMENT_EXTRACTION_FAILED" in result.error + assert "base64" not in result.error.lower() + + +def test_ooxml_member_expansion_budget_blocks_compression_bomb( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +): + import EvoScientist.document_extract as document_extract + + with zipfile.ZipFile( + tmp_path / "bomb.docx", "w", compression=zipfile.ZIP_DEFLATED + ) as archive: + archive.writestr( + "word/document.xml", + """ + expanded content + """, + ) + monkeypatch.setattr(document_extract, "MAX_OOXML_MEMBER_BYTES", 16) + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/bomb.docx") + + assert result.file_data is None + assert result.error is not None + assert "DOCUMENT_RESOURCE_LIMIT" in result.error + + +@pytest.mark.parametrize("member", ["../word/document.xml", "/word/document.xml"]) +def test_ooxml_rejects_unsafe_member_paths(tmp_path: Path, member: str): + with zipfile.ZipFile(tmp_path / "unsafe.docx", "w") as archive: + archive.writestr(member, "content") + archive.writestr( + "word/document.xml", + """ + safe""", + ) + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/unsafe.docx") + + assert result.file_data is None + assert result.error is not None + assert "DOCUMENT_RESOURCE_LIMIT" in result.error + + +def test_ooxml_rejects_duplicate_member_names(tmp_path: Path): + def write_duplicate_document(path: Path) -> None: + with zipfile.ZipFile(path, "w") as archive: + for text in ("first", "second"): + archive.writestr( + "word/document.xml", + f""" + {text}""", + ) + + with pytest.warns(UserWarning, match="Duplicate name"): + write_duplicate_document(tmp_path / "duplicate.docx") + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/duplicate.docx") + + assert result.file_data is None + assert result.error is not None + assert "DOCUMENT_RESOURCE_LIMIT" in result.error + + +def test_oversized_document_is_rejected_before_opening_content( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +): + from EvoScientist.document_extract import MAX_DOCUMENT_BYTES + + path = tmp_path / "oversized.pdf" + path.write_bytes(b"%PDF") + original_entry = RootedWorkspace.entry + + def oversized_entry(self, virtual_path): + entry = original_entry(self, virtual_path) + return type(entry)(entry.virtual_path, entry.is_dir, MAX_DOCUMENT_BYTES + 1, entry.modified_at) + + monkeypatch.setattr(RootedWorkspace, "entry", oversized_entry) + + result = ScopedFilesystemBackend(tmp_path).read("/workspace/oversized.pdf") + + assert result.file_data is None + assert result.error is not None + assert "DOCUMENT_TOO_LARGE" in result.error + + +def test_external_document_converter_timeout_is_structured( + monkeypatch: pytest.MonkeyPatch, +): + import subprocess + + import EvoScientist.document_extract as document_extract + + def timeout(*args, **kwargs): + raise subprocess.TimeoutExpired(cmd="anydoc", timeout=60) + + monkeypatch.setattr(document_extract.subprocess, "run", timeout) + + with pytest.raises( + document_extract.DocumentExtractionError, + match="DOCUMENT_CONVERSION_TIMEOUT", + ): + document_extract.extract_document_bytes(b"%PDF-minimal", "sample.pdf") + + +@pytest.mark.parametrize("filename", ["report.pptx", "archive.zip", "results.sqlite"]) +def test_text_write_cannot_create_or_corrupt_binary_container( + tmp_path: Path, filename: str +): + backend = ScopedFilesystemBackend(tmp_path) + + created = backend.write(f"/workspace/{filename}", "extracted text") + + assert created.error is not None + assert "binary container" in created.error.lower() + assert not (tmp_path / filename).exists() + + +def test_text_edit_cannot_modify_existing_binary_container(tmp_path: Path): + source = b"PK\x03\x04original-container" + (tmp_path / "report.pptx").write_bytes(source) + backend = ScopedFilesystemBackend(tmp_path) + + edited = backend.edit( + "/workspace/report.pptx", "original", "replacement" + ) + + assert edited.error is not None + assert "binary container" in edited.error.lower() + assert (tmp_path / "report.pptx").read_bytes() == source + + +def test_uploaded_text_is_read_only_to_file_tools(tmp_path: Path): + uploads = tmp_path / "uploads" + uploads.mkdir() + source = uploads / "notes.txt" + source.write_text("original", encoding="utf-8") + backend = ScopedFilesystemBackend(tmp_path) + + written = backend.write("/workspace/uploads/new.txt", "new") + edited = backend.edit("/workspace/uploads/notes.txt", "original", "changed") + + assert written.error is not None + assert "uploads" in written.error.lower() + assert edited.error is not None + assert "uploads" in edited.error.lower() + assert not (uploads / "new.txt").exists() + assert source.read_text(encoding="utf-8") == "original" + + +def test_utf8_sample_boundary_cut_is_not_binary(tmp_path: Path): + # Regression (2026-08-22): the 8192-byte UTF-8 probe can split a multi-byte + # CJK character at the sample boundary (req_v13.md cut at 8190/8191 split a + # 3-byte char). That raised UnicodeDecodeError -> misclassified as binary + # -> read_file returned a base64 file media block -> providers without file + # input replaced it with a placeholder -> the model retried forever. + backend = ScopedFilesystemBackend(tmp_path) + # 8190 ASCII bytes + one 3-byte CJK char, so the sample cuts mid-character. + content = ("a" * 8190 + "\u6e56" + "more text").encode("utf-8") + assert len(content) > 8192 + assert content[8190:8193] == "\u6e56".encode("utf-8") + backend.upload_files([("/workspace/cjk.md", content)]) + + result = backend.read("/workspace/cjk.md") + assert result.file_data is not None + assert result.file_data["encoding"] == "utf-8" + assert result.file_data["content"].startswith("a" * 10) + + +def test_mid_sample_invalid_bytes_still_binary(): + from EvoScientist.workspace_files import _is_binary_file + + # Invalid bytes well inside the sample are genuine garbage, not a cut. + assert _is_binary_file("/workspace/bad.raw", b"ok\xffi\xffd\xefmore") + # A boundary cut (error in the last 4 bytes that decodes clean when the + # dangling suffix is dropped) is text. + cut = ("a" * 8190 + "\u6e56").encode("utf-8")[:8192] + assert not _is_binary_file("/workspace/cut.md", cut) + # Same shape but the prefix itself is invalid -> stays binary. + assert _is_binary_file("/workspace/bad.md", b"\xff" * 8192) def test_symlink_targets_and_parents_are_rejected(tmp_path: Path): diff --git a/uv.lock b/uv.lock index 116eb3c..1ac7622 100644 --- a/uv.lock +++ b/uv.lock @@ -944,11 +944,11 @@ wheels = [ [[package]] name = "evoscientist" -version = "0.2.2" source = { editable = "." } dependencies = [ { name = "deepagents", extra = ["quickjs"] }, { name = "filelock" }, + { name = "firecrawl-anydoc" }, { name = "httpx" }, { name = "langchain" }, { name = "langchain-anthropic" }, @@ -963,6 +963,7 @@ dependencies = [ { name = "lazy-loader" }, { name = "markdownify" }, { name = "nest-asyncio" }, + { name = "pillow" }, { name = "prompt-toolkit" }, { name = "psutil" }, { name = "python-dotenv" }, @@ -1056,6 +1057,7 @@ requires-dist = [ { name = "discord-py", marker = "extra == 'discord'", specifier = ">=2.3" }, { name = "faster-whisper", marker = "extra == 'stt'", specifier = ">=1.0" }, { name = "filelock", specifier = ">=3.16" }, + { name = "firecrawl-anydoc", specifier = ">=0.1.6,<0.2" }, { name = "httpx", specifier = ">=0.28" }, { name = "langchain", specifier = ">=1.3" }, { name = "langchain-anthropic", specifier = ">=1.4" }, @@ -1072,6 +1074,7 @@ requires-dist = [ { name = "lazy-loader", specifier = ">=0.5" }, { name = "markdownify", specifier = ">=1.2" }, { name = "nest-asyncio", specifier = ">=1.6" }, + { name = "pillow", specifier = ">=10.0" }, { name = "pre-commit", marker = "extra == 'dev'", specifier = ">=3.5.0" }, { name = "prompt-toolkit", specifier = ">=3.0" }, { name = "psutil", specifier = ">=6.0" }, @@ -1317,6 +1320,21 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/18/79/1b8fa1bb3568781e84c9200f951c735f3f157429f44be0495da55894d620/filetype-1.2.0-py2.py3-none-any.whl", hash = "sha256:7ce71b6880181241cf7ac8697a2f1eb6a8bd9b429f7ad6d27b8db9ba5f1c2d25", size = 19970, upload-time = "2022-11-02T17:34:01.425Z" }, ] +[[package]] +name = "firecrawl-anydoc" +version = "0.1.9" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/aa/16/e37d4f284482a4f30e0502ce439a9fd8ccf013e6ae4438b4dfa209e48750/firecrawl_anydoc-0.1.9.tar.gz", hash = "sha256:0dfc64b82b4e971143dd6f0e4f39f8152fcc38be6523663682a241d8f3301b14", size = 197914, upload-time = "2026-08-13T21:48:27.989Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/42/37/866bf6270ad8552eaf6edf3b8e1a360acf3e8764140641361ae7b440514e/firecrawl_anydoc-0.1.9-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:ebd956831822d3ea1831253406736e0e4d9b217e3471eb68c9bf2eba9302ce73", size = 3469321, upload-time = "2026-08-13T21:48:16.775Z" }, + { url = "https://files.pythonhosted.org/packages/b1/9a/8fbe0726cdf69154e1eb05c119332b1fdfd8200eeafcfd1675741ac5f03d/firecrawl_anydoc-0.1.9-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:a2372d97826e8e7e68fd28cd415d08815f6aa66dcc09c6b423ac33dd2dfa91cf", size = 3287520, upload-time = "2026-08-13T21:48:18.37Z" }, + { url = "https://files.pythonhosted.org/packages/b0/ce/86a626224c365d8de65699d6ac8f4626ac05e7e75c3ae76c82f4904a07d2/firecrawl_anydoc-0.1.9-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ba1c19bdaa97f4850daf3fb4ca2855cefa146b5621ce7b22a71d158d3b84e54e", size = 3331094, upload-time = "2026-08-13T21:48:19.961Z" }, + { url = "https://files.pythonhosted.org/packages/0e/28/00d1d48fe205ab66eda4a2a22cce44aadf586ef1495e703349b791c8de32/firecrawl_anydoc-0.1.9-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:deea7a270fccda9be29c73235e938ca32e0f703e1ae120f08bf0fdb39301c6b4", size = 3555945, upload-time = "2026-08-13T21:48:21.822Z" }, + { url = "https://files.pythonhosted.org/packages/c8/25/ac52d98c5d082ef92396b2fcb398745a3b054816c67a12d19de86c1ba535/firecrawl_anydoc-0.1.9-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:de0858decd18188b0b88545689544fb3703ec08ade37f63c2857e4414e99bdf1", size = 3510861, upload-time = "2026-08-13T21:48:23.369Z" }, + { url = "https://files.pythonhosted.org/packages/ca/d3/e80314b5746c4ca85150f3ab2a527b2167012c16e706e148733211cb88b3/firecrawl_anydoc-0.1.9-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:e4e1765ed5574fa502931f8bfaec1f87af913afa5bb432aaaf5f9c094efc7f43", size = 3797896, upload-time = "2026-08-13T21:48:25.061Z" }, + { url = "https://files.pythonhosted.org/packages/8e/d0/e7b35c2365498d5e9f1bdb4b3ded78571b6dafbaba1880538490d27b71c9/firecrawl_anydoc-0.1.9-cp310-abi3-win_amd64.whl", hash = "sha256:aa6a5ca2e10939a87c9bd21c918ef8b146ad0061fedae661d4483ef5a5bbdf60", size = 3642084, upload-time = "2026-08-13T21:48:26.702Z" }, +] + [[package]] name = "flatbuffers" version = "25.12.19" @@ -3064,6 +3082,91 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/f1/d9/7fb5aa316bc299258e68c73ba3bddbc499654a07f151cba08f6153988714/pathspec-1.1.1-py3-none-any.whl", hash = "sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189", size = 57328, upload-time = "2026-04-27T01:46:07.06Z" }, ] +[[package]] +name = "pillow" +version = "12.3.0" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/1c/3d/bb7fca845737cf9d7dbde16ed1843984665ff2e0a518f5db43e77ec540b9/pillow-12.3.0.tar.gz", hash = "sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce", size = 47025035, upload-time = "2026-07-01T11:56:38.965Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/fb/c8/0a78b0e02d7ac54bc03e5321c9220da52f0c2ea83b21f7c40e7f3169c502/pillow-12.3.0-cp311-cp311-macosx_10_10_x86_64.whl", hash = "sha256:00808c5e14ef63ac5161091d242999076604ff74b883423a11e5d7bbb38bf756", size = 5392415, upload-time = "2026-07-01T11:53:47.162Z" }, + { url = "https://files.pythonhosted.org/packages/b2/5b/a02d30018abd97ced9f5a6c63d28597694a00d066516b9c1c6de45859fc9/pillow-12.3.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:37d6d0a00072fd2948eb22bce7e1475f34569d90c87c59f7a2ec59541b77f7a6", size = 4785266, upload-time = "2026-07-01T11:53:49.079Z" }, + { url = "https://files.pythonhosted.org/packages/c8/98/766667a4be768150a202836acd9fad19c06824ca86c4286d3cf6b274964e/pillow-12.3.0-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:bcb46e2f9feff8d06323983bd83ed00c201fdcab3d74973e7072a889b3979fcd", size = 6263814, upload-time = "2026-07-01T11:53:51.32Z" }, + { url = "https://files.pythonhosted.org/packages/3b/2d/ede717bc1144f63886c21fd349bb95860b0d1a21149ff16f2bb362b612b6/pillow-12.3.0-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:23d27a3e0307ec2244cc51e7287b919aa68d097504ebe19df4e76a98a3eea5bd", size = 6934408, upload-time = "2026-07-01T11:53:53.487Z" }, + { url = "https://files.pythonhosted.org/packages/a3/48/9c58b685e69d49c31af6c8eb9012055fab7e665785165c84796e2c73ce72/pillow-12.3.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:4f883547d4b7f0495ebe7056b0cc2aea76094e7a4abc8e933540f3271df27d9c", size = 6337160, upload-time = "2026-07-01T11:53:55.457Z" }, + { url = "https://files.pythonhosted.org/packages/ff/fa/dc2a5c0ba6df93f67c31d34b808b7ce440b40cdbf96f0b81cde1d1e6fa93/pillow-12.3.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:236ff70b9312fb68943c703aa842ca6a758abfa45ac187a5e7c1452e96ef72b5", size = 7045172, upload-time = "2026-07-01T11:53:57.736Z" }, + { url = "https://files.pythonhosted.org/packages/86/a5/444817a4d4c4c2417df00513086ca196f388d8f9ef40c2e4ccd1ad1af54b/pillow-12.3.0-cp311-cp311-win32.whl", hash = "sha256:10e41f0fbf1eec8cfd234b8fe17a4caac7c9d0db4c204d3c173a8f9f6ef3232b", size = 6472232, upload-time = "2026-07-01T11:53:59.767Z" }, + { url = "https://files.pythonhosted.org/packages/63/c6/4bad1b18d132a50b27e1365e1ab163616f7a5bb56d330f66f9d1d9d4f9d4/pillow-12.3.0-cp311-cp311-win_amd64.whl", hash = "sha256:8e95e1385e4998ae9694eeaa4730ba5457ff61185b3a55e2e7bea0880aef452a", size = 7233653, upload-time = "2026-07-01T11:54:02.066Z" }, + { url = "https://files.pythonhosted.org/packages/fd/16/00f91ab7760dc842f5aad55217e80fc4a7067a0604535249bc8a2d6d9870/pillow-12.3.0-cp311-cp311-win_arm64.whl", hash = "sha256:ebaea975e03d3141d9d3a507df75c9b3ec90fa9d2ffd07567b3a978d9d790b26", size = 2568195, upload-time = "2026-07-01T11:54:04.622Z" }, + { url = "https://files.pythonhosted.org/packages/37/bf/fb3ebff8ddcb76aac5a01389251bbbb9519922a9b520d8247c1ca864a25d/pillow-12.3.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965", size = 5345969, upload-time = "2026-07-01T11:54:06.397Z" }, + { url = "https://files.pythonhosted.org/packages/d8/66/9a386a92561f402389a4fc70c18838bf6d35eb5eb5c6850b4b2dc64f5048/pillow-12.3.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7", size = 4780323, upload-time = "2026-07-01T11:54:09.351Z" }, + { url = "https://files.pythonhosted.org/packages/25/27/ac8f99618ffd3dde21db0f4d4b1d2ab00c0880595bfd17df103f7f39fd0c/pillow-12.3.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9", size = 6266838, upload-time = "2026-07-01T11:54:11.71Z" }, + { url = "https://files.pythonhosted.org/packages/84/21/a35af28dcc61f37ed850a2d64c65c701321dfbf25085e469d5559360cbbf/pillow-12.3.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91", size = 6940830, upload-time = "2026-07-01T11:54:13.732Z" }, + { url = "https://files.pythonhosted.org/packages/eb/51/8b08617af3ad95e33ce6d7dd2c99ed6c8298f7fb131636303956be022e25/pillow-12.3.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c", size = 6344383, upload-time = "2026-07-01T11:54:15.756Z" }, + { url = "https://files.pythonhosted.org/packages/1d/72/cf78ac9780bb93c28328f408973845a309d4d145041665f734572ced1b52/pillow-12.3.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df", size = 7052934, upload-time = "2026-07-01T11:54:17.721Z" }, + { url = "https://files.pythonhosted.org/packages/20/20/25e0f4dc178a6bc0696793720055519a0de89e7661dae886992decbd2f81/pillow-12.3.0-cp312-cp312-win32.whl", hash = "sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f", size = 6472684, upload-time = "2026-07-01T11:54:19.839Z" }, + { url = "https://files.pythonhosted.org/packages/45/89/da2f7971a317f83d807fdd4065c0af40208e59e692cc43d315a71a0e96d1/pillow-12.3.0-cp312-cp312-win_amd64.whl", hash = "sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09", size = 7227137, upload-time = "2026-07-01T11:54:22.025Z" }, + { url = "https://files.pythonhosted.org/packages/de/47/4845a0a6c0dbf1db8456bd9fc791f13c5ced7ced20606d08a0aacfd25b49/pillow-12.3.0-cp312-cp312-win_arm64.whl", hash = "sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510", size = 2568267, upload-time = "2026-07-01T11:54:24.051Z" }, + { url = "https://files.pythonhosted.org/packages/9d/ac/31fb64e1e7efb5a4b50cd3d92049ba89ac6e4d8d3bb6a74e15048ca3353e/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89", size = 4161684, upload-time = "2026-07-01T11:54:25.934Z" }, + { url = "https://files.pythonhosted.org/packages/87/b4/9805e23d2b4d77842b468513841fda254ee42f0289d25088340e4ff46e2d/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace", size = 4255487, upload-time = "2026-07-01T11:54:27.935Z" }, + { url = "https://files.pythonhosted.org/packages/df/39/ecf519435a200c693fe053a6ee4d835b41cf963a4dfc2551c4e637cb2a71/pillow-12.3.0-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec", size = 3696433, upload-time = "2026-07-01T11:54:29.813Z" }, + { url = "https://files.pythonhosted.org/packages/42/92/2fc3ffad878ae8dd5469ec1bc8eb83b71f48e13efdf68f02709003982a32/pillow-12.3.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66", size = 5345889, upload-time = "2026-07-01T11:54:31.97Z" }, + { url = "https://files.pythonhosted.org/packages/10/76/8803c13605b763d33d156c4678fc77f8443389c0c51c8aef707bb02015f4/pillow-12.3.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35", size = 4780109, upload-time = "2026-07-01T11:54:34.026Z" }, + { url = "https://files.pythonhosted.org/packages/1f/01/e18aff37cb0b4aac47ac90f016d347a49aca667ef97f190b06ac2aabc928/pillow-12.3.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65", size = 6263736, upload-time = "2026-07-01T11:54:36.131Z" }, + { url = "https://files.pythonhosted.org/packages/f7/62/de5bdd77d935331f4f802edc11e4d82950f642caad6cb2f949837b8560e2/pillow-12.3.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3", size = 6937129, upload-time = "2026-07-01T11:54:38.216Z" }, + { url = "https://files.pythonhosted.org/packages/70/4d/105627a13300c5e0df1d174230b32fd1273062c96f7745fd552b945d1e1d/pillow-12.3.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a", size = 6339562, upload-time = "2026-07-01T11:54:40.354Z" }, + { url = "https://files.pythonhosted.org/packages/6b/1d/f13de01a553988ab895ba1c722e06cf3144d4f57656fd5b81b6d881f1179/pillow-12.3.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e", size = 7049439, upload-time = "2026-07-01T11:54:42.489Z" }, + { url = "https://files.pythonhosted.org/packages/c9/f9/066794cca041b969964f779ee5fa66a9498bbf34248ac39c5d7954e4198f/pillow-12.3.0-cp313-cp313-win32.whl", hash = "sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f", size = 6473287, upload-time = "2026-07-01T11:54:44.9Z" }, + { url = "https://files.pythonhosted.org/packages/a6/9b/7a58e61d62be561da3a356fe2384d4059a6345fc130e23ef1c36a5b81d24/pillow-12.3.0-cp313-cp313-win_amd64.whl", hash = "sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8", size = 7239691, upload-time = "2026-07-01T11:54:47.141Z" }, + { url = "https://files.pythonhosted.org/packages/aa/b0/c4ed4f0ef8f8fa5ee8351537db6650bb8189f7e118842978dd6589065692/pillow-12.3.0-cp313-cp313-win_arm64.whl", hash = "sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b", size = 2568185, upload-time = "2026-07-01T11:54:49.137Z" }, + { url = "https://files.pythonhosted.org/packages/dc/01/001f65b68192f0228cc1dbbc8d2530ab5d58b61037ba0587f946fea607cd/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330", size = 4161736, upload-time = "2026-07-01T11:54:51.156Z" }, + { url = "https://files.pythonhosted.org/packages/1a/d2/0219746d0fd16fc8a84498e79452375be3797d3ce4044596ce565164b84f/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217", size = 4255435, upload-time = "2026-07-01T11:54:53.414Z" }, + { url = "https://files.pythonhosted.org/packages/c8/02/8d0bc62ef0302318c46ff2a512822d2610e81c7aa46c9b3abe6cbaca5ad0/pillow-12.3.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930", size = 3696262, upload-time = "2026-07-01T11:54:55.739Z" }, + { url = "https://files.pythonhosted.org/packages/85/e2/73c77d218410b14f5f2d565e8a998d5317b7b9c75368d29985139f7a46f0/pillow-12.3.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8", size = 5350344, upload-time = "2026-07-01T11:54:57.657Z" }, + { url = "https://files.pythonhosted.org/packages/c7/da/32c752228ae345f489e3a42499d817b6c3996da7e8a3bc7a04fc806b243b/pillow-12.3.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0", size = 4780131, upload-time = "2026-07-01T11:54:59.713Z" }, + { url = "https://files.pythonhosted.org/packages/b1/9d/8b2c807dbef61a5197c047afe99823787eb66f63daf9fb2432f91d6f0462/pillow-12.3.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321", size = 6263757, upload-time = "2026-07-01T11:55:01.778Z" }, + { url = "https://files.pythonhosted.org/packages/5c/44/c85361f65dbe00eea8576ee467c768d25129989efb76e94f205e9ca9bb46/pillow-12.3.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b", size = 6936962, upload-time = "2026-07-01T11:55:03.93Z" }, + { url = "https://files.pythonhosted.org/packages/18/7e/e483414b35800b86b6f08dbbc7803fb5cd52c4d6f897f47d53ea2c7e6f65/pillow-12.3.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198", size = 6339171, upload-time = "2026-07-01T11:55:05.989Z" }, + { url = "https://files.pythonhosted.org/packages/f0/f4/68c491844841ede6bed70189546b3ee9731cf9f2cbad396faff5e1ccba45/pillow-12.3.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130", size = 7048116, upload-time = "2026-07-01T11:55:08.131Z" }, + { url = "https://files.pythonhosted.org/packages/a3/34/77f3f793fed8efc7d243f21b33c5a3f0d1c97ee70346d3db855587e155ff/pillow-12.3.0-cp314-cp314-win32.whl", hash = "sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a", size = 6467209, upload-time = "2026-07-01T11:55:10.408Z" }, + { url = "https://files.pythonhosted.org/packages/f1/e0/492879f69d94f91f60fc8cd05ba03650e9520afebb2fb7aa12777d7c7f38/pillow-12.3.0-cp314-cp314-win_amd64.whl", hash = "sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d", size = 7237707, upload-time = "2026-07-01T11:55:12.745Z" }, + { url = "https://files.pythonhosted.org/packages/c9/ac/6b11f2875f1c2ac040d84e1bbf9cf22a88038f901ca1037898b280b38365/pillow-12.3.0-cp314-cp314-win_arm64.whl", hash = "sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838", size = 2565995, upload-time = "2026-07-01T11:55:14.736Z" }, + { url = "https://files.pythonhosted.org/packages/52/69/c2208e56af9bfc1913afb24020297a691eb1d4ef688474c8a04913f65e04/pillow-12.3.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e", size = 5352503, upload-time = "2026-07-01T11:55:17.076Z" }, + { url = "https://files.pythonhosted.org/packages/07/70/e5686d753e898a45d778ff1718dba8516ead6ab6b95d85fc8c4b70650cf2/pillow-12.3.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17", size = 4782956, upload-time = "2026-07-01T11:55:19.448Z" }, + { url = "https://files.pythonhosted.org/packages/d5/37/25c6692f06927ee973ff18c8d9ee98ad0b4d84ee67a09610c2dd1447958e/pillow-12.3.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385", size = 6322855, upload-time = "2026-07-01T11:55:21.613Z" }, + { url = "https://files.pythonhosted.org/packages/cc/91/420637fcb8f1bc11029e403b4538e6694744428d8246118e45719f944556/pillow-12.3.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c", size = 6989642, upload-time = "2026-07-01T11:55:24.006Z" }, + { url = "https://files.pythonhosted.org/packages/10/08/b94d7811281ccf0d143a1cf768d1c49e1e54af63e7b708ab2ee3eb87face/pillow-12.3.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d", size = 6391281, upload-time = "2026-07-01T11:55:26.252Z" }, + { url = "https://files.pythonhosted.org/packages/d2/87/24233f785f55474dc02ce3e739c5528a77e3a862e9333d1dd7a25cc31f70/pillow-12.3.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931", size = 7096716, upload-time = "2026-07-01T11:55:28.318Z" }, + { url = "https://files.pythonhosted.org/packages/23/26/fcb2f6e37175b04f53570b59937867e2b80ee1685e744023153028fc14f9/pillow-12.3.0-cp314-cp314t-win32.whl", hash = "sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7", size = 6474125, upload-time = "2026-07-01T11:55:30.956Z" }, + { url = "https://files.pythonhosted.org/packages/90/de/3634abee5f1c9e13c56787b7d5517b0ba8d6de51700b95578cf338349c9f/pillow-12.3.0-cp314-cp314t-win_amd64.whl", hash = "sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c", size = 7242939, upload-time = "2026-07-01T11:55:34.044Z" }, + { url = "https://files.pythonhosted.org/packages/ce/2a/fd13f8eb24de5714a6eb444a3d67e2842c6c576e159a43793adf23051351/pillow-12.3.0-cp314-cp314t-win_arm64.whl", hash = "sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45", size = 2567506, upload-time = "2026-07-01T11:55:35.988Z" }, + { url = "https://files.pythonhosted.org/packages/5d/dc/8fdce34ec725a33c81c6ba122b904d6b9024e50ea9ac7bede62fab54506c/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphoneos.whl", hash = "sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139", size = 4162063, upload-time = "2026-07-01T11:55:37.941Z" }, + { url = "https://files.pythonhosted.org/packages/76/66/2044b9a63d3b84ff048228dfcb7cd9bf0df983e8470971bf7d4c57b693de/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402", size = 4255549, upload-time = "2026-07-01T11:55:40.022Z" }, + { url = "https://files.pythonhosted.org/packages/52/7e/1f67e6f4ece6b582ee4b539decbcc9f848dc245a93ed8cd7338bafef72f1/pillow-12.3.0-cp315-cp315-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c", size = 3696331, upload-time = "2026-07-01T11:55:41.98Z" }, + { url = "https://files.pythonhosted.org/packages/12/40/d306fc2c8e4d45d7f175c77edca7063be7b86fe7fe6e68f4353bf71d808c/pillow-12.3.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f", size = 5350370, upload-time = "2026-07-01T11:55:44.028Z" }, + { url = "https://files.pythonhosted.org/packages/dd/44/668fb1437e8ce420f62d6106eb66e44a5971602a4d794615bdf79315d82d/pillow-12.3.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701", size = 4780147, upload-time = "2026-07-01T11:55:46.073Z" }, + { url = "https://files.pythonhosted.org/packages/0c/08/93fa2e70e30a2d81547e481b6ee2bb9522117221fb1e0ce4b5df70967677/pillow-12.3.0-cp315-cp315-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace", size = 6273659, upload-time = "2026-07-01T11:55:48.264Z" }, + { url = "https://files.pythonhosted.org/packages/f8/6d/043e96ff814fc31a33077e4cba86082167db520c93632afdf2042febbb0c/pillow-12.3.0-cp315-cp315-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4", size = 6947439, upload-time = "2026-07-01T11:55:50.503Z" }, + { url = "https://files.pythonhosted.org/packages/af/92/ba71d2ee2ac0edf3fa33bd9d5ee9ee080da70b1766f3ca3934f9938ddac9/pillow-12.3.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39", size = 6353577, upload-time = "2026-07-01T11:55:52.697Z" }, + { url = "https://files.pythonhosted.org/packages/0f/ce/e63064e2122923ff687c8ad792d0d736a7b3920a56a46982e81a7fdd25d6/pillow-12.3.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71", size = 7060394, upload-time = "2026-07-01T11:55:55.149Z" }, + { url = "https://files.pythonhosted.org/packages/54/76/a09cc3ccc8d773a7283d34c38bec1708f9e3cc932093cbc4c5e71ac4060b/pillow-12.3.0-cp315-cp315-win32.whl", hash = "sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827", size = 6467375, upload-time = "2026-07-01T11:55:57.769Z" }, + { url = "https://files.pythonhosted.org/packages/3e/03/1846c49ba3b1d5550392a4bbd06d6fb4578e1cd91a803198b5c90f5f7d53/pillow-12.3.0-cp315-cp315-win_amd64.whl", hash = "sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5", size = 7237048, upload-time = "2026-07-01T11:55:59.975Z" }, + { url = "https://files.pythonhosted.org/packages/fb/bb/89f35dcc79610423f9f195504d7def7f0d1416a711541b42867e25fe3412/pillow-12.3.0-cp315-cp315-win_arm64.whl", hash = "sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658", size = 2566006, upload-time = "2026-07-01T11:56:02.143Z" }, + { url = "https://files.pythonhosted.org/packages/30/88/707027ba09942dfa2c28759b5c222d769290a41c6d20ea60ec250801941f/pillow-12.3.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf", size = 5352509, upload-time = "2026-07-01T11:56:04.2Z" }, + { url = "https://files.pythonhosted.org/packages/b0/6d/00352fa25332c2569cd387851f568cc5a4b75a9adbfb37ac4fbce4c02eec/pillow-12.3.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64", size = 4783167, upload-time = "2026-07-01T11:56:06.631Z" }, + { url = "https://files.pythonhosted.org/packages/13/4f/9e049dfa21af7c22427275720e2490267ba8138120add5c4c574deb69782/pillow-12.3.0-cp315-cp315t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e", size = 6329237, upload-time = "2026-07-01T11:56:08.868Z" }, + { url = "https://files.pythonhosted.org/packages/36/16/cf6eeaae8d0fce8dd390a33437cf68c5d5bd73834a2bc6e2f14efda0ab45/pillow-12.3.0-cp315-cp315t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777", size = 6997047, upload-time = "2026-07-01T11:56:11.379Z" }, + { url = "https://files.pythonhosted.org/packages/1e/69/dbf769bdd55f48bf5733cac28edc6364ffaa072ec9ba336266e4fe66be55/pillow-12.3.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1", size = 6400440, upload-time = "2026-07-01T11:56:13.908Z" }, + { url = "https://files.pythonhosted.org/packages/a0/e1/ffc9cfc2eea0d178da8018e18e959301ad9d6bc9f3edb7181e748a474b97/pillow-12.3.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9", size = 7105895, upload-time = "2026-07-01T11:56:16.575Z" }, + { url = "https://files.pythonhosted.org/packages/18/f0/a5595c1e8c3ae44b9828cb2f0fa8155e5095ef04d6327b8f61cf44a3df85/pillow-12.3.0-cp315-cp315t-win32.whl", hash = "sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8", size = 6474384, upload-time = "2026-07-01T11:56:18.855Z" }, + { url = "https://files.pythonhosted.org/packages/e4/04/62bcd9f844984c5938d3b05264a61d797a29d3e0812341a8204af70bbdee/pillow-12.3.0-cp315-cp315t-win_amd64.whl", hash = "sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418", size = 7243537, upload-time = "2026-07-01T11:56:21.214Z" }, + { url = "https://files.pythonhosted.org/packages/3d/68/1f3066acedf37673694a7141381d8f811ae97f30d34413d236abe7d489f1/pillow-12.3.0-cp315-cp315t-win_arm64.whl", hash = "sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59", size = 2567491, upload-time = "2026-07-01T11:56:23.506Z" }, + { url = "https://files.pythonhosted.org/packages/75/18/2e8b40223153ccbc60df07f9e8928dc0c76202aa4e55ae9f53962b6510d6/pillow-12.3.0-pp311-pypy311_pp73-macosx_10_15_x86_64.whl", hash = "sha256:b3c777e849237620b022f7f297dd67705f9f5cf1685f09f02e46f93e92725468", size = 5302510, upload-time = "2026-07-01T11:56:25.736Z" }, + { url = "https://files.pythonhosted.org/packages/46/3e/51fabf59d5ab801ceab709453d3ab6b180083496579549de4c45ced6528a/pillow-12.3.0-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:b343699e8308bdc51978310e1c959c584e7869cc8c40780058c87da7781a1e94", size = 4736058, upload-time = "2026-07-01T11:56:28.041Z" }, + { url = "https://files.pythonhosted.org/packages/bf/20/22fe9384b7949e25fb1293bcfc84fb82590ff4ea6b37c95b24d26d793d86/pillow-12.3.0-pp311-pypy311_pp73-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:fbd139c8447d25dd750ab79ee274cc5e1fe80fc56340ab10b18a195e1b6eca3e", size = 5237776, upload-time = "2026-07-01T11:56:30.263Z" }, + { url = "https://files.pythonhosted.org/packages/08/14/f6ba68107680ffa74b39985f3f30884e41318fbc4250caa423c79b4788bb/pillow-12.3.0-pp311-pypy311_pp73-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e7e480451b9fa137494bccd3a7d69adbe8ac65a87d97be61e11f1b1050a5bac3", size = 5860358, upload-time = "2026-07-01T11:56:32.68Z" }, + { url = "https://files.pythonhosted.org/packages/36/54/0169bc772ec491108b62f644f8ecf1fe5d8ae5ebafde2ee2142210166903/pillow-12.3.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:04f01d28a6aaff387bf842a13be313df23ba0597a44f1a976c9feb3c6ff4711a", size = 7231786, upload-time = "2026-07-01T11:56:35.046Z" }, +] + [[package]] name = "platformdirs" version = "4.9.6"