feat: prepare EvoScientist 0.3.0
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled

Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
This commit is contained in:
m4
2026-09-03 06:55:56 +08:00
parent ce0bec1f91
commit c683f6e739
35 changed files with 2806 additions and 67 deletions
+68 -7
View File
@@ -555,9 +555,10 @@ def _build_base_kwargs(
cfg = cfg if cfg is not None else _ensure_config()
tool_registry = {"think_tool": think_tool}
base_tools = [think_tool, skill_manager]
if os.environ.get("TAVILY_API_KEY"):
tool_registry["tavily_search"] = tavily_search
base_tools = [think_tool, skill_manager]
base_tools.append(tavily_search)
subs = load_subagents(
SUBAGENTS_CONFIG,
@@ -617,15 +618,54 @@ def load_mcp_and_build_kwargs(
)
tool_registry = {"think_tool": think_tool}
base_tools = [think_tool, skill_manager]
if os.environ.get("TAVILY_API_KEY"):
tool_registry["tavily_search"] = tavily_search
base_tools = [think_tool, skill_manager]
base_tools.append(tavily_search)
# Fresh tool registry — start from base tools + MCP tools
# DeepAgents installs these outside ``base_tools`` through middleware.
# MCP tools must never shadow them inside any one agent namespace.
middleware_tool_names = {
"ls",
"read_file",
"write_file",
"edit_file",
"glob",
"grep",
"execute",
"write_todos",
"task",
"start_async_task",
"check_async_task",
"update_async_task",
"cancel_async_task",
"list_async_tasks",
}
# Fresh tool registry — start from built-ins, then add one representative
# MCP implementation for YAML name resolution. A tool may be exposed to
# several agents, but no agent may contain duplicate names and no MCP tool
# may override a built-in implementation.
registry = dict(tool_registry)
for tools in mcp_by_agent.values():
builtin_names = {
*(str(getattr(tool, "name", "")) for tool in base_tools),
*middleware_tool_names,
}
for agent_name, tools in mcp_by_agent.items():
seen_for_agent: set[str] = set()
for t in tools:
registry[t.name] = t
tool_name = str(t.name)
if tool_name in builtin_names or tool_name in seen_for_agent:
from .llm.contracts import EvoRuntimeError
raise EvoRuntimeError(
"TOOL_REGISTRY_CONFLICT",
details=(
{"agent_name": str(agent_name), "tool_name": tool_name},
),
)
seen_for_agent.add(tool_name)
registry.setdefault(tool_name, t)
mcp_main = mcp_by_agent.pop("main", [])
@@ -639,10 +679,31 @@ def load_mcp_and_build_kwargs(
subs, workspace_dir=workspace_dir, cfg=cfg, chat_model=chat_model
)
# Inject MCP tools into subagents by name
# Inject MCP tools into subagents by name. YAML-resolved tools already
# belong to that agent namespace, so a second tool with the same name is a
# configuration conflict rather than an item to append silently.
for sa in subs:
if sa_tools := mcp_by_agent.get(sa["name"], []):
sa.setdefault("tools", []).extend(sa_tools)
target_tools = sa.setdefault("tools", [])
existing_names = {
str(getattr(tool, "name", tool)) for tool in target_tools
}
for tool in sa_tools:
tool_name = str(tool.name)
if tool_name in existing_names:
from .llm.contracts import EvoRuntimeError
raise EvoRuntimeError(
"TOOL_REGISTRY_CONFLICT",
details=(
{
"agent_name": str(sa["name"]),
"tool_name": tool_name,
},
),
)
existing_names.add(tool_name)
target_tools.append(tool)
# Swap selected sub-agents to AsyncSubAgent (must happen AFTER MCP injection
# since async sub-agents are remote graphs that load their own tools).
+3 -2
View File
@@ -9,7 +9,7 @@ from __future__ import annotations
from importlib import import_module
__version__ = "0.2.2"
from ._version import __version__
_EXPORTS: dict[str, tuple[str, str]] = {
# Agent graph (lazy to avoid expensive initialization at import time)
@@ -73,4 +73,5 @@ def __dir__() -> list[str]:
return sorted(set(globals()) | set(_EXPORTS))
__all__ = list(_EXPORTS)
__all__ = ["__version__"]
__all__.extend(_EXPORTS)
+3
View File
@@ -0,0 +1,3 @@
"""Package version shared by builds and runtime."""
__version__ = "0.3.0"
+466
View File
@@ -0,0 +1,466 @@
"""Bounded document extraction and non-text file policy for workspaces."""
from __future__ import annotations
import os
import subprocess
import sys
import tempfile
import zipfile
from pathlib import Path
from xml.etree import ElementTree as ET
MAX_DOCUMENT_BYTES = 50 * 1024 * 1024
MAX_DOCUMENT_RESULT_CHARS = 50_000
MAX_CONVERTED_DOCUMENT_BYTES = 10 * 1024 * 1024
DOCUMENT_CONVERSION_TIMEOUT_SECONDS = 60
MAX_IMAGE_BYTES = 25 * 1024 * 1024
MAX_IMAGE_EDGE = 2048
MAX_IMAGE_PIXELS = 40_000_000
MAX_OOXML_MEMBERS = 10_000
MAX_OOXML_MEMBER_BYTES = 50 * 1024 * 1024
MAX_OOXML_EXPANDED_BYTES = 200 * 1024 * 1024
MAX_OOXML_COMPRESSION_RATIO = 100
IMAGE_EXTENSIONS = frozenset(
{".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".png", ".tif", ".tiff", ".webp"}
)
DOCUMENT_EXTENSIONS = frozenset(
{
".doc",
".docm",
".docx",
".epub",
".odp",
".ods",
".odt",
".pdf",
".pot",
".pps",
".ppsm",
".ppsx",
".ppt",
".pptm",
".pptx",
".rtf",
".xls",
".xlsb",
".xlsm",
".xlsx",
}
)
ARCHIVE_EXTENSIONS = frozenset(
{".7z", ".bz2", ".gz", ".rar", ".tar", ".tgz", ".xz", ".zip"}
)
DATABASE_EXTENSIONS = frozenset({".db", ".sqlite", ".sqlite3"})
EXECUTABLE_EXTENSIONS = frozenset(
{".app", ".deb", ".dll", ".dylib", ".elf", ".exe", ".msi", ".rpm", ".so"}
)
DATASET_EXTENSIONS = frozenset(
{".arrow", ".feather", ".h5", ".hdf5", ".npy", ".npz", ".parquet"}
)
MEDIA_EXTENSIONS = frozenset(
{".aac", ".avi", ".flac", ".m4a", ".mkv", ".mov", ".mp3", ".mp4", ".ogg", ".wav", ".webm"}
)
_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
_A = "http://schemas.openxmlformats.org/drawingml/2006/main"
_S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
_R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
class DocumentExtractionError(RuntimeError):
"""A supported document could not be converted to bounded text."""
def classify_file(path: str, head: bytes = b"") -> str:
"""Classify a workspace file into one policy category."""
extension = Path(path).suffix.lower()
if extension in IMAGE_EXTENSIONS:
return "image"
if extension in DOCUMENT_EXTENSIONS:
return "document"
if extension in ARCHIVE_EXTENSIONS:
return "archive"
if extension in DATABASE_EXTENSIONS or head.startswith(b"SQLite format 3\x00"):
return "database"
if extension in EXECUTABLE_EXTENSIONS or head.startswith((b"MZ", b"\x7fELF")):
return "executable"
if extension in DATASET_EXTENSIONS:
return "dataset"
if extension in MEDIA_EXTENSIONS:
return "media"
return "unknown"
def binary_processing_guidance(path: str, kind: str, size_bytes: int) -> str:
"""Return bounded, actionable JSON-like guidance for Agent-side programming."""
import json
common = {
"code": "BINARY_PROCESSING_REQUIRED"
if kind != "binary"
else "UNSUPPORTED_BINARY_FILE",
"path": path,
"kind": kind,
"size_bytes": size_bytes,
}
if kind == "archive":
common.update(
action=(
"Use execute with Python to list and validate archive members before "
"selective extraction; never use extractall."
),
constraints={
"list_before_extract": True,
"max_members": 2000,
"max_total_uncompressed_bytes": 500 * 1024 * 1024,
"max_member_bytes": 100 * 1024 * 1024,
"max_compression_ratio": 100,
"reject_absolute_or_parent_paths": True,
"do_not_execute_members": True,
},
)
elif kind == "database":
common.update(
action=(
"Use execute with Python sqlite3 in read-only mode: "
"file:<path>?mode=ro&immutable=1; set PRAGMA query_only=ON; "
"inspect schema, run bounded SELECT queries with LIMIT, and write "
"large results under artifacts/."
),
constraints={
"read_only": True,
"mode": "mode=ro",
"query_only": True,
"max_rows": 1000,
"forbid_attach_database": True,
"forbid_load_extension": True,
},
)
elif kind == "executable":
common.update(
action=(
"Use execute only for bounded static metadata inspection (hash, file "
"headers, signature, imports, strings); this file must not be executed."
),
constraints={"must_not_be_executed": True, "static_analysis_only": True},
)
elif kind == "dataset":
common.update(
action=(
"Use execute with the appropriate library to inspect schema, dimensions, "
"statistics, and a bounded sample; do not serialize the whole dataset."
)
)
elif kind == "media":
common.update(
action=(
"Use execute with ffprobe/ffmpeg or an available transcription workflow "
"to inspect metadata and selected ranges; do not inline the complete file."
)
)
else:
common.update(
kind="binary",
action=(
"This is an unsupported binary file. Use execute only for bounded static "
"inspection; do not execute it or inline its bytes."
),
)
return json.dumps(common, ensure_ascii=False, sort_keys=True)
def prepare_image_bytes(data: bytes, path: str) -> bytes:
"""Validate and downsample an image before it becomes a model media block."""
import io
if len(data) > MAX_IMAGE_BYTES:
raise DocumentExtractionError(
f"IMAGE_TOO_LARGE: {len(data)} bytes exceeds {MAX_IMAGE_BYTES}"
)
try:
from PIL import Image
image = Image.open(io.BytesIO(data))
width, height = image.size
if width * height > MAX_IMAGE_PIXELS:
raise DocumentExtractionError(
f"IMAGE_PIXEL_BUDGET_EXCEEDED: {width}x{height} exceeds "
f"{MAX_IMAGE_PIXELS} pixels"
)
image.load()
except DocumentExtractionError:
raise
except Exception as exc:
raise DocumentExtractionError(
f"IMAGE_PROCESSING_FAILED: {path}: {type(exc).__name__}: {exc}"
) from exc
frame_count = int(getattr(image, "n_frames", 1) or 1)
if frame_count > 1:
image.seek(0)
work = image.convert("RGBA" if image.mode in {"RGBA", "LA"} else "RGB")
else:
work = image.copy()
if max(work.size) <= MAX_IMAGE_EDGE and frame_count == 1:
return data
work.thumbnail((MAX_IMAGE_EDGE, MAX_IMAGE_EDGE), Image.Resampling.LANCZOS)
has_alpha = work.mode in {"RGBA", "LA"} or (
work.mode == "P" and "transparency" in work.info
)
output = io.BytesIO()
if has_alpha:
if work.mode == "P":
work = work.convert("RGBA")
work.save(output, "PNG", optimize=True)
else:
if work.mode != "RGB":
work = work.convert("RGB")
work.save(output, "JPEG", quality=85, optimize=True)
return output.getvalue()
def extract_document_bytes(data: bytes, path: str) -> str:
"""Extract readable text from a supported document without exposing bytes."""
if len(data) > MAX_DOCUMENT_BYTES:
raise DocumentExtractionError(
f"DOCUMENT_TOO_LARGE: {len(data)} bytes exceeds {MAX_DOCUMENT_BYTES}"
)
extension = Path(path).suffix.lower()
try:
if extension == ".docx":
return _extract_docx(data)
if extension == ".pptx":
return _extract_pptx(data)
if extension == ".xlsx":
return _extract_xlsx(data)
return _extract_anydoc(data, extension)
except DocumentExtractionError:
raise
except Exception as exc:
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
) from exc
def paginate_document_text(text: str, *, offset: int, limit: int) -> str:
"""Apply line and character budgets to extracted document text."""
lines = text.splitlines(keepends=True)
if not lines:
return "(document contains no extractable text)"
if offset >= len(lines):
raise DocumentExtractionError(
f"Line offset {offset} exceeds extracted document length ({len(lines)} lines)"
)
selected = "".join(lines[offset : offset + limit])
if len(selected) <= MAX_DOCUMENT_RESULT_CHARS:
return selected
trimmed = selected[:MAX_DOCUMENT_RESULT_CHARS]
boundary = trimmed.rfind("\n")
if boundary > 0:
trimmed = trimmed[: boundary + 1]
consumed = max(1, len(trimmed.splitlines()))
return (
trimmed
+ f"\n[DOCUMENT_OUTPUT_TRUNCATED: use offset={offset + consumed} to continue; "
+ f"single-read limit is {MAX_DOCUMENT_RESULT_CHARS} characters]\n"
)
def _validated_ooxml_archive(data: bytes) -> zipfile.ZipFile:
try:
archive = zipfile.ZipFile(_bytes_path(data))
members = archive.infolist()
except zipfile.BadZipFile as exc:
raise DocumentExtractionError(
"DOCUMENT_EXTRACTION_FAILED: invalid OOXML container"
) from exc
expanded = 0
names: set[str] = set()
if len(members) > MAX_OOXML_MEMBERS:
archive.close()
raise DocumentExtractionError(
f"DOCUMENT_RESOURCE_LIMIT: OOXML has {len(members)} members; "
f"limit is {MAX_OOXML_MEMBERS}"
)
for member in members:
normalized = member.filename.replace("\\", "/")
parts = tuple(part for part in normalized.split("/") if part)
if (
normalized.startswith("/")
or ".." in parts
or member.filename in names
or bool(member.flag_bits & 0x1)
):
archive.close()
raise DocumentExtractionError(
"DOCUMENT_RESOURCE_LIMIT: OOXML contains an unsafe, duplicate, "
"or encrypted member"
)
names.add(member.filename)
expanded += member.file_size
ratio = member.file_size / max(member.compress_size, 1)
if (
member.file_size > MAX_OOXML_MEMBER_BYTES
or expanded > MAX_OOXML_EXPANDED_BYTES
or ratio > MAX_OOXML_COMPRESSION_RATIO
):
archive.close()
raise DocumentExtractionError(
"DOCUMENT_RESOURCE_LIMIT: OOXML member expansion exceeds safety limits"
)
return archive
def _zip_xml(data: bytes, member: str) -> ET.Element:
try:
with _validated_ooxml_archive(data) as archive:
raw = archive.read(member)
except KeyError as exc:
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: missing {member}"
) from exc
return ET.fromstring(raw)
def _bytes_path(data: bytes):
import io
return io.BytesIO(data)
def _ooxml_part_number(name: str) -> int:
stem = Path(name).stem
digits = "".join(character for character in stem if character.isdigit())
return int(digits) if digits else 0
def _extract_docx(data: bytes) -> str:
root = _zip_xml(data, "word/document.xml")
paragraphs: list[str] = []
for paragraph in root.iter(f"{{{_W}}}p"):
text = "".join(node.text or "" for node in paragraph.iter(f"{{{_W}}}t"))
if text:
paragraphs.append(text)
if not paragraphs:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: DOCX has no text")
return "\n".join(paragraphs) + "\n"
def _extract_pptx(data: bytes) -> str:
try:
with _validated_ooxml_archive(data) as archive:
names = sorted(
name
for name in archive.namelist()
if name.startswith("ppt/slides/slide") and name.endswith(".xml")
)
slides: list[str] = []
for index, name in enumerate(sorted(names, key=_ooxml_part_number), 1):
root = ET.fromstring(archive.read(name))
texts = [node.text or "" for node in root.iter(f"{{{_A}}}t")]
slides.append(f"## Slide {index}\n" + "\n".join(t for t in texts if t))
except zipfile.BadZipFile as exc:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid PPTX") from exc
if not slides:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: PPTX has no slides")
return "\n\n".join(slides) + "\n"
def _extract_xlsx(data: bytes) -> str:
try:
with _validated_ooxml_archive(data) as archive:
shared: list[str] = []
if "xl/sharedStrings.xml" in archive.namelist():
root = ET.fromstring(archive.read("xl/sharedStrings.xml"))
shared = [
"".join(node.text or "" for node in item.iter(f"{{{_S}}}t"))
for item in root.iter(f"{{{_S}}}si")
]
sheets = sorted(
name
for name in archive.namelist()
if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
)
output: list[str] = []
for index, name in enumerate(sheets, 1):
root = ET.fromstring(archive.read(name))
output.append(f"## Sheet {index}")
for row in root.iter(f"{{{_S}}}row"):
values: list[str] = []
for cell in row.iter(f"{{{_S}}}c"):
value_node = cell.find(f"{{{_S}}}v")
value = value_node.text if value_node is not None else ""
if cell.get("t") == "s" and value and value.isdigit():
shared_index = int(value)
value = shared[shared_index] if shared_index < len(shared) else value
values.append(value or "")
output.append("\t".join(values))
except zipfile.BadZipFile as exc:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid XLSX") from exc
if len(output) <= 1:
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: XLSX has no sheets")
return "\n".join(output) + "\n"
def _extract_anydoc(data: bytes, extension: str) -> str:
source = ""
output = ""
try:
with tempfile.NamedTemporaryFile(suffix=extension, delete=False) as handle:
handle.write(data)
source = handle.name
with tempfile.NamedTemporaryFile(suffix=".md", delete=False) as handle:
output = handle.name
script = (
"import pathlib,sys; import anydoc; "
"text=anydoc.to_markdown(sys.argv[1]); "
"pathlib.Path(sys.argv[2]).write_text(text, encoding='utf-8')"
)
subprocess.run(
[sys.executable, "-c", script, source, output],
check=True,
capture_output=True,
timeout=DOCUMENT_CONVERSION_TIMEOUT_SECONDS,
)
output_path = Path(output)
if output_path.stat().st_size > MAX_CONVERTED_DOCUMENT_BYTES:
raise DocumentExtractionError(
"DOCUMENT_RESOURCE_LIMIT: converted document exceeds output budget"
)
text = output_path.read_text(encoding="utf-8")
except subprocess.TimeoutExpired as exc:
raise DocumentExtractionError(
f"DOCUMENT_CONVERSION_TIMEOUT: exceeded {DOCUMENT_CONVERSION_TIMEOUT_SECONDS}s"
) from exc
except subprocess.CalledProcessError as exc:
detail = exc.stderr.decode("utf-8", errors="replace")[-1000:]
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: converter exited {exc.returncode}: {detail}"
) from exc
except Exception as exc:
if isinstance(exc, DocumentExtractionError):
raise
raise DocumentExtractionError(
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
) from exc
finally:
for temporary in (source, output):
if temporary:
try:
os.unlink(temporary)
except OSError:
pass
if not isinstance(text, str) or not text.strip():
raise DocumentExtractionError(
"DOCUMENT_EXTRACTION_FAILED: document contains no extractable text"
)
return text.rstrip("\n") + "\n"
+5
View File
@@ -55,6 +55,11 @@ class EvoRuntimeError(RuntimeError):
self.code = code
self.details = tuple(dict(item) for item in details)
def __repr__(self) -> str:
# LangGraph persists task failures using repr(exc). Keep that snapshot
# machine-readable without serializing provider messages or details.
return f"{type(self).__name__}(code={self.code!r})"
def now_ms() -> int:
return time.time_ns() // 1_000_000
+25 -3
View File
@@ -14,6 +14,8 @@ from langchain_core.tools import BaseTool
from langchain_core.utils.function_calling import convert_to_openai_tool
from pydantic import Field
from .contracts import EvoRuntimeError
class GatewayProxyChatModel(BaseChatModel):
gateway_url: str
@@ -78,7 +80,9 @@ class GatewayProxyChatModel(BaseChatModel):
) -> ChatResult:
del stop, kwargs
attempt_id = self._attempt_id(run_manager)
async with httpx.AsyncClient(timeout=httpx.Timeout(660.0, connect=5.0)) as client:
async with httpx.AsyncClient(
timeout=httpx.Timeout(660.0, connect=5.0)
) as client:
response = await client.post(
f"{self.gateway_url.rstrip('/')}/api/internal/recoverable-runs/model/invoke",
json={
@@ -115,7 +119,9 @@ class GatewayProxyChatModel(BaseChatModel):
"tool_choice": self.bound_tool_choice,
"stream": True,
}
async with httpx.AsyncClient(timeout=httpx.Timeout(660.0, connect=5.0)) as client:
async with httpx.AsyncClient(
timeout=httpx.Timeout(660.0, connect=5.0)
) as client:
async with client.stream(
"POST",
f"{self.gateway_url.rstrip('/')}/api/internal/recoverable-runs/model/stream",
@@ -126,11 +132,27 @@ class GatewayProxyChatModel(BaseChatModel):
async for line in response.aiter_lines():
if not line.startswith("data:"):
continue
data = line[len("data:"):].strip()
data = line[len("data:") :].strip()
if data == "[DONE]":
saw_done = True
break
chunk = json.loads(data)
if chunk.get("type") == "error":
code = str(chunk.get("code") or "MODEL_PROVIDER_ERROR")
message = str(chunk.get("message") or code)
details = {
key: value
for key, value in {
"http_status": chunk.get("status"),
"retryable": chunk.get("retryable"),
}.items()
if isinstance(value, int | bool)
}
raise EvoRuntimeError(
code,
message,
details=(details,) if details else (),
)
message = _chunk_to_message(chunk)
yield ChatGenerationChunk(
message=message,
+42 -1
View File
@@ -29,7 +29,7 @@ from __future__ import annotations
import hashlib
import os
from typing import Any
from typing import Any, cast
# ---------------------------------------------------------------------------
@@ -131,6 +131,47 @@ def _patch_openai_empty_sse_keepalive() -> None:
_patch_openai_empty_sse_keepalive()
# ---------------------------------------------------------------------------
# Patch: deepagents routes every non-text extension to a media block even when
# a backend has already extracted bounded UTF-8 text from the document. Keep
# real base64 binary data on the media path (the middleware checks encoding
# first), but route extracted Office/PDF results through a normal ToolMessage.
# ---------------------------------------------------------------------------
_deepagents_extracted_document_text_patched = False
def _patch_deepagents_extracted_document_text() -> None:
global _deepagents_extracted_document_text_patched
if _deepagents_extracted_document_text_patched:
return
try:
from pathlib import Path as _Path
import deepagents.middleware.filesystem as _filesystem
from EvoScientist.document_extract import DOCUMENT_EXTENSIONS
namespace = cast(dict[str, Any], vars(_filesystem))
original = namespace["_get_file_type"]
if getattr(original, "_evoscientist_document_text", False):
_deepagents_extracted_document_text_patched = True
return
def _document_text_type(path: str) -> str:
if _Path(path).suffix.lower() in DOCUMENT_EXTENSIONS:
return "text"
return original(path)
_document_text_type._evoscientist_document_text = True # type: ignore[attr-defined]
namespace["_get_file_type"] = _document_text_type
_deepagents_extracted_document_text_patched = True
except Exception:
pass
_patch_deepagents_extracted_document_text()
# ---------------------------------------------------------------------------
# Patch: ccproxy-api 0.2.7 Codex compatibility.
#
+2
View File
@@ -51,6 +51,7 @@ from .skill_context import (
DEFAULT_MAX_SKILLS_BYTES,
BudgetedSkillsMiddleware,
)
from .subagent_timeout import SubagentTimeoutMiddleware
from .tool_error_handler import ToolErrorHandlerMiddleware
from .tool_protocol_guard import ToolProtocolGuardMiddleware
from .tool_selector import create_tool_selector_middleware
@@ -82,6 +83,7 @@ __all__ = [
"RepetitiveToolCallGuardMiddleware",
"RuntimeContextMiddleware",
"SchedulerMiddleware",
"SubagentTimeoutMiddleware",
"ToolErrorHandlerMiddleware",
"ToolProtocolGuardMiddleware",
"collapse_repetitive_tool_rounds",
+20
View File
@@ -187,6 +187,17 @@ class DynamicReviewMiddleware(HumanInTheLoopMiddleware):
# A LangGraph resume continues at this interrupted node and does not
# re-run before_agent. Reusing a manual state is restrictive and is
# required for the existing resume child Run to complete.
# EXCEPTION: the gateway snapshots the thread's LIVE review mode
# into each resume child's envelope. When the user flipped the
# thread (or the current interrupt) to auto AFTER the parent run
# started, the injected ai4sci_review_mode context is "auto" —
# verify it and bypass HITL instead of interrupting again.
if review is not None and review.get("requested_mode") == "auto":
try:
_resolve_sync(current_run_id, review)
except AutoReviewVerificationError:
return super().after_model(state, runtime)
return None
return super().after_model(state, runtime)
raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID")
@@ -213,5 +224,14 @@ class DynamicReviewMiddleware(HumanInTheLoopMiddleware):
return super().after_model(state, runtime)
return None
if mode == "manual":
# Mirror the sync path: an injected auto context on a resume child
# (thread flipped to auto after the parent started) bypasses HITL
# after successful re-verification.
if review is not None and review.get("requested_mode") == "auto":
try:
await _resolve_async(current_run_id, review)
except AutoReviewVerificationError:
return super().after_model(state, runtime)
return None
return super().after_model(state, runtime)
raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID")
@@ -141,10 +141,13 @@ def _normalize(request: ModelRequest, exc: BaseException) -> ProviderStreamError
- ``AgentControlError`` — a platform-owned typed decision. Gateway route
fallback and canonical error mapping depend on its concrete type and
structured fields, so it must never become a provider incident.
- ``EvoRuntimeError`` — a stable host/runtime error that has already been
classified across the Gateway boundary and must retain its code.
- Models we don't recognize as a provider SDK.
"""
from langchain_core.exceptions import ContextOverflowError
from ..llm.contracts import EvoRuntimeError
from ..llm.errors import (
AgentControlError,
ProviderStreamError,
@@ -166,6 +169,9 @@ def _normalize(request: ModelRequest, exc: BaseException) -> ProviderStreamError
if isinstance(exc, AgentControlError):
return None
if isinstance(exc, EvoRuntimeError):
return None
# LangGraph control-flow / structural signals must propagate
# untouched, regardless of which caller invoked us.
if _should_pass_through(exc):
@@ -12,6 +12,8 @@ from langchain.agents.middleware.types import AgentMiddleware
from langchain_core.messages import ToolMessage, message_to_dict, messages_from_dict
from langgraph.types import Command
from EvoScientist.llm.contracts import EvoRuntimeError
if TYPE_CHECKING:
from langchain.agents.middleware.types import ToolCallRequest
@@ -91,6 +93,15 @@ async def _post(proxy: Mapping[str, str], phase: str, payload: dict[str, Any]) -
"envelope_signature": proxy["envelope_signature"],
},
)
if response.is_error:
try:
error_body = response.json()
except ValueError:
error_body = None
detail = error_body.get("detail") if isinstance(error_body, Mapping) else None
code = detail.get("code") if isinstance(detail, Mapping) else None
if isinstance(code, str) and code.isascii() and code.replace("_", "").isalnum():
raise EvoRuntimeError(code)
response.raise_for_status()
return dict(response.json())
@@ -0,0 +1,84 @@
"""Bound synchronous sub-agent calls in hosted Web runs."""
from __future__ import annotations
import asyncio
import json
from collections.abc import Awaitable, Callable
from typing import Any
from langchain.agents.middleware.types import AgentMiddleware, ToolCallRequest
from langchain_core.messages import ToolMessage
from langgraph.types import Command
class SubagentTimeoutMiddleware(AgentMiddleware):
"""Cancel a synchronous ``task`` call that exceeds the Web time budget."""
@property
def name(self) -> str:
return "subagent_timeout"
def __init__(self, timeout_seconds: float = 180.0) -> None:
super().__init__()
if timeout_seconds <= 0:
raise ValueError("timeout_seconds must be positive")
self.timeout_seconds = float(timeout_seconds)
@staticmethod
async def _cancel_task(task: asyncio.Future[Any]) -> None:
task.cancel()
try:
await asyncio.shield(task)
except asyncio.CancelledError:
if not task.done():
task.add_done_callback(SubagentTimeoutMiddleware._consume_task_result)
raise
except Exception:
pass
@staticmethod
def _consume_task_result(task: asyncio.Future[Any]) -> None:
try:
task.result()
except (asyncio.CancelledError, Exception):
pass
async def awrap_tool_call(
self,
request: ToolCallRequest,
handler: Callable[[ToolCallRequest], Awaitable[ToolMessage | Command[Any]]],
) -> ToolMessage | Command[Any]:
if str(request.tool_call.get("name") or "") != "task":
return await handler(request)
task = asyncio.ensure_future(handler(request))
try:
done, _pending = await asyncio.wait(
{task}, timeout=self.timeout_seconds
)
except asyncio.CancelledError:
await self._cancel_task(task)
raise
if done:
return task.result()
await self._cancel_task(task)
current = asyncio.current_task()
if current is not None and current.cancelling():
raise asyncio.CancelledError
payload = {
"code": "SUBAGENT_TIMEOUT",
"message": (
"The delegated sub-agent exceeded the hosted Web time limit. "
"Continue with available evidence or use the controlled web search tool directly."
),
"retryable": True,
"timeout_seconds": self.timeout_seconds,
}
return ToolMessage(
content=json.dumps(payload, ensure_ascii=False),
tool_call_id=str(request.tool_call.get("id") or "subagent_timeout"),
name="task",
status="error",
additional_kwargs={"error_code": "SUBAGENT_TIMEOUT"},
)
+43 -9
View File
@@ -132,6 +132,20 @@ def _node_version(node: Path) -> tuple[int, int, int]:
return parts
def _existing_resolved_paths(candidates: tuple[str, ...]) -> tuple[str, ...]:
resolved: list[str] = []
seen: set[str] = set()
for candidate in candidates:
try:
value = str(Path(candidate).resolve(strict=True))
except OSError:
continue
if value not in seen:
seen.add(value)
resolved.append(value)
return tuple(resolved)
def _system_read_paths() -> tuple[str, ...]:
candidates = (
(
@@ -162,7 +176,7 @@ def _system_read_paths() -> tuple[str, ...]:
"/dev/urandom",
)
)
return tuple(path for path in candidates if Path(path).exists())
return _existing_resolved_paths(candidates)
def _assert_install_contract() -> NativeSandboxInstallation:
@@ -262,6 +276,7 @@ def _sandbox_settings(
# srt adds these shared compatibility paths even when callers do
# not request them. Explicit deny wins over that built-in allow.
"denyWrite": [
str(files_dir / "uploads"),
"/tmp/claude",
"/private/tmp/claude",
"/dev/tty",
@@ -269,7 +284,10 @@ def _sandbox_settings(
"/dev/autofs_nowait",
],
},
"enableWeakerNestedSandbox": False,
"enableWeakerNestedSandbox": os.getenv(
"EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", ""
).strip().lower()
in {"1", "true", "yes", "on"},
"enableWeakerNetworkIsolation": False,
"allowAppleEvents": False,
"allowPty": False,
@@ -622,6 +640,19 @@ _READY_STATE = "unchecked"
_READY_ERROR: str | None = None
def _network_preflight_probe(port: int, unix_path: Path) -> str:
return (
"import os,socket;\n"
"assert 'OPENAI_API_KEY' not in os.environ\n"
f"t=socket.socket(); tcp=t.connect_ex(('127.0.0.1',{port})); t.close()\n"
"try:\n"
f" u=socket.socket(socket.AF_UNIX); unix=u.connect_ex({str(unix_path)!r}); u.close()\n"
"except OSError:\n"
" unix=1\n"
"assert tcp != 0 and unix != 0\n"
)
def _run_preflight(installation: NativeSandboxInstallation) -> None:
# AF_UNIX paths are limited to roughly 100 bytes on both target platforms;
# a deployment workspace path can already exceed that before the filename.
@@ -632,6 +663,10 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None:
files_dir.mkdir(mode=0o700)
runtime_dir.mkdir(mode=0o700)
(files_dir / "probe.txt").write_text("allowed", encoding="utf-8")
uploads_dir = files_dir / "uploads"
uploads_dir.mkdir(mode=0o700)
upload_source = uploads_dir / "source.txt"
upload_source.write_text("immutable", encoding="utf-8")
outside = root / "outside-secret.txt"
outside.write_text("secret", encoding="utf-8")
control_secret = runtime_dir / "control-secret.txt"
@@ -649,16 +684,13 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None:
unix_server.bind(str(unix_path))
unix_server.listen(1)
python_probe = (
"import os,socket,sys;"
"assert 'OPENAI_API_KEY' not in os.environ;"
f"t=socket.socket(); tcp=t.connect_ex(('127.0.0.1',{port})); t.close();"
f"u=socket.socket(socket.AF_UNIX); unix=u.connect_ex({str(unix_path)!r}); u.close();"
"sys.exit(0 if tcp != 0 and unix != 0 else 9)"
)
python_probe = _network_preflight_probe(port, unix_path)
command = " && ".join(
(
'test "$(cat probe.txt)" = allowed',
'test "$(cat uploads/source.txt)" = immutable',
"! sh -c 'printf changed > uploads/source.txt' 2>/dev/null",
"! rm uploads/source.txt 2>/dev/null",
"printf written > written.txt",
'printf temporary > "$TMPDIR/probe.tmp"',
"pandoc --version >/dev/null",
@@ -695,6 +727,8 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None:
)
if outside.read_text(encoding="utf-8") != "secret":
raise NativeSandboxUnavailable("native sandbox preflight escaped workspace")
if upload_source.read_text(encoding="utf-8") != "immutable":
raise NativeSandboxUnavailable("native sandbox preflight modified uploads")
def ensure_native_sandbox_ready() -> None:
+19
View File
@@ -267,6 +267,23 @@ echo "PID: $!" # check: ps -p <PID> · stop: kill <PID> · read
This prevents blocking the conversation during long operations."""
_FILE_PROGRAMMING_GUIDELINES = """# Safe Binary File Programming
- When `read_file` returns `BINARY_PROCESSING_REQUIRED`, use `execute` to inspect the referenced workspace file with a short, bounded program. Do not report the format as unsupported without trying the indicated safe workflow.
- For ZIP/TAR archives, list and validate members before selective extraction; never use `extractall` on an untrusted archive. Reject absolute paths, `..`, links escaping the destination, excessive member counts, expanded sizes, and compression ratios. Never execute files extracted from an archive.
- For SQLite, open `file:<path>?mode=ro&immutable=1` with `uri=True`, set `PRAGMA query_only=ON`, inspect `sqlite_master`, and use bounded `SELECT ... LIMIT ...` queries. Never use `ATTACH DATABASE`, enable extensions, or modify the source database.
- Do not modify original files under `uploads/`. Write extracted members, query results, converted documents, and other derived artifacts under `artifacts/` or `.ai4sci/`.
- Text extracted by `read_file` from PDF or Office files is a semantic view, not the original container bytes. Never write that text back to the source document with `write_file` or `edit_file`.
"""
_WEB_RESEARCH_GUIDELINES = """# Controlled Web Research
- `search_observations` searches local memory only. It does not access the internet and must never be presented as live web search.
- For current facts, public webpages, source discovery, or URL verification, call `tavily_search` first when it is available. Use `web_search` only when that configured search provider is available.
- Do not use `execute`, `curl`, or `httpx` to reach the public internet. The execution sandbox intentionally blocks raw networking; controlled search tools are the only supported network path.
- If controlled web search is unavailable or fails, report that specific limitation once and continue with clearly labeled non-live evidence. Do not repeatedly probe DNS, proxies, direct IPs, or local ports.
"""
# Sandbox (default) header: virtual `/` workspace.
_SHELL_GUIDELINES_SANDBOX_HEADER = """# Shell Execution Guidelines
@@ -460,6 +477,8 @@ def get_system_prompt(
REPORT_TEMPLATE,
WRITING_GUIDELINES,
shell_guidelines,
_FILE_PROGRAMMING_GUIDELINES,
_WEB_RESEARCH_GUIDELINES,
DELEGATION_STRATEGY,
ASYNC_NOTIFICATIONS,
]
+204
View File
@@ -1630,6 +1630,210 @@ async def db_stats(top_n: int = 5) -> dict[str, Any]:
return out
# ---------------------------------------------------------------------------
# History pruning / VACUUM (gateway timer + admin endpoints)
# ---------------------------------------------------------------------------
def _uuid6_unix_ts(checkpoint_id: str) -> float | None:
"""Extract the unix timestamp embedded in a UUIDv6 checkpoint id.
Returns ``None`` for non-UUID or non-v6 ids — legacy ids carry no usable
timestamp, and callers must skip those threads rather than guess an age.
"""
try:
u = uuid.UUID(checkpoint_id)
except (ValueError, AttributeError, TypeError):
return None
if u.version != 6:
return None
ts60 = ((u.int >> 80) << 12) | ((u.int >> 64) & 0x0FFF)
return ts60 / 10_000_000 - 12219292800
async def _count_thread_rows(
conn: aiosqlite.Connection, thread_id: str
) -> tuple[int, int]:
"""Return ``(checkpoints, writes)`` row counts for one thread."""
async with conn.execute(
"SELECT COUNT(*) FROM checkpoints WHERE thread_id = ?", (thread_id,)
) as cur:
row = await cur.fetchone()
ck = int(row[0]) if row else 0
wr = 0
if await _table_exists(conn, "writes"):
async with conn.execute(
"SELECT COUNT(*) FROM writes WHERE thread_id = ?", (thread_id,)
) as cur:
row = await cur.fetchone()
wr = int(row[0]) if row else 0
return ck, wr
async def _prune_thread_on_conn(
conn: aiosqlite.Connection, thread_id: str, keep_last: int
) -> tuple[int, int]:
"""Prune one thread on an open connection; return rows deleted per table.
Reuses :meth:`PruningCheckpointer._prune_after_put` per
``(thread_id, checkpoint_ns)`` group so retention is identical to the
live write path, including DeltaChannel snapshot-chain preservation.
Only ``metadata.agent_name == AGENT_NAME`` rows are ever touched.
"""
saver = PruningCheckpointer(conn, keep_per_ns=max(1, int(keep_last)))
before_ck, before_wr = await _count_thread_rows(conn, thread_id)
async with conn.execute(
"SELECT DISTINCT checkpoint_ns FROM checkpoints "
"WHERE thread_id = ? AND json_extract(metadata, '$.agent_name') = ?",
(thread_id, AGENT_NAME),
) as cur:
namespaces = [r[0] for r in await cur.fetchall()]
for ns in namespaces:
await saver._prune_after_put(thread_id, ns or "")
after_ck, after_wr = await _count_thread_rows(conn, thread_id)
return before_ck - after_ck, before_wr - after_wr
async def prune_thread_history(
thread_id: str, keep_last: int = 2, db_path: str | None = None
) -> dict[str, int]:
"""Prune one thread's history, keeping its ``keep_last`` most recent rows.
Returns ``{"deleted_checkpoints": int, "deleted_writes": int}``.
"""
path = str(db_path or get_db_path())
result = {"deleted_checkpoints": 0, "deleted_writes": 0}
if not Path(path).exists():
return result
async with aiosqlite.connect(path, timeout=30.0) as conn:
if not await _table_exists(conn, "checkpoints"):
return result
ck, wr = await _prune_thread_on_conn(conn, str(thread_id), keep_last)
result["deleted_checkpoints"] = ck
result["deleted_writes"] = wr
return result
async def prune_all_stale_threads(
max_age_hours: float = 72,
keep_last: int = 2,
db_path: str | None = None,
) -> dict[str, int]:
"""Prune every EvoScientist thread idle for at least ``max_age_hours``.
A thread is stale when its newest checkpoint (UUIDv6 timestamp) is older
than the cutoff. Threads whose newest id has no parseable timestamp are
skipped. Returns ``{"databases_processed", "threads_pruned",
"total_deleted_checkpoints", "total_deleted_writes"}``.
"""
path = str(db_path or get_db_path())
result = {
"databases_processed": 0,
"threads_pruned": 0,
"total_deleted_checkpoints": 0,
"total_deleted_writes": 0,
}
if not Path(path).exists():
return result
cutoff = time.time() - float(max_age_hours) * 3600.0
async with aiosqlite.connect(path, timeout=30.0) as conn:
if not await _table_exists(conn, "checkpoints"):
return result
result["databases_processed"] = 1
async with conn.execute(
"SELECT thread_id, MAX(checkpoint_id) FROM checkpoints "
"WHERE json_extract(metadata, '$.agent_name') = ? "
"GROUP BY thread_id",
(AGENT_NAME,),
) as cur:
rows = await cur.fetchall()
stale = [
tid
for tid, newest in rows
if (ts := _uuid6_unix_ts(newest)) is not None and ts < cutoff
]
for tid in stale:
ck, wr = await _prune_thread_on_conn(conn, tid, keep_last)
if ck or wr:
result["threads_pruned"] += 1
result["total_deleted_checkpoints"] += ck
result["total_deleted_writes"] += wr
return result
async def list_all_thread_ids(db_path: str | None = None) -> list[str]:
"""Return all EvoScientist thread ids in the sessions DB."""
path = str(db_path or get_db_path())
if not Path(path).exists():
return []
async with aiosqlite.connect(path, timeout=30.0) as conn:
if not await _table_exists(conn, "checkpoints"):
return []
async with conn.execute(
"SELECT DISTINCT thread_id FROM checkpoints "
"WHERE json_extract(metadata, '$.agent_name') = ?",
(AGENT_NAME,),
) as cur:
return [r[0] for r in await cur.fetchall()]
def list_all_session_db_paths() -> list[Path]:
"""Return existing session DB paths.
The current storage layout uses a single shared DB (``get_db_path()``),
so this returns a one-element list when it exists, else ``[]``.
"""
path = get_db_path()
return [path] if path.exists() else []
async def vacuum_db(db_path: str | None = None) -> dict[str, Any]:
"""Run ``VACUUM`` on the sessions DB to reclaim freed pages."""
path = str(db_path or get_db_path())
p = Path(path)
before = p.stat().st_size if p.exists() else 0
if p.exists():
async with aiosqlite.connect(path, timeout=120.0) as conn:
await conn.execute("VACUUM")
await conn.commit()
after = p.stat().st_size if p.exists() else 0
return {
"db_path": path,
"size_before_bytes": before,
"size_after_bytes": after,
}
async def get_aggregated_storage_stats() -> dict[str, Any]:
"""Return :func:`db_stats` plus per-thread checkpoint depth stats."""
stats = await db_stats()
depth: dict[str, Any] = {"min": 0, "max": 0, "avg": 0.0}
path = Path(stats["db_path"])
if path.exists():
try:
async with aiosqlite.connect(str(path), timeout=30.0) as conn:
if await _table_exists(conn, "checkpoints"):
async with conn.execute(
"SELECT MIN(n), MAX(n), AVG(n) FROM ("
" SELECT COUNT(*) AS n FROM checkpoints "
" WHERE json_extract(metadata, '$.agent_name') = ? "
" GROUP BY thread_id"
")",
(AGENT_NAME,),
) as cur:
row = await cur.fetchone()
if row and row[0] is not None:
depth = {
"min": int(row[0]),
"max": int(row[1]),
"avg": round(float(row[2]), 2),
}
except aiosqlite.Error:
# Read-only diagnostic — mirror db_stats and degrade to zeros.
pass
return {**stats, "thread_depth": depth}
# ---------------------------------------------------------------------------
# langgraph-api / WebUI checkpointer factory
# ---------------------------------------------------------------------------
+64 -17
View File
@@ -14,6 +14,16 @@ from tavily import TavilyClient
# Lazy initialization - only create client when needed
_tavily_client = None
MAX_SEARCH_RESULTS = 5
MAX_DISPLAY_QUERY_CHARS = 512
MAX_DISPLAY_TITLE_CHARS = 512
MAX_DISPLAY_URL_CHARS = 2_048
MAX_PAGE_CONTENT_CHARS = 4_000
MAX_SEARCH_RESULT_CHARS = 16_000
_TRUNCATION_MARKER = "\n\n[page content truncated]"
_SEARCH_TRUNCATION_MARKER = (
"\n[search result content truncated to preserve all titles and URLs]"
)
def _get_tavily_client() -> TavilyClient:
@@ -46,6 +56,12 @@ async def fetch_webpage_content(url: str, timeout: float = 10.0) -> str:
async with httpx.AsyncClient() as client:
response = await client.get(url, headers=headers, timeout=timeout)
response.raise_for_status()
content_type = response.headers.get("content-type", "").lower()
if not any(
allowed in content_type
for allowed in ("text/", "application/xhtml+xml")
):
return f"Error fetching content from {url}: unsupported content type {content_type or 'unknown'}"
return markdownify(response.text)
except Exception as e:
return f"Error fetching content from {url}: {e!s}"
@@ -72,40 +88,71 @@ async def tavily_search(
"""
def _sync_search() -> dict:
bounded_max_results = max(1, min(int(max_results), MAX_SEARCH_RESULTS))
return _get_tavily_client().search(
query,
max_results=max_results,
max_results=bounded_max_results,
topic=topic,
)
try:
# Run Tavily search asynchronously
# Run Tavily search asynchronously through the controlled host path.
search_results = await asyncio.to_thread(_sync_search)
from EvoScientist.runtime_integrations import record_service_usage
await record_service_usage("tavily", "search")
# Fetch full content for each URL concurrently
results = search_results.get("results", [])
if not results:
return f"No results found for '{query}'"
# Fetch all webpages concurrently
fetch_tasks = [fetch_webpage_content(r["url"]) for r in results]
contents = await asyncio.gather(*fetch_tasks)
# Format results
normalized = []
for result, fetched_content in zip(results, contents, strict=False):
title = str(result.get("title") or "Untitled")[:MAX_DISPLAY_TITLE_CHARS]
raw_url = str(result.get("url") or "")
url = (
raw_url
if len(raw_url) <= MAX_DISPLAY_URL_CHARS
else raw_url[: MAX_DISPLAY_URL_CHARS - len("...[URL truncated]")]
+ "...[URL truncated]"
)
tavily_summary = str(result.get("content") or "").strip()
fetch_failed = fetched_content.startswith("Error fetching content from ")
content = tavily_summary if fetch_failed and tavily_summary else fetched_content
fetch_note = (
"\n\n> Source page fetch failed; showing the Tavily-indexed summary."
if fetch_failed and tavily_summary
else ""
)
normalized.append((title, url, content, fetch_note))
display_query = query[:MAX_DISPLAY_QUERY_CHARS]
prefix = f"Found {len(normalized)} live web result(s) for '{display_query}':\n\n"
metadata_blocks = [f"## {title}\n**URL:** {url}\n\n" for title, url, _, _ in normalized]
fixed_chars = len(prefix) + sum(len(block) + len("\n\n---\n") for block in metadata_blocks)
remaining = max(0, MAX_SEARCH_RESULT_CHARS - fixed_chars - len(_SEARCH_TRUNCATION_MARKER))
per_result_budget = remaining // max(1, len(normalized))
result_texts = []
for result, content in zip(results, contents, strict=False):
result_text = f"""## {result["title"]}
**URL:** {result["url"]}
content_truncated = False
for metadata, (_, _, content, fetch_note) in zip(metadata_blocks, normalized, strict=True):
content_budget = max(0, min(MAX_PAGE_CONTENT_CHARS, per_result_budget - len(fetch_note)))
if len(content) > content_budget:
marker_budget = min(len(_TRUNCATION_MARKER), content_budget)
content = (
content[: content_budget - marker_budget]
+ _TRUNCATION_MARKER[:marker_budget]
)
content_truncated = True
result_texts.append(f"{metadata}{content}{fetch_note}\n\n---\n")
{content}
---
"""
result_texts.append(result_text)
return f"""Found {len(result_texts)} result(s) for '{query}':
{"".join(result_texts)}"""
formatted = prefix + "".join(result_texts)
if content_truncated:
formatted += _SEARCH_TRUNCATION_MARKER
return formatted
except Exception as e:
return f"Search failed: {e!s}"
+14 -1
View File
@@ -51,10 +51,23 @@ def web_tool_registry_manifest() -> tuple[tuple[dict[str, Any], ...], str]:
"additionalProperties": True,
"maxProperties": 32,
}
tool_descriptions = {
"tavily_search": (
"Search the live public web through the controlled Tavily service. "
"Use this for current facts, source discovery, and URL verification. "
"Do not use execute, curl, httpx, or raw sandbox networking instead."
),
"web_search": (
"Search the live public web through the configured MCP search provider. "
"Use this for current facts and source verification, not local memory recall."
),
}
manifest: tuple[dict[str, Any], ...] = tuple(
{
"name": name,
"description": "EvoScientist Web runtime tool",
"description": tool_descriptions.get(
name, "EvoScientist Web runtime tool"
),
"schema": schema,
}
for name in names
+133 -6
View File
@@ -375,7 +375,16 @@ def _standard_error(exc: Exception) -> str:
def _is_binary_file(path: str, raw: bytes) -> bool:
"""Classify common containers explicitly and fall back to content."""
"""Classify common containers explicitly and fall back to content.
The UTF-8 probe samples the first bytes, so a multi-byte character can be
cut in half at the sample boundary (e.g. an 8192-byte cut splitting a
3-byte CJK character). ``decode`` with ``errors="ignore"`` would hide real
garbage, so instead we re-probe on failure with the trailing partial
sequence removed: a decode error that vanishes once the (at most 4-byte)
dangling suffix is dropped is a truncation artifact, not binary content.
Files that still fail on the trimmed sample are genuinely not UTF-8.
"""
if Path(path).suffix.lower() in _BINARY_EXTENSIONS:
return True
@@ -384,7 +393,15 @@ def _is_binary_file(path: str, raw: bytes) -> bool:
return True
try:
sample.decode("utf-8")
except UnicodeDecodeError:
except UnicodeDecodeError as exc:
# Only a decode error at the very end of the sample can be a boundary
# cut. Errors positioned mid-sample are real invalid bytes.
if exc.start >= max(0, len(sample) - 4):
try:
sample[: exc.start].decode("utf-8")
except UnicodeDecodeError:
return True
return False
return True
return False
@@ -397,6 +414,13 @@ class ScopedFilesystemBackend(BackendProtocol):
root_dir, max_search_file_bytes=max_search_file_bytes
)
@staticmethod
def _uploads_are_read_only(file_path: str) -> bool:
normalized = "/" + file_path.replace("\\", "/").lstrip("/")
return normalized == "/workspace/uploads" or normalized.startswith(
"/workspace/uploads/"
)
@staticmethod
def _search_path(path: str | None) -> str:
if path in {None, "/"}:
@@ -426,15 +450,72 @@ class ScopedFilesystemBackend(BackendProtocol):
return LsResult(error=f"Cannot list '{path}': {_standard_error(exc)}")
def read(self, file_path: str, offset: int = 0, limit: int = 2000) -> ReadResult:
from .document_extract import (
MAX_DOCUMENT_BYTES,
DocumentExtractionError,
binary_processing_guidance,
classify_file,
extract_document_bytes,
paginate_document_text,
prepare_image_bytes,
)
try:
entry = self.workspace.entry(file_path)
with self.workspace.open_binary(file_path) as handle:
head = handle.read(4096)
kind = classify_file(file_path, head)
if kind == "document":
if entry.size > MAX_DOCUMENT_BYTES:
return ReadResult(
error=(
f"DOCUMENT_TOO_LARGE: '{file_path}' is {entry.size} bytes; "
f"document extraction limit is {MAX_DOCUMENT_BYTES} bytes"
)
)
handle.seek(0)
raw = handle.read(MAX_DOCUMENT_BYTES + 1)
try:
extracted = extract_document_bytes(raw, file_path)
content = paginate_document_text(
extracted, offset=max(0, offset), limit=max(1, limit)
)
except DocumentExtractionError as exc:
return ReadResult(error=str(exc))
return ReadResult(
file_data={"content": content, "encoding": "utf-8"}
)
if kind == "image":
handle.seek(0)
raw = handle.read()
try:
raw = prepare_image_bytes(raw, file_path)
except DocumentExtractionError as exc:
return ReadResult(error=str(exc))
return ReadResult(
file_data={
"content": base64.standard_b64encode(raw).decode("ascii"),
"encoding": "base64",
}
)
if kind in {"archive", "database", "executable", "dataset", "media"}:
return ReadResult(
error=binary_processing_guidance(file_path, kind, entry.size)
)
if _is_binary_file(file_path, head):
return ReadResult(
error=binary_processing_guidance(file_path, "binary", entry.size)
)
handle.seek(0)
raw = handle.read()
if _is_binary_file(file_path, raw):
return ReadResult(
file_data={
"content": base64.standard_b64encode(raw).decode("ascii"),
"encoding": "base64",
}
error=binary_processing_guidance(file_path, "binary", entry.size)
)
content = raw.decode("utf-8")
empty = check_empty_content(content)
@@ -455,6 +536,29 @@ class ScopedFilesystemBackend(BackendProtocol):
return ReadResult(error=f"Error reading file '{file_path}': {_standard_error(exc)}")
def write(self, file_path: str, content: str) -> WriteResult:
if self._uploads_are_read_only(file_path):
return WriteResult(
error=(
f"Cannot modify original upload '{file_path}'. "
"Write derived content under /workspace/artifacts/ or /workspace/.ai4sci/."
)
)
from .document_extract import (
ARCHIVE_EXTENSIONS,
DATABASE_EXTENSIONS,
DOCUMENT_EXTENSIONS,
)
if Path(file_path).suffix.lower() in (
DOCUMENT_EXTENSIONS | ARCHIVE_EXTENSIONS | DATABASE_EXTENSIONS
):
return WriteResult(
error=(
f"Cannot write plain text to binary container '{file_path}'. "
"Use execute with an appropriate document, archive, or database "
"library and write a derived file under artifacts/."
)
)
try:
self.workspace.write_new(file_path, content.encode("utf-8"))
return WriteResult(path=file_path)
@@ -472,6 +576,29 @@ class ScopedFilesystemBackend(BackendProtocol):
new_string: str,
replace_all: bool = False,
) -> EditResult:
if self._uploads_are_read_only(file_path):
return EditResult(
error=(
f"Cannot modify original upload '{file_path}'. "
"Write derived content under /workspace/artifacts/ or /workspace/.ai4sci/."
)
)
from .document_extract import (
ARCHIVE_EXTENSIONS,
DATABASE_EXTENSIONS,
DOCUMENT_EXTENSIONS,
)
if Path(file_path).suffix.lower() in (
DOCUMENT_EXTENSIONS | ARCHIVE_EXTENSIONS | DATABASE_EXTENSIONS
):
return EditResult(
error=(
f"Cannot edit binary container '{file_path}' with text replacement. "
"Use execute with an appropriate library and write a derived file "
"under artifacts/."
)
)
try:
with self.workspace.open_binary(file_path) as handle:
content = handle.read().decode("utf-8")
+2
View File
@@ -0,0 +1,2 @@
global-exclude *.py[cod]
prune **/__pycache__
@@ -0,0 +1,110 @@
# Ai4Sci 有界文件读取与 Agent 自主编程实施方案
> 状态:实施基线 v1.0
> 范围:EvoScientist Core + Ai4Sci-Web 必要接线
> 原则:程序提供事实和安全边界,主 Agent 使用现有 read_file/execute 完成渐进处理。
## 1. 目标
修复 Office/PDF 等二进制完整 Base64 进入模型上下文的问题,同时保留 Agent 对 ZIP、数据库和其他研究文件的自主编程能力。
本阶段不新增 inspect_file/process_file,不新增文件路由模型,不新增数据库表,不改变 SSE 事件协议。
## 2. 固定处理契约
| 类型 | read_file 行为 | 后续处理 |
|---|---|---|
| UTF-8文本/源码 | offset/limit 分页文本 | 主模型直接分析 |
| 图片 | 有界媒体块 | 多模态模型分析 |
| PDF/Office/ODF/RTF/EPUB | 提取为 Markdown/文本并分页 | 需要视觉信息时 Agent 用 execute 渲染指定页 |
| ZIP/TAR/7Z/RAR | 返回结构化事实和安全编程要求,不返回内容 | Agent 用 execute 先列清单,再选择性解压 |
| SQLite/DB | 返回结构化事实和只读查询要求 | Agent 用 sqlite3 mode=ro/query_only 编程查询 |
| 数据集/音视频 | 返回结构化事实和建议命令 | Agent 用现有库/CLI采样、转录或抽帧 |
| EXE/库/未知二进制 | 元数据引用;明确禁止执行 | 仅允许静态检查 |
## 3. 安全与预算
- 文档源文件最大 50 MiB,转换前检查。
- 单次提取文本最大 50,000 字符,按行截断并返回 next_offset 提示。
- ZIP禁止 extractall;先检查成员数、总展开量、单成员大小、压缩比、路径穿越、绝对路径和符号链接。
- ZIP建议上限:2000成员、500 MiB总展开、100 MiB单成员、100:1压缩比、3层嵌套。
- SQLite必须 `file:<path>?mode=ro&immutable=1`、`PRAGMA query_only=ON`,查询必须有LIMIT,结果写入artifacts/。
- EXE/DLL/ELF/Mach-O、宏和归档成员不得自动执行。
- 原始uploads由文件工具和Native Sandbox双层强制只读;派生文件写入artifacts/或.ai4sci/。
- read_file不得对任何非图片二进制返回完整Base64。
## 4. Core改动
1. 新增 `EvoScientist/document_extract.py`:
- 文件类型集合和分派;
- 50 MiB输入限制;
- OOXML/可选anydoc文档提取;
- 结构化失败信息。
2. 修改 `EvoScientist/workspace_files.py`:
- 覆盖 Ai4Sci Web full/scoped 的 `NativeWorkspaceBackend` 读取路径;
- 文件先分类;
- Office/PDF先提取再分页;
- 图片先解码校验,限制源文件、像素数和多帧输入,最大边降采样至 2048px 后再走现有Base64媒体契约;
- OOXML限制成员数、单成员/总展开量、压缩比,拒绝异常路径、重复和加密成员;
- PDF与旧Office转换在隔离子进程中执行,限制60秒和10MiB转换输出;
- ZIP/DB/其他二进制返回结构化错误指引;
- 避免先完整读取大文档再判断类型。
3. 修改 `EvoScientist/prompts.py`:
- 加入ZIP和数据库安全编程规则;
- 明确文档提取文本不可用普通写工具写回容器。
4. 显式声明 `firecrawl-anydoc` 依赖;若不可用或格式不支持,返回可操作错误,不回退Base64。
## 5. Gateway最小接线
- 上传阶段负责扩展名、MIME、配额、路径和Magic校验;拒绝未知 `application/octet-stream` 绕过。
- 显式允许 `.db/.sqlite/.sqlite3`,并将SQLite MIME别名视为等价;不在Gateway解析数据库。
- 附件输入至少保留virtual_path;Office/PDF提取、ZIP解包和数据库查询仍在Agent/execute侧。
- 工具事件归一化保留受限格式的 `error_code`,非终止型文件工具失败显示在现有ToolOutputItem工具卡中。
- 永久workspace准备失败通过现有ErrorItem和done事件投影为 `failed/incomplete/runtime_error`,并复用 `primary_error_code`;不新增平行终态协议。
## 6. 错误码
| code | recoverable | 含义 |
|---|---:|---|
| DOCUMENT_EXTRACTION_FAILED | true/视原因 | 文档损坏、加密或转换失败 |
| DOCUMENT_TOO_LARGE | false | 超过50 MiB输入限制 |
| BINARY_PROCESSING_REQUIRED | true | ZIP/DB/媒体等需Agent编程处理 |
| UNSUPPORTED_BINARY_FILE | false | 未知或不允许处理的二进制 |
| MODEL_RATE_LIMITED | true | 文档处理后模型调用受限 |
## 7. 测试与验收
### Core
- DOCX/PPTX/PDF不返回Base64。
- ZIP、SQLite、EXE、未知二进制不返回Base64或NUL乱码。
- 图片仍返回Base64图片媒体契约。
- 文档提取支持分页和50KB字符上限。
- 超过50MiB在转换前拒绝。
- 转换失败不回退Base64。
- UTF-8边界切分回归保持通过。
- prompt包含ZIP安全清单、禁止extractall、SQLite只读模板。
### 端到端
- 重放9.6MB PPTX时,任一ToolMessage/Checkpoint中不存在原文件Base64。
- Agent可读取提取文本并继续任务。
- ZIP先列目录后选择性解压;原ZIP哈希不变。
- SQLite只读查询;原DB哈希不变。
- 文件处理失败时Run不是正常completed,而是failed/incomplete并带稳定错误码。
## 8. 实施阶段
1. P0:文件分类、文档提取、阻断Base64、Core单测。
2. P0:ZIP/SQLite安全编程提示和测试。
3. P1:Gateway失败投影缺口(若独立评审确认存在)。
4. P1:真实PPTX/ZIP/SQLite集成重放。
5. 代码审查、静态扫描、回归修复和最终复验。
## 9. 非目标
- 不建设独立文件分析微服务。
- 不实现专用ZIP/数据库Agent工具。
- 不让隐藏模型生成文件处理计划。
- 不支持执行上传的可执行文件或宏。
- 不在本阶段实现全量视频理解或恶意软件动态沙箱。
+9 -1
View File
@@ -1,6 +1,6 @@
[project]
name = "EvoScientist"
version = "0.2.2"
dynamic = ["version"]
description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
readme = "README.md"
requires-python = ">=3.11"
@@ -17,6 +17,8 @@ classifiers = [
]
dependencies = [
"deepagents[quickjs]~=0.6.12",
"firecrawl-anydoc>=0.1.6,<0.2",
"pillow>=10.0",
"langchain>=1.3",
"langchain-anthropic>=1.4",
"langchain-openai>=1.2",
@@ -111,6 +113,9 @@ build-backend = "setuptools.build_meta"
[tool.setuptools.packages.find]
include = ["EvoScientist*"]
[tool.setuptools.dynamic]
version = { attr = "EvoScientist._version.__version__" }
[tool.setuptools.package-data]
EvoScientist = [
"subagents/*.yaml",
@@ -118,6 +123,9 @@ EvoScientist = [
"skills/**/*",
]
[tool.setuptools.exclude-package-data]
"*" = ["**/__pycache__/*", "**/*.pyc", "**/*.pyo"]
[tool.pytest.ini_options]
testpaths = ["tests"]
asyncio_mode = "auto"
@@ -0,0 +1,79 @@
from __future__ import annotations
import argparse
from pathlib import Path
_ORIGINAL = """ const rootSkip = new Set(['proc', 'dev', 'sys']);
for (const p of readConfig?.denyOnly || []) {
if (normalizePathForSandbox(p) === '/') {
for (const child of fs.readdirSync('/')) {
if (!rootSkip.has(child))
readDenyPaths.push('/' + child);
}
}
"""
_REPLACEMENT = """ const rootSkip = new Set(['proc', 'dev', 'sys']);
const rootChildIsAllowedSymlink = (childPath) => {
try {
if (!fs.lstatSync(childPath).isSymbolicLink())
return false;
const resolved = fs.realpathSync(childPath);
return readAllowPaths.some(allowPath => resolved === allowPath || resolved.startsWith(allowPath + '/'));
}
catch {
return false;
}
};
for (const p of readConfig?.denyOnly || []) {
if (normalizePathForSandbox(p) === '/') {
for (const child of fs.readdirSync('/')) {
const childPath = '/' + child;
if (!rootSkip.has(child) && !rootChildIsAllowedSymlink(childPath))
readDenyPaths.push(childPath);
}
}
"""
_TMPFS_ORIGINAL = """ args.push('--ro-bind', allowPath, allowPath);
logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`);
}
}
}
"""
_TMPFS_REPLACEMENT = """ args.push('--ro-bind', allowPath, allowPath);
logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`);
}
}
// A denyRead tmpfs must not become an unlisted writable location. Remount
// only the parent mount read-only; explicit writable child binds remain rw.
if (!allowedWritePaths.includes(normalizedPath)) {
args.push('--remount-ro', normalizedPath);
}
}
"""
def patch_file(path: Path) -> None:
source = path.read_text(encoding="utf-8")
if _REPLACEMENT in source or _TMPFS_REPLACEMENT in source:
raise RuntimeError("sandbox runtime merged-usr patch is already patched")
if source.count(_ORIGINAL) != 1:
raise RuntimeError("sandbox runtime merged-usr patch target does not match pinned source")
if source.count(_TMPFS_ORIGINAL) != 1:
raise RuntimeError("sandbox runtime read-only tmpfs patch target does not match pinned source")
patched = source.replace(_ORIGINAL, _REPLACEMENT)
patched = patched.replace(_TMPFS_ORIGINAL, _TMPFS_REPLACEMENT)
path.write_text(patched, encoding="utf-8")
def main() -> None:
parser = argparse.ArgumentParser()
parser.add_argument("path", type=Path)
args = parser.parse_args()
patch_file(args.path)
if __name__ == "__main__":
main()
+22 -1
View File
@@ -7,6 +7,7 @@ LANGGRAPH_CONFIG="${PROJECT_DIR}/EvoScientist/langgraph_dev/langgraph.json"
HOST="${EVOSCIENTIST_LANGGRAPH_HOST:-127.0.0.1}"
PORT="${EVOSCIENTIST_LANGGRAPH_DEV_PORT:-3076}"
WEB_ENV="${PROJECT_DIR}/../Ai4Sci-Web/.env"
N_JOBS="${EVOSCIENTIST_LANGGRAPH_JOBS_PER_WORKER:-6}"
if [[ ! -x "${PROJECT_DIR}/.venv/bin/langgraph" ]]; then
echo "LangGraph executable not found: ${PROJECT_DIR}/.venv/bin/langgraph" >&2
@@ -16,6 +17,26 @@ fi
cd "${PROJECT_DIR}"
# Ask before stopping the process listening on PORT. Matching every connection
# would also terminate Gateway while it is connected to this Runtime.
PIDS="$(lsof -tiTCP:"${PORT}" -sTCP:LISTEN 2>/dev/null || true)"
if [[ -n "${PIDS}" ]]; then
echo "Port ${PORT} is already in use by PID(s): ${PIDS}" >&2
ANSWER=""
read -r -p "Kill the process(es) using port ${PORT}? [y/N] " ANSWER || true
case "${ANSWER}" in
[yY]|[yY][eE][sS])
echo "Killing process(es) on port ${PORT}: ${PIDS}"
kill ${PIDS}
sleep 1
;;
*)
echo "LangGraph not started; process(es) on port ${PORT} were left running." >&2
exit 1
;;
esac
fi
# The Web Gateway always supplies a verified conversation workspace scope.
export EVOSCIENTIST_DEPLOY_MODE="${EVOSCIENTIST_DEPLOY_MODE:-full}"
export EVOSCIENTIST_WORKSPACE_DIR="${EVOSCIENTIST_WORKSPACE_DIR:-${PROJECT_DIR}/../.ai4sci/workspace}"
@@ -27,4 +48,4 @@ exec uv run --env-file "${WEB_ENV}" langgraph dev \
--no-browser \
--no-reload \
--allow-blocking \
--n-jobs-per-worker 1
--n-jobs-per-worker "${N_JOBS}"
@@ -15,6 +15,7 @@ from types import SimpleNamespace
import pytest
from EvoScientist.llm.contracts import EvoRuntimeError
from EvoScientist.llm.errors import (
AgentControlError,
ModelToolProtocolError,
@@ -169,6 +170,16 @@ class TestNormalize:
assert _normalize(req, error) is None
def test_stable_runtime_error_passes_through(self):
req = _request(_openai_model())
error = EvoRuntimeError(
"UPSTREAM_RATE_LIMITED",
"模型服务请求频率超限,请稍后重试或切换模型。",
details=({"http_status": 429},),
)
assert _normalize(req, error) is None
# ---------------------------------------------------------------------------
# _is_provider_error — used by tool selector to distinguish provider
@@ -397,6 +408,23 @@ class TestMiddleware:
assert excinfo.value.code == "MODEL_TOOL_PROTOCOL_INVALID"
assert excinfo.value.fallbackable is True
def test_awrap_preserves_stable_runtime_error_identity(self):
raised = EvoRuntimeError(
"UPSTREAM_RATE_LIMITED",
"模型服务请求频率超限,请稍后重试或切换模型。",
details=({"http_status": 429},),
)
async def handler(_req):
raise raised
req = _request(_openai_model())
with pytest.raises(EvoRuntimeError) as excinfo:
self._run_awrap(ErrorNormalizationMiddleware(), req, handler)
assert excinfo.value is raised
assert excinfo.value.code == "UPSTREAM_RATE_LIMITED"
def test_awrap_wraps_any_exception_from_recognized_model(self):
"""Any exception raised inside a call to a provider-recognized
model gets wrapped — including builtins like ``RuntimeError``.
+36 -2
View File
@@ -4,6 +4,7 @@ import httpx
import pytest
from langchain_core.messages import HumanMessage
from EvoScientist.llm.contracts import EvoRuntimeError
from EvoScientist.llm.gateway_proxy import GatewayProxyChatModel
@@ -43,6 +44,18 @@ class _FakeClient:
return _FakeStream(self._lines)
def test_runtime_error_repr_preserves_only_stable_code():
error = EvoRuntimeError(
"UPSTREAM_RATE_LIMITED",
"safe display message",
details=({"provider_request": "must-not-persist"},),
)
assert repr(error) == "EvoRuntimeError(code='UPSTREAM_RATE_LIMITED')"
assert "safe display message" not in repr(error)
assert "must-not-persist" not in repr(error)
@pytest.mark.anyio
async def test_astream_yields_chunks_from_sse(monkeypatch):
model = GatewayProxyChatModel(
@@ -52,7 +65,7 @@ async def test_astream_yields_chunks_from_sse(monkeypatch):
)
msg = {"type": "AIMessageChunk", "data": {"content": "hello"}}
lines = [
f'data: {json.dumps({"delta": {"message": msg}})}\n',
f"data: {json.dumps({'delta': {'message': msg}})}\n",
'data: {"delta": {"message": {"type": "AIMessageChunk", "data": {"content": " world"}}}}\n',
"data: [DONE]\n",
]
@@ -88,7 +101,7 @@ async def test_astream_roundtrips_streaming_tool_call_chunks(monkeypatch):
],
},
}
lines = [f'data: {json.dumps({"delta": {"message": msg}})}\n', "data: [DONE]\n"]
lines = [f"data: {json.dumps({'delta': {'message': msg}})}\n", "data: [DONE]\n"]
fake = _FakeClient(lines)
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: fake)
@@ -114,3 +127,24 @@ async def test_astream_raises_on_missing_done(monkeypatch):
with pytest.raises(RuntimeError, match="AI4SCI_MODEL_STREAM_INCOMPLETE"):
_ = [c async for c in model._astream([HumanMessage(content="hi")])]
@pytest.mark.anyio
async def test_astream_projects_gateway_error_frame(monkeypatch):
model = GatewayProxyChatModel(
gateway_url="http://gw",
run_id="run-1",
envelope_signature="sig",
)
lines = [
'data: {"type":"error","code":"UPSTREAM_RATE_LIMITED",'
'"status":429,"message":"模型服务请求频率超限,请稍后重试或切换模型。"}\n'
]
fake = _FakeClient(lines)
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: fake)
with pytest.raises(EvoRuntimeError) as exc_info:
_ = [c async for c in model._astream([HumanMessage(content="hi")])]
assert exc_info.value.code == "UPSTREAM_RATE_LIMITED"
assert exc_info.value.details == ({"http_status": 429},)
+57
View File
@@ -1,6 +1,8 @@
from __future__ import annotations
import json
import sys
import types
from pathlib import Path
import pytest
@@ -21,6 +23,18 @@ def _installation(tmp_path: Path) -> sandbox.NativeSandboxInstallation:
)
def test_existing_read_paths_resolve_and_deduplicate_symlink_aliases(tmp_path: Path):
usr = tmp_path / "usr"
usr.mkdir()
bin_alias = tmp_path / "bin"
bin_alias.symlink_to(usr, target_is_directory=True)
missing = tmp_path / "missing"
paths = sandbox._existing_resolved_paths((str(usr), str(bin_alias), str(missing)))
assert paths == (str(usr.resolve()),)
def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path):
files = tmp_path / "files"
command_tmp = tmp_path / "runtime" / "tmp" / "run"
@@ -38,6 +52,7 @@ def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path
"/dev/null",
]
assert policy["filesystem"]["denyWrite"] == [
str(files / "uploads"),
"/tmp/claude",
"/private/tmp/claude",
"/dev/tty",
@@ -50,6 +65,48 @@ def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path
assert "control" not in json.dumps(policy)
def test_weaker_nested_mode_requires_explicit_environment_opt_in(tmp_path: Path, monkeypatch):
monkeypatch.delenv("EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", raising=False)
files = tmp_path / "files"
command_tmp = tmp_path / "runtime" / "tmp" / "run"
files.mkdir()
command_tmp.mkdir(parents=True)
installation = _installation(tmp_path)
assert sandbox._sandbox_settings(installation, files, command_tmp)[
"enableWeakerNestedSandbox"
] is False
monkeypatch.setenv("EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", "true")
assert sandbox._sandbox_settings(installation, files, command_tmp)[
"enableWeakerNestedSandbox"
] is True
def test_network_preflight_probe_accepts_kernel_denied_unix_socket(monkeypatch):
monkeypatch.delenv("OPENAI_API_KEY", raising=False)
class FakeSocket:
def __init__(self, family=None, *_args):
if family == 1:
raise PermissionError("blocked by seccomp")
def connect_ex(self, _address):
return 1
def close(self):
return None
fake_socket = types.SimpleNamespace(
AF_UNIX=1,
socket=lambda family=None, *args: FakeSocket(family, *args),
)
monkeypatch.setitem(sys.modules, "socket", fake_socket)
namespace: dict[str, object] = {}
exec(sandbox._network_preflight_probe(1234, Path("/blocked.sock")), namespace)
def test_clean_environment_does_not_inherit_secrets(tmp_path: Path, monkeypatch):
command_tmp = tmp_path / "tmp"
(command_tmp / "home").mkdir(parents=True)
@@ -0,0 +1,62 @@
from __future__ import annotations
import runpy
from collections.abc import Callable
from pathlib import Path
from typing import cast
import pytest
PATCH_SCRIPT = Path(__file__).parents[1] / "runtime" / "native-sandbox" / "patch_merged_usr.py"
def test_patch_skips_only_symlink_aliases_covered_by_read_allow(tmp_path: Path):
namespace = runpy.run_path(str(PATCH_SCRIPT))
patch_file = cast(Callable[[Path], None], namespace["patch_file"])
source = tmp_path / "linux-sandbox-utils.js"
source.write_text(
"""function pushReadDenyDirMounts(args, normalizedPath, allowedWritePaths, readAllowPaths) {
const denySep = normalizedPath === '/' ? '/' : normalizedPath + '/';
args.push('--tmpfs', normalizedPath);
for (const writePath of allowedWritePaths) {
if (writePath.startsWith(denySep) || writePath === normalizedPath) {
args.push('--bind', writePath, writePath);
}
}
for (const allowPath of readAllowPaths) {
if (allowPath.startsWith(denySep) || allowPath === normalizedPath) {
if (!fs.existsSync(allowPath)) {
continue;
}
if (allowedWritePaths.some(w => (w.startsWith(denySep) || w === normalizedPath) &&
(allowPath === w || allowPath.startsWith(w + '/')))) {
continue;
}
args.push('--ro-bind', allowPath, allowPath);
logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`);
}
}
}
const rootSkip = new Set(['proc', 'dev', 'sys']);
for (const p of readConfig?.denyOnly || []) {
if (normalizePathForSandbox(p) === '/') {
for (const child of fs.readdirSync('/')) {
if (!rootSkip.has(child))
readDenyPaths.push('/' + child);
}
}
""",
encoding="utf-8",
)
patch_file(source)
patched = source.read_text(encoding="utf-8")
assert "isSymbolicLink()" in patched
assert "readAllowPaths.some" in patched
assert "resolved.startsWith(allowPath + '/')" in patched
assert "args.push('--remount-ro', normalizedPath)" in patched
assert "!allowedWritePaths.includes(normalizedPath)" in patched
with pytest.raises(RuntimeError, match="already patched"):
patch_file(source)
+17
View File
@@ -38,6 +38,23 @@ class TestGetSystemPrompt:
result = get_system_prompt()
assert "Shell Execution Guidelines" in result
def test_contains_safe_archive_and_sqlite_programming_contracts(self):
result = get_system_prompt(native_web_sandbox=True)
assert "never use `extractall`" in result
assert "mode=ro&immutable=1" in result
assert "PRAGMA query_only=ON" in result
assert "Never execute files extracted from an archive" in result
assert "Do not modify original files under `uploads/`" in result
def test_distinguishes_live_web_search_from_local_memory_search(self):
result = get_system_prompt(native_web_sandbox=True)
assert "tavily_search" in result
assert "search_observations" in result
assert "local memory" in result.lower()
assert "Do not use `execute`, `curl`, or `httpx`" in result
def test_contains_delegation(self):
result = get_system_prompt()
assert "Sub-Agent Delegation" in result
+37
View File
@@ -1,5 +1,9 @@
from __future__ import annotations
import httpx
import pytest
from EvoScientist.llm.contracts import EvoRuntimeError
from EvoScientist.middleware import recoverable_tools
@@ -51,3 +55,36 @@ def test_evomemory_never_falls_back_to_parent_model_proxy(monkeypatch):
assert proxy is None
assert metadata["run_kind"] == "evomemory_turn_worker"
@pytest.mark.asyncio
async def test_tool_effect_gateway_error_preserves_machine_code(monkeypatch):
class Client:
async def __aenter__(self):
return self
async def __aexit__(self, *_):
return None
async def post(self, url, json):
return httpx.Response(
409,
json={"detail": {"code": "RUN_FENCE_LOST"}},
request=httpx.Request("POST", url),
)
monkeypatch.setattr(recoverable_tools.httpx, "AsyncClient", lambda **_: Client())
with pytest.raises(EvoRuntimeError) as exc_info:
await recoverable_tools._post(
{
"gateway_url": "http://gateway",
"run_id": "run-1",
"envelope_signature": "signature",
},
"prepare",
{},
)
assert exc_info.value.code == "RUN_FENCE_LOST"
assert repr(exc_info.value) == "EvoRuntimeError(code='RUN_FENCE_LOST')"
+173
View File
@@ -22,13 +22,19 @@ from EvoScientist.sessions import (
delete_thread,
find_similar_threads,
generate_thread_id,
get_aggregated_storage_stats,
get_db_path,
get_most_recent,
get_thread_messages,
get_thread_metadata,
list_all_session_db_paths,
list_all_thread_ids,
list_threads,
prune_all_stale_threads,
prune_thread_history,
resolve_thread_id_prefix,
thread_exists,
vacuum_db,
)
@@ -2919,5 +2925,172 @@ class TestRestoreWebuiThreadsToGlobalStore(unittest.IsolatedAsyncioTestCase):
assert restore_called, "_restore_webui_threads_to_global_store must be called"
def _uuid6_from_unix(ts_unix: float) -> str:
"""Build a UUIDv6 (time-ordered checkpoint id) from a unix timestamp.
Production checkpoint ids are UUIDv6, so lexicographic order matches
insertion order and the timestamp is recoverable from the id itself.
"""
greg = int((ts_unix + 12219292800) * 10_000_000) & ((1 << 60) - 1)
high48, low12 = greg >> 12, greg & 0xFFF
rand = uuid.uuid4().int & ((1 << 62) - 1)
value = (high48 << 80) | (0x6 << 76) | (low12 << 64) | (0b10 << 62) | rand
return str(uuid.UUID(int=value))
class TestPruneFunctions(unittest.IsolatedAsyncioTestCase):
"""Tests for the prune/vacuum API used by the gateway timer and admin routes."""
async def asyncSetUp(self):
import time
import aiosqlite
self._tmpdir = tempfile.mkdtemp()
self.db_path = os.path.join(self._tmpdir, "prune_test.db")
now = time.time()
async with aiosqlite.connect(self.db_path) as conn:
await conn.execute("""
CREATE TABLE checkpoints (
thread_id TEXT NOT NULL,
checkpoint_ns TEXT NOT NULL DEFAULT '',
checkpoint_id TEXT NOT NULL,
parent_checkpoint_id TEXT,
type TEXT,
checkpoint BLOB,
metadata TEXT NOT NULL DEFAULT '{}',
PRIMARY KEY (thread_id, checkpoint_ns, checkpoint_id)
)
""")
await conn.execute("""
CREATE TABLE writes (
thread_id TEXT NOT NULL,
checkpoint_ns TEXT NOT NULL DEFAULT '',
checkpoint_id TEXT NOT NULL,
task_id TEXT NOT NULL,
idx INTEGER NOT NULL,
channel TEXT NOT NULL,
type TEXT,
value BLOB,
PRIMARY KEY (thread_id, checkpoint_ns, checkpoint_id, task_id, idx)
)
""")
await self._insert_thread(conn, "old_thread", 5, now - 10 * 86400)
await self._insert_thread(conn, "new_thread", 3, now - 60)
await self._insert_thread(
conn, "other", 4, now - 10 * 86400, agent="OtherAgent"
)
await conn.commit()
async def asyncTearDown(self):
try:
os.unlink(self.db_path)
os.rmdir(self._tmpdir)
except OSError:
pass
async def _insert_thread(self, conn, tid, count, ts_base, agent=AGENT_NAME):
serde = JsonPlusSerializer()
ctype, cblob = serde.dumps_typed(
{"channel_values": {"messages": [HumanMessage(content=f"seed-{tid}")]}}
)
prev = None
for i in range(count):
cid = _uuid6_from_unix(ts_base + i)
await conn.execute(
"INSERT INTO checkpoints (thread_id, checkpoint_ns, checkpoint_id,"
" parent_checkpoint_id, type, checkpoint, metadata)"
" VALUES (?, '', ?, ?, ?, ?, ?)",
(tid, cid, prev, ctype, cblob, json.dumps({"agent_name": agent})),
)
await conn.execute(
"INSERT INTO writes (thread_id, checkpoint_ns, checkpoint_id,"
" task_id, idx, channel, type, value)"
" VALUES (?, '', ?, 'task', 0, 'ch', 'str', ?)",
(tid, cid, b"x"),
)
prev = cid
async def _count(self, tid, table="checkpoints"):
import aiosqlite
async with aiosqlite.connect(self.db_path) as conn:
async with conn.execute(
f"SELECT COUNT(*) FROM {table} WHERE thread_id = ?", (tid,)
) as cur:
return (await cur.fetchone())[0]
async def test_prune_thread_history(self):
result = await prune_thread_history(
"old_thread", keep_last=2, db_path=self.db_path
)
# keep_last=2 anchors + 1 snapshot-seed ancestor preserved
assert result == {"deleted_checkpoints": 2, "deleted_writes": 2}
assert await self._count("old_thread") == 3
assert await self._count("old_thread", "writes") == 3
async def test_prune_thread_history_other_agent_untouched(self):
result = await prune_thread_history("other", keep_last=1, db_path=self.db_path)
assert result == {"deleted_checkpoints": 0, "deleted_writes": 0}
assert await self._count("other") == 4
async def test_prune_all_stale_threads(self):
result = await prune_all_stale_threads(
max_age_hours=24, keep_last=2, db_path=self.db_path
)
assert result["databases_processed"] == 1
assert result["threads_pruned"] == 1
assert result["total_deleted_checkpoints"] == 2
assert result["total_deleted_writes"] == 2
# fresh thread and foreign-agent thread untouched
assert await self._count("new_thread") == 3
assert await self._count("other") == 4
async def test_prune_all_stale_threads_none_stale(self):
result = await prune_all_stale_threads(
max_age_hours=24 * 365, keep_last=2, db_path=self.db_path
)
assert result["threads_pruned"] == 0
assert result["total_deleted_checkpoints"] == 0
assert await self._count("old_thread") == 5
async def test_list_all_thread_ids(self):
ids = await list_all_thread_ids(db_path=self.db_path)
assert sorted(ids) == ["new_thread", "old_thread"]
async def test_vacuum_db(self):
result = await vacuum_db(db_path=self.db_path)
assert result["size_after_bytes"] > 0
assert result["size_before_bytes"] >= result["size_after_bytes"]
async def test_get_aggregated_storage_stats(self):
with patch(
"EvoScientist.sessions.get_db_path",
return_value=_mock_path(self.db_path),
):
stats = await get_aggregated_storage_stats()
assert stats["thread_count"] == 2
assert stats["checkpoint_count"] == 8
assert stats["thread_depth"]["max"] == 5
assert stats["thread_depth"]["min"] == 3
async def test_list_all_session_db_paths(self):
with patch(
"EvoScientist.sessions.get_db_path",
return_value=_mock_path(self.db_path),
):
paths = list_all_session_db_paths()
assert len(paths) == 1
assert str(paths[0]) == self.db_path
missing = os.path.join(self._tmpdir, "nope.db")
with patch(
"EvoScientist.sessions.get_db_path",
return_value=_mock_path(missing),
):
assert list_all_session_db_paths() == []
if __name__ == "__main__":
unittest.main()
+174
View File
@@ -0,0 +1,174 @@
from __future__ import annotations
import asyncio
from unittest.mock import MagicMock
import pytest
from langchain_core.messages import ToolMessage
from EvoScientist.middleware.subagent_timeout import SubagentTimeoutMiddleware
def _request(name: str = "task"):
request = MagicMock()
request.tool_call = {"id": "call-1", "name": name, "args": {}}
return request
@pytest.mark.anyio
async def test_subagent_timeout_returns_stable_tool_error():
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
async def handler(_request):
await asyncio.sleep(1)
return ToolMessage(content="late", tool_call_id="call-1", name="task")
result = await middleware.awrap_tool_call(_request(), handler)
assert isinstance(result, ToolMessage)
assert result.status == "error"
assert result.name == "task"
assert result.additional_kwargs["error_code"] == "SUBAGENT_TIMEOUT"
assert "SUBAGENT_TIMEOUT" in result.content
@pytest.mark.anyio
async def test_subagent_timeout_passes_success_through():
middleware = SubagentTimeoutMiddleware(timeout_seconds=1)
expected = ToolMessage(content="done", tool_call_id="call-1", name="task")
async def handler(_request):
return expected
assert await middleware.awrap_tool_call(_request(), handler) is expected
@pytest.mark.anyio
async def test_subagent_timeout_does_not_bound_other_tools():
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
expected = ToolMessage(content="done", tool_call_id="call-1", name="read_file")
async def handler(_request):
await asyncio.sleep(0.02)
return expected
assert await middleware.awrap_tool_call(_request("read_file"), handler) is expected
@pytest.mark.anyio
async def test_subagent_internal_timeout_error_is_not_reclassified():
middleware = SubagentTimeoutMiddleware(timeout_seconds=1)
async def handler(_request):
raise TimeoutError("provider timed out immediately")
with pytest.raises(TimeoutError, match="provider timed out immediately"):
await middleware.awrap_tool_call(_request(), handler)
@pytest.mark.anyio
async def test_parent_cancellation_cancels_subagent_handler():
middleware = SubagentTimeoutMiddleware(timeout_seconds=10)
handler_cancelled = asyncio.Event()
async def handler(_request) -> ToolMessage:
try:
await asyncio.Event().wait()
finally:
handler_cancelled.set()
return ToolMessage(content="done", tool_call_id="call-1", name="task")
invocation = asyncio.create_task(
middleware.awrap_tool_call(_request(), handler)
)
await asyncio.sleep(0)
invocation.cancel()
with pytest.raises(asyncio.CancelledError):
await invocation
assert handler_cancelled.is_set()
@pytest.mark.anyio
async def test_parent_cancellation_wins_over_handler_cleanup_error():
middleware = SubagentTimeoutMiddleware(timeout_seconds=10)
async def handler(_request) -> ToolMessage:
try:
await asyncio.Event().wait()
except asyncio.CancelledError as exc:
raise RuntimeError("cleanup failed") from exc
return ToolMessage(content="done", tool_call_id="call-1", name="task")
invocation = asyncio.create_task(
middleware.awrap_tool_call(_request(), handler)
)
await asyncio.sleep(0)
invocation.cancel()
with pytest.raises(asyncio.CancelledError):
await invocation
@pytest.mark.anyio
async def test_deadline_wins_over_handler_cleanup_error():
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
async def handler(_request) -> ToolMessage:
try:
await asyncio.Event().wait()
except asyncio.CancelledError as exc:
raise RuntimeError("cleanup failed") from exc
return ToolMessage(content="done", tool_call_id="call-1", name="task")
result = await middleware.awrap_tool_call(_request(), handler)
assert isinstance(result, ToolMessage)
assert result.additional_kwargs["error_code"] == "SUBAGENT_TIMEOUT"
@pytest.mark.anyio
async def test_parent_cancellation_during_deadline_cleanup_is_not_swallowed():
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
cleanup_started = asyncio.Event()
release_cleanup = asyncio.Event()
async def handler(_request) -> ToolMessage:
try:
await asyncio.Event().wait()
except asyncio.CancelledError:
cleanup_started.set()
await release_cleanup.wait()
raise
return ToolMessage(content="done", tool_call_id="call-1", name="task")
invocation = asyncio.create_task(
middleware.awrap_tool_call(_request(), handler)
)
await cleanup_started.wait()
invocation.cancel()
release_cleanup.set()
with pytest.raises(asyncio.CancelledError):
await invocation
@pytest.mark.anyio
async def test_parent_cancellation_after_cleanup_before_timeout_return_wins(monkeypatch):
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
async def handler(_request) -> ToolMessage:
await asyncio.Event().wait()
return ToolMessage(content="done", tool_call_id="call-1", name="task")
async def finish_cleanup_then_cancel_parent(_task):
current = asyncio.current_task()
assert current is not None
current.cancel()
monkeypatch.setattr(middleware, "_cancel_task", finish_cleanup_then_cancel_parent)
invocation = asyncio.create_task(
middleware.awrap_tool_call(_request(), handler)
)
with pytest.raises(asyncio.CancelledError):
await invocation
+132
View File
@@ -0,0 +1,132 @@
from __future__ import annotations
import pytest
@pytest.mark.anyio
async def test_tavily_search_keeps_indexed_summary_when_source_fetch_fails(monkeypatch):
from EvoScientist.tools import search
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": "AIR staff profile",
"url": "https://air.cas.cn/example",
"content": "Indexed staff-profile summary.",
}
]
}
recorded: list[tuple[str, str]] = []
async def fetch_failed(_url: str, timeout: float = 10.0) -> str:
return "Error fetching content from https://air.cas.cn/example: DNS failed"
async def record(service: str, action: str) -> None:
recorded.append((service, action))
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", fetch_failed)
monkeypatch.setattr("EvoScientist.runtime_integrations.record_service_usage", record)
result = await search.tavily_search.ainvoke({"query": "高铭 空天院"})
assert "Indexed staff-profile summary." in result
assert "https://air.cas.cn/example" in result
assert "Tavily-indexed summary" in result
assert recorded == [("tavily", "search")]
@pytest.mark.anyio
async def test_tavily_search_bounds_fetched_page_content(monkeypatch):
from EvoScientist.tools import search
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": f"Result {index}",
"url": f"https://example.com/{index}",
"content": f"Indexed summary {index}",
}
for index in range(3)
]
}
async def huge_page(_url: str, timeout: float = 10.0) -> str:
return "page-content " * 10_000
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", huge_page)
result = await search.tavily_search.ainvoke({"query": "bounded search"})
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
for index in range(3):
assert f"https://example.com/{index}" in result
assert "[page content truncated]" in result
@pytest.mark.anyio
async def test_tavily_search_preserves_every_result_url_under_total_budget(monkeypatch):
from EvoScientist.tools import search
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": f"Result {index} " + ("very-long-title " * 800),
"url": f"https://example.com/result-{index}",
"content": f"Indexed summary {index}",
}
for index in range(3)
]
}
async def page(_url: str, timeout: float = 10.0) -> str:
return "page-content " * 1_000
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", page)
result = await search.tavily_search.ainvoke({"query": "preserve urls"})
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
for index in range(3):
assert f"https://example.com/result-{index}" in result
assert "[search result content truncated to preserve all titles and URLs]" in result
@pytest.mark.anyio
async def test_tavily_search_bounds_maliciously_long_url(monkeypatch):
from EvoScientist.tools import search
long_url = "https://example.com/" + ("a" * 20_000)
class _Client:
def search(self, *_args, **_kwargs):
return {
"results": [
{
"title": "Long URL result",
"url": long_url,
"content": "Indexed summary",
}
]
}
async def page(_url: str, timeout: float = 10.0) -> str:
return "page"
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
monkeypatch.setattr(search, "fetch_webpage_content", page)
result = await search.tavily_search.ainvoke({"query": "long url"})
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
assert "https://example.com/" in result
assert "[URL truncated]" in result
+178
View File
@@ -6,6 +6,184 @@ from EvoScientist.llm.contracts import EvoRuntimeError
from EvoScientist.web_runtime import _ToolRegistryFenceMiddleware
def test_web_registry_describes_tavily_as_controlled_live_search(monkeypatch):
monkeypatch.setenv("TAVILY_API_KEY", "test-key")
from EvoScientist.web_runtime import web_tool_registry_manifest
manifest, _revision = web_tool_registry_manifest()
tavily = next(item for item in manifest if item["name"] == "tavily_search")
assert "live public web" in tavily["description"].lower()
assert "execute" in tavily["description"].lower()
def test_base_kwargs_install_tavily_on_main_agent(monkeypatch):
import EvoScientist.EvoScientist as agent_module
monkeypatch.setenv("TAVILY_API_KEY", "test-key")
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
monkeypatch.setattr("EvoScientist.utils.load_subagents", lambda *_args, **_kwargs: [])
kwargs = agent_module._build_base_kwargs(
object(),
[],
cfg=object(),
chat_model=object(),
workspace_dir="/workspace",
)
assert "tavily_search" in {getattr(tool, "name", "") for tool in kwargs["tools"]}
def _stub_agent_build(monkeypatch, agent_module, subagents=None):
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
monkeypatch.setattr(
"EvoScientist.utils.load_subagents",
lambda *_args, **_kwargs: list(subagents or []),
)
@pytest.mark.parametrize(
"reserved_name",
[
"skill_manager",
"execute",
"start_async_task",
"check_async_task",
"update_async_task",
"cancel_async_task",
"list_async_tasks",
],
)
def test_mcp_cannot_override_reserved_tool(monkeypatch, reserved_name):
from types import SimpleNamespace
import EvoScientist.EvoScientist as agent_module
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
monkeypatch.setattr(
agent_module,
"_load_mcp_tools_cached",
lambda **_kwargs: {"main": [SimpleNamespace(name=reserved_name)]},
)
_stub_agent_build(monkeypatch, agent_module)
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"):
agent_module.load_mcp_and_build_kwargs(
object(),
[],
cfg=object(),
chat_model=object(),
workspace_dir="/workspace",
)
def test_mcp_cannot_duplicate_existing_subagent_tool(monkeypatch):
from types import SimpleNamespace
import EvoScientist.EvoScientist as agent_module
existing = SimpleNamespace(name="shared_search")
injected = SimpleNamespace(name="shared_search")
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
monkeypatch.setattr(
agent_module,
"_load_mcp_tools_cached",
lambda **_kwargs: {"research-agent": [injected]},
)
_stub_agent_build(
monkeypatch,
agent_module,
subagents=[{"name": "research-agent", "tools": [existing]}],
)
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"):
agent_module.load_mcp_and_build_kwargs(
object(),
[],
cfg=object(),
chat_model=object(),
workspace_dir="/workspace",
)
def test_mcp_cannot_override_builtin_tavily(monkeypatch):
from types import SimpleNamespace
import EvoScientist.EvoScientist as agent_module
monkeypatch.setenv("TAVILY_API_KEY", "test-key")
monkeypatch.setattr(
agent_module,
"_load_mcp_tools_cached",
lambda **_kwargs: {"main": [SimpleNamespace(name="tavily_search")]},
)
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
monkeypatch.setattr("EvoScientist.utils.load_subagents", lambda *_args, **_kwargs: [])
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"):
agent_module.load_mcp_and_build_kwargs(
object(),
[],
cfg=object(),
chat_model=object(),
workspace_dir="/workspace",
)
def test_same_mcp_tool_can_be_routed_to_multiple_agents(monkeypatch):
from types import SimpleNamespace
import EvoScientist.EvoScientist as agent_module
shared_main = SimpleNamespace(name="shared_search")
shared_research = SimpleNamespace(name="shared_search")
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
monkeypatch.setattr(
agent_module,
"_load_mcp_tools_cached",
lambda **_kwargs: {
"main": [shared_main],
"research-agent": [shared_research],
},
)
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
monkeypatch.setattr(
"EvoScientist.utils.load_subagents",
lambda *_args, **_kwargs: [
{"name": "research-agent", "tools": []}
],
)
kwargs = agent_module.load_mcp_and_build_kwargs(
object(),
[],
cfg=object(),
chat_model=object(),
workspace_dir="/workspace",
)
assert shared_main in kwargs["tools"]
research = next(
subagent for subagent in kwargs["subagents"]
if subagent["name"] == "research-agent"
)
assert shared_research in research["tools"]
def test_tool_dispatch_fence_rejects_changed_registry(monkeypatch):
monkeypatch.setattr(
"EvoScientist.web_runtime.web_tool_registry_manifest",
+374 -16
View File
@@ -2,10 +2,14 @@ from __future__ import annotations
import base64
import os
import sqlite3
import zipfile
from io import BytesIO
from pathlib import Path
import pytest
from deepagents.backends.protocol import ExecuteResponse
from PIL import Image
from EvoScientist.native_sandbox import (
NativeSandboxExecutor,
@@ -27,6 +31,19 @@ from EvoScientist.workspace_scope import (
)
def test_extracted_documents_route_through_deepagents_as_text():
import deepagents.middleware.filesystem as filesystem_middleware
from EvoScientist.llm.patches import _patch_deepagents_extracted_document_text
_patch_deepagents_extracted_document_text()
get_file_type = vars(filesystem_middleware)["_get_file_type"]
assert get_file_type("/workspace/report.pptx") == "text"
assert get_file_type("/workspace/report.pdf") == "text"
assert get_file_type("/workspace/image.png") == "image"
def test_normalize_workspace_path_is_strict():
assert normalize_workspace_path("/workspace") == ()
assert normalize_workspace_path("/workspace/reports/a.txt") == (
@@ -82,25 +99,366 @@ def test_root_lists_workspace_namespace(tmp_path: Path):
]
def test_docx_and_unknown_binary_read_with_base64_contract(tmp_path: Path):
def test_docx_read_extracts_text_instead_of_returning_base64(tmp_path: Path):
backend = ScopedFilesystemBackend(tmp_path)
docx = b"PK\x03\x04\x00word/document.xml"
unknown = b"custom\x00binary"
backend.upload_files(
[
("/workspace/input.docx", docx),
("/workspace/payload.custom", unknown),
]
docx = tmp_path / "input.docx"
with zipfile.ZipFile(docx, "w") as archive:
archive.writestr(
"word/document.xml",
"""<?xml version="1.0" encoding="UTF-8"?>
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>有界文档内容</w:t></w:r></w:p></w:body>
</w:document>""",
)
result = backend.read("/workspace/input.docx")
assert result.error is None
assert result.file_data is not None
assert result.file_data["encoding"] == "utf-8"
assert "有界文档内容" in result.file_data["content"]
assert "base64" not in result.file_data["content"]
@pytest.mark.parametrize(
("filename", "kind", "expected"),
[
("archive.zip", "archive", "list"),
("results.sqlite", "database", "mode=ro"),
("program.exe", "executable", "must not be executed"),
("payload.custom", "binary", "unsupported"),
],
)
def test_non_document_binary_returns_bounded_processing_guidance(
tmp_path: Path, filename: str, kind: str, expected: str
):
if filename.endswith(".zip"):
with zipfile.ZipFile(tmp_path / filename, "w") as archive:
archive.writestr("notes.txt", "hello")
elif filename.endswith(".sqlite"):
connection = sqlite3.connect(tmp_path / filename)
connection.execute("CREATE TABLE results(id INTEGER PRIMARY KEY, value TEXT)")
connection.commit()
connection.close()
elif filename.endswith(".exe"):
(tmp_path / filename).write_bytes(b"MZ\x00binary payload")
else:
(tmp_path / filename).write_bytes(b"custom\x00binary payload")
result = ScopedFilesystemBackend(tmp_path).read(f"/workspace/{filename}")
assert result.file_data is None
assert result.error is not None
assert "BINARY_PROCESSING_REQUIRED" in result.error or "UNSUPPORTED_BINARY_FILE" in result.error
assert f'"kind": "{kind}"' in result.error
assert expected.lower() in result.error.lower()
assert len(result.error) < 4000
def test_image_read_keeps_base64_media_contract(tmp_path: Path):
buffer = BytesIO()
Image.new("RGB", (32, 24), (1, 2, 3)).save(buffer, "PNG")
raw = buffer.getvalue()
(tmp_path / "image.png").write_bytes(raw)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/image.png")
assert result.error is None
assert result.file_data is not None
assert result.file_data["encoding"] == "base64"
def test_large_image_is_downsampled_before_base64_delivery(tmp_path: Path):
Image.new("RGB", (3000, 1200), (1, 2, 3)).save(tmp_path / "large.png", "PNG")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/large.png")
assert result.error is None
assert result.file_data is not None
decoded = base64.standard_b64decode(result.file_data["content"])
image = Image.open(BytesIO(decoded))
image.load()
assert image.size == (2048, 819)
assert image.format == "JPEG"
def test_corrupt_image_returns_error_instead_of_base64(tmp_path: Path):
(tmp_path / "broken.png").write_bytes(b"\x89PNG\r\nnot-decodable")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.png")
assert result.file_data is None
assert result.error is not None
assert "IMAGE_PROCESSING_FAILED" in result.error
def test_image_pixel_budget_is_enforced_before_model_delivery(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
import EvoScientist.document_extract as document_extract
Image.new("RGB", (100, 100), (1, 2, 3)).save(tmp_path / "pixels.png", "PNG")
monkeypatch.setattr(document_extract, "MAX_IMAGE_PIXELS", 9_999)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/pixels.png")
assert result.file_data is None
assert result.error is not None
assert "IMAGE_PIXEL_BUDGET_EXCEEDED" in result.error
def test_multiframe_image_is_reduced_to_first_frame(tmp_path: Path):
frames = [Image.new("RGB", (20, 10), color) for color in ((255, 0, 0), (0, 255, 0))]
frames[0].save(
tmp_path / "animated.gif",
format="GIF",
save_all=True,
append_images=frames[1:],
duration=100,
loop=0,
)
assert backend.read("/workspace/input.docx").file_data == {
"content": base64.standard_b64encode(docx).decode("ascii"),
"encoding": "base64",
}
assert backend.read("/workspace/payload.custom").file_data == {
"content": base64.standard_b64encode(unknown).decode("ascii"),
"encoding": "base64",
}
result = ScopedFilesystemBackend(tmp_path).read("/workspace/animated.gif")
assert result.error is None
assert result.file_data is not None
decoded = base64.standard_b64decode(result.file_data["content"])
image = Image.open(BytesIO(decoded))
image.load()
assert getattr(image, "n_frames", 1) == 1
def test_pptx_read_extracts_slide_text_without_base64(tmp_path: Path):
with zipfile.ZipFile(tmp_path / "deck.pptx", "w") as archive:
archive.writestr(
"ppt/slides/slide1.xml",
"""<p:sld xmlns:p="http://schemas.openxmlformats.org/presentationml/2006/main"
xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
<p:cSld><a:t>总体技术架构</a:t><a:t>核心能力说明</a:t></p:cSld>
</p:sld>""",
)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/deck.pptx")
assert result.error is None
assert result.file_data is not None
assert result.file_data["encoding"] == "utf-8"
assert "## Slide 1" in result.file_data["content"]
assert "总体技术架构" in result.file_data["content"]
def test_document_output_is_bounded_and_returns_continuation_hint(tmp_path: Path):
long_text = "\n".join(f"第{i:05d}行-" + "x" * 80 for i in range(2000))
with zipfile.ZipFile(tmp_path / "long.docx", "w") as archive:
paragraphs = "".join(
f"<w:p><w:r><w:t>{line}</w:t></w:r></w:p>"
for line in long_text.splitlines()
)
archive.writestr(
"word/document.xml",
f"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body>{paragraphs}</w:body></w:document>""",
)
result = ScopedFilesystemBackend(tmp_path).read(
"/workspace/long.docx", offset=0, limit=2000
)
assert result.error is None
assert result.file_data is not None
content = result.file_data["content"]
assert len(content) < 51_000
assert "DOCUMENT_OUTPUT_TRUNCATED" in content
assert "use offset=" in content
def test_corrupt_document_does_not_fall_back_to_base64(tmp_path: Path):
(tmp_path / "broken.pptx").write_bytes(b"PK\x03\x04not-a-real-presentation")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.pptx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_EXTRACTION_FAILED" in result.error
assert "base64" not in result.error.lower()
def test_ooxml_member_expansion_budget_blocks_compression_bomb(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
import EvoScientist.document_extract as document_extract
with zipfile.ZipFile(
tmp_path / "bomb.docx", "w", compression=zipfile.ZIP_DEFLATED
) as archive:
archive.writestr(
"word/document.xml",
"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>expanded content</w:t></w:r></w:p></w:body>
</w:document>""",
)
monkeypatch.setattr(document_extract, "MAX_OOXML_MEMBER_BYTES", 16)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/bomb.docx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
@pytest.mark.parametrize("member", ["../word/document.xml", "/word/document.xml"])
def test_ooxml_rejects_unsafe_member_paths(tmp_path: Path, member: str):
with zipfile.ZipFile(tmp_path / "unsafe.docx", "w") as archive:
archive.writestr(member, "content")
archive.writestr(
"word/document.xml",
"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>safe</w:t></w:r></w:p></w:body></w:document>""",
)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/unsafe.docx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
def test_ooxml_rejects_duplicate_member_names(tmp_path: Path):
def write_duplicate_document(path: Path) -> None:
with zipfile.ZipFile(path, "w") as archive:
for text in ("first", "second"):
archive.writestr(
"word/document.xml",
f"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>{text}</w:t></w:r></w:p></w:body></w:document>""",
)
with pytest.warns(UserWarning, match="Duplicate name"):
write_duplicate_document(tmp_path / "duplicate.docx")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/duplicate.docx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
def test_oversized_document_is_rejected_before_opening_content(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
from EvoScientist.document_extract import MAX_DOCUMENT_BYTES
path = tmp_path / "oversized.pdf"
path.write_bytes(b"%PDF")
original_entry = RootedWorkspace.entry
def oversized_entry(self, virtual_path):
entry = original_entry(self, virtual_path)
return type(entry)(entry.virtual_path, entry.is_dir, MAX_DOCUMENT_BYTES + 1, entry.modified_at)
monkeypatch.setattr(RootedWorkspace, "entry", oversized_entry)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/oversized.pdf")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_TOO_LARGE" in result.error
def test_external_document_converter_timeout_is_structured(
monkeypatch: pytest.MonkeyPatch,
):
import subprocess
import EvoScientist.document_extract as document_extract
def timeout(*args, **kwargs):
raise subprocess.TimeoutExpired(cmd="anydoc", timeout=60)
monkeypatch.setattr(document_extract.subprocess, "run", timeout)
with pytest.raises(
document_extract.DocumentExtractionError,
match="DOCUMENT_CONVERSION_TIMEOUT",
):
document_extract.extract_document_bytes(b"%PDF-minimal", "sample.pdf")
@pytest.mark.parametrize("filename", ["report.pptx", "archive.zip", "results.sqlite"])
def test_text_write_cannot_create_or_corrupt_binary_container(
tmp_path: Path, filename: str
):
backend = ScopedFilesystemBackend(tmp_path)
created = backend.write(f"/workspace/{filename}", "extracted text")
assert created.error is not None
assert "binary container" in created.error.lower()
assert not (tmp_path / filename).exists()
def test_text_edit_cannot_modify_existing_binary_container(tmp_path: Path):
source = b"PK\x03\x04original-container"
(tmp_path / "report.pptx").write_bytes(source)
backend = ScopedFilesystemBackend(tmp_path)
edited = backend.edit(
"/workspace/report.pptx", "original", "replacement"
)
assert edited.error is not None
assert "binary container" in edited.error.lower()
assert (tmp_path / "report.pptx").read_bytes() == source
def test_uploaded_text_is_read_only_to_file_tools(tmp_path: Path):
uploads = tmp_path / "uploads"
uploads.mkdir()
source = uploads / "notes.txt"
source.write_text("original", encoding="utf-8")
backend = ScopedFilesystemBackend(tmp_path)
written = backend.write("/workspace/uploads/new.txt", "new")
edited = backend.edit("/workspace/uploads/notes.txt", "original", "changed")
assert written.error is not None
assert "uploads" in written.error.lower()
assert edited.error is not None
assert "uploads" in edited.error.lower()
assert not (uploads / "new.txt").exists()
assert source.read_text(encoding="utf-8") == "original"
def test_utf8_sample_boundary_cut_is_not_binary(tmp_path: Path):
# Regression (2026-08-22): the 8192-byte UTF-8 probe can split a multi-byte
# CJK character at the sample boundary (req_v13.md cut at 8190/8191 split a
# 3-byte char). That raised UnicodeDecodeError -> misclassified as binary
# -> read_file returned a base64 file media block -> providers without file
# input replaced it with a placeholder -> the model retried forever.
backend = ScopedFilesystemBackend(tmp_path)
# 8190 ASCII bytes + one 3-byte CJK char, so the sample cuts mid-character.
content = ("a" * 8190 + "\u6e56" + "more text").encode("utf-8")
assert len(content) > 8192
assert content[8190:8193] == "\u6e56".encode("utf-8")
backend.upload_files([("/workspace/cjk.md", content)])
result = backend.read("/workspace/cjk.md")
assert result.file_data is not None
assert result.file_data["encoding"] == "utf-8"
assert result.file_data["content"].startswith("a" * 10)
def test_mid_sample_invalid_bytes_still_binary():
from EvoScientist.workspace_files import _is_binary_file
# Invalid bytes well inside the sample are genuine garbage, not a cut.
assert _is_binary_file("/workspace/bad.raw", b"ok\xffi\xffd\xefmore")
# A boundary cut (error in the last 4 bytes that decodes clean when the
# dangling suffix is dropped) is text.
cut = ("a" * 8190 + "\u6e56").encode("utf-8")[:8192]
assert not _is_binary_file("/workspace/cut.md", cut)
# Same shape but the prefix itself is invalid -> stays binary.
assert _is_binary_file("/workspace/bad.md", b"\xff" * 8192)
def test_symlink_targets_and_parents_are_rejected(tmp_path: Path):
Generated
+104 -1
View File
@@ -944,11 +944,11 @@ wheels = [
[[package]]
name = "evoscientist"
version = "0.2.2"
source = { editable = "." }
dependencies = [
{ name = "deepagents", extra = ["quickjs"] },
{ name = "filelock" },
{ name = "firecrawl-anydoc" },
{ name = "httpx" },
{ name = "langchain" },
{ name = "langchain-anthropic" },
@@ -963,6 +963,7 @@ dependencies = [
{ name = "lazy-loader" },
{ name = "markdownify" },
{ name = "nest-asyncio" },
{ name = "pillow" },
{ name = "prompt-toolkit" },
{ name = "psutil" },
{ name = "python-dotenv" },
@@ -1056,6 +1057,7 @@ requires-dist = [
{ name = "discord-py", marker = "extra == 'discord'", specifier = ">=2.3" },
{ name = "faster-whisper", marker = "extra == 'stt'", specifier = ">=1.0" },
{ name = "filelock", specifier = ">=3.16" },
{ name = "firecrawl-anydoc", specifier = ">=0.1.6,<0.2" },
{ name = "httpx", specifier = ">=0.28" },
{ name = "langchain", specifier = ">=1.3" },
{ name = "langchain-anthropic", specifier = ">=1.4" },
@@ -1072,6 +1074,7 @@ requires-dist = [
{ name = "lazy-loader", specifier = ">=0.5" },
{ name = "markdownify", specifier = ">=1.2" },
{ name = "nest-asyncio", specifier = ">=1.6" },
{ name = "pillow", specifier = ">=10.0" },
{ name = "pre-commit", marker = "extra == 'dev'", specifier = ">=3.5.0" },
{ name = "prompt-toolkit", specifier = ">=3.0" },
{ name = "psutil", specifier = ">=6.0" },
@@ -1317,6 +1320,21 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/18/79/1b8fa1bb3568781e84c9200f951c735f3f157429f44be0495da55894d620/filetype-1.2.0-py2.py3-none-any.whl", hash = "sha256:7ce71b6880181241cf7ac8697a2f1eb6a8bd9b429f7ad6d27b8db9ba5f1c2d25", size = 19970, upload-time = "2022-11-02T17:34:01.425Z" },
]
[[package]]
name = "firecrawl-anydoc"
version = "0.1.9"
source = { registry = "https://pypi.org/simple" }
sdist = { url = "https://files.pythonhosted.org/packages/aa/16/e37d4f284482a4f30e0502ce439a9fd8ccf013e6ae4438b4dfa209e48750/firecrawl_anydoc-0.1.9.tar.gz", hash = "sha256:0dfc64b82b4e971143dd6f0e4f39f8152fcc38be6523663682a241d8f3301b14", size = 197914, upload-time = "2026-08-13T21:48:27.989Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/42/37/866bf6270ad8552eaf6edf3b8e1a360acf3e8764140641361ae7b440514e/firecrawl_anydoc-0.1.9-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:ebd956831822d3ea1831253406736e0e4d9b217e3471eb68c9bf2eba9302ce73", size = 3469321, upload-time = "2026-08-13T21:48:16.775Z" },
{ url = "https://files.pythonhosted.org/packages/b1/9a/8fbe0726cdf69154e1eb05c119332b1fdfd8200eeafcfd1675741ac5f03d/firecrawl_anydoc-0.1.9-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:a2372d97826e8e7e68fd28cd415d08815f6aa66dcc09c6b423ac33dd2dfa91cf", size = 3287520, upload-time = "2026-08-13T21:48:18.37Z" },
{ url = "https://files.pythonhosted.org/packages/b0/ce/86a626224c365d8de65699d6ac8f4626ac05e7e75c3ae76c82f4904a07d2/firecrawl_anydoc-0.1.9-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ba1c19bdaa97f4850daf3fb4ca2855cefa146b5621ce7b22a71d158d3b84e54e", size = 3331094, upload-time = "2026-08-13T21:48:19.961Z" },
{ url = "https://files.pythonhosted.org/packages/0e/28/00d1d48fe205ab66eda4a2a22cce44aadf586ef1495e703349b791c8de32/firecrawl_anydoc-0.1.9-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:deea7a270fccda9be29c73235e938ca32e0f703e1ae120f08bf0fdb39301c6b4", size = 3555945, upload-time = "2026-08-13T21:48:21.822Z" },
{ url = "https://files.pythonhosted.org/packages/c8/25/ac52d98c5d082ef92396b2fcb398745a3b054816c67a12d19de86c1ba535/firecrawl_anydoc-0.1.9-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:de0858decd18188b0b88545689544fb3703ec08ade37f63c2857e4414e99bdf1", size = 3510861, upload-time = "2026-08-13T21:48:23.369Z" },
{ url = "https://files.pythonhosted.org/packages/ca/d3/e80314b5746c4ca85150f3ab2a527b2167012c16e706e148733211cb88b3/firecrawl_anydoc-0.1.9-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:e4e1765ed5574fa502931f8bfaec1f87af913afa5bb432aaaf5f9c094efc7f43", size = 3797896, upload-time = "2026-08-13T21:48:25.061Z" },
{ url = "https://files.pythonhosted.org/packages/8e/d0/e7b35c2365498d5e9f1bdb4b3ded78571b6dafbaba1880538490d27b71c9/firecrawl_anydoc-0.1.9-cp310-abi3-win_amd64.whl", hash = "sha256:aa6a5ca2e10939a87c9bd21c918ef8b146ad0061fedae661d4483ef5a5bbdf60", size = 3642084, upload-time = "2026-08-13T21:48:26.702Z" },
]
[[package]]
name = "flatbuffers"
version = "25.12.19"
@@ -3064,6 +3082,91 @@ wheels = [
{ url = "https://files.pythonhosted.org/packages/f1/d9/7fb5aa316bc299258e68c73ba3bddbc499654a07f151cba08f6153988714/pathspec-1.1.1-py3-none-any.whl", hash = "sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189", size = 57328, upload-time = "2026-04-27T01:46:07.06Z" },
]
[[package]]
name = "pillow"
version = "12.3.0"
source = { registry = "https://pypi.org/simple" }
sdist = { url = "https://files.pythonhosted.org/packages/1c/3d/bb7fca845737cf9d7dbde16ed1843984665ff2e0a518f5db43e77ec540b9/pillow-12.3.0.tar.gz", hash = "sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce", size = 47025035, upload-time = "2026-07-01T11:56:38.965Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/fb/c8/0a78b0e02d7ac54bc03e5321c9220da52f0c2ea83b21f7c40e7f3169c502/pillow-12.3.0-cp311-cp311-macosx_10_10_x86_64.whl", hash = "sha256:00808c5e14ef63ac5161091d242999076604ff74b883423a11e5d7bbb38bf756", size = 5392415, upload-time = "2026-07-01T11:53:47.162Z" },
{ url = "https://files.pythonhosted.org/packages/b2/5b/a02d30018abd97ced9f5a6c63d28597694a00d066516b9c1c6de45859fc9/pillow-12.3.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:37d6d0a00072fd2948eb22bce7e1475f34569d90c87c59f7a2ec59541b77f7a6", size = 4785266, upload-time = "2026-07-01T11:53:49.079Z" },
{ url = "https://files.pythonhosted.org/packages/c8/98/766667a4be768150a202836acd9fad19c06824ca86c4286d3cf6b274964e/pillow-12.3.0-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:bcb46e2f9feff8d06323983bd83ed00c201fdcab3d74973e7072a889b3979fcd", size = 6263814, upload-time = "2026-07-01T11:53:51.32Z" },
{ url = "https://files.pythonhosted.org/packages/3b/2d/ede717bc1144f63886c21fd349bb95860b0d1a21149ff16f2bb362b612b6/pillow-12.3.0-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:23d27a3e0307ec2244cc51e7287b919aa68d097504ebe19df4e76a98a3eea5bd", size = 6934408, upload-time = "2026-07-01T11:53:53.487Z" },
{ url = "https://files.pythonhosted.org/packages/a3/48/9c58b685e69d49c31af6c8eb9012055fab7e665785165c84796e2c73ce72/pillow-12.3.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:4f883547d4b7f0495ebe7056b0cc2aea76094e7a4abc8e933540f3271df27d9c", size = 6337160, upload-time = "2026-07-01T11:53:55.457Z" },
{ url = "https://files.pythonhosted.org/packages/ff/fa/dc2a5c0ba6df93f67c31d34b808b7ce440b40cdbf96f0b81cde1d1e6fa93/pillow-12.3.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:236ff70b9312fb68943c703aa842ca6a758abfa45ac187a5e7c1452e96ef72b5", size = 7045172, upload-time = "2026-07-01T11:53:57.736Z" },
{ url = "https://files.pythonhosted.org/packages/86/a5/444817a4d4c4c2417df00513086ca196f388d8f9ef40c2e4ccd1ad1af54b/pillow-12.3.0-cp311-cp311-win32.whl", hash = "sha256:10e41f0fbf1eec8cfd234b8fe17a4caac7c9d0db4c204d3c173a8f9f6ef3232b", size = 6472232, upload-time = "2026-07-01T11:53:59.767Z" },
{ url = "https://files.pythonhosted.org/packages/63/c6/4bad1b18d132a50b27e1365e1ab163616f7a5bb56d330f66f9d1d9d4f9d4/pillow-12.3.0-cp311-cp311-win_amd64.whl", hash = "sha256:8e95e1385e4998ae9694eeaa4730ba5457ff61185b3a55e2e7bea0880aef452a", size = 7233653, upload-time = "2026-07-01T11:54:02.066Z" },
{ url = "https://files.pythonhosted.org/packages/fd/16/00f91ab7760dc842f5aad55217e80fc4a7067a0604535249bc8a2d6d9870/pillow-12.3.0-cp311-cp311-win_arm64.whl", hash = "sha256:ebaea975e03d3141d9d3a507df75c9b3ec90fa9d2ffd07567b3a978d9d790b26", size = 2568195, upload-time = "2026-07-01T11:54:04.622Z" },
{ url = "https://files.pythonhosted.org/packages/37/bf/fb3ebff8ddcb76aac5a01389251bbbb9519922a9b520d8247c1ca864a25d/pillow-12.3.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965", size = 5345969, upload-time = "2026-07-01T11:54:06.397Z" },
{ url = "https://files.pythonhosted.org/packages/d8/66/9a386a92561f402389a4fc70c18838bf6d35eb5eb5c6850b4b2dc64f5048/pillow-12.3.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7", size = 4780323, upload-time = "2026-07-01T11:54:09.351Z" },
{ url = "https://files.pythonhosted.org/packages/25/27/ac8f99618ffd3dde21db0f4d4b1d2ab00c0880595bfd17df103f7f39fd0c/pillow-12.3.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9", size = 6266838, upload-time = "2026-07-01T11:54:11.71Z" },
{ url = "https://files.pythonhosted.org/packages/84/21/a35af28dcc61f37ed850a2d64c65c701321dfbf25085e469d5559360cbbf/pillow-12.3.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91", size = 6940830, upload-time = "2026-07-01T11:54:13.732Z" },
{ url = "https://files.pythonhosted.org/packages/eb/51/8b08617af3ad95e33ce6d7dd2c99ed6c8298f7fb131636303956be022e25/pillow-12.3.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c", size = 6344383, upload-time = "2026-07-01T11:54:15.756Z" },
{ url = "https://files.pythonhosted.org/packages/1d/72/cf78ac9780bb93c28328f408973845a309d4d145041665f734572ced1b52/pillow-12.3.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df", size = 7052934, upload-time = "2026-07-01T11:54:17.721Z" },
{ url = "https://files.pythonhosted.org/packages/20/20/25e0f4dc178a6bc0696793720055519a0de89e7661dae886992decbd2f81/pillow-12.3.0-cp312-cp312-win32.whl", hash = "sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f", size = 6472684, upload-time = "2026-07-01T11:54:19.839Z" },
{ url = "https://files.pythonhosted.org/packages/45/89/da2f7971a317f83d807fdd4065c0af40208e59e692cc43d315a71a0e96d1/pillow-12.3.0-cp312-cp312-win_amd64.whl", hash = "sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09", size = 7227137, upload-time = "2026-07-01T11:54:22.025Z" },
{ url = "https://files.pythonhosted.org/packages/de/47/4845a0a6c0dbf1db8456bd9fc791f13c5ced7ced20606d08a0aacfd25b49/pillow-12.3.0-cp312-cp312-win_arm64.whl", hash = "sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510", size = 2568267, upload-time = "2026-07-01T11:54:24.051Z" },
{ url = "https://files.pythonhosted.org/packages/9d/ac/31fb64e1e7efb5a4b50cd3d92049ba89ac6e4d8d3bb6a74e15048ca3353e/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89", size = 4161684, upload-time = "2026-07-01T11:54:25.934Z" },
{ url = "https://files.pythonhosted.org/packages/87/b4/9805e23d2b4d77842b468513841fda254ee42f0289d25088340e4ff46e2d/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace", size = 4255487, upload-time = "2026-07-01T11:54:27.935Z" },
{ url = "https://files.pythonhosted.org/packages/df/39/ecf519435a200c693fe053a6ee4d835b41cf963a4dfc2551c4e637cb2a71/pillow-12.3.0-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec", size = 3696433, upload-time = "2026-07-01T11:54:29.813Z" },
{ url = "https://files.pythonhosted.org/packages/42/92/2fc3ffad878ae8dd5469ec1bc8eb83b71f48e13efdf68f02709003982a32/pillow-12.3.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66", size = 5345889, upload-time = "2026-07-01T11:54:31.97Z" },
{ url = "https://files.pythonhosted.org/packages/10/76/8803c13605b763d33d156c4678fc77f8443389c0c51c8aef707bb02015f4/pillow-12.3.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35", size = 4780109, upload-time = "2026-07-01T11:54:34.026Z" },
{ url = "https://files.pythonhosted.org/packages/1f/01/e18aff37cb0b4aac47ac90f016d347a49aca667ef97f190b06ac2aabc928/pillow-12.3.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65", size = 6263736, upload-time = "2026-07-01T11:54:36.131Z" },
{ url = "https://files.pythonhosted.org/packages/f7/62/de5bdd77d935331f4f802edc11e4d82950f642caad6cb2f949837b8560e2/pillow-12.3.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3", size = 6937129, upload-time = "2026-07-01T11:54:38.216Z" },
{ url = "https://files.pythonhosted.org/packages/70/4d/105627a13300c5e0df1d174230b32fd1273062c96f7745fd552b945d1e1d/pillow-12.3.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a", size = 6339562, upload-time = "2026-07-01T11:54:40.354Z" },
{ url = "https://files.pythonhosted.org/packages/6b/1d/f13de01a553988ab895ba1c722e06cf3144d4f57656fd5b81b6d881f1179/pillow-12.3.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e", size = 7049439, upload-time = "2026-07-01T11:54:42.489Z" },
{ url = "https://files.pythonhosted.org/packages/c9/f9/066794cca041b969964f779ee5fa66a9498bbf34248ac39c5d7954e4198f/pillow-12.3.0-cp313-cp313-win32.whl", hash = "sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f", size = 6473287, upload-time = "2026-07-01T11:54:44.9Z" },
{ url = "https://files.pythonhosted.org/packages/a6/9b/7a58e61d62be561da3a356fe2384d4059a6345fc130e23ef1c36a5b81d24/pillow-12.3.0-cp313-cp313-win_amd64.whl", hash = "sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8", size = 7239691, upload-time = "2026-07-01T11:54:47.141Z" },
{ url = "https://files.pythonhosted.org/packages/aa/b0/c4ed4f0ef8f8fa5ee8351537db6650bb8189f7e118842978dd6589065692/pillow-12.3.0-cp313-cp313-win_arm64.whl", hash = "sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b", size = 2568185, upload-time = "2026-07-01T11:54:49.137Z" },
{ url = "https://files.pythonhosted.org/packages/dc/01/001f65b68192f0228cc1dbbc8d2530ab5d58b61037ba0587f946fea607cd/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330", size = 4161736, upload-time = "2026-07-01T11:54:51.156Z" },
{ url = "https://files.pythonhosted.org/packages/1a/d2/0219746d0fd16fc8a84498e79452375be3797d3ce4044596ce565164b84f/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217", size = 4255435, upload-time = "2026-07-01T11:54:53.414Z" },
{ url = "https://files.pythonhosted.org/packages/c8/02/8d0bc62ef0302318c46ff2a512822d2610e81c7aa46c9b3abe6cbaca5ad0/pillow-12.3.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930", size = 3696262, upload-time = "2026-07-01T11:54:55.739Z" },
{ url = "https://files.pythonhosted.org/packages/85/e2/73c77d218410b14f5f2d565e8a998d5317b7b9c75368d29985139f7a46f0/pillow-12.3.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8", size = 5350344, upload-time = "2026-07-01T11:54:57.657Z" },
{ url = "https://files.pythonhosted.org/packages/c7/da/32c752228ae345f489e3a42499d817b6c3996da7e8a3bc7a04fc806b243b/pillow-12.3.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0", size = 4780131, upload-time = "2026-07-01T11:54:59.713Z" },
{ url = "https://files.pythonhosted.org/packages/b1/9d/8b2c807dbef61a5197c047afe99823787eb66f63daf9fb2432f91d6f0462/pillow-12.3.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321", size = 6263757, upload-time = "2026-07-01T11:55:01.778Z" },
{ url = "https://files.pythonhosted.org/packages/5c/44/c85361f65dbe00eea8576ee467c768d25129989efb76e94f205e9ca9bb46/pillow-12.3.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b", size = 6936962, upload-time = "2026-07-01T11:55:03.93Z" },
{ url = "https://files.pythonhosted.org/packages/18/7e/e483414b35800b86b6f08dbbc7803fb5cd52c4d6f897f47d53ea2c7e6f65/pillow-12.3.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198", size = 6339171, upload-time = "2026-07-01T11:55:05.989Z" },
{ url = "https://files.pythonhosted.org/packages/f0/f4/68c491844841ede6bed70189546b3ee9731cf9f2cbad396faff5e1ccba45/pillow-12.3.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130", size = 7048116, upload-time = "2026-07-01T11:55:08.131Z" },
{ url = "https://files.pythonhosted.org/packages/a3/34/77f3f793fed8efc7d243f21b33c5a3f0d1c97ee70346d3db855587e155ff/pillow-12.3.0-cp314-cp314-win32.whl", hash = "sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a", size = 6467209, upload-time = "2026-07-01T11:55:10.408Z" },
{ url = "https://files.pythonhosted.org/packages/f1/e0/492879f69d94f91f60fc8cd05ba03650e9520afebb2fb7aa12777d7c7f38/pillow-12.3.0-cp314-cp314-win_amd64.whl", hash = "sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d", size = 7237707, upload-time = "2026-07-01T11:55:12.745Z" },
{ url = "https://files.pythonhosted.org/packages/c9/ac/6b11f2875f1c2ac040d84e1bbf9cf22a88038f901ca1037898b280b38365/pillow-12.3.0-cp314-cp314-win_arm64.whl", hash = "sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838", size = 2565995, upload-time = "2026-07-01T11:55:14.736Z" },
{ url = "https://files.pythonhosted.org/packages/52/69/c2208e56af9bfc1913afb24020297a691eb1d4ef688474c8a04913f65e04/pillow-12.3.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e", size = 5352503, upload-time = "2026-07-01T11:55:17.076Z" },
{ url = "https://files.pythonhosted.org/packages/07/70/e5686d753e898a45d778ff1718dba8516ead6ab6b95d85fc8c4b70650cf2/pillow-12.3.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17", size = 4782956, upload-time = "2026-07-01T11:55:19.448Z" },
{ url = "https://files.pythonhosted.org/packages/d5/37/25c6692f06927ee973ff18c8d9ee98ad0b4d84ee67a09610c2dd1447958e/pillow-12.3.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385", size = 6322855, upload-time = "2026-07-01T11:55:21.613Z" },
{ url = "https://files.pythonhosted.org/packages/cc/91/420637fcb8f1bc11029e403b4538e6694744428d8246118e45719f944556/pillow-12.3.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c", size = 6989642, upload-time = "2026-07-01T11:55:24.006Z" },
{ url = "https://files.pythonhosted.org/packages/10/08/b94d7811281ccf0d143a1cf768d1c49e1e54af63e7b708ab2ee3eb87face/pillow-12.3.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d", size = 6391281, upload-time = "2026-07-01T11:55:26.252Z" },
{ url = "https://files.pythonhosted.org/packages/d2/87/24233f785f55474dc02ce3e739c5528a77e3a862e9333d1dd7a25cc31f70/pillow-12.3.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931", size = 7096716, upload-time = "2026-07-01T11:55:28.318Z" },
{ url = "https://files.pythonhosted.org/packages/23/26/fcb2f6e37175b04f53570b59937867e2b80ee1685e744023153028fc14f9/pillow-12.3.0-cp314-cp314t-win32.whl", hash = "sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7", size = 6474125, upload-time = "2026-07-01T11:55:30.956Z" },
{ url = "https://files.pythonhosted.org/packages/90/de/3634abee5f1c9e13c56787b7d5517b0ba8d6de51700b95578cf338349c9f/pillow-12.3.0-cp314-cp314t-win_amd64.whl", hash = "sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c", size = 7242939, upload-time = "2026-07-01T11:55:34.044Z" },
{ url = "https://files.pythonhosted.org/packages/ce/2a/fd13f8eb24de5714a6eb444a3d67e2842c6c576e159a43793adf23051351/pillow-12.3.0-cp314-cp314t-win_arm64.whl", hash = "sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45", size = 2567506, upload-time = "2026-07-01T11:55:35.988Z" },
{ url = "https://files.pythonhosted.org/packages/5d/dc/8fdce34ec725a33c81c6ba122b904d6b9024e50ea9ac7bede62fab54506c/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphoneos.whl", hash = "sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139", size = 4162063, upload-time = "2026-07-01T11:55:37.941Z" },
{ url = "https://files.pythonhosted.org/packages/76/66/2044b9a63d3b84ff048228dfcb7cd9bf0df983e8470971bf7d4c57b693de/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402", size = 4255549, upload-time = "2026-07-01T11:55:40.022Z" },
{ url = "https://files.pythonhosted.org/packages/52/7e/1f67e6f4ece6b582ee4b539decbcc9f848dc245a93ed8cd7338bafef72f1/pillow-12.3.0-cp315-cp315-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c", size = 3696331, upload-time = "2026-07-01T11:55:41.98Z" },
{ url = "https://files.pythonhosted.org/packages/12/40/d306fc2c8e4d45d7f175c77edca7063be7b86fe7fe6e68f4353bf71d808c/pillow-12.3.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f", size = 5350370, upload-time = "2026-07-01T11:55:44.028Z" },
{ url = "https://files.pythonhosted.org/packages/dd/44/668fb1437e8ce420f62d6106eb66e44a5971602a4d794615bdf79315d82d/pillow-12.3.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701", size = 4780147, upload-time = "2026-07-01T11:55:46.073Z" },
{ url = "https://files.pythonhosted.org/packages/0c/08/93fa2e70e30a2d81547e481b6ee2bb9522117221fb1e0ce4b5df70967677/pillow-12.3.0-cp315-cp315-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace", size = 6273659, upload-time = "2026-07-01T11:55:48.264Z" },
{ url = "https://files.pythonhosted.org/packages/f8/6d/043e96ff814fc31a33077e4cba86082167db520c93632afdf2042febbb0c/pillow-12.3.0-cp315-cp315-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4", size = 6947439, upload-time = "2026-07-01T11:55:50.503Z" },
{ url = "https://files.pythonhosted.org/packages/af/92/ba71d2ee2ac0edf3fa33bd9d5ee9ee080da70b1766f3ca3934f9938ddac9/pillow-12.3.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39", size = 6353577, upload-time = "2026-07-01T11:55:52.697Z" },
{ url = "https://files.pythonhosted.org/packages/0f/ce/e63064e2122923ff687c8ad792d0d736a7b3920a56a46982e81a7fdd25d6/pillow-12.3.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71", size = 7060394, upload-time = "2026-07-01T11:55:55.149Z" },
{ url = "https://files.pythonhosted.org/packages/54/76/a09cc3ccc8d773a7283d34c38bec1708f9e3cc932093cbc4c5e71ac4060b/pillow-12.3.0-cp315-cp315-win32.whl", hash = "sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827", size = 6467375, upload-time = "2026-07-01T11:55:57.769Z" },
{ url = "https://files.pythonhosted.org/packages/3e/03/1846c49ba3b1d5550392a4bbd06d6fb4578e1cd91a803198b5c90f5f7d53/pillow-12.3.0-cp315-cp315-win_amd64.whl", hash = "sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5", size = 7237048, upload-time = "2026-07-01T11:55:59.975Z" },
{ url = "https://files.pythonhosted.org/packages/fb/bb/89f35dcc79610423f9f195504d7def7f0d1416a711541b42867e25fe3412/pillow-12.3.0-cp315-cp315-win_arm64.whl", hash = "sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658", size = 2566006, upload-time = "2026-07-01T11:56:02.143Z" },
{ url = "https://files.pythonhosted.org/packages/30/88/707027ba09942dfa2c28759b5c222d769290a41c6d20ea60ec250801941f/pillow-12.3.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf", size = 5352509, upload-time = "2026-07-01T11:56:04.2Z" },
{ url = "https://files.pythonhosted.org/packages/b0/6d/00352fa25332c2569cd387851f568cc5a4b75a9adbfb37ac4fbce4c02eec/pillow-12.3.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64", size = 4783167, upload-time = "2026-07-01T11:56:06.631Z" },
{ url = "https://files.pythonhosted.org/packages/13/4f/9e049dfa21af7c22427275720e2490267ba8138120add5c4c574deb69782/pillow-12.3.0-cp315-cp315t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e", size = 6329237, upload-time = "2026-07-01T11:56:08.868Z" },
{ url = "https://files.pythonhosted.org/packages/36/16/cf6eeaae8d0fce8dd390a33437cf68c5d5bd73834a2bc6e2f14efda0ab45/pillow-12.3.0-cp315-cp315t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777", size = 6997047, upload-time = "2026-07-01T11:56:11.379Z" },
{ url = "https://files.pythonhosted.org/packages/1e/69/dbf769bdd55f48bf5733cac28edc6364ffaa072ec9ba336266e4fe66be55/pillow-12.3.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1", size = 6400440, upload-time = "2026-07-01T11:56:13.908Z" },
{ url = "https://files.pythonhosted.org/packages/a0/e1/ffc9cfc2eea0d178da8018e18e959301ad9d6bc9f3edb7181e748a474b97/pillow-12.3.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9", size = 7105895, upload-time = "2026-07-01T11:56:16.575Z" },
{ url = "https://files.pythonhosted.org/packages/18/f0/a5595c1e8c3ae44b9828cb2f0fa8155e5095ef04d6327b8f61cf44a3df85/pillow-12.3.0-cp315-cp315t-win32.whl", hash = "sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8", size = 6474384, upload-time = "2026-07-01T11:56:18.855Z" },
{ url = "https://files.pythonhosted.org/packages/e4/04/62bcd9f844984c5938d3b05264a61d797a29d3e0812341a8204af70bbdee/pillow-12.3.0-cp315-cp315t-win_amd64.whl", hash = "sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418", size = 7243537, upload-time = "2026-07-01T11:56:21.214Z" },
{ url = "https://files.pythonhosted.org/packages/3d/68/1f3066acedf37673694a7141381d8f811ae97f30d34413d236abe7d489f1/pillow-12.3.0-cp315-cp315t-win_arm64.whl", hash = "sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59", size = 2567491, upload-time = "2026-07-01T11:56:23.506Z" },
{ url = "https://files.pythonhosted.org/packages/75/18/2e8b40223153ccbc60df07f9e8928dc0c76202aa4e55ae9f53962b6510d6/pillow-12.3.0-pp311-pypy311_pp73-macosx_10_15_x86_64.whl", hash = "sha256:b3c777e849237620b022f7f297dd67705f9f5cf1685f09f02e46f93e92725468", size = 5302510, upload-time = "2026-07-01T11:56:25.736Z" },
{ url = "https://files.pythonhosted.org/packages/46/3e/51fabf59d5ab801ceab709453d3ab6b180083496579549de4c45ced6528a/pillow-12.3.0-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:b343699e8308bdc51978310e1c959c584e7869cc8c40780058c87da7781a1e94", size = 4736058, upload-time = "2026-07-01T11:56:28.041Z" },
{ url = "https://files.pythonhosted.org/packages/bf/20/22fe9384b7949e25fb1293bcfc84fb82590ff4ea6b37c95b24d26d793d86/pillow-12.3.0-pp311-pypy311_pp73-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:fbd139c8447d25dd750ab79ee274cc5e1fe80fc56340ab10b18a195e1b6eca3e", size = 5237776, upload-time = "2026-07-01T11:56:30.263Z" },
{ url = "https://files.pythonhosted.org/packages/08/14/f6ba68107680ffa74b39985f3f30884e41318fbc4250caa423c79b4788bb/pillow-12.3.0-pp311-pypy311_pp73-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e7e480451b9fa137494bccd3a7d69adbe8ac65a87d97be61e11f1b1050a5bac3", size = 5860358, upload-time = "2026-07-01T11:56:32.68Z" },
{ url = "https://files.pythonhosted.org/packages/36/54/0169bc772ec491108b62f644f8ecf1fe5d8ae5ebafde2ee2142210166903/pillow-12.3.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:04f01d28a6aaff387bf842a13be313df23ba0597a44f1a976c9feb3c6ff4711a", size = 7231786, upload-time = "2026-07-01T11:56:35.046Z" },
]
[[package]]
name = "platformdirs"
version = "4.9.6"