feat: prepare EvoScientist 0.3.0
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
This commit is contained in:
@@ -555,9 +555,10 @@ def _build_base_kwargs(
|
||||
|
||||
cfg = cfg if cfg is not None else _ensure_config()
|
||||
tool_registry = {"think_tool": think_tool}
|
||||
base_tools = [think_tool, skill_manager]
|
||||
if os.environ.get("TAVILY_API_KEY"):
|
||||
tool_registry["tavily_search"] = tavily_search
|
||||
base_tools = [think_tool, skill_manager]
|
||||
base_tools.append(tavily_search)
|
||||
|
||||
subs = load_subagents(
|
||||
SUBAGENTS_CONFIG,
|
||||
@@ -617,15 +618,54 @@ def load_mcp_and_build_kwargs(
|
||||
)
|
||||
|
||||
tool_registry = {"think_tool": think_tool}
|
||||
base_tools = [think_tool, skill_manager]
|
||||
if os.environ.get("TAVILY_API_KEY"):
|
||||
tool_registry["tavily_search"] = tavily_search
|
||||
base_tools = [think_tool, skill_manager]
|
||||
base_tools.append(tavily_search)
|
||||
|
||||
# Fresh tool registry — start from base tools + MCP tools
|
||||
# DeepAgents installs these outside ``base_tools`` through middleware.
|
||||
# MCP tools must never shadow them inside any one agent namespace.
|
||||
middleware_tool_names = {
|
||||
"ls",
|
||||
"read_file",
|
||||
"write_file",
|
||||
"edit_file",
|
||||
"glob",
|
||||
"grep",
|
||||
"execute",
|
||||
"write_todos",
|
||||
"task",
|
||||
"start_async_task",
|
||||
"check_async_task",
|
||||
"update_async_task",
|
||||
"cancel_async_task",
|
||||
"list_async_tasks",
|
||||
}
|
||||
|
||||
# Fresh tool registry — start from built-ins, then add one representative
|
||||
# MCP implementation for YAML name resolution. A tool may be exposed to
|
||||
# several agents, but no agent may contain duplicate names and no MCP tool
|
||||
# may override a built-in implementation.
|
||||
registry = dict(tool_registry)
|
||||
for tools in mcp_by_agent.values():
|
||||
builtin_names = {
|
||||
*(str(getattr(tool, "name", "")) for tool in base_tools),
|
||||
*middleware_tool_names,
|
||||
}
|
||||
for agent_name, tools in mcp_by_agent.items():
|
||||
seen_for_agent: set[str] = set()
|
||||
for t in tools:
|
||||
registry[t.name] = t
|
||||
tool_name = str(t.name)
|
||||
if tool_name in builtin_names or tool_name in seen_for_agent:
|
||||
from .llm.contracts import EvoRuntimeError
|
||||
|
||||
raise EvoRuntimeError(
|
||||
"TOOL_REGISTRY_CONFLICT",
|
||||
details=(
|
||||
{"agent_name": str(agent_name), "tool_name": tool_name},
|
||||
),
|
||||
)
|
||||
seen_for_agent.add(tool_name)
|
||||
registry.setdefault(tool_name, t)
|
||||
|
||||
mcp_main = mcp_by_agent.pop("main", [])
|
||||
|
||||
@@ -639,10 +679,31 @@ def load_mcp_and_build_kwargs(
|
||||
subs, workspace_dir=workspace_dir, cfg=cfg, chat_model=chat_model
|
||||
)
|
||||
|
||||
# Inject MCP tools into subagents by name
|
||||
# Inject MCP tools into subagents by name. YAML-resolved tools already
|
||||
# belong to that agent namespace, so a second tool with the same name is a
|
||||
# configuration conflict rather than an item to append silently.
|
||||
for sa in subs:
|
||||
if sa_tools := mcp_by_agent.get(sa["name"], []):
|
||||
sa.setdefault("tools", []).extend(sa_tools)
|
||||
target_tools = sa.setdefault("tools", [])
|
||||
existing_names = {
|
||||
str(getattr(tool, "name", tool)) for tool in target_tools
|
||||
}
|
||||
for tool in sa_tools:
|
||||
tool_name = str(tool.name)
|
||||
if tool_name in existing_names:
|
||||
from .llm.contracts import EvoRuntimeError
|
||||
|
||||
raise EvoRuntimeError(
|
||||
"TOOL_REGISTRY_CONFLICT",
|
||||
details=(
|
||||
{
|
||||
"agent_name": str(sa["name"]),
|
||||
"tool_name": tool_name,
|
||||
},
|
||||
),
|
||||
)
|
||||
existing_names.add(tool_name)
|
||||
target_tools.append(tool)
|
||||
|
||||
# Swap selected sub-agents to AsyncSubAgent (must happen AFTER MCP injection
|
||||
# since async sub-agents are remote graphs that load their own tools).
|
||||
|
||||
@@ -9,7 +9,7 @@ from __future__ import annotations
|
||||
|
||||
from importlib import import_module
|
||||
|
||||
__version__ = "0.2.2"
|
||||
from ._version import __version__
|
||||
|
||||
_EXPORTS: dict[str, tuple[str, str]] = {
|
||||
# Agent graph (lazy to avoid expensive initialization at import time)
|
||||
@@ -73,4 +73,5 @@ def __dir__() -> list[str]:
|
||||
return sorted(set(globals()) | set(_EXPORTS))
|
||||
|
||||
|
||||
__all__ = list(_EXPORTS)
|
||||
__all__ = ["__version__"]
|
||||
__all__.extend(_EXPORTS)
|
||||
|
||||
@@ -0,0 +1,3 @@
|
||||
"""Package version shared by builds and runtime."""
|
||||
|
||||
__version__ = "0.3.0"
|
||||
@@ -0,0 +1,466 @@
|
||||
"""Bounded document extraction and non-text file policy for workspaces."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import tempfile
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
from xml.etree import ElementTree as ET
|
||||
|
||||
MAX_DOCUMENT_BYTES = 50 * 1024 * 1024
|
||||
MAX_DOCUMENT_RESULT_CHARS = 50_000
|
||||
MAX_CONVERTED_DOCUMENT_BYTES = 10 * 1024 * 1024
|
||||
DOCUMENT_CONVERSION_TIMEOUT_SECONDS = 60
|
||||
MAX_IMAGE_BYTES = 25 * 1024 * 1024
|
||||
MAX_IMAGE_EDGE = 2048
|
||||
MAX_IMAGE_PIXELS = 40_000_000
|
||||
MAX_OOXML_MEMBERS = 10_000
|
||||
MAX_OOXML_MEMBER_BYTES = 50 * 1024 * 1024
|
||||
MAX_OOXML_EXPANDED_BYTES = 200 * 1024 * 1024
|
||||
MAX_OOXML_COMPRESSION_RATIO = 100
|
||||
|
||||
IMAGE_EXTENSIONS = frozenset(
|
||||
{".bmp", ".gif", ".ico", ".jpeg", ".jpg", ".png", ".tif", ".tiff", ".webp"}
|
||||
)
|
||||
DOCUMENT_EXTENSIONS = frozenset(
|
||||
{
|
||||
".doc",
|
||||
".docm",
|
||||
".docx",
|
||||
".epub",
|
||||
".odp",
|
||||
".ods",
|
||||
".odt",
|
||||
".pdf",
|
||||
".pot",
|
||||
".pps",
|
||||
".ppsm",
|
||||
".ppsx",
|
||||
".ppt",
|
||||
".pptm",
|
||||
".pptx",
|
||||
".rtf",
|
||||
".xls",
|
||||
".xlsb",
|
||||
".xlsm",
|
||||
".xlsx",
|
||||
}
|
||||
)
|
||||
ARCHIVE_EXTENSIONS = frozenset(
|
||||
{".7z", ".bz2", ".gz", ".rar", ".tar", ".tgz", ".xz", ".zip"}
|
||||
)
|
||||
DATABASE_EXTENSIONS = frozenset({".db", ".sqlite", ".sqlite3"})
|
||||
EXECUTABLE_EXTENSIONS = frozenset(
|
||||
{".app", ".deb", ".dll", ".dylib", ".elf", ".exe", ".msi", ".rpm", ".so"}
|
||||
)
|
||||
DATASET_EXTENSIONS = frozenset(
|
||||
{".arrow", ".feather", ".h5", ".hdf5", ".npy", ".npz", ".parquet"}
|
||||
)
|
||||
MEDIA_EXTENSIONS = frozenset(
|
||||
{".aac", ".avi", ".flac", ".m4a", ".mkv", ".mov", ".mp3", ".mp4", ".ogg", ".wav", ".webm"}
|
||||
)
|
||||
|
||||
_W = "http://schemas.openxmlformats.org/wordprocessingml/2006/main"
|
||||
_A = "http://schemas.openxmlformats.org/drawingml/2006/main"
|
||||
_S = "http://schemas.openxmlformats.org/spreadsheetml/2006/main"
|
||||
_R = "http://schemas.openxmlformats.org/officeDocument/2006/relationships"
|
||||
|
||||
|
||||
class DocumentExtractionError(RuntimeError):
|
||||
"""A supported document could not be converted to bounded text."""
|
||||
|
||||
|
||||
def classify_file(path: str, head: bytes = b"") -> str:
|
||||
"""Classify a workspace file into one policy category."""
|
||||
|
||||
extension = Path(path).suffix.lower()
|
||||
if extension in IMAGE_EXTENSIONS:
|
||||
return "image"
|
||||
if extension in DOCUMENT_EXTENSIONS:
|
||||
return "document"
|
||||
if extension in ARCHIVE_EXTENSIONS:
|
||||
return "archive"
|
||||
if extension in DATABASE_EXTENSIONS or head.startswith(b"SQLite format 3\x00"):
|
||||
return "database"
|
||||
if extension in EXECUTABLE_EXTENSIONS or head.startswith((b"MZ", b"\x7fELF")):
|
||||
return "executable"
|
||||
if extension in DATASET_EXTENSIONS:
|
||||
return "dataset"
|
||||
if extension in MEDIA_EXTENSIONS:
|
||||
return "media"
|
||||
return "unknown"
|
||||
|
||||
|
||||
def binary_processing_guidance(path: str, kind: str, size_bytes: int) -> str:
|
||||
"""Return bounded, actionable JSON-like guidance for Agent-side programming."""
|
||||
|
||||
import json
|
||||
|
||||
common = {
|
||||
"code": "BINARY_PROCESSING_REQUIRED"
|
||||
if kind != "binary"
|
||||
else "UNSUPPORTED_BINARY_FILE",
|
||||
"path": path,
|
||||
"kind": kind,
|
||||
"size_bytes": size_bytes,
|
||||
}
|
||||
if kind == "archive":
|
||||
common.update(
|
||||
action=(
|
||||
"Use execute with Python to list and validate archive members before "
|
||||
"selective extraction; never use extractall."
|
||||
),
|
||||
constraints={
|
||||
"list_before_extract": True,
|
||||
"max_members": 2000,
|
||||
"max_total_uncompressed_bytes": 500 * 1024 * 1024,
|
||||
"max_member_bytes": 100 * 1024 * 1024,
|
||||
"max_compression_ratio": 100,
|
||||
"reject_absolute_or_parent_paths": True,
|
||||
"do_not_execute_members": True,
|
||||
},
|
||||
)
|
||||
elif kind == "database":
|
||||
common.update(
|
||||
action=(
|
||||
"Use execute with Python sqlite3 in read-only mode: "
|
||||
"file:<path>?mode=ro&immutable=1; set PRAGMA query_only=ON; "
|
||||
"inspect schema, run bounded SELECT queries with LIMIT, and write "
|
||||
"large results under artifacts/."
|
||||
),
|
||||
constraints={
|
||||
"read_only": True,
|
||||
"mode": "mode=ro",
|
||||
"query_only": True,
|
||||
"max_rows": 1000,
|
||||
"forbid_attach_database": True,
|
||||
"forbid_load_extension": True,
|
||||
},
|
||||
)
|
||||
elif kind == "executable":
|
||||
common.update(
|
||||
action=(
|
||||
"Use execute only for bounded static metadata inspection (hash, file "
|
||||
"headers, signature, imports, strings); this file must not be executed."
|
||||
),
|
||||
constraints={"must_not_be_executed": True, "static_analysis_only": True},
|
||||
)
|
||||
elif kind == "dataset":
|
||||
common.update(
|
||||
action=(
|
||||
"Use execute with the appropriate library to inspect schema, dimensions, "
|
||||
"statistics, and a bounded sample; do not serialize the whole dataset."
|
||||
)
|
||||
)
|
||||
elif kind == "media":
|
||||
common.update(
|
||||
action=(
|
||||
"Use execute with ffprobe/ffmpeg or an available transcription workflow "
|
||||
"to inspect metadata and selected ranges; do not inline the complete file."
|
||||
)
|
||||
)
|
||||
else:
|
||||
common.update(
|
||||
kind="binary",
|
||||
action=(
|
||||
"This is an unsupported binary file. Use execute only for bounded static "
|
||||
"inspection; do not execute it or inline its bytes."
|
||||
),
|
||||
)
|
||||
return json.dumps(common, ensure_ascii=False, sort_keys=True)
|
||||
|
||||
|
||||
def prepare_image_bytes(data: bytes, path: str) -> bytes:
|
||||
"""Validate and downsample an image before it becomes a model media block."""
|
||||
|
||||
import io
|
||||
|
||||
if len(data) > MAX_IMAGE_BYTES:
|
||||
raise DocumentExtractionError(
|
||||
f"IMAGE_TOO_LARGE: {len(data)} bytes exceeds {MAX_IMAGE_BYTES}"
|
||||
)
|
||||
try:
|
||||
from PIL import Image
|
||||
|
||||
image = Image.open(io.BytesIO(data))
|
||||
width, height = image.size
|
||||
if width * height > MAX_IMAGE_PIXELS:
|
||||
raise DocumentExtractionError(
|
||||
f"IMAGE_PIXEL_BUDGET_EXCEEDED: {width}x{height} exceeds "
|
||||
f"{MAX_IMAGE_PIXELS} pixels"
|
||||
)
|
||||
image.load()
|
||||
except DocumentExtractionError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise DocumentExtractionError(
|
||||
f"IMAGE_PROCESSING_FAILED: {path}: {type(exc).__name__}: {exc}"
|
||||
) from exc
|
||||
|
||||
frame_count = int(getattr(image, "n_frames", 1) or 1)
|
||||
if frame_count > 1:
|
||||
image.seek(0)
|
||||
work = image.convert("RGBA" if image.mode in {"RGBA", "LA"} else "RGB")
|
||||
else:
|
||||
work = image.copy()
|
||||
|
||||
if max(work.size) <= MAX_IMAGE_EDGE and frame_count == 1:
|
||||
return data
|
||||
|
||||
work.thumbnail((MAX_IMAGE_EDGE, MAX_IMAGE_EDGE), Image.Resampling.LANCZOS)
|
||||
has_alpha = work.mode in {"RGBA", "LA"} or (
|
||||
work.mode == "P" and "transparency" in work.info
|
||||
)
|
||||
output = io.BytesIO()
|
||||
if has_alpha:
|
||||
if work.mode == "P":
|
||||
work = work.convert("RGBA")
|
||||
work.save(output, "PNG", optimize=True)
|
||||
else:
|
||||
if work.mode != "RGB":
|
||||
work = work.convert("RGB")
|
||||
work.save(output, "JPEG", quality=85, optimize=True)
|
||||
return output.getvalue()
|
||||
|
||||
|
||||
def extract_document_bytes(data: bytes, path: str) -> str:
|
||||
"""Extract readable text from a supported document without exposing bytes."""
|
||||
|
||||
if len(data) > MAX_DOCUMENT_BYTES:
|
||||
raise DocumentExtractionError(
|
||||
f"DOCUMENT_TOO_LARGE: {len(data)} bytes exceeds {MAX_DOCUMENT_BYTES}"
|
||||
)
|
||||
extension = Path(path).suffix.lower()
|
||||
try:
|
||||
if extension == ".docx":
|
||||
return _extract_docx(data)
|
||||
if extension == ".pptx":
|
||||
return _extract_pptx(data)
|
||||
if extension == ".xlsx":
|
||||
return _extract_xlsx(data)
|
||||
return _extract_anydoc(data, extension)
|
||||
except DocumentExtractionError:
|
||||
raise
|
||||
except Exception as exc:
|
||||
raise DocumentExtractionError(
|
||||
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
|
||||
) from exc
|
||||
|
||||
|
||||
def paginate_document_text(text: str, *, offset: int, limit: int) -> str:
|
||||
"""Apply line and character budgets to extracted document text."""
|
||||
|
||||
lines = text.splitlines(keepends=True)
|
||||
if not lines:
|
||||
return "(document contains no extractable text)"
|
||||
if offset >= len(lines):
|
||||
raise DocumentExtractionError(
|
||||
f"Line offset {offset} exceeds extracted document length ({len(lines)} lines)"
|
||||
)
|
||||
selected = "".join(lines[offset : offset + limit])
|
||||
if len(selected) <= MAX_DOCUMENT_RESULT_CHARS:
|
||||
return selected
|
||||
trimmed = selected[:MAX_DOCUMENT_RESULT_CHARS]
|
||||
boundary = trimmed.rfind("\n")
|
||||
if boundary > 0:
|
||||
trimmed = trimmed[: boundary + 1]
|
||||
consumed = max(1, len(trimmed.splitlines()))
|
||||
return (
|
||||
trimmed
|
||||
+ f"\n[DOCUMENT_OUTPUT_TRUNCATED: use offset={offset + consumed} to continue; "
|
||||
+ f"single-read limit is {MAX_DOCUMENT_RESULT_CHARS} characters]\n"
|
||||
)
|
||||
|
||||
|
||||
def _validated_ooxml_archive(data: bytes) -> zipfile.ZipFile:
|
||||
try:
|
||||
archive = zipfile.ZipFile(_bytes_path(data))
|
||||
members = archive.infolist()
|
||||
except zipfile.BadZipFile as exc:
|
||||
raise DocumentExtractionError(
|
||||
"DOCUMENT_EXTRACTION_FAILED: invalid OOXML container"
|
||||
) from exc
|
||||
expanded = 0
|
||||
names: set[str] = set()
|
||||
if len(members) > MAX_OOXML_MEMBERS:
|
||||
archive.close()
|
||||
raise DocumentExtractionError(
|
||||
f"DOCUMENT_RESOURCE_LIMIT: OOXML has {len(members)} members; "
|
||||
f"limit is {MAX_OOXML_MEMBERS}"
|
||||
)
|
||||
for member in members:
|
||||
normalized = member.filename.replace("\\", "/")
|
||||
parts = tuple(part for part in normalized.split("/") if part)
|
||||
if (
|
||||
normalized.startswith("/")
|
||||
or ".." in parts
|
||||
or member.filename in names
|
||||
or bool(member.flag_bits & 0x1)
|
||||
):
|
||||
archive.close()
|
||||
raise DocumentExtractionError(
|
||||
"DOCUMENT_RESOURCE_LIMIT: OOXML contains an unsafe, duplicate, "
|
||||
"or encrypted member"
|
||||
)
|
||||
names.add(member.filename)
|
||||
expanded += member.file_size
|
||||
ratio = member.file_size / max(member.compress_size, 1)
|
||||
if (
|
||||
member.file_size > MAX_OOXML_MEMBER_BYTES
|
||||
or expanded > MAX_OOXML_EXPANDED_BYTES
|
||||
or ratio > MAX_OOXML_COMPRESSION_RATIO
|
||||
):
|
||||
archive.close()
|
||||
raise DocumentExtractionError(
|
||||
"DOCUMENT_RESOURCE_LIMIT: OOXML member expansion exceeds safety limits"
|
||||
)
|
||||
return archive
|
||||
|
||||
|
||||
def _zip_xml(data: bytes, member: str) -> ET.Element:
|
||||
try:
|
||||
with _validated_ooxml_archive(data) as archive:
|
||||
raw = archive.read(member)
|
||||
except KeyError as exc:
|
||||
raise DocumentExtractionError(
|
||||
f"DOCUMENT_EXTRACTION_FAILED: missing {member}"
|
||||
) from exc
|
||||
return ET.fromstring(raw)
|
||||
|
||||
|
||||
def _bytes_path(data: bytes):
|
||||
import io
|
||||
|
||||
return io.BytesIO(data)
|
||||
|
||||
|
||||
def _ooxml_part_number(name: str) -> int:
|
||||
stem = Path(name).stem
|
||||
digits = "".join(character for character in stem if character.isdigit())
|
||||
return int(digits) if digits else 0
|
||||
|
||||
|
||||
def _extract_docx(data: bytes) -> str:
|
||||
root = _zip_xml(data, "word/document.xml")
|
||||
paragraphs: list[str] = []
|
||||
for paragraph in root.iter(f"{{{_W}}}p"):
|
||||
text = "".join(node.text or "" for node in paragraph.iter(f"{{{_W}}}t"))
|
||||
if text:
|
||||
paragraphs.append(text)
|
||||
if not paragraphs:
|
||||
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: DOCX has no text")
|
||||
return "\n".join(paragraphs) + "\n"
|
||||
|
||||
|
||||
def _extract_pptx(data: bytes) -> str:
|
||||
try:
|
||||
with _validated_ooxml_archive(data) as archive:
|
||||
names = sorted(
|
||||
name
|
||||
for name in archive.namelist()
|
||||
if name.startswith("ppt/slides/slide") and name.endswith(".xml")
|
||||
)
|
||||
slides: list[str] = []
|
||||
for index, name in enumerate(sorted(names, key=_ooxml_part_number), 1):
|
||||
root = ET.fromstring(archive.read(name))
|
||||
texts = [node.text or "" for node in root.iter(f"{{{_A}}}t")]
|
||||
slides.append(f"## Slide {index}\n" + "\n".join(t for t in texts if t))
|
||||
except zipfile.BadZipFile as exc:
|
||||
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid PPTX") from exc
|
||||
if not slides:
|
||||
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: PPTX has no slides")
|
||||
return "\n\n".join(slides) + "\n"
|
||||
|
||||
|
||||
def _extract_xlsx(data: bytes) -> str:
|
||||
try:
|
||||
with _validated_ooxml_archive(data) as archive:
|
||||
shared: list[str] = []
|
||||
if "xl/sharedStrings.xml" in archive.namelist():
|
||||
root = ET.fromstring(archive.read("xl/sharedStrings.xml"))
|
||||
shared = [
|
||||
"".join(node.text or "" for node in item.iter(f"{{{_S}}}t"))
|
||||
for item in root.iter(f"{{{_S}}}si")
|
||||
]
|
||||
sheets = sorted(
|
||||
name
|
||||
for name in archive.namelist()
|
||||
if name.startswith("xl/worksheets/sheet") and name.endswith(".xml")
|
||||
)
|
||||
output: list[str] = []
|
||||
for index, name in enumerate(sheets, 1):
|
||||
root = ET.fromstring(archive.read(name))
|
||||
output.append(f"## Sheet {index}")
|
||||
for row in root.iter(f"{{{_S}}}row"):
|
||||
values: list[str] = []
|
||||
for cell in row.iter(f"{{{_S}}}c"):
|
||||
value_node = cell.find(f"{{{_S}}}v")
|
||||
value = value_node.text if value_node is not None else ""
|
||||
if cell.get("t") == "s" and value and value.isdigit():
|
||||
shared_index = int(value)
|
||||
value = shared[shared_index] if shared_index < len(shared) else value
|
||||
values.append(value or "")
|
||||
output.append("\t".join(values))
|
||||
except zipfile.BadZipFile as exc:
|
||||
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: invalid XLSX") from exc
|
||||
if len(output) <= 1:
|
||||
raise DocumentExtractionError("DOCUMENT_EXTRACTION_FAILED: XLSX has no sheets")
|
||||
return "\n".join(output) + "\n"
|
||||
|
||||
|
||||
def _extract_anydoc(data: bytes, extension: str) -> str:
|
||||
source = ""
|
||||
output = ""
|
||||
try:
|
||||
with tempfile.NamedTemporaryFile(suffix=extension, delete=False) as handle:
|
||||
handle.write(data)
|
||||
source = handle.name
|
||||
with tempfile.NamedTemporaryFile(suffix=".md", delete=False) as handle:
|
||||
output = handle.name
|
||||
script = (
|
||||
"import pathlib,sys; import anydoc; "
|
||||
"text=anydoc.to_markdown(sys.argv[1]); "
|
||||
"pathlib.Path(sys.argv[2]).write_text(text, encoding='utf-8')"
|
||||
)
|
||||
subprocess.run(
|
||||
[sys.executable, "-c", script, source, output],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
timeout=DOCUMENT_CONVERSION_TIMEOUT_SECONDS,
|
||||
)
|
||||
output_path = Path(output)
|
||||
if output_path.stat().st_size > MAX_CONVERTED_DOCUMENT_BYTES:
|
||||
raise DocumentExtractionError(
|
||||
"DOCUMENT_RESOURCE_LIMIT: converted document exceeds output budget"
|
||||
)
|
||||
text = output_path.read_text(encoding="utf-8")
|
||||
except subprocess.TimeoutExpired as exc:
|
||||
raise DocumentExtractionError(
|
||||
f"DOCUMENT_CONVERSION_TIMEOUT: exceeded {DOCUMENT_CONVERSION_TIMEOUT_SECONDS}s"
|
||||
) from exc
|
||||
except subprocess.CalledProcessError as exc:
|
||||
detail = exc.stderr.decode("utf-8", errors="replace")[-1000:]
|
||||
raise DocumentExtractionError(
|
||||
f"DOCUMENT_EXTRACTION_FAILED: converter exited {exc.returncode}: {detail}"
|
||||
) from exc
|
||||
except Exception as exc:
|
||||
if isinstance(exc, DocumentExtractionError):
|
||||
raise
|
||||
raise DocumentExtractionError(
|
||||
f"DOCUMENT_EXTRACTION_FAILED: {type(exc).__name__}: {exc}"
|
||||
) from exc
|
||||
finally:
|
||||
for temporary in (source, output):
|
||||
if temporary:
|
||||
try:
|
||||
os.unlink(temporary)
|
||||
except OSError:
|
||||
pass
|
||||
if not isinstance(text, str) or not text.strip():
|
||||
raise DocumentExtractionError(
|
||||
"DOCUMENT_EXTRACTION_FAILED: document contains no extractable text"
|
||||
)
|
||||
return text.rstrip("\n") + "\n"
|
||||
@@ -55,6 +55,11 @@ class EvoRuntimeError(RuntimeError):
|
||||
self.code = code
|
||||
self.details = tuple(dict(item) for item in details)
|
||||
|
||||
def __repr__(self) -> str:
|
||||
# LangGraph persists task failures using repr(exc). Keep that snapshot
|
||||
# machine-readable without serializing provider messages or details.
|
||||
return f"{type(self).__name__}(code={self.code!r})"
|
||||
|
||||
|
||||
def now_ms() -> int:
|
||||
return time.time_ns() // 1_000_000
|
||||
|
||||
@@ -14,6 +14,8 @@ from langchain_core.tools import BaseTool
|
||||
from langchain_core.utils.function_calling import convert_to_openai_tool
|
||||
from pydantic import Field
|
||||
|
||||
from .contracts import EvoRuntimeError
|
||||
|
||||
|
||||
class GatewayProxyChatModel(BaseChatModel):
|
||||
gateway_url: str
|
||||
@@ -78,7 +80,9 @@ class GatewayProxyChatModel(BaseChatModel):
|
||||
) -> ChatResult:
|
||||
del stop, kwargs
|
||||
attempt_id = self._attempt_id(run_manager)
|
||||
async with httpx.AsyncClient(timeout=httpx.Timeout(660.0, connect=5.0)) as client:
|
||||
async with httpx.AsyncClient(
|
||||
timeout=httpx.Timeout(660.0, connect=5.0)
|
||||
) as client:
|
||||
response = await client.post(
|
||||
f"{self.gateway_url.rstrip('/')}/api/internal/recoverable-runs/model/invoke",
|
||||
json={
|
||||
@@ -115,7 +119,9 @@ class GatewayProxyChatModel(BaseChatModel):
|
||||
"tool_choice": self.bound_tool_choice,
|
||||
"stream": True,
|
||||
}
|
||||
async with httpx.AsyncClient(timeout=httpx.Timeout(660.0, connect=5.0)) as client:
|
||||
async with httpx.AsyncClient(
|
||||
timeout=httpx.Timeout(660.0, connect=5.0)
|
||||
) as client:
|
||||
async with client.stream(
|
||||
"POST",
|
||||
f"{self.gateway_url.rstrip('/')}/api/internal/recoverable-runs/model/stream",
|
||||
@@ -126,11 +132,27 @@ class GatewayProxyChatModel(BaseChatModel):
|
||||
async for line in response.aiter_lines():
|
||||
if not line.startswith("data:"):
|
||||
continue
|
||||
data = line[len("data:"):].strip()
|
||||
data = line[len("data:") :].strip()
|
||||
if data == "[DONE]":
|
||||
saw_done = True
|
||||
break
|
||||
chunk = json.loads(data)
|
||||
if chunk.get("type") == "error":
|
||||
code = str(chunk.get("code") or "MODEL_PROVIDER_ERROR")
|
||||
message = str(chunk.get("message") or code)
|
||||
details = {
|
||||
key: value
|
||||
for key, value in {
|
||||
"http_status": chunk.get("status"),
|
||||
"retryable": chunk.get("retryable"),
|
||||
}.items()
|
||||
if isinstance(value, int | bool)
|
||||
}
|
||||
raise EvoRuntimeError(
|
||||
code,
|
||||
message,
|
||||
details=(details,) if details else (),
|
||||
)
|
||||
message = _chunk_to_message(chunk)
|
||||
yield ChatGenerationChunk(
|
||||
message=message,
|
||||
|
||||
@@ -29,7 +29,7 @@ from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import os
|
||||
from typing import Any
|
||||
from typing import Any, cast
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -131,6 +131,47 @@ def _patch_openai_empty_sse_keepalive() -> None:
|
||||
_patch_openai_empty_sse_keepalive()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Patch: deepagents routes every non-text extension to a media block even when
|
||||
# a backend has already extracted bounded UTF-8 text from the document. Keep
|
||||
# real base64 binary data on the media path (the middleware checks encoding
|
||||
# first), but route extracted Office/PDF results through a normal ToolMessage.
|
||||
# ---------------------------------------------------------------------------
|
||||
_deepagents_extracted_document_text_patched = False
|
||||
|
||||
|
||||
def _patch_deepagents_extracted_document_text() -> None:
|
||||
global _deepagents_extracted_document_text_patched
|
||||
if _deepagents_extracted_document_text_patched:
|
||||
return
|
||||
try:
|
||||
from pathlib import Path as _Path
|
||||
|
||||
import deepagents.middleware.filesystem as _filesystem
|
||||
|
||||
from EvoScientist.document_extract import DOCUMENT_EXTENSIONS
|
||||
|
||||
namespace = cast(dict[str, Any], vars(_filesystem))
|
||||
original = namespace["_get_file_type"]
|
||||
if getattr(original, "_evoscientist_document_text", False):
|
||||
_deepagents_extracted_document_text_patched = True
|
||||
return
|
||||
|
||||
def _document_text_type(path: str) -> str:
|
||||
if _Path(path).suffix.lower() in DOCUMENT_EXTENSIONS:
|
||||
return "text"
|
||||
return original(path)
|
||||
|
||||
_document_text_type._evoscientist_document_text = True # type: ignore[attr-defined]
|
||||
namespace["_get_file_type"] = _document_text_type
|
||||
_deepagents_extracted_document_text_patched = True
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
_patch_deepagents_extracted_document_text()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Patch: ccproxy-api 0.2.7 Codex compatibility.
|
||||
#
|
||||
|
||||
@@ -51,6 +51,7 @@ from .skill_context import (
|
||||
DEFAULT_MAX_SKILLS_BYTES,
|
||||
BudgetedSkillsMiddleware,
|
||||
)
|
||||
from .subagent_timeout import SubagentTimeoutMiddleware
|
||||
from .tool_error_handler import ToolErrorHandlerMiddleware
|
||||
from .tool_protocol_guard import ToolProtocolGuardMiddleware
|
||||
from .tool_selector import create_tool_selector_middleware
|
||||
@@ -82,6 +83,7 @@ __all__ = [
|
||||
"RepetitiveToolCallGuardMiddleware",
|
||||
"RuntimeContextMiddleware",
|
||||
"SchedulerMiddleware",
|
||||
"SubagentTimeoutMiddleware",
|
||||
"ToolErrorHandlerMiddleware",
|
||||
"ToolProtocolGuardMiddleware",
|
||||
"collapse_repetitive_tool_rounds",
|
||||
|
||||
@@ -187,6 +187,17 @@ class DynamicReviewMiddleware(HumanInTheLoopMiddleware):
|
||||
# A LangGraph resume continues at this interrupted node and does not
|
||||
# re-run before_agent. Reusing a manual state is restrictive and is
|
||||
# required for the existing resume child Run to complete.
|
||||
# EXCEPTION: the gateway snapshots the thread's LIVE review mode
|
||||
# into each resume child's envelope. When the user flipped the
|
||||
# thread (or the current interrupt) to auto AFTER the parent run
|
||||
# started, the injected ai4sci_review_mode context is "auto" —
|
||||
# verify it and bypass HITL instead of interrupting again.
|
||||
if review is not None and review.get("requested_mode") == "auto":
|
||||
try:
|
||||
_resolve_sync(current_run_id, review)
|
||||
except AutoReviewVerificationError:
|
||||
return super().after_model(state, runtime)
|
||||
return None
|
||||
return super().after_model(state, runtime)
|
||||
raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID")
|
||||
|
||||
@@ -213,5 +224,14 @@ class DynamicReviewMiddleware(HumanInTheLoopMiddleware):
|
||||
return super().after_model(state, runtime)
|
||||
return None
|
||||
if mode == "manual":
|
||||
# Mirror the sync path: an injected auto context on a resume child
|
||||
# (thread flipped to auto after the parent started) bypasses HITL
|
||||
# after successful re-verification.
|
||||
if review is not None and review.get("requested_mode") == "auto":
|
||||
try:
|
||||
await _resolve_async(current_run_id, review)
|
||||
except AutoReviewVerificationError:
|
||||
return super().after_model(state, runtime)
|
||||
return None
|
||||
return super().after_model(state, runtime)
|
||||
raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID")
|
||||
|
||||
@@ -141,10 +141,13 @@ def _normalize(request: ModelRequest, exc: BaseException) -> ProviderStreamError
|
||||
- ``AgentControlError`` — a platform-owned typed decision. Gateway route
|
||||
fallback and canonical error mapping depend on its concrete type and
|
||||
structured fields, so it must never become a provider incident.
|
||||
- ``EvoRuntimeError`` — a stable host/runtime error that has already been
|
||||
classified across the Gateway boundary and must retain its code.
|
||||
- Models we don't recognize as a provider SDK.
|
||||
"""
|
||||
from langchain_core.exceptions import ContextOverflowError
|
||||
|
||||
from ..llm.contracts import EvoRuntimeError
|
||||
from ..llm.errors import (
|
||||
AgentControlError,
|
||||
ProviderStreamError,
|
||||
@@ -166,6 +169,9 @@ def _normalize(request: ModelRequest, exc: BaseException) -> ProviderStreamError
|
||||
if isinstance(exc, AgentControlError):
|
||||
return None
|
||||
|
||||
if isinstance(exc, EvoRuntimeError):
|
||||
return None
|
||||
|
||||
# LangGraph control-flow / structural signals must propagate
|
||||
# untouched, regardless of which caller invoked us.
|
||||
if _should_pass_through(exc):
|
||||
|
||||
@@ -12,6 +12,8 @@ from langchain.agents.middleware.types import AgentMiddleware
|
||||
from langchain_core.messages import ToolMessage, message_to_dict, messages_from_dict
|
||||
from langgraph.types import Command
|
||||
|
||||
from EvoScientist.llm.contracts import EvoRuntimeError
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from langchain.agents.middleware.types import ToolCallRequest
|
||||
|
||||
@@ -91,6 +93,15 @@ async def _post(proxy: Mapping[str, str], phase: str, payload: dict[str, Any]) -
|
||||
"envelope_signature": proxy["envelope_signature"],
|
||||
},
|
||||
)
|
||||
if response.is_error:
|
||||
try:
|
||||
error_body = response.json()
|
||||
except ValueError:
|
||||
error_body = None
|
||||
detail = error_body.get("detail") if isinstance(error_body, Mapping) else None
|
||||
code = detail.get("code") if isinstance(detail, Mapping) else None
|
||||
if isinstance(code, str) and code.isascii() and code.replace("_", "").isalnum():
|
||||
raise EvoRuntimeError(code)
|
||||
response.raise_for_status()
|
||||
return dict(response.json())
|
||||
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
"""Bound synchronous sub-agent calls in hosted Web runs."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
import json
|
||||
from collections.abc import Awaitable, Callable
|
||||
from typing import Any
|
||||
|
||||
from langchain.agents.middleware.types import AgentMiddleware, ToolCallRequest
|
||||
from langchain_core.messages import ToolMessage
|
||||
from langgraph.types import Command
|
||||
|
||||
|
||||
class SubagentTimeoutMiddleware(AgentMiddleware):
|
||||
"""Cancel a synchronous ``task`` call that exceeds the Web time budget."""
|
||||
|
||||
@property
|
||||
def name(self) -> str:
|
||||
return "subagent_timeout"
|
||||
|
||||
def __init__(self, timeout_seconds: float = 180.0) -> None:
|
||||
super().__init__()
|
||||
if timeout_seconds <= 0:
|
||||
raise ValueError("timeout_seconds must be positive")
|
||||
self.timeout_seconds = float(timeout_seconds)
|
||||
|
||||
@staticmethod
|
||||
async def _cancel_task(task: asyncio.Future[Any]) -> None:
|
||||
task.cancel()
|
||||
try:
|
||||
await asyncio.shield(task)
|
||||
except asyncio.CancelledError:
|
||||
if not task.done():
|
||||
task.add_done_callback(SubagentTimeoutMiddleware._consume_task_result)
|
||||
raise
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
@staticmethod
|
||||
def _consume_task_result(task: asyncio.Future[Any]) -> None:
|
||||
try:
|
||||
task.result()
|
||||
except (asyncio.CancelledError, Exception):
|
||||
pass
|
||||
|
||||
async def awrap_tool_call(
|
||||
self,
|
||||
request: ToolCallRequest,
|
||||
handler: Callable[[ToolCallRequest], Awaitable[ToolMessage | Command[Any]]],
|
||||
) -> ToolMessage | Command[Any]:
|
||||
if str(request.tool_call.get("name") or "") != "task":
|
||||
return await handler(request)
|
||||
task = asyncio.ensure_future(handler(request))
|
||||
try:
|
||||
done, _pending = await asyncio.wait(
|
||||
{task}, timeout=self.timeout_seconds
|
||||
)
|
||||
except asyncio.CancelledError:
|
||||
await self._cancel_task(task)
|
||||
raise
|
||||
if done:
|
||||
return task.result()
|
||||
|
||||
await self._cancel_task(task)
|
||||
current = asyncio.current_task()
|
||||
if current is not None and current.cancelling():
|
||||
raise asyncio.CancelledError
|
||||
payload = {
|
||||
"code": "SUBAGENT_TIMEOUT",
|
||||
"message": (
|
||||
"The delegated sub-agent exceeded the hosted Web time limit. "
|
||||
"Continue with available evidence or use the controlled web search tool directly."
|
||||
),
|
||||
"retryable": True,
|
||||
"timeout_seconds": self.timeout_seconds,
|
||||
}
|
||||
return ToolMessage(
|
||||
content=json.dumps(payload, ensure_ascii=False),
|
||||
tool_call_id=str(request.tool_call.get("id") or "subagent_timeout"),
|
||||
name="task",
|
||||
status="error",
|
||||
additional_kwargs={"error_code": "SUBAGENT_TIMEOUT"},
|
||||
)
|
||||
@@ -132,6 +132,20 @@ def _node_version(node: Path) -> tuple[int, int, int]:
|
||||
return parts
|
||||
|
||||
|
||||
def _existing_resolved_paths(candidates: tuple[str, ...]) -> tuple[str, ...]:
|
||||
resolved: list[str] = []
|
||||
seen: set[str] = set()
|
||||
for candidate in candidates:
|
||||
try:
|
||||
value = str(Path(candidate).resolve(strict=True))
|
||||
except OSError:
|
||||
continue
|
||||
if value not in seen:
|
||||
seen.add(value)
|
||||
resolved.append(value)
|
||||
return tuple(resolved)
|
||||
|
||||
|
||||
def _system_read_paths() -> tuple[str, ...]:
|
||||
candidates = (
|
||||
(
|
||||
@@ -162,7 +176,7 @@ def _system_read_paths() -> tuple[str, ...]:
|
||||
"/dev/urandom",
|
||||
)
|
||||
)
|
||||
return tuple(path for path in candidates if Path(path).exists())
|
||||
return _existing_resolved_paths(candidates)
|
||||
|
||||
|
||||
def _assert_install_contract() -> NativeSandboxInstallation:
|
||||
@@ -262,6 +276,7 @@ def _sandbox_settings(
|
||||
# srt adds these shared compatibility paths even when callers do
|
||||
# not request them. Explicit deny wins over that built-in allow.
|
||||
"denyWrite": [
|
||||
str(files_dir / "uploads"),
|
||||
"/tmp/claude",
|
||||
"/private/tmp/claude",
|
||||
"/dev/tty",
|
||||
@@ -269,7 +284,10 @@ def _sandbox_settings(
|
||||
"/dev/autofs_nowait",
|
||||
],
|
||||
},
|
||||
"enableWeakerNestedSandbox": False,
|
||||
"enableWeakerNestedSandbox": os.getenv(
|
||||
"EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", ""
|
||||
).strip().lower()
|
||||
in {"1", "true", "yes", "on"},
|
||||
"enableWeakerNetworkIsolation": False,
|
||||
"allowAppleEvents": False,
|
||||
"allowPty": False,
|
||||
@@ -622,6 +640,19 @@ _READY_STATE = "unchecked"
|
||||
_READY_ERROR: str | None = None
|
||||
|
||||
|
||||
def _network_preflight_probe(port: int, unix_path: Path) -> str:
|
||||
return (
|
||||
"import os,socket;\n"
|
||||
"assert 'OPENAI_API_KEY' not in os.environ\n"
|
||||
f"t=socket.socket(); tcp=t.connect_ex(('127.0.0.1',{port})); t.close()\n"
|
||||
"try:\n"
|
||||
f" u=socket.socket(socket.AF_UNIX); unix=u.connect_ex({str(unix_path)!r}); u.close()\n"
|
||||
"except OSError:\n"
|
||||
" unix=1\n"
|
||||
"assert tcp != 0 and unix != 0\n"
|
||||
)
|
||||
|
||||
|
||||
def _run_preflight(installation: NativeSandboxInstallation) -> None:
|
||||
# AF_UNIX paths are limited to roughly 100 bytes on both target platforms;
|
||||
# a deployment workspace path can already exceed that before the filename.
|
||||
@@ -632,6 +663,10 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None:
|
||||
files_dir.mkdir(mode=0o700)
|
||||
runtime_dir.mkdir(mode=0o700)
|
||||
(files_dir / "probe.txt").write_text("allowed", encoding="utf-8")
|
||||
uploads_dir = files_dir / "uploads"
|
||||
uploads_dir.mkdir(mode=0o700)
|
||||
upload_source = uploads_dir / "source.txt"
|
||||
upload_source.write_text("immutable", encoding="utf-8")
|
||||
outside = root / "outside-secret.txt"
|
||||
outside.write_text("secret", encoding="utf-8")
|
||||
control_secret = runtime_dir / "control-secret.txt"
|
||||
@@ -649,16 +684,13 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None:
|
||||
unix_server.bind(str(unix_path))
|
||||
unix_server.listen(1)
|
||||
|
||||
python_probe = (
|
||||
"import os,socket,sys;"
|
||||
"assert 'OPENAI_API_KEY' not in os.environ;"
|
||||
f"t=socket.socket(); tcp=t.connect_ex(('127.0.0.1',{port})); t.close();"
|
||||
f"u=socket.socket(socket.AF_UNIX); unix=u.connect_ex({str(unix_path)!r}); u.close();"
|
||||
"sys.exit(0 if tcp != 0 and unix != 0 else 9)"
|
||||
)
|
||||
python_probe = _network_preflight_probe(port, unix_path)
|
||||
command = " && ".join(
|
||||
(
|
||||
'test "$(cat probe.txt)" = allowed',
|
||||
'test "$(cat uploads/source.txt)" = immutable',
|
||||
"! sh -c 'printf changed > uploads/source.txt' 2>/dev/null",
|
||||
"! rm uploads/source.txt 2>/dev/null",
|
||||
"printf written > written.txt",
|
||||
'printf temporary > "$TMPDIR/probe.tmp"',
|
||||
"pandoc --version >/dev/null",
|
||||
@@ -695,6 +727,8 @@ def _run_preflight(installation: NativeSandboxInstallation) -> None:
|
||||
)
|
||||
if outside.read_text(encoding="utf-8") != "secret":
|
||||
raise NativeSandboxUnavailable("native sandbox preflight escaped workspace")
|
||||
if upload_source.read_text(encoding="utf-8") != "immutable":
|
||||
raise NativeSandboxUnavailable("native sandbox preflight modified uploads")
|
||||
|
||||
|
||||
def ensure_native_sandbox_ready() -> None:
|
||||
|
||||
@@ -267,6 +267,23 @@ echo "PID: $!" # check: ps -p <PID> · stop: kill <PID> · read
|
||||
|
||||
This prevents blocking the conversation during long operations."""
|
||||
|
||||
_FILE_PROGRAMMING_GUIDELINES = """# Safe Binary File Programming
|
||||
|
||||
- When `read_file` returns `BINARY_PROCESSING_REQUIRED`, use `execute` to inspect the referenced workspace file with a short, bounded program. Do not report the format as unsupported without trying the indicated safe workflow.
|
||||
- For ZIP/TAR archives, list and validate members before selective extraction; never use `extractall` on an untrusted archive. Reject absolute paths, `..`, links escaping the destination, excessive member counts, expanded sizes, and compression ratios. Never execute files extracted from an archive.
|
||||
- For SQLite, open `file:<path>?mode=ro&immutable=1` with `uri=True`, set `PRAGMA query_only=ON`, inspect `sqlite_master`, and use bounded `SELECT ... LIMIT ...` queries. Never use `ATTACH DATABASE`, enable extensions, or modify the source database.
|
||||
- Do not modify original files under `uploads/`. Write extracted members, query results, converted documents, and other derived artifacts under `artifacts/` or `.ai4sci/`.
|
||||
- Text extracted by `read_file` from PDF or Office files is a semantic view, not the original container bytes. Never write that text back to the source document with `write_file` or `edit_file`.
|
||||
"""
|
||||
|
||||
_WEB_RESEARCH_GUIDELINES = """# Controlled Web Research
|
||||
|
||||
- `search_observations` searches local memory only. It does not access the internet and must never be presented as live web search.
|
||||
- For current facts, public webpages, source discovery, or URL verification, call `tavily_search` first when it is available. Use `web_search` only when that configured search provider is available.
|
||||
- Do not use `execute`, `curl`, or `httpx` to reach the public internet. The execution sandbox intentionally blocks raw networking; controlled search tools are the only supported network path.
|
||||
- If controlled web search is unavailable or fails, report that specific limitation once and continue with clearly labeled non-live evidence. Do not repeatedly probe DNS, proxies, direct IPs, or local ports.
|
||||
"""
|
||||
|
||||
# Sandbox (default) header: virtual `/` workspace.
|
||||
_SHELL_GUIDELINES_SANDBOX_HEADER = """# Shell Execution Guidelines
|
||||
|
||||
@@ -460,6 +477,8 @@ def get_system_prompt(
|
||||
REPORT_TEMPLATE,
|
||||
WRITING_GUIDELINES,
|
||||
shell_guidelines,
|
||||
_FILE_PROGRAMMING_GUIDELINES,
|
||||
_WEB_RESEARCH_GUIDELINES,
|
||||
DELEGATION_STRATEGY,
|
||||
ASYNC_NOTIFICATIONS,
|
||||
]
|
||||
|
||||
@@ -1630,6 +1630,210 @@ async def db_stats(top_n: int = 5) -> dict[str, Any]:
|
||||
return out
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# History pruning / VACUUM (gateway timer + admin endpoints)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
|
||||
def _uuid6_unix_ts(checkpoint_id: str) -> float | None:
|
||||
"""Extract the unix timestamp embedded in a UUIDv6 checkpoint id.
|
||||
|
||||
Returns ``None`` for non-UUID or non-v6 ids — legacy ids carry no usable
|
||||
timestamp, and callers must skip those threads rather than guess an age.
|
||||
"""
|
||||
try:
|
||||
u = uuid.UUID(checkpoint_id)
|
||||
except (ValueError, AttributeError, TypeError):
|
||||
return None
|
||||
if u.version != 6:
|
||||
return None
|
||||
ts60 = ((u.int >> 80) << 12) | ((u.int >> 64) & 0x0FFF)
|
||||
return ts60 / 10_000_000 - 12219292800
|
||||
|
||||
|
||||
async def _count_thread_rows(
|
||||
conn: aiosqlite.Connection, thread_id: str
|
||||
) -> tuple[int, int]:
|
||||
"""Return ``(checkpoints, writes)`` row counts for one thread."""
|
||||
async with conn.execute(
|
||||
"SELECT COUNT(*) FROM checkpoints WHERE thread_id = ?", (thread_id,)
|
||||
) as cur:
|
||||
row = await cur.fetchone()
|
||||
ck = int(row[0]) if row else 0
|
||||
wr = 0
|
||||
if await _table_exists(conn, "writes"):
|
||||
async with conn.execute(
|
||||
"SELECT COUNT(*) FROM writes WHERE thread_id = ?", (thread_id,)
|
||||
) as cur:
|
||||
row = await cur.fetchone()
|
||||
wr = int(row[0]) if row else 0
|
||||
return ck, wr
|
||||
|
||||
|
||||
async def _prune_thread_on_conn(
|
||||
conn: aiosqlite.Connection, thread_id: str, keep_last: int
|
||||
) -> tuple[int, int]:
|
||||
"""Prune one thread on an open connection; return rows deleted per table.
|
||||
|
||||
Reuses :meth:`PruningCheckpointer._prune_after_put` per
|
||||
``(thread_id, checkpoint_ns)`` group so retention is identical to the
|
||||
live write path, including DeltaChannel snapshot-chain preservation.
|
||||
Only ``metadata.agent_name == AGENT_NAME`` rows are ever touched.
|
||||
"""
|
||||
saver = PruningCheckpointer(conn, keep_per_ns=max(1, int(keep_last)))
|
||||
before_ck, before_wr = await _count_thread_rows(conn, thread_id)
|
||||
async with conn.execute(
|
||||
"SELECT DISTINCT checkpoint_ns FROM checkpoints "
|
||||
"WHERE thread_id = ? AND json_extract(metadata, '$.agent_name') = ?",
|
||||
(thread_id, AGENT_NAME),
|
||||
) as cur:
|
||||
namespaces = [r[0] for r in await cur.fetchall()]
|
||||
for ns in namespaces:
|
||||
await saver._prune_after_put(thread_id, ns or "")
|
||||
after_ck, after_wr = await _count_thread_rows(conn, thread_id)
|
||||
return before_ck - after_ck, before_wr - after_wr
|
||||
|
||||
|
||||
async def prune_thread_history(
|
||||
thread_id: str, keep_last: int = 2, db_path: str | None = None
|
||||
) -> dict[str, int]:
|
||||
"""Prune one thread's history, keeping its ``keep_last`` most recent rows.
|
||||
|
||||
Returns ``{"deleted_checkpoints": int, "deleted_writes": int}``.
|
||||
"""
|
||||
path = str(db_path or get_db_path())
|
||||
result = {"deleted_checkpoints": 0, "deleted_writes": 0}
|
||||
if not Path(path).exists():
|
||||
return result
|
||||
async with aiosqlite.connect(path, timeout=30.0) as conn:
|
||||
if not await _table_exists(conn, "checkpoints"):
|
||||
return result
|
||||
ck, wr = await _prune_thread_on_conn(conn, str(thread_id), keep_last)
|
||||
result["deleted_checkpoints"] = ck
|
||||
result["deleted_writes"] = wr
|
||||
return result
|
||||
|
||||
|
||||
async def prune_all_stale_threads(
|
||||
max_age_hours: float = 72,
|
||||
keep_last: int = 2,
|
||||
db_path: str | None = None,
|
||||
) -> dict[str, int]:
|
||||
"""Prune every EvoScientist thread idle for at least ``max_age_hours``.
|
||||
|
||||
A thread is stale when its newest checkpoint (UUIDv6 timestamp) is older
|
||||
than the cutoff. Threads whose newest id has no parseable timestamp are
|
||||
skipped. Returns ``{"databases_processed", "threads_pruned",
|
||||
"total_deleted_checkpoints", "total_deleted_writes"}``.
|
||||
"""
|
||||
path = str(db_path or get_db_path())
|
||||
result = {
|
||||
"databases_processed": 0,
|
||||
"threads_pruned": 0,
|
||||
"total_deleted_checkpoints": 0,
|
||||
"total_deleted_writes": 0,
|
||||
}
|
||||
if not Path(path).exists():
|
||||
return result
|
||||
cutoff = time.time() - float(max_age_hours) * 3600.0
|
||||
async with aiosqlite.connect(path, timeout=30.0) as conn:
|
||||
if not await _table_exists(conn, "checkpoints"):
|
||||
return result
|
||||
result["databases_processed"] = 1
|
||||
async with conn.execute(
|
||||
"SELECT thread_id, MAX(checkpoint_id) FROM checkpoints "
|
||||
"WHERE json_extract(metadata, '$.agent_name') = ? "
|
||||
"GROUP BY thread_id",
|
||||
(AGENT_NAME,),
|
||||
) as cur:
|
||||
rows = await cur.fetchall()
|
||||
stale = [
|
||||
tid
|
||||
for tid, newest in rows
|
||||
if (ts := _uuid6_unix_ts(newest)) is not None and ts < cutoff
|
||||
]
|
||||
for tid in stale:
|
||||
ck, wr = await _prune_thread_on_conn(conn, tid, keep_last)
|
||||
if ck or wr:
|
||||
result["threads_pruned"] += 1
|
||||
result["total_deleted_checkpoints"] += ck
|
||||
result["total_deleted_writes"] += wr
|
||||
return result
|
||||
|
||||
|
||||
async def list_all_thread_ids(db_path: str | None = None) -> list[str]:
|
||||
"""Return all EvoScientist thread ids in the sessions DB."""
|
||||
path = str(db_path or get_db_path())
|
||||
if not Path(path).exists():
|
||||
return []
|
||||
async with aiosqlite.connect(path, timeout=30.0) as conn:
|
||||
if not await _table_exists(conn, "checkpoints"):
|
||||
return []
|
||||
async with conn.execute(
|
||||
"SELECT DISTINCT thread_id FROM checkpoints "
|
||||
"WHERE json_extract(metadata, '$.agent_name') = ?",
|
||||
(AGENT_NAME,),
|
||||
) as cur:
|
||||
return [r[0] for r in await cur.fetchall()]
|
||||
|
||||
|
||||
def list_all_session_db_paths() -> list[Path]:
|
||||
"""Return existing session DB paths.
|
||||
|
||||
The current storage layout uses a single shared DB (``get_db_path()``),
|
||||
so this returns a one-element list when it exists, else ``[]``.
|
||||
"""
|
||||
path = get_db_path()
|
||||
return [path] if path.exists() else []
|
||||
|
||||
|
||||
async def vacuum_db(db_path: str | None = None) -> dict[str, Any]:
|
||||
"""Run ``VACUUM`` on the sessions DB to reclaim freed pages."""
|
||||
path = str(db_path or get_db_path())
|
||||
p = Path(path)
|
||||
before = p.stat().st_size if p.exists() else 0
|
||||
if p.exists():
|
||||
async with aiosqlite.connect(path, timeout=120.0) as conn:
|
||||
await conn.execute("VACUUM")
|
||||
await conn.commit()
|
||||
after = p.stat().st_size if p.exists() else 0
|
||||
return {
|
||||
"db_path": path,
|
||||
"size_before_bytes": before,
|
||||
"size_after_bytes": after,
|
||||
}
|
||||
|
||||
|
||||
async def get_aggregated_storage_stats() -> dict[str, Any]:
|
||||
"""Return :func:`db_stats` plus per-thread checkpoint depth stats."""
|
||||
stats = await db_stats()
|
||||
depth: dict[str, Any] = {"min": 0, "max": 0, "avg": 0.0}
|
||||
path = Path(stats["db_path"])
|
||||
if path.exists():
|
||||
try:
|
||||
async with aiosqlite.connect(str(path), timeout=30.0) as conn:
|
||||
if await _table_exists(conn, "checkpoints"):
|
||||
async with conn.execute(
|
||||
"SELECT MIN(n), MAX(n), AVG(n) FROM ("
|
||||
" SELECT COUNT(*) AS n FROM checkpoints "
|
||||
" WHERE json_extract(metadata, '$.agent_name') = ? "
|
||||
" GROUP BY thread_id"
|
||||
")",
|
||||
(AGENT_NAME,),
|
||||
) as cur:
|
||||
row = await cur.fetchone()
|
||||
if row and row[0] is not None:
|
||||
depth = {
|
||||
"min": int(row[0]),
|
||||
"max": int(row[1]),
|
||||
"avg": round(float(row[2]), 2),
|
||||
}
|
||||
except aiosqlite.Error:
|
||||
# Read-only diagnostic — mirror db_stats and degrade to zeros.
|
||||
pass
|
||||
return {**stats, "thread_depth": depth}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# langgraph-api / WebUI checkpointer factory
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
@@ -14,6 +14,16 @@ from tavily import TavilyClient
|
||||
|
||||
# Lazy initialization - only create client when needed
|
||||
_tavily_client = None
|
||||
MAX_SEARCH_RESULTS = 5
|
||||
MAX_DISPLAY_QUERY_CHARS = 512
|
||||
MAX_DISPLAY_TITLE_CHARS = 512
|
||||
MAX_DISPLAY_URL_CHARS = 2_048
|
||||
MAX_PAGE_CONTENT_CHARS = 4_000
|
||||
MAX_SEARCH_RESULT_CHARS = 16_000
|
||||
_TRUNCATION_MARKER = "\n\n[page content truncated]"
|
||||
_SEARCH_TRUNCATION_MARKER = (
|
||||
"\n[search result content truncated to preserve all titles and URLs]"
|
||||
)
|
||||
|
||||
|
||||
def _get_tavily_client() -> TavilyClient:
|
||||
@@ -46,6 +56,12 @@ async def fetch_webpage_content(url: str, timeout: float = 10.0) -> str:
|
||||
async with httpx.AsyncClient() as client:
|
||||
response = await client.get(url, headers=headers, timeout=timeout)
|
||||
response.raise_for_status()
|
||||
content_type = response.headers.get("content-type", "").lower()
|
||||
if not any(
|
||||
allowed in content_type
|
||||
for allowed in ("text/", "application/xhtml+xml")
|
||||
):
|
||||
return f"Error fetching content from {url}: unsupported content type {content_type or 'unknown'}"
|
||||
return markdownify(response.text)
|
||||
except Exception as e:
|
||||
return f"Error fetching content from {url}: {e!s}"
|
||||
@@ -72,40 +88,71 @@ async def tavily_search(
|
||||
"""
|
||||
|
||||
def _sync_search() -> dict:
|
||||
bounded_max_results = max(1, min(int(max_results), MAX_SEARCH_RESULTS))
|
||||
return _get_tavily_client().search(
|
||||
query,
|
||||
max_results=max_results,
|
||||
max_results=bounded_max_results,
|
||||
topic=topic,
|
||||
)
|
||||
|
||||
try:
|
||||
# Run Tavily search asynchronously
|
||||
# Run Tavily search asynchronously through the controlled host path.
|
||||
search_results = await asyncio.to_thread(_sync_search)
|
||||
from EvoScientist.runtime_integrations import record_service_usage
|
||||
|
||||
await record_service_usage("tavily", "search")
|
||||
|
||||
# Fetch full content for each URL concurrently
|
||||
results = search_results.get("results", [])
|
||||
if not results:
|
||||
return f"No results found for '{query}'"
|
||||
|
||||
# Fetch all webpages concurrently
|
||||
fetch_tasks = [fetch_webpage_content(r["url"]) for r in results]
|
||||
contents = await asyncio.gather(*fetch_tasks)
|
||||
|
||||
# Format results
|
||||
normalized = []
|
||||
for result, fetched_content in zip(results, contents, strict=False):
|
||||
title = str(result.get("title") or "Untitled")[:MAX_DISPLAY_TITLE_CHARS]
|
||||
raw_url = str(result.get("url") or "")
|
||||
url = (
|
||||
raw_url
|
||||
if len(raw_url) <= MAX_DISPLAY_URL_CHARS
|
||||
else raw_url[: MAX_DISPLAY_URL_CHARS - len("...[URL truncated]")]
|
||||
+ "...[URL truncated]"
|
||||
)
|
||||
tavily_summary = str(result.get("content") or "").strip()
|
||||
fetch_failed = fetched_content.startswith("Error fetching content from ")
|
||||
content = tavily_summary if fetch_failed and tavily_summary else fetched_content
|
||||
fetch_note = (
|
||||
"\n\n> Source page fetch failed; showing the Tavily-indexed summary."
|
||||
if fetch_failed and tavily_summary
|
||||
else ""
|
||||
)
|
||||
normalized.append((title, url, content, fetch_note))
|
||||
|
||||
display_query = query[:MAX_DISPLAY_QUERY_CHARS]
|
||||
prefix = f"Found {len(normalized)} live web result(s) for '{display_query}':\n\n"
|
||||
metadata_blocks = [f"## {title}\n**URL:** {url}\n\n" for title, url, _, _ in normalized]
|
||||
fixed_chars = len(prefix) + sum(len(block) + len("\n\n---\n") for block in metadata_blocks)
|
||||
remaining = max(0, MAX_SEARCH_RESULT_CHARS - fixed_chars - len(_SEARCH_TRUNCATION_MARKER))
|
||||
per_result_budget = remaining // max(1, len(normalized))
|
||||
|
||||
result_texts = []
|
||||
for result, content in zip(results, contents, strict=False):
|
||||
result_text = f"""## {result["title"]}
|
||||
**URL:** {result["url"]}
|
||||
content_truncated = False
|
||||
for metadata, (_, _, content, fetch_note) in zip(metadata_blocks, normalized, strict=True):
|
||||
content_budget = max(0, min(MAX_PAGE_CONTENT_CHARS, per_result_budget - len(fetch_note)))
|
||||
if len(content) > content_budget:
|
||||
marker_budget = min(len(_TRUNCATION_MARKER), content_budget)
|
||||
content = (
|
||||
content[: content_budget - marker_budget]
|
||||
+ _TRUNCATION_MARKER[:marker_budget]
|
||||
)
|
||||
content_truncated = True
|
||||
result_texts.append(f"{metadata}{content}{fetch_note}\n\n---\n")
|
||||
|
||||
{content}
|
||||
|
||||
---
|
||||
"""
|
||||
result_texts.append(result_text)
|
||||
|
||||
return f"""Found {len(result_texts)} result(s) for '{query}':
|
||||
|
||||
{"".join(result_texts)}"""
|
||||
formatted = prefix + "".join(result_texts)
|
||||
if content_truncated:
|
||||
formatted += _SEARCH_TRUNCATION_MARKER
|
||||
return formatted
|
||||
|
||||
except Exception as e:
|
||||
return f"Search failed: {e!s}"
|
||||
|
||||
@@ -51,10 +51,23 @@ def web_tool_registry_manifest() -> tuple[tuple[dict[str, Any], ...], str]:
|
||||
"additionalProperties": True,
|
||||
"maxProperties": 32,
|
||||
}
|
||||
tool_descriptions = {
|
||||
"tavily_search": (
|
||||
"Search the live public web through the controlled Tavily service. "
|
||||
"Use this for current facts, source discovery, and URL verification. "
|
||||
"Do not use execute, curl, httpx, or raw sandbox networking instead."
|
||||
),
|
||||
"web_search": (
|
||||
"Search the live public web through the configured MCP search provider. "
|
||||
"Use this for current facts and source verification, not local memory recall."
|
||||
),
|
||||
}
|
||||
manifest: tuple[dict[str, Any], ...] = tuple(
|
||||
{
|
||||
"name": name,
|
||||
"description": "EvoScientist Web runtime tool",
|
||||
"description": tool_descriptions.get(
|
||||
name, "EvoScientist Web runtime tool"
|
||||
),
|
||||
"schema": schema,
|
||||
}
|
||||
for name in names
|
||||
|
||||
@@ -375,7 +375,16 @@ def _standard_error(exc: Exception) -> str:
|
||||
|
||||
|
||||
def _is_binary_file(path: str, raw: bytes) -> bool:
|
||||
"""Classify common containers explicitly and fall back to content."""
|
||||
"""Classify common containers explicitly and fall back to content.
|
||||
|
||||
The UTF-8 probe samples the first bytes, so a multi-byte character can be
|
||||
cut in half at the sample boundary (e.g. an 8192-byte cut splitting a
|
||||
3-byte CJK character). ``decode`` with ``errors="ignore"`` would hide real
|
||||
garbage, so instead we re-probe on failure with the trailing partial
|
||||
sequence removed: a decode error that vanishes once the (at most 4-byte)
|
||||
dangling suffix is dropped is a truncation artifact, not binary content.
|
||||
Files that still fail on the trimmed sample are genuinely not UTF-8.
|
||||
"""
|
||||
|
||||
if Path(path).suffix.lower() in _BINARY_EXTENSIONS:
|
||||
return True
|
||||
@@ -384,7 +393,15 @@ def _is_binary_file(path: str, raw: bytes) -> bool:
|
||||
return True
|
||||
try:
|
||||
sample.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
except UnicodeDecodeError as exc:
|
||||
# Only a decode error at the very end of the sample can be a boundary
|
||||
# cut. Errors positioned mid-sample are real invalid bytes.
|
||||
if exc.start >= max(0, len(sample) - 4):
|
||||
try:
|
||||
sample[: exc.start].decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
return True
|
||||
return False
|
||||
return True
|
||||
return False
|
||||
|
||||
@@ -397,6 +414,13 @@ class ScopedFilesystemBackend(BackendProtocol):
|
||||
root_dir, max_search_file_bytes=max_search_file_bytes
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _uploads_are_read_only(file_path: str) -> bool:
|
||||
normalized = "/" + file_path.replace("\\", "/").lstrip("/")
|
||||
return normalized == "/workspace/uploads" or normalized.startswith(
|
||||
"/workspace/uploads/"
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _search_path(path: str | None) -> str:
|
||||
if path in {None, "/"}:
|
||||
@@ -426,15 +450,72 @@ class ScopedFilesystemBackend(BackendProtocol):
|
||||
return LsResult(error=f"Cannot list '{path}': {_standard_error(exc)}")
|
||||
|
||||
def read(self, file_path: str, offset: int = 0, limit: int = 2000) -> ReadResult:
|
||||
from .document_extract import (
|
||||
MAX_DOCUMENT_BYTES,
|
||||
DocumentExtractionError,
|
||||
binary_processing_guidance,
|
||||
classify_file,
|
||||
extract_document_bytes,
|
||||
paginate_document_text,
|
||||
prepare_image_bytes,
|
||||
)
|
||||
|
||||
try:
|
||||
entry = self.workspace.entry(file_path)
|
||||
with self.workspace.open_binary(file_path) as handle:
|
||||
head = handle.read(4096)
|
||||
kind = classify_file(file_path, head)
|
||||
|
||||
if kind == "document":
|
||||
if entry.size > MAX_DOCUMENT_BYTES:
|
||||
return ReadResult(
|
||||
error=(
|
||||
f"DOCUMENT_TOO_LARGE: '{file_path}' is {entry.size} bytes; "
|
||||
f"document extraction limit is {MAX_DOCUMENT_BYTES} bytes"
|
||||
)
|
||||
)
|
||||
handle.seek(0)
|
||||
raw = handle.read(MAX_DOCUMENT_BYTES + 1)
|
||||
try:
|
||||
extracted = extract_document_bytes(raw, file_path)
|
||||
content = paginate_document_text(
|
||||
extracted, offset=max(0, offset), limit=max(1, limit)
|
||||
)
|
||||
except DocumentExtractionError as exc:
|
||||
return ReadResult(error=str(exc))
|
||||
return ReadResult(
|
||||
file_data={"content": content, "encoding": "utf-8"}
|
||||
)
|
||||
|
||||
if kind == "image":
|
||||
handle.seek(0)
|
||||
raw = handle.read()
|
||||
try:
|
||||
raw = prepare_image_bytes(raw, file_path)
|
||||
except DocumentExtractionError as exc:
|
||||
return ReadResult(error=str(exc))
|
||||
return ReadResult(
|
||||
file_data={
|
||||
"content": base64.standard_b64encode(raw).decode("ascii"),
|
||||
"encoding": "base64",
|
||||
}
|
||||
)
|
||||
|
||||
if kind in {"archive", "database", "executable", "dataset", "media"}:
|
||||
return ReadResult(
|
||||
error=binary_processing_guidance(file_path, kind, entry.size)
|
||||
)
|
||||
|
||||
if _is_binary_file(file_path, head):
|
||||
return ReadResult(
|
||||
error=binary_processing_guidance(file_path, "binary", entry.size)
|
||||
)
|
||||
|
||||
handle.seek(0)
|
||||
raw = handle.read()
|
||||
if _is_binary_file(file_path, raw):
|
||||
return ReadResult(
|
||||
file_data={
|
||||
"content": base64.standard_b64encode(raw).decode("ascii"),
|
||||
"encoding": "base64",
|
||||
}
|
||||
error=binary_processing_guidance(file_path, "binary", entry.size)
|
||||
)
|
||||
content = raw.decode("utf-8")
|
||||
empty = check_empty_content(content)
|
||||
@@ -455,6 +536,29 @@ class ScopedFilesystemBackend(BackendProtocol):
|
||||
return ReadResult(error=f"Error reading file '{file_path}': {_standard_error(exc)}")
|
||||
|
||||
def write(self, file_path: str, content: str) -> WriteResult:
|
||||
if self._uploads_are_read_only(file_path):
|
||||
return WriteResult(
|
||||
error=(
|
||||
f"Cannot modify original upload '{file_path}'. "
|
||||
"Write derived content under /workspace/artifacts/ or /workspace/.ai4sci/."
|
||||
)
|
||||
)
|
||||
from .document_extract import (
|
||||
ARCHIVE_EXTENSIONS,
|
||||
DATABASE_EXTENSIONS,
|
||||
DOCUMENT_EXTENSIONS,
|
||||
)
|
||||
|
||||
if Path(file_path).suffix.lower() in (
|
||||
DOCUMENT_EXTENSIONS | ARCHIVE_EXTENSIONS | DATABASE_EXTENSIONS
|
||||
):
|
||||
return WriteResult(
|
||||
error=(
|
||||
f"Cannot write plain text to binary container '{file_path}'. "
|
||||
"Use execute with an appropriate document, archive, or database "
|
||||
"library and write a derived file under artifacts/."
|
||||
)
|
||||
)
|
||||
try:
|
||||
self.workspace.write_new(file_path, content.encode("utf-8"))
|
||||
return WriteResult(path=file_path)
|
||||
@@ -472,6 +576,29 @@ class ScopedFilesystemBackend(BackendProtocol):
|
||||
new_string: str,
|
||||
replace_all: bool = False,
|
||||
) -> EditResult:
|
||||
if self._uploads_are_read_only(file_path):
|
||||
return EditResult(
|
||||
error=(
|
||||
f"Cannot modify original upload '{file_path}'. "
|
||||
"Write derived content under /workspace/artifacts/ or /workspace/.ai4sci/."
|
||||
)
|
||||
)
|
||||
from .document_extract import (
|
||||
ARCHIVE_EXTENSIONS,
|
||||
DATABASE_EXTENSIONS,
|
||||
DOCUMENT_EXTENSIONS,
|
||||
)
|
||||
|
||||
if Path(file_path).suffix.lower() in (
|
||||
DOCUMENT_EXTENSIONS | ARCHIVE_EXTENSIONS | DATABASE_EXTENSIONS
|
||||
):
|
||||
return EditResult(
|
||||
error=(
|
||||
f"Cannot edit binary container '{file_path}' with text replacement. "
|
||||
"Use execute with an appropriate library and write a derived file "
|
||||
"under artifacts/."
|
||||
)
|
||||
)
|
||||
try:
|
||||
with self.workspace.open_binary(file_path) as handle:
|
||||
content = handle.read().decode("utf-8")
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
global-exclude *.py[cod]
|
||||
prune **/__pycache__
|
||||
@@ -0,0 +1,110 @@
|
||||
# Ai4Sci 有界文件读取与 Agent 自主编程实施方案
|
||||
|
||||
> 状态:实施基线 v1.0
|
||||
> 范围:EvoScientist Core + Ai4Sci-Web 必要接线
|
||||
> 原则:程序提供事实和安全边界,主 Agent 使用现有 read_file/execute 完成渐进处理。
|
||||
|
||||
## 1. 目标
|
||||
|
||||
修复 Office/PDF 等二进制完整 Base64 进入模型上下文的问题,同时保留 Agent 对 ZIP、数据库和其他研究文件的自主编程能力。
|
||||
|
||||
本阶段不新增 inspect_file/process_file,不新增文件路由模型,不新增数据库表,不改变 SSE 事件协议。
|
||||
|
||||
## 2. 固定处理契约
|
||||
|
||||
| 类型 | read_file 行为 | 后续处理 |
|
||||
|---|---|---|
|
||||
| UTF-8文本/源码 | offset/limit 分页文本 | 主模型直接分析 |
|
||||
| 图片 | 有界媒体块 | 多模态模型分析 |
|
||||
| PDF/Office/ODF/RTF/EPUB | 提取为 Markdown/文本并分页 | 需要视觉信息时 Agent 用 execute 渲染指定页 |
|
||||
| ZIP/TAR/7Z/RAR | 返回结构化事实和安全编程要求,不返回内容 | Agent 用 execute 先列清单,再选择性解压 |
|
||||
| SQLite/DB | 返回结构化事实和只读查询要求 | Agent 用 sqlite3 mode=ro/query_only 编程查询 |
|
||||
| 数据集/音视频 | 返回结构化事实和建议命令 | Agent 用现有库/CLI采样、转录或抽帧 |
|
||||
| EXE/库/未知二进制 | 元数据引用;明确禁止执行 | 仅允许静态检查 |
|
||||
|
||||
## 3. 安全与预算
|
||||
|
||||
- 文档源文件最大 50 MiB,转换前检查。
|
||||
- 单次提取文本最大 50,000 字符,按行截断并返回 next_offset 提示。
|
||||
- ZIP禁止 extractall;先检查成员数、总展开量、单成员大小、压缩比、路径穿越、绝对路径和符号链接。
|
||||
- ZIP建议上限:2000成员、500 MiB总展开、100 MiB单成员、100:1压缩比、3层嵌套。
|
||||
- SQLite必须 `file:<path>?mode=ro&immutable=1`、`PRAGMA query_only=ON`,查询必须有LIMIT,结果写入artifacts/。
|
||||
- EXE/DLL/ELF/Mach-O、宏和归档成员不得自动执行。
|
||||
- 原始uploads由文件工具和Native Sandbox双层强制只读;派生文件写入artifacts/或.ai4sci/。
|
||||
- read_file不得对任何非图片二进制返回完整Base64。
|
||||
|
||||
## 4. Core改动
|
||||
|
||||
1. 新增 `EvoScientist/document_extract.py`:
|
||||
- 文件类型集合和分派;
|
||||
- 50 MiB输入限制;
|
||||
- OOXML/可选anydoc文档提取;
|
||||
- 结构化失败信息。
|
||||
2. 修改 `EvoScientist/workspace_files.py`:
|
||||
- 覆盖 Ai4Sci Web full/scoped 的 `NativeWorkspaceBackend` 读取路径;
|
||||
- 文件先分类;
|
||||
- Office/PDF先提取再分页;
|
||||
- 图片先解码校验,限制源文件、像素数和多帧输入,最大边降采样至 2048px 后再走现有Base64媒体契约;
|
||||
- OOXML限制成员数、单成员/总展开量、压缩比,拒绝异常路径、重复和加密成员;
|
||||
- PDF与旧Office转换在隔离子进程中执行,限制60秒和10MiB转换输出;
|
||||
- ZIP/DB/其他二进制返回结构化错误指引;
|
||||
- 避免先完整读取大文档再判断类型。
|
||||
3. 修改 `EvoScientist/prompts.py`:
|
||||
- 加入ZIP和数据库安全编程规则;
|
||||
- 明确文档提取文本不可用普通写工具写回容器。
|
||||
4. 显式声明 `firecrawl-anydoc` 依赖;若不可用或格式不支持,返回可操作错误,不回退Base64。
|
||||
|
||||
## 5. Gateway最小接线
|
||||
|
||||
- 上传阶段负责扩展名、MIME、配额、路径和Magic校验;拒绝未知 `application/octet-stream` 绕过。
|
||||
- 显式允许 `.db/.sqlite/.sqlite3`,并将SQLite MIME别名视为等价;不在Gateway解析数据库。
|
||||
- 附件输入至少保留virtual_path;Office/PDF提取、ZIP解包和数据库查询仍在Agent/execute侧。
|
||||
- 工具事件归一化保留受限格式的 `error_code`,非终止型文件工具失败显示在现有ToolOutputItem工具卡中。
|
||||
- 永久workspace准备失败通过现有ErrorItem和done事件投影为 `failed/incomplete/runtime_error`,并复用 `primary_error_code`;不新增平行终态协议。
|
||||
|
||||
## 6. 错误码
|
||||
|
||||
| code | recoverable | 含义 |
|
||||
|---|---:|---|
|
||||
| DOCUMENT_EXTRACTION_FAILED | true/视原因 | 文档损坏、加密或转换失败 |
|
||||
| DOCUMENT_TOO_LARGE | false | 超过50 MiB输入限制 |
|
||||
| BINARY_PROCESSING_REQUIRED | true | ZIP/DB/媒体等需Agent编程处理 |
|
||||
| UNSUPPORTED_BINARY_FILE | false | 未知或不允许处理的二进制 |
|
||||
| MODEL_RATE_LIMITED | true | 文档处理后模型调用受限 |
|
||||
|
||||
## 7. 测试与验收
|
||||
|
||||
### Core
|
||||
|
||||
- DOCX/PPTX/PDF不返回Base64。
|
||||
- ZIP、SQLite、EXE、未知二进制不返回Base64或NUL乱码。
|
||||
- 图片仍返回Base64图片媒体契约。
|
||||
- 文档提取支持分页和50KB字符上限。
|
||||
- 超过50MiB在转换前拒绝。
|
||||
- 转换失败不回退Base64。
|
||||
- UTF-8边界切分回归保持通过。
|
||||
- prompt包含ZIP安全清单、禁止extractall、SQLite只读模板。
|
||||
|
||||
### 端到端
|
||||
|
||||
- 重放9.6MB PPTX时,任一ToolMessage/Checkpoint中不存在原文件Base64。
|
||||
- Agent可读取提取文本并继续任务。
|
||||
- ZIP先列目录后选择性解压;原ZIP哈希不变。
|
||||
- SQLite只读查询;原DB哈希不变。
|
||||
- 文件处理失败时Run不是正常completed,而是failed/incomplete并带稳定错误码。
|
||||
|
||||
## 8. 实施阶段
|
||||
|
||||
1. P0:文件分类、文档提取、阻断Base64、Core单测。
|
||||
2. P0:ZIP/SQLite安全编程提示和测试。
|
||||
3. P1:Gateway失败投影缺口(若独立评审确认存在)。
|
||||
4. P1:真实PPTX/ZIP/SQLite集成重放。
|
||||
5. 代码审查、静态扫描、回归修复和最终复验。
|
||||
|
||||
## 9. 非目标
|
||||
|
||||
- 不建设独立文件分析微服务。
|
||||
- 不实现专用ZIP/数据库Agent工具。
|
||||
- 不让隐藏模型生成文件处理计划。
|
||||
- 不支持执行上传的可执行文件或宏。
|
||||
- 不在本阶段实现全量视频理解或恶意软件动态沙箱。
|
||||
+9
-1
@@ -1,6 +1,6 @@
|
||||
[project]
|
||||
name = "EvoScientist"
|
||||
version = "0.2.2"
|
||||
dynamic = ["version"]
|
||||
description = "EvoScientist: Towards Self-Evolving AI Scientists for End-to-End Scientific Discovery"
|
||||
readme = "README.md"
|
||||
requires-python = ">=3.11"
|
||||
@@ -17,6 +17,8 @@ classifiers = [
|
||||
]
|
||||
dependencies = [
|
||||
"deepagents[quickjs]~=0.6.12",
|
||||
"firecrawl-anydoc>=0.1.6,<0.2",
|
||||
"pillow>=10.0",
|
||||
"langchain>=1.3",
|
||||
"langchain-anthropic>=1.4",
|
||||
"langchain-openai>=1.2",
|
||||
@@ -111,6 +113,9 @@ build-backend = "setuptools.build_meta"
|
||||
[tool.setuptools.packages.find]
|
||||
include = ["EvoScientist*"]
|
||||
|
||||
[tool.setuptools.dynamic]
|
||||
version = { attr = "EvoScientist._version.__version__" }
|
||||
|
||||
[tool.setuptools.package-data]
|
||||
EvoScientist = [
|
||||
"subagents/*.yaml",
|
||||
@@ -118,6 +123,9 @@ EvoScientist = [
|
||||
"skills/**/*",
|
||||
]
|
||||
|
||||
[tool.setuptools.exclude-package-data]
|
||||
"*" = ["**/__pycache__/*", "**/*.pyc", "**/*.pyo"]
|
||||
|
||||
[tool.pytest.ini_options]
|
||||
testpaths = ["tests"]
|
||||
asyncio_mode = "auto"
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
from pathlib import Path
|
||||
|
||||
_ORIGINAL = """ const rootSkip = new Set(['proc', 'dev', 'sys']);
|
||||
for (const p of readConfig?.denyOnly || []) {
|
||||
if (normalizePathForSandbox(p) === '/') {
|
||||
for (const child of fs.readdirSync('/')) {
|
||||
if (!rootSkip.has(child))
|
||||
readDenyPaths.push('/' + child);
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
_REPLACEMENT = """ const rootSkip = new Set(['proc', 'dev', 'sys']);
|
||||
const rootChildIsAllowedSymlink = (childPath) => {
|
||||
try {
|
||||
if (!fs.lstatSync(childPath).isSymbolicLink())
|
||||
return false;
|
||||
const resolved = fs.realpathSync(childPath);
|
||||
return readAllowPaths.some(allowPath => resolved === allowPath || resolved.startsWith(allowPath + '/'));
|
||||
}
|
||||
catch {
|
||||
return false;
|
||||
}
|
||||
};
|
||||
for (const p of readConfig?.denyOnly || []) {
|
||||
if (normalizePathForSandbox(p) === '/') {
|
||||
for (const child of fs.readdirSync('/')) {
|
||||
const childPath = '/' + child;
|
||||
if (!rootSkip.has(child) && !rootChildIsAllowedSymlink(childPath))
|
||||
readDenyPaths.push(childPath);
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
_TMPFS_ORIGINAL = """ args.push('--ro-bind', allowPath, allowPath);
|
||||
logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
_TMPFS_REPLACEMENT = """ args.push('--ro-bind', allowPath, allowPath);
|
||||
logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`);
|
||||
}
|
||||
}
|
||||
// A denyRead tmpfs must not become an unlisted writable location. Remount
|
||||
// only the parent mount read-only; explicit writable child binds remain rw.
|
||||
if (!allowedWritePaths.includes(normalizedPath)) {
|
||||
args.push('--remount-ro', normalizedPath);
|
||||
}
|
||||
}
|
||||
"""
|
||||
|
||||
|
||||
def patch_file(path: Path) -> None:
|
||||
source = path.read_text(encoding="utf-8")
|
||||
if _REPLACEMENT in source or _TMPFS_REPLACEMENT in source:
|
||||
raise RuntimeError("sandbox runtime merged-usr patch is already patched")
|
||||
if source.count(_ORIGINAL) != 1:
|
||||
raise RuntimeError("sandbox runtime merged-usr patch target does not match pinned source")
|
||||
if source.count(_TMPFS_ORIGINAL) != 1:
|
||||
raise RuntimeError("sandbox runtime read-only tmpfs patch target does not match pinned source")
|
||||
patched = source.replace(_ORIGINAL, _REPLACEMENT)
|
||||
patched = patched.replace(_TMPFS_ORIGINAL, _TMPFS_REPLACEMENT)
|
||||
path.write_text(patched, encoding="utf-8")
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("path", type=Path)
|
||||
args = parser.parse_args()
|
||||
patch_file(args.path)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+22
-1
@@ -7,6 +7,7 @@ LANGGRAPH_CONFIG="${PROJECT_DIR}/EvoScientist/langgraph_dev/langgraph.json"
|
||||
HOST="${EVOSCIENTIST_LANGGRAPH_HOST:-127.0.0.1}"
|
||||
PORT="${EVOSCIENTIST_LANGGRAPH_DEV_PORT:-3076}"
|
||||
WEB_ENV="${PROJECT_DIR}/../Ai4Sci-Web/.env"
|
||||
N_JOBS="${EVOSCIENTIST_LANGGRAPH_JOBS_PER_WORKER:-6}"
|
||||
|
||||
if [[ ! -x "${PROJECT_DIR}/.venv/bin/langgraph" ]]; then
|
||||
echo "LangGraph executable not found: ${PROJECT_DIR}/.venv/bin/langgraph" >&2
|
||||
@@ -16,6 +17,26 @@ fi
|
||||
|
||||
cd "${PROJECT_DIR}"
|
||||
|
||||
# Ask before stopping the process listening on PORT. Matching every connection
|
||||
# would also terminate Gateway while it is connected to this Runtime.
|
||||
PIDS="$(lsof -tiTCP:"${PORT}" -sTCP:LISTEN 2>/dev/null || true)"
|
||||
if [[ -n "${PIDS}" ]]; then
|
||||
echo "Port ${PORT} is already in use by PID(s): ${PIDS}" >&2
|
||||
ANSWER=""
|
||||
read -r -p "Kill the process(es) using port ${PORT}? [y/N] " ANSWER || true
|
||||
case "${ANSWER}" in
|
||||
[yY]|[yY][eE][sS])
|
||||
echo "Killing process(es) on port ${PORT}: ${PIDS}"
|
||||
kill ${PIDS}
|
||||
sleep 1
|
||||
;;
|
||||
*)
|
||||
echo "LangGraph not started; process(es) on port ${PORT} were left running." >&2
|
||||
exit 1
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
|
||||
# The Web Gateway always supplies a verified conversation workspace scope.
|
||||
export EVOSCIENTIST_DEPLOY_MODE="${EVOSCIENTIST_DEPLOY_MODE:-full}"
|
||||
export EVOSCIENTIST_WORKSPACE_DIR="${EVOSCIENTIST_WORKSPACE_DIR:-${PROJECT_DIR}/../.ai4sci/workspace}"
|
||||
@@ -27,4 +48,4 @@ exec uv run --env-file "${WEB_ENV}" langgraph dev \
|
||||
--no-browser \
|
||||
--no-reload \
|
||||
--allow-blocking \
|
||||
--n-jobs-per-worker 1
|
||||
--n-jobs-per-worker "${N_JOBS}"
|
||||
|
||||
@@ -15,6 +15,7 @@ from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from EvoScientist.llm.contracts import EvoRuntimeError
|
||||
from EvoScientist.llm.errors import (
|
||||
AgentControlError,
|
||||
ModelToolProtocolError,
|
||||
@@ -169,6 +170,16 @@ class TestNormalize:
|
||||
|
||||
assert _normalize(req, error) is None
|
||||
|
||||
def test_stable_runtime_error_passes_through(self):
|
||||
req = _request(_openai_model())
|
||||
error = EvoRuntimeError(
|
||||
"UPSTREAM_RATE_LIMITED",
|
||||
"模型服务请求频率超限,请稍后重试或切换模型。",
|
||||
details=({"http_status": 429},),
|
||||
)
|
||||
|
||||
assert _normalize(req, error) is None
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# _is_provider_error — used by tool selector to distinguish provider
|
||||
@@ -397,6 +408,23 @@ class TestMiddleware:
|
||||
assert excinfo.value.code == "MODEL_TOOL_PROTOCOL_INVALID"
|
||||
assert excinfo.value.fallbackable is True
|
||||
|
||||
def test_awrap_preserves_stable_runtime_error_identity(self):
|
||||
raised = EvoRuntimeError(
|
||||
"UPSTREAM_RATE_LIMITED",
|
||||
"模型服务请求频率超限,请稍后重试或切换模型。",
|
||||
details=({"http_status": 429},),
|
||||
)
|
||||
|
||||
async def handler(_req):
|
||||
raise raised
|
||||
|
||||
req = _request(_openai_model())
|
||||
with pytest.raises(EvoRuntimeError) as excinfo:
|
||||
self._run_awrap(ErrorNormalizationMiddleware(), req, handler)
|
||||
|
||||
assert excinfo.value is raised
|
||||
assert excinfo.value.code == "UPSTREAM_RATE_LIMITED"
|
||||
|
||||
def test_awrap_wraps_any_exception_from_recognized_model(self):
|
||||
"""Any exception raised inside a call to a provider-recognized
|
||||
model gets wrapped — including builtins like ``RuntimeError``.
|
||||
|
||||
@@ -4,6 +4,7 @@ import httpx
|
||||
import pytest
|
||||
from langchain_core.messages import HumanMessage
|
||||
|
||||
from EvoScientist.llm.contracts import EvoRuntimeError
|
||||
from EvoScientist.llm.gateway_proxy import GatewayProxyChatModel
|
||||
|
||||
|
||||
@@ -43,6 +44,18 @@ class _FakeClient:
|
||||
return _FakeStream(self._lines)
|
||||
|
||||
|
||||
def test_runtime_error_repr_preserves_only_stable_code():
|
||||
error = EvoRuntimeError(
|
||||
"UPSTREAM_RATE_LIMITED",
|
||||
"safe display message",
|
||||
details=({"provider_request": "must-not-persist"},),
|
||||
)
|
||||
|
||||
assert repr(error) == "EvoRuntimeError(code='UPSTREAM_RATE_LIMITED')"
|
||||
assert "safe display message" not in repr(error)
|
||||
assert "must-not-persist" not in repr(error)
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_astream_yields_chunks_from_sse(monkeypatch):
|
||||
model = GatewayProxyChatModel(
|
||||
@@ -52,7 +65,7 @@ async def test_astream_yields_chunks_from_sse(monkeypatch):
|
||||
)
|
||||
msg = {"type": "AIMessageChunk", "data": {"content": "hello"}}
|
||||
lines = [
|
||||
f'data: {json.dumps({"delta": {"message": msg}})}\n',
|
||||
f"data: {json.dumps({'delta': {'message': msg}})}\n",
|
||||
'data: {"delta": {"message": {"type": "AIMessageChunk", "data": {"content": " world"}}}}\n',
|
||||
"data: [DONE]\n",
|
||||
]
|
||||
@@ -88,7 +101,7 @@ async def test_astream_roundtrips_streaming_tool_call_chunks(monkeypatch):
|
||||
],
|
||||
},
|
||||
}
|
||||
lines = [f'data: {json.dumps({"delta": {"message": msg}})}\n', "data: [DONE]\n"]
|
||||
lines = [f"data: {json.dumps({'delta': {'message': msg}})}\n", "data: [DONE]\n"]
|
||||
fake = _FakeClient(lines)
|
||||
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: fake)
|
||||
|
||||
@@ -114,3 +127,24 @@ async def test_astream_raises_on_missing_done(monkeypatch):
|
||||
|
||||
with pytest.raises(RuntimeError, match="AI4SCI_MODEL_STREAM_INCOMPLETE"):
|
||||
_ = [c async for c in model._astream([HumanMessage(content="hi")])]
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_astream_projects_gateway_error_frame(monkeypatch):
|
||||
model = GatewayProxyChatModel(
|
||||
gateway_url="http://gw",
|
||||
run_id="run-1",
|
||||
envelope_signature="sig",
|
||||
)
|
||||
lines = [
|
||||
'data: {"type":"error","code":"UPSTREAM_RATE_LIMITED",'
|
||||
'"status":429,"message":"模型服务请求频率超限,请稍后重试或切换模型。"}\n'
|
||||
]
|
||||
fake = _FakeClient(lines)
|
||||
monkeypatch.setattr(httpx, "AsyncClient", lambda **kw: fake)
|
||||
|
||||
with pytest.raises(EvoRuntimeError) as exc_info:
|
||||
_ = [c async for c in model._astream([HumanMessage(content="hi")])]
|
||||
|
||||
assert exc_info.value.code == "UPSTREAM_RATE_LIMITED"
|
||||
assert exc_info.value.details == ({"http_status": 429},)
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import sys
|
||||
import types
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
@@ -21,6 +23,18 @@ def _installation(tmp_path: Path) -> sandbox.NativeSandboxInstallation:
|
||||
)
|
||||
|
||||
|
||||
def test_existing_read_paths_resolve_and_deduplicate_symlink_aliases(tmp_path: Path):
|
||||
usr = tmp_path / "usr"
|
||||
usr.mkdir()
|
||||
bin_alias = tmp_path / "bin"
|
||||
bin_alias.symlink_to(usr, target_is_directory=True)
|
||||
missing = tmp_path / "missing"
|
||||
|
||||
paths = sandbox._existing_resolved_paths((str(usr), str(bin_alias), str(missing)))
|
||||
|
||||
assert paths == (str(usr.resolve()),)
|
||||
|
||||
|
||||
def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path):
|
||||
files = tmp_path / "files"
|
||||
command_tmp = tmp_path / "runtime" / "tmp" / "run"
|
||||
@@ -38,6 +52,7 @@ def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path
|
||||
"/dev/null",
|
||||
]
|
||||
assert policy["filesystem"]["denyWrite"] == [
|
||||
str(files / "uploads"),
|
||||
"/tmp/claude",
|
||||
"/private/tmp/claude",
|
||||
"/dev/tty",
|
||||
@@ -50,6 +65,48 @@ def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path
|
||||
assert "control" not in json.dumps(policy)
|
||||
|
||||
|
||||
def test_weaker_nested_mode_requires_explicit_environment_opt_in(tmp_path: Path, monkeypatch):
|
||||
monkeypatch.delenv("EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", raising=False)
|
||||
files = tmp_path / "files"
|
||||
command_tmp = tmp_path / "runtime" / "tmp" / "run"
|
||||
files.mkdir()
|
||||
command_tmp.mkdir(parents=True)
|
||||
installation = _installation(tmp_path)
|
||||
|
||||
assert sandbox._sandbox_settings(installation, files, command_tmp)[
|
||||
"enableWeakerNestedSandbox"
|
||||
] is False
|
||||
|
||||
monkeypatch.setenv("EVOSCIENTIST_NATIVE_SANDBOX_WEAKER_NESTED", "true")
|
||||
assert sandbox._sandbox_settings(installation, files, command_tmp)[
|
||||
"enableWeakerNestedSandbox"
|
||||
] is True
|
||||
|
||||
|
||||
def test_network_preflight_probe_accepts_kernel_denied_unix_socket(monkeypatch):
|
||||
monkeypatch.delenv("OPENAI_API_KEY", raising=False)
|
||||
|
||||
class FakeSocket:
|
||||
def __init__(self, family=None, *_args):
|
||||
if family == 1:
|
||||
raise PermissionError("blocked by seccomp")
|
||||
|
||||
def connect_ex(self, _address):
|
||||
return 1
|
||||
|
||||
def close(self):
|
||||
return None
|
||||
|
||||
fake_socket = types.SimpleNamespace(
|
||||
AF_UNIX=1,
|
||||
socket=lambda family=None, *args: FakeSocket(family, *args),
|
||||
)
|
||||
monkeypatch.setitem(sys.modules, "socket", fake_socket)
|
||||
|
||||
namespace: dict[str, object] = {}
|
||||
exec(sandbox._network_preflight_probe(1234, Path("/blocked.sock")), namespace)
|
||||
|
||||
|
||||
def test_clean_environment_does_not_inherit_secrets(tmp_path: Path, monkeypatch):
|
||||
command_tmp = tmp_path / "tmp"
|
||||
(command_tmp / "home").mkdir(parents=True)
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import runpy
|
||||
from collections.abc import Callable
|
||||
from pathlib import Path
|
||||
from typing import cast
|
||||
|
||||
import pytest
|
||||
|
||||
PATCH_SCRIPT = Path(__file__).parents[1] / "runtime" / "native-sandbox" / "patch_merged_usr.py"
|
||||
|
||||
|
||||
def test_patch_skips_only_symlink_aliases_covered_by_read_allow(tmp_path: Path):
|
||||
namespace = runpy.run_path(str(PATCH_SCRIPT))
|
||||
patch_file = cast(Callable[[Path], None], namespace["patch_file"])
|
||||
|
||||
source = tmp_path / "linux-sandbox-utils.js"
|
||||
source.write_text(
|
||||
"""function pushReadDenyDirMounts(args, normalizedPath, allowedWritePaths, readAllowPaths) {
|
||||
const denySep = normalizedPath === '/' ? '/' : normalizedPath + '/';
|
||||
args.push('--tmpfs', normalizedPath);
|
||||
for (const writePath of allowedWritePaths) {
|
||||
if (writePath.startsWith(denySep) || writePath === normalizedPath) {
|
||||
args.push('--bind', writePath, writePath);
|
||||
}
|
||||
}
|
||||
for (const allowPath of readAllowPaths) {
|
||||
if (allowPath.startsWith(denySep) || allowPath === normalizedPath) {
|
||||
if (!fs.existsSync(allowPath)) {
|
||||
continue;
|
||||
}
|
||||
if (allowedWritePaths.some(w => (w.startsWith(denySep) || w === normalizedPath) &&
|
||||
(allowPath === w || allowPath.startsWith(w + '/')))) {
|
||||
continue;
|
||||
}
|
||||
args.push('--ro-bind', allowPath, allowPath);
|
||||
logForDebugging(`[Sandbox Linux] Re-allowed read access within denied region: ${allowPath}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
const rootSkip = new Set(['proc', 'dev', 'sys']);
|
||||
for (const p of readConfig?.denyOnly || []) {
|
||||
if (normalizePathForSandbox(p) === '/') {
|
||||
for (const child of fs.readdirSync('/')) {
|
||||
if (!rootSkip.has(child))
|
||||
readDenyPaths.push('/' + child);
|
||||
}
|
||||
}
|
||||
""",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
patch_file(source)
|
||||
patched = source.read_text(encoding="utf-8")
|
||||
|
||||
assert "isSymbolicLink()" in patched
|
||||
assert "readAllowPaths.some" in patched
|
||||
assert "resolved.startsWith(allowPath + '/')" in patched
|
||||
assert "args.push('--remount-ro', normalizedPath)" in patched
|
||||
assert "!allowedWritePaths.includes(normalizedPath)" in patched
|
||||
with pytest.raises(RuntimeError, match="already patched"):
|
||||
patch_file(source)
|
||||
@@ -38,6 +38,23 @@ class TestGetSystemPrompt:
|
||||
result = get_system_prompt()
|
||||
assert "Shell Execution Guidelines" in result
|
||||
|
||||
def test_contains_safe_archive_and_sqlite_programming_contracts(self):
|
||||
result = get_system_prompt(native_web_sandbox=True)
|
||||
|
||||
assert "never use `extractall`" in result
|
||||
assert "mode=ro&immutable=1" in result
|
||||
assert "PRAGMA query_only=ON" in result
|
||||
assert "Never execute files extracted from an archive" in result
|
||||
assert "Do not modify original files under `uploads/`" in result
|
||||
|
||||
def test_distinguishes_live_web_search_from_local_memory_search(self):
|
||||
result = get_system_prompt(native_web_sandbox=True)
|
||||
|
||||
assert "tavily_search" in result
|
||||
assert "search_observations" in result
|
||||
assert "local memory" in result.lower()
|
||||
assert "Do not use `execute`, `curl`, or `httpx`" in result
|
||||
|
||||
def test_contains_delegation(self):
|
||||
result = get_system_prompt()
|
||||
assert "Sub-Agent Delegation" in result
|
||||
|
||||
@@ -1,5 +1,9 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from EvoScientist.llm.contracts import EvoRuntimeError
|
||||
from EvoScientist.middleware import recoverable_tools
|
||||
|
||||
|
||||
@@ -51,3 +55,36 @@ def test_evomemory_never_falls_back_to_parent_model_proxy(monkeypatch):
|
||||
|
||||
assert proxy is None
|
||||
assert metadata["run_kind"] == "evomemory_turn_worker"
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_tool_effect_gateway_error_preserves_machine_code(monkeypatch):
|
||||
class Client:
|
||||
async def __aenter__(self):
|
||||
return self
|
||||
|
||||
async def __aexit__(self, *_):
|
||||
return None
|
||||
|
||||
async def post(self, url, json):
|
||||
return httpx.Response(
|
||||
409,
|
||||
json={"detail": {"code": "RUN_FENCE_LOST"}},
|
||||
request=httpx.Request("POST", url),
|
||||
)
|
||||
|
||||
monkeypatch.setattr(recoverable_tools.httpx, "AsyncClient", lambda **_: Client())
|
||||
|
||||
with pytest.raises(EvoRuntimeError) as exc_info:
|
||||
await recoverable_tools._post(
|
||||
{
|
||||
"gateway_url": "http://gateway",
|
||||
"run_id": "run-1",
|
||||
"envelope_signature": "signature",
|
||||
},
|
||||
"prepare",
|
||||
{},
|
||||
)
|
||||
|
||||
assert exc_info.value.code == "RUN_FENCE_LOST"
|
||||
assert repr(exc_info.value) == "EvoRuntimeError(code='RUN_FENCE_LOST')"
|
||||
|
||||
@@ -22,13 +22,19 @@ from EvoScientist.sessions import (
|
||||
delete_thread,
|
||||
find_similar_threads,
|
||||
generate_thread_id,
|
||||
get_aggregated_storage_stats,
|
||||
get_db_path,
|
||||
get_most_recent,
|
||||
get_thread_messages,
|
||||
get_thread_metadata,
|
||||
list_all_session_db_paths,
|
||||
list_all_thread_ids,
|
||||
list_threads,
|
||||
prune_all_stale_threads,
|
||||
prune_thread_history,
|
||||
resolve_thread_id_prefix,
|
||||
thread_exists,
|
||||
vacuum_db,
|
||||
)
|
||||
|
||||
|
||||
@@ -2919,5 +2925,172 @@ class TestRestoreWebuiThreadsToGlobalStore(unittest.IsolatedAsyncioTestCase):
|
||||
assert restore_called, "_restore_webui_threads_to_global_store must be called"
|
||||
|
||||
|
||||
def _uuid6_from_unix(ts_unix: float) -> str:
|
||||
"""Build a UUIDv6 (time-ordered checkpoint id) from a unix timestamp.
|
||||
|
||||
Production checkpoint ids are UUIDv6, so lexicographic order matches
|
||||
insertion order and the timestamp is recoverable from the id itself.
|
||||
"""
|
||||
greg = int((ts_unix + 12219292800) * 10_000_000) & ((1 << 60) - 1)
|
||||
high48, low12 = greg >> 12, greg & 0xFFF
|
||||
rand = uuid.uuid4().int & ((1 << 62) - 1)
|
||||
value = (high48 << 80) | (0x6 << 76) | (low12 << 64) | (0b10 << 62) | rand
|
||||
return str(uuid.UUID(int=value))
|
||||
|
||||
|
||||
class TestPruneFunctions(unittest.IsolatedAsyncioTestCase):
|
||||
"""Tests for the prune/vacuum API used by the gateway timer and admin routes."""
|
||||
|
||||
async def asyncSetUp(self):
|
||||
import time
|
||||
|
||||
import aiosqlite
|
||||
|
||||
self._tmpdir = tempfile.mkdtemp()
|
||||
self.db_path = os.path.join(self._tmpdir, "prune_test.db")
|
||||
now = time.time()
|
||||
|
||||
async with aiosqlite.connect(self.db_path) as conn:
|
||||
await conn.execute("""
|
||||
CREATE TABLE checkpoints (
|
||||
thread_id TEXT NOT NULL,
|
||||
checkpoint_ns TEXT NOT NULL DEFAULT '',
|
||||
checkpoint_id TEXT NOT NULL,
|
||||
parent_checkpoint_id TEXT,
|
||||
type TEXT,
|
||||
checkpoint BLOB,
|
||||
metadata TEXT NOT NULL DEFAULT '{}',
|
||||
PRIMARY KEY (thread_id, checkpoint_ns, checkpoint_id)
|
||||
)
|
||||
""")
|
||||
await conn.execute("""
|
||||
CREATE TABLE writes (
|
||||
thread_id TEXT NOT NULL,
|
||||
checkpoint_ns TEXT NOT NULL DEFAULT '',
|
||||
checkpoint_id TEXT NOT NULL,
|
||||
task_id TEXT NOT NULL,
|
||||
idx INTEGER NOT NULL,
|
||||
channel TEXT NOT NULL,
|
||||
type TEXT,
|
||||
value BLOB,
|
||||
PRIMARY KEY (thread_id, checkpoint_ns, checkpoint_id, task_id, idx)
|
||||
)
|
||||
""")
|
||||
await self._insert_thread(conn, "old_thread", 5, now - 10 * 86400)
|
||||
await self._insert_thread(conn, "new_thread", 3, now - 60)
|
||||
await self._insert_thread(
|
||||
conn, "other", 4, now - 10 * 86400, agent="OtherAgent"
|
||||
)
|
||||
await conn.commit()
|
||||
|
||||
async def asyncTearDown(self):
|
||||
try:
|
||||
os.unlink(self.db_path)
|
||||
os.rmdir(self._tmpdir)
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
async def _insert_thread(self, conn, tid, count, ts_base, agent=AGENT_NAME):
|
||||
serde = JsonPlusSerializer()
|
||||
ctype, cblob = serde.dumps_typed(
|
||||
{"channel_values": {"messages": [HumanMessage(content=f"seed-{tid}")]}}
|
||||
)
|
||||
prev = None
|
||||
for i in range(count):
|
||||
cid = _uuid6_from_unix(ts_base + i)
|
||||
await conn.execute(
|
||||
"INSERT INTO checkpoints (thread_id, checkpoint_ns, checkpoint_id,"
|
||||
" parent_checkpoint_id, type, checkpoint, metadata)"
|
||||
" VALUES (?, '', ?, ?, ?, ?, ?)",
|
||||
(tid, cid, prev, ctype, cblob, json.dumps({"agent_name": agent})),
|
||||
)
|
||||
await conn.execute(
|
||||
"INSERT INTO writes (thread_id, checkpoint_ns, checkpoint_id,"
|
||||
" task_id, idx, channel, type, value)"
|
||||
" VALUES (?, '', ?, 'task', 0, 'ch', 'str', ?)",
|
||||
(tid, cid, b"x"),
|
||||
)
|
||||
prev = cid
|
||||
|
||||
async def _count(self, tid, table="checkpoints"):
|
||||
import aiosqlite
|
||||
|
||||
async with aiosqlite.connect(self.db_path) as conn:
|
||||
async with conn.execute(
|
||||
f"SELECT COUNT(*) FROM {table} WHERE thread_id = ?", (tid,)
|
||||
) as cur:
|
||||
return (await cur.fetchone())[0]
|
||||
|
||||
async def test_prune_thread_history(self):
|
||||
result = await prune_thread_history(
|
||||
"old_thread", keep_last=2, db_path=self.db_path
|
||||
)
|
||||
# keep_last=2 anchors + 1 snapshot-seed ancestor preserved
|
||||
assert result == {"deleted_checkpoints": 2, "deleted_writes": 2}
|
||||
assert await self._count("old_thread") == 3
|
||||
assert await self._count("old_thread", "writes") == 3
|
||||
|
||||
async def test_prune_thread_history_other_agent_untouched(self):
|
||||
result = await prune_thread_history("other", keep_last=1, db_path=self.db_path)
|
||||
assert result == {"deleted_checkpoints": 0, "deleted_writes": 0}
|
||||
assert await self._count("other") == 4
|
||||
|
||||
async def test_prune_all_stale_threads(self):
|
||||
result = await prune_all_stale_threads(
|
||||
max_age_hours=24, keep_last=2, db_path=self.db_path
|
||||
)
|
||||
assert result["databases_processed"] == 1
|
||||
assert result["threads_pruned"] == 1
|
||||
assert result["total_deleted_checkpoints"] == 2
|
||||
assert result["total_deleted_writes"] == 2
|
||||
# fresh thread and foreign-agent thread untouched
|
||||
assert await self._count("new_thread") == 3
|
||||
assert await self._count("other") == 4
|
||||
|
||||
async def test_prune_all_stale_threads_none_stale(self):
|
||||
result = await prune_all_stale_threads(
|
||||
max_age_hours=24 * 365, keep_last=2, db_path=self.db_path
|
||||
)
|
||||
assert result["threads_pruned"] == 0
|
||||
assert result["total_deleted_checkpoints"] == 0
|
||||
assert await self._count("old_thread") == 5
|
||||
|
||||
async def test_list_all_thread_ids(self):
|
||||
ids = await list_all_thread_ids(db_path=self.db_path)
|
||||
assert sorted(ids) == ["new_thread", "old_thread"]
|
||||
|
||||
async def test_vacuum_db(self):
|
||||
result = await vacuum_db(db_path=self.db_path)
|
||||
assert result["size_after_bytes"] > 0
|
||||
assert result["size_before_bytes"] >= result["size_after_bytes"]
|
||||
|
||||
async def test_get_aggregated_storage_stats(self):
|
||||
with patch(
|
||||
"EvoScientist.sessions.get_db_path",
|
||||
return_value=_mock_path(self.db_path),
|
||||
):
|
||||
stats = await get_aggregated_storage_stats()
|
||||
assert stats["thread_count"] == 2
|
||||
assert stats["checkpoint_count"] == 8
|
||||
assert stats["thread_depth"]["max"] == 5
|
||||
assert stats["thread_depth"]["min"] == 3
|
||||
|
||||
async def test_list_all_session_db_paths(self):
|
||||
with patch(
|
||||
"EvoScientist.sessions.get_db_path",
|
||||
return_value=_mock_path(self.db_path),
|
||||
):
|
||||
paths = list_all_session_db_paths()
|
||||
assert len(paths) == 1
|
||||
assert str(paths[0]) == self.db_path
|
||||
|
||||
missing = os.path.join(self._tmpdir, "nope.db")
|
||||
with patch(
|
||||
"EvoScientist.sessions.get_db_path",
|
||||
return_value=_mock_path(missing),
|
||||
):
|
||||
assert list_all_session_db_paths() == []
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
@@ -0,0 +1,174 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from unittest.mock import MagicMock
|
||||
|
||||
import pytest
|
||||
from langchain_core.messages import ToolMessage
|
||||
|
||||
from EvoScientist.middleware.subagent_timeout import SubagentTimeoutMiddleware
|
||||
|
||||
|
||||
def _request(name: str = "task"):
|
||||
request = MagicMock()
|
||||
request.tool_call = {"id": "call-1", "name": name, "args": {}}
|
||||
return request
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_subagent_timeout_returns_stable_tool_error():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
|
||||
|
||||
async def handler(_request):
|
||||
await asyncio.sleep(1)
|
||||
return ToolMessage(content="late", tool_call_id="call-1", name="task")
|
||||
|
||||
result = await middleware.awrap_tool_call(_request(), handler)
|
||||
|
||||
assert isinstance(result, ToolMessage)
|
||||
assert result.status == "error"
|
||||
assert result.name == "task"
|
||||
assert result.additional_kwargs["error_code"] == "SUBAGENT_TIMEOUT"
|
||||
assert "SUBAGENT_TIMEOUT" in result.content
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_subagent_timeout_passes_success_through():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=1)
|
||||
expected = ToolMessage(content="done", tool_call_id="call-1", name="task")
|
||||
|
||||
async def handler(_request):
|
||||
return expected
|
||||
|
||||
assert await middleware.awrap_tool_call(_request(), handler) is expected
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_subagent_timeout_does_not_bound_other_tools():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
|
||||
expected = ToolMessage(content="done", tool_call_id="call-1", name="read_file")
|
||||
|
||||
async def handler(_request):
|
||||
await asyncio.sleep(0.02)
|
||||
return expected
|
||||
|
||||
assert await middleware.awrap_tool_call(_request("read_file"), handler) is expected
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_subagent_internal_timeout_error_is_not_reclassified():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=1)
|
||||
|
||||
async def handler(_request):
|
||||
raise TimeoutError("provider timed out immediately")
|
||||
|
||||
with pytest.raises(TimeoutError, match="provider timed out immediately"):
|
||||
await middleware.awrap_tool_call(_request(), handler)
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_parent_cancellation_cancels_subagent_handler():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=10)
|
||||
handler_cancelled = asyncio.Event()
|
||||
|
||||
async def handler(_request) -> ToolMessage:
|
||||
try:
|
||||
await asyncio.Event().wait()
|
||||
finally:
|
||||
handler_cancelled.set()
|
||||
return ToolMessage(content="done", tool_call_id="call-1", name="task")
|
||||
|
||||
invocation = asyncio.create_task(
|
||||
middleware.awrap_tool_call(_request(), handler)
|
||||
)
|
||||
await asyncio.sleep(0)
|
||||
invocation.cancel()
|
||||
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await invocation
|
||||
assert handler_cancelled.is_set()
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_parent_cancellation_wins_over_handler_cleanup_error():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=10)
|
||||
|
||||
async def handler(_request) -> ToolMessage:
|
||||
try:
|
||||
await asyncio.Event().wait()
|
||||
except asyncio.CancelledError as exc:
|
||||
raise RuntimeError("cleanup failed") from exc
|
||||
return ToolMessage(content="done", tool_call_id="call-1", name="task")
|
||||
|
||||
invocation = asyncio.create_task(
|
||||
middleware.awrap_tool_call(_request(), handler)
|
||||
)
|
||||
await asyncio.sleep(0)
|
||||
invocation.cancel()
|
||||
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await invocation
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_deadline_wins_over_handler_cleanup_error():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
|
||||
|
||||
async def handler(_request) -> ToolMessage:
|
||||
try:
|
||||
await asyncio.Event().wait()
|
||||
except asyncio.CancelledError as exc:
|
||||
raise RuntimeError("cleanup failed") from exc
|
||||
return ToolMessage(content="done", tool_call_id="call-1", name="task")
|
||||
|
||||
result = await middleware.awrap_tool_call(_request(), handler)
|
||||
|
||||
assert isinstance(result, ToolMessage)
|
||||
assert result.additional_kwargs["error_code"] == "SUBAGENT_TIMEOUT"
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_parent_cancellation_during_deadline_cleanup_is_not_swallowed():
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
|
||||
cleanup_started = asyncio.Event()
|
||||
release_cleanup = asyncio.Event()
|
||||
|
||||
async def handler(_request) -> ToolMessage:
|
||||
try:
|
||||
await asyncio.Event().wait()
|
||||
except asyncio.CancelledError:
|
||||
cleanup_started.set()
|
||||
await release_cleanup.wait()
|
||||
raise
|
||||
return ToolMessage(content="done", tool_call_id="call-1", name="task")
|
||||
|
||||
invocation = asyncio.create_task(
|
||||
middleware.awrap_tool_call(_request(), handler)
|
||||
)
|
||||
await cleanup_started.wait()
|
||||
invocation.cancel()
|
||||
release_cleanup.set()
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await invocation
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_parent_cancellation_after_cleanup_before_timeout_return_wins(monkeypatch):
|
||||
middleware = SubagentTimeoutMiddleware(timeout_seconds=0.01)
|
||||
|
||||
async def handler(_request) -> ToolMessage:
|
||||
await asyncio.Event().wait()
|
||||
return ToolMessage(content="done", tool_call_id="call-1", name="task")
|
||||
|
||||
async def finish_cleanup_then_cancel_parent(_task):
|
||||
current = asyncio.current_task()
|
||||
assert current is not None
|
||||
current.cancel()
|
||||
|
||||
monkeypatch.setattr(middleware, "_cancel_task", finish_cleanup_then_cancel_parent)
|
||||
|
||||
invocation = asyncio.create_task(
|
||||
middleware.awrap_tool_call(_request(), handler)
|
||||
)
|
||||
with pytest.raises(asyncio.CancelledError):
|
||||
await invocation
|
||||
@@ -0,0 +1,132 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import pytest
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_tavily_search_keeps_indexed_summary_when_source_fetch_fails(monkeypatch):
|
||||
from EvoScientist.tools import search
|
||||
|
||||
class _Client:
|
||||
def search(self, *_args, **_kwargs):
|
||||
return {
|
||||
"results": [
|
||||
{
|
||||
"title": "AIR staff profile",
|
||||
"url": "https://air.cas.cn/example",
|
||||
"content": "Indexed staff-profile summary.",
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
recorded: list[tuple[str, str]] = []
|
||||
|
||||
async def fetch_failed(_url: str, timeout: float = 10.0) -> str:
|
||||
return "Error fetching content from https://air.cas.cn/example: DNS failed"
|
||||
|
||||
async def record(service: str, action: str) -> None:
|
||||
recorded.append((service, action))
|
||||
|
||||
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
||||
monkeypatch.setattr(search, "fetch_webpage_content", fetch_failed)
|
||||
monkeypatch.setattr("EvoScientist.runtime_integrations.record_service_usage", record)
|
||||
|
||||
result = await search.tavily_search.ainvoke({"query": "高铭 空天院"})
|
||||
|
||||
assert "Indexed staff-profile summary." in result
|
||||
assert "https://air.cas.cn/example" in result
|
||||
assert "Tavily-indexed summary" in result
|
||||
assert recorded == [("tavily", "search")]
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_tavily_search_bounds_fetched_page_content(monkeypatch):
|
||||
from EvoScientist.tools import search
|
||||
|
||||
class _Client:
|
||||
def search(self, *_args, **_kwargs):
|
||||
return {
|
||||
"results": [
|
||||
{
|
||||
"title": f"Result {index}",
|
||||
"url": f"https://example.com/{index}",
|
||||
"content": f"Indexed summary {index}",
|
||||
}
|
||||
for index in range(3)
|
||||
]
|
||||
}
|
||||
|
||||
async def huge_page(_url: str, timeout: float = 10.0) -> str:
|
||||
return "page-content " * 10_000
|
||||
|
||||
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
||||
monkeypatch.setattr(search, "fetch_webpage_content", huge_page)
|
||||
|
||||
result = await search.tavily_search.ainvoke({"query": "bounded search"})
|
||||
|
||||
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
|
||||
for index in range(3):
|
||||
assert f"https://example.com/{index}" in result
|
||||
assert "[page content truncated]" in result
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_tavily_search_preserves_every_result_url_under_total_budget(monkeypatch):
|
||||
from EvoScientist.tools import search
|
||||
|
||||
class _Client:
|
||||
def search(self, *_args, **_kwargs):
|
||||
return {
|
||||
"results": [
|
||||
{
|
||||
"title": f"Result {index} " + ("very-long-title " * 800),
|
||||
"url": f"https://example.com/result-{index}",
|
||||
"content": f"Indexed summary {index}",
|
||||
}
|
||||
for index in range(3)
|
||||
]
|
||||
}
|
||||
|
||||
async def page(_url: str, timeout: float = 10.0) -> str:
|
||||
return "page-content " * 1_000
|
||||
|
||||
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
||||
monkeypatch.setattr(search, "fetch_webpage_content", page)
|
||||
|
||||
result = await search.tavily_search.ainvoke({"query": "preserve urls"})
|
||||
|
||||
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
|
||||
for index in range(3):
|
||||
assert f"https://example.com/result-{index}" in result
|
||||
assert "[search result content truncated to preserve all titles and URLs]" in result
|
||||
|
||||
|
||||
@pytest.mark.anyio
|
||||
async def test_tavily_search_bounds_maliciously_long_url(monkeypatch):
|
||||
from EvoScientist.tools import search
|
||||
|
||||
long_url = "https://example.com/" + ("a" * 20_000)
|
||||
|
||||
class _Client:
|
||||
def search(self, *_args, **_kwargs):
|
||||
return {
|
||||
"results": [
|
||||
{
|
||||
"title": "Long URL result",
|
||||
"url": long_url,
|
||||
"content": "Indexed summary",
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
async def page(_url: str, timeout: float = 10.0) -> str:
|
||||
return "page"
|
||||
|
||||
monkeypatch.setattr(search, "_get_tavily_client", lambda: _Client())
|
||||
monkeypatch.setattr(search, "fetch_webpage_content", page)
|
||||
|
||||
result = await search.tavily_search.ainvoke({"query": "long url"})
|
||||
|
||||
assert len(result) <= search.MAX_SEARCH_RESULT_CHARS
|
||||
assert "https://example.com/" in result
|
||||
assert "[URL truncated]" in result
|
||||
@@ -6,6 +6,184 @@ from EvoScientist.llm.contracts import EvoRuntimeError
|
||||
from EvoScientist.web_runtime import _ToolRegistryFenceMiddleware
|
||||
|
||||
|
||||
def test_web_registry_describes_tavily_as_controlled_live_search(monkeypatch):
|
||||
monkeypatch.setenv("TAVILY_API_KEY", "test-key")
|
||||
|
||||
from EvoScientist.web_runtime import web_tool_registry_manifest
|
||||
|
||||
manifest, _revision = web_tool_registry_manifest()
|
||||
tavily = next(item for item in manifest if item["name"] == "tavily_search")
|
||||
|
||||
assert "live public web" in tavily["description"].lower()
|
||||
assert "execute" in tavily["description"].lower()
|
||||
|
||||
|
||||
def test_base_kwargs_install_tavily_on_main_agent(monkeypatch):
|
||||
import EvoScientist.EvoScientist as agent_module
|
||||
|
||||
monkeypatch.setenv("TAVILY_API_KEY", "test-key")
|
||||
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
|
||||
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
|
||||
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
|
||||
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
|
||||
monkeypatch.setattr("EvoScientist.utils.load_subagents", lambda *_args, **_kwargs: [])
|
||||
|
||||
kwargs = agent_module._build_base_kwargs(
|
||||
object(),
|
||||
[],
|
||||
cfg=object(),
|
||||
chat_model=object(),
|
||||
workspace_dir="/workspace",
|
||||
)
|
||||
|
||||
assert "tavily_search" in {getattr(tool, "name", "") for tool in kwargs["tools"]}
|
||||
|
||||
|
||||
def _stub_agent_build(monkeypatch, agent_module, subagents=None):
|
||||
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
|
||||
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
|
||||
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
|
||||
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.utils.load_subagents",
|
||||
lambda *_args, **_kwargs: list(subagents or []),
|
||||
)
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"reserved_name",
|
||||
[
|
||||
"skill_manager",
|
||||
"execute",
|
||||
"start_async_task",
|
||||
"check_async_task",
|
||||
"update_async_task",
|
||||
"cancel_async_task",
|
||||
"list_async_tasks",
|
||||
],
|
||||
)
|
||||
def test_mcp_cannot_override_reserved_tool(monkeypatch, reserved_name):
|
||||
from types import SimpleNamespace
|
||||
|
||||
import EvoScientist.EvoScientist as agent_module
|
||||
|
||||
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
|
||||
monkeypatch.setattr(
|
||||
agent_module,
|
||||
"_load_mcp_tools_cached",
|
||||
lambda **_kwargs: {"main": [SimpleNamespace(name=reserved_name)]},
|
||||
)
|
||||
_stub_agent_build(monkeypatch, agent_module)
|
||||
|
||||
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"):
|
||||
agent_module.load_mcp_and_build_kwargs(
|
||||
object(),
|
||||
[],
|
||||
cfg=object(),
|
||||
chat_model=object(),
|
||||
workspace_dir="/workspace",
|
||||
)
|
||||
|
||||
|
||||
def test_mcp_cannot_duplicate_existing_subagent_tool(monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
|
||||
import EvoScientist.EvoScientist as agent_module
|
||||
|
||||
existing = SimpleNamespace(name="shared_search")
|
||||
injected = SimpleNamespace(name="shared_search")
|
||||
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
|
||||
monkeypatch.setattr(
|
||||
agent_module,
|
||||
"_load_mcp_tools_cached",
|
||||
lambda **_kwargs: {"research-agent": [injected]},
|
||||
)
|
||||
_stub_agent_build(
|
||||
monkeypatch,
|
||||
agent_module,
|
||||
subagents=[{"name": "research-agent", "tools": [existing]}],
|
||||
)
|
||||
|
||||
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"):
|
||||
agent_module.load_mcp_and_build_kwargs(
|
||||
object(),
|
||||
[],
|
||||
cfg=object(),
|
||||
chat_model=object(),
|
||||
workspace_dir="/workspace",
|
||||
)
|
||||
|
||||
|
||||
def test_mcp_cannot_override_builtin_tavily(monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
|
||||
import EvoScientist.EvoScientist as agent_module
|
||||
|
||||
monkeypatch.setenv("TAVILY_API_KEY", "test-key")
|
||||
monkeypatch.setattr(
|
||||
agent_module,
|
||||
"_load_mcp_tools_cached",
|
||||
lambda **_kwargs: {"main": [SimpleNamespace(name="tavily_search")]},
|
||||
)
|
||||
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
|
||||
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
|
||||
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
|
||||
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
|
||||
monkeypatch.setattr("EvoScientist.utils.load_subagents", lambda *_args, **_kwargs: [])
|
||||
|
||||
with pytest.raises(EvoRuntimeError, match="TOOL_REGISTRY_CONFLICT"):
|
||||
agent_module.load_mcp_and_build_kwargs(
|
||||
object(),
|
||||
[],
|
||||
cfg=object(),
|
||||
chat_model=object(),
|
||||
workspace_dir="/workspace",
|
||||
)
|
||||
|
||||
|
||||
def test_same_mcp_tool_can_be_routed_to_multiple_agents(monkeypatch):
|
||||
from types import SimpleNamespace
|
||||
|
||||
import EvoScientist.EvoScientist as agent_module
|
||||
|
||||
shared_main = SimpleNamespace(name="shared_search")
|
||||
shared_research = SimpleNamespace(name="shared_search")
|
||||
monkeypatch.delenv("TAVILY_API_KEY", raising=False)
|
||||
monkeypatch.setattr(
|
||||
agent_module,
|
||||
"_load_mcp_tools_cached",
|
||||
lambda **_kwargs: {
|
||||
"main": [shared_main],
|
||||
"research-agent": [shared_research],
|
||||
},
|
||||
)
|
||||
monkeypatch.setattr(agent_module, "_ensure_general_purpose_subagent", lambda _subs: None)
|
||||
monkeypatch.setattr(agent_module, "_inject_subagent_middleware", lambda *_args, **_kwargs: None)
|
||||
monkeypatch.setattr(agent_module, "_maybe_swap_async_subagents", lambda subs, *_args, **_kwargs: subs)
|
||||
monkeypatch.setattr(agent_module, "_configured_system_prompt", lambda _cfg: "prompt")
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.utils.load_subagents",
|
||||
lambda *_args, **_kwargs: [
|
||||
{"name": "research-agent", "tools": []}
|
||||
],
|
||||
)
|
||||
|
||||
kwargs = agent_module.load_mcp_and_build_kwargs(
|
||||
object(),
|
||||
[],
|
||||
cfg=object(),
|
||||
chat_model=object(),
|
||||
workspace_dir="/workspace",
|
||||
)
|
||||
|
||||
assert shared_main in kwargs["tools"]
|
||||
research = next(
|
||||
subagent for subagent in kwargs["subagents"]
|
||||
if subagent["name"] == "research-agent"
|
||||
)
|
||||
assert shared_research in research["tools"]
|
||||
|
||||
|
||||
def test_tool_dispatch_fence_rejects_changed_registry(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.web_runtime.web_tool_registry_manifest",
|
||||
|
||||
+374
-16
@@ -2,10 +2,14 @@ from __future__ import annotations
|
||||
|
||||
import base64
|
||||
import os
|
||||
import sqlite3
|
||||
import zipfile
|
||||
from io import BytesIO
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
from deepagents.backends.protocol import ExecuteResponse
|
||||
from PIL import Image
|
||||
|
||||
from EvoScientist.native_sandbox import (
|
||||
NativeSandboxExecutor,
|
||||
@@ -27,6 +31,19 @@ from EvoScientist.workspace_scope import (
|
||||
)
|
||||
|
||||
|
||||
def test_extracted_documents_route_through_deepagents_as_text():
|
||||
import deepagents.middleware.filesystem as filesystem_middleware
|
||||
|
||||
from EvoScientist.llm.patches import _patch_deepagents_extracted_document_text
|
||||
|
||||
_patch_deepagents_extracted_document_text()
|
||||
|
||||
get_file_type = vars(filesystem_middleware)["_get_file_type"]
|
||||
assert get_file_type("/workspace/report.pptx") == "text"
|
||||
assert get_file_type("/workspace/report.pdf") == "text"
|
||||
assert get_file_type("/workspace/image.png") == "image"
|
||||
|
||||
|
||||
def test_normalize_workspace_path_is_strict():
|
||||
assert normalize_workspace_path("/workspace") == ()
|
||||
assert normalize_workspace_path("/workspace/reports/a.txt") == (
|
||||
@@ -82,25 +99,366 @@ def test_root_lists_workspace_namespace(tmp_path: Path):
|
||||
]
|
||||
|
||||
|
||||
def test_docx_and_unknown_binary_read_with_base64_contract(tmp_path: Path):
|
||||
def test_docx_read_extracts_text_instead_of_returning_base64(tmp_path: Path):
|
||||
backend = ScopedFilesystemBackend(tmp_path)
|
||||
docx = b"PK\x03\x04\x00word/document.xml"
|
||||
unknown = b"custom\x00binary"
|
||||
backend.upload_files(
|
||||
[
|
||||
("/workspace/input.docx", docx),
|
||||
("/workspace/payload.custom", unknown),
|
||||
]
|
||||
docx = tmp_path / "input.docx"
|
||||
with zipfile.ZipFile(docx, "w") as archive:
|
||||
archive.writestr(
|
||||
"word/document.xml",
|
||||
"""<?xml version="1.0" encoding="UTF-8"?>
|
||||
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:body><w:p><w:r><w:t>有界文档内容</w:t></w:r></w:p></w:body>
|
||||
</w:document>""",
|
||||
)
|
||||
|
||||
result = backend.read("/workspace/input.docx")
|
||||
|
||||
assert result.error is None
|
||||
assert result.file_data is not None
|
||||
assert result.file_data["encoding"] == "utf-8"
|
||||
assert "有界文档内容" in result.file_data["content"]
|
||||
assert "base64" not in result.file_data["content"]
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("filename", "kind", "expected"),
|
||||
[
|
||||
("archive.zip", "archive", "list"),
|
||||
("results.sqlite", "database", "mode=ro"),
|
||||
("program.exe", "executable", "must not be executed"),
|
||||
("payload.custom", "binary", "unsupported"),
|
||||
],
|
||||
)
|
||||
def test_non_document_binary_returns_bounded_processing_guidance(
|
||||
tmp_path: Path, filename: str, kind: str, expected: str
|
||||
):
|
||||
if filename.endswith(".zip"):
|
||||
with zipfile.ZipFile(tmp_path / filename, "w") as archive:
|
||||
archive.writestr("notes.txt", "hello")
|
||||
elif filename.endswith(".sqlite"):
|
||||
connection = sqlite3.connect(tmp_path / filename)
|
||||
connection.execute("CREATE TABLE results(id INTEGER PRIMARY KEY, value TEXT)")
|
||||
connection.commit()
|
||||
connection.close()
|
||||
elif filename.endswith(".exe"):
|
||||
(tmp_path / filename).write_bytes(b"MZ\x00binary payload")
|
||||
else:
|
||||
(tmp_path / filename).write_bytes(b"custom\x00binary payload")
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read(f"/workspace/{filename}")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "BINARY_PROCESSING_REQUIRED" in result.error or "UNSUPPORTED_BINARY_FILE" in result.error
|
||||
assert f'"kind": "{kind}"' in result.error
|
||||
assert expected.lower() in result.error.lower()
|
||||
assert len(result.error) < 4000
|
||||
|
||||
|
||||
def test_image_read_keeps_base64_media_contract(tmp_path: Path):
|
||||
buffer = BytesIO()
|
||||
Image.new("RGB", (32, 24), (1, 2, 3)).save(buffer, "PNG")
|
||||
raw = buffer.getvalue()
|
||||
(tmp_path / "image.png").write_bytes(raw)
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/image.png")
|
||||
|
||||
assert result.error is None
|
||||
assert result.file_data is not None
|
||||
assert result.file_data["encoding"] == "base64"
|
||||
|
||||
|
||||
def test_large_image_is_downsampled_before_base64_delivery(tmp_path: Path):
|
||||
Image.new("RGB", (3000, 1200), (1, 2, 3)).save(tmp_path / "large.png", "PNG")
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/large.png")
|
||||
|
||||
assert result.error is None
|
||||
assert result.file_data is not None
|
||||
decoded = base64.standard_b64decode(result.file_data["content"])
|
||||
image = Image.open(BytesIO(decoded))
|
||||
image.load()
|
||||
assert image.size == (2048, 819)
|
||||
assert image.format == "JPEG"
|
||||
|
||||
|
||||
def test_corrupt_image_returns_error_instead_of_base64(tmp_path: Path):
|
||||
(tmp_path / "broken.png").write_bytes(b"\x89PNG\r\nnot-decodable")
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.png")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "IMAGE_PROCESSING_FAILED" in result.error
|
||||
|
||||
|
||||
def test_image_pixel_budget_is_enforced_before_model_delivery(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
import EvoScientist.document_extract as document_extract
|
||||
|
||||
Image.new("RGB", (100, 100), (1, 2, 3)).save(tmp_path / "pixels.png", "PNG")
|
||||
monkeypatch.setattr(document_extract, "MAX_IMAGE_PIXELS", 9_999)
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/pixels.png")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "IMAGE_PIXEL_BUDGET_EXCEEDED" in result.error
|
||||
|
||||
|
||||
def test_multiframe_image_is_reduced_to_first_frame(tmp_path: Path):
|
||||
frames = [Image.new("RGB", (20, 10), color) for color in ((255, 0, 0), (0, 255, 0))]
|
||||
frames[0].save(
|
||||
tmp_path / "animated.gif",
|
||||
format="GIF",
|
||||
save_all=True,
|
||||
append_images=frames[1:],
|
||||
duration=100,
|
||||
loop=0,
|
||||
)
|
||||
|
||||
assert backend.read("/workspace/input.docx").file_data == {
|
||||
"content": base64.standard_b64encode(docx).decode("ascii"),
|
||||
"encoding": "base64",
|
||||
}
|
||||
assert backend.read("/workspace/payload.custom").file_data == {
|
||||
"content": base64.standard_b64encode(unknown).decode("ascii"),
|
||||
"encoding": "base64",
|
||||
}
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/animated.gif")
|
||||
|
||||
assert result.error is None
|
||||
assert result.file_data is not None
|
||||
decoded = base64.standard_b64decode(result.file_data["content"])
|
||||
image = Image.open(BytesIO(decoded))
|
||||
image.load()
|
||||
assert getattr(image, "n_frames", 1) == 1
|
||||
|
||||
|
||||
def test_pptx_read_extracts_slide_text_without_base64(tmp_path: Path):
|
||||
with zipfile.ZipFile(tmp_path / "deck.pptx", "w") as archive:
|
||||
archive.writestr(
|
||||
"ppt/slides/slide1.xml",
|
||||
"""<p:sld xmlns:p="http://schemas.openxmlformats.org/presentationml/2006/main"
|
||||
xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
|
||||
<p:cSld><a:t>总体技术架构</a:t><a:t>核心能力说明</a:t></p:cSld>
|
||||
</p:sld>""",
|
||||
)
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/deck.pptx")
|
||||
|
||||
assert result.error is None
|
||||
assert result.file_data is not None
|
||||
assert result.file_data["encoding"] == "utf-8"
|
||||
assert "## Slide 1" in result.file_data["content"]
|
||||
assert "总体技术架构" in result.file_data["content"]
|
||||
|
||||
|
||||
def test_document_output_is_bounded_and_returns_continuation_hint(tmp_path: Path):
|
||||
long_text = "\n".join(f"第{i:05d}行-" + "x" * 80 for i in range(2000))
|
||||
with zipfile.ZipFile(tmp_path / "long.docx", "w") as archive:
|
||||
paragraphs = "".join(
|
||||
f"<w:p><w:r><w:t>{line}</w:t></w:r></w:p>"
|
||||
for line in long_text.splitlines()
|
||||
)
|
||||
archive.writestr(
|
||||
"word/document.xml",
|
||||
f"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:body>{paragraphs}</w:body></w:document>""",
|
||||
)
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read(
|
||||
"/workspace/long.docx", offset=0, limit=2000
|
||||
)
|
||||
|
||||
assert result.error is None
|
||||
assert result.file_data is not None
|
||||
content = result.file_data["content"]
|
||||
assert len(content) < 51_000
|
||||
assert "DOCUMENT_OUTPUT_TRUNCATED" in content
|
||||
assert "use offset=" in content
|
||||
|
||||
|
||||
def test_corrupt_document_does_not_fall_back_to_base64(tmp_path: Path):
|
||||
(tmp_path / "broken.pptx").write_bytes(b"PK\x03\x04not-a-real-presentation")
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.pptx")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "DOCUMENT_EXTRACTION_FAILED" in result.error
|
||||
assert "base64" not in result.error.lower()
|
||||
|
||||
|
||||
def test_ooxml_member_expansion_budget_blocks_compression_bomb(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
import EvoScientist.document_extract as document_extract
|
||||
|
||||
with zipfile.ZipFile(
|
||||
tmp_path / "bomb.docx", "w", compression=zipfile.ZIP_DEFLATED
|
||||
) as archive:
|
||||
archive.writestr(
|
||||
"word/document.xml",
|
||||
"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:body><w:p><w:r><w:t>expanded content</w:t></w:r></w:p></w:body>
|
||||
</w:document>""",
|
||||
)
|
||||
monkeypatch.setattr(document_extract, "MAX_OOXML_MEMBER_BYTES", 16)
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/bomb.docx")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
|
||||
|
||||
|
||||
@pytest.mark.parametrize("member", ["../word/document.xml", "/word/document.xml"])
|
||||
def test_ooxml_rejects_unsafe_member_paths(tmp_path: Path, member: str):
|
||||
with zipfile.ZipFile(tmp_path / "unsafe.docx", "w") as archive:
|
||||
archive.writestr(member, "content")
|
||||
archive.writestr(
|
||||
"word/document.xml",
|
||||
"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:body><w:p><w:r><w:t>safe</w:t></w:r></w:p></w:body></w:document>""",
|
||||
)
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/unsafe.docx")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
|
||||
|
||||
|
||||
def test_ooxml_rejects_duplicate_member_names(tmp_path: Path):
|
||||
def write_duplicate_document(path: Path) -> None:
|
||||
with zipfile.ZipFile(path, "w") as archive:
|
||||
for text in ("first", "second"):
|
||||
archive.writestr(
|
||||
"word/document.xml",
|
||||
f"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
|
||||
<w:body><w:p><w:r><w:t>{text}</w:t></w:r></w:p></w:body></w:document>""",
|
||||
)
|
||||
|
||||
with pytest.warns(UserWarning, match="Duplicate name"):
|
||||
write_duplicate_document(tmp_path / "duplicate.docx")
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/duplicate.docx")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
|
||||
|
||||
|
||||
def test_oversized_document_is_rejected_before_opening_content(
|
||||
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
|
||||
):
|
||||
from EvoScientist.document_extract import MAX_DOCUMENT_BYTES
|
||||
|
||||
path = tmp_path / "oversized.pdf"
|
||||
path.write_bytes(b"%PDF")
|
||||
original_entry = RootedWorkspace.entry
|
||||
|
||||
def oversized_entry(self, virtual_path):
|
||||
entry = original_entry(self, virtual_path)
|
||||
return type(entry)(entry.virtual_path, entry.is_dir, MAX_DOCUMENT_BYTES + 1, entry.modified_at)
|
||||
|
||||
monkeypatch.setattr(RootedWorkspace, "entry", oversized_entry)
|
||||
|
||||
result = ScopedFilesystemBackend(tmp_path).read("/workspace/oversized.pdf")
|
||||
|
||||
assert result.file_data is None
|
||||
assert result.error is not None
|
||||
assert "DOCUMENT_TOO_LARGE" in result.error
|
||||
|
||||
|
||||
def test_external_document_converter_timeout_is_structured(
|
||||
monkeypatch: pytest.MonkeyPatch,
|
||||
):
|
||||
import subprocess
|
||||
|
||||
import EvoScientist.document_extract as document_extract
|
||||
|
||||
def timeout(*args, **kwargs):
|
||||
raise subprocess.TimeoutExpired(cmd="anydoc", timeout=60)
|
||||
|
||||
monkeypatch.setattr(document_extract.subprocess, "run", timeout)
|
||||
|
||||
with pytest.raises(
|
||||
document_extract.DocumentExtractionError,
|
||||
match="DOCUMENT_CONVERSION_TIMEOUT",
|
||||
):
|
||||
document_extract.extract_document_bytes(b"%PDF-minimal", "sample.pdf")
|
||||
|
||||
|
||||
@pytest.mark.parametrize("filename", ["report.pptx", "archive.zip", "results.sqlite"])
|
||||
def test_text_write_cannot_create_or_corrupt_binary_container(
|
||||
tmp_path: Path, filename: str
|
||||
):
|
||||
backend = ScopedFilesystemBackend(tmp_path)
|
||||
|
||||
created = backend.write(f"/workspace/{filename}", "extracted text")
|
||||
|
||||
assert created.error is not None
|
||||
assert "binary container" in created.error.lower()
|
||||
assert not (tmp_path / filename).exists()
|
||||
|
||||
|
||||
def test_text_edit_cannot_modify_existing_binary_container(tmp_path: Path):
|
||||
source = b"PK\x03\x04original-container"
|
||||
(tmp_path / "report.pptx").write_bytes(source)
|
||||
backend = ScopedFilesystemBackend(tmp_path)
|
||||
|
||||
edited = backend.edit(
|
||||
"/workspace/report.pptx", "original", "replacement"
|
||||
)
|
||||
|
||||
assert edited.error is not None
|
||||
assert "binary container" in edited.error.lower()
|
||||
assert (tmp_path / "report.pptx").read_bytes() == source
|
||||
|
||||
|
||||
def test_uploaded_text_is_read_only_to_file_tools(tmp_path: Path):
|
||||
uploads = tmp_path / "uploads"
|
||||
uploads.mkdir()
|
||||
source = uploads / "notes.txt"
|
||||
source.write_text("original", encoding="utf-8")
|
||||
backend = ScopedFilesystemBackend(tmp_path)
|
||||
|
||||
written = backend.write("/workspace/uploads/new.txt", "new")
|
||||
edited = backend.edit("/workspace/uploads/notes.txt", "original", "changed")
|
||||
|
||||
assert written.error is not None
|
||||
assert "uploads" in written.error.lower()
|
||||
assert edited.error is not None
|
||||
assert "uploads" in edited.error.lower()
|
||||
assert not (uploads / "new.txt").exists()
|
||||
assert source.read_text(encoding="utf-8") == "original"
|
||||
|
||||
|
||||
def test_utf8_sample_boundary_cut_is_not_binary(tmp_path: Path):
|
||||
# Regression (2026-08-22): the 8192-byte UTF-8 probe can split a multi-byte
|
||||
# CJK character at the sample boundary (req_v13.md cut at 8190/8191 split a
|
||||
# 3-byte char). That raised UnicodeDecodeError -> misclassified as binary
|
||||
# -> read_file returned a base64 file media block -> providers without file
|
||||
# input replaced it with a placeholder -> the model retried forever.
|
||||
backend = ScopedFilesystemBackend(tmp_path)
|
||||
# 8190 ASCII bytes + one 3-byte CJK char, so the sample cuts mid-character.
|
||||
content = ("a" * 8190 + "\u6e56" + "more text").encode("utf-8")
|
||||
assert len(content) > 8192
|
||||
assert content[8190:8193] == "\u6e56".encode("utf-8")
|
||||
backend.upload_files([("/workspace/cjk.md", content)])
|
||||
|
||||
result = backend.read("/workspace/cjk.md")
|
||||
assert result.file_data is not None
|
||||
assert result.file_data["encoding"] == "utf-8"
|
||||
assert result.file_data["content"].startswith("a" * 10)
|
||||
|
||||
|
||||
def test_mid_sample_invalid_bytes_still_binary():
|
||||
from EvoScientist.workspace_files import _is_binary_file
|
||||
|
||||
# Invalid bytes well inside the sample are genuine garbage, not a cut.
|
||||
assert _is_binary_file("/workspace/bad.raw", b"ok\xffi\xffd\xefmore")
|
||||
# A boundary cut (error in the last 4 bytes that decodes clean when the
|
||||
# dangling suffix is dropped) is text.
|
||||
cut = ("a" * 8190 + "\u6e56").encode("utf-8")[:8192]
|
||||
assert not _is_binary_file("/workspace/cut.md", cut)
|
||||
# Same shape but the prefix itself is invalid -> stays binary.
|
||||
assert _is_binary_file("/workspace/bad.md", b"\xff" * 8192)
|
||||
|
||||
|
||||
def test_symlink_targets_and_parents_are_rejected(tmp_path: Path):
|
||||
|
||||
@@ -944,11 +944,11 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "evoscientist"
|
||||
version = "0.2.2"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "deepagents", extra = ["quickjs"] },
|
||||
{ name = "filelock" },
|
||||
{ name = "firecrawl-anydoc" },
|
||||
{ name = "httpx" },
|
||||
{ name = "langchain" },
|
||||
{ name = "langchain-anthropic" },
|
||||
@@ -963,6 +963,7 @@ dependencies = [
|
||||
{ name = "lazy-loader" },
|
||||
{ name = "markdownify" },
|
||||
{ name = "nest-asyncio" },
|
||||
{ name = "pillow" },
|
||||
{ name = "prompt-toolkit" },
|
||||
{ name = "psutil" },
|
||||
{ name = "python-dotenv" },
|
||||
@@ -1056,6 +1057,7 @@ requires-dist = [
|
||||
{ name = "discord-py", marker = "extra == 'discord'", specifier = ">=2.3" },
|
||||
{ name = "faster-whisper", marker = "extra == 'stt'", specifier = ">=1.0" },
|
||||
{ name = "filelock", specifier = ">=3.16" },
|
||||
{ name = "firecrawl-anydoc", specifier = ">=0.1.6,<0.2" },
|
||||
{ name = "httpx", specifier = ">=0.28" },
|
||||
{ name = "langchain", specifier = ">=1.3" },
|
||||
{ name = "langchain-anthropic", specifier = ">=1.4" },
|
||||
@@ -1072,6 +1074,7 @@ requires-dist = [
|
||||
{ name = "lazy-loader", specifier = ">=0.5" },
|
||||
{ name = "markdownify", specifier = ">=1.2" },
|
||||
{ name = "nest-asyncio", specifier = ">=1.6" },
|
||||
{ name = "pillow", specifier = ">=10.0" },
|
||||
{ name = "pre-commit", marker = "extra == 'dev'", specifier = ">=3.5.0" },
|
||||
{ name = "prompt-toolkit", specifier = ">=3.0" },
|
||||
{ name = "psutil", specifier = ">=6.0" },
|
||||
@@ -1317,6 +1320,21 @@ wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/18/79/1b8fa1bb3568781e84c9200f951c735f3f157429f44be0495da55894d620/filetype-1.2.0-py2.py3-none-any.whl", hash = "sha256:7ce71b6880181241cf7ac8697a2f1eb6a8bd9b429f7ad6d27b8db9ba5f1c2d25", size = 19970, upload-time = "2022-11-02T17:34:01.425Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "firecrawl-anydoc"
|
||||
version = "0.1.9"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/aa/16/e37d4f284482a4f30e0502ce439a9fd8ccf013e6ae4438b4dfa209e48750/firecrawl_anydoc-0.1.9.tar.gz", hash = "sha256:0dfc64b82b4e971143dd6f0e4f39f8152fcc38be6523663682a241d8f3301b14", size = 197914, upload-time = "2026-08-13T21:48:27.989Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/42/37/866bf6270ad8552eaf6edf3b8e1a360acf3e8764140641361ae7b440514e/firecrawl_anydoc-0.1.9-cp310-abi3-macosx_10_12_x86_64.whl", hash = "sha256:ebd956831822d3ea1831253406736e0e4d9b217e3471eb68c9bf2eba9302ce73", size = 3469321, upload-time = "2026-08-13T21:48:16.775Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/b1/9a/8fbe0726cdf69154e1eb05c119332b1fdfd8200eeafcfd1675741ac5f03d/firecrawl_anydoc-0.1.9-cp310-abi3-macosx_11_0_arm64.whl", hash = "sha256:a2372d97826e8e7e68fd28cd415d08815f6aa66dcc09c6b423ac33dd2dfa91cf", size = 3287520, upload-time = "2026-08-13T21:48:18.37Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/b0/ce/86a626224c365d8de65699d6ac8f4626ac05e7e75c3ae76c82f4904a07d2/firecrawl_anydoc-0.1.9-cp310-abi3-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:ba1c19bdaa97f4850daf3fb4ca2855cefa146b5621ce7b22a71d158d3b84e54e", size = 3331094, upload-time = "2026-08-13T21:48:19.961Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/0e/28/00d1d48fe205ab66eda4a2a22cce44aadf586ef1495e703349b791c8de32/firecrawl_anydoc-0.1.9-cp310-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl", hash = "sha256:deea7a270fccda9be29c73235e938ca32e0f703e1ae120f08bf0fdb39301c6b4", size = 3555945, upload-time = "2026-08-13T21:48:21.822Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/c8/25/ac52d98c5d082ef92396b2fcb398745a3b054816c67a12d19de86c1ba535/firecrawl_anydoc-0.1.9-cp310-abi3-musllinux_1_2_aarch64.whl", hash = "sha256:de0858decd18188b0b88545689544fb3703ec08ade37f63c2857e4414e99bdf1", size = 3510861, upload-time = "2026-08-13T21:48:23.369Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/ca/d3/e80314b5746c4ca85150f3ab2a527b2167012c16e706e148733211cb88b3/firecrawl_anydoc-0.1.9-cp310-abi3-musllinux_1_2_x86_64.whl", hash = "sha256:e4e1765ed5574fa502931f8bfaec1f87af913afa5bb432aaaf5f9c094efc7f43", size = 3797896, upload-time = "2026-08-13T21:48:25.061Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/8e/d0/e7b35c2365498d5e9f1bdb4b3ded78571b6dafbaba1880538490d27b71c9/firecrawl_anydoc-0.1.9-cp310-abi3-win_amd64.whl", hash = "sha256:aa6a5ca2e10939a87c9bd21c918ef8b146ad0061fedae661d4483ef5a5bbdf60", size = 3642084, upload-time = "2026-08-13T21:48:26.702Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "flatbuffers"
|
||||
version = "25.12.19"
|
||||
@@ -3064,6 +3082,91 @@ wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/f1/d9/7fb5aa316bc299258e68c73ba3bddbc499654a07f151cba08f6153988714/pathspec-1.1.1-py3-none-any.whl", hash = "sha256:a00ce642f577bf7f473932318056212bc4f8bfdf53128c78bbd5af0b9b20b189", size = 57328, upload-time = "2026-04-27T01:46:07.06Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "pillow"
|
||||
version = "12.3.0"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/1c/3d/bb7fca845737cf9d7dbde16ed1843984665ff2e0a518f5db43e77ec540b9/pillow-12.3.0.tar.gz", hash = "sha256:3b8182a766685eaa002637e28b4ec8d6b18819a0c71f579bf0dbaa5830297cce", size = 47025035, upload-time = "2026-07-01T11:56:38.965Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/fb/c8/0a78b0e02d7ac54bc03e5321c9220da52f0c2ea83b21f7c40e7f3169c502/pillow-12.3.0-cp311-cp311-macosx_10_10_x86_64.whl", hash = "sha256:00808c5e14ef63ac5161091d242999076604ff74b883423a11e5d7bbb38bf756", size = 5392415, upload-time = "2026-07-01T11:53:47.162Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/b2/5b/a02d30018abd97ced9f5a6c63d28597694a00d066516b9c1c6de45859fc9/pillow-12.3.0-cp311-cp311-macosx_11_0_arm64.whl", hash = "sha256:37d6d0a00072fd2948eb22bce7e1475f34569d90c87c59f7a2ec59541b77f7a6", size = 4785266, upload-time = "2026-07-01T11:53:49.079Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/c8/98/766667a4be768150a202836acd9fad19c06824ca86c4286d3cf6b274964e/pillow-12.3.0-cp311-cp311-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:bcb46e2f9feff8d06323983bd83ed00c201fdcab3d74973e7072a889b3979fcd", size = 6263814, upload-time = "2026-07-01T11:53:51.32Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/3b/2d/ede717bc1144f63886c21fd349bb95860b0d1a21149ff16f2bb362b612b6/pillow-12.3.0-cp311-cp311-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:23d27a3e0307ec2244cc51e7287b919aa68d097504ebe19df4e76a98a3eea5bd", size = 6934408, upload-time = "2026-07-01T11:53:53.487Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a3/48/9c58b685e69d49c31af6c8eb9012055fab7e665785165c84796e2c73ce72/pillow-12.3.0-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:4f883547d4b7f0495ebe7056b0cc2aea76094e7a4abc8e933540f3271df27d9c", size = 6337160, upload-time = "2026-07-01T11:53:55.457Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/ff/fa/dc2a5c0ba6df93f67c31d34b808b7ce440b40cdbf96f0b81cde1d1e6fa93/pillow-12.3.0-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:236ff70b9312fb68943c703aa842ca6a758abfa45ac187a5e7c1452e96ef72b5", size = 7045172, upload-time = "2026-07-01T11:53:57.736Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/86/a5/444817a4d4c4c2417df00513086ca196f388d8f9ef40c2e4ccd1ad1af54b/pillow-12.3.0-cp311-cp311-win32.whl", hash = "sha256:10e41f0fbf1eec8cfd234b8fe17a4caac7c9d0db4c204d3c173a8f9f6ef3232b", size = 6472232, upload-time = "2026-07-01T11:53:59.767Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/63/c6/4bad1b18d132a50b27e1365e1ab163616f7a5bb56d330f66f9d1d9d4f9d4/pillow-12.3.0-cp311-cp311-win_amd64.whl", hash = "sha256:8e95e1385e4998ae9694eeaa4730ba5457ff61185b3a55e2e7bea0880aef452a", size = 7233653, upload-time = "2026-07-01T11:54:02.066Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/fd/16/00f91ab7760dc842f5aad55217e80fc4a7067a0604535249bc8a2d6d9870/pillow-12.3.0-cp311-cp311-win_arm64.whl", hash = "sha256:ebaea975e03d3141d9d3a507df75c9b3ec90fa9d2ffd07567b3a978d9d790b26", size = 2568195, upload-time = "2026-07-01T11:54:04.622Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/37/bf/fb3ebff8ddcb76aac5a01389251bbbb9519922a9b520d8247c1ca864a25d/pillow-12.3.0-cp312-cp312-macosx_10_13_x86_64.whl", hash = "sha256:ba09209fbe443b4acccebe845d8a138b89a8f4fbaeedd44953490b5315d5e965", size = 5345969, upload-time = "2026-07-01T11:54:06.397Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/d8/66/9a386a92561f402389a4fc70c18838bf6d35eb5eb5c6850b4b2dc64f5048/pillow-12.3.0-cp312-cp312-macosx_11_0_arm64.whl", hash = "sha256:ffd0c5368496f41b0944be820fcb7a838aa6e623d250b01acf2643939c3f99d7", size = 4780323, upload-time = "2026-07-01T11:54:09.351Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/25/27/ac8f99618ffd3dde21db0f4d4b1d2ab00c0880595bfd17df103f7f39fd0c/pillow-12.3.0-cp312-cp312-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:d9c7f76c0673154f044e9d78c8655fb4213f6ca31a836df48b40fe5d187717b9", size = 6266838, upload-time = "2026-07-01T11:54:11.71Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/84/21/a35af28dcc61f37ed850a2d64c65c701321dfbf25085e469d5559360cbbf/pillow-12.3.0-cp312-cp312-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:78cb2c6865a35ab8ff8b75fd122f6033b92a62c82801110e48ddd6c936a45d91", size = 6940830, upload-time = "2026-07-01T11:54:13.732Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/eb/51/8b08617af3ad95e33ce6d7dd2c99ed6c8298f7fb131636303956be022e25/pillow-12.3.0-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:e491916b378fba47242221bb9ead245211b70d504f495d105d17b14a24b4907c", size = 6344383, upload-time = "2026-07-01T11:54:15.756Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/1d/72/cf78ac9780bb93c28328f408973845a309d4d145041665f734572ced1b52/pillow-12.3.0-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:0dd2064cbc55aaec028ef5fbb60fa47bb6c3e7918e07ff17935284b227a9d2df", size = 7052934, upload-time = "2026-07-01T11:54:17.721Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/20/20/25e0f4dc178a6bc0696793720055519a0de89e7661dae886992decbd2f81/pillow-12.3.0-cp312-cp312-win32.whl", hash = "sha256:dbce0b29841537a2fa4a214c2bbf14de3587c9680caa9b4e217568472490b28f", size = 6472684, upload-time = "2026-07-01T11:54:19.839Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/45/89/da2f7971a317f83d807fdd4065c0af40208e59e692cc43d315a71a0e96d1/pillow-12.3.0-cp312-cp312-win_amd64.whl", hash = "sha256:a2b55dd6b2a4c4b7d87ffa56bdb33fdc5fdb9a462173861a7bc097f17d91cb09", size = 7227137, upload-time = "2026-07-01T11:54:22.025Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/de/47/4845a0a6c0dbf1db8456bd9fc791f13c5ced7ced20606d08a0aacfd25b49/pillow-12.3.0-cp312-cp312-win_arm64.whl", hash = "sha256:331b624368d4f1d069149002f25f44bc61c8919ce8ddb3c45bdad8f6e2d89510", size = 2568267, upload-time = "2026-07-01T11:54:24.051Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/9d/ac/31fb64e1e7efb5a4b50cd3d92049ba89ac6e4d8d3bb6a74e15048ca3353e/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphoneos.whl", hash = "sha256:21900ce7ba264168cd50defae43cd75d25c833ad4ad6e73ffc5596d12e25ac89", size = 4161684, upload-time = "2026-07-01T11:54:25.934Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/87/b4/9805e23d2b4d77842b468513841fda254ee42f0289d25088340e4ff46e2d/pillow-12.3.0-cp313-cp313-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:4e8c2a84d977f50b9daed6eeaf3baef67d00d5d74d932288f02cb94518ee3ace", size = 4255487, upload-time = "2026-07-01T11:54:27.935Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/df/39/ecf519435a200c693fe053a6ee4d835b41cf963a4dfc2551c4e637cb2a71/pillow-12.3.0-cp313-cp313-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:ae26d61dfa7a47befdc7572b521024e8745f3d809bd95ca9505a7bba9ef849ec", size = 3696433, upload-time = "2026-07-01T11:54:29.813Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/42/92/2fc3ffad878ae8dd5469ec1bc8eb83b71f48e13efdf68f02709003982a32/pillow-12.3.0-cp313-cp313-macosx_10_13_x86_64.whl", hash = "sha256:7a743ff716f746fc19a9557f60dab1600d4613255f8a7aeb3cdde4db7eb15a66", size = 5345889, upload-time = "2026-07-01T11:54:31.97Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/10/76/8803c13605b763d33d156c4678fc77f8443389c0c51c8aef707bb02015f4/pillow-12.3.0-cp313-cp313-macosx_11_0_arm64.whl", hash = "sha256:d69141514cc30b774ceea5e3ed3a6635c8d8a96edf664689b890f4089111fb35", size = 4780109, upload-time = "2026-07-01T11:54:34.026Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/1f/01/e18aff37cb0b4aac47ac90f016d347a49aca667ef97f190b06ac2aabc928/pillow-12.3.0-cp313-cp313-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:f7401aebd7f581d7f83a439d87d474999317ee099218e5ad25d125290990ba65", size = 6263736, upload-time = "2026-07-01T11:54:36.131Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f7/62/de5bdd77d935331f4f802edc11e4d82950f642caad6cb2f949837b8560e2/pillow-12.3.0-cp313-cp313-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:0847a763afefb695bc912d7c131e7e0632d4edc1d8698f58ddabec8e46b8b6d3", size = 6937129, upload-time = "2026-07-01T11:54:38.216Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/70/4d/105627a13300c5e0df1d174230b32fd1273062c96f7745fd552b945d1e1d/pillow-12.3.0-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:571b9fcb07b97ef3a492028fb3d2dc0993ca23a06138b0315286566d29ef718a", size = 6339562, upload-time = "2026-07-01T11:54:40.354Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/6b/1d/f13de01a553988ab895ba1c722e06cf3144d4f57656fd5b81b6d881f1179/pillow-12.3.0-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:756c768d0c9c2955feb7a56c37ea24aea2e369f8d36a88da270b6a9f19e62b5e", size = 7049439, upload-time = "2026-07-01T11:54:42.489Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/c9/f9/066794cca041b969964f779ee5fa66a9498bbf34248ac39c5d7954e4198f/pillow-12.3.0-cp313-cp313-win32.whl", hash = "sha256:a876864214e136f0eb367788dbd7df045f4806801518e2cfe9e13229cfe06d8f", size = 6473287, upload-time = "2026-07-01T11:54:44.9Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a6/9b/7a58e61d62be561da3a356fe2384d4059a6345fc130e23ef1c36a5b81d24/pillow-12.3.0-cp313-cp313-win_amd64.whl", hash = "sha256:1cca606cd25738df4ed873d5ad46bbdb3d83b5cbca291f6b4ff13a4df6b0bbe8", size = 7239691, upload-time = "2026-07-01T11:54:47.141Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/aa/b0/c4ed4f0ef8f8fa5ee8351537db6650bb8189f7e118842978dd6589065692/pillow-12.3.0-cp313-cp313-win_arm64.whl", hash = "sha256:b629de27fda84b42cde7edef0d85f13b958b47f6e9bbcbba9b673c562a89bd8b", size = 2568185, upload-time = "2026-07-01T11:54:49.137Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/dc/01/001f65b68192f0228cc1dbbc8d2530ab5d58b61037ba0587f946fea607cd/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphoneos.whl", hash = "sha256:9cf95fe4d0f84c82d282745d9bb08ad9f926efa00be4697e767b814ce40d4330", size = 4161736, upload-time = "2026-07-01T11:54:51.156Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/1a/d2/0219746d0fd16fc8a84498e79452375be3797d3ce4044596ce565164b84f/pillow-12.3.0-cp314-cp314-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:8728f216dcdb6e6d555cf971cb34076139ad74b31fc2c14da4fafc741c5f6217", size = 4255435, upload-time = "2026-07-01T11:54:53.414Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/c8/02/8d0bc62ef0302318c46ff2a512822d2610e81c7aa46c9b3abe6cbaca5ad0/pillow-12.3.0-cp314-cp314-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:a45650e8ce7fafffd731db8550230db6b0d306d181a90b67d3e6bca2f1990930", size = 3696262, upload-time = "2026-07-01T11:54:55.739Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/85/e2/73c77d218410b14f5f2d565e8a998d5317b7b9c75368d29985139f7a46f0/pillow-12.3.0-cp314-cp314-macosx_10_15_x86_64.whl", hash = "sha256:ba54cfebe86920a559a7c4d6b9050791c20513650a1952ebe3368c7dc70306f8", size = 5350344, upload-time = "2026-07-01T11:54:57.657Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/c7/da/32c752228ae345f489e3a42499d817b6c3996da7e8a3bc7a04fc806b243b/pillow-12.3.0-cp314-cp314-macosx_11_0_arm64.whl", hash = "sha256:e158cb00350dc278f3b91551101aa7d12415a66ebf2c91d8d5ac14e56ddd3ad0", size = 4780131, upload-time = "2026-07-01T11:54:59.713Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/b1/9d/8b2c807dbef61a5197c047afe99823787eb66f63daf9fb2432f91d6f0462/pillow-12.3.0-cp314-cp314-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e9aeb04d6aef139de265b29683e119b638208f88cf73cdd1658aa07221165321", size = 6263757, upload-time = "2026-07-01T11:55:01.778Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/5c/44/c85361f65dbe00eea8576ee467c768d25129989efb76e94f205e9ca9bb46/pillow-12.3.0-cp314-cp314-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:251bf95b67017e27b13d82f5b326234ca62d70f9cf4c2b9032de2358a3b12c7b", size = 6936962, upload-time = "2026-07-01T11:55:03.93Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/18/7e/e483414b35800b86b6f08dbbc7803fb5cd52c4d6f897f47d53ea2c7e6f65/pillow-12.3.0-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:fe3cca2e4e8a592be0f269a1ca4835c25199d9f3ce815c8491048f785b0a0198", size = 6339171, upload-time = "2026-07-01T11:55:05.989Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f0/f4/68c491844841ede6bed70189546b3ee9731cf9f2cbad396faff5e1ccba45/pillow-12.3.0-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:23aceaa007d6172b02c277f0cd359c79492bbb14f7072b4ede9fbcaf20648130", size = 7048116, upload-time = "2026-07-01T11:55:08.131Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a3/34/77f3f793fed8efc7d243f21b33c5a3f0d1c97ee70346d3db855587e155ff/pillow-12.3.0-cp314-cp314-win32.whl", hash = "sha256:af8d94b0db561cf68b88a267c5c44b49e134f525d0dc2cb7ed413a66bc23559a", size = 6467209, upload-time = "2026-07-01T11:55:10.408Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f1/e0/492879f69d94f91f60fc8cd05ba03650e9520afebb2fb7aa12777d7c7f38/pillow-12.3.0-cp314-cp314-win_amd64.whl", hash = "sha256:fdafc9cce40277e0f7a0feabce0ee50dd2fa1800f3b38015e51296b5e814048d", size = 7237707, upload-time = "2026-07-01T11:55:12.745Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/c9/ac/6b11f2875f1c2ac040d84e1bbf9cf22a88038f901ca1037898b280b38365/pillow-12.3.0-cp314-cp314-win_arm64.whl", hash = "sha256:e91206ee562682b51b98ef4b26a6ef48fd84e15fd4c4bc5ec768eb641d206838", size = 2565995, upload-time = "2026-07-01T11:55:14.736Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/52/69/c2208e56af9bfc1913afb24020297a691eb1d4ef688474c8a04913f65e04/pillow-12.3.0-cp314-cp314t-macosx_10_15_x86_64.whl", hash = "sha256:164b31cd1a0490ab6efae01aa5df49da7061be0af1b30e035b6e9a1bfe34ee6e", size = 5352503, upload-time = "2026-07-01T11:55:17.076Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/07/70/e5686d753e898a45d778ff1718dba8516ead6ab6b95d85fc8c4b70650cf2/pillow-12.3.0-cp314-cp314t-macosx_11_0_arm64.whl", hash = "sha256:5afb51d599ea772b8365ae807ae557f18bccfe46ab261fd1c2a9ed700fc6eb17", size = 4782956, upload-time = "2026-07-01T11:55:19.448Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/d5/37/25c6692f06927ee973ff18c8d9ee98ad0b4d84ee67a09610c2dd1447958e/pillow-12.3.0-cp314-cp314t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:3edce1d53195db527e0191f84b71d02022de0540bf43a16ed734ed7537b07385", size = 6322855, upload-time = "2026-07-01T11:55:21.613Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/cc/91/420637fcb8f1bc11029e403b4538e6694744428d8246118e45719f944556/pillow-12.3.0-cp314-cp314t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:bf16ba1b4d0b6b7c8e534936632270cf70eb00dbe09005bc345b2677b726855c", size = 6989642, upload-time = "2026-07-01T11:55:24.006Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/10/08/b94d7811281ccf0d143a1cf768d1c49e1e54af63e7b708ab2ee3eb87face/pillow-12.3.0-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:24870b09b224f7ae3c39ed07d10e819d06f8720bc551847b1d623832b5b0e28d", size = 6391281, upload-time = "2026-07-01T11:55:26.252Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/d2/87/24233f785f55474dc02ce3e739c5528a77e3a862e9333d1dd7a25cc31f70/pillow-12.3.0-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:30f2aa603c41533cc25c05acd0da21636e84a315768feb631c937177db558931", size = 7096716, upload-time = "2026-07-01T11:55:28.318Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/23/26/fcb2f6e37175b04f53570b59937867e2b80ee1685e744023153028fc14f9/pillow-12.3.0-cp314-cp314t-win32.whl", hash = "sha256:4b0a7fe987b14c31ebda6083f74f22b561fd3739bc0ac51e019622e3d72668c7", size = 6474125, upload-time = "2026-07-01T11:55:30.956Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/90/de/3634abee5f1c9e13c56787b7d5517b0ba8d6de51700b95578cf338349c9f/pillow-12.3.0-cp314-cp314t-win_amd64.whl", hash = "sha256:962864dc93511324d51ddbb5b9f8731bf71675b93ca612a07441896f4688fb8c", size = 7242939, upload-time = "2026-07-01T11:55:34.044Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/ce/2a/fd13f8eb24de5714a6eb444a3d67e2842c6c576e159a43793adf23051351/pillow-12.3.0-cp314-cp314t-win_arm64.whl", hash = "sha256:0740a512dc522224c77d9aa5a8d70d8b7d73fb91f2c21125d8d025d3b8990e45", size = 2567506, upload-time = "2026-07-01T11:55:35.988Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/5d/dc/8fdce34ec725a33c81c6ba122b904d6b9024e50ea9ac7bede62fab54506c/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphoneos.whl", hash = "sha256:0feb2e9d6ad6c9e3c06effe9d00f3f1e618a6643273576b016f591e9315a7139", size = 4162063, upload-time = "2026-07-01T11:55:37.941Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/76/66/2044b9a63d3b84ff048228dfcb7cd9bf0df983e8470971bf7d4c57b693de/pillow-12.3.0-cp315-cp315-ios_13_0_arm64_iphonesimulator.whl", hash = "sha256:9e881fca225083806662a5c43d627d215f258ff43c890f831966c7d7ba9c7402", size = 4255549, upload-time = "2026-07-01T11:55:40.022Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/52/7e/1f67e6f4ece6b582ee4b539decbcc9f848dc245a93ed8cd7338bafef72f1/pillow-12.3.0-cp315-cp315-ios_13_0_x86_64_iphonesimulator.whl", hash = "sha256:4998562bf62a445225f22e07c896bb04b35b1b1f2eb6d760584c9c51d7a5f78c", size = 3696331, upload-time = "2026-07-01T11:55:41.98Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/12/40/d306fc2c8e4d45d7f175c77edca7063be7b86fe7fe6e68f4353bf71d808c/pillow-12.3.0-cp315-cp315-macosx_10_15_x86_64.whl", hash = "sha256:dc624f6bc473dacdf7ef7eb8678d0d08edf15cd94fad6ae5c7d6cc67a4e4902f", size = 5350370, upload-time = "2026-07-01T11:55:44.028Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/dd/44/668fb1437e8ce420f62d6106eb66e44a5971602a4d794615bdf79315d82d/pillow-12.3.0-cp315-cp315-macosx_11_0_arm64.whl", hash = "sha256:71d6097b330eea8fd15097780c8e89cb1a8ce7838669f48c5bacd6f663dd4701", size = 4780147, upload-time = "2026-07-01T11:55:46.073Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/0c/08/93fa2e70e30a2d81547e481b6ee2bb9522117221fb1e0ce4b5df70967677/pillow-12.3.0-cp315-cp315-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:28ce87c5ab450a9dd970b52e5aca5fe63ed432d18a2eaddd1979a00a1ba24ace", size = 6273659, upload-time = "2026-07-01T11:55:48.264Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/f8/6d/043e96ff814fc31a33077e4cba86082167db520c93632afdf2042febbb0c/pillow-12.3.0-cp315-cp315-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:6b02afb9b97f65fbca5f31db6a2a3ba21aa93030225f150fa3f249717e938fb4", size = 6947439, upload-time = "2026-07-01T11:55:50.503Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/af/92/ba71d2ee2ac0edf3fa33bd9d5ee9ee080da70b1766f3ca3934f9938ddac9/pillow-12.3.0-cp315-cp315-musllinux_1_2_aarch64.whl", hash = "sha256:1182d52bc2d5e5d7d0949503aa7e36d12f42205dc287e4883f407b1988820d39", size = 6353577, upload-time = "2026-07-01T11:55:52.697Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/0f/ce/e63064e2122923ff687c8ad792d0d736a7b3920a56a46982e81a7fdd25d6/pillow-12.3.0-cp315-cp315-musllinux_1_2_x86_64.whl", hash = "sha256:e795b7eb908249c4e43c7c99fac7c2c75dab0c43566e37db472a355f63693d71", size = 7060394, upload-time = "2026-07-01T11:55:55.149Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/54/76/a09cc3ccc8d773a7283d34c38bec1708f9e3cc932093cbc4c5e71ac4060b/pillow-12.3.0-cp315-cp315-win32.whl", hash = "sha256:57b3d78c95ba9059768b10e28b813002261d3f3dfc55cc48b0c988f625175827", size = 6467375, upload-time = "2026-07-01T11:55:57.769Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/3e/03/1846c49ba3b1d5550392a4bbd06d6fb4578e1cd91a803198b5c90f5f7d53/pillow-12.3.0-cp315-cp315-win_amd64.whl", hash = "sha256:fa4ecea169a355be7a3ade2c783e2ed12f0e40d2c5621cda8b3297faf7fbb9f5", size = 7237048, upload-time = "2026-07-01T11:55:59.975Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/fb/bb/89f35dcc79610423f9f195504d7def7f0d1416a711541b42867e25fe3412/pillow-12.3.0-cp315-cp315-win_arm64.whl", hash = "sha256:877c3f311ff35410f690861c4409e7ccbf0cd2f878e50628a28e5a0bb689e658", size = 2566006, upload-time = "2026-07-01T11:56:02.143Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/30/88/707027ba09942dfa2c28759b5c222d769290a41c6d20ea60ec250801941f/pillow-12.3.0-cp315-cp315t-macosx_10_15_x86_64.whl", hash = "sha256:e9871b1ffbfa9656b60aeee92ed5136a5742696006fa322b29ea3d8da0ecc9cf", size = 5352509, upload-time = "2026-07-01T11:56:04.2Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/b0/6d/00352fa25332c2569cd387851f568cc5a4b75a9adbfb37ac4fbce4c02eec/pillow-12.3.0-cp315-cp315t-macosx_11_0_arm64.whl", hash = "sha256:53aa02d20d10c3d814d536aa4e5ac9b84ca0ff5a88377963b085ad6822f93e64", size = 4783167, upload-time = "2026-07-01T11:56:06.631Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/13/4f/9e049dfa21af7c22427275720e2490267ba8138120add5c4c574deb69782/pillow-12.3.0-cp315-cp315t-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:446c34dcc4324b084a53b705127dc15717b22c5e140ae0a3c38349d4efec071e", size = 6329237, upload-time = "2026-07-01T11:56:08.868Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/36/16/cf6eeaae8d0fce8dd390a33437cf68c5d5bd73834a2bc6e2f14efda0ab45/pillow-12.3.0-cp315-cp315t-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:cf1845d02ad822a369a49f2bb9345b1614744267682e7a03527dc3bf6eea1777", size = 6997047, upload-time = "2026-07-01T11:56:11.379Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/1e/69/dbf769bdd55f48bf5733cac28edc6364ffaa072ec9ba336266e4fe66be55/pillow-12.3.0-cp315-cp315t-musllinux_1_2_aarch64.whl", hash = "sha256:186941b6aef820ad110fb01fb06eb925374dc3a21b17e37ec9a53b250c6fe2d1", size = 6400440, upload-time = "2026-07-01T11:56:13.908Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/a0/e1/ffc9cfc2eea0d178da8018e18e959301ad9d6bc9f3edb7181e748a474b97/pillow-12.3.0-cp315-cp315t-musllinux_1_2_x86_64.whl", hash = "sha256:f13c32a3abd6079a66d9526e18dad9b6d280384d49d7c54040cd57b6424041d9", size = 7105895, upload-time = "2026-07-01T11:56:16.575Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/18/f0/a5595c1e8c3ae44b9828cb2f0fa8155e5095ef04d6327b8f61cf44a3df85/pillow-12.3.0-cp315-cp315t-win32.whl", hash = "sha256:1657923d2d45afb66526e5b933e5b3052e6bdea196c90d3abb2424e18c77dae8", size = 6474384, upload-time = "2026-07-01T11:56:18.855Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/e4/04/62bcd9f844984c5938d3b05264a61d797a29d3e0812341a8204af70bbdee/pillow-12.3.0-cp315-cp315t-win_amd64.whl", hash = "sha256:8cd2f7bdda092d99c9fc2fb7391354f306d01443d22785d0cbfafa2e2c8bb418", size = 7243537, upload-time = "2026-07-01T11:56:21.214Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/3d/68/1f3066acedf37673694a7141381d8f811ae97f30d34413d236abe7d489f1/pillow-12.3.0-cp315-cp315t-win_arm64.whl", hash = "sha256:06ff022112bc9cbf83b60f8e028d94ad87b60621706487e65f673de61610ab59", size = 2567491, upload-time = "2026-07-01T11:56:23.506Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/75/18/2e8b40223153ccbc60df07f9e8928dc0c76202aa4e55ae9f53962b6510d6/pillow-12.3.0-pp311-pypy311_pp73-macosx_10_15_x86_64.whl", hash = "sha256:b3c777e849237620b022f7f297dd67705f9f5cf1685f09f02e46f93e92725468", size = 5302510, upload-time = "2026-07-01T11:56:25.736Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/46/3e/51fabf59d5ab801ceab709453d3ab6b180083496579549de4c45ced6528a/pillow-12.3.0-pp311-pypy311_pp73-macosx_11_0_arm64.whl", hash = "sha256:b343699e8308bdc51978310e1c959c584e7869cc8c40780058c87da7781a1e94", size = 4736058, upload-time = "2026-07-01T11:56:28.041Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/bf/20/22fe9384b7949e25fb1293bcfc84fb82590ff4ea6b37c95b24d26d793d86/pillow-12.3.0-pp311-pypy311_pp73-manylinux_2_27_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:fbd139c8447d25dd750ab79ee274cc5e1fe80fc56340ab10b18a195e1b6eca3e", size = 5237776, upload-time = "2026-07-01T11:56:30.263Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/08/14/f6ba68107680ffa74b39985f3f30884e41318fbc4250caa423c79b4788bb/pillow-12.3.0-pp311-pypy311_pp73-manylinux_2_27_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:e7e480451b9fa137494bccd3a7d69adbe8ac65a87d97be61e11f1b1050a5bac3", size = 5860358, upload-time = "2026-07-01T11:56:32.68Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/36/54/0169bc772ec491108b62f644f8ecf1fe5d8ae5ebafde2ee2142210166903/pillow-12.3.0-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:04f01d28a6aaff387bf842a13be313df23ba0597a44f1a976c9feb3c6ff4711a", size = 7231786, upload-time = "2026-07-01T11:56:35.046Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "platformdirs"
|
||||
version = "4.9.6"
|
||||
|
||||
Reference in New Issue
Block a user