Files
EvoScientist-Multi/tests/test_workspace_files.py
T
m4 c683f6e739
Docker / build (push) Has been cancelled
Build / build (push) Has been cancelled
Lint / ruff (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.11) (push) Has been cancelled
Test / pytest (ubuntu-latest, 3.12) (push) Has been cancelled
Test / pytest (windows-latest, 3.11) (push) Has been cancelled
Test / pytest (windows-latest, 3.12) (push) Has been cancelled
feat: prepare EvoScientist 0.3.0
Add bounded document ingestion, controlled web search, recoverable session support, subagent timeouts, and the native sandbox runtime contract. Unify package versioning and add release-focused regression coverage.
2026-09-03 06:55:56 +08:00

626 lines
22 KiB
Python

from __future__ import annotations
import base64
import os
import sqlite3
import zipfile
from io import BytesIO
from pathlib import Path
import pytest
from deepagents.backends.protocol import ExecuteResponse
from PIL import Image
from EvoScientist.native_sandbox import (
NativeSandboxExecutor,
NativeSandboxUnavailable,
NativeWorkspaceBackend,
)
from EvoScientist.workspace_files import (
RootedWorkspace,
ScopedFilesystemBackend,
WorkspacePathError,
normalize_workspace_path,
)
from EvoScientist.workspace_scope import (
DeferredScopedBackend,
_RuntimeScopeConfig,
conversation_files_dir,
delete_conversation_scope,
provision_conversation_scope,
)
def test_extracted_documents_route_through_deepagents_as_text():
import deepagents.middleware.filesystem as filesystem_middleware
from EvoScientist.llm.patches import _patch_deepagents_extracted_document_text
_patch_deepagents_extracted_document_text()
get_file_type = vars(filesystem_middleware)["_get_file_type"]
assert get_file_type("/workspace/report.pptx") == "text"
assert get_file_type("/workspace/report.pdf") == "text"
assert get_file_type("/workspace/image.png") == "image"
def test_normalize_workspace_path_is_strict():
assert normalize_workspace_path("/workspace") == ()
assert normalize_workspace_path("/workspace/reports/a.txt") == (
"reports",
"a.txt",
)
for invalid in (
"/etc/passwd",
"/workspace/../secret",
"/workspace/a//b",
"/workspace/./a",
"workspace/a",
"/workspace/a\\b",
):
with pytest.raises(WorkspacePathError):
normalize_workspace_path(invalid)
def test_scoped_filesystem_backend_complete_round_trip(tmp_path: Path):
backend = ScopedFilesystemBackend(tmp_path)
assert backend.write("/workspace/notes/a.txt", "alpha\nbeta\n").error is None
assert backend.read("/workspace/notes/a.txt").file_data == {
"content": "alpha\nbeta\n",
"encoding": "utf-8",
}
edit = backend.edit("/workspace/notes/a.txt", "beta", "gamma")
assert edit.error is None
assert edit.occurrences == 1
listing = backend.ls("/workspace/notes")
assert [item["path"] for item in listing.entries or []] == [
"/workspace/notes/a.txt"
]
assert [item["path"] for item in backend.glob("**/*.txt").matches or []] == [
"/workspace/notes/a.txt"
]
assert backend.grep("gamma").matches == [
{"path": "/workspace/notes/a.txt", "line": 2, "text": "gamma"}
]
upload = backend.upload_files([("/workspace/data.bin", b"\x00\x01")])[0]
assert upload.error is None
download = backend.download_files(["/workspace/data.bin"])[0]
assert download.error is None
assert download.content == b"\x00\x01"
def test_root_lists_workspace_namespace(tmp_path: Path):
backend = ScopedFilesystemBackend(tmp_path)
assert backend.ls("/").entries == [
{"path": "/workspace/", "is_dir": True, "size": 0}
]
def test_docx_read_extracts_text_instead_of_returning_base64(tmp_path: Path):
backend = ScopedFilesystemBackend(tmp_path)
docx = tmp_path / "input.docx"
with zipfile.ZipFile(docx, "w") as archive:
archive.writestr(
"word/document.xml",
"""<?xml version="1.0" encoding="UTF-8"?>
<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>有界文档内容</w:t></w:r></w:p></w:body>
</w:document>""",
)
result = backend.read("/workspace/input.docx")
assert result.error is None
assert result.file_data is not None
assert result.file_data["encoding"] == "utf-8"
assert "有界文档内容" in result.file_data["content"]
assert "base64" not in result.file_data["content"]
@pytest.mark.parametrize(
("filename", "kind", "expected"),
[
("archive.zip", "archive", "list"),
("results.sqlite", "database", "mode=ro"),
("program.exe", "executable", "must not be executed"),
("payload.custom", "binary", "unsupported"),
],
)
def test_non_document_binary_returns_bounded_processing_guidance(
tmp_path: Path, filename: str, kind: str, expected: str
):
if filename.endswith(".zip"):
with zipfile.ZipFile(tmp_path / filename, "w") as archive:
archive.writestr("notes.txt", "hello")
elif filename.endswith(".sqlite"):
connection = sqlite3.connect(tmp_path / filename)
connection.execute("CREATE TABLE results(id INTEGER PRIMARY KEY, value TEXT)")
connection.commit()
connection.close()
elif filename.endswith(".exe"):
(tmp_path / filename).write_bytes(b"MZ\x00binary payload")
else:
(tmp_path / filename).write_bytes(b"custom\x00binary payload")
result = ScopedFilesystemBackend(tmp_path).read(f"/workspace/{filename}")
assert result.file_data is None
assert result.error is not None
assert "BINARY_PROCESSING_REQUIRED" in result.error or "UNSUPPORTED_BINARY_FILE" in result.error
assert f'"kind": "{kind}"' in result.error
assert expected.lower() in result.error.lower()
assert len(result.error) < 4000
def test_image_read_keeps_base64_media_contract(tmp_path: Path):
buffer = BytesIO()
Image.new("RGB", (32, 24), (1, 2, 3)).save(buffer, "PNG")
raw = buffer.getvalue()
(tmp_path / "image.png").write_bytes(raw)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/image.png")
assert result.error is None
assert result.file_data is not None
assert result.file_data["encoding"] == "base64"
def test_large_image_is_downsampled_before_base64_delivery(tmp_path: Path):
Image.new("RGB", (3000, 1200), (1, 2, 3)).save(tmp_path / "large.png", "PNG")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/large.png")
assert result.error is None
assert result.file_data is not None
decoded = base64.standard_b64decode(result.file_data["content"])
image = Image.open(BytesIO(decoded))
image.load()
assert image.size == (2048, 819)
assert image.format == "JPEG"
def test_corrupt_image_returns_error_instead_of_base64(tmp_path: Path):
(tmp_path / "broken.png").write_bytes(b"\x89PNG\r\nnot-decodable")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.png")
assert result.file_data is None
assert result.error is not None
assert "IMAGE_PROCESSING_FAILED" in result.error
def test_image_pixel_budget_is_enforced_before_model_delivery(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
import EvoScientist.document_extract as document_extract
Image.new("RGB", (100, 100), (1, 2, 3)).save(tmp_path / "pixels.png", "PNG")
monkeypatch.setattr(document_extract, "MAX_IMAGE_PIXELS", 9_999)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/pixels.png")
assert result.file_data is None
assert result.error is not None
assert "IMAGE_PIXEL_BUDGET_EXCEEDED" in result.error
def test_multiframe_image_is_reduced_to_first_frame(tmp_path: Path):
frames = [Image.new("RGB", (20, 10), color) for color in ((255, 0, 0), (0, 255, 0))]
frames[0].save(
tmp_path / "animated.gif",
format="GIF",
save_all=True,
append_images=frames[1:],
duration=100,
loop=0,
)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/animated.gif")
assert result.error is None
assert result.file_data is not None
decoded = base64.standard_b64decode(result.file_data["content"])
image = Image.open(BytesIO(decoded))
image.load()
assert getattr(image, "n_frames", 1) == 1
def test_pptx_read_extracts_slide_text_without_base64(tmp_path: Path):
with zipfile.ZipFile(tmp_path / "deck.pptx", "w") as archive:
archive.writestr(
"ppt/slides/slide1.xml",
"""<p:sld xmlns:p="http://schemas.openxmlformats.org/presentationml/2006/main"
xmlns:a="http://schemas.openxmlformats.org/drawingml/2006/main">
<p:cSld><a:t>总体技术架构</a:t><a:t>核心能力说明</a:t></p:cSld>
</p:sld>""",
)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/deck.pptx")
assert result.error is None
assert result.file_data is not None
assert result.file_data["encoding"] == "utf-8"
assert "## Slide 1" in result.file_data["content"]
assert "总体技术架构" in result.file_data["content"]
def test_document_output_is_bounded_and_returns_continuation_hint(tmp_path: Path):
long_text = "\n".join(f"第{i:05d}行-" + "x" * 80 for i in range(2000))
with zipfile.ZipFile(tmp_path / "long.docx", "w") as archive:
paragraphs = "".join(
f"<w:p><w:r><w:t>{line}</w:t></w:r></w:p>"
for line in long_text.splitlines()
)
archive.writestr(
"word/document.xml",
f"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body>{paragraphs}</w:body></w:document>""",
)
result = ScopedFilesystemBackend(tmp_path).read(
"/workspace/long.docx", offset=0, limit=2000
)
assert result.error is None
assert result.file_data is not None
content = result.file_data["content"]
assert len(content) < 51_000
assert "DOCUMENT_OUTPUT_TRUNCATED" in content
assert "use offset=" in content
def test_corrupt_document_does_not_fall_back_to_base64(tmp_path: Path):
(tmp_path / "broken.pptx").write_bytes(b"PK\x03\x04not-a-real-presentation")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.pptx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_EXTRACTION_FAILED" in result.error
assert "base64" not in result.error.lower()
def test_ooxml_member_expansion_budget_blocks_compression_bomb(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
import EvoScientist.document_extract as document_extract
with zipfile.ZipFile(
tmp_path / "bomb.docx", "w", compression=zipfile.ZIP_DEFLATED
) as archive:
archive.writestr(
"word/document.xml",
"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>expanded content</w:t></w:r></w:p></w:body>
</w:document>""",
)
monkeypatch.setattr(document_extract, "MAX_OOXML_MEMBER_BYTES", 16)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/bomb.docx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
@pytest.mark.parametrize("member", ["../word/document.xml", "/word/document.xml"])
def test_ooxml_rejects_unsafe_member_paths(tmp_path: Path, member: str):
with zipfile.ZipFile(tmp_path / "unsafe.docx", "w") as archive:
archive.writestr(member, "content")
archive.writestr(
"word/document.xml",
"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>safe</w:t></w:r></w:p></w:body></w:document>""",
)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/unsafe.docx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
def test_ooxml_rejects_duplicate_member_names(tmp_path: Path):
def write_duplicate_document(path: Path) -> None:
with zipfile.ZipFile(path, "w") as archive:
for text in ("first", "second"):
archive.writestr(
"word/document.xml",
f"""<w:document xmlns:w="http://schemas.openxmlformats.org/wordprocessingml/2006/main">
<w:body><w:p><w:r><w:t>{text}</w:t></w:r></w:p></w:body></w:document>""",
)
with pytest.warns(UserWarning, match="Duplicate name"):
write_duplicate_document(tmp_path / "duplicate.docx")
result = ScopedFilesystemBackend(tmp_path).read("/workspace/duplicate.docx")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_RESOURCE_LIMIT" in result.error
def test_oversized_document_is_rejected_before_opening_content(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
from EvoScientist.document_extract import MAX_DOCUMENT_BYTES
path = tmp_path / "oversized.pdf"
path.write_bytes(b"%PDF")
original_entry = RootedWorkspace.entry
def oversized_entry(self, virtual_path):
entry = original_entry(self, virtual_path)
return type(entry)(entry.virtual_path, entry.is_dir, MAX_DOCUMENT_BYTES + 1, entry.modified_at)
monkeypatch.setattr(RootedWorkspace, "entry", oversized_entry)
result = ScopedFilesystemBackend(tmp_path).read("/workspace/oversized.pdf")
assert result.file_data is None
assert result.error is not None
assert "DOCUMENT_TOO_LARGE" in result.error
def test_external_document_converter_timeout_is_structured(
monkeypatch: pytest.MonkeyPatch,
):
import subprocess
import EvoScientist.document_extract as document_extract
def timeout(*args, **kwargs):
raise subprocess.TimeoutExpired(cmd="anydoc", timeout=60)
monkeypatch.setattr(document_extract.subprocess, "run", timeout)
with pytest.raises(
document_extract.DocumentExtractionError,
match="DOCUMENT_CONVERSION_TIMEOUT",
):
document_extract.extract_document_bytes(b"%PDF-minimal", "sample.pdf")
@pytest.mark.parametrize("filename", ["report.pptx", "archive.zip", "results.sqlite"])
def test_text_write_cannot_create_or_corrupt_binary_container(
tmp_path: Path, filename: str
):
backend = ScopedFilesystemBackend(tmp_path)
created = backend.write(f"/workspace/{filename}", "extracted text")
assert created.error is not None
assert "binary container" in created.error.lower()
assert not (tmp_path / filename).exists()
def test_text_edit_cannot_modify_existing_binary_container(tmp_path: Path):
source = b"PK\x03\x04original-container"
(tmp_path / "report.pptx").write_bytes(source)
backend = ScopedFilesystemBackend(tmp_path)
edited = backend.edit(
"/workspace/report.pptx", "original", "replacement"
)
assert edited.error is not None
assert "binary container" in edited.error.lower()
assert (tmp_path / "report.pptx").read_bytes() == source
def test_uploaded_text_is_read_only_to_file_tools(tmp_path: Path):
uploads = tmp_path / "uploads"
uploads.mkdir()
source = uploads / "notes.txt"
source.write_text("original", encoding="utf-8")
backend = ScopedFilesystemBackend(tmp_path)
written = backend.write("/workspace/uploads/new.txt", "new")
edited = backend.edit("/workspace/uploads/notes.txt", "original", "changed")
assert written.error is not None
assert "uploads" in written.error.lower()
assert edited.error is not None
assert "uploads" in edited.error.lower()
assert not (uploads / "new.txt").exists()
assert source.read_text(encoding="utf-8") == "original"
def test_utf8_sample_boundary_cut_is_not_binary(tmp_path: Path):
# Regression (2026-08-22): the 8192-byte UTF-8 probe can split a multi-byte
# CJK character at the sample boundary (req_v13.md cut at 8190/8191 split a
# 3-byte char). That raised UnicodeDecodeError -> misclassified as binary
# -> read_file returned a base64 file media block -> providers without file
# input replaced it with a placeholder -> the model retried forever.
backend = ScopedFilesystemBackend(tmp_path)
# 8190 ASCII bytes + one 3-byte CJK char, so the sample cuts mid-character.
content = ("a" * 8190 + "\u6e56" + "more text").encode("utf-8")
assert len(content) > 8192
assert content[8190:8193] == "\u6e56".encode("utf-8")
backend.upload_files([("/workspace/cjk.md", content)])
result = backend.read("/workspace/cjk.md")
assert result.file_data is not None
assert result.file_data["encoding"] == "utf-8"
assert result.file_data["content"].startswith("a" * 10)
def test_mid_sample_invalid_bytes_still_binary():
from EvoScientist.workspace_files import _is_binary_file
# Invalid bytes well inside the sample are genuine garbage, not a cut.
assert _is_binary_file("/workspace/bad.raw", b"ok\xffi\xffd\xefmore")
# A boundary cut (error in the last 4 bytes that decodes clean when the
# dangling suffix is dropped) is text.
cut = ("a" * 8190 + "\u6e56").encode("utf-8")[:8192]
assert not _is_binary_file("/workspace/cut.md", cut)
# Same shape but the prefix itself is invalid -> stays binary.
assert _is_binary_file("/workspace/bad.md", b"\xff" * 8192)
def test_symlink_targets_and_parents_are_rejected(tmp_path: Path):
outside = tmp_path.parent / "outside-secret.txt"
outside.write_text("secret", encoding="utf-8")
(tmp_path / "leak.txt").symlink_to(outside)
(tmp_path / "escape").symlink_to(tmp_path.parent, target_is_directory=True)
backend = ScopedFilesystemBackend(tmp_path)
assert backend.read("/workspace/leak.txt").error
assert backend.write("/workspace/escape/new.txt", "nope").error
assert backend.download_files(["/workspace/leak.txt"])[0].error == "invalid_path"
assert backend.ls("/workspace").entries == []
def test_rooted_workspace_returns_open_verified_handle(tmp_path: Path):
workspace = RootedWorkspace(tmp_path)
workspace.replace_file("/workspace/result.txt", b"result")
handle = workspace.open_binary("/workspace/result.txt")
os.unlink(tmp_path / "result.txt")
try:
assert handle.read() == b"result"
finally:
handle.close()
def test_rooted_workspace_entry_and_recursive_delete(tmp_path: Path):
workspace = RootedWorkspace(tmp_path)
workspace.replace_file("/workspace/report/assets/app.js", b"app()")
assert workspace.entry("/workspace/report").is_dir
assert workspace.entry("/workspace/report/assets/app.js").size == 5
workspace.delete("/workspace/report")
with pytest.raises(FileNotFoundError):
workspace.entry("/workspace/report")
def test_native_backend_delegates_validated_command_to_executor(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
files = tmp_path / "files"
runtime = tmp_path / "runtime"
files.mkdir()
runtime.mkdir()
captured: dict[str, object] = {}
def fake_execute(
self,
command,
*,
timeout=None,
skip_readiness_check=False,
cancel_event=None,
):
captured.update(command=command, timeout=timeout)
return ExecuteResponse(output="ok\n", exit_code=0, truncated=False)
monkeypatch.setattr(NativeSandboxExecutor, "execute", fake_execute)
backend = NativeWorkspaceBackend(files, runtime, timeout=30)
result = backend.execute("cat probe.txt", timeout=12)
assert result.exit_code == 0
assert captured == {"command": "cat probe.txt", "timeout": 12}
blocked = backend.execute("sudo cat probe.txt")
assert blocked.exit_code == 1
assert "blocked" in blocked.output.lower()
invalid = backend.execute("printf 'bad\x00command'")
assert invalid.exit_code == 1
def test_native_backend_fails_closed_when_executor_is_unavailable(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
):
files = tmp_path / "files"
runtime = tmp_path / "runtime"
files.mkdir()
runtime.mkdir()
def unavailable(*args, **kwargs):
raise NativeSandboxUnavailable("private deployment detail")
monkeypatch.setattr(NativeSandboxExecutor, "execute", unavailable)
result = NativeWorkspaceBackend(files, runtime, timeout=30).execute("echo ok")
assert result.exit_code == 125
assert "private deployment detail" not in result.output
@pytest.mark.asyncio
async def test_deferred_backend_preserves_async_executor_cancellation_path(
monkeypatch: pytest.MonkeyPatch,
):
proxy = DeferredScopedBackend(
_RuntimeScopeConfig(
scope_id="00000000-0000-0000-0000-000000000001",
owner_id="00000000-0000-0000-0000-000000000002",
thread_id="thread",
deployment_id="deployment",
),
dangerous=False,
)
captured: dict[str, object] = {}
class Delegate:
async def aexecute(self, command, *, timeout=None):
captured.update(command=command, timeout=timeout)
return ExecuteResponse(output="async", exit_code=0, truncated=False)
monkeypatch.setattr(proxy, "_delegate", lambda: Delegate())
result = await proxy.aexecute("sleep 1", timeout=7)
assert result.output == "async"
assert captured == {"command": "sleep 1", "timeout": 7}
def test_scope_delete_is_idempotent_and_removes_directory(tmp_path: Path):
thread_id = "0f88db64-720e-4f88-ac92-ea9a76b45596"
deployment_id = "test-workspace-delete"
record = provision_conversation_scope(
thread_id,
deployment_id=deployment_id,
workspace_root=tmp_path,
)
scope_root = tmp_path / ".evoscientist" / "conversations" / record.scope_id
(scope_root / "files" / "result.txt").write_text("done", encoding="utf-8")
deleted = delete_conversation_scope(
thread_id,
deployment_id=deployment_id,
workspace_root=tmp_path,
)
assert deleted.state == "deleted"
assert not scope_root.exists()
assert (
delete_conversation_scope(
thread_id,
deployment_id=deployment_id,
workspace_root=tmp_path,
).state
== "deleted"
)
def test_deleted_scope_can_be_reprovisioned_for_create_retry(tmp_path: Path):
thread_id = "a224305c-32ae-43c5-bdae-76b52912fb37"
deployment_id = "test-workspace-retry"
original = provision_conversation_scope(
thread_id, deployment_id=deployment_id, workspace_root=tmp_path
)
delete_conversation_scope(
thread_id, deployment_id=deployment_id, workspace_root=tmp_path
)
retried = provision_conversation_scope(
thread_id, deployment_id=deployment_id, workspace_root=tmp_path
)
assert retried.scope_id == original.scope_id
assert retried.state == "draft"
assert conversation_files_dir(retried.scope_id, tmp_path).is_dir()