from __future__ import annotations import base64 import os import sqlite3 import zipfile from io import BytesIO from pathlib import Path import pytest from deepagents.backends.protocol import ExecuteResponse from PIL import Image from EvoScientist.native_sandbox import ( NativeSandboxExecutor, NativeSandboxUnavailable, NativeWorkspaceBackend, ) from EvoScientist.workspace_files import ( RootedWorkspace, ScopedFilesystemBackend, WorkspacePathError, normalize_workspace_path, ) from EvoScientist.workspace_scope import ( DeferredScopedBackend, _RuntimeScopeConfig, conversation_files_dir, delete_conversation_scope, provision_conversation_scope, ) def test_extracted_documents_route_through_deepagents_as_text(): import deepagents.middleware.filesystem as filesystem_middleware from EvoScientist.llm.patches import _patch_deepagents_extracted_document_text _patch_deepagents_extracted_document_text() get_file_type = vars(filesystem_middleware)["_get_file_type"] assert get_file_type("/workspace/report.pptx") == "text" assert get_file_type("/workspace/report.pdf") == "text" assert get_file_type("/workspace/image.png") == "image" def test_normalize_workspace_path_is_strict(): assert normalize_workspace_path("/workspace") == () assert normalize_workspace_path("/workspace/reports/a.txt") == ( "reports", "a.txt", ) for invalid in ( "/etc/passwd", "/workspace/../secret", "/workspace/a//b", "/workspace/./a", "workspace/a", "/workspace/a\\b", ): with pytest.raises(WorkspacePathError): normalize_workspace_path(invalid) def test_scoped_filesystem_backend_complete_round_trip(tmp_path: Path): backend = ScopedFilesystemBackend(tmp_path) assert backend.write("/workspace/notes/a.txt", "alpha\nbeta\n").error is None assert backend.read("/workspace/notes/a.txt").file_data == { "content": "alpha\nbeta\n", "encoding": "utf-8", } edit = backend.edit("/workspace/notes/a.txt", "beta", "gamma") assert edit.error is None assert edit.occurrences == 1 listing = backend.ls("/workspace/notes") assert [item["path"] for item in listing.entries or []] == [ "/workspace/notes/a.txt" ] assert [item["path"] for item in backend.glob("**/*.txt").matches or []] == [ "/workspace/notes/a.txt" ] assert backend.grep("gamma").matches == [ {"path": "/workspace/notes/a.txt", "line": 2, "text": "gamma"} ] upload = backend.upload_files([("/workspace/data.bin", b"\x00\x01")])[0] assert upload.error is None download = backend.download_files(["/workspace/data.bin"])[0] assert download.error is None assert download.content == b"\x00\x01" def test_root_lists_workspace_namespace(tmp_path: Path): backend = ScopedFilesystemBackend(tmp_path) assert backend.ls("/").entries == [ {"path": "/workspace/", "is_dir": True, "size": 0} ] def test_docx_read_extracts_text_instead_of_returning_base64(tmp_path: Path): backend = ScopedFilesystemBackend(tmp_path) docx = tmp_path / "input.docx" with zipfile.ZipFile(docx, "w") as archive: archive.writestr( "word/document.xml", """ 有界文档内容 """, ) result = backend.read("/workspace/input.docx") assert result.error is None assert result.file_data is not None assert result.file_data["encoding"] == "utf-8" assert "有界文档内容" in result.file_data["content"] assert "base64" not in result.file_data["content"] @pytest.mark.parametrize( ("filename", "kind", "expected"), [ ("archive.zip", "archive", "list"), ("results.sqlite", "database", "mode=ro"), ("program.exe", "executable", "must not be executed"), ("payload.custom", "binary", "unsupported"), ], ) def test_non_document_binary_returns_bounded_processing_guidance( tmp_path: Path, filename: str, kind: str, expected: str ): if filename.endswith(".zip"): with zipfile.ZipFile(tmp_path / filename, "w") as archive: archive.writestr("notes.txt", "hello") elif filename.endswith(".sqlite"): connection = sqlite3.connect(tmp_path / filename) connection.execute("CREATE TABLE results(id INTEGER PRIMARY KEY, value TEXT)") connection.commit() connection.close() elif filename.endswith(".exe"): (tmp_path / filename).write_bytes(b"MZ\x00binary payload") else: (tmp_path / filename).write_bytes(b"custom\x00binary payload") result = ScopedFilesystemBackend(tmp_path).read(f"/workspace/{filename}") assert result.file_data is None assert result.error is not None assert "BINARY_PROCESSING_REQUIRED" in result.error or "UNSUPPORTED_BINARY_FILE" in result.error assert f'"kind": "{kind}"' in result.error assert expected.lower() in result.error.lower() assert len(result.error) < 4000 def test_image_read_keeps_base64_media_contract(tmp_path: Path): buffer = BytesIO() Image.new("RGB", (32, 24), (1, 2, 3)).save(buffer, "PNG") raw = buffer.getvalue() (tmp_path / "image.png").write_bytes(raw) result = ScopedFilesystemBackend(tmp_path).read("/workspace/image.png") assert result.error is None assert result.file_data is not None assert result.file_data["encoding"] == "base64" def test_large_image_is_downsampled_before_base64_delivery(tmp_path: Path): Image.new("RGB", (3000, 1200), (1, 2, 3)).save(tmp_path / "large.png", "PNG") result = ScopedFilesystemBackend(tmp_path).read("/workspace/large.png") assert result.error is None assert result.file_data is not None decoded = base64.standard_b64decode(result.file_data["content"]) image = Image.open(BytesIO(decoded)) image.load() assert image.size == (2048, 819) assert image.format == "JPEG" def test_corrupt_image_returns_error_instead_of_base64(tmp_path: Path): (tmp_path / "broken.png").write_bytes(b"\x89PNG\r\nnot-decodable") result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.png") assert result.file_data is None assert result.error is not None assert "IMAGE_PROCESSING_FAILED" in result.error def test_image_pixel_budget_is_enforced_before_model_delivery( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ): import EvoScientist.document_extract as document_extract Image.new("RGB", (100, 100), (1, 2, 3)).save(tmp_path / "pixels.png", "PNG") monkeypatch.setattr(document_extract, "MAX_IMAGE_PIXELS", 9_999) result = ScopedFilesystemBackend(tmp_path).read("/workspace/pixels.png") assert result.file_data is None assert result.error is not None assert "IMAGE_PIXEL_BUDGET_EXCEEDED" in result.error def test_multiframe_image_is_reduced_to_first_frame(tmp_path: Path): frames = [Image.new("RGB", (20, 10), color) for color in ((255, 0, 0), (0, 255, 0))] frames[0].save( tmp_path / "animated.gif", format="GIF", save_all=True, append_images=frames[1:], duration=100, loop=0, ) result = ScopedFilesystemBackend(tmp_path).read("/workspace/animated.gif") assert result.error is None assert result.file_data is not None decoded = base64.standard_b64decode(result.file_data["content"]) image = Image.open(BytesIO(decoded)) image.load() assert getattr(image, "n_frames", 1) == 1 def test_pptx_read_extracts_slide_text_without_base64(tmp_path: Path): with zipfile.ZipFile(tmp_path / "deck.pptx", "w") as archive: archive.writestr( "ppt/slides/slide1.xml", """ 总体技术架构核心能力说明 """, ) result = ScopedFilesystemBackend(tmp_path).read("/workspace/deck.pptx") assert result.error is None assert result.file_data is not None assert result.file_data["encoding"] == "utf-8" assert "## Slide 1" in result.file_data["content"] assert "总体技术架构" in result.file_data["content"] def test_document_output_is_bounded_and_returns_continuation_hint(tmp_path: Path): long_text = "\n".join(f"第{i:05d}行-" + "x" * 80 for i in range(2000)) with zipfile.ZipFile(tmp_path / "long.docx", "w") as archive: paragraphs = "".join( f"{line}" for line in long_text.splitlines() ) archive.writestr( "word/document.xml", f""" {paragraphs}""", ) result = ScopedFilesystemBackend(tmp_path).read( "/workspace/long.docx", offset=0, limit=2000 ) assert result.error is None assert result.file_data is not None content = result.file_data["content"] assert len(content) < 51_000 assert "DOCUMENT_OUTPUT_TRUNCATED" in content assert "use offset=" in content def test_corrupt_document_does_not_fall_back_to_base64(tmp_path: Path): (tmp_path / "broken.pptx").write_bytes(b"PK\x03\x04not-a-real-presentation") result = ScopedFilesystemBackend(tmp_path).read("/workspace/broken.pptx") assert result.file_data is None assert result.error is not None assert "DOCUMENT_EXTRACTION_FAILED" in result.error assert "base64" not in result.error.lower() def test_ooxml_member_expansion_budget_blocks_compression_bomb( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ): import EvoScientist.document_extract as document_extract with zipfile.ZipFile( tmp_path / "bomb.docx", "w", compression=zipfile.ZIP_DEFLATED ) as archive: archive.writestr( "word/document.xml", """ expanded content """, ) monkeypatch.setattr(document_extract, "MAX_OOXML_MEMBER_BYTES", 16) result = ScopedFilesystemBackend(tmp_path).read("/workspace/bomb.docx") assert result.file_data is None assert result.error is not None assert "DOCUMENT_RESOURCE_LIMIT" in result.error @pytest.mark.parametrize("member", ["../word/document.xml", "/word/document.xml"]) def test_ooxml_rejects_unsafe_member_paths(tmp_path: Path, member: str): with zipfile.ZipFile(tmp_path / "unsafe.docx", "w") as archive: archive.writestr(member, "content") archive.writestr( "word/document.xml", """ safe""", ) result = ScopedFilesystemBackend(tmp_path).read("/workspace/unsafe.docx") assert result.file_data is None assert result.error is not None assert "DOCUMENT_RESOURCE_LIMIT" in result.error def test_ooxml_rejects_duplicate_member_names(tmp_path: Path): def write_duplicate_document(path: Path) -> None: with zipfile.ZipFile(path, "w") as archive: for text in ("first", "second"): archive.writestr( "word/document.xml", f""" {text}""", ) with pytest.warns(UserWarning, match="Duplicate name"): write_duplicate_document(tmp_path / "duplicate.docx") result = ScopedFilesystemBackend(tmp_path).read("/workspace/duplicate.docx") assert result.file_data is None assert result.error is not None assert "DOCUMENT_RESOURCE_LIMIT" in result.error def test_oversized_document_is_rejected_before_opening_content( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ): from EvoScientist.document_extract import MAX_DOCUMENT_BYTES path = tmp_path / "oversized.pdf" path.write_bytes(b"%PDF") original_entry = RootedWorkspace.entry def oversized_entry(self, virtual_path): entry = original_entry(self, virtual_path) return type(entry)(entry.virtual_path, entry.is_dir, MAX_DOCUMENT_BYTES + 1, entry.modified_at) monkeypatch.setattr(RootedWorkspace, "entry", oversized_entry) result = ScopedFilesystemBackend(tmp_path).read("/workspace/oversized.pdf") assert result.file_data is None assert result.error is not None assert "DOCUMENT_TOO_LARGE" in result.error def test_external_document_converter_timeout_is_structured( monkeypatch: pytest.MonkeyPatch, ): import subprocess import EvoScientist.document_extract as document_extract def timeout(*args, **kwargs): raise subprocess.TimeoutExpired(cmd="anydoc", timeout=60) monkeypatch.setattr(document_extract.subprocess, "run", timeout) with pytest.raises( document_extract.DocumentExtractionError, match="DOCUMENT_CONVERSION_TIMEOUT", ): document_extract.extract_document_bytes(b"%PDF-minimal", "sample.pdf") @pytest.mark.parametrize("filename", ["report.pptx", "archive.zip", "results.sqlite"]) def test_text_write_cannot_create_or_corrupt_binary_container( tmp_path: Path, filename: str ): backend = ScopedFilesystemBackend(tmp_path) created = backend.write(f"/workspace/{filename}", "extracted text") assert created.error is not None assert "binary container" in created.error.lower() assert not (tmp_path / filename).exists() def test_text_edit_cannot_modify_existing_binary_container(tmp_path: Path): source = b"PK\x03\x04original-container" (tmp_path / "report.pptx").write_bytes(source) backend = ScopedFilesystemBackend(tmp_path) edited = backend.edit( "/workspace/report.pptx", "original", "replacement" ) assert edited.error is not None assert "binary container" in edited.error.lower() assert (tmp_path / "report.pptx").read_bytes() == source def test_uploaded_text_is_read_only_to_file_tools(tmp_path: Path): uploads = tmp_path / "uploads" uploads.mkdir() source = uploads / "notes.txt" source.write_text("original", encoding="utf-8") backend = ScopedFilesystemBackend(tmp_path) written = backend.write("/workspace/uploads/new.txt", "new") edited = backend.edit("/workspace/uploads/notes.txt", "original", "changed") assert written.error is not None assert "uploads" in written.error.lower() assert edited.error is not None assert "uploads" in edited.error.lower() assert not (uploads / "new.txt").exists() assert source.read_text(encoding="utf-8") == "original" def test_utf8_sample_boundary_cut_is_not_binary(tmp_path: Path): # Regression (2026-08-22): the 8192-byte UTF-8 probe can split a multi-byte # CJK character at the sample boundary (req_v13.md cut at 8190/8191 split a # 3-byte char). That raised UnicodeDecodeError -> misclassified as binary # -> read_file returned a base64 file media block -> providers without file # input replaced it with a placeholder -> the model retried forever. backend = ScopedFilesystemBackend(tmp_path) # 8190 ASCII bytes + one 3-byte CJK char, so the sample cuts mid-character. content = ("a" * 8190 + "\u6e56" + "more text").encode("utf-8") assert len(content) > 8192 assert content[8190:8193] == "\u6e56".encode("utf-8") backend.upload_files([("/workspace/cjk.md", content)]) result = backend.read("/workspace/cjk.md") assert result.file_data is not None assert result.file_data["encoding"] == "utf-8" assert result.file_data["content"].startswith("a" * 10) def test_mid_sample_invalid_bytes_still_binary(): from EvoScientist.workspace_files import _is_binary_file # Invalid bytes well inside the sample are genuine garbage, not a cut. assert _is_binary_file("/workspace/bad.raw", b"ok\xffi\xffd\xefmore") # A boundary cut (error in the last 4 bytes that decodes clean when the # dangling suffix is dropped) is text. cut = ("a" * 8190 + "\u6e56").encode("utf-8")[:8192] assert not _is_binary_file("/workspace/cut.md", cut) # Same shape but the prefix itself is invalid -> stays binary. assert _is_binary_file("/workspace/bad.md", b"\xff" * 8192) def test_symlink_targets_and_parents_are_rejected(tmp_path: Path): outside = tmp_path.parent / "outside-secret.txt" outside.write_text("secret", encoding="utf-8") (tmp_path / "leak.txt").symlink_to(outside) (tmp_path / "escape").symlink_to(tmp_path.parent, target_is_directory=True) backend = ScopedFilesystemBackend(tmp_path) assert backend.read("/workspace/leak.txt").error assert backend.write("/workspace/escape/new.txt", "nope").error assert backend.download_files(["/workspace/leak.txt"])[0].error == "invalid_path" assert backend.ls("/workspace").entries == [] def test_rooted_workspace_returns_open_verified_handle(tmp_path: Path): workspace = RootedWorkspace(tmp_path) workspace.replace_file("/workspace/result.txt", b"result") handle = workspace.open_binary("/workspace/result.txt") os.unlink(tmp_path / "result.txt") try: assert handle.read() == b"result" finally: handle.close() def test_rooted_workspace_entry_and_recursive_delete(tmp_path: Path): workspace = RootedWorkspace(tmp_path) workspace.replace_file("/workspace/report/assets/app.js", b"app()") assert workspace.entry("/workspace/report").is_dir assert workspace.entry("/workspace/report/assets/app.js").size == 5 workspace.delete("/workspace/report") with pytest.raises(FileNotFoundError): workspace.entry("/workspace/report") def test_native_backend_delegates_validated_command_to_executor( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ): files = tmp_path / "files" runtime = tmp_path / "runtime" files.mkdir() runtime.mkdir() captured: dict[str, object] = {} def fake_execute( self, command, *, timeout=None, skip_readiness_check=False, cancel_event=None, ): captured.update(command=command, timeout=timeout) return ExecuteResponse(output="ok\n", exit_code=0, truncated=False) monkeypatch.setattr(NativeSandboxExecutor, "execute", fake_execute) backend = NativeWorkspaceBackend(files, runtime, timeout=30) result = backend.execute("cat probe.txt", timeout=12) assert result.exit_code == 0 assert captured == {"command": "cat probe.txt", "timeout": 12} blocked = backend.execute("sudo cat probe.txt") assert blocked.exit_code == 1 assert "blocked" in blocked.output.lower() invalid = backend.execute("printf 'bad\x00command'") assert invalid.exit_code == 1 def test_native_backend_fails_closed_when_executor_is_unavailable( tmp_path: Path, monkeypatch: pytest.MonkeyPatch ): files = tmp_path / "files" runtime = tmp_path / "runtime" files.mkdir() runtime.mkdir() def unavailable(*args, **kwargs): raise NativeSandboxUnavailable("private deployment detail") monkeypatch.setattr(NativeSandboxExecutor, "execute", unavailable) result = NativeWorkspaceBackend(files, runtime, timeout=30).execute("echo ok") assert result.exit_code == 125 assert "private deployment detail" not in result.output @pytest.mark.asyncio async def test_deferred_backend_preserves_async_executor_cancellation_path( monkeypatch: pytest.MonkeyPatch, ): proxy = DeferredScopedBackend( _RuntimeScopeConfig( scope_id="00000000-0000-0000-0000-000000000001", owner_id="00000000-0000-0000-0000-000000000002", thread_id="thread", deployment_id="deployment", ), dangerous=False, ) captured: dict[str, object] = {} class Delegate: async def aexecute(self, command, *, timeout=None): captured.update(command=command, timeout=timeout) return ExecuteResponse(output="async", exit_code=0, truncated=False) monkeypatch.setattr(proxy, "_delegate", lambda: Delegate()) result = await proxy.aexecute("sleep 1", timeout=7) assert result.output == "async" assert captured == {"command": "sleep 1", "timeout": 7} def test_scope_delete_is_idempotent_and_removes_directory(tmp_path: Path): thread_id = "0f88db64-720e-4f88-ac92-ea9a76b45596" deployment_id = "test-workspace-delete" record = provision_conversation_scope( thread_id, deployment_id=deployment_id, workspace_root=tmp_path, ) scope_root = tmp_path / ".evoscientist" / "conversations" / record.scope_id (scope_root / "files" / "result.txt").write_text("done", encoding="utf-8") deleted = delete_conversation_scope( thread_id, deployment_id=deployment_id, workspace_root=tmp_path, ) assert deleted.state == "deleted" assert not scope_root.exists() assert ( delete_conversation_scope( thread_id, deployment_id=deployment_id, workspace_root=tmp_path, ).state == "deleted" ) def test_deleted_scope_can_be_reprovisioned_for_create_retry(tmp_path: Path): thread_id = "a224305c-32ae-43c5-bdae-76b52912fb37" deployment_id = "test-workspace-retry" original = provision_conversation_scope( thread_id, deployment_id=deployment_id, workspace_root=tmp_path ) delete_conversation_scope( thread_id, deployment_id=deployment_id, workspace_root=tmp_path ) retried = provision_conversation_scope( thread_id, deployment_id=deployment_id, workspace_root=tmp_path ) assert retried.scope_id == original.scope_id assert retried.state == "draft" assert conversation_files_dir(retried.scope_id, tmp_path).is_dir()