diff --git a/.gitignore b/.gitignore index 9186f4f..2f953f2 100644 --- a/.gitignore +++ b/.gitignore @@ -21,6 +21,7 @@ build/ venv/ bridge/node_modules/ bridge/package-lock.json +runtime/native-sandbox/node_modules/ # IDE / Tools .vscode/ diff --git a/EvoScientist/EvoScientist.py b/EvoScientist/EvoScientist.py index 7f8304e..ca038c5 100644 --- a/EvoScientist/EvoScientist.py +++ b/EvoScientist/EvoScientist.py @@ -284,6 +284,10 @@ def _configured_system_prompt(cfg) -> str: return get_system_prompt( dangerous=cfg.dangerous_mode, cwd=real_cwd, + native_web_sandbox=( + not cfg.dangerous_mode + and os.environ.get("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full" + ), ) @@ -946,25 +950,58 @@ def _get_default_agent(): from deepagents import create_deep_agent cfg = _ensure_config() + web_full = os.environ.get("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full" + if web_full: + # Refuse to expose a Web graph whose command isolation is missing or + # ineffective. CLI/stripped runtimes do not enter this path. + from .native_sandbox import ensure_native_sandbox_ready + + ensure_native_sandbox_ready() be = _get_default_backend() - mw = _get_default_middleware() + mw = ( + _get_default_middleware( + enable_background_execution=False, + enable_scheduler=False, + enable_memory_workers=False, + ) + if web_full + else _get_default_middleware() + ) # HITL on main agent only (mirrors create_cli_agent). Use middleware, # not interrupt_on= kwarg — the kwarg propagates to every subagent and # breaks parallel execute calls (multi-pending-interrupt LangGraph # error). See PR #202. - if not cfg.auto_approve: - mw.append( - HumanInTheLoopMiddleware( - interrupt_on={ - "execute": True, - "run_in_background": True, - "schedule_task": True, - } - ) - ) + from .middleware import DynamicReviewMiddleware - if os.environ.get("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "stripped": + mw.append( + DynamicReviewMiddleware( + interrupt_on={ + "execute": True, + "run_in_background": True, + "schedule_task": True, + } + ) + ) + + if web_full: + kwargs = _build_base_kwargs( + be, + mw, + workspace_dir="/workspace", + ) + # Web file access must stay on the scoped backend. Host MCP, + # subagents and global skill mutation remain unavailable here. + kwargs = { + **kwargs, + "subagents": [], + "tools": [ + tool + for tool in kwargs.get("tools", []) + if getattr(tool, "name", "") != "skill_manager" + ], + } + elif os.environ.get("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "stripped": kwargs = _build_base_kwargs( be, mw, diff --git a/EvoScientist/langgraph_dev/http.py b/EvoScientist/langgraph_dev/http.py index c7dcf55..e3906ca 100644 --- a/EvoScientist/langgraph_dev/http.py +++ b/EvoScientist/langgraph_dev/http.py @@ -23,11 +23,9 @@ memory. from __future__ import annotations import asyncio -import hashlib import json import os import secrets -from pathlib import Path, PurePosixPath from typing import Any from uuid import UUID @@ -194,6 +192,31 @@ async def get_workspace_scope(request: Request) -> JSONResponse: return JSONResponse(_scope_payload(record)) +async def delete_workspace_scope(request: Request) -> JSONResponse: + if denied := _scope_service_authorized(request): + return denied + try: + from EvoScientist.workspace_scope import ( + current_deployment_id, + delete_conversation_scope, + ) + + record = await asyncio.to_thread( + delete_conversation_scope, + str(request.path_params["thread_id"]), + deployment_id=current_deployment_id(), + ) + except Exception as exc: + code = ( + "WORKSPACE_SCOPE_NOT_FOUND" + if type(exc).__name__ == "ScopeNotFoundError" + else "WORKSPACE_SCOPE_DELETE_FAILED" + ) + status = 404 if code == "WORKSPACE_SCOPE_NOT_FOUND" else 409 + return JSONResponse({"code": code, "message": str(exc)}, status_code=status) + return JSONResponse(_scope_payload(record)) + + async def reserve_workspace_run(request: Request) -> JSONResponse: if denied := _scope_service_authorized(request): return denied @@ -247,69 +270,6 @@ async def bind_workspace_run(request: Request) -> JSONResponse: return JSONResponse(_run_payload(run)) -def _materialize_target(scope_id: str, raw_path: str) -> Path: - from EvoScientist.workspace_scope import conversation_files_dir - - path = PurePosixPath(raw_path.replace("\\", "/")) - if path.is_absolute() or not path.parts or path.parts[0] != "uploads": - raise ValueError("only uploads/ paths are accepted") - if any(part in {"", ".", ".."} for part in path.parts): - raise ValueError("invalid upload path") - root = conversation_files_dir(scope_id).resolve() - target = root.joinpath(*path.parts) - target.parent.mkdir(parents=True, exist_ok=True) - try: - target.parent.resolve().relative_to(root) - except ValueError as exc: - raise ValueError("upload path escapes workspace scope") from exc - current = root - for part in path.parts[:-1]: - current = current / part - if current.is_symlink(): - raise ValueError("symlink parents are rejected") - if target.is_symlink(): - raise ValueError("symlink targets are rejected") - return target - - -async def materialize_workspace_file(request: Request) -> JSONResponse: - if denied := _scope_service_authorized(request): - return denied - scope_id = str(request.path_params["scope_id"]) - try: - await asyncio.to_thread(_registry_call, "get", scope_id) - target = await asyncio.to_thread( - _materialize_target, scope_id, str(request.path_params["path"]) - ) - except Exception as exc: - return JSONResponse({"code": "WORKSPACE_PATH_INVALID", "message": str(exc)}, status_code=400) - expected_hash = request.headers.get("x-content-sha256", "").lower() - expected_size = int(request.headers.get("content-length") or 0) - if expected_size > 100 * 1024 * 1024: - return JSONResponse({"code": "WORKSPACE_FILE_TOO_LARGE"}, status_code=413) - temporary = target.with_name(f".{target.name}.{secrets.token_hex(8)}.tmp") - digest = hashlib.sha256() - size = 0 - try: - with temporary.open("xb") as handle: - async for chunk in request.stream(): - size += len(chunk) - if size > 100 * 1024 * 1024: - raise ValueError("workspace file exceeds 100 MiB") - digest.update(chunk) - handle.write(chunk) - handle.flush() - os.fsync(handle.fileno()) - actual_hash = digest.hexdigest() - if expected_hash and not secrets.compare_digest(actual_hash, expected_hash): - raise ValueError("workspace file hash mismatch") - os.replace(temporary, target) - except Exception as exc: - temporary.unlink(missing_ok=True) - return JSONResponse({"code": "WORKSPACE_FILE_INVALID", "message": str(exc)}, status_code=409) - return JSONResponse({"virtual_path": str(request.path_params["path"]), "size": size, "sha256": actual_hash}) - - async def create_recoverable_run(request: Request) -> JSONResponse: """Create a LangGraph Run with a caller-owned deterministic UUID. @@ -421,8 +381,8 @@ app = Starlette( ), Route("/internal/workspace-scopes/provision", provision_workspace_scope, methods=["POST"]), Route("/internal/workspace-scopes/by-thread/{thread_id}", get_workspace_scope, methods=["GET"]), + Route("/internal/workspace-scopes/by-thread/{thread_id}", delete_workspace_scope, methods=["DELETE"]), Route("/internal/workspace-scopes/{scope_id}/runs/reserve", reserve_workspace_run, methods=["POST"]), Route("/internal/workspace-scopes/{scope_id}/runs/{run_request_id}", bind_workspace_run, methods=["PATCH"]), - Route("/internal/workspace-scopes/{scope_id}/files/{path:path}", materialize_workspace_file, methods=["PUT"]), ] ) diff --git a/EvoScientist/middleware/__init__.py b/EvoScientist/middleware/__init__.py index d2aa82e..22b07b4 100644 --- a/EvoScientist/middleware/__init__.py +++ b/EvoScientist/middleware/__init__.py @@ -19,6 +19,7 @@ from .context_editing import ( ) from .context_overflow import ContextOverflowMapperMiddleware from .disable_subagent import DisableSubagentToolMiddleware +from .dynamic_review import DynamicReviewMiddleware from .error_normalization import ErrorNormalizationMiddleware from .memory import ( EvoMemoryMiddleware, @@ -69,6 +70,7 @@ __all__ = [ "ConfigurableModelMiddleware", "ContextOverflowMapperMiddleware", "DisableSubagentToolMiddleware", + "DynamicReviewMiddleware", "ErrorNormalizationMiddleware", "EvoMemoryLifecycleMiddleware", "EvoMemoryMiddleware", diff --git a/EvoScientist/middleware/dynamic_review.py b/EvoScientist/middleware/dynamic_review.py new file mode 100644 index 0000000..075b7b1 --- /dev/null +++ b/EvoScientist/middleware/dynamic_review.py @@ -0,0 +1,217 @@ +"""Run-scoped automatic review for Ai4Sci Web executions.""" + +from __future__ import annotations + +import os +from collections.abc import Mapping +from typing import Annotated, Any, NotRequired + +import httpx +from langchain.agents.middleware import HumanInTheLoopMiddleware +from langchain.agents.middleware.types import AgentState, OmitFromSchema +from langgraph.config import get_config + + +class DynamicReviewState(AgentState[Any]): + _verified_review_mode: NotRequired[ + Annotated[dict[str, Any], OmitFromSchema(input=True, output=True)] + ] + + +class AutoReviewVerificationError(RuntimeError): + """Raised when an automatic review request cannot be verified.""" + + +def _review_context() -> tuple[str, Mapping[str, Any] | None]: + try: + config = get_config() + except RuntimeError: + return "", None + configurable = config.get("configurable") if isinstance(config, Mapping) else None + if not isinstance(configurable, Mapping): + return "", None + run_id = str(configurable.get("ai4sci_run_id") or "") + review = configurable.get("ai4sci_review_mode") + return run_id, review if isinstance(review, Mapping) else None + + +def _manual_state(run_id: str, review: Mapping[str, Any] | None) -> dict[str, Any]: + revision = review.get("review_mode_revision") if review is not None else 0 + return { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": run_id, + "mode": "manual", + "revision": revision if isinstance(revision, int) and revision >= 0 else 0, + } + + +def _request_payload( + run_id: str, review: Mapping[str, Any] +) -> tuple[str, dict[str, str]]: + configured_url = str(review.get("gateway_url") or "").rstrip("/") + gateway_url = os.environ.get("GATEWAY_INTERNAL_URL", "").strip().rstrip("/") + gateway_url = gateway_url or configured_url + payload = { + "run_id": run_id, + "envelope_digest": str(review.get("envelope_digest") or ""), + "envelope_signature": str(review.get("envelope_signature") or ""), + } + if not gateway_url or not all(payload.values()): + raise AutoReviewVerificationError("AUTO_REVIEW_CONTEXT_INVALID") + return gateway_url, payload + + +def _service_headers() -> dict[str, str]: + token = os.environ.get("EVOSCIENTIST_BACKEND_SERVICE_TOKEN", "").strip() + return {"X-Ai4Sci-Service-Token": token} if token else {} + + +def _validated_auto_state( + run_id: str, + review: Mapping[str, Any], + resolved: Mapping[str, Any], +) -> dict[str, Any]: + requested_revision = review.get("review_mode_revision") + if ( + not run_id + or review.get("requested_mode") != "auto" + or not isinstance(requested_revision, int) + or requested_revision < 0 + or resolved.get("protocol") != "resolved-review-mode-v1" + or str(resolved.get("run_id") or "") != run_id + or str(resolved.get("envelope_digest") or "") + != str(review.get("envelope_digest") or "") + or resolved.get("mode") != "auto" + or resolved.get("revision") != requested_revision + ): + raise AutoReviewVerificationError("AUTO_REVIEW_RESPONSE_INVALID") + return { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": run_id, + "mode": "auto", + "revision": requested_revision, + } + + +def _resolve_sync(run_id: str, review: Mapping[str, Any]) -> dict[str, Any]: + gateway_url, payload = _request_payload(run_id, review) + try: + response = httpx.post( + f"{gateway_url}/api/internal/recoverable-runs/review-mode/resolve", + json=payload, + headers=_service_headers(), + timeout=httpx.Timeout(10.0, connect=3.0), + ) + response.raise_for_status() + resolved = response.json() + except (httpx.HTTPError, ValueError, TypeError) as exc: + raise AutoReviewVerificationError("AUTO_REVIEW_VERIFICATION_FAILED") from exc + if not isinstance(resolved, Mapping): + raise AutoReviewVerificationError("AUTO_REVIEW_RESPONSE_INVALID") + return _validated_auto_state(run_id, review, resolved) + + +async def _resolve_async(run_id: str, review: Mapping[str, Any]) -> dict[str, Any]: + gateway_url, payload = _request_payload(run_id, review) + try: + async with httpx.AsyncClient( + timeout=httpx.Timeout(10.0, connect=3.0) + ) as client: + response = await client.post( + f"{gateway_url}/api/internal/recoverable-runs/review-mode/resolve", + json=payload, + headers=_service_headers(), + ) + response.raise_for_status() + resolved = response.json() + except (httpx.HTTPError, ValueError, TypeError) as exc: + raise AutoReviewVerificationError("AUTO_REVIEW_VERIFICATION_FAILED") from exc + if not isinstance(resolved, Mapping): + raise AutoReviewVerificationError("AUTO_REVIEW_RESPONSE_INVALID") + return _validated_auto_state(run_id, review, resolved) + + +class DynamicReviewMiddleware(HumanInTheLoopMiddleware): + """Use HITL for manual Runs and bypass it only for verified automatic Runs.""" + + state_schema = DynamicReviewState + + def before_agent(self, state: DynamicReviewState, runtime: Any) -> dict[str, Any]: + del state, runtime + run_id, review = _review_context() + if review is None or review.get("requested_mode") != "auto": + return {"_verified_review_mode": _manual_state(run_id, review)} + return {"_verified_review_mode": _resolve_sync(run_id, review)} + + async def abefore_agent( + self, state: DynamicReviewState, runtime: Any + ) -> dict[str, Any]: + del state, runtime + run_id, review = _review_context() + if review is None or review.get("requested_mode") != "auto": + return {"_verified_review_mode": _manual_state(run_id, review)} + return {"_verified_review_mode": await _resolve_async(run_id, review)} + + def after_model( + self, state: DynamicReviewState, runtime: Any + ) -> dict[str, Any] | None: + verified = state.get("_verified_review_mode") + if not isinstance(verified, Mapping): + raise AutoReviewVerificationError("REVIEW_MODE_STATE_MISSING") + mode = verified.get("mode") + current_run_id, review = _review_context() + if mode == "auto": + if not current_run_id: + raise AutoReviewVerificationError("REVIEW_MODE_RUN_MISMATCH") + if verified.get("execution_run_id") == current_run_id: + return None + # A LangGraph resume continues at the interrupted node and does not + # re-run before_agent, so execution_run_id still names the parent + # Run that established the verified auto-mode state. Re-verify auto + # approval against the review context the gateway injected for the + # resume child. + if review is None: + # Legacy resume without injected context: trust the parent's + # already-verified auto approval for the remainder of the turn. + return None + if review.get("requested_mode") != "auto": + # The gateway now requires manual approval for this turn. + return super().after_model(state, runtime) + try: + _resolve_sync(current_run_id, review) + except AutoReviewVerificationError: + # Auto approval could not be re-verified; fall back to review. + return super().after_model(state, runtime) + return None + if mode == "manual": + # A LangGraph resume continues at this interrupted node and does not + # re-run before_agent. Reusing a manual state is restrictive and is + # required for the existing resume child Run to complete. + return super().after_model(state, runtime) + raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID") + + async def aafter_model( + self, state: DynamicReviewState, runtime: Any + ) -> dict[str, Any] | None: + verified = state.get("_verified_review_mode") + if not isinstance(verified, Mapping): + raise AutoReviewVerificationError("REVIEW_MODE_STATE_MISSING") + mode = verified.get("mode") + current_run_id, review = _review_context() + if mode == "auto": + if not current_run_id: + raise AutoReviewVerificationError("REVIEW_MODE_RUN_MISMATCH") + if verified.get("execution_run_id") == current_run_id: + return None + if review is None: + return None + if review.get("requested_mode") != "auto": + return super().after_model(state, runtime) + try: + await _resolve_async(current_run_id, review) + except AutoReviewVerificationError: + return super().after_model(state, runtime) + return None + if mode == "manual": + return super().after_model(state, runtime) + raise AutoReviewVerificationError("REVIEW_MODE_STATE_INVALID") diff --git a/EvoScientist/native_sandbox.py b/EvoScientist/native_sandbox.py new file mode 100644 index 0000000..6b677c5 --- /dev/null +++ b/EvoScientist/native_sandbox.py @@ -0,0 +1,743 @@ +"""Native OS process sandbox for Web conversation workspaces.""" + +from __future__ import annotations + +import asyncio +import json +import os +import platform +import selectors +import shlex +import shutil +import signal +import socket +import subprocess +import sys +import tempfile +import threading +import time +import uuid +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from deepagents.backends.protocol import ExecuteResponse, SandboxBackendProtocol + +from .workspace_files import ScopedFilesystemBackend + +_PINNED_SRT_VERSION = "0.0.73" +_DEFAULT_TIMEOUT = 300 +_MAX_TIMEOUT = 3600 +_DEFAULT_OUTPUT_LIMIT = 100_000 +_DEFAULT_FILE_SIZE_LIMIT = 100 * 1024 * 1024 +_INIT_ERROR_MARKERS = ( + "could not load settings", + "failed to initialize", + "sandbox initialization failed", + "failed to generate sandbox", + "sandbox-exec: sandbox_apply", +) + + +class NativeSandboxUnavailable(RuntimeError): + """Raised when the required native sandbox cannot enforce its policy.""" + + +@dataclass(frozen=True, slots=True) +class NativeSandboxInstallation: + srt: Path + package_root: Path + path_env: str + system_read_paths: tuple[str, ...] + seccomp_helper: Path | None = None + bwrap: Path | None = None + socat: Path | None = None + + +def _repo_root() -> Path: + return Path(__file__).resolve().parents[1] + + +def _runtime_root() -> Path: + configured = os.getenv("EVOSCIENTIST_NATIVE_SANDBOX_ROOT", "").strip() + return ( + Path(configured).expanduser().resolve() + if configured + else (_repo_root() / "runtime" / "native-sandbox").resolve() + ) + + +def _positive_int(name: str, default: int) -> int: + raw = os.getenv(name, str(default)).strip() + try: + value = int(raw) + except ValueError as exc: + raise NativeSandboxUnavailable(f"{name} must be an integer") from exc + if value < 1: + raise NativeSandboxUnavailable(f"{name} must be positive") + return value + + +def _tool_path() -> str: + candidates = ( + [ + "/opt/homebrew/bin", + "/usr/local/bin", + "/usr/bin", + "/bin", + "/usr/sbin", + "/sbin", + ] + if sys.platform == "darwin" + else [ + "/usr/local/bin", + "/usr/bin", + "/bin", + "/usr/sbin", + "/sbin", + "/opt/evoscientist-tools/bin", + ] + ) + return os.pathsep.join(path for path in candidates if Path(path).is_dir()) + + +def _which_required(command: str, path_env: str) -> Path: + found = shutil.which(command, path=path_env) + if not found: + raise NativeSandboxUnavailable(f"native sandbox requires {command}") + return Path(found).resolve() + + +def _node_version(node: Path) -> tuple[int, int, int]: + try: + result = subprocess.run( + [str(node), "--version"], + check=False, + capture_output=True, + text=True, + timeout=5, + env={"PATH": str(node.parent)}, + ) + except (OSError, subprocess.TimeoutExpired) as exc: + raise NativeSandboxUnavailable("native sandbox cannot start node") from exc + value = result.stdout.strip().removeprefix("v") + try: + parts = tuple(int(part) for part in value.split(".")[:3]) + except ValueError as exc: + raise NativeSandboxUnavailable( + "native sandbox cannot determine node version" + ) from exc + if len(parts) != 3: + raise NativeSandboxUnavailable("native sandbox cannot determine node version") + return parts + + +def _system_read_paths() -> tuple[str, ...]: + candidates = ( + ( + "/System", + "/usr", + "/bin", + "/sbin", + "/opt/homebrew", + "/usr/local", + "/private/etc/ssl", + "/private/var/select/sh", + "/dev/null", + "/dev/zero", + "/dev/urandom", + ) + if sys.platform == "darwin" + else ( + "/usr", + "/bin", + "/sbin", + "/lib", + "/lib64", + "/opt/evoscientist-tools", + "/etc/ssl", + "/etc/ld.so.cache", + "/dev/null", + "/dev/zero", + "/dev/urandom", + ) + ) + return tuple(path for path in candidates if Path(path).exists()) + + +def _assert_install_contract() -> NativeSandboxInstallation: + if sys.platform not in {"darwin", "linux"}: + raise NativeSandboxUnavailable( + f"native sandbox does not support {platform.system() or sys.platform}" + ) + + root = _runtime_root() + package_root = root / "node_modules" / "@anthropic-ai" / "sandbox-runtime" + package_json = package_root / "package.json" + srt = root / "node_modules" / ".bin" / "srt" + try: + metadata = json.loads(package_json.read_text(encoding="utf-8")) + except (OSError, ValueError) as exc: + raise NativeSandboxUnavailable( + "pinned native sandbox dependency is not installed; run npm ci --omit=dev --prefix runtime/native-sandbox" + ) from exc + if metadata.get("version") != _PINNED_SRT_VERSION: + raise NativeSandboxUnavailable( + f"native sandbox requires @anthropic-ai/sandbox-runtime {_PINNED_SRT_VERSION}" + ) + if not srt.is_file() or not os.access(srt, os.X_OK): + raise NativeSandboxUnavailable("pinned srt executable is missing") + + path_env = _tool_path() + node = _which_required("node", path_env) + if _node_version(node) < (20, 11, 0): + raise NativeSandboxUnavailable("native sandbox requires node >= 20.11.0") + for tool in ("bash", "rg", "python3", "pandoc"): + _which_required(tool, path_env) + + seccomp_helper: Path | None = None + bwrap: Path | None = None + socat: Path | None = None + if sys.platform == "darwin": + sandbox_exec = Path("/usr/bin/sandbox-exec") + if not sandbox_exec.is_file() or not os.access(sandbox_exec, os.X_OK): + raise NativeSandboxUnavailable("native sandbox requires macOS sandbox-exec") + else: + bwrap = _which_required("bwrap", path_env) + socat = _which_required("socat", path_env) + arch = platform.machine().lower() + vendor_arch = { + "x86_64": "x64", + "amd64": "x64", + "arm64": "arm64", + "aarch64": "arm64", + }.get(arch) + if vendor_arch is None: + raise NativeSandboxUnavailable( + f"native sandbox does not support Linux architecture {arch}" + ) + candidate = package_root / "vendor" / "seccomp" / vendor_arch / "apply-seccomp" + try: + seccomp_helper = candidate.resolve(strict=True) + seccomp_helper.relative_to(package_root.resolve(strict=True)) + except (OSError, ValueError) as exc: + raise NativeSandboxUnavailable( + "native sandbox seccomp helper is invalid" + ) from exc + if not seccomp_helper.is_file() or not os.access(seccomp_helper, os.X_OK): + raise NativeSandboxUnavailable( + "native sandbox seccomp helper is not executable" + ) + + return NativeSandboxInstallation( + srt=srt.resolve(), + package_root=package_root.resolve(), + path_env=path_env, + system_read_paths=_system_read_paths(), + seccomp_helper=seccomp_helper, + bwrap=bwrap, + socat=socat, + ) + + +def _sandbox_settings( + installation: NativeSandboxInstallation, + files_dir: Path, + command_tmp: Path, +) -> dict[str, Any]: + allow_read = [str(files_dir), str(command_tmp), *installation.system_read_paths] + settings: dict[str, Any] = { + "network": { + "allowedDomains": [], + "deniedDomains": [], + "strictAllowlist": True, + "allowUnixSockets": [], + "allowAllUnixSockets": False, + "allowLocalBinding": False, + }, + "filesystem": { + "denyRead": ["/"], + "allowRead": list(dict.fromkeys(allow_read)), + "allowWrite": [str(files_dir), str(command_tmp), "/dev/null"], + # srt adds these shared compatibility paths even when callers do + # not request them. Explicit deny wins over that built-in allow. + "denyWrite": [ + "/tmp/claude", + "/private/tmp/claude", + "/dev/tty", + "/dev/dtracehelper", + "/dev/autofs_nowait", + ], + }, + "enableWeakerNestedSandbox": False, + "enableWeakerNetworkIsolation": False, + "allowAppleEvents": False, + "allowPty": False, + "ripgrep": {"command": "rg"}, + } + if sys.platform == "linux": + if ( + installation.seccomp_helper is None + or installation.bwrap is None + or installation.socat is None + ): + raise NativeSandboxUnavailable( + "native sandbox Linux helpers are incomplete" + ) + settings["filesystem"]["allowRead"].append(str(installation.seccomp_helper)) + settings["seccomp"] = {"applyPath": str(installation.seccomp_helper)} + settings["bwrapPath"] = str(installation.bwrap) + settings["socatPath"] = str(installation.socat) + return settings + + +def _clean_environment( + installation: NativeSandboxInstallation, command_tmp: Path +) -> dict[str, str]: + locale = "en_US.UTF-8" if sys.platform == "darwin" else "C.UTF-8" + return { + "PATH": installation.path_env, + "HOME": str(command_tmp / "home"), + "TMPDIR": str(command_tmp / "tmp"), + "WORKSPACE": ".", + "LANG": locale, + "LC_ALL": locale, + } + + +def _terminate_process_group(process: subprocess.Popen[bytes]) -> None: + try: + os.killpg(process.pid, signal.SIGTERM) + except (ProcessLookupError, PermissionError): + return + try: + process.wait(timeout=0.5) + except subprocess.TimeoutExpired: + pass + try: + os.killpg(process.pid, signal.SIGKILL) + except (ProcessLookupError, PermissionError): + pass + + +def _kill_remaining_process_group(process: subprocess.Popen[bytes]) -> None: + """Remove descendants left behind after the srt parent has exited.""" + + try: + os.killpg(process.pid, signal.SIGKILL) + except (ProcessLookupError, PermissionError): + pass + + +def _collect_process( + process: subprocess.Popen[bytes], + *, + timeout: int, + output_limit: int, + cancel_event: threading.Event | None = None, +) -> tuple[bytes, int, bool, bool, bool]: + selector = selectors.DefaultSelector() + streams = [ + stream for stream in (process.stdout, process.stderr) if stream is not None + ] + for stream in streams: + os.set_blocking(stream.fileno(), False) + selector.register(stream, selectors.EVENT_READ) + + chunks: list[bytes] = [] + retained = 0 + truncated = False + timed_out = False + cancelled = False + deadline = time.monotonic() + timeout + exited_at: float | None = None + try: + while selector.get_map(): + now = time.monotonic() + if not timed_out and now >= deadline: + timed_out = True + _terminate_process_group(process) + if ( + not timed_out + and not cancelled + and cancel_event is not None + and cancel_event.is_set() + ): + cancelled = True + _terminate_process_group(process) + events = selector.select(0.1) + for key, _ in events: + try: + data = os.read(key.fileobj.fileno(), 65_536) + except BlockingIOError: + continue + if not data: + selector.unregister(key.fileobj) + continue + available = max(0, output_limit - retained) + if available: + kept = data[:available] + chunks.append(kept) + retained += len(kept) + if len(data) > available: + truncated = True + + if process.poll() is not None: + exited_at = exited_at or time.monotonic() + # A detached descendant must not keep inherited pipes open forever. + if not events and time.monotonic() - exited_at > 0.25: + for key in list(selector.get_map().values()): + selector.unregister(key.fileobj) + break + if process.poll() is None: + _terminate_process_group(process) + return_code = process.wait(timeout=1) + except subprocess.TimeoutExpired: + _terminate_process_group(process) + return_code = process.wait(timeout=1) + finally: + selector.close() + for stream in streams: + stream.close() + + _kill_remaining_process_group(process) + + if timed_out: + return_code = 124 + elif cancelled: + return_code = 130 + elif return_code < 0: + return_code = 128 + abs(return_code) + return b"".join(chunks), return_code, truncated, timed_out, cancelled + + +class NativeSandboxExecutor: + """Execute one shell command with a scope-specific mandatory OS policy.""" + + def __init__( + self, + files_dir: str | Path, + runtime_dir: str | Path, + *, + timeout: int = _DEFAULT_TIMEOUT, + installation: NativeSandboxInstallation | None = None, + ) -> None: + self.files_dir = Path(files_dir).resolve(strict=True) + self.runtime_dir = Path(runtime_dir).resolve(strict=True) + if self.files_dir == self.runtime_dir or self.runtime_dir.is_relative_to( + self.files_dir + ): + raise NativeSandboxUnavailable( + "sandbox control directory must be outside workspace files" + ) + self.timeout = max(1, min(int(timeout), _MAX_TIMEOUT)) + self._installation = installation + + def execute( + self, + command: str, + *, + timeout: int | None = None, + skip_readiness_check: bool = False, + cancel_event: threading.Event | None = None, + ) -> ExecuteResponse: + if not skip_readiness_check: + assert_native_sandbox_ready() + installation = self._installation or _assert_install_contract() + effective_timeout = max(1, min(timeout or self.timeout, _MAX_TIMEOUT)) + output_limit = _positive_int( + "EVOSCIENTIST_SANDBOX_MAX_OUTPUT_BYTES", _DEFAULT_OUTPUT_LIMIT + ) + file_size_limit = _positive_int( + "EVOSCIENTIST_SANDBOX_FILE_SIZE_BYTES", _DEFAULT_FILE_SIZE_LIMIT + ) + + execution_id = uuid.uuid4().hex + control_dir = self.runtime_dir / "control" / execution_id + command_tmp = self.runtime_dir / "tmp" / execution_id + settings_path = control_dir / "settings.json" + try: + control_dir.mkdir(mode=0o700, parents=True) + (command_tmp / "home").mkdir(mode=0o700, parents=True) + (command_tmp / "tmp").mkdir(mode=0o700) + settings = _sandbox_settings(installation, self.files_dir, command_tmp) + settings_path.write_text( + json.dumps(settings, ensure_ascii=True, separators=(",", ":")), + encoding="utf-8", + ) + settings_path.chmod(0o600) + + entrypoint = command_tmp / "entrypoint.sh" + entrypoint.write_text( + "#!/bin/sh\n" + "exec /usr/bin/env -i " + 'PATH="$PATH" HOME="$HOME" TMPDIR="$TMPDIR" ' + 'WORKSPACE="$WORKSPACE" LANG="$LANG" LC_ALL="$LC_ALL" ' + '/bin/sh -c "$1"\n', + encoding="ascii", + ) + entrypoint.chmod(0o700) + + helper = Path(__file__).with_name("sandbox_exec_helper.py") + invocation = [ + sys.executable, + str(helper), + str(file_size_limit), + "--", + str(installation.srt), + "--settings", + str(settings_path), + str(entrypoint), + command, + ] + environment = _clean_environment(installation, command_tmp) + # srt reads this trusted compatibility variable while generating + # its wrapper. entrypoint.sh removes it before the user command. + environment["CLAUDE_CODE_TMPDIR"] = str(command_tmp / "tmp") + try: + process = subprocess.Popen( + invocation, + cwd=self.files_dir, + env=environment, + stdin=subprocess.DEVNULL, + stdout=subprocess.PIPE, + stderr=subprocess.PIPE, + start_new_session=True, + close_fds=True, + ) + except OSError as exc: + raise NativeSandboxUnavailable( + "native sandbox process could not start" + ) from exc + + raw, return_code, truncated, timed_out, _cancelled = _collect_process( + process, + timeout=effective_timeout, + output_limit=output_limit, + cancel_event=cancel_event, + ) + output = raw.decode("utf-8", errors="replace") + output = output.replace(str(control_dir), "") + lowered = output.lower() + if any(marker in lowered for marker in _INIT_ERROR_MARKERS): + return ExecuteResponse( + output="Native sandbox failed to initialize.", + exit_code=125, + truncated=False, + ) + if timed_out: + suffix = f"Command timed out after {effective_timeout} seconds." + output = f"{output.rstrip()}\n{suffix}" if output else suffix + return ExecuteResponse( + output=output, + exit_code=return_code, + truncated=truncated, + ) + finally: + shutil.rmtree(control_dir, ignore_errors=True) + shutil.rmtree(command_tmp, ignore_errors=True) + + +class NativeWorkspaceBackend(ScopedFilesystemBackend, SandboxBackendProtocol): + """DeepAgents backend sharing one scope between file tools and native shell.""" + + def __init__(self, root_dir: Path, runtime_dir: Path, *, timeout: int) -> None: + super().__init__(root_dir) + self._executor = NativeSandboxExecutor(root_dir, runtime_dir, timeout=timeout) + self._sandbox_id = f"native-scope-{uuid.uuid4().hex[:8]}" + + @property + def id(self) -> str: + return self._sandbox_id + + @staticmethod + def _command_error(command: str) -> ExecuteResponse | None: + from .backends import validate_command + + if ( + not isinstance(command, str) + or not command + or "\x00" in command + or len(command) > 100_000 + ): + return ExecuteResponse( + output="Command blocked: invalid command length.", + exit_code=1, + truncated=False, + ) + error = validate_command(command, dangerous=True) + if error: + return ExecuteResponse(output=error, exit_code=1, truncated=False) + return None + + def _execute( + self, + command: str, + *, + timeout: int | None, + cancel_event: threading.Event | None = None, + ) -> ExecuteResponse: + error = self._command_error(command) + if error is not None: + return error + try: + return self._executor.execute( + command, timeout=timeout, cancel_event=cancel_event + ) + except (NativeSandboxUnavailable, OSError): + return ExecuteResponse( + output="Native sandbox is unavailable; command execution was refused.", + exit_code=125, + truncated=False, + ) + + def execute(self, command: str, *, timeout: int | None = None) -> ExecuteResponse: + return self._execute(command, timeout=timeout) + + async def aexecute( + self, command: str, *, timeout: int | None = None + ) -> ExecuteResponse: + cancel_event = threading.Event() + task = asyncio.create_task( + asyncio.to_thread( + self._execute, + command, + timeout=timeout, + cancel_event=cancel_event, + ) + ) + try: + return await asyncio.shield(task) + except asyncio.CancelledError: + cancel_event.set() + try: + await asyncio.wait_for(asyncio.shield(task), timeout=2) + except (TimeoutError, asyncio.CancelledError): + pass + raise + + +_READY_CONDITION = threading.Condition() +_READY_STATE = "unchecked" +_READY_ERROR: str | None = None + + +def _run_preflight(installation: NativeSandboxInstallation) -> None: + # AF_UNIX paths are limited to roughly 100 bytes on both target platforms; + # a deployment workspace path can already exceed that before the filename. + with tempfile.TemporaryDirectory(prefix="evosci-srt-") as temporary: + root = Path(temporary) + files_dir = root / "files" + runtime_dir = root / "runtime" + files_dir.mkdir(mode=0o700) + runtime_dir.mkdir(mode=0o700) + (files_dir / "probe.txt").write_text("allowed", encoding="utf-8") + outside = root / "outside-secret.txt" + outside.write_text("secret", encoding="utf-8") + control_secret = runtime_dir / "control-secret.txt" + control_secret.write_text("control", encoding="utf-8") + + tcp = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + unix_server: socket.socket | None = None + unix_path = files_dir / "blocked.sock" + try: + tcp.bind(("127.0.0.1", 0)) + tcp.listen(1) + port = int(tcp.getsockname()[1]) + if hasattr(socket, "AF_UNIX"): + unix_server = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) + unix_server.bind(str(unix_path)) + unix_server.listen(1) + + python_probe = ( + "import os,socket,sys;" + "assert 'OPENAI_API_KEY' not in os.environ;" + f"t=socket.socket(); tcp=t.connect_ex(('127.0.0.1',{port})); t.close();" + f"u=socket.socket(socket.AF_UNIX); unix=u.connect_ex({str(unix_path)!r}); u.close();" + "sys.exit(0 if tcp != 0 and unix != 0 else 9)" + ) + command = " && ".join( + ( + 'test "$(cat probe.txt)" = allowed', + "printf written > written.txt", + 'printf temporary > "$TMPDIR/probe.tmp"', + "pandoc --version >/dev/null", + f"python3 -c {shlex.quote(python_probe)}", + f"! cat {shlex.quote(str(outside))} >/dev/null 2>&1", + f"! cat {shlex.quote(str(control_secret))} >/dev/null 2>&1", + f"! sh -c {shlex.quote('printf escaped > ' + str(outside))} 2>/dev/null", + ) + ) + result = NativeSandboxExecutor( + files_dir, + runtime_dir, + timeout=20, + installation=installation, + ).execute(command, skip_readiness_check=True) + finally: + tcp.close() + if unix_server is not None: + unix_server.close() + try: + unix_path.unlink() + except FileNotFoundError: + pass + + if result.exit_code != 0: + detail = result.output.replace(str(root), "").strip()[:500] + raise NativeSandboxUnavailable( + f"native sandbox preflight failed (exit {result.exit_code})" + + (f": {detail}" if detail else "") + ) + if (files_dir / "written.txt").read_text(encoding="utf-8") != "written": + raise NativeSandboxUnavailable( + "native sandbox preflight could not write workspace" + ) + if outside.read_text(encoding="utf-8") != "secret": + raise NativeSandboxUnavailable("native sandbox preflight escaped workspace") + + +def ensure_native_sandbox_ready() -> None: + """Run the real isolation preflight once and cache the process-level result.""" + + global _READY_ERROR, _READY_STATE + with _READY_CONDITION: + while _READY_STATE == "checking": + _READY_CONDITION.wait() + if _READY_STATE == "ready": + return + if _READY_STATE == "failed": + raise NativeSandboxUnavailable( + _READY_ERROR or "native sandbox preflight previously failed" + ) + _READY_STATE = "checking" + + try: + installation = _assert_install_contract() + _run_preflight(installation) + except Exception as exc: + message = str(exc) or "native sandbox preflight failed" + with _READY_CONDITION: + _READY_ERROR = message + _READY_STATE = "failed" + _READY_CONDITION.notify_all() + if isinstance(exc, NativeSandboxUnavailable): + raise + raise NativeSandboxUnavailable(message) from exc + else: + with _READY_CONDITION: + _READY_ERROR = None + _READY_STATE = "ready" + _READY_CONDITION.notify_all() + + +def assert_native_sandbox_ready() -> None: + ensure_native_sandbox_ready() + + +def _reset_native_sandbox_readiness_for_tests() -> None: + global _READY_ERROR, _READY_STATE + with _READY_CONDITION: + _READY_ERROR = None + _READY_STATE = "unchecked" + _READY_CONDITION.notify_all() diff --git a/EvoScientist/prompts.py b/EvoScientist/prompts.py index 726d657..8efd417 100644 --- a/EvoScientist/prompts.py +++ b/EvoScientist/prompts.py @@ -289,6 +289,22 @@ _SHELL_GUIDELINES_DANGEROUS_FOOTER = """ **Still blocked even here**: privileged/system commands (`sudo`, `chmod`, `chown`, `mkfs`, `dd`, `shutdown`, `reboot`) and `rm -rf /` are rejected regardless of mode.""" +_SHELL_GUIDELINES_NATIVE_WEB = """# Shell Execution Guidelines + +The `execute` tool runs each command in the current conversation's native OS sandbox. + +- The command starts in the conversation workspace. Use relative paths such as `uploads/input.docx`; `$WORKSPACE` is `.` and refers to the same directory that file tools expose as `/workspace`. +- Do not use `/workspace` inside shell commands. It is a file-tool virtual path, not a mounted shell path. +- Commands may read and write only this conversation workspace and their private temporary directory. System tools are read-only, network and Unix sockets are blocked, and Runtime credentials are absent. +- Commands run in the foreground. Do not append `&`, detach children, or use background-process tools in Web mode. +- Commands default to a 300s timeout and 100 KB retained output. For a known longer command, pass `timeout` up to 3600 seconds. + +For document processing, use the allowed installed tools directly, for example: +```bash +pandoc "$WORKSPACE/uploads/input.docx" -o "$WORKSPACE/results/report.html" +python3 "$WORKSPACE/scripts/analyze.py" "$WORKSPACE/uploads/input.docx" +```""" + def _build_shell_guidelines(*, dangerous: bool = False, cwd: str | None = None) -> str: """Assemble the shell guidelines from the shared core + per-mode header/footer.""" @@ -401,6 +417,7 @@ def get_system_prompt( *, dangerous: bool = False, cwd: str | None = None, + native_web_sandbox: bool = False, ) -> str: """Generate the complete static system prompt. @@ -425,15 +442,18 @@ def get_system_prompt( (no virtual workspace) instead of the sandboxed default. cwd: Real absolute working directory shown to the agent in dangerous mode. Falls back to ``.`` when not provided. + native_web_sandbox: Use foreground-only, relative-path guidance for + the Web native process sandbox. Returns: Combined static system prompt string. """ - shell_guidelines = ( - _build_shell_guidelines(dangerous=True, cwd=cwd) - if dangerous - else SHELL_GUIDELINES - ) + if dangerous: + shell_guidelines = _build_shell_guidelines(dangerous=True, cwd=cwd) + elif native_web_sandbox: + shell_guidelines = _SHELL_GUIDELINES_NATIVE_WEB + else: + shell_guidelines = SHELL_GUIDELINES sections = [ EVOSCIENTIST_IDENTITY, EXPERIMENT_WORKFLOW, diff --git a/EvoScientist/sandbox_exec_helper.py b/EvoScientist/sandbox_exec_helper.py new file mode 100644 index 0000000..4262ee8 --- /dev/null +++ b/EvoScientist/sandbox_exec_helper.py @@ -0,0 +1,28 @@ +"""Apply process resource limits before entering the native sandbox runtime.""" + +from __future__ import annotations + +import os +import resource +import sys + + +def main() -> None: + if len(sys.argv) < 4 or sys.argv[2] != "--": + raise SystemExit("usage: sandbox_exec_helper.py FILE_SIZE -- COMMAND [ARG ...]") + + try: + file_size = int(sys.argv[1]) + except ValueError as exc: + raise SystemExit("FILE_SIZE must be an integer") from exc + if file_size < 1: + raise SystemExit("FILE_SIZE must be positive") + + resource.setrlimit(resource.RLIMIT_FSIZE, (file_size, file_size)) + os.umask(0o077) + command = sys.argv[3:] + os.execvpe(command[0], command, os.environ) + + +if __name__ == "__main__": + main() diff --git a/EvoScientist/scope_registry.py b/EvoScientist/scope_registry.py index 7b2117a..18f614e 100644 --- a/EvoScientist/scope_registry.py +++ b/EvoScientist/scope_registry.py @@ -492,6 +492,32 @@ class ScopeRegistry: if existing is not None: if scope_id and str(existing["scope_id"]) != requested_scope: raise ScopeConflictError("thread already belongs to another scope") + if str(existing["state"]) == "deleted": + conn.execute( + """ + UPDATE scopes + SET state = ?, revision = revision + 1, updated_at = ?, deleted_at = NULL + WHERE deployment_id = ? AND scope_id = ? + """, + (state, now, deployment_id, str(existing["scope_id"])), + ) + conn.execute( + """ + UPDATE scope_owners + SET state = 'active', updated_at = ?, terminal_at = NULL + WHERE deployment_id = ? AND scope_id = ? + AND owner_type = 'primary_thread' + """, + (now, deployment_id, str(existing["scope_id"])), + ) + existing = conn.execute( + """ + SELECT * FROM scopes + WHERE deployment_id = ? AND primary_thread_id = ? + """, + (deployment_id, primary_thread_id), + ).fetchone() + assert existing is not None primary_owner = self._primary_owner( conn, deployment_id, str(existing["scope_id"]) ) diff --git a/EvoScientist/stream/emitter.py b/EvoScientist/stream/emitter.py index 714d23c..d9d3201 100644 --- a/EvoScientist/stream/emitter.py +++ b/EvoScientist/stream/emitter.py @@ -9,7 +9,6 @@ from typing import Any STREAM_PROTOCOL_CAPABILITIES = frozenset( { - "task_snapshot_v1", "complete_tool_call_v1", "correlated_tool_call_id_v1", "final_invalid_tool_call_v1", @@ -167,14 +166,6 @@ class StreamEventEmitter: }, ) - @staticmethod - def task_snapshot(source: str, items: list[dict[str, Any]]) -> StreamEvent: - """Emit the complete root-agent task state without product-specific IDs.""" - return StreamEvent( - "task_snapshot", - {"type": "task_snapshot", "source": source, "items": items}, - ) - @staticmethod def interrupt( interrupt_id: str, diff --git a/EvoScientist/stream/events.py b/EvoScientist/stream/events.py index 410505e..2efd7df 100644 --- a/EvoScientist/stream/events.py +++ b/EvoScientist/stream/events.py @@ -272,7 +272,6 @@ class _V3EventProcessor: ] = {} self._emitted_interrupts: set[str] = set() self._pending_invalid_tool_calls: dict[str, tuple[str, str]] = {} - self._last_task_snapshot: tuple[tuple[str, str], ...] | None = None self._selector = _ToolSelectionSuppressor(emitter) @staticmethod @@ -300,9 +299,7 @@ class _V3EventProcessor: if method == "tools": return self._process_tool_event(namespace, _event_data(event), subagent) if method == "updates": - return self._process_update_event( - _event_data(event), namespace=namespace, source="update" - ) + return self._process_update_event(_event_data(event), namespace=namespace) if method == "values": events: list[dict[str, Any]] = [] params = event.get("params") or {} @@ -312,12 +309,11 @@ class _V3EventProcessor: self._process_update_event( {"__interrupt__": interrupts}, namespace=namespace, - source="values", ) ) events.extend( self._process_update_event( - _event_data(event), namespace=namespace, source="values" + _event_data(event), namespace=namespace ) ) if self._process_value_message_snapshots and not namespace: @@ -699,66 +695,17 @@ class _V3EventProcessor: return [] - @staticmethod - def _normalize_task_items(value: object) -> list[dict[str, str]] | None: - if not isinstance(value, list): - return None - aliases = { - "todo": "pending", - "pending": "pending", - "active": "in_progress", - "in-progress": "in_progress", - "in_progress": "in_progress", - "done": "completed", - "completed": "completed", - } - items: list[dict[str, str]] = [] - for raw in value: - raw_map = _as_raw_map(raw) - if raw_map is None: - continue - content = str(raw_map.get("content") or raw_map.get("task") or "").strip() - if not content: - continue - status = aliases.get(str(raw_map.get("status") or "pending").lower()) - if status is None: - continue - items.append({"content": content, "status": status}) - return items - - @classmethod - def _find_task_items(cls, data: object) -> list[dict[str, str]] | None: - data_map = _as_raw_map(data) - if data_map is None: - return None - if "todos" in data_map: - return cls._normalize_task_items(data_map["todos"]) - for value in data_map.values(): - nested = _as_raw_map(value) - if nested is not None and "todos" in nested: - return cls._normalize_task_items(nested["todos"]) - return None - def _process_update_event( self, data: object, *, namespace: tuple[str, ...] = (), - source: str = "update", ) -> list[dict[str, Any]]: events: list[dict[str, Any]] = [] data_map = _as_raw_map(data) if data_map is not None and "__interrupt__" in data_map: events.extend(self._process_interrupts(data_map["__interrupt__"])) - if not namespace: - items = self._find_task_items(data) - if items is not None: - signature = tuple((item["content"], item["status"]) for item in items) - if signature != self._last_task_snapshot: - self._last_task_snapshot = signature - events.append(self.emitter.task_snapshot(source, items).data) - summarization_event = _find_summarization_event_payload(data) if summarization_event and not self._summarization_in_progress: signature = _summarization_event_signature(summarization_event) diff --git a/EvoScientist/workspace_files.py b/EvoScientist/workspace_files.py new file mode 100644 index 0000000..9a9b0dc --- /dev/null +++ b/EvoScientist/workspace_files.py @@ -0,0 +1,576 @@ +"""Descriptor-rooted file access for Web conversation workspaces.""" + +from __future__ import annotations + +import base64 +import errno +import os +import stat +import uuid +from collections.abc import Iterator +from contextlib import contextmanager +from dataclasses import dataclass +from datetime import UTC, datetime +from pathlib import Path +from typing import BinaryIO + +import wcmatch.glob as wcglob +from deepagents.backends.protocol import ( + FILE_NOT_FOUND, + INVALID_PATH, + IS_DIRECTORY, + PERMISSION_DENIED, + BackendProtocol, + EditResult, + FileDownloadResponse, + FileUploadResponse, + GlobResult, + GrepResult, + LsResult, + ReadResult, + WriteResult, +) +from deepagents.backends.utils import check_empty_content, perform_string_replacement + +WORKSPACE_PREFIX = "/workspace" +_GLOB_FLAGS = wcglob.BRACE | wcglob.GLOBSTAR +_BINARY_EXTENSIONS = { + ".7z", + ".aac", + ".arrow", + ".avi", + ".bin", + ".bmp", + ".bz2", + ".db", + ".doc", + ".docx", + ".feather", + ".gif", + ".gz", + ".h5", + ".hdf5", + ".ico", + ".jpeg", + ".jpg", + ".m4a", + ".mkv", + ".mov", + ".mp3", + ".mp4", + ".npy", + ".npz", + ".ogg", + ".parquet", + ".pdf", + ".pickle", + ".pkl", + ".png", + ".ppt", + ".pptx", + ".pyc", + ".rar", + ".sqlite", + ".sqlite3", + ".tar", + ".tif", + ".tiff", + ".wav", + ".wasm", + ".webm", + ".webp", + ".xls", + ".xlsb", + ".xlsx", + ".xz", + ".zip", +} + + +class WorkspacePathError(ValueError): + """Raised when a virtual workspace path is invalid or unsafe.""" + + +@dataclass(frozen=True, slots=True) +class WorkspaceEntry: + virtual_path: str + is_dir: bool + size: int + modified_at: str + + +def normalize_workspace_path(path: str, *, allow_root: bool = True) -> tuple[str, ...]: + """Return safe path components for one canonical ``/workspace`` path.""" + + if not isinstance(path, str) or not path or "\x00" in path or "\\" in path: + raise WorkspacePathError("invalid workspace path") + if path == WORKSPACE_PREFIX: + if allow_root: + return () + raise WorkspacePathError("workspace root is not a file") + prefix = WORKSPACE_PREFIX + "/" + if not path.startswith(prefix): + raise WorkspacePathError("path must be under /workspace") + relative = path[len(prefix) :] + parts = tuple(relative.split("/")) + if not parts or any(part in {"", ".", ".."} for part in parts): + raise WorkspacePathError("invalid workspace path") + return parts + + +def _virtual(parts: tuple[str, ...]) -> str: + return WORKSPACE_PREFIX + ("/" + "/".join(parts) if parts else "") + + +def _modified_at(value: os.stat_result) -> str: + return datetime.fromtimestamp(value.st_mtime, UTC).isoformat() + + +def _open_flags(*, directory: bool = False) -> int: + flags = os.O_RDONLY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0) + if directory: + flags |= getattr(os, "O_DIRECTORY", 0) + return flags + + +class RootedWorkspace: + """Open and mutate files relative to one trusted directory descriptor.""" + + def __init__(self, root_dir: str | Path, *, max_search_file_bytes: int = 10 * 1024 * 1024): + self.root_dir = Path(root_dir) + self.max_search_file_bytes = max_search_file_bytes + + def _open_root(self) -> int: + return os.open(self.root_dir, _open_flags(directory=True)) + + def _open_dir_parts(self, parts: tuple[str, ...], *, create: bool = False) -> int: + current = self._open_root() + try: + for part in parts: + if create: + try: + os.mkdir(part, mode=0o700, dir_fd=current) + except FileExistsError: + pass + child = os.open(part, _open_flags(directory=True), dir_fd=current) + os.close(current) + current = child + return current + except Exception: + os.close(current) + raise + + @contextmanager + def _parent(self, path: str, *, create: bool = False) -> Iterator[tuple[int, str]]: + parts = normalize_workspace_path(path, allow_root=False) + parent = self._open_dir_parts(parts[:-1], create=create) + try: + yield parent, parts[-1] + finally: + os.close(parent) + + def open_binary(self, path: str) -> BinaryIO: + """Return an already-verified regular-file handle; caller closes it.""" + + with self._parent(path) as (parent, name): + fd = os.open(name, _open_flags(), dir_fd=parent) + try: + if not stat.S_ISREG(os.fstat(fd).st_mode): + raise WorkspacePathError("workspace path is not a regular file") + return os.fdopen(fd, "rb") + except Exception: + os.close(fd) + raise + + def _write_temporary(self, parent: int, data: bytes) -> str: + name = f".workspace-{uuid.uuid4().hex}.tmp" + flags = ( + os.O_WRONLY + | os.O_CREAT + | os.O_EXCL + | getattr(os, "O_CLOEXEC", 0) + | getattr(os, "O_NOFOLLOW", 0) + ) + fd = os.open(name, flags, 0o600, dir_fd=parent) + try: + with os.fdopen(fd, "wb", closefd=False) as handle: + handle.write(data) + handle.flush() + os.fsync(handle.fileno()) + finally: + os.close(fd) + return name + + def write_new(self, path: str, data: bytes) -> None: + """Atomically publish a new file and fail when it already exists.""" + + with self._parent(path, create=True) as (parent, name): + temporary = self._write_temporary(parent, data) + try: + os.link( + temporary, + name, + src_dir_fd=parent, + dst_dir_fd=parent, + follow_symlinks=False, + ) + finally: + os.unlink(temporary, dir_fd=parent) + + def replace_file(self, path: str, data: bytes, *, require_existing: bool = False) -> None: + """Atomically create or replace a regular file without following links.""" + + with self._parent(path, create=True) as (parent, name): + try: + current = os.stat(name, dir_fd=parent, follow_symlinks=False) + except FileNotFoundError: + if require_existing: + raise + else: + if not stat.S_ISREG(current.st_mode): + raise WorkspacePathError("workspace target is not a regular file") + temporary = self._write_temporary(parent, data) + try: + os.replace( + temporary, + name, + src_dir_fd=parent, + dst_dir_fd=parent, + ) + except Exception: + os.unlink(temporary, dir_fd=parent) + raise + + def unlink_file(self, path: str) -> None: + with self._parent(path) as (parent, name): + current = os.stat(name, dir_fd=parent, follow_symlinks=False) + if not stat.S_ISREG(current.st_mode): + raise WorkspacePathError("workspace target is not a regular file") + os.unlink(name, dir_fd=parent) + + def entry(self, path: str) -> WorkspaceEntry: + """Return metadata without following any component or final symlink.""" + + parts = normalize_workspace_path(path) + if not parts: + value = os.stat(self.root_dir, follow_symlinks=False) + return WorkspaceEntry(WORKSPACE_PREFIX + "/", True, 0, _modified_at(value)) + with self._parent(path) as (parent, name): + value = os.stat(name, dir_fd=parent, follow_symlinks=False) + is_dir = stat.S_ISDIR(value.st_mode) + if not is_dir and not stat.S_ISREG(value.st_mode): + raise WorkspacePathError("workspace path is not a regular file or directory") + return WorkspaceEntry( + _virtual(parts) + ("/" if is_dir else ""), + is_dir, + 0 if is_dir else int(value.st_size), + _modified_at(value), + ) + + def delete(self, path: str) -> None: + """Delete one regular file or directory tree without following symlinks.""" + + with self._parent(path) as (parent, name): + value = os.stat(name, dir_fd=parent, follow_symlinks=False) + if stat.S_ISREG(value.st_mode): + os.unlink(name, dir_fd=parent) + return + if not stat.S_ISDIR(value.st_mode): + raise WorkspacePathError("workspace path is not a regular file or directory") + directory = os.open(name, _open_flags(directory=True), dir_fd=parent) + try: + self._delete_directory_contents(directory) + finally: + os.close(directory) + os.rmdir(name, dir_fd=parent) + + def _delete_directory_contents(self, directory: int) -> None: + for name in os.listdir(directory): + value = os.stat(name, dir_fd=directory, follow_symlinks=False) + if stat.S_ISREG(value.st_mode): + os.unlink(name, dir_fd=directory) + elif stat.S_ISDIR(value.st_mode): + child = os.open(name, _open_flags(directory=True), dir_fd=directory) + try: + self._delete_directory_contents(child) + finally: + os.close(child) + os.rmdir(name, dir_fd=directory) + else: + raise WorkspacePathError("workspace contains an unsupported entry") + + def list_dir(self, path: str = WORKSPACE_PREFIX) -> list[WorkspaceEntry]: + parts = normalize_workspace_path(path) + directory = self._open_dir_parts(parts) + entries: list[WorkspaceEntry] = [] + try: + for name in sorted(os.listdir(directory), key=str.casefold): + try: + value = os.stat(name, dir_fd=directory, follow_symlinks=False) + except FileNotFoundError: + continue + is_dir = stat.S_ISDIR(value.st_mode) + if not is_dir and not stat.S_ISREG(value.st_mode): + continue + child_parts = (*parts, name) + entries.append( + WorkspaceEntry( + virtual_path=_virtual(child_parts) + ("/" if is_dir else ""), + is_dir=is_dir, + size=0 if is_dir else int(value.st_size), + modified_at=_modified_at(value), + ) + ) + finally: + os.close(directory) + return entries + + def walk_files(self, path: str = WORKSPACE_PREFIX) -> list[WorkspaceEntry]: + parts = normalize_workspace_path(path) + root = self._open_dir_parts(parts) + results: list[WorkspaceEntry] = [] + + def walk(directory: int, relative: tuple[str, ...]) -> None: + for name in sorted(os.listdir(directory), key=str.casefold): + try: + value = os.stat(name, dir_fd=directory, follow_symlinks=False) + except FileNotFoundError: + continue + child = (*relative, name) + if stat.S_ISREG(value.st_mode): + results.append( + WorkspaceEntry( + virtual_path=_virtual((*parts, *child)), + is_dir=False, + size=int(value.st_size), + modified_at=_modified_at(value), + ) + ) + elif stat.S_ISDIR(value.st_mode): + child_fd = os.open(name, _open_flags(directory=True), dir_fd=directory) + try: + walk(child_fd, child) + finally: + os.close(child_fd) + + try: + walk(root, ()) + finally: + os.close(root) + return results + + +def _standard_error(exc: Exception) -> str: + if isinstance(exc, FileNotFoundError): + return FILE_NOT_FOUND + if isinstance(exc, IsADirectoryError): + return IS_DIRECTORY + if isinstance(exc, PermissionError): + return PERMISSION_DENIED + if isinstance(exc, (WorkspacePathError, ValueError)): + return INVALID_PATH + if isinstance(exc, OSError) and exc.errno in {errno.ELOOP, errno.ENOTDIR}: + return INVALID_PATH + return str(exc) + + +def _is_binary_file(path: str, raw: bytes) -> bool: + """Classify common containers explicitly and fall back to content.""" + + if Path(path).suffix.lower() in _BINARY_EXTENSIONS: + return True + sample = raw[:8192] + if b"\x00" in sample: + return True + try: + sample.decode("utf-8") + except UnicodeDecodeError: + return True + return False + + +class ScopedFilesystemBackend(BackendProtocol): + """DeepAgents file backend exposing exactly one ``/workspace`` tree.""" + + def __init__(self, root_dir: str | Path, *, max_search_file_bytes: int = 10 * 1024 * 1024): + self.workspace = RootedWorkspace( + root_dir, max_search_file_bytes=max_search_file_bytes + ) + + @staticmethod + def _search_path(path: str | None) -> str: + if path in {None, "/"}: + return WORKSPACE_PREFIX + return str(path) + + def ls(self, path: str) -> LsResult: + if path == "/": + return LsResult( + entries=[ + {"path": WORKSPACE_PREFIX + "/", "is_dir": True, "size": 0} + ] + ) + try: + return LsResult( + entries=[ + { + "path": item.virtual_path, + "is_dir": item.is_dir, + "size": item.size, + "modified_at": item.modified_at, + } + for item in self.workspace.list_dir(path) + ] + ) + except Exception as exc: + return LsResult(error=f"Cannot list '{path}': {_standard_error(exc)}") + + def read(self, file_path: str, offset: int = 0, limit: int = 2000) -> ReadResult: + try: + with self.workspace.open_binary(file_path) as handle: + raw = handle.read() + if _is_binary_file(file_path, raw): + return ReadResult( + file_data={ + "content": base64.standard_b64encode(raw).decode("ascii"), + "encoding": "base64", + } + ) + content = raw.decode("utf-8") + empty = check_empty_content(content) + if empty: + return ReadResult(file_data={"content": empty, "encoding": "utf-8"}) + lines = content.splitlines(keepends=True) + if offset >= len(lines): + return ReadResult( + error=f"Line offset {offset} exceeds file length ({len(lines)} lines)" + ) + return ReadResult( + file_data={ + "content": "".join(lines[offset : offset + limit]), + "encoding": "utf-8", + } + ) + except (UnicodeDecodeError, OSError, ValueError) as exc: + return ReadResult(error=f"Error reading file '{file_path}': {_standard_error(exc)}") + + def write(self, file_path: str, content: str) -> WriteResult: + try: + self.workspace.write_new(file_path, content.encode("utf-8")) + return WriteResult(path=file_path) + except FileExistsError: + return WriteResult( + error=f"Cannot write to {file_path} because it already exists. Read and then make an edit, or write to a new path." + ) + except (OSError, UnicodeEncodeError, ValueError) as exc: + return WriteResult(error=f"Error writing file '{file_path}': {_standard_error(exc)}") + + def edit( + self, + file_path: str, + old_string: str, + new_string: str, + replace_all: bool = False, + ) -> EditResult: + try: + with self.workspace.open_binary(file_path) as handle: + content = handle.read().decode("utf-8") + old_string = old_string.replace("\r\n", "\n").replace("\r", "\n") + new_string = new_string.replace("\r\n", "\n").replace("\r", "\n") + replacement = perform_string_replacement( + content, old_string, new_string, replace_all + ) + if isinstance(replacement, str): + return EditResult(error=replacement) + new_content, occurrences = replacement + self.workspace.replace_file( + file_path, new_content.encode("utf-8"), require_existing=True + ) + return EditResult(path=file_path, occurrences=int(occurrences)) + except (OSError, UnicodeError, ValueError) as exc: + return EditResult(error=f"Error editing file '{file_path}': {_standard_error(exc)}") + + @staticmethod + def _pattern(pattern: str, base: str) -> tuple[str, str]: + if pattern == WORKSPACE_PREFIX: + return WORKSPACE_PREFIX, "*" + prefix = WORKSPACE_PREFIX + "/" + if pattern.startswith(prefix): + return WORKSPACE_PREFIX, pattern[len(prefix) :] + return base, pattern.lstrip("/") + + def glob(self, pattern: str, path: str | None = None) -> GlobResult: + base = self._search_path(path) + base, relative_pattern = self._pattern(pattern, base) + try: + base_parts = normalize_workspace_path(base) + matches = [] + for item in self.workspace.walk_files(base): + item_parts = normalize_workspace_path(item.virtual_path) + relative = "/".join(item_parts[len(base_parts) :]) + if wcglob.globmatch(relative, relative_pattern, flags=_GLOB_FLAGS): + matches.append( + { + "path": item.virtual_path, + "is_dir": False, + "size": item.size, + "modified_at": item.modified_at, + } + ) + return GlobResult(matches=matches) + except (OSError, ValueError) as exc: + return GlobResult(error=f"Error globbing '{base}': {_standard_error(exc)}", matches=[]) + + def grep( + self, pattern: str, path: str | None = None, glob: str | None = None + ) -> GrepResult: + base = self._search_path(path) + try: + base_parts = normalize_workspace_path(base) + matches = [] + for item in self.workspace.walk_files(base): + if item.size > self.workspace.max_search_file_bytes: + continue + item_parts = normalize_workspace_path(item.virtual_path) + relative = "/".join(item_parts[len(base_parts) :]) + if glob and not wcglob.globmatch(relative, glob.lstrip("/"), flags=_GLOB_FLAGS): + continue + try: + with self.workspace.open_binary(item.virtual_path) as handle: + text = handle.read().decode("utf-8") + except (UnicodeDecodeError, OSError, ValueError): + continue + for line_number, line in enumerate(text.splitlines(), start=1): + if pattern in line: + matches.append( + { + "path": item.virtual_path, + "line": line_number, + "text": line, + } + ) + return GrepResult(matches=matches) + except (OSError, ValueError) as exc: + return GrepResult(error=f"Error grepping '{base}': {_standard_error(exc)}") + + def upload_files(self, files: list[tuple[str, bytes]]) -> list[FileUploadResponse]: + responses = [] + for path, content in files: + try: + self.workspace.replace_file(path, content) + responses.append(FileUploadResponse(path=path)) + except Exception as exc: + responses.append(FileUploadResponse(path=path, error=_standard_error(exc))) + return responses + + def download_files(self, paths: list[str]) -> list[FileDownloadResponse]: + responses = [] + for path in paths: + try: + with self.workspace.open_binary(path) as handle: + responses.append(FileDownloadResponse(path=path, content=handle.read())) + except Exception as exc: + responses.append( + FileDownloadResponse(path=path, content=None, error=_standard_error(exc)) + ) + return responses diff --git a/EvoScientist/workspace_scope.py b/EvoScientist/workspace_scope.py index 95d7fc9..37bae8e 100644 --- a/EvoScientist/workspace_scope.py +++ b/EvoScientist/workspace_scope.py @@ -2,9 +2,9 @@ from __future__ import annotations +import asyncio import os import shutil -import subprocess import threading import uuid from collections.abc import Callable @@ -49,28 +49,11 @@ def is_required() -> bool: def verify_required_executor() -> None: - """Fail startup unless the pinned scope executor is locally usable.""" + """Fail startup unless the pinned native executor passes its real preflight.""" - docker = shutil.which("docker") - if not docker: - raise RuntimeError("required workspace isolation needs the docker OCI runtime") - image = os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_IMAGE", "").strip() - if "@sha256:" not in image: - raise RuntimeError("required workspace isolation needs an OCI image pinned by digest") - try: - probe = subprocess.run( - [docker, "image", "inspect", image], - check=False, - capture_output=True, - text=True, - timeout=10, - ) - except (OSError, subprocess.TimeoutExpired) as exc: - raise RuntimeError("required workspace isolation cannot verify the OCI executor") from exc - if probe.returncode != 0: - raise RuntimeError( - f"required workspace isolation needs local OCI image {image!r}" - ) + from .native_sandbox import ensure_native_sandbox_ready + + ensure_native_sandbox_ready() def current_deployment_id() -> str: @@ -111,88 +94,6 @@ class _RuntimeScopeConfig: deployment_id: str | None -class ScopedContainerBackend: - """Filesystem backend whose shell commands execute in a scope-only OCI container.""" - - def __init__(self, root_dir: Path, *, timeout: int) -> None: - from .backends import CustomSandboxBackend - - # Reuse the hardened filesystem operations; only ``execute`` is - # replaced so no agent shell runs in the host process. - self._filesystem = CustomSandboxBackend( - root_dir=str(root_dir), virtual_mode=True, timeout=timeout, dangerous=False - ) - self._root_dir = root_dir - self._timeout = timeout - - def __getattr__(self, name: str) -> Any: - return getattr(self._filesystem, name) - - def execute(self, command: str, *, timeout: int | None = None) -> Any: - from .backends import ExecuteResponse, prepare_sandbox_command - - command, error = prepare_sandbox_command( - command, self._filesystem.cwd, virtual_mode=True, dangerous=False - ) - if error: - return ExecuteResponse(output=error, exit_code=1, truncated=False) - image = os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_IMAGE", "").strip() - if "@sha256:" not in image: - return ExecuteResponse( - output="Required workspace isolation needs an OCI image pinned by digest.", - exit_code=125, - truncated=False, - ) - effective_timeout = max(1, min(timeout or self._timeout, 3600)) - invocation = [ - "docker", - "run", - "--rm", - "--network", - "none", - "--read-only", - "--tmpfs", - "/tmp:rw,noexec,nosuid,size=64m", - "--cap-drop", - "ALL", - "--security-opt", - "no-new-privileges", - "--pids-limit", - os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_PIDS", "128"), - "--memory", - os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_MEMORY", "1g"), - "--cpus", - os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_CPUS", "1"), - "--mount", - f"type=bind,src={self._root_dir},dst=/workspace", - "--workdir", - "/workspace", - image, - "sh", - "-lc", - command, - ] - try: - completed = subprocess.run( - invocation, - check=False, - capture_output=True, - text=True, - timeout=effective_timeout, - ) - except FileNotFoundError: - return ExecuteResponse( - output="Required workspace isolation needs an OCI runtime (docker was not found).", - exit_code=127, - truncated=False, - ) - except subprocess.TimeoutExpired as exc: - output = (exc.stdout or "") + (exc.stderr or "") - return ExecuteResponse(output=output, exit_code=124, truncated=False) - output = completed.stdout + completed.stderr - return ExecuteResponse(output=output, exit_code=completed.returncode, truncated=False) - - def _configurable(runtime: ToolRuntime[Any, Any] | Any | None) -> dict[str, Any]: """Return the active runnable config, with a non-graph fallback. @@ -363,7 +264,43 @@ def provision_conversation_scope( return record -def _build_backend(root_dir: Path, *, dangerous: bool) -> Any: +def delete_conversation_scope( + thread_id: str, + *, + deployment_id: str | None = None, + workspace_root: Path | None = None, +) -> ScopeRecord: + """Idempotently revoke and remove the conversation scope for one Thread.""" + + root = (workspace_root or paths.WORKSPACE_ROOT).expanduser() + deployment_id = deployment_id or deployment_id_for_workspace(root) + registry = get_scope_registry(root) + record = registry.get_by_thread(deployment_id, thread_id) + if record.state == "deleted": + return record + if record.state != "deleting": + record = registry.transition_scope( + deployment_id, + record.scope_id, + expected_revision=record.revision, + state="deleting", + ) + + conversations = (root / ".evoscientist" / "conversations").resolve() + scope_root = conversation_root(record.scope_id, root) + if scope_root.parent.resolve() != conversations: + raise ScopeAccessError("workspace directory escapes its conversations root") + if scope_root.exists(): + shutil.rmtree(scope_root) + return registry.transition_scope( + deployment_id, + record.scope_id, + expected_revision=record.revision, + state="deleted", + ) + + +def _build_backend(root_dir: Path, runtime_dir: Path, *, dangerous: bool) -> Any: from deepagents.backends import CompositeBackend from .backends import ( @@ -375,8 +312,13 @@ def _build_backend(root_dir: Path, *, dangerous: bool) -> Any: cfg_timeout = int(os.getenv("EVOSCIENTIST_SANDBOX_EXECUTE_TIMEOUT", "300")) ws_backend: Any - if is_required(): - ws_backend = ScopedContainerBackend(root_dir, timeout=cfg_timeout) + web_full = os.getenv("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full" + if is_required() or web_full: + from .native_sandbox import NativeWorkspaceBackend + + ws_backend = NativeWorkspaceBackend( + root_dir, runtime_dir, timeout=cfg_timeout + ) else: ws_backend = CustomSandboxBackend( root_dir=str(root_dir), @@ -441,7 +383,7 @@ class DeferredScopedBackend(SandboxBackendProtocol): context = _resolve_scope_context(self._config) if context is None: raise ScopeAccessError("scoped backend lost its workspace scope") - if is_required() and self._dangerous: + if (is_required() or os.getenv("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full") and self._dangerous: raise ScopeAccessError( "dangerous_mode is incompatible with required isolation" ) @@ -449,7 +391,9 @@ class DeferredScopedBackend(SandboxBackendProtocol): with self._lock: if self._backend is None or self._backend_key != key: self._backend = _build_backend( - context.files_dir, dangerous=self._dangerous + context.files_dir, + context.runtime_dir, + dangerous=self._dangerous, ) self._backend_key = key return self._backend @@ -489,6 +433,12 @@ class DeferredScopedBackend(SandboxBackendProtocol): def execute(self, command: str, *, timeout: int | None = None) -> ExecuteResponse: return self._delegate().execute(command, timeout=timeout) + async def aexecute( + self, command: str, *, timeout: int | None = None + ) -> ExecuteResponse: + backend = await asyncio.to_thread(self._delegate) + return await backend.aexecute(command, timeout=timeout) + def create_workspace_backend( runtime: ToolRuntime[Any, Any], @@ -506,7 +456,7 @@ def create_workspace_backend( "deployed graph runs require a workspace scope" ) return legacy_backend() - if is_required() and dangerous: + if (is_required() or os.getenv("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full") and dangerous: raise ScopeAccessError("dangerous_mode is incompatible with required isolation") return DeferredScopedBackend(config, dangerous=dangerous) diff --git a/docs/architecture/Web会话任务一级数据项设计方案.md b/docs/architecture/Web会话任务一级数据项设计方案.md new file mode 100644 index 0000000..f33e8a4 --- /dev/null +++ b/docs/architecture/Web会话任务一级数据项设计方案.md @@ -0,0 +1,486 @@ +# Web 会话任务一级数据项设计方案 + +> 状态:已实施,定向验证通过 +> 版本:1.2 +> 日期:2026-08-17 +> 方案:方案一(最小改造版) +> 适用范围:Ai4Sci-Web 会话数据和 EvoScientist Web Runtime 投影链路 +> 历史兼容:开发阶段直接切换,不读取或写入旧 `task_snapshot` + +## 1. 目标与边界 + +本次只解决一个问题:把任务放入会话消息的统一 `items` 流,使任务可以和消息、推理、工具调用统一排序、持久化、恢复和展示。 + +本次不增加: + +- 任务数据库表、任务 API 或独立 Checkpoint; +- 独立任务状态通道、线程级任务缓存或 snapshot reducer; +- `parent_task_id`、任务树、任务组; +- `progress` 百分比和进度聚合。 + +任务的唯一持久化来源是: + +```text +ConversationMessage.items[type="task"] +``` + +## 2. 核心决策 + +```text +ConversationMessage +└── items[] + ├── message + ├── reasoning + ├── tool_call + ├── tool_output + ├── task + └── artifact +``` + +任务属于明确的助手消息。任务事件必须携带现有 `ProjectionEvent.message_id`,不得由前端猜测或把任务挂到线程级状态。 + +任务 ID 的作用域是 `message_id`:同一个 `task_id` 只要求在同一条助手消息内唯一。前端归并必须使用 `(message_id, task_id)`,禁止跨消息只按 `task_id` 归并。 + +`task_id` 由服务端适配器生成并写入事件。模型输出的任意同名字段不能直接作为任务 ID;如果上游提供 `id`,只能作为经过规范化、消息作用域隔离和哈希处理的非可信 `task_key`。相同运行重放时,适配器必须根据同一 `message_id + task_key` 生成同一 ID;不同消息即使任务文本相同,也必须得到不同的作用域。 + +## 3. 数据模型 + +在 `gateway/contracts/conversation_items.py` 增加并加入 `ConversationItem` 判别联合: + +```python +class TaskConversationItem(ItemBase): + type: Literal["task"] = "task" + task_id: str = Field(pattern=r"^task_[0-9a-f]{24}$") + content: str = Field(min_length=1, max_length=20_000) + # pending 在公共投影生命周期中表现为 in_progress。 + status: Literal["in_progress", "completed", "failed", "cancelled"] + task_status: Literal[ + "pending", "in_progress", "completed", "failed", "cancelled" + ] +``` + +任务条目的 `status` 是上述模型中显式覆盖的字段;`pending` 任务使用 +`status="in_progress"`,`cancelled` 任务使用 `status="cancelled"`。 +前端公共 `ItemStatus` 同步增加 `cancelled`,但其他条目类型不产生该值。 + +字段职责: + +| 字段 | 作用 | +| --- | --- | +| `item_id` | 会话条目 ID,由投影器分配 | +| `item_sequence` | 任务首次出现时在消息 `items` 中的顺序 | +| `revision` | 同一任务条目的更新版本,由投影器递增 | +| `actor` | 任务来源主体,沿用 `ItemActor` | +| `source` | 事件来源和 lane 序号,沿用 `ItemSourceRange` | +| `status` | 统一投影生命周期,不作为任务业务状态读取 | +| `task_id` | 在当前 `message_id` 内稳定的任务业务 ID | +| `content` | 面向用户的任务文本 | +| `task_status` | 任务业务状态 | + +`progress` 和 `parent_task_id` 不进入本阶段模型。需要层级或进度时另行设计,不在本次任务归类改造中隐式引入。 + +### 3.1 `status` 与 `task_status` 映射 + +任务条目覆盖 `ItemBase.status` 的类型,增加任务专属的 `cancelled` 生命周期状态。这样取消不会伪装成成功完成;公共条目消费者仍可按生命周期处理任务条目。 + +| `task_status` | `ItemBase.status` | 说明 | +| --- | --- | --- | +| `pending` | `in_progress` | 任务条目仍可更新 | +| `in_progress` | `in_progress` | 任务条目仍可更新 | +| `completed` | `completed` | 任务成功完成 | +| `failed` | `failed` | 任务失败 | +| `cancelled` | `cancelled` | 条目写入已结束,但不表示成功 | + +前端任务展示、筛选和统计只能读取 `task_status`,不能读取公共 `status` 判断任务是否成功。 + +## 4. 事件协议 + +删除 `ProjectionEvent.type="task_snapshot"`,增加批量的 +`ProjectionEvent.type="task_update"`。一次 `write_todos` 工具调用只产生一个 +`task_update`,一次 patch 内完成全部任务的追加、更新和取消,避免为同一原始事件分配多个 +`source_sequence`。 + +`ProjectionEvent.type` 的判别联合必须删除 `"task_snapshot"` 并加入 +`"task_update"`;`ProjectionEvent.payload` 仍保持通用字典,由任务事件分支显式通过 +`TaskUpdatePayload` 校验。 + +事件使用现有外层字段: + +```json +{ + "message_id": "msg_...", + "source": { + "protocol": "run-v1", + "source_key": "run:...", + "run_id": "run_...", + "stream_id": "stream_...", + "source_sequence": 42 + }, + "type": "task_update", + "actor": {"type": "root", "id": "root"}, + "payload": { + "mode": "replace", + "items": [ + { + "task_id": "task_91a...", + "task_key": "todo:0", + "content": "分析需求文档", + "task_status": "in_progress" + } + ] + } +} +``` + +严格合同: + +```python +class TaskUpdateEntry(ConversationContractModel): + task_id: str = Field(pattern=r"^task_[0-9a-f]{24}$") + task_key: str = Field(min_length=1, max_length=256) + content: str = Field(min_length=1, max_length=20_000) + task_status: Literal[ + "pending", "in_progress", "completed", "failed", "cancelled" + ] + + +class TaskUpdatePayload(ConversationContractModel): + mode: Literal["replace"] = "replace" + items: list[TaskUpdateEntry] = Field(max_length=1_000) +``` + +`mode="replace"` 表示当前 `write_todos` 列表是该消息任务集合的权威状态。列表中缺失的已有活动任务由投影器自动更新为 `cancelled`;已有终态任务保留原终态。 + +约束: + +- `message_id` 必须是目标助手消息;事件不得跨消息投影。 +- 目标助手消息记录必须先于事件创建,但可以还没有正文;任务允许成为该消息的第一个 item。 +- 已经完成或失败的消息不再接受新的实时任务事件;重放恢复除外。 +- `items` 中的 `task_id`、`task_key`、`content`、`task_status` 必填;第一阶段不接受 `progress` 和 `parent_task_id`。 +- `source_sequence` 使用现有 lane 全局序号,不能使用每个任务单独计数器。 +- 同一 lane、同一 `source_sequence` 的事件按现有 lane 游标幂等处理;重复事件直接忽略。生产端不得为不同 payload 复用同一序号。 +- `task_update` 只能由服务端的任务适配器生成,不能直接信任模型输出的任务事件。 + +### 4.1 任务事件生产链路 + +```text +write_todos 工具调用 + ↓ +服务端 todo 适配器 + ↓ 使用既有 message_id,生成完整 items、task_id、task_status +task_update ProjectionEvent + ↓ +ConversationItemProjector + ↓ +message_item_patch + ↓ +Web UI message.items +``` + +适配器规则: + +1. 只接受可信的 `write_todos` 工具调用,不接受任意模型文本中的任务字段。 +2. 对任务内容做 NFC 规范化和空白清理。 +3. `task_key` 优先取上游稳定的 `id`/`task_id`;没有稳定键时使用原列表位置,例如 `todo:0`。重排列表在没有稳定键时视为任务身份变化。 +4. 使用 `sha256(message_id + "\\0" + task_key).hexdigest()[:24]` 生成 `task_id`。同一次重放必须复用同一 `message_id` 和任务键。 +5. 状态别名在适配器转换,至少支持 `pending`、`in_progress`、`completed`、`failed`、`cancelled`;投影器只接受五种规范状态。 +6. `write_todos.todos` 必须是当前消息的完整任务列表,不支持增量列表;缺少完整列表时拒绝生成任务事件。 +7. 将完整规范化列表放入一个 `TaskUpdatePayload(mode="replace")`,不为列表中的每个任务另造事件序号。 + +Web event adapter 的映射必须固定为: + +```text +raw tool_call(name="write_todos") -> task_update +raw task_snapshot -> reject +其他普通 tool_call -> tool_call +``` + +Runtime emitter 不再产生 `task_snapshot`;`write_todos` 的原始工具事件只进入一次上述适配,不能同时走普通 `tool_call` 分支。 + +## 5. 投影规则 + +`ConversationItemProjector` 增加: + +```python +self.task_index: dict[str, int] +``` + +`task_index` 的键只在当前 projector 的 `message_id` 范围内有效。`from_message()` 从已有 `items` 扫描任务条目重建索引。 + +处理 `task_update`: + +1. 对尚未消费的序号先使用 `TaskUpdatePayload` 严格校验 payload;校验失败不推进 lane 游标、不产生 patch,并记录可诊断的投影错误。 +2. 校验通过后执行现有 `_accept_source`,按 lane 的全局 `source_sequence` 去重和拒绝乱序事件。 +3. 校验 `items` 内 `task_id` 和 `task_key` 不重复,并校验 `task_id` 是否等于服务端根据 `message_id + task_key` 计算的值。 +4. 按 `task_id` 查找当前消息中的任务:不存在时创建 `TaskConversationItem`,分配 `item_id`、`item_sequence`、`revision=1`,产生 `append`。 +5. 已存在时保留 `item_id` 和 `item_sequence`,只更新 `content`、`task_status`、`actor`、`source`,并递增 `revision`,产生 `update`。 +6. `mode="replace"` 下,当前消息中已有但不在 incoming `items` 的活动任务更新为 `cancelled`;已有终态任务不回退、不重复更新。 +7. 字段没有变化时不产生 patch,但仍保留已接受的 lane 游标。 +8. 每个任务在同一条消息中只能有一个 `item_id`;发现快照或索引重复时中止恢复并报告数据损坏。 + +### 5.1 状态转换 + +允许的业务状态转换: + +```text +pending -> pending | in_progress | completed | failed | cancelled +in_progress -> in_progress | completed | failed | cancelled +completed -> completed +failed -> failed +cancelled -> cancelled +``` + +终态任务不能回退、不能改写为其他终态,也不能修改 `content`。重复的同状态事件视为幂等事件;终态冲突事件报告协议错误。 + +### 5.2 幂等、乱序和冲突 + +- lane 级去重沿用 `last_source_cursor_by_lane`,不新增任务级全局游标。 +- 同一任务的业务变更由 `task_id` 定位、由 `revision` 递增;一次批量事件产生一个包含多个 item 操作的 patch。 +- 重复事件不产生新 patch。 +- 已被 lane 游标接受但内容与当前任务完全相同的事件不产生新 patch。 +- `UpdateItemOperation.expected_revision` 不匹配时返回 projection conflict,由恢复流程重新加载消息快照后重放未处理事件。 +- source 序号由生产端保证单调递增;投影器对已消费序号统一按幂等重复处理,不重新分配序号。 + +## 6. 运行结束和异常收敛 + +任务适配器应在运行结束前为每个已创建任务发送明确的终态 `task_update`。投影器仍必须提供最终收敛兜底,不能因为流中断而留下活动任务。 + +运行协议规定: + +- 正常完成:所有已创建任务必须为 `completed`、`failed` 或 `cancelled`,不能残留 `pending`/`in_progress`。 +- 运行取消:未收到终态的任务由 `finalize` 兜底更新为 `cancelled`。 +- 运行中断等待审批:这是逻辑消息的暂停,不是最终终态;任务保留 + `pending`/`in_progress`,以便同一 assistant `message_id` 的 resume Run 继续更新。 +- 运行失败:未收到终态的任务由 `finalize` 兜底更新为 `failed`,错误原因另由普通 `error` 条目承载。 +- 正常完成但仍有活动任务时,`finalize` 兜底更新为 `cancelled`,绝不伪造 `completed`。 +- `finalize` 必须跳过普通公共状态批量更新逻辑,使用任务专属状态更新,同时递增任务 `revision`;该兜底更新沿用当前 `source`,不伪造新的外部 `source_sequence`。 +- `finalize` 完成后将 `finalized=true` 持久化。之后除恢复重放外,新的实时事件拒绝投影。 + +这样可以避免出现“助手消息已完成但任务仍执行中”或“任务被错误标记为成功”。 + +## 7. 消息快照、Checkpoint 和恢复 + +删除并禁止继续写入: + +- `AssistantMessageV2.task_snapshot`; +- `ConversationItemProjector.task_snapshot`; +- `snapshot()` 返回的 `task_snapshot`; +- `task_panel_visible` metadata; +- `TaskSnapshot`、`TaskItem` 旁路模型。 + +消息快照只保存 `items`: + +```json +{ + "message_schema_version": 2, + "message_id": "msg_...", + "items": [ + {"type": "message"}, + {"type": "task", "task_id": "task_..."}, + {"type": "tool_call"} + ] +} +``` + +恢复流程: + +```text +数据库 message.items + ↓ +ConversationItemProjector.from_message() + ↓ +扫描 type == "task" + ↓ +重建 task_index、source lane 游标和 finalized 状态 + ↓ +继续接收 task_update +``` + +开发阶段切换前必须清理旧会话、checkpoint、recoverable run 和缓存数据。启动后不得再接受含 `task_snapshot` 的消息或事件;发现旧字段应明确报错,不能静默降级。 + +`ConversationItemProjector` 增加并在 `from_message()`/`snapshot()` 中保存 `finalized: bool`。`apply()` 先按 source 游标处理重放和重复事件,再拒绝已 finalized 消息的新实时事件;resume Run 在复用被中断的 assistant `message_id` 时显式调用 `reopen_for_resume()`,因此审批暂停不会阻断后续投影,真正完成/失败/取消后仍拒绝新任务。 + +`AssistantMessageV2` 同步增加 `finalized: bool = False`,否则严格消息合同会拒绝包含该字段的恢复快照。前端消息模型可以读取但不需要单独维护该状态。 + +## 8. 前端设计 + +在 `conversation-items.ts` 增加: + +```ts +export interface TaskConversationItem extends ItemBase { + type: "task"; + task_id: string; + content: string; + status: "in_progress" | "completed" | "failed" | "cancelled"; + task_status: + | "pending" + | "in_progress" + | "completed" + | "failed" + | "cancelled"; +} +``` + +加入 `ConversationItem` 联合类型,并将公共 `ItemStatus` 增加 `cancelled`。 +删除 `TaskSnapshot`、`TaskPanelState`、`taskStateByThread`、`tasksThreadId` 和 snapshot reducer 分支。 + +任务从消息条目派生,但必须保留消息作用域: + +```ts +function selectCurrentMessageTasks( + messages: ChatMessage[], + activeTurnMessageId: string | null, +): Array<{ key: string; messageId: string; task: TaskConversationItem }> { + const current = activeTurnMessageId + ? messages.find((message) => message.messageId === activeTurnMessageId) + : undefined; + const source = current ?? [...messages].reverse().find( + (message) => message.role === "assistant" + && (message.items ?? []).some((item) => item.type === "task"), + ); + return (source?.items ?? []) + .filter((item): item is TaskConversationItem => item.type === "task") + .map((task) => ({ + key: `${source!.messageId}:${task.task_id}`, + messageId: source!.messageId, + task, + })); +} +``` + +`TaskPanel` 保持接收 `threadId`,但内部改用 `selectCurrentMessageTasks(messages, activeTurnMessageId)`;不再从线程级任务状态读取。历史消息中的任务在对应消息条目中展示,任务面板默认只显示当前助手消息的任务。若需要会话级列表,必须按 `message_id` 分组展示。 + +`applyItemPatch()` 只负责将 `append`/`update` 写入目标消息的 `items`,不再读取或写入任何任务 metadata。`use-evo-stream.ts` 删除 `hideCompletedTaskSnapshot()` 调用;任务面板在任务全部完成后只改变折叠展示,不修改持久化数据。 + +所有 UI 状态判断和计数使用 `task.task_status`,公共 `status` 只用于通用条目生命周期: + +| `task_status` | 展示 | +| --- | --- | +| `pending` | 待处理 | +| `in_progress` | 执行中 | +| `completed` | 已完成 | +| `failed` | 失败 | +| `cancelled` | 已取消 | + +任务不进入 Markdown 正文拼接。无任务的消息不显示任务区域。 + +## 9. 错误处理 + +以下情况拒绝投影并记录错误: + +- `message_id` 不属于当前 projector; +- `task_id`、`content` 或 `task_status` 不合法; +- `task_id` 与 `message_id + task_key` 的服务端计算结果不一致; +- `items` 内存在重复 `task_id` 或 `task_key`; +- payload 含第一阶段未定义的 `progress`、`parent_task_id` 等字段; +- 终态任务回退或修改内容; +- `UpdateItemOperation.expected_revision` 不匹配。 + +错误不会创建任务条目。投影校验失败抛出统一的 `ProjectionProtocolError`,由 stream handler 转换为协议错误日志和 `error_raised`,不能静默丢弃。面向用户的执行失败通过 `task_status="failed"` 表示;需要详细原因时追加普通 `error` 条目,并通过关联的 `source` 或 `operation_id` 说明上下文。 + +## 10. 与工具调用的关系 + +```text +task = 计划或执行单元 +tool_call = 实际工具调用 +tool_output = 工具结果 +``` + +任务和工具调用是两个独立的一级条目,按各自事件到达顺序进入同一 `items` 列表。`write_todos` 是任务事件的唯一默认来源;普通工具调用不会自动生成任务。`write_todos` 本身是任务控制事件,不再额外生成 `tool_call` 条目,避免同一原始事件重复消耗 `source_sequence`;实际执行工具仍生成 `tool_call`/`tool_output`。 + +## 11. 必须删除的旧结构和代码清单 + +开发阶段直接删除,不保留兼容分支。以下文件必须完成迁移或删除旧字段引用: + +- 后端合同:`gateway/contracts/messages.py`、`gateway/contracts/run.py`、`gateway/contracts/__init__.py`; +- 投影和适配:`gateway/services/conversation_item_projector.py`、`gateway/services/projection_event_adapter.py`、`gateway/services/todo_state.py`; +- 恢复和派生:`gateway/services/recoverable_runs.py`、`gateway/services/evomemory_jobs.py`; +- Runtime 事件:`EvoScientist/stream/emitter.py`、`EvoScientist/stream/events.py`; +- 前端类型和状态:`frontend/src/types/tasks.ts`、`frontend/src/types/conversation-items.ts`、`frontend/src/store/chat-store.ts`、`frontend/src/store/conversation-store.ts`、`frontend/src/lib/conversation-reducer.ts`、`frontend/src/lib/api.ts`; +- 前端调用点:`frontend/src/components/chat/task-panel.tsx`、`frontend/src/app/(dashboard)/chat/[threadId]/page.tsx`、`frontend/src/hooks/use-evo-stream.ts`; +- fixtures 和旧测试:前端 e2e fixtures、chat/conversation store tests、后端 todo/projector/snapshot tests。 + +删除内容包括 `TaskItem`、`TaskSnapshot`、`task_snapshot` 事件和消息字段、`task_panel_visible` metadata、`ConversationItemProjector.task_snapshot`、线程级任务状态和旧 snapshot reducer。 + +完成后执行全仓库检查,`task_snapshot` 只允许出现在迁移说明或测试断言中,不得出现在运行时代码路径。 + +## 12. 测试和验收 + +### 后端合同和投影测试 + +- `task_update` 可被 `ProjectionEvent` 接受,`task_snapshot` 被拒绝; +- 一个 `write_todos` 列表只产生一个 `task_update`,并在一个 patch 中完成多个任务操作; +- `write_todos` 不再额外生成 `tool_call`,普通执行工具仍保留 `tool_call`/`tool_output`; +- 增量或结构不完整的 `write_todos` 列表被拒绝; +- 缺失的活动任务自动变为 `cancelled`,已有终态任务不回退; +- `TaskConversationItem` 判别联合、字段长度和状态枚举校验; +- 稳定 `task_key`、ID 计算和重放得到相同 `task_id`; +- 首次事件产生一个 `append`;后续事件只产生同一 `item_id` 的 `update`; +- 同消息相同任务 ID 不重复追加;跨消息相同任务 ID 不互相覆盖; +- 同状态重复、lane 乱序和重复重放; +- 终态转换矩阵和内容不可变规则; +- 任务事件早于其他消息条目、消息已结束和错误消息绑定; +- 非法 payload 通过 `ProjectionProtocolError` 进入协议错误处理,不静默丢失; +- 非法新事件不会推进 source 游标,修复后可以重放; +- `from_message()` 重建 `task_index`、lane 游标和 `finalized`; +- 运行成功、取消、中断、失败时任务终态收敛; +- 快照和 patch 中不再出现 `task_snapshot` 或 `task_panel_visible`。 + +### 前端测试 + +- 正确识别 `task` 条目并使用 `task_status`; +- 同一消息多 revision 只显示最新版本; +- 不同消息相同 `task_id` 仍分别显示; +- 按 `item_sequence` 排序; +- 五种任务状态图标和文案正确; +- 刷新、断线重连和历史消息恢复后任务不丢失; +- 任务不混入助手正文; +- 无任务时任务区域隐藏。 + +### 端到端验收 + +1. 一个 `write_todos` 列表产生一个批量 `task_update`。 +2. Web UI 收到一个 `message_item_patch`,并将所有任务放入目标助手消息的 `items`。 +3. 多次更新只改变同一个任务的 `revision`,列表中消失的活动任务变为 `cancelled`。 +4. 同一文本出现在不同消息时,两个任务互不覆盖。 +5. 断线重连或刷新后仍显示最新任务状态和 `finalized` 状态。 +6. 运行结束时不存在未收敛的 `pending`/`in_progress` 任务。 +7. 运行结束后再来的实时任务事件被拒绝,恢复重放仍可幂等处理。 +8. 运行代码和开发数据中不再出现旧 `task_snapshot`。 + +## 13. 实施顺序 + +1. 修改后端合同:增加 `TaskConversationItem`、`TaskUpdateEntry`、`TaskUpdatePayload`,替换事件类型。 +2. 修改 `write_todos` 适配器:完成稳定 `task_key`、`task_id` 计算,并一次生成完整 `items` 列表。 +3. 修改 Runtime emitter 和 Web event adapter:一个原始工具事件只分配一个 `source_sequence`。 +4. 修改 projector:完成批量 diff、缺失任务取消、状态机、幂等和严格 payload 校验。 +5. 增加 projector 的 `finalized` 持久化;修改 `from_message()`、snapshot、recovery 和 `finalize` 兜底。 +6. 删除后端旧 snapshot 模型、字段、metadata 和读取路径,按第 11 节文件清单逐项清零。 +7. 修改前端公共 `ItemStatus`、联合类型、消息条目渲染、按消息 selector 和任务面板。 +8. 删除前端旁路任务状态、旧 snapshot reducer、done 回调和相关 API 类型。 +9. 执行后端、前端和端到端测试,并增加 `task_snapshot` 静态扫描。 +10. 清理开发环境旧会话、checkpoint、recoverable run 和缓存数据。 +11. 启动前后端,验证批量事件、重连恢复、终态收敛、终态后拒绝和旧字段拒绝。 + +## 14. 验收标准 + +- `task` 是 `ConversationItem` 一级类型; +- 任务与消息、推理、工具调用统一存储和排序; +- 任务事件明确绑定 `message_id`; +- 同一消息同一任务只产生一个 `item_id`; +- 不同消息不会因相同 `task_id` 或文本而合并; +- 更新只递增同一条目的 `revision`; +- 五种任务业务状态有明确转换规则; +- 运行结束不存在未定义的活动任务状态; +- Checkpoint、刷新、重连和恢复后任务状态不丢失; +- 前端不依赖独立任务 snapshot 或线程旁路状态; +- 不新增数据库表、任务 API 或额外状态通道; +- 旧 `task_snapshot` 从运行时代码、开发数据和事件协议中移除。 + +## 15. 本阶段明确不做 + +- 任务父子层级和任务组; +- 百分比进度、总进度和加权聚合; +- 跨消息任务合并; +- 独立任务详情页、任务 API 和任务数据库; +- 历史旧 snapshot 自动迁移。 diff --git a/docs/architecture/Web原生操作系统进程沙盒设计方案.md b/docs/architecture/Web原生操作系统进程沙盒设计方案.md new file mode 100644 index 0000000..42ea909 --- /dev/null +++ b/docs/architecture/Web原生操作系统进程沙盒设计方案.md @@ -0,0 +1,699 @@ +# Web 原生操作系统进程沙盒设计方案 + +> 状态:已实施(macOS 验证通过,Linux 等待目标平台 CI 验证) +> 版本:1.4 +> 日期:2026-08-17 +> 适用范围:EvoScientist Web LangGraph Runtime +> 不适用范围:EvoScientist CLI、Web UI、Gateway 协议、审批流程 +> 兼容策略:开发阶段直接切换,不保留旧命令执行后端 + +## 1. 决策 + +Web Runtime 采用原生操作系统进程沙盒执行命令: + +- macOS 使用 Seatbelt; +- Linux 使用 bubblewrap 和 seccomp; +- 使用固定版本的 `@anthropic-ai/sandbox-runtime`(`srt`)统一生成并应用平台策略; +- 不再使用 Docker 作为 Web 命令执行后端; +- 沙盒初始化失败时直接拒绝执行,禁止回退宿主 Shell。 + +本方案的安全目标只有一个:**命令只能访问当前会话文件,以及部署明确放行的只读系统和工具链目录**。 +它保证访问权限隔离,不保证只能执行少数命令,也不承诺在 macOS 上隐藏当前会话目录的宿主物理路径。 + +文件工具继续使用现有的会话虚拟目录和 rooted accessor。文件工具与命令执行器必须从同一个 +`WorkspaceRegistry` 取得当前 Thread 的物理目录: + +```text +/.evoscientist/conversations//files/ +``` + +本方案替换《Web 虚拟目录与会话沙盒最小改造方案》中的容器命令执行设计,不改变该方案已经确定的 +Thread、scope、文件上传、下载、预览和目录描述符访问模型。 + +## 2. 必须接受的平台限制 + +Linux mount namespace 可以把会话目录绑定到真实的 `/workspace`,macOS Seatbelt 只能限制访问权限, +不能为进程建立新的文件系统命名空间,也不能把任意目录透明挂载成 `/workspace`。 + +为了保证 macOS 和 Linux 的 Shell 行为一致,本方案不在 Linux 单独提供真实 `/workspace` 挂载,而是采用 +统一合同: + +- 文件 API 和文件工具继续使用 `/workspace/**`; +- 命令进程的当前目录就是同一个会话工作区; +- Shell 命令只使用相对路径,例如 `uploads/input.docx`; +- Shell 中的 `.` 与文件工具中的 `/workspace` 指向同一个目录; +- 不解析或改写任意 Shell 文本中的 `/workspace`。 + +示例: + +```text +文件工具路径:/workspace/uploads/input.docx +Shell 路径: uploads/input.docx +物理文件: /files/uploads/input.docx +``` + +如果产品要求任意 Shell 命令中必须存在真实的 `/workspace` 绝对路径,则原生 macOS 沙盒无法满足,应该 +恢复选择 OCI 容器方案。通过字符串替换把 `/workspace` 改成宿主路径不属于可接受实现,因为它不能安全 +处理引号、变量、重定向、子 Shell 和脚本内容,还会向模型暴露物理路径。 + +当前产品接受原生沙盒的这一限制:命令可以通过 `pwd` 或 `getcwd()` 知道当前 scope 的物理路径,但即使 +知道路径也不能读取父目录、其他 scope 或其他宿主文件。服务端生成的常规错误和日志仍使用 +`/workspace` 表达当前目录,这是界面合同,不是安全边界。 + +## 3. 目标和非目标 + +### 3.1 目标 + +1. 命令只能读取当前 Thread 工作区和部署配置中明确列出的只读系统、工具链目录。 +2. 命令只能写入当前 Thread 工作区和本次执行的私有临时目录。 +3. 子进程自动继承相同限制,不能通过启动 Python、pandoc 或另一个 Shell 绕过沙盒。 +4. 默认禁止公网、内网、回环地址和 Unix Socket。 +5. 不向命令传递 Runtime 的密钥、数据库配置和服务凭证。 +6. 文件工具与命令执行器操作同一个物理目录。 +7. 沙盒不可用、策略生成失败或预检失败时拒绝执行。 +8. 修复 DOCX 被当作 UTF-8 文本读取的问题,使上传文档可由允许的宿主工具处理。 +9. 保留墙钟超时、流式输出保留上限和单文件大小等最低资源限制。 + +### 3.2 非目标 + +- 不建设容器池、任务队列或远程执行平台。 +- 不修改自动审批、人工审批、Checkpoint 或 LangGraph state channel。 +- 不修改 Gateway API、前端或数据库结构。 +- 不迁移历史 workspace 和历史执行记录。 +- 不允许模型动态扩大文件、网络或环境变量权限。 +- 不建设通用配额平台,不在第一版承诺跨平台内存和进程数量硬限制。 +- 不支持依赖宿主用户 Home 目录、项目虚拟环境或 Docker Socket 的命令。 +- 不把 macOS 后台进程绝对零残留作为文件访问隔离的成立条件。 + +## 4. 威胁模型 + +模型生成的命令和工作区内的文件全部视为不可信输入。方案必须防止: + +- `../`、绝对路径、符号链接和竞态条件造成目录逃逸; +- 读取其他 Thread、项目源码、用户 Home、SSH 密钥和云凭证; +- 写入 Runtime 源码、启动脚本、系统目录或其他 Thread; +- 通过子进程、Python 标准库、pandoc filter 或 Shell 重定向绕过限制; +- 通过网络、Unix Socket、Docker Socket、Apple Events 或 GUI 程序向外传输数据; +- 通过父进程环境变量和继承的文件描述符取得密钥; +- 后台子进程持续消耗宿主资源。 + +本方案不把提示词、审批结果、命令字符串检查或路径字符串替换作为安全边界。安全边界由 rooted 文件 +访问和操作系统强制执行的进程策略共同构成。 + +物理路径本身不属于本方案保护的秘密。受保护的是路径对应的数据和操作权限:知道宿主路径不能转化为 +对工作区外文件的读取或写入能力。 + +## 5. 目标架构 + +```text +Authenticated Web Thread + | + v + WorkspaceRegistry + | + +------------------------+ + | | + v v + RootedWorkspace NativeSandboxExecutor + file tools/API | + | v + | pinned srt CLI + | / \ + | macOS Seatbelt Linux bubblewrap + | | + +------------+-----------+ + v + //files +``` + +### 5.1 唯一工作区来源 + +执行器不能根据请求参数、模型输入或命令内容决定物理目录。它只能接收由 `WorkspaceRegistry` 解析完成的 +内部 `WorkspaceScope`,至少包含: + +```text +scope_id +thread_id +files_dir +``` + +`files_dir` 必须经过 Registry 的根目录约束校验。服务端不能主动把物理路径写入工具参数或结构化 API +字段;命令自身产生的 stdout、stderr 和文件内容属于不可信输出,本方案不承诺从中清除物理路径。 + +### 5.2 新执行器 + +新增 `NativeSandboxExecutor`,职责限定为: + +1. 接收可信 `WorkspaceScope` 和不可信命令字符串; +2. 为本次调用生成最小权限策略和私有临时目录; +3. 使用固定 argv 启动 `srt`,不让 Python Runtime 自己通过 `shell=True` 拼接启动命令; +4. 收集输出、处理超时并管理完整进程树生命周期; +5. 将底层错误转换为稳定的执行错误码; +6. 清理本次执行产生的策略文件和临时目录。 + +命令本身仍由沙盒内的 `/bin/sh -c` 解释,这是工具的预期能力。不能使用 `-l` 登录 Shell,避免读取 +profile 并修改白名单环境;调用 `srt` 的 Python 外层必须使用 argv 数组,不能使用 `shell=True`。 + +当前 `srt` 在 macOS/Linux 内部仍会生成受信任的包装字符串并通过宿主 Shell 启动 Seatbelt/bubblewrap。 +该层属于固定版本依赖的可信计算基,不能宣称整个启动链完全不经过宿主 Shell。每次升级必须重新验证 +引号、换行、变量展开、重定向和多字节文件名不会在进入沙盒前被错误解释。 + +## 6. 依赖管理 + +### 6.1 固定 sandbox-runtime + +在 Runtime 仓库中增加独立的原生沙盒工具目录和 lockfile,固定 +`@anthropic-ai/sandbox-runtime` 的精确版本。部署或启动时直接使用项目固定的 `srt` 可执行文件: + +```text +runtime/native-sandbox/package.json +runtime/native-sandbox/package-lock.json +runtime/native-sandbox/node_modules/.bin/srt +``` + +禁止每次执行时使用 `npx` 在线下载,禁止依赖未固定版本的全局 `srt`。Runtime 通过内部配置读取 +可执行文件位置,该配置不能来自模型或 Thread 参数。 + +环境准备阶段使用 lockfile 执行一次: + +```bash +npm ci --omit=dev --prefix runtime/native-sandbox +``` + +`node_modules` 不提交版本库,Runtime 启动和每次命令都不能执行依赖下载;缺少依赖只返回预检错误。 + +平台依赖是安装合同的一部分,Runtime 只检测、不在启动时安装: + +```text +macOS: node, rg, pinned srt +Linux: node, bwrap, socat, rg, pinned srt, apply-seccomp +``` + +Linux 的 `apply-seccomp` 会在 bubblewrap 内执行。由于 pinned `srt` 位于项目目录,而项目目录整体禁止读取, +策略构造器必须从固定包解析出 helper 的 canonical 路径,只把 helper 文件及其必需的动态库精确加入 +`srt_support_read_paths`。不能为了 helper 放行 `runtime/native-sandbox`、`node_modules` 或项目根目录。 +解析结果必须位于 lockfile 安装得到的 sandbox-runtime 包内;越出固定包目录、是符号链接到未知位置或 +缺少可执行权限时预检失败。 + +### 6.2 宿主工具链 + +允许的 Python、pandoc 等工具必须安装在受控的只读系统工具目录,例如: + +```text +macOS: /opt/homebrew/bin, /opt/homebrew/lib +Linux: /usr/bin, /usr/lib, /opt/evoscientist-tools +``` + +不放行 `/Users//.../.venv` 或项目源码目录。模型命令所需 Python 包应安装到现有 Homebrew Python +或部署配置的工具目录,而不是通过放开整个 Home 或项目目录解决。Runtime 自己继续使用项目 `.venv`; +该 `.venv` 不需要暴露给沙盒内命令。 + +`PATH` 只用于稳定查找工具,不是可执行文件安全白名单。命令可以用绝对路径执行任何位于允许读取目录 +中的系统二进制;安全性来自这些进程仍受相同的文件、网络、环境和资源策略限制。部署应尽量缩小工具链 +目录,但不增加脆弱的命令名称过滤器。 + +## 7. 文件系统策略 + +每次调用都根据当前 `files_dir` 生成策略。控制文件与命令临时文件必须使用两个不同目录: + +```text +/runtime/control// + settings.json + controller metadata + +/runtime/tmp// + HOME/ + TMPDIR/ +``` + +`control` 不进入命令的 `allowRead` 或 `allowWrite`;`tmp` 才作为 +`execution-private-temp-dir` 精确放行。两者使用随机 `execution_id` 和 `0700` 权限,不能放入 +`files_dir`。虽然控制进程与命令使用同一 OS 账号,操作系统沙盒仍必须拒绝命令访问 `control`。 + +`srt` 从 `settings.json` 完成策略加载后,Runtime 必须关闭命令不需要的控制文件描述符。策略路径不能通过 +环境变量传给目标命令,控制目录在命令结束后单独清理。 + +`sandbox-runtime` 的读取规则默认允许、写入规则默认拒绝,因此实现必须显式建立默认拒绝的读取策略, +不能把下面的 `allowRead` 清单误认为会自动禁止其他路径。等价策略为: + +```text +filesystem: + denyRead: + - / + allowRead: + - + - + - + - + - + allowWrite: + - + - +``` + +读取规则中精确的 `allowRead` 覆盖根级 `denyRead`;写入仍保持 allow-only。策略构造器必须使用 Registry +返回的 canonical scope 路径和部署拥有的静态工具链配置,不能接受模型、Thread 或命令提供附加路径。 + +### 7.1 允许读取 + +- 当前 Thread 的 `files_dir`; +- 本次执行的私有临时目录; +- 操作系统运行二进制和动态库所需的只读目录,例如 macOS 的 `/System`、`/usr`、`/bin`,Linux 的 + `/usr`、`/bin`、`/lib` 和 `/lib64`; +- 固定的只读工具链目录,例如 `/opt/homebrew` 或 `/opt/evoscientist-tools`; +- Linux 中 pinned `srt` 解析出的 `apply-seccomp` helper 及其必需运行库; +- 工具运行所需的最小安全设备节点,例如 `/dev/null`、`/dev/zero` 和 `/dev/urandom`; +- `srt` 在当前平台正常启动所需的最小系统路径。 + +工具链和系统根在启动时全部 canonicalize,并验证不位于 workspace 内。部署操作者和宿主已安装工具链 +属于可信边界,不要求目录必须由 root 或专用服务账号持有;真正的要求是沙盒策略必须拒绝命令写入这些 +目录。除上述清单外的 `/Users`、`/home`、`/private`、`/etc`、`/var`、workspace 总根及项目目录均保持 +不可读;若某个工具确实缺少运行依赖,应增加一个精确、只读、经过评审的路径,而不是放开其父目录。 + +### 7.2 允许写入 + +- 当前 Thread 的 `files_dir`; +- 本次执行的私有临时目录。 + +若工具需要 `/dev/null` 等固定安全设备节点,策略构造器按平台精确放行对应操作;不能放行整个 `/dev` +目录,也不能允许任意设备节点。 + +### 7.3 明确禁止 + +- 用户 Home 和其他 `/Users/**` 或 `/home/**` 数据; +- workspace 总根目录和其他会话目录; +- EvoScientist、Ai4Sci-Web 源码和配置目录; +- SSH、Git、AWS、Kubernetes、Docker 等用户配置; +- `/Volumes` 中的外部卷; +- 系统、工具链和 Runtime 安装目录的写入; +- 符号链接指向的工作区外目标; +- 设备文件、FIFO 和宿主 Unix Socket。 + +macOS 使用“拒绝根目录读取,再精确放行”的 Seatbelt 策略。Linux 使用 bubblewrap 只暴露必要系统目录 +和当前 scope;系统目录只读,当前 scope 和私有临时目录可写。两种平台必须通过相同的黑盒访问用例, +不能仅凭生成的策略文本判断隔离成功。 + +## 8. 网络和进程能力策略 + +默认策略为完全断网: + +- `allowedDomains` 为空; +- 禁止本地端口监听; +- 禁止回环和内网连接; +- 禁止 Unix Socket; +- 不挂载 Docker Socket; +- macOS 设置 `allowAppleEvents=false`; +- 不启用 weaker nested sandbox; +- 不启用 weaker network isolation; +- Linux 使用 bubblewrap network namespace 和 sandbox-runtime 的 seccomp 规则。 + +当前需求不需要按工具或命令动态放行网络。未来确有联网需求时必须作为独立设计评审,不能由模型在 +命令参数中声明域名白名单。 + +## 9. 环境变量和进程启动 + +执行器从空环境构造白名单环境,不复制 `os.environ`: + +```text +PATH= +HOME= +TMPDIR= +WORKSPACE=. +LANG= +LC_ALL= +``` + +固定版本 `srt` 会为了其网络代理和兼容临时目录向直接子进程注入额外变量。Runtime 使用放在本次 +command tmp 中的受信任入口脚本进入 Seatbelt/bubblewrap 后,再通过 `env -i` 只传递上述六项变量给 +用户 Shell。因此代理认证信息、`SANDBOX_RUNTIME` 和兼容变量不会到达模型命令。`srt` 强制增加的共享 +`/tmp/claude` 等默认写路径由 `denyWrite` 显式否决,实际 `TMPDIR` 指向本次 execution 私有目录。 + +明确不能传递: + +```text +OPENAI_API_KEY +ANTHROPIC_API_KEY +DATABASE_URL +AWS_* +SSH_* +GITHUB_TOKEN +KUBECONFIG +DOCKER_HOST +代理服务器凭证和其他 Runtime secrets +``` + +启动要求: + +- `cwd` 固定为当前 scope 的 `files_dir`; +- `stdin` 默认关闭; +- 关闭所有非必要继承文件描述符; +- 创建独立进程组或 session; +- stdout 和 stderr 以 bytes 并发流式读取,合计只保留配置的最大字节数; +- 使用现有执行超时; +- 超时、取消和 Runtime 关闭时先终止进程组,宽限期后强制终止; +- Linux 依靠 bubblewrap PID namespace 清理 namespace 内剩余进程; +- macOS 在超时、异步取消以及 `srt` 正常退出后终止对应进程组;主动 `setsid()` 脱离进程组的进程不承诺 + 仅凭 `srt + killpg` 绝对清理,但仍继承 Seatbelt 权限; +- 服务端自己生成的错误将当前 scope 前缀显示为 `/workspace`,但不把输出字符串替换当作安全机制。 + +### 9.1 最低资源限制 + +`srt` 负责文件和网络权限,不负责替代 Docker 的全部资源控制。第一版只保留对当前问题有直接价值且 +跨平台行为相对稳定的限制: + +```text +Runtime supervisor + -> in-package limit-and-exec helper + -> srt + -> /bin/sh -c +``` + +第一版至少限制: + +- 单文件大小; +- stdout/stderr 保留字节数; +- 墙钟超时。 + +Runtime 必须使用流式 `Popen` 并发读取 stdout/stderr,不能使用会无限缓存输出的 +`subprocess.run(capture_output=True)`。达到保留上限后继续 drain 两个管道但丢弃新增字节,让正常文件生成 +继续完成;结果设置 `truncated=True`。只有超时、取消或 Runtime 关闭才终止命令,持续输出由墙钟超时 +最终收敛。输出按 bytes 计数,结束后使用 UTF-8 `errors="replace"` 解码,避免二进制输出触发异常。 + +单文件限制由一个很小的 Runtime 内置 helper 在执行 `srt` 前设置 `RLIMIT_FSIZE` 后 `exec`;helper 不需要 +复制到 `/opt`,也不要求专用系统账号。 + +第一版不强制 `RLIMIT_NOFILE`、`RLIMIT_AS` 和 `RLIMIT_NPROC`,因为它们可能同时限制 `srt`、proxy 或同一 +用户的其他进程,不能提供稳定的每命令语义。生产 Linux 可以在应用外增加 cgroup/PID 限制,但不影响 +本期文件访问隔离验收。 + +## 10. 命令路径合同 + +Runtime 的执行工具说明必须明确: + +```text +The command starts in the workspace root. Use relative paths in shell commands. +File tools expose the same directory as /workspace. +The WORKSPACE environment variable is set to '.'. +``` + +允许: + +```bash +ls uploads +mkdir -p "$WORKSPACE/results" +pandoc "$WORKSPACE/uploads/湖南火电 需求说明书.docx" -o "$WORKSPACE/results/report.html" +python3 "$WORKSPACE/scripts/analyze.py" "$WORKSPACE/uploads/input.docx" +``` + +不支持: + +```bash +ls /workspace/uploads +cat /Users/user/secret.txt +cd ../other-scope +``` + +`/workspace` 在 macOS Shell 中不是实际挂载点,因此该路径按普通不存在路径处理。执行器不扫描、不拒绝、 +也不改写命令中的 `/workspace` 字符串;相对路径合同由工具说明表达,真正的越界访问由操作系统策略拒绝。 +文件工具结果使用 `/workspace/**`,模型转入 Shell 时应去掉该前缀或使用 `$WORKSPACE/`。 +命令可以取得真实 `cwd`,但不能据此访问 scope 之外的数据。 + +现有通用 Shell 提示中的虚拟根 `/`、`/output.log` 和 `run_in_background` 不适用于 Web full。Web full 使用 +独立的执行提示:日志写到 `$WORKSPACE/output.log`;不暴露后台执行工具,也不鼓励手工 `&`/`nohup`; +长命令通过 execute 的受限 `timeout` 前台运行。CLI 和 dangerous mode 的原提示保持不变。 + +## 11. DOCX 和二进制文件修复 + +命令沙盒与文件读取类型是两个独立问题,必须同时修复: + +1. 文件后端将 `.docx`、`.xlsx`、`.pptx`、PDF、图片和压缩包识别为二进制文件; +2. 二进制文件不能进入 UTF-8 文本读取分支; +3. `UnicodeDecodeError` 不能被通用 `ValueError` 捕获并映射成 `invalid_path`; +4. `read_file` 沿用现有 BackendProtocol,对识别出的二进制文件返回 base64,不新增 + `unsupported_binary_file` 协议; +5. 模型需要文档文本时使用同一 scope 内的 pandoc/Python 解析,不把 base64 当作文本处理; +6. pandoc/Python 使用 Shell 相对路径读取同一个 scope 内的文件; +7. 中文、空格和长文件名通过 argv/Shell 正确引用,不进行路径编码变体尝试。 + +## 12. 启动预检与安全测试 + +### 12.1 启动预检 + +Runtime 在接受请求之前执行一次轻量真实隔离预检。预检在系统短临时路径中创建测试 scope 和工作区外 +的测试哨兵文件;使用短路径是因为 macOS/Linux 的 Unix Socket 地址长度约为 100 字节,而部署 workspace +物理路径可能已经超过该限制。测试 scope 仍使用与正式会话完全相同的策略,只验证启动所需的关键能力: + +1. `srt` 存在且版本与 lockfile/配置一致; +2. 当前平台受支持; +3. macOS 的 Node、rg 和 Seatbelt 能力可用,或 Linux 的 Node、bwrap、socat、rg、用户 namespace 和 + `apply-seccomp` 可用; +4. 工作区内读写成功; +5. 允许的系统目录和工具链只能读取、不能写入; +6. `/etc/passwd`、用户 Home、项目源码、workspace 总根、其他 scope 中的哨兵文件读取失败; +7. 工作区和本次私有临时目录之外的写入失败; +8. 子 Shell 仍不能读取哨兵文件; +9. 对预检临时启动的宿主回环 TCP listener 和 Unix Socket 的访问失败; +10. Python、pandoc 等配置允许的工具能够启动; +11. Linux 在真实 `denyRead: ["/"]` 策略下仍能在 bubblewrap 内启动 `apply-seccomp` 并拒绝 Unix Socket; +12. 命令不能读取 `control/settings.json`,但能读写独立的 command tmp; +13. 临时文件和策略文件在测试后被清理。 + +任一关键检查失败,Runtime 启动失败并报告具体原因。不能记录警告后退回 `LocalShellBackend`。 +Linux 依赖检查报告 seccomp 不可用时,即使上游只给 warning,本 Runtime 也必须把它提升为启动失败,避免 +Unix Socket 在不知情的情况下变成可用。 + +Linux 部署需要显式检查 bubblewrap 所需的 unprivileged user namespace。Ubuntu 等启用额外 AppArmor 限制 +的环境应在部署层提供经过评审的 AppArmor 配置,应用代码不能尝试自行关闭系统安全策略。 + +### 12.2 调用位置 + +新增进程级 `ensure_native_sandbox_ready()`,接入点固定为: + +1. `EvoScientist._get_default_agent()` 检测到 `EVOSCIENTIST_DEPLOY_MODE=full` 后,在创建 backend 和 agent + 之前调用; +2. 检查成功后在当前 Runtime 进程内缓存 ready 状态和 pinned `srt` 版本,不为每个 Thread 重复执行; +3. 检查失败立即使 Web full graph 构建失败,不能创建可接收请求的 agent; +4. `NativeSandboxExecutor.execute()` 在启动命令前调用 `assert_native_sandbox_ready()`;ready 状态命中时 + 不重新执行探测,failed 状态直接拒绝; +5. CLI、dangerous 和 stripped 模式不调用该预检,保持现有行为。 + +`start-langgraph.sh` 可以额外提供诊断输出,但不能成为唯一调用点,因为 Web full 还可能从其他部署入口 +启动。安全闭环必须位于 Web full Runtime 进程内部。 + +### 12.3 CI 和平台验收 + +以下重型或具有副作用的用例只在 CI、安装验收和 `sandbox doctor` 中执行,不阻塞每次普通启动: + +- CPU、内存压力和进程爆发; +- 单文件、句柄和输出超限; +- 后台化、双重 fork 和 `setsid()`; +- 符号链接替换与并发目录竞态; +- 大量 stdout/stderr 和超时终止; +- macOS/Linux 完整逃逸矩阵。 + +## 13. 运行时错误模型 + +执行器对外返回稳定错误类型: + +| 错误码 | 含义 | +| --- | --- | +| `NATIVE_SANDBOX_UNAVAILABLE` | 当前平台或 `srt` 不可用 | +| `SANDBOX_POLICY_INIT_FAILED` | 本次策略生成或加载失败 | +| `EXECUTABLE_NOT_FOUND` | 允许工具链中不存在该命令 | +| `EXECUTION_TIMEOUT` | 执行超过超时 | +| `COMMAND_FAILED` | 命令正常启动但返回非零状态 | + +沙盒控制器自己生成的错误只返回虚拟路径和必要诊断,不能主动附带 scope 物理路径、策略内容或宿主环境 +信息。命令自身的 stdout/stderr 按不可信输出处理,不承诺路径脱敏。权限错误不能继续映射成 +`invalid_path`。 + +使用 `srt` CLI 时,命令收到的 `EPERM`、网络拒绝和工具自己的权限错误不能稳定区分,统一返回 +`COMMAND_FAILED` 并保留经过大小限制的 stderr。第一版不解析易变的日志文本来伪造细粒度错误码;若以后 +确实需要结构化 violation,另行增加基于 `SandboxManager` 库接口的 Node bridge。 + +## 14. 代码改动范围 + +预计修改: + +```text +EvoScientist/workspace_scope.py + - 用 NativeSandboxExecutor 替换 Web 的 ScopedContainerBackend + - 保留 Registry 和 rooted 文件边界 + +EvoScientist/native_sandbox.py(新增) + - 平台检查 + - 策略构造 + - control/command tmp 分离 + - srt 启动 + - 环境清理 + - 超时、进程树清理和错误映射 + +EvoScientist/sandbox_exec_helper.py(新增) + - 在进入 srt 前设置 RLIMIT_FSIZE + - exec 固定版本的 srt,不安装第二套系统工具链 + +EvoScientist/EvoScientist.py + - Web full 创建 agent 前调用 ensure_native_sandbox_ready() + - CLI 和 stripped 模式不受影响 + +EvoScientist/workspace_files.py + - 修复 DOCX 等二进制类型识别和错误分类 + +EvoScientist/prompts.py + - Web full 的 execute 使用相对路径和 WORKSPACE=. + - Web full 不再引用 /output.log 或不存在的后台执行工具 + +runtime/native-sandbox/(新增) + - 固定 sandbox-runtime 依赖及 lockfile + +tests/ + - 单元测试、真实沙盒集成测试和回归测试 +``` + +不修改 Web UI、Gateway 文件协议、审批 middleware、Checkpoint 和数据库 schema。 + +## 15. 测试设计 + +### 15.1 文件边界 + +- 当前 scope 普通文件可读写; +- 另一个 scope 不可读写; +- `../` 和宿主绝对路径不能逃逸; +- 工作区内外符号链接均不能突破边界; +- 子进程与父命令具有相同访问结果; +- 并发替换路径或符号链接不能绕过 rooted accessor。 +- `/etc`、用户 Home、项目目录、workspace 总根和其他 scope 均不可读; +- 系统及工具链清单可读但不可写; + +### 15.2 网络和凭证 + +- DNS、公网、回环和内网连接失败; +- Unix Socket 和 Docker Socket 不可用; +- Apple Events、`open` 和 `osascript` 不能逃出沙盒; +- 命令环境中不存在测试 API Key; +- 继承文件描述符不能读取 Runtime 已打开的敏感文件。 + +### 15.3 工具和文档 + +- Python 和 pandoc 可从受控工具链启动; +- 中文、空格文件名的 DOCX 可读取和转换; +- `.docx` 不再进入 UTF-8 解码分支,`read_file` 返回现有 base64 合同; +- 生成的 HTML 位于当前 scope,能被现有文件列表和下载接口发现; +- 相对路径执行成功,`/workspace` 字符串不会被扫描或改写。 +- `$WORKSPACE/uploads/**` 与 Shell 相对路径指向同一文件; +- 引号、换行、`$()`、重定向、空格、中文文件名在 `srt` 包装前后语义一致; +- Web full 提示词不再生成 `/output.log`,日志写入 `$WORKSPACE/output.log`; +- Web full 不再提示使用未暴露的 `run_in_background`,长命令使用受限 timeout 前台执行; + +### 15.4 生命周期 + +- 超时终止完整进程组; +- Runtime 取消请求后,同一进程组内无残留进程; +- macOS 后台化和尝试脱离 session 的进程触发残留检测与告警,不能突破原有 Seatbelt 权限; +- 策略加载失败不会启动未隔离命令; +- 启动预检失败时 Runtime 不接受请求; +- 并行 Thread 的策略不会使用错误的 scope。 +- 单文件限制有真实超限测试;输出超过保留上限后命令继续完成且结果标记 `truncated=True`;CPU、内存、 + 句柄和进程压力进入 CI 观察项。 + +## 16. 验收标准 + +以下条件必须全部成立: + +1. 上传到当前 Thread 的中文 DOCX 能被 pandoc 或 Python 处理。 +2. 命令输出写入当前 Thread 后,Web 文件接口立即可见。 +3. 文件工具与命令执行器读写同一个物理目录。 +4. 命令无法读取其他 Thread、用户 Home、项目源码和服务密钥。 +5. 命令无法写入工作区和私有临时目录之外。 +6. 子进程不能绕过文件、网络和环境变量限制。 +7. 网络、Unix Socket、Docker Socket 和 Apple Events 默认不可用。 +8. 超时和取消后,同一进程组内不存在残留进程;macOS 检测到脱离进程时记录安全告警。 +9. 沙盒不可用时命令失败,绝不降级为宿主直接执行。 +10. 命令即使知道当前 scope 的物理路径,也不能读取或写入 scope 之外的数据。 +11. 每次命令的超时和单文件限制真实生效;输出保留有界且不会仅因截断而终止命令。 +12. 服务端生成的常规错误使用 `/workspace` 表达当前目录,不承诺清理任意命令输出或生成文件中的路径。 +13. macOS 和 Linux 的 Shell 都使用工作区相对路径合同。 +14. 现有 Web UI、Gateway API 和审批流程不需要因本次改造变化。 + +## 17. 实施顺序 + +1. 固定 `sandbox-runtime` 依赖和平台依赖清单,解析 Linux `srt_support_read_paths`。 +2. 实现 control/command tmp 分层、默认拒绝读取策略和 `NativeSandboxExecutor`。 +3. 实现流式输出、超时监督和轻量 exec helper,设置单文件限制。 +4. 将 Web Runtime 的 execute 路由切换到新执行器,删除 Web 未隔离宿主 Shell 和容器回退路径。 +5. 在 Web full agent 创建前接入进程级预检,并在 execute 增加 ready 断言。 +6. 更新 Web full 执行提示,统一使用 `$WORKSPACE`/相对路径、非登录 Shell 和前台执行合同。 +7. 修复 DOCX 和其他二进制文件类型识别,保持二进制 base64 协议。 +8. 用真实上传 DOCX 完成上传、解析、生成、列表、预览/下载闭环验证。 +9. 在 macOS 和目标 Linux 环境分别运行 `sandbox doctor` 与完整 CI 隔离测试后再合入。 + +由于处于开发阶段,不增加 feature flag、双执行后端、历史迁移和兼容分支。 + +## 18. 风险与约束 + +### 18.1 sandbox-runtime 成熟度 + +`sandbox-runtime` 当前属于实验性项目。必须固定版本,不能自动升级;每次升级都要重新运行完整逃逸测试并 +评审 macOS/Linux 策略差异。 + +### 18.2 macOS 没有 mount namespace + +Shell 使用相对路径是本方案成立的必要合同,不是临时实现细节。若后续必须支持真实 `/workspace`,应重新 +选择容器,而不是添加命令重写层。 + +本方案明确接受命令能够取得当前 scope 的物理路径。Seatbelt 必须保证该知识不能转化为工作区外访问; +若未来把“宿主路径不可见”升级为安全要求,本方案立即不再适用。 + +### 18.3 宿主工具链 + +部署操作者和已安装的系统工具链属于可信边界,可以由当前登录用户持有。沙盒策略必须使工具链对命令 +只读,但不要求为本功能新建系统账号或复制第二套工具链。若某个模型工具只能从项目 `.venv` 运行,应把 +所需依赖安装进现有 Homebrew Python 或专用工具目录,而不是放开整个项目目录。 + +### 18.4 平台策略差异 + +Seatbelt 与 bubblewrap 的机制不同,不能只依赖单元测试证明安全。两个目标平台都必须运行相同的黑盒 +逃逸用例,以行为结果而不是策略文本作为验收依据。 + +### 18.5 读取规则语义 + +`sandbox-runtime` 读取权限默认允许,并且 `allowRead` 优先于 `denyRead`。固定版本升级后必须重新验证根级 +拒绝加精确放行仍保持相同语义;如果语义变化,启动预检必须阻止 Runtime 启动。 + +### 18.6 macOS 进程清理 + +Seatbelt 限制会被子进程继承,但 `srt` CLI 和进程组信号不能严格保证清除已经脱离 session 的进程。 +第一版保证文件和网络权限不会因后台化而扩大,并对已知进程树进行清理和告警;它不把 macOS 绝对零残留 +作为安全承诺。若后续需要该承诺,应改用具备生命周期命名空间的容器或 VM。 + +### 18.7 CLI 错误能力 + +`srt` CLI 不提供足够稳定的结构化 violation 合同,因此第一版不区分文件拒绝和网络拒绝,统一保留为 +受大小限制的 `COMMAND_FAILED`。只有在引入基于库接口的 Node bridge 后,才能新增更细的拒绝错误码。 + +## 19. 参考资料 + +- [Anthropic sandbox-runtime](https://github.com/anthropic-experimental/sandbox-runtime) +- [bubblewrap](https://github.com/containers/bubblewrap) + +## 20. 实施与验证记录 + +2026-08-17 已完成: + +- 固定 `@anthropic-ai/sandbox-runtime@0.0.73` 并提交 package lock; +- Web full 后端已从 `ScopedContainerBackend` 切换为 `NativeWorkspaceBackend`,没有 Docker 或宿主 Shell + 回退; +- Web full graph 在 agent/backend 创建前执行进程级真实预检; +- 文件工具与命令执行器共享同一个 scope `files_dir`,control 与 command tmp 分离; +- 目标命令使用 `env -i`,Runtime secret 和 `srt` 代理凭证不会进入命令环境; +- 增加有界流式输出、`RLIMIT_FSIZE`、墙钟超时、异步取消和退出后进程组清理; +- `.docx`、`.xlsx`、压缩包及未知含 NUL 文件返回现有 base64 合同; +- Web full 提示词只使用相对路径/`WORKSPACE=.`,不再建议不存在的后台工具; +- macOS 真实黑盒测试覆盖 scope 逃逸、control 读取、TCP/Unix Socket、环境清理、输出截断、文件大小、 + 超时、异步取消、后台子进程清理,以及中文空格 DOCX 经 pandoc 生成和回读闭环; +- 全仓回归结果为 `3030 passed, 16 skipped`;原生沙盒专用黑盒结果为 `6 passed`; +- Web LangGraph Runtime 已在 `127.0.0.1:3076` 重启并成功导入 `EvoScientist` 主 graph。 + +未在本机宣称完成的事项:Linux bubblewrap/seccomp 行为验证。Linux 代码包含固定 `apply-seccomp` 路径、 +依赖预检和 fail-closed 策略,但合入 Linux 部署前必须设置 +`EVOSCIENTIST_RUN_NATIVE_SANDBOX_TESTS=1` 在目标 Linux 主机运行同一黑盒套件。 diff --git a/docs/architecture/Web虚拟目录与会话沙盒完整改造方案.md b/docs/architecture/Web虚拟目录与会话沙盒完整改造方案.md new file mode 100644 index 0000000..29f57fe --- /dev/null +++ b/docs/architecture/Web虚拟目录与会话沙盒完整改造方案.md @@ -0,0 +1,591 @@ +# Web 虚拟目录与会话沙盒最小改造方案 + +> 状态:已实施并完成闭环验证 +> 版本:2.4 +> 日期:2026-08-16 +> 适用范围:Ai4Sci-Web Frontend/Gateway、EvoScientist Web LangGraph Runtime +> 不适用范围:EvoScientist CLI +> 兼容策略:开发阶段直接切换,不兼容旧 Web workspace 和旧会话文件 + +## 1. 结论 + +Web 中一个 Thread 只对应一个真实会话目录,模型和用户只使用统一的虚拟根目录 +`/workspace`。文件工具直接在该会话目录内工作;命令工具在容器中执行,并且只把该会话目录挂载为 +`/workspace`。 + +```text +Thread + -> conversation scope directory + -> mounted/resolved as /workspace + -> file tools, execute and Web file APIs all use this directory +``` + +这就是本次改造的全部核心。生成文件已经位于: + +```text +.ai4sci/workspace/.evoscientist/conversations//files/ +``` + +该物理位置本身没有问题。问题是目前不同工具没有共同遵守这个根目录,而且 Web 文件 API 没有把它作为 +当前 Thread 的文件源。 + +本方案不引入 Artifact 状态机、文件提交批次、Run 完成事务、恢复快照、常驻容器、远程执行协议或双向 +文件流。 + +## 2. 目标与边界 + +### 2.1 必须实现 + +1. 一个 Web Thread 在创建时绑定一个唯一 `scope_id` 和一个唯一会话目录。 +2. 用户文件在模型、工具调用参数、消息和页面中只使用 `/workspace/**` 虚拟路径。 +3. 文件工具只能读写当前会话目录,不能访问父目录、其他 Thread 或宿主机任意路径。 +4. `execute` 只能在容器内运行,容器只挂载当前会话目录到 `/workspace`。 +5. Gateway 的文件列表、下载和预览直接读取同一个会话目录,不再维护第二份 workspace 副本。 +6. Web 没有安全 executor 时禁止执行命令,不允许回退宿主 shell。 +7. 自动审批和人工审批使用完全相同的目录和执行边界。 +8. CLI 保持现状,不进入本次改造。 + +### 2.2 非目标 + +- 不兼容或迁移历史 Web workspace。 +- 不建设通用容器编排平台。 +- 不支持 Gateway 与 Runtime 分布式部署或远程文件传输。 +- 不新增 Artifact 表、批次表、版本状态或完成态事务。 +- 不新增 Run snapshot、tmpfs 输出区或文件冻结流程。 +- 不重新设计审批、Checkpoint、消息投影和 HTML 预览安全策略。 +- 不修改 Recoverable Run 的 `workspace_state` 枚举、数据库约束和 `bound` 派发门槛;只把准备阶段的 + “复制附件”改为“原位验证附件”。 +- 不解决多个并发 Run 修改同名文件的业务冲突;沿用当前 Thread 语义。 +- 不在应用层实现 scope 总容量状态机。workspace 所在卷必须由部署配置容量上限或文件系统 quota;本期 + 只增加容器单文件大小限制。 + +## 3. 唯一目录模型 + +### 3.1 物理目录 + +每个 Thread 使用一个目录: + +```text +/.evoscientist/conversations//files/ +``` + +目录内不再强制拆分 `uploads`、`artifacts` 和 `work`。上传文件、工具修改和模型生成文件都位于同一个 +目录树中,目录结构由用户和模型正常组织。 + +### 3.2 虚拟目录 + +上述物理目录在所有 Web 接口中统一表示为: + +```text +/workspace +``` + +示例: + +```text +物理路径:/files/report/result.html +虚拟路径:/workspace/report/result.html +``` + +浏览器不能提交物理路径或 `scope_id` 来选目录。Gateway 必须根据已鉴权的 `user_id + thread_id` +取得 Thread 记录,再使用其中的 `workspace_dir`。 + +Thread 创建时完成以下操作,不再延迟到第一次 Run: + +```text +create Thread + -> Gateway 调用现有 ensure_scope(thread_id) + -> Runtime 返回 scope_id 并创建 scope/files + -> Gateway 按共享 workspace_root + scope_id 推导 files 目录 + -> Gateway 校验目录存在且位于 conversations 根下 + -> 将该目录写入现有 threads.workspace_dir +``` + +`scope_id` 继续由现有 Runtime Scope Registry 管理,不新增 Gateway scope 表。Gateway 与 Runtime 共用一个 +`scope_files_dir(workspace_root, scope_id)` 路径规则;内部 scope API 不向浏览器返回物理路径。 + +物理 `workspace_dir` 只允许保存在服务端 Thread 记录中。Thread API、提示词、消息和前端状态不得返回该 +字段的真实值;对外统一返回 `/workspace`,文件响应统一返回 `/workspace/`。 + +### 3.3 唯一权威 + +会话目录就是当前 Thread 文件的唯一权威数据源: + +- 上传文件直接写入会话目录; +- 文件工具直接读写会话目录; +- `execute` 通过容器挂载读写会话目录; +- 文件列表直接扫描会话目录; +- 下载和预览直接读取会话目录; +- 删除 Thread 时删除对应会话目录。 + +不再把文件从 Gateway workspace 复制到 Runtime scope,也不再把生成文件从 Runtime scope 提交回 +Gateway Storage。这样不会出现两份同名文件内容不一致的问题。 + +对于 Web Thread,会话文件路由不再根据 `storage_service.backend_name` 分流: + +- 上传、列表、下载、预览和归档统一使用 `threads.workspace_dir`; +- 上传仍执行现有大小、MIME、magic bytes 和用户配额校验; +- 上传响应以 `virtual_path` 为文件标识,`file_id` 可以为空; +- 通用 Storage Service 继续服务其他业务,但不再作为 Web 会话文件的数据源或索引; +- 模型生成文件不需要注册 `user_files` 或 Artifact 记录。 + +本版本只支持 Gateway 与 Runtime 共享同一 workspace 根目录的部署方式。若部署时目录不可共享,启动 +检查必须报告配置错误;本次不为该场景增加网络文件协议。 + +### 3.4 Run 附件合同 + +附件在上传时已经位于当前 Thread 的 `/workspace/uploads/**`,因此创建 Run 时不再复制字节: + +```text +Run body.files: /workspace/uploads/a.pdf + -> Gateway 使用当前 Thread rooted accessor 打开并验证文件 + -> 校验文件存在、是普通文件,并且仍属于当前 workspace + -> 保留现有 bind_run(scope_id, run_request_id, run_id) + -> 保留现有 mark_workspace_bound(run_id) + -> workspace_state=bound 后正常派发 +``` + +现有 `workspace_state`、lease、重试和 `bound` 派发门槛保持不变。只修改 +`_prepare_workspace_once()` 的附件分支:不再要求 `storage_service`,不再调用 `open_workspace_input()` 或 +`WorkspaceScopeClient.materialize()`。无附件的 Run 仍直接执行 bind 和 bound。 + +附件描述符只信任 `virtual_path`,其值必须是 `/workspace/uploads/**`;旧 `file_id`、object key 和物理路径 +不参与解析。retry/resume 复用同一个虚拟路径,并在每次 bind 前重新验证文件仍然存在。 + +## 4. 最小目标架构 + +```text +Browser + | thread_id + /workspace/relative/path + v +Gateway + | authenticate user and load thread.workspace_dir + | list / upload / download / preview + v +Conversation directory (single source of truth) + ^ ^ + | rooted file backend | bind mount + | | +LangGraph file tools command container + /workspace +``` + +用户文件边界只需要三个现有职责: + +1. Gateway 使用 Thread 所有权鉴权选择唯一 scope。 +2. 文件 API 和文件工具使用同一 rooted file accessor。 +3. shell/解释器进入容器,依靠 mount namespace 隔离宿主文件系统。 + +路径字符串替换、提示词和审批规则都不是安全边界。 + +## 5. 文件访问规则 + +所有 Web 文件入口共用一个 rooted file accessor,输入只能是 `/workspace` 或 +`/workspace/`。不能采用“先 `resolve()`、再用路径重新 `open()`”的实现,因为容器可以在 +两步之间替换符号链接。 + +解析规则: + +1. 去掉固定前缀 `/workspace`,得到相对路径。 +2. `/workspace` 本身表示根目录;其他输入拒绝空字节、`.`、`..`、空路径段和非 `/workspace` 绝对路径。 +3. 打开当前 Thread scope 根目录的目录描述符,后续操作始终相对于该描述符。 +4. 逐级使用 `dir_fd + O_NOFOLLOW` 打开目录和文件;任何一级是符号链接都拒绝。 +5. 创建文件或目录时,使用已经验证的父目录描述符执行相对创建,不重新拼接宿主绝对路径。 +6. 列表使用 `lstat`/不跟随链接的目录遍历;符号链接不进入列表,也不能下载、预览或归档。 +7. 只接受普通文件和普通目录,不接受设备文件、FIFO 和 socket。 +8. 错误信息只返回虚拟路径,不返回宿主物理路径。 + +Web 会话目录不支持符号链接,包括指向 scope 内部的符号链接。该限制用很小的功能代价消除了宿主 +Gateway 与并发容器之间的 TOCTOU 逃逸窗口。 + +安全打开后必须继续使用同一个文件描述符:下载和预览使用已经打开的 file handle 进行流式响应,不能把 +校验后的路径交给 `FileResponse(path)` 再次打开;归档也必须通过 accessor 遍历和读取,不能先收集宿主 +路径再打包。 + +以下访问必须失败: + +```text +/etc/passwd +/Users/... +/workspace/../other +/workspace/link-to-host-secret +其他 Thread 的物理目录或 scope_id +``` + +这里限制的是宿主文件。容器镜像自身可以存在 `/etc/passwd`、`/proc` 等正常系统文件,但它们不是 +宿主机的同名文件。 + +## 6. 命令执行规则 + +保留现有的每次命令启动一次容器的实现方向,不引入常驻容器生命周期。 + +等价执行模型: + +```text +docker run --rm + --network none + --read-only + --tmpfs /tmp:rw,noexec,nosuid,size=64m + --cap-drop ALL + --security-opt no-new-privileges + --user : + --env HOME=/tmp + --pids-limit + --memory + --cpus + --ulimit fsize=: + --mount type=bind,src=,dst=/workspace,rw + --workdir /workspace + + +``` + +具体要求: + +- 只挂载当前 scope 的 `files` 目录,不挂载 workspace 总根、项目源码、用户 home 或 Docker socket。 +- 不继承 Gateway/LangGraph 的密钥环境,只传入明确允许的普通环境变量。 +- 使用 digest 固定镜像;镜像不存在或 Docker 不可用时调用失败。 +- 禁止回退到 `LocalShellBackend`、`shell=True` 或宿主 `subprocess`。 +- 命令的当前目录固定为 `/workspace`。 +- 容器使用 Runtime 进程的数字 UID/GID,值只能来自 `os.getuid()`/`os.getgid()`,不能来自模型、请求或 + Thread 参数;容器生成文件必须继续由 Gateway 和 Runtime 读写、覆盖和删除。 +- 容器 `HOME` 固定为私有可写 `/tmp`,不挂载或注入宿主 home。 +- 保留现有的只读 rootfs、私有 `/tmp`、`no-new-privileges`、PID、内存和 CPU 限制。 +- 使用配置化的 `fsize` soft/hard limit 限制单个生成文件;超限命令失败并返回普通执行错误。 +- stdout/stderr 中发现宿主 workspace 前缀时进行错误处理,不能把物理路径写入对话消息。 + +`ScopedContainerBackend.execute()` 不得调用面向宿主 shell 的 `prepare_sandbox_command()` 或 +`convert_virtual_paths_in_command()`。容器内 `/workspace` 是真实路径,必须原样保留: + +```text +/workspace/report.html -> /workspace/report.html +``` + +容器命令只经过一个很小的 container command validator:保留命令长度、灾难性删除和平台禁用命令检查, +不做路径改写,不把 `/skills`、`/memories` 解析成宿主路径,也不把任意绝对路径转换为相对路径。即使命令 +引用 `/Users/**` 或 `/etc/**`,它看到的也只是容器镜像文件系统;宿主隔离由 mount namespace 保证。 + +容器命令可以修改 `/workspace`,修改结果立即出现在同一个会话目录中,文件列表无需额外提交即可看到。 + +`fsize` 只能限制单文件,不能限制大量小文件的总量。scope 总容量由部署负责:`workspace_root` 必须位于 +有明确容量上限、监控和告警的独立卷,生产环境应使用该文件系统提供的 quota。该限制是本版本明确接受的 +运维边界,不引入 tmpfs 输出区、Artifact 提交或应用层容量账本。 + +## 7. 工具边界 + +### 7.1 纳入同一目录的工具 + +以下 Web 能力必须使用当前 scope 的 rooted backend: + +- `ls`、`read_file`、`write_file`、`edit_file` 等文件工具; +- 文档、代码和数据处理工具对工作文件的访问; +- `execute` 的工作目录挂载; +- Gateway 文件列表、上传、下载和预览。 + +### 7.2 暂时禁用的旁路 + +现有 Web 路径如果不能接收当前 scope,就不能继续暴露: + +- 直接使用宿主 `Popen` 的后台命令; +- 能选择任意宿主目录的本地 shell fallback; +- 能自行创建另一套 backend 的子代理执行路径; +- 能访问宿主文件系统的 stdio MCP 工具。 + +本次只做关闭或从 Web 工具集中移除,不为这些能力设计新协议。后续确有需求时,再让它们显式接收同一个 +scope backend。 + +`/workspace` 是唯一的用户文件命名空间。现有 `/skills` 和 `/memories` 是平台能力命名空间: + +- 不属于 Thread 会话文件; +- 不出现在 Gateway 文件列表、下载、预览或归档接口; +- 不挂载到 `execute` 容器; +- 不能接收或返回宿主物理路径; +- `/skills` 只读,Memory 继续通过现有受控 backend 访问。 + +审批状态和普通数据库工具同样不是会话文件系统,不需要伪装成 `/workspace`。 + +## 8. 审批关系 + +审批只决定工具调用是否暂停等待用户确认: + +```text +tool request + -> review middleware decides auto/manual approval + -> approved request enters the same scoped backend/container +``` + +因此: + +- 自动审批不会扩大可访问目录; +- 人工审批也不能允许访问 scope 外路径; +- 路径越界在审批之前或执行入口处直接拒绝; +- 无论自动或人工,最终都使用同一个 `scope_id`。 + +本方案不修改已有自动审批设计。 + +## 9. 生命周期 + +只保留最小生命周期,并复用 Registry 已有状态: + +1. 创建 Thread 时创建 `scope_id` 和会话目录,并写入 `threads.workspace_dir`。 +2. Thread 后续 Run 复用同一个 `scope_id` 和目录。 +3. 每次文件/命令请求验证当前用户拥有该 Thread。 +4. 删除 Thread 时,Runtime 把现有 scope 转为 `deleting`,使新的工具操作立即失败。 +5. Gateway 取消该 Thread 的活动 Run,然后 Runtime 删除 scope 目录并转为 `deleted`。 +6. 删除接口必须幂等;中途失败可以用同一 `thread_id` 重试清理。 + +删除过程中仍在运行的单次容器最多继续访问已经挂载的旧目录 inode,不能访问其他 scope;删除完成后不再 +存在可由 Thread 重新解析的路径。无需为此引入常驻容器管理器。 + +不新增 lease、fencing token、owner 状态机或 Run 专属文件目录。现有 scope registry 只承担 +`thread_id -> scope_id -> physical root` 的绑定、所有权校验以及已有的 `deleting/deleted` 状态转换。 + +## 10. 现有代码调整 + +### 10.1 EvoScientist Web Runtime + +在 `EvoScientist/workspace_scope.py` 中: + +- 保留并收敛 `ScopedContainerBackend`; +- 所有文件方法绑定到当前 scope 的 `files` 根目录; +- `execute` 只调用容器 executor; +- 删除 Web 对 `CustomSandboxBackend` 宿主 shell 行为的委托; +- 删除 Web 宿主 shell fallback; +- 统一输出虚拟路径 `/workspace/**`。 + +新增一个小型 `EvoScientist/workspace_files.py`: + +- 实现虚拟路径规范化和基于目录描述符的 `read/write/edit/ls/glob/grep/upload/download`; +- `glob/grep` 只能使用不跟随符号链接的安全 walker,不能回退 `Path.rglob()`; +- `write/edit/upload` 使用已验证父目录描述符和相对临时文件完成原子替换; +- `download` 返回已打开 file handle,调用方负责在流结束后关闭; +- Runtime 文件 backend 与 Gateway 会话文件路由共同复用该实现; +- 不包含 Storage、Artifact、审批或 Run 状态逻辑。 + +新增轻量 `ScopedFilesystemBackend` 适配 DeepAgents backend 协议。`ScopedContainerBackend` 组合该文件 +backend 并只覆盖 `execute`,不再通过 `__getattr__` 把文件方法委托给 `CustomSandboxBackend`。 + +在 `EvoScientist/scope_registry.py` 中: + +- 保留 Thread、scope 与物理目录的简单绑定; +- Web 操作只接受当前 Thread 的有效 scope; +- 复用已有 `deleting/deleted` 状态,不增加新的生命周期状态或租约协议。 + +在 Runtime 内部 workspace scope HTTP 接口中: + +- 保留现有 `ensure_scope(thread_id)`; +- 增加幂等的按 Thread 删除 scope 接口,内部复用已有状态转换; +- 不向浏览器暴露该接口,也不返回物理路径。 + +在 Web graph/tool 装配处: + +- 所有文件工具注入同一个 scoped backend; +- 移除不能遵守该 backend 的 Web 执行型工具; +- CLI 装配不变。 + +### 10.2 Ai4Sci-Web Gateway + +在 Gateway workspace/file 服务中: + +- 创建 Thread 时调用 `ensure_scope(thread_id)`,并把推导出的 scope `files` 目录写入现有 + `threads.workspace_dir`;scope 创建失败则 Thread 创建失败; +- Thread 数据库写入失败时调用幂等 scope 删除接口清理刚创建的目录; +- 鉴权后从 Thread 记录取得 `workspace_dir`,并再次校验其位于配置的 conversations 根目录内; +- 把上传、列表、下载、预览和归档改为同一 rooted file accessor; +- API 和响应只使用 `/workspace/**` 虚拟路径,不接受或返回物理路径;路由内部去掉固定前缀后再调用 + accessor; +- 对 Web Thread 删除 `storage_service.backend_name` 文件分支、附件物化和 Artifact 注册路径; +- 保留现有上传校验和配额检查,但字节直接写入 `workspace_dir/uploads/`; +- 下载和预览使用 accessor 返回的已打开 file handle,不再使用会重新按路径打开的 `FileResponse(path)`; +- 删除 Thread 时调用 Runtime 的幂等 scope 删除接口,不再直接信任数据库路径执行裸 `rmtree`。 + +在 Recoverable Run coordinator 中: + +- 保留 `workspace_state`、workspace lease、`bind_run()`、`mark_workspace_bound()` 和现有派发条件; +- `_prepare_workspace_once()` 使用 Thread rooted accessor 原位验证 `body.files`; +- 删除 `storage_service is None` 错误分支、`open_workspace_input()` 和 `client.materialize()`; +- 验证失败沿用现有 `workspace_prepare_retry()` 和 `failed` 处理,不新增状态。 + +在上传配额统计中: + +- 查询该用户未删除的 Web Thread,从每条记录取得 `workspace_dir`; +- 校验目录位于 conversations 根下,再通过 rooted accessor 统计 `/workspace/uploads/**`; +- 不再依赖 FileGateway 元数据统计 Web Thread 上传,也不统计模型生成文件; +- 开发阶段不再统计或兼容旧全局上传目录; +- 不新增配额表或文件索引。 + +Gateway 与 Runtime 必须运行在同一 POSIX 主机、同一 mount namespace、同一服务 UID/GID,并读取同一个 +本机绝对 `workspace_root`。两个启动脚本显式导出相同的 `EVOSCIENTIST_WORKSPACE_DIR`;Runtime 继续用 +`0700` 创建 scope,命令容器使用相同的数字 UID/GID。Thread 创建后 Gateway 必须在 Runtime 刚创建的 +scope 中完成一次随机探针文件的创建、读取和删除,该探针是实际 mount/权限的一致性硬检查。任一步失败 +则创建 Thread 失败,不回退旧 workspace。Windows 和不同 UID/ACL 部署不属于本版本支持范围。 + +### 10.3 Ai4Sci-Web Frontend + +- Web 会话文件以 `thread_id + virtual_path` 标识,不依赖 `file_id`; +- 上传、文件树、下载、预览和消息附件统一使用 `/workspace/**`; +- 不读取或显示物理 `workspace_dir`; +- 不改变工具调用分组和自动审批 UI。 + +### 10.4 不修改 + +- 自动审批 middleware 和 review rules; +- Checkpoint、Recoverable Run 状态枚举和数据库约束; +- Run completed 的投影流程; +- Artifact 数据库模型,但 Web 会话文件不再写入该模型; +- CLI backend 和 CLI 文件语义; +- HTML 预览的 CSP 与独立安全加固。 + +## 11. 实施顺序 + +### 阶段 A:统一目录 + +1. Thread 创建时调用 `ensure_scope` 并写入新的 `workspace_dir`。 +2. 实现基于目录描述符且禁止符号链接的 rooted file accessor。 +3. 文件工具切换到 rooted backend。 +4. Gateway 上传、列表、下载、预览和归档切换到同一 accessor。 +5. 前端文件身份切换为 `thread_id + /workspace/**`。 +6. Web 会话文件退出通用 Storage Service 分支。 +7. Run workspace 准备阶段从附件复制改为 scope 原位验证,并继续进入 `bound`。 + +### 阶段 B:关闭执行旁路 + +1. `execute` 只允许 `ScopedContainerBackend`。 +2. 容器命令改用不重写路径的 container command validator。 +3. 删除 Web 宿主 shell fallback。 +4. 从 Web 工具集移除后台宿主命令及其他文件系统旁路。 +5. Docker 或镜像不可用时 fail closed。 + +### 阶段 C:删除旧逻辑 + +1. 删除 Gateway workspace 到 scope 的附件复制。 +2. 删除未被使用的 multi-root/global/peer-root Web 路径。 +3. 接入幂等 scope 删除接口。 +4. 删除旧 Web workspace 数据;开发阶段不迁移。 + +任何中间版本都不能出现“文件工具已切换,但 `execute` 仍运行宿主 shell”的组合。阶段 A 与阶段 B 必须 +在同一次 Web 发布中启用。 + +## 12. 验收清单 + +### 12.1 功能 + +- 上传文件后,`ls /workspace` 能立即看到。 +- 第一次 Run 之前上传的文件与 Run 内看到的文件来自同一目录。 +- 带附件的 Run 无需复制文件即可从 `workspace_state` 进入 `bound` 并正常派发。 +- 文件工具写入 `/workspace/result.html` 后,页面文件列表能立即看到。 +- 页面可以下载和预览该文件,内容与工具读取一致。 +- Thread API、文件 API、消息和页面均不出现宿主物理目录。 +- `execute` 创建或修改文件后,无需 Artifact 提交步骤即可在页面看到。 +- 同一 Thread 的后续 Run 能继续访问已有文件。 +- CLI 行为没有变化。 + +### 12.2 隔离 + +- 文件工具和 Gateway 文件 API 拒绝 `..`、宿主绝对路径及所有符号链接。 +- 容器并发替换目录项时,Gateway 仍不能读到 scope 外文件。 +- Thread A 不能读取 Thread B 的文件。 +- 容器不能读取宿主 home、项目源码、其他 scope 或宿主密钥环境。 +- 容器只对当前 `/workspace` 具有持久写权限。 +- Gateway 上传的文件可由容器修改,容器生成的文件可由 Gateway 读取、覆盖和删除。 +- 命令中的 `/workspace/a` 在容器内保持 `/workspace/a`,不会变成 `/workspace/workspace/a`。 +- 单个文件超过配置的 `fsize` 上限时命令失败,不产生可绕过上限的完整文件。 +- Docker 不可用时命令失败,且没有宿主 shell fallback。 +- 后台命令、子代理和 MCP 不存在绕过当前 scope 的宿主文件访问路径。 + +### 12.3 审批一致性 + +- 同一工具在自动审批和人工审批下访问范围完全一致。 +- 越界请求即使被人工批准也不能执行。 +- review rules 的变化不影响路径隔离测试结果。 + +## 13. 必需测试 + +只新增与本次边界直接相关的测试: + +1. 虚拟路径解析:正常路径、`..`、绝对路径、空字节和编码边界。 +2. 符号链接:任意路径层级的符号链接均拒绝,并覆盖 resolve/open 并发替换测试。 +3. 跨 Thread:相同用户和不同用户均不能通过路径或 `scope_id` 越权。 +4. 容器隔离:宿主 sentinel、home、项目源码和敏感环境不可见。 +5. 生成文件闭环:容器写文件后,列表、读取、下载和预览读取同一字节。 +6. executor 故障:Docker/镜像缺失时失败且不回退。 +7. 审批一致性:自动与人工审批执行同一组隔离用例。 +8. CLI 回归:现有 CLI 文件与执行测试继续通过。 +9. 创建时序:第一次 Run 前上传、Run 内读取和页面下载读取同一字节。 +10. 删除:活动 Run 存在时删除 Thread,后续工具操作失败且清理可以幂等重试。 +11. 响应竞态:安全打开后替换目录项,下载、预览和归档仍读取原文件描述符且不越界。 +12. 信息泄漏:Thread API、文件 API、SSE、错误和模型消息均不包含 workspace 物理前缀。 +13. Run 附件:无 Storage Service 时,附件原位验证后仍能 bind、进入 `bound` 并派发;文件缺失则沿用 + workspace prepare 失败流程。 +14. 容器路径:执行读取 `/workspace/probe.txt`,断言没有产生嵌套 `workspace/workspace`。 +15. 文件工具全集:`read/write/edit/ls/glob/grep/upload/download` 分别覆盖路径逃逸和符号链接竞态。 +16. 上传配额:多个 Thread 的 `uploads/` 会合并计入同一用户额度,模型输出不计入上传额度。 +17. 共享权限:Runtime 创建 `0700` scope 后,Gateway 探针创建、读取和删除成功;UID 或根目录不一致时 + fail closed。 +18. 容器身份:Gateway 上传文件后容器可以修改;容器生成文件后 Gateway 可以读取、覆盖和删除;生成 + 文件的宿主 UID/GID 与服务进程一致。 +19. 文件大小:容器写入超过 `fsize` 限制的单个文件时执行失败;测试不把该限制误认为 scope 总容量 + quota。 + +## 14. 明确删除的原方案设计 + +以下内容不属于“一个虚拟目录”的必要条件,全部从方案移除: + +- Gateway/Runtime 双权威与文件双写; +- Artifact pending batch、commit、abort 和查询协议; +- Artifact 数据库状态迁移和版本状态机; +- Run completed 与 Artifact committed 的事务绑定; +- 每个 Run 的 uploads/artifacts/work/tmpfs 分区; +- 文件 freeze、drain barrier 和输出流; +- 输入 manifest pinning 和恢复快照; +- 常驻容器、container supervisor、lease 和 fencing; +- local/remote executor provider 抽象; +- 为分布式部署设计的认证文件流; +- 独立 HTML preview origin 改造; +- 历史 scope 和 workspace 迁移兼容。 + +这些能力如果未来出现独立的业务需求,应分别立项,不能重新塞回虚拟目录基础改造。 + +## 15. 最终判定 + +该方案解决的是两个直接问题: + +1. 所有 Web 文件操作是否只认当前 Thread 的 `/workspace`。 +2. 模型执行命令时是否无法越过这个目录访问宿主机。 + +第一个问题由“单一会话目录 + rooted path resolver”解决;第二个问题由“仅挂载该目录的容器”解决。 +除此以外不增加新的系统职责。 + +## 16. 实施与验证记录 + +本方案已在 EvoScientist Runtime、Ai4Sci-Web Gateway 和 Frontend 同一次改造中落地: + +- Runtime 新增 descriptor-rooted `workspace_files.py`,文件工具全集与 `execute` 共用当前 scope; +- `ScopedContainerBackend` 只挂载当前 scope 到 `/workspace`,并关闭宿主 shell、MCP、后台执行和子代理旁路; +- Gateway 新增 `ConversationWorkspace`,Thread 创建、上传、列表、下载、预览、归档、配额、Run 附件准备和 + Thread 删除均使用同一物理目录; +- Recoverable Run 保留原 `workspace_state=bound` 门槛,但附件改为 `/workspace/uploads/**` 原位验证; +- Frontend 文件身份统一为 `thread_id + /workspace/**`,预览和下载路由不再产生双斜杠; +- `start-gateway.sh` 与 `start-langgraph.sh` 显式使用同一个 workspace 根。 + +自动化验证结果: + +- EvoScientist:`3023 passed, 10 skipped`; +- Ai4Sci-Web Gateway:`688 passed, 4 skipped`; +- Frontend Vitest:`169 passed`; +- Frontend TypeScript:通过; +- 本次涉及文件的 Ruff 与 `git diff --check`:通过。 + +Playwright 真实页面验证完成以下链路: + +```text +create Thread 200 -> workspace_dir=/workspace +upload 200 -> virtual_path=/workspace/uploads/, file_id=null +create Run 201 -> body.files 复用同一 virtual_path +read_file -> 返回上传文件原始第一行 +list / preview / download -> 内容一致 +delete Thread 200 -> Registry state=deleted -> scope 物理目录不存在 +``` + +调试过程中,Gateway 与 Runtime workspace 根不一致曾使 Thread 创建返回 +`WORKSPACE_SCOPE_INVALID`。该 fail-closed 行为符合设计,并由启动脚本统一根目录后通过同一创建探针。 diff --git a/runtime/native-sandbox/README.md b/runtime/native-sandbox/README.md new file mode 100644 index 0000000..c1e995d --- /dev/null +++ b/runtime/native-sandbox/README.md @@ -0,0 +1,24 @@ +# EvoScientist Native Sandbox Runtime + +This directory pins the native Web command sandbox dependency. It is a deployment +dependency, not an npm application and not a CLI execution backend. + +Install exactly what is recorded in the lockfile during environment preparation: + +```bash +npm ci --omit=dev --prefix runtime/native-sandbox +``` + +Web full startup then performs a real fail-closed isolation preflight. It never +downloads dependencies and never falls back to Docker or the host shell. + +Platform tools: + +- macOS: Node 20.11+, `bash`, `rg`, `python3`, `pandoc`, and `/usr/bin/sandbox-exec` +- Linux: Node 20.11+, `bash`, `rg`, `python3`, `pandoc`, `bwrap`, and `socat` + +Run the host-specific black-box suite after installation: + +```bash +EVOSCIENTIST_RUN_NATIVE_SANDBOX_TESTS=1 uv run pytest -q tests/test_native_sandbox_integration.py +``` diff --git a/runtime/native-sandbox/package-lock.json b/runtime/native-sandbox/package-lock.json new file mode 100644 index 0000000..c7507ce --- /dev/null +++ b/runtime/native-sandbox/package-lock.json @@ -0,0 +1,66 @@ +{ + "name": "evoscientist-native-sandbox-runtime", + "version": "1.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "evoscientist-native-sandbox-runtime", + "version": "1.0.0", + "dependencies": { + "@anthropic-ai/sandbox-runtime": "0.0.73" + } + }, + "node_modules/@anthropic-ai/sandbox-runtime": { + "version": "0.0.73", + "resolved": "https://registry.npmjs.org/@anthropic-ai/sandbox-runtime/-/sandbox-runtime-0.0.73.tgz", + "integrity": "sha512-F608iUirrCqwvInZYGRRgJWDQj0tt6fNVE9aPagpotLJ5LhC4JbrMFIIZww5MFjb+HRCkpE0+xdI79c30tdVYg==", + "license": "Apache-2.0", + "dependencies": { + "@pondwader/socks5-server": "^1.0.10", + "commander": "^12.1.0", + "node-forge": "^1.4.0", + "zod": "^3.24.1" + }, + "bin": { + "srt": "dist/cli.js" + }, + "engines": { + "node": ">=20.11.0" + } + }, + "node_modules/@pondwader/socks5-server": { + "version": "1.0.10", + "resolved": "https://registry.npmjs.org/@pondwader/socks5-server/-/socks5-server-1.0.10.tgz", + "integrity": "sha512-bQY06wzzR8D2+vVCUoBsr5QS2U6UgPUQRmErNwtsuI6vLcyRKkafjkr3KxbtGFf9aBBIV2mcvlsKD1UYaIV+sg==", + "license": "MIT" + }, + "node_modules/commander": { + "version": "12.1.0", + "resolved": "https://registry.npmjs.org/commander/-/commander-12.1.0.tgz", + "integrity": "sha512-Vw8qHK3bZM9y/P10u3Vib8o/DdkvA2OtPtZvD871QKjy74Wj1WSKFILMPRPSdUSx5RFK1arlJzEtA4PkFgnbuA==", + "license": "MIT", + "engines": { + "node": ">=18" + } + }, + "node_modules/node-forge": { + "version": "1.4.0", + "resolved": "https://registry.npmjs.org/node-forge/-/node-forge-1.4.0.tgz", + "integrity": "sha512-LarFH0+6VfriEhqMMcLX2F7SwSXeWwnEAJEsYm5QKWchiVYVvJyV9v7UDvUv+w5HO23ZpQTXDv/GxdDdMyOuoQ==", + "license": "(BSD-3-Clause OR GPL-2.0)", + "engines": { + "node": ">= 6.13.0" + } + }, + "node_modules/zod": { + "version": "3.25.76", + "resolved": "https://registry.npmjs.org/zod/-/zod-3.25.76.tgz", + "integrity": "sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==", + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/colinhacks" + } + } + } +} diff --git a/runtime/native-sandbox/package.json b/runtime/native-sandbox/package.json new file mode 100644 index 0000000..04e7db4 --- /dev/null +++ b/runtime/native-sandbox/package.json @@ -0,0 +1,8 @@ +{ + "name": "evoscientist-native-sandbox-runtime", + "private": true, + "version": "1.0.0", + "dependencies": { + "@anthropic-ai/sandbox-runtime": "0.0.73" + } +} diff --git a/tests/test_dynamic_review_middleware.py b/tests/test_dynamic_review_middleware.py new file mode 100644 index 0000000..353fdc5 --- /dev/null +++ b/tests/test_dynamic_review_middleware.py @@ -0,0 +1,374 @@ +from __future__ import annotations + +from types import SimpleNamespace +from unittest.mock import patch + +import pytest +from langchain.agents import create_agent +from langchain.agents.middleware import HumanInTheLoopMiddleware +from langchain_core.language_models.fake_chat_models import FakeMessagesListChatModel +from langchain_core.messages import AIMessage, HumanMessage, ToolMessage +from langchain_core.tools import tool + +import EvoScientist.middleware.dynamic_review as dynamic_review +from EvoScientist.middleware.dynamic_review import ( + AutoReviewVerificationError, + DynamicReviewMiddleware, +) + + +def _config(run_id: str, mode: str, revision: int = 3) -> dict: + return { + "configurable": { + "ai4sci_run_id": run_id, + "ai4sci_review_mode": { + "gateway_url": "http://127.0.0.1:8065", + "requested_mode": mode, + "review_mode_revision": revision, + "run_id": run_id, + "envelope_digest": "d" * 64, + "envelope_signature": "s" * 64, + }, + } + } + + +def _middleware() -> DynamicReviewMiddleware: + return DynamicReviewMiddleware(interrupt_on={"execute": False}) + + +class _ToolCallingFakeModel(FakeMessagesListChatModel): + def bind_tools(self, _tools, *, tool_choice=None, **_kwargs): + return self + + +def test_manual_before_agent_overwrites_prior_auto_state(monkeypatch): + monkeypatch.setattr( + dynamic_review, "get_config", lambda: _config("run-manual", "manual") + ) + middleware = _middleware() + + update = middleware.before_agent( + { + "messages": [], + "_verified_review_mode": { + "execution_run_id": "run-old", + "mode": "auto", + "revision": 2, + }, + }, + SimpleNamespace(), + ) + + assert update == { + "_verified_review_mode": { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": "run-manual", + "mode": "manual", + "revision": 3, + } + } + + +@pytest.mark.asyncio +async def test_auto_before_agent_resolves_once_and_after_model_skips_hitl(monkeypatch): + monkeypatch.setattr( + dynamic_review, "get_config", lambda: _config("run-auto", "auto") + ) + calls = 0 + + async def resolve(run_id, review): + nonlocal calls + calls += 1 + assert run_id == "run-auto" + assert review["requested_mode"] == "auto" + return { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": run_id, + "mode": "auto", + "revision": 3, + } + + monkeypatch.setattr(dynamic_review, "_resolve_async", resolve) + middleware = _middleware() + update = await middleware.abefore_agent({"messages": []}, SimpleNamespace()) + + with patch.object(HumanInTheLoopMiddleware, "after_model") as parent: + assert ( + middleware.after_model({"messages": [], **update}, SimpleNamespace()) + is None + ) + assert calls == 1 + parent.assert_not_called() + + +@pytest.mark.asyncio +async def test_real_langgraph_auto_mode_executes_tool_without_interrupt(monkeypatch): + calls = [] + + @tool + def execute(command: str) -> str: + """Execute a test command.""" + calls.append(command) + return "ok" + + async def resolve(run_id, _review): + return { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": run_id, + "mode": "auto", + "revision": 3, + } + + monkeypatch.setattr(dynamic_review, "_resolve_async", resolve) + agent = create_agent( + model=_ToolCallingFakeModel( + responses=[ + AIMessage( + content="", + tool_calls=[ + { + "name": "execute", + "args": {"command": "pwd"}, + "id": "call-1", + "type": "tool_call", + } + ], + ), + AIMessage(content="done"), + ] + ), + tools=[execute], + middleware=[ + DynamicReviewMiddleware( + interrupt_on={"execute": {"allowed_decisions": ["approve", "reject"]}} + ) + ], + ) + + result = await agent.ainvoke( + {"messages": [HumanMessage(content="run pwd")]}, + config=_config("run-auto", "auto"), + ) + + assert calls == ["pwd"] + assert any( + isinstance(message, ToolMessage) and message.tool_call_id == "call-1" + for message in result["messages"] + ) + + +def test_manual_after_model_uses_existing_hitl_even_for_resume_child(monkeypatch): + monkeypatch.setattr( + dynamic_review, "get_config", lambda: _config("child-run", "manual") + ) + middleware = _middleware() + state = { + "messages": [], + "_verified_review_mode": { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": "parent-run", + "mode": "manual", + "revision": 2, + }, + } + + with patch.object( + HumanInTheLoopMiddleware, "after_model", return_value={"manual": True} + ) as parent: + assert middleware.after_model(state, SimpleNamespace()) == {"manual": True} + parent.assert_called_once() + + +def test_auto_after_model_requires_current_run_context(monkeypatch): + monkeypatch.setattr(dynamic_review, "get_config", lambda: {"configurable": {}}) + middleware = _middleware() + + with pytest.raises(AutoReviewVerificationError, match="REVIEW_MODE_RUN_MISMATCH"): + middleware.after_model( + { + "messages": [], + "_verified_review_mode": { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": "run-old", + "mode": "auto", + "revision": 3, + }, + }, + SimpleNamespace(), + ) + + +def test_auto_after_model_resume_reverifies_injected_auto_context(monkeypatch): + monkeypatch.setattr( + dynamic_review, "get_config", lambda: _config("child-run", "auto") + ) + calls = [] + + def resolve(run_id, review): + calls.append(run_id) + assert review["requested_mode"] == "auto" + return { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": run_id, + "mode": "auto", + "revision": review["review_mode_revision"], + } + + monkeypatch.setattr(dynamic_review, "_resolve_sync", resolve) + middleware = _middleware() + + with patch.object(HumanInTheLoopMiddleware, "after_model") as parent: + assert ( + middleware.after_model( + { + "messages": [], + "_verified_review_mode": { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": "parent-run", + "mode": "auto", + "revision": 3, + }, + }, + SimpleNamespace(), + ) + is None + ) + parent.assert_not_called() + assert calls == ["child-run"] + + +def test_auto_after_model_resume_downgrades_to_hitl_when_manual(monkeypatch): + monkeypatch.setattr( + dynamic_review, "get_config", lambda: _config("child-run", "manual") + ) + middleware = _middleware() + + with patch.object( + HumanInTheLoopMiddleware, "after_model", return_value={"manual": True} + ) as parent: + assert middleware.after_model( + { + "messages": [], + "_verified_review_mode": { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": "parent-run", + "mode": "auto", + "revision": 3, + }, + }, + SimpleNamespace(), + ) == {"manual": True} + parent.assert_called_once() + + +def test_auto_after_model_resume_falls_back_to_hitl_on_verify_failure(monkeypatch): + monkeypatch.setattr( + dynamic_review, "get_config", lambda: _config("child-run", "auto") + ) + + def resolve(_run_id, _review): + raise AutoReviewVerificationError("AUTO_REVIEW_VERIFICATION_FAILED") + + monkeypatch.setattr(dynamic_review, "_resolve_sync", resolve) + middleware = _middleware() + + with patch.object( + HumanInTheLoopMiddleware, "after_model", return_value={"manual": True} + ) as parent: + assert middleware.after_model( + { + "messages": [], + "_verified_review_mode": { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": "parent-run", + "mode": "auto", + "revision": 3, + }, + }, + SimpleNamespace(), + ) == {"manual": True} + parent.assert_called_once() + + +def test_auto_after_model_inherits_parent_state_for_resume_child(monkeypatch): + # A legacy resume child has no ai4sci_review_mode injected by the gateway; + # the verified auto state is inherited from the parent run and must be + # reused for the remainder of the turn. + monkeypatch.setattr( + dynamic_review, + "get_config", + lambda: {"configurable": {"ai4sci_run_id": "child-run"}}, + ) + middleware = _middleware() + + with patch.object(HumanInTheLoopMiddleware, "after_model") as parent: + assert ( + middleware.after_model( + { + "messages": [], + "_verified_review_mode": { + "protocol": "verified-review-mode-state-v1", + "execution_run_id": "parent-run", + "mode": "auto", + "revision": 3, + }, + }, + SimpleNamespace(), + ) + is None + ) + parent.assert_not_called() + + +def test_auto_response_must_match_signed_request(): + review = _config("run-auto", "auto")["configurable"]["ai4sci_review_mode"] + with pytest.raises( + AutoReviewVerificationError, match="AUTO_REVIEW_RESPONSE_INVALID" + ): + dynamic_review._validated_auto_state( + "run-auto", + review, + { + "protocol": "resolved-review-mode-v1", + "run_id": "run-auto", + "envelope_digest": "d" * 64, + "mode": "manual", + "revision": 3, + }, + ) + + +@pytest.mark.parametrize("auto_approve", [False, True]) +def test_default_web_graph_installs_only_dynamic_hitl(monkeypatch, auto_approve): + import EvoScientist.EvoScientist as agent_module + + captured = {} + + class Agent: + def with_config(self, _config): + return self + + cfg = SimpleNamespace(auto_approve=auto_approve, recursion_limit=50) + monkeypatch.setattr(agent_module, "_EvoScientist_agent", None) + monkeypatch.setattr(agent_module, "_ensure_config", lambda: cfg) + monkeypatch.setattr(agent_module, "_get_default_backend", lambda: object()) + monkeypatch.setattr(agent_module, "_get_default_middleware", lambda: []) + monkeypatch.setattr( + agent_module, + "load_mcp_and_build_kwargs", + lambda _backend, middleware, **_kwargs: captured.setdefault( + "kwargs", {"middleware": middleware} + ), + ) + monkeypatch.setattr( + agent_module, "_apply_budgeted_skill_context", lambda kwargs, _backend: kwargs + ) + monkeypatch.setattr("deepagents.create_deep_agent", lambda **_kwargs: Agent()) + monkeypatch.delenv("EVOSCIENTIST_DEPLOY_MODE", raising=False) + + agent_module._get_default_agent() + + middleware = captured["kwargs"]["middleware"] + assert sum(isinstance(item, DynamicReviewMiddleware) for item in middleware) == 1 + assert not any(type(item) is HumanInTheLoopMiddleware for item in middleware) diff --git a/tests/test_langgraph_dev_http.py b/tests/test_langgraph_dev_http.py index 4b83250..ef97585 100644 --- a/tests/test_langgraph_dev_http.py +++ b/tests/test_langgraph_dev_http.py @@ -210,8 +210,8 @@ def test_workspace_scope_routes_require_service_token(): assert response.status_code == 401 -def test_workspace_materialize_rejects_path_escape(): - from EvoScientist.langgraph_dev.http import _materialize_target +def test_workspace_path_rejects_escape(): + from EvoScientist.workspace_files import normalize_workspace_path - with pytest.raises(ValueError, match="uploads"): - _materialize_target(str(uuid4()), "../secret.txt") + with pytest.raises(ValueError, match="workspace"): + normalize_workspace_path("/workspace/../secret.txt") diff --git a/tests/test_native_sandbox.py b/tests/test_native_sandbox.py new file mode 100644 index 0000000..f1235f5 --- /dev/null +++ b/tests/test_native_sandbox.py @@ -0,0 +1,107 @@ +from __future__ import annotations + +import json +from pathlib import Path + +import pytest + +import EvoScientist.native_sandbox as sandbox + + +def _installation(tmp_path: Path) -> sandbox.NativeSandboxInstallation: + package = tmp_path / "package" + package.mkdir() + srt = tmp_path / "srt" + srt.touch(mode=0o700) + return sandbox.NativeSandboxInstallation( + srt=srt, + package_root=package, + path_env="/usr/bin:/bin", + system_read_paths=("/usr", "/bin", "/dev/null"), + ) + + +def test_policy_denies_root_and_only_writes_scope_and_command_tmp(tmp_path: Path): + files = tmp_path / "files" + command_tmp = tmp_path / "runtime" / "tmp" / "run" + files.mkdir() + command_tmp.mkdir(parents=True) + + policy = sandbox._sandbox_settings(_installation(tmp_path), files, command_tmp) + + assert policy["filesystem"]["denyRead"] == ["/"] + assert str(files) in policy["filesystem"]["allowRead"] + assert str(command_tmp) in policy["filesystem"]["allowRead"] + assert policy["filesystem"]["allowWrite"] == [ + str(files), + str(command_tmp), + "/dev/null", + ] + assert policy["filesystem"]["denyWrite"] == [ + "/tmp/claude", + "/private/tmp/claude", + "/dev/tty", + "/dev/dtracehelper", + "/dev/autofs_nowait", + ] + assert policy["network"]["allowedDomains"] == [] + assert policy["network"]["allowAllUnixSockets"] is False + assert policy["allowAppleEvents"] is False + assert "control" not in json.dumps(policy) + + +def test_clean_environment_does_not_inherit_secrets(tmp_path: Path, monkeypatch): + command_tmp = tmp_path / "tmp" + (command_tmp / "home").mkdir(parents=True) + (command_tmp / "tmp").mkdir() + monkeypatch.setenv("OPENAI_API_KEY", "secret") + + environment = sandbox._clean_environment(_installation(tmp_path), command_tmp) + + assert set(environment) == {"PATH", "HOME", "TMPDIR", "WORKSPACE", "LANG", "LC_ALL"} + assert "OPENAI_API_KEY" not in environment + assert environment["WORKSPACE"] == "." + + +def test_executor_requires_control_directory_outside_files(tmp_path: Path): + files = tmp_path / "files" + files.mkdir() + runtime = files / "runtime" + runtime.mkdir() + + with pytest.raises(sandbox.NativeSandboxUnavailable): + sandbox.NativeSandboxExecutor(files, runtime) + + +def test_readiness_is_cached_and_failure_is_fail_closed(monkeypatch): + sandbox._reset_native_sandbox_readiness_for_tests() + calls = {"install": 0, "preflight": 0} + + def install(): + calls["install"] += 1 + return object() + + def preflight(_installation): + calls["preflight"] += 1 + + monkeypatch.setattr(sandbox, "_assert_install_contract", install) + monkeypatch.setattr(sandbox, "_run_preflight", preflight) + sandbox.ensure_native_sandbox_ready() + sandbox.ensure_native_sandbox_ready() + assert calls == {"install": 1, "preflight": 1} + + sandbox._reset_native_sandbox_readiness_for_tests() + monkeypatch.setattr( + sandbox, + "_run_preflight", + lambda _installation: (_ for _ in ()).throw( + sandbox.NativeSandboxUnavailable("failed once") + ), + ) + with pytest.raises(sandbox.NativeSandboxUnavailable, match="failed once"): + sandbox.ensure_native_sandbox_ready() + with pytest.raises(sandbox.NativeSandboxUnavailable, match="failed once"): + sandbox.ensure_native_sandbox_ready() + assert calls["install"] == 2 + + sandbox._reset_native_sandbox_readiness_for_tests() diff --git a/tests/test_native_sandbox_integration.py b/tests/test_native_sandbox_integration.py new file mode 100644 index 0000000..8547384 --- /dev/null +++ b/tests/test_native_sandbox_integration.py @@ -0,0 +1,139 @@ +from __future__ import annotations + +import asyncio +import os +import shlex +import time +from pathlib import Path + +import pytest + +from EvoScientist.native_sandbox import ( + NativeSandboxExecutor, + NativeWorkspaceBackend, + ensure_native_sandbox_ready, +) + +pytestmark = pytest.mark.skipif( + os.getenv("EVOSCIENTIST_RUN_NATIVE_SANDBOX_TESTS") != "1", + reason="set EVOSCIENTIST_RUN_NATIVE_SANDBOX_TESTS=1 on a supported host", +) + + +@pytest.fixture +def executor(tmp_path: Path) -> NativeSandboxExecutor: + ensure_native_sandbox_ready() + files = tmp_path / "files" + runtime = tmp_path / "runtime" + files.mkdir() + runtime.mkdir() + return NativeSandboxExecutor(files, runtime, timeout=5) + + +def test_real_sandbox_scrubs_secrets_and_keeps_command_tmp_private( + executor: NativeSandboxExecutor, monkeypatch: pytest.MonkeyPatch +): + monkeypatch.setenv("OPENAI_API_KEY", "must-not-leak") + result = executor.execute( + 'env | sort; printf result > result.txt; printf temp > "$TMPDIR/value"' + ) + + assert result.exit_code == 0 + assert "OPENAI_API_KEY" not in result.output + assert "PROXY_PASSWORD" not in result.output + assert "HTTP_PROXY" not in result.output + assert (executor.files_dir / "result.txt").read_text(encoding="utf-8") == "result" + assert not list((executor.runtime_dir / "tmp").glob("*/tmp/value")) + + +def test_real_sandbox_drains_but_caps_output( + executor: NativeSandboxExecutor, monkeypatch: pytest.MonkeyPatch +): + monkeypatch.setenv("EVOSCIENTIST_SANDBOX_MAX_OUTPUT_BYTES", "64") + code = "import sys; sys.stdout.write('x' * 10000)" + + result = executor.execute(f"python3 -c {shlex.quote(code)}") + + assert result.exit_code == 0 + assert result.truncated is True + assert len(result.output.encode("utf-8")) <= 64 + + +def test_real_sandbox_enforces_timeout_and_file_limit( + executor: NativeSandboxExecutor, monkeypatch: pytest.MonkeyPatch +): + started = time.monotonic() + timed_out = executor.execute("sleep 5", timeout=1) + assert timed_out.exit_code == 124 + assert time.monotonic() - started < 3 + + monkeypatch.setenv("EVOSCIENTIST_SANDBOX_FILE_SIZE_BYTES", "1024") + code = "f=open('large.bin','wb'); f.write(b'x' * 4096); f.flush()" + limited = executor.execute(f"python3 -c {shlex.quote(code)}") + assert limited.exit_code != 0 + assert (executor.files_dir / "large.bin").stat().st_size <= 1024 + + +def test_real_sandbox_kills_background_descendants(executor: NativeSandboxExecutor): + result = executor.execute("sleep 30 >/dev/null 2>&1 & echo $! > child.pid") + assert result.exit_code == 0 + child_pid = int((executor.files_dir / "child.pid").read_text(encoding="ascii")) + time.sleep(0.1) + with pytest.raises(ProcessLookupError): + os.kill(child_pid, 0) + + +def test_file_tools_and_pandoc_share_one_workspace(tmp_path: Path): + ensure_native_sandbox_ready() + files = tmp_path / "files" + runtime = tmp_path / "runtime" + files.mkdir() + runtime.mkdir() + backend = NativeWorkspaceBackend(files, runtime, timeout=20) + assert ( + backend.write("/workspace/source.md", "# 火电分析\n\n同一会话文件。\n").error + is None + ) + + result = backend.execute( + "mkdir -p uploads results && " + 'pandoc source.md -o "uploads/湖南 火电.docx" && ' + 'pandoc "uploads/湖南 火电.docx" -o results/report.html' + ) + + assert result.exit_code == 0, result.output + assert ( + backend.read("/workspace/uploads/湖南 火电.docx").file_data["encoding"] + == "base64" + ) + report = backend.read("/workspace/results/report.html") + assert report.error is None + assert "火电分析" in report.file_data["content"] + assert backend.download_files(["/workspace/results/report.html"])[0].content + + +@pytest.mark.asyncio +async def test_async_cancellation_terminates_process_group(tmp_path: Path): + ensure_native_sandbox_ready() + files = tmp_path / "files" + runtime = tmp_path / "runtime" + files.mkdir() + runtime.mkdir() + backend = NativeWorkspaceBackend(files, runtime, timeout=30) + + task = asyncio.create_task( + backend.aexecute("echo $$ > shell.pid; sleep 30", timeout=30) + ) + for _ in range(50): + if (files / "shell.pid").exists(): + break + await asyncio.sleep(0.02) + assert (files / "shell.pid").exists() + shell_pid = int((files / "shell.pid").read_text(encoding="ascii")) + + task.cancel() + with pytest.raises(asyncio.CancelledError): + await task + await asyncio.sleep(0.1) + with pytest.raises(ProcessLookupError): + os.kill(shell_pid, 0) diff --git a/tests/test_prompts.py b/tests/test_prompts.py index 02b1f26..8777339 100644 --- a/tests/test_prompts.py +++ b/tests/test_prompts.py @@ -183,3 +183,13 @@ class TestDangerousShellGuidelines: def test_dangerous_without_cwd_falls_back(self): result = get_system_prompt(dangerous=True) assert "DANGEROUS MODE" in result + + +class TestNativeWebShellGuidelines: + def test_uses_relative_foreground_contract(self): + result = get_system_prompt(native_web_sandbox=True) + assert "$WORKSPACE` is `.`" in result + assert "Do not use `/workspace` inside shell commands" in result + assert "Do not append `&`" in result + assert "run_in_background" not in result + assert "> /output.log" not in result diff --git a/tests/test_stream_events.py b/tests/test_stream_events.py index dfb8f1b..258e744 100644 --- a/tests/test_stream_events.py +++ b/tests/test_stream_events.py @@ -1123,7 +1123,7 @@ class TestUsageStatsExtraction: class TestCanonicalSourceCapabilities: - async def test_root_update_emits_full_task_snapshot_and_empty_clear(self): + async def test_root_update_does_not_emit_legacy_task_snapshot(self): agent = FakeV3Agent( [ protocol_event( @@ -1134,15 +1134,7 @@ class TestCanonicalSourceCapabilities: ] ) events = await collect_events(agent) - snapshots = [event for event in events if event.get("type") == "task_snapshot"] - assert snapshots == [ - { - "type": "task_snapshot", - "source": "update", - "items": [{"content": "Inspect", "status": "in_progress"}], - }, - {"type": "task_snapshot", "source": "update", "items": []}, - ] + assert not any(event.get("type") == "task_snapshot" for event in events) async def test_subagent_todos_do_not_replace_root_snapshot(self): agent = FakeV3Agent( @@ -1182,7 +1174,6 @@ class TestCanonicalSourceCapabilities: def test_stream_capabilities_are_explicit(self): assert STREAM_PROTOCOL_CAPABILITIES == frozenset( { - "task_snapshot_v1", "complete_tool_call_v1", "correlated_tool_call_id_v1", "final_invalid_tool_call_v1", diff --git a/tests/test_workspace_files.py b/tests/test_workspace_files.py new file mode 100644 index 0000000..2e1c1fa --- /dev/null +++ b/tests/test_workspace_files.py @@ -0,0 +1,267 @@ +from __future__ import annotations + +import base64 +import os +from pathlib import Path + +import pytest +from deepagents.backends.protocol import ExecuteResponse + +from EvoScientist.native_sandbox import ( + NativeSandboxExecutor, + NativeSandboxUnavailable, + NativeWorkspaceBackend, +) +from EvoScientist.workspace_files import ( + RootedWorkspace, + ScopedFilesystemBackend, + WorkspacePathError, + normalize_workspace_path, +) +from EvoScientist.workspace_scope import ( + DeferredScopedBackend, + _RuntimeScopeConfig, + conversation_files_dir, + delete_conversation_scope, + provision_conversation_scope, +) + + +def test_normalize_workspace_path_is_strict(): + assert normalize_workspace_path("/workspace") == () + assert normalize_workspace_path("/workspace/reports/a.txt") == ( + "reports", + "a.txt", + ) + for invalid in ( + "/etc/passwd", + "/workspace/../secret", + "/workspace/a//b", + "/workspace/./a", + "workspace/a", + "/workspace/a\\b", + ): + with pytest.raises(WorkspacePathError): + normalize_workspace_path(invalid) + + +def test_scoped_filesystem_backend_complete_round_trip(tmp_path: Path): + backend = ScopedFilesystemBackend(tmp_path) + + assert backend.write("/workspace/notes/a.txt", "alpha\nbeta\n").error is None + assert backend.read("/workspace/notes/a.txt").file_data == { + "content": "alpha\nbeta\n", + "encoding": "utf-8", + } + edit = backend.edit("/workspace/notes/a.txt", "beta", "gamma") + assert edit.error is None + assert edit.occurrences == 1 + + listing = backend.ls("/workspace/notes") + assert [item["path"] for item in listing.entries or []] == [ + "/workspace/notes/a.txt" + ] + assert [item["path"] for item in backend.glob("**/*.txt").matches or []] == [ + "/workspace/notes/a.txt" + ] + assert backend.grep("gamma").matches == [ + {"path": "/workspace/notes/a.txt", "line": 2, "text": "gamma"} + ] + + upload = backend.upload_files([("/workspace/data.bin", b"\x00\x01")])[0] + assert upload.error is None + download = backend.download_files(["/workspace/data.bin"])[0] + assert download.error is None + assert download.content == b"\x00\x01" + + +def test_root_lists_workspace_namespace(tmp_path: Path): + backend = ScopedFilesystemBackend(tmp_path) + assert backend.ls("/").entries == [ + {"path": "/workspace/", "is_dir": True, "size": 0} + ] + + +def test_docx_and_unknown_binary_read_with_base64_contract(tmp_path: Path): + backend = ScopedFilesystemBackend(tmp_path) + docx = b"PK\x03\x04\x00word/document.xml" + unknown = b"custom\x00binary" + backend.upload_files( + [ + ("/workspace/input.docx", docx), + ("/workspace/payload.custom", unknown), + ] + ) + + assert backend.read("/workspace/input.docx").file_data == { + "content": base64.standard_b64encode(docx).decode("ascii"), + "encoding": "base64", + } + assert backend.read("/workspace/payload.custom").file_data == { + "content": base64.standard_b64encode(unknown).decode("ascii"), + "encoding": "base64", + } + + +def test_symlink_targets_and_parents_are_rejected(tmp_path: Path): + outside = tmp_path.parent / "outside-secret.txt" + outside.write_text("secret", encoding="utf-8") + (tmp_path / "leak.txt").symlink_to(outside) + (tmp_path / "escape").symlink_to(tmp_path.parent, target_is_directory=True) + backend = ScopedFilesystemBackend(tmp_path) + + assert backend.read("/workspace/leak.txt").error + assert backend.write("/workspace/escape/new.txt", "nope").error + assert backend.download_files(["/workspace/leak.txt"])[0].error == "invalid_path" + assert backend.ls("/workspace").entries == [] + + +def test_rooted_workspace_returns_open_verified_handle(tmp_path: Path): + workspace = RootedWorkspace(tmp_path) + workspace.replace_file("/workspace/result.txt", b"result") + handle = workspace.open_binary("/workspace/result.txt") + os.unlink(tmp_path / "result.txt") + try: + assert handle.read() == b"result" + finally: + handle.close() + + +def test_rooted_workspace_entry_and_recursive_delete(tmp_path: Path): + workspace = RootedWorkspace(tmp_path) + workspace.replace_file("/workspace/report/assets/app.js", b"app()") + + assert workspace.entry("/workspace/report").is_dir + assert workspace.entry("/workspace/report/assets/app.js").size == 5 + + workspace.delete("/workspace/report") + + with pytest.raises(FileNotFoundError): + workspace.entry("/workspace/report") + + +def test_native_backend_delegates_validated_command_to_executor( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +): + files = tmp_path / "files" + runtime = tmp_path / "runtime" + files.mkdir() + runtime.mkdir() + captured: dict[str, object] = {} + + def fake_execute( + self, + command, + *, + timeout=None, + skip_readiness_check=False, + cancel_event=None, + ): + captured.update(command=command, timeout=timeout) + return ExecuteResponse(output="ok\n", exit_code=0, truncated=False) + + monkeypatch.setattr(NativeSandboxExecutor, "execute", fake_execute) + + backend = NativeWorkspaceBackend(files, runtime, timeout=30) + result = backend.execute("cat probe.txt", timeout=12) + assert result.exit_code == 0 + assert captured == {"command": "cat probe.txt", "timeout": 12} + + blocked = backend.execute("sudo cat probe.txt") + assert blocked.exit_code == 1 + assert "blocked" in blocked.output.lower() + + invalid = backend.execute("printf 'bad\x00command'") + assert invalid.exit_code == 1 + + +def test_native_backend_fails_closed_when_executor_is_unavailable( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +): + files = tmp_path / "files" + runtime = tmp_path / "runtime" + files.mkdir() + runtime.mkdir() + + def unavailable(*args, **kwargs): + raise NativeSandboxUnavailable("private deployment detail") + + monkeypatch.setattr(NativeSandboxExecutor, "execute", unavailable) + result = NativeWorkspaceBackend(files, runtime, timeout=30).execute("echo ok") + + assert result.exit_code == 125 + assert "private deployment detail" not in result.output + + +@pytest.mark.asyncio +async def test_deferred_backend_preserves_async_executor_cancellation_path( + monkeypatch: pytest.MonkeyPatch, +): + proxy = DeferredScopedBackend( + _RuntimeScopeConfig( + scope_id="00000000-0000-0000-0000-000000000001", + owner_id="00000000-0000-0000-0000-000000000002", + thread_id="thread", + deployment_id="deployment", + ), + dangerous=False, + ) + captured: dict[str, object] = {} + + class Delegate: + async def aexecute(self, command, *, timeout=None): + captured.update(command=command, timeout=timeout) + return ExecuteResponse(output="async", exit_code=0, truncated=False) + + monkeypatch.setattr(proxy, "_delegate", lambda: Delegate()) + result = await proxy.aexecute("sleep 1", timeout=7) + + assert result.output == "async" + assert captured == {"command": "sleep 1", "timeout": 7} + + +def test_scope_delete_is_idempotent_and_removes_directory(tmp_path: Path): + thread_id = "0f88db64-720e-4f88-ac92-ea9a76b45596" + deployment_id = "test-workspace-delete" + record = provision_conversation_scope( + thread_id, + deployment_id=deployment_id, + workspace_root=tmp_path, + ) + scope_root = tmp_path / ".evoscientist" / "conversations" / record.scope_id + (scope_root / "files" / "result.txt").write_text("done", encoding="utf-8") + + deleted = delete_conversation_scope( + thread_id, + deployment_id=deployment_id, + workspace_root=tmp_path, + ) + assert deleted.state == "deleted" + assert not scope_root.exists() + assert ( + delete_conversation_scope( + thread_id, + deployment_id=deployment_id, + workspace_root=tmp_path, + ).state + == "deleted" + ) + + +def test_deleted_scope_can_be_reprovisioned_for_create_retry(tmp_path: Path): + thread_id = "a224305c-32ae-43c5-bdae-76b52912fb37" + deployment_id = "test-workspace-retry" + original = provision_conversation_scope( + thread_id, deployment_id=deployment_id, workspace_root=tmp_path + ) + delete_conversation_scope( + thread_id, deployment_id=deployment_id, workspace_root=tmp_path + ) + + retried = provision_conversation_scope( + thread_id, deployment_id=deployment_id, workspace_root=tmp_path + ) + + assert retried.scope_id == original.scope_id + assert retried.state == "draft" + assert conversation_files_dir(retried.scope_id, tmp_path).is_dir()