Files
EvoScientist/EvoScientist/workspace_scope.py
T

542 lines
19 KiB
Python

"""Conversation-scoped workspace resolution and DeepAgents backend factory."""
from __future__ import annotations
import os
import shutil
import subprocess
import threading
import uuid
from collections.abc import Callable
from dataclasses import dataclass
from pathlib import Path
from typing import Any
from deepagents.backends.protocol import (
EditResult,
ExecuteResponse,
FileDownloadResponse,
FileUploadResponse,
GlobResult,
GrepResult,
LsResult,
ReadResult,
SandboxBackendProtocol,
WriteResult,
)
from langchain.tools import ToolRuntime
from . import paths
from .scope_registry import (
ScopeAccessError,
ScopeRecord,
deployment_id_for_workspace,
get_scope_registry,
)
IsolationMode = str
def workspace_isolation_mode() -> IsolationMode:
value = os.getenv("EVOSCIENTIST_WORKSPACE_ISOLATION", "optional").strip().lower()
if value not in {"legacy", "optional", "required"}:
raise RuntimeError("EVOSCIENTIST_WORKSPACE_ISOLATION must be legacy, optional or required")
return value
def is_required() -> bool:
return workspace_isolation_mode() == "required"
def verify_required_executor() -> None:
"""Fail startup unless the pinned scope executor is locally usable."""
docker = shutil.which("docker")
if not docker:
raise RuntimeError("required workspace isolation needs the docker OCI runtime")
image = os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_IMAGE", "").strip()
if "@sha256:" not in image:
raise RuntimeError("required workspace isolation needs an OCI image pinned by digest")
try:
probe = subprocess.run(
[docker, "image", "inspect", image],
check=False,
capture_output=True,
text=True,
timeout=10,
)
except (OSError, subprocess.TimeoutExpired) as exc:
raise RuntimeError("required workspace isolation cannot verify the OCI executor") from exc
if probe.returncode != 0:
raise RuntimeError(
f"required workspace isolation needs local OCI image {image!r}"
)
def current_deployment_id() -> str:
return deployment_id_for_workspace(paths.WORKSPACE_ROOT)
def conversation_root(scope_id: str, workspace_root: Path | None = None) -> Path:
scope = str(uuid.UUID(scope_id))
# The deploy process supplies an absolute workspace root. This helper is
# called from synchronous DeepAgents backend factories on the ASGI loop.
root = (workspace_root or paths.WORKSPACE_ROOT).expanduser()
return root / ".evoscientist" / "conversations" / scope
def conversation_files_dir(scope_id: str, workspace_root: Path | None = None) -> Path:
return conversation_root(scope_id, workspace_root) / "files"
@dataclass(frozen=True, slots=True)
class ScopeContext:
deployment_id: str
scope_id: str
owner_id: str
thread_id: str
revision: int
files_dir: Path
runtime_dir: Path
primary_thread_id: str
@dataclass(frozen=True, slots=True)
class _RuntimeScopeConfig:
"""Untrusted runtime identifiers parsed without filesystem or Registry I/O."""
scope_id: str
owner_id: str
thread_id: str
deployment_id: str | None
class ScopedContainerBackend:
"""Filesystem backend whose shell commands execute in a scope-only OCI container."""
def __init__(self, root_dir: Path, *, timeout: int) -> None:
from .backends import CustomSandboxBackend
# Reuse the hardened filesystem operations; only ``execute`` is
# replaced so no agent shell runs in the host process.
self._filesystem = CustomSandboxBackend(
root_dir=str(root_dir), virtual_mode=True, timeout=timeout, dangerous=False
)
self._root_dir = root_dir
self._timeout = timeout
def __getattr__(self, name: str) -> Any:
return getattr(self._filesystem, name)
def execute(self, command: str, *, timeout: int | None = None) -> Any:
from .backends import ExecuteResponse, prepare_sandbox_command
command, error = prepare_sandbox_command(
command, self._filesystem.cwd, virtual_mode=True, dangerous=False
)
if error:
return ExecuteResponse(output=error, exit_code=1, truncated=False)
image = os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_IMAGE", "").strip()
if "@sha256:" not in image:
return ExecuteResponse(
output="Required workspace isolation needs an OCI image pinned by digest.",
exit_code=125,
truncated=False,
)
effective_timeout = max(1, min(timeout or self._timeout, 3600))
invocation = [
"docker",
"run",
"--rm",
"--network",
"none",
"--read-only",
"--tmpfs",
"/tmp:rw,noexec,nosuid,size=64m",
"--cap-drop",
"ALL",
"--security-opt",
"no-new-privileges",
"--pids-limit",
os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_PIDS", "128"),
"--memory",
os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_MEMORY", "1g"),
"--cpus",
os.getenv("EVOSCIENTIST_STRICT_EXECUTOR_CPUS", "1"),
"--mount",
f"type=bind,src={self._root_dir},dst=/workspace",
"--workdir",
"/workspace",
image,
"sh",
"-lc",
command,
]
try:
completed = subprocess.run(
invocation,
check=False,
capture_output=True,
text=True,
timeout=effective_timeout,
)
except FileNotFoundError:
return ExecuteResponse(
output="Required workspace isolation needs an OCI runtime (docker was not found).",
exit_code=127,
truncated=False,
)
except subprocess.TimeoutExpired as exc:
output = (exc.stdout or "") + (exc.stderr or "")
return ExecuteResponse(output=output, exit_code=124, truncated=False)
output = completed.stdout + completed.stderr
return ExecuteResponse(output=output, exit_code=completed.returncode, truncated=False)
def _configurable(runtime: ToolRuntime[Any, Any] | Any | None) -> dict[str, Any]:
"""Return the active runnable config, with a non-graph fallback.
``ToolRuntime`` deliberately does not expose ``RunnableConfig`` during a
graph execution. LangGraph keeps it in a context variable instead. The
fallback preserves direct callers and unit tests that supply a lightweight
runtime object outside a runnable context.
"""
config: Any = None
try:
from langgraph.config import get_config
config = get_config()
except (ImportError, LookupError, RuntimeError):
pass
if not isinstance(config, dict) and runtime is not None:
config = getattr(runtime, "config", None) or {}
if not isinstance(config, dict):
return {}
configurable = config.get("configurable") or {}
return dict(configurable) if isinstance(configurable, dict) else {}
def _required_string(configurable: dict[str, Any], key: str) -> str:
value = configurable.get(key)
if not isinstance(value, str) or not value:
raise ScopeAccessError(f"missing {key}")
return value
def _runtime_scope_config(
runtime: ToolRuntime[Any, Any] | Any | None,
*,
kind: str,
) -> _RuntimeScopeConfig | None:
"""Parse scope identifiers without treating config as an authorization grant."""
configurable = _configurable(runtime)
scope_id = configurable.get("workspace_scope_id")
owner_id = configurable.get("workspace_scope_owner_id")
thread_id = configurable.get("thread_id")
if scope_id is None and owner_id is None:
if workspace_isolation_mode() == "required":
raise ScopeAccessError(f"{kind} requires a workspace scope")
return None
if (
not isinstance(scope_id, str)
or not isinstance(owner_id, str)
or not isinstance(thread_id, str)
):
raise ScopeAccessError(f"{kind} has an incomplete workspace scope")
try:
canonical_scope_id = str(uuid.UUID(scope_id))
canonical_owner_id = str(uuid.UUID(owner_id))
except ValueError as exc:
raise ScopeAccessError(f"{kind} has an invalid workspace scope") from exc
deployment_id = configurable.get("workspace_deployment_id")
if deployment_id is not None and (
not isinstance(deployment_id, str) or not deployment_id
):
raise ScopeAccessError(f"{kind} has an invalid workspace deployment")
return _RuntimeScopeConfig(
scope_id=canonical_scope_id,
owner_id=canonical_owner_id,
thread_id=thread_id,
deployment_id=deployment_id,
)
def _validated_scope_directories(scope_id: str) -> tuple[Path, Path]:
"""Return canonical private directories after preventing symlink escape.
This function intentionally resolves paths and must run only from a
filesystem-operation worker, never from the runtime backend factory.
"""
conversations_dir = (
paths.WORKSPACE_ROOT.expanduser() / ".evoscientist" / "conversations"
).resolve(strict=True)
scope_root = (conversations_dir / scope_id).resolve(strict=True)
files_dir = (scope_root / "files").resolve(strict=True)
runtime_dir = (scope_root / "runtime").resolve(strict=True)
if (
scope_root.parent != conversations_dir
or files_dir.parent != scope_root
or runtime_dir.parent != scope_root
):
raise ScopeAccessError("workspace directory escapes its scope")
if not files_dir.is_dir() or not runtime_dir.is_dir():
raise ScopeAccessError("workspace directory is missing")
return files_dir, runtime_dir
def _resolve_scope_context(config: _RuntimeScopeConfig | None) -> ScopeContext | None:
"""Validate parsed scope identifiers against the active registry."""
if config is None:
return None
deployment_id = config.deployment_id or current_deployment_id()
registry = get_scope_registry(paths.WORKSPACE_ROOT)
if registry.active_lock(deployment_id, "workspace-cutover") is not None:
raise ScopeAccessError("workspace cutover is in progress")
record = registry.assert_runtime(
deployment_id, config.scope_id, config.thread_id, config.owner_id
)
files_dir, runtime_dir = _validated_scope_directories(record.scope_id)
return ScopeContext(
deployment_id=deployment_id,
scope_id=record.scope_id,
owner_id=config.owner_id,
thread_id=config.thread_id,
revision=record.revision,
files_dir=files_dir,
runtime_dir=runtime_dir,
primary_thread_id=record.primary_thread_id,
)
def require_scoped_runtime(
runtime: ToolRuntime[Any, Any] | Any | None,
*,
kind: str = "tool",
) -> ScopeContext | None:
"""Resolve and validate a runtime scope.
``optional`` retains legacy CLI compatibility when no scope has been
injected. ``required`` never falls back to ``WORKSPACE_ROOT``.
"""
config = _runtime_scope_config(runtime, kind=kind)
return _resolve_scope_context(config)
def provision_conversation_scope(
thread_id: str,
*,
deployment_id: str | None = None,
scope_id: str | None = None,
workspace_root: Path | None = None,
lock_operation_id: str | None = None,
) -> ScopeRecord:
"""Create the registry mapping and private directory for a primary thread."""
root = (workspace_root or paths.WORKSPACE_ROOT).expanduser()
deployment_id = deployment_id or deployment_id_for_workspace(root)
registry = get_scope_registry(root)
record = registry.provision(
deployment_id,
thread_id,
scope_id=scope_id,
lock_operation_id=lock_operation_id,
)
root = conversation_root(record.scope_id, root)
try:
(root / "files").mkdir(mode=0o700, parents=True, exist_ok=True)
(root / "runtime").mkdir(mode=0o700, parents=True, exist_ok=True)
for directory in (root, root / "files", root / "runtime"):
try:
directory.chmod(0o700)
except OSError:
pass
except OSError:
# Keep the durable reservation for the recovery job; it is safer than
# silently falling back to the shared deployment root.
raise
return record
def _build_backend(root_dir: Path, *, dangerous: bool) -> Any:
from deepagents.backends import CompositeBackend
from .backends import (
CustomSandboxBackend,
MemoryFilesystemBackend,
MergedSkillsBackend,
)
from .EvoScientist import SKILLS_DIR
cfg_timeout = int(os.getenv("EVOSCIENTIST_SANDBOX_EXECUTE_TIMEOUT", "300"))
ws_backend: Any
if is_required():
ws_backend = ScopedContainerBackend(root_dir, timeout=cfg_timeout)
else:
ws_backend = CustomSandboxBackend(
root_dir=str(root_dir),
virtual_mode=True,
timeout=cfg_timeout,
dangerous=dangerous,
)
return CompositeBackend(
default=ws_backend,
routes={
"/skills/": MergedSkillsBackend(
primary_dir=str(paths.USER_SKILLS_DIR),
global_dir=str(paths.GLOBAL_SKILLS_DIR),
secondary_dir=SKILLS_DIR,
),
"/memories/": MemoryFilesystemBackend(
root_dir=str(paths.MEMORIES_DIR), virtual_mode=True
),
},
)
class DeferredScopedBackend(SandboxBackendProtocol):
"""Resolve the scoped filesystem backend only from a worker thread.
DeepAgents invokes its deprecated backend factory from async middleware.
Its concrete filesystem backends synchronously call ``Path.resolve()`` in
their constructors, so doing that work in the factory makes every run fail
under LangGraph's blocking-call detector. This proxy itself is I/O-free;
the inherited async methods dispatch the synchronous operations to a
thread, where Registry validation and concrete backend construction occur.
"""
def __init__(
self,
config: _RuntimeScopeConfig,
*,
dangerous: bool,
) -> None:
self._config = config
self._dangerous = dangerous
self._backend: Any | None = None
self._backend_key: tuple[str, str, str, int] | None = None
self._lock = threading.RLock()
@property
def id(self) -> str:
# This is queried while composing the model request; do not initialize
# the real backend or touch the Registry here.
return f"scope-{self._config.scope_id[:8]}-{self._config.owner_id[:8]}"
def _delegate(self) -> Any:
"""Validate the current scope and return a concrete backend.
Every operation enters here, so a deleted scope or stale owner cannot
keep using a backend constructed before the lifecycle transition.
"""
# Async backend methods run this code in a worker thread. LangGraph's
# RunnableConfig context variable is not available there, so validate
# the immutable scope parsed by the factory on the graph thread.
context = _resolve_scope_context(self._config)
if context is None:
raise ScopeAccessError("scoped backend lost its workspace scope")
if is_required() and self._dangerous:
raise ScopeAccessError(
"dangerous_mode is incompatible with required isolation"
)
key = (context.scope_id, context.owner_id, context.thread_id, context.revision)
with self._lock:
if self._backend is None or self._backend_key != key:
self._backend = _build_backend(
context.files_dir, dangerous=self._dangerous
)
self._backend_key = key
return self._backend
def ls(self, path: str) -> LsResult:
return self._delegate().ls(path)
def read(self, file_path: str, offset: int = 0, limit: int = 2000) -> ReadResult:
return self._delegate().read(file_path, offset, limit)
def grep(
self, pattern: str, path: str | None = None, glob: str | None = None
) -> GrepResult:
return self._delegate().grep(pattern, path, glob)
def glob(self, pattern: str, path: str | None = None) -> GlobResult:
return self._delegate().glob(pattern, path)
def write(self, file_path: str, content: str) -> WriteResult:
return self._delegate().write(file_path, content)
def edit(
self,
file_path: str,
old_string: str,
new_string: str,
replace_all: bool = False,
) -> EditResult:
return self._delegate().edit(file_path, old_string, new_string, replace_all)
def upload_files(self, files: list[tuple[str, bytes]]) -> list[FileUploadResponse]:
return self._delegate().upload_files(files)
def download_files(self, paths: list[str]) -> list[FileDownloadResponse]:
return self._delegate().download_files(paths)
def execute(self, command: str, *, timeout: int | None = None) -> ExecuteResponse:
return self._delegate().execute(command, timeout=timeout)
def create_workspace_backend(
runtime: ToolRuntime[Any, Any],
*,
legacy_backend: Callable[[], Any],
dangerous: bool = False,
allow_unscoped_legacy: bool = True,
) -> Any:
"""Return a backend handle without blocking the Agent event loop."""
config = _runtime_scope_config(runtime, kind="filesystem backend")
if config is None:
if not allow_unscoped_legacy:
raise ScopeAccessError(
"deployed graph runs require a workspace scope"
)
return legacy_backend()
if is_required() and dangerous:
raise ScopeAccessError("dangerous_mode is incompatible with required isolation")
return DeferredScopedBackend(config, dangerous=dangerous)
def create_workspace_backend_factory(
legacy_backend: Callable[[], Any],
*,
dangerous: bool = False,
allow_unscoped_legacy: bool = True,
) -> Callable[[ToolRuntime[Any, Any]], Any]:
def factory(runtime: ToolRuntime[Any, Any]) -> Any:
return create_workspace_backend(
runtime,
legacy_backend=legacy_backend,
dangerous=dangerous,
allow_unscoped_legacy=allow_unscoped_legacy,
)
return factory
def workspace_metadata(record: ScopeRecord) -> dict[str, str | int]:
"""Metadata mirrored onto the LangGraph primary thread by trusted callers."""
return {
"workspace_schema_version": 1,
"workspace_scope_id": record.scope_id,
"workspace_status": record.state,
"workspace_scope_owner_id": record.primary_owner_id,
"workspace_scope_revision": record.revision,
"workspace_deployment_id": record.deployment_id,
}