diff --git a/EvoScientist/EvoScientist.py b/EvoScientist/EvoScientist.py index 19afb25..954f8d1 100644 --- a/EvoScientist/EvoScientist.py +++ b/EvoScientist/EvoScientist.py @@ -509,16 +509,27 @@ def _get_default_middleware(*, for_async_subagent: bool = False): def _get_default_agent(): - """Build the default agent (with MCP, no checkpointer) on first access. + """Build the default agent (no checkpointer) on first access. - When invoked from the langgraph dev subprocess (env var - ``EVOSCIENTIST_DEPLOYED_NO_MCP=true``, set by - ``langgraph_dev.manager.start_langgraph_dev``), MCP loading is skipped to - avoid duplicating the CLI's MCP server pool — the deployed main agent - is currently only reachable via HTTP for Web UI / SDK clients (none in - use yet), so paying for a second copy of the same MCP servers is pure - waste. Re-enable later by removing the env var when MCP-needing remote - callers are introduced. + MCP loading depends on which subprocess mode (if any) this agent is + being built in. ``langgraph_dev.manager.start_langgraph_dev`` injects + ``EVOSCIENTIST_DEPLOY_MODE`` into the subprocess with one of two values: + + - ``EVOSCIENTIST_DEPLOY_MODE=full`` — set by ``EvoSci deploy``. The + subprocess is the *primary* programmatic entry point (Python scripts, + Jupyter, integration tests via ``langgraph_sdk``), so it needs the full + configuration: **load MCP**, and ``_ASYNC_SUBAGENTS_AVAILABLE`` flips on + at module load so async sub-agents self-loop through this same + langgraph dev server. + + - ``EVOSCIENTIST_DEPLOY_MODE=stripped`` — set by ``EvoSci`` / ``EvoSci + serve``. The CLI's in-process main agent already loaded MCP; this + subprocess only services async sub-agent self-loops, so **skip MCP** + to avoid running a second copy of the same servers. + + Plain ``from EvoScientist import EvoScientist_agent`` (env var unset) + loads MCP. Async sub-agents stay disabled in that case because there is + no langgraph dev server to self-loop into. """ global _EvoScientist_agent if _EvoScientist_agent is None: @@ -535,7 +546,7 @@ def _get_default_agent(): if not cfg.auto_approve: mw.append(HumanInTheLoopMiddleware(interrupt_on={"execute": True})) - if os.environ.get("EVOSCIENTIST_DEPLOYED_NO_MCP", "").lower() == "true": + if os.environ.get("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "stripped": kwargs = _build_base_kwargs(be, mw) else: kwargs = load_mcp_and_build_kwargs(be, mw) diff --git a/EvoScientist/cli/__init__.py b/EvoScientist/cli/__init__.py index 1ac27f4..2b12d55 100644 --- a/EvoScientist/cli/__init__.py +++ b/EvoScientist/cli/__init__.py @@ -8,6 +8,7 @@ only pay their cost when someone actually touches those names. from __future__ import annotations +from .. import deploy as _deploy_pkg # noqa: F401 — registers `deploy` @app.command from . import commands # noqa: F401 — registers @app.command decorators from ._app import app diff --git a/EvoScientist/cli/commands.py b/EvoScientist/cli/commands.py index 20ac814..770dccb 100644 --- a/EvoScientist/cli/commands.py +++ b/EvoScientist/cli/commands.py @@ -192,16 +192,27 @@ def _ensure_async_subagent_server(config: Any, *, workspace_dir: str) -> None: Shared by both the interactive entry and the serve entry — keeps the user-visible status message and the conditional in one place. + + Raises ``typer.Exit(1)`` (after surfacing a red error) when an + externally-managed langgraph dev is already running for a different + workspace — e.g., ``EvoSci deploy --workdir /A`` is up and the user + is starting ``EvoSci`` / ``EvoSci serve`` in /B. Continuing in that + state would route async sub-agent calls to a process pinned to /A + while the main agent runs in /B. """ if not getattr(config, "enable_async_subagents", False): return - from ..langgraph_dev.manager import ensure_langgraph_dev + from ..langgraph_dev.manager import WorkspaceMismatchError, ensure_langgraph_dev - with console.status( - "[dim]Starting async sub-agent server (langgraph dev)...[/dim]", - spinner="dots", - ): - ensure_langgraph_dev(config, workspace_dir=workspace_dir) + try: + with console.status( + "[dim]Starting async sub-agent server (langgraph dev)...[/dim]", + spinner="dots", + ): + ensure_langgraph_dev(config, workspace_dir=workspace_dir) + except WorkspaceMismatchError as exc: + console.print(f"[red]{exc}[/red]") + raise typer.Exit(1) from exc def _resolve_context_window( diff --git a/EvoScientist/cli/interactive.py b/EvoScientist/cli/interactive.py index dc293db..0d7bd3e 100644 --- a/EvoScientist/cli/interactive.py +++ b/EvoScientist/cli/interactive.py @@ -628,7 +628,6 @@ def cmd_interactive( callback mutates REPL state, reloads the agent, and renders conversation history.""" if workspace_dir: - state["workspace_dir"] = workspace_dir # Sync the langgraph dev subprocess to the resumed # workspace so deployed sub-agents (writing-agent etc.) # don't operate on the previous workspace's files. The @@ -638,19 +637,34 @@ def cmd_interactive( # doesn't think the CLI is frozen, and run the sync call # in a worker thread so the asyncio event loop keeps # serving channel polls / MCP heartbeats during the wait. + # + # State mutation happens AFTER this sync succeeds so a + # WorkspaceMismatchError leaves the session's existing + # workspace_dir / thread_id untouched. if getattr(config, "enable_async_subagents", False): - from ..langgraph_dev.manager import ensure_langgraph_dev + from ..langgraph_dev.manager import ( + WorkspaceMismatchError, + ensure_langgraph_dev, + ) - with console.status( - "[dim]Syncing async sub-agent server to resumed " - "workspace...[/dim]", - spinner="dots", - ): - await asyncio.to_thread( - ensure_langgraph_dev, - config, - workspace_dir=workspace_dir, - ) + try: + with console.status( + "[dim]Syncing async sub-agent server to resumed " + "workspace...[/dim]", + spinner="dots", + ): + await asyncio.to_thread( + ensure_langgraph_dev, + config, + workspace_dir=workspace_dir, + ) + except WorkspaceMismatchError as exc: + # Another EvoSci process owns the langgraph dev + # server for a different workspace. Abort the + # resume without mutating session state. + console.print(f"[red]{exc}[/red]") + return + state["workspace_dir"] = workspace_dir state["thread_id"] = thread_id state["resumed"] = True state["status_started_at"] = datetime.now() @@ -705,18 +719,28 @@ def cmd_interactive( # the sync call in a worker thread so the asyncio # event loop stays responsive. if getattr(config, "enable_async_subagents", False): - from ..langgraph_dev.manager import ensure_langgraph_dev + from ..langgraph_dev.manager import ( + WorkspaceMismatchError, + ensure_langgraph_dev, + ) - with console.status( - "[dim]Syncing async sub-agent server to " - "resumed workspace...[/dim]", - spinner="dots", - ): - await asyncio.to_thread( - ensure_langgraph_dev, - config, - workspace_dir=ws, - ) + try: + with console.status( + "[dim]Syncing async sub-agent server to " + "resumed workspace...[/dim]", + spinner="dots", + ): + await asyncio.to_thread( + ensure_langgraph_dev, + config, + workspace_dir=ws, + ) + except WorkspaceMismatchError as exc: + # Startup --resume into a workspace owned by + # a different EvoSci process: refuse to start + # the CLI so the user can resolve the conflict. + console.print(f"[red]{exc}[/red]") + raise typer.Exit(1) from exc else: # Resolution failed (ambiguous/not-found); the user's raw # input is still seeded in state["thread_id"] from init. diff --git a/EvoScientist/cli/tui_interactive.py b/EvoScientist/cli/tui_interactive.py index 6891cc2..61ccd41 100644 --- a/EvoScientist/cli/tui_interactive.py +++ b/EvoScientist/cli/tui_interactive.py @@ -611,7 +611,6 @@ def run_textual_interactive( self, thread_id: str, workspace_dir: str | None = None ) -> None: if workspace_dir: - self._workspace_dir = workspace_dir # Mirror the Rich CLI fix: when a /resume restores a thread # whose workspace differs from the one the langgraph dev # subprocess was launched with, the deployed sub-agents @@ -622,11 +621,19 @@ def run_textual_interactive( # loop keeps refreshing the UI during the up-to-60s wait, and # show a live timer widget (like /compact) so the user sees # progress instead of a frozen static line. + # + # ``self._workspace_dir`` is mutated AFTER the sync succeeds + # so a WorkspaceMismatchError leaves the session pointing at + # the existing workspace instead of half-resuming into the + # conflicting one. from ..config import load_config _resume_cfg = load_config() if getattr(_resume_cfg, "enable_async_subagents", False): - from ..langgraph_dev.manager import ensure_langgraph_dev + from ..langgraph_dev.manager import ( + WorkspaceMismatchError, + ensure_langgraph_dev, + ) from .widgets.workspace_sync_widget import WorkspaceSyncWidget sync_widget = WorkspaceSyncWidget() @@ -639,8 +646,15 @@ def run_textual_interactive( _resume_cfg, workspace_dir=workspace_dir, ) + except WorkspaceMismatchError as exc: + # Another EvoSci process owns the langgraph dev for a + # different workspace. Abort the resume without + # mutating session state. + self.notify(str(exc), severity="error", timeout=10) + return finally: await sync_widget.cleanup() + self._workspace_dir = workspace_dir self._conversation_tid = thread_id # Background reload: history renders immediately; next turn awaits. @@ -2976,6 +2990,7 @@ def run_textual_interactive( if resolved: meta = await get_thread_metadata(resolved) ws = (meta or {}).get("workspace_dir", "") + mismatch_aborted = False if ws: effective_workspace = ws # Sync langgraph dev subprocess to the resumed @@ -2984,10 +2999,14 @@ def run_textual_interactive( # Without this, --resume against a thread from a # different workspace would leave deployed sub-agents # operating on the launch directory's files. + from ..stream.console import console as _resume_console + try: from ..config import load_config - from ..langgraph_dev.manager import ensure_langgraph_dev - from ..stream.console import console as _resume_console + from ..langgraph_dev.manager import ( + WorkspaceMismatchError, + ensure_langgraph_dev, + ) _ws_cfg = load_config() if getattr(_ws_cfg, "enable_async_subagents", False): @@ -3001,6 +3020,15 @@ def run_textual_interactive( _ws_cfg, workspace_dir=ws, ) + except WorkspaceMismatchError as _ws_mismatch_exc: + # Surface the user-actionable message via the + # Rich console (TUI hasn't taken over the terminal + # yet) and abort the resume so the TUI doesn't + # start up pointing at the conflicting workspace + # with async sub-agents routed at a server pinned + # to a different workspace. + _resume_console.print(f"[red]{_ws_mismatch_exc}[/red]") + mismatch_aborted = True except Exception as _ws_sync_exc: # Non-fatal at startup — async sub-agents fall back # to sync via the manager's own availability flag. @@ -3013,8 +3041,20 @@ def run_textual_interactive( "to in-process sync delegation for this session.", _ws_sync_exc, ) - effective_thread_id = resolved - resumed = True + if mismatch_aborted: + # Revert the workspace mutation and fall through to the + # ``effective_thread_id is None`` branch below, which + # generates a fresh thread in the original launch + # workspace. User keeps the TUI session; the requested + # ``--thread-id`` resume is what we refuse. + effective_workspace = workspace_dir + resume_warning = ( + f"Resume of thread '{thread_id}' aborted due to " + "workspace conflict. Starting new session." + ) + else: + effective_thread_id = resolved + resumed = True elif matches: resume_warning = ( f"Thread prefix '{thread_id}' is ambiguous " diff --git a/EvoScientist/deploy/__init__.py b/EvoScientist/deploy/__init__.py new file mode 100644 index 0000000..ab26b96 --- /dev/null +++ b/EvoScientist/deploy/__init__.py @@ -0,0 +1,22 @@ +"""EvoScientist deploy mode. + +Standalone LangGraph server launcher. ``EvoSci deploy`` starts a +``langgraph dev`` subprocess hosting the fully-equipped main agent +(MCP + async sub-agents enabled), exposed at +``http://localhost:{port}`` for external LangChain-compatible UIs +and SDK clients. + +This module is intentionally separate from ``EvoScientist.cli``: +deploy mode does NOT load an in-process CLI agent, session DB, channel +runtime, or TUI — those are all TUI / serve concerns. Deploy only +manages the lifecycle of the langgraph dev subprocess (and ccproxy if +OAuth is configured). +""" + +from __future__ import annotations + +# Importing the server submodule registers the ``@app.command()`` decorator +# on the shared Typer ``app`` from ``EvoScientist.cli._app``. +from . import server + +__all__ = ["server"] diff --git a/EvoScientist/deploy/server.py b/EvoScientist/deploy/server.py new file mode 100644 index 0000000..e015cda --- /dev/null +++ b/EvoScientist/deploy/server.py @@ -0,0 +1,254 @@ +"""``EvoSci deploy`` — start a standalone langgraph dev server. + +Hosts the fully-equipped EvoScientist main agent (MCP + async sub-agents) +for consumption by external LangChain-compatible UIs (deep-agents-ui, +agent-chat-ui, LangSmith Studio) and SDK clients. + +Differs from ``EvoSci`` / ``EvoSci serve``: no in-process CLI agent, +no session DB, no channel runtime, no TUI. The terminal only shows +startup progress, the Ready banner, and then blocks until Ctrl+C. + +Mode dispatch happens via the ``EVOSCIENTIST_DEPLOY_MODE`` env var +injected by ``start_langgraph_dev``: ``full`` for the deploy subprocess +(this command), ``stripped`` for CLI/serve subprocesses, unset for the +parent process. The subprocess reads this at module-load time +(``langgraph_dev/manager.py``) to flip ``_ASYNC_SUBAGENTS_AVAILABLE``, +and the agent build code (``EvoScientist.py:_get_default_agent``) +loads or skips MCP based on the value. +""" + +from __future__ import annotations + +import atexit +import os +import signal +import threading +from pathlib import Path +from typing import Any + +import typer # type: ignore[import-untyped] +from rich.panel import Panel +from rich.text import Text + +from ..cli._app import app +from ..stream.console import console + + +@app.command() +def deploy( + workdir: str | None = typer.Option( + None, + "--workdir", + help="Workspace directory (default: config.default_workdir or cwd)", + ), + port: int | None = typer.Option( + None, + "--port", + help="Port for langgraph dev (default: config.langgraph_dev_port = 6174)", + ), + debug: bool = typer.Option( + False, + "--debug", + help="Enable debug logging", + ), +): + """Deploy EvoScientist main agent as a standalone LangGraph dev server. + + Starts ``langgraph dev`` in deploy mode (full MCP + async sub-agents). + Connect any LangChain-compatible UI or SDK client to the printed + endpoint. Press Ctrl+C to stop. + """ + from ..config import apply_config_to_env, get_effective_config + from ..langgraph_dev.manager import ( + _DEFAULT_PORT, + _LOG_FILE, + _is_port_occupied, + is_langgraph_dev_running, + start_langgraph_dev, + stop_langgraph_dev, + ) + + # 1. Load config (no CLI overrides here — deploy is opinionated about + # full MCP + async; user-facing flags are workspace/port/debug only). + cli_overrides: dict[str, Any] = {} + if debug: + cli_overrides["log_level"] = "DEBUG" + config = get_effective_config(cli_overrides) + if debug: + os.environ["EVOSCIENTIST_LOG_LEVEL"] = "DEBUG" + from ..cli.commands import _configure_logging + + _configure_logging() + apply_config_to_env(config) + + # 2. Resolve workspace (CLI > config.default_workdir > cwd) + if workdir: + ws = os.path.abspath(os.path.expanduser(workdir)) + elif config.default_workdir: + ws = os.path.abspath(os.path.expanduser(config.default_workdir)) + else: + ws = os.getcwd() + # Subprocess inherits this path via EVOSCIENTIST_WORKSPACE_DIR (set inside + # start_langgraph_dev). Ensure the dir exists; do NOT mutate the parent + # process's paths module state — the deploy parent has no in-process agent. + os.makedirs(ws, exist_ok=True) + + # 3. Resolve port (explicit None check — don't treat --port 0 as "unset"), + # then validate range so misconfigurations fail fast with a clear message + # instead of an opaque socket error from langgraph dev later. + effective_port = ( + int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT)) + if port is None + else port + ) + if not (1 <= effective_port <= 65535): + console.print( + f"[red]Invalid port {effective_port}. Use an integer in [1, 65535].[/red]" + ) + raise typer.Exit(1) + + # 4. Pre-flight port check — refuse to start if a non-EvoSci process is + # holding the port. If an existing EvoSci langgraph dev is already up, + # also refuse (deploy is the "primary server" — running multiple on the + # same port is a configuration error). + if _is_port_occupied(effective_port): + if is_langgraph_dev_running(port=effective_port): + console.print( + f"[red]Port {effective_port} is already serving a langgraph dev " + f"instance.[/red]" + ) + console.print( + "[dim]Stop the existing EvoSci/serve session first, or use " + "[bold]--port[/bold] to deploy on a different port.[/dim]" + ) + else: + console.print( + f"[red]Port {effective_port} is occupied by another process.[/red]" + ) + console.print( + f"[dim]Run [bold]lsof -i :{effective_port}[/bold] to inspect, " + f"or use [bold]--port[/bold] to pick a different port.[/dim]" + ) + raise typer.Exit(1) + + # 5. Startup banner + _auth_label = _describe_auth(config) + console.print( + Panel( + Text.from_markup( + f"[bold]Workspace:[/bold] {_shorten(ws)}\n" + f"[bold]Port:[/bold] {effective_port}\n" + f"[bold]Auth:[/bold] {_auth_label}" + ), + title="[bold cyan]EvoScientist Deploy[/bold cyan]", + border_style="cyan", + ) + ) + + # 6. ccproxy lifecycle (only if any provider uses OAuth) + _ccproxy_proc = None + if config.anthropic_auth_mode == "oauth" or config.openai_auth_mode == "oauth": + try: + from ..ccproxy_manager import maybe_start_ccproxy, stop_ccproxy + + with console.status( + "[dim]Starting ccproxy (OAuth proxy)...[/dim]", spinner="dots" + ): + _ccproxy_proc = maybe_start_ccproxy(config) + if _ccproxy_proc: + atexit.register(stop_ccproxy, _ccproxy_proc) + console.print("[green]✓[/green] ccproxy started") + except RuntimeError as exc: + console.print(f"[red]ccproxy startup failed:[/red] {exc}") + raise typer.Exit(1) from exc + + # 7. Start langgraph dev (deploy mode → full MCP + async) + jobs_per_worker = int(getattr(config, "langgraph_dev_jobs_per_worker", 10)) + file_persistence = bool(getattr(config, "langgraph_dev_file_persistence", True)) + try: + with console.status( + "[dim]Starting langgraph dev (deploy mode: MCP + async)...[/dim]", + spinner="dots", + ): + proc = start_langgraph_dev( + workspace_dir=Path(ws), + port=effective_port, + file_persistence=file_persistence, + jobs_per_worker=jobs_per_worker, + deploy_mode=True, + ) + atexit.register(stop_langgraph_dev, proc) + except Exception as exc: + console.print(f"[red]langgraph dev startup failed:[/red] {exc}") + raise typer.Exit(1) from exc + + # start_langgraph_dev already health-polled before returning; if we got + # here, the subprocess is up. + console.print("[green]✓[/green] langgraph dev ready") + + # 9. Ready banner + log_hint = _shorten(str(_LOG_FILE)) + console.print( + Panel( + Text.from_markup( + f"[bold]Endpoint:[/bold] " + f"http://localhost:{effective_port}\n" + f"[bold]Assistant ID:[/bold] EvoScientist\n" + f"[bold]Connect via:[/bold] any LangChain SDK / " + f"LangGraph-compatible UI\n" + f"[bold]Logs:[/bold] {log_hint}\n\n" + f"[dim]Press Ctrl+C to stop.[/dim]" + ), + title="[bold green]✓ Ready[/bold green]", + border_style="green", + ) + ) + + # 10. Block on signal — mirror serve's dual-gate (threading.Event + + # explicit SIGINT/SIGTERM handlers) so SIGTERM (no default raise) also + # triggers clean shutdown. + shutdown_event = threading.Event() + + def _handle_shutdown(signum: int, _frame: Any) -> None: + shutdown_event.set() + if signum == signal.SIGINT: + signal.default_int_handler(signum, _frame) + + _orig_sigint = signal.signal(signal.SIGINT, _handle_shutdown) + _orig_sigterm = signal.signal(signal.SIGTERM, _handle_shutdown) + + try: + while not shutdown_event.is_set(): + shutdown_event.wait(timeout=0.5) + except KeyboardInterrupt: + shutdown_event.set() + finally: + signal.signal(signal.SIGINT, _orig_sigint) + signal.signal(signal.SIGTERM, _orig_sigterm) + # stop_langgraph_dev + stop_ccproxy run via atexit during interpreter + # shutdown, so subprocess teardown happens AFTER this print returns. + # Don't claim "Stopped." here — that would be a lie until atexit fires. + console.print( + "\n[dim]Shutting down (background cleanup may take a few seconds)...[/dim]" + ) + + +def _describe_auth(config: Any) -> str: + """Render a one-line auth summary for the startup banner.""" + anth = getattr(config, "anthropic_auth_mode", "api_key") + oai = getattr(config, "openai_auth_mode", "api_key") + if anth == "oauth" and oai == "oauth": + return "OAuth (Anthropic + OpenAI via ccproxy)" + if anth == "oauth": + return "OAuth (Anthropic via ccproxy) + API key (OpenAI)" + if oai == "oauth": + return "API key (Anthropic) + OAuth (OpenAI via ccproxy)" + return "API key" + + +def _shorten(path: str) -> str: + """Replace ``$HOME`` prefix with ``~`` for compact display.""" + home = os.path.expanduser("~") + if path.startswith(home): + return "~" + path[len(home) :] + return path diff --git a/EvoScientist/langgraph_dev/manager.py b/EvoScientist/langgraph_dev/manager.py index 24bc7e1..7e9e2db 100644 --- a/EvoScientist/langgraph_dev/manager.py +++ b/EvoScientist/langgraph_dev/manager.py @@ -12,6 +12,7 @@ Mirrors the lifecycle pattern used by ``ccproxy_manager.py``. from __future__ import annotations import atexit +import json import logging import os import shutil @@ -53,6 +54,87 @@ _PID_DIR = Path.home() / ".config" / "evoscientist" _PID_FILE = _PID_DIR / "langgraph_dev.pid" _LOG_FILE = _PID_DIR / "langgraph_dev.log" +# Workspace fingerprint sidecar — JSON recording the workspace + pid of the +# running langgraph dev. Cross-process callers (e.g. TUI starting up while +# ``EvoSci deploy`` is already running) read this on the reuse path to refuse +# silently operating on a different workspace's files. Missing/corrupt sidecar +# degrades gracefully to a log warning for backward compatibility with +# langgraph devs started before this protocol existed. +_WORKSPACE_SIDECAR = _PID_DIR / "langgraph_dev.workspace.json" + + +class WorkspaceMismatchError(RuntimeError): + """Raised when a caller would reuse a langgraph dev whose recorded + workspace differs from the workspace the caller requested. + + Surfaced by ``ensure_langgraph_dev`` on the cross-process reuse path so + callers (CLI / serve) can print a clear refuse-with-hint message instead + of silently routing async sub-agent calls to a process pinned to a + different workspace. + """ + + +def _write_workspace_sidecar(workspace_dir: Path, pid: int) -> None: + """Record the workspace + pid of the langgraph dev we just started. + + Atomic write via temp-file + ``os.replace``: without this, a concurrent + reader could observe a partially-written file, fail JSON parse, and + silently downgrade to the "no sidecar" fallback path — which skips the + workspace mismatch check entirely. ``os.replace`` is atomic on POSIX + and on Windows; the temp file lives in the same directory so the rename + stays within one filesystem. + + Best-effort: failures are logged and swallowed. A missing sidecar + degrades gracefully to the pre-feature behavior (log-warning only) in + ``ensure_langgraph_dev``. + """ + try: + _PID_DIR.mkdir(parents=True, exist_ok=True) + tmp = _WORKSPACE_SIDECAR.with_suffix(".json.tmp") + tmp.write_text(json.dumps({"workspace": str(workspace_dir), "pid": pid})) + os.replace(tmp, _WORKSPACE_SIDECAR) + except OSError as exc: + logger.warning( + "Failed to write workspace sidecar %s: %s", _WORKSPACE_SIDECAR, exc + ) + + +def _read_workspace_sidecar() -> dict | None: + """Read the workspace sidecar. Returns None if missing, corrupt, or + structurally wrong (must be a dict whose ``workspace`` value is a + non-empty string). + + Schema validation matters because the reuse branch in + ``_ensure_langgraph_dev_locked`` runs ``Path(sidecar["workspace"]).resolve()`` + directly — without the value-type check, a payload like + ``{"workspace": null}`` or ``{"workspace": []}`` would parse fine, pass + a naive ``"workspace" in data`` check, then raise ``TypeError`` inside + ``Path(...)`` and surface as an unhandled exception instead of the + documented log-warning fallback. + """ + if not _WORKSPACE_SIDECAR.exists(): + return None + try: + data = json.loads(_WORKSPACE_SIDECAR.read_text()) + except (OSError, ValueError): + return None + if not isinstance(data, dict): + return None + workspace = data.get("workspace") + if not isinstance(workspace, str) or not workspace: + return None + return data + + +def _unlink_workspace_sidecar() -> None: + """Best-effort sidecar removal — called alongside every ``_PID_FILE.unlink()`` + so the workspace fingerprint never outlives the PID file it pairs with.""" + try: + _WORKSPACE_SIDECAR.unlink() + except OSError: + pass + + # Cross-process file lock for ``ensure_langgraph_dev``. Without this, two # concurrent CLI shells racing on the cold-start window can SIGKILL each # other's still-booting subprocesses (Shell B sees Shell A's port-bound but @@ -73,12 +155,23 @@ _PROCESS: subprocess.Popen | None = None # sub-agents' cwd / EVOSCIENTIST_WORKSPACE_DIR env match the new workspace. _PROCESS_WORKSPACE: Path | None = None -# Whether async sub-agents are usable in this CLI process. Only True after -# ``ensure_langgraph_dev`` confirms the subprocess is healthy (or already -# running). Stays False on startup failure so ``_maybe_swap_async_subagents`` -# can fall back to in-process sync delegation instead of routing tool calls -# at a dead URL. -_ASYNC_SUBAGENTS_AVAILABLE: bool = False +# Whether async sub-agents are usable in this process. +# +# - CLI / serve parent process: starts False; flipped True after +# ``ensure_langgraph_dev`` confirms the subprocess is healthy. Stays False +# on startup failure so ``_maybe_swap_async_subagents`` can fall back to +# in-process sync delegation instead of routing tool calls at a dead URL. +# - langgraph dev subprocess spawned by ``EvoSci deploy``: starts True via +# ``EVOSCIENTIST_DEPLOY_MODE=full`` env var. The deployed main agent IS the +# langgraph dev server, so http://localhost:{port} is always reachable for +# self-loop async sub-agent dispatch. +# - langgraph dev subprocess spawned by ``EvoSci`` / ``EvoSci serve``: env +# var is ``stripped``, stays False — the deployed main agent in that +# subprocess is dead code (only sub-agent graphs are invoked), so async +# swap is unnecessary. +_ASYNC_SUBAGENTS_AVAILABLE: bool = ( + os.environ.get("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "full" +) def is_async_subagents_available() -> bool: @@ -265,6 +358,7 @@ def _kill_owned_stale_process(port: int) -> bool: _PID_FILE.unlink() except OSError: pass + _unlink_workspace_sidecar() return False except psutil.AccessDenied: return False @@ -294,6 +388,7 @@ def _kill_owned_stale_process(port: int) -> bool: _PID_FILE.unlink() except OSError: pass + _unlink_workspace_sidecar() return False try: @@ -304,6 +399,7 @@ def _kill_owned_stale_process(port: int) -> bool: _PID_FILE.unlink() except OSError: pass + _unlink_workspace_sidecar() return True @@ -330,6 +426,7 @@ def start_langgraph_dev( port: int = _DEFAULT_PORT, file_persistence: bool = True, jobs_per_worker: int = 10, + deploy_mode: bool = False, ) -> subprocess.Popen: """Start langgraph dev as a background subprocess. @@ -450,17 +547,28 @@ def start_langgraph_dev( if not file_persistence: sub_env["LANGGRAPH_DISABLE_FILE_PERSISTENCE"] = "true" - # Skip MCP loading inside the langgraph dev subprocess. The CLI's main - # agent already loaded MCP servers in the foreground process; without - # this guard, ``main_graph.py`` would import ``EvoScientist_agent`` and - # trigger ``_get_default_agent`` → ``load_mcp_and_build_kwargs`` → - # spawning a SECOND copy of every MCP server in the subprocess. - # The deployed main agent is currently only reachable via HTTP (for - # future Web UI / SDK clients), and none of those are in use, so the - # duplicate MCP pool is pure waste. Async sub-agents don't load MCP at - # all (their factory bypasses ``load_mcp_and_build_kwargs``), so they - # are unaffected. - sub_env["EVOSCIENTIST_DEPLOYED_NO_MCP"] = "true" + # Subprocess mode flag — single env var with enum values: + # + # - ``EVOSCIENTIST_DEPLOY_MODE=full`` (deploy_mode=True): set by + # ``EvoSci deploy``. Subprocess is the primary programmatic entry + # point; main agent loads MCP and ``_ASYNC_SUBAGENTS_AVAILABLE`` + # flips to True at module load, enabling self-loop async dispatch. + # + # - ``EVOSCIENTIST_DEPLOY_MODE=stripped`` (deploy_mode=False): set by + # ``EvoSci`` / ``EvoSci serve``. The CLI's main agent already loaded + # MCP in the foreground process; the subprocess skips MCP to avoid + # spawning a SECOND copy of every MCP server. The deployed main + # agent in this mode is dead code — only sub-agent graphs are + # invoked over HTTP — so the duplicate MCP pool would be pure waste. + # + # - (unset): parent process or plain ``import EvoScientist``. Loads + # MCP normally; async sub-agents stay disabled (no langgraph dev + # server to self-loop into). + # + # Strip any inherited value first so a stray export in the user's shell + # cannot override the mode resolved by this caller. + sub_env.pop("EVOSCIENTIST_DEPLOY_MODE", None) + sub_env["EVOSCIENTIST_DEPLOY_MODE"] = "full" if deploy_mode else "stripped" try: proc = subprocess.Popen( @@ -489,6 +597,7 @@ def start_langgraph_dev( except Exception: pass _PID_FILE.write_text(str(proc.pid)) + _write_workspace_sidecar(workspace_dir=workspace_dir, pid=proc.pid) global _PROCESS_WORKSPACE _PROCESS = proc _PROCESS_WORKSPACE = workspace_dir @@ -543,52 +652,57 @@ def stop_langgraph_dev(proc: subprocess.Popen | None = None) -> None: with _LOCK: proc = proc if proc is not None else _PROCESS if proc is None: - return - - if proc.poll() is None: - # Cross-platform process-tree shutdown: walk children explicitly - # because POSIX process groups (``os.killpg``) don't exist on - # Windows. ``psutil.Process.children(recursive=True)`` works on - # both — we mirror the previous SIGTERM-then-SIGKILL escalation. - try: - parent = psutil.Process(proc.pid) - descendants = parent.children(recursive=True) - for child in descendants: - try: - child.terminate() - except (psutil.NoSuchProcess, psutil.AccessDenied): - pass - parent.terminate() - proc.wait(timeout=5) - except psutil.NoSuchProcess: - pass - except subprocess.TimeoutExpired: + # No live process to stop, but stale PID/sidecar files may still + # be on disk from a previous run that died unexpectedly — fall + # through to the unconditional file cleanup below so subsequent + # ensure_langgraph_dev calls don't read stale workspace info. + pass + else: + if proc.poll() is None: + # Cross-platform process-tree shutdown: walk children explicitly + # because POSIX process groups (``os.killpg``) don't exist on + # Windows. ``psutil.Process.children(recursive=True)`` works on + # both — we mirror the previous SIGTERM-then-SIGKILL escalation. try: parent = psutil.Process(proc.pid) - for child in parent.children(recursive=True): + descendants = parent.children(recursive=True) + for child in descendants: try: - child.kill() + child.terminate() except (psutil.NoSuchProcess, psutil.AccessDenied): pass - parent.kill() + parent.terminate() + proc.wait(timeout=5) except psutil.NoSuchProcess: pass - # Reap the Popen handle so we don't leave a zombie until - # the CLI itself exits. Short timeout because parent.kill() - # above already issued SIGKILL to the process tree. - try: - proc.wait(timeout=2) except subprocess.TimeoutExpired: - pass + try: + parent = psutil.Process(proc.pid) + for child in parent.children(recursive=True): + try: + child.kill() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + parent.kill() + except psutil.NoSuchProcess: + pass + # Reap the Popen handle so we don't leave a zombie until + # the CLI itself exits. Short timeout because parent.kill() + # above already issued SIGKILL to the process tree. + try: + proc.wait(timeout=2) + except subprocess.TimeoutExpired: + pass - if proc is _PROCESS: - _PROCESS = None - _PROCESS_WORKSPACE = None + if proc is _PROCESS: + _PROCESS = None + _PROCESS_WORKSPACE = None if _PID_FILE.exists(): try: _PID_FILE.unlink() except OSError: pass + _unlink_workspace_sidecar() # Note: ``.langgraph_api/`` is intentionally NOT removed — it holds # langgraph dev's persisted async-task / scheduler / Store state that @@ -699,20 +813,46 @@ def _ensure_langgraph_dev_locked( _ASYNC_SUBAGENTS_AVAILABLE = False # cleared until restart succeeds if is_langgraph_dev_running(port=port): - # If WE own the running process, workspace was already verified above - # via _PROCESS_WORKSPACE comparison. If we DON'T own it (some other - # langgraph dev started by the user / another CLI), we have no way to - # confirm its workspace matches what was just requested — async - # sub-agents could end up operating on a different project's files. - # Warn loudly so the user notices. - if _PROCESS is None and ws_path is not None: - logger.warning( - "Reusing externally-managed langgraph dev on %s — cannot verify " - "its workspace matches the requested %s. Async sub-agents may " - "operate on a different workspace's files.", - _base_url(port), - ws_path, - ) + # If WE own the running process AND it's still alive, workspace was + # already verified above via _PROCESS_WORKSPACE comparison. Otherwise + # — we never owned it (EvoSci deploy in another terminal, or a + # langgraph dev the user spawned manually) OR our handle is stale (our + # subprocess died and a different one rebound the port) — check the + # workspace sidecar, the only cross-process source of truth for the + # running instance's workspace. A stale non-None _PROCESS must NOT + # short-circuit this check, or we'd silently reuse a wrong-workspace + # server. + owned_running = _PROCESS is not None and _PROCESS.poll() is None + if not owned_running and ws_path is not None: + sidecar = _read_workspace_sidecar() + if sidecar is not None: + recorded = Path(sidecar["workspace"]).resolve() + if recorded != ws_path.resolve(): + raise WorkspaceMismatchError( + f"An EvoSci langgraph dev is already running on " + f"{_base_url(port)} for workspace {recorded}, but the " + f"current process requested workspace {ws_path}. " + f"Stop the other EvoSci session (deploy / TUI / serve) " + f"or rerun with --workdir {recorded}." + ) + logger.info( + "Reusing externally-managed langgraph dev on %s; sidecar " + "confirms matching workspace %s.", + _base_url(port), + recorded, + ) + else: + # Pre-feature langgraph dev — no sidecar to verify against. + # Fall back to the original log-warning behavior so users + # running an older subprocess don't get bricked. + logger.warning( + "Reusing externally-managed langgraph dev on %s — no " + "workspace sidecar, cannot verify it matches the requested " + "%s. Async sub-agents may operate on a different workspace's " + "files.", + _base_url(port), + ws_path, + ) else: logger.info("langgraph dev already running on %s, reusing", _base_url(port)) _ASYNC_SUBAGENTS_AVAILABLE = True diff --git a/README.md b/README.md index 002f4b4..cb8f950 100644 --- a/README.md +++ b/README.md @@ -383,6 +383,7 @@ EvoSci --workdir /path/to/project # open in a specific directory EvoSci -m run # isolated per-session workspace EvoSci --ui cli # classic CLI (lightweight) EvoSci serve # headless mode — channels only, no interactive prompt +EvoSci deploy # standalone LangGraph server for external UIs / SDK clients ``` diff --git a/README.zh-CN.md b/README.zh-CN.md index 4c8570a..cca5410 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -392,6 +392,7 @@ EvoSci --workdir /path/to/project # 在指定目录下启动 EvoSci -m run # 隔离的会话级工作区 EvoSci --ui cli # 经典 CLI(轻量) EvoSci serve # 无头模式——仅渠道,无交互提示符 +EvoSci deploy # 独立 LangGraph 服务器——供外部 UI / SDK 客户端使用 ``` diff --git a/tests/test_cli_deploy.py b/tests/test_cli_deploy.py new file mode 100644 index 0000000..7b18508 --- /dev/null +++ b/tests/test_cli_deploy.py @@ -0,0 +1,396 @@ +"""Tests for ``EvoSci deploy`` command flow. + +Verifies the orchestration: +- workspace resolution (CLI > config > cwd) +- port resolution (CLI > config > default) +- port collision pre-flight +- ccproxy lifecycle (only if OAuth configured) +- ``start_langgraph_dev(deploy_mode=True)`` invocation +- clean shutdown via signal handler / KeyboardInterrupt +""" + +from __future__ import annotations + +from types import SimpleNamespace +from typing import Any + +import pytest +import typer + +from EvoScientist.deploy import server as deploy_server + + +def _make_config( + *, + default_workdir: str = "", + langgraph_dev_port: int = 6174, + anthropic_auth_mode: str = "api_key", + openai_auth_mode: str = "api_key", + log_level: str = "warning", + langgraph_dev_jobs_per_worker: int = 10, + langgraph_dev_file_persistence: bool = True, +): + return SimpleNamespace( + default_workdir=default_workdir, + langgraph_dev_port=langgraph_dev_port, + anthropic_auth_mode=anthropic_auth_mode, + openai_auth_mode=openai_auth_mode, + log_level=log_level, + langgraph_dev_jobs_per_worker=langgraph_dev_jobs_per_worker, + langgraph_dev_file_persistence=langgraph_dev_file_persistence, + ) + + +class _ImmediateEvent: + """Fake ``threading.Event``: first ``is_set()`` returns False so the + while-loop body runs once; subsequent calls return True, exiting the + loop. ``wait`` is a no-op (no real blocking).""" + + def __init__(self): + self._called = 0 + self.set_was_called = False + + def is_set(self) -> bool: + self._called += 1 + return self._called > 1 + + def wait(self, timeout: float | None = None): + return None + + def set(self): + self.set_was_called = True + self._called = 99 + + +def _run_deploy_once( + monkeypatch, + config, + *, + workdir: str | None = None, + port: int | None = None, + debug: bool = False, + cwd: str | None = None, + port_occupied: bool = False, + langgraph_dev_running: bool = True, # health-check passes after start +): + """Run ``deploy()`` end-to-end with all external dependencies mocked. + Returns a ``captured`` dict with observation points.""" + import EvoScientist.config as config_mod + + captured: dict[str, Any] = { + "ccproxy_started": False, + "ccproxy_stopped": False, + "langgraph_dev_started": False, + "langgraph_dev_stopped": False, + "deploy_mode_passed": None, + "workspace_passed": None, + "port_passed": None, + "atexit_callbacks": [], + } + + def _fake_get_effective_config(cli_overrides=None): + captured["cli_overrides"] = dict(cli_overrides or {}) + merged = vars(config).copy() + merged.update(cli_overrides or {}) + return SimpleNamespace(**merged) + + monkeypatch.setattr(config_mod, "get_effective_config", _fake_get_effective_config) + monkeypatch.setattr(config_mod, "apply_config_to_env", lambda _cfg: None) + + monkeypatch.setattr(deploy_server, "console", _SilentConsole()) + + # Workspace setup mocks + from EvoScientist import paths as paths_mod + + monkeypatch.setattr(paths_mod, "set_workspace_root", lambda _p: None) + monkeypatch.setattr(paths_mod, "ensure_dirs", lambda: None) + + # langgraph_dev.manager mocks + from EvoScientist.langgraph_dev import manager as lgm + + monkeypatch.setattr(lgm, "_is_port_occupied", lambda _p: port_occupied) + monkeypatch.setattr( + lgm, + "is_langgraph_dev_running", + lambda **_kw: langgraph_dev_running, + ) + + def _fake_start_langgraph_dev( + workspace_dir=None, + *, + port=None, + file_persistence=True, + jobs_per_worker=10, + deploy_mode=False, + ): + captured["langgraph_dev_started"] = True + captured["workspace_passed"] = str(workspace_dir) if workspace_dir else None + captured["port_passed"] = port + captured["deploy_mode_passed"] = deploy_mode + captured["jobs_per_worker_passed"] = jobs_per_worker + captured["file_persistence_passed"] = file_persistence + return SimpleNamespace(pid=99999) + + def _fake_stop_langgraph_dev(_proc=None): + captured["langgraph_dev_stopped"] = True + + monkeypatch.setattr(lgm, "start_langgraph_dev", _fake_start_langgraph_dev) + monkeypatch.setattr(lgm, "stop_langgraph_dev", _fake_stop_langgraph_dev) + + # ccproxy mocks + from EvoScientist import ccproxy_manager as ccp + + def _fake_maybe_start_ccproxy(_cfg): + captured["ccproxy_started"] = True + return SimpleNamespace(pid=88888) + + def _fake_stop_ccproxy(_proc): + captured["ccproxy_stopped"] = True + + monkeypatch.setattr(ccp, "maybe_start_ccproxy", _fake_maybe_start_ccproxy) + monkeypatch.setattr(ccp, "stop_ccproxy", _fake_stop_ccproxy) + + # atexit mock — capture without executing (don't pollute test process) + import atexit + + def _fake_atexit_register(fn, *args, **kwargs): + captured["atexit_callbacks"].append((fn.__name__, args, kwargs)) + return fn + + monkeypatch.setattr(atexit, "register", _fake_atexit_register) + + # signal mock — capture handlers so tests can exercise _handle_shutdown. + # Returning a no-op original means deploy()'s finally block restores + # something harmless onto the real signal module. + import signal + + # deploy() calls signal.signal twice per signum: first to install + # _handle_shutdown, then in finally to restore the original. We want the + # first (real) handler — keep first-write-wins semantics. + captured["signal_handlers"] = {} + + def _capture_signal(signum, handler): + if signum not in captured["signal_handlers"]: + captured["signal_handlers"][signum] = handler + return lambda *_a, **_kw: None + + monkeypatch.setattr(signal, "signal", _capture_signal) + + # threading.Event mock — exits the wait loop after one iteration; + # factory captures the instance so tests can inspect set() calls. + import threading + + def _make_event(): + ev = _ImmediateEvent() + captured["event_instance"] = ev + return ev + + monkeypatch.setattr(threading, "Event", _make_event) + + # os.makedirs / os.getcwd + import os + + monkeypatch.setattr(os, "makedirs", lambda *a, **k: None) + if cwd is not None: + monkeypatch.setattr(os, "getcwd", lambda: cwd) + + deploy_server.deploy(workdir=workdir, port=port, debug=debug) + return captured + + +class _SilentConsole: + """Stand-in for the Rich console — swallows all output so test runs + don't spew ANSI to the captured pytest output (but doesn't break the + code paths that call ``console.print`` / ``console.status``).""" + + def print(self, *args, **kwargs): + pass + + def status(self, *args, **kwargs): + class _Ctx: + def __enter__(self_inner): + return self_inner + + def __exit__(self_inner, *a): + return False + + return _Ctx() + + +# ============================================================================= +# Tests +# ============================================================================= + + +def test_deploy_starts_langgraph_dev_with_deploy_mode_true(monkeypatch, tmp_path): + config = _make_config(default_workdir=str(tmp_path)) + captured = _run_deploy_once(monkeypatch, config) + + assert captured["langgraph_dev_started"] is True + assert captured["deploy_mode_passed"] is True, ( + "deploy command MUST call start_langgraph_dev with deploy_mode=True" + ) + + +def test_deploy_workdir_cli_arg_beats_config(monkeypatch, tmp_path): + cli_ws = tmp_path / "cli_ws" + cfg_ws = tmp_path / "cfg_ws" + config = _make_config(default_workdir=str(cfg_ws)) + captured = _run_deploy_once(monkeypatch, config, workdir=str(cli_ws)) + + assert captured["workspace_passed"] == str(cli_ws) + + +def test_deploy_workdir_config_beats_cwd(monkeypatch, tmp_path): + cfg_ws = tmp_path / "cfg_ws" + config = _make_config(default_workdir=str(cfg_ws)) + captured = _run_deploy_once(monkeypatch, config, cwd="/tmp/should_not_be_used") + + assert captured["workspace_passed"] == str(cfg_ws) + + +def test_deploy_workdir_falls_back_to_cwd(monkeypatch, tmp_path): + config = _make_config(default_workdir="") + cwd = str(tmp_path / "cwd") + captured = _run_deploy_once(monkeypatch, config, cwd=cwd) + + assert captured["workspace_passed"] == cwd + + +def test_deploy_port_cli_arg_beats_config(monkeypatch, tmp_path): + config = _make_config(default_workdir=str(tmp_path), langgraph_dev_port=6174) + captured = _run_deploy_once(monkeypatch, config, port=7000) + + assert captured["port_passed"] == 7000 + + +def test_deploy_port_defaults_to_config(monkeypatch, tmp_path): + config = _make_config(default_workdir=str(tmp_path), langgraph_dev_port=6543) + captured = _run_deploy_once(monkeypatch, config) + + assert captured["port_passed"] == 6543 + + +@pytest.mark.parametrize("bad_port", [0, -1, 70000]) +def test_deploy_refuses_invalid_port(monkeypatch, tmp_path, bad_port): + """CLI must reject out-of-range ports (port=0 was the original silent-fail + case: ``port or default`` treated 0 as falsy).""" + config = _make_config(default_workdir=str(tmp_path)) + with pytest.raises(typer.Exit) as exc: + _run_deploy_once(monkeypatch, config, port=bad_port) + assert exc.value.exit_code == 1 + + +def test_deploy_no_ccproxy_when_api_key_auth(monkeypatch, tmp_path): + config = _make_config( + default_workdir=str(tmp_path), + anthropic_auth_mode="api_key", + openai_auth_mode="api_key", + ) + captured = _run_deploy_once(monkeypatch, config) + + assert captured["ccproxy_started"] is False + + +def test_deploy_starts_ccproxy_when_anthropic_oauth(monkeypatch, tmp_path): + config = _make_config(default_workdir=str(tmp_path), anthropic_auth_mode="oauth") + captured = _run_deploy_once(monkeypatch, config) + + assert captured["ccproxy_started"] is True + + +def test_deploy_starts_ccproxy_when_openai_oauth(monkeypatch, tmp_path): + config = _make_config(default_workdir=str(tmp_path), openai_auth_mode="oauth") + captured = _run_deploy_once(monkeypatch, config) + + assert captured["ccproxy_started"] is True + + +def test_deploy_refuses_when_port_occupied_by_evosci(monkeypatch, tmp_path): + """If port is occupied AND it's serving an EvoSci langgraph dev, + refuse with exit code 1.""" + config = _make_config(default_workdir=str(tmp_path)) + with pytest.raises(typer.Exit) as exc: + _run_deploy_once( + monkeypatch, + config, + port_occupied=True, + langgraph_dev_running=True, # /ok responds → existing EvoSci instance + ) + assert exc.value.exit_code == 1 + + +def test_deploy_refuses_when_port_occupied_by_foreign(monkeypatch, tmp_path): + """If port is occupied but /ok doesn't respond, treat as foreign process + and refuse.""" + config = _make_config(default_workdir=str(tmp_path)) + with pytest.raises(typer.Exit) as exc: + _run_deploy_once( + monkeypatch, + config, + port_occupied=True, + langgraph_dev_running=False, # foreign process holds the port + ) + assert exc.value.exit_code == 1 + + +def test_deploy_registers_cleanup_for_langgraph_dev(monkeypatch, tmp_path): + config = _make_config(default_workdir=str(tmp_path)) + captured = _run_deploy_once(monkeypatch, config) + + names = [name for (name, _args, _kw) in captured["atexit_callbacks"]] + assert "_fake_stop_langgraph_dev" in names, ( + "deploy must register stop_langgraph_dev via atexit for clean shutdown" + ) + + +def test_deploy_registers_cleanup_for_ccproxy_when_oauth(monkeypatch, tmp_path): + config = _make_config(default_workdir=str(tmp_path), anthropic_auth_mode="oauth") + captured = _run_deploy_once(monkeypatch, config) + + names = [name for (name, _args, _kw) in captured["atexit_callbacks"]] + assert "_fake_stop_ccproxy" in names + + +def test_deploy_registers_signal_handlers_for_shutdown(monkeypatch, tmp_path): + """deploy() must register SIGINT and SIGTERM handlers so external signals + trigger clean shutdown via the shutdown_event wait loop.""" + import signal + + config = _make_config(default_workdir=str(tmp_path)) + captured = _run_deploy_once(monkeypatch, config) + + handlers = captured["signal_handlers"] + assert signal.SIGINT in handlers, "deploy must register a SIGINT handler" + assert signal.SIGTERM in handlers, "deploy must register a SIGTERM handler" + assert callable(handlers[signal.SIGINT]) + assert callable(handlers[signal.SIGTERM]) + + +def test_handle_shutdown_sigterm_sets_shutdown_event(monkeypatch, tmp_path): + """Invoking the captured SIGTERM handler must set deploy()'s shutdown_event. + + SIGTERM is used (not SIGINT) because the SIGINT branch of _handle_shutdown + calls signal.default_int_handler which raises KeyboardInterrupt — that + would terminate the test process rather than exercise the event path. + """ + import signal + + config = _make_config(default_workdir=str(tmp_path)) + captured = _run_deploy_once(monkeypatch, config) + + event = captured["event_instance"] + assert event is not None, "deploy() must have constructed a threading.Event" + # During the normal helper run the wait loop exits via is_set() flipping + # to True (not via set()), so set_was_called should still be False here. + assert event.set_was_called is False, ( + "Sanity check: helper's _ImmediateEvent should exit naturally without " + "set() being called; if this fails, the helper changed behavior." + ) + + sigterm_handler = captured["signal_handlers"][signal.SIGTERM] + sigterm_handler(signal.SIGTERM, None) + + assert event.set_was_called is True, ( + "_handle_shutdown(SIGTERM, None) must call shutdown_event.set()" + ) diff --git a/tests/test_langgraph_dev_deploy_mode.py b/tests/test_langgraph_dev_deploy_mode.py new file mode 100644 index 0000000..56fd1af --- /dev/null +++ b/tests/test_langgraph_dev_deploy_mode.py @@ -0,0 +1,256 @@ +"""Tests for ``start_langgraph_dev(deploy_mode=...)`` env var injection. + +Verifies the single-env-var enum routing: +- ``deploy_mode=True`` → ``EVOSCIENTIST_DEPLOY_MODE=full`` +- ``deploy_mode=False`` → ``EVOSCIENTIST_DEPLOY_MODE=stripped`` +- (parent process / plain import) → ``EVOSCIENTIST_DEPLOY_MODE`` unset +""" + +from __future__ import annotations + +import subprocess +from pathlib import Path + +import pytest + +from EvoScientist.langgraph_dev import manager + + +class _PopenAbort(Exception): + """Raised by the fake ``Popen`` to short-circuit ``start_langgraph_dev`` + after the env dict is constructed but before health-polling runs.""" + + +def _patch_start_prereqs(monkeypatch, tmp_path: Path) -> dict: + """Mock everything ``start_langgraph_dev`` does before ``subprocess.Popen`` + so we can run it end-to-end up to the point where the env dict is captured. + Returns a ``captured`` dict that the test populates from the fake Popen.""" + captured: dict = {} + + monkeypatch.setattr(manager, "_langgraph_exe", lambda: "/usr/bin/langgraph") + + fake_config = tmp_path / "langgraph.json" + fake_config.write_text("{}") + monkeypatch.setattr(manager, "_packaged_langgraph_config", lambda: fake_config) + + # No conflicts, no stale process — straight to spawn. + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_: False) + monkeypatch.setattr(manager, "_is_port_occupied", lambda _port: False) + monkeypatch.setattr(manager, "_wait_for_port_bindable", lambda _port: True) + monkeypatch.setattr(manager, "_kill_owned_stale_process", lambda _port: False) + monkeypatch.setattr( + manager, "_wait_for_port_release", lambda _port, timeout=10.0: True + ) + + # _PID_DIR / _LOG_FILE point under user dir — redirect to tmp. + pid_dir = tmp_path / "pid_dir" + monkeypatch.setattr(manager, "_PID_DIR", pid_dir) + monkeypatch.setattr(manager, "_LOG_FILE", tmp_path / "langgraph_dev.log") + + def _fake_popen(args, **kwargs): + captured["args"] = args + captured["env"] = kwargs.get("env", {}) + captured["cwd"] = kwargs.get("cwd") + raise _PopenAbort("env captured") + + monkeypatch.setattr(subprocess, "Popen", _fake_popen) + return captured + + +def test_deploy_mode_true_sets_full(monkeypatch, tmp_path): + captured = _patch_start_prereqs(monkeypatch, tmp_path) + + with pytest.raises(_PopenAbort): + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16174, + deploy_mode=True, + ) + + env = captured["env"] + assert env.get("EVOSCIENTIST_DEPLOY_MODE") == "full", ( + "deploy_mode=True must inject EVOSCIENTIST_DEPLOY_MODE=full" + ) + + +def test_deploy_mode_false_default_sets_stripped(monkeypatch, tmp_path): + captured = _patch_start_prereqs(monkeypatch, tmp_path) + + with pytest.raises(_PopenAbort): + # deploy_mode omitted → defaults to False + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16175, + ) + + env = captured["env"] + assert env.get("EVOSCIENTIST_DEPLOY_MODE") == "stripped", ( + "deploy_mode=False (default) must inject EVOSCIENTIST_DEPLOY_MODE=stripped" + ) + + +def test_deploy_mode_explicitly_false_sets_stripped(monkeypatch, tmp_path): + """Same as default, but with deploy_mode=False stated explicitly.""" + captured = _patch_start_prereqs(monkeypatch, tmp_path) + + with pytest.raises(_PopenAbort): + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16176, + deploy_mode=False, + ) + + env = captured["env"] + assert env.get("EVOSCIENTIST_DEPLOY_MODE") == "stripped" + + +def test_deploy_mode_always_set_to_one_of_full_or_stripped(monkeypatch, tmp_path): + """Regression: the subprocess always sees exactly one of the two enum + values for ``EVOSCIENTIST_DEPLOY_MODE`` — never unset, never garbage.""" + for deploy_mode, expected in ((True, "full"), (False, "stripped")): + captured = _patch_start_prereqs(monkeypatch, tmp_path) + with pytest.raises(_PopenAbort): + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16177, + deploy_mode=deploy_mode, + ) + env = captured["env"] + assert env.get("EVOSCIENTIST_DEPLOY_MODE") == expected, ( + f"deploy_mode={deploy_mode}: expected EVOSCIENTIST_DEPLOY_MODE=" + f"{expected!r}, got {env.get('EVOSCIENTIST_DEPLOY_MODE')!r}" + ) + + +def test_inherited_stripped_overridden_when_deploy_mode_true(monkeypatch, tmp_path): + """If the parent process exports ``EVOSCIENTIST_DEPLOY_MODE=stripped`` and + we ask for deploy mode, the subprocess env must see the resolved value + (``full``), not the stale inherited one.""" + monkeypatch.setenv("EVOSCIENTIST_DEPLOY_MODE", "stripped") + captured = _patch_start_prereqs(monkeypatch, tmp_path) + + with pytest.raises(_PopenAbort): + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16180, + deploy_mode=True, + ) + + env = captured["env"] + assert env.get("EVOSCIENTIST_DEPLOY_MODE") == "full", ( + "inherited stripped value must be overridden when deploy_mode=True" + ) + + +def test_inherited_full_overridden_when_deploy_mode_false(monkeypatch, tmp_path): + """Symmetric: parent exports ``EVOSCIENTIST_DEPLOY_MODE=full``, CLI/serve + calls start_langgraph_dev with default (deploy_mode=False), inherited + value must be overridden to ``stripped``.""" + monkeypatch.setenv("EVOSCIENTIST_DEPLOY_MODE", "full") + captured = _patch_start_prereqs(monkeypatch, tmp_path) + + with pytest.raises(_PopenAbort): + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16181, + ) + + env = captured["env"] + assert env.get("EVOSCIENTIST_DEPLOY_MODE") == "stripped", ( + "inherited full value must be overridden when deploy_mode=False" + ) + + +def test_inherited_arbitrary_value_overridden(monkeypatch, tmp_path): + """Defense against an unexpected inherited value (e.g. legacy ``true`` + from before the enum rename, or any user-set garbage). The resolved + deploy_mode always wins.""" + for inherited in ("true", "garbage", "FULL", ""): + for deploy_mode, expected in ((True, "full"), (False, "stripped")): + monkeypatch.setenv("EVOSCIENTIST_DEPLOY_MODE", inherited) + captured = _patch_start_prereqs(monkeypatch, tmp_path) + + with pytest.raises(_PopenAbort): + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16182, + deploy_mode=deploy_mode, + ) + + env = captured["env"] + assert env.get("EVOSCIENTIST_DEPLOY_MODE") == expected, ( + f"inherited={inherited!r}, deploy_mode={deploy_mode}: " + f"expected EVOSCIENTIST_DEPLOY_MODE={expected!r}, " + f"got {env.get('EVOSCIENTIST_DEPLOY_MODE')!r}" + ) + + +def test_workspace_dir_env_var_set_regardless_of_mode(monkeypatch, tmp_path): + """EVOSCIENTIST_WORKSPACE_DIR is independent of deploy_mode.""" + for deploy_mode in (True, False): + captured = _patch_start_prereqs(monkeypatch, tmp_path) + with pytest.raises(_PopenAbort): + manager.start_langgraph_dev( + workspace_dir=tmp_path, + port=16178, + deploy_mode=deploy_mode, + ) + assert captured["env"].get("EVOSCIENTIST_WORKSPACE_DIR") == str(tmp_path) + + +# ============================================================================= +# Module-load behavior — _ASYNC_SUBAGENTS_AVAILABLE reads env var on import +# ============================================================================= + + +def test_async_subagents_available_init_from_env_full(monkeypatch): + """When ``EVOSCIENTIST_DEPLOY_MODE=full`` is set in the env at module + import time, ``_ASYNC_SUBAGENTS_AVAILABLE`` initializes to True so the + deployed main agent's ``_maybe_swap_async_subagents`` swaps eagerly + without waiting for ``start_langgraph_dev`` to flip the flag (which it + can't — the deploy subprocess never calls that function on itself).""" + monkeypatch.setenv("EVOSCIENTIST_DEPLOY_MODE", "full") + # Re-import the module to re-run the module-level initialization. + import importlib + + import EvoScientist.langgraph_dev.manager as mgr + + reloaded = importlib.reload(mgr) + try: + assert reloaded._ASYNC_SUBAGENTS_AVAILABLE is True + assert reloaded.is_async_subagents_available() is True + finally: + # Restore: reload again without the env var so subsequent tests + # see the normal initialization. + monkeypatch.delenv("EVOSCIENTIST_DEPLOY_MODE", raising=False) + importlib.reload(mgr) + + +def test_async_subagents_available_init_false_for_stripped(monkeypatch): + """``stripped`` is the CLI/serve subprocess mode — async sub-agents stay + disabled at module-load time (they get enabled later by ``ensure_langgraph_dev`` + in the parent process, NOT by the subprocess flipping its own flag).""" + monkeypatch.setenv("EVOSCIENTIST_DEPLOY_MODE", "stripped") + import importlib + + import EvoScientist.langgraph_dev.manager as mgr + + reloaded = importlib.reload(mgr) + try: + assert reloaded._ASYNC_SUBAGENTS_AVAILABLE is False + finally: + monkeypatch.delenv("EVOSCIENTIST_DEPLOY_MODE", raising=False) + importlib.reload(mgr) + + +def test_async_subagents_available_init_false_without_env(monkeypatch): + """When the env var is unset, ``_ASYNC_SUBAGENTS_AVAILABLE`` initializes + to False — the pre-existing safety behavior (fall back to sync if + langgraph dev isn't reachable).""" + monkeypatch.delenv("EVOSCIENTIST_DEPLOY_MODE", raising=False) + import importlib + + import EvoScientist.langgraph_dev.manager as mgr + + reloaded = importlib.reload(mgr) + assert reloaded._ASYNC_SUBAGENTS_AVAILABLE is False diff --git a/tests/test_langgraph_dev_workspace_sidecar.py b/tests/test_langgraph_dev_workspace_sidecar.py new file mode 100644 index 0000000..56e9b57 --- /dev/null +++ b/tests/test_langgraph_dev_workspace_sidecar.py @@ -0,0 +1,183 @@ +"""Tests for the workspace-fingerprint sidecar protocol. + +When langgraph dev is reused across processes (e.g., TUI / serve detects +a deploy-started instance on the configured port), the workspace recorded +in the sidecar JSON must match the workspace requested by the caller. On +mismatch we raise ``WorkspaceMismatchError`` so callers can surface a +clear refuse-with-hint error rather than silently operating on the wrong +project's files. + +Background: ``EvoSci deploy --workdir /A`` running + ``EvoSci`` (TUI) in +/B previously took the "reuse externally-managed langgraph dev" branch +in ``ensure_langgraph_dev`` and only logged a warning. The deployed +sub-agents stayed pinned to /A while the TUI's main agent ran in /B, +breaking ``task()`` delegations. +""" + +from __future__ import annotations + +import json + +import pytest + +from EvoScientist.langgraph_dev import manager + + +def test_sidecar_path_is_next_to_pid_file(): + """Sidecar JSON lives at _PID_DIR / 'langgraph_dev.workspace.json'.""" + assert ( + manager._WORKSPACE_SIDECAR == manager._PID_DIR / "langgraph_dev.workspace.json" + ) + + +def test_write_workspace_sidecar_records_workspace_and_pid(tmp_path, monkeypatch): + """``_write_workspace_sidecar`` writes JSON with workspace + pid.""" + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", tmp_path / "ws.json") + workspace = tmp_path / "some" / "ws" + manager._write_workspace_sidecar(workspace_dir=workspace, pid=12345) + data = json.loads((tmp_path / "ws.json").read_text()) + assert data["workspace"] == str(workspace) + assert data["pid"] == 12345 + + +def test_read_workspace_sidecar_returns_none_when_missing(tmp_path, monkeypatch): + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", tmp_path / "absent.json") + assert manager._read_workspace_sidecar() is None + + +def test_read_workspace_sidecar_returns_none_on_corrupt_json(tmp_path, monkeypatch): + sidecar = tmp_path / "bad.json" + sidecar.write_text("not json at all") + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", sidecar) + assert manager._read_workspace_sidecar() is None + + +@pytest.mark.parametrize( + "payload", + [ + "[]", # valid JSON, wrong top-level type + "{}", # valid JSON dict but missing "workspace" + '{"pid": 12345}', # valid JSON dict but missing "workspace" + '"just a string"', # valid JSON scalar + "null", # valid JSON null + '{"workspace": null}', # workspace present but null → Path(None) TypeError + '{"workspace": []}', # workspace present but list → Path([]) TypeError + '{"workspace": 12345}', # workspace present but int → Path(int) TypeError + '{"workspace": ""}', # workspace present but empty string → resolves to cwd + ], +) +def test_read_workspace_sidecar_returns_none_on_wrong_schema( + payload, tmp_path, monkeypatch +): + """JSON that parses but doesn't match the expected schema must degrade + to None — otherwise the reuse branch's ``Path(sidecar["workspace"]).resolve()`` + would raise KeyError/TypeError or silently resolve to cwd, surfacing as + an unhandled exception or producing a misleading match check.""" + sidecar = tmp_path / "schema.json" + sidecar.write_text(payload) + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", sidecar) + assert manager._read_workspace_sidecar() is None + + +def test_read_workspace_sidecar_round_trip(tmp_path, monkeypatch): + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", tmp_path / "rt.json") + workspace = tmp_path / "x" / "y" + manager._write_workspace_sidecar(workspace_dir=workspace, pid=42) + data = manager._read_workspace_sidecar() + assert data == {"workspace": str(workspace), "pid": 42} + + +def test_workspace_mismatch_error_is_runtime_error_subclass(): + assert issubclass(manager.WorkspaceMismatchError, RuntimeError) + + +def test_ensure_langgraph_dev_refuses_on_workspace_mismatch(tmp_path, monkeypatch): + """Cross-process reuse with sidecar workspace ≠ requested → raises.""" + ws_a = tmp_path / "A" + ws_b = tmp_path / "B" + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", tmp_path / "ws.json") + manager._write_workspace_sidecar(workspace_dir=ws_a, pid=99999) + + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_kw: True) + monkeypatch.setattr(manager, "_PROCESS", None) + monkeypatch.setattr(manager, "_PROCESS_WORKSPACE", None) + + cfg = manager.EvoScientistConfig() + cfg.enable_async_subagents = True + with pytest.raises(manager.WorkspaceMismatchError) as exc: + manager.ensure_langgraph_dev(cfg, workspace_dir=ws_b) + assert str(ws_a.resolve()) in str(exc.value) + assert str(ws_b) in str(exc.value) + + +def test_ensure_langgraph_dev_refuses_on_mismatch_with_stale_process( + tmp_path, monkeypatch +): + """A non-None but dead ``_PROCESS`` handle must NOT short-circuit the + sidecar check. Regression for the case where our subprocess exited and a + different langgraph dev rebound the port.""" + ws_a = tmp_path / "A" + ws_b = tmp_path / "B" + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", tmp_path / "ws.json") + manager._write_workspace_sidecar(workspace_dir=ws_a, pid=99999) + + class _DeadProc: + def poll(self): + return 1 # non-None → process has exited + + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_kw: True) + monkeypatch.setattr(manager, "_PROCESS", _DeadProc()) + # _PROCESS_WORKSPACE matches ws_b so the earlier owned-restart branch (which + # also gates on _PROCESS.poll() is None) doesn't fire on this dead handle. + monkeypatch.setattr(manager, "_PROCESS_WORKSPACE", ws_b) + + cfg = manager.EvoScientistConfig() + cfg.enable_async_subagents = True + with pytest.raises(manager.WorkspaceMismatchError): + manager.ensure_langgraph_dev(cfg, workspace_dir=ws_b) + + +def test_ensure_langgraph_dev_reuses_when_workspace_matches(tmp_path, monkeypatch): + """Cross-process reuse with matching sidecar workspace → no raise.""" + ws_a = tmp_path / "A" + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", tmp_path / "ws.json") + manager._write_workspace_sidecar(workspace_dir=ws_a, pid=99999) + + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_kw: True) + monkeypatch.setattr(manager, "_PROCESS", None) + monkeypatch.setattr(manager, "_PROCESS_WORKSPACE", None) + + cfg = manager.EvoScientistConfig() + cfg.enable_async_subagents = True + # Should NOT raise. + manager.ensure_langgraph_dev(cfg, workspace_dir=ws_a) + + +def test_ensure_langgraph_dev_reuses_when_sidecar_missing(tmp_path, monkeypatch): + """Backward compat: pre-feature langgraph dev (no sidecar) falls back to + the existing log-warning behavior rather than refusing.""" + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", tmp_path / "absent.json") + monkeypatch.setattr(manager, "is_langgraph_dev_running", lambda **_kw: True) + monkeypatch.setattr(manager, "_PROCESS", None) + monkeypatch.setattr(manager, "_PROCESS_WORKSPACE", None) + + cfg = manager.EvoScientistConfig() + cfg.enable_async_subagents = True + # Should NOT raise — degrades to the prior reuse-with-warning branch. + manager.ensure_langgraph_dev(cfg, workspace_dir=tmp_path / "B") + + +def test_stop_langgraph_dev_removes_sidecar(tmp_path, monkeypatch): + """``stop_langgraph_dev`` should unlink the sidecar alongside the PID file.""" + sidecar = tmp_path / "ws.json" + pid_file = tmp_path / "pid.txt" + monkeypatch.setattr(manager, "_WORKSPACE_SIDECAR", sidecar) + monkeypatch.setattr(manager, "_PID_FILE", pid_file) + manager._write_workspace_sidecar(workspace_dir=tmp_path / "x", pid=42) + assert sidecar.exists() + + # _PROCESS is None so stop_langgraph_dev shouldn't try to kill anything; + # we're only verifying the sidecar cleanup path here. + monkeypatch.setattr(manager, "_PROCESS", None) + manager.stop_langgraph_dev() + assert not sidecar.exists()