diff --git a/.gitignore b/.gitignore index 1a8ce31..1295760 100644 --- a/.gitignore +++ b/.gitignore @@ -9,6 +9,7 @@ dist/ build/ *.egg *.pytest_cache/ +.benchmarks/ .coverage .ipynb_checkpoints/ diff --git a/EvoScientist/EvoScientist.py b/EvoScientist/EvoScientist.py index 3a57630..2151bec 100644 --- a/EvoScientist/EvoScientist.py +++ b/EvoScientist/EvoScientist.py @@ -37,7 +37,7 @@ logging.getLogger("deepagents.middleware.skills").setLevel(logging.ERROR) # Constants # ============================================================================= -SUBAGENTS_CONFIG = Path(__file__).parent / "subagent.yaml" +SUBAGENTS_CONFIG = Path(__file__).parent / "subagents" SKILLS_DIR = str(Path(__file__).parent / "skills") # ============================================================================= @@ -221,6 +221,85 @@ def _build_prompt_refs() -> dict: } +def _maybe_swap_async_subagents(subs: list) -> list: + """Replace ``_async``-flagged sub-agents with ``AsyncSubAgent`` specs when enabled. + + Reads the ``_async`` field carried through by ``utils.load_subagents._build_one`` + (sourced from each yaml's ``async: true`` flag). When + ``config.enable_async_subagents`` is also set, those sub-agents are + swapped from synchronous in-process dicts to ``AsyncSubAgent`` references + pointing at the langgraph dev graph of the same name. + + The deployed graphs live in ``EvoScientist.langgraph_dev.graphs`` and + are registered in ``EvoScientist/langgraph_dev/langgraph.json``. + + Adding a new async sub-agent requires no change here — flip + ``async: true`` in its yaml and create the matching deployment graph. + + All return paths strip the internal ``_async`` field from sub-agent dicts + before handoff, since deepagents may schema-validate the kwarg. + """ + cfg = _ensure_config() + if not getattr(cfg, "enable_async_subagents", False): + # Async fully disabled — strip the internal flag before handoff. + for s in subs: + s.pop("_async", None) + return subs + + # Guard: if the langgraph dev subprocess never came up (port conflict, + # binary missing, etc.), routing sub-agents to a dead URL produces hangs + # and confusing tool errors. Fall back to in-process sync delegation. + from .langgraph_dev.manager import is_async_subagents_available + + if not is_async_subagents_available(): + logging.getLogger(__name__).warning( + "enable_async_subagents=true but langgraph dev is not reachable; " + "falling back to in-process sync delegation for all sub-agents." + ) + # Strip the internal ``_async`` flag (carried from ``load_subagents``) + # before sub-agents reach deepagents — it's never a deepagents key. + for s in subs: + s.pop("_async", None) + return subs + + # The ``_async`` flag was set by ``utils.load_subagents._build_one`` from + # each yaml's ``async:`` field. No need to re-parse the yaml files here. + async_specs: dict[str, str] = { + s["name"]: s.get("description", "") for s in subs if s.get("_async") + } + + if not async_specs: + for s in subs: + s.pop("_async", None) + return subs + + from deepagents import AsyncSubAgent + + port = int(getattr(cfg, "langgraph_dev_port", 6174)) + out = [] + # MCP tools routed to async sub-agents (via ``expose_to: `` in + # mcp.yaml) ARE delivered — the deployed factory + # ``subagents/_factory.py:build_async_subagent_graph`` loads its own MCP + # connection per server (cost: one extra MCP server subprocess per + # exposed server, since stdio transports can't share across processes). + for s in subs: + name = s.get("name") + if name in async_specs: + out.append( + AsyncSubAgent( + name=name, + description=async_specs[name], + graph_id=name, + url=f"http://localhost:{port}", + ) + ) + else: + # Strip the internal flag before handoff to deepagents. + s.pop("_async", None) + out.append(s) + return out + + def _build_base_kwargs(base_backend, base_middleware): """Build agent kwargs *without* MCP (fast, no subprocess spawning).""" from .tools import skill_manager, tavily_search, think_tool @@ -237,6 +316,7 @@ def _build_base_kwargs(base_backend, base_middleware): prompt_refs=_build_prompt_refs(), ) _inject_subagent_middleware(subs) + subs = _maybe_swap_async_subagents(subs) return { "name": "EvoScientist", "model": _ensure_chat_model(), @@ -292,6 +372,10 @@ def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress= if sa_tools := mcp_by_agent.get(sa["name"], []): sa.setdefault("tools", []).extend(sa_tools) + # Swap selected sub-agents to AsyncSubAgent (must happen AFTER MCP injection + # since async sub-agents are remote graphs that load their own tools). + subs = _maybe_swap_async_subagents(subs) + return { "name": "EvoScientist", "model": _ensure_chat_model(), @@ -373,17 +457,40 @@ def _get_default_middleware(): def _get_default_agent(): - """Build the default agent (with MCP, no checkpointer) on first access.""" + """Build the default agent (with MCP, no checkpointer) on first access. + + When invoked from the langgraph dev subprocess (env var + ``EVOSCIENTIST_DEPLOYED_NO_MCP=true``, set by + ``langgraph_dev.manager.start_langgraph_dev``), MCP loading is skipped to + avoid duplicating the CLI's MCP server pool — the deployed main agent + is currently only reachable via HTTP for Web UI / SDK clients (none in + use yet), so paying for a second copy of the same MCP servers is pure + waste. Re-enable later by removing the env var when MCP-needing remote + callers are introduced. + """ global _EvoScientist_agent if _EvoScientist_agent is None: from deepagents import create_deep_agent + cfg = _ensure_config() be = _get_default_backend() mw = _get_default_middleware() - kwargs = load_mcp_and_build_kwargs(be, mw) - _EvoScientist_agent = create_deep_agent(**kwargs).with_config( - {"recursion_limit": 1000} - ) + + # HITL on main agent only (mirrors create_cli_agent). Use middleware, + # not interrupt_on= kwarg — the kwarg propagates to every subagent and + # breaks parallel execute calls (multi-pending-interrupt LangGraph + # error). See PR #202. + if not cfg.auto_approve: + mw.append(HumanInTheLoopMiddleware(interrupt_on={"execute": True})) + + if os.environ.get("EVOSCIENTIST_DEPLOYED_NO_MCP", "").lower() == "true": + kwargs = _build_base_kwargs(be, mw) + else: + kwargs = load_mcp_and_build_kwargs(be, mw) + + _EvoScientist_agent = create_deep_agent( + **kwargs, + ).with_config({"recursion_limit": cfg.recursion_limit}) return _EvoScientist_agent @@ -515,4 +622,4 @@ def create_cli_agent( return create_deep_agent( **kwargs, checkpointer=checkpointer, - ).with_config({"recursion_limit": 1000}) + ).with_config({"recursion_limit": cfg.recursion_limit}) diff --git a/EvoScientist/cli/commands.py b/EvoScientist/cli/commands.py index bf883ce..eb9ee62 100644 --- a/EvoScientist/cli/commands.py +++ b/EvoScientist/cli/commands.py @@ -187,6 +187,23 @@ class CompactSummaryRenderable: yield render_compact_summary_panel(self.summary_text) +def _ensure_async_subagent_server(config: Any, *, workspace_dir: str) -> None: + """Conditionally start the langgraph dev subprocess for async sub-agents. + + Shared by both the interactive entry and the serve entry — keeps the + user-visible status message and the conditional in one place. + """ + if not getattr(config, "enable_async_subagents", False): + return + from ..langgraph_dev.manager import ensure_langgraph_dev + + with console.status( + "[dim]Starting async sub-agent server (langgraph dev)...[/dim]", + spinner="dots", + ): + ensure_langgraph_dev(config, workspace_dir=workspace_dir) + + def _resolve_context_window( model: Any, fallback: int = _COMPACT_CONTEXT_WINDOW_FALLBACK ) -> int: @@ -868,6 +885,10 @@ def serve( set_workspace_root(ws) ensure_dirs() + # Auto-start langgraph dev (after workspace resolution, so deployed + # async sub-agents inherit the CLI's workspace via EVOSCIENTIST_WORKSPACE_DIR). + _ensure_async_subagent_server(config, workspace_dir=ws) + console.print("[dim]Loading agent...[/dim]") agent = _load_agent(workspace_dir=ws, config=config) from ..sessions import generate_thread_id @@ -1574,6 +1595,10 @@ def _main_callback( # Ensure memory and skills subdirs exist in workspace ensure_dirs() + # Auto-start langgraph dev (after workspace resolution, so deployed + # async sub-agents inherit the CLI's workspace via EVOSCIENTIST_WORKSPACE_DIR). + _ensure_async_subagent_server(config, workspace_dir=workspace_dir) + if prompt: # Single-shot mode: wrap in persistent checkpointer import asyncio @@ -1584,6 +1609,7 @@ def _main_callback( resolve_thread_id_prefix, ) from .interactive import cmd_run + from .resume_hint import print_resume_hint async def _single_shot(): async with get_checkpointer() as checkpointer: @@ -1613,15 +1639,21 @@ def _main_callback( checkpointer=checkpointer, config=config, ) - cmd_run( - agent, - prompt, - thread_id=tid, - show_thinking=show_thinking, - workspace_dir=workspace_dir, - model=config.model, - ui_backend=config.ui_backend, - ) + try: + cmd_run( + agent, + prompt, + thread_id=tid, + show_thinking=show_thinking, + workspace_dir=workspace_dir, + model=config.model, + ui_backend=config.ui_backend, + ) + finally: + try: + print_resume_hint(tid, console=console) + except Exception: + pass import nest_asyncio # type: ignore[import-untyped] diff --git a/EvoScientist/cli/interactive.py b/EvoScientist/cli/interactive.py index a3f72d4..3ccc9be 100644 --- a/EvoScientist/cli/interactive.py +++ b/EvoScientist/cli/interactive.py @@ -629,6 +629,28 @@ def cmd_interactive( renders conversation history.""" if workspace_dir: state["workspace_dir"] = workspace_dir + # Sync the langgraph dev subprocess to the resumed + # workspace so deployed sub-agents (writing-agent etc.) + # don't operate on the previous workspace's files. The + # manager auto-detects the change and restarts; no-ops + # if async subagents are disabled or workspace unchanged. + # Restart can take 10-15s — show a spinner so the user + # doesn't think the CLI is frozen, and run the sync call + # in a worker thread so the asyncio event loop keeps + # serving channel polls / MCP heartbeats during the wait. + if getattr(config, "enable_async_subagents", False): + from ..langgraph_dev.manager import ensure_langgraph_dev + + with console.status( + "[dim]Syncing async sub-agent server to resumed " + "workspace...[/dim]", + spinner="dots", + ): + await asyncio.to_thread( + ensure_langgraph_dev, + config, + workspace_dir=workspace_dir, + ) state["thread_id"] = thread_id state["resumed"] = True state["status_started_at"] = datetime.now() @@ -676,6 +698,25 @@ def cmd_interactive( state["status_last_input_tokens"] = None if ws: state["workspace_dir"] = ws + # CLI-startup --resume path: sync langgraph dev + # subprocess to the thread's saved workspace if it + # differs from the one we initially launched it with. + # Show a spinner during the 10-15s restart, and run + # the sync call in a worker thread so the asyncio + # event loop stays responsive. + if getattr(config, "enable_async_subagents", False): + from ..langgraph_dev.manager import ensure_langgraph_dev + + with console.status( + "[dim]Syncing async sub-agent server to " + "resumed workspace...[/dim]", + spinner="dots", + ): + await asyncio.to_thread( + ensure_langgraph_dev, + config, + workspace_dir=ws, + ) else: # Resolution failed (ambiguous/not-found); the user's raw # input is still seeded in state["thread_id"] from init. diff --git a/EvoScientist/cli/tui_interactive.py b/EvoScientist/cli/tui_interactive.py index 07b11cb..475a8a6 100644 --- a/EvoScientist/cli/tui_interactive.py +++ b/EvoScientist/cli/tui_interactive.py @@ -612,6 +612,35 @@ def run_textual_interactive( ) -> None: if workspace_dir: self._workspace_dir = workspace_dir + # Mirror the Rich CLI fix: when a /resume restores a thread + # whose workspace differs from the one the langgraph dev + # subprocess was launched with, the deployed sub-agents + # would otherwise keep operating on the previous workspace. + # Sync the subprocess to the new workspace; the manager + # auto-detects the change and restarts (or no-ops if disabled + # or unchanged). Run in a worker thread so the Textual event + # loop keeps refreshing the UI during the up-to-60s wait, and + # show a live timer widget (like /compact) so the user sees + # progress instead of a frozen static line. + from ..config import load_config + + _resume_cfg = load_config() + if getattr(_resume_cfg, "enable_async_subagents", False): + from ..langgraph_dev.manager import ensure_langgraph_dev + from .widgets.workspace_sync_widget import WorkspaceSyncWidget + + sync_widget = WorkspaceSyncWidget() + container = self.query_one("#chat", VerticalScroll) + await container.mount(sync_widget) + container.scroll_end(animate=False) + try: + await asyncio.to_thread( + ensure_langgraph_dev, + _resume_cfg, + workspace_dir=workspace_dir, + ) + finally: + await sync_widget.cleanup() self._conversation_tid = thread_id # Background reload: history renders immediately; next turn awaits. @@ -2759,6 +2788,41 @@ def run_textual_interactive( ws = (meta or {}).get("workspace_dir", "") if ws: effective_workspace = ws + # Sync langgraph dev subprocess to the resumed + # workspace BEFORE the Textual app takes over the + # terminal. Mirrors interactive.py's Rich-CLI fix. + # Without this, --resume against a thread from a + # different workspace would leave deployed sub-agents + # operating on the launch directory's files. + try: + from ..config import load_config + from ..langgraph_dev.manager import ensure_langgraph_dev + from ..stream.console import console as _resume_console + + _ws_cfg = load_config() + if getattr(_ws_cfg, "enable_async_subagents", False): + with _resume_console.status( + "[dim]Syncing async sub-agent server to " + "resumed workspace...[/dim]", + spinner="dots", + ): + await asyncio.to_thread( + ensure_langgraph_dev, + _ws_cfg, + workspace_dir=ws, + ) + except Exception as _ws_sync_exc: + # Non-fatal at startup — async sub-agents fall back + # to sync via the manager's own availability flag. + # Surface the exception so unexpected failures + # (import errors, regressions in + # ensure_langgraph_dev, etc.) don't hide silently. + logging.getLogger(__name__).warning( + "TUI startup workspace sync to langgraph dev " + "failed: %s. Async sub-agents will fall back " + "to in-process sync delegation for this session.", + _ws_sync_exc, + ) effective_thread_id = resolved resumed = True elif matches: diff --git a/EvoScientist/cli/widgets/workspace_sync_widget.py b/EvoScientist/cli/widgets/workspace_sync_widget.py new file mode 100644 index 0000000..99fd87d --- /dev/null +++ b/EvoScientist/cli/widgets/workspace_sync_widget.py @@ -0,0 +1,39 @@ +"""Transient widget shown while a /resume restarts the langgraph dev subprocess. + +Mirrors ``CompactingWidget`` — a timer-backed status line that ticks elapsed +seconds so the user has live feedback during the up-to-60s langgraph dev +workspace sync (subprocess stop + restart so deployed sub-agents see the +resumed thread's workspace). +""" + +from __future__ import annotations + +from .timed_status_widget import TimedStatusWidget + + +class WorkspaceSyncWidget(TimedStatusWidget): + """Timer-backed status line for an in-progress workspace sync.""" + + DEFAULT_CSS = """ + WorkspaceSyncWidget { + height: auto; + color: #94a3b8; + padding: 0 0; + margin: 0 0 1 0; + } + """ + + def __init__(self) -> None: + super().__init__() + + def _refresh_display(self) -> None: + self.update( + f"Syncing async sub-agent server to resumed workspace... " + f"({self.elapsed_seconds}s)" + ) + + async def cleanup(self) -> None: + """Stop timer and remove from DOM.""" + self._stop_timer() + if self.is_mounted: + await self.remove() diff --git a/EvoScientist/config/onboard.py b/EvoScientist/config/onboard.py index 1223c3f..ee79946 100644 --- a/EvoScientist/config/onboard.py +++ b/EvoScientist/config/onboard.py @@ -103,6 +103,7 @@ def _checkbox_ask(choices, message: str, **kwargs): STEPS = [ "UI", + "LangGraph Port", "Provider", "API Key", "Model", @@ -661,6 +662,90 @@ def _step_ui_backend(config: EvoScientistConfig) -> str: return backend +def _step_langgraph_dev_port(config: EvoScientistConfig) -> int: + """Step 0.5: Choose the local TCP port for the langgraph dev subprocess. + + EvoSci auto-starts a ``langgraph dev`` server in the background to host + deployed sub-agents (writing-agent, data-analysis-agent) when + ``enable_async_subagents`` is True. This step lets the user pick a free + port, with a live conflict check on the configured default. + + Returns the chosen port; caller assigns it to ``config.langgraph_dev_port``. + """ + if not getattr(config, "enable_async_subagents", True): + # User has async disabled — port is irrelevant, no prompt. + return getattr(config, "langgraph_dev_port", 6174) + + from ..langgraph_dev.manager import _is_port_occupied, is_langgraph_dev_running + + current_port = getattr(config, "langgraph_dev_port", 6174) + current_occupied = _is_port_occupied(current_port) + if current_occupied and is_langgraph_dev_running(port=current_port): + # Another EvoSci shell is already serving on this port — reuse, don't + # force the user to renumber. + current_occupied = False + + # Bake the live status into the prompt label so the user sees it WITH + # the question, not as a side-effect line that prints before input. + # Single set of parens, no nesting (mirrors ccproxy's prompt style). + if current_occupied: + prompt_label = ( + f"Enter port for EvoScientist server " + f"(Current: {current_port}, occupied, pick another):" + ) + else: + prompt_label = ( + f"Enter port for EvoScientist server " + f"(Current: {current_port}, available, Enter to keep):" + ) + + def valid_port(value: str) -> bool: + if not value: + # Allow keeping the default only if it's actually free; otherwise + # require the user to pick something else. + return not current_occupied + try: + port = int(value) + except (ValueError, TypeError): + return False + if not (1024 < port < 65536): + return False + # Reject user-typed ports that are already occupied UNLESS the + # occupier is our own langgraph dev (e.g., another EvoSci shell) — + # in that case the runtime will reuse it. + if not _is_port_occupied(port): + return True + return is_langgraph_dev_running(port=port) + + raw = questionary.text( + prompt_label, + validate=valid_port, + style=WIZARD_STYLE, + qmark=QMARK, + ).ask() + + if raw is None: + raise KeyboardInterrupt() + + port = int(raw) if raw else current_port + + # Final probe — warn (don't fail) if the chosen port is still occupied + # by something OTHER than our own langgraph dev. Reuse of an existing + # EvoSci server on that port is fine. They can always change later via: + # EvoSci config set langgraph_dev_port + if _is_port_occupied(port) and not is_langgraph_dev_running(port=port): + console.print( + f" [yellow]⚠ Port {port} is occupied. EvoSci may fail to start its " + f"server. Free the port or change later with: " + f"EvoSci config set langgraph_dev_port [/yellow]" + ) + else: + console.print( + f" [green]✓ EvoScientist will run on http://127.0.0.1:{port}[/green]" + ) + return port + + def _step_provider(config: EvoScientistConfig) -> str: """Step 1: Select LLM provider. @@ -2893,6 +2978,9 @@ def run_onboard(skip_validation: bool = False) -> bool: ui_backend = _step_ui_backend(config) config.ui_backend = ui_backend + # Step 0.5: langgraph dev port (with live conflict check) + config.langgraph_dev_port = _step_langgraph_dev_port(config) + # Step 1: Provider provider = _step_provider(config) config.provider = provider diff --git a/EvoScientist/config/settings.py b/EvoScientist/config/settings.py index 5e366d9..c9acb38 100644 --- a/EvoScientist/config/settings.py +++ b/EvoScientist/config/settings.py @@ -87,6 +87,54 @@ class EvoScientistConfig: provider: str = "anthropic" model: str = "claude-sonnet-4-5" + # Async Sub-agent Settings + # When True (default), the EvoSci CLI auto-starts a langgraph dev subprocess + # so any sub-agent flagged ``async: true`` in subagents/.yaml runs + # non-blocking via AsyncSubAgent. Currently affects writing-agent and + # data-analysis-agent. Adds ~10-15s to CLI startup (langgraph dev cold + # start, mostly MCP server spawn time). + # + # Set False to run fully in-process — saves the startup cost in scenarios + # where async isn't useful: short scripted EvoSci runs (CI / one-shot + # ``-p "..."``), low-RAM environments, or workflows that only need the + # synchronous sub-agents (planner / research / code / debug). + enable_async_subagents: bool = True + + # Port for the auto-started langgraph dev subprocess. 6174 is Kaprekar's + # constant — a memorable EvoScientist-themed default that avoids collisions + # with common dev ports (3000/5000/8000/8080) and the langgraph CLI default + # 2024. Override if it conflicts with another local service. + langgraph_dev_port: int = 6174 + + # Whether langgraph dev persists its runtime state to .langgraph_api/ next + # to the subprocess cwd. True (default) keeps async-task, scheduler, and + # Store API state across subprocess restarts — useful for future + # cross-session async, cron, and Store features. Set False to suppress + # writes (workspace stays cleaner; state is in-memory only and lost on + # CLI exit). EvoScientist's main thread persistence uses sessions.db + # regardless of this setting. + langgraph_dev_file_persistence: bool = True + + # Concurrency: how many runs each langgraph dev worker processes in parallel. + # 10 is the langgraph dev recommended default and works well on a typical + # dev machine. Lower it (e.g., 4) on memory-constrained or low-core + # machines if multiple async sub-agents in flight cause noticeable + # slowdown. + langgraph_dev_jobs_per_worker: int = 10 + + # Max LangGraph super-steps (LLM call / tool call / sub-agent delegation + # each count as 1) before raising GraphRecursionError. Resets on every + # ``agent.invoke()`` — i.e., this is per-turn, NOT per-conversation. For + # long conversations the relevant mechanisms are checkpointer persistence + # (sessions.db), ContextEditingMiddleware (window management), and + # EvoMemoryMiddleware (cross-turn memory). + # + # 1,000,000 is "effectively unlimited" — typical research turns use + # 200-1000 steps; reaching 1M would cost ~$10K in tokens, by which point + # rate limits, context overflow, or API quota errors would trip first. + # Lower (e.g., 5000) if you want a tighter safety net against runaway loops. + recursion_limit: int = 1_000_000 + # Workspace Settings default_mode: Literal["daemon", "run"] = "daemon" default_workdir: str = "" @@ -393,6 +441,11 @@ _ENV_MAPPINGS = { "ccproxy_port": "EVOSCIENTIST_CCPROXY_PORT", "use_responses_api": "EVOSCIENTIST_USE_RESPONSES_API", "checkpoint_keep_per_thread": "EVOSCIENTIST_CHECKPOINT_KEEP_PER_THREAD", + "enable_async_subagents": "EVOSCIENTIST_ENABLE_ASYNC_SUBAGENTS", + "langgraph_dev_port": "EVOSCIENTIST_LANGGRAPH_DEV_PORT", + "langgraph_dev_file_persistence": "EVOSCIENTIST_LANGGRAPH_DEV_FILE_PERSISTENCE", + "langgraph_dev_jobs_per_worker": "EVOSCIENTIST_LANGGRAPH_DEV_JOBS_PER_WORKER", + "recursion_limit": "EVOSCIENTIST_RECURSION_LIMIT", } diff --git a/EvoScientist/langgraph_dev/__init__.py b/EvoScientist/langgraph_dev/__init__.py new file mode 100644 index 0000000..462badc --- /dev/null +++ b/EvoScientist/langgraph_dev/__init__.py @@ -0,0 +1,17 @@ +"""``langgraph dev`` deployment surface. + +Holds everything needed to run the EvoScientist agent ecosystem on a local +``langgraph dev`` subprocess: + +- ``manager`` — subprocess lifecycle (auto-start, port management, cleanup). +- ``langgraph.json`` — graph manifest consumed by ``langgraph dev --config``. +- ``main_graph`` — re-export of the lazy-loaded ``EvoScientist_agent``. +- ``graphs`` — module-level bindings for every yaml-flagged async sub-agent + (``async: true`` in ``EvoScientist/subagents/.yaml``). + +The graphs themselves are built by ``EvoScientist.subagents._factory. +build_async_subagent_graph`` from the canonical yaml definitions, so this +package only owns the *deployment* concern. Adding a new async sub-agent +takes three steps: flip the yaml flag, add a one-line binding in +``graphs.py``, and register it in ``langgraph.json``. +""" diff --git a/EvoScientist/langgraph_dev/graphs.py b/EvoScientist/langgraph_dev/graphs.py new file mode 100644 index 0000000..0ecafa3 --- /dev/null +++ b/EvoScientist/langgraph_dev/graphs.py @@ -0,0 +1,28 @@ +"""Deployed graphs for all yaml-flagged async sub-agents. + +One module-level binding per ``async: true`` entry in +``EvoScientist/subagents/.yaml``. Each binding is a graph compiled by +``build_async_subagent_graph`` (which reads the yaml, wires tools/skills/ +backend/middleware identical to the in-process sync version, and returns a +runnable langgraph). + +To add a new async sub-agent: + + 1. Set ``async: true`` in ``EvoScientist/subagents/.yaml``. + 2. Add a one-line binding here:: + + _agent = build_async_subagent_graph("") + + 3. Register it in ``EvoScientist/langgraph_dev/langgraph.json``:: + + "": "EvoScientist.langgraph_dev.graphs:_agent" + +The deployed main agent (``EvoScientist_agent``) lives in ``main_graph.py`` +because it follows a different mechanism (re-exporting a lazily-constructed +attribute), not the yaml-driven factory. +""" + +from EvoScientist.subagents._factory import build_async_subagent_graph + +writing_agent = build_async_subagent_graph("writing-agent") +data_analysis_agent = build_async_subagent_graph("data-analysis-agent") diff --git a/EvoScientist/langgraph_dev/langgraph.json b/EvoScientist/langgraph_dev/langgraph.json new file mode 100644 index 0000000..00dd45d --- /dev/null +++ b/EvoScientist/langgraph_dev/langgraph.json @@ -0,0 +1,11 @@ +{ + "dependencies": ["."], + "graphs": { + "EvoScientist": "EvoScientist.langgraph_dev.main_graph:EvoScientist_agent", + "writing-agent": "EvoScientist.langgraph_dev.graphs:writing_agent", + "data-analysis-agent": "EvoScientist.langgraph_dev.graphs:data_analysis_agent" + }, + "config": { + "recursion_limit": 1000000 + } +} diff --git a/EvoScientist/langgraph_dev/main_graph.py b/EvoScientist/langgraph_dev/main_graph.py new file mode 100644 index 0000000..fc1bc41 --- /dev/null +++ b/EvoScientist/langgraph_dev/main_graph.py @@ -0,0 +1,12 @@ +"""Deployed graph entry for the main EvoScientist agent. + +The main ``EvoScientist_agent`` is exposed via ``__getattr__`` lazy loading +in ``EvoScientist/EvoScientist.py`` so it doesn't construct on plain +``import EvoScientist``. ``langgraph dev`` 's symbol resolver inspects +module attributes directly and doesn't trigger ``__getattr__``, so we +re-export here to make it visible. +""" + +from EvoScientist.EvoScientist import EvoScientist_agent + +__all__ = ["EvoScientist_agent"] diff --git a/EvoScientist/langgraph_dev/manager.py b/EvoScientist/langgraph_dev/manager.py new file mode 100644 index 0000000..24bc7e1 --- /dev/null +++ b/EvoScientist/langgraph_dev/manager.py @@ -0,0 +1,742 @@ +"""langgraph dev lifecycle management for async sub-agent support. + +Provides functions to start/stop/health-check a ``langgraph dev`` subprocess +that hosts the EvoScientist main agent and async sub-agents (e.g. +``writing-agent``). The CLI calls ``ensure_langgraph_dev(config, ...)`` at +startup so users can run ``EvoSci -p "..."`` without manually managing the +langgraph dev server. + +Mirrors the lifecycle pattern used by ``ccproxy_manager.py``. +""" + +from __future__ import annotations + +import atexit +import logging +import os +import shutil +import subprocess +import threading +import time +from pathlib import Path + +import httpx +import psutil +from filelock import FileLock +from filelock import Timeout as FileLockTimeout + +from EvoScientist.config import EvoScientistConfig + +logger = logging.getLogger(__name__) + + +# Reentrant lock guarding ``_PROCESS`` / ``_PROCESS_WORKSPACE`` / +# ``_ASYNC_SUBAGENTS_AVAILABLE`` mutations and the ``ensure_langgraph_dev`` +# decision/start/stop flow. Reentrant because ``ensure_langgraph_dev`` can call +# ``stop_langgraph_dev`` from inside its own critical section during a +# workspace-driven restart, and both mutate the same module-level state. +_LOCK = threading.RLock() + + +# Default port (Kaprekar's constant — see config/settings.py for the rationale). +# Overridable per-call via ``start_langgraph_dev(port=...)`` / +# ``ensure_langgraph_dev`` (which reads ``config.langgraph_dev_port``) and the +# corresponding url= field on AsyncSubAgent specs. +_DEFAULT_PORT = 6174 + + +def _base_url(port: int = _DEFAULT_PORT) -> str: + return f"http://localhost:{port}" + + +_PID_DIR = Path.home() / ".config" / "evoscientist" +_PID_FILE = _PID_DIR / "langgraph_dev.pid" +_LOG_FILE = _PID_DIR / "langgraph_dev.log" + +# Cross-process file lock for ``ensure_langgraph_dev``. Without this, two +# concurrent CLI shells racing on the cold-start window can SIGKILL each +# other's still-booting subprocesses (Shell B sees Shell A's port-bound but +# not-yet-/ok subprocess as a "stale process to clean up"). With the lock, +# Shell B blocks until Shell A's health-check finishes, then sees the +# healthy server and reuses it. ``threading.RLock`` is process-local and +# can't coordinate across CLI invocations. +_FILE_LOCK_PATH = _PID_DIR / "langgraph_dev.lock" +_FILE_LOCK_TIMEOUT = 120.0 # 60s cold-start health-check + buffer + +# Module-level handle to the langgraph dev subprocess we started, if any. +# Stays None when we reused an existing process (managed by the user). +_PROCESS: subprocess.Popen | None = None + +# Workspace directory the running subprocess was launched with. Used by +# ``ensure_langgraph_dev`` to detect a workspace switch (e.g., on /resume of +# a thread from a different workspace) and trigger a restart so the deployed +# sub-agents' cwd / EVOSCIENTIST_WORKSPACE_DIR env match the new workspace. +_PROCESS_WORKSPACE: Path | None = None + +# Whether async sub-agents are usable in this CLI process. Only True after +# ``ensure_langgraph_dev`` confirms the subprocess is healthy (or already +# running). Stays False on startup failure so ``_maybe_swap_async_subagents`` +# can fall back to in-process sync delegation instead of routing tool calls +# at a dead URL. +_ASYNC_SUBAGENTS_AVAILABLE: bool = False + + +def is_async_subagents_available() -> bool: + """Return True if the langgraph dev subprocess is up and reachable. + + Used by ``_maybe_swap_async_subagents`` to decide whether to swap dict + sub-agents to ``AsyncSubAgent`` references. False means a graceful + fallback to synchronous in-process delegation. + """ + return _ASYNC_SUBAGENTS_AVAILABLE + + +# ============================================================================= +# Availability & health +# ============================================================================= + + +def _langgraph_exe() -> str | None: + """Return the path to the langgraph CLI binary, or None if not found.""" + found = shutil.which("langgraph") + if found: + return found + import sys as _sys + + candidate = os.path.join(os.path.dirname(_sys.executable), "langgraph") + if os.path.isfile(candidate) and os.access(candidate, os.X_OK): + return candidate + return None + + +def is_langgraph_dev_available() -> bool: + """Check whether the ``langgraph`` CLI binary is available.""" + return _langgraph_exe() is not None + + +def is_langgraph_dev_running( + base_url: str | None = None, + *, + port: int = _DEFAULT_PORT, +) -> bool: + """Check whether a langgraph dev API is already serving at ``base_url``. + + ``base_url`` overrides ``port`` when given. + """ + url = base_url or _base_url(port) + try: + return httpx.get(f"{url}/ok", timeout=1.0).status_code == 200 + except (httpx.TransportError, OSError): + return False + + +def _is_port_occupied(port: int) -> bool: + """Return True if anything is listening on ``port`` (TCP, IPv4).""" + import socket as _socket + + s = _socket.socket(_socket.AF_INET, _socket.SOCK_STREAM) + try: + s.settimeout(0.5) + # connect_ex returns 0 on success (something accepted), nonzero otherwise + return s.connect_ex(("127.0.0.1", port)) == 0 + finally: + s.close() + + +def _wait_for_port_release(port: int, timeout: float = 10.0) -> bool: + """Poll until ``port`` is released or ``timeout`` elapses. + + Used after ``stop_langgraph_dev`` / ``_kill_owned_stale_process`` to + bridge the kernel's TIME_WAIT delay before we try to bind again. Returns + True if the port is free, False on timeout. + """ + deadline = time.monotonic() + timeout + while _is_port_occupied(port) and time.monotonic() < deadline: + time.sleep(0.5) + return not _is_port_occupied(port) + + +def _can_bind_port(port: int) -> bool: + """Return True if a fresh ``bind()`` to ``port`` succeeds right now. + + More reliable than ``_is_port_occupied`` when the previous listener has + just exited: ``connect_ex`` can already report "free" while ``bind()`` + still fails because the kernel hasn't fully released the socket + (TIME_WAIT for accepted connections, SO_REUSEADDR rules, etc.). This + actually attempts the bind that langgraph dev would attempt, then + closes immediately. + """ + import socket as _socket + + s = _socket.socket(_socket.AF_INET, _socket.SOCK_STREAM) + try: + s.bind(("127.0.0.1", port)) + return True + except OSError: + return False + finally: + try: + s.close() + except Exception: + pass + + +def _wait_for_port_bindable(port: int, timeout: float = 60.0) -> bool: + """Poll until a real ``bind()`` to ``port`` can succeed, or timeout. + + Use this immediately before ``subprocess.Popen("langgraph dev")`` — + matches the strictness of the bind langgraph dev itself will perform, + so we don't pass the lighter ``_is_port_occupied`` gate only to fail + on the actual bind a few seconds later. + + Default 60s timeout matches macOS's TCP TIME_WAIT duration — a port + held by an exited listener is genuinely unbindable for up to that long + on a tight CLI exit + restart cycle. Shorter timeouts give up too early. + """ + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if _can_bind_port(port): + return True + time.sleep(0.5) + return False + + +def _list_pids_on_port(port: int) -> list[int]: + """Return list of PIDs bound to ``port``, or empty list on lookup failure. + + Read-only; never sends signals. Use this to *inspect* port state before + deciding what (if anything) to clean up. + + Cross-platform via ``psutil.net_connections`` — works on POSIX and Windows + without depending on ``lsof`` / ``netstat`` shell tools. + """ + try: + return list( + { + conn.pid + for conn in psutil.net_connections(kind="inet") + if conn.laddr and conn.laddr.port == port and conn.pid is not None + } + ) + except (psutil.AccessDenied, psutil.Error): + return [] + + +def _kill_owned_stale_process(port: int) -> bool: + """Kill ONLY a previously-owned langgraph dev process bound to ``port``. + + "Owned" means the PID written to ``_PID_FILE`` by an earlier + ``start_langgraph_dev`` invocation in this user account, AND the live + process at that PID still has ``langgraph`` in its command line (defense + against PID recycling). Returns True if a stale-but-owned process was + cleaned up; returns False (without sending any signals) if the port is + occupied by an unowned process or the PID has been recycled — caller + should treat that as a hard conflict and refuse to start. + + Why this matters: + 1. ``net_connections`` may report any process bound to the port, + including user-run dev servers that legitimately took 6174. + SIGKILL'ing those is a data-loss event. + 2. Even with PID-file ownership, the OS may have recycled the PID + to an unrelated process between sessions (e.g., after a SIGKILL'd + CLI left the PID file behind). The cmdline check rules that out. + """ + if not _PID_FILE.exists(): + return False + try: + owned_pid = int(_PID_FILE.read_text().strip()) + except (OSError, ValueError): + return False + + occupiers = _list_pids_on_port(port) + if owned_pid not in occupiers: + return False # Port is held by a different process now. + + # Defense-in-depth: PID could have been recycled to an unrelated process. + # Verify the live process at that PID still looks like langgraph dev + # before sending any signals. + try: + proc = psutil.Process(owned_pid) + cmdline = proc.cmdline() + except psutil.NoSuchProcess: + # PID file points at a dead process — clean up the file but don't + # try to kill anything. + try: + _PID_FILE.unlink() + except OSError: + pass + return False + except psutil.AccessDenied: + return False + + # Loose substring match by design: PID-file ownership is the primary + # guard; this check only hardens against PID recycling between sessions. + # A foreign process happening to have "langgraph" in its argv (e.g., a + # text editor with langgraph_dev.py open) would slip through, but the + # ownership check above already excluded externally-owned PIDs, so the + # window is the narrow case where our exact PID was reused. Keeping the + # match loose avoids version skew with langgraph CLI invocation styles. + if not any("langgraph" in arg for arg in cmdline): + # PID was recycled by an unrelated process. Refuse to kill it, but + # still clean up the PID file — our original langgraph dev with that + # PID is definitely gone (PIDs are only recycled after the original + # process exits), so the file's claim is stale. Mirrors the cleanup + # in the NoSuchProcess branch above. + logger.warning( + "PID file %s claims pid %d for langgraph dev, but that pid now " + "points at a different process (cmdline=%s). Refusing to kill, " + "removing stale PID file.", + _PID_FILE, + owned_pid, + cmdline, + ) + try: + _PID_FILE.unlink() + except OSError: + pass + return False + + try: + proc.kill() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + try: + _PID_FILE.unlink() + except OSError: + pass + return True + + +def _packaged_langgraph_config() -> Path: + """Return path to the package-shipped ``langgraph.json``. + + Lives at ``EvoScientist/langgraph_dev/langgraph.json`` and is included + in the wheel via ``pyproject.toml`` ``package-data`` so it's available + regardless of how EvoScientist was installed (pip / editable / source). + """ + import EvoScientist.langgraph_dev as _pkg + + return Path(_pkg.__file__).resolve().parent / "langgraph.json" + + +# ============================================================================= +# Process management +# ============================================================================= + + +def start_langgraph_dev( + workspace_dir: Path | None = None, + *, + port: int = _DEFAULT_PORT, + file_persistence: bool = True, + jobs_per_worker: int = 10, +) -> subprocess.Popen: + """Start langgraph dev as a background subprocess. + + Args: + workspace_dir: Working directory for the subprocess (subprocess ``cwd``). + Determines where deployed agents' filesystem operations land + (``CustomSandboxBackend`` derives its workspace root from cwd via + ``paths.WORKSPACE_ROOT``). Defaults to ``Path.cwd()``. + port: TCP port to bind. Defaults to 6174 (Kaprekar's constant). + file_persistence: When True (default), langgraph dev writes its full + ``.langgraph_api/`` cache so async-task / Store / scheduler state + survives subprocess restarts. Set False to suppress periodic + flushes (workspace stays cleaner; state is in-memory only). + + Returns: + The Popen handle for the langgraph dev process. + + Raises: + FileNotFoundError: If the langgraph CLI or packaged ``langgraph.json`` + is missing. + RuntimeError: If langgraph dev exits early or never becomes healthy. + """ + global _PROCESS + + exe = _langgraph_exe() + if exe is None: + raise FileNotFoundError( + "langgraph CLI not found. Reinstall EvoScientist (langgraph-cli is " + "a hard dependency): pip install -e '.[dev]'" + ) + + config_file = _packaged_langgraph_config() + if not config_file.exists(): + raise FileNotFoundError( + f"Packaged langgraph.json not found at {config_file}. " + "This indicates a broken EvoScientist installation — reinstall." + ) + + workspace_dir = workspace_dir or Path.cwd() + + # Defensive: handle a port that's occupied but not serving /ok. + # Three cases: + # (a) Our own previous langgraph dev (PID matches _PID_FILE) — kill it. + # (b) Our own previous langgraph dev exited but the kernel still holds + # the socket in TIME_WAIT — no live PID for lsof to match, and the + # PID file may already be gone (stop_langgraph_dev unlinks it). The + # bind poll below correctly waits this out. + # (c) Foreign process legitimately holds the port — we must NOT kill it. + # The bind poll will keep failing and raise an actionable error. + # We don't try to disambiguate (b) vs (c) here: ``_kill_owned_stale_process`` + # only verifies PID-file ownership, so absence of a match conflates "stale + # TIME_WAIT" with "foreign process". Falling through to the bind poll + # disambiguates by behavior — TIME_WAIT clears, foreign listeners don't. + if not is_langgraph_dev_running(port=port) and _is_port_occupied(port): + if _kill_owned_stale_process(port): + logger.warning( + "Cleaned up stale langgraph dev (pid from %s) on port %d", + _PID_FILE, + port, + ) + # After SIGKILL the kernel may keep the port in TIME_WAIT for + # several seconds before fully releasing it. Poll until the port + # is genuinely free so the upcoming bind() doesn't race a + # half-released socket and crash with "Port already in use". + _wait_for_port_release(port) + else: + # No owned stale PID — could be foreign or kernel-only TIME_WAIT + # from a previous subprocess. Defer to the bind poll below. + logger.info( + "Port %d occupied with no owned stale PID — waiting for " + "kernel TIME_WAIT release (or bind-poll timeout if a " + "foreign process holds it).", + port, + ) + + # Final defense: poll until a real ``bind()`` to ``port`` succeeds before + # spawning langgraph dev. ``_is_port_occupied`` (connect-based) can report + # the port as "free" while langgraph dev's stricter bind still fails — + # that mismatch is what makes back-to-back CLI exit + restart show + # "Port already in use" even though our pre-checks passed. By probing + # the same operation langgraph dev will do, we either wait it out or + # fail clearly with an actionable message. 60s covers macOS TIME_WAIT. + if not _wait_for_port_bindable(port): + raise RuntimeError( + f"Port {port} cannot be bound after waiting 60s (kernel TIME_WAIT " + f"or another process holds it). Free the port with `lsof -ti:{port}`, " + f"or change ports with: `EvoSci config set langgraph_dev_port `" + ) + + _PID_DIR.mkdir(parents=True, exist_ok=True) + # Open the log file once and hand it to subprocess.Popen as stdout/stderr. + # Popen duplicates the fd into the child via fork+exec, so closing our + # parent-side handle in the finally below releases this process's fd + # without affecting the child. Without the close, every restart leaks + # one fd — a problem on heavy ``/resume`` cycling that could eventually + # exhaust the process's open-file limit. + log_handle = open(_LOG_FILE, "ab") # closed in finally below + + # Propagate workspace to the subprocess so deployed sub-agents resolve + # paths.WORKSPACE_ROOT to the same dir as the CLI's main agent. cwd alone + # is fragile (relative paths in MCP configs etc.); env var is explicit. + # + # Note: ``EVOSCIENTIST_WORKSPACE_DIR`` serves a dual role in this codebase. + # config/settings.py:_ENV_MAPPINGS reads it as a user-facing override of + # ``default_workdir`` (parent process). Here we WRITE it on the subprocess + # env to propagate the resolved workspace into langgraph dev. Both + # purposes mean "this is the user's workspace", so they don't conflict; + # the explicit write below always wins for the subprocess regardless of + # what the parent had inherited from its own environment. + sub_env = os.environ.copy() + sub_env["EVOSCIENTIST_WORKSPACE_DIR"] = str(workspace_dir) + + # By default, let langgraph dev write its full ``.langgraph_api/`` cache + # so future use cases — cross-session async tasks, Store API persistence, + # cron job state across CLI restarts — work without further changes. Users + # who want a clean workspace can opt out via: + # EvoSci config set langgraph_dev_file_persistence false + if not file_persistence: + sub_env["LANGGRAPH_DISABLE_FILE_PERSISTENCE"] = "true" + + # Skip MCP loading inside the langgraph dev subprocess. The CLI's main + # agent already loaded MCP servers in the foreground process; without + # this guard, ``main_graph.py`` would import ``EvoScientist_agent`` and + # trigger ``_get_default_agent`` → ``load_mcp_and_build_kwargs`` → + # spawning a SECOND copy of every MCP server in the subprocess. + # The deployed main agent is currently only reachable via HTTP (for + # future Web UI / SDK clients), and none of those are in use, so the + # duplicate MCP pool is pure waste. Async sub-agents don't load MCP at + # all (their factory bypasses ``load_mcp_and_build_kwargs``), so they + # are unaffected. + sub_env["EVOSCIENTIST_DEPLOYED_NO_MCP"] = "true" + + try: + proc = subprocess.Popen( + [ + exe, + "dev", + "--config", + str(config_file), + "--port", + str(port), + "--n-jobs-per-worker", + str(jobs_per_worker), + "--no-browser", + ], + cwd=str(workspace_dir), + stdout=log_handle, + stderr=log_handle, + env=sub_env, + start_new_session=True, + ) + finally: + # The child has its own copy of the fd; closing ours prevents an + # accumulating leak across restarts. Run even if Popen raises. + try: + log_handle.close() + except Exception: + pass + _PID_FILE.write_text(str(proc.pid)) + global _PROCESS_WORKSPACE + _PROCESS = proc + _PROCESS_WORKSPACE = workspace_dir + + # langgraph dev cold-starts in ~10-15s normally; first-time npx-based MCP + # servers can push this to 30-60s while npm fetches packages, so the budget + # is generous. Subsequent runs are much faster thanks to npm cache. + deadline = time.monotonic() + 60 + while time.monotonic() < deadline: + if proc.poll() is not None: + tail = "" + try: + tail = _LOG_FILE.read_text()[-2000:] + except Exception: + pass + # Subprocess died on its own — clear our module-level bookkeeping + # (``_PROCESS``, ``_PROCESS_WORKSPACE``, ``_PID_FILE``) before + # raising. Without this, ``_PROCESS`` would keep pointing at the + # dead handle and ``_PID_FILE`` at a non-existent PID, leading + # the next ``ensure_langgraph_dev`` to misjudge state. Pass + # ``proc`` directly so ``stop_langgraph_dev`` works against the + # one we just spawned even if the global state was overwritten. + stop_langgraph_dev(proc) + raise RuntimeError( + f"langgraph dev exited immediately with code {proc.returncode}.\n" + f"Log tail:\n{tail}" + ) + if is_langgraph_dev_running(port=port): + logger.info( + "langgraph dev started on %s (pid=%d)", _base_url(port), proc.pid + ) + return proc + time.sleep(0.5) + + stop_langgraph_dev(proc) + raise RuntimeError( + f"langgraph dev did not become healthy within 60 seconds. Check {_LOG_FILE}" + ) + + +def stop_langgraph_dev(proc: subprocess.Popen | None = None) -> None: + """Gracefully stop a langgraph dev process. + + Sends SIGTERM to the process group (langgraph dev spawns worker children), + falling back to SIGKILL after 5 seconds. Safe to call with ``None``. + + Acquires ``_LOCK`` (reentrant) before mutating ``_PROCESS`` / + ``_PROCESS_WORKSPACE`` so concurrent ``ensure_langgraph_dev`` callers + (which also hold ``_LOCK``) don't observe partially-cleared state. + """ + global _PROCESS, _PROCESS_WORKSPACE + with _LOCK: + proc = proc if proc is not None else _PROCESS + if proc is None: + return + + if proc.poll() is None: + # Cross-platform process-tree shutdown: walk children explicitly + # because POSIX process groups (``os.killpg``) don't exist on + # Windows. ``psutil.Process.children(recursive=True)`` works on + # both — we mirror the previous SIGTERM-then-SIGKILL escalation. + try: + parent = psutil.Process(proc.pid) + descendants = parent.children(recursive=True) + for child in descendants: + try: + child.terminate() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + parent.terminate() + proc.wait(timeout=5) + except psutil.NoSuchProcess: + pass + except subprocess.TimeoutExpired: + try: + parent = psutil.Process(proc.pid) + for child in parent.children(recursive=True): + try: + child.kill() + except (psutil.NoSuchProcess, psutil.AccessDenied): + pass + parent.kill() + except psutil.NoSuchProcess: + pass + # Reap the Popen handle so we don't leave a zombie until + # the CLI itself exits. Short timeout because parent.kill() + # above already issued SIGKILL to the process tree. + try: + proc.wait(timeout=2) + except subprocess.TimeoutExpired: + pass + + if proc is _PROCESS: + _PROCESS = None + _PROCESS_WORKSPACE = None + if _PID_FILE.exists(): + try: + _PID_FILE.unlink() + except OSError: + pass + + # Note: ``.langgraph_api/`` is intentionally NOT removed — it holds + # langgraph dev's persisted async-task / scheduler / Store state that + # may be useful across CLI restarts. Users who want a clean workspace + # can ``rm -rf .langgraph_api/`` manually or set + # ``langgraph_dev_file_persistence: false`` in config to suppress writes. + + +# ============================================================================= +# High-level orchestration +# ============================================================================= + + +def ensure_langgraph_dev( + config: EvoScientistConfig, + workspace_dir: Path | str | None = None, +) -> subprocess.Popen | None: + """Conditionally start langgraph dev based on ``config.enable_async_subagents``. + + Behavior: + - flag false: no-op, returns None + - flag true + already running on the configured port: reuse, returns None + (we don't own it; warns if the workspace can't be verified) + - flag true + not running: start subprocess, register atexit cleanup, return Popen + + Args: + config: Active EvoScientistConfig. + workspace_dir: Workspace to inherit on the subprocess. Set to the CLI's + resolved workspace so deployed async sub-agents see the same files + as the main in-process agent. If None, the subprocess uses its + own ``Path.cwd()`` (the CLI's launch directory). + + Errors during startup are logged but don't abort the CLI — the user can + still chat with sync sub-agents; only async sub-agent calls will fail. + """ + global _ASYNC_SUBAGENTS_AVAILABLE + if not getattr(config, "enable_async_subagents", False): + return None + + # Two layers of locking: + # 1. ``FileLock`` — cross-process coordination. Without it, two CLI + # shells (TUI + ``-p`` + ``serve``) racing on the cold-start window + # can SIGKILL each other's still-booting subprocesses via + # ``_kill_owned_stale_process`` (Shell A's PID is in the file and + # bound to the port, but ``/ok`` isn't responding yet, so Shell B + # thinks it's stale). + # 2. ``_LOCK`` (in-process RLock) — serializes intra-process callers + # (rapid ``/resume`` in succession, channel threads). Reentrant so + # the workspace-restart path can call ``stop_langgraph_dev`` from + # inside the critical section. + _PID_DIR.mkdir(parents=True, exist_ok=True) + try: + with FileLock(str(_FILE_LOCK_PATH), timeout=_FILE_LOCK_TIMEOUT): + with _LOCK: + return _ensure_langgraph_dev_locked(config, workspace_dir) + except FileLockTimeout: + logger.warning( + "Timed out waiting %.0fs for cross-process langgraph dev lock at %s. " + "Another CLI shell may be stuck during cold-start. Falling back to " + "sync sub-agent delegation for this session.", + _FILE_LOCK_TIMEOUT, + _FILE_LOCK_PATH, + ) + _ASYNC_SUBAGENTS_AVAILABLE = False + return None + + +def _ensure_langgraph_dev_locked( + config: EvoScientistConfig, + workspace_dir: Path | str | None, +) -> subprocess.Popen | None: + """Locked critical section of ``ensure_langgraph_dev`` — must hold ``_LOCK``.""" + global _ASYNC_SUBAGENTS_AVAILABLE + port = int(getattr(config, "langgraph_dev_port", _DEFAULT_PORT)) + file_persistence = bool(getattr(config, "langgraph_dev_file_persistence", True)) + jobs_per_worker = int(getattr(config, "langgraph_dev_jobs_per_worker", 10)) + + ws_path = Path(workspace_dir) if workspace_dir is not None else None + + # If a subprocess we own is running with a *different* workspace than what + # was just requested (typical trigger: user just /resumed a thread from a + # different workspace), the deployed sub-agents' cwd / EVOSCIENTIST_WORKSPACE_DIR + # are stale. Stop it so the start-fresh path below relaunches with the right + # workspace. We only act when WE own the process — never kill an externally- + # managed langgraph dev. + if ( + ws_path is not None + and _PROCESS is not None + and _PROCESS.poll() is None + and _PROCESS_WORKSPACE is not None + and _PROCESS_WORKSPACE.resolve() != ws_path.resolve() + ): + logger.info( + "Workspace changed (%s -> %s); restarting langgraph dev so deployed " + "sub-agents pick up the new workspace.", + _PROCESS_WORKSPACE, + ws_path, + ) + stop_langgraph_dev() + # Crucial: stop_langgraph_dev unlinks the PID file. If we then fell + # through with the port still in TIME_WAIT, the next defensive + # ``_kill_owned_stale_process`` call inside start_langgraph_dev would + # see no PID file, treat the lingering socket as a foreign process, + # and abort with a hard "non-langgraph process" error — turning a + # clean owned restart into a permanent async-disable. Wait inline for + # the kernel to release the port before continuing. + _wait_for_port_release(port) + _ASYNC_SUBAGENTS_AVAILABLE = False # cleared until restart succeeds + + if is_langgraph_dev_running(port=port): + # If WE own the running process, workspace was already verified above + # via _PROCESS_WORKSPACE comparison. If we DON'T own it (some other + # langgraph dev started by the user / another CLI), we have no way to + # confirm its workspace matches what was just requested — async + # sub-agents could end up operating on a different project's files. + # Warn loudly so the user notices. + if _PROCESS is None and ws_path is not None: + logger.warning( + "Reusing externally-managed langgraph dev on %s — cannot verify " + "its workspace matches the requested %s. Async sub-agents may " + "operate on a different workspace's files.", + _base_url(port), + ws_path, + ) + else: + logger.info("langgraph dev already running on %s, reusing", _base_url(port)) + _ASYNC_SUBAGENTS_AVAILABLE = True + return None + + try: + proc = start_langgraph_dev( + workspace_dir=ws_path, + port=port, + file_persistence=file_persistence, + jobs_per_worker=jobs_per_worker, + ) + except (FileNotFoundError, RuntimeError) as exc: + # Startup failed — keep async subagents disabled so the main agent + # falls back to in-process sync delegation rather than routing tool + # calls at a dead URL. + _ASYNC_SUBAGENTS_AVAILABLE = False + logger.warning( + "Failed to start langgraph dev — async sub-agents disabled, " + "falling back to in-process sync delegation. %s", + exc, + ) + return None + + _ASYNC_SUBAGENTS_AVAILABLE = True + atexit.register(stop_langgraph_dev, proc) + return proc diff --git a/EvoScientist/mcp/README.md b/EvoScientist/mcp/README.md index d49f499..a4d0c98 100644 --- a/EvoScientist/mcp/README.md +++ b/EvoScientist/mcp/README.md @@ -184,7 +184,7 @@ github: | `writing-agent` | Report writing | > [!NOTE] -> Tools routed to sub-agents are injected automatically — no need to edit `subagent.yaml`. +> Tools routed to sub-agents are injected automatically — no need to edit any `EvoScientist/subagents/*.yaml` file. ## 🔍 Tool Filtering with Wildcards diff --git a/EvoScientist/subagent.yaml b/EvoScientist/subagent.yaml deleted file mode 100644 index ba92e75..0000000 --- a/EvoScientist/subagent.yaml +++ /dev/null @@ -1,160 +0,0 @@ -planner-agent: - description: "Plan experiments: stages, success signals, and dependencies (no web search, no implementation)." - tools: [think_tool] - skills: ["/skills/"] - system_prompt: | - You are the planner-agent. You do NOT implement code. You create and update experimental plans - that are practical to run locally. - - Before planning, check `/memories/ideation-memory.md` and `/memories/experiment-memory.md` - for prior knowledge from past research cycles. Incorporate relevant entries into - your plan (e.g., proven strategies, known failed directions). Skip if these files - do not exist yet. - - You may be invoked in two modes: - 1) PLAN MODE: produce an initial experimental plan. - 2) REFLECTION MODE: update the plan based on stage results. - - The caller should start the task with either: - - MODE: PLAN - - MODE: REFLECTION - If MODE is not specified, assume PLAN. - - PLAN MODE output (Markdown): - 1) Assumptions & scope - 2) Stages (numbered). For each stage include: - - goal - - success signals (metrics/thresholds or qualitative checks) - - what to run (scripts/commands at a high level) - - expected artifacts (tables/plots/logs) - 3) Dependencies (data, compute, environment) - 4) Iteration triggers (when to change dataset/model/objective) - 5) Evaluation protocol (splits, primary metrics, baselines) and data quality checks - 6) Environment preflight (GPU/CUDA/VRAM/disk) and required dependencies (pip packages) - - REFLECTION MODE output (JSON only, no extra text): - { - "completed": ["..."], - "unmet_success_signals": ["..."], - "skill_suggestions": ["..."], - "stage_modifications": [ - {"stage": "Stage name or index", "change": "What to adjust and why"} - ], - "new_stages": [ - { - "title": "...", - "goal": "...", - "success_signals": ["..."], - "what_to_run": ["..."], - "expected_artifacts": ["..."] - } - ], - "todo_updates": ["..."] - } - - Empty arrays are valid. If no changes are needed, return the JSON with empty arrays. - "skill_suggestions" should use skill names from your available skills listing. - - Keep the structure flexible (not rigid templates). If model size is unspecified, default to - <=7B-class models and lightweight baselines. - -research-agent: - description: "Web research for methods/baselines/datasets (one topic at a time, return actionable notes + sources)." - tools: [tavily_search, think_tool] - skills: ["/skills/"] - system_prompt_ref: RESEARCHER_INSTRUCTIONS - -code-agent: - description: "Implement experiment code and runnable scripts; keep changes minimal and reproducible." - tools: [think_tool] - skills: ["/skills/"] - system_prompt: | - You are the code-agent. Implement experiment code in the workspace and keep changes minimal, - reproducible, and easy to run. - - Guidelines: - - Prefer small scripts and clear entry points. - - Record exact commands to run and where outputs are written. - - Write outputs under /artifacts/ (recommended) and log key params to /experiment_log.md (optional). - - Do not modify /skills/. - - If a relevant local skill exists, read its SKILL.md and follow its workflow instead of reinventing. - - Check `/memories/experiment-memory.md` for proven strategies from past cycles before implementing. - Skip if the file does not exist yet. - - Before heavy runs, confirm GPU/CUDA/VRAM availability and required packages. - - Suggested preflight commands: - - nvidia-smi - - python -c "import torch; print(torch.cuda.is_available(), torch.version.cuda, torch.cuda.get_device_name(0))" - - When responding, include: - - Files changed - - Commands to run - - Output paths - - Any remaining issues/next steps - -debug-agent: - description: "Debug runtime failures and fix bugs with minimal, verifiable patches." - tools: [think_tool] - skills: ["/skills/"] - system_prompt: | - You are the debug-agent. Reproduce failures, identify root causes, apply minimal fixes, and provide - concise diagnostics. - - Guidelines: - - Prefer small, safe changes. - - Explain the root cause in one paragraph. - - Provide how to reproduce and how to verify the fix. - - Do not modify /skills/. - - If a relevant local skill exists, read its SKILL.md and use it as a diagnostic checklist. - - When responding, include: - - Root cause - - Fix summary (files/changes) - - Repro steps - - Verification steps - -data-analysis-agent: - description: "Analyze experiment outputs: compute metrics, make plots, summarize insights." - tools: [think_tool] - skills: ["/skills/"] - system_prompt: | - You are the data-analysis-agent. Analyze experiment outputs, compute metrics, and create - publication-friendly plots. - - Guidelines: - - Do not invent numbers; compute from files or state what is missing. - - Save figures/tables under /artifacts/ (recommended) and reference paths. - - Summarize insights and provide 1-3 recommended next experiments. - - If a relevant local skill exists (evaluation, logging, plotting), read its SKILL.md and follow it. - - Report effect sizes and uncertainty (confidence intervals/error bars) when applicable. - - Apply multiple-testing corrections when comparing many conditions. - - Distinguish exploratory vs confirmatory findings. - - When responding, include: - - Metrics computed (with definitions) - - Figures/tables produced (paths) - - Interpretation and next steps - -writing-agent: - description: "Draft a paper-ready Markdown experiment report (no fabricated results/citations)." - tools: [think_tool] - skills: ["/skills/"] - system_prompt: | - You are the writing-agent. Draft a clear Markdown experimental report suitable for later paper writing. - - Guidelines: - - Use the experiment plan, logs, and artifacts. Reference file paths for figures/tables. - - Do not fabricate results or citations. - - If something is missing, add a TODO with the exact command needed to generate it. - - If a relevant local skill exists (e.g., paper-writing, reporting conventions), read its SKILL.md and follow it. - - Report uncertainty, effect sizes, and statistical corrections when relevant. - - Include negative results and clear limitations. - - Document evaluation protocol (splits/metrics/baselines) and data QC checks. - - Preferred sections: - 1) Summary & goals - 2) Experiment plan (stages + success signals) - 3) Setup (data, model, environment, parameters) - 4) Baselines and comparisons - 5) Results (with artifact paths) - 6) Analysis, limitations, and next steps - 7) Sources (only if web research was used) diff --git a/EvoScientist/subagents/__init__.py b/EvoScientist/subagents/__init__.py new file mode 100644 index 0000000..dc82e6d --- /dev/null +++ b/EvoScientist/subagents/__init__.py @@ -0,0 +1,12 @@ +"""Sub-agent definitions (YAML, one file per agent). + +Each ``.yaml`` here describes one sub-agent in the form expected by +``EvoScientist.utils.load_subagents``. The directory is the canonical +single source of truth for sub-agent prompts, tools, skills, and metadata. + +Optional ``async: true`` on a sub-agent's yaml routes it through +``langgraph dev`` as an AsyncSubAgent when ``config.enable_async_subagents`` +is set; the matching deployment binding lives in +``EvoScientist/langgraph_dev/graphs.py``, built by +``EvoScientist.subagents._factory.build_async_subagent_graph``. +""" diff --git a/EvoScientist/subagents/_factory.py b/EvoScientist/subagents/_factory.py new file mode 100644 index 0000000..e95ed53 --- /dev/null +++ b/EvoScientist/subagents/_factory.py @@ -0,0 +1,107 @@ +"""Factory for building deployable sub-agent graphs from yaml definitions. + +Lives in ``EvoScientist/subagents/`` next to the canonical yaml entries +because the factory is "build a graph from a sub-agent name" — a generic +construction utility, not a deployment concern. Any deployment surface +(``EvoScientist/langgraph_dev/``, future ``langgraph_platform/``, custom +servers) can call ``build_async_subagent_graph(name)`` to materialize the +runnable graph. + +Reuses the main EvoScientist agent's chat model, backend, and middleware so +the deployed sub-agent has full capability parity with its in-process +synchronous counterpart: same workspace files, same ``/skills/`` and +``/memory/`` routes, same error-handling and context-overflow middleware. +""" + +from __future__ import annotations + +import os +from typing import Any + + +def build_async_subagent_graph(name: str) -> Any: + """Build a deployable graph for the ``name`` sub-agent defined in yaml. + + Args: + name: The sub-agent's key in one of the ``EvoScientist/subagents/*.yaml`` + files (e.g. ``"writing-agent"``). + + Returns: + A compiled ``langgraph`` graph ready for registration in ``langgraph.json``. + + Raises: + ValueError: If ``name`` is not defined under ``EvoScientist/subagents/``. + """ + # Lazy imports — the factory is invoked at langgraph dev startup time, so + # all heavy modules (deepagents, llm, MCP) are pulled in here rather than + # at package import. + from deepagents import create_deep_agent + + from EvoScientist.config import apply_config_to_env, get_effective_config + from EvoScientist.EvoScientist import ( + SUBAGENTS_CONFIG, + _build_prompt_refs, + _ensure_chat_model, + _get_default_backend, + _get_default_middleware, + ) + from EvoScientist.tools import tavily_search, think_tool + from EvoScientist.utils import load_subagents + + # Surface API keys as env vars so downstream SDKs (openai, anthropic, …) + # find them on subprocess invocations from langgraph dev. + cfg = get_effective_config() + apply_config_to_env(cfg) + + # Mirror the tool registry constructed in EvoScientist._build_base_kwargs. + tool_registry = {"think_tool": think_tool} + if os.environ.get("TAVILY_API_KEY"): + tool_registry["tavily_search"] = tavily_search + + # Use the official loader so resolved tools, prompt_refs, and skills are + # all wired the same way as the in-process sync version. + specs = load_subagents( + SUBAGENTS_CONFIG, + tool_registry=tool_registry, + prompt_refs=_build_prompt_refs(), + ) + spec = next((s for s in specs if s.get("name") == name), None) + if spec is None: + raise ValueError( + f"Sub-agent {name!r} not found in {SUBAGENTS_CONFIG}. " + f"Available: {[s.get('name') for s in specs]}" + ) + + # Load MCP tools routed to THIS agent via ``expose_to: `` in + # ``mcp.yaml``. Use the cached helper so multiple ``build_async_subagent_graph`` + # calls in the same langgraph dev subprocess (one per registered async graph) + # share a single MCP connection set per server instead of re-spawning. + from EvoScientist.EvoScientist import _load_mcp_tools_cached + + mcp_tools_by_agent = _load_mcp_tools_cached() + agent_mcp_tools = mcp_tools_by_agent.get(name, []) + + # NOTE on HITL: async sub-agents intentionally do NOT set ``interrupt_on``, + # even though the deployed main agent does. They run as standalone graphs + # on the langgraph dev subprocess; the parent (CLI main agent) only sees a + # ``task_id`` from ``start_async_task`` and has no UI path to surface a + # paused-on-interrupt child to the user. Setting ``interrupt_on`` here + # would hang the sub-agent on its first ``execute`` call with no one to + # approve. The user-visible HITL boundary is the parent's + # ``start_async_task`` decision; restrict the child's reach by limiting + # ``tools`` in ``subagents/.yaml`` instead. + # Memory middleware is included so async sub-agents can READ + # /memory/MEMORY.md, but the extraction trigger (20+ human messages, + # see middleware/memory.py) never fires here — sub-agents only receive + # the parent's task delegation as a system prompt, not human messages. + # Net effect: sub-agents have read-only memory access. Memory writes + # happen exclusively from the main agent's user-facing conversation. + return create_deep_agent( + name=name, + model=_ensure_chat_model(), + system_prompt=spec.get("system_prompt", ""), + tools=spec.get("tools", []) + agent_mcp_tools, + skills=spec.get("skills"), + backend=_get_default_backend(), + middleware=_get_default_middleware(), + ).with_config({"recursion_limit": cfg.recursion_limit}) diff --git a/EvoScientist/subagents/code.yaml b/EvoScientist/subagents/code.yaml new file mode 100644 index 0000000..4a9f8e2 --- /dev/null +++ b/EvoScientist/subagents/code.yaml @@ -0,0 +1,26 @@ +code-agent: + description: "Implement experiment code and runnable scripts; keep changes minimal and reproducible." + tools: [think_tool] + skills: ["/skills/"] + system_prompt: | + You are the code-agent. Implement experiment code in the workspace and keep changes minimal, + reproducible, and easy to run. + + Guidelines: + - Prefer small scripts and clear entry points. + - Record exact commands to run and where outputs are written. + - Write outputs under /artifacts/ (recommended) and log key params to /experiment_log.md (optional). + - Do not modify /skills/. + - If a relevant local skill exists, read its SKILL.md and follow its workflow instead of reinventing. + - Check `/memories/experiment-memory.md` for proven strategies from past cycles before implementing. + Skip if the file does not exist yet. + - Before heavy runs, confirm GPU/CUDA/VRAM availability and required packages. + - Suggested preflight commands: + - nvidia-smi + - python -c "import torch; print(torch.cuda.is_available(), torch.version.cuda, torch.cuda.get_device_name(0))" + + When responding, include: + - Files changed + - Commands to run + - Output paths + - Any remaining issues/next steps diff --git a/EvoScientist/subagents/data_analysis.yaml b/EvoScientist/subagents/data_analysis.yaml new file mode 100644 index 0000000..923ff58 --- /dev/null +++ b/EvoScientist/subagents/data_analysis.yaml @@ -0,0 +1,25 @@ +data-analysis-agent: + description: "Analyze experiment outputs: compute metrics, make plots, summarize insights." + tools: [think_tool] + skills: ["/skills/"] + # Long-running (typically 30-120s for compute + plots). Routed through + # langgraph dev as an AsyncSubAgent when config.enable_async_subagents is set. + # Deployment graph: EvoScientist.langgraph_dev.graphs:data_analysis_agent + async: true + system_prompt: | + You are the data-analysis-agent. Analyze experiment outputs, compute metrics, and create + publication-friendly plots. + + Guidelines: + - Do not invent numbers; compute from files or state what is missing. + - Save figures/tables under /artifacts/ (recommended) and reference paths. + - Summarize insights and provide 1-3 recommended next experiments. + - If a relevant local skill exists (evaluation, logging, plotting), read its SKILL.md and follow it. + - Report effect sizes and uncertainty (confidence intervals/error bars) when applicable. + - Apply multiple-testing corrections when comparing many conditions. + - Distinguish exploratory vs confirmatory findings. + + When responding, include: + - Metrics computed (with definitions) + - Figures/tables produced (paths) + - Interpretation and next steps diff --git a/EvoScientist/subagents/debug.yaml b/EvoScientist/subagents/debug.yaml new file mode 100644 index 0000000..198b7fd --- /dev/null +++ b/EvoScientist/subagents/debug.yaml @@ -0,0 +1,20 @@ +debug-agent: + description: "Debug runtime failures and fix bugs with minimal, verifiable patches." + tools: [think_tool] + skills: ["/skills/"] + system_prompt: | + You are the debug-agent. Reproduce failures, identify root causes, apply minimal fixes, and provide + concise diagnostics. + + Guidelines: + - Prefer small, safe changes. + - Explain the root cause in one paragraph. + - Provide how to reproduce and how to verify the fix. + - Do not modify /skills/. + - If a relevant local skill exists, read its SKILL.md and use it as a diagnostic checklist. + + When responding, include: + - Root cause + - Fix summary (files/changes) + - Repro steps + - Verification steps diff --git a/EvoScientist/subagents/planner.yaml b/EvoScientist/subagents/planner.yaml new file mode 100644 index 0000000..6ea3a9f --- /dev/null +++ b/EvoScientist/subagents/planner.yaml @@ -0,0 +1,59 @@ +planner-agent: + description: "Plan experiments: stages, success signals, and dependencies (no web search, no implementation)." + tools: [think_tool] + skills: ["/skills/"] + system_prompt: | + You are the planner-agent. You do NOT implement code. You create and update experimental plans + that are practical to run locally. + + Before planning, check `/memories/ideation-memory.md` and `/memories/experiment-memory.md` + for prior knowledge from past research cycles. Incorporate relevant entries into + your plan (e.g., proven strategies, known failed directions). Skip if these files + do not exist yet. + + You may be invoked in two modes: + 1) PLAN MODE: produce an initial experimental plan. + 2) REFLECTION MODE: update the plan based on stage results. + + The caller should start the task with either: + - MODE: PLAN + - MODE: REFLECTION + If MODE is not specified, assume PLAN. + + PLAN MODE output (Markdown): + 1) Assumptions & scope + 2) Stages (numbered). For each stage include: + - goal + - success signals (metrics/thresholds or qualitative checks) + - what to run (scripts/commands at a high level) + - expected artifacts (tables/plots/logs) + 3) Dependencies (data, compute, environment) + 4) Iteration triggers (when to change dataset/model/objective) + 5) Evaluation protocol (splits, primary metrics, baselines) and data quality checks + 6) Environment preflight (GPU/CUDA/VRAM/disk) and required dependencies (pip packages) + + REFLECTION MODE output (JSON only, no extra text): + { + "completed": ["..."], + "unmet_success_signals": ["..."], + "skill_suggestions": ["..."], + "stage_modifications": [ + {"stage": "Stage name or index", "change": "What to adjust and why"} + ], + "new_stages": [ + { + "title": "...", + "goal": "...", + "success_signals": ["..."], + "what_to_run": ["..."], + "expected_artifacts": ["..."] + } + ], + "todo_updates": ["..."] + } + + Empty arrays are valid. If no changes are needed, return the JSON with empty arrays. + "skill_suggestions" should use skill names from your available skills listing. + + Keep the structure flexible (not rigid templates). If model size is unspecified, default to + <=7B-class models and lightweight baselines. diff --git a/EvoScientist/subagents/research.yaml b/EvoScientist/subagents/research.yaml new file mode 100644 index 0000000..a895be9 --- /dev/null +++ b/EvoScientist/subagents/research.yaml @@ -0,0 +1,5 @@ +research-agent: + description: "Web research for methods/baselines/datasets (one topic at a time, return actionable notes + sources)." + tools: [tavily_search, think_tool] + skills: ["/skills/"] + system_prompt_ref: RESEARCHER_INSTRUCTIONS diff --git a/EvoScientist/subagents/writing.yaml b/EvoScientist/subagents/writing.yaml new file mode 100644 index 0000000..a461a56 --- /dev/null +++ b/EvoScientist/subagents/writing.yaml @@ -0,0 +1,28 @@ +writing-agent: + description: "Draft a paper-ready Markdown experiment report (no fabricated results/citations)." + tools: [think_tool] + skills: ["/skills/"] + # Long-running (typically 30-180s for full report). Routed through langgraph + # dev as an AsyncSubAgent when config.enable_async_subagents is set. + # Deployment graph: EvoScientist.langgraph_dev.graphs:writing_agent + async: true + system_prompt: | + You are the writing-agent. Draft a clear Markdown experimental report suitable for later paper writing. + + Guidelines: + - Use the experiment plan, logs, and artifacts. Reference file paths for figures/tables. + - Do not fabricate results or citations. + - If something is missing, add a TODO with the exact command needed to generate it. + - If a relevant local skill exists (e.g., paper-writing, reporting conventions), read its SKILL.md and follow it. + - Report uncertainty, effect sizes, and statistical corrections when relevant. + - Include negative results and clear limitations. + - Document evaluation protocol (splits/metrics/baselines) and data QC checks. + + Preferred sections: + 1) Summary & goals + 2) Experiment plan (stages + success signals) + 3) Setup (data, model, environment, parameters) + 4) Baselines and comparisons + 5) Results (with artifact paths) + 6) Analysis, limitations, and next steps + 7) Sources (only if web research was used) diff --git a/EvoScientist/utils.py b/EvoScientist/utils.py index 3e6ed86..60cd7bd 100644 --- a/EvoScientist/utils.py +++ b/EvoScientist/utils.py @@ -115,40 +115,77 @@ def load_subagents( tool_registry: dict[str, Any], prompt_refs: dict[str, str] | None = None, ) -> list[dict[str, Any]]: - """Load subagent definitions from YAML and wire up tools. + """Load subagent definitions from a directory of YAML files and wire up tools. NOTE: This is a custom utility. deepagents does not natively load subagents from files - they're normally defined inline in the create_deep_agent() call. We externalize to YAML here to keep configuration separate from code. - Supported YAML schemas: + ``config_path`` must be a directory containing one ``.yaml`` per + sub-agent. All ``*.yaml`` files are merged into a single mapping. Files + starting with ``.`` (dotfiles, editor swap files) or ``_`` (private / + disabled) are ignored. ``.yml`` is intentionally not supported — keeps + one canonical extension and avoids the dev-vs-wheel packaging mismatch. - 1) Mapping style (recommended): - planner-agent: - description: "..." - tools: [think_tool] - system_prompt: | - ... - research-agent: - description: "..." - tools: [tavily_search, think_tool] - system_prompt_ref: RESEARCHER_INSTRUCTIONS + Each file's top level must be a mapping ``{: }``:: - 2) List style (legacy): - subagents: - - name: planner-agent - description: "..." - tools: [think_tool] - system_prompt: | - ... + planner-agent: + description: "..." + tools: [think_tool] + system_prompt: | + ... + research-agent: + description: "..." + tools: [tavily_search, think_tool] + system_prompt_ref: RESEARCHER_INSTRUCTIONS """ prompt_refs = prompt_refs or {} - with config_path.open(encoding="utf-8") as f: - config = yaml.safe_load(f) or {} + if not config_path.is_dir(): + raise ValueError( + f"{config_path}: sub-agent config must be a directory " + f"containing one .yaml per agent" + ) - if not isinstance(config, dict) or not config: - raise ValueError("subagent.yaml must be a mapping or contain 'subagents:'") + # Only ``.yaml`` is supported (canonical extension). Skip files starting + # with ``_`` (private / disabled) or ``.`` (dotfiles, editor swap files + # like ``.foo.yaml.swp``). ``.yml`` is intentionally not loaded — keeps + # one canonical extension across the project, simplifies packaging + # (no need for a parallel ``subagents/*.yml`` entry in ``pyproject.toml`` + # ``package-data``), and matches every existing yaml file in this repo. + config: dict[str, Any] = {} + for yml in sorted(config_path.glob("*.yaml")): + if yml.name.startswith(".") or yml.name.startswith("_"): + continue + with yml.open(encoding="utf-8") as f: + data = yaml.safe_load(f) + if data is None: + # Empty file — skip silently (matches dotfile/underscore skip behavior) + continue + if not isinstance(data, dict): + raise ValueError( + f"{yml}: top-level must be a mapping (one entry per sub-agent)" + ) + # Detect duplicate keys across files + for key in data: + if key in config: + raise ValueError( + f"Sub-agent {key!r} defined in multiple files; " + f"second occurrence in {yml.name}" + ) + for key, spec in data.items(): + if not isinstance(spec, dict): + raise ValueError( + f"{yml}: sub-agent {key!r} must map to a spec dict, " + f"got {type(spec).__name__}" + ) + config.update(data) + + if not config: + raise ValueError( + f"{config_path}: no sub-agent definitions found " + f"(expected one or more .yaml files)" + ) subagents: list[dict[str, Any]] = [] @@ -185,26 +222,24 @@ def load_subagents( ) subagent["tools"] = resolved + # Internal field: carries the ``async:`` yaml flag through to + # ``_maybe_swap_async_subagents`` so the swap doesn't need a second + # yaml pass to discover async-flagged agents. Underscore prefix marks + # it as internal — must be popped before passing to deepagents. + async_val = spec.get("async", False) + if not isinstance(async_val, bool): + # Reject quoted-string yaml values like ``async: "false"`` — + # ``bool("false")`` is ``True`` (non-empty string), which silently + # flips the agent into async mode. Fail loud instead. + raise ValueError( + f"Subagent {name!r}: 'async' must be a boolean, " + f"got {type(async_val).__name__}: {async_val!r}" + ) + subagent["_async"] = async_val + return subagent - # Legacy list style - if "subagents" in config: - items = config.get("subagents") - if not isinstance(items, list) or not items: - raise ValueError("subagent.yaml must contain a non-empty 'subagents:' list") - for item in items: - if not isinstance(item, dict): - continue - name = item.get("name") - if not name: - raise ValueError("Each subagent entry must have a 'name'") - subagents.append(_build_one(name, item)) - return subagents - - # Mapping style: {: } for name, spec in config.items(): - if not isinstance(spec, dict): - continue subagents.append(_build_one(name, spec)) return subagents diff --git a/pyproject.toml b/pyproject.toml index 1ae42eb..d458b9f 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -34,6 +34,8 @@ dependencies = [ "langgraph-cli[inmem]>=0.4", "langgraph-checkpoint-sqlite>=3.0", "httpx>=0.28", + "psutil>=6.0", + "filelock>=3.16", "lazy-loader>=0.5", "markdownify>=1.2", "nest-asyncio>=1.6", @@ -97,7 +99,11 @@ build-backend = "setuptools.build_meta" include = ["EvoScientist*"] [tool.setuptools.package-data] -EvoScientist = ["subagent.yaml", "skills/**/*"] +EvoScientist = [ + "subagents/*.yaml", + "langgraph_dev/*.json", + "skills/**/*", +] [tool.pytest.ini_options] testpaths = ["tests"] diff --git a/tests/test_async_subagent_swap.py b/tests/test_async_subagent_swap.py new file mode 100644 index 0000000..382fc77 --- /dev/null +++ b/tests/test_async_subagent_swap.py @@ -0,0 +1,173 @@ +"""Tests for ``EvoScientist._maybe_swap_async_subagents``. + +Covers the fallback / swap / strip-internal-flag paths that decide whether +sub-agents are routed in-process (sync ``task`` tool) or to the langgraph +dev subprocess (``AsyncSubAgent`` over HTTP). +""" + +from __future__ import annotations + +from types import SimpleNamespace +from unittest.mock import patch + +from EvoScientist.EvoScientist import _maybe_swap_async_subagents + + +def _sub(name: str, *, async_flag: bool, description: str = "desc") -> dict: + """Build a sub-agent dict shaped like ``utils.load_subagents`` output.""" + return { + "name": name, + "description": description, + "system_prompt": "x", + "tools": [], + "_async": async_flag, + } + + +# ============================================================================= +# Async disabled in config → return unchanged +# ============================================================================= + + +def test_returns_unchanged_when_async_disabled_and_strips_flag(): + """No swap when config.enable_async_subagents is False (the default). + + Even in this disabled-path, the internal ``_async`` flag must be stripped + before sub-agents reach deepagents (which may schema-validate the dicts). + """ + cfg = SimpleNamespace(enable_async_subagents=False) + subs = [ + _sub("planner-agent", async_flag=False), + _sub("writing-agent", async_flag=True), + ] + with patch("EvoScientist.EvoScientist._ensure_config", return_value=cfg): + out = _maybe_swap_async_subagents(subs) + assert out is subs + for s in out: + assert "_async" not in s, f"_async leaked into {s['name']}" + + +# ============================================================================= +# Async enabled but langgraph dev unreachable → strip flag, return as sync +# ============================================================================= + + +class TestFallbackPath: + def _setup(self): + return SimpleNamespace( + enable_async_subagents=True, + langgraph_dev_port=6174, + ) + + def test_returns_subs_unchanged(self): + cfg = self._setup() + subs = [ + _sub("planner-agent", async_flag=False), + _sub("writing-agent", async_flag=True), + ] + with ( + patch("EvoScientist.EvoScientist._ensure_config", return_value=cfg), + patch( + "EvoScientist.langgraph_dev.manager.is_async_subagents_available", + return_value=False, + ), + ): + out = _maybe_swap_async_subagents(subs) + assert out is subs + + def test_strips_async_flag_from_all_subs(self): + """Even fallback path must strip _async before deepagents handoff.""" + cfg = self._setup() + subs = [ + _sub("planner-agent", async_flag=False), + _sub("writing-agent", async_flag=True), + ] + with ( + patch("EvoScientist.EvoScientist._ensure_config", return_value=cfg), + patch( + "EvoScientist.langgraph_dev.manager.is_async_subagents_available", + return_value=False, + ), + ): + out = _maybe_swap_async_subagents(subs) + for s in out: + assert "_async" not in s, f"_async leaked into {s['name']}" + + +# ============================================================================= +# Async enabled + reachable + nothing flagged async → return all as sync (stripped) +# ============================================================================= + + +def test_no_async_flagged_subs_strips_and_returns(): + cfg = SimpleNamespace(enable_async_subagents=True, langgraph_dev_port=6174) + subs = [ + _sub("planner-agent", async_flag=False), + _sub("research-agent", async_flag=False), + ] + with ( + patch("EvoScientist.EvoScientist._ensure_config", return_value=cfg), + patch( + "EvoScientist.langgraph_dev.manager.is_async_subagents_available", + return_value=True, + ), + ): + out = _maybe_swap_async_subagents(subs) + assert out is subs # nothing to swap, returned as-is + for s in out: + assert "_async" not in s + + +# ============================================================================= +# Async enabled + reachable + has async-flagged subs → swap to AsyncSubAgent +# ============================================================================= + + +def test_swaps_async_flagged_subs(): + cfg = SimpleNamespace(enable_async_subagents=True, langgraph_dev_port=6174) + subs = [ + _sub("planner-agent", async_flag=False, description="plan"), + _sub("writing-agent", async_flag=True, description="write report"), + _sub("data-analysis-agent", async_flag=True, description="analyze"), + ] + with ( + patch("EvoScientist.EvoScientist._ensure_config", return_value=cfg), + patch( + "EvoScientist.langgraph_dev.manager.is_async_subagents_available", + return_value=True, + ), + ): + out = _maybe_swap_async_subagents(subs) + + assert len(out) == 3 + + # Sync sub kept as a plain dict, _async stripped. + by_name = {s["name"]: s for s in out} + planner = by_name["planner-agent"] + assert isinstance(planner, dict) + assert "_async" not in planner + + # Async subs are AsyncSubAgent specs (TypedDict) pointing at the right URL. + writing = by_name["writing-agent"] + assert writing["graph_id"] == "writing-agent" + assert writing["url"] == "http://localhost:6174" + assert writing["description"] == "write report" + + data = by_name["data-analysis-agent"] + assert data["graph_id"] == "data-analysis-agent" + assert data["url"] == "http://localhost:6174" + + +def test_swap_uses_configured_port(): + """AsyncSubAgent.url should reflect cfg.langgraph_dev_port, not hardcoded.""" + cfg = SimpleNamespace(enable_async_subagents=True, langgraph_dev_port=9999) + subs = [_sub("writing-agent", async_flag=True)] + with ( + patch("EvoScientist.EvoScientist._ensure_config", return_value=cfg), + patch( + "EvoScientist.langgraph_dev.manager.is_async_subagents_available", + return_value=True, + ), + ): + out = _maybe_swap_async_subagents(subs) + assert out[0]["url"] == "http://localhost:9999" diff --git a/tests/test_langgraph_manager.py b/tests/test_langgraph_manager.py new file mode 100644 index 0000000..5650758 --- /dev/null +++ b/tests/test_langgraph_manager.py @@ -0,0 +1,234 @@ +"""Happy-path tests for langgraph_dev.manager. + +Mocks httpx, psutil, subprocess.Popen, and module-level state so the tests +run on CI without requiring the langgraph CLI to be installed or any port +to be available. +""" + +from __future__ import annotations + +from types import SimpleNamespace +from unittest.mock import MagicMock, patch + +import httpx +import pytest + +from EvoScientist.langgraph_dev import manager + + +@pytest.fixture(autouse=True) +def reset_module_state(): + """Reset manager module globals before each test for isolation.""" + manager._PROCESS = None + manager._PROCESS_WORKSPACE = None + manager._ASYNC_SUBAGENTS_AVAILABLE = False + yield + manager._PROCESS = None + manager._PROCESS_WORKSPACE = None + manager._ASYNC_SUBAGENTS_AVAILABLE = False + + +# ============================================================================= +# is_langgraph_dev_running +# ============================================================================= + + +class TestIsLanggraphDevRunning: + @patch("EvoScientist.langgraph_dev.manager.httpx.get") + def test_returns_false_on_connect_error(self, mock_get): + mock_get.side_effect = httpx.ConnectError("refused") + assert manager.is_langgraph_dev_running(port=6174) is False + + @patch("EvoScientist.langgraph_dev.manager.httpx.get") + def test_returns_false_on_timeout(self, mock_get): + mock_get.side_effect = httpx.TimeoutException("slow") + assert manager.is_langgraph_dev_running(port=6174) is False + + @patch("EvoScientist.langgraph_dev.manager.httpx.get") + def test_returns_true_on_200(self, mock_get): + mock_get.return_value = MagicMock(status_code=200) + assert manager.is_langgraph_dev_running(port=6174) is True + # Verify it probed /ok at the configured port. + called_url = mock_get.call_args[0][0] + assert called_url == "http://localhost:6174/ok" + + @patch("EvoScientist.langgraph_dev.manager.httpx.get") + def test_returns_false_on_non_200(self, mock_get): + mock_get.return_value = MagicMock(status_code=503) + assert manager.is_langgraph_dev_running(port=6174) is False + + +# ============================================================================= +# _list_pids_on_port +# ============================================================================= + + +class TestListPidsOnPort: + @patch("EvoScientist.langgraph_dev.manager.psutil.net_connections") + def test_empty_when_no_connections(self, mock_net): + mock_net.return_value = [] + assert manager._list_pids_on_port(6174) == [] + + @patch("EvoScientist.langgraph_dev.manager.psutil.net_connections") + def test_returns_pid_for_matching_port(self, mock_net): + mock_net.return_value = [ + SimpleNamespace(laddr=SimpleNamespace(port=6174), pid=12345), + SimpleNamespace(laddr=SimpleNamespace(port=8080), pid=99999), + ] + result = manager._list_pids_on_port(6174) + assert result == [12345] + + @patch("EvoScientist.langgraph_dev.manager.psutil.net_connections") + def test_filters_none_pid(self, mock_net): + mock_net.return_value = [ + SimpleNamespace(laddr=SimpleNamespace(port=6174), pid=None), + SimpleNamespace(laddr=SimpleNamespace(port=6174), pid=12345), + ] + result = manager._list_pids_on_port(6174) + assert result == [12345] + + def test_returns_empty_on_access_denied(self): + with patch.object( + manager.psutil, + "net_connections", + side_effect=manager.psutil.AccessDenied(), + ): + assert manager._list_pids_on_port(6174) == [] + + +# ============================================================================= +# _kill_owned_stale_process +# ============================================================================= + + +class TestKillOwnedStaleProcess: + def test_returns_false_if_no_pid_file(self, tmp_path): + with patch.object(manager, "_PID_FILE", tmp_path / "missing.pid"): + assert manager._kill_owned_stale_process(6174) is False + + def test_returns_false_if_pid_file_unreadable(self, tmp_path): + pid_file = tmp_path / "bad.pid" + pid_file.write_text("not-a-number") + with patch.object(manager, "_PID_FILE", pid_file): + assert manager._kill_owned_stale_process(6174) is False + + def test_returns_false_if_pid_not_in_occupiers(self, tmp_path): + pid_file = tmp_path / "lg.pid" + pid_file.write_text("12345") + with ( + patch.object(manager, "_PID_FILE", pid_file), + patch.object(manager, "_list_pids_on_port", return_value=[99999]), + ): + assert manager._kill_owned_stale_process(6174) is False + # PID file should be left intact — the port is held by someone + # else, not a stale ours. + assert pid_file.exists() + + def test_refuses_to_kill_recycled_pid(self, tmp_path): + """PID matches but cmdline doesn't contain 'langgraph' → don't kill.""" + pid_file = tmp_path / "lg.pid" + pid_file.write_text("12345") + fake_proc = MagicMock() + fake_proc.cmdline.return_value = ["bash", "-c", "echo hi"] + with ( + patch.object(manager, "_PID_FILE", pid_file), + patch.object(manager, "_list_pids_on_port", return_value=[12345]), + patch.object(manager.psutil, "Process", return_value=fake_proc), + ): + assert manager._kill_owned_stale_process(6174) is False + fake_proc.kill.assert_not_called() + # PID file should be removed — the entry is stale (our process is + # gone, PID was recycled by an unrelated process). + assert not pid_file.exists() + + def test_kills_when_cmdline_matches_langgraph(self, tmp_path): + """Owned PID + cmdline contains 'langgraph' → kill + cleanup PID file.""" + pid_file = tmp_path / "lg.pid" + pid_file.write_text("12345") + fake_proc = MagicMock() + fake_proc.cmdline.return_value = [ + "/usr/bin/python", + "/usr/bin/langgraph", + "dev", + ] + with ( + patch.object(manager, "_PID_FILE", pid_file), + patch.object(manager, "_list_pids_on_port", return_value=[12345]), + patch.object(manager.psutil, "Process", return_value=fake_proc), + ): + assert manager._kill_owned_stale_process(6174) is True + fake_proc.kill.assert_called_once() + assert not pid_file.exists() + + def test_handles_dead_pid(self, tmp_path): + """PID file claims a PID but the process is gone → cleanup PID file, no error.""" + pid_file = tmp_path / "lg.pid" + pid_file.write_text("12345") + with ( + patch.object(manager, "_PID_FILE", pid_file), + patch.object(manager, "_list_pids_on_port", return_value=[12345]), + patch.object( + manager.psutil, + "Process", + side_effect=manager.psutil.NoSuchProcess(12345), + ), + ): + assert manager._kill_owned_stale_process(6174) is False + assert not pid_file.exists() + + +# ============================================================================= +# ensure_langgraph_dev — high-level orchestration +# ============================================================================= + + +class TestEnsureLanggraphDev: + def test_returns_none_when_async_disabled(self): + cfg = SimpleNamespace(enable_async_subagents=False) + assert manager.ensure_langgraph_dev(cfg) is None + # And the availability flag should remain False. + assert manager.is_async_subagents_available() is False + + def test_reuses_existing_healthy_subprocess(self, tmp_path): + """When the subprocess is already running, no new Popen call.""" + cfg = SimpleNamespace( + enable_async_subagents=True, + langgraph_dev_port=6174, + langgraph_dev_file_persistence=True, + ) + with ( + patch.object( + manager, "is_langgraph_dev_running", return_value=True + ) as mock_running, + patch.object(manager, "start_langgraph_dev") as mock_start, + patch.object(manager, "_FILE_LOCK_PATH", tmp_path / "lg.lock"), + # Isolate from real ``~/.config/evoscientist/`` — without this + # patch, the FileLock setup would mkdir the user's actual config + # dir as a test side-effect. + patch.object(manager, "_PID_DIR", tmp_path / "pids"), + ): + result = manager.ensure_langgraph_dev(cfg, workspace_dir=tmp_path) + # We didn't spawn anything — there's already a healthy server. + mock_start.assert_not_called() + # Reuse path returns None (we don't own the existing process). + assert result is None + # is_async_subagents_available was flipped True. + assert manager.is_async_subagents_available() is True + # Health check was called at least once. + assert mock_running.called + + +# ============================================================================= +# is_async_subagents_available — module state +# ============================================================================= + + +class TestIsAsyncSubagentsAvailable: + def test_starts_false(self): + assert manager.is_async_subagents_available() is False + + def test_reflects_module_state(self): + manager._ASYNC_SUBAGENTS_AVAILABLE = True + assert manager.is_async_subagents_available() is True + manager._ASYNC_SUBAGENTS_AVAILABLE = False + assert manager.is_async_subagents_available() is False diff --git a/tests/test_load_subagents.py b/tests/test_load_subagents.py new file mode 100644 index 0000000..fb12999 --- /dev/null +++ b/tests/test_load_subagents.py @@ -0,0 +1,141 @@ +"""Tests for ``EvoScientist.utils.load_subagents``. + +Focused on schema-validation paths that are easy to silently misuse from +yaml — primarily the ``async:`` flag type check that prevents quoted-string +or integer values from being misinterpreted as booleans. +""" + +from __future__ import annotations + +import textwrap + +import pytest + +from EvoScientist.utils import load_subagents + + +def _write_yaml(tmp_path, name: str, body: str): + """Write ``body`` to ``tmp_path/name`` and return the directory path.""" + (tmp_path / name).write_text(textwrap.dedent(body)) + return tmp_path + + +def test_async_flag_accepts_real_bool(tmp_path): + """``async: true`` (real yaml boolean) is accepted and carried through.""" + config_path = _write_yaml( + tmp_path, + "writing.yaml", + """ + writing-agent: + description: Drafts reports + system_prompt: "" + tools: [] + async: true + """, + ) + subs = load_subagents(config_path, tool_registry={}) + assert len(subs) == 1 + assert subs[0]["name"] == "writing-agent" + assert subs[0]["_async"] is True + + +def test_async_flag_defaults_to_false_when_omitted(tmp_path): + """No ``async:`` field → ``_async`` defaults to False.""" + config_path = _write_yaml( + tmp_path, + "planner.yaml", + """ + planner-agent: + description: Plans experiments + system_prompt: "" + tools: [] + """, + ) + subs = load_subagents(config_path, tool_registry={}) + assert subs[0]["_async"] is False + + +def test_async_flag_rejects_quoted_string(tmp_path): + """``async: "false"`` (quoted) is a real user trap — bool("false") is True. + + Without the explicit isinstance check, this would silently flip the agent + into async mode. We require the validator to fail loud instead. + """ + config_path = _write_yaml( + tmp_path, + "bad.yaml", + """ + bad-agent: + description: "" + system_prompt: "" + tools: [] + async: "false" + """, + ) + with pytest.raises(ValueError, match=r"'async' must be a boolean"): + load_subagents(config_path, tool_registry={}) + + +def test_async_flag_rejects_integer(tmp_path): + """``async: 1`` is also rejected — yaml integers are not booleans.""" + config_path = _write_yaml( + tmp_path, + "bad.yaml", + """ + bad-agent: + description: "" + system_prompt: "" + tools: [] + async: 1 + """, + ) + with pytest.raises(ValueError, match=r"'async' must be a boolean"): + load_subagents(config_path, tool_registry={}) + + +def test_async_flag_error_includes_agent_name(tmp_path): + """Error message must include the offending agent name for triage.""" + config_path = _write_yaml( + tmp_path, + "bad.yaml", + """ + my-bad-agent: + description: "" + system_prompt: "" + tools: [] + async: "yes" + """, + ) + with pytest.raises(ValueError, match=r"my-bad-agent"): + load_subagents(config_path, tool_registry={}) + + +def test_non_dict_spec_raises(tmp_path): + """Yaml entries that aren't mappings must fail loud, not be silently dropped. + + Previously ``_build_one`` had a ``if not isinstance(spec, dict): continue`` + fallback that swallowed malformed entries — users would see their agent + quietly disappear with no error. Now caught during the merge loop. + """ + config_path = _write_yaml( + tmp_path, + "bad.yaml", + """ + bad-agent: 123 + """, + ) + with pytest.raises(ValueError, match=r"must map to a spec dict"): + load_subagents(config_path, tool_registry={}) + + +def test_non_dict_spec_error_includes_filename_and_name(tmp_path): + """Error must surface BOTH the offending file path and agent name.""" + config_path = _write_yaml( + tmp_path, + "weird.yaml", + """ + weird-agent: "just a string" + """, + ) + with pytest.raises(ValueError, match=r"weird\.yaml.*weird-agent"): + load_subagents(config_path, tool_registry={}) diff --git a/tests/test_onboard.py b/tests/test_onboard.py index 4957330..33b2fdc 100644 --- a/tests/test_onboard.py +++ b/tests/test_onboard.py @@ -21,11 +21,12 @@ from EvoScientist.config.onboard import ( class TestConstants: - def test_steps_has_eleven_items(self): - """Test that STEPS contains exactly 11 steps.""" - assert len(STEPS) == 11 + def test_steps_has_twelve_items(self): + """Test that STEPS contains exactly 12 steps.""" + assert len(STEPS) == 12 assert STEPS == [ "UI", + "LangGraph Port", "Provider", "API Key", "Model",