diff --git a/EvoScientist/EvoScientist.py b/EvoScientist/EvoScientist.py index ba5016f..e7aae4f 100644 --- a/EvoScientist/EvoScientist.py +++ b/EvoScientist/EvoScientist.py @@ -20,7 +20,6 @@ Usage: import json import logging import os -from datetime import datetime from pathlib import Path from langchain.agents.middleware import AgentMiddleware, HumanInTheLoopMiddleware @@ -28,7 +27,7 @@ from langchain.agents.middleware import AgentMiddleware, HumanInTheLoopMiddlewar from . import paths as _paths_mod from .config import apply_config_to_env, get_effective_config from .paths import set_active_workspace, set_workspace_root -from .prompts import RESEARCHER_INSTRUCTIONS, get_system_prompt +from .prompts import get_system_prompt # Suppress noisy warnings from deepagents skill loader (non-string frontmatter fields, etc.) logging.getLogger("deepagents.middleware.skills").setLevel(logging.ERROR) @@ -198,6 +197,7 @@ def _inject_subagent_middleware(subs: list[dict]) -> None: ContextOverflowMapperMiddleware, ToolErrorHandlerMiddleware, create_context_editing_middleware, + create_runtime_context_middleware, ) for sa in subs: @@ -206,21 +206,13 @@ def _inject_subagent_middleware(subs: list[dict]) -> None: # No ``model=`` — subagents share the main agent's model, # so defer to the factory's ``_ensure_chat_model()`` fallback. create_context_editing_middleware(), + create_runtime_context_middleware(), ToolErrorHandlerMiddleware(), ContextOverflowMapperMiddleware(), ] ) -def _build_prompt_refs() -> dict: - """Build prompt references with the current date (not frozen at import).""" - return { - "RESEARCHER_INSTRUCTIONS": RESEARCHER_INSTRUCTIONS.format( - date=datetime.now().strftime("%Y-%m-%d"), - ), - } - - def _maybe_swap_async_subagents(subs: list, middleware: list | None = None) -> list: """Replace ``_async``-flagged sub-agents with ``AsyncSubAgent`` specs when enabled. @@ -333,7 +325,6 @@ def _build_base_kwargs(base_backend, base_middleware): subs = load_subagents( SUBAGENTS_CONFIG, tool_registry=tool_registry, - prompt_refs=_build_prompt_refs(), ) _inject_subagent_middleware(subs) subs = _maybe_swap_async_subagents(subs, base_middleware) @@ -382,7 +373,6 @@ def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress= subs = load_subagents( SUBAGENTS_CONFIG, tool_registry=registry, - prompt_refs=_build_prompt_refs(), ) _inject_subagent_middleware(subs) @@ -476,6 +466,7 @@ def _get_default_middleware( create_code_interpreter_middleware, create_context_editing_middleware, create_memory_middleware, + create_runtime_context_middleware, create_tool_selector_middleware, load_fallback_chain, ) @@ -496,6 +487,7 @@ def _get_default_middleware( ContextOverflowMapperMiddleware(), ToolErrorHandlerMiddleware(), *create_tool_selector_middleware(model=model), + create_runtime_context_middleware(), create_memory_middleware(memory_dir, workspace_dir=workspace_dir), ] diff --git a/EvoScientist/__init__.py b/EvoScientist/__init__.py index afc4c08..485f1e8 100644 --- a/EvoScientist/__init__.py +++ b/EvoScientist/__init__.py @@ -29,7 +29,6 @@ _EXPORTS: dict[str, tuple[str, str]] = { "DEFAULT_MODEL": (".llm", "DEFAULT_MODEL"), # Prompts "get_system_prompt": (".prompts", "get_system_prompt"), - "RESEARCHER_INSTRUCTIONS": (".prompts", "RESEARCHER_INSTRUCTIONS"), # Tools "tavily_search": (".tools", "tavily_search"), "think_tool": (".tools", "think_tool"), diff --git a/EvoScientist/middleware/__init__.py b/EvoScientist/middleware/__init__.py index 607340c..7419a1c 100644 --- a/EvoScientist/middleware/__init__.py +++ b/EvoScientist/middleware/__init__.py @@ -23,6 +23,7 @@ from .memory import ( create_memory_middleware, ) from .model_fallback import ModelFallbackMiddleware, load_fallback_chain +from .runtime_context import RuntimeContextMiddleware, create_runtime_context_middleware from .tool_error_handler import ToolErrorHandlerMiddleware from .tool_selector import create_tool_selector_middleware from .utils import disable_thinking @@ -37,11 +38,13 @@ __all__ = [ "EvoMemoryMiddleware", "ModelFallbackMiddleware", "Question", + "RuntimeContextMiddleware", "ToolErrorHandlerMiddleware", "compute_context_editing_trigger", "create_code_interpreter_middleware", "create_context_editing_middleware", "create_memory_middleware", + "create_runtime_context_middleware", "create_tool_selector_middleware", "disable_thinking", "load_fallback_chain", diff --git a/EvoScientist/middleware/runtime_context.py b/EvoScientist/middleware/runtime_context.py new file mode 100644 index 0000000..d53f7c3 --- /dev/null +++ b/EvoScientist/middleware/runtime_context.py @@ -0,0 +1,86 @@ +"""Runtime context middleware for EvoScientist.""" + +from __future__ import annotations + +from collections.abc import Awaitable, Callable +from datetime import datetime + +from langchain.agents.middleware.types import ( + AgentMiddleware, + ModelRequest, + ModelResponse, +) + +RUNTIME_CONTEXT_TEMPLATE = """ +Current date: {date} +Local timezone: {timezone} + +Use this context to resolve relative time references like today, yesterday, and +next week. +""" + + +def _format_timezone(now: datetime) -> str: + """Return a compact local timezone label for prompt injection.""" + offset = now.utcoffset() + if offset is None: + return "local" + + total_seconds = int(offset.total_seconds()) + sign = "+" if total_seconds >= 0 else "-" + total_seconds = abs(total_seconds) + hours, remainder = divmod(total_seconds, 3600) + minutes = remainder // 60 + offset_text = f"UTC{sign}{hours:02d}:{minutes:02d}" + + name = now.tzname() + if name and name != offset_text: + return f"{name} ({offset_text})" + return offset_text + + +class RuntimeContextMiddleware(AgentMiddleware): + """Inject per-turn runtime context into model calls.""" + + def __init__(self, *, now_fn: Callable[[], datetime] | None = None) -> None: + self._now_fn = now_fn or (lambda: datetime.now().astimezone()) + + def _runtime_context(self) -> str: + now = self._now_fn() + return RUNTIME_CONTEXT_TEMPLATE.format( + date=now.strftime("%Y-%m-%d"), + timezone=_format_timezone(now), + ) + + def modify_request(self, request: ModelRequest) -> ModelRequest: + """Append runtime context to the system prompt.""" + from deepagents.middleware._utils import append_to_system_message + + new_system = append_to_system_message( + request.system_message, + self._runtime_context(), + ) + return request.override(system_message=new_system) + + def wrap_model_call( + self, + request: ModelRequest, + handler: Callable[[ModelRequest], ModelResponse], + ) -> ModelResponse: + """Inject runtime context before the sync model handler.""" + return handler(self.modify_request(request)) + + async def awrap_model_call( + self, + request: ModelRequest, + handler: Callable[[ModelRequest], Awaitable[ModelResponse]], + ) -> ModelResponse: + """Inject runtime context before the async model handler.""" + return await handler(self.modify_request(request)) + + +def create_runtime_context_middleware( + *, now_fn: Callable[[], datetime] | None = None +) -> RuntimeContextMiddleware: + """Build runtime-context middleware.""" + return RuntimeContextMiddleware(now_fn=now_fn) diff --git a/EvoScientist/prompts.py b/EvoScientist/prompts.py index 6942d78..a31dca2 100644 --- a/EvoScientist/prompts.py +++ b/EvoScientist/prompts.py @@ -13,8 +13,7 @@ The main agent's system prompt is assembled by :func:`get_system_prompt` from: - :data:`ASYNC_NOTIFICATIONS` — how to triage `[Async tasks update]` signals from async sub-agents -:data:`RESEARCHER_INSTRUCTIONS` is the research-agent sub-agent prompt, -loaded via ``_build_prompt_refs`` in ``EvoScientist.py``. +Built-in sub-agent prompts live in ``EvoScientist/subagents/*.yaml``. Style notes ----------- @@ -346,56 +345,6 @@ For EACH task in the batch, independently: It is fine to fetch one task and defer another from the same batch. """ -# ============================================================================= -# Sub-agent research instructions -# ============================================================================= - -RESEARCHER_INSTRUCTIONS = """You are a research assistant. Today's date is {date}. - -## Task -Use tools to gather information on the assigned topic (methods, baselines, datasets, or prior results) to support experimental planning or iteration. Prefer actionable details: datasets, metrics, code availability, and common pitfalls. Do not fabricate citations or URLs. Capture evaluation protocols (splits, metrics, calibration) and known failure modes. - -## Available Tools -- `think_tool` — Reflect on findings and plan next steps -- `read_file` — Read skill instructions when a skill matches the task (paths shown in your available skills listing) -- Optionally, some web search tools to find information online. - -**CRITICAL:** Use `think_tool` after each search - -## Research Strategy -1. Read the question carefully -2. Start with broad searches -3. After each search, reflect: Do I have enough? What's missing? -4. Narrow searches to fill gaps -5. Stop when you can answer confidently - -## Hard Limits -- Simple queries: 2-3 searches maximum -- Complex queries: up to 5 searches maximum -- Stop after 5 searches regardless - -## Stop When -- You can answer comprehensively -- You have 3+ relevant sources -- Last 2 searches returned similar information - -## Response Format -Structure findings with clear headings and cite sources inline: - -``` -## Key Findings - -Finding one with context [1]. Another insight [2]. - -## Recommended Next Experiments -- One actionable experiment suggestion with motivation and expected outcome. - -### Sources -[1] Title: URL -[2] Title: URL -``` -""" - # ============================================================================= # Combined exports # ============================================================================= @@ -414,10 +363,10 @@ def get_system_prompt() -> str: 6. :data:`DELEGATION_STRATEGY` 7. :data:`ASYNC_NOTIFICATIONS` - The current date is injected per-turn by - :class:`EvoScientist.middleware.EvoMemoryMiddleware` (piggy-backing on its - existing ```` injection), so the static prefix here remains - byte-stable across midnight rollover and across long-running daemons. + Runtime context is injected per-turn by + :class:`EvoScientist.middleware.RuntimeContextMiddleware`, so the static + prefix here remains byte-stable across midnight rollover and across + long-running daemons. Returns: Combined static system prompt string. diff --git a/EvoScientist/subagents/_factory.py b/EvoScientist/subagents/_factory.py index 12714fd..1c659f5 100644 --- a/EvoScientist/subagents/_factory.py +++ b/EvoScientist/subagents/_factory.py @@ -40,7 +40,6 @@ def build_async_subagent_graph(name: str) -> Any: from EvoScientist.config import apply_config_to_env, get_effective_config from EvoScientist.EvoScientist import ( SUBAGENTS_CONFIG, - _build_prompt_refs, _ensure_chat_model, _get_default_backend, _get_default_middleware, @@ -63,7 +62,6 @@ def build_async_subagent_graph(name: str) -> Any: specs = load_subagents( SUBAGENTS_CONFIG, tool_registry=tool_registry, - prompt_refs=_build_prompt_refs(), ) spec = next((s for s in specs if s.get("name") == name), None) if spec is None: diff --git a/EvoScientist/subagents/research.yaml b/EvoScientist/subagents/research.yaml index a895be9..05b4187 100644 --- a/EvoScientist/subagents/research.yaml +++ b/EvoScientist/subagents/research.yaml @@ -2,4 +2,48 @@ research-agent: description: "Web research for methods/baselines/datasets (one topic at a time, return actionable notes + sources)." tools: [tavily_search, think_tool] skills: ["/skills/"] - system_prompt_ref: RESEARCHER_INSTRUCTIONS + system_prompt: | + You are a research assistant. + + ## Task + Use tools to gather information on the assigned topic (methods, baselines, datasets, or prior results) to support experimental planning or iteration. Prefer actionable details: datasets, metrics, code availability, and common pitfalls. Do not fabricate citations or URLs. Capture evaluation protocols (splits, metrics, calibration) and known failure modes. + + ## Available Tools + - `think_tool` - Reflect on findings and plan next steps + - `read_file` - Read skill instructions when a skill matches the task (paths shown in your available skills listing) + - Optionally, some web search tools to find information online. + + **CRITICAL:** Use `think_tool` after each search + + ## Research Strategy + 1. Read the question carefully + 2. Start with broad searches + 3. After each search, reflect: Do I have enough? What's missing? + 4. Narrow searches to fill gaps + 5. Stop when you can answer confidently + + ## Hard Limits + - Simple queries: 2-3 searches maximum + - Complex queries: up to 5 searches maximum + - Stop after 5 searches regardless + + ## Stop When + - You can answer comprehensively + - You have 3+ relevant sources + - Last 2 searches returned similar information + + ## Response Format + Structure findings with clear headings and cite sources inline: + + ``` + ## Key Findings + + Finding one with context [1]. Another insight [2]. + + ## Recommended Next Experiments + - One actionable experiment suggestion with motivation and expected outcome. + + ### Sources + [1] Title: URL + [2] Title: URL + ``` diff --git a/EvoScientist/utils.py b/EvoScientist/utils.py index 60cd7bd..520a2fa 100644 --- a/EvoScientist/utils.py +++ b/EvoScientist/utils.py @@ -137,7 +137,8 @@ def load_subagents( research-agent: description: "..." tools: [tavily_search, think_tool] - system_prompt_ref: RESEARCHER_INSTRUCTIONS + system_prompt: | + ... """ prompt_refs = prompt_refs or {} diff --git a/tests/test_async_subagent_factory.py b/tests/test_async_subagent_factory.py index 62d23ea..6f88f36 100644 --- a/tests/test_async_subagent_factory.py +++ b/tests/test_async_subagent_factory.py @@ -17,7 +17,6 @@ from unittest.mock import MagicMock, patch @patch("EvoScientist.EvoScientist._get_default_middleware", return_value=[]) @patch("EvoScientist.EvoScientist._get_default_backend") @patch("EvoScientist.EvoScientist._ensure_chat_model") -@patch("EvoScientist.EvoScientist._build_prompt_refs", return_value={}) @patch("EvoScientist.utils.load_subagents") @patch("EvoScientist.config.apply_config_to_env") @patch("EvoScientist.config.get_effective_config") @@ -25,7 +24,6 @@ def test_factory_requests_async_safe_middleware( mock_get_cfg, mock_apply_env, mock_load_subs, - mock_prompt_refs, mock_chat, mock_backend, mock_get_mw, diff --git a/tests/test_prompts.py b/tests/test_prompts.py index 66a372c..d8c507b 100644 --- a/tests/test_prompts.py +++ b/tests/test_prompts.py @@ -63,16 +63,15 @@ class TestGetSystemPrompt: assert 0 <= idx_identity < idx_workflow < idx_delegation def test_does_not_contain_static_date(self): - """Date is injected per-turn by EvoMemoryMiddleware, not baked into static prompt. + """Date is injected per-turn by runtime context, not baked into static prompt. Static prompt must stay byte-stable across midnight so the cache prefix - survives. See EvoMemoryMiddleware.modify_request for runtime injection. + survives. See RuntimeContextMiddleware for runtime injection. """ import re result = get_system_prompt() - # No literal "Today's date is YYYY-MM-DD." in the static prompt. - assert not re.search(r"Today's date is \d{4}-\d{2}-\d{2}", result) + assert not re.search(r"Current date: \d{4}-\d{2}-\d{2}", result) def test_mentions_skill_manager_for_discovery(self): """Agent must know it can browse/install skills from the EvoSkills catalog.""" diff --git a/tests/test_runtime_context_middleware.py b/tests/test_runtime_context_middleware.py new file mode 100644 index 0000000..7cb560d --- /dev/null +++ b/tests/test_runtime_context_middleware.py @@ -0,0 +1,113 @@ +from __future__ import annotations + +from datetime import UTC, datetime +from types import SimpleNamespace +from unittest.mock import MagicMock, patch + +from langchain_core.messages import SystemMessage + +from EvoScientist.middleware.runtime_context import ( + RuntimeContextMiddleware, + create_runtime_context_middleware, +) + + +def _request(): + request = SimpleNamespace( + state={}, + runtime=object(), + system_message=SystemMessage(content="base system"), + ) + request.override = lambda **kwargs: SimpleNamespace( + **{ + "state": request.state, + "runtime": request.runtime, + "system_message": kwargs.get("system_message", request.system_message), + } + ) + return request + + +def _system_text(modified) -> str: + system_message = modified.system_message + assert system_message is not None + return str(system_message.content) + + +def _mock_config(): + cfg = MagicMock() + cfg.enable_ask_user = False + cfg.auto_mode = False + cfg.auto_approve = False + cfg.model_fallbacks = None + cfg.code_interpreter_timeout = 60 + cfg.code_interpreter_max_result_chars = 6000 + return cfg + + +def test_runtime_context_injects_current_date_and_timezone(): + middleware = RuntimeContextMiddleware( + now_fn=lambda: datetime(2026, 6, 2, 12, 0, tzinfo=UTC) + ) + + modified = middleware.modify_request(_request()) + system_text = _system_text(modified) + + assert "" in system_text + assert "Current date: 2026-06-02" in system_text + assert "Local timezone: UTC (UTC+00:00)" in system_text + assert "today" in system_text + + +@patch( + "EvoScientist.middleware.create_tool_selector_middleware", + return_value=[MagicMock(), MagicMock()], +) +@patch("EvoScientist.EvoScientist._ensure_chat_model") +@patch("EvoScientist.EvoScientist._ensure_config") +def test_default_middleware_includes_runtime_context( + mock_config, mock_model, mock_tool_selector +): + mock_config.return_value = _mock_config() + mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000}) + + from EvoScientist.EvoScientist import _get_default_middleware + + middleware = _get_default_middleware() + + assert any(isinstance(m, RuntimeContextMiddleware) for m in middleware) + + +@patch( + "EvoScientist.middleware.create_tool_selector_middleware", + return_value=[MagicMock(), MagicMock()], +) +@patch("EvoScientist.EvoScientist._ensure_chat_model") +@patch("EvoScientist.EvoScientist._ensure_config") +def test_async_subagent_middleware_includes_runtime_context( + mock_config, mock_model, mock_tool_selector +): + mock_config.return_value = _mock_config() + mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000}) + + from EvoScientist.EvoScientist import _get_default_middleware + + middleware = _get_default_middleware(for_async_subagent=True) + + assert any(isinstance(m, RuntimeContextMiddleware) for m in middleware) + + +@patch("EvoScientist.EvoScientist._ensure_chat_model") +def test_configured_subagent_middleware_includes_runtime_context(mock_model): + mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000}) + + from EvoScientist.EvoScientist import _inject_subagent_middleware + + subs = [{"name": "test-agent"}] + _inject_subagent_middleware(subs) + + assert any(isinstance(m, RuntimeContextMiddleware) for m in subs[0]["middleware"]) + + +def test_runtime_context_factory_returns_middleware(): + assert isinstance(create_runtime_context_middleware(), RuntimeContextMiddleware)