Add runtime context middleware (#255)
This commit is contained in:
@@ -20,7 +20,6 @@ Usage:
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
from datetime import datetime
|
||||
from pathlib import Path
|
||||
|
||||
from langchain.agents.middleware import AgentMiddleware, HumanInTheLoopMiddleware
|
||||
@@ -28,7 +27,7 @@ from langchain.agents.middleware import AgentMiddleware, HumanInTheLoopMiddlewar
|
||||
from . import paths as _paths_mod
|
||||
from .config import apply_config_to_env, get_effective_config
|
||||
from .paths import set_active_workspace, set_workspace_root
|
||||
from .prompts import RESEARCHER_INSTRUCTIONS, get_system_prompt
|
||||
from .prompts import get_system_prompt
|
||||
|
||||
# Suppress noisy warnings from deepagents skill loader (non-string frontmatter fields, etc.)
|
||||
logging.getLogger("deepagents.middleware.skills").setLevel(logging.ERROR)
|
||||
@@ -198,6 +197,7 @@ def _inject_subagent_middleware(subs: list[dict]) -> None:
|
||||
ContextOverflowMapperMiddleware,
|
||||
ToolErrorHandlerMiddleware,
|
||||
create_context_editing_middleware,
|
||||
create_runtime_context_middleware,
|
||||
)
|
||||
|
||||
for sa in subs:
|
||||
@@ -206,21 +206,13 @@ def _inject_subagent_middleware(subs: list[dict]) -> None:
|
||||
# No ``model=`` — subagents share the main agent's model,
|
||||
# so defer to the factory's ``_ensure_chat_model()`` fallback.
|
||||
create_context_editing_middleware(),
|
||||
create_runtime_context_middleware(),
|
||||
ToolErrorHandlerMiddleware(),
|
||||
ContextOverflowMapperMiddleware(),
|
||||
]
|
||||
)
|
||||
|
||||
|
||||
def _build_prompt_refs() -> dict:
|
||||
"""Build prompt references with the current date (not frozen at import)."""
|
||||
return {
|
||||
"RESEARCHER_INSTRUCTIONS": RESEARCHER_INSTRUCTIONS.format(
|
||||
date=datetime.now().strftime("%Y-%m-%d"),
|
||||
),
|
||||
}
|
||||
|
||||
|
||||
def _maybe_swap_async_subagents(subs: list, middleware: list | None = None) -> list:
|
||||
"""Replace ``_async``-flagged sub-agents with ``AsyncSubAgent`` specs when enabled.
|
||||
|
||||
@@ -333,7 +325,6 @@ def _build_base_kwargs(base_backend, base_middleware):
|
||||
subs = load_subagents(
|
||||
SUBAGENTS_CONFIG,
|
||||
tool_registry=tool_registry,
|
||||
prompt_refs=_build_prompt_refs(),
|
||||
)
|
||||
_inject_subagent_middleware(subs)
|
||||
subs = _maybe_swap_async_subagents(subs, base_middleware)
|
||||
@@ -382,7 +373,6 @@ def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress=
|
||||
subs = load_subagents(
|
||||
SUBAGENTS_CONFIG,
|
||||
tool_registry=registry,
|
||||
prompt_refs=_build_prompt_refs(),
|
||||
)
|
||||
|
||||
_inject_subagent_middleware(subs)
|
||||
@@ -476,6 +466,7 @@ def _get_default_middleware(
|
||||
create_code_interpreter_middleware,
|
||||
create_context_editing_middleware,
|
||||
create_memory_middleware,
|
||||
create_runtime_context_middleware,
|
||||
create_tool_selector_middleware,
|
||||
load_fallback_chain,
|
||||
)
|
||||
@@ -496,6 +487,7 @@ def _get_default_middleware(
|
||||
ContextOverflowMapperMiddleware(),
|
||||
ToolErrorHandlerMiddleware(),
|
||||
*create_tool_selector_middleware(model=model),
|
||||
create_runtime_context_middleware(),
|
||||
create_memory_middleware(memory_dir, workspace_dir=workspace_dir),
|
||||
]
|
||||
|
||||
|
||||
@@ -29,7 +29,6 @@ _EXPORTS: dict[str, tuple[str, str]] = {
|
||||
"DEFAULT_MODEL": (".llm", "DEFAULT_MODEL"),
|
||||
# Prompts
|
||||
"get_system_prompt": (".prompts", "get_system_prompt"),
|
||||
"RESEARCHER_INSTRUCTIONS": (".prompts", "RESEARCHER_INSTRUCTIONS"),
|
||||
# Tools
|
||||
"tavily_search": (".tools", "tavily_search"),
|
||||
"think_tool": (".tools", "think_tool"),
|
||||
|
||||
@@ -23,6 +23,7 @@ from .memory import (
|
||||
create_memory_middleware,
|
||||
)
|
||||
from .model_fallback import ModelFallbackMiddleware, load_fallback_chain
|
||||
from .runtime_context import RuntimeContextMiddleware, create_runtime_context_middleware
|
||||
from .tool_error_handler import ToolErrorHandlerMiddleware
|
||||
from .tool_selector import create_tool_selector_middleware
|
||||
from .utils import disable_thinking
|
||||
@@ -37,11 +38,13 @@ __all__ = [
|
||||
"EvoMemoryMiddleware",
|
||||
"ModelFallbackMiddleware",
|
||||
"Question",
|
||||
"RuntimeContextMiddleware",
|
||||
"ToolErrorHandlerMiddleware",
|
||||
"compute_context_editing_trigger",
|
||||
"create_code_interpreter_middleware",
|
||||
"create_context_editing_middleware",
|
||||
"create_memory_middleware",
|
||||
"create_runtime_context_middleware",
|
||||
"create_tool_selector_middleware",
|
||||
"disable_thinking",
|
||||
"load_fallback_chain",
|
||||
|
||||
@@ -0,0 +1,86 @@
|
||||
"""Runtime context middleware for EvoScientist."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Awaitable, Callable
|
||||
from datetime import datetime
|
||||
|
||||
from langchain.agents.middleware.types import (
|
||||
AgentMiddleware,
|
||||
ModelRequest,
|
||||
ModelResponse,
|
||||
)
|
||||
|
||||
RUNTIME_CONTEXT_TEMPLATE = """<runtime_context>
|
||||
Current date: {date}
|
||||
Local timezone: {timezone}
|
||||
|
||||
Use this context to resolve relative time references like today, yesterday, and
|
||||
next week.
|
||||
</runtime_context>"""
|
||||
|
||||
|
||||
def _format_timezone(now: datetime) -> str:
|
||||
"""Return a compact local timezone label for prompt injection."""
|
||||
offset = now.utcoffset()
|
||||
if offset is None:
|
||||
return "local"
|
||||
|
||||
total_seconds = int(offset.total_seconds())
|
||||
sign = "+" if total_seconds >= 0 else "-"
|
||||
total_seconds = abs(total_seconds)
|
||||
hours, remainder = divmod(total_seconds, 3600)
|
||||
minutes = remainder // 60
|
||||
offset_text = f"UTC{sign}{hours:02d}:{minutes:02d}"
|
||||
|
||||
name = now.tzname()
|
||||
if name and name != offset_text:
|
||||
return f"{name} ({offset_text})"
|
||||
return offset_text
|
||||
|
||||
|
||||
class RuntimeContextMiddleware(AgentMiddleware):
|
||||
"""Inject per-turn runtime context into model calls."""
|
||||
|
||||
def __init__(self, *, now_fn: Callable[[], datetime] | None = None) -> None:
|
||||
self._now_fn = now_fn or (lambda: datetime.now().astimezone())
|
||||
|
||||
def _runtime_context(self) -> str:
|
||||
now = self._now_fn()
|
||||
return RUNTIME_CONTEXT_TEMPLATE.format(
|
||||
date=now.strftime("%Y-%m-%d"),
|
||||
timezone=_format_timezone(now),
|
||||
)
|
||||
|
||||
def modify_request(self, request: ModelRequest) -> ModelRequest:
|
||||
"""Append runtime context to the system prompt."""
|
||||
from deepagents.middleware._utils import append_to_system_message
|
||||
|
||||
new_system = append_to_system_message(
|
||||
request.system_message,
|
||||
self._runtime_context(),
|
||||
)
|
||||
return request.override(system_message=new_system)
|
||||
|
||||
def wrap_model_call(
|
||||
self,
|
||||
request: ModelRequest,
|
||||
handler: Callable[[ModelRequest], ModelResponse],
|
||||
) -> ModelResponse:
|
||||
"""Inject runtime context before the sync model handler."""
|
||||
return handler(self.modify_request(request))
|
||||
|
||||
async def awrap_model_call(
|
||||
self,
|
||||
request: ModelRequest,
|
||||
handler: Callable[[ModelRequest], Awaitable[ModelResponse]],
|
||||
) -> ModelResponse:
|
||||
"""Inject runtime context before the async model handler."""
|
||||
return await handler(self.modify_request(request))
|
||||
|
||||
|
||||
def create_runtime_context_middleware(
|
||||
*, now_fn: Callable[[], datetime] | None = None
|
||||
) -> RuntimeContextMiddleware:
|
||||
"""Build runtime-context middleware."""
|
||||
return RuntimeContextMiddleware(now_fn=now_fn)
|
||||
+5
-56
@@ -13,8 +13,7 @@ The main agent's system prompt is assembled by :func:`get_system_prompt` from:
|
||||
- :data:`ASYNC_NOTIFICATIONS` — how to triage `[Async tasks update]` signals
|
||||
from async sub-agents
|
||||
|
||||
:data:`RESEARCHER_INSTRUCTIONS` is the research-agent sub-agent prompt,
|
||||
loaded via ``_build_prompt_refs`` in ``EvoScientist.py``.
|
||||
Built-in sub-agent prompts live in ``EvoScientist/subagents/*.yaml``.
|
||||
|
||||
Style notes
|
||||
-----------
|
||||
@@ -346,56 +345,6 @@ For EACH task in the batch, independently:
|
||||
It is fine to fetch one task and defer another from the same batch.
|
||||
"""
|
||||
|
||||
# =============================================================================
|
||||
# Sub-agent research instructions
|
||||
# =============================================================================
|
||||
|
||||
RESEARCHER_INSTRUCTIONS = """You are a research assistant. Today's date is {date}.
|
||||
|
||||
## Task
|
||||
Use tools to gather information on the assigned topic (methods, baselines, datasets, or prior results) to support experimental planning or iteration. Prefer actionable details: datasets, metrics, code availability, and common pitfalls. Do not fabricate citations or URLs. Capture evaluation protocols (splits, metrics, calibration) and known failure modes.
|
||||
|
||||
## Available Tools
|
||||
- `think_tool` — Reflect on findings and plan next steps
|
||||
- `read_file` — Read skill instructions when a skill matches the task (paths shown in your available skills listing)
|
||||
- Optionally, some web search tools to find information online.
|
||||
|
||||
**CRITICAL:** Use `think_tool` after each search
|
||||
|
||||
## Research Strategy
|
||||
1. Read the question carefully
|
||||
2. Start with broad searches
|
||||
3. After each search, reflect: Do I have enough? What's missing?
|
||||
4. Narrow searches to fill gaps
|
||||
5. Stop when you can answer confidently
|
||||
|
||||
## Hard Limits
|
||||
- Simple queries: 2-3 searches maximum
|
||||
- Complex queries: up to 5 searches maximum
|
||||
- Stop after 5 searches regardless
|
||||
|
||||
## Stop When
|
||||
- You can answer comprehensively
|
||||
- You have 3+ relevant sources
|
||||
- Last 2 searches returned similar information
|
||||
|
||||
## Response Format
|
||||
Structure findings with clear headings and cite sources inline:
|
||||
|
||||
```
|
||||
## Key Findings
|
||||
|
||||
Finding one with context [1]. Another insight [2].
|
||||
|
||||
## Recommended Next Experiments
|
||||
- One actionable experiment suggestion with motivation and expected outcome.
|
||||
|
||||
### Sources
|
||||
[1] Title: URL
|
||||
[2] Title: URL
|
||||
```
|
||||
"""
|
||||
|
||||
# =============================================================================
|
||||
# Combined exports
|
||||
# =============================================================================
|
||||
@@ -414,10 +363,10 @@ def get_system_prompt() -> str:
|
||||
6. :data:`DELEGATION_STRATEGY`
|
||||
7. :data:`ASYNC_NOTIFICATIONS`
|
||||
|
||||
The current date is injected per-turn by
|
||||
:class:`EvoScientist.middleware.EvoMemoryMiddleware` (piggy-backing on its
|
||||
existing ``<evo_memory>`` injection), so the static prefix here remains
|
||||
byte-stable across midnight rollover and across long-running daemons.
|
||||
Runtime context is injected per-turn by
|
||||
:class:`EvoScientist.middleware.RuntimeContextMiddleware`, so the static
|
||||
prefix here remains byte-stable across midnight rollover and across
|
||||
long-running daemons.
|
||||
|
||||
Returns:
|
||||
Combined static system prompt string.
|
||||
|
||||
@@ -40,7 +40,6 @@ def build_async_subagent_graph(name: str) -> Any:
|
||||
from EvoScientist.config import apply_config_to_env, get_effective_config
|
||||
from EvoScientist.EvoScientist import (
|
||||
SUBAGENTS_CONFIG,
|
||||
_build_prompt_refs,
|
||||
_ensure_chat_model,
|
||||
_get_default_backend,
|
||||
_get_default_middleware,
|
||||
@@ -63,7 +62,6 @@ def build_async_subagent_graph(name: str) -> Any:
|
||||
specs = load_subagents(
|
||||
SUBAGENTS_CONFIG,
|
||||
tool_registry=tool_registry,
|
||||
prompt_refs=_build_prompt_refs(),
|
||||
)
|
||||
spec = next((s for s in specs if s.get("name") == name), None)
|
||||
if spec is None:
|
||||
|
||||
@@ -2,4 +2,48 @@ research-agent:
|
||||
description: "Web research for methods/baselines/datasets (one topic at a time, return actionable notes + sources)."
|
||||
tools: [tavily_search, think_tool]
|
||||
skills: ["/skills/"]
|
||||
system_prompt_ref: RESEARCHER_INSTRUCTIONS
|
||||
system_prompt: |
|
||||
You are a research assistant.
|
||||
|
||||
## Task
|
||||
Use tools to gather information on the assigned topic (methods, baselines, datasets, or prior results) to support experimental planning or iteration. Prefer actionable details: datasets, metrics, code availability, and common pitfalls. Do not fabricate citations or URLs. Capture evaluation protocols (splits, metrics, calibration) and known failure modes.
|
||||
|
||||
## Available Tools
|
||||
- `think_tool` - Reflect on findings and plan next steps
|
||||
- `read_file` - Read skill instructions when a skill matches the task (paths shown in your available skills listing)
|
||||
- Optionally, some web search tools to find information online.
|
||||
|
||||
**CRITICAL:** Use `think_tool` after each search
|
||||
|
||||
## Research Strategy
|
||||
1. Read the question carefully
|
||||
2. Start with broad searches
|
||||
3. After each search, reflect: Do I have enough? What's missing?
|
||||
4. Narrow searches to fill gaps
|
||||
5. Stop when you can answer confidently
|
||||
|
||||
## Hard Limits
|
||||
- Simple queries: 2-3 searches maximum
|
||||
- Complex queries: up to 5 searches maximum
|
||||
- Stop after 5 searches regardless
|
||||
|
||||
## Stop When
|
||||
- You can answer comprehensively
|
||||
- You have 3+ relevant sources
|
||||
- Last 2 searches returned similar information
|
||||
|
||||
## Response Format
|
||||
Structure findings with clear headings and cite sources inline:
|
||||
|
||||
```
|
||||
## Key Findings
|
||||
|
||||
Finding one with context [1]. Another insight [2].
|
||||
|
||||
## Recommended Next Experiments
|
||||
- One actionable experiment suggestion with motivation and expected outcome.
|
||||
|
||||
### Sources
|
||||
[1] Title: URL
|
||||
[2] Title: URL
|
||||
```
|
||||
|
||||
@@ -137,7 +137,8 @@ def load_subagents(
|
||||
research-agent:
|
||||
description: "..."
|
||||
tools: [tavily_search, think_tool]
|
||||
system_prompt_ref: RESEARCHER_INSTRUCTIONS
|
||||
system_prompt: |
|
||||
...
|
||||
"""
|
||||
prompt_refs = prompt_refs or {}
|
||||
|
||||
|
||||
@@ -17,7 +17,6 @@ from unittest.mock import MagicMock, patch
|
||||
@patch("EvoScientist.EvoScientist._get_default_middleware", return_value=[])
|
||||
@patch("EvoScientist.EvoScientist._get_default_backend")
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
@patch("EvoScientist.EvoScientist._build_prompt_refs", return_value={})
|
||||
@patch("EvoScientist.utils.load_subagents")
|
||||
@patch("EvoScientist.config.apply_config_to_env")
|
||||
@patch("EvoScientist.config.get_effective_config")
|
||||
@@ -25,7 +24,6 @@ def test_factory_requests_async_safe_middleware(
|
||||
mock_get_cfg,
|
||||
mock_apply_env,
|
||||
mock_load_subs,
|
||||
mock_prompt_refs,
|
||||
mock_chat,
|
||||
mock_backend,
|
||||
mock_get_mw,
|
||||
|
||||
@@ -63,16 +63,15 @@ class TestGetSystemPrompt:
|
||||
assert 0 <= idx_identity < idx_workflow < idx_delegation
|
||||
|
||||
def test_does_not_contain_static_date(self):
|
||||
"""Date is injected per-turn by EvoMemoryMiddleware, not baked into static prompt.
|
||||
"""Date is injected per-turn by runtime context, not baked into static prompt.
|
||||
|
||||
Static prompt must stay byte-stable across midnight so the cache prefix
|
||||
survives. See EvoMemoryMiddleware.modify_request for runtime injection.
|
||||
survives. See RuntimeContextMiddleware for runtime injection.
|
||||
"""
|
||||
import re
|
||||
|
||||
result = get_system_prompt()
|
||||
# No literal "Today's date is YYYY-MM-DD." in the static prompt.
|
||||
assert not re.search(r"Today's date is \d{4}-\d{2}-\d{2}", result)
|
||||
assert not re.search(r"Current date: \d{4}-\d{2}-\d{2}", result)
|
||||
|
||||
def test_mentions_skill_manager_for_discovery(self):
|
||||
"""Agent must know it can browse/install skills from the EvoSkills catalog."""
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
from __future__ import annotations
|
||||
|
||||
from datetime import UTC, datetime
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from langchain_core.messages import SystemMessage
|
||||
|
||||
from EvoScientist.middleware.runtime_context import (
|
||||
RuntimeContextMiddleware,
|
||||
create_runtime_context_middleware,
|
||||
)
|
||||
|
||||
|
||||
def _request():
|
||||
request = SimpleNamespace(
|
||||
state={},
|
||||
runtime=object(),
|
||||
system_message=SystemMessage(content="base system"),
|
||||
)
|
||||
request.override = lambda **kwargs: SimpleNamespace(
|
||||
**{
|
||||
"state": request.state,
|
||||
"runtime": request.runtime,
|
||||
"system_message": kwargs.get("system_message", request.system_message),
|
||||
}
|
||||
)
|
||||
return request
|
||||
|
||||
|
||||
def _system_text(modified) -> str:
|
||||
system_message = modified.system_message
|
||||
assert system_message is not None
|
||||
return str(system_message.content)
|
||||
|
||||
|
||||
def _mock_config():
|
||||
cfg = MagicMock()
|
||||
cfg.enable_ask_user = False
|
||||
cfg.auto_mode = False
|
||||
cfg.auto_approve = False
|
||||
cfg.model_fallbacks = None
|
||||
cfg.code_interpreter_timeout = 60
|
||||
cfg.code_interpreter_max_result_chars = 6000
|
||||
return cfg
|
||||
|
||||
|
||||
def test_runtime_context_injects_current_date_and_timezone():
|
||||
middleware = RuntimeContextMiddleware(
|
||||
now_fn=lambda: datetime(2026, 6, 2, 12, 0, tzinfo=UTC)
|
||||
)
|
||||
|
||||
modified = middleware.modify_request(_request())
|
||||
system_text = _system_text(modified)
|
||||
|
||||
assert "<runtime_context>" in system_text
|
||||
assert "Current date: 2026-06-02" in system_text
|
||||
assert "Local timezone: UTC (UTC+00:00)" in system_text
|
||||
assert "today" in system_text
|
||||
|
||||
|
||||
@patch(
|
||||
"EvoScientist.middleware.create_tool_selector_middleware",
|
||||
return_value=[MagicMock(), MagicMock()],
|
||||
)
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
@patch("EvoScientist.EvoScientist._ensure_config")
|
||||
def test_default_middleware_includes_runtime_context(
|
||||
mock_config, mock_model, mock_tool_selector
|
||||
):
|
||||
mock_config.return_value = _mock_config()
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
from EvoScientist.EvoScientist import _get_default_middleware
|
||||
|
||||
middleware = _get_default_middleware()
|
||||
|
||||
assert any(isinstance(m, RuntimeContextMiddleware) for m in middleware)
|
||||
|
||||
|
||||
@patch(
|
||||
"EvoScientist.middleware.create_tool_selector_middleware",
|
||||
return_value=[MagicMock(), MagicMock()],
|
||||
)
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
@patch("EvoScientist.EvoScientist._ensure_config")
|
||||
def test_async_subagent_middleware_includes_runtime_context(
|
||||
mock_config, mock_model, mock_tool_selector
|
||||
):
|
||||
mock_config.return_value = _mock_config()
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
from EvoScientist.EvoScientist import _get_default_middleware
|
||||
|
||||
middleware = _get_default_middleware(for_async_subagent=True)
|
||||
|
||||
assert any(isinstance(m, RuntimeContextMiddleware) for m in middleware)
|
||||
|
||||
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
def test_configured_subagent_middleware_includes_runtime_context(mock_model):
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
from EvoScientist.EvoScientist import _inject_subagent_middleware
|
||||
|
||||
subs = [{"name": "test-agent"}]
|
||||
_inject_subagent_middleware(subs)
|
||||
|
||||
assert any(isinstance(m, RuntimeContextMiddleware) for m in subs[0]["middleware"])
|
||||
|
||||
|
||||
def test_runtime_context_factory_returns_middleware():
|
||||
assert isinstance(create_runtime_context_middleware(), RuntimeContextMiddleware)
|
||||
Reference in New Issue
Block a user