Add runtime context middleware (#255)

This commit is contained in:
dinos
2026-06-02 19:47:26 +02:00
committed by GitHub
parent 9285c6dad8
commit d348076f40
11 changed files with 262 additions and 80 deletions
+5 -13
View File
@@ -20,7 +20,6 @@ Usage:
import json
import logging
import os
from datetime import datetime
from pathlib import Path
from langchain.agents.middleware import AgentMiddleware, HumanInTheLoopMiddleware
@@ -28,7 +27,7 @@ from langchain.agents.middleware import AgentMiddleware, HumanInTheLoopMiddlewar
from . import paths as _paths_mod
from .config import apply_config_to_env, get_effective_config
from .paths import set_active_workspace, set_workspace_root
from .prompts import RESEARCHER_INSTRUCTIONS, get_system_prompt
from .prompts import get_system_prompt
# Suppress noisy warnings from deepagents skill loader (non-string frontmatter fields, etc.)
logging.getLogger("deepagents.middleware.skills").setLevel(logging.ERROR)
@@ -198,6 +197,7 @@ def _inject_subagent_middleware(subs: list[dict]) -> None:
ContextOverflowMapperMiddleware,
ToolErrorHandlerMiddleware,
create_context_editing_middleware,
create_runtime_context_middleware,
)
for sa in subs:
@@ -206,21 +206,13 @@ def _inject_subagent_middleware(subs: list[dict]) -> None:
# No ``model=`` — subagents share the main agent's model,
# so defer to the factory's ``_ensure_chat_model()`` fallback.
create_context_editing_middleware(),
create_runtime_context_middleware(),
ToolErrorHandlerMiddleware(),
ContextOverflowMapperMiddleware(),
]
)
def _build_prompt_refs() -> dict:
"""Build prompt references with the current date (not frozen at import)."""
return {
"RESEARCHER_INSTRUCTIONS": RESEARCHER_INSTRUCTIONS.format(
date=datetime.now().strftime("%Y-%m-%d"),
),
}
def _maybe_swap_async_subagents(subs: list, middleware: list | None = None) -> list:
"""Replace ``_async``-flagged sub-agents with ``AsyncSubAgent`` specs when enabled.
@@ -333,7 +325,6 @@ def _build_base_kwargs(base_backend, base_middleware):
subs = load_subagents(
SUBAGENTS_CONFIG,
tool_registry=tool_registry,
prompt_refs=_build_prompt_refs(),
)
_inject_subagent_middleware(subs)
subs = _maybe_swap_async_subagents(subs, base_middleware)
@@ -382,7 +373,6 @@ def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress=
subs = load_subagents(
SUBAGENTS_CONFIG,
tool_registry=registry,
prompt_refs=_build_prompt_refs(),
)
_inject_subagent_middleware(subs)
@@ -476,6 +466,7 @@ def _get_default_middleware(
create_code_interpreter_middleware,
create_context_editing_middleware,
create_memory_middleware,
create_runtime_context_middleware,
create_tool_selector_middleware,
load_fallback_chain,
)
@@ -496,6 +487,7 @@ def _get_default_middleware(
ContextOverflowMapperMiddleware(),
ToolErrorHandlerMiddleware(),
*create_tool_selector_middleware(model=model),
create_runtime_context_middleware(),
create_memory_middleware(memory_dir, workspace_dir=workspace_dir),
]
-1
View File
@@ -29,7 +29,6 @@ _EXPORTS: dict[str, tuple[str, str]] = {
"DEFAULT_MODEL": (".llm", "DEFAULT_MODEL"),
# Prompts
"get_system_prompt": (".prompts", "get_system_prompt"),
"RESEARCHER_INSTRUCTIONS": (".prompts", "RESEARCHER_INSTRUCTIONS"),
# Tools
"tavily_search": (".tools", "tavily_search"),
"think_tool": (".tools", "think_tool"),
+3
View File
@@ -23,6 +23,7 @@ from .memory import (
create_memory_middleware,
)
from .model_fallback import ModelFallbackMiddleware, load_fallback_chain
from .runtime_context import RuntimeContextMiddleware, create_runtime_context_middleware
from .tool_error_handler import ToolErrorHandlerMiddleware
from .tool_selector import create_tool_selector_middleware
from .utils import disable_thinking
@@ -37,11 +38,13 @@ __all__ = [
"EvoMemoryMiddleware",
"ModelFallbackMiddleware",
"Question",
"RuntimeContextMiddleware",
"ToolErrorHandlerMiddleware",
"compute_context_editing_trigger",
"create_code_interpreter_middleware",
"create_context_editing_middleware",
"create_memory_middleware",
"create_runtime_context_middleware",
"create_tool_selector_middleware",
"disable_thinking",
"load_fallback_chain",
@@ -0,0 +1,86 @@
"""Runtime context middleware for EvoScientist."""
from __future__ import annotations
from collections.abc import Awaitable, Callable
from datetime import datetime
from langchain.agents.middleware.types import (
AgentMiddleware,
ModelRequest,
ModelResponse,
)
RUNTIME_CONTEXT_TEMPLATE = """<runtime_context>
Current date: {date}
Local timezone: {timezone}
Use this context to resolve relative time references like today, yesterday, and
next week.
</runtime_context>"""
def _format_timezone(now: datetime) -> str:
"""Return a compact local timezone label for prompt injection."""
offset = now.utcoffset()
if offset is None:
return "local"
total_seconds = int(offset.total_seconds())
sign = "+" if total_seconds >= 0 else "-"
total_seconds = abs(total_seconds)
hours, remainder = divmod(total_seconds, 3600)
minutes = remainder // 60
offset_text = f"UTC{sign}{hours:02d}:{minutes:02d}"
name = now.tzname()
if name and name != offset_text:
return f"{name} ({offset_text})"
return offset_text
class RuntimeContextMiddleware(AgentMiddleware):
"""Inject per-turn runtime context into model calls."""
def __init__(self, *, now_fn: Callable[[], datetime] | None = None) -> None:
self._now_fn = now_fn or (lambda: datetime.now().astimezone())
def _runtime_context(self) -> str:
now = self._now_fn()
return RUNTIME_CONTEXT_TEMPLATE.format(
date=now.strftime("%Y-%m-%d"),
timezone=_format_timezone(now),
)
def modify_request(self, request: ModelRequest) -> ModelRequest:
"""Append runtime context to the system prompt."""
from deepagents.middleware._utils import append_to_system_message
new_system = append_to_system_message(
request.system_message,
self._runtime_context(),
)
return request.override(system_message=new_system)
def wrap_model_call(
self,
request: ModelRequest,
handler: Callable[[ModelRequest], ModelResponse],
) -> ModelResponse:
"""Inject runtime context before the sync model handler."""
return handler(self.modify_request(request))
async def awrap_model_call(
self,
request: ModelRequest,
handler: Callable[[ModelRequest], Awaitable[ModelResponse]],
) -> ModelResponse:
"""Inject runtime context before the async model handler."""
return await handler(self.modify_request(request))
def create_runtime_context_middleware(
*, now_fn: Callable[[], datetime] | None = None
) -> RuntimeContextMiddleware:
"""Build runtime-context middleware."""
return RuntimeContextMiddleware(now_fn=now_fn)
+5 -56
View File
@@ -13,8 +13,7 @@ The main agent's system prompt is assembled by :func:`get_system_prompt` from:
- :data:`ASYNC_NOTIFICATIONS` — how to triage `[Async tasks update]` signals
from async sub-agents
:data:`RESEARCHER_INSTRUCTIONS` is the research-agent sub-agent prompt,
loaded via ``_build_prompt_refs`` in ``EvoScientist.py``.
Built-in sub-agent prompts live in ``EvoScientist/subagents/*.yaml``.
Style notes
-----------
@@ -346,56 +345,6 @@ For EACH task in the batch, independently:
It is fine to fetch one task and defer another from the same batch.
"""
# =============================================================================
# Sub-agent research instructions
# =============================================================================
RESEARCHER_INSTRUCTIONS = """You are a research assistant. Today's date is {date}.
## Task
Use tools to gather information on the assigned topic (methods, baselines, datasets, or prior results) to support experimental planning or iteration. Prefer actionable details: datasets, metrics, code availability, and common pitfalls. Do not fabricate citations or URLs. Capture evaluation protocols (splits, metrics, calibration) and known failure modes.
## Available Tools
- `think_tool` — Reflect on findings and plan next steps
- `read_file` — Read skill instructions when a skill matches the task (paths shown in your available skills listing)
- Optionally, some web search tools to find information online.
**CRITICAL:** Use `think_tool` after each search
## Research Strategy
1. Read the question carefully
2. Start with broad searches
3. After each search, reflect: Do I have enough? What's missing?
4. Narrow searches to fill gaps
5. Stop when you can answer confidently
## Hard Limits
- Simple queries: 2-3 searches maximum
- Complex queries: up to 5 searches maximum
- Stop after 5 searches regardless
## Stop When
- You can answer comprehensively
- You have 3+ relevant sources
- Last 2 searches returned similar information
## Response Format
Structure findings with clear headings and cite sources inline:
```
## Key Findings
Finding one with context [1]. Another insight [2].
## Recommended Next Experiments
- One actionable experiment suggestion with motivation and expected outcome.
### Sources
[1] Title: URL
[2] Title: URL
```
"""
# =============================================================================
# Combined exports
# =============================================================================
@@ -414,10 +363,10 @@ def get_system_prompt() -> str:
6. :data:`DELEGATION_STRATEGY`
7. :data:`ASYNC_NOTIFICATIONS`
The current date is injected per-turn by
:class:`EvoScientist.middleware.EvoMemoryMiddleware` (piggy-backing on its
existing ``<evo_memory>`` injection), so the static prefix here remains
byte-stable across midnight rollover and across long-running daemons.
Runtime context is injected per-turn by
:class:`EvoScientist.middleware.RuntimeContextMiddleware`, so the static
prefix here remains byte-stable across midnight rollover and across
long-running daemons.
Returns:
Combined static system prompt string.
-2
View File
@@ -40,7 +40,6 @@ def build_async_subagent_graph(name: str) -> Any:
from EvoScientist.config import apply_config_to_env, get_effective_config
from EvoScientist.EvoScientist import (
SUBAGENTS_CONFIG,
_build_prompt_refs,
_ensure_chat_model,
_get_default_backend,
_get_default_middleware,
@@ -63,7 +62,6 @@ def build_async_subagent_graph(name: str) -> Any:
specs = load_subagents(
SUBAGENTS_CONFIG,
tool_registry=tool_registry,
prompt_refs=_build_prompt_refs(),
)
spec = next((s for s in specs if s.get("name") == name), None)
if spec is None:
+45 -1
View File
@@ -2,4 +2,48 @@ research-agent:
description: "Web research for methods/baselines/datasets (one topic at a time, return actionable notes + sources)."
tools: [tavily_search, think_tool]
skills: ["/skills/"]
system_prompt_ref: RESEARCHER_INSTRUCTIONS
system_prompt: |
You are a research assistant.
## Task
Use tools to gather information on the assigned topic (methods, baselines, datasets, or prior results) to support experimental planning or iteration. Prefer actionable details: datasets, metrics, code availability, and common pitfalls. Do not fabricate citations or URLs. Capture evaluation protocols (splits, metrics, calibration) and known failure modes.
## Available Tools
- `think_tool` - Reflect on findings and plan next steps
- `read_file` - Read skill instructions when a skill matches the task (paths shown in your available skills listing)
- Optionally, some web search tools to find information online.
**CRITICAL:** Use `think_tool` after each search
## Research Strategy
1. Read the question carefully
2. Start with broad searches
3. After each search, reflect: Do I have enough? What's missing?
4. Narrow searches to fill gaps
5. Stop when you can answer confidently
## Hard Limits
- Simple queries: 2-3 searches maximum
- Complex queries: up to 5 searches maximum
- Stop after 5 searches regardless
## Stop When
- You can answer comprehensively
- You have 3+ relevant sources
- Last 2 searches returned similar information
## Response Format
Structure findings with clear headings and cite sources inline:
```
## Key Findings
Finding one with context [1]. Another insight [2].
## Recommended Next Experiments
- One actionable experiment suggestion with motivation and expected outcome.
### Sources
[1] Title: URL
[2] Title: URL
```
+2 -1
View File
@@ -137,7 +137,8 @@ def load_subagents(
research-agent:
description: "..."
tools: [tavily_search, think_tool]
system_prompt_ref: RESEARCHER_INSTRUCTIONS
system_prompt: |
...
"""
prompt_refs = prompt_refs or {}
-2
View File
@@ -17,7 +17,6 @@ from unittest.mock import MagicMock, patch
@patch("EvoScientist.EvoScientist._get_default_middleware", return_value=[])
@patch("EvoScientist.EvoScientist._get_default_backend")
@patch("EvoScientist.EvoScientist._ensure_chat_model")
@patch("EvoScientist.EvoScientist._build_prompt_refs", return_value={})
@patch("EvoScientist.utils.load_subagents")
@patch("EvoScientist.config.apply_config_to_env")
@patch("EvoScientist.config.get_effective_config")
@@ -25,7 +24,6 @@ def test_factory_requests_async_safe_middleware(
mock_get_cfg,
mock_apply_env,
mock_load_subs,
mock_prompt_refs,
mock_chat,
mock_backend,
mock_get_mw,
+3 -4
View File
@@ -63,16 +63,15 @@ class TestGetSystemPrompt:
assert 0 <= idx_identity < idx_workflow < idx_delegation
def test_does_not_contain_static_date(self):
"""Date is injected per-turn by EvoMemoryMiddleware, not baked into static prompt.
"""Date is injected per-turn by runtime context, not baked into static prompt.
Static prompt must stay byte-stable across midnight so the cache prefix
survives. See EvoMemoryMiddleware.modify_request for runtime injection.
survives. See RuntimeContextMiddleware for runtime injection.
"""
import re
result = get_system_prompt()
# No literal "Today's date is YYYY-MM-DD." in the static prompt.
assert not re.search(r"Today's date is \d{4}-\d{2}-\d{2}", result)
assert not re.search(r"Current date: \d{4}-\d{2}-\d{2}", result)
def test_mentions_skill_manager_for_discovery(self):
"""Agent must know it can browse/install skills from the EvoSkills catalog."""
+113
View File
@@ -0,0 +1,113 @@
from __future__ import annotations
from datetime import UTC, datetime
from types import SimpleNamespace
from unittest.mock import MagicMock, patch
from langchain_core.messages import SystemMessage
from EvoScientist.middleware.runtime_context import (
RuntimeContextMiddleware,
create_runtime_context_middleware,
)
def _request():
request = SimpleNamespace(
state={},
runtime=object(),
system_message=SystemMessage(content="base system"),
)
request.override = lambda **kwargs: SimpleNamespace(
**{
"state": request.state,
"runtime": request.runtime,
"system_message": kwargs.get("system_message", request.system_message),
}
)
return request
def _system_text(modified) -> str:
system_message = modified.system_message
assert system_message is not None
return str(system_message.content)
def _mock_config():
cfg = MagicMock()
cfg.enable_ask_user = False
cfg.auto_mode = False
cfg.auto_approve = False
cfg.model_fallbacks = None
cfg.code_interpreter_timeout = 60
cfg.code_interpreter_max_result_chars = 6000
return cfg
def test_runtime_context_injects_current_date_and_timezone():
middleware = RuntimeContextMiddleware(
now_fn=lambda: datetime(2026, 6, 2, 12, 0, tzinfo=UTC)
)
modified = middleware.modify_request(_request())
system_text = _system_text(modified)
assert "<runtime_context>" in system_text
assert "Current date: 2026-06-02" in system_text
assert "Local timezone: UTC (UTC+00:00)" in system_text
assert "today" in system_text
@patch(
"EvoScientist.middleware.create_tool_selector_middleware",
return_value=[MagicMock(), MagicMock()],
)
@patch("EvoScientist.EvoScientist._ensure_chat_model")
@patch("EvoScientist.EvoScientist._ensure_config")
def test_default_middleware_includes_runtime_context(
mock_config, mock_model, mock_tool_selector
):
mock_config.return_value = _mock_config()
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
from EvoScientist.EvoScientist import _get_default_middleware
middleware = _get_default_middleware()
assert any(isinstance(m, RuntimeContextMiddleware) for m in middleware)
@patch(
"EvoScientist.middleware.create_tool_selector_middleware",
return_value=[MagicMock(), MagicMock()],
)
@patch("EvoScientist.EvoScientist._ensure_chat_model")
@patch("EvoScientist.EvoScientist._ensure_config")
def test_async_subagent_middleware_includes_runtime_context(
mock_config, mock_model, mock_tool_selector
):
mock_config.return_value = _mock_config()
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
from EvoScientist.EvoScientist import _get_default_middleware
middleware = _get_default_middleware(for_async_subagent=True)
assert any(isinstance(m, RuntimeContextMiddleware) for m in middleware)
@patch("EvoScientist.EvoScientist._ensure_chat_model")
def test_configured_subagent_middleware_includes_runtime_context(mock_model):
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
from EvoScientist.EvoScientist import _inject_subagent_middleware
subs = [{"name": "test-agent"}]
_inject_subagent_middleware(subs)
assert any(isinstance(m, RuntimeContextMiddleware) for m in subs[0]["middleware"])
def test_runtime_context_factory_returns_middleware():
assert isinstance(create_runtime_context_middleware(), RuntimeContextMiddleware)