feat(memory): add observation memory lifecycle (#259)
* feat(memory): add observation memory lifecycle Add file-backed observation memory with deterministic markdown records, structured record_observation tooling, startup indexing, and profile/observation prompt guidance. Launch post-turn and post-subagent EvoMemory workers through LangGraph dev so completed runs can update profile memory, save durable observations, and write subagent execution summaries without blocking the active agent. Wire memory middleware into the main agent, subagents, async graphs, TUI status reporting, worker activity accounting, and observation-aware research prompts, with regression coverage for storage, lifecycle scheduling, graph registration, status display, and stream reset behavior. * fix(cli): sync background agent server on resume Resume flows now need to keep the LangGraph dev background server aligned with the active workspace even when async subagents are disabled. EvoMemory workers use that server too, so gating resume-time sync on enable_async_subagents could leave workers pinned to the launch workspace after resuming a thread from another workspace. Run workspace sync unconditionally for Rich CLI and Textual resume paths, while preserving WorkspaceMismatchError handling so failed sync aborts the resume before mutating the active thread or workspace. Propagate aborted resume callbacks through the command UI so channel-issued /resume commands do not send false success or history output. Channel slash dispatch now treats CommandManager-caught command errors as command errors and skips completion hooks for those failed commands. Add regression coverage for disabled async subagents, callback aborts, and channel command error reporting. * fix(cli): prepare serve resume workspace before adopting Load the resumed workspace agent and sync the background server as a single pre-adoption step. Restore the previous active workspace if preparation fails so serve mode keeps using the old session consistently. * fix(memory): untrack abandoned worker status watches Stop treating watcher shutdown as confirmed worker completion. Terminal worker statuses still count memory deltas, while poll failures or watcher setup failures now remove the active run without crediting partial outputs. * fix(cli): report channel command failures accurately Treat command_error as a None sentinel so empty error strings still fail, and let TUI resumes continue only on non-mismatch background-server sync failures while reporting degraded mode. * fix(stream): clear memory counters for resume streams Reset completed-memory counters for every new agent stream, including Command-based HITL and resume streams, so saved-memory indicators do not leak across turns. * docs(tools): make observation recording guidance conditional Clarify that agents should call record_observation only when the observation tool is available, preserving the existing durability and usefulness criteria. * feat(config): add controls for profile and observation memory Add config flags for profile memory, observation memory, observation writer placement, and background memory workers. Wire the controls through main agents, subagents, EvoMemory middleware, and memory lifecycle workers so observation writes can be assigned to the live agent, subagent worker, both, or neither. Keep turn memory workers profile-only and make prompts reflect the available observation read/write paths. Skip langgraph dev startup when neither async subagents nor memory workers need the background server. Add coverage for config parsing, prompt gating, middleware wiring, and worker tool availability. * test(cli): include memory defaults in serve config stubs * fix(memory): offload async worker launch blocking calls Run the langgraph-dev health check and memory-output snapshot in worker threads from the async EvoMemory launcher so it does not block the event loop. * chore(memory): harden turn worker subagent guardrail * chore(memory): refresh profile context per request * fix(memory): offload async profile file reads * fix(memory): offload async worker completion accounting
This commit is contained in:
@@ -37,6 +37,7 @@ bridge/package-lock.json
|
||||
workspace/
|
||||
skills/
|
||||
memory/
|
||||
!EvoScientist/memory/
|
||||
media/
|
||||
conversation_history/
|
||||
.deno_cache/
|
||||
|
||||
+158
-25
@@ -25,7 +25,13 @@ from pathlib import Path
|
||||
from langchain.agents.middleware import AgentMiddleware, HumanInTheLoopMiddleware
|
||||
|
||||
from . import paths as _paths_mod
|
||||
from .config import apply_config_to_env, get_effective_config
|
||||
from .config import (
|
||||
MemoryControls,
|
||||
MemoryObservationTarget,
|
||||
apply_config_to_env,
|
||||
get_effective_config,
|
||||
)
|
||||
from .memory import MemorySourceType
|
||||
from .paths import set_active_workspace, set_workspace_root
|
||||
from .prompts import get_system_prompt
|
||||
|
||||
@@ -38,6 +44,7 @@ logging.getLogger("deepagents.middleware.skills").setLevel(logging.ERROR)
|
||||
|
||||
SUBAGENTS_CONFIG = Path(__file__).parent / "subagents"
|
||||
SKILLS_DIR = str(Path(__file__).parent / "skills")
|
||||
DEFAULT_SKILL_SOURCES = ("/skills/",)
|
||||
|
||||
# =============================================================================
|
||||
# Lazy state — initialized on first use, not at import time
|
||||
@@ -186,7 +193,21 @@ def _load_mcp_tools_cached(on_progress=None) -> dict[str, list]:
|
||||
# =============================================================================
|
||||
|
||||
|
||||
def _inject_subagent_middleware(subs: list[dict]) -> None:
|
||||
def _configured_system_prompt(cfg) -> str:
|
||||
memory_controls = MemoryControls.from_config(cfg)
|
||||
return get_system_prompt(
|
||||
enable_observation_memory=memory_controls.observations_enabled,
|
||||
enable_observation_writes=memory_controls.observation_tool_enabled(
|
||||
MemoryObservationTarget.AGENT
|
||||
),
|
||||
)
|
||||
|
||||
|
||||
def _inject_subagent_middleware(
|
||||
subs: list[dict],
|
||||
*,
|
||||
workspace_dir: str | Path | None = None,
|
||||
) -> None:
|
||||
"""Ensure every subagent gets error handling and context management middleware.
|
||||
|
||||
Without this, subagent tool errors are caught by LangGraph's default
|
||||
@@ -195,22 +216,69 @@ def _inject_subagent_middleware(subs: list[dict]) -> None:
|
||||
"""
|
||||
from .middleware import (
|
||||
ContextOverflowMapperMiddleware,
|
||||
MemoryLifecycleRole,
|
||||
ToolErrorHandlerMiddleware,
|
||||
create_context_editing_middleware,
|
||||
create_memory_lifecycle_middleware,
|
||||
create_memory_middleware,
|
||||
create_runtime_context_middleware,
|
||||
)
|
||||
|
||||
cfg = _ensure_config()
|
||||
memory_controls = MemoryControls.from_config(cfg)
|
||||
memory_dir = str(_paths_mod.MEMORIES_DIR)
|
||||
for sa in subs:
|
||||
sa.setdefault("middleware", []).extend(
|
||||
[
|
||||
# No ``model=`` — subagents share the main agent's model,
|
||||
# so defer to the factory's ``_ensure_chat_model()`` fallback.
|
||||
create_context_editing_middleware(),
|
||||
create_runtime_context_middleware(),
|
||||
ToolErrorHandlerMiddleware(),
|
||||
ContextOverflowMapperMiddleware(),
|
||||
]
|
||||
name = str(sa.get("name") or "sub-agent")
|
||||
source_type = MemorySourceType.SUBAGENT
|
||||
memory_middleware = create_memory_middleware(
|
||||
memory_dir,
|
||||
workspace_dir=workspace_dir,
|
||||
source_type=source_type,
|
||||
source_agent=name,
|
||||
enable_profile_memory=memory_controls.profile_enabled,
|
||||
enable_observation_memory=memory_controls.observations_enabled,
|
||||
enable_observation_tool=memory_controls.observation_tool_enabled(
|
||||
MemoryObservationTarget.AGENT
|
||||
),
|
||||
)
|
||||
middleware = [
|
||||
# No ``model=`` — subagents share the main agent's model,
|
||||
# so defer to the factory's ``_ensure_chat_model()`` fallback.
|
||||
create_context_editing_middleware(),
|
||||
create_runtime_context_middleware(),
|
||||
ToolErrorHandlerMiddleware(),
|
||||
ContextOverflowMapperMiddleware(),
|
||||
]
|
||||
if memory_controls.memory_enabled:
|
||||
middleware.append(memory_middleware)
|
||||
if memory_controls.worker_needed(MemoryObservationTarget.SUBAGENT_WORKER):
|
||||
middleware.append(
|
||||
create_memory_lifecycle_middleware(
|
||||
memory_dir,
|
||||
workspace_dir=workspace_dir,
|
||||
project_id=memory_middleware.project_id,
|
||||
role=MemoryLifecycleRole.SUBAGENT,
|
||||
source_agent=name,
|
||||
)
|
||||
)
|
||||
sa.setdefault("middleware", []).extend(middleware)
|
||||
|
||||
|
||||
def _ensure_general_purpose_subagent(subs: list[dict]) -> None:
|
||||
"""Materialize DeepAgents' default subagent so our middleware wraps it."""
|
||||
from deepagents.middleware.subagents import GENERAL_PURPOSE_SUBAGENT
|
||||
|
||||
name = GENERAL_PURPOSE_SUBAGENT["name"]
|
||||
if any(sa.get("name") == name for sa in subs):
|
||||
return
|
||||
|
||||
subs.insert(
|
||||
0,
|
||||
{
|
||||
**GENERAL_PURPOSE_SUBAGENT,
|
||||
"skills": list(DEFAULT_SKILL_SOURCES),
|
||||
},
|
||||
)
|
||||
|
||||
|
||||
def _maybe_swap_async_subagents(subs: list, middleware: list | None = None) -> list:
|
||||
@@ -312,7 +380,7 @@ def _maybe_swap_async_subagents(subs: list, middleware: list | None = None) -> l
|
||||
return out
|
||||
|
||||
|
||||
def _build_base_kwargs(base_backend, base_middleware):
|
||||
def _build_base_kwargs(base_backend, base_middleware, *, workspace_dir=None):
|
||||
"""Build agent kwargs *without* MCP (fast, no subprocess spawning)."""
|
||||
from .tools import skill_manager, tavily_search, think_tool
|
||||
from .utils import load_subagents
|
||||
@@ -326,7 +394,8 @@ def _build_base_kwargs(base_backend, base_middleware):
|
||||
SUBAGENTS_CONFIG,
|
||||
tool_registry=tool_registry,
|
||||
)
|
||||
_inject_subagent_middleware(subs)
|
||||
_ensure_general_purpose_subagent(subs)
|
||||
_inject_subagent_middleware(subs, workspace_dir=workspace_dir)
|
||||
subs = _maybe_swap_async_subagents(subs, base_middleware)
|
||||
return {
|
||||
"name": "EvoScientist",
|
||||
@@ -335,12 +404,18 @@ def _build_base_kwargs(base_backend, base_middleware):
|
||||
"backend": base_backend,
|
||||
"subagents": subs,
|
||||
"middleware": base_middleware,
|
||||
"system_prompt": get_system_prompt(),
|
||||
"skills": ["/skills/"],
|
||||
"system_prompt": _configured_system_prompt(_ensure_config()),
|
||||
"skills": list(DEFAULT_SKILL_SOURCES),
|
||||
}
|
||||
|
||||
|
||||
def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress=None):
|
||||
def load_mcp_and_build_kwargs(
|
||||
base_backend,
|
||||
base_middleware,
|
||||
*,
|
||||
on_mcp_progress=None,
|
||||
workspace_dir=None,
|
||||
):
|
||||
"""Load MCP tools (cached by config) and build agent kwargs.
|
||||
|
||||
Re-connects to MCP servers only when the effective MCP config changes.
|
||||
@@ -355,7 +430,11 @@ def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress=
|
||||
|
||||
mcp_by_agent = _load_mcp_tools_cached(on_progress=on_mcp_progress)
|
||||
if not mcp_by_agent:
|
||||
return _build_base_kwargs(base_backend, base_middleware)
|
||||
return _build_base_kwargs(
|
||||
base_backend,
|
||||
base_middleware,
|
||||
workspace_dir=workspace_dir,
|
||||
)
|
||||
|
||||
tool_registry = {"think_tool": think_tool}
|
||||
if os.environ.get("TAVILY_API_KEY"):
|
||||
@@ -375,7 +454,8 @@ def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress=
|
||||
tool_registry=registry,
|
||||
)
|
||||
|
||||
_inject_subagent_middleware(subs)
|
||||
_ensure_general_purpose_subagent(subs)
|
||||
_inject_subagent_middleware(subs, workspace_dir=workspace_dir)
|
||||
|
||||
# Inject MCP tools into subagents by name
|
||||
for sa in subs:
|
||||
@@ -393,8 +473,8 @@ def load_mcp_and_build_kwargs(base_backend, base_middleware, *, on_mcp_progress=
|
||||
"backend": base_backend,
|
||||
"subagents": subs,
|
||||
"middleware": base_middleware,
|
||||
"system_prompt": get_system_prompt(),
|
||||
"skills": ["/skills/"],
|
||||
"system_prompt": _configured_system_prompt(_ensure_config()),
|
||||
"skills": list(DEFAULT_SKILL_SOURCES),
|
||||
}
|
||||
|
||||
|
||||
@@ -443,6 +523,7 @@ def _get_default_middleware(
|
||||
*,
|
||||
for_async_subagent: bool = False,
|
||||
workspace_dir: str | Path | None = None,
|
||||
memory_source_agent: str = "EvoScientist",
|
||||
):
|
||||
"""Build the default middleware list.
|
||||
|
||||
@@ -457,14 +538,18 @@ def _get_default_middleware(
|
||||
``subagents/_factory.py`` deliberately skips ``interrupt_on=`` on
|
||||
the deepagents level. Defaults to False (full middleware list)
|
||||
for the CLI's in-process agent.
|
||||
memory_source_agent: Attribution name for profile/observation writes.
|
||||
Async sub-agent factories pass their deployed agent name here.
|
||||
"""
|
||||
from .middleware import (
|
||||
ConfigurableModelMiddleware,
|
||||
ContextOverflowMapperMiddleware,
|
||||
MemoryLifecycleRole,
|
||||
ModelFallbackMiddleware,
|
||||
ToolErrorHandlerMiddleware,
|
||||
create_code_interpreter_middleware,
|
||||
create_context_editing_middleware,
|
||||
create_memory_lifecycle_middleware,
|
||||
create_memory_middleware,
|
||||
create_runtime_context_middleware,
|
||||
create_tool_selector_middleware,
|
||||
@@ -476,10 +561,30 @@ def _get_default_middleware(
|
||||
load_fallback_chain(cfg.model_fallbacks)
|
||||
model = _ensure_chat_model()
|
||||
memory_dir = str(_paths_mod.MEMORIES_DIR)
|
||||
source_type = (
|
||||
MemorySourceType.SUBAGENT if for_async_subagent else MemorySourceType.TURN
|
||||
)
|
||||
memory_controls = MemoryControls.from_config(cfg)
|
||||
worker_target = (
|
||||
MemoryObservationTarget.SUBAGENT_WORKER
|
||||
if for_async_subagent
|
||||
else MemoryObservationTarget.TURN_WORKER
|
||||
)
|
||||
# ``ConfigurableModelMiddleware`` is placed first so it wraps
|
||||
# ``ModelFallbackMiddleware``: a configurable.model override sets the
|
||||
# PRIMARY model only, leaving the fallback chain free to try its own
|
||||
# alternatives instead of re-overriding every retry to the same model.
|
||||
memory_middleware = create_memory_middleware(
|
||||
memory_dir,
|
||||
workspace_dir=workspace_dir,
|
||||
source_type=source_type,
|
||||
source_agent=memory_source_agent,
|
||||
enable_profile_memory=memory_controls.profile_enabled,
|
||||
enable_observation_memory=memory_controls.observations_enabled,
|
||||
enable_observation_tool=memory_controls.observation_tool_enabled(
|
||||
MemoryObservationTarget.AGENT
|
||||
),
|
||||
)
|
||||
mw = [
|
||||
ConfigurableModelMiddleware(),
|
||||
create_context_editing_middleware(model),
|
||||
@@ -488,8 +593,23 @@ def _get_default_middleware(
|
||||
ToolErrorHandlerMiddleware(),
|
||||
*create_tool_selector_middleware(model=model),
|
||||
create_runtime_context_middleware(),
|
||||
create_memory_middleware(memory_dir, workspace_dir=workspace_dir),
|
||||
]
|
||||
if memory_controls.memory_enabled:
|
||||
mw.append(memory_middleware)
|
||||
if memory_controls.worker_needed(worker_target):
|
||||
mw.append(
|
||||
create_memory_lifecycle_middleware(
|
||||
memory_dir,
|
||||
workspace_dir=workspace_dir,
|
||||
project_id=memory_middleware.project_id,
|
||||
role=(
|
||||
MemoryLifecycleRole.SUBAGENT
|
||||
if for_async_subagent
|
||||
else MemoryLifecycleRole.TURN
|
||||
),
|
||||
source_agent=memory_source_agent,
|
||||
)
|
||||
)
|
||||
|
||||
if cfg.enable_ask_user and not cfg.auto_mode and not for_async_subagent:
|
||||
from .middleware.ask_user import AskUserMiddleware
|
||||
@@ -556,9 +676,17 @@ def _get_default_agent():
|
||||
)
|
||||
|
||||
if os.environ.get("EVOSCIENTIST_DEPLOY_MODE", "").lower() == "stripped":
|
||||
kwargs = _build_base_kwargs(be, mw)
|
||||
kwargs = _build_base_kwargs(
|
||||
be,
|
||||
mw,
|
||||
workspace_dir=str(_paths_mod.WORKSPACE_ROOT),
|
||||
)
|
||||
else:
|
||||
kwargs = load_mcp_and_build_kwargs(be, mw)
|
||||
kwargs = load_mcp_and_build_kwargs(
|
||||
be,
|
||||
mw,
|
||||
workspace_dir=str(_paths_mod.WORKSPACE_ROOT),
|
||||
)
|
||||
|
||||
_EvoScientist_agent = create_deep_agent(
|
||||
**kwargs,
|
||||
@@ -573,7 +701,7 @@ def __getattr__(name: str):
|
||||
if name == "chat_model":
|
||||
return _ensure_chat_model()
|
||||
if name == "SYSTEM_PROMPT":
|
||||
return get_system_prompt()
|
||||
return _configured_system_prompt(_ensure_config())
|
||||
if name == "backend":
|
||||
return _get_default_backend()
|
||||
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
||||
@@ -678,7 +806,12 @@ def create_cli_agent(
|
||||
)
|
||||
|
||||
# Re-load MCP tools from current config (picks up /mcp add changes)
|
||||
kwargs = load_mcp_and_build_kwargs(be, mw, on_mcp_progress=on_mcp_progress)
|
||||
kwargs = load_mcp_and_build_kwargs(
|
||||
be,
|
||||
mw,
|
||||
on_mcp_progress=on_mcp_progress,
|
||||
workspace_dir=workspace_dir,
|
||||
)
|
||||
|
||||
return create_deep_agent(
|
||||
**kwargs,
|
||||
|
||||
@@ -404,6 +404,11 @@ async def _dispatch_channel_slash_impl(
|
||||
return True # must return — do NOT fall through to the agent
|
||||
|
||||
if cmd_executed:
|
||||
if ctx.command_error is not None:
|
||||
details = ctx.command_error or "(no details)"
|
||||
_set_channel_response(msg.msg_id, f"Command error: {details}")
|
||||
return True
|
||||
|
||||
if on_cmd_completed is not None:
|
||||
try:
|
||||
# Command output already flushed by ``cmd_manager.execute``
|
||||
|
||||
+181
-26
@@ -8,14 +8,15 @@ from collections.abc import Awaitable, Callable
|
||||
from datetime import datetime
|
||||
from importlib.metadata import version as _pkg_version
|
||||
from pathlib import Path
|
||||
from typing import Annotated, Any
|
||||
from typing import Annotated, Any, cast
|
||||
|
||||
import typer # type: ignore[import-untyped]
|
||||
import typer
|
||||
from rich.markup import escape
|
||||
from rich.table import Table
|
||||
|
||||
from ..commands.base import Command, CommandContext
|
||||
from ..llm.context_window import DEFAULT_CONTEXT_WINDOW_FALLBACK, resolve_context_window
|
||||
from ..paths import ensure_dirs, set_workspace_root
|
||||
from ..paths import ensure_dirs, set_active_workspace, set_workspace_root
|
||||
from ..stream.console import console
|
||||
from ._app import app, channel_app, config_app, configure_app, mcp_app, sessions_app
|
||||
from ._constants import build_metadata
|
||||
@@ -435,10 +436,11 @@ class CompactSummaryRenderable:
|
||||
|
||||
|
||||
def _ensure_async_subagent_server(config: Any, *, workspace_dir: str) -> None:
|
||||
"""Conditionally start the langgraph dev subprocess for async sub-agents.
|
||||
"""Start the langgraph dev subprocess for background agent work.
|
||||
|
||||
Shared by both the interactive entry and the serve entry — keeps the
|
||||
user-visible status message and the conditional in one place.
|
||||
Shared by both the interactive entry and the serve entry so the
|
||||
user-visible status message and workspace-mismatch handling stay in one
|
||||
place.
|
||||
|
||||
Raises ``typer.Exit(1)`` (after surfacing a red error) when an
|
||||
externally-managed langgraph dev is already running for a different
|
||||
@@ -447,13 +449,11 @@ def _ensure_async_subagent_server(config: Any, *, workspace_dir: str) -> None:
|
||||
state would route async sub-agent calls to a process pinned to /A
|
||||
while the main agent runs in /B.
|
||||
"""
|
||||
if not getattr(config, "enable_async_subagents", False):
|
||||
return
|
||||
from ..langgraph_dev.manager import WorkspaceMismatchError, ensure_langgraph_dev
|
||||
|
||||
try:
|
||||
with console.status(
|
||||
"[dim]Starting async sub-agent server (langgraph dev)...[/dim]",
|
||||
"[dim]Starting background agent server (langgraph dev)...[/dim]",
|
||||
spinner="dots",
|
||||
):
|
||||
ensure_langgraph_dev(config, workspace_dir=workspace_dir)
|
||||
@@ -462,6 +462,33 @@ def _ensure_async_subagent_server(config: Any, *, workspace_dir: str) -> None:
|
||||
raise typer.Exit(1) from exc
|
||||
|
||||
|
||||
async def _sync_background_agent_server_workspace(
|
||||
config: Any,
|
||||
*,
|
||||
workspace_dir: str,
|
||||
status_message: str = (
|
||||
"[dim]Syncing background agent server to resumed workspace...[/dim]"
|
||||
),
|
||||
) -> None:
|
||||
"""Sync langgraph dev to a resumed workspace for background agent work.
|
||||
|
||||
``ensure_langgraph_dev`` is intentionally always called: EvoMemory
|
||||
background workers require the server even when async subagents are disabled.
|
||||
WorkspaceMismatchError is left for callers to handle according to their UI
|
||||
flow.
|
||||
"""
|
||||
import asyncio
|
||||
|
||||
from ..langgraph_dev.manager import ensure_langgraph_dev
|
||||
|
||||
with console.status(status_message, spinner="dots"):
|
||||
await asyncio.to_thread(
|
||||
ensure_langgraph_dev,
|
||||
config,
|
||||
workspace_dir=workspace_dir,
|
||||
)
|
||||
|
||||
|
||||
def _resolve_context_window(
|
||||
model: Any, fallback: int = _COMPACT_CONTEXT_WINDOW_FALLBACK
|
||||
) -> int:
|
||||
@@ -592,8 +619,9 @@ async def compact_conversation(
|
||||
return CompactResult("noop", "Nothing to compact — start a conversation first.")
|
||||
|
||||
from langchain_core.messages.utils import count_tokens_approximately
|
||||
from langchain_core.runnables import RunnableConfig
|
||||
|
||||
config = {"configurable": {"thread_id": thread_id}}
|
||||
config: RunnableConfig = {"configurable": {"thread_id": thread_id}}
|
||||
|
||||
try:
|
||||
state_snapshot = await agent.aget_state(config)
|
||||
@@ -714,7 +742,12 @@ async def compact_conversation(
|
||||
finally:
|
||||
var_child_runnable_config.reset(_token)
|
||||
|
||||
summary_msg = middleware._build_new_messages_with_path(summary, file_path)[0]
|
||||
from langchain_core.messages import HumanMessage
|
||||
|
||||
summary_msg = cast(
|
||||
HumanMessage,
|
||||
middleware._build_new_messages_with_path(summary, file_path)[0],
|
||||
)
|
||||
|
||||
# Compute token savings (message-level, used for pct calculation)
|
||||
tokens_summary = count_tokens_approximately([summary_msg])
|
||||
@@ -805,9 +838,101 @@ def _make_serve_start_new_session_cb(
|
||||
return _cb
|
||||
|
||||
|
||||
def _serve_resume_config(
|
||||
agent_holder: dict[str, Any],
|
||||
config: Any | None,
|
||||
) -> Any | None:
|
||||
"""Return the effective config to use for serve-mode resume sync."""
|
||||
return config if config is not None else agent_holder.get("config")
|
||||
|
||||
|
||||
async def _apply_serve_resume_state(
|
||||
agent_holder: dict[str, Any],
|
||||
channel_runtime: Any | None,
|
||||
*,
|
||||
thread_id: str,
|
||||
workspace_dir: str | None,
|
||||
config: Any | None = None,
|
||||
) -> None:
|
||||
"""Adopt a resumed thread/workspace into serve-mode runtime state.
|
||||
|
||||
Workspace-bound resources are rebuilt and synced before mutating the shared
|
||||
holder. The agent is loaded before syncing the external server so a load
|
||||
failure cannot move the server away from the currently active session.
|
||||
"""
|
||||
import asyncio
|
||||
|
||||
old_workspace = agent_holder.get("workspace_dir")
|
||||
new_workspace = (
|
||||
workspace_dir if workspace_dir and workspace_dir != old_workspace else None
|
||||
)
|
||||
new_agent: Any | None = None
|
||||
|
||||
if new_workspace is not None:
|
||||
effective_config = _serve_resume_config(agent_holder, config)
|
||||
if effective_config is None:
|
||||
raise RuntimeError(
|
||||
"Cannot resume into a different workspace in serve mode without "
|
||||
"the effective configuration."
|
||||
)
|
||||
try:
|
||||
new_agent = await asyncio.to_thread(
|
||||
_load_agent,
|
||||
workspace_dir=new_workspace,
|
||||
config=effective_config,
|
||||
)
|
||||
await _sync_background_agent_server_workspace(
|
||||
effective_config,
|
||||
workspace_dir=new_workspace,
|
||||
)
|
||||
except Exception:
|
||||
if old_workspace:
|
||||
set_active_workspace(old_workspace)
|
||||
raise
|
||||
|
||||
old_thread_id = agent_holder.get("thread_id")
|
||||
thread_changed = bool(thread_id) and thread_id != old_thread_id
|
||||
if thread_changed:
|
||||
forget_channel_origin(old_thread_id)
|
||||
agent_holder["thread_id"] = thread_id
|
||||
if channel_runtime is not None:
|
||||
channel_runtime.thread_id = thread_id
|
||||
|
||||
if new_workspace is not None:
|
||||
agent_holder["workspace_dir"] = new_workspace
|
||||
agent_holder["agent"] = new_agent
|
||||
if channel_runtime is not None:
|
||||
channel_runtime.agent = new_agent
|
||||
|
||||
|
||||
def _make_serve_handle_session_resume_cb(
|
||||
agent_holder: dict[str, Any],
|
||||
channel_runtime: Any | None = None,
|
||||
*,
|
||||
config: Any | None = None,
|
||||
):
|
||||
"""Build the ChannelCommandUI resume callback for serve mode."""
|
||||
|
||||
async def _cb(thread_id: str, workspace_dir: str | None = None) -> None:
|
||||
old_thread_id = agent_holder.get("thread_id")
|
||||
await _apply_serve_resume_state(
|
||||
agent_holder,
|
||||
channel_runtime,
|
||||
thread_id=thread_id,
|
||||
workspace_dir=workspace_dir,
|
||||
config=config,
|
||||
)
|
||||
if thread_id and thread_id != old_thread_id:
|
||||
agent_holder["_resume_warning_thread_id"] = thread_id
|
||||
|
||||
return _cb
|
||||
|
||||
|
||||
def _make_serve_cmd_completed_hook(
|
||||
agent_holder: dict[str, Any],
|
||||
channel_runtime: Any | None = None,
|
||||
*,
|
||||
config: Any | None = None,
|
||||
):
|
||||
"""Build the ``on_cmd_completed`` hook used by serve mode.
|
||||
|
||||
@@ -827,12 +952,18 @@ def _make_serve_cmd_completed_hook(
|
||||
without spinning up the whole serve loop.
|
||||
"""
|
||||
|
||||
async def _hook(ctx: Any, original_agent: Any, cmd: Any) -> None:
|
||||
async def _hook(ctx: CommandContext, original_agent: Any, cmd: Command) -> None:
|
||||
if ctx.agent is not None and ctx.agent is not original_agent:
|
||||
agent_holder["agent"] = ctx.agent
|
||||
if channel_runtime is not None:
|
||||
channel_runtime.agent = ctx.agent
|
||||
|
||||
old_thread_id = agent_holder.get("thread_id")
|
||||
resume_warning_thread_id = agent_holder.pop(
|
||||
"_resume_warning_thread_id",
|
||||
None,
|
||||
)
|
||||
|
||||
# ``/resume`` mutates ``ctx.thread_id`` directly (its UI callback
|
||||
# is a no-op in serve mode since there's no REPL to reset). Pick
|
||||
# up the new id here so subsequent messages run on the resumed
|
||||
@@ -841,23 +972,32 @@ def _make_serve_cmd_completed_hook(
|
||||
# ``ctx.thread_id`` unchanged — ``thread_changed`` gates both
|
||||
# the adoption and the user-facing warning so neither fires in
|
||||
# that case.
|
||||
new_tid = getattr(ctx, "thread_id", None)
|
||||
thread_changed = bool(new_tid) and new_tid != agent_holder.get("thread_id")
|
||||
if thread_changed:
|
||||
forget_channel_origin(agent_holder.get("thread_id"))
|
||||
agent_holder["thread_id"] = new_tid
|
||||
if channel_runtime is not None:
|
||||
channel_runtime.thread_id = new_tid
|
||||
new_tid = ctx.thread_id
|
||||
if cmd.name == "/resume":
|
||||
await _apply_serve_resume_state(
|
||||
agent_holder,
|
||||
channel_runtime,
|
||||
thread_id=new_tid,
|
||||
workspace_dir=ctx.workspace_dir,
|
||||
config=config,
|
||||
)
|
||||
else:
|
||||
thread_changed = bool(new_tid) and new_tid != old_thread_id
|
||||
if thread_changed:
|
||||
forget_channel_origin(old_thread_id)
|
||||
agent_holder["thread_id"] = new_tid
|
||||
if channel_runtime is not None:
|
||||
channel_runtime.thread_id = new_tid
|
||||
|
||||
new_workspace = getattr(ctx, "workspace_dir", None)
|
||||
if new_workspace and new_workspace != agent_holder.get("workspace_dir"):
|
||||
agent_holder["workspace_dir"] = new_workspace
|
||||
thread_changed = bool(new_tid) and new_tid != old_thread_id
|
||||
|
||||
# Surface the in-memory-state limitation to the channel user
|
||||
# for ``/resume`` so the missing history isn't silent. Flush
|
||||
# is required because ``cmd_manager.execute`` already flushed
|
||||
# the command's own output before calling this hook.
|
||||
if getattr(cmd, "name", None) == "/resume" and thread_changed:
|
||||
if cmd.name == "/resume" and (
|
||||
thread_changed or resume_warning_thread_id == new_tid
|
||||
):
|
||||
try:
|
||||
ctx.ui.append_system(
|
||||
"Note: serve mode uses in-memory state — "
|
||||
@@ -879,6 +1019,7 @@ def _serve_process_message(
|
||||
workspace_dir: str,
|
||||
show_thinking: bool,
|
||||
on_cmd_completed: Callable[..., Awaitable[None]] | None = None,
|
||||
handle_session_resume_cb: Callable[..., Awaitable[None]] | None = None,
|
||||
start_new_session_cb: Callable[[], None] | None = None,
|
||||
channel_runtime: Any | None = None,
|
||||
) -> None:
|
||||
@@ -1004,8 +1145,17 @@ def _serve_process_message(
|
||||
append_system=lambda t, s="dim": console.print(t, style=s),
|
||||
start_new_session_cb=start_new_session_cb
|
||||
or _make_serve_start_new_session_cb(agent_holder, channel_runtime),
|
||||
handle_session_resume_cb=handle_session_resume_cb
|
||||
or _make_serve_handle_session_resume_cb(
|
||||
agent_holder,
|
||||
channel_runtime,
|
||||
),
|
||||
on_cmd_completed=on_cmd_completed
|
||||
or _make_serve_cmd_completed_hook(agent_holder, channel_runtime),
|
||||
or _make_serve_cmd_completed_hook(
|
||||
agent_holder,
|
||||
channel_runtime,
|
||||
config=agent_holder.get("config"),
|
||||
),
|
||||
channel_runtime=channel_runtime,
|
||||
)
|
||||
)
|
||||
@@ -1254,6 +1404,7 @@ def serve(
|
||||
"agent": agent,
|
||||
"thread_id": tid,
|
||||
"workspace_dir": ws,
|
||||
"config": config,
|
||||
}
|
||||
|
||||
from ..commands.base import ChannelRuntime
|
||||
@@ -1264,7 +1415,10 @@ def serve(
|
||||
# them for every inbound message. Without this hoist each message
|
||||
# would allocate a fresh closure pair.
|
||||
_serve_on_cmd_completed = _make_serve_cmd_completed_hook(
|
||||
agent_holder, channel_runtime
|
||||
agent_holder, channel_runtime, config=config
|
||||
)
|
||||
_serve_handle_session_resume_cb = _make_serve_handle_session_resume_cb(
|
||||
agent_holder, channel_runtime, config=config
|
||||
)
|
||||
_serve_start_new_session_cb = _make_serve_start_new_session_cb(
|
||||
agent_holder, channel_runtime
|
||||
@@ -1323,6 +1477,7 @@ def serve(
|
||||
workspace_dir=ws,
|
||||
show_thinking=effective_channel_thinking,
|
||||
on_cmd_completed=_serve_on_cmd_completed,
|
||||
handle_session_resume_cb=_serve_handle_session_resume_cb,
|
||||
start_new_session_cb=_serve_start_new_session_cb,
|
||||
channel_runtime=channel_runtime,
|
||||
)
|
||||
@@ -2049,7 +2204,7 @@ def _main_callback(
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
import nest_asyncio # type: ignore[import-untyped]
|
||||
import nest_asyncio
|
||||
|
||||
nest_asyncio.apply()
|
||||
asyncio.get_event_loop().run_until_complete(_single_shot())
|
||||
|
||||
@@ -31,7 +31,7 @@ from rich.text import Text
|
||||
|
||||
import EvoScientist.cli.channel as _ch_mod
|
||||
|
||||
from ..commands.base import CommandContext
|
||||
from ..commands.base import Command, CommandContext
|
||||
from ..commands.manager import manager as cmd_manager
|
||||
from ..sessions import (
|
||||
generate_thread_id,
|
||||
@@ -269,6 +269,7 @@ def cmd_interactive(
|
||||
thread_id=thread_id,
|
||||
load_agent=load_agent,
|
||||
create_session_workspace=_create_session_workspace,
|
||||
config=config,
|
||||
)
|
||||
return
|
||||
|
||||
@@ -630,10 +631,9 @@ def cmd_interactive(
|
||||
renders conversation history."""
|
||||
if workspace_dir:
|
||||
# Sync the langgraph dev subprocess to the resumed
|
||||
# workspace so deployed sub-agents (writing-agent etc.)
|
||||
# workspace so background workers and deployed sub-agents
|
||||
# don't operate on the previous workspace's files. The
|
||||
# manager auto-detects the change and restarts; no-ops
|
||||
# if async subagents are disabled or workspace unchanged.
|
||||
# manager auto-detects the change and restarts when needed.
|
||||
# Restart can take 10-15s — show a spinner so the user
|
||||
# doesn't think the CLI is frozen, and run the sync call
|
||||
# in a worker thread so the asyncio event loop keeps
|
||||
@@ -642,29 +642,21 @@ def cmd_interactive(
|
||||
# State mutation happens AFTER this sync succeeds so a
|
||||
# WorkspaceMismatchError leaves the session's existing
|
||||
# workspace_dir / thread_id untouched.
|
||||
if getattr(config, "enable_async_subagents", False):
|
||||
from ..langgraph_dev.manager import (
|
||||
WorkspaceMismatchError,
|
||||
ensure_langgraph_dev,
|
||||
)
|
||||
from ..langgraph_dev.manager import WorkspaceMismatchError
|
||||
from .commands import _sync_background_agent_server_workspace
|
||||
|
||||
try:
|
||||
with console.status(
|
||||
"[dim]Syncing async sub-agent server to resumed "
|
||||
"workspace...[/dim]",
|
||||
spinner="dots",
|
||||
):
|
||||
await asyncio.to_thread(
|
||||
ensure_langgraph_dev,
|
||||
config,
|
||||
workspace_dir=workspace_dir,
|
||||
)
|
||||
except WorkspaceMismatchError as exc:
|
||||
# Another EvoSci process owns the langgraph dev
|
||||
# server for a different workspace. Abort the
|
||||
# resume without mutating session state.
|
||||
console.print(f"[red]{exc}[/red]")
|
||||
return
|
||||
try:
|
||||
await _sync_background_agent_server_workspace(
|
||||
config,
|
||||
workspace_dir=workspace_dir,
|
||||
)
|
||||
except WorkspaceMismatchError as exc:
|
||||
# Another EvoSci process owns the langgraph dev
|
||||
# server for a different workspace. Abort the
|
||||
# resume without mutating session state. Raise so
|
||||
# command UIs, including channel UI, report failure
|
||||
# instead of continuing with success/history output.
|
||||
raise RuntimeError(str(exc)) from exc
|
||||
state["workspace_dir"] = workspace_dir
|
||||
if thread_id != state.get("thread_id"):
|
||||
# Only drop the origin on a real thread change — resuming
|
||||
@@ -724,29 +716,20 @@ def cmd_interactive(
|
||||
# Show a spinner during the 10-15s restart, and run
|
||||
# the sync call in a worker thread so the asyncio
|
||||
# event loop stays responsive.
|
||||
if getattr(config, "enable_async_subagents", False):
|
||||
from ..langgraph_dev.manager import (
|
||||
WorkspaceMismatchError,
|
||||
ensure_langgraph_dev,
|
||||
)
|
||||
from ..langgraph_dev.manager import WorkspaceMismatchError
|
||||
from .commands import _sync_background_agent_server_workspace
|
||||
|
||||
try:
|
||||
with console.status(
|
||||
"[dim]Syncing async sub-agent server to "
|
||||
"resumed workspace...[/dim]",
|
||||
spinner="dots",
|
||||
):
|
||||
await asyncio.to_thread(
|
||||
ensure_langgraph_dev,
|
||||
config,
|
||||
workspace_dir=ws,
|
||||
)
|
||||
except WorkspaceMismatchError as exc:
|
||||
# Startup --resume into a workspace owned by
|
||||
# a different EvoSci process: refuse to start
|
||||
# the CLI so the user can resolve the conflict.
|
||||
console.print(f"[red]{exc}[/red]")
|
||||
raise typer.Exit(1) from exc
|
||||
try:
|
||||
await _sync_background_agent_server_workspace(
|
||||
config,
|
||||
workspace_dir=ws,
|
||||
)
|
||||
except WorkspaceMismatchError as exc:
|
||||
# Startup --resume into a workspace owned by
|
||||
# a different EvoSci process: refuse to start
|
||||
# the CLI so the user can resolve the conflict.
|
||||
console.print(f"[red]{exc}[/red]")
|
||||
raise typer.Exit(1) from exc
|
||||
else:
|
||||
# Resolution failed (ambiguous/not-found); the user's raw
|
||||
# input is still seeded in state["thread_id"] from init.
|
||||
@@ -890,7 +873,7 @@ def cmd_interactive(
|
||||
# etc. sent via iMessage actually execute instead of being
|
||||
# fed to the LLM as a plain prompt.
|
||||
async def _on_channel_cmd_completed(
|
||||
ctx: Any, original_agent: Any, cmd: Any
|
||||
ctx: CommandContext, original_agent: Any, cmd: Command
|
||||
) -> None:
|
||||
"""Mirror the REPL adoption block at
|
||||
``interactive.py:1005-1030`` so ``/model`` and similar
|
||||
@@ -931,7 +914,7 @@ def cmd_interactive(
|
||||
# status snapshot re-rendered even when the agent
|
||||
# didn't swap. ``/resume`` refreshes inline in its
|
||||
# own async callback.
|
||||
if agent_swapped or getattr(cmd, "name", None) in (
|
||||
if agent_swapped or cmd.name in (
|
||||
"/compact",
|
||||
"/new",
|
||||
):
|
||||
|
||||
@@ -13,6 +13,7 @@ from ..llm.context_window import (
|
||||
DEFAULT_CONTEXT_WINDOW_FALLBACK,
|
||||
resolve_context_window,
|
||||
)
|
||||
from ..memory.worker_activity import MemoryWorkerStatusSnapshot, memory_worker_status
|
||||
from ..sessions import get_thread_messages
|
||||
|
||||
_FALLBACK_CONTEXT_WINDOW = DEFAULT_CONTEXT_WINDOW_FALLBACK
|
||||
@@ -91,15 +92,15 @@ def format_token_count_compact(value: int) -> str:
|
||||
"""Format large token counts into a compact human-readable form."""
|
||||
abs_value = abs(int(value))
|
||||
if abs_value >= 1_000_000:
|
||||
num = value / 1_000_000
|
||||
num = float(value) / 1_000_000
|
||||
suffix = "M"
|
||||
elif abs_value >= 1_000:
|
||||
num = value / 1_000
|
||||
num = float(value) / 1_000
|
||||
suffix = "K"
|
||||
else:
|
||||
return str(value)
|
||||
|
||||
if num.is_integer():
|
||||
if num == int(num):
|
||||
return f"{int(num)}{suffix}"
|
||||
return f"{num:.1f}{suffix}"
|
||||
|
||||
@@ -160,15 +161,10 @@ def trim_status_text(text: str, max_width: int) -> str:
|
||||
if max_width <= ellipsis_width:
|
||||
return ellipsis[:max_width]
|
||||
|
||||
try:
|
||||
from prompt_toolkit.utils import get_cwidth
|
||||
except Exception:
|
||||
get_cwidth = None
|
||||
|
||||
out: list[str] = []
|
||||
width = 0
|
||||
for ch in text:
|
||||
ch_width = get_cwidth(ch) if get_cwidth else len(ch)
|
||||
ch_width = _display_width(ch)
|
||||
if width + ch_width + ellipsis_width > max_width:
|
||||
break
|
||||
out.append(ch)
|
||||
@@ -176,13 +172,70 @@ def trim_status_text(text: str, max_width: int) -> str:
|
||||
return "".join(out).rstrip() + ellipsis
|
||||
|
||||
|
||||
def get_memory_worker_status() -> MemoryWorkerStatusSnapshot | None:
|
||||
"""Read completed EvoMemory save counts without making rendering fail."""
|
||||
try:
|
||||
return memory_worker_status()
|
||||
except Exception:
|
||||
return None
|
||||
|
||||
|
||||
def _plural(count: int, singular: str, plural: str | None = None) -> str:
|
||||
word = singular if count == 1 else (plural or f"{singular}s")
|
||||
return f"{count} {word}"
|
||||
|
||||
|
||||
def _memory_worker_label(status: MemoryWorkerStatusSnapshot) -> str:
|
||||
parts: list[str] = []
|
||||
if status.is_running:
|
||||
parts.append("🧠")
|
||||
|
||||
saved: list[str] = []
|
||||
if status.profile_updates:
|
||||
saved.append(_plural(status.profile_updates, "profile edit"))
|
||||
if status.observations_recorded:
|
||||
saved.append(_plural(status.observations_recorded, "observation"))
|
||||
if saved:
|
||||
parts.append(f"Saved {', '.join(saved)}")
|
||||
|
||||
return " ".join(parts)
|
||||
|
||||
|
||||
def _append_memory_worker_indicator(
|
||||
frags: list[tuple[str, str]],
|
||||
*,
|
||||
status: MemoryWorkerStatusSnapshot | None,
|
||||
width: int,
|
||||
) -> None:
|
||||
if status is None:
|
||||
return
|
||||
|
||||
label = _memory_worker_label(status)
|
||||
if not label:
|
||||
return
|
||||
|
||||
tail: list[tuple[str, str]] = []
|
||||
if frags and frags[-1] == ("class:status-bar", " "):
|
||||
tail.append(frags.pop())
|
||||
|
||||
separator = " │ " if width >= 76 else " · "
|
||||
frags.extend(
|
||||
[
|
||||
("class:status-bar-dim", separator),
|
||||
("class:status-bar-warn", label),
|
||||
]
|
||||
)
|
||||
frags.extend(tail)
|
||||
|
||||
|
||||
def build_status_fragments(
|
||||
snapshot: SessionStatusSnapshot,
|
||||
started_at: datetime,
|
||||
width: int,
|
||||
) -> list[tuple[str, str]]:
|
||||
"""Build prompt_toolkit formatted-text fragments for the status bar."""
|
||||
duration_label = format_duration_compact(started_at)
|
||||
now = datetime.now()
|
||||
duration_label = format_duration_compact(started_at, now=now)
|
||||
percent = snapshot.context_percent
|
||||
percent_label = f"{percent}%"
|
||||
if width < 52:
|
||||
@@ -220,6 +273,12 @@ def build_status_fragments(
|
||||
("class:status-bar", " "),
|
||||
]
|
||||
|
||||
_append_memory_worker_indicator(
|
||||
frags,
|
||||
status=get_memory_worker_status(),
|
||||
width=width,
|
||||
)
|
||||
|
||||
total_width = sum(_display_width(text) for _, text in frags)
|
||||
if total_width > width:
|
||||
plain_text = "".join(text for _, text in frags)
|
||||
|
||||
@@ -22,7 +22,7 @@ from rich.text import Text
|
||||
import EvoScientist.cli.channel as _ch_mod
|
||||
from EvoScientist.cli.widgets.thread_selector import ThreadPickerWidget
|
||||
|
||||
from ..commands import CommandContext
|
||||
from ..commands import Command, CommandContext
|
||||
from ..commands import manager as cmd_manager
|
||||
from ..paths import DATA_DIR
|
||||
from ..sessions import (
|
||||
@@ -191,7 +191,7 @@ async def _sync_tui_command_completion(
|
||||
app: Any,
|
||||
ctx: CommandContext,
|
||||
original_agent: Any,
|
||||
cmd: Any,
|
||||
cmd: Command,
|
||||
) -> None:
|
||||
"""Adopt successful command-side state changes back into the TUI app."""
|
||||
agent_swapped = ctx.agent is not None and ctx.agent is not original_agent
|
||||
@@ -260,8 +260,14 @@ def run_textual_interactive(
|
||||
thread_id: str | None,
|
||||
load_agent: Callable[..., Any],
|
||||
create_session_workspace: Callable[[str | None], str],
|
||||
config: Any | None = None,
|
||||
) -> None:
|
||||
"""Run full-screen Textual interactive chat loop."""
|
||||
if config is None:
|
||||
from ..config import get_effective_config
|
||||
|
||||
config = get_effective_config()
|
||||
|
||||
try:
|
||||
from textual.app import App, ComposeResult
|
||||
from textual.binding import Binding
|
||||
@@ -642,47 +648,57 @@ def run_textual_interactive(
|
||||
if workspace_dir:
|
||||
# Mirror the Rich CLI fix: when a /resume restores a thread
|
||||
# whose workspace differs from the one the langgraph dev
|
||||
# subprocess was launched with, the deployed sub-agents
|
||||
# would otherwise keep operating on the previous workspace.
|
||||
# Sync the subprocess to the new workspace; the manager
|
||||
# auto-detects the change and restarts (or no-ops if disabled
|
||||
# or unchanged). Run in a worker thread so the Textual event
|
||||
# loop keeps refreshing the UI during the up-to-60s wait, and
|
||||
# show a live timer widget (like /compact) so the user sees
|
||||
# progress instead of a frozen static line.
|
||||
# subprocess was launched with, background workers and
|
||||
# deployed sub-agents would otherwise keep operating on the
|
||||
# previous workspace. Sync the subprocess to the new workspace;
|
||||
# the manager auto-detects the change and restarts when needed.
|
||||
# Run in a worker thread so the Textual event loop keeps
|
||||
# refreshing the UI during the up-to-60s wait, and show a live
|
||||
# timer widget (like /compact) so the user sees progress
|
||||
# instead of a frozen static line.
|
||||
#
|
||||
# ``self._workspace_dir`` is mutated AFTER the sync succeeds
|
||||
# so a WorkspaceMismatchError leaves the session pointing at
|
||||
# the existing workspace instead of half-resuming into the
|
||||
# conflicting one.
|
||||
from ..config import load_config
|
||||
# ``self._workspace_dir`` is mutated AFTER mismatch checks so
|
||||
# WorkspaceMismatchError leaves the session pointing at the
|
||||
# existing workspace. Other sync failures resume locally in
|
||||
# the TUI while background workers may be unavailable.
|
||||
from ..langgraph_dev.manager import (
|
||||
WorkspaceMismatchError,
|
||||
ensure_langgraph_dev,
|
||||
)
|
||||
from .widgets.workspace_sync_widget import WorkspaceSyncWidget
|
||||
|
||||
_resume_cfg = load_config()
|
||||
if getattr(_resume_cfg, "enable_async_subagents", False):
|
||||
from ..langgraph_dev.manager import (
|
||||
WorkspaceMismatchError,
|
||||
sync_widget = WorkspaceSyncWidget()
|
||||
container = self.query_one("#chat", VerticalScroll)
|
||||
await container.mount(sync_widget)
|
||||
container.scroll_end(animate=False)
|
||||
try:
|
||||
await asyncio.to_thread(
|
||||
ensure_langgraph_dev,
|
||||
config,
|
||||
workspace_dir=workspace_dir,
|
||||
)
|
||||
from .widgets.workspace_sync_widget import WorkspaceSyncWidget
|
||||
|
||||
sync_widget = WorkspaceSyncWidget()
|
||||
container = self.query_one("#chat", VerticalScroll)
|
||||
await container.mount(sync_widget)
|
||||
container.scroll_end(animate=False)
|
||||
try:
|
||||
await asyncio.to_thread(
|
||||
ensure_langgraph_dev,
|
||||
_resume_cfg,
|
||||
workspace_dir=workspace_dir,
|
||||
)
|
||||
except WorkspaceMismatchError as exc:
|
||||
# Another EvoSci process owns the langgraph dev for a
|
||||
# different workspace. Abort the resume without
|
||||
# mutating session state.
|
||||
self.notify(str(exc), severity="error", timeout=10)
|
||||
return
|
||||
finally:
|
||||
await sync_widget.cleanup()
|
||||
except WorkspaceMismatchError as exc:
|
||||
# Another EvoSci process owns the langgraph dev for a
|
||||
# different workspace. Abort the resume without mutating
|
||||
# session state. Raise so command UIs, including channel
|
||||
# UI, report failure instead of continuing with
|
||||
# success/history output.
|
||||
raise RuntimeError(str(exc)) from exc
|
||||
except Exception:
|
||||
_channel_logger.warning(
|
||||
"Failed to sync background agent server for resumed "
|
||||
"workspace %s; continuing resume in degraded mode",
|
||||
workspace_dir,
|
||||
exc_info=True,
|
||||
)
|
||||
self.append_system(
|
||||
"Background agent server sync failed; resumed local "
|
||||
"session, but async subagents and EvoMemory workers "
|
||||
"may be unavailable.",
|
||||
style="yellow",
|
||||
)
|
||||
finally:
|
||||
await sync_widget.cleanup()
|
||||
self._workspace_dir = workspace_dir
|
||||
|
||||
if thread_id != self._conversation_tid:
|
||||
@@ -975,7 +991,7 @@ def run_textual_interactive(
|
||||
self,
|
||||
ctx: CommandContext,
|
||||
original_agent: Any,
|
||||
cmd: Any,
|
||||
cmd: Command,
|
||||
) -> None:
|
||||
await _sync_tui_command_completion(self, ctx, original_agent, cmd)
|
||||
|
||||
@@ -3132,48 +3148,42 @@ def run_textual_interactive(
|
||||
# workspace BEFORE the Textual app takes over the
|
||||
# terminal. Mirrors interactive.py's Rich-CLI fix.
|
||||
# Without this, --resume against a thread from a
|
||||
# different workspace would leave deployed sub-agents
|
||||
# operating on the launch directory's files.
|
||||
# different workspace would leave background workers
|
||||
# and deployed sub-agents operating on the launch
|
||||
# directory's files.
|
||||
from ..stream.console import console as _resume_console
|
||||
|
||||
try:
|
||||
from ..config import load_config
|
||||
from ..langgraph_dev.manager import (
|
||||
WorkspaceMismatchError,
|
||||
ensure_langgraph_dev,
|
||||
from ..langgraph_dev.manager import WorkspaceMismatchError
|
||||
from .commands import (
|
||||
_sync_background_agent_server_workspace,
|
||||
)
|
||||
|
||||
_ws_cfg = load_config()
|
||||
if getattr(_ws_cfg, "enable_async_subagents", False):
|
||||
with _resume_console.status(
|
||||
"[dim]Syncing async sub-agent server to "
|
||||
"resumed workspace...[/dim]",
|
||||
spinner="dots",
|
||||
):
|
||||
await asyncio.to_thread(
|
||||
ensure_langgraph_dev,
|
||||
_ws_cfg,
|
||||
workspace_dir=ws,
|
||||
)
|
||||
await _sync_background_agent_server_workspace(
|
||||
config,
|
||||
workspace_dir=ws,
|
||||
)
|
||||
except WorkspaceMismatchError as _ws_mismatch_exc:
|
||||
# Surface the user-actionable message via the
|
||||
# Rich console (TUI hasn't taken over the terminal
|
||||
# yet) and abort the resume so the TUI doesn't
|
||||
# start up pointing at the conflicting workspace
|
||||
# with async sub-agents routed at a server pinned
|
||||
# to a different workspace.
|
||||
# with background work routed at a server pinned to
|
||||
# a different workspace.
|
||||
_resume_console.print(f"[red]{_ws_mismatch_exc}[/red]")
|
||||
mismatch_aborted = True
|
||||
except Exception as _ws_sync_exc:
|
||||
# Non-fatal at startup — async sub-agents fall back
|
||||
# to sync via the manager's own availability flag.
|
||||
# to sync via the manager's own availability flag,
|
||||
# and memory workers skip while the server is down.
|
||||
# Surface the exception so unexpected failures
|
||||
# (import errors, regressions in
|
||||
# ensure_langgraph_dev, etc.) don't hide silently.
|
||||
logging.getLogger(__name__).warning(
|
||||
"TUI startup workspace sync to langgraph dev "
|
||||
"failed: %s. Async sub-agents will fall back "
|
||||
"to in-process sync delegation for this session.",
|
||||
"to in-process sync delegation and EvoMemory "
|
||||
"workers will skip for this session.",
|
||||
_ws_sync_exc,
|
||||
)
|
||||
if mismatch_aborted:
|
||||
|
||||
@@ -78,6 +78,7 @@ class CommandContext:
|
||||
checkpointer: Any = None
|
||||
config: Any = None
|
||||
channel_runtime: ChannelRuntime | None = None
|
||||
command_error: str | None = None
|
||||
# Real LLM input token count from last usage_metadata (includes system
|
||||
# prompt + tool schemas). Used by /compact for accurate display.
|
||||
input_tokens_hint: int | None = None
|
||||
|
||||
@@ -93,12 +93,14 @@ class CommandManager:
|
||||
if not cmd:
|
||||
return False
|
||||
|
||||
ctx.command_error = None
|
||||
try:
|
||||
await cmd.execute(ctx, args)
|
||||
await ctx.ui.flush()
|
||||
return True
|
||||
except Exception as e:
|
||||
_logger.exception(f"Error executing command {cmd_name}: {e}")
|
||||
ctx.command_error = str(e)
|
||||
ctx.ui.append_system(f"Error executing {cmd_name}: {e}", style="red")
|
||||
await ctx.ui.flush()
|
||||
return True
|
||||
|
||||
@@ -10,6 +10,9 @@ The onboard module is loaded lazily because it pulls in heavy dependencies
|
||||
|
||||
from .settings import (
|
||||
EvoScientistConfig,
|
||||
MemoryControls,
|
||||
MemoryObservationTarget,
|
||||
MemoryObservationWriter,
|
||||
apply_config_to_env,
|
||||
get_config_dir,
|
||||
get_config_path,
|
||||
@@ -24,6 +27,9 @@ from .settings import (
|
||||
|
||||
__all__ = [
|
||||
"EvoScientistConfig",
|
||||
"MemoryControls",
|
||||
"MemoryObservationTarget",
|
||||
"MemoryObservationWriter",
|
||||
"apply_config_to_env",
|
||||
# settings
|
||||
"get_config_dir",
|
||||
|
||||
@@ -10,6 +10,7 @@ from __future__ import annotations
|
||||
import logging
|
||||
import os
|
||||
from dataclasses import asdict, dataclass, fields
|
||||
from enum import StrEnum
|
||||
from pathlib import Path
|
||||
from typing import Any, Literal
|
||||
|
||||
@@ -22,6 +23,40 @@ from dotenv import find_dotenv, load_dotenv
|
||||
# `interrupt_on` set in EvoScientist.py.
|
||||
HITL_SHELL_TOOLS = ("execute", "run_in_background")
|
||||
|
||||
|
||||
class MemoryObservationTarget(StrEnum):
|
||||
"""Runtime locations that can receive `record_observation`."""
|
||||
|
||||
AGENT = "agent"
|
||||
TURN_WORKER = "turn_worker"
|
||||
SUBAGENT_WORKER = "subagent_worker"
|
||||
|
||||
|
||||
class MemoryObservationWriter(StrEnum):
|
||||
"""Configured observation-writing policy."""
|
||||
|
||||
OFF = "off"
|
||||
AGENT = "agent"
|
||||
WORKER = "worker"
|
||||
ALL = "all"
|
||||
|
||||
def enables(self, target: MemoryObservationTarget) -> bool:
|
||||
match self:
|
||||
case MemoryObservationWriter.OFF:
|
||||
return False
|
||||
case MemoryObservationWriter.AGENT:
|
||||
return target == MemoryObservationTarget.AGENT
|
||||
case MemoryObservationWriter.WORKER:
|
||||
return target == MemoryObservationTarget.SUBAGENT_WORKER
|
||||
case MemoryObservationWriter.ALL:
|
||||
return target in (
|
||||
MemoryObservationTarget.AGENT,
|
||||
MemoryObservationTarget.SUBAGENT_WORKER,
|
||||
)
|
||||
|
||||
|
||||
DEFAULT_MEMORY_OBSERVATION_WRITER = MemoryObservationWriter.ALL
|
||||
|
||||
# =============================================================================
|
||||
# Configuration paths
|
||||
# =============================================================================
|
||||
@@ -149,6 +184,24 @@ class EvoScientistConfig:
|
||||
# Lower (e.g., 5000) if you want a tighter safety net against runaway loops.
|
||||
recursion_limit: int = 1_000_000
|
||||
|
||||
# Memory Settings
|
||||
# Profile memory injects and maintains `/memories/profile/...` files.
|
||||
memory_profile_enabled: bool = True
|
||||
# Observation memory indexes `/memories/observations/...` and adds
|
||||
# observation-read guidance/context. Writes require this switch plus an
|
||||
# allowed `memory_observation_writer` role below.
|
||||
memory_observations_enabled: bool = True
|
||||
# Which observation-writing path receives the `record_observation` tool:
|
||||
# "off" disables writes; "agent" means live agents; "worker" means the
|
||||
# subagent memory worker; "all" means live agents and the subagent memory
|
||||
# worker. The turn memory worker remains profile-only.
|
||||
memory_observation_writer: MemoryObservationWriter = (
|
||||
DEFAULT_MEMORY_OBSERVATION_WRITER
|
||||
)
|
||||
# Post-turn and post-subagent memory workers. Disable for no-background-memory
|
||||
# controls while still allowing live agents to read configured memory.
|
||||
memory_workers_enabled: bool = True
|
||||
|
||||
# Workspace Settings
|
||||
default_mode: Literal["daemon", "run"] = "daemon"
|
||||
default_workdir: str = ""
|
||||
@@ -325,6 +378,56 @@ class EvoScientistConfig:
|
||||
)
|
||||
self.sandbox_execute_timeout = 300
|
||||
|
||||
try:
|
||||
writer = MemoryObservationWriter(
|
||||
str(self.memory_observation_writer).strip().lower()
|
||||
)
|
||||
except ValueError:
|
||||
logging.getLogger(__name__).warning(
|
||||
"Invalid memory_observation_writer %r; falling back to %s.",
|
||||
self.memory_observation_writer,
|
||||
DEFAULT_MEMORY_OBSERVATION_WRITER.value,
|
||||
)
|
||||
writer = DEFAULT_MEMORY_OBSERVATION_WRITER
|
||||
self.memory_observation_writer = writer
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MemoryControls:
|
||||
"""Resolved memory feature switches used by agent and worker wiring."""
|
||||
|
||||
profile_enabled: bool
|
||||
observations_enabled: bool
|
||||
observation_writer: MemoryObservationWriter
|
||||
workers_enabled: bool
|
||||
|
||||
@classmethod
|
||||
def from_config(cls, config: EvoScientistConfig) -> MemoryControls:
|
||||
return cls(
|
||||
profile_enabled=config.memory_profile_enabled,
|
||||
observations_enabled=config.memory_observations_enabled,
|
||||
observation_writer=config.memory_observation_writer,
|
||||
workers_enabled=config.memory_workers_enabled,
|
||||
)
|
||||
|
||||
@property
|
||||
def memory_enabled(self) -> bool:
|
||||
return self.profile_enabled or self.observations_enabled
|
||||
|
||||
def observation_tool_enabled(self, target: MemoryObservationTarget) -> bool:
|
||||
return self.observations_enabled and self.observation_writer.enables(target)
|
||||
|
||||
def worker_needed(self, target: MemoryObservationTarget) -> bool:
|
||||
if not self.workers_enabled:
|
||||
return False
|
||||
match target:
|
||||
case MemoryObservationTarget.TURN_WORKER:
|
||||
return self.profile_enabled
|
||||
case MemoryObservationTarget.SUBAGENT_WORKER:
|
||||
return self.profile_enabled or self.observation_tool_enabled(target)
|
||||
case MemoryObservationTarget.AGENT:
|
||||
return False
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Config file operations
|
||||
@@ -366,7 +469,7 @@ def save_config(config: EvoScientistConfig) -> None:
|
||||
config_path = get_config_path()
|
||||
config_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
data = asdict(config)
|
||||
data = _config_to_dict(config)
|
||||
|
||||
# Save all fields including empty API keys (users can set them via env vars instead)
|
||||
with open(config_path, "w") as f:
|
||||
@@ -380,6 +483,13 @@ def reset_config() -> None:
|
||||
config_path.unlink()
|
||||
|
||||
|
||||
def _config_to_dict(config: EvoScientistConfig) -> dict[str, Any]:
|
||||
"""Return a plain serializable config dict."""
|
||||
data = asdict(config)
|
||||
data["memory_observation_writer"] = config.memory_observation_writer.value
|
||||
return data
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Config value operations
|
||||
# =============================================================================
|
||||
@@ -420,7 +530,10 @@ def get_config_value(key: str) -> Any:
|
||||
The value, or None if key doesn't exist.
|
||||
"""
|
||||
config = load_config()
|
||||
return getattr(config, key, None)
|
||||
value = getattr(config, key, None)
|
||||
if isinstance(value, MemoryObservationWriter):
|
||||
return value.value
|
||||
return value
|
||||
|
||||
|
||||
def set_config_value(key: str, value: Any) -> bool:
|
||||
@@ -455,6 +568,11 @@ def set_config_value(key: str, value: Any) -> bool:
|
||||
|
||||
if key == "sandbox_execute_timeout" and value <= 0:
|
||||
return False
|
||||
if key == "memory_observation_writer":
|
||||
try:
|
||||
value = MemoryObservationWriter(str(value).strip().lower())
|
||||
except ValueError:
|
||||
return False
|
||||
|
||||
setattr(config, key, value)
|
||||
save_config(config)
|
||||
@@ -467,7 +585,7 @@ def list_config() -> dict[str, Any]:
|
||||
Returns:
|
||||
Dictionary of all configuration key-value pairs.
|
||||
"""
|
||||
return asdict(load_config())
|
||||
return _config_to_dict(load_config())
|
||||
|
||||
|
||||
# =============================================================================
|
||||
@@ -518,6 +636,10 @@ _ENV_MAPPINGS = {
|
||||
"langgraph_dev_file_persistence": "EVOSCIENTIST_LANGGRAPH_DEV_FILE_PERSISTENCE",
|
||||
"langgraph_dev_jobs_per_worker": "EVOSCIENTIST_LANGGRAPH_DEV_JOBS_PER_WORKER",
|
||||
"recursion_limit": "EVOSCIENTIST_RECURSION_LIMIT",
|
||||
"memory_profile_enabled": "EVOSCIENTIST_MEMORY_PROFILE_ENABLED",
|
||||
"memory_observations_enabled": "EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED",
|
||||
"memory_observation_writer": "EVOSCIENTIST_MEMORY_OBSERVATION_WRITER",
|
||||
"memory_workers_enabled": "EVOSCIENTIST_MEMORY_WORKERS_ENABLED",
|
||||
}
|
||||
|
||||
|
||||
@@ -542,7 +664,7 @@ def get_effective_config(
|
||||
|
||||
# Start with file config (includes defaults for missing values)
|
||||
config = load_config()
|
||||
data = asdict(config)
|
||||
data = _config_to_dict(config)
|
||||
|
||||
# Apply environment variable overrides
|
||||
for config_key, env_key in _ENV_MAPPINGS.items():
|
||||
|
||||
@@ -22,7 +22,13 @@ because it follows a different mechanism (re-exporting a lazily-constructed
|
||||
attribute), not the yaml-driven factory.
|
||||
"""
|
||||
|
||||
from EvoScientist.middleware.memory_lifecycle import (
|
||||
MemoryLifecycleRole,
|
||||
build_memory_worker_graph,
|
||||
)
|
||||
from EvoScientist.subagents._factory import build_async_subagent_graph
|
||||
|
||||
writing_agent = build_async_subagent_graph("writing-agent")
|
||||
data_analysis_agent = build_async_subagent_graph("data-analysis-agent")
|
||||
evomemory_subagent_worker = build_memory_worker_graph(MemoryLifecycleRole.SUBAGENT)
|
||||
evomemory_turn_worker = build_memory_worker_graph(MemoryLifecycleRole.TURN)
|
||||
|
||||
@@ -3,7 +3,9 @@
|
||||
"graphs": {
|
||||
"EvoScientist": "EvoScientist.langgraph_dev.main_graph:EvoScientist_agent",
|
||||
"writing-agent": "EvoScientist.langgraph_dev.graphs:writing_agent",
|
||||
"data-analysis-agent": "EvoScientist.langgraph_dev.graphs:data_analysis_agent"
|
||||
"data-analysis-agent": "EvoScientist.langgraph_dev.graphs:data_analysis_agent",
|
||||
"evomemory-subagent-worker": "EvoScientist.langgraph_dev.graphs:evomemory_subagent_worker",
|
||||
"evomemory-turn-worker": "EvoScientist.langgraph_dev.graphs:evomemory_turn_worker"
|
||||
},
|
||||
"config": {
|
||||
"recursion_limit": 1000000
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
"""langgraph dev lifecycle management for async sub-agent support.
|
||||
"""langgraph dev lifecycle management for background agent support.
|
||||
|
||||
Provides functions to start/stop/health-check a ``langgraph dev`` subprocess
|
||||
that hosts the EvoScientist main agent and async sub-agents (e.g.
|
||||
``writing-agent``). The CLI calls ``ensure_langgraph_dev(config, ...)`` at
|
||||
startup so users can run ``EvoSci -p "..."`` without manually managing the
|
||||
langgraph dev server.
|
||||
that hosts the EvoScientist main agent, async sub-agents (e.g.
|
||||
``writing-agent``), and EvoMemory background workers. The CLI calls
|
||||
``ensure_langgraph_dev(config, ...)`` at startup so users can run
|
||||
``EvoSci -p "..."`` without manually managing the langgraph dev server.
|
||||
|
||||
Mirrors the lifecycle pattern used by ``ccproxy_manager.py``.
|
||||
"""
|
||||
@@ -26,11 +26,25 @@ import psutil
|
||||
from filelock import FileLock
|
||||
from filelock import Timeout as FileLockTimeout
|
||||
|
||||
from EvoScientist.config import EvoScientistConfig
|
||||
from EvoScientist.config import (
|
||||
EvoScientistConfig,
|
||||
MemoryControls,
|
||||
MemoryObservationTarget,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def needs_langgraph_dev(config: EvoScientistConfig) -> bool:
|
||||
"""Return whether this config needs the background langgraph dev server."""
|
||||
if config.enable_async_subagents:
|
||||
return True
|
||||
memory_controls = MemoryControls.from_config(config)
|
||||
return memory_controls.worker_needed(
|
||||
MemoryObservationTarget.TURN_WORKER
|
||||
) or memory_controls.worker_needed(MemoryObservationTarget.SUBAGENT_WORKER)
|
||||
|
||||
|
||||
# Reentrant lock guarding ``_PROCESS`` / ``_PROCESS_WORKSPACE`` /
|
||||
# ``_ASYNC_SUBAGENTS_AVAILABLE`` mutations and the ``ensure_langgraph_dev``
|
||||
# decision/start/stop flow. Reentrant because ``ensure_langgraph_dev`` can call
|
||||
@@ -720,13 +734,12 @@ def ensure_langgraph_dev(
|
||||
config: EvoScientistConfig,
|
||||
workspace_dir: Path | str | None = None,
|
||||
) -> subprocess.Popen | None:
|
||||
"""Conditionally start langgraph dev based on ``config.enable_async_subagents``.
|
||||
"""Start or reuse langgraph dev for async/background agent work.
|
||||
|
||||
Behavior:
|
||||
- flag false: no-op, returns None
|
||||
- flag true + already running on the configured port: reuse, returns None
|
||||
- already running on the configured port: reuse, returns None
|
||||
(we don't own it; warns if the workspace can't be verified)
|
||||
- flag true + not running: start subprocess, register atexit cleanup, return Popen
|
||||
- not running: start subprocess, register atexit cleanup, return Popen
|
||||
|
||||
Args:
|
||||
config: Active EvoScientistConfig.
|
||||
@@ -736,10 +749,13 @@ def ensure_langgraph_dev(
|
||||
own ``Path.cwd()`` (the CLI's launch directory).
|
||||
|
||||
Errors during startup are logged but don't abort the CLI — the user can
|
||||
still chat with sync sub-agents; only async sub-agent calls will fail.
|
||||
still chat with sync sub-agents; only async sub-agent calls and EvoMemory
|
||||
background workers will fail.
|
||||
"""
|
||||
global _ASYNC_SUBAGENTS_AVAILABLE
|
||||
if not getattr(config, "enable_async_subagents", False):
|
||||
|
||||
if not needs_langgraph_dev(config):
|
||||
_ASYNC_SUBAGENTS_AVAILABLE = False
|
||||
return None
|
||||
|
||||
# Two layers of locking:
|
||||
@@ -868,11 +884,13 @@ def _ensure_langgraph_dev_locked(
|
||||
except (FileNotFoundError, RuntimeError) as exc:
|
||||
# Startup failed — keep async subagents disabled so the main agent
|
||||
# falls back to in-process sync delegation rather than routing tool
|
||||
# calls at a dead URL.
|
||||
# calls at a dead URL. EvoMemory workers will also skip until the
|
||||
# server is reachable.
|
||||
_ASYNC_SUBAGENTS_AVAILABLE = False
|
||||
logger.warning(
|
||||
"Failed to start langgraph dev — async sub-agents disabled, "
|
||||
"falling back to in-process sync delegation. %s",
|
||||
"Failed to start langgraph dev — async sub-agents will fall back "
|
||||
"to in-process delegation, and EvoMemory background workers will "
|
||||
"not run. %s",
|
||||
exc,
|
||||
)
|
||||
return None
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
"""File-backed memory helpers used by EvoScientist middleware."""
|
||||
|
||||
from .observations import (
|
||||
OBSERVATION_DIR,
|
||||
MemoryScope,
|
||||
MemorySourceType,
|
||||
MemoryType,
|
||||
ObservationRecordResult,
|
||||
RecordObservationArgs,
|
||||
create_record_observation_tool,
|
||||
record_observation_file,
|
||||
)
|
||||
|
||||
__all__ = [
|
||||
"OBSERVATION_DIR",
|
||||
"MemoryScope",
|
||||
"MemorySourceType",
|
||||
"MemoryType",
|
||||
"ObservationRecordResult",
|
||||
"RecordObservationArgs",
|
||||
"create_record_observation_tool",
|
||||
"record_observation_file",
|
||||
]
|
||||
@@ -0,0 +1,429 @@
|
||||
"""File-backed observation memory.
|
||||
|
||||
Observations are small markdown files under `/memories/observations/`. Each
|
||||
file has stable frontmatter for future indexing plus a short body that agents
|
||||
can grep and read with ordinary file tools today.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import json
|
||||
from collections.abc import Mapping
|
||||
from dataclasses import dataclass
|
||||
from datetime import UTC, datetime
|
||||
from enum import StrEnum
|
||||
from pathlib import Path
|
||||
from typing import Annotated, NotRequired, TypedDict
|
||||
|
||||
from langchain.tools import ToolRuntime
|
||||
from langchain_core.tools import BaseTool, InjectedToolArg, StructuredTool
|
||||
from pydantic import BaseModel, ConfigDict, Field
|
||||
|
||||
OBSERVATION_DIR = "/observations"
|
||||
|
||||
|
||||
class MemoryType(StrEnum):
|
||||
"""Kinds of reusable memory an observation can represent."""
|
||||
|
||||
SEMANTIC = "semantic"
|
||||
PROCEDURAL = "procedural"
|
||||
EPISODIC = "episodic"
|
||||
|
||||
|
||||
class MemoryScope(StrEnum):
|
||||
"""Whether an observation is global or tied to the active project."""
|
||||
|
||||
GLOBAL = "global"
|
||||
PROJECT = "project"
|
||||
|
||||
|
||||
class MemorySourceType(StrEnum):
|
||||
"""Where an observation came from in the agent lifecycle."""
|
||||
|
||||
SUBAGENT = "subagent"
|
||||
TURN = "turn"
|
||||
|
||||
|
||||
class ObservationRecordResult(TypedDict):
|
||||
"""Result returned by `record_observation`."""
|
||||
|
||||
observation_id: str
|
||||
path: str
|
||||
created: bool
|
||||
memory_type: MemoryType
|
||||
scope: MemoryScope
|
||||
project_id: NotRequired[str]
|
||||
|
||||
|
||||
class RecordObservationArgs(BaseModel):
|
||||
"""Model-facing arguments for the `record_observation` tool."""
|
||||
|
||||
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||
|
||||
memory_type: MemoryType = Field(
|
||||
description=(
|
||||
"semantic for reusable facts/findings; procedural for reusable "
|
||||
"commands, tool constraints, workarounds, or operating recipes; "
|
||||
"episodic only for notable one-time session events needed for "
|
||||
"future debugging or handoff."
|
||||
),
|
||||
)
|
||||
summary: str = Field(
|
||||
min_length=1,
|
||||
description=(
|
||||
"One-line agent-generated summary used in the observation index. "
|
||||
"Make it specific enough to decide whether to read the full file."
|
||||
),
|
||||
)
|
||||
observation: str = Field(
|
||||
min_length=1,
|
||||
description=(
|
||||
"Concise reusable memory. Do not include raw traces, long citation "
|
||||
"dumps, or claims that are not supported by the trajectory."
|
||||
),
|
||||
)
|
||||
why_it_matters: str = Field(
|
||||
min_length=1,
|
||||
description=(
|
||||
"Why this will matter in future work, including compact evidence "
|
||||
"or provenance when relevant."
|
||||
),
|
||||
)
|
||||
evidence: str | None = Field(
|
||||
default=None,
|
||||
description=(
|
||||
"Optional compact support such as source URLs, arXiv IDs, artifact "
|
||||
"paths, exact commands, or 'observed in this run'. Use this for "
|
||||
"bibliographic, benchmark, or date-sensitive claims."
|
||||
),
|
||||
)
|
||||
scope: MemoryScope = Field(
|
||||
description=(
|
||||
"global for cross-project findings and general tool/platform "
|
||||
"behavior; project only for workspace-specific facts, commands, "
|
||||
"or conventions."
|
||||
),
|
||||
)
|
||||
runtime: Annotated[ToolRuntime | None, InjectedToolArg] = None
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _ObservationContext:
|
||||
"""Concrete source metadata attached to an observation file."""
|
||||
|
||||
project_id: str
|
||||
source_session_id: str
|
||||
source_agent: str
|
||||
source_trajectory_digest: str | None
|
||||
record_tool_call_id: str | None
|
||||
record_worker_agent: str
|
||||
|
||||
|
||||
def _normalize(text: str) -> str:
|
||||
"""Collapse whitespace before deriving the dedupe id."""
|
||||
return " ".join(text.strip().split())
|
||||
|
||||
|
||||
def _observation_id(
|
||||
*,
|
||||
memory_type: MemoryType,
|
||||
scope: MemoryScope,
|
||||
observation: str,
|
||||
why_it_matters: str,
|
||||
) -> str:
|
||||
"""Return a deterministic id for semantically identical observations."""
|
||||
key = "\n".join(
|
||||
[
|
||||
memory_type.value,
|
||||
scope.value,
|
||||
_normalize(observation).casefold(),
|
||||
_normalize(why_it_matters).casefold(),
|
||||
]
|
||||
)
|
||||
digest = hashlib.sha256(key.encode("utf-8")).hexdigest()[:16]
|
||||
return f"O-{digest}"
|
||||
|
||||
|
||||
def _agent_path(memory_path: str) -> str:
|
||||
"""Translate a memory-relative path to the virtual path agents see."""
|
||||
return f"/memories{memory_path}"
|
||||
|
||||
|
||||
def _memory_path(
|
||||
*,
|
||||
observation_id: str,
|
||||
scope: MemoryScope,
|
||||
project_id: str,
|
||||
) -> str:
|
||||
"""Return the memory-relative path for an observation id."""
|
||||
if scope == MemoryScope.PROJECT:
|
||||
return f"{OBSERVATION_DIR}/projects/{project_id}/{observation_id}.md"
|
||||
return f"{OBSERVATION_DIR}/global/{observation_id}.md"
|
||||
|
||||
|
||||
def _json_string(value: str) -> str:
|
||||
"""Render a string as a YAML-safe JSON scalar."""
|
||||
return json.dumps(value, ensure_ascii=False)
|
||||
|
||||
|
||||
def _format_frontmatter(
|
||||
*,
|
||||
observation_id: str,
|
||||
created_at: str,
|
||||
memory_type: MemoryType,
|
||||
summary: str,
|
||||
scope: MemoryScope,
|
||||
source_type: MemorySourceType,
|
||||
source_agent: str,
|
||||
project_id: str,
|
||||
) -> str:
|
||||
"""Build the frontmatter block for an observation file."""
|
||||
lines = [
|
||||
"---",
|
||||
f"id: {_json_string(observation_id)}",
|
||||
f"created_at: {_json_string(created_at)}",
|
||||
f"summary: {_json_string(summary)}",
|
||||
f"memory_type: {memory_type.value}",
|
||||
f"scope: {scope.value}",
|
||||
]
|
||||
if scope == MemoryScope.PROJECT:
|
||||
lines.append(f"project_id: {_json_string(project_id)}")
|
||||
lines.extend(
|
||||
[
|
||||
"source:",
|
||||
f" type: {source_type.value}",
|
||||
f" agent: {_json_string(source_agent)}",
|
||||
]
|
||||
)
|
||||
lines.append("---")
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def _format_observation_markdown(
|
||||
*,
|
||||
observation_id: str,
|
||||
created_at: str,
|
||||
memory_type: MemoryType,
|
||||
summary: str,
|
||||
observation: str,
|
||||
why_it_matters: str,
|
||||
evidence: str | None,
|
||||
scope: MemoryScope,
|
||||
source_type: MemorySourceType,
|
||||
source_agent: str,
|
||||
project_id: str,
|
||||
) -> str:
|
||||
"""Render a complete observation markdown document."""
|
||||
frontmatter = _format_frontmatter(
|
||||
observation_id=observation_id,
|
||||
created_at=created_at,
|
||||
memory_type=memory_type,
|
||||
summary=summary,
|
||||
scope=scope,
|
||||
source_type=source_type,
|
||||
source_agent=source_agent,
|
||||
project_id=project_id,
|
||||
)
|
||||
body = (
|
||||
f"{frontmatter}\n\n"
|
||||
"## Observation\n\n"
|
||||
f"{observation.strip()}\n\n"
|
||||
"## Why It Matters\n\n"
|
||||
f"{why_it_matters.strip()}\n"
|
||||
)
|
||||
if evidence and evidence.strip():
|
||||
body += f"\n## Evidence\n\n{evidence.strip()}\n"
|
||||
return body
|
||||
|
||||
|
||||
def _runtime_config_value(runtime: ToolRuntime | None, key: str) -> str | None:
|
||||
"""Read one optional string override from runtime configurable config."""
|
||||
if runtime is None:
|
||||
return None
|
||||
config = runtime.config or {}
|
||||
if not isinstance(config, Mapping):
|
||||
return None
|
||||
configurable = config.get("configurable", {})
|
||||
if not isinstance(configurable, Mapping):
|
||||
return None
|
||||
value = configurable.get(key)
|
||||
return value if isinstance(value, str) and value else None
|
||||
|
||||
|
||||
def _runtime_session_id(runtime: ToolRuntime | None) -> str:
|
||||
"""Extract the source thread id from tool runtime metadata when present."""
|
||||
source_session_id = _runtime_config_value(runtime, "evomemory_source_session_id")
|
||||
if source_session_id:
|
||||
return source_session_id
|
||||
if runtime is not None:
|
||||
if runtime.execution_info and runtime.execution_info.thread_id:
|
||||
return str(runtime.execution_info.thread_id)
|
||||
thread_id = _runtime_config_value(runtime, "thread_id")
|
||||
if thread_id:
|
||||
return thread_id
|
||||
return "unknown"
|
||||
|
||||
|
||||
def _runtime_tool_call_id(runtime: ToolRuntime | None) -> str | None:
|
||||
"""Extract the active tool call id from runtime metadata when present."""
|
||||
if runtime is None or not runtime.tool_call_id:
|
||||
return None
|
||||
return str(runtime.tool_call_id)
|
||||
|
||||
|
||||
def _resolve_observation_context(
|
||||
runtime: ToolRuntime | None,
|
||||
*,
|
||||
project_id: str,
|
||||
source_agent: str,
|
||||
source_tool_call_id: str | None,
|
||||
) -> _ObservationContext:
|
||||
"""Resolve required observation metadata from fixed values and runtime."""
|
||||
return _ObservationContext(
|
||||
project_id=_runtime_config_value(runtime, "evomemory_project_id") or project_id,
|
||||
source_session_id=_runtime_session_id(runtime),
|
||||
source_agent=_runtime_config_value(runtime, "evomemory_source_agent")
|
||||
or source_agent,
|
||||
source_trajectory_digest=_runtime_config_value(
|
||||
runtime, "evomemory_trajectory_digest"
|
||||
),
|
||||
record_tool_call_id=source_tool_call_id
|
||||
if source_tool_call_id is not None
|
||||
else _runtime_tool_call_id(runtime),
|
||||
record_worker_agent=source_agent,
|
||||
)
|
||||
|
||||
|
||||
def record_observation_file(
|
||||
*,
|
||||
memory_dir: str | Path,
|
||||
project_id: str,
|
||||
memory_type: MemoryType,
|
||||
summary: str,
|
||||
observation: str,
|
||||
why_it_matters: str,
|
||||
scope: MemoryScope,
|
||||
source_type: MemorySourceType,
|
||||
source_session_id: str,
|
||||
source_agent: str,
|
||||
source_trajectory_digest: str | None = None,
|
||||
source_tool_call_id: str | None = None,
|
||||
record_worker_agent: str | None = None,
|
||||
evidence: str | None = None,
|
||||
) -> ObservationRecordResult:
|
||||
"""Create an observation markdown file unless an equivalent one exists.
|
||||
|
||||
The id is derived from the normalized observation text, rationale, type, and
|
||||
scope, so repeated attempts to save the same observation return the existing
|
||||
path instead of creating duplicates.
|
||||
"""
|
||||
|
||||
summary_text = summary.strip()
|
||||
observation_text = observation.strip()
|
||||
why_text = why_it_matters.strip()
|
||||
if not summary_text:
|
||||
raise ValueError("summary must not be empty")
|
||||
if not observation_text:
|
||||
raise ValueError("observation must not be empty")
|
||||
if not why_text:
|
||||
raise ValueError("why_it_matters must not be empty")
|
||||
|
||||
observation_id = _observation_id(
|
||||
memory_type=memory_type,
|
||||
scope=scope,
|
||||
observation=observation_text,
|
||||
why_it_matters=why_text,
|
||||
)
|
||||
memory_path = _memory_path(
|
||||
observation_id=observation_id,
|
||||
scope=scope,
|
||||
project_id=project_id,
|
||||
)
|
||||
path = Path(memory_dir).expanduser() / memory_path.lstrip("/")
|
||||
created = False
|
||||
if not path.exists():
|
||||
created_at = datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
|
||||
content = _format_observation_markdown(
|
||||
observation_id=observation_id,
|
||||
created_at=created_at,
|
||||
memory_type=memory_type,
|
||||
summary=summary_text,
|
||||
observation=observation_text,
|
||||
why_it_matters=why_text,
|
||||
evidence=evidence.strip() if evidence else None,
|
||||
scope=scope,
|
||||
source_type=source_type,
|
||||
source_agent=source_agent,
|
||||
project_id=project_id,
|
||||
)
|
||||
path.parent.mkdir(parents=True, exist_ok=True)
|
||||
path.write_text(content, encoding="utf-8")
|
||||
created = True
|
||||
|
||||
result: ObservationRecordResult = {
|
||||
"observation_id": observation_id,
|
||||
"path": _agent_path(memory_path),
|
||||
"created": created,
|
||||
"memory_type": memory_type,
|
||||
"scope": scope,
|
||||
}
|
||||
if scope == MemoryScope.PROJECT:
|
||||
result["project_id"] = project_id
|
||||
return result
|
||||
|
||||
|
||||
def create_record_observation_tool(
|
||||
*,
|
||||
memory_dir: str | Path,
|
||||
project_id: str,
|
||||
source_type: MemorySourceType,
|
||||
source_agent: str,
|
||||
source_tool_call_id: str | None = None,
|
||||
) -> BaseTool:
|
||||
"""Build the `record_observation` tool for one agent context."""
|
||||
|
||||
def _record_observation(
|
||||
memory_type: MemoryType,
|
||||
summary: str,
|
||||
observation: str,
|
||||
why_it_matters: str,
|
||||
scope: MemoryScope,
|
||||
evidence: str | None = None,
|
||||
runtime: ToolRuntime | None = None,
|
||||
) -> str:
|
||||
context = _resolve_observation_context(
|
||||
runtime,
|
||||
project_id=project_id,
|
||||
source_agent=source_agent,
|
||||
source_tool_call_id=source_tool_call_id,
|
||||
)
|
||||
result = record_observation_file(
|
||||
memory_dir=memory_dir,
|
||||
project_id=context.project_id,
|
||||
memory_type=memory_type,
|
||||
summary=summary,
|
||||
observation=observation,
|
||||
why_it_matters=why_it_matters,
|
||||
evidence=evidence,
|
||||
scope=scope,
|
||||
source_type=source_type,
|
||||
source_session_id=context.source_session_id,
|
||||
source_agent=context.source_agent,
|
||||
source_trajectory_digest=context.source_trajectory_digest,
|
||||
source_tool_call_id=context.record_tool_call_id,
|
||||
record_worker_agent=context.record_worker_agent,
|
||||
)
|
||||
return json.dumps(result, ensure_ascii=False, sort_keys=True)
|
||||
|
||||
return StructuredTool.from_function(
|
||||
func=_record_observation,
|
||||
name="record_observation",
|
||||
description=(
|
||||
"Record compact reusable memory as a structured EvoMemory "
|
||||
"observation markdown file. Use procedural/global for reusable "
|
||||
"tool or platform behavior unless it is project-specific."
|
||||
),
|
||||
args_schema=RecordObservationArgs,
|
||||
infer_schema=False,
|
||||
)
|
||||
@@ -0,0 +1,164 @@
|
||||
"""Process-local activity tracking for EvoMemory workers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import hashlib
|
||||
import threading
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MemoryWorkerStatusSnapshot:
|
||||
"""Completed memory writes shown in the status bar."""
|
||||
|
||||
is_running: bool = False
|
||||
profile_updates: int = 0
|
||||
observations_recorded: int = 0
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class MemoryOutputSnapshot:
|
||||
profile_files: dict[str, str]
|
||||
observation_files: frozenset[str]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class _ActiveMemoryWorker:
|
||||
memory_dir: Path
|
||||
before_outputs: MemoryOutputSnapshot
|
||||
|
||||
|
||||
_active_runs: dict[tuple[str, str], _ActiveMemoryWorker] = {}
|
||||
_active_lock = threading.Lock()
|
||||
_profile_updates = 0
|
||||
_observations_recorded = 0
|
||||
_counted_profile_versions: set[tuple[str, str, str]] = set()
|
||||
_counted_observation_files: set[tuple[str, str]] = set()
|
||||
|
||||
|
||||
def _file_digest(path: Path) -> str | None:
|
||||
try:
|
||||
return hashlib.sha256(path.read_bytes()).hexdigest()
|
||||
except OSError:
|
||||
return None
|
||||
|
||||
|
||||
def snapshot_memory_outputs(memory_dir: str | Path) -> MemoryOutputSnapshot:
|
||||
root = Path(memory_dir).expanduser()
|
||||
profile_root = root / "profile"
|
||||
observation_root = root / "observations"
|
||||
|
||||
profile_files: dict[str, str] = {}
|
||||
if profile_root.exists():
|
||||
for path in profile_root.rglob("*.md"):
|
||||
if not path.is_file():
|
||||
continue
|
||||
digest = _file_digest(path)
|
||||
if digest is not None:
|
||||
profile_files[str(path.relative_to(root))] = digest
|
||||
|
||||
observation_files: set[str] = set()
|
||||
if observation_root.exists():
|
||||
for path in observation_root.rglob("*.md"):
|
||||
if path.is_file():
|
||||
observation_files.add(str(path.relative_to(root)))
|
||||
|
||||
return MemoryOutputSnapshot(
|
||||
profile_files=profile_files,
|
||||
observation_files=frozenset(observation_files),
|
||||
)
|
||||
|
||||
|
||||
def _memory_output_delta(
|
||||
memory_dir: str | Path,
|
||||
before: MemoryOutputSnapshot,
|
||||
after: MemoryOutputSnapshot,
|
||||
) -> tuple[set[tuple[str, str, str]], set[tuple[str, str]]]:
|
||||
root_key = str(Path(memory_dir).expanduser().resolve())
|
||||
profile_versions = {
|
||||
(root_key, path, digest)
|
||||
for path, digest in after.profile_files.items()
|
||||
if before.profile_files.get(path) != digest
|
||||
}
|
||||
observation_files = {
|
||||
(root_key, path) for path in after.observation_files - before.observation_files
|
||||
}
|
||||
return profile_versions, observation_files
|
||||
|
||||
|
||||
def memory_worker_status() -> MemoryWorkerStatusSnapshot:
|
||||
with _active_lock:
|
||||
return MemoryWorkerStatusSnapshot(
|
||||
is_running=bool(_active_runs),
|
||||
profile_updates=_profile_updates,
|
||||
observations_recorded=_observations_recorded,
|
||||
)
|
||||
|
||||
|
||||
def clear_memory_worker_saved_counts() -> None:
|
||||
"""Clear completed memory-save counters while preserving active workers."""
|
||||
global _observations_recorded, _profile_updates
|
||||
|
||||
with _active_lock:
|
||||
_profile_updates = 0
|
||||
_observations_recorded = 0
|
||||
|
||||
|
||||
def mark_memory_worker_started(
|
||||
*,
|
||||
thread_id: str,
|
||||
run_id: str,
|
||||
memory_dir: str | Path,
|
||||
before_outputs: MemoryOutputSnapshot | None = None,
|
||||
) -> None:
|
||||
memory_root = Path(memory_dir).expanduser()
|
||||
before = before_outputs or snapshot_memory_outputs(memory_root)
|
||||
with _active_lock:
|
||||
_active_runs[(thread_id, run_id)] = _ActiveMemoryWorker(
|
||||
memory_dir=memory_root,
|
||||
before_outputs=before,
|
||||
)
|
||||
|
||||
|
||||
def forget_memory_worker(thread_id: str, run_id: str) -> None:
|
||||
"""Stop tracking a worker without counting memory-output deltas."""
|
||||
with _active_lock:
|
||||
_active_runs.pop((thread_id, run_id), None)
|
||||
|
||||
|
||||
def mark_memory_worker_finished(thread_id: str, run_id: str) -> None:
|
||||
global _observations_recorded, _profile_updates
|
||||
|
||||
with _active_lock:
|
||||
worker = _active_runs.pop((thread_id, run_id), None)
|
||||
if worker is None:
|
||||
return
|
||||
|
||||
after = snapshot_memory_outputs(worker.memory_dir)
|
||||
profile_versions, observation_files = _memory_output_delta(
|
||||
worker.memory_dir,
|
||||
worker.before_outputs,
|
||||
after,
|
||||
)
|
||||
if not profile_versions and not observation_files:
|
||||
return
|
||||
|
||||
with _active_lock:
|
||||
new_profile_versions = profile_versions - _counted_profile_versions
|
||||
new_observation_files = observation_files - _counted_observation_files
|
||||
_counted_profile_versions.update(new_profile_versions)
|
||||
_counted_observation_files.update(new_observation_files)
|
||||
_profile_updates += len(new_profile_versions)
|
||||
_observations_recorded += len(new_observation_files)
|
||||
|
||||
|
||||
def reset_memory_worker_status_for_tests() -> None:
|
||||
global _observations_recorded, _profile_updates
|
||||
|
||||
with _active_lock:
|
||||
_active_runs.clear()
|
||||
_counted_profile_versions.clear()
|
||||
_counted_observation_files.clear()
|
||||
_profile_updates = 0
|
||||
_observations_recorded = 0
|
||||
@@ -22,6 +22,11 @@ from .memory import (
|
||||
EvoMemoryMiddleware,
|
||||
create_memory_middleware,
|
||||
)
|
||||
from .memory_lifecycle import (
|
||||
EvoMemoryLifecycleMiddleware,
|
||||
MemoryLifecycleRole,
|
||||
create_memory_lifecycle_middleware,
|
||||
)
|
||||
from .model_fallback import ModelFallbackMiddleware, load_fallback_chain
|
||||
from .runtime_context import RuntimeContextMiddleware, create_runtime_context_middleware
|
||||
from .tool_error_handler import ToolErrorHandlerMiddleware
|
||||
@@ -35,7 +40,9 @@ __all__ = [
|
||||
"Choice",
|
||||
"ConfigurableModelMiddleware",
|
||||
"ContextOverflowMapperMiddleware",
|
||||
"EvoMemoryLifecycleMiddleware",
|
||||
"EvoMemoryMiddleware",
|
||||
"MemoryLifecycleRole",
|
||||
"ModelFallbackMiddleware",
|
||||
"Question",
|
||||
"RuntimeContextMiddleware",
|
||||
@@ -43,6 +50,7 @@ __all__ = [
|
||||
"compute_context_editing_trigger",
|
||||
"create_code_interpreter_middleware",
|
||||
"create_context_editing_middleware",
|
||||
"create_memory_lifecycle_middleware",
|
||||
"create_memory_middleware",
|
||||
"create_runtime_context_middleware",
|
||||
"create_tool_selector_middleware",
|
||||
|
||||
@@ -2,9 +2,10 @@
|
||||
|
||||
The middleware owns the markdown files under ``/memories/profile/``: it creates
|
||||
them when missing, migrates the old ``/memories/MEMORY.md`` file when present,
|
||||
and injects either profile contents or profile file pointers into model calls.
|
||||
Agents still read and edit the files through their normal ``/memories/...``
|
||||
tools; this middleware only handles setup and prompt context.
|
||||
injects either profile contents or profile file pointers into model calls, and
|
||||
points agents at observation memory under ``/memories/observations/``. Agents
|
||||
still read and edit profile files through their normal ``/memories/...`` tools;
|
||||
observation writes go through the structured ``record_observation`` tool.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
@@ -14,8 +15,10 @@ import logging
|
||||
import re
|
||||
import subprocess
|
||||
from collections.abc import Awaitable, Callable
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
|
||||
import yaml
|
||||
from langchain.agents.middleware.types import (
|
||||
AgentMiddleware,
|
||||
ModelRequest,
|
||||
@@ -23,10 +26,17 @@ from langchain.agents.middleware.types import (
|
||||
)
|
||||
|
||||
from .. import paths as _paths
|
||||
from ..memory import (
|
||||
MemoryScope,
|
||||
MemorySourceType,
|
||||
MemoryType,
|
||||
create_record_observation_tool,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_MAX_INLINE_PROFILE_CHARS = 24_000
|
||||
DEFAULT_MAX_INLINE_OBSERVATION_INDEX_CHARS = 12_000
|
||||
_LEGACY_MEMORY_FILENAME = "MEMORY.md"
|
||||
_LEGACY_IMPORT_HEADING = "Imported from legacy MEMORY.md"
|
||||
|
||||
@@ -34,8 +44,9 @@ _LEGACY_IMPORT_HEADING = "Imported from legacy MEMORY.md"
|
||||
PROFILE_INJECTION_TEMPLATE = """<profile_memory>
|
||||
{profile_content}
|
||||
</profile_memory>
|
||||
{observation_memory}
|
||||
|
||||
<profile_memory_instructions>
|
||||
<memory_instructions>
|
||||
These profile notes live under `/memories/profile/`.
|
||||
Every agent can read and update them with normal file tools.
|
||||
|
||||
@@ -47,7 +58,59 @@ Use these files for:
|
||||
|
||||
Read the relevant file before editing it. Add small bullets under existing
|
||||
headings, skip duplicates, and leave out temporary task state.
|
||||
</profile_memory_instructions>"""
|
||||
|
||||
Profile update scope:
|
||||
- Review the profile context above and the latest trajectory for stable changes
|
||||
to user preferences, research taste, collaboration style, or project
|
||||
conventions.
|
||||
- Do not infer profile facts from task content alone. Profile updates need
|
||||
stable evidence about the user, their preferences, or this project.
|
||||
- When a profile update is warranted, edit the relevant
|
||||
`/memories/profile/...` file with a small deduplicated bullet under an
|
||||
existing heading.
|
||||
- When the turn only contains task progress, subagent findings, search results,
|
||||
command output, or temporary run context, leave profile files unchanged.
|
||||
{observation_instructions}
|
||||
</memory_instructions>"""
|
||||
|
||||
OBSERVATION_MEMORY_READ_INSTRUCTIONS = """
|
||||
Observation memory lives under `/memories/observations/`:
|
||||
- `/memories/observations/global/`: cross-project observations.
|
||||
- `/memories/observations/projects/{project_id}/`: observations for this workspace.
|
||||
|
||||
Memory preflight:
|
||||
- For main-agent and subagent work, before planning, running commands,
|
||||
implementing, debugging, analyzing, or writing reports, run a quick search of
|
||||
observation memory unless the task is clearly trivial or the observation
|
||||
directories are empty.
|
||||
- Use file tools, not shell paths: start with `grep` on `/memories/observations/`
|
||||
using task keywords, then `read_file` the relevant hits by id/path. Use
|
||||
`glob` or `ls` only to inspect what exists when grep returns nothing useful.
|
||||
- When the task calls for a specific kind of memory, grep frontmatter first:
|
||||
`memory_type: procedural` for reusable commands/workarounds, `memory_type:
|
||||
semantic` for reusable facts/findings, `scope: project` for workspace-local
|
||||
notes, and `scope: global` for cross-project notes.
|
||||
- Mention the result briefly in your plan or handoff: which observation mattered,
|
||||
or that no relevant observation was found. Do not let this become a long detour.
|
||||
"""
|
||||
|
||||
OBSERVATION_MEMORY_WRITE_INSTRUCTIONS = """
|
||||
Call `record_observation` only for durable, non-obvious, evidence-backed
|
||||
information that is not already in memory and is likely to change future behavior:
|
||||
recurring constraints, important decisions, failed approaches future agents might
|
||||
repeat, verified evaluator outcomes, or tool/workflow workarounds.
|
||||
Provide a one-line `summary` that is specific enough for future agents to decide
|
||||
whether to read the full observation.
|
||||
|
||||
Distill reusable insight rather than saving raw task output or a transcript of
|
||||
what happened.
|
||||
|
||||
Use procedural/global for general tool or platform behavior that can recur
|
||||
outside this workspace; use project scope only for workspace-specific facts,
|
||||
commands, datasets, benchmarks, or config. Do not hand-write observation files.
|
||||
Do not record routine progress, raw traces, ordinary command output, citation
|
||||
lists without synthesis, simple filesystem listings, temporary paths/run ids,
|
||||
one-off environment discoveries, or task summaries."""
|
||||
|
||||
PROFILE_TEMPLATES: dict[str, str] = {
|
||||
"/profile/SOUL.md": """# EvoScientist soul
|
||||
@@ -74,8 +137,7 @@ Things worth remembering about the person using EvoScientist.
|
||||
""",
|
||||
"/profile/RESEARCH_TASTE.md": """# Research taste
|
||||
|
||||
Research taste to keep in mind: interests, standards, methods that tend to fit,
|
||||
and things to avoid.
|
||||
Research taste to keep in mind: interests, standards, methods that tend to fit, and things to avoid.
|
||||
|
||||
## Interests
|
||||
|
||||
@@ -207,6 +269,17 @@ def _append_imported_section(content: str, body: str) -> str:
|
||||
return content.rstrip() + f"\n\n## {_LEGACY_IMPORT_HEADING}\n\n{body.strip()}\n"
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ObservationIndexRecord:
|
||||
"""One summary-bearing observation listed in the system prompt index."""
|
||||
|
||||
observation_id: str
|
||||
memory_path: str
|
||||
memory_type: MemoryType
|
||||
scope: MemoryScope
|
||||
summary: str
|
||||
|
||||
|
||||
class EvoMemoryMiddleware(AgentMiddleware):
|
||||
"""Middleware that maintains the profile memory files used by EvoScientist.
|
||||
|
||||
@@ -220,10 +293,17 @@ class EvoMemoryMiddleware(AgentMiddleware):
|
||||
memory_dir: str | Path,
|
||||
workspace_dir: str | Path | None = None,
|
||||
max_inline_profile_chars: int = DEFAULT_MAX_INLINE_PROFILE_CHARS,
|
||||
source_type: MemorySourceType = MemorySourceType.TURN,
|
||||
source_agent: str = "EvoScientist",
|
||||
enable_profile_memory: bool = True,
|
||||
enable_observation_memory: bool = True,
|
||||
enable_observation_tool: bool = True,
|
||||
) -> None:
|
||||
self._memory_dir = Path(memory_dir).expanduser()
|
||||
workspace = Path(workspace_dir or _paths.WORKSPACE_ROOT).expanduser()
|
||||
self._project_id = _resolve_project_id(workspace)
|
||||
self._enable_profile_memory = enable_profile_memory
|
||||
self._enable_observation_memory = enable_observation_memory
|
||||
self._profile_specs = _profile_specs(self._project_id)
|
||||
pointer_lines = ["Profile files are available at:"]
|
||||
pointer_lines.extend(
|
||||
@@ -231,6 +311,36 @@ class EvoMemoryMiddleware(AgentMiddleware):
|
||||
)
|
||||
self._profile_pointer_context = "\n".join(pointer_lines)
|
||||
self._max_inline_profile_chars = max_inline_profile_chars
|
||||
self._enable_observation_tool = (
|
||||
enable_observation_memory and enable_observation_tool
|
||||
)
|
||||
self.tools = (
|
||||
[
|
||||
create_record_observation_tool(
|
||||
memory_dir=self._memory_dir,
|
||||
project_id=self._project_id,
|
||||
source_type=source_type,
|
||||
source_agent=source_agent,
|
||||
)
|
||||
]
|
||||
if self._enable_observation_tool
|
||||
else []
|
||||
)
|
||||
self._observation_index_records = []
|
||||
self._observation_index_context = ""
|
||||
if not enable_observation_memory:
|
||||
return
|
||||
|
||||
self._ensure_observation_dirs()
|
||||
self._observation_index_records = self._read_observation_index_records()
|
||||
self._observation_index_context = self._observation_index_context_from_records(
|
||||
self._observation_index_records
|
||||
)
|
||||
|
||||
@property
|
||||
def project_id(self) -> str:
|
||||
"""Stable project id used for this middleware's project memory paths."""
|
||||
return self._project_id
|
||||
|
||||
def _file_path(self, memory_path: str) -> Path:
|
||||
"""Resolve a memory-relative path against the memory directory."""
|
||||
@@ -267,6 +377,17 @@ class EvoMemoryMiddleware(AgentMiddleware):
|
||||
return False
|
||||
return True
|
||||
|
||||
def _ensure_observation_dirs(self) -> None:
|
||||
"""Create the observation directories agents are prompted to search."""
|
||||
for memory_path in (
|
||||
"/observations/global",
|
||||
f"/observations/projects/{self._project_id}",
|
||||
):
|
||||
try:
|
||||
self._file_path(memory_path).mkdir(parents=True, exist_ok=True)
|
||||
except OSError as e:
|
||||
logger.warning("Failed to create observation memory dir: %s", e)
|
||||
|
||||
def _ensure_profile_files(self) -> list[tuple[str, str]]:
|
||||
"""Create the expected profile files if needed and return their contents."""
|
||||
records = []
|
||||
@@ -338,8 +459,7 @@ class EvoMemoryMiddleware(AgentMiddleware):
|
||||
|
||||
return self._delete_legacy_memory(legacy_path)
|
||||
|
||||
def _read_profile_records(self) -> list[tuple[str, str]]:
|
||||
"""Load all profile files after bootstrapping and legacy migration."""
|
||||
def _read_bootstrapped_profile_records(self) -> list[tuple[str, str]]:
|
||||
records = self._ensure_profile_files()
|
||||
if self._migrate_legacy_memory():
|
||||
records = [
|
||||
@@ -348,6 +468,14 @@ class EvoMemoryMiddleware(AgentMiddleware):
|
||||
]
|
||||
return records
|
||||
|
||||
def _read_profile_records(self) -> list[tuple[str, str]]:
|
||||
"""Load all profile files after bootstrapping and legacy migration."""
|
||||
if not self._enable_observation_memory:
|
||||
return self._read_bootstrapped_profile_records()
|
||||
|
||||
self._ensure_observation_dirs()
|
||||
return self._read_bootstrapped_profile_records()
|
||||
|
||||
def _profile_context_from_records(self, records: list[tuple[str, str]]) -> str:
|
||||
"""Inline profile contents unless they exceed the prompt budget."""
|
||||
full = "\n\n".join(
|
||||
@@ -360,40 +488,243 @@ class EvoMemoryMiddleware(AgentMiddleware):
|
||||
return self._profile_pointer_context
|
||||
|
||||
def _read_profile_memory(self) -> str:
|
||||
"""Return profile context, falling back to file pointers on setup errors."""
|
||||
"""Return profile context, falling back to file pointers."""
|
||||
try:
|
||||
records = self._read_profile_records()
|
||||
return self._profile_context_from_records(records)
|
||||
return (
|
||||
self._profile_context_from_records(records)
|
||||
or self._profile_pointer_context
|
||||
)
|
||||
except Exception as e:
|
||||
logger.debug("Failed to read profile memory: %s", e)
|
||||
return self._profile_pointer_context
|
||||
|
||||
def _observation_memory_paths(self) -> list[Path]:
|
||||
"""Return summary-indexable observation files for this project context."""
|
||||
paths: list[Path] = []
|
||||
for memory_path in (
|
||||
"/observations/global",
|
||||
f"/observations/projects/{self._project_id}",
|
||||
):
|
||||
directory = self._file_path(memory_path)
|
||||
try:
|
||||
paths.extend(sorted(directory.glob("*.md")))
|
||||
except OSError as e:
|
||||
logger.warning("Failed to list observation memory %s: %s", directory, e)
|
||||
return paths
|
||||
|
||||
def _read_observation_frontmatter(self, path: Path) -> dict[str, object] | None:
|
||||
"""Read explicit YAML frontmatter for an observation file."""
|
||||
try:
|
||||
text = path.read_text(encoding="utf-8")
|
||||
except (OSError, UnicodeDecodeError) as e:
|
||||
logger.warning("Failed to read observation memory %s: %s", path, e)
|
||||
return None
|
||||
if not text.startswith("---\n"):
|
||||
return None
|
||||
try:
|
||||
frontmatter, _body = text.removeprefix("---\n").split("\n---\n", 1)
|
||||
metadata = yaml.safe_load(frontmatter)
|
||||
except (ValueError, yaml.YAMLError):
|
||||
return None
|
||||
if not isinstance(metadata, dict):
|
||||
return None
|
||||
return {key: value for key, value in metadata.items() if isinstance(key, str)}
|
||||
|
||||
def _observation_index_record_from_path(
|
||||
self, path: Path
|
||||
) -> ObservationIndexRecord | None:
|
||||
"""Return an index record only when explicit summary metadata exists."""
|
||||
metadata = self._read_observation_frontmatter(path)
|
||||
if metadata is None:
|
||||
return None
|
||||
|
||||
observation_id = str(metadata.get("id") or "").strip()
|
||||
summary = str(metadata.get("summary") or "").strip()
|
||||
memory_type_value = str(metadata.get("memory_type") or "").strip()
|
||||
scope_value = str(metadata.get("scope") or "").strip()
|
||||
if (
|
||||
not observation_id
|
||||
or not summary
|
||||
or not memory_type_value
|
||||
or not scope_value
|
||||
):
|
||||
return None
|
||||
try:
|
||||
memory_type = MemoryType(memory_type_value)
|
||||
scope = MemoryScope(scope_value)
|
||||
except ValueError:
|
||||
return None
|
||||
|
||||
try:
|
||||
memory_path = "/" + path.relative_to(self._memory_dir).as_posix()
|
||||
except ValueError:
|
||||
return None
|
||||
return ObservationIndexRecord(
|
||||
observation_id=observation_id,
|
||||
memory_path=memory_path,
|
||||
memory_type=memory_type,
|
||||
scope=scope,
|
||||
summary=summary,
|
||||
)
|
||||
|
||||
def _read_observation_index_records(self) -> list[ObservationIndexRecord]:
|
||||
"""Load summary-bearing observation records for prompt indexing."""
|
||||
records = [
|
||||
record
|
||||
for path in self._observation_memory_paths()
|
||||
if (record := self._observation_index_record_from_path(path)) is not None
|
||||
]
|
||||
return sorted(records, key=lambda record: record.observation_id)
|
||||
|
||||
def _observation_index_count_line(
|
||||
self, records: list[ObservationIndexRecord]
|
||||
) -> str:
|
||||
"""Return compact observation counts by scope and memory type."""
|
||||
scope_counts = dict.fromkeys(MemoryScope, 0)
|
||||
type_counts = dict.fromkeys(MemoryType, 0)
|
||||
for record in records:
|
||||
scope_counts[record.scope] += 1
|
||||
type_counts[record.memory_type] += 1
|
||||
return (
|
||||
f"Counts: total={len(records)}; "
|
||||
f"scope global={scope_counts[MemoryScope.GLOBAL]}, "
|
||||
f"project={scope_counts[MemoryScope.PROJECT]}; "
|
||||
f"type semantic={type_counts[MemoryType.SEMANTIC]}, "
|
||||
f"procedural={type_counts[MemoryType.PROCEDURAL]}, "
|
||||
f"episodic={type_counts[MemoryType.EPISODIC]}."
|
||||
)
|
||||
|
||||
def _observation_search_hints(self) -> str:
|
||||
"""Return stable search hints for observation memory."""
|
||||
return "\n".join(
|
||||
[
|
||||
"Search hints:",
|
||||
"- Grep by id when you already know it from the index.",
|
||||
(
|
||||
"- Grep frontmatter by type when appropriate: "
|
||||
"`memory_type: procedural`, `memory_type: semantic`, or "
|
||||
"`memory_type: episodic`."
|
||||
),
|
||||
(
|
||||
"- Grep frontmatter by scope when appropriate: "
|
||||
"`scope: project` or `scope: global`."
|
||||
),
|
||||
"- Combine those with task keywords, then read relevant hits.",
|
||||
]
|
||||
)
|
||||
|
||||
def _observation_index_context_from_records(
|
||||
self,
|
||||
records: list[ObservationIndexRecord],
|
||||
*,
|
||||
max_inline_chars: int = DEFAULT_MAX_INLINE_OBSERVATION_INDEX_CHARS,
|
||||
) -> str:
|
||||
"""Build the static observation index injected into the system prompt."""
|
||||
header = "\n".join(
|
||||
[
|
||||
"<observation_memory>",
|
||||
"Observation index loaded at agent start.",
|
||||
self._observation_index_count_line(records),
|
||||
]
|
||||
)
|
||||
if not records:
|
||||
return "\n".join(
|
||||
[header, self._observation_search_hints(), "</observation_memory>"]
|
||||
)
|
||||
|
||||
lines = [
|
||||
f"- {record.observation_id} "
|
||||
f"[{record.memory_type.value}/{record.scope.value}] "
|
||||
f"{_agent_path(record.memory_path)}: {record.summary}"
|
||||
for record in records
|
||||
]
|
||||
full = "\n".join(
|
||||
[
|
||||
header,
|
||||
"Indexed observations:",
|
||||
*lines,
|
||||
self._observation_search_hints(),
|
||||
"</observation_memory>",
|
||||
]
|
||||
)
|
||||
if len(full) <= max_inline_chars:
|
||||
return full
|
||||
return "\n".join(
|
||||
[
|
||||
header,
|
||||
"Observation summaries are too large to inline; search on demand.",
|
||||
self._observation_search_hints(),
|
||||
"</observation_memory>",
|
||||
]
|
||||
)
|
||||
|
||||
def _observation_memory_instructions(self) -> str:
|
||||
if not self._enable_observation_memory:
|
||||
return ""
|
||||
|
||||
instructions = OBSERVATION_MEMORY_READ_INSTRUCTIONS.format(
|
||||
project_id=self._project_id
|
||||
)
|
||||
if not self._enable_observation_tool:
|
||||
return instructions
|
||||
return instructions + OBSERVATION_MEMORY_WRITE_INSTRUCTIONS
|
||||
|
||||
def _inject_profile_context(
|
||||
self, request: ModelRequest, profile_content: str
|
||||
) -> ModelRequest:
|
||||
"""Append profile context and editing guidance to the system prompt."""
|
||||
from deepagents.middleware._utils import append_to_system_message
|
||||
|
||||
if not self._enable_profile_memory and not self._enable_observation_memory:
|
||||
return request
|
||||
|
||||
observation_instructions = self._observation_memory_instructions()
|
||||
|
||||
if not self._enable_profile_memory:
|
||||
injection = "\n\n".join(
|
||||
part
|
||||
for part in (
|
||||
self._observation_index_context,
|
||||
(
|
||||
"<memory_instructions>\n"
|
||||
f"{observation_instructions.strip()}\n"
|
||||
"</memory_instructions>"
|
||||
)
|
||||
if observation_instructions.strip()
|
||||
else "",
|
||||
)
|
||||
if part
|
||||
)
|
||||
new_system = append_to_system_message(request.system_message, injection)
|
||||
return request.override(system_message=new_system)
|
||||
|
||||
injection = PROFILE_INJECTION_TEMPLATE.format(
|
||||
profile_content=profile_content,
|
||||
observation_memory=self._observation_index_context,
|
||||
project_id=self._project_id,
|
||||
observation_instructions=observation_instructions,
|
||||
)
|
||||
new_system = append_to_system_message(request.system_message, injection)
|
||||
return request.override(system_message=new_system)
|
||||
|
||||
def _profile_context_for_request(self) -> str:
|
||||
if not self._enable_profile_memory:
|
||||
return ""
|
||||
return self._read_profile_memory()
|
||||
|
||||
def modify_request(self, request: ModelRequest) -> ModelRequest:
|
||||
"""Apply profile memory injection for synchronous model calls."""
|
||||
profile_content = self._read_profile_memory()
|
||||
if not profile_content:
|
||||
profile_content = self._profile_pointer_context
|
||||
return self._inject_profile_context(request, profile_content)
|
||||
"""Apply memory injection for synchronous model calls."""
|
||||
return self._inject_profile_context(
|
||||
request, self._profile_context_for_request()
|
||||
)
|
||||
|
||||
async def amodify_request(self, request: ModelRequest) -> ModelRequest:
|
||||
"""Async profile injection; file reads run off the event loop."""
|
||||
profile_content = await asyncio.to_thread(self._read_profile_memory)
|
||||
if not profile_content:
|
||||
profile_content = self._profile_pointer_context
|
||||
return self._inject_profile_context(request, profile_content)
|
||||
"""Apply memory injection for asynchronous model calls."""
|
||||
profile_context = ""
|
||||
if self._enable_profile_memory:
|
||||
profile_context = await asyncio.to_thread(self._read_profile_memory)
|
||||
return self._inject_profile_context(request, profile_context)
|
||||
|
||||
def wrap_model_call(
|
||||
self,
|
||||
@@ -416,6 +747,11 @@ def create_memory_middleware(
|
||||
memory_dir: str | None = None,
|
||||
workspace_dir: str | Path | None = None,
|
||||
max_inline_profile_chars: int = DEFAULT_MAX_INLINE_PROFILE_CHARS,
|
||||
source_type: MemorySourceType = MemorySourceType.TURN,
|
||||
source_agent: str = "EvoScientist",
|
||||
enable_profile_memory: bool = True,
|
||||
enable_observation_memory: bool = True,
|
||||
enable_observation_tool: bool = True,
|
||||
) -> EvoMemoryMiddleware:
|
||||
"""Build profile-memory middleware, defaulting to the shared memories directory."""
|
||||
|
||||
@@ -426,4 +762,9 @@ def create_memory_middleware(
|
||||
memory_dir=memory_dir,
|
||||
workspace_dir=workspace_dir,
|
||||
max_inline_profile_chars=max_inline_profile_chars,
|
||||
source_type=source_type,
|
||||
source_agent=source_agent,
|
||||
enable_profile_memory=enable_profile_memory,
|
||||
enable_observation_memory=enable_observation_memory,
|
||||
enable_observation_tool=enable_observation_tool,
|
||||
)
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
+72
-26
@@ -45,7 +45,26 @@ You help researchers move from question to publishable contribution. That spans
|
||||
# own constants below to keep this section focused on flow)
|
||||
# =============================================================================
|
||||
|
||||
EXPERIMENT_WORKFLOW = """# Experiment Workflow
|
||||
_OBSERVATION_MEMORY_INTAKE_STEP = (
|
||||
"- When prior work may matter, search `/memories/observations/` for saved "
|
||||
"findings, failed attempts, commands, and decisions. Incorporate relevant "
|
||||
"observations into planning. Skip this when there is no useful memory yet."
|
||||
)
|
||||
|
||||
_MEMORY_EVOLUTION_SECTION = """### Memory Evolution (after significant outcomes)
|
||||
After meaningful research, implementation, evaluation, or debugging outcomes,
|
||||
consider whether a compact reusable note passes the memory bar before calling
|
||||
`record_observation`. Most outcomes should stay in the final answer, artifacts,
|
||||
or execution summary. Use observation memory only for durable, non-obvious,
|
||||
evidence-backed findings, decisions, failed approaches, tool constraints,
|
||||
evaluator outcomes, or project lessons that are likely to change future
|
||||
behavior. Distill reusable insight rather than saving raw task output or a
|
||||
transcript of what happened. When you call `record_observation`, include a
|
||||
one-line `summary` that lets future agents decide whether to read the full
|
||||
observation.
|
||||
"""
|
||||
|
||||
_EXPERIMENT_WORKFLOW_PREAMBLE = """# Experiment Workflow
|
||||
|
||||
When the task is to plan, run, or report on experiments, follow the workflow below.
|
||||
|
||||
@@ -76,14 +95,21 @@ Not every project needs all steps. Match the starting point to what the user alr
|
||||
- Apply multiple-testing correction when comparing many conditions.
|
||||
- State limitations, negative results, and sensitivity to key parameters.
|
||||
- Track reproducibility (seeds, versions, configs, and exact commands).
|
||||
"""
|
||||
|
||||
## Step 1: Intake & Scope
|
||||
- Read the proposal and extract goals, datasets, constraints, and evaluation metrics.
|
||||
- Capture key assumptions and open questions.
|
||||
- Check `/memories/` for prior research knowledge: `ideation-memory.md` (known promising and failed directions) and `experiment-memory.md` (proven strategies from past cycles). Incorporate relevant findings into planning. Skip if these files do not exist yet.
|
||||
- Save the original proposal to `/research_request.md`.
|
||||
|
||||
## Step 2: Plan (Recommended Structure)
|
||||
def _build_intake_scope(*, enable_observation_memory: bool) -> str:
|
||||
bullets = [
|
||||
"- Read the proposal and extract goals, datasets, constraints, and evaluation metrics.",
|
||||
"- Capture key assumptions and open questions.",
|
||||
]
|
||||
if enable_observation_memory:
|
||||
bullets.append(_OBSERVATION_MEMORY_INTAKE_STEP)
|
||||
bullets.append("- Save the original proposal to `/research_request.md`.")
|
||||
return "\n".join(["## Step 1: Intake & Scope", *bullets])
|
||||
|
||||
|
||||
_EXPERIMENT_WORKFLOW_EXECUTION = """## Step 2: Plan (Recommended Structure)
|
||||
- Create experiment stages with success signals (flexible, not rigid).
|
||||
- Identify resource/data dependencies and baseline requirements.
|
||||
- Use `write_todos` to track the execution plan and updates.
|
||||
@@ -131,17 +157,10 @@ Before delegating code tasks to code-agent, ask the user which code generation m
|
||||
- Prefer evidence-driven iteration: error analysis, sanity checks, and minimal ablations.
|
||||
- Update `/todos.md` to reflect new iterations.
|
||||
- Stop iterating when evidence is sufficient or diminishing returns appear.
|
||||
"""
|
||||
|
||||
### Memory Evolution (after significant outcomes)
|
||||
At these trigger points, invoke the `evo-memory` skill (read `/skills/evo-memory/SKILL.md` for the protocols, I/O specs, and classification rules):
|
||||
|
||||
- After **research-ideation** completes
|
||||
- After **experiment-pipeline** fails
|
||||
- After **experiment-pipeline** succeeds
|
||||
|
||||
If the `evo-memory` skill is not installed, manually log key learnings to `/memories/`: what worked, what failed, and why.
|
||||
|
||||
### Stage Reflection (Recommended Checkpoint)
|
||||
_EXPERIMENT_WORKFLOW_REFLECTION_AND_CLOSE = """### Stage Reflection (Recommended Checkpoint)
|
||||
After any meaningful experimental stage (baseline, new dataset, new training recipe, etc.), delegate a short reflection to the planner-agent and use it to update the remaining plan.
|
||||
|
||||
Trigger this checkpoint when:
|
||||
@@ -193,6 +212,26 @@ Empty arrays are valid. If no changes are needed, return the JSON with empty arr
|
||||
- Confirm the report answers the proposal and documents key settings/results.
|
||||
"""
|
||||
|
||||
|
||||
def _build_experiment_workflow(
|
||||
*,
|
||||
enable_observation_memory: bool = True,
|
||||
enable_observation_writes: bool = True,
|
||||
) -> str:
|
||||
"""Build the workflow section with memory instructions matching config."""
|
||||
sections = [
|
||||
_EXPERIMENT_WORKFLOW_PREAMBLE,
|
||||
_build_intake_scope(enable_observation_memory=enable_observation_memory),
|
||||
_EXPERIMENT_WORKFLOW_EXECUTION,
|
||||
]
|
||||
if enable_observation_memory and enable_observation_writes:
|
||||
sections.append(_MEMORY_EVOLUTION_SECTION)
|
||||
sections.append(_EXPERIMENT_WORKFLOW_REFLECTION_AND_CLOSE)
|
||||
return "\n\n".join(section.strip() for section in sections)
|
||||
|
||||
|
||||
EXPERIMENT_WORKFLOW = _build_experiment_workflow()
|
||||
|
||||
# =============================================================================
|
||||
# Report template (single source of truth — referenced from Step 5)
|
||||
# =============================================================================
|
||||
@@ -225,11 +264,10 @@ WRITING_GUIDELINES = """# Writing Guidelines
|
||||
# =============================================================================
|
||||
|
||||
# NOTE: the "300s" default below is intentionally hardcoded static text, not
|
||||
# templated from config. get_system_prompt() must stay byte-stable for prompt
|
||||
# caching, so the configured value is NOT injected here. The actually-enforced
|
||||
# timeout is cfg.sandbox_execute_timeout (CustomSandboxBackend); this number is
|
||||
# just the documented default, and the per-command `timeout` override is the
|
||||
# mechanism that matters to the agent.
|
||||
# templated from config. The actually-enforced timeout is
|
||||
# cfg.sandbox_execute_timeout (CustomSandboxBackend); this number is just the
|
||||
# documented default, and the per-command `timeout` override is the mechanism
|
||||
# that matters to the agent.
|
||||
SHELL_GUIDELINES = """# Shell Execution Guidelines
|
||||
|
||||
When using the `execute` tool for shell commands:
|
||||
@@ -350,7 +388,11 @@ It is fine to fetch one task and defer another from the same batch.
|
||||
# =============================================================================
|
||||
|
||||
|
||||
def get_system_prompt() -> str:
|
||||
def get_system_prompt(
|
||||
*,
|
||||
enable_observation_memory: bool = True,
|
||||
enable_observation_writes: bool = True,
|
||||
) -> str:
|
||||
"""Generate the complete static system prompt.
|
||||
|
||||
Sections are concatenated in this order:
|
||||
@@ -364,16 +406,20 @@ def get_system_prompt() -> str:
|
||||
7. :data:`ASYNC_NOTIFICATIONS`
|
||||
|
||||
Runtime context is injected per-turn by
|
||||
:class:`EvoScientist.middleware.RuntimeContextMiddleware`, so the static
|
||||
prefix here remains byte-stable across midnight rollover and across
|
||||
long-running daemons.
|
||||
:class:`EvoScientist.middleware.RuntimeContextMiddleware`, so dates and
|
||||
similar per-turn values are not baked into this prompt. Memory-related
|
||||
workflow sections can vary with the configured memory controls.
|
||||
|
||||
Returns:
|
||||
Combined static system prompt string.
|
||||
"""
|
||||
workflow = _build_experiment_workflow(
|
||||
enable_observation_memory=enable_observation_memory,
|
||||
enable_observation_writes=enable_observation_writes,
|
||||
)
|
||||
sections = [
|
||||
EVOSCIENTIST_IDENTITY,
|
||||
EXPERIMENT_WORKFLOW,
|
||||
workflow,
|
||||
REPORT_TEMPLATE,
|
||||
WRITING_GUIDELINES,
|
||||
SHELL_GUIDELINES,
|
||||
|
||||
@@ -17,6 +17,7 @@ from langchain_core.messages import ( # type: ignore[import-untyped]
|
||||
AIMessageChunk,
|
||||
)
|
||||
|
||||
from ..memory.worker_activity import clear_memory_worker_saved_counts
|
||||
from .emitter import StreamEventEmitter
|
||||
from .tracker import ToolCallTracker
|
||||
from .utils import DisplayLimits, is_success
|
||||
@@ -385,6 +386,7 @@ async def stream_agent_events(
|
||||
return "sub-agent"
|
||||
|
||||
# Build input for agent.astream()
|
||||
clear_memory_worker_saved_counts()
|
||||
if isinstance(message, str):
|
||||
# Build user message content: text + inline images + file path references
|
||||
user_content: str | list[dict[str, Any]] = message
|
||||
|
||||
@@ -41,8 +41,10 @@ def build_async_subagent_graph(name: str) -> Any:
|
||||
from EvoScientist.EvoScientist import (
|
||||
SUBAGENTS_CONFIG,
|
||||
_ensure_chat_model,
|
||||
_ensure_general_purpose_subagent,
|
||||
_get_default_backend,
|
||||
_get_default_middleware,
|
||||
_inject_subagent_middleware,
|
||||
)
|
||||
from EvoScientist.tools import tavily_search, think_tool
|
||||
from EvoScientist.utils import load_subagents
|
||||
@@ -96,6 +98,10 @@ def build_async_subagent_graph(name: str) -> Any:
|
||||
#
|
||||
# Memory middleware is included so async sub-agents get the same profile
|
||||
# context and `/memories/profile/...` file guidance as the main agent.
|
||||
subagents = []
|
||||
_ensure_general_purpose_subagent(subagents)
|
||||
_inject_subagent_middleware(subagents)
|
||||
|
||||
return create_deep_agent(
|
||||
name=name,
|
||||
model=_ensure_chat_model(),
|
||||
@@ -103,5 +109,9 @@ def build_async_subagent_graph(name: str) -> Any:
|
||||
tools=spec.get("tools", []) + agent_mcp_tools,
|
||||
skills=spec.get("skills"),
|
||||
backend=_get_default_backend(),
|
||||
middleware=_get_default_middleware(for_async_subagent=True),
|
||||
middleware=_get_default_middleware(
|
||||
for_async_subagent=True,
|
||||
memory_source_agent=name,
|
||||
),
|
||||
subagents=subagents,
|
||||
).with_config({"recursion_limit": cfg.recursion_limit})
|
||||
|
||||
@@ -12,8 +12,8 @@ code-agent:
|
||||
- Write outputs under /artifacts/ (recommended) and log key params to /experiment_log.md (optional).
|
||||
- Do not modify /skills/.
|
||||
- If a relevant local skill exists, read its SKILL.md and follow its workflow instead of reinventing.
|
||||
- Check `/memories/experiment-memory.md` for proven strategies from past cycles before implementing.
|
||||
Skip if the file does not exist yet.
|
||||
- Search `/memories/observations/` for relevant proven commands, failed attempts,
|
||||
tool constraints, or project lessons before implementing. Skip if there is no useful memory yet.
|
||||
- Before heavy runs, confirm GPU/CUDA/VRAM availability and required packages.
|
||||
- Suggested preflight commands:
|
||||
- nvidia-smi
|
||||
|
||||
@@ -6,10 +6,9 @@ planner-agent:
|
||||
You are the planner-agent. You do NOT implement code. You create and update experimental plans
|
||||
that are practical to run locally.
|
||||
|
||||
Before planning, check `/memories/ideation-memory.md` and `/memories/experiment-memory.md`
|
||||
for prior knowledge from past research cycles. Incorporate relevant entries into
|
||||
your plan (e.g., proven strategies, known failed directions). Skip if these files
|
||||
do not exist yet.
|
||||
Before planning, search `/memories/observations/` for relevant prior findings,
|
||||
failed directions, commands, and decisions from past research cycles. Incorporate
|
||||
relevant observations into your plan. Skip if there is no useful memory yet.
|
||||
|
||||
You may be invoked in two modes:
|
||||
1) PLAN MODE: produce an initial experimental plan.
|
||||
|
||||
@@ -28,12 +28,13 @@ def think_tool(reflection: str) -> str:
|
||||
the relevant `SKILL.md` for full instructions. Skills cover various research
|
||||
phases — ideation, experiment execution, paper writing, review, and more.
|
||||
Follow a skill's workflow rather than improvising when one is available.
|
||||
4. Prior knowledge — Have I checked research memory before starting?
|
||||
`/memories/ideation-memory.md` records promising and failed research directions.
|
||||
`/memories/experiment-memory.md` records proven strategies from past cycles.
|
||||
Read these at the start of new work. After completing or failing a task,
|
||||
consider whether the outcome should be recorded back into memory.
|
||||
Skip this if the memory files do not exist yet.
|
||||
4. Prior knowledge — Have I checked observation memory when it may matter?
|
||||
`/memories/observations/` records saved findings, failed attempts,
|
||||
commands, decisions, and other reusable notes. Search it with `grep` or
|
||||
`glob` before repeating substantial work. After completing or failing a
|
||||
task, call `record_observation` (when available) if the outcome is durable,
|
||||
non-obvious, evidence-backed, not already in memory, and likely to
|
||||
change future behavior. Skip this when there is no useful memory yet.
|
||||
5. Strategy — Should I continue the current approach, adjust it, or try
|
||||
something different? What evidence supports this decision?
|
||||
6. Handoff — Is this phase complete? What artifacts and results does the
|
||||
|
||||
@@ -11,6 +11,29 @@ from __future__ import annotations
|
||||
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
from EvoScientist.config import MemoryObservationWriter
|
||||
|
||||
|
||||
def _single_middleware(subagent: dict, class_name: str):
|
||||
matches = [m for m in subagent["middleware"] if type(m).__name__ == class_name]
|
||||
assert len(matches) == 1
|
||||
return matches[0]
|
||||
|
||||
|
||||
def _assert_subagent_memory_middleware(subagent: dict, *, source_agent: str) -> None:
|
||||
from EvoScientist.middleware.memory_lifecycle import MemoryLifecycleRole
|
||||
|
||||
memory_middleware = _single_middleware(subagent, "EvoMemoryMiddleware")
|
||||
lifecycle_middleware = _single_middleware(
|
||||
subagent,
|
||||
"EvoMemoryLifecycleMiddleware",
|
||||
)
|
||||
|
||||
assert [tool.name for tool in memory_middleware.tools] == ["record_observation"]
|
||||
assert lifecycle_middleware._role == MemoryLifecycleRole.SUBAGENT
|
||||
assert lifecycle_middleware._source_agent == source_agent
|
||||
assert lifecycle_middleware._project_id == memory_middleware.project_id
|
||||
|
||||
|
||||
@patch("deepagents.create_deep_agent")
|
||||
@patch("EvoScientist.EvoScientist._load_mcp_tools_cached", return_value={})
|
||||
@@ -40,6 +63,10 @@ def test_factory_requests_async_safe_middleware(
|
||||
# Minimal config stub so factory's `cfg.recursion_limit` access works.
|
||||
cfg = MagicMock()
|
||||
cfg.recursion_limit = 1_000_000
|
||||
cfg.memory_profile_enabled = True
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.ALL
|
||||
cfg.memory_workers_enabled = True
|
||||
mock_get_cfg.return_value = cfg
|
||||
# Factory looks up the requested name in the loaded subagent specs;
|
||||
# any matching name is fine.
|
||||
@@ -59,8 +86,140 @@ def test_factory_requests_async_safe_middleware(
|
||||
|
||||
build_async_subagent_graph("writing-agent")
|
||||
|
||||
# The contract: factory MUST pass ``for_async_subagent=True``.
|
||||
mock_get_mw.assert_called_once_with(for_async_subagent=True)
|
||||
# The contract: factory MUST pass async-safe mode and the source agent name.
|
||||
mock_get_mw.assert_called_once_with(
|
||||
for_async_subagent=True,
|
||||
memory_source_agent="writing-agent",
|
||||
)
|
||||
subagents = mock_create.call_args.kwargs["subagents"]
|
||||
assert subagents[0]["name"] == "general-purpose"
|
||||
_assert_subagent_memory_middleware(
|
||||
subagents[0],
|
||||
source_agent="general-purpose",
|
||||
)
|
||||
|
||||
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
def test_inject_subagent_adds_memory_middleware(mock_model, tmp_path):
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
from EvoScientist.EvoScientist import _inject_subagent_middleware
|
||||
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
subs = [{"name": "test-agent"}]
|
||||
|
||||
_inject_subagent_middleware(subs, workspace_dir=workspace)
|
||||
|
||||
_assert_subagent_memory_middleware(subs[0], source_agent="test-agent")
|
||||
|
||||
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
@patch("EvoScientist.EvoScientist._ensure_config")
|
||||
def test_inject_subagent_omits_memory_middleware_when_memory_disabled(
|
||||
mock_config, mock_model, tmp_path
|
||||
):
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
cfg = MagicMock()
|
||||
cfg.memory_profile_enabled = False
|
||||
cfg.memory_observations_enabled = False
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.ALL
|
||||
cfg.memory_workers_enabled = True
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _inject_subagent_middleware
|
||||
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
subs = [{"name": "test-agent"}]
|
||||
|
||||
_inject_subagent_middleware(subs, workspace_dir=workspace)
|
||||
|
||||
assert not [
|
||||
m
|
||||
for m in subs[0]["middleware"]
|
||||
if type(m).__name__ in {"EvoMemoryMiddleware", "EvoMemoryLifecycleMiddleware"}
|
||||
]
|
||||
|
||||
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
@patch("EvoScientist.EvoScientist._ensure_config")
|
||||
def test_inject_subagent_worker_only_observation_writer_keeps_live_tool_off(
|
||||
mock_config, mock_model, tmp_path
|
||||
):
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
cfg = MagicMock()
|
||||
cfg.memory_profile_enabled = False
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.WORKER
|
||||
cfg.memory_workers_enabled = True
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _inject_subagent_middleware
|
||||
from EvoScientist.middleware.memory_lifecycle import MemoryLifecycleRole
|
||||
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
subs = [{"name": "test-agent"}]
|
||||
|
||||
_inject_subagent_middleware(subs, workspace_dir=workspace)
|
||||
|
||||
memory_middleware = _single_middleware(subs[0], "EvoMemoryMiddleware")
|
||||
lifecycle_middleware = _single_middleware(
|
||||
subs[0],
|
||||
"EvoMemoryLifecycleMiddleware",
|
||||
)
|
||||
assert memory_middleware.tools == []
|
||||
assert lifecycle_middleware._role == MemoryLifecycleRole.SUBAGENT
|
||||
|
||||
|
||||
@patch(
|
||||
"EvoScientist.middleware.create_tool_selector_middleware",
|
||||
return_value=[MagicMock()],
|
||||
)
|
||||
@patch("EvoScientist.EvoScientist._ensure_chat_model")
|
||||
@patch("EvoScientist.EvoScientist._ensure_config")
|
||||
def test_all_observation_writer_skips_turn_worker_without_profile_memory(
|
||||
mock_config, mock_chat, mock_tool_selector
|
||||
):
|
||||
cfg = MagicMock()
|
||||
cfg.enable_ask_user = False
|
||||
cfg.auto_mode = False
|
||||
cfg.auto_approve = False
|
||||
cfg.model_fallbacks = None
|
||||
cfg.memory_profile_enabled = False
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.ALL
|
||||
cfg.memory_workers_enabled = True
|
||||
mock_config.return_value = cfg
|
||||
mock_chat.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
from EvoScientist.EvoScientist import _get_default_middleware
|
||||
|
||||
middleware = _get_default_middleware()
|
||||
memory_middleware = next(
|
||||
m for m in middleware if type(m).__name__ == "EvoMemoryMiddleware"
|
||||
)
|
||||
|
||||
assert [tool.name for tool in memory_middleware.tools] == ["record_observation"]
|
||||
assert not any(
|
||||
type(m).__name__ == "EvoMemoryLifecycleMiddleware" for m in middleware
|
||||
)
|
||||
|
||||
|
||||
def test_configured_system_prompt_matches_live_observation_tool():
|
||||
cfg = MagicMock()
|
||||
cfg.memory_profile_enabled = True
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.WORKER
|
||||
cfg.memory_workers_enabled = True
|
||||
|
||||
from EvoScientist.EvoScientist import _configured_system_prompt
|
||||
|
||||
prompt = _configured_system_prompt(cfg)
|
||||
|
||||
assert "/memories/observations/" in prompt
|
||||
assert "record_observation" not in prompt
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
@@ -96,6 +255,10 @@ def test_async_subagent_mode_filters_ask_user(
|
||||
cfg.auto_mode = False
|
||||
cfg.auto_approve = False
|
||||
cfg.model_fallbacks = None
|
||||
cfg.memory_profile_enabled = True
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.ALL
|
||||
cfg.memory_workers_enabled = True
|
||||
mock_config.return_value = cfg
|
||||
mock_chat.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
|
||||
@@ -4,6 +4,8 @@ import asyncio
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from EvoScientist.commands.channel_ui import ChannelCommandUI
|
||||
from tests.conftest import run_async as _run
|
||||
|
||||
@@ -77,6 +79,24 @@ def test_handle_session_resume_sends_history_back_to_channel_without_local_dupli
|
||||
assert "EvoScientist: Here is the saved answer." in text
|
||||
|
||||
|
||||
def test_handle_session_resume_propagates_callback_abort_without_history():
|
||||
callback = AsyncMock(side_effect=RuntimeError("workspace conflict"))
|
||||
bus_ref = SimpleNamespace(publish_outbound=AsyncMock())
|
||||
ui, captured = _make_ui(callback=callback, bus_ref=bus_ref)
|
||||
|
||||
with patch(
|
||||
"EvoScientist.sessions.get_thread_messages",
|
||||
new=AsyncMock(),
|
||||
) as get_messages:
|
||||
with pytest.raises(RuntimeError, match="workspace conflict"):
|
||||
_run(_run_resume(ui, "thread-42", "/workspace"))
|
||||
|
||||
callback.assert_awaited_once_with("thread-42", "/workspace")
|
||||
get_messages.assert_not_awaited()
|
||||
bus_ref.publish_outbound.assert_not_awaited()
|
||||
assert captured == []
|
||||
|
||||
|
||||
def test_handle_session_resume_reports_history_load_error():
|
||||
callback = AsyncMock()
|
||||
bus_ref = SimpleNamespace(publish_outbound=AsyncMock())
|
||||
|
||||
@@ -373,6 +373,86 @@ def test_on_cmd_completed_skipped_on_fall_through_and_error():
|
||||
completed.assert_not_called()
|
||||
|
||||
|
||||
def test_command_error_skips_completion_hook_and_reports_error():
|
||||
"""A command caught as failed by CommandManager must not look successful."""
|
||||
msg = _make_msg(content="/resume abc")
|
||||
fake_cmd = MagicMock()
|
||||
fake_cmd.needs_agent.return_value = False
|
||||
completed = AsyncMock()
|
||||
|
||||
async def _execute(_command, ctx):
|
||||
ctx.command_error = "workspace conflict"
|
||||
ctx.ui.append_system("Error executing /resume: workspace conflict", style="red")
|
||||
await ctx.ui.flush()
|
||||
return True
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.commands.manager.manager.resolve",
|
||||
return_value=(fake_cmd, ["abc"]),
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.commands.manager.manager.execute",
|
||||
side_effect=_execute,
|
||||
),
|
||||
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
||||
):
|
||||
handled = _run(
|
||||
dispatch_channel_slash_command(
|
||||
msg,
|
||||
agent=None,
|
||||
thread_id="old-thread",
|
||||
workspace_dir="/old-workspace",
|
||||
checkpointer=None,
|
||||
append_system=MagicMock(),
|
||||
on_cmd_completed=completed,
|
||||
)
|
||||
)
|
||||
|
||||
assert handled is True
|
||||
completed.assert_not_awaited()
|
||||
mock_set_resp.assert_called_once_with("msg-1", "Command error: workspace conflict")
|
||||
|
||||
|
||||
def test_empty_command_error_still_reports_error():
|
||||
"""An empty string error is still a command failure sentinel."""
|
||||
msg = _make_msg(content="/resume abc")
|
||||
fake_cmd = MagicMock()
|
||||
fake_cmd.needs_agent.return_value = False
|
||||
completed = AsyncMock()
|
||||
|
||||
async def _execute(_command, ctx):
|
||||
ctx.command_error = ""
|
||||
return True
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.commands.manager.manager.resolve",
|
||||
return_value=(fake_cmd, ["abc"]),
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.commands.manager.manager.execute",
|
||||
side_effect=_execute,
|
||||
),
|
||||
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
||||
):
|
||||
handled = _run(
|
||||
dispatch_channel_slash_command(
|
||||
msg,
|
||||
agent=None,
|
||||
thread_id="old-thread",
|
||||
workspace_dir="/old-workspace",
|
||||
checkpointer=None,
|
||||
append_system=MagicMock(),
|
||||
on_cmd_completed=completed,
|
||||
)
|
||||
)
|
||||
|
||||
assert handled is True
|
||||
completed.assert_not_awaited()
|
||||
mock_set_resp.assert_called_once_with("msg-1", "Command error: (no details)")
|
||||
|
||||
|
||||
def test_on_cmd_completed_exception_is_absorbed():
|
||||
"""A raising hook must NOT prevent the channel response from being set."""
|
||||
msg = _make_msg()
|
||||
|
||||
@@ -6,6 +6,7 @@ import os
|
||||
from types import SimpleNamespace
|
||||
|
||||
from EvoScientist.cli import commands
|
||||
from EvoScientist.config import MemoryObservationWriter
|
||||
|
||||
|
||||
def _make_config(
|
||||
@@ -27,6 +28,11 @@ def _make_config(
|
||||
auto_approve=auto_approve,
|
||||
auto_mode=auto_mode,
|
||||
enable_ask_user=enable_ask_user,
|
||||
enable_async_subagents=False,
|
||||
memory_profile_enabled=True,
|
||||
memory_observations_enabled=True,
|
||||
memory_observation_writer=MemoryObservationWriter.ALL,
|
||||
memory_workers_enabled=False,
|
||||
provider="anthropic",
|
||||
anthropic_auth_mode="api_key",
|
||||
openai_auth_mode="api_key",
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
"""Tests for CLI interactive UI backend dispatch."""
|
||||
|
||||
import asyncio
|
||||
from types import SimpleNamespace
|
||||
|
||||
import pytest
|
||||
|
||||
from EvoScientist.cli.commands import _is_fresh_interactive_session
|
||||
@@ -77,16 +80,65 @@ def test_main_callback_resume_falls_back_to_cli(monkeypatch):
|
||||
assert calls.get("dispatch") == ("cli", "cli")
|
||||
|
||||
|
||||
def test_background_agent_server_starts_even_when_async_subagents_disabled(
|
||||
monkeypatch,
|
||||
):
|
||||
import EvoScientist.cli.commands as cmds
|
||||
|
||||
calls = []
|
||||
|
||||
def fake_ensure(config, *, workspace_dir):
|
||||
calls.append((config, workspace_dir))
|
||||
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.langgraph_dev.manager.ensure_langgraph_dev",
|
||||
fake_ensure,
|
||||
)
|
||||
|
||||
config = SimpleNamespace(enable_async_subagents=False)
|
||||
cmds._ensure_async_subagent_server(config, workspace_dir="/tmp/workspace")
|
||||
|
||||
assert calls == [(config, "/tmp/workspace")]
|
||||
|
||||
|
||||
def test_resume_workspace_sync_runs_even_when_async_subagents_disabled(
|
||||
monkeypatch,
|
||||
):
|
||||
import EvoScientist.cli.commands as cmds
|
||||
|
||||
calls = []
|
||||
|
||||
def fake_ensure(config, *, workspace_dir):
|
||||
calls.append((config, workspace_dir))
|
||||
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.langgraph_dev.manager.ensure_langgraph_dev",
|
||||
fake_ensure,
|
||||
)
|
||||
|
||||
config = SimpleNamespace(enable_async_subagents=False)
|
||||
asyncio.run(
|
||||
cmds._sync_background_agent_server_workspace(
|
||||
config,
|
||||
workspace_dir="/tmp/resumed-workspace",
|
||||
)
|
||||
)
|
||||
|
||||
assert calls == [(config, "/tmp/resumed-workspace")]
|
||||
|
||||
|
||||
def test_cmd_interactive_dispatches_to_textual(monkeypatch):
|
||||
captured: dict[str, object] = {}
|
||||
captured_kwargs: list[dict[str, object]] = []
|
||||
effective_config = SimpleNamespace(langgraph_dev_port=9999)
|
||||
|
||||
def _fake_resolve_ui_backend(value, *, warn_fallback=False):
|
||||
captured["resolved_input"] = value
|
||||
captured["warn_fallback"] = warn_fallback
|
||||
return "tui"
|
||||
|
||||
def _fake_run_textual_interactive(**kwargs):
|
||||
captured["kwargs"] = kwargs
|
||||
def _fake_run_textual_interactive(**kwargs: object):
|
||||
captured_kwargs.append(kwargs)
|
||||
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.cli.interactive.resolve_ui_backend",
|
||||
@@ -108,13 +160,14 @@ def test_cmd_interactive_dispatches_to_textual(monkeypatch):
|
||||
run_name="demo-run",
|
||||
thread_id="thread-1",
|
||||
ui_backend="tui",
|
||||
config=effective_config,
|
||||
)
|
||||
|
||||
assert captured["resolved_input"] == "tui"
|
||||
assert captured["warn_fallback"] is True
|
||||
|
||||
kwargs = captured["kwargs"]
|
||||
assert isinstance(kwargs, dict)
|
||||
assert len(captured_kwargs) == 1
|
||||
kwargs = captured_kwargs[0]
|
||||
assert kwargs["workspace_dir"] == "/tmp/workspace"
|
||||
assert kwargs["workspace_fixed"] is True
|
||||
assert kwargs["mode"] == "daemon"
|
||||
@@ -122,6 +175,7 @@ def test_cmd_interactive_dispatches_to_textual(monkeypatch):
|
||||
assert kwargs["provider"] == "demo-provider"
|
||||
assert kwargs["run_name"] == "demo-run"
|
||||
assert kwargs["thread_id"] == "thread-1"
|
||||
assert kwargs["config"] is effective_config
|
||||
assert kwargs["channel_send_thinking"] is True
|
||||
assert callable(kwargs["load_agent"])
|
||||
assert callable(kwargs["create_session_workspace"])
|
||||
|
||||
@@ -8,6 +8,9 @@ import yaml
|
||||
|
||||
from EvoScientist.config import (
|
||||
EvoScientistConfig,
|
||||
MemoryControls,
|
||||
MemoryObservationTarget,
|
||||
MemoryObservationWriter,
|
||||
apply_config_to_env,
|
||||
get_config_dir,
|
||||
get_config_path,
|
||||
@@ -43,6 +46,10 @@ def temp_config_dir(tmp_path, monkeypatch):
|
||||
"EVOSCIENTIST_DEFAULT_MODE",
|
||||
"EVOSCIENTIST_WORKSPACE_DIR",
|
||||
"EVOSCIENTIST_UI_BACKEND",
|
||||
"EVOSCIENTIST_MEMORY_PROFILE_ENABLED",
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED",
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATION_WRITER",
|
||||
"EVOSCIENTIST_MEMORY_WORKERS_ENABLED",
|
||||
]:
|
||||
monkeypatch.delenv(key, raising=False)
|
||||
return config_dir
|
||||
@@ -58,6 +65,10 @@ def clean_env(monkeypatch):
|
||||
"EVOSCIENTIST_DEFAULT_MODE",
|
||||
"EVOSCIENTIST_WORKSPACE_DIR",
|
||||
"EVOSCIENTIST_UI_BACKEND",
|
||||
"EVOSCIENTIST_MEMORY_PROFILE_ENABLED",
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED",
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATION_WRITER",
|
||||
"EVOSCIENTIST_MEMORY_WORKERS_ENABLED",
|
||||
]:
|
||||
monkeypatch.delenv(key, raising=False)
|
||||
|
||||
@@ -83,6 +94,10 @@ class TestEvoScientistConfig:
|
||||
assert config.ui_backend == "tui"
|
||||
assert config.log_level == "warning"
|
||||
assert config.reasoning_effort == "high"
|
||||
assert config.memory_profile_enabled is True
|
||||
assert config.memory_observations_enabled is True
|
||||
assert config.memory_observation_writer == MemoryObservationWriter.ALL
|
||||
assert config.memory_workers_enabled is True
|
||||
assert config.ollama_base_url == ""
|
||||
assert config.channel_debug_tracing is False
|
||||
assert config.imessage_enabled is False
|
||||
@@ -291,6 +306,56 @@ class TestGetSetValues:
|
||||
set_config_value("channel_debug_tracing", "false")
|
||||
assert get_config_value("channel_debug_tracing") is False
|
||||
|
||||
def test_set_memory_observation_writer_validates_value(
|
||||
self, temp_config_dir, clean_env
|
||||
):
|
||||
"""Observation writer mode accepts only the supported product controls."""
|
||||
save_config(
|
||||
EvoScientistConfig(memory_observation_writer=MemoryObservationWriter.ALL)
|
||||
)
|
||||
|
||||
assert set_config_value("memory_observation_writer", "worker") is True
|
||||
assert get_config_value("memory_observation_writer") == "worker"
|
||||
assert set_config_value("memory_observation_writer", "AGENT") is True
|
||||
assert get_config_value("memory_observation_writer") == "agent"
|
||||
assert set_config_value("memory_observation_writer", "subagent") is False
|
||||
assert get_config_value("memory_observation_writer") == "agent"
|
||||
|
||||
def test_memory_controls_observation_writer_targets(self):
|
||||
"""MemoryControls centralizes observation writer target semantics."""
|
||||
worker_controls = MemoryControls.from_config(
|
||||
EvoScientistConfig(
|
||||
memory_profile_enabled=False,
|
||||
memory_observations_enabled=True,
|
||||
memory_observation_writer=MemoryObservationWriter.WORKER,
|
||||
memory_workers_enabled=True,
|
||||
)
|
||||
)
|
||||
all_controls = MemoryControls.from_config(
|
||||
EvoScientistConfig(
|
||||
memory_profile_enabled=False,
|
||||
memory_observations_enabled=True,
|
||||
memory_observation_writer=MemoryObservationWriter.ALL,
|
||||
memory_workers_enabled=True,
|
||||
)
|
||||
)
|
||||
|
||||
assert not worker_controls.observation_tool_enabled(
|
||||
MemoryObservationTarget.TURN_WORKER
|
||||
)
|
||||
assert not worker_controls.worker_needed(MemoryObservationTarget.TURN_WORKER)
|
||||
assert worker_controls.observation_tool_enabled(
|
||||
MemoryObservationTarget.SUBAGENT_WORKER
|
||||
)
|
||||
assert not worker_controls.observation_tool_enabled(
|
||||
MemoryObservationTarget.AGENT
|
||||
)
|
||||
assert not all_controls.worker_needed(MemoryObservationTarget.TURN_WORKER)
|
||||
assert all_controls.observation_tool_enabled(MemoryObservationTarget.AGENT)
|
||||
assert all_controls.observation_tool_enabled(
|
||||
MemoryObservationTarget.SUBAGENT_WORKER
|
||||
)
|
||||
|
||||
def test_list_config(self, temp_config_dir, clean_env):
|
||||
"""Test listing all config values."""
|
||||
config = EvoScientistConfig(provider="openai", model="gpt-4o")
|
||||
@@ -381,6 +446,27 @@ class TestPriorityChain:
|
||||
config = get_effective_config()
|
||||
assert config.channel_debug_tracing is True
|
||||
|
||||
def test_env_memory_controls_override(self, temp_config_dir, monkeypatch):
|
||||
"""Memory controls can be selected via environment variables."""
|
||||
save_config(
|
||||
EvoScientistConfig(
|
||||
memory_profile_enabled=True,
|
||||
memory_observations_enabled=True,
|
||||
memory_observation_writer=MemoryObservationWriter.ALL,
|
||||
memory_workers_enabled=True,
|
||||
)
|
||||
)
|
||||
monkeypatch.setenv("EVOSCIENTIST_MEMORY_PROFILE_ENABLED", "false")
|
||||
monkeypatch.setenv("EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED", "false")
|
||||
monkeypatch.setenv("EVOSCIENTIST_MEMORY_OBSERVATION_WRITER", "worker")
|
||||
monkeypatch.setenv("EVOSCIENTIST_MEMORY_WORKERS_ENABLED", "false")
|
||||
|
||||
config = get_effective_config()
|
||||
assert config.memory_profile_enabled is False
|
||||
assert config.memory_observations_enabled is False
|
||||
assert config.memory_observation_writer == MemoryObservationWriter.WORKER
|
||||
assert config.memory_workers_enabled is False
|
||||
|
||||
def test_sandbox_execute_timeout_default(self, temp_config_dir, clean_env):
|
||||
"""Sandbox execute timeout defaults to 300 seconds."""
|
||||
assert EvoScientistConfig().sandbox_execute_timeout == 300
|
||||
|
||||
@@ -13,6 +13,7 @@ from unittest.mock import MagicMock, patch
|
||||
import httpx
|
||||
import pytest
|
||||
|
||||
from EvoScientist.config.settings import EvoScientistConfig
|
||||
from EvoScientist.langgraph_dev import manager
|
||||
|
||||
|
||||
@@ -183,19 +184,52 @@ class TestKillOwnedStaleProcess:
|
||||
|
||||
|
||||
class TestEnsureLanggraphDev:
|
||||
def test_returns_none_when_async_disabled(self):
|
||||
cfg = SimpleNamespace(enable_async_subagents=False)
|
||||
assert manager.ensure_langgraph_dev(cfg) is None
|
||||
# And the availability flag should remain False.
|
||||
def test_starts_when_async_disabled_but_memory_workers_enabled(self, tmp_path):
|
||||
"""EvoMemory workers can require langgraph dev even without async subagents."""
|
||||
cfg = EvoScientistConfig()
|
||||
cfg.enable_async_subagents = False
|
||||
cfg.memory_workers_enabled = True
|
||||
cfg.langgraph_dev_port = 6174
|
||||
cfg.langgraph_dev_file_persistence = True
|
||||
proc = MagicMock()
|
||||
with (
|
||||
patch.object(manager, "is_langgraph_dev_running", return_value=False),
|
||||
patch.object(manager, "start_langgraph_dev", return_value=proc) as start,
|
||||
patch.object(manager, "_FILE_LOCK_PATH", tmp_path / "lg.lock"),
|
||||
patch.object(manager, "_PID_DIR", tmp_path / "pids"),
|
||||
):
|
||||
result = manager.ensure_langgraph_dev(cfg, workspace_dir=tmp_path)
|
||||
|
||||
assert result is proc
|
||||
start.assert_called_once()
|
||||
assert manager.is_async_subagents_available() is True
|
||||
|
||||
def test_skips_when_async_and_memory_workers_disabled(self, tmp_path):
|
||||
"""No background server is needed without async subagents or workers."""
|
||||
cfg = EvoScientistConfig()
|
||||
cfg.enable_async_subagents = False
|
||||
cfg.memory_workers_enabled = False
|
||||
cfg.langgraph_dev_port = 6174
|
||||
cfg.langgraph_dev_file_persistence = True
|
||||
with (
|
||||
patch.object(manager, "is_langgraph_dev_running") as mock_running,
|
||||
patch.object(manager, "start_langgraph_dev") as start,
|
||||
patch.object(manager, "_FILE_LOCK_PATH", tmp_path / "lg.lock"),
|
||||
patch.object(manager, "_PID_DIR", tmp_path / "pids"),
|
||||
):
|
||||
result = manager.ensure_langgraph_dev(cfg, workspace_dir=tmp_path)
|
||||
|
||||
assert result is None
|
||||
mock_running.assert_not_called()
|
||||
start.assert_not_called()
|
||||
assert manager.is_async_subagents_available() is False
|
||||
|
||||
def test_reuses_existing_healthy_subprocess(self, tmp_path):
|
||||
"""When the subprocess is already running, no new Popen call."""
|
||||
cfg = SimpleNamespace(
|
||||
enable_async_subagents=True,
|
||||
langgraph_dev_port=6174,
|
||||
langgraph_dev_file_persistence=True,
|
||||
)
|
||||
cfg = EvoScientistConfig()
|
||||
cfg.enable_async_subagents = True
|
||||
cfg.langgraph_dev_port = 6174
|
||||
cfg.langgraph_dev_file_persistence = True
|
||||
with (
|
||||
patch.object(
|
||||
manager, "is_langgraph_dev_running", return_value=True
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,11 +1,19 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import threading
|
||||
from types import SimpleNamespace
|
||||
|
||||
from blockbuster import BlockBuster
|
||||
from langchain_core.messages import SystemMessage
|
||||
|
||||
import EvoScientist.middleware.memory as memory_module
|
||||
from EvoScientist import paths
|
||||
from EvoScientist.memory.observations import (
|
||||
MemoryScope,
|
||||
MemorySourceType,
|
||||
MemoryType,
|
||||
record_observation_file,
|
||||
)
|
||||
|
||||
|
||||
def _request():
|
||||
@@ -24,12 +32,6 @@ def _request():
|
||||
return request
|
||||
|
||||
|
||||
def _system_text(modified) -> str:
|
||||
system_message = modified.system_message
|
||||
assert system_message is not None
|
||||
return str(system_message.content)
|
||||
|
||||
|
||||
def _path_project_id(workspace) -> str:
|
||||
return memory_module._resolve_project_id(workspace)
|
||||
|
||||
@@ -48,17 +50,181 @@ def test_profile_memory_bootstraps_and_injects_profile_files(tmp_path, monkeypat
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
|
||||
middleware = memory_module.create_memory_middleware(str(memories))
|
||||
modified = middleware.modify_request(_request())
|
||||
system_text = _system_text(modified)
|
||||
middleware.modify_request(_request())
|
||||
|
||||
assert "Today's date" not in system_text
|
||||
assert "<profile_memory>" in system_text
|
||||
assert "# User profile" in system_text
|
||||
assert "/memories/profile/USER_PROFILE.md" in system_text
|
||||
assert [tool.name for tool in middleware.tools] == ["record_observation"]
|
||||
assert (memories / "profile" / "SOUL.md").exists()
|
||||
assert (memories / "profile" / "USER_PROFILE.md").exists()
|
||||
assert (memories / "profile" / "RESEARCH_TASTE.md").exists()
|
||||
assert list((memories / "profile" / "projects").glob("*/PROJECT_PROFILE.md"))
|
||||
assert (memories / "observations" / "global").is_dir()
|
||||
assert list((memories / "observations" / "projects").glob("P-*"))
|
||||
|
||||
|
||||
def test_profile_memory_can_disable_observation_tool(tmp_path, monkeypatch):
|
||||
memories = tmp_path / "memories"
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
|
||||
middleware = memory_module.create_memory_middleware(
|
||||
str(memories),
|
||||
enable_observation_tool=False,
|
||||
)
|
||||
middleware.modify_request(_request())
|
||||
|
||||
assert middleware.tools == []
|
||||
assert (memories / "profile" / "USER_PROFILE.md").exists()
|
||||
|
||||
|
||||
def test_memory_middleware_can_disable_all_memory_injection(tmp_path, monkeypatch):
|
||||
memories = tmp_path / "memories"
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
|
||||
middleware = memory_module.create_memory_middleware(
|
||||
str(memories),
|
||||
enable_profile_memory=False,
|
||||
enable_observation_memory=False,
|
||||
)
|
||||
request = _request()
|
||||
modified = middleware.modify_request(request)
|
||||
|
||||
assert modified is request
|
||||
assert middleware.tools == []
|
||||
assert not (memories / "profile").exists()
|
||||
assert not (memories / "observations").exists()
|
||||
|
||||
|
||||
def test_observation_memory_can_be_read_only_without_profile(tmp_path, monkeypatch):
|
||||
memories = tmp_path / "memories"
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
|
||||
middleware = memory_module.create_memory_middleware(
|
||||
str(memories),
|
||||
enable_profile_memory=False,
|
||||
enable_observation_memory=True,
|
||||
enable_observation_tool=False,
|
||||
)
|
||||
modified = middleware.modify_request(_request())
|
||||
content = str(modified.system_message.content)
|
||||
|
||||
assert middleware.tools == []
|
||||
assert not (memories / "profile").exists()
|
||||
assert (memories / "observations" / "global").is_dir()
|
||||
assert list((memories / "observations" / "projects").glob("P-*"))
|
||||
assert "<observation_memory>" in content
|
||||
assert "Memory preflight:" in content
|
||||
assert "record_observation" not in content
|
||||
|
||||
|
||||
def test_observation_index_loads_summary_frontmatter_once(tmp_path, monkeypatch):
|
||||
memories = tmp_path / "memories"
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
project_id = _path_project_id(workspace)
|
||||
global_result = record_observation_file(
|
||||
memory_dir=memories,
|
||||
project_id=project_id,
|
||||
memory_type=MemoryType.SEMANTIC,
|
||||
summary="A global fact is available for future lookup.",
|
||||
observation="A global fact should be indexed.",
|
||||
why_it_matters="Future agents can decide whether to read it.",
|
||||
scope=MemoryScope.GLOBAL,
|
||||
source_type=MemorySourceType.SUBAGENT,
|
||||
source_session_id="thread-1",
|
||||
source_agent="research-agent",
|
||||
)
|
||||
project_result = record_observation_file(
|
||||
memory_dir=memories,
|
||||
project_id=project_id,
|
||||
memory_type=MemoryType.PROCEDURAL,
|
||||
summary="A project recipe is available for future lookup.",
|
||||
observation="A project recipe should be indexed.",
|
||||
why_it_matters="Future agents can choose it for this workspace.",
|
||||
scope=MemoryScope.PROJECT,
|
||||
source_type=MemorySourceType.SUBAGENT,
|
||||
source_session_id="thread-1",
|
||||
source_agent="code-agent",
|
||||
)
|
||||
(memories / "observations" / "global" / "O-old.md").write_text(
|
||||
"\n".join(
|
||||
[
|
||||
"---",
|
||||
'id: "O-old"',
|
||||
"memory_type: semantic",
|
||||
"scope: global",
|
||||
"---",
|
||||
"",
|
||||
"## Observation",
|
||||
"",
|
||||
"Old observations without summary are not indexed.",
|
||||
]
|
||||
),
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
middleware = memory_module.create_memory_middleware(str(memories))
|
||||
indexed = {
|
||||
record.observation_id: (record.memory_type, record.scope, record.summary)
|
||||
for record in middleware._observation_index_records
|
||||
}
|
||||
record_observation_file(
|
||||
memory_dir=memories,
|
||||
project_id=project_id,
|
||||
memory_type=MemoryType.SEMANTIC,
|
||||
summary="This later observation is not in the cached index.",
|
||||
observation="Observation written after middleware construction.",
|
||||
why_it_matters="Prompt memory should stay stable during the agent run.",
|
||||
scope=MemoryScope.GLOBAL,
|
||||
source_type=MemorySourceType.SUBAGENT,
|
||||
source_session_id="thread-2",
|
||||
source_agent="research-agent",
|
||||
)
|
||||
|
||||
assert indexed == {
|
||||
global_result["observation_id"]: (
|
||||
MemoryType.SEMANTIC,
|
||||
MemoryScope.GLOBAL,
|
||||
"A global fact is available for future lookup.",
|
||||
),
|
||||
project_result["observation_id"]: (
|
||||
MemoryType.PROCEDURAL,
|
||||
MemoryScope.PROJECT,
|
||||
"A project recipe is available for future lookup.",
|
||||
),
|
||||
}
|
||||
assert {
|
||||
record.observation_id for record in middleware._observation_index_records
|
||||
} == set(indexed)
|
||||
|
||||
|
||||
def test_observation_index_omits_summaries_when_budget_exceeded(tmp_path, monkeypatch):
|
||||
memories = tmp_path / "memories"
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
middleware = memory_module.create_memory_middleware(str(memories))
|
||||
records = [
|
||||
memory_module.ObservationIndexRecord(
|
||||
observation_id="O-large",
|
||||
memory_path="/observations/global/O-large.md",
|
||||
memory_type=MemoryType.PROCEDURAL,
|
||||
scope=MemoryScope.GLOBAL,
|
||||
summary="Do not inline this summary when the index exceeds budget.",
|
||||
)
|
||||
]
|
||||
|
||||
context = middleware._observation_index_context_from_records(
|
||||
records,
|
||||
max_inline_chars=1,
|
||||
)
|
||||
|
||||
assert "Do not inline this summary" not in context
|
||||
|
||||
|
||||
def test_profile_memory_uses_path_pointers_when_profiles_exceed_budget(
|
||||
@@ -72,12 +238,12 @@ def test_profile_memory_uses_path_pointers_when_profiles_exceed_budget(
|
||||
middleware = memory_module.create_memory_middleware(
|
||||
str(memories), max_inline_profile_chars=10
|
||||
)
|
||||
modified = middleware.modify_request(_request())
|
||||
system_text = _system_text(modified)
|
||||
middleware.modify_request(_request())
|
||||
records = middleware._read_profile_records()
|
||||
|
||||
assert "Profile files are available at:" in system_text
|
||||
assert "File: /memories/profile/SOUL.md" not in system_text
|
||||
assert "/memories/profile/USER_PROFILE.md" in system_text
|
||||
assert middleware._profile_context_from_records(records) == (
|
||||
middleware._profile_pointer_context
|
||||
)
|
||||
|
||||
|
||||
def test_profile_memory_async_path_bootstraps_and_injects(
|
||||
@@ -92,11 +258,8 @@ def test_profile_memory_async_path_bootstraps_and_injects(
|
||||
return request
|
||||
|
||||
middleware = memory_module.create_memory_middleware(str(memories))
|
||||
modified = run_async(middleware.awrap_model_call(_request(), _handler))
|
||||
system_text = _system_text(modified)
|
||||
run_async(middleware.awrap_model_call(_request(), _handler))
|
||||
|
||||
assert "<profile_memory>" in system_text
|
||||
assert "/memories/profile/USER_PROFILE.md" in system_text
|
||||
assert (memories / "profile" / "USER_PROFILE.md").exists()
|
||||
|
||||
|
||||
@@ -106,15 +269,15 @@ def test_profile_memory_write_failure_uses_path_pointers(tmp_path, monkeypatch):
|
||||
workspace.mkdir()
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
|
||||
monkeypatch.setattr(
|
||||
memory_module.EvoMemoryMiddleware,
|
||||
"_write_text",
|
||||
lambda _self, _path, _content: False,
|
||||
)
|
||||
middleware = memory_module.create_memory_middleware(str(memories))
|
||||
monkeypatch.setattr(middleware, "_write_text", lambda _path, _content: False)
|
||||
|
||||
modified = middleware.modify_request(_request())
|
||||
system_text = _system_text(modified)
|
||||
middleware.modify_request(_request())
|
||||
|
||||
assert "Profile files are available at:" in system_text
|
||||
assert "File: /memories/profile/SOUL.md" not in system_text
|
||||
assert "# User profile" not in system_text
|
||||
assert not (memories / "profile" / "USER_PROFILE.md").exists()
|
||||
|
||||
|
||||
@@ -133,13 +296,56 @@ def test_profile_memory_read_failure_uses_path_pointers_without_overwriting(
|
||||
soul_path.write_bytes(original_bytes)
|
||||
|
||||
middleware = memory_module.create_memory_middleware(str(memories))
|
||||
modified = middleware.modify_request(_request())
|
||||
system_text = _system_text(modified)
|
||||
middleware.modify_request(_request())
|
||||
|
||||
assert "Profile files are available at:" in system_text
|
||||
assert soul_path.read_bytes() == original_bytes
|
||||
|
||||
|
||||
def test_profile_memory_async_path_inlines_content_under_blockbuster(
|
||||
tmp_path, monkeypatch, run_async
|
||||
):
|
||||
memories = tmp_path / "memories"
|
||||
workspace = tmp_path / "workspace"
|
||||
workspace.mkdir()
|
||||
monkeypatch.setattr(paths, "WORKSPACE_ROOT", workspace)
|
||||
|
||||
middleware = memory_module.create_memory_middleware(str(memories))
|
||||
middleware.modify_request(_request())
|
||||
user_profile = memories / "profile" / "USER_PROFILE.md"
|
||||
user_profile.write_text(
|
||||
user_profile.read_text(encoding="utf-8")
|
||||
+ "\n\n- Async profile content should be inlined.",
|
||||
encoding="utf-8",
|
||||
)
|
||||
|
||||
call_threads = []
|
||||
original_read = middleware._read_profile_memory
|
||||
|
||||
def tracked_read_profile_memory():
|
||||
call_threads.append(threading.get_ident())
|
||||
return original_read()
|
||||
|
||||
monkeypatch.setattr(middleware, "_read_profile_memory", tracked_read_profile_memory)
|
||||
|
||||
async def run():
|
||||
event_loop_thread = threading.get_ident()
|
||||
blocker = BlockBuster(scanned_modules=memory_module)
|
||||
blocker.activate()
|
||||
try:
|
||||
modified = await middleware.amodify_request(_request())
|
||||
finally:
|
||||
blocker.deactivate()
|
||||
return event_loop_thread, modified
|
||||
|
||||
event_loop_thread, modified = run_async(run())
|
||||
|
||||
assert call_threads
|
||||
assert all(thread_id != event_loop_thread for thread_id in call_threads)
|
||||
assert "Async profile content should be inlined." in str(
|
||||
modified.system_message.content
|
||||
)
|
||||
|
||||
|
||||
def test_profile_memory_migrates_legacy_memory_once(tmp_path, monkeypatch):
|
||||
memories = tmp_path / "memories"
|
||||
memories.mkdir()
|
||||
@@ -217,20 +423,14 @@ def test_profile_memory_uses_explicit_workspace_for_project_profile(
|
||||
middleware = memory_module.create_memory_middleware(
|
||||
str(memories), workspace_dir=str(active_workspace)
|
||||
)
|
||||
modified = middleware.modify_request(_request())
|
||||
system_text = _system_text(modified)
|
||||
middleware.modify_request(_request())
|
||||
|
||||
expected_project_id = _path_project_id(active_workspace)
|
||||
wrong_project_id = _path_project_id(global_workspace)
|
||||
|
||||
assert (
|
||||
f"/memories/profile/projects/{expected_project_id}/PROJECT_PROFILE.md"
|
||||
in system_text
|
||||
)
|
||||
assert (
|
||||
memories / "profile" / "projects" / expected_project_id / "PROJECT_PROFILE.md"
|
||||
).exists()
|
||||
assert wrong_project_id not in system_text
|
||||
assert not (
|
||||
memories / "profile" / "projects" / wrong_project_id / "PROJECT_PROFILE.md"
|
||||
).exists()
|
||||
@@ -253,17 +453,14 @@ def test_profile_memory_resolves_project_id_once_per_middleware(
|
||||
middleware = memory_module.create_memory_middleware(
|
||||
str(memories), workspace_dir=str(workspace), max_inline_profile_chars=10
|
||||
)
|
||||
sync_modified = middleware.modify_request(_request())
|
||||
async_modified = run_async(middleware.amodify_request(_request()))
|
||||
middleware.modify_request(_request())
|
||||
run_async(middleware.amodify_request(_request()))
|
||||
|
||||
assert calls == [workspace]
|
||||
assert (
|
||||
"/memories/profile/projects/P-cached-project/PROJECT_PROFILE.md"
|
||||
in _system_text(sync_modified)
|
||||
)
|
||||
assert (
|
||||
"/memories/profile/projects/P-cached-project/PROJECT_PROFILE.md"
|
||||
in _system_text(async_modified)
|
||||
assert middleware.project_id == "P-cached-project"
|
||||
assert any(
|
||||
path == "/profile/projects/P-cached-project/PROJECT_PROFILE.md"
|
||||
for path, _template in middleware._profile_specs
|
||||
)
|
||||
|
||||
|
||||
|
||||
+21
-1
@@ -84,7 +84,7 @@ class TestGetSystemPrompt:
|
||||
|
||||
Anything sent to `/memory/...` falls through to CustomSandboxBackend
|
||||
(workspace files), bypassing the persistent FilesystemBackend that
|
||||
owns ideation-memory.md / experiment-memory.md.
|
||||
owns persistent memory files.
|
||||
"""
|
||||
result = get_system_prompt()
|
||||
# `/memory/` as a filesystem path (after a backtick or whitespace, before
|
||||
@@ -96,6 +96,26 @@ class TestGetSystemPrompt:
|
||||
"Found `/memory/<file>` in system prompt — should be `/memories/<file>`"
|
||||
)
|
||||
|
||||
def test_observation_writes_can_be_removed(self):
|
||||
result = get_system_prompt(
|
||||
enable_observation_memory=True,
|
||||
enable_observation_writes=False,
|
||||
)
|
||||
|
||||
assert "/memories/observations/" in result
|
||||
assert "record_observation" not in result
|
||||
assert "Memory Evolution" not in result
|
||||
|
||||
def test_observation_memory_can_be_removed(self):
|
||||
result = get_system_prompt(
|
||||
enable_observation_memory=False,
|
||||
enable_observation_writes=False,
|
||||
)
|
||||
|
||||
assert "/memories/observations/" not in result
|
||||
assert "record_observation" not in result
|
||||
assert "Memory Evolution" not in result
|
||||
|
||||
|
||||
class TestEvoScientistIdentity:
|
||||
def test_constant_not_empty(self):
|
||||
|
||||
@@ -8,12 +8,15 @@ captured at startup.
|
||||
|
||||
from unittest.mock import AsyncMock, MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
from EvoScientist.cli.channel import (
|
||||
ChannelMessage,
|
||||
_register_channel_request,
|
||||
)
|
||||
from EvoScientist.cli.commands import (
|
||||
_make_serve_cmd_completed_hook,
|
||||
_make_serve_handle_session_resume_cb,
|
||||
_make_serve_start_new_session_cb,
|
||||
_serve_process_message,
|
||||
)
|
||||
@@ -102,6 +105,7 @@ def test_hook_updates_thread_id_on_resume():
|
||||
ctx = MagicMock()
|
||||
ctx.agent = "a" # no agent swap
|
||||
ctx.thread_id = "new-tid"
|
||||
ctx.workspace_dir = None
|
||||
cmd = MagicMock()
|
||||
cmd.name = "/resume"
|
||||
|
||||
@@ -111,20 +115,39 @@ def test_hook_updates_thread_id_on_resume():
|
||||
|
||||
|
||||
def test_hook_updates_workspace_dir_on_resume():
|
||||
"""`/resume` can restore a different workspace; serve must adopt it."""
|
||||
holder = {"agent": "a", "thread_id": "original-tid", "workspace_dir": "/old-ws"}
|
||||
hook = _make_serve_cmd_completed_hook(holder)
|
||||
"""`/resume` can restore a different workspace; serve must reload for it."""
|
||||
cfg = object()
|
||||
holder = {
|
||||
"agent": "old-agent",
|
||||
"thread_id": "original-tid",
|
||||
"workspace_dir": "/old-ws",
|
||||
"config": cfg,
|
||||
}
|
||||
hook = _make_serve_cmd_completed_hook(holder, config=cfg)
|
||||
|
||||
ctx = MagicMock()
|
||||
ctx.agent = "a"
|
||||
ctx.agent = "old-agent"
|
||||
ctx.thread_id = "new-tid"
|
||||
ctx.workspace_dir = "/restored-ws"
|
||||
cmd = MagicMock()
|
||||
cmd.name = "/resume"
|
||||
|
||||
_run(hook(ctx, "a", cmd))
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.cli.commands._sync_background_agent_server_workspace",
|
||||
new=AsyncMock(),
|
||||
) as sync_server,
|
||||
patch(
|
||||
"EvoScientist.cli.commands._load_agent",
|
||||
return_value="reloaded-agent",
|
||||
) as load_agent,
|
||||
):
|
||||
_run(hook(ctx, "old-agent", cmd))
|
||||
|
||||
sync_server.assert_awaited_once_with(cfg, workspace_dir="/restored-ws")
|
||||
load_agent.assert_called_once_with(workspace_dir="/restored-ws", config=cfg)
|
||||
assert holder["workspace_dir"] == "/restored-ws"
|
||||
assert holder["agent"] == "reloaded-agent"
|
||||
|
||||
|
||||
def test_hook_syncs_channel_runtime_thread_id():
|
||||
@@ -137,6 +160,7 @@ def test_hook_syncs_channel_runtime_thread_id():
|
||||
ctx = MagicMock()
|
||||
ctx.agent = "a"
|
||||
ctx.thread_id = "new-tid"
|
||||
ctx.workspace_dir = None
|
||||
cmd = MagicMock()
|
||||
cmd.name = "/resume"
|
||||
|
||||
@@ -171,6 +195,7 @@ def test_hook_skips_resume_warning_when_thread_unchanged():
|
||||
ctx = MagicMock()
|
||||
ctx.agent = "a"
|
||||
ctx.thread_id = "original-tid" # unchanged — bare /resume case
|
||||
ctx.workspace_dir = None
|
||||
cmd = MagicMock()
|
||||
cmd.name = "/resume"
|
||||
|
||||
@@ -191,6 +216,7 @@ def test_hook_emits_resume_warning_when_thread_changed():
|
||||
ctx.ui.flush = AsyncMock()
|
||||
ctx.agent = "a"
|
||||
ctx.thread_id = "abc12345-resumed-tid"
|
||||
ctx.workspace_dir = None
|
||||
cmd = MagicMock()
|
||||
cmd.name = "/resume"
|
||||
|
||||
@@ -239,6 +265,164 @@ def test_start_new_session_cb_leaves_agent_alone():
|
||||
assert holder["agent"] == "a"
|
||||
|
||||
|
||||
def test_serve_resume_callback_syncs_reloads_and_adopts_workspace():
|
||||
cfg = object()
|
||||
holder = {
|
||||
"agent": "old-agent",
|
||||
"thread_id": "old-tid",
|
||||
"workspace_dir": "/old-ws",
|
||||
"config": cfg,
|
||||
}
|
||||
runtime = ChannelRuntime(agent="old-agent", thread_id="old-tid")
|
||||
cb = _make_serve_handle_session_resume_cb(holder, runtime, config=cfg)
|
||||
call_order: list[str] = []
|
||||
|
||||
def _load_agent(**_kwargs):
|
||||
call_order.append("load")
|
||||
return "reloaded-agent"
|
||||
|
||||
async def _sync_server(*_args, **_kwargs):
|
||||
call_order.append("sync")
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.cli.commands._sync_background_agent_server_workspace",
|
||||
new=AsyncMock(side_effect=_sync_server),
|
||||
) as sync_server,
|
||||
patch(
|
||||
"EvoScientist.cli.commands._load_agent",
|
||||
side_effect=_load_agent,
|
||||
) as load_agent,
|
||||
):
|
||||
_run(cb("new-tid", "/new-ws"))
|
||||
|
||||
sync_server.assert_awaited_once_with(cfg, workspace_dir="/new-ws")
|
||||
load_agent.assert_called_once_with(workspace_dir="/new-ws", config=cfg)
|
||||
assert call_order == ["load", "sync"]
|
||||
assert holder["thread_id"] == "new-tid"
|
||||
assert holder["workspace_dir"] == "/new-ws"
|
||||
assert holder["agent"] == "reloaded-agent"
|
||||
assert runtime.thread_id == "new-tid"
|
||||
assert runtime.agent == "reloaded-agent"
|
||||
|
||||
|
||||
def test_hook_emits_resume_warning_after_resume_callback_adopts_thread():
|
||||
cfg = object()
|
||||
holder = {
|
||||
"agent": "old-agent",
|
||||
"thread_id": "old-tid",
|
||||
"workspace_dir": "/old-ws",
|
||||
"config": cfg,
|
||||
}
|
||||
runtime = ChannelRuntime(agent="old-agent", thread_id="old-tid")
|
||||
cb = _make_serve_handle_session_resume_cb(holder, runtime, config=cfg)
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.cli.commands._sync_background_agent_server_workspace",
|
||||
new=AsyncMock(),
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.cli.commands._load_agent",
|
||||
return_value="reloaded-agent",
|
||||
),
|
||||
):
|
||||
_run(cb("abc12345-resumed-tid", "/new-ws"))
|
||||
|
||||
hook = _make_serve_cmd_completed_hook(holder, runtime, config=cfg)
|
||||
ctx = MagicMock()
|
||||
ctx.ui.flush = AsyncMock()
|
||||
ctx.agent = "reloaded-agent"
|
||||
ctx.thread_id = "abc12345-resumed-tid"
|
||||
ctx.workspace_dir = "/new-ws"
|
||||
cmd = MagicMock()
|
||||
cmd.name = "/resume"
|
||||
|
||||
_run(hook(ctx, "reloaded-agent", cmd))
|
||||
|
||||
ctx.ui.append_system.assert_called_once()
|
||||
assert "in-memory state" in ctx.ui.append_system.call_args.args[0]
|
||||
ctx.ui.flush.assert_awaited_once()
|
||||
|
||||
|
||||
def test_serve_resume_callback_preserves_state_when_sync_fails():
|
||||
cfg = object()
|
||||
holder = {
|
||||
"agent": "old-agent",
|
||||
"thread_id": "old-tid",
|
||||
"workspace_dir": "/old-ws",
|
||||
"config": cfg,
|
||||
}
|
||||
runtime = ChannelRuntime(agent="old-agent", thread_id="old-tid")
|
||||
cb = _make_serve_handle_session_resume_cb(holder, runtime, config=cfg)
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.cli.commands._sync_background_agent_server_workspace",
|
||||
new=AsyncMock(side_effect=RuntimeError("workspace conflict")),
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.cli.commands._load_agent",
|
||||
return_value="loaded-but-not-adopted",
|
||||
) as load_agent,
|
||||
patch("EvoScientist.cli.commands.set_active_workspace") as set_active,
|
||||
pytest.raises(RuntimeError, match="workspace conflict"),
|
||||
):
|
||||
_run(cb("new-tid", "/new-ws"))
|
||||
|
||||
load_agent.assert_called_once_with(workspace_dir="/new-ws", config=cfg)
|
||||
set_active.assert_called_once_with("/old-ws")
|
||||
assert "loaded-but-not-adopted" not in holder.values()
|
||||
assert "_resume_warning_thread_id" not in holder
|
||||
assert holder == {
|
||||
"agent": "old-agent",
|
||||
"thread_id": "old-tid",
|
||||
"workspace_dir": "/old-ws",
|
||||
"config": cfg,
|
||||
}
|
||||
assert runtime.agent == "old-agent"
|
||||
assert runtime.thread_id == "old-tid"
|
||||
|
||||
|
||||
def test_serve_resume_callback_load_failure_does_not_sync_or_adopt():
|
||||
cfg = object()
|
||||
holder = {
|
||||
"agent": "old-agent",
|
||||
"thread_id": "old-tid",
|
||||
"workspace_dir": "/old-ws",
|
||||
"config": cfg,
|
||||
}
|
||||
runtime = ChannelRuntime(agent="old-agent", thread_id="old-tid")
|
||||
cb = _make_serve_handle_session_resume_cb(holder, runtime, config=cfg)
|
||||
|
||||
with (
|
||||
patch(
|
||||
"EvoScientist.cli.commands._load_agent",
|
||||
side_effect=RuntimeError("load failed"),
|
||||
) as load_agent,
|
||||
patch("EvoScientist.cli.commands.set_active_workspace") as set_active,
|
||||
patch(
|
||||
"EvoScientist.cli.commands._sync_background_agent_server_workspace",
|
||||
new=AsyncMock(),
|
||||
) as sync_server,
|
||||
pytest.raises(RuntimeError, match="load failed"),
|
||||
):
|
||||
_run(cb("new-tid", "/new-ws"))
|
||||
|
||||
load_agent.assert_called_once_with(workspace_dir="/new-ws", config=cfg)
|
||||
set_active.assert_called_once_with("/old-ws")
|
||||
sync_server.assert_not_awaited()
|
||||
assert "_resume_warning_thread_id" not in holder
|
||||
assert holder == {
|
||||
"agent": "old-agent",
|
||||
"thread_id": "old-tid",
|
||||
"workspace_dir": "/old-ws",
|
||||
"config": cfg,
|
||||
}
|
||||
assert runtime.agent == "old-agent"
|
||||
assert runtime.thread_id == "old-tid"
|
||||
|
||||
|
||||
def test_hook_handles_both_agent_and_thread_swap():
|
||||
"""Edge case: a command that changes both (hypothetical). Both
|
||||
updates must land in the holder."""
|
||||
|
||||
@@ -4,6 +4,7 @@ from __future__ import annotations
|
||||
|
||||
import asyncio
|
||||
from datetime import datetime, timedelta
|
||||
from types import SimpleNamespace
|
||||
from typing import ClassVar
|
||||
|
||||
from langchain_core.messages import HumanMessage
|
||||
@@ -50,6 +51,90 @@ def test_build_status_fragments_wide_layout():
|
||||
assert "3m" in rendered
|
||||
|
||||
|
||||
def test_build_status_fragments_shows_memory_worker_indicator(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.cli.status_bar.get_memory_worker_status",
|
||||
lambda: SimpleNamespace(
|
||||
is_running=False,
|
||||
profile_updates=4,
|
||||
observations_recorded=5,
|
||||
),
|
||||
)
|
||||
snapshot = SessionStatusSnapshot(
|
||||
model_full="openai/gpt-6",
|
||||
model_short="gpt-6",
|
||||
context_tokens=12_345,
|
||||
context_window=128_000,
|
||||
context_percent=10,
|
||||
)
|
||||
|
||||
fragments = build_status_fragments(
|
||||
snapshot,
|
||||
datetime.now() - timedelta(minutes=3),
|
||||
100,
|
||||
)
|
||||
|
||||
assert any(
|
||||
style == "class:status-bar-warn" and text.strip() for style, text in fragments
|
||||
)
|
||||
|
||||
|
||||
def test_build_status_fragments_shows_memory_worker_indicator_when_running(
|
||||
monkeypatch,
|
||||
):
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.cli.status_bar.get_memory_worker_status",
|
||||
lambda: SimpleNamespace(
|
||||
is_running=True,
|
||||
profile_updates=0,
|
||||
observations_recorded=0,
|
||||
),
|
||||
)
|
||||
snapshot = SessionStatusSnapshot(
|
||||
model_full="openai/gpt-6",
|
||||
model_short="gpt-6",
|
||||
context_tokens=12_345,
|
||||
context_window=128_000,
|
||||
context_percent=10,
|
||||
)
|
||||
|
||||
fragments = build_status_fragments(
|
||||
snapshot,
|
||||
datetime.now() - timedelta(minutes=3),
|
||||
100,
|
||||
)
|
||||
|
||||
assert any(
|
||||
style == "class:status-bar-warn" and text.strip() for style, text in fragments
|
||||
)
|
||||
|
||||
|
||||
def test_build_status_fragments_hides_memory_indicator_when_idle(monkeypatch):
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.cli.status_bar.get_memory_worker_status",
|
||||
lambda: SimpleNamespace(
|
||||
is_running=False,
|
||||
profile_updates=0,
|
||||
observations_recorded=0,
|
||||
),
|
||||
)
|
||||
snapshot = SessionStatusSnapshot(
|
||||
model_full="openai/gpt-6",
|
||||
model_short="gpt-6",
|
||||
context_tokens=12_345,
|
||||
context_window=128_000,
|
||||
context_percent=10,
|
||||
)
|
||||
|
||||
fragments = build_status_fragments(
|
||||
snapshot,
|
||||
datetime.now() - timedelta(minutes=3),
|
||||
100,
|
||||
)
|
||||
|
||||
assert not any(style == "class:status-bar-warn" for style, _text in fragments)
|
||||
|
||||
|
||||
def test_build_status_fragments_medium_layout():
|
||||
snapshot = SessionStatusSnapshot(
|
||||
model_full="openai/gpt-5.4",
|
||||
|
||||
@@ -5,6 +5,7 @@ from types import SimpleNamespace
|
||||
from unittest.mock import AsyncMock, MagicMock
|
||||
|
||||
from langchain_core.messages import AIMessageChunk
|
||||
from langgraph.types import Command
|
||||
|
||||
from EvoScientist.stream.events import (
|
||||
_extract_summary_message_text,
|
||||
@@ -235,6 +236,34 @@ class TestMultiModeChunkUnpacking:
|
||||
assert len(text_events) == 1
|
||||
assert text_events[0]["content"] == "should appear"
|
||||
|
||||
def test_user_message_clears_memory_worker_saved_counts(self, monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.stream.events.clear_memory_worker_saved_counts",
|
||||
lambda: calls.append(True),
|
||||
)
|
||||
mock_agent = AsyncMock()
|
||||
mock_agent.astream = MagicMock(return_value=_async_iter([]))
|
||||
|
||||
_collect_events(mock_agent, message="new user turn")
|
||||
|
||||
assert calls == [True]
|
||||
|
||||
def test_command_message_clears_memory_worker_saved_counts(self, monkeypatch):
|
||||
calls = []
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.stream.events.clear_memory_worker_saved_counts",
|
||||
lambda: calls.append(True),
|
||||
)
|
||||
mock_agent = AsyncMock()
|
||||
mock_agent.astream = MagicMock(return_value=_async_iter([]))
|
||||
resume_command = Command(resume={"decisions": [{"type": "approve"}]})
|
||||
|
||||
_collect_events(mock_agent, message=resume_command)
|
||||
|
||||
assert calls == [True]
|
||||
assert mock_agent.astream.call_args.args[0] is resume_command
|
||||
|
||||
def test_summarization_filtered(self):
|
||||
"""Chunks with lc_source=summarization metadata are filtered out."""
|
||||
chunk_real = _make_ai_chunk("real content")
|
||||
|
||||
Reference in New Issue
Block a user