92d95dee68
* feat(memory): add observation memory lifecycle Add file-backed observation memory with deterministic markdown records, structured record_observation tooling, startup indexing, and profile/observation prompt guidance. Launch post-turn and post-subagent EvoMemory workers through LangGraph dev so completed runs can update profile memory, save durable observations, and write subagent execution summaries without blocking the active agent. Wire memory middleware into the main agent, subagents, async graphs, TUI status reporting, worker activity accounting, and observation-aware research prompts, with regression coverage for storage, lifecycle scheduling, graph registration, status display, and stream reset behavior. * fix(cli): sync background agent server on resume Resume flows now need to keep the LangGraph dev background server aligned with the active workspace even when async subagents are disabled. EvoMemory workers use that server too, so gating resume-time sync on enable_async_subagents could leave workers pinned to the launch workspace after resuming a thread from another workspace. Run workspace sync unconditionally for Rich CLI and Textual resume paths, while preserving WorkspaceMismatchError handling so failed sync aborts the resume before mutating the active thread or workspace. Propagate aborted resume callbacks through the command UI so channel-issued /resume commands do not send false success or history output. Channel slash dispatch now treats CommandManager-caught command errors as command errors and skips completion hooks for those failed commands. Add regression coverage for disabled async subagents, callback aborts, and channel command error reporting. * fix(cli): prepare serve resume workspace before adopting Load the resumed workspace agent and sync the background server as a single pre-adoption step. Restore the previous active workspace if preparation fails so serve mode keeps using the old session consistently. * fix(memory): untrack abandoned worker status watches Stop treating watcher shutdown as confirmed worker completion. Terminal worker statuses still count memory deltas, while poll failures or watcher setup failures now remove the active run without crediting partial outputs. * fix(cli): report channel command failures accurately Treat command_error as a None sentinel so empty error strings still fail, and let TUI resumes continue only on non-mismatch background-server sync failures while reporting degraded mode. * fix(stream): clear memory counters for resume streams Reset completed-memory counters for every new agent stream, including Command-based HITL and resume streams, so saved-memory indicators do not leak across turns. * docs(tools): make observation recording guidance conditional Clarify that agents should call record_observation only when the observation tool is available, preserving the existing durability and usefulness criteria. * feat(config): add controls for profile and observation memory Add config flags for profile memory, observation memory, observation writer placement, and background memory workers. Wire the controls through main agents, subagents, EvoMemory middleware, and memory lifecycle workers so observation writes can be assigned to the live agent, subagent worker, both, or neither. Keep turn memory workers profile-only and make prompts reflect the available observation read/write paths. Skip langgraph dev startup when neither async subagents nor memory workers need the background server. Add coverage for config parsing, prompt gating, middleware wiring, and worker tool availability. * test(cli): include memory defaults in serve config stubs * fix(memory): offload async worker launch blocking calls Run the langgraph-dev health check and memory-output snapshot in worker threads from the async EvoMemory launcher so it does not block the event loop. * chore(memory): harden turn worker subagent guardrail * chore(memory): refresh profile context per request * fix(memory): offload async profile file reads * fix(memory): offload async worker completion accounting
249 lines
7.5 KiB
Python
249 lines
7.5 KiB
Python
"""Tests for CLI serve/channel glue behavior."""
|
|
|
|
from __future__ import annotations
|
|
|
|
import os
|
|
from types import SimpleNamespace
|
|
|
|
from EvoScientist.cli import commands
|
|
from EvoScientist.config import MemoryObservationWriter
|
|
|
|
|
|
def _make_config(
|
|
*,
|
|
default_workdir: str = "",
|
|
channel_send_thinking: bool = True,
|
|
log_level: str = "warning",
|
|
channel_debug_tracing: bool = False,
|
|
auto_approve: bool = False,
|
|
auto_mode: bool = False,
|
|
enable_ask_user: bool = True,
|
|
):
|
|
return SimpleNamespace(
|
|
channel_enabled="telegram",
|
|
default_workdir=default_workdir,
|
|
channel_send_thinking=channel_send_thinking,
|
|
log_level=log_level,
|
|
channel_debug_tracing=channel_debug_tracing,
|
|
auto_approve=auto_approve,
|
|
auto_mode=auto_mode,
|
|
enable_ask_user=enable_ask_user,
|
|
enable_async_subagents=False,
|
|
memory_profile_enabled=True,
|
|
memory_observations_enabled=True,
|
|
memory_observation_writer=MemoryObservationWriter.ALL,
|
|
memory_workers_enabled=False,
|
|
provider="anthropic",
|
|
anthropic_auth_mode="api_key",
|
|
openai_auth_mode="api_key",
|
|
)
|
|
|
|
|
|
def _run_serve_once(
|
|
monkeypatch,
|
|
config,
|
|
*,
|
|
workdir: str | None = None,
|
|
no_thinking: bool = False,
|
|
debug: bool = False,
|
|
cwd: str | None = None,
|
|
auto_approve: bool = False,
|
|
auto_mode: bool = False,
|
|
ask_user: bool = False,
|
|
):
|
|
import EvoScientist.config as config_mod
|
|
|
|
order: list[tuple[str, str | None]] = []
|
|
captured: dict[str, object] = {}
|
|
|
|
def _fake_set_workspace_root(path):
|
|
order.append(("set_workspace_root", str(path)))
|
|
|
|
def _fake_ensure_dirs():
|
|
order.append(("ensure_dirs", None))
|
|
|
|
def _fake_load_agent(workspace_dir=None, checkpointer=None, config=None):
|
|
captured["workspace_dir"] = workspace_dir
|
|
return object()
|
|
|
|
def _fake_start_channels_bus_mode(cfg, agent, thread_id, *, send_thinking=None):
|
|
captured["started"] = True
|
|
captured["send_thinking"] = send_thinking
|
|
captured["thread_id"] = thread_id
|
|
|
|
def _fake_channels_stop(channel_type=None, *, runtime=None):
|
|
captured["stopped"] = True
|
|
|
|
class _InterruptQueue:
|
|
"""A fake queue whose get() immediately raises KeyboardInterrupt."""
|
|
|
|
def get(self, timeout=None):
|
|
raise KeyboardInterrupt()
|
|
|
|
monkeypatch.setattr(commands, "set_workspace_root", _fake_set_workspace_root)
|
|
monkeypatch.setattr(commands, "ensure_dirs", _fake_ensure_dirs)
|
|
monkeypatch.setattr(commands, "_load_agent", _fake_load_agent)
|
|
monkeypatch.setattr(
|
|
commands, "_start_channels_bus_mode", _fake_start_channels_bus_mode
|
|
)
|
|
monkeypatch.setattr(commands, "_channels_stop", _fake_channels_stop)
|
|
monkeypatch.setattr(commands, "_message_queue", _InterruptQueue())
|
|
|
|
def _fake_get_effective_config(cli_overrides=None):
|
|
captured["cli_overrides"] = dict(cli_overrides or {})
|
|
merged = vars(config).copy()
|
|
merged.update(cli_overrides or {})
|
|
return SimpleNamespace(**merged)
|
|
|
|
monkeypatch.setattr(config_mod, "get_effective_config", _fake_get_effective_config)
|
|
monkeypatch.setattr(config_mod, "apply_config_to_env", lambda _cfg: None)
|
|
|
|
if cwd is not None:
|
|
monkeypatch.setattr(commands.os, "getcwd", lambda: cwd)
|
|
|
|
commands.serve(
|
|
no_thinking=no_thinking,
|
|
workdir=workdir,
|
|
debug=debug,
|
|
auto_approve=auto_approve,
|
|
auto_mode=auto_mode,
|
|
ask_user=ask_user,
|
|
)
|
|
return order, captured
|
|
|
|
|
|
def test_serve_workdir_has_highest_priority_and_sets_root_before_ensure(
|
|
monkeypatch, tmp_path
|
|
):
|
|
cfg_ws = tmp_path / "cfg_ws"
|
|
cli_ws = tmp_path / "cli_ws"
|
|
config = _make_config(default_workdir=str(cfg_ws), channel_send_thinking=True)
|
|
|
|
order, captured = _run_serve_once(
|
|
monkeypatch,
|
|
config,
|
|
workdir=str(cli_ws),
|
|
)
|
|
|
|
expected = str(cli_ws.resolve())
|
|
assert captured["workspace_dir"] == expected
|
|
assert any(step == ("set_workspace_root", expected) for step in order)
|
|
set_idx = next(i for i, step in enumerate(order) if step[0] == "set_workspace_root")
|
|
ensure_idx = next(i for i, step in enumerate(order) if step[0] == "ensure_dirs")
|
|
assert set_idx < ensure_idx
|
|
|
|
|
|
def test_serve_uses_config_default_workdir_when_no_cli_workdir(monkeypatch, tmp_path):
|
|
cfg_ws = tmp_path / "cfg_ws"
|
|
config = _make_config(default_workdir=str(cfg_ws), channel_send_thinking=True)
|
|
|
|
order, captured = _run_serve_once(monkeypatch, config)
|
|
|
|
expected = str(cfg_ws.resolve())
|
|
assert captured["workspace_dir"] == expected
|
|
assert ("set_workspace_root", expected) in order
|
|
|
|
|
|
def test_serve_uses_cwd_when_no_workdir_config(monkeypatch, tmp_path):
|
|
cwd = str(tmp_path.resolve())
|
|
config = _make_config(default_workdir="", channel_send_thinking=True)
|
|
|
|
order, captured = _run_serve_once(
|
|
monkeypatch,
|
|
config,
|
|
cwd=cwd,
|
|
)
|
|
|
|
assert captured["workspace_dir"] == cwd
|
|
assert ("set_workspace_root", cwd) in order
|
|
|
|
|
|
def test_serve_channel_thinking_respects_config_and_no_thinking(monkeypatch, tmp_path):
|
|
ws = str((tmp_path / "ws").resolve())
|
|
|
|
_, captured_default = _run_serve_once(
|
|
monkeypatch,
|
|
_make_config(default_workdir=ws, channel_send_thinking=True),
|
|
)
|
|
assert captured_default["send_thinking"] is True
|
|
|
|
_, captured_cfg_off = _run_serve_once(
|
|
monkeypatch,
|
|
_make_config(default_workdir=ws, channel_send_thinking=False),
|
|
)
|
|
assert captured_cfg_off["send_thinking"] is False
|
|
|
|
_, captured_cli_off = _run_serve_once(
|
|
monkeypatch,
|
|
_make_config(default_workdir=ws, channel_send_thinking=True),
|
|
no_thinking=True,
|
|
)
|
|
assert captured_cli_off["send_thinking"] is False
|
|
|
|
|
|
def test_serve_debug_sets_log_level_and_channel_trace(monkeypatch, tmp_path):
|
|
ws = str((tmp_path / "ws").resolve())
|
|
config = _make_config(default_workdir=ws, channel_send_thinking=True)
|
|
configure_calls: list[tuple[str | None, str | None]] = []
|
|
monkeypatch.delenv("EVOSCIENTIST_LOG_LEVEL", raising=False)
|
|
monkeypatch.delenv("EVOSCIENTIST_CHANNEL_DEBUG_TRACING", raising=False)
|
|
|
|
monkeypatch.setattr(
|
|
commands,
|
|
"_configure_logging",
|
|
lambda: configure_calls.append(
|
|
(
|
|
os.environ.get("EVOSCIENTIST_LOG_LEVEL"),
|
|
os.environ.get("EVOSCIENTIST_CHANNEL_DEBUG_TRACING"),
|
|
)
|
|
),
|
|
)
|
|
|
|
_, captured = _run_serve_once(
|
|
monkeypatch,
|
|
config,
|
|
workdir=ws,
|
|
debug=True,
|
|
)
|
|
|
|
assert captured["cli_overrides"] == {
|
|
"log_level": "DEBUG",
|
|
"channel_debug_tracing": True,
|
|
}
|
|
assert configure_calls == [("DEBUG", "true")]
|
|
|
|
|
|
def test_serve_auto_approve_only_sets_auto_approve(monkeypatch, tmp_path):
|
|
ws = str((tmp_path / "ws").resolve())
|
|
config = _make_config(default_workdir=ws, enable_ask_user=True)
|
|
|
|
_, captured = _run_serve_once(
|
|
monkeypatch,
|
|
config,
|
|
workdir=ws,
|
|
auto_approve=True,
|
|
)
|
|
|
|
assert captured["cli_overrides"] == {"auto_approve": True}
|
|
|
|
|
|
def test_serve_auto_mode_implies_auto_approve_and_disables_ask_user(
|
|
monkeypatch, tmp_path
|
|
):
|
|
ws = str((tmp_path / "ws").resolve())
|
|
config = _make_config(default_workdir=ws, enable_ask_user=True)
|
|
|
|
_, captured = _run_serve_once(
|
|
monkeypatch,
|
|
config,
|
|
workdir=ws,
|
|
auto_mode=True,
|
|
ask_user=True,
|
|
)
|
|
|
|
assert captured["cli_overrides"] == {
|
|
"auto_mode": True,
|
|
"auto_approve": True,
|
|
"enable_ask_user": False,
|
|
}
|