92d95dee68
* feat(memory): add observation memory lifecycle Add file-backed observation memory with deterministic markdown records, structured record_observation tooling, startup indexing, and profile/observation prompt guidance. Launch post-turn and post-subagent EvoMemory workers through LangGraph dev so completed runs can update profile memory, save durable observations, and write subagent execution summaries without blocking the active agent. Wire memory middleware into the main agent, subagents, async graphs, TUI status reporting, worker activity accounting, and observation-aware research prompts, with regression coverage for storage, lifecycle scheduling, graph registration, status display, and stream reset behavior. * fix(cli): sync background agent server on resume Resume flows now need to keep the LangGraph dev background server aligned with the active workspace even when async subagents are disabled. EvoMemory workers use that server too, so gating resume-time sync on enable_async_subagents could leave workers pinned to the launch workspace after resuming a thread from another workspace. Run workspace sync unconditionally for Rich CLI and Textual resume paths, while preserving WorkspaceMismatchError handling so failed sync aborts the resume before mutating the active thread or workspace. Propagate aborted resume callbacks through the command UI so channel-issued /resume commands do not send false success or history output. Channel slash dispatch now treats CommandManager-caught command errors as command errors and skips completion hooks for those failed commands. Add regression coverage for disabled async subagents, callback aborts, and channel command error reporting. * fix(cli): prepare serve resume workspace before adopting Load the resumed workspace agent and sync the background server as a single pre-adoption step. Restore the previous active workspace if preparation fails so serve mode keeps using the old session consistently. * fix(memory): untrack abandoned worker status watches Stop treating watcher shutdown as confirmed worker completion. Terminal worker statuses still count memory deltas, while poll failures or watcher setup failures now remove the active run without crediting partial outputs. * fix(cli): report channel command failures accurately Treat command_error as a None sentinel so empty error strings still fail, and let TUI resumes continue only on non-mismatch background-server sync failures while reporting degraded mode. * fix(stream): clear memory counters for resume streams Reset completed-memory counters for every new agent stream, including Command-based HITL and resume streams, so saved-memory indicators do not leak across turns. * docs(tools): make observation recording guidance conditional Clarify that agents should call record_observation only when the observation tool is available, preserving the existing durability and usefulness criteria. * feat(config): add controls for profile and observation memory Add config flags for profile memory, observation memory, observation writer placement, and background memory workers. Wire the controls through main agents, subagents, EvoMemory middleware, and memory lifecycle workers so observation writes can be assigned to the live agent, subagent worker, both, or neither. Keep turn memory workers profile-only and make prompts reflect the available observation read/write paths. Skip langgraph dev startup when neither async subagents nor memory workers need the background server. Add coverage for config parsing, prompt gating, middleware wiring, and worker tool availability. * test(cli): include memory defaults in serve config stubs * fix(memory): offload async worker launch blocking calls Run the langgraph-dev health check and memory-output snapshot in worker threads from the async EvoMemory launcher so it does not block the event loop. * chore(memory): harden turn worker subagent guardrail * chore(memory): refresh profile context per request * fix(memory): offload async profile file reads * fix(memory): offload async worker completion accounting
554 lines
18 KiB
Python
554 lines
18 KiB
Python
"""Tests for ``dispatch_channel_slash_command`` in ``cli/channel.py``.
|
|
|
|
Regression coverage for issue #181 — slash commands arriving over a
|
|
messaging channel must route through ``cmd_manager`` instead of being
|
|
fed to the LLM as a plain prompt, on every UI surface (Rich CLI, TUI,
|
|
headless ``serve``).
|
|
"""
|
|
|
|
from unittest.mock import AsyncMock, MagicMock, patch
|
|
|
|
from EvoScientist.cli.channel import (
|
|
ChannelMessage,
|
|
dispatch_channel_slash_command,
|
|
)
|
|
from tests.conftest import run_async as _run
|
|
|
|
|
|
def _make_msg(
|
|
content: str = "/evoskills core", msg_id: str = "msg-1"
|
|
) -> ChannelMessage:
|
|
return ChannelMessage(
|
|
msg_id=msg_id,
|
|
content=content,
|
|
sender="+44XXXXXX",
|
|
channel_type="imessage",
|
|
metadata={},
|
|
channel_ref=None,
|
|
bus_ref=None,
|
|
chat_id="+44XXXXXX",
|
|
message_id="ts-1",
|
|
)
|
|
|
|
|
|
def test_non_slash_returns_false():
|
|
"""Plain text messages must fall through to the agent."""
|
|
msg = _make_msg(content="hello agent")
|
|
append = MagicMock()
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=append,
|
|
)
|
|
)
|
|
assert handled is False
|
|
append.assert_not_called()
|
|
|
|
|
|
def test_unresolved_slash_returns_false():
|
|
"""Unknown slash commands must fall through (matches TUI behavior)."""
|
|
msg = _make_msg(content="/unknown-cmd")
|
|
append = MagicMock()
|
|
with patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=None,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=append,
|
|
)
|
|
)
|
|
assert handled is False
|
|
|
|
|
|
def test_successful_slash_execution_sets_response_and_breadcrumb():
|
|
"""Known slash command: cmd_manager.execute ran, helper returns True,
|
|
sends a confirmation to the channel user, and appends a local log line."""
|
|
msg = _make_msg()
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
append = MagicMock()
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, ["core"]),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=AsyncMock(return_value=True),
|
|
) as mock_execute,
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent="fake-agent",
|
|
thread_id="t1",
|
|
workspace_dir="/tmp",
|
|
checkpointer=None,
|
|
append_system=append,
|
|
)
|
|
)
|
|
assert handled is True
|
|
mock_execute.assert_awaited_once()
|
|
mock_set_resp.assert_called_once()
|
|
assert mock_set_resp.call_args[0][0] == "msg-1"
|
|
assert "Command executed" in mock_set_resp.call_args[0][1]
|
|
breadcrumbs = [call.args[0] for call in append.call_args_list]
|
|
assert any("Executed command from" in t for t in breadcrumbs)
|
|
|
|
|
|
def test_needs_agent_awaits_loader_and_passes_result():
|
|
"""Commands with needs_agent=True must await the loader and the
|
|
resulting agent must flow through the CommandContext."""
|
|
msg = _make_msg()
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = True
|
|
append = MagicMock()
|
|
await_called = MagicMock()
|
|
|
|
async def _await_ready():
|
|
await_called()
|
|
return "ready-agent"
|
|
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=AsyncMock(return_value=True),
|
|
) as mock_execute,
|
|
patch("EvoScientist.cli.channel._set_channel_response"),
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=append,
|
|
await_agent_ready=_await_ready,
|
|
)
|
|
)
|
|
assert handled is True
|
|
await_called.assert_called_once()
|
|
ctx_arg = mock_execute.await_args.args[1]
|
|
assert ctx_arg.agent == "ready-agent"
|
|
|
|
|
|
def test_await_agent_ready_failure_sets_error_response():
|
|
msg = _make_msg()
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = True
|
|
append = MagicMock()
|
|
|
|
async def _await_ready():
|
|
raise RuntimeError("agent blew up")
|
|
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=append,
|
|
await_agent_ready=_await_ready,
|
|
)
|
|
)
|
|
assert handled is True
|
|
mock_set_resp.assert_called_once()
|
|
resp_text = mock_set_resp.call_args[0][1]
|
|
assert "Command error" in resp_text
|
|
assert "agent blew up" in resp_text
|
|
|
|
|
|
def test_cmd_manager_raises_returns_true_with_error():
|
|
"""If cmd_manager.execute raises past its own try/except, the helper
|
|
must absorb it, return True, and report via _set_channel_response."""
|
|
msg = _make_msg()
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
append = MagicMock()
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=AsyncMock(side_effect=RuntimeError("boom")),
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=append,
|
|
)
|
|
)
|
|
assert handled is True
|
|
mock_set_resp.assert_called_once()
|
|
resp_text = mock_set_resp.call_args[0][1]
|
|
assert "Command error" in resp_text
|
|
assert "boom" in resp_text
|
|
|
|
|
|
def test_on_cmd_completed_awaited_with_ctx_original_agent_and_cmd():
|
|
"""After a successful slash execute, the on_cmd_completed hook must
|
|
be awaited with (ctx, original_agent, cmd) so Rich CLI can adopt an
|
|
``/model`` agent swap and refresh status for state-mutating commands."""
|
|
msg = _make_msg()
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
fake_cmd.name = "/model"
|
|
append = MagicMock()
|
|
captured: dict[str, object] = {}
|
|
|
|
async def _fake_execute(command_str, ctx):
|
|
# Simulate /model swapping ctx.agent to a new handle.
|
|
ctx.agent = "swapped-agent"
|
|
return True
|
|
|
|
async def _on_completed(ctx, original_agent, cmd):
|
|
captured["ctx_agent"] = ctx.agent
|
|
captured["original_agent"] = original_agent
|
|
captured["cmd_name"] = getattr(cmd, "name", None)
|
|
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=_fake_execute,
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response"),
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent="original-agent",
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=append,
|
|
on_cmd_completed=_on_completed,
|
|
)
|
|
)
|
|
assert handled is True
|
|
assert captured["ctx_agent"] == "swapped-agent"
|
|
assert captured["original_agent"] == "original-agent"
|
|
assert captured["cmd_name"] == "/model"
|
|
|
|
|
|
def test_on_cmd_completed_receives_cmd_for_new_and_compact():
|
|
"""``/new`` / ``/compact`` invoked via channel must flow the cmd into
|
|
the hook so the callback can still refresh status when the agent
|
|
didn't swap — mirrors REPL ``interactive.py:1027-1030``."""
|
|
for cmd_name in ("/new", "/compact"):
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
fake_cmd.name = cmd_name
|
|
captured: dict[str, str | None] = {}
|
|
|
|
async def _on_completed(ctx, original_agent, cmd, _captured=captured):
|
|
_captured["cmd_name"] = getattr(cmd, "name", None)
|
|
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=AsyncMock(return_value=True),
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response"),
|
|
):
|
|
_run(
|
|
dispatch_channel_slash_command(
|
|
_make_msg(content=cmd_name),
|
|
agent="same-agent",
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
on_cmd_completed=_on_completed,
|
|
)
|
|
)
|
|
assert captured["cmd_name"] == cmd_name, cmd_name
|
|
|
|
|
|
def test_on_cmd_completed_skipped_on_fall_through_and_error():
|
|
"""The hook must NOT fire for unresolved slash, non-slash text, or
|
|
when cmd_manager.execute raised."""
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
completed = MagicMock()
|
|
|
|
async def _noop(*args, **kwargs):
|
|
completed(*args, **kwargs)
|
|
|
|
# Non-slash
|
|
with patch("EvoScientist.cli.channel._set_channel_response"):
|
|
_run(
|
|
dispatch_channel_slash_command(
|
|
_make_msg(content="hi"),
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
on_cmd_completed=_noop,
|
|
)
|
|
)
|
|
# Unresolved slash
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=None,
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response"),
|
|
):
|
|
_run(
|
|
dispatch_channel_slash_command(
|
|
_make_msg(content="/nope"),
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
on_cmd_completed=_noop,
|
|
)
|
|
)
|
|
# Execute raises
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=AsyncMock(side_effect=RuntimeError("boom")),
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response"),
|
|
):
|
|
_run(
|
|
dispatch_channel_slash_command(
|
|
_make_msg(),
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
on_cmd_completed=_noop,
|
|
)
|
|
)
|
|
|
|
completed.assert_not_called()
|
|
|
|
|
|
def test_command_error_skips_completion_hook_and_reports_error():
|
|
"""A command caught as failed by CommandManager must not look successful."""
|
|
msg = _make_msg(content="/resume abc")
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
completed = AsyncMock()
|
|
|
|
async def _execute(_command, ctx):
|
|
ctx.command_error = "workspace conflict"
|
|
ctx.ui.append_system("Error executing /resume: workspace conflict", style="red")
|
|
await ctx.ui.flush()
|
|
return True
|
|
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, ["abc"]),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
side_effect=_execute,
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="old-thread",
|
|
workspace_dir="/old-workspace",
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
on_cmd_completed=completed,
|
|
)
|
|
)
|
|
|
|
assert handled is True
|
|
completed.assert_not_awaited()
|
|
mock_set_resp.assert_called_once_with("msg-1", "Command error: workspace conflict")
|
|
|
|
|
|
def test_empty_command_error_still_reports_error():
|
|
"""An empty string error is still a command failure sentinel."""
|
|
msg = _make_msg(content="/resume abc")
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
completed = AsyncMock()
|
|
|
|
async def _execute(_command, ctx):
|
|
ctx.command_error = ""
|
|
return True
|
|
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, ["abc"]),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
side_effect=_execute,
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="old-thread",
|
|
workspace_dir="/old-workspace",
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
on_cmd_completed=completed,
|
|
)
|
|
)
|
|
|
|
assert handled is True
|
|
completed.assert_not_awaited()
|
|
mock_set_resp.assert_called_once_with("msg-1", "Command error: (no details)")
|
|
|
|
|
|
def test_on_cmd_completed_exception_is_absorbed():
|
|
"""A raising hook must NOT prevent the channel response from being set."""
|
|
msg = _make_msg()
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
|
|
async def _boom(ctx, original_agent, cmd):
|
|
raise RuntimeError("hook blew up")
|
|
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=AsyncMock(return_value=True),
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent="orig",
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
on_cmd_completed=_boom,
|
|
)
|
|
)
|
|
assert handled is True
|
|
mock_set_resp.assert_called_once()
|
|
assert "Command executed" in mock_set_resp.call_args[0][1]
|
|
|
|
|
|
def test_top_level_exception_is_absorbed():
|
|
"""Last-ditch safety net: if anything inside the dispatch pipeline
|
|
raises unexpectedly (lazy import failure, ChannelCommandUI ctor,
|
|
terminal I/O from append_system, ...), the helper must NOT
|
|
propagate — it sets an error response and returns True so the
|
|
caller's polling loop stays alive and doesn't fall through to the
|
|
agent streaming path."""
|
|
msg = _make_msg()
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
side_effect=RuntimeError("exploded during resolve"),
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=MagicMock(),
|
|
)
|
|
)
|
|
assert handled is True
|
|
mock_set_resp.assert_called_once()
|
|
resp_text = mock_set_resp.call_args[0][1]
|
|
assert "Command error" in resp_text
|
|
assert "exploded during resolve" in resp_text
|
|
|
|
|
|
def test_cmd_execute_returning_false_falls_through():
|
|
"""When cmd_manager.execute returns False (empty/unparseable input),
|
|
the helper must return False so the caller falls through to the agent."""
|
|
msg = _make_msg(content="/")
|
|
fake_cmd = MagicMock()
|
|
fake_cmd.needs_agent.return_value = False
|
|
append = MagicMock()
|
|
with (
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.resolve",
|
|
return_value=(fake_cmd, []),
|
|
),
|
|
patch(
|
|
"EvoScientist.commands.manager.manager.execute",
|
|
new=AsyncMock(return_value=False),
|
|
),
|
|
patch("EvoScientist.cli.channel._set_channel_response") as mock_set_resp,
|
|
):
|
|
handled = _run(
|
|
dispatch_channel_slash_command(
|
|
msg,
|
|
agent=None,
|
|
thread_id="t1",
|
|
workspace_dir=None,
|
|
checkpointer=None,
|
|
append_system=append,
|
|
)
|
|
)
|
|
assert handled is False
|
|
mock_set_resp.assert_not_called()
|