92d95dee68
* feat(memory): add observation memory lifecycle Add file-backed observation memory with deterministic markdown records, structured record_observation tooling, startup indexing, and profile/observation prompt guidance. Launch post-turn and post-subagent EvoMemory workers through LangGraph dev so completed runs can update profile memory, save durable observations, and write subagent execution summaries without blocking the active agent. Wire memory middleware into the main agent, subagents, async graphs, TUI status reporting, worker activity accounting, and observation-aware research prompts, with regression coverage for storage, lifecycle scheduling, graph registration, status display, and stream reset behavior. * fix(cli): sync background agent server on resume Resume flows now need to keep the LangGraph dev background server aligned with the active workspace even when async subagents are disabled. EvoMemory workers use that server too, so gating resume-time sync on enable_async_subagents could leave workers pinned to the launch workspace after resuming a thread from another workspace. Run workspace sync unconditionally for Rich CLI and Textual resume paths, while preserving WorkspaceMismatchError handling so failed sync aborts the resume before mutating the active thread or workspace. Propagate aborted resume callbacks through the command UI so channel-issued /resume commands do not send false success or history output. Channel slash dispatch now treats CommandManager-caught command errors as command errors and skips completion hooks for those failed commands. Add regression coverage for disabled async subagents, callback aborts, and channel command error reporting. * fix(cli): prepare serve resume workspace before adopting Load the resumed workspace agent and sync the background server as a single pre-adoption step. Restore the previous active workspace if preparation fails so serve mode keeps using the old session consistently. * fix(memory): untrack abandoned worker status watches Stop treating watcher shutdown as confirmed worker completion. Terminal worker statuses still count memory deltas, while poll failures or watcher setup failures now remove the active run without crediting partial outputs. * fix(cli): report channel command failures accurately Treat command_error as a None sentinel so empty error strings still fail, and let TUI resumes continue only on non-mismatch background-server sync failures while reporting degraded mode. * fix(stream): clear memory counters for resume streams Reset completed-memory counters for every new agent stream, including Command-based HITL and resume streams, so saved-memory indicators do not leak across turns. * docs(tools): make observation recording guidance conditional Clarify that agents should call record_observation only when the observation tool is available, preserving the existing durability and usefulness criteria. * feat(config): add controls for profile and observation memory Add config flags for profile memory, observation memory, observation writer placement, and background memory workers. Wire the controls through main agents, subagents, EvoMemory middleware, and memory lifecycle workers so observation writes can be assigned to the live agent, subagent worker, both, or neither. Keep turn memory workers profile-only and make prompts reflect the available observation read/write paths. Skip langgraph dev startup when neither async subagents nor memory workers need the background server. Add coverage for config parsing, prompt gating, middleware wiring, and worker tool availability. * test(cli): include memory defaults in serve config stubs * fix(memory): offload async worker launch blocking calls Run the langgraph-dev health check and memory-output snapshot in worker threads from the async EvoMemory launcher so it does not block the event loop. * chore(memory): harden turn worker subagent guardrail * chore(memory): refresh profile context per request * fix(memory): offload async profile file reads * fix(memory): offload async worker completion accounting
185 lines
6.4 KiB
Python
185 lines
6.4 KiB
Python
"""Tests for EvoScientist/prompts.py."""
|
|
|
|
from EvoScientist.prompts import (
|
|
DELEGATION_STRATEGY,
|
|
EVOSCIENTIST_IDENTITY,
|
|
EXPERIMENT_WORKFLOW,
|
|
REPORT_TEMPLATE,
|
|
SHELL_GUIDELINES,
|
|
WRITING_GUIDELINES,
|
|
get_system_prompt,
|
|
)
|
|
|
|
|
|
class TestGetSystemPrompt:
|
|
def test_returns_non_empty(self):
|
|
result = get_system_prompt()
|
|
assert isinstance(result, str)
|
|
assert len(result) > 100
|
|
|
|
def test_contains_identity(self):
|
|
result = get_system_prompt()
|
|
assert "EvoScientist" in result
|
|
assert "self-evolving" in result
|
|
|
|
def test_contains_workflow(self):
|
|
result = get_system_prompt()
|
|
assert "Experiment Workflow" in result
|
|
|
|
def test_contains_report_template(self):
|
|
result = get_system_prompt()
|
|
assert "Experiment Report Template" in result
|
|
|
|
def test_contains_writing_guidelines(self):
|
|
result = get_system_prompt()
|
|
assert "Writing Guidelines" in result
|
|
|
|
def test_contains_shell_guidelines(self):
|
|
result = get_system_prompt()
|
|
assert "Shell Execution Guidelines" in result
|
|
|
|
def test_contains_delegation(self):
|
|
result = get_system_prompt()
|
|
assert "Sub-Agent Delegation" in result
|
|
|
|
def test_no_numeric_limits(self):
|
|
result = get_system_prompt()
|
|
assert "{max_concurrent}" not in result
|
|
assert "{max_iterations}" not in result
|
|
|
|
def test_workflow_constant_not_empty(self):
|
|
assert len(EXPERIMENT_WORKFLOW) > 0
|
|
|
|
def test_delegation_no_placeholders(self):
|
|
assert "{max_concurrent}" not in DELEGATION_STRATEGY
|
|
assert "{max_iterations}" not in DELEGATION_STRATEGY
|
|
|
|
def test_section_ordering(self):
|
|
"""Identity must precede workflow; workflow must precede delegation."""
|
|
result = get_system_prompt()
|
|
idx_identity = result.find("# Identity")
|
|
idx_workflow = result.find("# Experiment Workflow")
|
|
idx_delegation = result.find("# Sub-Agent Delegation")
|
|
assert 0 <= idx_identity < idx_workflow < idx_delegation
|
|
|
|
def test_does_not_contain_static_date(self):
|
|
"""Date is injected per-turn by runtime context, not baked into static prompt.
|
|
|
|
Static prompt must stay byte-stable across midnight so the cache prefix
|
|
survives. See RuntimeContextMiddleware for runtime injection.
|
|
"""
|
|
import re
|
|
|
|
result = get_system_prompt()
|
|
assert not re.search(r"Current date: \d{4}-\d{2}-\d{2}", result)
|
|
|
|
def test_mentions_skill_manager_for_discovery(self):
|
|
"""Agent must know it can browse/install skills from the EvoSkills catalog."""
|
|
result = get_system_prompt()
|
|
assert "skill_manager" in result
|
|
assert "EvoSkills" in result
|
|
|
|
def test_no_stale_memory_path_singular(self):
|
|
"""Backend route is `/memories/`, not `/memory/`. Catch silent-bug regressions.
|
|
|
|
Anything sent to `/memory/...` falls through to CustomSandboxBackend
|
|
(workspace files), bypassing the persistent FilesystemBackend that
|
|
owns persistent memory files.
|
|
"""
|
|
result = get_system_prompt()
|
|
# `/memory/` as a filesystem path (after a backtick or whitespace, before
|
|
# an alpha char or another /). Excludes word-list usages like
|
|
# "context/memory/web search".
|
|
import re
|
|
|
|
assert not re.search(r"[\s`]/memory/[a-zA-Z]", result), (
|
|
"Found `/memory/<file>` in system prompt — should be `/memories/<file>`"
|
|
)
|
|
|
|
def test_observation_writes_can_be_removed(self):
|
|
result = get_system_prompt(
|
|
enable_observation_memory=True,
|
|
enable_observation_writes=False,
|
|
)
|
|
|
|
assert "/memories/observations/" in result
|
|
assert "record_observation" not in result
|
|
assert "Memory Evolution" not in result
|
|
|
|
def test_observation_memory_can_be_removed(self):
|
|
result = get_system_prompt(
|
|
enable_observation_memory=False,
|
|
enable_observation_writes=False,
|
|
)
|
|
|
|
assert "/memories/observations/" not in result
|
|
assert "record_observation" not in result
|
|
assert "Memory Evolution" not in result
|
|
|
|
|
|
class TestEvoScientistIdentity:
|
|
def test_constant_not_empty(self):
|
|
assert len(EVOSCIENTIST_IDENTITY) > 0
|
|
|
|
def test_states_role(self):
|
|
assert "You are EvoScientist" in EVOSCIENTIST_IDENTITY
|
|
|
|
def test_mentions_human_on_the_loop_paradigm(self):
|
|
# Behavioral cue: agent should know it isn't asking permission for every action
|
|
assert "on-the-loop" in EVOSCIENTIST_IDENTITY
|
|
|
|
|
|
class TestReportTemplate:
|
|
def test_constant_not_empty(self):
|
|
assert len(REPORT_TEMPLATE) > 0
|
|
|
|
def test_contains_six_sections(self):
|
|
# Match the six recommended sections (lowercased to be tolerant of phrasing)
|
|
body = REPORT_TEMPLATE.lower()
|
|
for section in (
|
|
"summary",
|
|
"experiment plan",
|
|
"setup",
|
|
"baselines",
|
|
"results",
|
|
"limitations",
|
|
):
|
|
assert section in body, section
|
|
|
|
def test_not_duplicated_in_workflow_step5(self):
|
|
"""Step 5 should reference REPORT_TEMPLATE, not redefine the schema."""
|
|
# Positive: Step 5 must actually reference the report template.
|
|
assert "Experiment Report Template" in EXPERIMENT_WORKFLOW
|
|
# Negative: section headers unique to REPORT_TEMPLATE must not appear
|
|
# inlined inside EXPERIMENT_WORKFLOW (would mean the schema was
|
|
# duplicated again, regardless of indentation style).
|
|
assert "Baselines and comparisons" not in EXPERIMENT_WORKFLOW
|
|
|
|
|
|
class TestWritingGuidelines:
|
|
def test_constant_not_empty(self):
|
|
assert len(WRITING_GUIDELINES) > 0
|
|
|
|
def test_mentions_first_person_avoidance(self):
|
|
assert (
|
|
"first-person" in WRITING_GUIDELINES.lower()
|
|
or "I ..." in WRITING_GUIDELINES
|
|
)
|
|
|
|
|
|
class TestShellGuidelines:
|
|
def test_constant_not_empty(self):
|
|
assert len(SHELL_GUIDELINES) > 0
|
|
|
|
def test_mentions_timeout_limit(self):
|
|
assert "300" in SHELL_GUIDELINES # default timeout
|
|
assert "3600" in SHELL_GUIDELINES # per-command override ceiling
|
|
|
|
def test_mentions_background_execution(self):
|
|
assert "background" in SHELL_GUIDELINES.lower()
|
|
|
|
def test_not_duplicated_in_workflow(self):
|
|
"""SHELL_GUIDELINES content should live ONLY in its own constant."""
|
|
# Sentinel phrase unique to SHELL_GUIDELINES
|
|
assert "Sandbox limits" not in EXPERIMENT_WORKFLOW
|