Files
EvoScientist/tests/test_prompts.py
T
dinos 92d95dee68 feat(memory): add observation memory lifecycle (#259)
* feat(memory): add observation memory lifecycle

Add file-backed observation memory with deterministic markdown records,
structured record_observation tooling, startup indexing, and
profile/observation prompt guidance.

Launch post-turn and post-subagent EvoMemory workers through LangGraph
dev so completed runs can update profile memory, save durable
observations, and write subagent execution summaries without blocking
the active agent.

Wire memory middleware into the main agent, subagents, async graphs, TUI
status reporting, worker activity accounting, and observation-aware
research prompts, with regression coverage for storage, lifecycle
scheduling, graph registration, status display, and stream reset
behavior.

* fix(cli): sync background agent server on resume

Resume flows now need to keep the LangGraph dev background server
aligned with the active workspace even when async subagents are
disabled. EvoMemory workers use that server too, so gating resume-time
sync on enable_async_subagents could leave workers pinned to the launch
workspace after resuming a thread from another workspace.

Run workspace sync unconditionally for Rich CLI and Textual resume
paths, while preserving WorkspaceMismatchError handling so failed sync
aborts the resume before mutating the active thread or workspace.

Propagate aborted resume callbacks through the command UI so
channel-issued /resume commands do not send false success or history
output. Channel slash dispatch now treats CommandManager-caught command
errors as command errors and skips completion hooks for those failed
commands.

Add regression coverage for disabled async subagents, callback aborts,
and channel command error reporting.

* fix(cli): prepare serve resume workspace before adopting

Load the resumed workspace agent and sync the background server as a
single pre-adoption step. Restore the previous active workspace if
preparation fails so serve mode keeps using the old session
consistently.

* fix(memory): untrack abandoned worker status watches

Stop treating watcher shutdown as confirmed worker completion. Terminal
worker statuses still count memory deltas, while poll failures or
watcher setup failures now remove the active run without crediting
partial outputs.

* fix(cli): report channel command failures accurately

Treat command_error as a None sentinel so empty error strings still
fail, and let TUI resumes continue only on non-mismatch
background-server sync failures while reporting degraded mode.

* fix(stream): clear memory counters for resume streams

Reset completed-memory counters for every new agent stream, including
Command-based HITL and resume streams, so saved-memory indicators do not
leak across turns.

* docs(tools): make observation recording guidance conditional

Clarify that agents should call record_observation only when the
observation tool is available, preserving the existing durability and
usefulness criteria.

* feat(config): add controls for profile and observation memory

Add config flags for profile memory, observation memory, observation
writer placement, and background memory workers.

Wire the controls through main agents, subagents, EvoMemory middleware,
and memory lifecycle workers so observation writes can be assigned to
the live agent, subagent worker, both, or neither. Keep turn memory
workers profile-only and make prompts reflect the available observation
read/write paths. Skip langgraph dev startup when neither async
subagents nor memory workers need the background server.

Add coverage for config parsing, prompt gating, middleware wiring, and
worker tool availability.

* test(cli): include memory defaults in serve config stubs

* fix(memory): offload async worker launch blocking calls

Run the langgraph-dev health check and memory-output snapshot in worker
threads from the async EvoMemory launcher so it does not block the event
loop.

* chore(memory): harden turn worker subagent guardrail

* chore(memory): refresh profile context per request

* fix(memory): offload async profile file reads

* fix(memory): offload async worker completion accounting
2026-06-05 15:11:20 +01:00

185 lines
6.4 KiB
Python

"""Tests for EvoScientist/prompts.py."""
from EvoScientist.prompts import (
DELEGATION_STRATEGY,
EVOSCIENTIST_IDENTITY,
EXPERIMENT_WORKFLOW,
REPORT_TEMPLATE,
SHELL_GUIDELINES,
WRITING_GUIDELINES,
get_system_prompt,
)
class TestGetSystemPrompt:
def test_returns_non_empty(self):
result = get_system_prompt()
assert isinstance(result, str)
assert len(result) > 100
def test_contains_identity(self):
result = get_system_prompt()
assert "EvoScientist" in result
assert "self-evolving" in result
def test_contains_workflow(self):
result = get_system_prompt()
assert "Experiment Workflow" in result
def test_contains_report_template(self):
result = get_system_prompt()
assert "Experiment Report Template" in result
def test_contains_writing_guidelines(self):
result = get_system_prompt()
assert "Writing Guidelines" in result
def test_contains_shell_guidelines(self):
result = get_system_prompt()
assert "Shell Execution Guidelines" in result
def test_contains_delegation(self):
result = get_system_prompt()
assert "Sub-Agent Delegation" in result
def test_no_numeric_limits(self):
result = get_system_prompt()
assert "{max_concurrent}" not in result
assert "{max_iterations}" not in result
def test_workflow_constant_not_empty(self):
assert len(EXPERIMENT_WORKFLOW) > 0
def test_delegation_no_placeholders(self):
assert "{max_concurrent}" not in DELEGATION_STRATEGY
assert "{max_iterations}" not in DELEGATION_STRATEGY
def test_section_ordering(self):
"""Identity must precede workflow; workflow must precede delegation."""
result = get_system_prompt()
idx_identity = result.find("# Identity")
idx_workflow = result.find("# Experiment Workflow")
idx_delegation = result.find("# Sub-Agent Delegation")
assert 0 <= idx_identity < idx_workflow < idx_delegation
def test_does_not_contain_static_date(self):
"""Date is injected per-turn by runtime context, not baked into static prompt.
Static prompt must stay byte-stable across midnight so the cache prefix
survives. See RuntimeContextMiddleware for runtime injection.
"""
import re
result = get_system_prompt()
assert not re.search(r"Current date: \d{4}-\d{2}-\d{2}", result)
def test_mentions_skill_manager_for_discovery(self):
"""Agent must know it can browse/install skills from the EvoSkills catalog."""
result = get_system_prompt()
assert "skill_manager" in result
assert "EvoSkills" in result
def test_no_stale_memory_path_singular(self):
"""Backend route is `/memories/`, not `/memory/`. Catch silent-bug regressions.
Anything sent to `/memory/...` falls through to CustomSandboxBackend
(workspace files), bypassing the persistent FilesystemBackend that
owns persistent memory files.
"""
result = get_system_prompt()
# `/memory/` as a filesystem path (after a backtick or whitespace, before
# an alpha char or another /). Excludes word-list usages like
# "context/memory/web search".
import re
assert not re.search(r"[\s`]/memory/[a-zA-Z]", result), (
"Found `/memory/<file>` in system prompt — should be `/memories/<file>`"
)
def test_observation_writes_can_be_removed(self):
result = get_system_prompt(
enable_observation_memory=True,
enable_observation_writes=False,
)
assert "/memories/observations/" in result
assert "record_observation" not in result
assert "Memory Evolution" not in result
def test_observation_memory_can_be_removed(self):
result = get_system_prompt(
enable_observation_memory=False,
enable_observation_writes=False,
)
assert "/memories/observations/" not in result
assert "record_observation" not in result
assert "Memory Evolution" not in result
class TestEvoScientistIdentity:
def test_constant_not_empty(self):
assert len(EVOSCIENTIST_IDENTITY) > 0
def test_states_role(self):
assert "You are EvoScientist" in EVOSCIENTIST_IDENTITY
def test_mentions_human_on_the_loop_paradigm(self):
# Behavioral cue: agent should know it isn't asking permission for every action
assert "on-the-loop" in EVOSCIENTIST_IDENTITY
class TestReportTemplate:
def test_constant_not_empty(self):
assert len(REPORT_TEMPLATE) > 0
def test_contains_six_sections(self):
# Match the six recommended sections (lowercased to be tolerant of phrasing)
body = REPORT_TEMPLATE.lower()
for section in (
"summary",
"experiment plan",
"setup",
"baselines",
"results",
"limitations",
):
assert section in body, section
def test_not_duplicated_in_workflow_step5(self):
"""Step 5 should reference REPORT_TEMPLATE, not redefine the schema."""
# Positive: Step 5 must actually reference the report template.
assert "Experiment Report Template" in EXPERIMENT_WORKFLOW
# Negative: section headers unique to REPORT_TEMPLATE must not appear
# inlined inside EXPERIMENT_WORKFLOW (would mean the schema was
# duplicated again, regardless of indentation style).
assert "Baselines and comparisons" not in EXPERIMENT_WORKFLOW
class TestWritingGuidelines:
def test_constant_not_empty(self):
assert len(WRITING_GUIDELINES) > 0
def test_mentions_first_person_avoidance(self):
assert (
"first-person" in WRITING_GUIDELINES.lower()
or "I ..." in WRITING_GUIDELINES
)
class TestShellGuidelines:
def test_constant_not_empty(self):
assert len(SHELL_GUIDELINES) > 0
def test_mentions_timeout_limit(self):
assert "300" in SHELL_GUIDELINES # default timeout
assert "3600" in SHELL_GUIDELINES # per-command override ceiling
def test_mentions_background_execution(self):
assert "background" in SHELL_GUIDELINES.lower()
def test_not_duplicated_in_workflow(self):
"""SHELL_GUIDELINES content should live ONLY in its own constant."""
# Sentinel phrase unique to SHELL_GUIDELINES
assert "Sandbox limits" not in EXPERIMENT_WORKFLOW