Merge remote-tracking branch 'origin/main' into feature/channel-unification

# Conflicts:
#	EvoScientist/cli/commands.py
This commit is contained in:
MuXinCG
2026-02-15 18:10:24 +08:00
13 changed files with 55 additions and 1320 deletions
+1 -3
View File
@@ -28,7 +28,6 @@ from .prompts import RESEARCHER_INSTRUCTIONS, get_system_prompt
from .utils import load_subagents
from .tools import tavily_search, think_tool, skill_manager, view_image
from .paths import (
ensure_dirs,
default_workspace_dir,
set_active_workspace,
MEMORY_DIR as _MEMORY_DIR_PATH,
@@ -47,8 +46,7 @@ apply_config_to_env(_config)
MAX_CONCURRENT = _config.max_concurrent
MAX_ITERATIONS = _config.max_iterations
# Workspace settings
ensure_dirs()
# Workspace settings (defer dir creation to CLI; here we just resolve paths)
WORKSPACE_DIR = str(default_workspace_dir())
set_active_workspace(WORKSPACE_DIR)
MEMORY_DIR = str(_MEMORY_DIR_PATH) # Shared across sessions (not per-session)
+8 -9
View File
@@ -448,13 +448,12 @@ def _main_callback(
if not re.fullmatch(r"[A-Za-z0-9_-]+", name):
raise typer.BadParameter("--name may only contain letters, digits, hyphens, and underscores")
ensure_dirs()
# Resolve effective mode from config (CLI mode already applied via overrides)
effective_mode: str | None = None # None means explicit --workdir/--use-cwd was used
# Resolve workspace directory for this session
# Priority: --use-cwd > --workdir > --mode (explicit) > default_workdir > default_mode
# Priority: --workdir > --mode (explicit) > default_workdir > default_mode > cwd
# --use-cwd is kept for backward compat but is now the default behavior
if use_cwd:
workspace_dir = os.getcwd()
workspace_fixed = True
@@ -465,7 +464,7 @@ def _main_callback(
elif mode:
# Explicit --mode overrides default_workdir
effective_mode = mode
workspace_root = config.default_workdir or str(default_workspace_dir())
workspace_root = config.default_workdir or os.getcwd()
workspace_root = os.path.abspath(os.path.expanduser(workspace_root))
if effective_mode == "run":
runs_dir = Path(workspace_root, "runs")
@@ -475,7 +474,6 @@ def _main_callback(
workspace_fixed = False
else: # daemon
workspace_dir = workspace_root
os.makedirs(workspace_dir, exist_ok=True)
workspace_fixed = True
elif config.default_workdir:
# Use configured default workdir with configured mode
@@ -489,18 +487,19 @@ def _main_callback(
workspace_fixed = False
else: # daemon
workspace_dir = workspace_root
os.makedirs(workspace_dir, exist_ok=True)
workspace_fixed = True
else:
effective_mode = config.default_mode
if effective_mode == "run":
workspace_dir = _create_session_workspace(name)
workspace_fixed = False
else: # daemon mode (default)
workspace_dir = str(default_workspace_dir())
os.makedirs(workspace_dir, exist_ok=True)
else: # daemon mode (default) — use current directory
workspace_dir = os.getcwd()
workspace_fixed = True
# Ensure memory and skills subdirs exist in workspace
ensure_dirs()
if prompt:
# Single-shot mode: wrap in persistent checkpointer
import asyncio
+20 -7
View File
@@ -5,23 +5,36 @@ from .agent import _shorten_path
def _cmd_list_skills() -> None:
"""List installed user skills."""
"""List all available skills (user and system)."""
from ..tools.skills_manager import list_skills
from ..paths import USER_SKILLS_DIR
skills = list_skills(include_system=False)
skills = list_skills(include_system=True)
if not skills:
console.print("[dim]No user-installed skills.[/dim]")
console.print("[dim]No skills available.[/dim]")
console.print("[dim]Install with:[/dim] /install-skill <path-or-url>")
console.print(f"[dim]Skills directory:[/dim] [cyan]{_shorten_path(str(USER_SKILLS_DIR))}[/cyan]")
console.print()
return
console.print(f"[bold]User-Installed Skills[/bold] ({len(skills)}):")
for skill in skills:
console.print(f" [green]{skill.name}[/green] - {skill.description}")
console.print(f"\n[dim]Location:[/dim] [cyan]{_shorten_path(str(USER_SKILLS_DIR))}[/cyan]")
user_skills = [s for s in skills if s.source == "user"]
system_skills = [s for s in skills if s.source == "system"]
if user_skills:
console.print(f"[bold]User Skills[/bold] ({len(user_skills)}):")
for skill in user_skills:
console.print(f" [green]{skill.name}[/green] - {skill.description}")
if user_skills and system_skills:
console.print()
if system_skills:
console.print(f"[bold]Built-in Skills[/bold] ({len(system_skills)}):")
for skill in system_skills:
console.print(f" [cyan]{skill.name}[/cyan] - {skill.description}")
console.print(f"\n[dim]User skills folder:[/dim] [green]{_shorten_path(str(USER_SKILLS_DIR))}[/green]")
console.print()
+10 -19
View File
@@ -644,13 +644,15 @@ def _step_workspace(config: EvoScientistConfig) -> tuple[str, str]:
Tuple of (mode, workdir).
"""
# Mode selection
cwd = os.getcwd()
cwd_short = os.path.basename(cwd) or cwd
mode_choices = [
Choice(
title="Daemon (persistent workspace ./workspace/)",
title="Daemon (persistent workspace)",
value="daemon",
),
Choice(
title="Run (isolated per-session ./workspace/runs/<timestamp>/)",
title="Run (isolated per-session)",
value="run",
),
]
@@ -668,27 +670,16 @@ def _step_workspace(config: EvoScientistConfig) -> tuple[str, str]:
raise KeyboardInterrupt()
# Custom workdir (optional)
use_custom = questionary.confirm(
"Use custom workspace directory? (default: ./workspace/)",
default=bool(config.default_workdir),
current_default = config.default_workdir or ""
workdir = questionary.text(
f"Workspace directory (Enter to use ./{cwd_short}/):",
default=current_default,
style=WIZARD_STYLE,
qmark=QMARK,
).ask()
if use_custom is None:
if workdir is None:
raise KeyboardInterrupt()
workdir = ""
if use_custom:
workdir = questionary.text(
"Workspace directory path:",
default=config.default_workdir or "",
style=WIZARD_STYLE,
qmark=QMARK,
).ask()
if workdir is None:
raise KeyboardInterrupt()
workdir = workdir.strip()
workdir = workdir.strip()
return mode, workdir
+8 -4
View File
@@ -18,8 +18,8 @@ def _env_path(key: str) -> Path | None:
return _expand(value)
# Workspace root: directly under cwd (no hidden .evoscientist layer)
WORKSPACE_ROOT = _env_path("EVOSCIENTIST_WORKSPACE_DIR") or (Path.cwd() / "workspace")
# Workspace root: current working directory by default (user's project dir)
WORKSPACE_ROOT = _env_path("EVOSCIENTIST_WORKSPACE_DIR") or Path.cwd()
RUNS_DIR = _env_path("EVOSCIENTIST_RUNS_DIR") or (WORKSPACE_ROOT / "runs")
MEMORY_DIR = _env_path("EVOSCIENTIST_MEMORY_DIR") or (WORKSPACE_ROOT / "memory")
@@ -27,8 +27,12 @@ USER_SKILLS_DIR = _env_path("EVOSCIENTIST_SKILLS_DIR") or (WORKSPACE_ROOT / "ski
def ensure_dirs() -> None:
"""Create runtime directories if they do not exist."""
for path in (WORKSPACE_ROOT, RUNS_DIR, MEMORY_DIR, USER_SKILLS_DIR):
"""Create runtime subdirectories (memory, skills) if they do not exist.
Does NOT create the workspace root itself — it should already exist
(either the user's cwd or a directory they specified).
"""
for path in (MEMORY_DIR, USER_SKILLS_DIR):
path.mkdir(parents=True, exist_ok=True)
-21
View File
@@ -1,21 +0,0 @@
MIT License
Copyright (c) 2026 EvoScientist
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in all
copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
SOFTWARE.
-318
View File
@@ -1,318 +0,0 @@
---
name: sci-swarm
description: Spawn parallel EvoSci experiments, control interactive REPLs, and orchestrate multi-agent workflows via tmux sessions. Use when the user wants to run multiple experiments simultaneously, needs an interactive Python/shell REPL, or wants to coordinate independent research tasks.
license: Complete terms in LICENSE.txt
metadata:
version: "0.1.0"
author: Xi Zhang
tags: [parallel, experiments, tmux, orchestration, multi-agent]
allowed-tools: "execute read_file write_file ls"
---
# sci-swarm — Multi-Agent Experiment Orchestration
Spawn independent EvoSci CLI instances in isolated tmux sessions for parallel experiments, interactive REPL control, and collaborative research workflows.
## Sandbox Path Warning
The EvoScientist sandbox converts literal `/path` to `./path` in shell commands. **Always use shell variables for absolute paths**, never hardcoded literals like `/tmp/...`.
```bash
# CORRECT - uses shell variable
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
SOCKET="$SOCKET_DIR/evosci.sock"
tmux -S "$SOCKET" new -d -s my-session
```
WRONG — literal absolute path will be converted to `./tmp/...` by the sandbox:
```
tmux -S /tmp/evosci-tmux-sockets/evosci.sock new -d -s my-session
```
## Quick Start
Create a session, send a command, and capture output:
```bash
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
mkdir -p "$SOCKET_DIR"
SOCKET="$SOCKET_DIR/evosci.sock"
SESSION=evosci-demo
tmux -S "$SOCKET" new -d -s "$SESSION" -n shell
tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 -- 'echo hello from tmux' Enter
sleep 1
tmux -S "$SOCKET" capture-pane -p -J -t "$SESSION":0.0 -S -200
```
After starting a session, always print monitor commands:
```
To monitor:
tmux -S "$SOCKET" attach -t "$SESSION"
tmux -S "$SOCKET" capture-pane -p -J -t "$SESSION":0.0 -S -200
```
## First-Time Script Setup
The skill includes helper scripts under `/skills/sci-swarm/scripts/`. Because the skills directory is read-only and shell CWD differs from the virtual path, copy them to the workspace first:
```bash
mkdir -p tmux-scripts
```
Then use `read_file` and `write_file` to copy each script:
1. `read_file("/skills/sci-swarm/scripts/spawn-evosci.sh")` then `write_file("/tmux-scripts/spawn-evosci.sh", content)`
2. `read_file("/skills/sci-swarm/scripts/find-sessions.sh")` then `write_file("/tmux-scripts/find-sessions.sh", content)`
3. `read_file("/skills/sci-swarm/scripts/wait-for-text.sh")` then `write_file("/tmux-scripts/wait-for-text.sh", content)`
4. `read_file("/skills/sci-swarm/scripts/collect-results.sh")` then `write_file("/tmux-scripts/collect-results.sh", content)`
Run scripts with `bash` (no `chmod` needed):
```bash
bash tmux-scripts/spawn-evosci.sh experiment-1 "Run baseline comparison"
```
## Workspace Mode: Always Use --workdir
**Critical**: Always use `--workdir` to spawn sub-agents. Never use `-m run` (it nests `workspace/runs/` inside the main workspace, breaking read_file paths).
| Mode | Effect | Verdict |
|------|--------|---------|
| `-m run -n "xxx"` | Nests `workspace/runs/xxx/` with copied memory/skills | Avoid — path nesting breaks |
| `--use-cwd` | Agent works in main agent's root directory | Pollutes root directory |
| `--workdir <path>` | Agent works in specified subdirectory | Best practice — isolated yet accessible |
## Path Perspective Asymmetry
Each spawned EvoSci instance sees its `--workdir` as `/` (virtual root). When the main agent needs to read a sub-agent's output, it must use the **main agent's path perspective**:
- Sub-agent writes: `write_file("/results.md", ...)` → file lands in its workdir
- Main agent reads: `read_file("/agents/experiment-1/results.md")` → using main agent's path
This applies to all file operations (read_file, ls, grep, glob).
## Spawning EvoSci Instances
### Using the spawn script (recommended)
```bash
bash tmux-scripts/spawn-evosci.sh experiment-1 "Investigate the effect of learning rate on convergence"
bash tmux-scripts/spawn-evosci.sh experiment-2 "Compare Adam vs SGD optimizers" --model claude-sonnet-4-5
```
Add `--no-thinking` to save tokens in automated workflows:
```bash
bash tmux-scripts/spawn-evosci.sh experiment-1 "Run baseline comparison" --no-thinking
```
Each instance gets its own workspace at `workspace/runs/<name>` and shares `/memory/` with all other instances.
### Stateful multi-round sessions with --thread-id
Use `--thread-id` to maintain conversation context across multiple invocations to the same session. The sub-agent remembers previous instructions and can build on prior work:
```bash
bash tmux-scripts/spawn-evosci.sh builder "Step 1: Set up the data pipeline" --thread-id pipeline-001
```
After the first call completes, send follow-up prompts to the same session with the same thread ID:
```bash
bash tmux-scripts/spawn-evosci.sh builder "Step 2: Train the model using the pipeline from step 1" --thread-id pipeline-001
```
| Without --thread-id | With --thread-id |
|---------------------|------------------|
| Each call is a fresh instance, no memory | Preserves full conversation history |
| Best for one-shot tasks | Best for multi-step iterative experiments |
### Inline (without helper script)
```bash
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
mkdir -p "$SOCKET_DIR"
SOCKET="$SOCKET_DIR/evosci.sock"
NAME=experiment-1
WORKDIR="$(pwd)/workspace/runs/$NAME"
mkdir -p "$WORKDIR"
tmux -S "$SOCKET" new-session -d -s "evosci-$NAME" -n shell
tmux -S "$SOCKET" send-keys -t "evosci-$NAME":0.0 -- "EvoSci -p 'Your research prompt here' --workdir $WORKDIR --no-thinking" Enter
```
### Spawning multiple experiments
```bash
for exp in baseline-lr0.001 baseline-lr0.01 baseline-lr0.1; do
bash tmux-scripts/spawn-evosci.sh "$exp" "Train model with learning rate ${exp##*-}" --no-thinking
done
```
## Monitoring Progress
### Capture pane output
```bash
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
SOCKET="$SOCKET_DIR/evosci.sock"
tmux -S "$SOCKET" capture-pane -p -J -t "evosci-experiment-1":0.0 -S -200
```
### Detect completion
EvoSci prints "Goodbye!" when it exits. Check for it:
```bash
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
SOCKET="$SOCKET_DIR/evosci.sock"
if tmux -S "$SOCKET" capture-pane -p -t "evosci-experiment-1":0.0 -S -5 | grep -q "Goodbye!"; then
echo "experiment-1: DONE"
else
echo "experiment-1: still running"
fi
```
### Wait for text (blocking)
```bash
bash tmux-scripts/wait-for-text.sh -S "$SOCKET" -t "evosci-experiment-1":0.0 -p 'Goodbye!' -T 300
```
Options:
- `-t` target pane (required)
- `-p` regex pattern to match (required)
- `-S` tmux socket path
- `-F` treat pattern as fixed string
- `-T` timeout in seconds (default: 15)
- `-i` poll interval (default: 0.5)
- `-l` history lines to search (default: 1000)
## Collecting Results
### Using the collect script
```bash
bash tmux-scripts/collect-results.sh experiment-1
bash tmux-scripts/collect-results.sh experiment-1 --pane-output
```
This shows session status, lists output files, and optionally captures the pane.
### Manual collection
Read output files from completed workspaces:
```bash
read_file("/runs/experiment-1/results.md")
read_file("/runs/experiment-1/analysis.py")
```
List workspace contents:
```bash
ls("/runs/experiment-1/")
```
## Interactive REPL Control
### Python REPL
Use `PYTHON_BASIC_REPL=1` to prevent the enhanced REPL from breaking send-keys:
```bash
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
SOCKET="$SOCKET_DIR/evosci.sock"
SESSION=evosci-python
tmux -S "$SOCKET" new -d -s "$SESSION" -n shell
tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 -- 'PYTHON_BASIC_REPL=1 python3 -q' Enter
sleep 1
tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 -l -- 'print("hello")' && sleep 0.1 && tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 Enter
sleep 1
tmux -S "$SOCKET" capture-pane -p -J -t "$SESSION":0.0 -S -50
```
### Sending input safely
- Prefer literal sends: `tmux -S "$SOCKET" send-keys -t target -l -- "$cmd"`
- Control keys: `tmux -S "$SOCKET" send-keys -t target C-c`
- For TUI apps (interactive CLIs), separate text and Enter with a delay:
```bash
tmux -S "$SOCKET" send-keys -t target -l -- "$cmd" && sleep 0.1 && tmux -S "$SOCKET" send-keys -t target Enter
```
## Agent Communication
For collaborative research across spawned instances, use file-based message passing via shared memory:
```bash
# Agent A writes a message
write_file("/memory/messages/from-experiment-1.md", "## Finding\nLearning rate 0.01 converges fastest.")
# Agent B reads it
read_file("/memory/messages/from-experiment-1.md")
```
All EvoSci instances share `/memory/` regardless of their workspace directory.
## Finding & Cleaning Sessions
### List sessions
```bash
bash tmux-scripts/find-sessions.sh -S "$SOCKET"
bash tmux-scripts/find-sessions.sh --all
```
Options:
- `-L` socket name (tmux -L)
- `-S` socket path (tmux -S)
- `-A`/`--all` scan all sockets under `EVOSCI_TMUX_SOCKET_DIR`
- `-q` filter session names
### Kill a session
```bash
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
SOCKET="$SOCKET_DIR/evosci.sock"
tmux -S "$SOCKET" kill-session -t "evosci-experiment-1"
```
### Kill all sessions
```bash
SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
SOCKET="$SOCKET_DIR/evosci.sock"
tmux -S "$SOCKET" kill-server
```
## Swarm Conventions
### Session registry
The spawn script maintains a registry at `$EVOSCI_TMUX_SOCKET_DIR/registry.json` tracking all spawned sessions, their workspaces, prompts, and status.
### Limits
| Variable | Default | Description |
|----------|---------|-------------|
| `EVOSCI_MAX_TMUX_SESSIONS` | 5 | Maximum concurrent sessions |
| `EVOSCI_TMUX_DEPTH` | 0 | Current recursion depth (auto-incremented) |
| `EVOSCI_MAX_TMUX_DEPTH` | 3 | Maximum recursion depth |
| `EVOSCI_TMUX_SOCKET_DIR` | `$TMPDIR/evosci-tmux-sockets` | Socket directory |
### Naming
- Session names: `evosci-<name>` (e.g., `evosci-experiment-1`)
- Workspaces: `workspace/runs/<name>/`
- Keep names short, alphanumeric with hyphens
### Targeting panes
- Target format: `session:window.pane` (defaults to `:0.0`)
- Inspect: `tmux -S "$SOCKET" list-sessions`, `tmux -S "$SOCKET" list-panes -a`
@@ -1,150 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
usage() {
cat <<'USAGE'
Usage: collect-results.sh <name> [options]
Collect results from a spawned EvoSci session.
Arguments:
name Session name suffix (matches evosci-<name>)
Options:
--socket-path <path> tmux socket path (default: $EVOSCI_TMUX_SOCKET_DIR/evosci.sock)
--workdir <path> Workspace directory (default: workspace/runs/<name>)
--pane-output Also capture the final pane output
-h, --help Show this help
USAGE
}
# --- Defaults ---
name=""
socket_path=""
workdir=""
pane_output=false
socket_dir="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
# --- Parse args ---
while [[ $# -gt 0 ]]; do
case "$1" in
--socket-path) socket_path="${2-}"; shift 2 ;;
--workdir) workdir="${2-}"; shift 2 ;;
--pane-output) pane_output=true; shift ;;
-h|--help) usage; exit 0 ;;
-*) echo "Unknown option: $1" >&2; usage; exit 1 ;;
*)
if [[ -z "$name" ]]; then
name="$1"
else
echo "Unexpected argument: $1" >&2; usage; exit 1
fi
shift
;;
esac
done
if [[ -z "$name" ]]; then
echo "Error: name is required" >&2
usage
exit 1
fi
# --- Resolve paths ---
if [[ -z "$socket_path" ]]; then
socket_path="$socket_dir/evosci.sock"
fi
if [[ -z "$workdir" ]]; then
# Try registry first
registry_file="$socket_dir/registry.json"
if [[ -f "$registry_file" ]]; then
found_workdir=$(EVOSCI_REG_FILE="$registry_file" EVOSCI_REG_NAME="$name" python3 -c "
import json, os
with open(os.environ['EVOSCI_REG_FILE']) as f:
registry = json.load(f)
for entry in registry:
if entry.get('name') == os.environ['EVOSCI_REG_NAME']:
print(entry.get('workdir', ''))
break
" 2>/dev/null || true)
if [[ -n "$found_workdir" ]]; then
workdir="$found_workdir"
fi
fi
# Fallback to convention
if [[ -z "$workdir" ]]; then
workdir="$(pwd)/workspace/runs/${name}"
fi
fi
# --- Check session status ---
session_name="evosci-${name}"
session_running=false
if tmux -S "$socket_path" has-session -t "$session_name" 2>/dev/null; then
session_running=true
fi
if [[ "$session_running" == true ]]; then
echo "Status: RUNNING"
echo "Session '$session_name' is still active."
else
echo "Status: COMPLETED"
echo "Session '$session_name' has ended."
# Update registry status
registry_file="$socket_dir/registry.json"
if [[ -f "$registry_file" ]]; then
EVOSCI_REG_FILE="$registry_file" EVOSCI_REG_NAME="$name" python3 -c "
import json, os, datetime
reg_file = os.environ['EVOSCI_REG_FILE']
reg_name = os.environ['EVOSCI_REG_NAME']
with open(reg_file) as f:
registry = json.load(f)
for entry in registry:
if entry.get('name') == reg_name and entry.get('status') == 'running':
entry['status'] = 'completed'
entry['completed_at'] = datetime.datetime.now().isoformat()
with open(reg_file, 'w') as f:
json.dump(registry, f, indent=2)
" 2>/dev/null || true
fi
fi
echo ""
# --- Check workspace ---
if [[ ! -d "$workdir" ]]; then
echo "Workspace not found: $workdir"
exit 1
fi
echo "Workspace: $workdir"
echo ""
# --- List output files ---
echo "Output files:"
file_count=0
for ext in md py csv png json ipynb txt pdf; do
while IFS= read -r -d '' file; do
rel_path="${file#"$workdir"/}"
size=$(wc -c < "$file" 2>/dev/null | tr -d ' ')
printf ' %-50s %s bytes\n' "$rel_path" "$size"
file_count=$((file_count + 1))
done < <(find "$workdir" -maxdepth 3 -name "*.${ext}" -print0 2>/dev/null)
done
if [[ "$file_count" -eq 0 ]]; then
echo " (no output files found)"
fi
# --- Optionally capture pane output ---
if [[ "$pane_output" == true && "$session_running" == true ]]; then
echo ""
echo "--- Pane Output (last 500 lines) ---"
tmux -S "$socket_path" capture-pane -p -J -t "$session_name":0.0 -S -500 2>/dev/null || echo "(could not capture pane)"
fi
echo ""
echo "To read a specific file:"
echo " read_file \"/runs/${name}/<filename>\""
@@ -1,112 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
usage() {
cat <<'USAGE'
Usage: find-sessions.sh [-L socket-name|-S socket-path|-A] [-q pattern]
List tmux sessions on a socket (default tmux socket if none provided).
Options:
-L, --socket tmux socket name (passed to tmux -L)
-S, --socket-path tmux socket path (passed to tmux -S)
-A, --all scan all sockets under EVOSCI_TMUX_SOCKET_DIR
-q, --query case-insensitive substring to filter session names
-h, --help show this help
USAGE
}
socket_name=""
socket_path=""
query=""
scan_all=false
socket_dir="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
while [[ $# -gt 0 ]]; do
case "$1" in
-L|--socket) socket_name="${2-}"; shift 2 ;;
-S|--socket-path) socket_path="${2-}"; shift 2 ;;
-A|--all) scan_all=true; shift ;;
-q|--query) query="${2-}"; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) echo "Unknown option: $1" >&2; usage; exit 1 ;;
esac
done
if [[ "$scan_all" == true && ( -n "$socket_name" || -n "$socket_path" ) ]]; then
echo "Cannot combine --all with -L or -S" >&2
exit 1
fi
if [[ -n "$socket_name" && -n "$socket_path" ]]; then
echo "Use either -L or -S, not both" >&2
exit 1
fi
if ! command -v tmux >/dev/null 2>&1; then
echo "tmux not found in PATH" >&2
exit 1
fi
list_sessions() {
local label="$1"; shift
local tmux_cmd=(tmux "$@")
if ! sessions="$("${tmux_cmd[@]}" list-sessions -F '#{session_name}\t#{session_attached}\t#{session_created_string}' 2>/dev/null)"; then
echo "No tmux server found on $label" >&2
return 1
fi
if [[ -n "$query" ]]; then
sessions="$(printf '%s\n' "$sessions" | grep -i -- "$query" || true)"
fi
if [[ -z "$sessions" ]]; then
echo "No sessions found on $label"
return 0
fi
echo "Sessions on $label:"
printf '%s\n' "$sessions" | while IFS=$'\t' read -r name attached created; do
attached_label=$([[ "$attached" == "1" ]] && echo "attached" || echo "detached")
printf ' - %s (%s, started %s)\n' "$name" "$attached_label" "$created"
done
}
if [[ "$scan_all" == true ]]; then
if [[ ! -d "$socket_dir" ]]; then
echo "Socket directory not found: $socket_dir" >&2
exit 1
fi
shopt -s nullglob
sockets=("$socket_dir"/*)
shopt -u nullglob
if [[ "${#sockets[@]}" -eq 0 ]]; then
echo "No sockets found under $socket_dir" >&2
exit 1
fi
exit_code=0
for sock in "${sockets[@]}"; do
if [[ ! -S "$sock" ]]; then
continue
fi
list_sessions "socket path '$sock'" -S "$sock" || exit_code=$?
done
exit "$exit_code"
fi
tmux_cmd=(tmux)
socket_label="default socket"
if [[ -n "$socket_name" ]]; then
tmux_cmd+=(-L "$socket_name")
socket_label="socket name '$socket_name'"
elif [[ -n "$socket_path" ]]; then
tmux_cmd+=(-S "$socket_path")
socket_label="socket path '$socket_path'"
fi
list_sessions "$socket_label" "${tmux_cmd[@]:1}"
@@ -1,209 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
usage() {
cat <<'USAGE'
Usage: spawn-evosci.sh <name> "<prompt>" [options]
Spawn an EvoSci CLI instance in an isolated tmux session.
Arguments:
name Session suffix (session will be named evosci-<name>)
prompt Research prompt to pass to EvoSci -p
Options:
--socket-path <path> tmux socket path (default: $EVOSCI_TMUX_SOCKET_DIR/evosci.sock)
--workdir <path> Workspace directory (default: workspace/runs/<name>)
--model <model> Model to use (e.g., claude-sonnet-4-5)
--thread-id <id> Thread ID for stateful multi-round conversations
--no-thinking Disable thinking output (saves tokens)
--max-sessions <n> Max allowed sessions (default: $EVOSCI_MAX_TMUX_SESSIONS or 5)
--max-depth <n> Max recursion depth (default: $EVOSCI_MAX_TMUX_DEPTH or 3)
-h, --help Show this help
USAGE
}
# --- Defaults ---
name=""
prompt=""
socket_path=""
workdir=""
model=""
thread_id=""
no_thinking=false
max_sessions="${EVOSCI_MAX_TMUX_SESSIONS:-5}"
max_depth="${EVOSCI_MAX_TMUX_DEPTH:-3}"
current_depth="${EVOSCI_TMUX_DEPTH:-0}"
socket_dir="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}"
# --- Parse args ---
while [[ $# -gt 0 ]]; do
case "$1" in
--socket-path) socket_path="${2-}"; shift 2 ;;
--workdir) workdir="${2-}"; shift 2 ;;
--model) model="${2-}"; shift 2 ;;
--thread-id) thread_id="${2-}"; shift 2 ;;
--no-thinking) no_thinking=true; shift ;;
--max-sessions) max_sessions="${2-}"; shift 2 ;;
--max-depth) max_depth="${2-}"; shift 2 ;;
-h|--help) usage; exit 0 ;;
-*) echo "Unknown option: $1" >&2; usage; exit 1 ;;
*)
if [[ -z "$name" ]]; then
name="$1"
elif [[ -z "$prompt" ]]; then
prompt="$1"
else
echo "Unexpected argument: $1" >&2; usage; exit 1
fi
shift
;;
esac
done
# --- Validate required args ---
if [[ -z "$name" || -z "$prompt" ]]; then
echo "Error: name and prompt are required" >&2
usage
exit 1
fi
# Validate name (alphanumeric, hyphens, underscores only)
if ! [[ "$name" =~ ^[a-zA-Z0-9][a-zA-Z0-9_-]*$ ]]; then
echo "Error: name must be alphanumeric with hyphens/underscores (got: $name)" >&2
exit 1
fi
# --- Check prerequisites ---
if ! command -v tmux >/dev/null 2>&1; then
echo "Error: tmux not found in PATH" >&2
exit 1
fi
if ! command -v EvoSci >/dev/null 2>&1; then
echo "Error: EvoSci not found in PATH (install with: pip install -e '.[dev]')" >&2
exit 1
fi
# --- Check recursion depth ---
if (( current_depth >= max_depth )); then
echo "Error: recursion depth limit reached (current=$current_depth, max=$max_depth)" >&2
echo "Set EVOSCI_MAX_TMUX_DEPTH to increase the limit." >&2
exit 1
fi
# --- Resolve socket ---
mkdir -p "$socket_dir"
if [[ -z "$socket_path" ]]; then
socket_path="$socket_dir/evosci.sock"
fi
# --- Check session limit ---
session_count=0
if tmux -S "$socket_path" list-sessions 2>/dev/null | grep -c '^' > /dev/null 2>&1; then
session_count=$(tmux -S "$socket_path" list-sessions 2>/dev/null | grep -c '^' || echo 0)
fi
if (( session_count >= max_sessions )); then
echo "Error: session limit reached ($session_count/$max_sessions)" >&2
echo "Kill completed sessions or increase --max-sessions." >&2
exit 1
fi
# --- Check duplicate session ---
session_name="evosci-${name}"
if tmux -S "$socket_path" has-session -t "$session_name" 2>/dev/null; then
if [[ -n "$thread_id" ]]; then
# With --thread-id, reuse is intentional: kill the stale session
# (EvoSci has exited but the tmux shell remains)
tmux -S "$socket_path" kill-session -t "$session_name" 2>/dev/null || true
else
echo "Error: session '$session_name' already exists" >&2
echo "Use a different name or kill the existing session:" >&2
echo " tmux -S \"$socket_path\" kill-session -t \"$session_name\"" >&2
exit 1
fi
fi
# --- Resolve workspace ---
if [[ -z "$workdir" ]]; then
workdir="$(pwd)/workspace/runs/${name}"
fi
# Make workdir absolute if relative
if [[ "$workdir" != /* ]]; then
workdir="$(pwd)/$workdir"
fi
mkdir -p "$workdir"
# --- Build EvoSci command ---
evosci_cmd="EvoSci -p $(printf '%q' "$prompt") --workdir $(printf '%q' "$workdir")"
if [[ -n "$model" ]]; then
evosci_cmd+=" --model $(printf '%q' "$model")"
fi
if [[ -n "$thread_id" ]]; then
evosci_cmd+=" --thread-id $(printf '%q' "$thread_id")"
fi
if [[ "$no_thinking" == true ]]; then
evosci_cmd+=" --no-thinking"
fi
# --- Create tmux session ---
next_depth=$((current_depth + 1))
if ! tmux -S "$socket_path" new-session -d -s "$session_name" -n shell \
-e "EVOSCI_TMUX_DEPTH=$next_depth" \
-e "EVOSCI_TMUX_SOCKET_DIR=$socket_dir" 2>/dev/null; then
echo "Error: failed to create tmux session '$session_name'" >&2
exit 2
fi
# --- Launch EvoSci ---
tmux -S "$socket_path" send-keys -t "$session_name":0.0 -- "$evosci_cmd" Enter
# --- Update registry ---
registry_file="$socket_dir/registry.json"
EVOSCI_REG_FILE="$registry_file" \
EVOSCI_REG_NAME="$name" \
EVOSCI_REG_SESSION="$session_name" \
EVOSCI_REG_SOCKET="$socket_path" \
EVOSCI_REG_WORKDIR="$workdir" \
EVOSCI_REG_PROMPT="$prompt" \
EVOSCI_REG_DEPTH="$next_depth" \
EVOSCI_REG_THREAD="$thread_id" \
python3 -c "
import json, os, datetime
registry_path = os.environ['EVOSCI_REG_FILE']
try:
with open(registry_path) as f:
registry = json.load(f)
except (FileNotFoundError, json.JSONDecodeError):
registry = []
entry = {
'name': os.environ['EVOSCI_REG_NAME'],
'session': os.environ['EVOSCI_REG_SESSION'],
'socket': os.environ['EVOSCI_REG_SOCKET'],
'workdir': os.environ['EVOSCI_REG_WORKDIR'],
'prompt': os.environ['EVOSCI_REG_PROMPT'],
'status': 'running',
'started_at': datetime.datetime.now().isoformat(),
'depth': int(os.environ['EVOSCI_REG_DEPTH'])
}
thread_id = os.environ.get('EVOSCI_REG_THREAD', '')
if thread_id:
entry['thread_id'] = thread_id
registry.append(entry)
with open(registry_path, 'w') as f:
json.dump(registry, f, indent=2)
" 2>/dev/null || true
# --- Print confirmation ---
echo "Spawned EvoSci session: $session_name"
echo " Workspace: $workdir"
if [[ -n "$thread_id" ]]; then
echo " Thread ID: $thread_id (stateful)"
fi
echo " Depth: $next_depth/$max_depth"
echo ""
echo "Monitor commands:"
echo " tmux -S \"$socket_path\" attach -t \"$session_name\""
echo " tmux -S \"$socket_path\" capture-pane -p -J -t \"$session_name\":0.0 -S -200"
@@ -1,92 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
usage() {
cat <<'USAGE'
Usage: wait-for-text.sh -t target -p pattern [options]
Poll a tmux pane for text and exit when found.
Options:
-t, --target tmux target (session:window.pane), required
-p, --pattern regex pattern to look for, required
-F, --fixed treat pattern as a fixed string (grep -F)
-S, --socket tmux socket path (passed to tmux -S)
-T, --timeout seconds to wait (integer, default: 15)
-i, --interval poll interval in seconds (default: 0.5)
-l, --lines number of history lines to inspect (integer, default: 1000)
-h, --help show this help
USAGE
}
target=""
pattern=""
grep_flag="-E"
socket_path=""
timeout=15
interval=0.5
lines=1000
while [[ $# -gt 0 ]]; do
case "$1" in
-t|--target) target="${2-}"; shift 2 ;;
-p|--pattern) pattern="${2-}"; shift 2 ;;
-F|--fixed) grep_flag="-F"; shift ;;
-S|--socket) socket_path="${2-}"; shift 2 ;;
-T|--timeout) timeout="${2-}"; shift 2 ;;
-i|--interval) interval="${2-}"; shift 2 ;;
-l|--lines) lines="${2-}"; shift 2 ;;
-h|--help) usage; exit 0 ;;
*) echo "Unknown option: $1" >&2; usage; exit 1 ;;
esac
done
if [[ -z "$target" || -z "$pattern" ]]; then
echo "target and pattern are required" >&2
usage
exit 1
fi
if ! [[ "$timeout" =~ ^[0-9]+$ ]]; then
echo "timeout must be an integer number of seconds" >&2
exit 1
fi
if ! [[ "$lines" =~ ^[0-9]+$ ]]; then
echo "lines must be an integer" >&2
exit 1
fi
if ! command -v tmux >/dev/null 2>&1; then
echo "tmux not found in PATH" >&2
exit 1
fi
# Build base tmux command with optional socket
tmux_base=(tmux)
if [[ -n "$socket_path" ]]; then
tmux_base+=(-S "$socket_path")
fi
# End time in epoch seconds (integer, good enough for polling)
start_epoch=$(date +%s)
deadline=$((start_epoch + timeout))
while true; do
# -J joins wrapped lines, -S uses negative index to read last N lines
pane_text="$("${tmux_base[@]}" capture-pane -p -J -t "$target" -S "-${lines}" 2>/dev/null || true)"
if printf '%s\n' "$pane_text" | grep $grep_flag -- "$pattern" >/dev/null 2>&1; then
exit 0
fi
now=$(date +%s)
if (( now >= deadline )); then
echo "Timed out after ${timeout}s waiting for pattern: $pattern" >&2
echo "Last ${lines} lines from $target:" >&2
printf '%s\n' "$pane_text" >&2
exit 1
fi
sleep "$interval"
done
+8 -6
View File
@@ -257,14 +257,14 @@ class TestStepModel:
class TestStepWorkspace:
def test_returns_mode_and_empty_workdir(self):
"""Test workspace step with no custom directory."""
"""Test workspace step with no custom directory (user keeps cwd default)."""
from EvoScientist.config.onboard import _step_workspace
config = EvoScientistConfig()
with patch("EvoScientist.config.onboard.questionary") as mock_q:
mock_q.select.return_value.ask.return_value = "daemon"
mock_q.confirm.return_value.ask.return_value = False # No custom dir
mock_q.text.return_value.ask.return_value = "" # Keep default (empty = use cwd)
result = _step_workspace(config)
assert result == ("daemon", "")
@@ -277,7 +277,6 @@ class TestStepWorkspace:
with patch("EvoScientist.config.onboard.questionary") as mock_q:
mock_q.select.return_value.ask.return_value = "run"
mock_q.confirm.return_value.ask.return_value = True # Use custom dir
mock_q.text.return_value.ask.return_value = "/custom/path"
result = _step_workspace(config)
@@ -686,10 +685,10 @@ class TestRunOnboard:
"", # Tavily key (keep current)
]
mock_q.confirm.return_value.ask.side_effect = [
False, # Custom workdir
True, # Save config
]
mock_q.text.return_value.ask.side_effect = [
"", # Workspace directory (empty = use cwd)
"3", # Max concurrent
"3", # Max iterations
]
@@ -735,10 +734,13 @@ class TestRunOnboard:
]
mock_q.password.return_value.ask.side_effect = ["", ""]
mock_q.confirm.return_value.ask.side_effect = [
False, # Custom workdir
False, # Save config - NO
]
mock_q.text.return_value.ask.side_effect = ["3", "3"]
mock_q.text.return_value.ask.side_effect = [
"", # Workspace directory (empty = use cwd)
"3", # Max concurrent
"3", # Max iterations
]
mock_q.checkbox.return_value.ask.return_value = [] # Skills: skip
result = run_onboard(skip_validation=True)
-370
View File
@@ -1,370 +0,0 @@
"""Tests for the sci-swarm skill.
Validates SKILL.md structure, script syntax, naming conventions,
and sandbox compatibility. No tmux or API keys required.
"""
import re
import subprocess
from pathlib import Path
import pytest
import yaml
# Skill directory
SKILL_DIR = Path(__file__).parent.parent / "EvoScientist" / "skills" / "sci-swarm"
SCRIPTS_DIR = SKILL_DIR / "scripts"
SKILL_MD = SKILL_DIR / "SKILL.md"
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def parse_frontmatter(text: str) -> dict:
"""Extract YAML frontmatter from a markdown file."""
match = re.match(r"^---\n(.+?)\n---", text, re.DOTALL)
if not match:
return {}
return yaml.safe_load(match.group(1))
def extract_code_blocks(text: str) -> list[str]:
"""Extract bash/sh fenced code block contents from markdown."""
return re.findall(r"```(?:bash|sh)\n(.*?)```", text, re.DOTALL)
# ---------------------------------------------------------------------------
# SKILL.md frontmatter
# ---------------------------------------------------------------------------
class TestSkillMdFrontmatter:
"""SKILL.md YAML frontmatter parses correctly."""
@pytest.fixture(autouse=True)
def _load(self):
self.text = SKILL_MD.read_text()
self.meta = parse_frontmatter(self.text)
def test_name(self):
assert self.meta.get("name") == "sci-swarm"
def test_description_present(self):
desc = self.meta.get("description", "")
assert len(desc) > 20, "description should be meaningful"
def test_license_reference(self):
license_ref = self.meta.get("license", "")
assert "LICENSE.txt" in license_ref
def test_allowed_tools_in_metadata(self):
metadata = self.meta.get("metadata", {})
tools = metadata.get("allowed-tools", "")
assert "execute" in tools
# ---------------------------------------------------------------------------
# SKILL.md sections
# ---------------------------------------------------------------------------
class TestSkillMdSections:
"""SKILL.md contains all expected sections."""
EXPECTED_HEADINGS = [
"Sandbox Path Warning",
"Quick Start",
"First-Time Script Setup",
"Workspace Mode: Always Use --workdir",
"Path Perspective Asymmetry",
"Spawning EvoSci Instances",
"Monitoring Progress",
"Collecting Results",
"Interactive REPL Control",
"Agent Communication",
"Finding & Cleaning Sessions",
"Swarm Conventions",
]
@pytest.fixture(autouse=True)
def _load(self):
self.text = SKILL_MD.read_text()
@pytest.mark.parametrize("heading", EXPECTED_HEADINGS)
def test_section_present(self, heading):
assert heading in self.text, f"Missing section: {heading}"
# ---------------------------------------------------------------------------
# Script syntax validation
# ---------------------------------------------------------------------------
class TestScriptSyntax:
"""All shell scripts pass bash -n syntax check."""
SCRIPTS = list(SCRIPTS_DIR.glob("*.sh"))
@pytest.mark.parametrize("script", SCRIPTS, ids=lambda s: s.name)
def test_bash_syntax(self, script):
result = subprocess.run(
["bash", "-n", str(script)],
capture_output=True,
text=True,
)
assert result.returncode == 0, (
f"{script.name} has syntax errors:\n{result.stderr}"
)
# ---------------------------------------------------------------------------
# No literal absolute paths in SKILL.md code blocks
# ---------------------------------------------------------------------------
class TestNoLiteralAbsolutePaths:
"""Code blocks must not contain hardcoded absolute paths.
The sandbox converts /path to ./path, so all code examples must
use shell variables ($SOCKET, $SOCKET_DIR, etc.) for absolute paths.
"""
# Patterns that indicate a literal absolute path (not in a variable assignment)
FORBIDDEN_PATTERNS = [
# Literal /tmp/ not part of a variable expansion or default value
r'(?<!\$\{TMPDIR:-)/tmp/',
# Literal /Users/ or /home/
r'/Users/',
r'/home/',
]
@pytest.fixture(autouse=True)
def _load(self):
self.text = SKILL_MD.read_text()
self.code_blocks = extract_code_blocks(self.text)
def test_no_literal_tmp_paths(self):
"""No hardcoded /tmp/ outside of ${TMPDIR:-/tmp} defaults and WRONG examples."""
for i, block in enumerate(self.code_blocks):
# Determine if this block is in the "WRONG" example section
block_pos = self.text.find(block)
preceding = self.text[:block_pos]
is_wrong_example = "# WRONG" in preceding[preceding.rfind("```"):] if "```" in preceding else False
for line in block.strip().splitlines():
line_stripped = line.strip()
# Skip comments
if line_stripped.startswith("#"):
continue
# Skip lines that are variable assignments with defaults
if "TMPDIR:-/tmp" in line:
continue
# Skip lines in the WRONG example block
if is_wrong_example:
continue
if "/tmp/" in line:
pytest.fail(
f"Code block {i} has literal /tmp/ path "
f"(use shell variable instead):\n {line}"
)
def test_no_literal_user_paths(self):
"""No hardcoded /Users/ or /home/ paths in code blocks."""
for i, block in enumerate(self.code_blocks):
for line in block.strip().splitlines():
if line.strip().startswith("#"):
continue
for pattern in ["/Users/", "/home/"]:
if pattern in line:
pytest.fail(
f"Code block {i} has literal {pattern} path:\n {line}"
)
# ---------------------------------------------------------------------------
# Scripts reference correct env vars (not openclaw)
# ---------------------------------------------------------------------------
class TestEnvVarNaming:
"""Scripts must use EVOSCI_ prefixed env vars, not OPENCLAW_ or CLAWDBOT_."""
SCRIPTS = list(SCRIPTS_DIR.glob("*.sh"))
@pytest.mark.parametrize("script", SCRIPTS, ids=lambda s: s.name)
def test_no_openclaw_vars(self, script):
content = script.read_text()
assert "OPENCLAW_" not in content, (
f"{script.name} references OPENCLAW_ env vars"
)
assert "CLAWDBOT_" not in content, (
f"{script.name} references CLAWDBOT_ env vars"
)
# wait-for-text.sh doesn't need EVOSCI_TMUX_SOCKET_DIR — it takes
# the socket path via -S argument from the caller.
SCRIPTS_NEEDING_SOCKET_DIR = [
s for s in SCRIPTS if s.name != "wait-for-text.sh"
]
@pytest.mark.parametrize(
"script", SCRIPTS_NEEDING_SOCKET_DIR, ids=lambda s: s.name
)
def test_uses_evosci_socket_dir(self, script):
content = script.read_text()
assert "EVOSCI_TMUX_SOCKET_DIR" in content, (
f"{script.name} should reference EVOSCI_TMUX_SOCKET_DIR"
)
# ---------------------------------------------------------------------------
# Inline commands pass validate_command()
# ---------------------------------------------------------------------------
class TestSandboxCompatibility:
"""Inline commands from SKILL.md should pass the sandbox validator."""
@pytest.fixture(autouse=True)
def _load(self):
self.text = SKILL_MD.read_text()
self.code_blocks = extract_code_blocks(self.text)
def _extract_commands(self) -> list[str]:
"""Extract executable commands from code blocks."""
commands = []
for block in self.code_blocks:
for line in block.strip().splitlines():
line = line.strip()
# Skip comments, empty lines, variable assignments, control flow
if not line or line.startswith("#"):
continue
if line.startswith("if ") or line.startswith("fi"):
continue
if line.startswith("for ") or line.startswith("done"):
continue
if line.startswith("else") or line.startswith("then"):
continue
# Skip pure variable assignments (VAR=value with no command)
if re.match(r'^[A-Z_]+=', line) and "tmux" not in line:
continue
# Skip non-shell directives
if line.startswith("read_file") or line.startswith("write_file"):
continue
if line.startswith("ls(") or line.startswith("To monitor"):
continue
commands.append(line)
return commands
def test_no_blocked_commands(self):
"""No commands use sudo, chmod, or other blocked operations."""
# Import validate_command directly to avoid relative import issues
import importlib.util
spec = importlib.util.spec_from_file_location(
"backends",
Path(__file__).parent.parent / "EvoScientist" / "backends.py",
)
mod = importlib.util.module_from_spec(spec)
spec.loader.exec_module(mod)
validate_command = mod.validate_command
commands = self._extract_commands()
assert len(commands) > 0, "Should find some commands to validate"
for cmd in commands:
# Only validate simple commands (not multi-line or complex pipes)
if "&&" in cmd:
# Validate each part separately
parts = cmd.split("&&")
for part in parts:
part = part.strip()
if part:
error = validate_command(part)
if error:
pytest.fail(f"Command blocked: {part}\n Error: {error}")
else:
error = validate_command(cmd)
if error:
pytest.fail(f"Command blocked: {cmd}\n Error: {error}")
# ---------------------------------------------------------------------------
# Script file completeness
# ---------------------------------------------------------------------------
class TestScriptCompleteness:
"""All expected scripts exist."""
EXPECTED_SCRIPTS = [
"find-sessions.sh",
"wait-for-text.sh",
"spawn-evosci.sh",
"collect-results.sh",
]
@pytest.mark.parametrize("script_name", EXPECTED_SCRIPTS)
def test_script_exists(self, script_name):
script = SCRIPTS_DIR / script_name
assert script.exists(), f"Missing script: {script_name}"
assert script.stat().st_size > 100, f"Script too small: {script_name}"
def test_all_scripts_have_shebang(self):
for script in SCRIPTS_DIR.glob("*.sh"):
first_line = script.read_text().splitlines()[0]
assert first_line.startswith("#!/"), (
f"{script.name} missing shebang line"
)
def test_all_scripts_have_set_euo(self):
for script in SCRIPTS_DIR.glob("*.sh"):
content = script.read_text()
assert "set -euo pipefail" in content, (
f"{script.name} missing 'set -euo pipefail'"
)
def test_all_scripts_have_usage(self):
for script in SCRIPTS_DIR.glob("*.sh"):
content = script.read_text()
assert "usage()" in content, (
f"{script.name} missing usage() function"
)
# ---------------------------------------------------------------------------
# spawn-evosci.sh specifics
# ---------------------------------------------------------------------------
class TestSpawnScript:
"""spawn-evosci.sh has required safety features."""
@pytest.fixture(autouse=True)
def _load(self):
self.content = (SCRIPTS_DIR / "spawn-evosci.sh").read_text()
def test_checks_recursion_depth(self):
assert "EVOSCI_TMUX_DEPTH" in self.content
def test_checks_max_depth(self):
assert "EVOSCI_MAX_TMUX_DEPTH" in self.content
def test_checks_session_limit(self):
assert "EVOSCI_MAX_TMUX_SESSIONS" in self.content
def test_checks_duplicate_session(self):
assert "has-session" in self.content
def test_creates_workspace(self):
assert "mkdir -p" in self.content
def test_updates_registry(self):
assert "registry.json" in self.content
def test_uses_workdir_flag(self):
assert "--workdir" in self.content
def test_validates_name(self):
# Should validate name format
assert "alphanumeric" in self.content.lower() or re.search(
r'\[a-zA-Z0-9\]', self.content
)
def test_supports_thread_id(self):
assert "--thread-id" in self.content
def test_supports_no_thinking(self):
assert "--no-thinking" in self.content