diff --git a/EvoScientist/skills/sci-swarm/LICENSE.txt b/EvoScientist/skills/sci-swarm/LICENSE.txt deleted file mode 100644 index 4fe30c9..0000000 --- a/EvoScientist/skills/sci-swarm/LICENSE.txt +++ /dev/null @@ -1,21 +0,0 @@ -MIT License - -Copyright (c) 2026 EvoScientist - -Permission is hereby granted, free of charge, to any person obtaining a copy -of this software and associated documentation files (the "Software"), to deal -in the Software without restriction, including without limitation the rights -to use, copy, modify, merge, publish, distribute, sublicense, and/or sell -copies of the Software, and to permit persons to whom the Software is -furnished to do so, subject to the following conditions: - -The above copyright notice and this permission notice shall be included in all -copies or substantial portions of the Software. - -THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR -IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, -FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE -AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER -LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, -OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE -SOFTWARE. diff --git a/EvoScientist/skills/sci-swarm/SKILL.md b/EvoScientist/skills/sci-swarm/SKILL.md deleted file mode 100644 index 257077a..0000000 --- a/EvoScientist/skills/sci-swarm/SKILL.md +++ /dev/null @@ -1,318 +0,0 @@ ---- -name: sci-swarm -description: Spawn parallel EvoSci experiments, control interactive REPLs, and orchestrate multi-agent workflows via tmux sessions. Use when the user wants to run multiple experiments simultaneously, needs an interactive Python/shell REPL, or wants to coordinate independent research tasks. -license: Complete terms in LICENSE.txt -metadata: - version: "0.1.0" - author: Xi Zhang - tags: [parallel, experiments, tmux, orchestration, multi-agent] - allowed-tools: "execute read_file write_file ls" ---- - -# sci-swarm — Multi-Agent Experiment Orchestration - -Spawn independent EvoSci CLI instances in isolated tmux sessions for parallel experiments, interactive REPL control, and collaborative research workflows. - -## Sandbox Path Warning - -The EvoScientist sandbox converts literal `/path` to `./path` in shell commands. **Always use shell variables for absolute paths**, never hardcoded literals like `/tmp/...`. - -```bash -# CORRECT - uses shell variable -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -SOCKET="$SOCKET_DIR/evosci.sock" -tmux -S "$SOCKET" new -d -s my-session -``` - -WRONG — literal absolute path will be converted to `./tmp/...` by the sandbox: - -``` -tmux -S /tmp/evosci-tmux-sockets/evosci.sock new -d -s my-session -``` - -## Quick Start - -Create a session, send a command, and capture output: - -```bash -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -mkdir -p "$SOCKET_DIR" -SOCKET="$SOCKET_DIR/evosci.sock" -SESSION=evosci-demo - -tmux -S "$SOCKET" new -d -s "$SESSION" -n shell -tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 -- 'echo hello from tmux' Enter -sleep 1 -tmux -S "$SOCKET" capture-pane -p -J -t "$SESSION":0.0 -S -200 -``` - -After starting a session, always print monitor commands: - -``` -To monitor: - tmux -S "$SOCKET" attach -t "$SESSION" - tmux -S "$SOCKET" capture-pane -p -J -t "$SESSION":0.0 -S -200 -``` - -## First-Time Script Setup - -The skill includes helper scripts under `/skills/sci-swarm/scripts/`. Because the skills directory is read-only and shell CWD differs from the virtual path, copy them to the workspace first: - -```bash -mkdir -p tmux-scripts -``` - -Then use `read_file` and `write_file` to copy each script: - -1. `read_file("/skills/sci-swarm/scripts/spawn-evosci.sh")` then `write_file("/tmux-scripts/spawn-evosci.sh", content)` -2. `read_file("/skills/sci-swarm/scripts/find-sessions.sh")` then `write_file("/tmux-scripts/find-sessions.sh", content)` -3. `read_file("/skills/sci-swarm/scripts/wait-for-text.sh")` then `write_file("/tmux-scripts/wait-for-text.sh", content)` -4. `read_file("/skills/sci-swarm/scripts/collect-results.sh")` then `write_file("/tmux-scripts/collect-results.sh", content)` - -Run scripts with `bash` (no `chmod` needed): - -```bash -bash tmux-scripts/spawn-evosci.sh experiment-1 "Run baseline comparison" -``` - -## Workspace Mode: Always Use --workdir - -**Critical**: Always use `--workdir` to spawn sub-agents. Never use `-m run` (it nests `workspace/runs/` inside the main workspace, breaking read_file paths). - -| Mode | Effect | Verdict | -|------|--------|---------| -| `-m run -n "xxx"` | Nests `workspace/runs/xxx/` with copied memory/skills | Avoid — path nesting breaks | -| `--use-cwd` | Agent works in main agent's root directory | Pollutes root directory | -| `--workdir ` | Agent works in specified subdirectory | Best practice — isolated yet accessible | - -## Path Perspective Asymmetry - -Each spawned EvoSci instance sees its `--workdir` as `/` (virtual root). When the main agent needs to read a sub-agent's output, it must use the **main agent's path perspective**: - -- Sub-agent writes: `write_file("/results.md", ...)` → file lands in its workdir -- Main agent reads: `read_file("/agents/experiment-1/results.md")` → using main agent's path - -This applies to all file operations (read_file, ls, grep, glob). - -## Spawning EvoSci Instances - -### Using the spawn script (recommended) - -```bash -bash tmux-scripts/spawn-evosci.sh experiment-1 "Investigate the effect of learning rate on convergence" -bash tmux-scripts/spawn-evosci.sh experiment-2 "Compare Adam vs SGD optimizers" --model claude-sonnet-4-5 -``` - -Add `--no-thinking` to save tokens in automated workflows: - -```bash -bash tmux-scripts/spawn-evosci.sh experiment-1 "Run baseline comparison" --no-thinking -``` - -Each instance gets its own workspace at `workspace/runs/` and shares `/memory/` with all other instances. - -### Stateful multi-round sessions with --thread-id - -Use `--thread-id` to maintain conversation context across multiple invocations to the same session. The sub-agent remembers previous instructions and can build on prior work: - -```bash -bash tmux-scripts/spawn-evosci.sh builder "Step 1: Set up the data pipeline" --thread-id pipeline-001 -``` - -After the first call completes, send follow-up prompts to the same session with the same thread ID: - -```bash -bash tmux-scripts/spawn-evosci.sh builder "Step 2: Train the model using the pipeline from step 1" --thread-id pipeline-001 -``` - -| Without --thread-id | With --thread-id | -|---------------------|------------------| -| Each call is a fresh instance, no memory | Preserves full conversation history | -| Best for one-shot tasks | Best for multi-step iterative experiments | - -### Inline (without helper script) - -```bash -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -mkdir -p "$SOCKET_DIR" -SOCKET="$SOCKET_DIR/evosci.sock" -NAME=experiment-1 -WORKDIR="$(pwd)/workspace/runs/$NAME" -mkdir -p "$WORKDIR" - -tmux -S "$SOCKET" new-session -d -s "evosci-$NAME" -n shell -tmux -S "$SOCKET" send-keys -t "evosci-$NAME":0.0 -- "EvoSci -p 'Your research prompt here' --workdir $WORKDIR --no-thinking" Enter -``` - -### Spawning multiple experiments - -```bash -for exp in baseline-lr0.001 baseline-lr0.01 baseline-lr0.1; do - bash tmux-scripts/spawn-evosci.sh "$exp" "Train model with learning rate ${exp##*-}" --no-thinking -done -``` - -## Monitoring Progress - -### Capture pane output - -```bash -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -SOCKET="$SOCKET_DIR/evosci.sock" -tmux -S "$SOCKET" capture-pane -p -J -t "evosci-experiment-1":0.0 -S -200 -``` - -### Detect completion - -EvoSci prints "Goodbye!" when it exits. Check for it: - -```bash -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -SOCKET="$SOCKET_DIR/evosci.sock" -if tmux -S "$SOCKET" capture-pane -p -t "evosci-experiment-1":0.0 -S -5 | grep -q "Goodbye!"; then - echo "experiment-1: DONE" -else - echo "experiment-1: still running" -fi -``` - -### Wait for text (blocking) - -```bash -bash tmux-scripts/wait-for-text.sh -S "$SOCKET" -t "evosci-experiment-1":0.0 -p 'Goodbye!' -T 300 -``` - -Options: -- `-t` target pane (required) -- `-p` regex pattern to match (required) -- `-S` tmux socket path -- `-F` treat pattern as fixed string -- `-T` timeout in seconds (default: 15) -- `-i` poll interval (default: 0.5) -- `-l` history lines to search (default: 1000) - -## Collecting Results - -### Using the collect script - -```bash -bash tmux-scripts/collect-results.sh experiment-1 -bash tmux-scripts/collect-results.sh experiment-1 --pane-output -``` - -This shows session status, lists output files, and optionally captures the pane. - -### Manual collection - -Read output files from completed workspaces: - -```bash -read_file("/runs/experiment-1/results.md") -read_file("/runs/experiment-1/analysis.py") -``` - -List workspace contents: - -```bash -ls("/runs/experiment-1/") -``` - -## Interactive REPL Control - -### Python REPL - -Use `PYTHON_BASIC_REPL=1` to prevent the enhanced REPL from breaking send-keys: - -```bash -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -SOCKET="$SOCKET_DIR/evosci.sock" -SESSION=evosci-python - -tmux -S "$SOCKET" new -d -s "$SESSION" -n shell -tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 -- 'PYTHON_BASIC_REPL=1 python3 -q' Enter -sleep 1 -tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 -l -- 'print("hello")' && sleep 0.1 && tmux -S "$SOCKET" send-keys -t "$SESSION":0.0 Enter -sleep 1 -tmux -S "$SOCKET" capture-pane -p -J -t "$SESSION":0.0 -S -50 -``` - -### Sending input safely - -- Prefer literal sends: `tmux -S "$SOCKET" send-keys -t target -l -- "$cmd"` -- Control keys: `tmux -S "$SOCKET" send-keys -t target C-c` -- For TUI apps (interactive CLIs), separate text and Enter with a delay: - -```bash -tmux -S "$SOCKET" send-keys -t target -l -- "$cmd" && sleep 0.1 && tmux -S "$SOCKET" send-keys -t target Enter -``` - -## Agent Communication - -For collaborative research across spawned instances, use file-based message passing via shared memory: - -```bash -# Agent A writes a message -write_file("/memory/messages/from-experiment-1.md", "## Finding\nLearning rate 0.01 converges fastest.") - -# Agent B reads it -read_file("/memory/messages/from-experiment-1.md") -``` - -All EvoSci instances share `/memory/` regardless of their workspace directory. - -## Finding & Cleaning Sessions - -### List sessions - -```bash -bash tmux-scripts/find-sessions.sh -S "$SOCKET" -bash tmux-scripts/find-sessions.sh --all -``` - -Options: -- `-L` socket name (tmux -L) -- `-S` socket path (tmux -S) -- `-A`/`--all` scan all sockets under `EVOSCI_TMUX_SOCKET_DIR` -- `-q` filter session names - -### Kill a session - -```bash -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -SOCKET="$SOCKET_DIR/evosci.sock" -tmux -S "$SOCKET" kill-session -t "evosci-experiment-1" -``` - -### Kill all sessions - -```bash -SOCKET_DIR="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" -SOCKET="$SOCKET_DIR/evosci.sock" -tmux -S "$SOCKET" kill-server -``` - -## Swarm Conventions - -### Session registry - -The spawn script maintains a registry at `$EVOSCI_TMUX_SOCKET_DIR/registry.json` tracking all spawned sessions, their workspaces, prompts, and status. - -### Limits - -| Variable | Default | Description | -|----------|---------|-------------| -| `EVOSCI_MAX_TMUX_SESSIONS` | 5 | Maximum concurrent sessions | -| `EVOSCI_TMUX_DEPTH` | 0 | Current recursion depth (auto-incremented) | -| `EVOSCI_MAX_TMUX_DEPTH` | 3 | Maximum recursion depth | -| `EVOSCI_TMUX_SOCKET_DIR` | `$TMPDIR/evosci-tmux-sockets` | Socket directory | - -### Naming - -- Session names: `evosci-` (e.g., `evosci-experiment-1`) -- Workspaces: `workspace/runs//` -- Keep names short, alphanumeric with hyphens - -### Targeting panes - -- Target format: `session:window.pane` (defaults to `:0.0`) -- Inspect: `tmux -S "$SOCKET" list-sessions`, `tmux -S "$SOCKET" list-panes -a` diff --git a/EvoScientist/skills/sci-swarm/scripts/collect-results.sh b/EvoScientist/skills/sci-swarm/scripts/collect-results.sh deleted file mode 100644 index 966a81c..0000000 --- a/EvoScientist/skills/sci-swarm/scripts/collect-results.sh +++ /dev/null @@ -1,150 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -usage() { - cat <<'USAGE' -Usage: collect-results.sh [options] - -Collect results from a spawned EvoSci session. - -Arguments: - name Session name suffix (matches evosci-) - -Options: - --socket-path tmux socket path (default: $EVOSCI_TMUX_SOCKET_DIR/evosci.sock) - --workdir Workspace directory (default: workspace/runs/) - --pane-output Also capture the final pane output - -h, --help Show this help -USAGE -} - -# --- Defaults --- -name="" -socket_path="" -workdir="" -pane_output=false -socket_dir="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" - -# --- Parse args --- -while [[ $# -gt 0 ]]; do - case "$1" in - --socket-path) socket_path="${2-}"; shift 2 ;; - --workdir) workdir="${2-}"; shift 2 ;; - --pane-output) pane_output=true; shift ;; - -h|--help) usage; exit 0 ;; - -*) echo "Unknown option: $1" >&2; usage; exit 1 ;; - *) - if [[ -z "$name" ]]; then - name="$1" - else - echo "Unexpected argument: $1" >&2; usage; exit 1 - fi - shift - ;; - esac -done - -if [[ -z "$name" ]]; then - echo "Error: name is required" >&2 - usage - exit 1 -fi - -# --- Resolve paths --- -if [[ -z "$socket_path" ]]; then - socket_path="$socket_dir/evosci.sock" -fi - -if [[ -z "$workdir" ]]; then - # Try registry first - registry_file="$socket_dir/registry.json" - if [[ -f "$registry_file" ]]; then - found_workdir=$(EVOSCI_REG_FILE="$registry_file" EVOSCI_REG_NAME="$name" python3 -c " -import json, os -with open(os.environ['EVOSCI_REG_FILE']) as f: - registry = json.load(f) -for entry in registry: - if entry.get('name') == os.environ['EVOSCI_REG_NAME']: - print(entry.get('workdir', '')) - break -" 2>/dev/null || true) - if [[ -n "$found_workdir" ]]; then - workdir="$found_workdir" - fi - fi - # Fallback to convention - if [[ -z "$workdir" ]]; then - workdir="$(pwd)/workspace/runs/${name}" - fi -fi - -# --- Check session status --- -session_name="evosci-${name}" -session_running=false -if tmux -S "$socket_path" has-session -t "$session_name" 2>/dev/null; then - session_running=true -fi - -if [[ "$session_running" == true ]]; then - echo "Status: RUNNING" - echo "Session '$session_name' is still active." -else - echo "Status: COMPLETED" - echo "Session '$session_name' has ended." - - # Update registry status - registry_file="$socket_dir/registry.json" - if [[ -f "$registry_file" ]]; then - EVOSCI_REG_FILE="$registry_file" EVOSCI_REG_NAME="$name" python3 -c " -import json, os, datetime -reg_file = os.environ['EVOSCI_REG_FILE'] -reg_name = os.environ['EVOSCI_REG_NAME'] -with open(reg_file) as f: - registry = json.load(f) -for entry in registry: - if entry.get('name') == reg_name and entry.get('status') == 'running': - entry['status'] = 'completed' - entry['completed_at'] = datetime.datetime.now().isoformat() -with open(reg_file, 'w') as f: - json.dump(registry, f, indent=2) -" 2>/dev/null || true - fi -fi - -echo "" - -# --- Check workspace --- -if [[ ! -d "$workdir" ]]; then - echo "Workspace not found: $workdir" - exit 1 -fi - -echo "Workspace: $workdir" -echo "" - -# --- List output files --- -echo "Output files:" -file_count=0 -for ext in md py csv png json ipynb txt pdf; do - while IFS= read -r -d '' file; do - rel_path="${file#"$workdir"/}" - size=$(wc -c < "$file" 2>/dev/null | tr -d ' ') - printf ' %-50s %s bytes\n' "$rel_path" "$size" - file_count=$((file_count + 1)) - done < <(find "$workdir" -maxdepth 3 -name "*.${ext}" -print0 2>/dev/null) -done - -if [[ "$file_count" -eq 0 ]]; then - echo " (no output files found)" -fi - -# --- Optionally capture pane output --- -if [[ "$pane_output" == true && "$session_running" == true ]]; then - echo "" - echo "--- Pane Output (last 500 lines) ---" - tmux -S "$socket_path" capture-pane -p -J -t "$session_name":0.0 -S -500 2>/dev/null || echo "(could not capture pane)" -fi - -echo "" -echo "To read a specific file:" -echo " read_file \"/runs/${name}/\"" diff --git a/EvoScientist/skills/sci-swarm/scripts/find-sessions.sh b/EvoScientist/skills/sci-swarm/scripts/find-sessions.sh deleted file mode 100644 index 47f60a6..0000000 --- a/EvoScientist/skills/sci-swarm/scripts/find-sessions.sh +++ /dev/null @@ -1,112 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -usage() { - cat <<'USAGE' -Usage: find-sessions.sh [-L socket-name|-S socket-path|-A] [-q pattern] - -List tmux sessions on a socket (default tmux socket if none provided). - -Options: - -L, --socket tmux socket name (passed to tmux -L) - -S, --socket-path tmux socket path (passed to tmux -S) - -A, --all scan all sockets under EVOSCI_TMUX_SOCKET_DIR - -q, --query case-insensitive substring to filter session names - -h, --help show this help -USAGE -} - -socket_name="" -socket_path="" -query="" -scan_all=false -socket_dir="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" - -while [[ $# -gt 0 ]]; do - case "$1" in - -L|--socket) socket_name="${2-}"; shift 2 ;; - -S|--socket-path) socket_path="${2-}"; shift 2 ;; - -A|--all) scan_all=true; shift ;; - -q|--query) query="${2-}"; shift 2 ;; - -h|--help) usage; exit 0 ;; - *) echo "Unknown option: $1" >&2; usage; exit 1 ;; - esac -done - -if [[ "$scan_all" == true && ( -n "$socket_name" || -n "$socket_path" ) ]]; then - echo "Cannot combine --all with -L or -S" >&2 - exit 1 -fi - -if [[ -n "$socket_name" && -n "$socket_path" ]]; then - echo "Use either -L or -S, not both" >&2 - exit 1 -fi - -if ! command -v tmux >/dev/null 2>&1; then - echo "tmux not found in PATH" >&2 - exit 1 -fi - -list_sessions() { - local label="$1"; shift - local tmux_cmd=(tmux "$@") - - if ! sessions="$("${tmux_cmd[@]}" list-sessions -F '#{session_name}\t#{session_attached}\t#{session_created_string}' 2>/dev/null)"; then - echo "No tmux server found on $label" >&2 - return 1 - fi - - if [[ -n "$query" ]]; then - sessions="$(printf '%s\n' "$sessions" | grep -i -- "$query" || true)" - fi - - if [[ -z "$sessions" ]]; then - echo "No sessions found on $label" - return 0 - fi - - echo "Sessions on $label:" - printf '%s\n' "$sessions" | while IFS=$'\t' read -r name attached created; do - attached_label=$([[ "$attached" == "1" ]] && echo "attached" || echo "detached") - printf ' - %s (%s, started %s)\n' "$name" "$attached_label" "$created" - done -} - -if [[ "$scan_all" == true ]]; then - if [[ ! -d "$socket_dir" ]]; then - echo "Socket directory not found: $socket_dir" >&2 - exit 1 - fi - - shopt -s nullglob - sockets=("$socket_dir"/*) - shopt -u nullglob - - if [[ "${#sockets[@]}" -eq 0 ]]; then - echo "No sockets found under $socket_dir" >&2 - exit 1 - fi - - exit_code=0 - for sock in "${sockets[@]}"; do - if [[ ! -S "$sock" ]]; then - continue - fi - list_sessions "socket path '$sock'" -S "$sock" || exit_code=$? - done - exit "$exit_code" -fi - -tmux_cmd=(tmux) -socket_label="default socket" - -if [[ -n "$socket_name" ]]; then - tmux_cmd+=(-L "$socket_name") - socket_label="socket name '$socket_name'" -elif [[ -n "$socket_path" ]]; then - tmux_cmd+=(-S "$socket_path") - socket_label="socket path '$socket_path'" -fi - -list_sessions "$socket_label" "${tmux_cmd[@]:1}" diff --git a/EvoScientist/skills/sci-swarm/scripts/spawn-evosci.sh b/EvoScientist/skills/sci-swarm/scripts/spawn-evosci.sh deleted file mode 100644 index a40ab0a..0000000 --- a/EvoScientist/skills/sci-swarm/scripts/spawn-evosci.sh +++ /dev/null @@ -1,209 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -usage() { - cat <<'USAGE' -Usage: spawn-evosci.sh "" [options] - -Spawn an EvoSci CLI instance in an isolated tmux session. - -Arguments: - name Session suffix (session will be named evosci-) - prompt Research prompt to pass to EvoSci -p - -Options: - --socket-path tmux socket path (default: $EVOSCI_TMUX_SOCKET_DIR/evosci.sock) - --workdir Workspace directory (default: workspace/runs/) - --model Model to use (e.g., claude-sonnet-4-5) - --thread-id Thread ID for stateful multi-round conversations - --no-thinking Disable thinking output (saves tokens) - --max-sessions Max allowed sessions (default: $EVOSCI_MAX_TMUX_SESSIONS or 5) - --max-depth Max recursion depth (default: $EVOSCI_MAX_TMUX_DEPTH or 3) - -h, --help Show this help -USAGE -} - -# --- Defaults --- -name="" -prompt="" -socket_path="" -workdir="" -model="" -thread_id="" -no_thinking=false -max_sessions="${EVOSCI_MAX_TMUX_SESSIONS:-5}" -max_depth="${EVOSCI_MAX_TMUX_DEPTH:-3}" -current_depth="${EVOSCI_TMUX_DEPTH:-0}" -socket_dir="${EVOSCI_TMUX_SOCKET_DIR:-${TMPDIR:-/tmp}/evosci-tmux-sockets}" - -# --- Parse args --- -while [[ $# -gt 0 ]]; do - case "$1" in - --socket-path) socket_path="${2-}"; shift 2 ;; - --workdir) workdir="${2-}"; shift 2 ;; - --model) model="${2-}"; shift 2 ;; - --thread-id) thread_id="${2-}"; shift 2 ;; - --no-thinking) no_thinking=true; shift ;; - --max-sessions) max_sessions="${2-}"; shift 2 ;; - --max-depth) max_depth="${2-}"; shift 2 ;; - -h|--help) usage; exit 0 ;; - -*) echo "Unknown option: $1" >&2; usage; exit 1 ;; - *) - if [[ -z "$name" ]]; then - name="$1" - elif [[ -z "$prompt" ]]; then - prompt="$1" - else - echo "Unexpected argument: $1" >&2; usage; exit 1 - fi - shift - ;; - esac -done - -# --- Validate required args --- -if [[ -z "$name" || -z "$prompt" ]]; then - echo "Error: name and prompt are required" >&2 - usage - exit 1 -fi - -# Validate name (alphanumeric, hyphens, underscores only) -if ! [[ "$name" =~ ^[a-zA-Z0-9][a-zA-Z0-9_-]*$ ]]; then - echo "Error: name must be alphanumeric with hyphens/underscores (got: $name)" >&2 - exit 1 -fi - -# --- Check prerequisites --- -if ! command -v tmux >/dev/null 2>&1; then - echo "Error: tmux not found in PATH" >&2 - exit 1 -fi - -if ! command -v EvoSci >/dev/null 2>&1; then - echo "Error: EvoSci not found in PATH (install with: pip install -e '.[dev]')" >&2 - exit 1 -fi - -# --- Check recursion depth --- -if (( current_depth >= max_depth )); then - echo "Error: recursion depth limit reached (current=$current_depth, max=$max_depth)" >&2 - echo "Set EVOSCI_MAX_TMUX_DEPTH to increase the limit." >&2 - exit 1 -fi - -# --- Resolve socket --- -mkdir -p "$socket_dir" -if [[ -z "$socket_path" ]]; then - socket_path="$socket_dir/evosci.sock" -fi - -# --- Check session limit --- -session_count=0 -if tmux -S "$socket_path" list-sessions 2>/dev/null | grep -c '^' > /dev/null 2>&1; then - session_count=$(tmux -S "$socket_path" list-sessions 2>/dev/null | grep -c '^' || echo 0) -fi -if (( session_count >= max_sessions )); then - echo "Error: session limit reached ($session_count/$max_sessions)" >&2 - echo "Kill completed sessions or increase --max-sessions." >&2 - exit 1 -fi - -# --- Check duplicate session --- -session_name="evosci-${name}" -if tmux -S "$socket_path" has-session -t "$session_name" 2>/dev/null; then - if [[ -n "$thread_id" ]]; then - # With --thread-id, reuse is intentional: kill the stale session - # (EvoSci has exited but the tmux shell remains) - tmux -S "$socket_path" kill-session -t "$session_name" 2>/dev/null || true - else - echo "Error: session '$session_name' already exists" >&2 - echo "Use a different name or kill the existing session:" >&2 - echo " tmux -S \"$socket_path\" kill-session -t \"$session_name\"" >&2 - exit 1 - fi -fi - -# --- Resolve workspace --- -if [[ -z "$workdir" ]]; then - workdir="$(pwd)/workspace/runs/${name}" -fi - -# Make workdir absolute if relative -if [[ "$workdir" != /* ]]; then - workdir="$(pwd)/$workdir" -fi -mkdir -p "$workdir" - -# --- Build EvoSci command --- -evosci_cmd="EvoSci -p $(printf '%q' "$prompt") --workdir $(printf '%q' "$workdir")" -if [[ -n "$model" ]]; then - evosci_cmd+=" --model $(printf '%q' "$model")" -fi -if [[ -n "$thread_id" ]]; then - evosci_cmd+=" --thread-id $(printf '%q' "$thread_id")" -fi -if [[ "$no_thinking" == true ]]; then - evosci_cmd+=" --no-thinking" -fi - -# --- Create tmux session --- -next_depth=$((current_depth + 1)) - -if ! tmux -S "$socket_path" new-session -d -s "$session_name" -n shell \ - -e "EVOSCI_TMUX_DEPTH=$next_depth" \ - -e "EVOSCI_TMUX_SOCKET_DIR=$socket_dir" 2>/dev/null; then - echo "Error: failed to create tmux session '$session_name'" >&2 - exit 2 -fi - -# --- Launch EvoSci --- -tmux -S "$socket_path" send-keys -t "$session_name":0.0 -- "$evosci_cmd" Enter - -# --- Update registry --- -registry_file="$socket_dir/registry.json" -EVOSCI_REG_FILE="$registry_file" \ -EVOSCI_REG_NAME="$name" \ -EVOSCI_REG_SESSION="$session_name" \ -EVOSCI_REG_SOCKET="$socket_path" \ -EVOSCI_REG_WORKDIR="$workdir" \ -EVOSCI_REG_PROMPT="$prompt" \ -EVOSCI_REG_DEPTH="$next_depth" \ -EVOSCI_REG_THREAD="$thread_id" \ -python3 -c " -import json, os, datetime -registry_path = os.environ['EVOSCI_REG_FILE'] -try: - with open(registry_path) as f: - registry = json.load(f) -except (FileNotFoundError, json.JSONDecodeError): - registry = [] -entry = { - 'name': os.environ['EVOSCI_REG_NAME'], - 'session': os.environ['EVOSCI_REG_SESSION'], - 'socket': os.environ['EVOSCI_REG_SOCKET'], - 'workdir': os.environ['EVOSCI_REG_WORKDIR'], - 'prompt': os.environ['EVOSCI_REG_PROMPT'], - 'status': 'running', - 'started_at': datetime.datetime.now().isoformat(), - 'depth': int(os.environ['EVOSCI_REG_DEPTH']) -} -thread_id = os.environ.get('EVOSCI_REG_THREAD', '') -if thread_id: - entry['thread_id'] = thread_id -registry.append(entry) -with open(registry_path, 'w') as f: - json.dump(registry, f, indent=2) -" 2>/dev/null || true - -# --- Print confirmation --- -echo "Spawned EvoSci session: $session_name" -echo " Workspace: $workdir" -if [[ -n "$thread_id" ]]; then - echo " Thread ID: $thread_id (stateful)" -fi -echo " Depth: $next_depth/$max_depth" -echo "" -echo "Monitor commands:" -echo " tmux -S \"$socket_path\" attach -t \"$session_name\"" -echo " tmux -S \"$socket_path\" capture-pane -p -J -t \"$session_name\":0.0 -S -200" diff --git a/EvoScientist/skills/sci-swarm/scripts/wait-for-text.sh b/EvoScientist/skills/sci-swarm/scripts/wait-for-text.sh deleted file mode 100644 index 3020cb0..0000000 --- a/EvoScientist/skills/sci-swarm/scripts/wait-for-text.sh +++ /dev/null @@ -1,92 +0,0 @@ -#!/usr/bin/env bash -set -euo pipefail - -usage() { - cat <<'USAGE' -Usage: wait-for-text.sh -t target -p pattern [options] - -Poll a tmux pane for text and exit when found. - -Options: - -t, --target tmux target (session:window.pane), required - -p, --pattern regex pattern to look for, required - -F, --fixed treat pattern as a fixed string (grep -F) - -S, --socket tmux socket path (passed to tmux -S) - -T, --timeout seconds to wait (integer, default: 15) - -i, --interval poll interval in seconds (default: 0.5) - -l, --lines number of history lines to inspect (integer, default: 1000) - -h, --help show this help -USAGE -} - -target="" -pattern="" -grep_flag="-E" -socket_path="" -timeout=15 -interval=0.5 -lines=1000 - -while [[ $# -gt 0 ]]; do - case "$1" in - -t|--target) target="${2-}"; shift 2 ;; - -p|--pattern) pattern="${2-}"; shift 2 ;; - -F|--fixed) grep_flag="-F"; shift ;; - -S|--socket) socket_path="${2-}"; shift 2 ;; - -T|--timeout) timeout="${2-}"; shift 2 ;; - -i|--interval) interval="${2-}"; shift 2 ;; - -l|--lines) lines="${2-}"; shift 2 ;; - -h|--help) usage; exit 0 ;; - *) echo "Unknown option: $1" >&2; usage; exit 1 ;; - esac -done - -if [[ -z "$target" || -z "$pattern" ]]; then - echo "target and pattern are required" >&2 - usage - exit 1 -fi - -if ! [[ "$timeout" =~ ^[0-9]+$ ]]; then - echo "timeout must be an integer number of seconds" >&2 - exit 1 -fi - -if ! [[ "$lines" =~ ^[0-9]+$ ]]; then - echo "lines must be an integer" >&2 - exit 1 -fi - -if ! command -v tmux >/dev/null 2>&1; then - echo "tmux not found in PATH" >&2 - exit 1 -fi - -# Build base tmux command with optional socket -tmux_base=(tmux) -if [[ -n "$socket_path" ]]; then - tmux_base+=(-S "$socket_path") -fi - -# End time in epoch seconds (integer, good enough for polling) -start_epoch=$(date +%s) -deadline=$((start_epoch + timeout)) - -while true; do - # -J joins wrapped lines, -S uses negative index to read last N lines - pane_text="$("${tmux_base[@]}" capture-pane -p -J -t "$target" -S "-${lines}" 2>/dev/null || true)" - - if printf '%s\n' "$pane_text" | grep $grep_flag -- "$pattern" >/dev/null 2>&1; then - exit 0 - fi - - now=$(date +%s) - if (( now >= deadline )); then - echo "Timed out after ${timeout}s waiting for pattern: $pattern" >&2 - echo "Last ${lines} lines from $target:" >&2 - printf '%s\n' "$pane_text" >&2 - exit 1 - fi - - sleep "$interval" -done diff --git a/tests/test_sci_swarm_skill.py b/tests/test_sci_swarm_skill.py deleted file mode 100644 index 4ee2170..0000000 --- a/tests/test_sci_swarm_skill.py +++ /dev/null @@ -1,370 +0,0 @@ -"""Tests for the sci-swarm skill. - -Validates SKILL.md structure, script syntax, naming conventions, -and sandbox compatibility. No tmux or API keys required. -""" - -import re -import subprocess -from pathlib import Path - -import pytest -import yaml - -# Skill directory -SKILL_DIR = Path(__file__).parent.parent / "EvoScientist" / "skills" / "sci-swarm" -SCRIPTS_DIR = SKILL_DIR / "scripts" -SKILL_MD = SKILL_DIR / "SKILL.md" - - -# --------------------------------------------------------------------------- -# Helpers -# --------------------------------------------------------------------------- - -def parse_frontmatter(text: str) -> dict: - """Extract YAML frontmatter from a markdown file.""" - match = re.match(r"^---\n(.+?)\n---", text, re.DOTALL) - if not match: - return {} - return yaml.safe_load(match.group(1)) - - -def extract_code_blocks(text: str) -> list[str]: - """Extract bash/sh fenced code block contents from markdown.""" - return re.findall(r"```(?:bash|sh)\n(.*?)```", text, re.DOTALL) - - -# --------------------------------------------------------------------------- -# SKILL.md frontmatter -# --------------------------------------------------------------------------- - -class TestSkillMdFrontmatter: - """SKILL.md YAML frontmatter parses correctly.""" - - @pytest.fixture(autouse=True) - def _load(self): - self.text = SKILL_MD.read_text() - self.meta = parse_frontmatter(self.text) - - def test_name(self): - assert self.meta.get("name") == "sci-swarm" - - def test_description_present(self): - desc = self.meta.get("description", "") - assert len(desc) > 20, "description should be meaningful" - - def test_license_reference(self): - license_ref = self.meta.get("license", "") - assert "LICENSE.txt" in license_ref - - def test_allowed_tools_in_metadata(self): - metadata = self.meta.get("metadata", {}) - tools = metadata.get("allowed-tools", "") - assert "execute" in tools - - -# --------------------------------------------------------------------------- -# SKILL.md sections -# --------------------------------------------------------------------------- - -class TestSkillMdSections: - """SKILL.md contains all expected sections.""" - - EXPECTED_HEADINGS = [ - "Sandbox Path Warning", - "Quick Start", - "First-Time Script Setup", - "Workspace Mode: Always Use --workdir", - "Path Perspective Asymmetry", - "Spawning EvoSci Instances", - "Monitoring Progress", - "Collecting Results", - "Interactive REPL Control", - "Agent Communication", - "Finding & Cleaning Sessions", - "Swarm Conventions", - ] - - @pytest.fixture(autouse=True) - def _load(self): - self.text = SKILL_MD.read_text() - - @pytest.mark.parametrize("heading", EXPECTED_HEADINGS) - def test_section_present(self, heading): - assert heading in self.text, f"Missing section: {heading}" - - -# --------------------------------------------------------------------------- -# Script syntax validation -# --------------------------------------------------------------------------- - -class TestScriptSyntax: - """All shell scripts pass bash -n syntax check.""" - - SCRIPTS = list(SCRIPTS_DIR.glob("*.sh")) - - @pytest.mark.parametrize("script", SCRIPTS, ids=lambda s: s.name) - def test_bash_syntax(self, script): - result = subprocess.run( - ["bash", "-n", str(script)], - capture_output=True, - text=True, - ) - assert result.returncode == 0, ( - f"{script.name} has syntax errors:\n{result.stderr}" - ) - - -# --------------------------------------------------------------------------- -# No literal absolute paths in SKILL.md code blocks -# --------------------------------------------------------------------------- - -class TestNoLiteralAbsolutePaths: - """Code blocks must not contain hardcoded absolute paths. - - The sandbox converts /path to ./path, so all code examples must - use shell variables ($SOCKET, $SOCKET_DIR, etc.) for absolute paths. - """ - - # Patterns that indicate a literal absolute path (not in a variable assignment) - FORBIDDEN_PATTERNS = [ - # Literal /tmp/ not part of a variable expansion or default value - r'(? list[str]: - """Extract executable commands from code blocks.""" - commands = [] - for block in self.code_blocks: - for line in block.strip().splitlines(): - line = line.strip() - # Skip comments, empty lines, variable assignments, control flow - if not line or line.startswith("#"): - continue - if line.startswith("if ") or line.startswith("fi"): - continue - if line.startswith("for ") or line.startswith("done"): - continue - if line.startswith("else") or line.startswith("then"): - continue - # Skip pure variable assignments (VAR=value with no command) - if re.match(r'^[A-Z_]+=', line) and "tmux" not in line: - continue - # Skip non-shell directives - if line.startswith("read_file") or line.startswith("write_file"): - continue - if line.startswith("ls(") or line.startswith("To monitor"): - continue - commands.append(line) - return commands - - def test_no_blocked_commands(self): - """No commands use sudo, chmod, or other blocked operations.""" - # Import validate_command directly to avoid relative import issues - import importlib.util - spec = importlib.util.spec_from_file_location( - "backends", - Path(__file__).parent.parent / "EvoScientist" / "backends.py", - ) - mod = importlib.util.module_from_spec(spec) - spec.loader.exec_module(mod) - validate_command = mod.validate_command - - commands = self._extract_commands() - assert len(commands) > 0, "Should find some commands to validate" - - for cmd in commands: - # Only validate simple commands (not multi-line or complex pipes) - if "&&" in cmd: - # Validate each part separately - parts = cmd.split("&&") - for part in parts: - part = part.strip() - if part: - error = validate_command(part) - if error: - pytest.fail(f"Command blocked: {part}\n Error: {error}") - else: - error = validate_command(cmd) - if error: - pytest.fail(f"Command blocked: {cmd}\n Error: {error}") - - -# --------------------------------------------------------------------------- -# Script file completeness -# --------------------------------------------------------------------------- - -class TestScriptCompleteness: - """All expected scripts exist.""" - - EXPECTED_SCRIPTS = [ - "find-sessions.sh", - "wait-for-text.sh", - "spawn-evosci.sh", - "collect-results.sh", - ] - - @pytest.mark.parametrize("script_name", EXPECTED_SCRIPTS) - def test_script_exists(self, script_name): - script = SCRIPTS_DIR / script_name - assert script.exists(), f"Missing script: {script_name}" - assert script.stat().st_size > 100, f"Script too small: {script_name}" - - def test_all_scripts_have_shebang(self): - for script in SCRIPTS_DIR.glob("*.sh"): - first_line = script.read_text().splitlines()[0] - assert first_line.startswith("#!/"), ( - f"{script.name} missing shebang line" - ) - - def test_all_scripts_have_set_euo(self): - for script in SCRIPTS_DIR.glob("*.sh"): - content = script.read_text() - assert "set -euo pipefail" in content, ( - f"{script.name} missing 'set -euo pipefail'" - ) - - def test_all_scripts_have_usage(self): - for script in SCRIPTS_DIR.glob("*.sh"): - content = script.read_text() - assert "usage()" in content, ( - f"{script.name} missing usage() function" - ) - - -# --------------------------------------------------------------------------- -# spawn-evosci.sh specifics -# --------------------------------------------------------------------------- - -class TestSpawnScript: - """spawn-evosci.sh has required safety features.""" - - @pytest.fixture(autouse=True) - def _load(self): - self.content = (SCRIPTS_DIR / "spawn-evosci.sh").read_text() - - def test_checks_recursion_depth(self): - assert "EVOSCI_TMUX_DEPTH" in self.content - - def test_checks_max_depth(self): - assert "EVOSCI_MAX_TMUX_DEPTH" in self.content - - def test_checks_session_limit(self): - assert "EVOSCI_MAX_TMUX_SESSIONS" in self.content - - def test_checks_duplicate_session(self): - assert "has-session" in self.content - - def test_creates_workspace(self): - assert "mkdir -p" in self.content - - def test_updates_registry(self): - assert "registry.json" in self.content - - def test_uses_workdir_flag(self): - assert "--workdir" in self.content - - def test_validates_name(self): - # Should validate name format - assert "alphanumeric" in self.content.lower() or re.search( - r'\[a-zA-Z0-9\]', self.content - ) - - def test_supports_thread_id(self): - assert "--thread-id" in self.content - - def test_supports_no_thinking(self): - assert "--no-thinking" in self.content