feat: enhance timeout handling with recovery guidance and resource estimation in middleware
This commit is contained in:
@@ -403,4 +403,23 @@ class CustomSandboxBackend(LocalShellBackend):
|
||||
)
|
||||
|
||||
# Delegate to parent for subprocess execution
|
||||
return super().execute(command, timeout=timeout)
|
||||
response = super().execute(command, timeout=timeout)
|
||||
|
||||
# Enhance timeout errors with actionable recovery guidance
|
||||
if response.exit_code == 124:
|
||||
cmd_words = command.split()
|
||||
grep_hint = cmd_words[0] if cmd_words else "process"
|
||||
bg_cmd = f"{command} > /output.log 2>&1 &"
|
||||
response = ExecuteResponse(
|
||||
output=(
|
||||
f"{response.output}\n\n"
|
||||
f"Recovery: re-run in background to avoid the sandbox timeout:\n"
|
||||
f" {bg_cmd}\n"
|
||||
f"Then check progress: ps aux | grep {grep_hint}\n"
|
||||
f"Read results: cat /output.log"
|
||||
),
|
||||
exit_code=response.exit_code,
|
||||
truncated=response.truncated,
|
||||
)
|
||||
|
||||
return response
|
||||
|
||||
@@ -152,7 +152,8 @@ Each question can be:
|
||||
- "multiple_choice": User selects from predefined options (an "Other" option is always appended)
|
||||
|
||||
Use when: dataset/model/framework selection is ambiguous, experiment parameters unclear,
|
||||
research scope needs clarification, or significant plan needs confirmation.
|
||||
research scope needs clarification, significant plan needs confirmation,
|
||||
resource estimation before heavy compute, or execution failures need user-guided recovery.
|
||||
Do NOT use for trivial decisions or questions answerable from context/memory/web search."""
|
||||
|
||||
|
||||
@@ -172,6 +173,16 @@ or available tools.
|
||||
- **Ambiguous instructions**: When the user's request has multiple valid interpretations
|
||||
- **Resource constraints**: When the approach depends on available compute, time, or data
|
||||
|
||||
### Resource & execution awareness (`ask_user` is especially valuable here):
|
||||
- **Pre-execution estimation**: Before heavy compute (training, large-scale eval),
|
||||
estimate time/memory/cost and confirm. E.g. "Training needs ~2h and ~16GB GPU.
|
||||
Proceed, or reduce model size?"
|
||||
- **Timeout & failure recovery**: When a command times out (exit code 124) or fails
|
||||
with OOM/CUDA errors, present recovery options. E.g. "Training timed out. Options:
|
||||
(A) run in background, (B) reduce epochs, (C) switch to smaller model"
|
||||
- **Intermediate checkpoints**: When results diverge from expectations, ask before
|
||||
continuing. E.g. "Baseline accuracy 62% vs expected 80%. Investigate or proceed?"
|
||||
|
||||
### When NOT to use `ask_user`:
|
||||
- Simple yes/no decisions — proceed with your best judgment
|
||||
- Information already provided in the conversation or memory
|
||||
|
||||
@@ -151,7 +151,7 @@ Use this to personalize your responses and avoid re-asking known information.
|
||||
- An experiment completes with notable conclusions
|
||||
|
||||
**How to update memory:**
|
||||
- If /memory/MEMORY.md does not exist yet, use `write_file` to create it
|
||||
- If `/memory/MEMORY.md` does not exist yet, use `write_file` to create it
|
||||
- If it already exists, use `edit_file` to update specific sections
|
||||
- Use this markdown structure:
|
||||
|
||||
@@ -657,7 +657,7 @@ class EvoMemoryMiddleware(AgentMiddleware):
|
||||
logger.debug("Failed to load memory during modify_request: %s", e)
|
||||
# Use placeholder when memory file doesn't exist yet
|
||||
if not memory_content:
|
||||
memory_content = "(No memory saved yet. Create /memory/MEMORY.md when you learn important information.)"
|
||||
memory_content = "(No memory saved yet. Create `/memory/MEMORY.md` when you learn important information.)"
|
||||
|
||||
from deepagents.middleware._utils import append_to_system_message
|
||||
injection = MEMORY_INJECTION_TEMPLATE.format(memory_content=memory_content)
|
||||
|
||||
+19
-8
@@ -14,7 +14,7 @@ into reproducible experiments and a paper-ready experimental report.
|
||||
- Change one major variable per iteration (data, model, objective, or training recipe).
|
||||
- Never invent results. If you cannot run something, say so and propose the smallest next step.
|
||||
- Delegate aggressively using the `task` tool. Prefer the research sub-agent for web search.
|
||||
- Use local skills when they match the task. Your available skills are listed in the system prompt — read the relevant SKILL.md for full instructions.
|
||||
- Use local skills when they match the task. Your available skills are listed in the system prompt — read the relevant `SKILL.md` for full instructions.
|
||||
All skills are available under `/skills/` (read-only).
|
||||
|
||||
## Research Lifecycle (when applicable)
|
||||
@@ -27,7 +27,7 @@ For end-to-end research projects, the recommended skill sequence is:
|
||||
6. `paper-review` — Self-review across quality dimensions
|
||||
7. `paper-rebuttal` — Respond to reviewer comments (if applicable)
|
||||
Not every project needs all steps. Match the starting point to what the user already has.
|
||||
Read the appropriate skill's SKILL.md for workflow guidance at each phase.
|
||||
Read the appropriate skill's `SKILL.md` for workflow guidance at each phase.
|
||||
|
||||
## Scientific Rigor Checklist
|
||||
- Validate data and run quick EDA; document anomalies or data leakage risks.
|
||||
@@ -50,7 +50,7 @@ Read the appropriate skill's SKILL.md for workflow guidance at each phase.
|
||||
- Identify resource/data dependencies and baseline requirements
|
||||
- Use `write_todos` to track the execution plan and updates
|
||||
- If delegating planning to planner-agent, start your message with: `MODE: PLAN`
|
||||
- If a stage matches an existing skill, note the skill name in the plan and read its SKILL.md before implementation.
|
||||
- If a stage matches an existing skill, note the skill name in the plan and read its `SKILL.md` before implementation.
|
||||
-- Save the plan to `/todos.md` (recommended). Include per-stage:
|
||||
- objective and success signals
|
||||
- what to run (commands/scripts)
|
||||
@@ -69,7 +69,7 @@ Read the appropriate skill's SKILL.md for workflow guidance at each phase.
|
||||
- Report drafting → writing-agent
|
||||
- Prefer the research-agent for web search; avoid searching directly
|
||||
- Use `execute` for shell commands when running experiments
|
||||
- When a task matches an existing skill, read its SKILL.md and follow it rather than reinventing the workflow.
|
||||
- When a task matches an existing skill, read its `SKILL.md` and follow it rather than reinventing the workflow.
|
||||
- Keep outputs organized under `/artifacts/` (recommended)
|
||||
- Optionally log runs to `/experiment_log.md` (params, seeds, env, outputs)
|
||||
|
||||
@@ -175,6 +175,9 @@ Then revise `/todos.md` accordingly.
|
||||
## Shell Execution Guidelines
|
||||
When using the `execute` tool for shell commands:
|
||||
|
||||
**Sandbox limits**: Commands time out after 300 seconds (exit code 124) and output is
|
||||
truncated at 100 KB. Plan accordingly.
|
||||
|
||||
**Short commands** (< 30 seconds): Run directly
|
||||
```bash
|
||||
python script.py
|
||||
@@ -193,6 +196,14 @@ ps aux | grep long_task
|
||||
cat /output.log
|
||||
```
|
||||
|
||||
**Before heavy compute**: Estimate runtime. If likely > 5 minutes, use background
|
||||
execution from the start. If GPU memory is uncertain, start with a small test run
|
||||
(1 epoch, small batch) before the full run.
|
||||
|
||||
**After a timeout (exit code 124)**: Do NOT re-run the same command. Instead:
|
||||
1. Re-launch in background with output logging
|
||||
2. Or reduce the workload (fewer epochs, smaller model, subset of data)
|
||||
|
||||
This prevents blocking the conversation during long operations.
|
||||
"""
|
||||
|
||||
@@ -275,11 +286,11 @@ Do not fabricate citations or URLs.
|
||||
Capture evaluation protocols (splits, metrics, calibration) and known failure modes.
|
||||
|
||||
## Available Tools
|
||||
1. **tavily_search** - Web search for information
|
||||
2. **think_tool** - Reflect on findings and plan next steps
|
||||
3. **read_file** - Read skill instructions when a skill matches the task (paths shown in your available skills listing)
|
||||
1. `tavily_search` — Web search for information
|
||||
2. `think_tool` — Reflect on findings and plan next steps
|
||||
3. `read_file` — Read skill instructions when a skill matches the task (paths shown in your available skills listing)
|
||||
|
||||
**CRITICAL: Use think_tool after each search**
|
||||
**CRITICAL:** Use `think_tool` after each search
|
||||
|
||||
## Research Strategy
|
||||
1. Read the question carefully
|
||||
|
||||
@@ -25,7 +25,7 @@ def think_tool(reflection: str) -> str:
|
||||
Would a critical reviewer accept it, or are there gaps to fill?
|
||||
3. Skills leverage — Is there an installed skill that provides a structured
|
||||
workflow for what I'm doing? Check your available skills listing and read
|
||||
the relevant SKILL.md for full instructions. Skills cover various research
|
||||
the relevant `SKILL.md` for full instructions. Skills cover various research
|
||||
phases — ideation, experiment execution, paper writing, review, and more.
|
||||
Follow a skill's workflow rather than improvising when one is available.
|
||||
4. Prior knowledge — Have I checked research memory before starting?
|
||||
@@ -38,10 +38,16 @@ def think_tool(reflection: str) -> str:
|
||||
something different? What evidence supports this decision?
|
||||
6. Handoff — Is this phase complete? What artifacts and results does the
|
||||
next phase or the caller need? Am I leaving clear, well-organized outputs?
|
||||
7. Resource & compute — Before heavy operations (training, large evals),
|
||||
estimate runtime and memory. The sandbox has a 300s execution timeout
|
||||
and 100KB output limit. For tasks likely exceeding these, plan background
|
||||
execution with log files. After a timeout or OOM, reflect on whether to
|
||||
retry with reduced parameters (smaller model, fewer epochs, data subset)
|
||||
or switch to background execution.
|
||||
|
||||
Not every reflection needs all six dimensions. Pick the ones relevant to
|
||||
Not every reflection needs all seven dimensions. Pick the ones relevant to
|
||||
the current moment. A focused two or three dimension reflection is better
|
||||
than a shallow pass over all six.
|
||||
than a shallow pass over all seven.
|
||||
|
||||
Args:
|
||||
reflection: Your structured reflection addressing the relevant dimensions above
|
||||
|
||||
@@ -279,6 +279,21 @@ class TestAskUserMiddleware:
|
||||
mw = AskUserMiddleware(system_prompt="custom prompt")
|
||||
assert mw.system_prompt == "custom prompt"
|
||||
|
||||
def test_system_prompt_mentions_resource_estimation(self):
|
||||
from EvoScientist.middleware.ask_user import ASK_USER_SYSTEM_PROMPT
|
||||
|
||||
assert "estimation" in ASK_USER_SYSTEM_PROMPT.lower()
|
||||
|
||||
def test_system_prompt_mentions_timeout(self):
|
||||
from EvoScientist.middleware.ask_user import ASK_USER_SYSTEM_PROMPT
|
||||
|
||||
assert "timeout" in ASK_USER_SYSTEM_PROMPT.lower()
|
||||
|
||||
def test_tool_description_mentions_resource(self):
|
||||
from EvoScientist.middleware.ask_user import ASK_USER_TOOL_DESCRIPTION
|
||||
|
||||
assert "resource" in ASK_USER_TOOL_DESCRIPTION.lower()
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stream event emitter
|
||||
|
||||
@@ -367,3 +367,31 @@ class TestPipelineCommandValidation:
|
||||
def test_quoted_pipe_not_split(self):
|
||||
"""Pipe inside quotes is not a shell operator."""
|
||||
assert validate_command("echo 'hello | world'") is None
|
||||
|
||||
|
||||
# === execute() timeout recovery guidance ===
|
||||
|
||||
class TestExecuteTimeoutRecovery:
|
||||
def test_timeout_includes_recovery_guidance(self, tmp_workspace):
|
||||
backend = CustomSandboxBackend(root_dir=tmp_workspace, timeout=1)
|
||||
resp = backend.execute("sleep 10")
|
||||
assert resp.exit_code == 124
|
||||
assert "Recovery" in resp.output
|
||||
assert "background" in resp.output.lower()
|
||||
|
||||
def test_timeout_includes_background_command(self, tmp_workspace):
|
||||
backend = CustomSandboxBackend(root_dir=tmp_workspace, timeout=1)
|
||||
resp = backend.execute("sleep 10")
|
||||
assert "sleep 10" in resp.output
|
||||
assert "> /output.log 2>&1 &" in resp.output
|
||||
|
||||
def test_timeout_preserves_original_error(self, tmp_workspace):
|
||||
backend = CustomSandboxBackend(root_dir=tmp_workspace, timeout=1)
|
||||
resp = backend.execute("sleep 10")
|
||||
assert "timed out" in resp.output.lower()
|
||||
|
||||
def test_non_timeout_not_enhanced(self, tmp_workspace):
|
||||
backend = CustomSandboxBackend(root_dir=tmp_workspace)
|
||||
resp = backend.execute("python3 -c 'raise SystemExit(1)'")
|
||||
assert resp.exit_code == 1
|
||||
assert "Recovery" not in resp.output
|
||||
|
||||
@@ -28,3 +28,10 @@ class TestGetSystemPrompt:
|
||||
def test_delegation_no_placeholders(self):
|
||||
assert "{max_concurrent}" not in DELEGATION_STRATEGY
|
||||
assert "{max_iterations}" not in DELEGATION_STRATEGY
|
||||
|
||||
def test_shell_guidelines_mention_timeout_limit(self):
|
||||
assert "300" in EXPERIMENT_WORKFLOW
|
||||
assert "124" in EXPERIMENT_WORKFLOW
|
||||
|
||||
def test_shell_guidelines_mention_background(self):
|
||||
assert "background" in EXPERIMENT_WORKFLOW.lower()
|
||||
|
||||
@@ -16,3 +16,7 @@ class TestThinkTool:
|
||||
def test_empty_reflection(self):
|
||||
result = think_tool.invoke({"reflection": ""})
|
||||
assert "Reflection recorded" in result
|
||||
|
||||
def test_docstring_has_resource_dimension(self):
|
||||
assert "resource" in think_tool.description.lower()
|
||||
assert "300" in think_tool.description
|
||||
|
||||
Reference in New Issue
Block a user