feat: enhance timeout handling with recovery guidance and resource estimation in middleware

This commit is contained in:
X-iZhang
2026-03-11 00:03:18 +00:00
parent 1625caa194
commit 5ff000a9b2
9 changed files with 116 additions and 15 deletions
+20 -1
View File
@@ -403,4 +403,23 @@ class CustomSandboxBackend(LocalShellBackend):
)
# Delegate to parent for subprocess execution
return super().execute(command, timeout=timeout)
response = super().execute(command, timeout=timeout)
# Enhance timeout errors with actionable recovery guidance
if response.exit_code == 124:
cmd_words = command.split()
grep_hint = cmd_words[0] if cmd_words else "process"
bg_cmd = f"{command} > /output.log 2>&1 &"
response = ExecuteResponse(
output=(
f"{response.output}\n\n"
f"Recovery: re-run in background to avoid the sandbox timeout:\n"
f" {bg_cmd}\n"
f"Then check progress: ps aux | grep {grep_hint}\n"
f"Read results: cat /output.log"
),
exit_code=response.exit_code,
truncated=response.truncated,
)
return response
+12 -1
View File
@@ -152,7 +152,8 @@ Each question can be:
- "multiple_choice": User selects from predefined options (an "Other" option is always appended)
Use when: dataset/model/framework selection is ambiguous, experiment parameters unclear,
research scope needs clarification, or significant plan needs confirmation.
research scope needs clarification, significant plan needs confirmation,
resource estimation before heavy compute, or execution failures need user-guided recovery.
Do NOT use for trivial decisions or questions answerable from context/memory/web search."""
@@ -172,6 +173,16 @@ or available tools.
- **Ambiguous instructions**: When the user's request has multiple valid interpretations
- **Resource constraints**: When the approach depends on available compute, time, or data
### Resource & execution awareness (`ask_user` is especially valuable here):
- **Pre-execution estimation**: Before heavy compute (training, large-scale eval),
estimate time/memory/cost and confirm. E.g. "Training needs ~2h and ~16GB GPU.
Proceed, or reduce model size?"
- **Timeout & failure recovery**: When a command times out (exit code 124) or fails
with OOM/CUDA errors, present recovery options. E.g. "Training timed out. Options:
(A) run in background, (B) reduce epochs, (C) switch to smaller model"
- **Intermediate checkpoints**: When results diverge from expectations, ask before
continuing. E.g. "Baseline accuracy 62% vs expected 80%. Investigate or proceed?"
### When NOT to use `ask_user`:
- Simple yes/no decisions — proceed with your best judgment
- Information already provided in the conversation or memory
+2 -2
View File
@@ -151,7 +151,7 @@ Use this to personalize your responses and avoid re-asking known information.
- An experiment completes with notable conclusions
**How to update memory:**
- If /memory/MEMORY.md does not exist yet, use `write_file` to create it
- If `/memory/MEMORY.md` does not exist yet, use `write_file` to create it
- If it already exists, use `edit_file` to update specific sections
- Use this markdown structure:
@@ -657,7 +657,7 @@ class EvoMemoryMiddleware(AgentMiddleware):
logger.debug("Failed to load memory during modify_request: %s", e)
# Use placeholder when memory file doesn't exist yet
if not memory_content:
memory_content = "(No memory saved yet. Create /memory/MEMORY.md when you learn important information.)"
memory_content = "(No memory saved yet. Create `/memory/MEMORY.md` when you learn important information.)"
from deepagents.middleware._utils import append_to_system_message
injection = MEMORY_INJECTION_TEMPLATE.format(memory_content=memory_content)
+19 -8
View File
@@ -14,7 +14,7 @@ into reproducible experiments and a paper-ready experimental report.
- Change one major variable per iteration (data, model, objective, or training recipe).
- Never invent results. If you cannot run something, say so and propose the smallest next step.
- Delegate aggressively using the `task` tool. Prefer the research sub-agent for web search.
- Use local skills when they match the task. Your available skills are listed in the system prompt — read the relevant SKILL.md for full instructions.
- Use local skills when they match the task. Your available skills are listed in the system prompt — read the relevant `SKILL.md` for full instructions.
All skills are available under `/skills/` (read-only).
## Research Lifecycle (when applicable)
@@ -27,7 +27,7 @@ For end-to-end research projects, the recommended skill sequence is:
6. `paper-review` — Self-review across quality dimensions
7. `paper-rebuttal` — Respond to reviewer comments (if applicable)
Not every project needs all steps. Match the starting point to what the user already has.
Read the appropriate skill's SKILL.md for workflow guidance at each phase.
Read the appropriate skill's `SKILL.md` for workflow guidance at each phase.
## Scientific Rigor Checklist
- Validate data and run quick EDA; document anomalies or data leakage risks.
@@ -50,7 +50,7 @@ Read the appropriate skill's SKILL.md for workflow guidance at each phase.
- Identify resource/data dependencies and baseline requirements
- Use `write_todos` to track the execution plan and updates
- If delegating planning to planner-agent, start your message with: `MODE: PLAN`
- If a stage matches an existing skill, note the skill name in the plan and read its SKILL.md before implementation.
- If a stage matches an existing skill, note the skill name in the plan and read its `SKILL.md` before implementation.
-- Save the plan to `/todos.md` (recommended). Include per-stage:
- objective and success signals
- what to run (commands/scripts)
@@ -69,7 +69,7 @@ Read the appropriate skill's SKILL.md for workflow guidance at each phase.
- Report drafting → writing-agent
- Prefer the research-agent for web search; avoid searching directly
- Use `execute` for shell commands when running experiments
- When a task matches an existing skill, read its SKILL.md and follow it rather than reinventing the workflow.
- When a task matches an existing skill, read its `SKILL.md` and follow it rather than reinventing the workflow.
- Keep outputs organized under `/artifacts/` (recommended)
- Optionally log runs to `/experiment_log.md` (params, seeds, env, outputs)
@@ -175,6 +175,9 @@ Then revise `/todos.md` accordingly.
## Shell Execution Guidelines
When using the `execute` tool for shell commands:
**Sandbox limits**: Commands time out after 300 seconds (exit code 124) and output is
truncated at 100 KB. Plan accordingly.
**Short commands** (< 30 seconds): Run directly
```bash
python script.py
@@ -193,6 +196,14 @@ ps aux | grep long_task
cat /output.log
```
**Before heavy compute**: Estimate runtime. If likely > 5 minutes, use background
execution from the start. If GPU memory is uncertain, start with a small test run
(1 epoch, small batch) before the full run.
**After a timeout (exit code 124)**: Do NOT re-run the same command. Instead:
1. Re-launch in background with output logging
2. Or reduce the workload (fewer epochs, smaller model, subset of data)
This prevents blocking the conversation during long operations.
"""
@@ -275,11 +286,11 @@ Do not fabricate citations or URLs.
Capture evaluation protocols (splits, metrics, calibration) and known failure modes.
## Available Tools
1. **tavily_search** - Web search for information
2. **think_tool** - Reflect on findings and plan next steps
3. **read_file** - Read skill instructions when a skill matches the task (paths shown in your available skills listing)
1. `tavily_search` — Web search for information
2. `think_tool` — Reflect on findings and plan next steps
3. `read_file` — Read skill instructions when a skill matches the task (paths shown in your available skills listing)
**CRITICAL: Use think_tool after each search**
**CRITICAL:** Use `think_tool` after each search
## Research Strategy
1. Read the question carefully
+9 -3
View File
@@ -25,7 +25,7 @@ def think_tool(reflection: str) -> str:
Would a critical reviewer accept it, or are there gaps to fill?
3. Skills leverage — Is there an installed skill that provides a structured
workflow for what I'm doing? Check your available skills listing and read
the relevant SKILL.md for full instructions. Skills cover various research
the relevant `SKILL.md` for full instructions. Skills cover various research
phases — ideation, experiment execution, paper writing, review, and more.
Follow a skill's workflow rather than improvising when one is available.
4. Prior knowledge — Have I checked research memory before starting?
@@ -38,10 +38,16 @@ def think_tool(reflection: str) -> str:
something different? What evidence supports this decision?
6. Handoff — Is this phase complete? What artifacts and results does the
next phase or the caller need? Am I leaving clear, well-organized outputs?
7. Resource & compute — Before heavy operations (training, large evals),
estimate runtime and memory. The sandbox has a 300s execution timeout
and 100KB output limit. For tasks likely exceeding these, plan background
execution with log files. After a timeout or OOM, reflect on whether to
retry with reduced parameters (smaller model, fewer epochs, data subset)
or switch to background execution.
Not every reflection needs all six dimensions. Pick the ones relevant to
Not every reflection needs all seven dimensions. Pick the ones relevant to
the current moment. A focused two or three dimension reflection is better
than a shallow pass over all six.
than a shallow pass over all seven.
Args:
reflection: Your structured reflection addressing the relevant dimensions above
+15
View File
@@ -279,6 +279,21 @@ class TestAskUserMiddleware:
mw = AskUserMiddleware(system_prompt="custom prompt")
assert mw.system_prompt == "custom prompt"
def test_system_prompt_mentions_resource_estimation(self):
from EvoScientist.middleware.ask_user import ASK_USER_SYSTEM_PROMPT
assert "estimation" in ASK_USER_SYSTEM_PROMPT.lower()
def test_system_prompt_mentions_timeout(self):
from EvoScientist.middleware.ask_user import ASK_USER_SYSTEM_PROMPT
assert "timeout" in ASK_USER_SYSTEM_PROMPT.lower()
def test_tool_description_mentions_resource(self):
from EvoScientist.middleware.ask_user import ASK_USER_TOOL_DESCRIPTION
assert "resource" in ASK_USER_TOOL_DESCRIPTION.lower()
# ---------------------------------------------------------------------------
# Stream event emitter
+28
View File
@@ -367,3 +367,31 @@ class TestPipelineCommandValidation:
def test_quoted_pipe_not_split(self):
"""Pipe inside quotes is not a shell operator."""
assert validate_command("echo 'hello | world'") is None
# === execute() timeout recovery guidance ===
class TestExecuteTimeoutRecovery:
def test_timeout_includes_recovery_guidance(self, tmp_workspace):
backend = CustomSandboxBackend(root_dir=tmp_workspace, timeout=1)
resp = backend.execute("sleep 10")
assert resp.exit_code == 124
assert "Recovery" in resp.output
assert "background" in resp.output.lower()
def test_timeout_includes_background_command(self, tmp_workspace):
backend = CustomSandboxBackend(root_dir=tmp_workspace, timeout=1)
resp = backend.execute("sleep 10")
assert "sleep 10" in resp.output
assert "> /output.log 2>&1 &" in resp.output
def test_timeout_preserves_original_error(self, tmp_workspace):
backend = CustomSandboxBackend(root_dir=tmp_workspace, timeout=1)
resp = backend.execute("sleep 10")
assert "timed out" in resp.output.lower()
def test_non_timeout_not_enhanced(self, tmp_workspace):
backend = CustomSandboxBackend(root_dir=tmp_workspace)
resp = backend.execute("python3 -c 'raise SystemExit(1)'")
assert resp.exit_code == 1
assert "Recovery" not in resp.output
+7
View File
@@ -28,3 +28,10 @@ class TestGetSystemPrompt:
def test_delegation_no_placeholders(self):
assert "{max_concurrent}" not in DELEGATION_STRATEGY
assert "{max_iterations}" not in DELEGATION_STRATEGY
def test_shell_guidelines_mention_timeout_limit(self):
assert "300" in EXPERIMENT_WORKFLOW
assert "124" in EXPERIMENT_WORKFLOW
def test_shell_guidelines_mention_background(self):
assert "background" in EXPERIMENT_WORKFLOW.lower()
+4
View File
@@ -16,3 +16,7 @@ class TestThinkTool:
def test_empty_reflection(self):
result = think_tool.invoke({"reflection": ""})
assert "Reflection recorded" in result
def test_docstring_has_resource_dimension(self):
assert "resource" in think_tool.description.lower()
assert "300" in think_tool.description