refactor(config): remove max_concurrent and max_iterations from config and onboarding steps
feat(prompts): update system prompt to eliminate numeric limits for sub-agents and delegation rounds test(tests): adjust tests to reflect changes in configuration and onboarding logic
This commit is contained in:
@@ -88,11 +88,7 @@ def _ensure_system_prompt():
|
||||
"""Return cached system prompt, creating it on first call."""
|
||||
global _system_prompt
|
||||
if _system_prompt is None:
|
||||
cfg = _ensure_config()
|
||||
_system_prompt = get_system_prompt(
|
||||
max_concurrent=cfg.max_concurrent,
|
||||
max_iterations=cfg.max_iterations,
|
||||
)
|
||||
_system_prompt = get_system_prompt()
|
||||
return _system_prompt
|
||||
|
||||
|
||||
|
||||
@@ -90,7 +90,7 @@ def _checkbox_ask(choices, message: str, **kwargs):
|
||||
finally:
|
||||
InquirerControl._get_choice_tokens = original
|
||||
|
||||
STEPS = ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Parameters", "Skills", "MCP Servers", "Channels"]
|
||||
STEPS = ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Thinking", "Skills", "MCP Servers", "Channels"]
|
||||
|
||||
|
||||
# =============================================================================
|
||||
@@ -804,44 +804,15 @@ def _step_workspace(config: EvoScientistConfig) -> tuple[str, str]:
|
||||
return mode, workdir
|
||||
|
||||
|
||||
def _step_parameters(config: EvoScientistConfig) -> tuple[int, int, bool]:
|
||||
"""Step 6: Configure agent parameters.
|
||||
def _step_thinking(config: EvoScientistConfig) -> bool:
|
||||
"""Step 6: Configure thinking panel visibility.
|
||||
|
||||
Args:
|
||||
config: Current configuration.
|
||||
|
||||
Returns:
|
||||
Tuple of (max_concurrent, max_iterations, show_thinking).
|
||||
Whether to show thinking panels in CLI.
|
||||
"""
|
||||
# Max concurrent
|
||||
max_concurrent_str = questionary.text(
|
||||
"Max concurrent sub-agents (1-10):",
|
||||
default=str(config.max_concurrent),
|
||||
style=WIZARD_STYLE,
|
||||
qmark=QMARK,
|
||||
validate=lambda x: x.strip() == "" or (x.strip().isdigit() and 1 <= int(x.strip()) <= 10),
|
||||
).ask()
|
||||
|
||||
if max_concurrent_str is None:
|
||||
raise KeyboardInterrupt()
|
||||
|
||||
max_concurrent = int(max_concurrent_str.strip()) if max_concurrent_str.strip() else config.max_concurrent
|
||||
|
||||
# Max iterations
|
||||
max_iterations_str = questionary.text(
|
||||
"Max delegation iterations (1-10):",
|
||||
default=str(config.max_iterations),
|
||||
style=WIZARD_STYLE,
|
||||
qmark=QMARK,
|
||||
validate=lambda x: x.strip() == "" or (x.strip().isdigit() and 1 <= int(x.strip()) <= 10),
|
||||
).ask()
|
||||
|
||||
if max_iterations_str is None:
|
||||
raise KeyboardInterrupt()
|
||||
|
||||
max_iterations = int(max_iterations_str.strip()) if max_iterations_str.strip() else config.max_iterations
|
||||
|
||||
# Show thinking
|
||||
thinking_choices = [
|
||||
Choice(title="On (show model reasoning)", value=True),
|
||||
Choice(title="Off (hide model reasoning)", value=False),
|
||||
@@ -859,7 +830,7 @@ def _step_parameters(config: EvoScientistConfig) -> tuple[int, int, bool]:
|
||||
if show_thinking is None:
|
||||
raise KeyboardInterrupt()
|
||||
|
||||
return max_concurrent, max_iterations, show_thinking
|
||||
return show_thinking
|
||||
|
||||
|
||||
_RECOMMENDED_SKILLS = [
|
||||
@@ -1828,10 +1799,8 @@ def run_onboard(skip_validation: bool = False) -> bool:
|
||||
config.default_mode = mode
|
||||
config.default_workdir = workdir
|
||||
|
||||
# Step 6: Parameters
|
||||
max_concurrent, max_iterations, show_thinking = _step_parameters(config)
|
||||
config.max_concurrent = max_concurrent
|
||||
config.max_iterations = max_iterations
|
||||
# Step 6: Thinking
|
||||
show_thinking = _step_thinking(config)
|
||||
config.show_thinking = show_thinking
|
||||
|
||||
# Step 7: Skills
|
||||
|
||||
@@ -53,8 +53,6 @@ class EvoScientistConfig:
|
||||
model: Default model name (short name or full ID).
|
||||
default_mode: Default workspace mode ('daemon' or 'run').
|
||||
default_workdir: Default workspace directory (empty = use current working directory).
|
||||
max_concurrent: Maximum concurrent sub-agents.
|
||||
max_iterations: Maximum delegation iterations.
|
||||
show_thinking: Whether to show thinking panels in CLI.
|
||||
"""
|
||||
|
||||
@@ -78,10 +76,6 @@ class EvoScientistConfig:
|
||||
default_mode: Literal["daemon", "run"] = "daemon"
|
||||
default_workdir: str = ""
|
||||
|
||||
# Agent Parameters
|
||||
max_concurrent: int = 3
|
||||
max_iterations: int = 3
|
||||
|
||||
# UI Settings
|
||||
show_thinking: bool = True
|
||||
ui_backend: Literal["rich", "textual"] = "rich"
|
||||
|
||||
+40
-24
@@ -172,6 +172,12 @@ This prevents blocking the conversation during long operations.
|
||||
|
||||
DELEGATION_STRATEGY = """# Sub-Agent Delegation
|
||||
|
||||
## Mindset
|
||||
Treat every experiment as a submission draft. Each claim requires sufficient
|
||||
evidence: reproducible numbers, controlled comparisons, and identified failure
|
||||
modes. Iterate until a critical reviewer would accept the results — not for a
|
||||
fixed number of rounds.
|
||||
|
||||
## Default: Use 1 Sub-Agent
|
||||
For most tasks, a single sub-agent is sufficient:
|
||||
- "Plan experimental stages" → planner-agent
|
||||
@@ -187,24 +193,42 @@ For most tasks, a single sub-agent is sufficient:
|
||||
- Provide concrete file paths, commands, and success signals in each task
|
||||
so the sub-agent can respond precisely
|
||||
|
||||
## Parallelize Only When Necessary
|
||||
Use multiple sub-agents ONLY for:
|
||||
## When to Parallelize
|
||||
Launch multiple sub-agents only when experiments are independent:
|
||||
|
||||
**Explicit comparisons** (1 per method/baseline):
|
||||
- "Compare A vs B vs C" → 3 parallel sub-agents
|
||||
**Parallel** (no dependency between results):
|
||||
- Comparing Method A vs B vs C on the same data → one agent per method
|
||||
- Running the same method on Dataset X, Y, Z → one agent per dataset
|
||||
- Literature search while implementing a baseline → two agents
|
||||
|
||||
**Distinct experiments** with separate datasets or setups:
|
||||
- "Run baselines on X and Y" → 2 parallel sub-agents
|
||||
**Sequential** (each step depends on the previous):
|
||||
- Hyperparameter tuning — each round uses the previous result
|
||||
- Debug → fix → re-run — must observe the outcome before proceeding
|
||||
- Ablation design — requires knowing which components matter first
|
||||
|
||||
## Limits
|
||||
- Maximum {max_concurrent} parallel sub-agents per round
|
||||
- Maximum {max_iterations} delegation rounds total
|
||||
- Stop when evidence is sufficient
|
||||
## When to Stop Iterating
|
||||
After each stage, ask: "Would a critical reviewer accept this evidence?"
|
||||
|
||||
**Stop** when ALL of the following hold:
|
||||
- A baseline is established and documented
|
||||
- The primary metric is consistent across runs (≥3 seeds or folds, with
|
||||
confidence intervals or error bars)
|
||||
- Ablations confirm each key component's contribution
|
||||
- Results are compared against relevant baselines from the literature
|
||||
- Failure cases and limitations are identified and documented
|
||||
- All success signals defined in the plan are satisfied
|
||||
|
||||
**Keep iterating** if ANY of the following is true:
|
||||
- Results vary widely across runs (high variance, no uncertainty estimate)
|
||||
- A necessary comparison or ablation is missing
|
||||
- The method fails on straightforward cases without explanation
|
||||
- A reviewer would reasonably ask "did you try X?" and X is feasible
|
||||
|
||||
## Key Principles
|
||||
- Bias towards a single sub-agent (token-efficient)
|
||||
- Avoid premature decomposition
|
||||
- Each sub-agent returns focused, self-contained findings
|
||||
- Bias towards a single sub-agent — add concurrency only when the workload
|
||||
is genuinely independent
|
||||
- Avoid premature decomposition — one focused task per sub-agent
|
||||
- Each sub-agent returns self-contained findings with concrete artifacts
|
||||
"""
|
||||
|
||||
# =============================================================================
|
||||
@@ -264,18 +288,10 @@ Finding one with context [1]. Another insight [2].
|
||||
# Combined exports
|
||||
# =============================================================================
|
||||
|
||||
def get_system_prompt(max_concurrent: int = 3, max_iterations: int = 3) -> str:
|
||||
"""Generate the complete system prompt with configured limits.
|
||||
|
||||
Args:
|
||||
max_concurrent: Maximum number of concurrent sub-agents.
|
||||
max_iterations: Maximum number of delegation rounds.
|
||||
def get_system_prompt() -> str:
|
||||
"""Generate the complete system prompt.
|
||||
|
||||
Returns:
|
||||
Combined system prompt string.
|
||||
"""
|
||||
delegation = DELEGATION_STRATEGY.format(
|
||||
max_concurrent=max_concurrent,
|
||||
max_iterations=max_iterations,
|
||||
)
|
||||
return EXPERIMENT_WORKFLOW + "\n" + delegation
|
||||
return EXPERIMENT_WORKFLOW + "\n" + DELEGATION_STRATEGY
|
||||
|
||||
+1
-15
@@ -75,8 +75,6 @@ class TestEvoScientistConfig:
|
||||
assert config.model == "claude-sonnet-4-5"
|
||||
assert config.default_mode == "daemon"
|
||||
assert config.default_workdir == ""
|
||||
assert config.max_concurrent == 3
|
||||
assert config.max_iterations == 3
|
||||
assert config.show_thinking is True
|
||||
assert config.ui_backend == "rich"
|
||||
assert config.ollama_base_url == ""
|
||||
@@ -90,14 +88,12 @@ class TestEvoScientistConfig:
|
||||
provider="openai",
|
||||
model="gpt-4o",
|
||||
default_mode="run",
|
||||
max_concurrent=5,
|
||||
)
|
||||
|
||||
assert config.anthropic_api_key == "sk-ant-test"
|
||||
assert config.provider == "openai"
|
||||
assert config.model == "gpt-4o"
|
||||
assert config.default_mode == "run"
|
||||
assert config.max_concurrent == 5
|
||||
|
||||
|
||||
# =============================================================================
|
||||
@@ -154,14 +150,12 @@ class TestLoadSaveReset:
|
||||
original = EvoScientistConfig(
|
||||
anthropic_api_key="test-key",
|
||||
provider="openai",
|
||||
max_concurrent=7,
|
||||
)
|
||||
save_config(original)
|
||||
|
||||
loaded = load_config()
|
||||
assert loaded.anthropic_api_key == "test-key"
|
||||
assert loaded.provider == "openai"
|
||||
assert loaded.max_concurrent == 7
|
||||
|
||||
def test_reset_deletes_config_file(self, temp_config_dir, clean_env):
|
||||
"""Test that reset deletes the config file."""
|
||||
@@ -240,13 +234,6 @@ class TestGetSetValues:
|
||||
set_config_value("show_thinking", "yes")
|
||||
assert get_config_value("show_thinking") is True
|
||||
|
||||
def test_set_config_value_type_coercion_int(self, temp_config_dir, clean_env):
|
||||
"""Test that integer values are coerced correctly."""
|
||||
save_config(EvoScientistConfig())
|
||||
|
||||
set_config_value("max_concurrent", "5")
|
||||
assert get_config_value("max_concurrent") == 5
|
||||
|
||||
def test_set_imessage_enabled_coercion(self, temp_config_dir, clean_env):
|
||||
"""Test that imessage_enabled is coerced from string to bool."""
|
||||
save_config(EvoScientistConfig())
|
||||
@@ -299,11 +286,10 @@ class TestPriorityChain:
|
||||
|
||||
def test_file_overrides_defaults(self, temp_config_dir, clean_env):
|
||||
"""Test file config overrides defaults."""
|
||||
save_config(EvoScientistConfig(provider="openai", max_concurrent=7))
|
||||
save_config(EvoScientistConfig(provider="openai"))
|
||||
|
||||
config = get_effective_config()
|
||||
assert config.provider == "openai"
|
||||
assert config.max_concurrent == 7
|
||||
|
||||
def test_env_overrides_file(self, temp_config_dir, monkeypatch):
|
||||
"""Test environment variables override file config."""
|
||||
|
||||
+15
-21
@@ -25,7 +25,7 @@ class TestConstants:
|
||||
def test_steps_has_ten_items(self):
|
||||
"""Test that STEPS contains exactly 10 steps."""
|
||||
assert len(STEPS) == 10
|
||||
assert STEPS == ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Parameters", "Skills", "MCP Servers", "Channels"]
|
||||
assert STEPS == ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Thinking", "Skills", "MCP Servers", "Channels"]
|
||||
|
||||
def test_wizard_style_is_style_instance(self):
|
||||
"""Test that WIZARD_STYLE is a prompt_toolkit Style."""
|
||||
@@ -702,32 +702,30 @@ class TestStepMcpServersNpxFailure:
|
||||
mock_add.assert_not_called()
|
||||
|
||||
|
||||
class TestStepParameters:
|
||||
def test_returns_parameters(self):
|
||||
"""Test parameters step returns all values."""
|
||||
from EvoScientist.config.onboard import _step_parameters
|
||||
class TestStepThinking:
|
||||
def test_returns_show_thinking(self):
|
||||
"""Test thinking step returns selected value."""
|
||||
from EvoScientist.config.onboard import _step_thinking
|
||||
|
||||
config = EvoScientistConfig()
|
||||
|
||||
with patch("EvoScientist.config.onboard.questionary") as mock_q:
|
||||
mock_q.text.return_value.ask.side_effect = ["5", "3"]
|
||||
mock_q.select.return_value.ask.return_value = True # show_thinking
|
||||
result = _step_parameters(config)
|
||||
mock_q.select.return_value.ask.return_value = True
|
||||
result = _step_thinking(config)
|
||||
|
||||
assert result == (5, 3, True)
|
||||
assert result is True
|
||||
|
||||
def test_uses_defaults_on_empty_input(self):
|
||||
"""Test that empty input uses defaults."""
|
||||
from EvoScientist.config.onboard import _step_parameters
|
||||
def test_returns_false_when_off(self):
|
||||
"""Test thinking step returns False when user selects Off."""
|
||||
from EvoScientist.config.onboard import _step_thinking
|
||||
|
||||
config = EvoScientistConfig(max_concurrent=2, max_iterations=4, show_thinking=False)
|
||||
config = EvoScientistConfig(show_thinking=False)
|
||||
|
||||
with patch("EvoScientist.config.onboard.questionary") as mock_q:
|
||||
mock_q.text.return_value.ask.side_effect = ["", ""] # Empty inputs
|
||||
mock_q.select.return_value.ask.return_value = False # show_thinking
|
||||
result = _step_parameters(config)
|
||||
mock_q.select.return_value.ask.return_value = False
|
||||
result = _step_thinking(config)
|
||||
|
||||
assert result == (2, 4, False)
|
||||
assert result is False
|
||||
|
||||
|
||||
# =============================================================================
|
||||
@@ -765,8 +763,6 @@ class TestRunOnboard:
|
||||
]
|
||||
mock_q.text.return_value.ask.side_effect = [
|
||||
"", # Workspace directory (empty = use cwd)
|
||||
"3", # Max concurrent
|
||||
"3", # Max iterations
|
||||
]
|
||||
mock_q.checkbox.return_value.ask.return_value = [] # Skills: skip
|
||||
|
||||
@@ -815,8 +811,6 @@ class TestRunOnboard:
|
||||
]
|
||||
mock_q.text.return_value.ask.side_effect = [
|
||||
"", # Workspace directory (empty = use cwd)
|
||||
"3", # Max concurrent
|
||||
"3", # Max iterations
|
||||
]
|
||||
mock_q.checkbox.return_value.ask.return_value = [] # Skills: skip
|
||||
|
||||
|
||||
+6
-11
@@ -17,19 +17,14 @@ class TestGetSystemPrompt:
|
||||
result = get_system_prompt()
|
||||
assert "Sub-Agent Delegation" in result
|
||||
|
||||
def test_default_params_interpolated(self):
|
||||
def test_no_numeric_limits(self):
|
||||
result = get_system_prompt()
|
||||
assert "3 parallel sub-agents" in result
|
||||
assert "3 delegation rounds" in result
|
||||
|
||||
def test_custom_params(self):
|
||||
result = get_system_prompt(max_concurrent=5, max_iterations=10)
|
||||
assert "5 parallel sub-agents" in result
|
||||
assert "10 delegation rounds" in result
|
||||
assert "{max_concurrent}" not in result
|
||||
assert "{max_iterations}" not in result
|
||||
|
||||
def test_workflow_constant_not_empty(self):
|
||||
assert len(EXPERIMENT_WORKFLOW) > 0
|
||||
|
||||
def test_delegation_has_placeholders(self):
|
||||
assert "{max_concurrent}" in DELEGATION_STRATEGY
|
||||
assert "{max_iterations}" in DELEGATION_STRATEGY
|
||||
def test_delegation_no_placeholders(self):
|
||||
assert "{max_concurrent}" not in DELEGATION_STRATEGY
|
||||
assert "{max_iterations}" not in DELEGATION_STRATEGY
|
||||
|
||||
Reference in New Issue
Block a user