refactor(config): remove max_concurrent and max_iterations from config and onboarding steps

feat(prompts): update system prompt to eliminate numeric limits for sub-agents and delegation rounds
test(tests): adjust tests to reflect changes in configuration and onboarding logic
This commit is contained in:
X-iZhang
2026-02-27 18:57:47 +00:00
parent 1130ddd8e9
commit 2f6cf824f2
7 changed files with 70 additions and 120 deletions
+1 -5
View File
@@ -88,11 +88,7 @@ def _ensure_system_prompt():
"""Return cached system prompt, creating it on first call."""
global _system_prompt
if _system_prompt is None:
cfg = _ensure_config()
_system_prompt = get_system_prompt(
max_concurrent=cfg.max_concurrent,
max_iterations=cfg.max_iterations,
)
_system_prompt = get_system_prompt()
return _system_prompt
+7 -38
View File
@@ -90,7 +90,7 @@ def _checkbox_ask(choices, message: str, **kwargs):
finally:
InquirerControl._get_choice_tokens = original
STEPS = ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Parameters", "Skills", "MCP Servers", "Channels"]
STEPS = ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Thinking", "Skills", "MCP Servers", "Channels"]
# =============================================================================
@@ -804,44 +804,15 @@ def _step_workspace(config: EvoScientistConfig) -> tuple[str, str]:
return mode, workdir
def _step_parameters(config: EvoScientistConfig) -> tuple[int, int, bool]:
"""Step 6: Configure agent parameters.
def _step_thinking(config: EvoScientistConfig) -> bool:
"""Step 6: Configure thinking panel visibility.
Args:
config: Current configuration.
Returns:
Tuple of (max_concurrent, max_iterations, show_thinking).
Whether to show thinking panels in CLI.
"""
# Max concurrent
max_concurrent_str = questionary.text(
"Max concurrent sub-agents (1-10):",
default=str(config.max_concurrent),
style=WIZARD_STYLE,
qmark=QMARK,
validate=lambda x: x.strip() == "" or (x.strip().isdigit() and 1 <= int(x.strip()) <= 10),
).ask()
if max_concurrent_str is None:
raise KeyboardInterrupt()
max_concurrent = int(max_concurrent_str.strip()) if max_concurrent_str.strip() else config.max_concurrent
# Max iterations
max_iterations_str = questionary.text(
"Max delegation iterations (1-10):",
default=str(config.max_iterations),
style=WIZARD_STYLE,
qmark=QMARK,
validate=lambda x: x.strip() == "" or (x.strip().isdigit() and 1 <= int(x.strip()) <= 10),
).ask()
if max_iterations_str is None:
raise KeyboardInterrupt()
max_iterations = int(max_iterations_str.strip()) if max_iterations_str.strip() else config.max_iterations
# Show thinking
thinking_choices = [
Choice(title="On (show model reasoning)", value=True),
Choice(title="Off (hide model reasoning)", value=False),
@@ -859,7 +830,7 @@ def _step_parameters(config: EvoScientistConfig) -> tuple[int, int, bool]:
if show_thinking is None:
raise KeyboardInterrupt()
return max_concurrent, max_iterations, show_thinking
return show_thinking
_RECOMMENDED_SKILLS = [
@@ -1828,10 +1799,8 @@ def run_onboard(skip_validation: bool = False) -> bool:
config.default_mode = mode
config.default_workdir = workdir
# Step 6: Parameters
max_concurrent, max_iterations, show_thinking = _step_parameters(config)
config.max_concurrent = max_concurrent
config.max_iterations = max_iterations
# Step 6: Thinking
show_thinking = _step_thinking(config)
config.show_thinking = show_thinking
# Step 7: Skills
-6
View File
@@ -53,8 +53,6 @@ class EvoScientistConfig:
model: Default model name (short name or full ID).
default_mode: Default workspace mode ('daemon' or 'run').
default_workdir: Default workspace directory (empty = use current working directory).
max_concurrent: Maximum concurrent sub-agents.
max_iterations: Maximum delegation iterations.
show_thinking: Whether to show thinking panels in CLI.
"""
@@ -78,10 +76,6 @@ class EvoScientistConfig:
default_mode: Literal["daemon", "run"] = "daemon"
default_workdir: str = ""
# Agent Parameters
max_concurrent: int = 3
max_iterations: int = 3
# UI Settings
show_thinking: bool = True
ui_backend: Literal["rich", "textual"] = "rich"
+40 -24
View File
@@ -172,6 +172,12 @@ This prevents blocking the conversation during long operations.
DELEGATION_STRATEGY = """# Sub-Agent Delegation
## Mindset
Treat every experiment as a submission draft. Each claim requires sufficient
evidence: reproducible numbers, controlled comparisons, and identified failure
modes. Iterate until a critical reviewer would accept the results — not for a
fixed number of rounds.
## Default: Use 1 Sub-Agent
For most tasks, a single sub-agent is sufficient:
- "Plan experimental stages" → planner-agent
@@ -187,24 +193,42 @@ For most tasks, a single sub-agent is sufficient:
- Provide concrete file paths, commands, and success signals in each task
so the sub-agent can respond precisely
## Parallelize Only When Necessary
Use multiple sub-agents ONLY for:
## When to Parallelize
Launch multiple sub-agents only when experiments are independent:
**Explicit comparisons** (1 per method/baseline):
- "Compare A vs B vs C" → 3 parallel sub-agents
**Parallel** (no dependency between results):
- Comparing Method A vs B vs C on the same data → one agent per method
- Running the same method on Dataset X, Y, Z → one agent per dataset
- Literature search while implementing a baseline → two agents
**Distinct experiments** with separate datasets or setups:
- "Run baselines on X and Y" → 2 parallel sub-agents
**Sequential** (each step depends on the previous):
- Hyperparameter tuning — each round uses the previous result
- Debug → fix → re-run — must observe the outcome before proceeding
- Ablation design — requires knowing which components matter first
## Limits
- Maximum {max_concurrent} parallel sub-agents per round
- Maximum {max_iterations} delegation rounds total
- Stop when evidence is sufficient
## When to Stop Iterating
After each stage, ask: "Would a critical reviewer accept this evidence?"
**Stop** when ALL of the following hold:
- A baseline is established and documented
- The primary metric is consistent across runs (≥3 seeds or folds, with
confidence intervals or error bars)
- Ablations confirm each key component's contribution
- Results are compared against relevant baselines from the literature
- Failure cases and limitations are identified and documented
- All success signals defined in the plan are satisfied
**Keep iterating** if ANY of the following is true:
- Results vary widely across runs (high variance, no uncertainty estimate)
- A necessary comparison or ablation is missing
- The method fails on straightforward cases without explanation
- A reviewer would reasonably ask "did you try X?" and X is feasible
## Key Principles
- Bias towards a single sub-agent (token-efficient)
- Avoid premature decomposition
- Each sub-agent returns focused, self-contained findings
- Bias towards a single sub-agent — add concurrency only when the workload
is genuinely independent
- Avoid premature decomposition — one focused task per sub-agent
- Each sub-agent returns self-contained findings with concrete artifacts
"""
# =============================================================================
@@ -264,18 +288,10 @@ Finding one with context [1]. Another insight [2].
# Combined exports
# =============================================================================
def get_system_prompt(max_concurrent: int = 3, max_iterations: int = 3) -> str:
"""Generate the complete system prompt with configured limits.
Args:
max_concurrent: Maximum number of concurrent sub-agents.
max_iterations: Maximum number of delegation rounds.
def get_system_prompt() -> str:
"""Generate the complete system prompt.
Returns:
Combined system prompt string.
"""
delegation = DELEGATION_STRATEGY.format(
max_concurrent=max_concurrent,
max_iterations=max_iterations,
)
return EXPERIMENT_WORKFLOW + "\n" + delegation
return EXPERIMENT_WORKFLOW + "\n" + DELEGATION_STRATEGY
+1 -15
View File
@@ -75,8 +75,6 @@ class TestEvoScientistConfig:
assert config.model == "claude-sonnet-4-5"
assert config.default_mode == "daemon"
assert config.default_workdir == ""
assert config.max_concurrent == 3
assert config.max_iterations == 3
assert config.show_thinking is True
assert config.ui_backend == "rich"
assert config.ollama_base_url == ""
@@ -90,14 +88,12 @@ class TestEvoScientistConfig:
provider="openai",
model="gpt-4o",
default_mode="run",
max_concurrent=5,
)
assert config.anthropic_api_key == "sk-ant-test"
assert config.provider == "openai"
assert config.model == "gpt-4o"
assert config.default_mode == "run"
assert config.max_concurrent == 5
# =============================================================================
@@ -154,14 +150,12 @@ class TestLoadSaveReset:
original = EvoScientistConfig(
anthropic_api_key="test-key",
provider="openai",
max_concurrent=7,
)
save_config(original)
loaded = load_config()
assert loaded.anthropic_api_key == "test-key"
assert loaded.provider == "openai"
assert loaded.max_concurrent == 7
def test_reset_deletes_config_file(self, temp_config_dir, clean_env):
"""Test that reset deletes the config file."""
@@ -240,13 +234,6 @@ class TestGetSetValues:
set_config_value("show_thinking", "yes")
assert get_config_value("show_thinking") is True
def test_set_config_value_type_coercion_int(self, temp_config_dir, clean_env):
"""Test that integer values are coerced correctly."""
save_config(EvoScientistConfig())
set_config_value("max_concurrent", "5")
assert get_config_value("max_concurrent") == 5
def test_set_imessage_enabled_coercion(self, temp_config_dir, clean_env):
"""Test that imessage_enabled is coerced from string to bool."""
save_config(EvoScientistConfig())
@@ -299,11 +286,10 @@ class TestPriorityChain:
def test_file_overrides_defaults(self, temp_config_dir, clean_env):
"""Test file config overrides defaults."""
save_config(EvoScientistConfig(provider="openai", max_concurrent=7))
save_config(EvoScientistConfig(provider="openai"))
config = get_effective_config()
assert config.provider == "openai"
assert config.max_concurrent == 7
def test_env_overrides_file(self, temp_config_dir, monkeypatch):
"""Test environment variables override file config."""
+15 -21
View File
@@ -25,7 +25,7 @@ class TestConstants:
def test_steps_has_ten_items(self):
"""Test that STEPS contains exactly 10 steps."""
assert len(STEPS) == 10
assert STEPS == ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Parameters", "Skills", "MCP Servers", "Channels"]
assert STEPS == ["UI", "Provider", "API Key", "Model", "Tavily Key", "Workspace", "Thinking", "Skills", "MCP Servers", "Channels"]
def test_wizard_style_is_style_instance(self):
"""Test that WIZARD_STYLE is a prompt_toolkit Style."""
@@ -702,32 +702,30 @@ class TestStepMcpServersNpxFailure:
mock_add.assert_not_called()
class TestStepParameters:
def test_returns_parameters(self):
"""Test parameters step returns all values."""
from EvoScientist.config.onboard import _step_parameters
class TestStepThinking:
def test_returns_show_thinking(self):
"""Test thinking step returns selected value."""
from EvoScientist.config.onboard import _step_thinking
config = EvoScientistConfig()
with patch("EvoScientist.config.onboard.questionary") as mock_q:
mock_q.text.return_value.ask.side_effect = ["5", "3"]
mock_q.select.return_value.ask.return_value = True # show_thinking
result = _step_parameters(config)
mock_q.select.return_value.ask.return_value = True
result = _step_thinking(config)
assert result == (5, 3, True)
assert result is True
def test_uses_defaults_on_empty_input(self):
"""Test that empty input uses defaults."""
from EvoScientist.config.onboard import _step_parameters
def test_returns_false_when_off(self):
"""Test thinking step returns False when user selects Off."""
from EvoScientist.config.onboard import _step_thinking
config = EvoScientistConfig(max_concurrent=2, max_iterations=4, show_thinking=False)
config = EvoScientistConfig(show_thinking=False)
with patch("EvoScientist.config.onboard.questionary") as mock_q:
mock_q.text.return_value.ask.side_effect = ["", ""] # Empty inputs
mock_q.select.return_value.ask.return_value = False # show_thinking
result = _step_parameters(config)
mock_q.select.return_value.ask.return_value = False
result = _step_thinking(config)
assert result == (2, 4, False)
assert result is False
# =============================================================================
@@ -765,8 +763,6 @@ class TestRunOnboard:
]
mock_q.text.return_value.ask.side_effect = [
"", # Workspace directory (empty = use cwd)
"3", # Max concurrent
"3", # Max iterations
]
mock_q.checkbox.return_value.ask.return_value = [] # Skills: skip
@@ -815,8 +811,6 @@ class TestRunOnboard:
]
mock_q.text.return_value.ask.side_effect = [
"", # Workspace directory (empty = use cwd)
"3", # Max concurrent
"3", # Max iterations
]
mock_q.checkbox.return_value.ask.return_value = [] # Skills: skip
+6 -11
View File
@@ -17,19 +17,14 @@ class TestGetSystemPrompt:
result = get_system_prompt()
assert "Sub-Agent Delegation" in result
def test_default_params_interpolated(self):
def test_no_numeric_limits(self):
result = get_system_prompt()
assert "3 parallel sub-agents" in result
assert "3 delegation rounds" in result
def test_custom_params(self):
result = get_system_prompt(max_concurrent=5, max_iterations=10)
assert "5 parallel sub-agents" in result
assert "10 delegation rounds" in result
assert "{max_concurrent}" not in result
assert "{max_iterations}" not in result
def test_workflow_constant_not_empty(self):
assert len(EXPERIMENT_WORKFLOW) > 0
def test_delegation_has_placeholders(self):
assert "{max_concurrent}" in DELEGATION_STRATEGY
assert "{max_iterations}" in DELEGATION_STRATEGY
def test_delegation_no_placeholders(self):
assert "{max_concurrent}" not in DELEGATION_STRATEGY
assert "{max_iterations}" not in DELEGATION_STRATEGY