Release/v0.1.4 (#266)
* feat(middleware): reposition code interpreter middleware in the stack * feat(models): add qwen3.7-plus model entry and update context window comment * feat(models): add qwen3.7-max and qwen3.7-plus model entries for DashScope * feat(auxiliary): implement auxiliary model support for background tasks and tool selection - Added auxiliary model configuration to EvoScientistConfig. - Introduced _ensure_auxiliary_chat_model function to manage auxiliary model instances. - Updated onboarding steps to include auxiliary model selection. - Modified middleware to route tool selection to the auxiliary model when applicable. - Enhanced tests to cover auxiliary model functionality and configuration. * feat(steps): update UI backend selection options and descriptions * Refactor code structure for improved readability and maintainability * feat(patches): implement OpenRouter response reasoning item stripping to prevent multi-turn errors * feat: update version to v0.1.4 in badges, README, and pyproject.toml; adjust skill counts in steps.py * feat(config): add auxiliary model and provider environment variables to test setup
This commit is contained in:
@@ -411,6 +411,8 @@ def test_auto_approve_still_includes_ask_user_middleware(
|
||||
cfg.enable_ask_user = True
|
||||
cfg.auto_approve = True
|
||||
cfg.auto_mode = False
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
@@ -430,6 +432,8 @@ def test_auto_mode_disables_ask_user_middleware(
|
||||
cfg.enable_ask_user = True
|
||||
cfg.auto_approve = True
|
||||
cfg.auto_mode = True
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
@@ -458,6 +462,8 @@ def test_for_async_subagent_omits_ask_user_middleware(
|
||||
cfg.enable_ask_user = True
|
||||
cfg.auto_approve = False
|
||||
cfg.auto_mode = False
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
mock_model.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
|
||||
@@ -125,6 +125,8 @@ def test_inject_subagent_omits_memory_middleware_when_memory_disabled(
|
||||
cfg.memory_observations_enabled = False
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.ALL
|
||||
cfg.memory_workers_enabled = True
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _inject_subagent_middleware
|
||||
@@ -153,6 +155,8 @@ def test_inject_subagent_worker_only_observation_writer_keeps_live_tool_off(
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.WORKER
|
||||
cfg.memory_workers_enabled = True
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _inject_subagent_middleware
|
||||
@@ -191,6 +195,8 @@ def test_all_observation_writer_skips_turn_worker_without_profile_memory(
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.ALL
|
||||
cfg.memory_workers_enabled = True
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
mock_chat.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
@@ -259,6 +265,8 @@ def test_async_subagent_mode_filters_ask_user(
|
||||
cfg.memory_observations_enabled = True
|
||||
cfg.memory_observation_writer = MemoryObservationWriter.ALL
|
||||
cfg.memory_workers_enabled = True
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
mock_chat.return_value = MagicMock(profile={"max_input_tokens": 200_000})
|
||||
|
||||
|
||||
@@ -0,0 +1,159 @@
|
||||
"""Tests for the auxiliary-model resolver and its middleware scoping.
|
||||
|
||||
Covers ``EvoScientist.EvoScientist._ensure_auxiliary_chat_model`` (fallback to
|
||||
the main model when unset) and the wiring in ``_get_default_middleware`` that
|
||||
routes the main agent's tool selector to the auxiliary model while keeping
|
||||
context editing — and async sub-agents — on the main model.
|
||||
"""
|
||||
|
||||
from types import SimpleNamespace
|
||||
from unittest.mock import MagicMock, patch
|
||||
|
||||
import pytest
|
||||
|
||||
import EvoScientist.EvoScientist as E
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _reset_model_caches(monkeypatch):
|
||||
"""Isolate the module-level model caches per test."""
|
||||
monkeypatch.setattr(E, "_chat_model", None, raising=False)
|
||||
monkeypatch.setattr(E, "_chat_model_key", None, raising=False)
|
||||
monkeypatch.setattr(E, "_auxiliary_chat_model", None, raising=False)
|
||||
monkeypatch.setattr(E, "_auxiliary_chat_model_key", None, raising=False)
|
||||
|
||||
|
||||
def _cfg(**over):
|
||||
base = {
|
||||
"model": "main-m",
|
||||
"provider": "main-p",
|
||||
"auxiliary_model": "",
|
||||
"auxiliary_provider": "",
|
||||
}
|
||||
base.update(over)
|
||||
return SimpleNamespace(**base)
|
||||
|
||||
|
||||
class TestAuxiliaryResolver:
|
||||
def test_empty_returns_main_instance(self, monkeypatch):
|
||||
main = object()
|
||||
monkeypatch.setattr(E, "_ensure_config", lambda config=None: _cfg())
|
||||
monkeypatch.setattr(E, "_ensure_chat_model", lambda: main)
|
||||
assert E._ensure_auxiliary_chat_model() is main
|
||||
|
||||
def test_aux_equal_to_main_reuses_main_instance(self, monkeypatch):
|
||||
main = object()
|
||||
monkeypatch.setattr(
|
||||
E,
|
||||
"_ensure_config",
|
||||
lambda config=None: _cfg(
|
||||
auxiliary_model="main-m", auxiliary_provider="main-p"
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr(E, "_ensure_chat_model", lambda: main)
|
||||
assert E._ensure_auxiliary_chat_model() is main
|
||||
|
||||
def test_set_builds_auxiliary(self, monkeypatch):
|
||||
fake = object()
|
||||
get_chat_model = MagicMock(return_value=fake)
|
||||
monkeypatch.setattr(
|
||||
E,
|
||||
"_ensure_config",
|
||||
lambda config=None: _cfg(
|
||||
auxiliary_model="aux-m", auxiliary_provider="aux-p"
|
||||
),
|
||||
)
|
||||
monkeypatch.setattr("EvoScientist.llm.get_chat_model", get_chat_model)
|
||||
assert E._ensure_auxiliary_chat_model() is fake
|
||||
get_chat_model.assert_called_once_with(model="aux-m", provider="aux-p")
|
||||
|
||||
def test_empty_provider_falls_back_to_main_provider(self, monkeypatch):
|
||||
get_chat_model = MagicMock(return_value=object())
|
||||
monkeypatch.setattr(
|
||||
E,
|
||||
"_ensure_config",
|
||||
lambda config=None: _cfg(auxiliary_model="aux-m", auxiliary_provider=""),
|
||||
)
|
||||
monkeypatch.setattr("EvoScientist.llm.get_chat_model", get_chat_model)
|
||||
E._ensure_auxiliary_chat_model()
|
||||
get_chat_model.assert_called_once_with(model="aux-m", provider="main-p")
|
||||
|
||||
def test_set_chat_model_resets_aux_cache(self, monkeypatch):
|
||||
monkeypatch.setattr(E, "_auxiliary_chat_model", object(), raising=False)
|
||||
monkeypatch.setattr(E, "_auxiliary_chat_model_key", ("x", "y"), raising=False)
|
||||
monkeypatch.setattr(
|
||||
"EvoScientist.llm.get_chat_model", MagicMock(return_value=object())
|
||||
)
|
||||
E.set_chat_model("new-m", "new-p")
|
||||
assert E._auxiliary_chat_model is None
|
||||
assert E._auxiliary_chat_model_key is None
|
||||
|
||||
|
||||
def _mock_cfg():
|
||||
cfg = MagicMock()
|
||||
cfg.enable_ask_user = False
|
||||
cfg.auto_mode = False
|
||||
cfg.auto_approve = False
|
||||
cfg.model_fallbacks = None
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
return cfg
|
||||
|
||||
|
||||
class TestAuxiliaryMiddlewareScope:
|
||||
"""``_get_default_middleware`` routes only the right components to aux."""
|
||||
|
||||
def _capture(self):
|
||||
cap: dict[str, object] = {}
|
||||
|
||||
def fake_tool_selector(*args, model=None, **kwargs):
|
||||
cap["tool_selector"] = model
|
||||
return [MagicMock()]
|
||||
|
||||
def fake_context_editing(model=None, *args, **kwargs):
|
||||
cap["context_editing"] = model
|
||||
return MagicMock()
|
||||
|
||||
return cap, fake_tool_selector, fake_context_editing
|
||||
|
||||
def test_main_agent_tool_selector_aux_context_editing_main(self):
|
||||
cap, fake_ts, fake_ce = self._capture()
|
||||
main_model, aux_model = object(), object()
|
||||
with (
|
||||
patch.object(E, "_ensure_config", return_value=_mock_cfg()),
|
||||
patch.object(E, "_ensure_chat_model", return_value=main_model),
|
||||
patch.object(E, "_ensure_auxiliary_chat_model", return_value=aux_model),
|
||||
patch(
|
||||
"EvoScientist.middleware.create_tool_selector_middleware",
|
||||
side_effect=fake_ts,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.middleware.create_context_editing_middleware",
|
||||
side_effect=fake_ce,
|
||||
),
|
||||
):
|
||||
E._get_default_middleware()
|
||||
|
||||
assert cap["tool_selector"] is aux_model
|
||||
assert cap["context_editing"] is main_model
|
||||
|
||||
def test_async_subagent_tool_selector_stays_main(self):
|
||||
cap, fake_ts, fake_ce = self._capture()
|
||||
main_model, aux_model = object(), object()
|
||||
with (
|
||||
patch.object(E, "_ensure_config", return_value=_mock_cfg()),
|
||||
patch.object(E, "_ensure_chat_model", return_value=main_model),
|
||||
patch.object(E, "_ensure_auxiliary_chat_model", return_value=aux_model),
|
||||
patch(
|
||||
"EvoScientist.middleware.create_tool_selector_middleware",
|
||||
side_effect=fake_ts,
|
||||
),
|
||||
patch(
|
||||
"EvoScientist.middleware.create_context_editing_middleware",
|
||||
side_effect=fake_ce,
|
||||
),
|
||||
):
|
||||
E._get_default_middleware(for_async_subagent=True)
|
||||
|
||||
assert cap["tool_selector"] is main_model
|
||||
assert cap["context_editing"] is main_model
|
||||
@@ -50,6 +50,8 @@ def temp_config_dir(tmp_path, monkeypatch):
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED",
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATION_WRITER",
|
||||
"EVOSCIENTIST_MEMORY_WORKERS_ENABLED",
|
||||
"EVOSCIENTIST_AUXILIARY_MODEL",
|
||||
"EVOSCIENTIST_AUXILIARY_PROVIDER",
|
||||
]:
|
||||
monkeypatch.delenv(key, raising=False)
|
||||
return config_dir
|
||||
@@ -69,6 +71,8 @@ def clean_env(monkeypatch):
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATIONS_ENABLED",
|
||||
"EVOSCIENTIST_MEMORY_OBSERVATION_WRITER",
|
||||
"EVOSCIENTIST_MEMORY_WORKERS_ENABLED",
|
||||
"EVOSCIENTIST_AUXILIARY_MODEL",
|
||||
"EVOSCIENTIST_AUXILIARY_PROVIDER",
|
||||
]:
|
||||
monkeypatch.delenv(key, raising=False)
|
||||
|
||||
@@ -590,3 +594,39 @@ class TestApplyConfigToEnv:
|
||||
apply_config_to_env(config)
|
||||
|
||||
assert os.environ.get("OLLAMA_BASE_URL") == "http://existing:11434"
|
||||
|
||||
|
||||
class TestAuxiliaryModelConfig:
|
||||
"""auxiliary_model / auxiliary_provider config fields (plain str, optional)."""
|
||||
|
||||
def test_defaults_empty(self):
|
||||
cfg = EvoScientistConfig()
|
||||
assert cfg.auxiliary_model == ""
|
||||
assert cfg.auxiliary_provider == ""
|
||||
|
||||
def test_save_and_load_round_trip(self, temp_config_dir, clean_env):
|
||||
save_config(
|
||||
EvoScientistConfig(
|
||||
auxiliary_model="claude-haiku-4-5",
|
||||
auxiliary_provider="anthropic",
|
||||
)
|
||||
)
|
||||
loaded = load_config()
|
||||
assert loaded.auxiliary_model == "claude-haiku-4-5"
|
||||
assert loaded.auxiliary_provider == "anthropic"
|
||||
|
||||
def test_get_set_value(self, temp_config_dir, clean_env):
|
||||
save_config(EvoScientistConfig())
|
||||
assert set_config_value("auxiliary_model", "qwen3.6-flash") is True
|
||||
assert set_config_value("auxiliary_provider", "dashscope") is True
|
||||
assert get_config_value("auxiliary_model") == "qwen3.6-flash"
|
||||
assert get_config_value("auxiliary_provider") == "dashscope"
|
||||
|
||||
def test_env_overrides_file(self, temp_config_dir, monkeypatch):
|
||||
save_config(EvoScientistConfig(auxiliary_model="claude-haiku-4-5"))
|
||||
monkeypatch.setenv("EVOSCIENTIST_AUXILIARY_MODEL", "gpt-5.5")
|
||||
monkeypatch.setenv("EVOSCIENTIST_AUXILIARY_PROVIDER", "openai")
|
||||
|
||||
config = get_effective_config()
|
||||
assert config.auxiliary_model == "gpt-5.5"
|
||||
assert config.auxiliary_provider == "openai"
|
||||
|
||||
@@ -114,6 +114,8 @@ def test_default_middleware_includes_context_editing(mock_config, mock_model, mo
|
||||
cfg = MagicMock()
|
||||
cfg.enable_ask_user = False
|
||||
cfg.auto_approve = False
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _get_default_middleware
|
||||
@@ -148,6 +150,8 @@ def test_context_editing_before_overflow_mapper(mock_config, mock_model, mock_ts
|
||||
cfg = MagicMock()
|
||||
cfg.enable_ask_user = False
|
||||
cfg.auto_approve = False
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _get_default_middleware
|
||||
|
||||
+140
-2
@@ -402,13 +402,15 @@ class TestThirdPartyRouting:
|
||||
"""OpenRouter should use native 'openrouter' provider via init_chat_model."""
|
||||
mock_init.return_value = "mock_model"
|
||||
monkeypatch.setenv("OPENROUTER_API_KEY", "or-key-456")
|
||||
# Assert the DEFAULT effort, so isolate from any leaked env override.
|
||||
monkeypatch.delenv("EVOSCIENTIST_REASONING_EFFORT", raising=False)
|
||||
|
||||
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
||||
|
||||
call_kwargs = mock_init.call_args[1]
|
||||
assert call_kwargs["model_provider"] == "openrouter"
|
||||
assert call_kwargs["api_key"] == "or-key-456"
|
||||
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "disabled"}
|
||||
assert call_kwargs["reasoning"] == {"effort": "high", "summary": "auto"}
|
||||
|
||||
@patch("EvoScientist.llm.models.init_chat_model")
|
||||
def test_openrouter_reasoning_user_override(self, mock_init, monkeypatch):
|
||||
@@ -435,7 +437,7 @@ class TestThirdPartyRouting:
|
||||
get_chat_model("x-ai/grok-4.3", provider="openrouter")
|
||||
|
||||
call_kwargs = mock_init.call_args[1]
|
||||
assert call_kwargs["reasoning"] == {"effort": "medium", "summary": "disabled"}
|
||||
assert call_kwargs["reasoning"] == {"effort": "medium", "summary": "auto"}
|
||||
|
||||
@patch("EvoScientist.llm.models.init_chat_model")
|
||||
def test_custom_routes_through_openai(self, mock_init, monkeypatch):
|
||||
@@ -1994,6 +1996,142 @@ class TestPatchOpenAICaptureReasoningContent:
|
||||
assert msg.additional_kwargs.get("reasoning_content") == "use the tool"
|
||||
|
||||
|
||||
class TestIsResponsesReasoningItem:
|
||||
"""_is_responses_reasoning_item flags encrypted OpenAI-Responses items."""
|
||||
|
||||
def test_rs_id_is_responses_item(self):
|
||||
from EvoScientist.llm.patches import _is_responses_reasoning_item
|
||||
|
||||
assert _is_responses_reasoning_item({"id": "rs_09363d42", "type": "x"})
|
||||
|
||||
def test_encrypted_data_is_responses_item(self):
|
||||
from EvoScientist.llm.patches import _is_responses_reasoning_item
|
||||
|
||||
assert _is_responses_reasoning_item({"data": "gAAAAAB...", "type": "x"})
|
||||
|
||||
def test_plain_text_reasoning_is_not_responses_item(self):
|
||||
from EvoScientist.llm.patches import _is_responses_reasoning_item
|
||||
|
||||
assert not _is_responses_reasoning_item(
|
||||
{"type": "reasoning.text", "text": "thinking", "index": 0}
|
||||
)
|
||||
assert not _is_responses_reasoning_item("not a dict")
|
||||
|
||||
|
||||
class TestPatchOpenrouterStripResponsesReasoning:
|
||||
"""OpenAI-Responses encrypted reasoning items (`rs_` id / encrypted data)
|
||||
are stripped from outgoing OpenRouter assistant messages, preventing the
|
||||
multi-turn "Item with id 'rs_...' not found" 400 (store=false; #37777).
|
||||
"""
|
||||
|
||||
def _apply(self):
|
||||
import langchain_openrouter.chat_models as mod
|
||||
|
||||
import EvoScientist.llm.patches as patches
|
||||
|
||||
orig = mod._convert_message_to_dict
|
||||
orig_flag = patches._openrouter_reasoning_strip_patched
|
||||
patches._openrouter_reasoning_strip_patched = False
|
||||
patches._patch_openrouter_strip_responses_reasoning()
|
||||
return patches, mod, orig, orig_flag
|
||||
|
||||
@staticmethod
|
||||
def _restore(patches, mod, orig, orig_flag):
|
||||
mod._convert_message_to_dict = orig
|
||||
patches._openrouter_reasoning_strip_patched = orig_flag
|
||||
|
||||
def test_strips_encrypted_item_drops_key_when_empty(self):
|
||||
from langchain_core.messages import AIMessage
|
||||
|
||||
patches, mod, orig, orig_flag = self._apply()
|
||||
try:
|
||||
msg = AIMessage(
|
||||
content="done",
|
||||
additional_kwargs={
|
||||
"reasoning_details": [
|
||||
{
|
||||
"type": "reasoning.summary",
|
||||
"format": "openai-responses-v1",
|
||||
"id": "rs_09363d42b054",
|
||||
"data": "gAAAAAB...",
|
||||
"summary": "real reasoning text",
|
||||
"index": 0,
|
||||
}
|
||||
],
|
||||
},
|
||||
)
|
||||
result = mod._convert_message_to_dict(msg)
|
||||
# sole entry was an rs_ item → reasoning_details removed entirely.
|
||||
assert "reasoning_details" not in result
|
||||
finally:
|
||||
self._restore(patches, mod, orig, orig_flag)
|
||||
|
||||
def test_keeps_plain_text_reasoning(self):
|
||||
from langchain_core.messages import AIMessage
|
||||
|
||||
patches, mod, orig, orig_flag = self._apply()
|
||||
try:
|
||||
msg = AIMessage(
|
||||
content="done",
|
||||
additional_kwargs={
|
||||
"reasoning_details": [
|
||||
{"type": "reasoning.text", "text": "thinking", "index": 0},
|
||||
{"id": "rs_abc", "data": "blob", "index": 1},
|
||||
],
|
||||
},
|
||||
)
|
||||
result = mod._convert_message_to_dict(msg)
|
||||
kept = result["reasoning_details"]
|
||||
assert len(kept) == 1
|
||||
assert kept[0]["type"] == "reasoning.text"
|
||||
finally:
|
||||
self._restore(patches, mod, orig, orig_flag)
|
||||
|
||||
def test_does_not_mutate_original_message(self):
|
||||
from langchain_core.messages import AIMessage
|
||||
|
||||
patches, mod, orig, orig_flag = self._apply()
|
||||
try:
|
||||
details = [{"id": "rs_abc", "data": "blob"}]
|
||||
msg = AIMessage(
|
||||
content="x", additional_kwargs={"reasoning_details": details}
|
||||
)
|
||||
mod._convert_message_to_dict(msg)
|
||||
# stored history untouched — we filter a fresh list, not in place.
|
||||
assert details == [{"id": "rs_abc", "data": "blob"}]
|
||||
finally:
|
||||
self._restore(patches, mod, orig, orig_flag)
|
||||
|
||||
def test_patch_is_idempotent(self):
|
||||
patches, mod, orig, orig_flag = self._apply()
|
||||
try:
|
||||
wrapper = mod._convert_message_to_dict
|
||||
# Second call is guarded by the flag → must not re-wrap.
|
||||
patches._patch_openrouter_strip_responses_reasoning()
|
||||
assert mod._convert_message_to_dict is wrapper
|
||||
finally:
|
||||
self._restore(patches, mod, orig, orig_flag)
|
||||
|
||||
def test_non_dict_entry_is_kept(self):
|
||||
from langchain_core.messages import AIMessage
|
||||
|
||||
patches, mod, orig, orig_flag = self._apply()
|
||||
try:
|
||||
msg = AIMessage(
|
||||
content="done",
|
||||
additional_kwargs={
|
||||
"reasoning_details": [
|
||||
"opaque", # non-dict slipped in → kept, not crashed on
|
||||
{"id": "rs_abc", "data": "blob", "index": 1},
|
||||
],
|
||||
},
|
||||
)
|
||||
result = mod._convert_message_to_dict(msg)
|
||||
assert result["reasoning_details"] == ["opaque"]
|
||||
finally:
|
||||
self._restore(patches, mod, orig, orig_flag)
|
||||
|
||||
|
||||
# =============================================================================
|
||||
# Test _apply_auto_config
|
||||
# =============================================================================
|
||||
|
||||
+155
-5
@@ -46,15 +46,16 @@ def _patch_all_questionary(mock_q):
|
||||
|
||||
|
||||
class TestConstants:
|
||||
def test_steps_has_twelve_items(self):
|
||||
"""Test that STEPS contains exactly 12 steps."""
|
||||
assert len(STEPS) == 12
|
||||
def test_steps_has_thirteen_items(self):
|
||||
"""Test that STEPS contains exactly 13 steps."""
|
||||
assert len(STEPS) == 13
|
||||
assert STEPS == [
|
||||
"UI",
|
||||
"LangGraph Port",
|
||||
"Provider",
|
||||
"API Key",
|
||||
"Model",
|
||||
"Auxiliary Model",
|
||||
"Tavily Key",
|
||||
"Workspace",
|
||||
"Thinking",
|
||||
@@ -353,6 +354,20 @@ class TestStepProvider:
|
||||
assert result == "anthropic"
|
||||
mock_q.select.assert_called_once()
|
||||
|
||||
def test_default_value_and_label_override(self):
|
||||
"""default_value preselects a provider (re-run co-pilot default) and
|
||||
label customizes the prompt text."""
|
||||
from EvoScientist.config.onboard.steps import _step_provider
|
||||
|
||||
config = EvoScientistConfig(provider="anthropic")
|
||||
with patch("EvoScientist.config.onboard.steps.questionary") as mock_q:
|
||||
mock_q.select.return_value.ask.return_value = "openai"
|
||||
_step_provider(config, label="co-pilot", default_value="openrouter")
|
||||
|
||||
call = mock_q.select.call_args
|
||||
assert call.kwargs["default"] == "openrouter" # override, not config.provider
|
||||
assert "co-pilot" in call.args[0]
|
||||
|
||||
def test_raises_keyboard_interrupt_on_cancel(self):
|
||||
"""Test that _step_provider raises KeyboardInterrupt on cancel."""
|
||||
from EvoScientist.config.onboard.steps import _step_provider
|
||||
@@ -378,6 +393,38 @@ class TestStepModel:
|
||||
|
||||
assert result == "claude-sonnet-4-6"
|
||||
|
||||
def test_main_model_not_in_provider_list_defaults_to_first(self):
|
||||
"""Reset/main flow: a config.model that isn't in the chosen provider's
|
||||
list (e.g. provider switched to google-genai) defaults to that
|
||||
provider's first model, NOT the custom 'Type a model name...' entry."""
|
||||
from EvoScientist.config.onboard.steps import _step_model
|
||||
from EvoScientist.llm.models import get_models_for_provider
|
||||
|
||||
config = EvoScientistConfig(model="claude-sonnet-4-6")
|
||||
entries = get_models_for_provider("google-genai")
|
||||
with patch("EvoScientist.config.onboard.steps.questionary") as mock_q:
|
||||
mock_q.select.return_value.ask.return_value = entries[0][0]
|
||||
_step_model(config, "google-genai")
|
||||
|
||||
default = mock_q.select.call_args.kwargs["default"]
|
||||
assert default == entries[0][0]
|
||||
assert default != "__custom__"
|
||||
|
||||
def test_custom_default_value_preselects_and_prefills(self):
|
||||
"""Co-pilot re-run: a saved custom (non-registry) model preselects and
|
||||
prefills the 'Type a model name...' entry."""
|
||||
from EvoScientist.config.onboard.steps import _step_model
|
||||
|
||||
config = EvoScientistConfig()
|
||||
with patch("EvoScientist.config.onboard.steps.questionary") as mock_q:
|
||||
mock_q.select.return_value.ask.return_value = "__custom__"
|
||||
mock_q.text.return_value.ask.return_value = "my-private/model"
|
||||
result = _step_model(config, "openrouter", default_value="my-private/model")
|
||||
|
||||
assert mock_q.select.call_args.kwargs["default"] == "__custom__"
|
||||
assert mock_q.text.call_args.kwargs["default"] == "my-private/model"
|
||||
assert result == "my-private/model"
|
||||
|
||||
def test_raises_keyboard_interrupt_on_cancel(self):
|
||||
"""Test that _step_model raises KeyboardInterrupt on cancel."""
|
||||
from EvoScientist.config.onboard.steps import _step_model
|
||||
@@ -1143,6 +1190,7 @@ class TestRunOnboard:
|
||||
"anthropic", # Provider
|
||||
"api_key", # Anthropic auth mode (API key, not OAuth)
|
||||
"claude-sonnet-4-6", # Model
|
||||
"skip", # Auxiliary: Skip (single driver)
|
||||
"daemon", # Workspace mode
|
||||
True, # Show thinking
|
||||
]
|
||||
@@ -1173,6 +1221,101 @@ class TestRunOnboard:
|
||||
assert final_config.ui_backend == "tui"
|
||||
assert final_config.default_mode == "daemon"
|
||||
|
||||
def test_auxiliary_model_enabled_collects_provider_and_key(self):
|
||||
"""Enabling the auxiliary step stores its provider, model, and the
|
||||
chosen provider's API key (a different company than the main agent)."""
|
||||
from EvoScientist.config.onboard.wizard import run_onboard
|
||||
|
||||
mock_q = MagicMock()
|
||||
with (
|
||||
_patch_all_questionary(mock_q),
|
||||
patch("EvoScientist.config.onboard.wizard.load_config") as mock_load,
|
||||
patch("EvoScientist.config.onboard.wizard.save_config") as mock_save,
|
||||
patch("EvoScientist.config.onboard.wizard.console"),
|
||||
patch("EvoScientist.config.onboard.steps.console"),
|
||||
patch("EvoScientist.config.onboard.channels.console"),
|
||||
patch("EvoScientist.config.onboard.helpers.console"),
|
||||
patch("EvoScientist.config.onboard.wizard._step_tinytex"),
|
||||
):
|
||||
mock_load.return_value = EvoScientistConfig()
|
||||
|
||||
mock_q.select.return_value.ask.side_effect = [
|
||||
"tui", # UI backend
|
||||
"anthropic", # Provider
|
||||
"api_key", # Anthropic auth mode
|
||||
"claude-sonnet-4-6", # Model
|
||||
"assemble", # Auxiliary: Assemble
|
||||
"openai", # Auxiliary provider (a different company)
|
||||
"gpt-5.5", # Auxiliary model
|
||||
"daemon", # Workspace mode
|
||||
True, # Show thinking
|
||||
]
|
||||
mock_q.password.return_value.ask.side_effect = [
|
||||
"", # Main provider API key (keep current)
|
||||
"sk-aux-openai", # Auxiliary provider API key
|
||||
"", # Tavily key (keep current)
|
||||
]
|
||||
mock_q.confirm.return_value.ask.side_effect = [
|
||||
True, # Save config
|
||||
]
|
||||
mock_q.text.return_value.ask.side_effect = [
|
||||
"", # Workspace directory
|
||||
]
|
||||
mock_q.checkbox.return_value.ask.return_value = [] # Skills: skip
|
||||
|
||||
result = run_onboard(skip_validation=True)
|
||||
|
||||
assert result is True
|
||||
final_config = mock_save.call_args_list[-1].args[0]
|
||||
assert final_config.auxiliary_provider == "openai"
|
||||
assert final_config.auxiliary_model == "gpt-5.5"
|
||||
# The auxiliary provider's key is stored in its per-provider field.
|
||||
assert final_config.openai_api_key == "sk-aux-openai"
|
||||
# Main agent is untouched.
|
||||
assert final_config.provider == "anthropic"
|
||||
assert final_config.model == "claude-sonnet-4-6"
|
||||
|
||||
def test_auxiliary_custom_provider_collects_base_url(self):
|
||||
"""Regression for the custom-provider fix: a custom auxiliary provider
|
||||
collects its base URL (provider -> base URL -> key -> model order)."""
|
||||
from EvoScientist.config.onboard.wizard import run_onboard
|
||||
|
||||
mock_q = MagicMock()
|
||||
with (
|
||||
_patch_all_questionary(mock_q),
|
||||
patch("EvoScientist.config.onboard.wizard.load_config") as mock_load,
|
||||
patch("EvoScientist.config.onboard.wizard.save_config") as mock_save,
|
||||
patch("EvoScientist.config.onboard.wizard.console"),
|
||||
patch("EvoScientist.config.onboard.steps.console"),
|
||||
patch("EvoScientist.config.onboard.channels.console"),
|
||||
patch("EvoScientist.config.onboard.helpers.console"),
|
||||
):
|
||||
mock_load.return_value = EvoScientistConfig()
|
||||
mock_q.select.return_value.ask.side_effect = [
|
||||
"assemble", # Auxiliary: Assemble
|
||||
"custom-openai", # Auxiliary provider
|
||||
"gpt-5.5", # Auxiliary model (from the custom-openai registry)
|
||||
]
|
||||
mock_q.text.return_value.ask.side_effect = [
|
||||
"https://my-endpoint/v1", # Auxiliary base URL (custom provider)
|
||||
]
|
||||
mock_q.password.return_value.ask.side_effect = [
|
||||
"sk-aux-custom", # Auxiliary provider API key
|
||||
]
|
||||
mock_q.confirm.return_value.ask.side_effect = [True] # Save
|
||||
|
||||
result = run_onboard(
|
||||
skip_validation=True, only_sections={"auxiliary_model"}
|
||||
)
|
||||
|
||||
assert result is True
|
||||
final_config = mock_save.call_args_list[-1].args[0]
|
||||
assert final_config.auxiliary_provider == "custom-openai"
|
||||
assert final_config.auxiliary_model == "gpt-5.5"
|
||||
# Base URL must be collected for the custom auxiliary provider.
|
||||
assert final_config.custom_openai_base_url == "https://my-endpoint/v1"
|
||||
assert final_config.custom_openai_api_key == "sk-aux-custom"
|
||||
|
||||
def test_returns_false_on_cancel(self):
|
||||
"""Test that run_onboard returns False when cancelled."""
|
||||
from EvoScientist.config.onboard.wizard import run_onboard
|
||||
@@ -1218,6 +1361,7 @@ class TestRunOnboard:
|
||||
"anthropic", # Provider
|
||||
"api_key", # Anthropic auth mode
|
||||
"claude-sonnet-4-6", # Model
|
||||
"skip", # Auxiliary: Skip (single driver)
|
||||
"daemon", # Workspace mode
|
||||
True, # Show thinking
|
||||
]
|
||||
@@ -1288,11 +1432,14 @@ class TestRunOnboard:
|
||||
"anthropic",
|
||||
"api_key",
|
||||
"claude-sonnet-4-6",
|
||||
"skip", # Auxiliary: Skip (single driver)
|
||||
"daemon",
|
||||
True,
|
||||
]
|
||||
mock_q.password.return_value.ask.side_effect = ["", ""]
|
||||
mock_q.confirm.return_value.ask.side_effect = [False] # Save? = No
|
||||
mock_q.confirm.return_value.ask.side_effect = [
|
||||
False, # Save? = No
|
||||
]
|
||||
mock_q.text.return_value.ask.side_effect = [""]
|
||||
mock_q.checkbox.return_value.ask.return_value = []
|
||||
|
||||
@@ -1335,11 +1482,14 @@ class TestRunOnboard:
|
||||
"anthropic",
|
||||
"api_key",
|
||||
"claude-sonnet-4-6",
|
||||
"skip", # Auxiliary: Skip (single driver)
|
||||
"daemon",
|
||||
True,
|
||||
]
|
||||
mock_q.password.return_value.ask.side_effect = ["", ""]
|
||||
mock_q.confirm.return_value.ask.side_effect = [False] # Save? = No
|
||||
mock_q.confirm.return_value.ask.side_effect = [
|
||||
False, # Save? = No
|
||||
]
|
||||
mock_q.text.return_value.ask.side_effect = [""]
|
||||
mock_q.checkbox.return_value.ask.return_value = []
|
||||
|
||||
|
||||
@@ -40,6 +40,8 @@ def _mock_config():
|
||||
cfg.auto_mode = False
|
||||
cfg.auto_approve = False
|
||||
cfg.model_fallbacks = None
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
cfg.code_interpreter_timeout = 60
|
||||
cfg.code_interpreter_max_result_chars = 6000
|
||||
return cfg
|
||||
|
||||
@@ -161,6 +161,8 @@ def test_default_middleware_includes_tool_selector(mock_config, mock_model, mock
|
||||
cfg = MagicMock()
|
||||
cfg.enable_ask_user = False
|
||||
cfg.auto_approve = False
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _get_default_middleware
|
||||
@@ -196,6 +198,8 @@ def test_tool_selector_ordering(mock_config, mock_model, mock_ts):
|
||||
cfg = MagicMock()
|
||||
cfg.enable_ask_user = False
|
||||
cfg.auto_approve = False
|
||||
cfg.auxiliary_model = ""
|
||||
cfg.auxiliary_provider = ""
|
||||
mock_config.return_value = cfg
|
||||
|
||||
from EvoScientist.EvoScientist import _get_default_middleware
|
||||
|
||||
Reference in New Issue
Block a user